457 lines
18 KiB
Python
457 lines
18 KiB
Python
"""Kanban board watcher methods for GatewayRunner.
|
|
|
|
Background loops that subscribe to kanban boards, deliver notifications and
|
|
artifacts, and drive the multi-agent dispatcher. They use only ``self`` state,
|
|
so they live on a mixin ``GatewayRunner`` inherits. Per-tick work lives in
|
|
``kanban_watchers_notifier`` / ``kanban_watchers_dispatcher``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import contextlib
|
|
import os
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any, Callable, Optional
|
|
|
|
from gateway.kanban_watchers_common import _to_thread_process_service, logger # noqa: F401 (tests import via origin)
|
|
from gateway.kanban_watchers_notifier import ( # noqa: F401 (_wake_scope_id: tests import via origin)
|
|
_KanbanNotification,
|
|
_notifier_collect,
|
|
_wake_scope_id,
|
|
)
|
|
from gateway.kanban_watchers_dispatcher import (
|
|
_KanbanDispatcher,
|
|
_log_spawn_results,
|
|
_resolve_dispatcher_settings,
|
|
)
|
|
|
|
|
|
def _resolve_auto_decompose_settings(
|
|
load_config: Callable[[], Any],
|
|
) -> "tuple[bool, int]":
|
|
"""Resolve the live (enabled, per_tick) auto-decompose settings.
|
|
|
|
Read fresh every dispatcher tick so flipping ``kanban.auto_decompose:
|
|
false`` stops runaway fan-out on the next tick without a restart. Fails
|
|
safe: a config read error returns ``(False, 3)`` rather than re-enabling
|
|
a feature the user turned off. ``per_tick`` is clamped to ``>= 1``.
|
|
"""
|
|
try:
|
|
cfg = load_config()
|
|
except Exception:
|
|
return False, 3
|
|
kcfg = cfg.get("kanban", {}) if isinstance(cfg, dict) else {}
|
|
enabled = bool(kcfg.get("auto_decompose", True))
|
|
try:
|
|
per_tick = int(kcfg.get("auto_decompose_per_tick", 3) or 3)
|
|
except (TypeError, ValueError):
|
|
per_tick = 3
|
|
return enabled, max(per_tick, 1)
|
|
|
|
|
|
def _kanban_dispatch_allowed() -> bool:
|
|
"""Return False while the global emergency stop (`hermes pause`) is engaged.
|
|
|
|
Checked every tick before spawning, so a pause applies on the next tick;
|
|
in-flight workers are never touched. Fails open if estop is unimportable.
|
|
"""
|
|
try:
|
|
from agent.estop import check_paused
|
|
except ImportError:
|
|
return True
|
|
return not check_paused("kanban", logger)
|
|
|
|
|
|
def _acquire_singleton_lock(lock_path) -> "tuple[Optional[object], str]":
|
|
"""Take the exclusive, non-blocking advisory lock for the sole dispatcher.
|
|
|
|
Only one gateway machine-wide may run the embedded dispatcher: concurrent
|
|
dispatchers double reclaim frequency and claim events, and with
|
|
``wal_autocheckpoint=0`` concurrent manual checkpoints can corrupt index
|
|
pages. ``dispatch_in_gateway`` is the primary control; this is the backstop.
|
|
|
|
Returns ``(handle, "held")`` (caller must release via
|
|
:func:`_release_singleton_lock`), ``(None, "contended")`` when another
|
|
process holds it (caller must NOT dispatch), or ``(None, "unavailable")``
|
|
when locking cannot be performed (caller falls back to config control).
|
|
"""
|
|
try:
|
|
from gateway.status import _try_acquire_file_lock # deferred; same package
|
|
except ImportError:
|
|
return None, "unavailable"
|
|
try:
|
|
Path(lock_path).parent.mkdir(parents=True, exist_ok=True)
|
|
handle = open(str(lock_path), "a+", encoding="utf-8")
|
|
except OSError:
|
|
return None, "unavailable"
|
|
if not _try_acquire_file_lock(handle):
|
|
handle.close()
|
|
return None, "contended"
|
|
return handle, "held"
|
|
|
|
|
|
def _release_singleton_lock(handle) -> None:
|
|
"""Release a lock acquired via :func:`_acquire_singleton_lock`."""
|
|
if handle is None:
|
|
return
|
|
try:
|
|
from gateway.status import _release_file_lock
|
|
_release_file_lock(handle)
|
|
except Exception:
|
|
pass
|
|
with contextlib.suppress(Exception):
|
|
handle.close()
|
|
|
|
|
|
def _gc_retention_days() -> int:
|
|
"""``kanban.done_sub_retention_days`` (default 30; 0 disables), re-read per sweep; fails safe to 30."""
|
|
try:
|
|
from hermes_cli.config import load_config as _load_cfg
|
|
|
|
_kanban_cfg = (_load_cfg() or {}).get("kanban") or {}
|
|
return int(_kanban_cfg.get("done_sub_retention_days", 30))
|
|
except Exception:
|
|
return 30
|
|
|
|
|
|
class GatewayKanbanWatchersMixin:
|
|
"""Kanban watcher / notifier / dispatcher loops for GatewayRunner."""
|
|
|
|
def _owns_kanban_dispatcher_lock(self) -> bool:
|
|
"""Return whether this gateway currently owns the singleton lock."""
|
|
return getattr(self, "_kanban_dispatcher_lock_handle", None) is not None
|
|
|
|
def _release_kanban_dispatcher_lock(self) -> None:
|
|
"""Clear notifier-visible ownership before releasing the OS lock."""
|
|
handle = getattr(self, "_kanban_dispatcher_lock_handle", None)
|
|
self._kanban_dispatcher_lock_handle = None
|
|
_release_singleton_lock(handle)
|
|
|
|
async def _sleep_between_ticks(self, interval: float) -> None:
|
|
"""Sleep *interval* (floored to 1s) in 1s slices so stop() never waits a full interval."""
|
|
interval = max(interval, 1.0)
|
|
slept = 0.0
|
|
while slept < interval and self._running:
|
|
await asyncio.sleep(min(1.0, interval - slept))
|
|
slept += 1.0
|
|
|
|
async def _kanban_notifier_watcher(self, interval: float = 5.0) -> None:
|
|
"""Poll ``kanban_notify_subs`` and deliver terminal events to users.
|
|
|
|
Per subscription, claims ``task_events`` newer than the stored cursor
|
|
(kinds in TERMINAL_KINDS), sends one message per event to
|
|
``(platform, chat_id, thread_id)``, then advances the cursor. The
|
|
subscription is removed only when the task is ``archived``: ``done``
|
|
is reversible (review/continuation), so the cursor — not unsubscribing
|
|
— is the dedup mechanism. Earlier unsub-on-terminal silently dropped
|
|
users when the dispatcher respawned a crashed task.
|
|
|
|
All SQLite work runs in a thread; one tick's failure never stops the
|
|
next. Iterates every board on disk per tick.
|
|
"""
|
|
from gateway.config import Platform as _Platform
|
|
try:
|
|
from hermes_cli import kanban_db as _kb
|
|
except Exception:
|
|
logger.warning("kanban notifier: kanban_db not importable; notifier disabled")
|
|
return
|
|
|
|
sub_fail_counts: dict[tuple, int] = getattr(
|
|
self, "_kanban_sub_fail_counts", {}
|
|
)
|
|
self._kanban_sub_fail_counts = sub_fail_counts
|
|
notifier_profile = getattr(self, "_kanban_notifier_profile", None)
|
|
if not notifier_profile:
|
|
notifier_profile = self._active_profile_name()
|
|
self._kanban_notifier_profile = notifier_profile
|
|
|
|
# Initial delay so the gateway can finish wiring adapters.
|
|
await asyncio.sleep(5)
|
|
|
|
# Stale done-sub GC: subs survive ``done``, so boards that never
|
|
# archive would accumulate rows scanned every tick. One DELETE per
|
|
# board, at startup and at most hourly.
|
|
_GC_INTERVAL_SECONDS = 3600.0
|
|
_gc_next_at = 0.0 # 0 → sweep on the first tick after startup
|
|
|
|
while self._running:
|
|
try:
|
|
_gc_due = time.monotonic() >= _gc_next_at
|
|
_retention = 30
|
|
if _gc_due:
|
|
_gc_next_at = time.monotonic() + _GC_INTERVAL_SECONDS
|
|
_retention = _gc_retention_days()
|
|
|
|
deliveries = await asyncio.to_thread(
|
|
_notifier_collect, self, _kb,
|
|
notifier_profile=notifier_profile,
|
|
gc_due=_gc_due,
|
|
gc_retention_days=_retention,
|
|
)
|
|
for d in deliveries:
|
|
await _KanbanNotification(
|
|
self, d, platform_cls=_Platform, sub_fail_counts=sub_fail_counts,
|
|
).deliver()
|
|
except Exception as exc:
|
|
logger.warning("kanban notifier tick failed: %s", exc)
|
|
await self._sleep_between_ticks(interval)
|
|
|
|
def _kanban_sub_op(self, board: Optional[str], op: str, sub: dict, **extra: Any) -> None:
|
|
"""Sync helper (runs in to_thread): call ``kanban_db.<op>`` for one subscription on its board."""
|
|
from hermes_cli import kanban_db as _kb
|
|
conn = _kb.connect(board=board)
|
|
try:
|
|
getattr(_kb, op)(
|
|
conn,
|
|
task_id=sub["task_id"],
|
|
platform=sub["platform"],
|
|
chat_id=sub["chat_id"],
|
|
thread_id=sub.get("thread_id") or "",
|
|
**extra,
|
|
)
|
|
finally:
|
|
conn.close()
|
|
|
|
def _kanban_advance(
|
|
self, sub: dict, cursor: int, board: Optional[str] = None,
|
|
) -> None:
|
|
"""Advance a subscription's cursor on the board that owns it."""
|
|
self._kanban_sub_op(board, "advance_notify_cursor", sub, new_cursor=cursor)
|
|
|
|
def _kanban_unsub(self, sub: dict, board: Optional[str] = None) -> None:
|
|
self._kanban_sub_op(board, "remove_notify_sub", sub)
|
|
|
|
def _kanban_rewind(
|
|
self,
|
|
sub: dict,
|
|
claimed_cursor: int,
|
|
old_cursor: int,
|
|
board: Optional[str] = None,
|
|
) -> None:
|
|
"""Undo a claimed notification cursor after send failure."""
|
|
self._kanban_sub_op(
|
|
board, "rewind_notify_cursor", sub,
|
|
claimed_cursor=claimed_cursor, old_cursor=old_cursor,
|
|
)
|
|
|
|
async def _deliver_kanban_artifacts(
|
|
self,
|
|
*,
|
|
adapter,
|
|
chat_id: str,
|
|
metadata: dict,
|
|
event_payload: Optional[dict],
|
|
task,
|
|
) -> None:
|
|
"""Upload artifact files referenced by a completed kanban task.
|
|
|
|
Sources, in priority order: ``event_payload['artifacts']``,
|
|
``event_payload['summary']``, then ``task.result`` (legacy). Paths are
|
|
deduplicated, missing files are skipped (may be mentioned for
|
|
reference only), and upload errors are logged, never raised.
|
|
"""
|
|
candidates: list[str] = []
|
|
seen: set[str] = set()
|
|
|
|
def _add(path: str) -> None:
|
|
if not path:
|
|
return
|
|
expanded = os.path.expanduser(path)
|
|
if expanded in seen or not os.path.isfile(expanded):
|
|
return
|
|
seen.add(expanded)
|
|
candidates.append(expanded)
|
|
|
|
if isinstance(event_payload, dict):
|
|
raw = event_payload.get("artifacts")
|
|
if isinstance(raw, (list, tuple)):
|
|
for item in raw:
|
|
if isinstance(item, str):
|
|
_add(item)
|
|
|
|
summary = event_payload.get("summary")
|
|
if isinstance(summary, str) and summary:
|
|
paths, _ = adapter.extract_local_files(summary)
|
|
for p in paths:
|
|
_add(p)
|
|
|
|
if task is not None and getattr(task, "result", None):
|
|
paths, _ = adapter.extract_local_files(str(task.result))
|
|
for p in paths:
|
|
_add(p)
|
|
|
|
if not candidates:
|
|
return
|
|
|
|
from gateway.platforms.base import BasePlatformAdapter
|
|
candidates = BasePlatformAdapter.filter_local_delivery_paths(candidates)
|
|
if not candidates:
|
|
return
|
|
|
|
_IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp"}
|
|
_VIDEO_EXTS = {".mp4", ".mov", ".avi", ".mkv", ".webm", ".3gp"}
|
|
|
|
from urllib.parse import quote as _quote
|
|
|
|
# Images ride one send_multiple_images call (batch uploads on Signal/Slack).
|
|
image_paths = [p for p in candidates if Path(p).suffix.lower() in _IMAGE_EXTS]
|
|
other_paths = [p for p in candidates if Path(p).suffix.lower() not in _IMAGE_EXTS]
|
|
|
|
if image_paths:
|
|
try:
|
|
batch = [(f"file://{_quote(p)}", "") for p in image_paths]
|
|
await adapter.send_multiple_images(
|
|
chat_id=chat_id, images=batch, metadata=metadata,
|
|
)
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"kanban notifier: image batch upload failed: %s", exc,
|
|
)
|
|
|
|
for path in other_paths:
|
|
try:
|
|
if Path(path).suffix.lower() in _VIDEO_EXTS:
|
|
await adapter.send_video(
|
|
chat_id=chat_id, video_path=path, metadata=metadata,
|
|
)
|
|
else:
|
|
await adapter.send_document(
|
|
chat_id=chat_id, file_path=path, metadata=metadata,
|
|
)
|
|
except Exception as exc:
|
|
logger.warning(
|
|
"kanban notifier: artifact upload (%s) failed: %s",
|
|
path, exc,
|
|
)
|
|
|
|
async def _kanban_dispatcher_watcher(self) -> None:
|
|
"""Embedded kanban dispatcher — one tick every `dispatch_interval_seconds`.
|
|
|
|
Gated by `kanban.dispatch_in_gateway` (default True); when false the
|
|
loop exits and an external `hermes kanban daemon` is expected. Each
|
|
tick runs :func:`kanban_db.dispatch_once` in a thread; one tick's
|
|
failure never stops the next. Shutdown: ``self._running`` is checked
|
|
between ticks and the in-flight ``to_thread`` returns on its own.
|
|
"""
|
|
# Config is read once at boot (restart to apply), except the
|
|
# auto-decompose toggle which is re-read every tick. The env var is
|
|
# an escape hatch to disable without editing YAML.
|
|
try:
|
|
from hermes_cli.config import load_config as _load_config
|
|
except Exception:
|
|
logger.warning("kanban dispatcher: config loader unavailable; disabled")
|
|
return
|
|
env_override = os.environ.get("HERMES_KANBAN_DISPATCH_IN_GATEWAY", "").strip().lower()
|
|
if env_override in {"0", "false", "no", "off"}:
|
|
logger.info("kanban dispatcher: disabled via HERMES_KANBAN_DISPATCH_IN_GATEWAY env")
|
|
return
|
|
|
|
try:
|
|
cfg = _load_config()
|
|
except Exception as exc:
|
|
logger.warning("kanban dispatcher: cannot load config (%s); disabled", exc)
|
|
return
|
|
kanban_cfg = cfg.get("kanban", {}) if isinstance(cfg, dict) else {}
|
|
if not kanban_cfg.get("dispatch_in_gateway", True):
|
|
logger.info(
|
|
"kanban dispatcher: disabled via config kanban.dispatch_in_gateway=false"
|
|
)
|
|
return
|
|
|
|
try:
|
|
from hermes_cli import kanban_db as _kb
|
|
except Exception:
|
|
logger.warning("kanban dispatcher: kanban_db not importable; dispatcher disabled")
|
|
return
|
|
|
|
# Single-dispatcher backstop (see _acquire_singleton_lock). The lock
|
|
# lives at the machine-global kanban root, so it serialises ALL gateways.
|
|
self._kanban_dispatcher_lock_handle = None
|
|
_lock_path = _kb.kanban_home() / "kanban" / ".dispatcher.lock"
|
|
_lock_handle, _lock_state = _acquire_singleton_lock(_lock_path)
|
|
if _lock_state == "contended":
|
|
logger.info(
|
|
"kanban dispatcher: another gateway already holds the dispatcher "
|
|
"lock (%s); this gateway will NOT dispatch.", _lock_path,
|
|
)
|
|
return
|
|
if _lock_state == "held":
|
|
self._kanban_dispatcher_lock_handle = _lock_handle # hold for process lifetime
|
|
logger.info("kanban dispatcher: holding singleton dispatcher lock (%s)", _lock_path)
|
|
else:
|
|
logger.warning(
|
|
"kanban dispatcher: advisory lock unavailable at %s; proceeding "
|
|
"on config control alone.", _lock_path,
|
|
)
|
|
|
|
settings = _resolve_dispatcher_settings(kanban_cfg, _kb)
|
|
interval = settings.interval
|
|
|
|
# Initial delay so adapters are wired before workers spawn (matches the notifier).
|
|
await asyncio.sleep(5)
|
|
|
|
# Health telemetry (mirrors `_cmd_daemon`): warn when the ready queue
|
|
# is non-empty but spawns are 0 for N consecutive ticks — usually a
|
|
# broken PATH, missing venv, or credential loss.
|
|
HEALTH_WINDOW = 6
|
|
bad_ticks = 0
|
|
last_warn_at = 0
|
|
dispatcher = _KanbanDispatcher(_kb, settings)
|
|
|
|
logger.info(
|
|
"kanban dispatcher: embedded in gateway (interval=%.1fs)", interval
|
|
)
|
|
while self._running:
|
|
try:
|
|
# Reap zombies before per-board work so a board DB failure
|
|
# cannot block cleanup of unrelated workers.
|
|
pids = await _to_thread_process_service(_kb.reap_worker_zombies)
|
|
if pids:
|
|
logger.info(
|
|
"kanban dispatcher: reaped %d zombie worker(s), pids=%s",
|
|
len(pids),
|
|
pids,
|
|
)
|
|
except Exception:
|
|
logger.exception("kanban dispatcher: zombie reaper failed")
|
|
|
|
try:
|
|
# Emergency stop (`hermes pause`): no auto-decompose or
|
|
# dispatch while paused; running workers finish naturally.
|
|
if not _kanban_dispatch_allowed():
|
|
bad_ticks = 0
|
|
else:
|
|
# Re-read the auto-decompose toggle live so disabling it
|
|
# takes effect on the next tick, not on restart.
|
|
_ad_enabled, _ad_per_tick = _resolve_auto_decompose_settings(_load_config)
|
|
if _ad_enabled:
|
|
await _to_thread_process_service(dispatcher.auto_decompose_tick, _ad_per_tick)
|
|
results = await _to_thread_process_service(dispatcher.tick_once)
|
|
any_spawned = _log_spawn_results(results)
|
|
# Health telemetry (aggregate across boards)
|
|
ready_pending = await _to_thread_process_service(dispatcher.ready_nonempty)
|
|
bad_ticks = bad_ticks + 1 if ready_pending and not any_spawned else 0
|
|
if bad_ticks >= HEALTH_WINDOW:
|
|
now = int(time.time())
|
|
if now - last_warn_at >= 300:
|
|
logger.warning(
|
|
"kanban dispatcher stuck: ready queue non-empty for "
|
|
"%d consecutive ticks but 0 workers spawned. Check "
|
|
"profile health (venv, PATH, credentials) and "
|
|
"`hermes kanban list --status ready`.",
|
|
bad_ticks,
|
|
)
|
|
last_warn_at = now
|
|
except asyncio.CancelledError:
|
|
logger.debug("kanban dispatcher: cancelled")
|
|
self._release_kanban_dispatcher_lock()
|
|
raise
|
|
except Exception:
|
|
logger.exception("kanban dispatcher: unexpected watcher error")
|
|
|
|
await self._sleep_between_ticks(interval)
|
|
|
|
self._release_kanban_dispatcher_lock()
|