fix(kanban): text dispatch output and both "dispatcher stuck" warnings name the hold reason

`hermes kanban dispatch` (plain output), the standalone daemon's stuck warning
and the gateway's embedded dispatcher stuck warning all reported a bare
`Spawned: 0` / "0 workers spawned" while the respawn guard held every ready
card — the reason existed only as a `respawn_guarded` task event visible via
`hermes kanban tail`. Operators watching the gateway health warning for 73+
ticks (#111910) had nothing to act on.

- `kanban_db_dispatch.describe_suppression()` renders the guard reasons per
  task plus rate_limited / skipped_locked / memory_pressure for one or more
  DispatchResults, so the CLI daemon and gateway warnings share one wording:
  `Last tick held back: active_pr=1, memory_pressure=elevated.`
- plain `dispatch` output prints `Guarded (<reason>): <task id>` and the
  tick-level holds, mirroring the JSON fields.
- kanban docs: how to see why a ready card is not spawning.

Co-authored-by: Steven Saehrig <trac3r726@users.noreply.github.com>

Part of #111910
This commit is contained in:
teknium1
2026-09-15 12:09:41 -07:00
committed by Teknium
parent ec64ec0d24
commit c7f4bc5bd7
6 changed files with 134 additions and 3 deletions
+30
View File
@@ -20,6 +20,7 @@ from dataclasses import field
from pathlib import Path
from typing import Any
from typing import Callable
from typing import Iterable
from typing import Mapping
from typing import Optional
from typing import TYPE_CHECKING
@@ -135,6 +136,35 @@ class DispatchResult:
Reclaim/promotion bookkeeping still ran; deferred tasks stay queued."""
def describe_suppression(results: Iterable[Optional["DispatchResult"]]) -> str:
"""One line naming why the tick(s) held ready work back, or ``""``.
``active_pr=1, recent_success=2, rate_limited=1, skipped_locked=1,
memory_pressure=critical`` — the respawn-guard reasons counted per task
plus the tick-level holds. Feeds the "dispatcher stuck" warnings of the
CLI daemon and the embedded gateway dispatcher, which otherwise report a
bare zero-spawn count while ``hermes kanban tail`` is the only place the
guard reason is written (#111910).
"""
counts: dict[str, int] = {}
pressure: Optional[str] = None
for res in results:
if res is None:
continue
for _task_id, reason in res.respawn_guarded:
counts[reason] = counts.get(reason, 0) + 1
if res.rate_limited:
counts["rate_limited"] = counts.get("rate_limited", 0) + len(res.rate_limited)
if res.skipped_locked:
counts["skipped_locked"] = counts.get("skipped_locked", 0) + 1
if res.memory_pressure:
pressure = res.memory_pressure
parts = [f"{k}={v}" for k, v in sorted(counts.items())]
if pressure:
parts.append(f"memory_pressure={pressure}")
return ", ".join(parts)
# Bounded registry of recently-reaped worker exits, filled by the reap loop in
# ``dispatch_once`` and read by ``detect_crashed_workers`` to classify a dead-pid
# task. Entry: ``pid -> (raw_wait_status, reaped_at_epoch)``; raw status kept so
+11 -1
View File
@@ -143,6 +143,14 @@ def _cmd_dispatch(args: argparse.Namespace) -> int:
f"Skipped (non-spawnable assignee — terminal lane, OK): "
f"{', '.join(res.skipped_nonspawnable)}"
)
for tid, reason in res.respawn_guarded:
print(f"Guarded ({reason}): {tid}")
if res.rate_limited:
print(f"Rate-limited (released to ready, no failure counted): {', '.join(res.rate_limited)}")
if res.skipped_locked:
print("Skipped: another dispatcher holds this board's lock (no writes this tick)")
if res.memory_pressure:
print(f"Memory pressure {res.memory_pressure}: new workers restricted this tick")
return 0
@@ -212,10 +220,12 @@ def _cmd_daemon(args: argparse.Namespace) -> int:
if health_state["bad_ticks"] >= HEALTH_WINDOW:
now = int(time.time())
if now - health_state["last_warn_at"] >= 300:
held = kbd.describe_suppression([res])
held = f" Last tick held back: {held}." if held else ""
print(
f"[{_fmt_ts(now)}] WARN dispatcher stuck: ready queue non-empty for "
f"{health_state['bad_ticks']} consecutive ticks but 0 workers spawned "
f"successfully. Check profile health (venv, PATH, credentials) and `hermes "
f"successfully.{held} Check profile health (venv, PATH, credentials) and `hermes "
f"kanban list --status ready` / `hermes kanban list --status blocked` for "
f"recent spawn_failed tasks.",
file=sys.stderr, flush=True,