Files
hermes-agent/agent/review_idle_queue.py
T
Teknium c408601937 refactor(agent/review): simplify curator, background_review, verify, insights, title and learning modules (-22% LOC)
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.

- Dead code: 27 private helpers with zero references removed
  (_auto_title_session, _resolve_review_model, _parse_make_targets,
  _filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
  _merge_runs, learning_graph_render bucket/period/node helpers,
  _memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
  _retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
  _ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
  _write_file/_write_json (4 curator report writers), _msg_text
  (background_review <- side_question), _report_failure/_notify_title
  (title_generator instant/auto paths), _is_under (verification_evidence),
  _scoped SQL pair builder + _query (insights), _optional_lock
  (background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
  curator_backup summary bits, learning_graph_render buckets, insights
  section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
  comments collapsed; every non-obvious WHY/invariant kept in compact form.

Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
2026-09-02 13:30:25 -07:00

265 lines
11 KiB
Python

"""Idle deferral for background reviews on the managed local runtime.
When the review runtime IS the managed llama-server, the post-turn review fork
monopolizes the GPU the user's next prompt needs, for minutes — and the next
live turn cancels it, so an active session pays the decode cost AND loses the
learning. This module keeps the decision to learn where it was (turn end, nudge
intervals, full model, full transcript) and moves only the execution moment:
reviews bound for the managed local endpoint are queued and dispatched when the
machine is quiet. Everything else runs immediately.
Policy (auxiliary.background_review.defer): ``auto`` (default) defers exactly
when the resolved review runtime targets the managed local server; ``never`` is
the old behavior. Explicit /refine (focus set) never defers.
Queue semantics:
- One slot per session, newest snapshot wins (a review replays the whole
conversation, so coalescing is deduplication, not loss).
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn wrapper
observing the run token's cancel flag.
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless of
idleness — deferral may delay learning, never lose it.
- In-memory, best-effort: dropped on process exit, like the immediate fork.
Idle truth comes from the supervisor's /slots (machine-level, sees every client
incl. other profiles) and must hold for a settle window so a review is not
launched between two quick prompts. In-process turn liveness is tracked via
note_turn_started/note_turn_finished from run_conversation.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.request
from typing import Any, Callable, Dict, Optional
logger = logging.getLogger(__name__)
# Sustained-quiet window before dispatch: long enough that two back-to-back
# prompts do not look idle, short enough that a coffee break runs the queue.
_IDLE_SETTLE_S = 15.0
# Poll cadence while the queue is non-empty. The thread parks when empty.
_POLL_INTERVAL_S = 5.0
# Age at which a queued review dispatches regardless of idleness.
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
return raw if raw in ("auto", "never") else "auto"
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
try:
value = float(raw)
except (TypeError, ValueError):
return _MAX_AGE_DEFAULT_S
return value if value > 0 else _MAX_AGE_DEFAULT_S
def review_targets_managed_local(agent: Any,
task_cfg: Optional[Dict[str, Any]]) -> bool:
"""Would this review fork decode on the llama-server WE manage?
Resolves the review runtime as the fork will and exact-matches its netloc
against the supervisor state file (cannot false-positive on external local
servers). Any failure reads False: immediate spawn is the safe default.
The netloc probe (one TTL-cached state-file read) runs FIRST so cloud-only
installs return False without resolving the runtime on the turn's tail.
"""
try:
from agent.auxiliary_client import (
_is_managed_local_endpoint,
_managed_local_netloc,
)
if not _managed_local_netloc():
return False
from agent.background_review import _resolve_review_runtime
runtime = _resolve_review_runtime(agent, task_cfg)
return _is_managed_local_endpoint(runtime.get("base_url"))
except Exception: # noqa: BLE001
return False
class _PendingReview:
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
self.agent = agent
self.session_key = session_key
self.kwargs = kwargs
self.enqueued_at = time.monotonic()
class ReviewIdleQueue:
"""Session-coalescing queue + idle-gated dispatcher thread."""
def __init__(self) -> None:
self._lock = threading.Lock()
self._pending: Dict[str, _PendingReview] = {}
self._wake = threading.Event()
self._thread: Optional[threading.Thread] = None
self._live_turns = 0
self._quiet_since: Optional[float] = None
# Test seams — replaced by unit tests, never in production.
self._now: Callable[[], float] = time.monotonic
self._server_idle: Callable[[], bool] = _managed_server_idle
# ── turn liveness (this process) ────────────────────────────
def note_turn_started(self) -> None:
with self._lock:
self._live_turns += 1
self._quiet_since = None
def note_turn_finished(self) -> None:
with self._lock:
self._live_turns = max(0, self._live_turns - 1)
if self._live_turns == 0:
self._quiet_since = self._now()
self._wake.set()
# ── queue ────────────────────────────────────────────────────
def enqueue(self, agent: Any, session_key: str,
kwargs: Dict[str, Any]) -> None:
"""Add (or replace — newest snapshot wins) a session's pending review."""
with self._lock:
existing = self._pending.get(session_key)
item = _PendingReview(agent, session_key, kwargs)
# Stamp through the queue's clock (test seam); keep the ORIGINAL
# enqueue time on coalesce so a busy session cannot push its
# review's age-out forever.
item.enqueued_at = (existing.enqueued_at if existing is not None
else self._now())
self._pending[session_key] = item
self._ensure_thread()
self._wake.set()
logger.info("Background review deferred (session=%s, queued=%d)",
session_key[-12:], len(self._pending))
def pending_count(self) -> int:
with self._lock:
return len(self._pending)
# ── dispatcher ───────────────────────────────────────────────
def _ensure_thread(self) -> None:
with self._lock:
if self._thread is None or not self._thread.is_alive():
self._thread = threading.Thread(
target=self._run, daemon=True, name="bg-review-idle-queue")
self._thread.start()
def _quiet_for(self) -> float:
"""Seconds this process has been turn-free (0 while a turn runs)."""
with self._lock:
if self._live_turns > 0 or self._quiet_since is None:
return 0.0
return self._now() - self._quiet_since
def _pop_dispatchable(self) -> Optional[_PendingReview]:
"""Oldest aged-out item, else any item once quiet+idle hold."""
with self._lock:
if not self._pending:
return None
items = sorted(self._pending.values(),
key=lambda p: p.enqueued_at)
aged = [p for p in items
if self._now() - p.enqueued_at
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
candidate = aged[0] if aged else None
if candidate is None:
if self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle():
return None
with self._lock:
if not self._pending:
return None
candidate = min(self._pending.values(),
key=lambda p: p.enqueued_at)
with self._lock:
return self._pending.pop(candidate.session_key, None)
def _run(self) -> None:
while True:
self._wake.wait()
with self._lock:
if not self._pending:
self._wake.clear()
continue
item = None
try:
item = self._pop_dispatchable()
if item is not None:
if not self._still_enabled(item):
logger.info(
"Deferred background review dropped: reviews "
"were disabled while it was queued (session=%s)",
item.session_key[-12:])
continue
logger.info(
"Dispatching deferred background review "
"(session=%s, waited=%.0fs, queued=%d)",
item.session_key[-12:],
self._now() - item.enqueued_at,
self.pending_count())
item.agent._spawn_background_review_now(**item.kwargs)
except Exception: # noqa: BLE001 — dispatcher must survive anything
logger.warning("Deferred review dispatch failed",
exc_info=True)
if item is None:
time.sleep(_POLL_INTERVAL_S)
@staticmethod
def _still_enabled(item: _PendingReview) -> bool:
"""Re-check the enabled gate at DISPATCH time: minutes may pass in the
queue, and disabling reviews meanwhile must not be resurrected. Fail-open
like the gate itself."""
try:
from agent.background_review import load_background_review_settings
return load_background_review_settings()[0]
except Exception: # noqa: BLE001
return True
def _managed_server_idle() -> bool:
"""Machine-level idle: no processing slot on any loaded model of the
managed router. Unreachable/no state file reads idle (nothing to
contend with). One /models + one /slots call per loaded model."""
try:
from hermes_cli.local_runtime.supervisor import state_path
from urllib.parse import quote
state = json.loads(state_path().read_text(encoding="utf-8"))
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
headers = {"Authorization": f"Bearer {state.get('api_key', '')}"}
if not base:
return True
def _get(path: str) -> Any:
req = urllib.request.Request(f"{base}{path}", headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
return json.loads(r.read())
loaded = [m["id"] for m in _get("/models").get("data", [])
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
for mid in loaded:
if any(s.get("is_processing") for s in _get(f"/slots?model={quote(mid)}")
if isinstance(s, dict)):
return False
return True
except Exception: # noqa: BLE001
return True
# Module singleton — one queue per process, like the load-progress watcher.
QUEUE = ReviewIdleQueue()