c408601937
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.
- Dead code: 27 private helpers with zero references removed
(_auto_title_session, _resolve_review_model, _parse_make_targets,
_filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
_merge_runs, learning_graph_render bucket/period/node helpers,
_memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
_retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
_ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
_write_file/_write_json (4 curator report writers), _msg_text
(background_review <- side_question), _report_failure/_notify_title
(title_generator instant/auto paths), _is_under (verification_evidence),
_scoped SQL pair builder + _query (insights), _optional_lock
(background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
curator_backup summary bits, learning_graph_render buckets, insights
section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
comments collapsed; every non-obvious WHY/invariant kept in compact form.
Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
265 lines
11 KiB
Python
265 lines
11 KiB
Python
"""Idle deferral for background reviews on the managed local runtime.
|
|
|
|
When the review runtime IS the managed llama-server, the post-turn review fork
|
|
monopolizes the GPU the user's next prompt needs, for minutes — and the next
|
|
live turn cancels it, so an active session pays the decode cost AND loses the
|
|
learning. This module keeps the decision to learn where it was (turn end, nudge
|
|
intervals, full model, full transcript) and moves only the execution moment:
|
|
reviews bound for the managed local endpoint are queued and dispatched when the
|
|
machine is quiet. Everything else runs immediately.
|
|
|
|
Policy (auxiliary.background_review.defer): ``auto`` (default) defers exactly
|
|
when the resolved review runtime targets the managed local server; ``never`` is
|
|
the old behavior. Explicit /refine (focus set) never defers.
|
|
|
|
Queue semantics:
|
|
- One slot per session, newest snapshot wins (a review replays the whole
|
|
conversation, so coalescing is deduplication, not loss).
|
|
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn wrapper
|
|
observing the run token's cancel flag.
|
|
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless of
|
|
idleness — deferral may delay learning, never lose it.
|
|
- In-memory, best-effort: dropped on process exit, like the immediate fork.
|
|
|
|
Idle truth comes from the supervisor's /slots (machine-level, sees every client
|
|
incl. other profiles) and must hold for a settle window so a review is not
|
|
launched between two quick prompts. In-process turn liveness is tracked via
|
|
note_turn_started/note_turn_finished from run_conversation.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import threading
|
|
import time
|
|
import urllib.request
|
|
from typing import Any, Callable, Dict, Optional
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Sustained-quiet window before dispatch: long enough that two back-to-back
|
|
# prompts do not look idle, short enough that a coffee break runs the queue.
|
|
_IDLE_SETTLE_S = 15.0
|
|
# Poll cadence while the queue is non-empty. The thread parks when empty.
|
|
_POLL_INTERVAL_S = 5.0
|
|
# Age at which a queued review dispatches regardless of idleness.
|
|
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
|
|
|
|
|
|
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
|
|
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
|
|
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
|
|
return raw if raw in ("auto", "never") else "auto"
|
|
|
|
|
|
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
|
|
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError):
|
|
return _MAX_AGE_DEFAULT_S
|
|
return value if value > 0 else _MAX_AGE_DEFAULT_S
|
|
|
|
|
|
def review_targets_managed_local(agent: Any,
|
|
task_cfg: Optional[Dict[str, Any]]) -> bool:
|
|
"""Would this review fork decode on the llama-server WE manage?
|
|
|
|
Resolves the review runtime as the fork will and exact-matches its netloc
|
|
against the supervisor state file (cannot false-positive on external local
|
|
servers). Any failure reads False: immediate spawn is the safe default.
|
|
The netloc probe (one TTL-cached state-file read) runs FIRST so cloud-only
|
|
installs return False without resolving the runtime on the turn's tail.
|
|
"""
|
|
try:
|
|
from agent.auxiliary_client import (
|
|
_is_managed_local_endpoint,
|
|
_managed_local_netloc,
|
|
)
|
|
|
|
if not _managed_local_netloc():
|
|
return False
|
|
from agent.background_review import _resolve_review_runtime
|
|
|
|
runtime = _resolve_review_runtime(agent, task_cfg)
|
|
return _is_managed_local_endpoint(runtime.get("base_url"))
|
|
except Exception: # noqa: BLE001
|
|
return False
|
|
|
|
|
|
class _PendingReview:
|
|
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
|
|
|
|
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
|
|
self.agent = agent
|
|
self.session_key = session_key
|
|
self.kwargs = kwargs
|
|
self.enqueued_at = time.monotonic()
|
|
|
|
|
|
class ReviewIdleQueue:
|
|
"""Session-coalescing queue + idle-gated dispatcher thread."""
|
|
|
|
def __init__(self) -> None:
|
|
self._lock = threading.Lock()
|
|
self._pending: Dict[str, _PendingReview] = {}
|
|
self._wake = threading.Event()
|
|
self._thread: Optional[threading.Thread] = None
|
|
self._live_turns = 0
|
|
self._quiet_since: Optional[float] = None
|
|
# Test seams — replaced by unit tests, never in production.
|
|
self._now: Callable[[], float] = time.monotonic
|
|
self._server_idle: Callable[[], bool] = _managed_server_idle
|
|
|
|
# ── turn liveness (this process) ────────────────────────────
|
|
|
|
def note_turn_started(self) -> None:
|
|
with self._lock:
|
|
self._live_turns += 1
|
|
self._quiet_since = None
|
|
|
|
def note_turn_finished(self) -> None:
|
|
with self._lock:
|
|
self._live_turns = max(0, self._live_turns - 1)
|
|
if self._live_turns == 0:
|
|
self._quiet_since = self._now()
|
|
self._wake.set()
|
|
|
|
# ── queue ────────────────────────────────────────────────────
|
|
|
|
def enqueue(self, agent: Any, session_key: str,
|
|
kwargs: Dict[str, Any]) -> None:
|
|
"""Add (or replace — newest snapshot wins) a session's pending review."""
|
|
with self._lock:
|
|
existing = self._pending.get(session_key)
|
|
item = _PendingReview(agent, session_key, kwargs)
|
|
# Stamp through the queue's clock (test seam); keep the ORIGINAL
|
|
# enqueue time on coalesce so a busy session cannot push its
|
|
# review's age-out forever.
|
|
item.enqueued_at = (existing.enqueued_at if existing is not None
|
|
else self._now())
|
|
self._pending[session_key] = item
|
|
self._ensure_thread()
|
|
self._wake.set()
|
|
logger.info("Background review deferred (session=%s, queued=%d)",
|
|
session_key[-12:], len(self._pending))
|
|
|
|
def pending_count(self) -> int:
|
|
with self._lock:
|
|
return len(self._pending)
|
|
|
|
# ── dispatcher ───────────────────────────────────────────────
|
|
|
|
def _ensure_thread(self) -> None:
|
|
with self._lock:
|
|
if self._thread is None or not self._thread.is_alive():
|
|
self._thread = threading.Thread(
|
|
target=self._run, daemon=True, name="bg-review-idle-queue")
|
|
self._thread.start()
|
|
|
|
def _quiet_for(self) -> float:
|
|
"""Seconds this process has been turn-free (0 while a turn runs)."""
|
|
with self._lock:
|
|
if self._live_turns > 0 or self._quiet_since is None:
|
|
return 0.0
|
|
return self._now() - self._quiet_since
|
|
|
|
def _pop_dispatchable(self) -> Optional[_PendingReview]:
|
|
"""Oldest aged-out item, else any item once quiet+idle hold."""
|
|
with self._lock:
|
|
if not self._pending:
|
|
return None
|
|
items = sorted(self._pending.values(),
|
|
key=lambda p: p.enqueued_at)
|
|
aged = [p for p in items
|
|
if self._now() - p.enqueued_at
|
|
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
|
|
candidate = aged[0] if aged else None
|
|
if candidate is None:
|
|
if self._quiet_for() < _IDLE_SETTLE_S or not self._server_idle():
|
|
return None
|
|
with self._lock:
|
|
if not self._pending:
|
|
return None
|
|
candidate = min(self._pending.values(),
|
|
key=lambda p: p.enqueued_at)
|
|
with self._lock:
|
|
return self._pending.pop(candidate.session_key, None)
|
|
|
|
def _run(self) -> None:
|
|
while True:
|
|
self._wake.wait()
|
|
with self._lock:
|
|
if not self._pending:
|
|
self._wake.clear()
|
|
continue
|
|
item = None
|
|
try:
|
|
item = self._pop_dispatchable()
|
|
if item is not None:
|
|
if not self._still_enabled(item):
|
|
logger.info(
|
|
"Deferred background review dropped: reviews "
|
|
"were disabled while it was queued (session=%s)",
|
|
item.session_key[-12:])
|
|
continue
|
|
logger.info(
|
|
"Dispatching deferred background review "
|
|
"(session=%s, waited=%.0fs, queued=%d)",
|
|
item.session_key[-12:],
|
|
self._now() - item.enqueued_at,
|
|
self.pending_count())
|
|
item.agent._spawn_background_review_now(**item.kwargs)
|
|
except Exception: # noqa: BLE001 — dispatcher must survive anything
|
|
logger.warning("Deferred review dispatch failed",
|
|
exc_info=True)
|
|
if item is None:
|
|
time.sleep(_POLL_INTERVAL_S)
|
|
|
|
@staticmethod
|
|
def _still_enabled(item: _PendingReview) -> bool:
|
|
"""Re-check the enabled gate at DISPATCH time: minutes may pass in the
|
|
queue, and disabling reviews meanwhile must not be resurrected. Fail-open
|
|
like the gate itself."""
|
|
try:
|
|
from agent.background_review import load_background_review_settings
|
|
|
|
return load_background_review_settings()[0]
|
|
except Exception: # noqa: BLE001
|
|
return True
|
|
|
|
|
|
def _managed_server_idle() -> bool:
|
|
"""Machine-level idle: no processing slot on any loaded model of the
|
|
managed router. Unreachable/no state file reads idle (nothing to
|
|
contend with). One /models + one /slots call per loaded model."""
|
|
try:
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
from urllib.parse import quote
|
|
|
|
state = json.loads(state_path().read_text(encoding="utf-8"))
|
|
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
|
|
headers = {"Authorization": f"Bearer {state.get('api_key', '')}"}
|
|
if not base:
|
|
return True
|
|
|
|
def _get(path: str) -> Any:
|
|
req = urllib.request.Request(f"{base}{path}", headers=headers)
|
|
with urllib.request.urlopen(req, timeout=3) as r:
|
|
return json.loads(r.read())
|
|
|
|
loaded = [m["id"] for m in _get("/models").get("data", [])
|
|
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
|
|
for mid in loaded:
|
|
if any(s.get("is_processing") for s in _get(f"/slots?model={quote(mid)}")
|
|
if isinstance(s, dict)):
|
|
return False
|
|
return True
|
|
except Exception: # noqa: BLE001
|
|
return True
|
|
|
|
|
|
# Module singleton — one queue per process, like the load-progress watcher.
|
|
QUEUE = ReviewIdleQueue()
|