Files
hermes-agent/agent/curator_backup.py
T
Teknium c408601937 refactor(agent/review): simplify curator, background_review, verify, insights, title and learning modules (-22% LOC)
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.

- Dead code: 27 private helpers with zero references removed
  (_auto_title_session, _resolve_review_model, _parse_make_targets,
  _filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
  _merge_runs, learning_graph_render bucket/period/node helpers,
  _memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
  _retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
  _ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
  _write_file/_write_json (4 curator report writers), _msg_text
  (background_review <- side_question), _report_failure/_notify_title
  (title_generator instant/auto paths), _is_under (verification_evidence),
  _scoped SQL pair builder + _query (insights), _optional_lock
  (background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
  curator_backup summary bits, learning_graph_render buckets, insights
  section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
  comments collapsed; every non-obvious WHY/invariant kept in compact form.

Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
2026-09-02 13:30:25 -07:00

573 lines
23 KiB
Python

"""Curator snapshot + rollback.
Before any mutating curator pass, ``~/.hermes/skills/`` is tar.gz'd under
``~/.hermes/skills/.curator_backups/<utc-iso>/`` with a ``manifest.json``.
Rollback first snapshots the CURRENT tree (so it is itself undoable), then
extracts the chosen snapshot into place.
Excluded: ``.curator_backups/`` (would recurse), ``.hub/`` (hub-managed) and
``.git/`` (repository metadata — managed by git, not the curator).
Included: every skill dir, ``.usage.json``, ``.archive/``, ``.curator_state``
(so rollback also restores last-run-at and the curator doesn't re-fire),
``.bundled_manifest`` and ``.curator_suppressed``.
Each snapshot also copies ``~/.hermes/cron/jobs.json`` as ``cron-jobs.json``:
the consolidation pass rewrites cron ``skills``/``skill`` references in place,
so without it rolling back the skills tree would leave jobs pointing at
umbrellas. Rollback restores only those two fields; the rest is live state.
"""
from __future__ import annotations
import json
import logging
import os
import re
import shutil
import tarfile
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional, Set, Tuple
from hermes_constants import get_hermes_home
from agent.skill_utils import is_excluded_skill_path
from agent.curator import _read_config_section
from hermes_cli.sizefmt import format_bytes
logger = logging.getLogger(__name__)
DEFAULT_KEEP = 5
# Never rolled into a snapshot: .hub/ is owned by the skills hub (rolling it
# back breaks lockfile invariants); .curator_backups is the backup dir itself;
# .git is repository metadata — rolling it back breaks git tracking, and
# snapshots that include it grow with the full history (once backups are
# committed back, each snapshot contains the prior ones: 38MB of skills
# inflated to 24GB in weeks). ``_tar_filter`` applies the same set to nested
# paths, so a ``.git`` inside an individual skill dir is skipped too.
_EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"}
# Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename).
# Optional ``-NN`` suffix disambiguates two snapshots in the same second.
_ID_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}-\d{2}-\d{2}Z(-\d{2})?$")
CRON_JOBS_FILENAME = "cron-jobs.json"
_ARCHIVE_NAME = "skills.tar.gz"
def _skills_dir() -> Path:
return get_hermes_home() / "skills"
def _backups_dir() -> Path:
return _skills_dir() / ".curator_backups"
def _jobs_list(parsed: Any) -> Optional[list]:
"""jobs.json is ``{"jobs": [...], "updated_at": ...}``; also accept a bare
list for forward compat. None when neither shape matches."""
if isinstance(parsed, dict):
parsed = parsed.get("jobs")
return parsed if isinstance(parsed, list) else None
def _backup_cron_jobs_into(dest: Path) -> Dict[str, Any]:
"""Copy the live ``~/.hermes/cron/jobs.json`` into ``dest`` as ``cron-jobs.json``.
Never raises: a missing/unreadable file yields ``backed_up=False`` plus a
reason, and the snapshot proceeds."""
src = get_hermes_home() / "cron" / "jobs.json"
info: Dict[str, Any] = {"backed_up": False, "jobs_count": 0}
if not src.exists():
info["reason"] = "no cron/jobs.json present"
return info
try:
# utf-8-sig, same dialect as cron/jobs.load_jobs: a Windows-editor BOM
# would otherwise break json.loads AND be written into the backup.
raw = src.read_text(encoding="utf-8-sig")
except OSError as e:
logger.debug("Failed to read cron/jobs.json for backup: %s", e)
info["reason"] = f"read error: {e}"
return info
# jobs_count is a diagnostic only — an unparseable file is still stored raw.
try:
jobs = _jobs_list(json.loads(raw))
if jobs is not None:
info["jobs_count"] = len(jobs)
except (json.JSONDecodeError, TypeError):
info["jobs_count"] = 0
info["parse_warning"] = "jobs.json was not valid JSON at snapshot time"
try:
(dest / CRON_JOBS_FILENAME).write_text(raw, encoding="utf-8")
except OSError as e:
logger.debug("Failed to write cron backup file: %s", e)
info["reason"] = f"write error: {e}"
return info
info["backed_up"] = True
return info
def _utc_id(now: Optional[datetime] = None) -> str:
"""UTC ISO-ish filesystem-safe timestamp: ``2026-05-01T13-05-42Z``."""
s = (datetime.now(timezone.utc) if now is None else now).replace(microsecond=0).isoformat()
return s.removesuffix("+00:00").replace(":", "-") + "Z"
def _load_config() -> Dict[str, Any]:
return _read_config_section("curator", "backup", label="curator backup", log=logger)
def is_enabled() -> bool:
"""Default ON — the whole point of the backup is safety by default."""
return bool(_load_config().get("enabled", True))
def get_keep() -> int:
try:
n = int(_load_config().get("keep", DEFAULT_KEEP))
except (TypeError, ValueError):
n = DEFAULT_KEEP
return max(1, n)
# --- Snapshot ---
def _count_skill_files(base: Path) -> int:
try:
return sum(1 for p in base.rglob("SKILL.md") if not is_excluded_skill_path(p))
except OSError:
return 0
def _write_manifest(dest: Path, reason: str, archive_path: Path, skills_counted: int,
cron_info: Dict[str, Any]) -> None:
cron_jobs: Dict[str, Any] = {
"backed_up": bool(cron_info.get("backed_up", False)), "jobs_count": int(cron_info.get("jobs_count", 0)),
}
if not cron_info.get("backed_up"):
cron_jobs["reason"] = cron_info.get("reason", "not captured")
if cron_info.get("parse_warning"):
cron_jobs["parse_warning"] = cron_info["parse_warning"]
manifest = {
"id": dest.name, "reason": reason, "created_at": datetime.now(timezone.utc).isoformat(),
"archive": archive_path.name, "archive_bytes": archive_path.stat().st_size,
"skill_files": skills_counted, "cron_jobs": cron_jobs,
}
(dest / "manifest.json").write_text(json.dumps(manifest, indent=2, sort_keys=True), encoding="utf-8")
def _rmtree_quiet(path: Path) -> None:
shutil.rmtree(path, ignore_errors=True)
def snapshot_skills(reason: str = "manual", *, protect_ids: Optional[Set[str]] = None) -> Optional[Path]:
"""Create a tar.gz snapshot of ``~/.hermes/skills/`` and prune old ones.
Returns the snapshot dir, or None when skipped (disabled, skills dir missing,
IO error) — logged at debug so the curator never aborts a pass over a backup
failure. ``protect_ids`` survive the prune step (rollback protects its target)."""
if not is_enabled():
logger.debug("Curator backup disabled by config; skipping snapshot")
return None
skills = _skills_dir()
if not skills.exists():
logger.debug("No ~/.hermes/skills/ directory — nothing to back up")
return None
backups = _backups_dir()
try:
backups.mkdir(parents=True, exist_ok=True)
except OSError as e:
logger.debug("Failed to create backups dir %s: %s", backups, e)
return None
# Two curator runs in the same second must not clobber each other.
base_id = _utc_id()
snap_id = base_id
counter = 1
while (backups / snap_id).exists():
snap_id = f"{base_id}-{counter:02d}"
counter += 1
dest = backups / snap_id
try:
dest.mkdir(parents=True, exist_ok=False)
except OSError as e:
logger.debug("Failed to create snapshot dir %s: %s", dest, e)
return None
archive = dest / _ARCHIVE_NAME
def _tar_filter(tarinfo: tarfile.TarInfo) -> Optional[tarfile.TarInfo]:
parts = Path(tarinfo.name).parts
return None if any(p in _EXCLUDE_TOP_LEVEL for p in parts) else tarinfo
try:
with tarfile.open(archive, "w:gz", compresslevel=6) as tf:
for entry in sorted(skills.iterdir()):
if entry.name in _EXCLUDE_TOP_LEVEL:
continue
# arcname relative to skills/ so extraction drops back in cleanly.
tf.add(str(entry), arcname=entry.name, recursive=True, filter=_tar_filter)
# Cron capture is additive and never fails the snapshot; the manifest
# records whether it happened so rollback can say "no cron data".
_write_manifest(dest, reason, archive, _count_skill_files(skills), _backup_cron_jobs_into(dest))
except (OSError, tarfile.TarError) as e:
logger.debug("Curator snapshot failed: %s", e, exc_info=True)
_rmtree_quiet(dest) # clean up partial snapshot
return None
_prune_old(keep=get_keep(), protect=protect_ids)
logger.info("Curator snapshot created: %s (%s)", snap_id, reason)
return dest
def _prune_old(keep: int, protect: Optional[Set[str]] = None) -> List[str]:
"""Delete regular snapshots beyond the newest *keep*; returns deleted ids.
Ids in *protect* are never deleted — rollback() uses this so the mandatory
pre-rollback safety snapshot cannot evict the snapshot being restored. Stale
``.rollback-staging-*`` dirs (crashed rollback) are cleaned up on every call."""
protect = protect or set()
backups = _backups_dir()
if not backups.exists():
return []
dirs = [c for c in backups.iterdir() if c.is_dir()]
stale_staging = [c for c in dirs if c.name.startswith(".rollback-staging-")]
# Newest first (lexicographic works because the id is UTC ISO).
entries = sorted((c for c in dirs if _ID_RE.match(c.name)), key=lambda c: c.name, reverse=True)
deleted: List[str] = []
for path in entries[keep:]:
if path.name in protect:
continue
try:
shutil.rmtree(path)
deleted.append(path.name)
except OSError as e:
logger.debug("Failed to prune %s: %s", path, e)
for path in stale_staging:
try:
shutil.rmtree(path)
except OSError as e:
logger.debug("Failed to clean stale staging dir %s: %s", path, e)
return deleted
# --- List + rollback ---
def _read_manifest(snap_dir: Path) -> Dict[str, Any]:
try:
return json.loads((snap_dir / "manifest.json").read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return {}
def _is_restorable(child: Path) -> bool:
"""A real snapshot dir with a tarball (excludes ``.rollback-staging-*``)."""
return bool(child.is_dir() and _ID_RE.match(child.name) and (child / _ARCHIVE_NAME).exists())
def _restorable_snapshots() -> List[Path]:
"""Restorable snapshot dirs, newest first."""
backups = _backups_dir()
if not backups.exists():
return []
return [c for c in sorted(backups.iterdir(), reverse=True) if _is_restorable(c)]
def list_backups() -> List[Dict[str, Any]]:
"""All restorable snapshots (manifest dicts), newest first."""
out: List[Dict[str, Any]] = []
for child in _restorable_snapshots():
mf = _read_manifest(child)
mf.setdefault("id", child.name)
mf.setdefault("path", str(child))
if "archive_bytes" not in mf:
try:
mf["archive_bytes"] = (child / _ARCHIVE_NAME).stat().st_size
except OSError:
mf["archive_bytes"] = 0
out.append(mf)
return out
def _resolve_backup(backup_id: Optional[str]) -> Optional[Path]:
"""Path of the requested backup (newest if *backup_id* is None); None if no match."""
if backup_id:
target = _backups_dir() / backup_id
return target if _ID_RE.match(backup_id) and _is_restorable(target) else None
candidates = _restorable_snapshots()
return candidates[0] if candidates else None
def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]:
"""Reconcile backed-up cron skill links into the live ``cron/jobs.json``.
Only ``skills``/``skill`` are restored, and only on jobs that still exist
live (by ``id``) — everything else is live state. Backup-only jobs are
skipped and reported; live-only jobs untouched. Never raises; writes through
``cron.jobs`` under the scheduler's lock so we don't race tick()."""
report: Dict[str, Any] = {"attempted": False, "restored": [], "skipped_missing": [], "unchanged": 0, "error": None}
backup_file = snapshot_dir / CRON_JOBS_FILENAME
if not backup_file.exists():
report["error"] = f"snapshot has no {CRON_JOBS_FILENAME}"
return report
try:
backup_jobs = _jobs_list(json.loads(backup_file.read_text(encoding="utf-8")))
except (OSError, json.JSONDecodeError) as e:
report["error"] = f"failed to load backed-up jobs: {e}"
return report
if backup_jobs is None:
report["error"] = "backed-up cron-jobs.json has no jobs list"
return report
# Backed-up skill state keyed by job id (legacy single + modern list field).
backup_by_id: Dict[str, Dict[str, Any]] = {
job["id"]: {"skills": job.get("skills"), "skill": job.get("skill"), "name": job.get("name") or job["id"]}
for job in backup_jobs
if isinstance(job, dict) and isinstance(job.get("id"), str) and job.get("id")
}
if not backup_by_id:
report["attempted"] = True # we tried but there was nothing to do
return report
try:
from cron.jobs import load_jobs, save_jobs, _jobs_lock
except ImportError as e:
report["error"] = f"cron module unavailable: {e}"
return report
report["attempted"] = True
try:
with _jobs_lock():
live_jobs = load_jobs()
changed = False
live_ids = set()
for live in live_jobs:
if not isinstance(live, dict):
continue
jid = live.get("id")
if not isinstance(jid, str) or not jid:
continue
live_ids.add(jid)
backup = backup_by_id.get(jid)
if backup is None:
continue # live job didn't exist at snapshot time
cur = {"skills": live.get("skills"), "skill": live.get("skill")}
bkp = {"skills": backup.get("skills"), "skill": backup.get("skill")}
if cur == bkp:
report["unchanged"] += 1
continue
# Restore, preserving absence (don't add a key the backup lacked).
for key, value in bkp.items():
if value is None:
live.pop(key, None)
else:
live[key] = value
report["restored"].append({"job_id": jid, "job_name": backup.get("name") or jid, "from": cur, "to": bkp})
changed = True
# Jobs in backup but not live = user deleted them after the snapshot.
report["skipped_missing"] = [
{"job_id": jid, "job_name": backup.get("name") or jid}
for jid, backup in backup_by_id.items() if jid not in live_ids
]
if changed:
save_jobs(live_jobs)
except Exception as e: # noqa: BLE001 — rollback must not die mid-restore
logger.debug("Cron skill-link restore failed: %s", e, exc_info=True)
report["error"] = f"restore failed mid-flight: {e}"
return report
def _remove_entry(entry: Path) -> None:
if entry.is_dir() and not entry.is_symlink():
shutil.rmtree(entry)
elif entry.exists() or entry.is_symlink():
entry.unlink()
def _restore_excluded_subtrees(staged: Path, skills: Path) -> None:
"""Move excluded entries (nested ``.git``/``.hub``/...) from *staged* back
under *skills* after a successful extract. Snapshots never contain these, so
the staged copy of the live tree is the only source. ``.git`` may be a dir
or a file (submodule / worktree ``gitdir:`` pointer) — both are moved.
Best-effort and conditional: an entry is carried only when its parent skill
dir was restored and nothing sits at the target. If the target snapshot
predates the skill, the entry is dropped with the staging dir rather than
left orphaned; the safety snapshot excludes these paths too, so that case
is not undoable."""
def _carry(src: Path) -> None:
dest = skills / src.relative_to(staged)
if dest.parent.is_dir() and not dest.exists():
try:
shutil.move(str(src), str(dest))
except OSError as e:
logger.debug("Could not restore excluded entry %s: %s", src, e)
for dirpath, dirnames, filenames in os.walk(staged):
keep = []
for name in dirnames:
if name in _EXCLUDE_TOP_LEVEL:
_carry(Path(dirpath) / name)
else:
keep.append(name)
dirnames[:] = keep
for name in filenames:
if name in _EXCLUDE_TOP_LEVEL:
_carry(Path(dirpath) / name)
def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]:
"""Move staged entries back to their original paths; returns names that could
not be restored. ``shutil.move`` moves *into* an existing destination dir, so
partial-extract debris would bury the real skill (``skills/foo/foo/``) —
clear each original path first. The staged copy is authoritative."""
failed: List[str] = []
for orig, dest in moved:
try:
_remove_entry(orig)
shutil.move(str(dest), str(orig))
except OSError:
failed.append(orig.name)
return failed
def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]:
"""Restore ``~/.hermes/skills/`` from a snapshot (explicit id or newest):
safety-snapshot the CURRENT tree; stage current top-level entries; extract;
on failure move staged entries back. Returns ``(ok, message, snapshot_path)``."""
target = _resolve_backup(backup_id)
if target is None:
return (
False,
"no matching backup found"
+ (f" for id '{backup_id}'" if backup_id else "")
+ " (use `hermes curator rollback --list` to see available snapshots)",
None,
)
archive = target / _ARCHIVE_NAME
if not archive.exists():
return (False, f"snapshot {target.name} has no skills.tar.gz — corrupted?", None)
skills = _skills_dir()
skills.mkdir(parents=True, exist_ok=True)
backups = _backups_dir()
backups.mkdir(parents=True, exist_ok=True)
# Safety snapshot FIRST; bail if it fails, else a failed extract could leave
# the user with no skills. Protect the target from this snapshot's prune step.
try:
safety_snapshot = snapshot_skills(
reason=f"pre-rollback to {target.name}",
protect_ids={target.name},
)
except Exception as e:
return (False, f"pre-rollback safety snapshot failed: {e}", None)
if safety_snapshot is None:
return (
False,
"pre-rollback safety snapshot failed; backups may be disabled "
"or unavailable, and current skills were not changed",
None,
)
# Stage current entries so the extract lands in an empty tree; the safety
# snapshot above (not staging) is the user-facing undo handle.
staged = backups / f".rollback-staging-{_utc_id()}"
try:
staged.mkdir(parents=True, exist_ok=False)
except OSError as e:
return (False, f"failed to create staging dir: {e}", None)
moved: List[Tuple[Path, Path]] = []
try:
for entry in list(skills.iterdir()):
if entry.name in _EXCLUDE_TOP_LEVEL:
continue
dest = staged / entry.name
shutil.move(str(entry), str(dest))
moved.append((entry, dest))
except OSError as e:
_unstage(moved)
_rmtree_quiet(staged)
return (False, f"failed to stage current skills: {e}", None)
try:
with tarfile.open(archive, "r:gz") as tf:
# Reject absolute paths and ".." defensively; Python 3.12+ also
# gets filter='data', older interpreters fall back unfiltered.
for member in tf.getmembers():
if member.name.startswith("/") or ".." in Path(member.name).parts:
raise tarfile.TarError(f"refusing to extract unsafe path: {member.name!r}")
try:
tf.extractall(str(skills), filter="data") # type: ignore[call-arg]
except TypeError:
tf.extractall(str(skills)) # Python < 3.12 — no filter kwarg
except (OSError, tarfile.TarError) as e:
# A partial extract can leave entries the original tree never had;
# drop those first or the "restored" tree is skills + a slice of snapshot.
staged_names = {orig.name for orig, _ in moved}
for entry in list(skills.iterdir()):
if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names:
continue
try:
_remove_entry(entry)
except OSError:
pass
unrestored = _unstage(moved)
if unrestored:
# Don't claim a clean restore; keep the staging dir for hand recovery.
return (
False,
f"snapshot extract failed: {e} - could not restore "
f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}",
None,
)
_rmtree_quiet(staged)
return (False, f"snapshot extract failed (state restored): {e}", None)
# Snapshots never contain excluded subtrees (nested ``.git``, ``.hub``, ...),
# so carry them over from the staged live tree (top-level ``.git`` is never
# staged). Then staging is done; the undo handle is the safety snapshot.
_restore_excluded_subtrees(staged, skills)
_rmtree_quiet(staged)
# Cron reconciliation failures don't fail the rollback — the skills tree
# (the main guarantee) is already restored.
cron_report = _restore_cron_skill_links(target)
summary_bits = [f"restored from snapshot {target.name}"]
if cron_report.get("attempted"):
if cron_report.get("error"):
summary_bits.append(f"cron links: error — {cron_report['error']}")
else:
# (attempted with nothing matched — empty snapshot or no overlapping ids — says nothing)
parts = [f"{n} {label}" for n, label in (
(len(cron_report.get("restored") or []), "job(s) had skill links restored"),
(len(cron_report.get("skipped_missing") or []), "backed-up job(s) no longer exist (skipped)"),
(cron_report.get("unchanged", 0), "already matched"),
) if n]
if parts:
summary_bits.append("cron links: " + ", ".join(parts))
logger.info("Curator rollback: restored from %s (cron_report=%s)",
target.name, cron_report)
return (True, "; ".join(summary_bits), target)
# --- Human-readable summary for CLI ---
def summarize_backups() -> str:
rows = list_backups()
if not rows:
return "No curator snapshots yet."
header = f"{'id':<24} {'reason':<40} {'skills':>6} {'size':>8}"
lines = [header, "─" * len(header)] + [
f"{r.get('id','?'):<24} {(r.get('reason','?') or '?')[:40]:<40} "
f"{r.get('skill_files', 0):>6} {format_bytes(int(r.get('archive_bytes', 0))):>8}"
for r in rows
]
return "\n".join(lines)