c408601937
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.
- Dead code: 27 private helpers with zero references removed
(_auto_title_session, _resolve_review_model, _parse_make_targets,
_filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
_merge_runs, learning_graph_render bucket/period/node helpers,
_memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
_retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
_ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
_write_file/_write_json (4 curator report writers), _msg_text
(background_review <- side_question), _report_failure/_notify_title
(title_generator instant/auto paths), _is_under (verification_evidence),
_scoped SQL pair builder + _query (insights), _optional_lock
(background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
curator_backup summary bits, learning_graph_render buckets, insights
section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
comments collapsed; every non-obvious WHY/invariant kept in compact form.
Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
573 lines
23 KiB
Python
573 lines
23 KiB
Python
"""Curator snapshot + rollback.
|
|
|
|
Before any mutating curator pass, ``~/.hermes/skills/`` is tar.gz'd under
|
|
``~/.hermes/skills/.curator_backups/<utc-iso>/`` with a ``manifest.json``.
|
|
Rollback first snapshots the CURRENT tree (so it is itself undoable), then
|
|
extracts the chosen snapshot into place.
|
|
|
|
Excluded: ``.curator_backups/`` (would recurse), ``.hub/`` (hub-managed) and
|
|
``.git/`` (repository metadata — managed by git, not the curator).
|
|
Included: every skill dir, ``.usage.json``, ``.archive/``, ``.curator_state``
|
|
(so rollback also restores last-run-at and the curator doesn't re-fire),
|
|
``.bundled_manifest`` and ``.curator_suppressed``.
|
|
|
|
Each snapshot also copies ``~/.hermes/cron/jobs.json`` as ``cron-jobs.json``:
|
|
the consolidation pass rewrites cron ``skills``/``skill`` references in place,
|
|
so without it rolling back the skills tree would leave jobs pointing at
|
|
umbrellas. Rollback restores only those two fields; the rest is live state.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import shutil
|
|
import tarfile
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
|
|
from hermes_constants import get_hermes_home
|
|
from agent.skill_utils import is_excluded_skill_path
|
|
from agent.curator import _read_config_section
|
|
from hermes_cli.sizefmt import format_bytes
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
DEFAULT_KEEP = 5
|
|
|
|
# Never rolled into a snapshot: .hub/ is owned by the skills hub (rolling it
|
|
# back breaks lockfile invariants); .curator_backups is the backup dir itself;
|
|
# .git is repository metadata — rolling it back breaks git tracking, and
|
|
# snapshots that include it grow with the full history (once backups are
|
|
# committed back, each snapshot contains the prior ones: 38MB of skills
|
|
# inflated to 24GB in weeks). ``_tar_filter`` applies the same set to nested
|
|
# paths, so a ``.git`` inside an individual skill dir is skipped too.
|
|
_EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"}
|
|
|
|
# Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename).
|
|
# Optional ``-NN`` suffix disambiguates two snapshots in the same second.
|
|
_ID_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}-\d{2}-\d{2}Z(-\d{2})?$")
|
|
|
|
CRON_JOBS_FILENAME = "cron-jobs.json"
|
|
_ARCHIVE_NAME = "skills.tar.gz"
|
|
|
|
|
|
def _skills_dir() -> Path:
|
|
return get_hermes_home() / "skills"
|
|
|
|
|
|
def _backups_dir() -> Path:
|
|
return _skills_dir() / ".curator_backups"
|
|
|
|
|
|
def _jobs_list(parsed: Any) -> Optional[list]:
|
|
"""jobs.json is ``{"jobs": [...], "updated_at": ...}``; also accept a bare
|
|
list for forward compat. None when neither shape matches."""
|
|
if isinstance(parsed, dict):
|
|
parsed = parsed.get("jobs")
|
|
return parsed if isinstance(parsed, list) else None
|
|
|
|
|
|
def _backup_cron_jobs_into(dest: Path) -> Dict[str, Any]:
|
|
"""Copy the live ``~/.hermes/cron/jobs.json`` into ``dest`` as ``cron-jobs.json``.
|
|
Never raises: a missing/unreadable file yields ``backed_up=False`` plus a
|
|
reason, and the snapshot proceeds."""
|
|
src = get_hermes_home() / "cron" / "jobs.json"
|
|
info: Dict[str, Any] = {"backed_up": False, "jobs_count": 0}
|
|
if not src.exists():
|
|
info["reason"] = "no cron/jobs.json present"
|
|
return info
|
|
try:
|
|
# utf-8-sig, same dialect as cron/jobs.load_jobs: a Windows-editor BOM
|
|
# would otherwise break json.loads AND be written into the backup.
|
|
raw = src.read_text(encoding="utf-8-sig")
|
|
except OSError as e:
|
|
logger.debug("Failed to read cron/jobs.json for backup: %s", e)
|
|
info["reason"] = f"read error: {e}"
|
|
return info
|
|
# jobs_count is a diagnostic only — an unparseable file is still stored raw.
|
|
try:
|
|
jobs = _jobs_list(json.loads(raw))
|
|
if jobs is not None:
|
|
info["jobs_count"] = len(jobs)
|
|
except (json.JSONDecodeError, TypeError):
|
|
info["jobs_count"] = 0
|
|
info["parse_warning"] = "jobs.json was not valid JSON at snapshot time"
|
|
try:
|
|
(dest / CRON_JOBS_FILENAME).write_text(raw, encoding="utf-8")
|
|
except OSError as e:
|
|
logger.debug("Failed to write cron backup file: %s", e)
|
|
info["reason"] = f"write error: {e}"
|
|
return info
|
|
info["backed_up"] = True
|
|
return info
|
|
|
|
|
|
def _utc_id(now: Optional[datetime] = None) -> str:
|
|
"""UTC ISO-ish filesystem-safe timestamp: ``2026-05-01T13-05-42Z``."""
|
|
s = (datetime.now(timezone.utc) if now is None else now).replace(microsecond=0).isoformat()
|
|
return s.removesuffix("+00:00").replace(":", "-") + "Z"
|
|
|
|
|
|
def _load_config() -> Dict[str, Any]:
|
|
return _read_config_section("curator", "backup", label="curator backup", log=logger)
|
|
|
|
|
|
def is_enabled() -> bool:
|
|
"""Default ON — the whole point of the backup is safety by default."""
|
|
return bool(_load_config().get("enabled", True))
|
|
|
|
|
|
def get_keep() -> int:
|
|
try:
|
|
n = int(_load_config().get("keep", DEFAULT_KEEP))
|
|
except (TypeError, ValueError):
|
|
n = DEFAULT_KEEP
|
|
return max(1, n)
|
|
|
|
|
|
# --- Snapshot ---
|
|
|
|
def _count_skill_files(base: Path) -> int:
|
|
try:
|
|
return sum(1 for p in base.rglob("SKILL.md") if not is_excluded_skill_path(p))
|
|
except OSError:
|
|
return 0
|
|
|
|
|
|
def _write_manifest(dest: Path, reason: str, archive_path: Path, skills_counted: int,
|
|
cron_info: Dict[str, Any]) -> None:
|
|
cron_jobs: Dict[str, Any] = {
|
|
"backed_up": bool(cron_info.get("backed_up", False)), "jobs_count": int(cron_info.get("jobs_count", 0)),
|
|
}
|
|
if not cron_info.get("backed_up"):
|
|
cron_jobs["reason"] = cron_info.get("reason", "not captured")
|
|
if cron_info.get("parse_warning"):
|
|
cron_jobs["parse_warning"] = cron_info["parse_warning"]
|
|
manifest = {
|
|
"id": dest.name, "reason": reason, "created_at": datetime.now(timezone.utc).isoformat(),
|
|
"archive": archive_path.name, "archive_bytes": archive_path.stat().st_size,
|
|
"skill_files": skills_counted, "cron_jobs": cron_jobs,
|
|
}
|
|
(dest / "manifest.json").write_text(json.dumps(manifest, indent=2, sort_keys=True), encoding="utf-8")
|
|
|
|
|
|
def _rmtree_quiet(path: Path) -> None:
|
|
shutil.rmtree(path, ignore_errors=True)
|
|
|
|
|
|
def snapshot_skills(reason: str = "manual", *, protect_ids: Optional[Set[str]] = None) -> Optional[Path]:
|
|
"""Create a tar.gz snapshot of ``~/.hermes/skills/`` and prune old ones.
|
|
Returns the snapshot dir, or None when skipped (disabled, skills dir missing,
|
|
IO error) — logged at debug so the curator never aborts a pass over a backup
|
|
failure. ``protect_ids`` survive the prune step (rollback protects its target)."""
|
|
if not is_enabled():
|
|
logger.debug("Curator backup disabled by config; skipping snapshot")
|
|
return None
|
|
|
|
skills = _skills_dir()
|
|
if not skills.exists():
|
|
logger.debug("No ~/.hermes/skills/ directory — nothing to back up")
|
|
return None
|
|
|
|
backups = _backups_dir()
|
|
try:
|
|
backups.mkdir(parents=True, exist_ok=True)
|
|
except OSError as e:
|
|
logger.debug("Failed to create backups dir %s: %s", backups, e)
|
|
return None
|
|
|
|
# Two curator runs in the same second must not clobber each other.
|
|
base_id = _utc_id()
|
|
snap_id = base_id
|
|
counter = 1
|
|
while (backups / snap_id).exists():
|
|
snap_id = f"{base_id}-{counter:02d}"
|
|
counter += 1
|
|
|
|
dest = backups / snap_id
|
|
try:
|
|
dest.mkdir(parents=True, exist_ok=False)
|
|
except OSError as e:
|
|
logger.debug("Failed to create snapshot dir %s: %s", dest, e)
|
|
return None
|
|
|
|
archive = dest / _ARCHIVE_NAME
|
|
|
|
def _tar_filter(tarinfo: tarfile.TarInfo) -> Optional[tarfile.TarInfo]:
|
|
parts = Path(tarinfo.name).parts
|
|
return None if any(p in _EXCLUDE_TOP_LEVEL for p in parts) else tarinfo
|
|
|
|
try:
|
|
with tarfile.open(archive, "w:gz", compresslevel=6) as tf:
|
|
for entry in sorted(skills.iterdir()):
|
|
if entry.name in _EXCLUDE_TOP_LEVEL:
|
|
continue
|
|
# arcname relative to skills/ so extraction drops back in cleanly.
|
|
tf.add(str(entry), arcname=entry.name, recursive=True, filter=_tar_filter)
|
|
# Cron capture is additive and never fails the snapshot; the manifest
|
|
# records whether it happened so rollback can say "no cron data".
|
|
_write_manifest(dest, reason, archive, _count_skill_files(skills), _backup_cron_jobs_into(dest))
|
|
except (OSError, tarfile.TarError) as e:
|
|
logger.debug("Curator snapshot failed: %s", e, exc_info=True)
|
|
_rmtree_quiet(dest) # clean up partial snapshot
|
|
return None
|
|
|
|
_prune_old(keep=get_keep(), protect=protect_ids)
|
|
logger.info("Curator snapshot created: %s (%s)", snap_id, reason)
|
|
return dest
|
|
|
|
|
|
def _prune_old(keep: int, protect: Optional[Set[str]] = None) -> List[str]:
|
|
"""Delete regular snapshots beyond the newest *keep*; returns deleted ids.
|
|
Ids in *protect* are never deleted — rollback() uses this so the mandatory
|
|
pre-rollback safety snapshot cannot evict the snapshot being restored. Stale
|
|
``.rollback-staging-*`` dirs (crashed rollback) are cleaned up on every call."""
|
|
protect = protect or set()
|
|
backups = _backups_dir()
|
|
if not backups.exists():
|
|
return []
|
|
dirs = [c for c in backups.iterdir() if c.is_dir()]
|
|
stale_staging = [c for c in dirs if c.name.startswith(".rollback-staging-")]
|
|
# Newest first (lexicographic works because the id is UTC ISO).
|
|
entries = sorted((c for c in dirs if _ID_RE.match(c.name)), key=lambda c: c.name, reverse=True)
|
|
deleted: List[str] = []
|
|
for path in entries[keep:]:
|
|
if path.name in protect:
|
|
continue
|
|
try:
|
|
shutil.rmtree(path)
|
|
deleted.append(path.name)
|
|
except OSError as e:
|
|
logger.debug("Failed to prune %s: %s", path, e)
|
|
for path in stale_staging:
|
|
try:
|
|
shutil.rmtree(path)
|
|
except OSError as e:
|
|
logger.debug("Failed to clean stale staging dir %s: %s", path, e)
|
|
return deleted
|
|
|
|
|
|
# --- List + rollback ---
|
|
|
|
def _read_manifest(snap_dir: Path) -> Dict[str, Any]:
|
|
try:
|
|
return json.loads((snap_dir / "manifest.json").read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return {}
|
|
|
|
|
|
def _is_restorable(child: Path) -> bool:
|
|
"""A real snapshot dir with a tarball (excludes ``.rollback-staging-*``)."""
|
|
return bool(child.is_dir() and _ID_RE.match(child.name) and (child / _ARCHIVE_NAME).exists())
|
|
|
|
|
|
def _restorable_snapshots() -> List[Path]:
|
|
"""Restorable snapshot dirs, newest first."""
|
|
backups = _backups_dir()
|
|
if not backups.exists():
|
|
return []
|
|
return [c for c in sorted(backups.iterdir(), reverse=True) if _is_restorable(c)]
|
|
|
|
|
|
def list_backups() -> List[Dict[str, Any]]:
|
|
"""All restorable snapshots (manifest dicts), newest first."""
|
|
out: List[Dict[str, Any]] = []
|
|
for child in _restorable_snapshots():
|
|
mf = _read_manifest(child)
|
|
mf.setdefault("id", child.name)
|
|
mf.setdefault("path", str(child))
|
|
if "archive_bytes" not in mf:
|
|
try:
|
|
mf["archive_bytes"] = (child / _ARCHIVE_NAME).stat().st_size
|
|
except OSError:
|
|
mf["archive_bytes"] = 0
|
|
out.append(mf)
|
|
return out
|
|
|
|
|
|
def _resolve_backup(backup_id: Optional[str]) -> Optional[Path]:
|
|
"""Path of the requested backup (newest if *backup_id* is None); None if no match."""
|
|
if backup_id:
|
|
target = _backups_dir() / backup_id
|
|
return target if _ID_RE.match(backup_id) and _is_restorable(target) else None
|
|
candidates = _restorable_snapshots()
|
|
return candidates[0] if candidates else None
|
|
|
|
|
|
def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]:
|
|
"""Reconcile backed-up cron skill links into the live ``cron/jobs.json``.
|
|
Only ``skills``/``skill`` are restored, and only on jobs that still exist
|
|
live (by ``id``) — everything else is live state. Backup-only jobs are
|
|
skipped and reported; live-only jobs untouched. Never raises; writes through
|
|
``cron.jobs`` under the scheduler's lock so we don't race tick()."""
|
|
report: Dict[str, Any] = {"attempted": False, "restored": [], "skipped_missing": [], "unchanged": 0, "error": None}
|
|
backup_file = snapshot_dir / CRON_JOBS_FILENAME
|
|
if not backup_file.exists():
|
|
report["error"] = f"snapshot has no {CRON_JOBS_FILENAME}"
|
|
return report
|
|
|
|
try:
|
|
backup_jobs = _jobs_list(json.loads(backup_file.read_text(encoding="utf-8")))
|
|
except (OSError, json.JSONDecodeError) as e:
|
|
report["error"] = f"failed to load backed-up jobs: {e}"
|
|
return report
|
|
if backup_jobs is None:
|
|
report["error"] = "backed-up cron-jobs.json has no jobs list"
|
|
return report
|
|
|
|
# Backed-up skill state keyed by job id (legacy single + modern list field).
|
|
backup_by_id: Dict[str, Dict[str, Any]] = {
|
|
job["id"]: {"skills": job.get("skills"), "skill": job.get("skill"), "name": job.get("name") or job["id"]}
|
|
for job in backup_jobs
|
|
if isinstance(job, dict) and isinstance(job.get("id"), str) and job.get("id")
|
|
}
|
|
|
|
if not backup_by_id:
|
|
report["attempted"] = True # we tried but there was nothing to do
|
|
return report
|
|
|
|
try:
|
|
from cron.jobs import load_jobs, save_jobs, _jobs_lock
|
|
except ImportError as e:
|
|
report["error"] = f"cron module unavailable: {e}"
|
|
return report
|
|
|
|
report["attempted"] = True
|
|
try:
|
|
with _jobs_lock():
|
|
live_jobs = load_jobs()
|
|
changed = False
|
|
live_ids = set()
|
|
for live in live_jobs:
|
|
if not isinstance(live, dict):
|
|
continue
|
|
jid = live.get("id")
|
|
if not isinstance(jid, str) or not jid:
|
|
continue
|
|
live_ids.add(jid)
|
|
backup = backup_by_id.get(jid)
|
|
if backup is None:
|
|
continue # live job didn't exist at snapshot time
|
|
cur = {"skills": live.get("skills"), "skill": live.get("skill")}
|
|
bkp = {"skills": backup.get("skills"), "skill": backup.get("skill")}
|
|
if cur == bkp:
|
|
report["unchanged"] += 1
|
|
continue
|
|
# Restore, preserving absence (don't add a key the backup lacked).
|
|
for key, value in bkp.items():
|
|
if value is None:
|
|
live.pop(key, None)
|
|
else:
|
|
live[key] = value
|
|
report["restored"].append({"job_id": jid, "job_name": backup.get("name") or jid, "from": cur, "to": bkp})
|
|
changed = True
|
|
|
|
# Jobs in backup but not live = user deleted them after the snapshot.
|
|
report["skipped_missing"] = [
|
|
{"job_id": jid, "job_name": backup.get("name") or jid}
|
|
for jid, backup in backup_by_id.items() if jid not in live_ids
|
|
]
|
|
if changed:
|
|
save_jobs(live_jobs)
|
|
except Exception as e: # noqa: BLE001 — rollback must not die mid-restore
|
|
logger.debug("Cron skill-link restore failed: %s", e, exc_info=True)
|
|
report["error"] = f"restore failed mid-flight: {e}"
|
|
|
|
return report
|
|
|
|
|
|
def _remove_entry(entry: Path) -> None:
|
|
if entry.is_dir() and not entry.is_symlink():
|
|
shutil.rmtree(entry)
|
|
elif entry.exists() or entry.is_symlink():
|
|
entry.unlink()
|
|
|
|
|
|
def _restore_excluded_subtrees(staged: Path, skills: Path) -> None:
|
|
"""Move excluded entries (nested ``.git``/``.hub``/...) from *staged* back
|
|
under *skills* after a successful extract. Snapshots never contain these, so
|
|
the staged copy of the live tree is the only source. ``.git`` may be a dir
|
|
or a file (submodule / worktree ``gitdir:`` pointer) — both are moved.
|
|
Best-effort and conditional: an entry is carried only when its parent skill
|
|
dir was restored and nothing sits at the target. If the target snapshot
|
|
predates the skill, the entry is dropped with the staging dir rather than
|
|
left orphaned; the safety snapshot excludes these paths too, so that case
|
|
is not undoable."""
|
|
def _carry(src: Path) -> None:
|
|
dest = skills / src.relative_to(staged)
|
|
if dest.parent.is_dir() and not dest.exists():
|
|
try:
|
|
shutil.move(str(src), str(dest))
|
|
except OSError as e:
|
|
logger.debug("Could not restore excluded entry %s: %s", src, e)
|
|
|
|
for dirpath, dirnames, filenames in os.walk(staged):
|
|
keep = []
|
|
for name in dirnames:
|
|
if name in _EXCLUDE_TOP_LEVEL:
|
|
_carry(Path(dirpath) / name)
|
|
else:
|
|
keep.append(name)
|
|
dirnames[:] = keep
|
|
for name in filenames:
|
|
if name in _EXCLUDE_TOP_LEVEL:
|
|
_carry(Path(dirpath) / name)
|
|
|
|
|
|
def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]:
|
|
"""Move staged entries back to their original paths; returns names that could
|
|
not be restored. ``shutil.move`` moves *into* an existing destination dir, so
|
|
partial-extract debris would bury the real skill (``skills/foo/foo/``) —
|
|
clear each original path first. The staged copy is authoritative."""
|
|
failed: List[str] = []
|
|
for orig, dest in moved:
|
|
try:
|
|
_remove_entry(orig)
|
|
shutil.move(str(dest), str(orig))
|
|
except OSError:
|
|
failed.append(orig.name)
|
|
return failed
|
|
|
|
|
|
def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]:
|
|
"""Restore ``~/.hermes/skills/`` from a snapshot (explicit id or newest):
|
|
safety-snapshot the CURRENT tree; stage current top-level entries; extract;
|
|
on failure move staged entries back. Returns ``(ok, message, snapshot_path)``."""
|
|
target = _resolve_backup(backup_id)
|
|
if target is None:
|
|
return (
|
|
False,
|
|
"no matching backup found"
|
|
+ (f" for id '{backup_id}'" if backup_id else "")
|
|
+ " (use `hermes curator rollback --list` to see available snapshots)",
|
|
None,
|
|
)
|
|
archive = target / _ARCHIVE_NAME
|
|
if not archive.exists():
|
|
return (False, f"snapshot {target.name} has no skills.tar.gz — corrupted?", None)
|
|
|
|
skills = _skills_dir()
|
|
skills.mkdir(parents=True, exist_ok=True)
|
|
backups = _backups_dir()
|
|
backups.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Safety snapshot FIRST; bail if it fails, else a failed extract could leave
|
|
# the user with no skills. Protect the target from this snapshot's prune step.
|
|
try:
|
|
safety_snapshot = snapshot_skills(
|
|
reason=f"pre-rollback to {target.name}",
|
|
protect_ids={target.name},
|
|
)
|
|
except Exception as e:
|
|
return (False, f"pre-rollback safety snapshot failed: {e}", None)
|
|
if safety_snapshot is None:
|
|
return (
|
|
False,
|
|
"pre-rollback safety snapshot failed; backups may be disabled "
|
|
"or unavailable, and current skills were not changed",
|
|
None,
|
|
)
|
|
|
|
# Stage current entries so the extract lands in an empty tree; the safety
|
|
# snapshot above (not staging) is the user-facing undo handle.
|
|
staged = backups / f".rollback-staging-{_utc_id()}"
|
|
try:
|
|
staged.mkdir(parents=True, exist_ok=False)
|
|
except OSError as e:
|
|
return (False, f"failed to create staging dir: {e}", None)
|
|
|
|
moved: List[Tuple[Path, Path]] = []
|
|
try:
|
|
for entry in list(skills.iterdir()):
|
|
if entry.name in _EXCLUDE_TOP_LEVEL:
|
|
continue
|
|
dest = staged / entry.name
|
|
shutil.move(str(entry), str(dest))
|
|
moved.append((entry, dest))
|
|
except OSError as e:
|
|
_unstage(moved)
|
|
_rmtree_quiet(staged)
|
|
return (False, f"failed to stage current skills: {e}", None)
|
|
|
|
try:
|
|
with tarfile.open(archive, "r:gz") as tf:
|
|
# Reject absolute paths and ".." defensively; Python 3.12+ also
|
|
# gets filter='data', older interpreters fall back unfiltered.
|
|
for member in tf.getmembers():
|
|
if member.name.startswith("/") or ".." in Path(member.name).parts:
|
|
raise tarfile.TarError(f"refusing to extract unsafe path: {member.name!r}")
|
|
try:
|
|
tf.extractall(str(skills), filter="data") # type: ignore[call-arg]
|
|
except TypeError:
|
|
tf.extractall(str(skills)) # Python < 3.12 — no filter kwarg
|
|
except (OSError, tarfile.TarError) as e:
|
|
# A partial extract can leave entries the original tree never had;
|
|
# drop those first or the "restored" tree is skills + a slice of snapshot.
|
|
staged_names = {orig.name for orig, _ in moved}
|
|
for entry in list(skills.iterdir()):
|
|
if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names:
|
|
continue
|
|
try:
|
|
_remove_entry(entry)
|
|
except OSError:
|
|
pass
|
|
unrestored = _unstage(moved)
|
|
if unrestored:
|
|
# Don't claim a clean restore; keep the staging dir for hand recovery.
|
|
return (
|
|
False,
|
|
f"snapshot extract failed: {e} - could not restore "
|
|
f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}",
|
|
None,
|
|
)
|
|
_rmtree_quiet(staged)
|
|
return (False, f"snapshot extract failed (state restored): {e}", None)
|
|
|
|
# Snapshots never contain excluded subtrees (nested ``.git``, ``.hub``, ...),
|
|
# so carry them over from the staged live tree (top-level ``.git`` is never
|
|
# staged). Then staging is done; the undo handle is the safety snapshot.
|
|
_restore_excluded_subtrees(staged, skills)
|
|
_rmtree_quiet(staged)
|
|
|
|
# Cron reconciliation failures don't fail the rollback — the skills tree
|
|
# (the main guarantee) is already restored.
|
|
cron_report = _restore_cron_skill_links(target)
|
|
|
|
summary_bits = [f"restored from snapshot {target.name}"]
|
|
if cron_report.get("attempted"):
|
|
if cron_report.get("error"):
|
|
summary_bits.append(f"cron links: error — {cron_report['error']}")
|
|
else:
|
|
# (attempted with nothing matched — empty snapshot or no overlapping ids — says nothing)
|
|
parts = [f"{n} {label}" for n, label in (
|
|
(len(cron_report.get("restored") or []), "job(s) had skill links restored"),
|
|
(len(cron_report.get("skipped_missing") or []), "backed-up job(s) no longer exist (skipped)"),
|
|
(cron_report.get("unchanged", 0), "already matched"),
|
|
) if n]
|
|
if parts:
|
|
summary_bits.append("cron links: " + ", ".join(parts))
|
|
|
|
logger.info("Curator rollback: restored from %s (cron_report=%s)",
|
|
target.name, cron_report)
|
|
return (True, "; ".join(summary_bits), target)
|
|
|
|
|
|
# --- Human-readable summary for CLI ---
|
|
|
|
def summarize_backups() -> str:
|
|
rows = list_backups()
|
|
if not rows:
|
|
return "No curator snapshots yet."
|
|
header = f"{'id':<24} {'reason':<40} {'skills':>6} {'size':>8}"
|
|
lines = [header, "─" * len(header)] + [
|
|
f"{r.get('id','?'):<24} {(r.get('reason','?') or '?')[:40]:<40} "
|
|
f"{r.get('skill_files', 0):>6} {format_bytes(int(r.get('archive_bytes', 0))):>8}"
|
|
for r in rows
|
|
]
|
|
return "\n".join(lines)
|