From 15751166291cfa82b20b233d33aac61c1a07ade6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 04:55:50 -0700 Subject: [PATCH] fix(update): pre-update snapshots now cover every profile, not just the invoking one (#66140) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The code swap and gateway fleet restart touch all profiles, but the pre-update quick snapshot photographed only the invoking profile's home — siblings had no snapshot for the post-update safety nets or manual restore to draw on. - backup.py: create_pre_update_snapshots_all_profiles() — the SAME snapshot set, per-file 1GiB cap, and keep policy as the invoking profile (no partial tier, no new restore-coherence class), each into the sibling's own state-snapshots/; restore_cron_jobs_all_profiles() runs the #34600 cron-loss safety net per profile against its OWN snapshot (same-generation by construction). - update_cmd.py: sibling snapshots taken right after the invoking profile's (best-effort, receipt-recorded); post-update cron restore extended to every sibling. - Docs: updating.md pre-update snapshot step now states the per-profile behavior and the file-loss-recovery vs rollback contract. - 9 unit tests + E2E (real files: sibling snapshot on disk, clobbered jobs.json restored 7/7 from the sibling's own snapshot, keep=1 prune). --- hermes_cli/backup.py | 98 ++++++++++++++ hermes_cli/update_cmd.py | 58 +++++++++ tests/hermes_cli/test_backup_all_profiles.py | 129 +++++++++++++++++++ website/docs/getting-started/updating.md | 2 +- 4 files changed, 286 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_backup_all_profiles.py diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py index 7742b0540a..30e3b6cce3 100644 --- a/hermes_cli/backup.py +++ b/hermes_cli/backup.py @@ -1776,6 +1776,104 @@ def restore_cron_jobs_if_emptied( return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id} +def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]: + """(name, home) for every OTHER profile on this install. Never raises. + + The update's code swap and gateway fleet restart touch every profile, + so the pre-update snapshot must too (#66140). The invoking profile is + excluded — its snapshot is taken by the existing call. + """ + homes: list[tuple[str, Path]] = [] + try: + from hermes_cli.profiles import ( + _get_default_hermes_home, + _get_profiles_root, + _PROFILE_ID_RE, + ) + + invoking = invoking_home.resolve() + default_home = _get_default_hermes_home() + if default_home.is_dir() and default_home.resolve() != invoking: + homes.append(("default", default_home)) + root = _get_profiles_root() + if root.is_dir(): + for entry in sorted(root.iterdir()): + if ( + entry.is_dir() + and entry.name != "default" + and _PROFILE_ID_RE.match(entry.name) + and entry.resolve() != invoking + ): + homes.append((entry.name, entry)) + except Exception as exc: + logger.debug("Sibling profile enumeration failed: %s", exc) + return homes + + +def create_pre_update_snapshots_all_profiles( + invoking_home: Optional[Path] = None, + keep: Optional[int] = None, + max_file_size: Optional[int] = None, +) -> Dict[str, str]: + """Pre-update quick snapshots for every SIBLING profile (#66140). + + Same snapshot set, same per-file size cap, same keep policy as the + invoking profile's snapshot — identical semantics per profile, no + partial-tier coherence class. Each sibling's snapshot lands under its + OWN ``/state-snapshots/`` so per-profile restore tooling finds + it where it expects. Returns ``{profile_name: snapshot_id}`` for the + siblings that snapshotted successfully. Never raises. + """ + results: Dict[str, str] = {} + home = invoking_home or get_hermes_home() + for name, profile_home in _sibling_profile_homes(home): + try: + snap_id = create_quick_snapshot( + label="pre-update", + hermes_home=profile_home, + keep=keep, + max_file_size=max_file_size, + ) + if snap_id: + results[name] = snap_id + except Exception as exc: + logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc) + return results + + +def restore_cron_jobs_all_profiles( + profile_snapshots: Dict[str, str], + invoking_home: Optional[Path] = None, +) -> list[Dict[str, Any]]: + """Run the cron-jobs safety net for every sibling profile (#66140). + + ``profile_snapshots`` is the map returned by + :func:`create_pre_update_snapshots_all_profiles`. Each profile's live + ``cron/jobs.json`` is compared against ITS OWN snapshot — restores are + same-generation by construction (the snapshot was taken minutes ago by + this update run). Returns one result dict per restored profile, each + with a ``profile`` key added. Never raises. + """ + restored: list[Dict[str, Any]] = [] + if not profile_snapshots: + return restored + home = invoking_home or get_hermes_home() + by_name = dict(_sibling_profile_homes(home)) + for name, snap_id in profile_snapshots.items(): + profile_home = by_name.get(name) + if profile_home is None: + continue + try: + result = restore_cron_jobs_if_emptied(snap_id, hermes_home=profile_home) + except Exception as exc: + logger.debug("Cron restore check for profile %s failed: %s", name, exc) + continue + if result: + result["profile"] = name + restored.append(result) + return restored + + def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int: """Remove oldest quick snapshots beyond the keep limit. Returns count deleted.""" if not root.exists(): diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index f41a37fb67..f7a0239298 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -3230,6 +3230,11 @@ def _ensure_acp_launcher() -> None: print(f" ✓ Installed hermes-acp launcher → {acp_cmd}") _PRE_UPDATE_SNAPSHOT_KEEP = 1 +# Sibling-profile snapshot ids from the current run's pre-update backup +# ({profile: snapshot_id}) — consumed by the post-update per-profile +# cron-jobs safety net (#66140). Module-level because the snapshot and the +# restore run in the same process but far apart in _cmd_update_impl. +_LAST_SIBLING_SNAPSHOTS: dict = {} # Per-file size cap for the pre-update quick snapshot. Anything larger is # skipped with a warning: the snapshot exists to protect small, hard-to- @@ -3379,6 +3384,40 @@ def _run_pre_update_backup(args) -> Optional[str]: print() if snapshot_id: print(f"◆ Pre-update snapshot: {snapshot_id}") + + # #66140: the code swap + fleet restart touch EVERY profile, so + # every profile gets the same snapshot (same set, same 1GiB cap, + # keep=1) under its own state-snapshots/. Best-effort per profile. + try: + from hermes_cli.backup import create_pre_update_snapshots_all_profiles + + _sibling_snaps = create_pre_update_snapshots_all_profiles( + keep=_PRE_UPDATE_SNAPSHOT_KEEP, + max_file_size=_PRE_UPDATE_SNAPSHOT_MAX_FILE_SIZE, + ) + if _sibling_snaps: + print( + f"◆ Sibling profile snapshot(s): " + + ", ".join(sorted(_sibling_snaps)) + ) + try: + from hermes_cli.update_receipt import record_step + + record_step( + "sibling_profile_snapshots", + True, + ", ".join( + f"{k}={v}" for k, v in sorted(_sibling_snaps.items()) + ), + ) + except Exception: + pass + global _LAST_SIBLING_SNAPSHOTS + _LAST_SIBLING_SNAPSHOTS = _sibling_snaps + except Exception as _sib_exc: + logging.getLogger(__name__).debug( + "Sibling profile snapshots failed: %s", _sib_exc + ) except Exception as exc: # Never let a snapshot failure block an update. logging.getLogger(__name__).debug("Pre-update snapshot failed: %s", exc) @@ -6435,6 +6474,25 @@ def _cmd_update_impl(args, gateway_mode: bool): # Never let the cron safety net break an otherwise-good update. logger.debug("Cron jobs auto-restore check failed: %s", exc) + # #66140: run the same cron-jobs safety net for every sibling + # profile against ITS OWN pre-update snapshot (same-generation by + # construction — both taken by this run). + try: + from hermes_cli.backup import restore_cron_jobs_all_profiles + + for _restored in restore_cron_jobs_all_profiles( + _LAST_SIBLING_SNAPSHOTS + ): + print() + print( + f" ⚠️ Profile '{_restored['profile']}': cron/jobs.json " + f"lost jobs during this update — restored " + f"{_restored['job_count']} job(s) from pre-update " + f"snapshot {_restored['snapshot_id']}." + ) + except Exception as exc: + logger.debug("Sibling cron auto-restore check failed: %s", exc) + _print_update_summary( node_failures=node_failures, desktop_build_ok=desktop_build_ok, diff --git a/tests/hermes_cli/test_backup_all_profiles.py b/tests/hermes_cli/test_backup_all_profiles.py new file mode 100644 index 0000000000..aaebe442cd --- /dev/null +++ b/tests/hermes_cli/test_backup_all_profiles.py @@ -0,0 +1,129 @@ +"""Tests for the #66140 fix: pre-update snapshots cover every profile.""" + +import json +import re +from pathlib import Path + +import pytest + +import hermes_cli.backup as backup + + +def _mk_profile(home: Path, jobs: int = 0) -> Path: + home.mkdir(parents=True, exist_ok=True) + (home / "config.yaml").write_text("model: {}\n", encoding="utf-8") + if jobs: + cron = home / "cron" + cron.mkdir(exist_ok=True) + payload = {"jobs": [{"id": f"j{i}"} for i in range(jobs)]} + (cron / "jobs.json").write_text(json.dumps(payload), encoding="utf-8") + return home + + +@pytest.fixture() +def profiles(monkeypatch, tmp_path): + """default (invoking) + work + sparks profile homes.""" + default_home = _mk_profile(tmp_path / "home", jobs=3) + work = _mk_profile(tmp_path / "home" / "profiles" / "work", jobs=5) + sparks = _mk_profile(tmp_path / "home" / "profiles" / "sparks", jobs=0) + monkeypatch.setattr( + "hermes_cli.profiles._get_default_hermes_home", lambda: default_home + ) + monkeypatch.setattr( + "hermes_cli.profiles._get_profiles_root", lambda: tmp_path / "home" / "profiles" + ) + monkeypatch.setattr( + "hermes_cli.profiles._PROFILE_ID_RE", + re.compile(r"^[a-z0-9][a-z0-9_-]*$"), + raising=False, + ) + return {"default": default_home, "work": work, "sparks": sparks} + + +class TestSiblingEnumeration: + def test_excludes_invoking_profile(self, profiles): + names = [n for n, _ in backup._sibling_profile_homes(profiles["default"])] + assert names == ["sparks", "work"] + + def test_invoked_from_named_profile_includes_default(self, profiles): + names = [n for n, _ in backup._sibling_profile_homes(profiles["work"])] + assert names == ["default", "sparks"] + + def test_never_raises(self, monkeypatch, tmp_path): + def _boom(): + raise RuntimeError("no profiles module") + + monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", _boom) + assert backup._sibling_profile_homes(tmp_path) == [] + + +class TestAllProfileSnapshots: + def test_each_sibling_snapshotted_into_own_home(self, profiles): + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + assert set(result) == {"work", "sparks"} + for name, snap_id in result.items(): + snap_dir = profiles[name] / "state-snapshots" / snap_id + assert snap_dir.is_dir() + assert (snap_dir / "config.yaml").is_file() + assert "pre-update" in snap_id + # invoking profile untouched by THIS call + assert not (profiles["default"] / "state-snapshots").exists() + + def test_size_cap_forwarded(self, profiles): + big = profiles["work"] / "state.db" + big.write_bytes(b"\x00" * 4096) + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1, max_file_size=1024 + ) + snap_dir = profiles["work"] / "state-snapshots" / result["work"] + assert not (snap_dir / "state.db").exists() # capped out + assert (snap_dir / "config.yaml").is_file() # small files captured + + def test_one_failing_sibling_does_not_block_others(self, profiles, monkeypatch): + real = backup.create_quick_snapshot + + def _flaky(label=None, hermes_home=None, keep=None, max_file_size=None): + if hermes_home == profiles["work"]: + raise OSError("disk full") + return real( + label=label, hermes_home=hermes_home, keep=keep, + max_file_size=max_file_size, + ) + + monkeypatch.setattr(backup, "create_quick_snapshot", _flaky) + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"] + ) + assert "sparks" in result and "work" not in result + + +class TestPerProfileCronRestore: + def test_lost_jobs_restored_from_own_snapshot(self, profiles): + snaps = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + # simulate the migration emptying work's jobs.json + jobs_path = profiles["work"] / "cron" / "jobs.json" + jobs_path.write_text(json.dumps({"jobs": []}), encoding="utf-8") + + restored = backup.restore_cron_jobs_all_profiles( + snaps, invoking_home=profiles["default"] + ) + assert len(restored) == 1 + assert restored[0]["profile"] == "work" + assert restored[0]["job_count"] == 5 + live = json.loads(jobs_path.read_text(encoding="utf-8")) + assert len(live["jobs"]) == 5 + + def test_healthy_profiles_untouched(self, profiles): + snaps = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + assert backup.restore_cron_jobs_all_profiles( + snaps, invoking_home=profiles["default"] + ) == [] + + def test_empty_map_is_noop(self, profiles): + assert backup.restore_cron_jobs_all_profiles({}) == [] diff --git a/website/docs/getting-started/updating.md b/website/docs/getting-started/updating.md index aa8e9d9f58..0353e9540e 100644 --- a/website/docs/getting-started/updating.md +++ b/website/docs/getting-started/updating.md @@ -24,7 +24,7 @@ This pulls the latest code from `main`, updates dependencies, and prompts you to When you run `hermes update`, the following steps occur: -1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). +1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Because the code swap and gateway restarts touch every profile, the same snapshot is taken for **every profile** on the install — each into its own `state-snapshots/` directory — and the post-update cron-jobs safety net checks each profile against its own snapshot. Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). Quick snapshots are file-loss recovery, not code-rollback insurance — for a coherent point-in-time rollback use `--backup` (full mode). 2. **Git pull** — pulls the latest code from the `main` branch and updates submodules 3. **Post-pull syntax validation + auto-rollback** — after the pull, Hermes compiles the nine critical files every `hermes` invocation imports at startup. If any fails to parse (e.g. an orphan merge-conflict marker, an accidentally truncated file), Hermes runs `git reset --hard ` to roll the install back so your shell stays bootable. Re-run `hermes update` once the upstream fix lands. 4. **Dependency install** — runs `uv pip install -e ".[all]"` to pick up new or changed dependencies