fix(update): pre-update snapshots now cover every profile, not just the invoking one (#66140)

The code swap and gateway fleet restart touch all profiles, but the
pre-update quick snapshot photographed only the invoking profile's home
— siblings had no snapshot for the post-update safety nets or manual
restore to draw on.

- backup.py: create_pre_update_snapshots_all_profiles() — the SAME
  snapshot set, per-file 1GiB cap, and keep policy as the invoking
  profile (no partial tier, no new restore-coherence class), each into
  the sibling's own state-snapshots/; restore_cron_jobs_all_profiles()
  runs the #34600 cron-loss safety net per profile against its OWN
  snapshot (same-generation by construction).
- update_cmd.py: sibling snapshots taken right after the invoking
  profile's (best-effort, receipt-recorded); post-update cron restore
  extended to every sibling.
- Docs: updating.md pre-update snapshot step now states the per-profile
  behavior and the file-loss-recovery vs rollback contract.
- 9 unit tests + E2E (real files: sibling snapshot on disk, clobbered
  jobs.json restored 7/7 from the sibling's own snapshot, keep=1 prune).
This commit is contained in:
Teknium
2026-08-21 04:55:50 -07:00
parent b96a9b7408
commit 1575116629
4 changed files with 286 additions and 1 deletions
+98
View File
@@ -1776,6 +1776,104 @@ def restore_cron_jobs_if_emptied(
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
"""(name, home) for every OTHER profile on this install. Never raises.
The update's code swap and gateway fleet restart touch every profile,
so the pre-update snapshot must too (#66140). The invoking profile is
excluded — its snapshot is taken by the existing call.
"""
homes: list[tuple[str, Path]] = []
try:
from hermes_cli.profiles import (
_get_default_hermes_home,
_get_profiles_root,
_PROFILE_ID_RE,
)
invoking = invoking_home.resolve()
default_home = _get_default_hermes_home()
if default_home.is_dir() and default_home.resolve() != invoking:
homes.append(("default", default_home))
root = _get_profiles_root()
if root.is_dir():
for entry in sorted(root.iterdir()):
if (
entry.is_dir()
and entry.name != "default"
and _PROFILE_ID_RE.match(entry.name)
and entry.resolve() != invoking
):
homes.append((entry.name, entry))
except Exception as exc:
logger.debug("Sibling profile enumeration failed: %s", exc)
return homes
def create_pre_update_snapshots_all_profiles(
invoking_home: Optional[Path] = None,
keep: Optional[int] = None,
max_file_size: Optional[int] = None,
) -> Dict[str, str]:
"""Pre-update quick snapshots for every SIBLING profile (#66140).
Same snapshot set, same per-file size cap, same keep policy as the
invoking profile's snapshot — identical semantics per profile, no
partial-tier coherence class. Each sibling's snapshot lands under its
OWN ``<home>/state-snapshots/`` so per-profile restore tooling finds
it where it expects. Returns ``{profile_name: snapshot_id}`` for the
siblings that snapshotted successfully. Never raises.
"""
results: Dict[str, str] = {}
home = invoking_home or get_hermes_home()
for name, profile_home in _sibling_profile_homes(home):
try:
snap_id = create_quick_snapshot(
label="pre-update",
hermes_home=profile_home,
keep=keep,
max_file_size=max_file_size,
)
if snap_id:
results[name] = snap_id
except Exception as exc:
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
return results
def restore_cron_jobs_all_profiles(
profile_snapshots: Dict[str, str],
invoking_home: Optional[Path] = None,
) -> list[Dict[str, Any]]:
"""Run the cron-jobs safety net for every sibling profile (#66140).
``profile_snapshots`` is the map returned by
:func:`create_pre_update_snapshots_all_profiles`. Each profile's live
``cron/jobs.json`` is compared against ITS OWN snapshot — restores are
same-generation by construction (the snapshot was taken minutes ago by
this update run). Returns one result dict per restored profile, each
with a ``profile`` key added. Never raises.
"""
restored: list[Dict[str, Any]] = []
if not profile_snapshots:
return restored
home = invoking_home or get_hermes_home()
by_name = dict(_sibling_profile_homes(home))
for name, snap_id in profile_snapshots.items():
profile_home = by_name.get(name)
if profile_home is None:
continue
try:
result = restore_cron_jobs_if_emptied(snap_id, hermes_home=profile_home)
except Exception as exc:
logger.debug("Cron restore check for profile %s failed: %s", name, exc)
continue
if result:
result["profile"] = name
restored.append(result)
return restored
def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int:
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
if not root.exists():
+58
View File
@@ -3230,6 +3230,11 @@ def _ensure_acp_launcher() -> None:
print(f" ✓ Installed hermes-acp launcher → {acp_cmd}")
_PRE_UPDATE_SNAPSHOT_KEEP = 1
# Sibling-profile snapshot ids from the current run's pre-update backup
# ({profile: snapshot_id}) — consumed by the post-update per-profile
# cron-jobs safety net (#66140). Module-level because the snapshot and the
# restore run in the same process but far apart in _cmd_update_impl.
_LAST_SIBLING_SNAPSHOTS: dict = {}
# Per-file size cap for the pre-update quick snapshot. Anything larger is
# skipped with a warning: the snapshot exists to protect small, hard-to-
@@ -3379,6 +3384,40 @@ def _run_pre_update_backup(args) -> Optional[str]:
print()
if snapshot_id:
print(f"◆ Pre-update snapshot: {snapshot_id}")
# #66140: the code swap + fleet restart touch EVERY profile, so
# every profile gets the same snapshot (same set, same 1GiB cap,
# keep=1) under its own state-snapshots/. Best-effort per profile.
try:
from hermes_cli.backup import create_pre_update_snapshots_all_profiles
_sibling_snaps = create_pre_update_snapshots_all_profiles(
keep=_PRE_UPDATE_SNAPSHOT_KEEP,
max_file_size=_PRE_UPDATE_SNAPSHOT_MAX_FILE_SIZE,
)
if _sibling_snaps:
print(
f"◆ Sibling profile snapshot(s): "
+ ", ".join(sorted(_sibling_snaps))
)
try:
from hermes_cli.update_receipt import record_step
record_step(
"sibling_profile_snapshots",
True,
", ".join(
f"{k}={v}" for k, v in sorted(_sibling_snaps.items())
),
)
except Exception:
pass
global _LAST_SIBLING_SNAPSHOTS
_LAST_SIBLING_SNAPSHOTS = _sibling_snaps
except Exception as _sib_exc:
logging.getLogger(__name__).debug(
"Sibling profile snapshots failed: %s", _sib_exc
)
except Exception as exc:
# Never let a snapshot failure block an update.
logging.getLogger(__name__).debug("Pre-update snapshot failed: %s", exc)
@@ -6435,6 +6474,25 @@ def _cmd_update_impl(args, gateway_mode: bool):
# Never let the cron safety net break an otherwise-good update.
logger.debug("Cron jobs auto-restore check failed: %s", exc)
# #66140: run the same cron-jobs safety net for every sibling
# profile against ITS OWN pre-update snapshot (same-generation by
# construction — both taken by this run).
try:
from hermes_cli.backup import restore_cron_jobs_all_profiles
for _restored in restore_cron_jobs_all_profiles(
_LAST_SIBLING_SNAPSHOTS
):
print()
print(
f" ⚠️ Profile '{_restored['profile']}': cron/jobs.json "
f"lost jobs during this update — restored "
f"{_restored['job_count']} job(s) from pre-update "
f"snapshot {_restored['snapshot_id']}."
)
except Exception as exc:
logger.debug("Sibling cron auto-restore check failed: %s", exc)
_print_update_summary(
node_failures=node_failures,
desktop_build_ok=desktop_build_ok,
@@ -0,0 +1,129 @@
"""Tests for the #66140 fix: pre-update snapshots cover every profile."""
import json
import re
from pathlib import Path
import pytest
import hermes_cli.backup as backup
def _mk_profile(home: Path, jobs: int = 0) -> Path:
home.mkdir(parents=True, exist_ok=True)
(home / "config.yaml").write_text("model: {}\n", encoding="utf-8")
if jobs:
cron = home / "cron"
cron.mkdir(exist_ok=True)
payload = {"jobs": [{"id": f"j{i}"} for i in range(jobs)]}
(cron / "jobs.json").write_text(json.dumps(payload), encoding="utf-8")
return home
@pytest.fixture()
def profiles(monkeypatch, tmp_path):
"""default (invoking) + work + sparks profile homes."""
default_home = _mk_profile(tmp_path / "home", jobs=3)
work = _mk_profile(tmp_path / "home" / "profiles" / "work", jobs=5)
sparks = _mk_profile(tmp_path / "home" / "profiles" / "sparks", jobs=0)
monkeypatch.setattr(
"hermes_cli.profiles._get_default_hermes_home", lambda: default_home
)
monkeypatch.setattr(
"hermes_cli.profiles._get_profiles_root", lambda: tmp_path / "home" / "profiles"
)
monkeypatch.setattr(
"hermes_cli.profiles._PROFILE_ID_RE",
re.compile(r"^[a-z0-9][a-z0-9_-]*$"),
raising=False,
)
return {"default": default_home, "work": work, "sparks": sparks}
class TestSiblingEnumeration:
def test_excludes_invoking_profile(self, profiles):
names = [n for n, _ in backup._sibling_profile_homes(profiles["default"])]
assert names == ["sparks", "work"]
def test_invoked_from_named_profile_includes_default(self, profiles):
names = [n for n, _ in backup._sibling_profile_homes(profiles["work"])]
assert names == ["default", "sparks"]
def test_never_raises(self, monkeypatch, tmp_path):
def _boom():
raise RuntimeError("no profiles module")
monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", _boom)
assert backup._sibling_profile_homes(tmp_path) == []
class TestAllProfileSnapshots:
def test_each_sibling_snapshotted_into_own_home(self, profiles):
result = backup.create_pre_update_snapshots_all_profiles(
invoking_home=profiles["default"], keep=1
)
assert set(result) == {"work", "sparks"}
for name, snap_id in result.items():
snap_dir = profiles[name] / "state-snapshots" / snap_id
assert snap_dir.is_dir()
assert (snap_dir / "config.yaml").is_file()
assert "pre-update" in snap_id
# invoking profile untouched by THIS call
assert not (profiles["default"] / "state-snapshots").exists()
def test_size_cap_forwarded(self, profiles):
big = profiles["work"] / "state.db"
big.write_bytes(b"\x00" * 4096)
result = backup.create_pre_update_snapshots_all_profiles(
invoking_home=profiles["default"], keep=1, max_file_size=1024
)
snap_dir = profiles["work"] / "state-snapshots" / result["work"]
assert not (snap_dir / "state.db").exists() # capped out
assert (snap_dir / "config.yaml").is_file() # small files captured
def test_one_failing_sibling_does_not_block_others(self, profiles, monkeypatch):
real = backup.create_quick_snapshot
def _flaky(label=None, hermes_home=None, keep=None, max_file_size=None):
if hermes_home == profiles["work"]:
raise OSError("disk full")
return real(
label=label, hermes_home=hermes_home, keep=keep,
max_file_size=max_file_size,
)
monkeypatch.setattr(backup, "create_quick_snapshot", _flaky)
result = backup.create_pre_update_snapshots_all_profiles(
invoking_home=profiles["default"]
)
assert "sparks" in result and "work" not in result
class TestPerProfileCronRestore:
def test_lost_jobs_restored_from_own_snapshot(self, profiles):
snaps = backup.create_pre_update_snapshots_all_profiles(
invoking_home=profiles["default"], keep=1
)
# simulate the migration emptying work's jobs.json
jobs_path = profiles["work"] / "cron" / "jobs.json"
jobs_path.write_text(json.dumps({"jobs": []}), encoding="utf-8")
restored = backup.restore_cron_jobs_all_profiles(
snaps, invoking_home=profiles["default"]
)
assert len(restored) == 1
assert restored[0]["profile"] == "work"
assert restored[0]["job_count"] == 5
live = json.loads(jobs_path.read_text(encoding="utf-8"))
assert len(live["jobs"]) == 5
def test_healthy_profiles_untouched(self, profiles):
snaps = backup.create_pre_update_snapshots_all_profiles(
invoking_home=profiles["default"], keep=1
)
assert backup.restore_cron_jobs_all_profiles(
snaps, invoking_home=profiles["default"]
) == []
def test_empty_map_is_noop(self, profiles):
assert backup.restore_cron_jobs_all_profiles({}) == []
+1 -1
View File
@@ -24,7 +24,7 @@ This pulls the latest code from `main`, updates dependencies, and prompts you to
When you run `hermes update`, the following steps occur:
1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md).
1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Because the code swap and gateway restarts touch every profile, the same snapshot is taken for **every profile** on the install — each into its own `state-snapshots/` directory — and the post-update cron-jobs safety net checks each profile against its own snapshot. Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). Quick snapshots are file-loss recovery, not code-rollback insurance — for a coherent point-in-time rollback use `--backup` (full mode).
2. **Git pull** — pulls the latest code from the `main` branch and updates submodules
3. **Post-pull syntax validation + auto-rollback** — after the pull, Hermes compiles the nine critical files every `hermes` invocation imports at startup. If any fails to parse (e.g. an orphan merge-conflict marker, an accidentally truncated file), Hermes runs `git reset --hard <pre-pull-sha>` to roll the install back so your shell stays bootable. Re-run `hermes update` once the upstream fix lands.
4. **Dependency install** — runs `uv pip install -e ".[all]"` to pick up new or changed dependencies