fix(update): pre-update snapshots now cover every profile, not just the invoking one (#66140)
The code swap and gateway fleet restart touch all profiles, but the pre-update quick snapshot photographed only the invoking profile's home — siblings had no snapshot for the post-update safety nets or manual restore to draw on. - backup.py: create_pre_update_snapshots_all_profiles() — the SAME snapshot set, per-file 1GiB cap, and keep policy as the invoking profile (no partial tier, no new restore-coherence class), each into the sibling's own state-snapshots/; restore_cron_jobs_all_profiles() runs the #34600 cron-loss safety net per profile against its OWN snapshot (same-generation by construction). - update_cmd.py: sibling snapshots taken right after the invoking profile's (best-effort, receipt-recorded); post-update cron restore extended to every sibling. - Docs: updating.md pre-update snapshot step now states the per-profile behavior and the file-loss-recovery vs rollback contract. - 9 unit tests + E2E (real files: sibling snapshot on disk, clobbered jobs.json restored 7/7 from the sibling's own snapshot, keep=1 prune).
This commit is contained in:
@@ -1776,6 +1776,104 @@ def restore_cron_jobs_if_emptied(
|
||||
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
|
||||
|
||||
|
||||
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
|
||||
"""(name, home) for every OTHER profile on this install. Never raises.
|
||||
|
||||
The update's code swap and gateway fleet restart touch every profile,
|
||||
so the pre-update snapshot must too (#66140). The invoking profile is
|
||||
excluded — its snapshot is taken by the existing call.
|
||||
"""
|
||||
homes: list[tuple[str, Path]] = []
|
||||
try:
|
||||
from hermes_cli.profiles import (
|
||||
_get_default_hermes_home,
|
||||
_get_profiles_root,
|
||||
_PROFILE_ID_RE,
|
||||
)
|
||||
|
||||
invoking = invoking_home.resolve()
|
||||
default_home = _get_default_hermes_home()
|
||||
if default_home.is_dir() and default_home.resolve() != invoking:
|
||||
homes.append(("default", default_home))
|
||||
root = _get_profiles_root()
|
||||
if root.is_dir():
|
||||
for entry in sorted(root.iterdir()):
|
||||
if (
|
||||
entry.is_dir()
|
||||
and entry.name != "default"
|
||||
and _PROFILE_ID_RE.match(entry.name)
|
||||
and entry.resolve() != invoking
|
||||
):
|
||||
homes.append((entry.name, entry))
|
||||
except Exception as exc:
|
||||
logger.debug("Sibling profile enumeration failed: %s", exc)
|
||||
return homes
|
||||
|
||||
|
||||
def create_pre_update_snapshots_all_profiles(
|
||||
invoking_home: Optional[Path] = None,
|
||||
keep: Optional[int] = None,
|
||||
max_file_size: Optional[int] = None,
|
||||
) -> Dict[str, str]:
|
||||
"""Pre-update quick snapshots for every SIBLING profile (#66140).
|
||||
|
||||
Same snapshot set, same per-file size cap, same keep policy as the
|
||||
invoking profile's snapshot — identical semantics per profile, no
|
||||
partial-tier coherence class. Each sibling's snapshot lands under its
|
||||
OWN ``<home>/state-snapshots/`` so per-profile restore tooling finds
|
||||
it where it expects. Returns ``{profile_name: snapshot_id}`` for the
|
||||
siblings that snapshotted successfully. Never raises.
|
||||
"""
|
||||
results: Dict[str, str] = {}
|
||||
home = invoking_home or get_hermes_home()
|
||||
for name, profile_home in _sibling_profile_homes(home):
|
||||
try:
|
||||
snap_id = create_quick_snapshot(
|
||||
label="pre-update",
|
||||
hermes_home=profile_home,
|
||||
keep=keep,
|
||||
max_file_size=max_file_size,
|
||||
)
|
||||
if snap_id:
|
||||
results[name] = snap_id
|
||||
except Exception as exc:
|
||||
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
|
||||
return results
|
||||
|
||||
|
||||
def restore_cron_jobs_all_profiles(
|
||||
profile_snapshots: Dict[str, str],
|
||||
invoking_home: Optional[Path] = None,
|
||||
) -> list[Dict[str, Any]]:
|
||||
"""Run the cron-jobs safety net for every sibling profile (#66140).
|
||||
|
||||
``profile_snapshots`` is the map returned by
|
||||
:func:`create_pre_update_snapshots_all_profiles`. Each profile's live
|
||||
``cron/jobs.json`` is compared against ITS OWN snapshot — restores are
|
||||
same-generation by construction (the snapshot was taken minutes ago by
|
||||
this update run). Returns one result dict per restored profile, each
|
||||
with a ``profile`` key added. Never raises.
|
||||
"""
|
||||
restored: list[Dict[str, Any]] = []
|
||||
if not profile_snapshots:
|
||||
return restored
|
||||
home = invoking_home or get_hermes_home()
|
||||
by_name = dict(_sibling_profile_homes(home))
|
||||
for name, snap_id in profile_snapshots.items():
|
||||
profile_home = by_name.get(name)
|
||||
if profile_home is None:
|
||||
continue
|
||||
try:
|
||||
result = restore_cron_jobs_if_emptied(snap_id, hermes_home=profile_home)
|
||||
except Exception as exc:
|
||||
logger.debug("Cron restore check for profile %s failed: %s", name, exc)
|
||||
continue
|
||||
if result:
|
||||
result["profile"] = name
|
||||
restored.append(result)
|
||||
return restored
|
||||
|
||||
|
||||
def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int:
|
||||
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
|
||||
if not root.exists():
|
||||
|
||||
@@ -3230,6 +3230,11 @@ def _ensure_acp_launcher() -> None:
|
||||
print(f" ✓ Installed hermes-acp launcher → {acp_cmd}")
|
||||
|
||||
_PRE_UPDATE_SNAPSHOT_KEEP = 1
|
||||
# Sibling-profile snapshot ids from the current run's pre-update backup
|
||||
# ({profile: snapshot_id}) — consumed by the post-update per-profile
|
||||
# cron-jobs safety net (#66140). Module-level because the snapshot and the
|
||||
# restore run in the same process but far apart in _cmd_update_impl.
|
||||
_LAST_SIBLING_SNAPSHOTS: dict = {}
|
||||
|
||||
# Per-file size cap for the pre-update quick snapshot. Anything larger is
|
||||
# skipped with a warning: the snapshot exists to protect small, hard-to-
|
||||
@@ -3379,6 +3384,40 @@ def _run_pre_update_backup(args) -> Optional[str]:
|
||||
print()
|
||||
if snapshot_id:
|
||||
print(f"◆ Pre-update snapshot: {snapshot_id}")
|
||||
|
||||
# #66140: the code swap + fleet restart touch EVERY profile, so
|
||||
# every profile gets the same snapshot (same set, same 1GiB cap,
|
||||
# keep=1) under its own state-snapshots/. Best-effort per profile.
|
||||
try:
|
||||
from hermes_cli.backup import create_pre_update_snapshots_all_profiles
|
||||
|
||||
_sibling_snaps = create_pre_update_snapshots_all_profiles(
|
||||
keep=_PRE_UPDATE_SNAPSHOT_KEEP,
|
||||
max_file_size=_PRE_UPDATE_SNAPSHOT_MAX_FILE_SIZE,
|
||||
)
|
||||
if _sibling_snaps:
|
||||
print(
|
||||
f"◆ Sibling profile snapshot(s): "
|
||||
+ ", ".join(sorted(_sibling_snaps))
|
||||
)
|
||||
try:
|
||||
from hermes_cli.update_receipt import record_step
|
||||
|
||||
record_step(
|
||||
"sibling_profile_snapshots",
|
||||
True,
|
||||
", ".join(
|
||||
f"{k}={v}" for k, v in sorted(_sibling_snaps.items())
|
||||
),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
global _LAST_SIBLING_SNAPSHOTS
|
||||
_LAST_SIBLING_SNAPSHOTS = _sibling_snaps
|
||||
except Exception as _sib_exc:
|
||||
logging.getLogger(__name__).debug(
|
||||
"Sibling profile snapshots failed: %s", _sib_exc
|
||||
)
|
||||
except Exception as exc:
|
||||
# Never let a snapshot failure block an update.
|
||||
logging.getLogger(__name__).debug("Pre-update snapshot failed: %s", exc)
|
||||
@@ -6435,6 +6474,25 @@ def _cmd_update_impl(args, gateway_mode: bool):
|
||||
# Never let the cron safety net break an otherwise-good update.
|
||||
logger.debug("Cron jobs auto-restore check failed: %s", exc)
|
||||
|
||||
# #66140: run the same cron-jobs safety net for every sibling
|
||||
# profile against ITS OWN pre-update snapshot (same-generation by
|
||||
# construction — both taken by this run).
|
||||
try:
|
||||
from hermes_cli.backup import restore_cron_jobs_all_profiles
|
||||
|
||||
for _restored in restore_cron_jobs_all_profiles(
|
||||
_LAST_SIBLING_SNAPSHOTS
|
||||
):
|
||||
print()
|
||||
print(
|
||||
f" ⚠️ Profile '{_restored['profile']}': cron/jobs.json "
|
||||
f"lost jobs during this update — restored "
|
||||
f"{_restored['job_count']} job(s) from pre-update "
|
||||
f"snapshot {_restored['snapshot_id']}."
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Sibling cron auto-restore check failed: %s", exc)
|
||||
|
||||
_print_update_summary(
|
||||
node_failures=node_failures,
|
||||
desktop_build_ok=desktop_build_ok,
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Tests for the #66140 fix: pre-update snapshots cover every profile."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.backup as backup
|
||||
|
||||
|
||||
def _mk_profile(home: Path, jobs: int = 0) -> Path:
|
||||
home.mkdir(parents=True, exist_ok=True)
|
||||
(home / "config.yaml").write_text("model: {}\n", encoding="utf-8")
|
||||
if jobs:
|
||||
cron = home / "cron"
|
||||
cron.mkdir(exist_ok=True)
|
||||
payload = {"jobs": [{"id": f"j{i}"} for i in range(jobs)]}
|
||||
(cron / "jobs.json").write_text(json.dumps(payload), encoding="utf-8")
|
||||
return home
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def profiles(monkeypatch, tmp_path):
|
||||
"""default (invoking) + work + sparks profile homes."""
|
||||
default_home = _mk_profile(tmp_path / "home", jobs=3)
|
||||
work = _mk_profile(tmp_path / "home" / "profiles" / "work", jobs=5)
|
||||
sparks = _mk_profile(tmp_path / "home" / "profiles" / "sparks", jobs=0)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.profiles._get_default_hermes_home", lambda: default_home
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.profiles._get_profiles_root", lambda: tmp_path / "home" / "profiles"
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.profiles._PROFILE_ID_RE",
|
||||
re.compile(r"^[a-z0-9][a-z0-9_-]*$"),
|
||||
raising=False,
|
||||
)
|
||||
return {"default": default_home, "work": work, "sparks": sparks}
|
||||
|
||||
|
||||
class TestSiblingEnumeration:
|
||||
def test_excludes_invoking_profile(self, profiles):
|
||||
names = [n for n, _ in backup._sibling_profile_homes(profiles["default"])]
|
||||
assert names == ["sparks", "work"]
|
||||
|
||||
def test_invoked_from_named_profile_includes_default(self, profiles):
|
||||
names = [n for n, _ in backup._sibling_profile_homes(profiles["work"])]
|
||||
assert names == ["default", "sparks"]
|
||||
|
||||
def test_never_raises(self, monkeypatch, tmp_path):
|
||||
def _boom():
|
||||
raise RuntimeError("no profiles module")
|
||||
|
||||
monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", _boom)
|
||||
assert backup._sibling_profile_homes(tmp_path) == []
|
||||
|
||||
|
||||
class TestAllProfileSnapshots:
|
||||
def test_each_sibling_snapshotted_into_own_home(self, profiles):
|
||||
result = backup.create_pre_update_snapshots_all_profiles(
|
||||
invoking_home=profiles["default"], keep=1
|
||||
)
|
||||
assert set(result) == {"work", "sparks"}
|
||||
for name, snap_id in result.items():
|
||||
snap_dir = profiles[name] / "state-snapshots" / snap_id
|
||||
assert snap_dir.is_dir()
|
||||
assert (snap_dir / "config.yaml").is_file()
|
||||
assert "pre-update" in snap_id
|
||||
# invoking profile untouched by THIS call
|
||||
assert not (profiles["default"] / "state-snapshots").exists()
|
||||
|
||||
def test_size_cap_forwarded(self, profiles):
|
||||
big = profiles["work"] / "state.db"
|
||||
big.write_bytes(b"\x00" * 4096)
|
||||
result = backup.create_pre_update_snapshots_all_profiles(
|
||||
invoking_home=profiles["default"], keep=1, max_file_size=1024
|
||||
)
|
||||
snap_dir = profiles["work"] / "state-snapshots" / result["work"]
|
||||
assert not (snap_dir / "state.db").exists() # capped out
|
||||
assert (snap_dir / "config.yaml").is_file() # small files captured
|
||||
|
||||
def test_one_failing_sibling_does_not_block_others(self, profiles, monkeypatch):
|
||||
real = backup.create_quick_snapshot
|
||||
|
||||
def _flaky(label=None, hermes_home=None, keep=None, max_file_size=None):
|
||||
if hermes_home == profiles["work"]:
|
||||
raise OSError("disk full")
|
||||
return real(
|
||||
label=label, hermes_home=hermes_home, keep=keep,
|
||||
max_file_size=max_file_size,
|
||||
)
|
||||
|
||||
monkeypatch.setattr(backup, "create_quick_snapshot", _flaky)
|
||||
result = backup.create_pre_update_snapshots_all_profiles(
|
||||
invoking_home=profiles["default"]
|
||||
)
|
||||
assert "sparks" in result and "work" not in result
|
||||
|
||||
|
||||
class TestPerProfileCronRestore:
|
||||
def test_lost_jobs_restored_from_own_snapshot(self, profiles):
|
||||
snaps = backup.create_pre_update_snapshots_all_profiles(
|
||||
invoking_home=profiles["default"], keep=1
|
||||
)
|
||||
# simulate the migration emptying work's jobs.json
|
||||
jobs_path = profiles["work"] / "cron" / "jobs.json"
|
||||
jobs_path.write_text(json.dumps({"jobs": []}), encoding="utf-8")
|
||||
|
||||
restored = backup.restore_cron_jobs_all_profiles(
|
||||
snaps, invoking_home=profiles["default"]
|
||||
)
|
||||
assert len(restored) == 1
|
||||
assert restored[0]["profile"] == "work"
|
||||
assert restored[0]["job_count"] == 5
|
||||
live = json.loads(jobs_path.read_text(encoding="utf-8"))
|
||||
assert len(live["jobs"]) == 5
|
||||
|
||||
def test_healthy_profiles_untouched(self, profiles):
|
||||
snaps = backup.create_pre_update_snapshots_all_profiles(
|
||||
invoking_home=profiles["default"], keep=1
|
||||
)
|
||||
assert backup.restore_cron_jobs_all_profiles(
|
||||
snaps, invoking_home=profiles["default"]
|
||||
) == []
|
||||
|
||||
def test_empty_map_is_noop(self, profiles):
|
||||
assert backup.restore_cron_jobs_all_profiles({}) == []
|
||||
@@ -24,7 +24,7 @@ This pulls the latest code from `main`, updates dependencies, and prompts you to
|
||||
|
||||
When you run `hermes update`, the following steps occur:
|
||||
|
||||
1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md).
|
||||
1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Because the code swap and gateway restarts touch every profile, the same snapshot is taken for **every profile** on the install — each into its own `state-snapshots/` directory — and the post-update cron-jobs safety net checks each profile against its own snapshot. Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). Quick snapshots are file-loss recovery, not code-rollback insurance — for a coherent point-in-time rollback use `--backup` (full mode).
|
||||
2. **Git pull** — pulls the latest code from the `main` branch and updates submodules
|
||||
3. **Post-pull syntax validation + auto-rollback** — after the pull, Hermes compiles the nine critical files every `hermes` invocation imports at startup. If any fails to parse (e.g. an orphan merge-conflict marker, an accidentally truncated file), Hermes runs `git reset --hard <pre-pull-sha>` to roll the install back so your shell stays bootable. Re-run `hermes update` once the upstream fix lands.
|
||||
4. **Dependency install** — runs `uv pip install -e ".[all]"` to pick up new or changed dependencies
|
||||
|
||||
Reference in New Issue
Block a user