fix(update): make the in-progress marker a real cross-process lock
Three surfaces start updates against one checkout: a terminal "hermes update", the dashboard's Update button (which spawns that same command detached), and the desktop's, which hands off to the Tauri updater. Only the Tauri updater published the in-progress marker, and only Electron read it -- to gate backend startup, not to stop a second updater. So a dashboard-spawned update and an installer-driven git checkout could mutate the same tree concurrently, rewriting source under a live interpreter. Claim the same marker from cmd_update rather than adding a second mechanism: same path, same pid+started_at payload the Rust and Electron readers already parse. A marker only counts as live when its pid is alive and it is inside the shared age ceiling, so a crashed updater self-heals instead of wedging every future update. Release only removes a marker we still own, leaving a handoff partner's claim intact. Refusing exits 2, matching the existing concurrent-instance contract the Tauri updater already recognizes.
This commit is contained in:
@@ -0,0 +1,216 @@
|
||||
"""Cross-process mutual exclusion for in-flight Hermes updates.
|
||||
|
||||
Three different surfaces can start an update of the same install tree:
|
||||
|
||||
* ``hermes update`` from a terminal,
|
||||
* the dashboard's Update button (``POST /api/hermes/update`` →
|
||||
``_spawn_hermes_action(["update"])``, detached),
|
||||
* the desktop's Update button, which hands off to the Tauri
|
||||
``hermes-setup --update`` and, on its failure screen, to install-mode
|
||||
bootstrap (``install.ps1`` / ``install.sh``).
|
||||
|
||||
Until now only the Tauri updater published an "update in progress" marker
|
||||
(``UpdateMarkerGuard`` in ``apps/bootstrap-installer/src-tauri/src/update.rs``),
|
||||
and only the Electron desktop consumed it (``electron/update-marker.ts``, to
|
||||
gate local backend startup). Nothing stopped two *updaters* from running at
|
||||
once — so a dashboard-spawned ``hermes update`` and an installer-driven
|
||||
``git checkout`` could mutate the same checkout concurrently, rewriting source
|
||||
under a live interpreter and leaving the tree half-updated.
|
||||
|
||||
This module makes that same marker the single lock for **all** update
|
||||
entrypoints instead of adding a fourth mechanism. Format and location are
|
||||
unchanged and remain byte-compatible with the Rust and Electron readers:
|
||||
|
||||
<HERMES_HOME>/.hermes-update-in-progress body: "<pid>\\n<started_at_unix>"
|
||||
|
||||
A marker only counts as a live update when its pid is alive AND it is younger
|
||||
than :data:`UPDATE_MARKER_MAX_AGE_MS` — mirroring ``readLiveUpdateMarker`` so a
|
||||
crashed updater self-heals instead of wedging every future update. A stale
|
||||
marker is removed on read by whoever notices it first.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Keep in sync with UPDATE_MARKER_MAX_AGE_MS in
|
||||
# apps/desktop/electron/update-marker.ts — the same marker is read by both, and
|
||||
# a shorter ceiling here would let Python steal a lock Electron still considers
|
||||
# live. A full update (git pull + uv sync + desktop rebuild) is minutes.
|
||||
UPDATE_MARKER_MAX_AGE_SECONDS = 20 * 60
|
||||
|
||||
MARKER_NAME = ".hermes-update-in-progress"
|
||||
|
||||
# Exit code meaning "another updater/instance owns this install right now".
|
||||
# Already the de-facto contract: the Windows shim + venv-holder guards in
|
||||
# _cmd_update_impl exit 2, and the Tauri updater matches on it
|
||||
# (UPDATE_EXIT_CONCURRENT in apps/bootstrap-installer/src-tauri/src/update.rs)
|
||||
# to show "Hermes is still running" instead of a generic failure. Naming it
|
||||
# here keeps the concurrent-update refusal on that same understood contract.
|
||||
UPDATE_EXIT_CONCURRENT = 2
|
||||
|
||||
|
||||
def update_marker_path() -> Path:
|
||||
"""Path of the shared update marker.
|
||||
|
||||
Uses the *process* Hermes home (never the context-local profile override):
|
||||
the Rust updater resolves ``$HERMES_HOME`` or the platform default, and the
|
||||
desktop pins that same value into the updater's env. A profile-scoped path
|
||||
here would put the lock somewhere the other two owners never look.
|
||||
"""
|
||||
from hermes_constants import get_process_hermes_home
|
||||
|
||||
return get_process_hermes_home() / MARKER_NAME
|
||||
|
||||
|
||||
def _pid_alive(pid: int) -> bool:
|
||||
"""True when a process with ``pid`` currently exists.
|
||||
|
||||
``os.kill(pid, 0)`` is POSIX-only in spirit but CPython implements the
|
||||
signal-0 existence probe on Windows too. ``PermissionError`` means the pid
|
||||
exists but is owned by another user — still alive for our purposes.
|
||||
|
||||
A value too large for the platform's ``pid_t`` raises ``OverflowError``
|
||||
(not ``OSError``), so a corrupt marker would otherwise crash every update
|
||||
that reads it. Treat any unusable pid as dead: the marker is garbage and
|
||||
must not wedge the lock.
|
||||
"""
|
||||
if pid <= 0:
|
||||
return False
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except ProcessLookupError:
|
||||
return False
|
||||
except PermissionError:
|
||||
return True
|
||||
except (OverflowError, ValueError):
|
||||
return False
|
||||
except OSError:
|
||||
# Unexpected errno: assume alive rather than stomping a live updater.
|
||||
return True
|
||||
return True
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UpdateHolder:
|
||||
"""A confirmed-live update currently holding the lock."""
|
||||
|
||||
pid: int
|
||||
age_seconds: float
|
||||
|
||||
|
||||
def read_live_update(*, path: Path | None = None) -> UpdateHolder | None:
|
||||
"""Return the live update holding the lock, or ``None``.
|
||||
|
||||
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent,
|
||||
unreadable, malformed, dead-pid, and past-the-ceiling all mean "no live
|
||||
update", and a stale marker file is deleted so it can't strand future runs.
|
||||
Never raises.
|
||||
"""
|
||||
marker = path or update_marker_path()
|
||||
try:
|
||||
raw = marker.read_text(encoding="utf-8")
|
||||
except OSError:
|
||||
return None # absent or unreadable => no live update
|
||||
|
||||
lines = raw.splitlines()
|
||||
try:
|
||||
pid = int(lines[0].strip())
|
||||
except (IndexError, ValueError):
|
||||
pid = -1
|
||||
try:
|
||||
started_at = float(lines[1].strip())
|
||||
except (IndexError, ValueError):
|
||||
started_at = float("-inf")
|
||||
|
||||
age = time.time() - started_at
|
||||
if not _pid_alive(pid) or age > UPDATE_MARKER_MAX_AGE_SECONDS:
|
||||
try:
|
||||
marker.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
|
||||
return UpdateHolder(pid=pid, age_seconds=age)
|
||||
|
||||
|
||||
def describe_holder(holder: UpdateHolder) -> str:
|
||||
"""One-line, user-facing explanation of who holds the update lock."""
|
||||
minutes, seconds = divmod(int(max(holder.age_seconds, 0)), 60)
|
||||
elapsed = f"{minutes}m {seconds}s" if minutes else f"{seconds}s"
|
||||
return (
|
||||
f"✗ Another Hermes update is already running (PID {holder.pid}, "
|
||||
f"started {elapsed} ago).\n"
|
||||
"\n"
|
||||
" Two updates mutating the same checkout corrupt it: one rewrites\n"
|
||||
" source while the other is mid-install. Wait for it to finish, or\n"
|
||||
" close the window/dashboard tab that started it, then retry."
|
||||
)
|
||||
|
||||
|
||||
class UpdateLock:
|
||||
"""Context manager owning the shared update marker for this process.
|
||||
|
||||
``acquired`` is False when another live update already holds it — callers
|
||||
decide whether that's a hard refusal (CLI/dashboard) or a wait. Releasing
|
||||
only removes the marker when *we* still own it, so a marker rewritten by a
|
||||
handoff partner (the Tauri updater overwrites it with its own pid) is never
|
||||
deleted out from under its new owner.
|
||||
"""
|
||||
|
||||
def __init__(self, *, path: Path | None = None) -> None:
|
||||
self.path = path or update_marker_path()
|
||||
self.acquired = False
|
||||
self.holder: UpdateHolder | None = None
|
||||
|
||||
def acquire(self) -> bool:
|
||||
"""Claim the lock. Returns False (and sets ``holder``) if it's taken."""
|
||||
existing = read_live_update(path=self.path)
|
||||
if existing is not None:
|
||||
self.holder = existing
|
||||
return False
|
||||
try:
|
||||
self.path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self.path.write_text(
|
||||
f"{os.getpid()}\n{int(time.time())}\n", encoding="utf-8"
|
||||
)
|
||||
except OSError as exc:
|
||||
# Best-effort, exactly like the Rust guard: an unwritable marker
|
||||
# must not block the update itself (that would be a worse failure
|
||||
# than the race it prevents). Degrade to the pre-lock behavior.
|
||||
logger.debug("Could not write update marker %s: %s", self.path, exc)
|
||||
return True
|
||||
self.acquired = True
|
||||
return True
|
||||
|
||||
def release(self) -> None:
|
||||
"""Drop the marker if this process still owns it. Never raises."""
|
||||
if not self.acquired:
|
||||
return
|
||||
self.acquired = False
|
||||
try:
|
||||
raw = self.path.read_text(encoding="utf-8")
|
||||
owner = int(raw.splitlines()[0].strip())
|
||||
except (OSError, IndexError, ValueError):
|
||||
return
|
||||
if owner != os.getpid():
|
||||
# A handoff partner took ownership (e.g. the Tauri updater wrote
|
||||
# its own pid). Leave it alone — it's still a live update.
|
||||
return
|
||||
try:
|
||||
self.path.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def __enter__(self) -> "UpdateLock":
|
||||
self.acquire()
|
||||
return self
|
||||
|
||||
def __exit__(self, *_exc) -> None:
|
||||
self.release()
|
||||
Reference in New Issue
Block a user