1904 lines
72 KiB
Rust
1904 lines
72 KiB
Rust
//! Update orchestration.
|
|
//!
|
|
//! Driven when the installer is launched as `Hermes-Setup.exe --update` (see
|
|
//! `AppMode` in lib.rs). The desktop app hands off to us — it exits, then we:
|
|
//!
|
|
//! 1. wait for the old Hermes desktop process to fully exit (so both the
|
|
//! venv shim and packaged app.asar are free; otherwise `hermes update`
|
|
//! or repair bootstrap can race locked files),
|
|
//! 2. run `hermes update --yes --gateway` (Python/repo update; this does NOT
|
|
//! rebuild apps/desktop by design — see cmd_update in hermes_cli/main.py),
|
|
//! 3. run `hermes desktop --build-only` (the rebuild step update skips),
|
|
//! 4. launch the freshly-built desktop (reuses bootstrap::launch logic).
|
|
//!
|
|
//! We reuse the `BootstrapEvent` channel + the existing progress UI by
|
|
//! emitting a synthetic multi-stage manifest (handoff → update → rebuild, plus
|
|
//! an install stage on macOS). To the frontend an update looks like a short
|
|
//! bootstrap, broken into the real operations run_update performs so the user
|
|
//! sees discrete steps (with the live log underneath) instead of one bar.
|
|
//!
|
|
//! Cross-platform note: `hermes update` already handles macOS/Linux (git/pip).
|
|
//! The only OS-specific bits here are the venv shim path (resolve_hermes) and
|
|
//! the no-window creation flag — both already cfg-gated. Keep new logic
|
|
//! OS-agnostic so the mac/linux port stays "fill in the paths".
|
|
|
|
use std::env;
|
|
use std::ffi::OsString;
|
|
use std::path::{Path, PathBuf};
|
|
use std::process::Stdio;
|
|
use std::sync::atomic::{AtomicBool, Ordering};
|
|
use std::time::{Duration, Instant};
|
|
|
|
use anyhow::{anyhow, Result};
|
|
use tauri::{AppHandle, Emitter};
|
|
use tokio::process::Command;
|
|
|
|
use crate::events::{BootstrapEvent, LogStream, StageInfo, StageState};
|
|
use crate::powershell::{pump_child, DRAIN_GRACE};
|
|
|
|
/// `hermes update` exit code meaning "another hermes process is holding the
|
|
/// venv shim open / dirty precondition" — see _cmd_update_impl in
|
|
/// hermes_cli/main.py (sys.exit(2)). We surface a targeted message for this.
|
|
const UPDATE_EXIT_CONCURRENT: i32 = 2;
|
|
|
|
/// How long to wait for the old desktop process to release files under the
|
|
/// install tree before giving up and letting `hermes update`'s own guard decide.
|
|
const DESKTOP_EXIT_WAIT: Duration = Duration::from_secs(20);
|
|
const DESKTOP_EXIT_POLL: Duration = Duration::from_millis(500);
|
|
|
|
/// Guards against concurrent update runs. The frontend kicks `startUpdate()`
|
|
/// from a mount effect, which can fire more than once (React strict-mode
|
|
/// double-invokes effects in dev; a window reload or stray re-init can do it
|
|
/// in prod). Two `run_update` tasks racing on `git stash` corrupt the working
|
|
/// tree — one stashes the changes the other then can't find. Exactly one task
|
|
/// may hold this flag at a time.
|
|
static UPDATE_RUNNING: AtomicBool = AtomicBool::new(false);
|
|
|
|
/// Frontend → Rust: kick off the update flow. Mirrors `start_bootstrap`'s
|
|
/// fire-and-forget shape; progress arrives on the `bootstrap` event channel.
|
|
#[tauri::command]
|
|
pub async fn start_update(app: AppHandle) -> Result<(), String> {
|
|
// Re-entrancy guard (see UPDATE_RUNNING). compare_exchange lets exactly one
|
|
// caller flip false→true; any concurrent caller no-ops instead of spawning
|
|
// a second racing update.
|
|
if UPDATE_RUNNING
|
|
.compare_exchange(false, true, Ordering::SeqCst, Ordering::SeqCst)
|
|
.is_err()
|
|
{
|
|
// Already running: re-emit the manifest so a duplicate startUpdate()
|
|
// call (which resets the frontend store) can recover its stage list.
|
|
let target_app = if cfg!(target_os = "macos") {
|
|
target_app_from_args(std::env::args().skip(1))
|
|
} else {
|
|
None
|
|
};
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Manifest {
|
|
stages: update_stages(target_app.is_some()),
|
|
protocol_version: None,
|
|
},
|
|
);
|
|
return Ok(());
|
|
}
|
|
tokio::spawn(async move {
|
|
if let Err(err) = run_update(app.clone()).await {
|
|
// run_update already emits a Failed event on the paths that matter;
|
|
// this catches anything that escaped. Emit defensively.
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: None,
|
|
error: format!("{err:#}"),
|
|
},
|
|
);
|
|
}
|
|
UPDATE_RUNNING.store(false, Ordering::SeqCst);
|
|
});
|
|
Ok(())
|
|
}
|
|
|
|
/// RAII guard that owns the "update in progress" marker (see
|
|
/// `paths::update_in_progress_marker`). Created at the top of `run_update`;
|
|
/// its `Drop` removes the marker on EVERY exit path — success, early
|
|
/// `return Err`, or a panic that unwinds through `run_update` — so a crashed
|
|
/// or aborted updater can never permanently strand the marker and block
|
|
/// future desktop launches. The marker payload is `{pid}\n{started_at_unix}`
|
|
/// so the desktop's launch gate can detect a stale marker (dead PID / past a
|
|
/// hard ceiling) and self-heal rather than wait forever.
|
|
///
|
|
/// The marker is also the cross-process update lock: `hermes update` claims
|
|
/// the same file (see `hermes_cli/update_lock.py`) so a dashboard-spawned
|
|
/// update and this updater can't mutate one checkout at the same time.
|
|
/// `acquire` therefore REFUSES when a live foreign owner holds it rather than
|
|
/// overwriting — the pre-fix clobber is what let a dashboard `hermes update`
|
|
/// keep running while install-mode bootstrap rewrote the tree underneath it.
|
|
struct UpdateMarkerGuard {
|
|
path: PathBuf,
|
|
/// False when a live foreign updater already owns the marker: we hold no
|
|
/// claim, so `Drop` must not delete their marker.
|
|
owned: bool,
|
|
}
|
|
|
|
/// Never treat a marker older than this as a live update. Mirrors
|
|
/// UPDATE_MARKER_MAX_AGE_MS in apps/desktop/electron/update-marker.ts and
|
|
/// UPDATE_MARKER_MAX_AGE_SECONDS in hermes_cli/update_lock.py — all three read
|
|
/// this one file, so a shorter ceiling in any of them would steal a lock the
|
|
/// others still consider live.
|
|
const UPDATE_MARKER_MAX_AGE_SECS: u64 = 20 * 60;
|
|
|
|
/// The pid + age of a confirmed-live update holding the marker.
|
|
struct MarkerOwner {
|
|
pid: u32,
|
|
age_secs: u64,
|
|
}
|
|
|
|
/// Read the marker and report a live owner, if any. `None` for every "no live
|
|
/// update" case — absent, unreadable, malformed, dead pid, or past the ceiling
|
|
/// — matching `readLiveUpdateMarker` in the Electron gate. Never panics.
|
|
///
|
|
/// Self-PID is returned so `acquire` can adopt the desktop's pre-written claim
|
|
/// without refreshing its acquisition time (#74761). A foreign live pid (e.g.
|
|
/// a dashboard-spawned `hermes update`) still blocks.
|
|
fn live_marker_owner(path: &Path) -> Option<MarkerOwner> {
|
|
let raw = std::fs::read_to_string(path).ok()?;
|
|
let mut lines = raw.lines();
|
|
let pid: u32 = lines.next()?.trim().parse().ok()?;
|
|
let started_at: u64 = lines.next().unwrap_or("").trim().parse().unwrap_or(0);
|
|
let now = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0);
|
|
let age_secs = now.saturating_sub(started_at);
|
|
if age_secs > UPDATE_MARKER_MAX_AGE_SECS || !pid_is_alive(pid) {
|
|
return None;
|
|
}
|
|
Some(MarkerOwner { pid, age_secs })
|
|
}
|
|
|
|
/// True when the on-disk marker names THIS process as its owner.
|
|
///
|
|
/// A raw read is used instead of `live_marker_owner` on purpose: that
|
|
/// helper folds in age and liveness policy (and, since the #74761
|
|
/// adoption work, self-ownership handling has changed shape more than
|
|
/// once). The exit-2 self-heal below needs exactly one raw fact — does
|
|
/// the marker name our PID — because a `hermes update` child that
|
|
/// refuses over OUR marker is a handoff-recognition failure in a stale
|
|
/// checkout, not a real concurrent update.
|
|
fn marker_owned_by_self(path: &Path) -> bool {
|
|
std::fs::read_to_string(path)
|
|
.ok()
|
|
.and_then(|raw| {
|
|
raw.lines()
|
|
.next()
|
|
.and_then(|line| line.trim().parse::<u32>().ok())
|
|
})
|
|
== Some(std::process::id())
|
|
}
|
|
|
|
/// The exit-2 heal decision (#75788), extracted so the contract is testable.
|
|
///
|
|
/// True only when BOTH hold: the child exited with the concurrent-update
|
|
/// refusal code, AND the on-disk marker names THIS process. That combination
|
|
/// means the child refused over its own parent's claim — a stale checkout
|
|
/// without handoff recognition — so dropping the claim and retrying once is
|
|
/// safe. Any other owner (live foreign updater, garbage, missing marker) or
|
|
/// any other exit code must leave the refusal untouched.
|
|
fn should_heal_self_marker_refusal(exit_code: Option<i32>, marker_path: &Path) -> bool {
|
|
exit_code == Some(UPDATE_EXIT_CONCURRENT) && marker_owned_by_self(marker_path)
|
|
}
|
|
|
|
/// True when a process with `pid` currently exists.
|
|
#[cfg(windows)]
|
|
fn pid_is_alive(pid: u32) -> bool {
|
|
use windows_sys::Win32::Foundation::{CloseHandle, STILL_ACTIVE};
|
|
use windows_sys::Win32::System::Threading::{
|
|
GetExitCodeProcess, OpenProcess, PROCESS_QUERY_LIMITED_INFORMATION,
|
|
};
|
|
|
|
unsafe {
|
|
let handle = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, 0, pid);
|
|
if handle.is_null() {
|
|
// Either the pid is gone or we lack rights to open it. A pid we
|
|
// can't inspect is treated as dead so an unopenable straggler
|
|
// can't wedge every future update.
|
|
return false;
|
|
}
|
|
let mut code: u32 = 0;
|
|
let ok = GetExitCodeProcess(handle, &mut code);
|
|
CloseHandle(handle);
|
|
ok != 0 && code == STILL_ACTIVE as u32
|
|
}
|
|
}
|
|
|
|
#[cfg(not(windows))]
|
|
fn pid_is_alive(pid: u32) -> bool {
|
|
// signal 0 delivers nothing; it only probes existence/permission.
|
|
// ESRCH => dead. EPERM => alive but owned by another user.
|
|
let rc = unsafe { libc::kill(pid as libc::pid_t, 0) };
|
|
if rc == 0 {
|
|
return true;
|
|
}
|
|
std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM)
|
|
}
|
|
|
|
impl UpdateMarkerGuard {
|
|
/// Claim the marker, or report the live updater that already owns it.
|
|
///
|
|
/// Writing is best-effort: a write failure must NOT abort the update (the
|
|
/// gate degrades to "no marker => proceed", i.e. exactly the pre-marker
|
|
/// behavior), so we log and carry on with a guard that still attempts
|
|
/// cleanup of whatever may exist at the path.
|
|
fn acquire(path: PathBuf) -> Result<Self, MarkerOwner> {
|
|
let pid = std::process::id();
|
|
if let Some(owner) = live_marker_owner(&path) {
|
|
if owner.pid == pid {
|
|
// Repeated acquisition in this process is intentionally
|
|
// re-entrant because the desktop may have pre-written our pid.
|
|
// The desktop races ahead and pre-writes our pid. Adopt that
|
|
// claim verbatim: rewriting started_at here lets retries reset
|
|
// a wedged updater's age before the stale ceiling can clear it.
|
|
return Ok(Self { path, owned: true });
|
|
}
|
|
return Err(owner);
|
|
}
|
|
let started_at = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0);
|
|
if let Some(parent) = path.parent() {
|
|
let _ = std::fs::create_dir_all(parent);
|
|
}
|
|
if let Err(err) = std::fs::write(&path, format!("{pid}\n{started_at}")) {
|
|
tracing::warn!(?path, %err, "could not write update-in-progress marker");
|
|
}
|
|
Ok(Self { path, owned: true })
|
|
}
|
|
|
|
/// Release the marker as soon as every mutating stage has completed.
|
|
///
|
|
/// The updater still owns a Tauri/Cocoa event loop while it relaunches the
|
|
/// desktop, and that loop can outlive `app.exit(0)`. Relying on `Drop`
|
|
/// alone therefore leaves a *successful* update looking active — a live
|
|
/// pid holding a fresh marker — which blocks desktop startup and every
|
|
/// other updater for the full age ceiling. Idempotent: `Drop` still runs
|
|
/// and tolerates an already-removed marker.
|
|
fn complete(&self) {
|
|
if !self.owned {
|
|
return;
|
|
}
|
|
if let Err(err) = std::fs::remove_file(&self.path) {
|
|
if err.kind() != std::io::ErrorKind::NotFound {
|
|
tracing::warn!(path = ?self.path, %err, "could not remove completed update marker");
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
impl Drop for UpdateMarkerGuard {
|
|
fn drop(&mut self) {
|
|
self.complete();
|
|
}
|
|
}
|
|
|
|
async fn run_update(app: AppHandle) -> Result<()> {
|
|
let hermes_home = crate::paths::hermes_home();
|
|
let install_root = hermes_home.join("hermes-agent");
|
|
|
|
// Mutual exclusion (#50238): publish an "update in progress" marker for the
|
|
// entire duration of this update. A desktop instance the user relaunches
|
|
// mid-update consults this before spawning its own local backend — without
|
|
// it, that backend re-locks the venv shim, our `force_kill_other_hermes`
|
|
// straggler-cleanup kills it, and the relaunch/kill cycle loops. The guard
|
|
// removes the marker on every exit path (incl. early returns / panics).
|
|
//
|
|
// The same marker is the cross-process update lock (hermes_cli/
|
|
// update_lock.py claims it too), so a live foreign owner means another
|
|
// updater — most often a dashboard-spawned `hermes update` — is already
|
|
// mutating this checkout. Refuse instead of running a second one over it.
|
|
let _update_marker = match UpdateMarkerGuard::acquire(
|
|
crate::paths::update_in_progress_marker(),
|
|
) {
|
|
Ok(guard) => guard,
|
|
Err(owner) => {
|
|
let mins = owner.age_secs / 60;
|
|
let secs = owner.age_secs % 60;
|
|
let elapsed = if mins > 0 {
|
|
format!("{mins}m {secs}s")
|
|
} else {
|
|
format!("{secs}s")
|
|
};
|
|
let msg = format!(
|
|
"Another Hermes update is already running (PID {}, started {} ago). \
|
|
Wait for it to finish, or close the window or dashboard tab that \
|
|
started it, then try again.",
|
|
owner.pid, elapsed
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: None,
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
return Err(anyhow!(msg));
|
|
}
|
|
};
|
|
|
|
let update_branch = update_branch_from_args(std::env::args().skip(1))
|
|
.or_else(|| option_env_string("BUILD_PIN_BRANCH"))
|
|
.unwrap_or_else(|| "main".to_string());
|
|
let target_app = if cfg!(target_os = "macos") {
|
|
target_app_from_args(std::env::args().skip(1))
|
|
} else {
|
|
None
|
|
};
|
|
|
|
let hermes = resolve_hermes(&install_root).ok_or_else(|| {
|
|
let msg = format!(
|
|
"Could not find the hermes CLI under {}. Is Hermes installed? \
|
|
Re-run the installer to repair the install.",
|
|
install_root.display()
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: None,
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
anyhow!(msg)
|
|
})?;
|
|
|
|
// Synthetic manifest so the existing progress UI renders our stages.
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Manifest {
|
|
stages: update_stages(target_app.is_some()),
|
|
protocol_version: None,
|
|
},
|
|
);
|
|
|
|
// ---- stage 1: wait for the old desktop to die ------------------------
|
|
// The desktop exec'd us then called app.exit(), but process teardown is
|
|
// async on Windows. If it still holds the venv shim, `hermes update`
|
|
// aborts with exit 2. If it still holds the packaged app.asar,
|
|
// install.ps1's repair/re-clone path cannot move/remove the install tree.
|
|
// Give both handles a bounded window to clear. Surfaced as its own stage
|
|
// (rather than a silent pre-step) so a slow close / force-kill reads as
|
|
// real progress instead of a frozen first bar.
|
|
let started = Instant::now();
|
|
emit_stage(&app, "handoff", StageState::Running, None, None);
|
|
wait_for_install_locks_free(&install_root, &app, "handoff").await;
|
|
emit_stage(
|
|
&app,
|
|
"handoff",
|
|
StageState::Succeeded,
|
|
Some(started.elapsed().as_millis() as u64),
|
|
None,
|
|
);
|
|
|
|
// ---- stage 2: hermes update -----------------------------------------
|
|
// Pass --branch so `hermes update` targets the branch this installer was
|
|
// built/pinned against (BUILD_PIN_BRANCH), NOT its built-in default of
|
|
// `main`. The install was a detached-HEAD checkout of a specific commit;
|
|
// without --branch, `hermes update` switches the checkout to `main` (a
|
|
// divergent branch that may not even have the desktop CLI command), then
|
|
// reports "already up to date" against the wrong branch. The desktop
|
|
// detected the update against this same branch, so we must update against
|
|
// it too.
|
|
emit_log(
|
|
&app,
|
|
Some("update"),
|
|
LogStream::Stdout,
|
|
&format!("[update] updating against branch {update_branch}"),
|
|
);
|
|
let child_env = update_child_env(&install_root);
|
|
let mut update_args: Vec<String> =
|
|
vec!["update".into(), "--yes".into(), "--gateway".into()];
|
|
// --force skips `hermes update`'s Windows running-exe guard (which would
|
|
// `sys.exit(2)` and dead-end the handoff). By contract the desktop has
|
|
// already exited and waited for the install locks to clear before launching
|
|
// us, and wait_for_install_locks_free below force-kills any straggler — so by the
|
|
// time `hermes update` runs there is no legitimate hermes.exe to protect,
|
|
// and the guard would only produce a false "Hermes is still running" stop.
|
|
//
|
|
// NOTE: --force does NOT bypass the venv-python holder guard (that needs
|
|
// an explicit `--force-venv`, which we deliberately do not pass). Our lock
|
|
// probe only checks the hermes.exe shim and app.asar, so an external venv
|
|
// python holding a native .pyd (a user terminal, an unmanaged gateway)
|
|
// could still be alive here — mutating the venv under it would strand the
|
|
// install half-updated. If that guard fires, it exits 2 and the match arm
|
|
// below surfaces the correct "close all Hermes windows" message.
|
|
update_args.push("--force".into());
|
|
update_args.push("--branch".into());
|
|
update_args.push(update_branch);
|
|
|
|
emit_stage(&app, "update", StageState::Running, None, None);
|
|
let started = Instant::now();
|
|
let mut update = run_streamed(
|
|
&app,
|
|
&hermes,
|
|
&update_args,
|
|
&install_root,
|
|
&child_env,
|
|
Some("update"),
|
|
)
|
|
.await?;
|
|
|
|
// Retry-once for the update-boundary crash. `hermes update` lazily imports
|
|
// the FRESHLY PULLED modules, but the dependency-install step still runs the
|
|
// already-in-memory pre-pull code for one invocation. A release that changed
|
|
// an updater-path contract across that boundary (e.g. #39780's `_UvResult`,
|
|
// whose `__iter__` injected a bool into the argv and crashed Windows
|
|
// `list2cmdline` with `TypeError: sequence item 1: expected str instance,
|
|
// bool found`, fixed in #39820) therefore kills the FIRST update on the
|
|
// parked population — even though the fix is already on disk by then. A
|
|
// second `hermes update` runs clean because the now-current module is loaded
|
|
// from the start. Rather than make the parked user click Update twice (and
|
|
// stare at a scary crash first), retry once automatically. Skip the retry
|
|
// for the concurrent-instance guard (exit 2) — that's a "close Hermes" state
|
|
// a retry can't fix.
|
|
if !matches!(update.exit_code, Some(0) | Some(UPDATE_EXIT_CONCURRENT)) {
|
|
emit_log(
|
|
&app,
|
|
Some("update"),
|
|
LogStream::Stdout,
|
|
"[update] first update attempt failed; retrying once (the fix it just \
|
|
pulled loads on the second run)…",
|
|
);
|
|
update = run_streamed(
|
|
&app,
|
|
&hermes,
|
|
&update_args,
|
|
&install_root,
|
|
&child_env,
|
|
Some("update"),
|
|
)
|
|
.await?;
|
|
}
|
|
|
|
// Self-owned-marker heal (#75788). Exit 2 means the child refused over a
|
|
// live update marker with a foreign owner. When that "foreign" owner is
|
|
// THIS process, the child simply failed to recognize the handoff — a
|
|
// checkout predating the HERMES_UPDATE_HANDOFF_PID env fix (8c76fe19f)
|
|
// and the ancestor-pid fallback runs its pre-pull update_lock.py, reads
|
|
// our marker, and exits 2 every time. The refusal loop is unbreakable
|
|
// from the user's side because the update being refused is the one that
|
|
// ships the fix. The marker exists to serialize updates and this process
|
|
// IS the update: drop our claim and retry once with the marker absent.
|
|
// The guard re-removes on Drop (idempotent), and the desktop is already
|
|
// gone at this point, so nothing races the brief marker-free window.
|
|
if should_heal_self_marker_refusal(
|
|
update.exit_code,
|
|
&crate::paths::update_in_progress_marker(),
|
|
) {
|
|
emit_log(
|
|
&app,
|
|
Some("update"),
|
|
LogStream::Stdout,
|
|
"[update] child refused over this updater's own marker (stale \
|
|
checkout without handoff recognition); clearing the claim and \
|
|
retrying once…",
|
|
);
|
|
_update_marker.complete();
|
|
update = run_streamed(
|
|
&app,
|
|
&hermes,
|
|
&update_args,
|
|
&install_root,
|
|
&child_env,
|
|
Some("update"),
|
|
)
|
|
.await?;
|
|
}
|
|
let update_ms = started.elapsed().as_millis() as u64;
|
|
|
|
match update.exit_code {
|
|
Some(0) => {
|
|
emit_stage(&app, "update", StageState::Succeeded, Some(update_ms), None);
|
|
}
|
|
Some(code) if code == UPDATE_EXIT_CONCURRENT => {
|
|
let msg = "Hermes is still running. Close all Hermes windows and try \
|
|
the update again."
|
|
.to_string();
|
|
emit_stage(
|
|
&app,
|
|
"update",
|
|
StageState::Failed,
|
|
Some(update_ms),
|
|
Some(msg.clone()),
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: Some("update".into()),
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
return Err(anyhow!(msg));
|
|
}
|
|
other => {
|
|
let msg = format!(
|
|
"hermes update failed (exit {:?}). See {} for details.",
|
|
other,
|
|
crate::paths::hermes_home()
|
|
.join("logs")
|
|
.join("update.log")
|
|
.display()
|
|
);
|
|
emit_stage(
|
|
&app,
|
|
"update",
|
|
StageState::Failed,
|
|
Some(update_ms),
|
|
Some(msg.clone()),
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: Some("update".into()),
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
return Err(anyhow!(msg));
|
|
}
|
|
}
|
|
|
|
// ---- stage 3: hermes desktop --build-only ----------------------------
|
|
// `hermes update` deliberately does NOT build apps/desktop (it installs
|
|
// repo-root deps with --workspaces=false). This is the rebuild it skips.
|
|
emit_stage(&app, "rebuild", StageState::Running, None, None);
|
|
let started = Instant::now();
|
|
let rebuild_args: Vec<String> = vec!["desktop".into(), "--build-only".into()];
|
|
let mut rebuild = run_streamed(
|
|
&app,
|
|
&hermes,
|
|
&rebuild_args,
|
|
&install_root,
|
|
&child_env,
|
|
Some("rebuild"),
|
|
)
|
|
.await?;
|
|
|
|
// Retry-once: the first `--build-only` can return nonzero on a still-settling
|
|
// post-update tree or a network-blocked Electron fetch that our self-heal
|
|
// repaired mid-run. A second attempt then builds clean off the healed dist
|
|
// (the content-hash stamp makes it a near-no-op when the first actually
|
|
// succeeded). Without this the updater bails here and never reaches the
|
|
// relaunch below — the app updates but doesn't restart. Matches the
|
|
// retry-once `hermes update` already does above, and `hermes update`'s own
|
|
// desktop rebuild in cmd_update.
|
|
if rebuild_needs_retry(rebuild.exit_code) {
|
|
emit_log(
|
|
&app,
|
|
Some("rebuild"),
|
|
LogStream::Stdout,
|
|
"[rebuild] first desktop rebuild failed; retrying once (a self-healed \
|
|
Electron download builds clean on the second run)…",
|
|
);
|
|
rebuild = run_streamed(
|
|
&app,
|
|
&hermes,
|
|
&rebuild_args,
|
|
&install_root,
|
|
&child_env,
|
|
Some("rebuild"),
|
|
)
|
|
.await?;
|
|
}
|
|
let rebuild_ms = started.elapsed().as_millis() as u64;
|
|
|
|
if rebuild.exit_code != Some(0) {
|
|
let msg = format!(
|
|
"Rebuilding the desktop app failed (exit {:?}). The update was \
|
|
applied but the app could not be rebuilt; run `hermes desktop` \
|
|
from a terminal to see the error.",
|
|
rebuild.exit_code
|
|
);
|
|
emit_stage(
|
|
&app,
|
|
"rebuild",
|
|
StageState::Failed,
|
|
Some(rebuild_ms),
|
|
Some(msg.clone()),
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: Some("rebuild".into()),
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
return Err(anyhow!(msg));
|
|
}
|
|
emit_stage(&app, "rebuild", StageState::Succeeded, Some(rebuild_ms), None);
|
|
|
|
let launch_target = if let Some(target_app) = target_app {
|
|
let started = Instant::now();
|
|
emit_stage(&app, "install", StageState::Running, None, None);
|
|
match install_macos_app_update(&app, &install_root, &target_app).await {
|
|
Ok(installed_app) => {
|
|
emit_stage(
|
|
&app,
|
|
"install",
|
|
StageState::Succeeded,
|
|
Some(started.elapsed().as_millis() as u64),
|
|
None,
|
|
);
|
|
Some(installed_app)
|
|
}
|
|
Err(err) => {
|
|
let msg = format!("{err:#}");
|
|
emit_stage(
|
|
&app,
|
|
"install",
|
|
StageState::Failed,
|
|
Some(started.elapsed().as_millis() as u64),
|
|
Some(msg.clone()),
|
|
);
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Failed {
|
|
stage: Some("install".into()),
|
|
error: msg.clone(),
|
|
},
|
|
);
|
|
return Err(anyhow!(msg));
|
|
}
|
|
}
|
|
} else {
|
|
None
|
|
};
|
|
|
|
// ---- done: signal complete, then launch the fresh desktop ------------
|
|
emit(
|
|
&app,
|
|
BootstrapEvent::Complete {
|
|
install_root: install_root.to_string_lossy().into_owned(),
|
|
marker: None,
|
|
},
|
|
);
|
|
// Every install-tree mutation is finished. Release the lock BEFORE the
|
|
// relaunch: this process can stay wedged in its native event loop even
|
|
// after a successful app.exit(), and a live pid on a fresh marker would
|
|
// make a completed update look active — blocking desktop startup and
|
|
// every other updater until the age ceiling expires.
|
|
_update_marker.complete();
|
|
|
|
if let Some(target_app) = launch_target {
|
|
if let Err(err) = launch_macos_app_and_exit(&app, &target_app).await {
|
|
emit_log(
|
|
&app,
|
|
None,
|
|
LogStream::Stderr,
|
|
&format!("[update] could not auto-launch desktop: {err}. Launch Hermes manually."),
|
|
);
|
|
}
|
|
} else if let Err(err) =
|
|
crate::bootstrap::launch_hermes_desktop(app.clone(), install_root.to_string_lossy().into_owned()).await
|
|
{
|
|
// Launch failed: don't hard-fail the update (it succeeded); surface a
|
|
// log line so the success screen can still tell the user to launch
|
|
// manually.
|
|
emit_log(
|
|
&app,
|
|
None,
|
|
LogStream::Stdout,
|
|
&format!("[update] could not auto-launch desktop: {err}. Launch Hermes manually."),
|
|
);
|
|
}
|
|
|
|
// The launch helpers normally request exit themselves, but their failure
|
|
// paths must still close a successful updater. A native event loop can
|
|
// ignore that graceful request, so arm a process-exit fallback now that
|
|
// all update state and the marker have been settled.
|
|
exit_after_success(&app);
|
|
Ok(())
|
|
}
|
|
|
|
/// Ask the app to exit, with a hard `process::exit` fallback for a native
|
|
/// event loop that ignores the graceful request. Without it a finished updater
|
|
/// can linger as a live pid forever.
|
|
fn exit_after_success(app: &AppHandle) {
|
|
std::thread::spawn(|| {
|
|
std::thread::sleep(std::time::Duration::from_secs(3));
|
|
tracing::warn!("graceful updater exit timed out; forcing process exit");
|
|
std::process::exit(0);
|
|
});
|
|
app.exit(0);
|
|
}
|
|
|
|
/// Poll until the venv shim AND packaged desktop app bundle are no longer locked
|
|
/// (Windows) or a bounded timeout elapses. On non-Windows this is a short fixed
|
|
/// grace since file locking isn't the failure mode there.
|
|
pub(crate) async fn wait_for_install_locks_free(install_root: &Path, app: &AppHandle, stage: &str) {
|
|
let lock_targets = install_lock_probe_paths(install_root);
|
|
let deadline = Instant::now() + DESKTOP_EXIT_WAIT;
|
|
|
|
emit_log(app, Some(stage), LogStream::Stdout, "[handoff] waiting for Hermes to exit…");
|
|
|
|
loop {
|
|
let locked = locked_paths(&lock_targets);
|
|
if locked.is_empty() {
|
|
return;
|
|
}
|
|
if Instant::now() >= deadline {
|
|
// Last resort: a backend shim can still hold update-sensitive
|
|
// files when the desktop's shutdown races a detached child. Only
|
|
// target the shim at this install root: the desktop binary is also
|
|
// Hermes.exe, so an image-name kill would tear down the app itself.
|
|
emit_log(
|
|
app,
|
|
Some(stage),
|
|
LogStream::Stdout,
|
|
&format!(
|
|
"[handoff] Hermes still holding install files ({}); locating backend shims…",
|
|
format_locked_paths(&locked)
|
|
),
|
|
);
|
|
let shim = venv_hermes(install_root);
|
|
let shim_pids = backend_shim_pids(&shim);
|
|
if shim_pids.is_empty() {
|
|
emit_log(
|
|
app,
|
|
Some(stage),
|
|
LogStream::Stdout,
|
|
"[handoff] no installed backend shim matched the force-kill fallback",
|
|
);
|
|
} else {
|
|
for pid in &shim_pids {
|
|
emit_log(
|
|
app,
|
|
Some(stage),
|
|
LogStream::Stdout,
|
|
&format!(
|
|
"[handoff] force-killing backend shim PID {pid} ({})",
|
|
shim.display()
|
|
),
|
|
);
|
|
}
|
|
force_kill_process_trees(&shim_pids);
|
|
}
|
|
tokio::time::sleep(Duration::from_millis(800)).await;
|
|
let locked_after_kill = locked_paths(&lock_targets);
|
|
if locked_after_kill.is_empty() {
|
|
emit_log(
|
|
app,
|
|
Some(stage),
|
|
LogStream::Stdout,
|
|
"[handoff] install files freed after force-kill",
|
|
);
|
|
} else {
|
|
emit_log(
|
|
app,
|
|
Some(stage),
|
|
LogStream::Stdout,
|
|
&format!(
|
|
"[handoff] install files still locked ({}); proceeding (--force + quarantine will handle it)",
|
|
format_locked_paths(&locked_after_kill)
|
|
),
|
|
);
|
|
}
|
|
return;
|
|
}
|
|
tokio::time::sleep(DESKTOP_EXIT_POLL).await;
|
|
}
|
|
}
|
|
|
|
fn install_lock_probe_paths(install_root: &Path) -> Vec<PathBuf> {
|
|
let mut paths = vec![venv_hermes(install_root)];
|
|
paths.extend(desktop_app_payload_paths(install_root));
|
|
paths
|
|
}
|
|
|
|
fn desktop_app_payload_paths(install_root: &Path) -> Vec<PathBuf> {
|
|
let release = install_root.join("apps").join("desktop").join("release");
|
|
if cfg!(target_os = "windows") {
|
|
vec![
|
|
release.join("win-unpacked").join("resources").join("app.asar"),
|
|
release.join("win-arm64-unpacked").join("resources").join("app.asar"),
|
|
]
|
|
} else if cfg!(target_os = "macos") {
|
|
vec![
|
|
release.join("mac").join("Hermes.app").join("Contents").join("Resources").join("app.asar"),
|
|
release.join("mac-arm64").join("Hermes.app").join("Contents").join("Resources").join("app.asar"),
|
|
]
|
|
} else {
|
|
vec![release.join("linux-unpacked").join("resources").join("app.asar")]
|
|
}
|
|
}
|
|
|
|
fn locked_paths(paths: &[PathBuf]) -> Vec<PathBuf> {
|
|
paths.iter().filter(|p| is_locked(p)).cloned().collect()
|
|
}
|
|
|
|
fn format_locked_paths(paths: &[PathBuf]) -> String {
|
|
paths.iter().map(|p| p.display().to_string()).collect::<Vec<_>>().join(", ")
|
|
}
|
|
|
|
/// Find processes running the exact `venv\Scripts\hermes.exe` shim for this
|
|
/// installation. Windows image names are case-insensitive and the desktop is
|
|
/// also Hermes.exe, so matching by image name alone is unsafe.
|
|
#[cfg(windows)]
|
|
fn backend_shim_pids(shim: &Path) -> Vec<u32> {
|
|
use std::ffi::OsString;
|
|
use std::mem::{size_of, zeroed};
|
|
use std::os::windows::ffi::OsStringExt;
|
|
use windows_sys::Win32::Foundation::{CloseHandle, INVALID_HANDLE_VALUE};
|
|
use windows_sys::Win32::System::Diagnostics::ToolHelp::{
|
|
CreateToolhelp32Snapshot, Process32FirstW, Process32NextW, PROCESSENTRY32W,
|
|
TH32CS_SNAPPROCESS,
|
|
};
|
|
use windows_sys::Win32::System::Threading::{
|
|
OpenProcess, QueryFullProcessImageNameW, PROCESS_QUERY_LIMITED_INFORMATION,
|
|
};
|
|
|
|
const MAX_PATH_CHARS: usize = 32_768;
|
|
|
|
fn image_path_for_pid(pid: u32) -> Option<PathBuf> {
|
|
unsafe {
|
|
let handle = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, 0, pid);
|
|
if handle.is_null() {
|
|
return None;
|
|
}
|
|
let mut path = vec![0_u16; MAX_PATH_CHARS];
|
|
let mut len = path.len() as u32;
|
|
let ok = QueryFullProcessImageNameW(handle, 0, path.as_mut_ptr(), &mut len);
|
|
CloseHandle(handle);
|
|
(ok != 0).then(|| PathBuf::from(OsString::from_wide(&path[..len as usize])))
|
|
}
|
|
}
|
|
|
|
let snapshot = unsafe { CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0) };
|
|
if snapshot == INVALID_HANDLE_VALUE {
|
|
return Vec::new();
|
|
}
|
|
|
|
let mut entry: PROCESSENTRY32W = unsafe { zeroed() };
|
|
entry.dwSize = size_of::<PROCESSENTRY32W>() as u32;
|
|
let mut pids = Vec::new();
|
|
let mut inspected_candidates = 0_u32;
|
|
let own_pid = std::process::id();
|
|
let mut has_entry = unsafe { Process32FirstW(snapshot, &mut entry) } != 0;
|
|
while has_entry {
|
|
let pid = entry.th32ProcessID;
|
|
if pid != own_pid {
|
|
if let Some(path) = image_path_for_pid(pid) {
|
|
inspected_candidates += 1;
|
|
if same_windows_path(&path, shim) {
|
|
pids.push(pid);
|
|
}
|
|
}
|
|
}
|
|
has_entry = unsafe { Process32NextW(snapshot, &mut entry) } != 0;
|
|
}
|
|
unsafe { CloseHandle(snapshot) };
|
|
if pids.is_empty() && inspected_candidates > 0 {
|
|
tracing::debug!(
|
|
expected_shim = %shim.display(),
|
|
inspected_candidates,
|
|
"no queryable process image matched the backend shim path"
|
|
);
|
|
}
|
|
pids
|
|
}
|
|
|
|
#[cfg(not(windows))]
|
|
fn backend_shim_pids(_shim: &Path) -> Vec<u32> {
|
|
Vec::new()
|
|
}
|
|
|
|
fn same_windows_path(actual: &Path, expected: &Path) -> bool {
|
|
actual
|
|
.to_string_lossy()
|
|
.eq_ignore_ascii_case(&expected.to_string_lossy())
|
|
}
|
|
|
|
#[cfg(windows)]
|
|
fn force_kill_process_trees(pids: &[u32]) {
|
|
for pid in pids {
|
|
let _ = std::process::Command::new("taskkill")
|
|
.args(["/F", "/T", "/PID", &pid.to_string()])
|
|
.stdout(std::process::Stdio::null())
|
|
.stderr(std::process::Stdio::null())
|
|
.status();
|
|
}
|
|
}
|
|
|
|
#[cfg(not(windows))]
|
|
fn force_kill_process_trees(_pids: &[u32]) {}
|
|
|
|
/// Best-effort lock probe: try to open the file for read+write. On Windows an
|
|
/// exclusively-held running .exe refuses the open with a sharing violation.
|
|
/// On Unix this almost always succeeds (no mandatory locking), which is fine —
|
|
/// the venv-shim contention is a Windows-only problem.
|
|
fn is_locked(path: &Path) -> bool {
|
|
if !path.exists() {
|
|
return false;
|
|
}
|
|
match std::fs::OpenOptions::new().read(true).write(true).open(path) {
|
|
Ok(_) => false,
|
|
Err(_) => true,
|
|
}
|
|
}
|
|
|
|
/// Whether the `desktop --build-only` rebuild should be retried once. Any
|
|
/// non-success exit qualifies: the common cause is a transient first-attempt
|
|
/// failure (still-settling tree / self-healed Electron download) that a clean
|
|
/// second run resolves.
|
|
fn rebuild_needs_retry(exit_code: Option<i32>) -> bool {
|
|
exit_code != Some(0)
|
|
}
|
|
|
|
/// Spawn `hermes <args>` from `cwd`, stream stdout/stderr as Log events on the
|
|
/// bootstrap channel, and return the exit code. Mirrors powershell::run_script
|
|
/// but for an arbitrary command (no install.ps1 -File wrapping).
|
|
async fn run_streamed(
|
|
app: &AppHandle,
|
|
program: &Path,
|
|
args: &[String],
|
|
cwd: &Path,
|
|
envs: &[(String, OsString)],
|
|
stage: Option<&str>,
|
|
) -> Result<CmdResult> {
|
|
let mut cmd = Command::new(program);
|
|
cmd.args(args)
|
|
.current_dir(cwd)
|
|
.stdin(Stdio::null())
|
|
.stdout(Stdio::piped())
|
|
.stderr(Stdio::piped());
|
|
for (key, value) in envs {
|
|
cmd.env(key, value);
|
|
}
|
|
|
|
#[cfg(target_os = "windows")]
|
|
{
|
|
use std::os::windows::process::CommandExt;
|
|
// CREATE_NO_WINDOW = 0x08000000 — no flashing console behind the GUI.
|
|
cmd.creation_flags(0x0800_0000);
|
|
}
|
|
|
|
let mut child = cmd
|
|
.spawn()
|
|
.map_err(|e| anyhow!("spawning {} {:?}: {e}", program.display(), args))?;
|
|
|
|
// Same non-UTF-8-safe decode path as powershell::run_script (#67193), and
|
|
// the same rule about pipe EOF: `hermes update` is precisely the shape that
|
|
// leaves resident descendants holding an inherited stdout handle, and every
|
|
// stage this drives sits downstream of the read.
|
|
let stage_owned = stage.map(|s| s.to_string());
|
|
let outcome = pump_child(
|
|
&mut child,
|
|
|l| emit_log(app, stage_owned.as_deref(), LogStream::Stdout, l),
|
|
|l| emit_log(app, stage_owned.as_deref(), LogStream::Stderr, l),
|
|
&mut None,
|
|
DRAIN_GRACE,
|
|
)
|
|
.await
|
|
.map_err(|e| anyhow!("streaming {} {:?}: {e}", program.display(), args))?;
|
|
|
|
if outcome.abandoned {
|
|
let note = format!(
|
|
"{} exited but a surviving descendant still holds its stdout/stderr; \
|
|
gave up on the last {}s of output (#90455)",
|
|
program.display(),
|
|
DRAIN_GRACE.as_secs()
|
|
);
|
|
tracing::warn!("{note}");
|
|
emit_log(app, stage_owned.as_deref(), LogStream::Stderr, ¬e);
|
|
}
|
|
|
|
Ok(CmdResult {
|
|
exit_code: outcome.exit_code,
|
|
})
|
|
}
|
|
|
|
struct CmdResult {
|
|
exit_code: Option<i32>,
|
|
}
|
|
|
|
/// Path to the venv hermes shim under an install root, regardless of existence.
|
|
fn venv_hermes(install_root: &Path) -> PathBuf {
|
|
if cfg!(target_os = "windows") {
|
|
install_root.join("venv").join("Scripts").join("hermes.exe")
|
|
} else {
|
|
install_root.join("venv").join("bin").join("hermes")
|
|
}
|
|
}
|
|
|
|
/// Resolve the hermes CLI to drive. Prefer the venv shim in the install we
|
|
/// just updated; fall back to `hermes` on PATH.
|
|
fn resolve_hermes(install_root: &Path) -> Option<PathBuf> {
|
|
let shim = venv_hermes(install_root);
|
|
if shim.exists() {
|
|
return Some(shim);
|
|
}
|
|
// PATH fallback. which-style probe via env, kept dependency-free.
|
|
let exe = if cfg!(target_os = "windows") { "hermes.exe" } else { "hermes" };
|
|
if let Ok(path) = std::env::var("PATH") {
|
|
let sep = if cfg!(target_os = "windows") { ';' } else { ':' };
|
|
for dir in path.split(sep) {
|
|
let cand = Path::new(dir).join(exe);
|
|
if cand.exists() {
|
|
return Some(cand);
|
|
}
|
|
}
|
|
}
|
|
None
|
|
}
|
|
|
|
fn update_child_env(install_root: &Path) -> Vec<(String, OsString)> {
|
|
let hermes_home = crate::paths::hermes_home();
|
|
let mut envs = vec![(
|
|
"HERMES_HOME".to_string(),
|
|
hermes_home.as_os_str().to_os_string(),
|
|
)];
|
|
// `hermes update` is a Python CLI writing to a pipe here, so CPython
|
|
// block-buffers its stdout: nothing reaches run_streamed (and the live
|
|
// log UI) until 8 KB accumulate or the process exits. Long quiet steps —
|
|
// the pre-update backup can zip multi-GB archives for minutes — render as
|
|
// a frozen stage, and users cancel a healthy update. Force line-by-line
|
|
// output instead.
|
|
envs.push(("PYTHONUNBUFFERED".to_string(), OsString::from("1")));
|
|
// We hold the update-in-progress marker for this whole run, and the
|
|
// `hermes update` child claims that SAME lock (hermes_cli/update_lock.py).
|
|
// Name our pid so the child recognizes the live holder as its own
|
|
// orchestrator and runs under our claim — without this every GUI update
|
|
// refuses its parent's marker with exit 2 ("Hermes is still running")
|
|
// and no number of retries can ever succeed. Keep the variable name in
|
|
// sync with HANDOFF_PID_ENV in hermes_cli/update_lock.py.
|
|
envs.push((
|
|
"HERMES_UPDATE_HANDOFF_PID".to_string(),
|
|
OsString::from(std::process::id().to_string()),
|
|
));
|
|
if let Some(path) = path_with_prepended_entries(&[
|
|
hermes_home.join("node").join("bin"),
|
|
venv_bin_dir(install_root),
|
|
]) {
|
|
envs.push(("PATH".to_string(), path));
|
|
}
|
|
envs
|
|
}
|
|
|
|
fn venv_bin_dir(install_root: &Path) -> PathBuf {
|
|
if cfg!(target_os = "windows") {
|
|
install_root.join("venv").join("Scripts")
|
|
} else {
|
|
install_root.join("venv").join("bin")
|
|
}
|
|
}
|
|
|
|
fn path_with_prepended_entries(entries: &[PathBuf]) -> Option<OsString> {
|
|
let mut parts: Vec<PathBuf> = entries.to_vec();
|
|
if let Some(existing) = env::var_os("PATH") {
|
|
parts.extend(env::split_paths(&existing));
|
|
}
|
|
env::join_paths(parts).ok()
|
|
}
|
|
|
|
fn update_branch_from_args<I, S>(args: I) -> Option<String>
|
|
where
|
|
I: IntoIterator<Item = S>,
|
|
S: AsRef<str>,
|
|
{
|
|
arg_value_from_args(args, "--branch")
|
|
.map(|s| s.trim().to_string())
|
|
.filter(|s| !s.is_empty())
|
|
}
|
|
|
|
fn target_app_from_args<I, S>(args: I) -> Option<PathBuf>
|
|
where
|
|
I: IntoIterator<Item = S>,
|
|
S: AsRef<str>,
|
|
{
|
|
arg_value_from_args(args, "--target-app")
|
|
.map(PathBuf::from)
|
|
.filter(|p| p.extension().and_then(|e| e.to_str()) == Some("app"))
|
|
}
|
|
|
|
fn arg_value_from_args<I, S>(args: I, name: &str) -> Option<String>
|
|
where
|
|
I: IntoIterator<Item = S>,
|
|
S: AsRef<str>,
|
|
{
|
|
let mut iter = args.into_iter().map(|s| s.as_ref().to_string()).peekable();
|
|
while let Some(arg) = iter.next() {
|
|
if arg == name {
|
|
return iter.next();
|
|
}
|
|
if let Some(value) = arg.strip_prefix(&format!("{name}=")) {
|
|
return Some(value.to_string());
|
|
}
|
|
}
|
|
None
|
|
}
|
|
|
|
#[cfg(target_os = "macos")]
|
|
async fn install_macos_app_update(
|
|
app: &AppHandle,
|
|
install_root: &Path,
|
|
target_app: &Path,
|
|
) -> Result<PathBuf> {
|
|
if target_app.extension().and_then(|e| e.to_str()) != Some("app") {
|
|
return Err(anyhow!(
|
|
"refusing to install update into non-app path: {}",
|
|
target_app.display()
|
|
));
|
|
}
|
|
|
|
let rebuilt_app = crate::bootstrap::resolve_hermes_desktop_app(install_root).ok_or_else(|| {
|
|
anyhow!(
|
|
"desktop rebuild succeeded but no Hermes.app was found under {}",
|
|
install_root.join("apps").join("desktop").join("release").display()
|
|
)
|
|
})?;
|
|
|
|
let same = match (rebuilt_app.canonicalize(), target_app.canonicalize()) {
|
|
(Ok(a), Ok(b)) => a == b,
|
|
_ => rebuilt_app == target_app,
|
|
};
|
|
if same {
|
|
emit_log(
|
|
app,
|
|
Some("install"),
|
|
LogStream::Stdout,
|
|
&format!(
|
|
"[update] rebuilt app is already the launch target: {}",
|
|
target_app.display()
|
|
),
|
|
);
|
|
return Ok(target_app.to_path_buf());
|
|
}
|
|
|
|
emit_log(
|
|
app,
|
|
Some("install"),
|
|
LogStream::Stdout,
|
|
&format!(
|
|
"[update] installing rebuilt app {} -> {}",
|
|
rebuilt_app.display(),
|
|
target_app.display()
|
|
),
|
|
);
|
|
|
|
if let Some(parent) = target_app.parent() {
|
|
tokio::fs::create_dir_all(parent).await?;
|
|
}
|
|
let tmp = PathBuf::from(format!("{}.hermes-update-new", target_app.display()));
|
|
let old = PathBuf::from(format!("{}.hermes-update-old", target_app.display()));
|
|
remove_dir_if_exists(&tmp).await;
|
|
remove_dir_if_exists(&old).await;
|
|
|
|
let ditto = Command::new("/usr/bin/ditto")
|
|
.arg(&rebuilt_app)
|
|
.arg(&tmp)
|
|
.current_dir(crate::paths::hermes_home())
|
|
.status()
|
|
.await
|
|
.map_err(|e| anyhow!("running ditto: {e}"))?;
|
|
if !ditto.success() {
|
|
return Err(anyhow!(
|
|
"ditto failed while copying updated app into {}",
|
|
tmp.display()
|
|
));
|
|
}
|
|
|
|
// Atomic-as-possible swap with rollback. Extracted so the invariant
|
|
// (target is never left deleted-with-no-replacement) can be unit-tested
|
|
// without ditto / a real .app bundle.
|
|
swap_in_new_bundle(&tmp, target_app, &old).await?;
|
|
|
|
let _ = Command::new("/usr/bin/xattr")
|
|
.arg("-dr")
|
|
.arg("com.apple.quarantine")
|
|
.arg(target_app)
|
|
.current_dir(crate::paths::hermes_home())
|
|
.status()
|
|
.await;
|
|
|
|
Ok(target_app.to_path_buf())
|
|
}
|
|
|
|
/// Move a freshly-staged bundle (`tmp`) into place at `target`, parking any
|
|
/// existing bundle at `old` so the move can succeed (macOS `rename` won't
|
|
/// overwrite a non-empty directory).
|
|
///
|
|
/// Invariant: on ANY failure path, `target` is left pointing at a working
|
|
/// bundle — either the original (rolled back from `old`) or untouched — and we
|
|
/// never delete the running app with no replacement in place. The staged `tmp`
|
|
/// copy is cleaned up on failure.
|
|
async fn swap_in_new_bundle(tmp: &Path, target: &Path, old: &Path) -> Result<()> {
|
|
let moved_old = if target.exists() {
|
|
if let Err(err) = tokio::fs::rename(target, old).await {
|
|
// Could not move the existing app aside. Leave it untouched and
|
|
// bail — a failed update must not brick the install.
|
|
remove_dir_if_exists(tmp).await;
|
|
return Err(anyhow!(
|
|
"could not move existing app aside at {} (leaving it in place): {err}",
|
|
target.display()
|
|
));
|
|
}
|
|
true
|
|
} else {
|
|
false
|
|
};
|
|
if let Err(err) = tokio::fs::rename(tmp, target).await {
|
|
// Restore the original app from the backup so the user keeps a working
|
|
// install, and clean up the staged copy.
|
|
if moved_old {
|
|
let _ = tokio::fs::rename(old, target).await;
|
|
}
|
|
remove_dir_if_exists(tmp).await;
|
|
return Err(anyhow!("installing updated app at {}: {err}", target.display()));
|
|
}
|
|
remove_dir_if_exists(old).await;
|
|
Ok(())
|
|
}
|
|
|
|
#[cfg(not(target_os = "macos"))]
|
|
async fn install_macos_app_update(
|
|
_app: &AppHandle,
|
|
_install_root: &Path,
|
|
target_app: &Path,
|
|
) -> Result<PathBuf> {
|
|
Ok(target_app.to_path_buf())
|
|
}
|
|
|
|
async fn remove_dir_if_exists(path: &Path) {
|
|
if path.exists() {
|
|
let _ = tokio::fs::remove_dir_all(path).await;
|
|
}
|
|
}
|
|
|
|
#[cfg(target_os = "macos")]
|
|
async fn launch_macos_app_and_exit(app: &AppHandle, target_app: &Path) -> Result<()> {
|
|
crate::bootstrap::open_macos_app_detached(target_app)
|
|
.map_err(|e| anyhow!("launching {}: {e}", target_app.display()))?;
|
|
tokio::time::sleep(std::time::Duration::from_millis(150)).await;
|
|
app.exit(0);
|
|
Ok(())
|
|
}
|
|
|
|
#[cfg(not(target_os = "macos"))]
|
|
async fn launch_macos_app_and_exit(_app: &AppHandle, _target_app: &Path) -> Result<()> {
|
|
Ok(())
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Event helpers — keep emit shape identical to bootstrap.rs so the UI is reused
|
|
// ---------------------------------------------------------------------------
|
|
|
|
fn stage_info(name: &str, title: &str) -> StageInfo {
|
|
StageInfo {
|
|
name: name.to_string(),
|
|
title: title.to_string(),
|
|
category: "update".to_string(),
|
|
needs_user_input: false,
|
|
}
|
|
}
|
|
|
|
/// The synthetic update manifest. Mirrors the real operations `run_update`
|
|
/// performs so the progress UI shows them as discrete steps (with the live log
|
|
/// underneath) instead of one monolithic bar. `include_install` adds the macOS
|
|
/// app-swap stage. Both the happy path and the re-entrancy guard build the
|
|
/// manifest here so the two can never drift apart.
|
|
fn update_stages(include_install: bool) -> Vec<StageInfo> {
|
|
let mut stages = vec![
|
|
stage_info("handoff", "Preparing to update"),
|
|
stage_info("update", "Downloading the latest version"),
|
|
stage_info("rebuild", "Rebuilding the desktop app"),
|
|
];
|
|
if include_install {
|
|
stages.push(stage_info("install", "Installing the update"));
|
|
}
|
|
stages
|
|
}
|
|
|
|
// option_env! only accepts string literals, so the build-time pins are read
|
|
// by their literal names here. Mirrors bootstrap.rs's helper of the same name
|
|
// (kept local rather than shared because option_env! can't be parameterized).
|
|
fn option_env_string(key: &str) -> Option<String> {
|
|
let val = match key {
|
|
"BUILD_PIN_COMMIT" => option_env!("BUILD_PIN_COMMIT"),
|
|
"BUILD_PIN_BRANCH" => option_env!("BUILD_PIN_BRANCH"),
|
|
_ => None,
|
|
};
|
|
val.map(|s| s.to_string())
|
|
}
|
|
|
|
fn emit(app: &AppHandle, event: BootstrapEvent) {
|
|
if let Err(e) = app.emit(BootstrapEvent::CHANNEL, &event) {
|
|
tracing::warn!(?e, "failed to emit update event");
|
|
}
|
|
}
|
|
|
|
fn emit_stage(
|
|
app: &AppHandle,
|
|
name: &str,
|
|
state: StageState,
|
|
duration_ms: Option<u64>,
|
|
error: Option<String>,
|
|
) {
|
|
tracing::info!(stage = %name, ?state, ?duration_ms, ?error, "update stage");
|
|
emit(
|
|
app,
|
|
BootstrapEvent::Stage {
|
|
name: name.to_string(),
|
|
state,
|
|
duration_ms,
|
|
result: None,
|
|
error,
|
|
},
|
|
);
|
|
}
|
|
|
|
fn emit_log(app: &AppHandle, stage: Option<&str>, stream: LogStream, line: &str) {
|
|
match stage {
|
|
Some(s) => tracing::info!(target: "bootstrap.log", stage = %s, "{line}"),
|
|
None => tracing::info!(target: "bootstrap.log", "{line}"),
|
|
}
|
|
emit(
|
|
app,
|
|
BootstrapEvent::Log {
|
|
stage: stage.map(|s| s.to_string()),
|
|
line: line.to_string(),
|
|
stream,
|
|
},
|
|
);
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn venv_hermes_is_under_install_root() {
|
|
let root = Path::new("/x/hermes-agent");
|
|
let shim = venv_hermes(root);
|
|
assert!(shim.starts_with(root));
|
|
assert!(shim.to_string_lossy().contains("venv"));
|
|
}
|
|
|
|
#[test]
|
|
fn missing_file_is_not_locked() {
|
|
assert!(!is_locked(Path::new("/nonexistent/does/not/exist/xyz")));
|
|
}
|
|
|
|
#[test]
|
|
fn update_child_env_forces_unbuffered_python() {
|
|
let envs = update_child_env(Path::new("/x/hermes-agent"));
|
|
assert!(
|
|
envs.iter()
|
|
.any(|(k, v)| k == "PYTHONUNBUFFERED" && v.to_str() == Some("1")),
|
|
"update children must run unbuffered so long steps stream to the live log"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn update_child_env_names_our_pid_for_the_lock_handoff() {
|
|
let envs = update_child_env(Path::new("/x/hermes-agent"));
|
|
assert!(
|
|
envs.iter().any(|(k, v)| k == "HERMES_UPDATE_HANDOFF_PID"
|
|
&& v.to_str() == Some(std::process::id().to_string().as_str())),
|
|
"the hermes update child claims the same marker we hold; without our pid \
|
|
it refuses its own parent's lock and every GUI update dead-ends on exit 2"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn lock_probe_paths_include_desktop_app_payload() {
|
|
let root = Path::new("/x/hermes-agent");
|
|
let probes = install_lock_probe_paths(root);
|
|
|
|
assert!(
|
|
probes.iter().any(|p| p == &venv_hermes(root)),
|
|
"venv shim remains part of the update lock probe"
|
|
);
|
|
assert!(
|
|
// Windows/Linux payloads live under `resources/`, the macOS bundle
|
|
// under `Contents/Resources/` — Path::ends_with is case-sensitive.
|
|
probes.iter().any(|p| {
|
|
p.ends_with(Path::new("resources/app.asar"))
|
|
|| p.ends_with(Path::new("Resources/app.asar"))
|
|
}),
|
|
"packaged app.asar must be probed so repair/re-clone waits for the old desktop to exit"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn locked_paths_ignores_missing_payloads() {
|
|
let root = Path::new("/nonexistent/hermes-agent");
|
|
let probes = install_lock_probe_paths(root);
|
|
|
|
assert!(locked_paths(&probes).is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn same_windows_path_accepts_case_only_difference() {
|
|
assert!(same_windows_path(
|
|
Path::new(r"C:\Users\tester\.hermes\hermes-agent\venv\scripts\HERMES.EXE"),
|
|
Path::new(r"c:\users\tester\.hermes\hermes-agent\venv\Scripts\hermes.exe"),
|
|
));
|
|
}
|
|
|
|
#[test]
|
|
fn same_windows_path_rejects_desktop_binary() {
|
|
assert!(!same_windows_path(
|
|
Path::new(r"C:\Users\tester\.hermes\hermes-agent\apps\desktop\Hermes.exe"),
|
|
Path::new(r"C:\Users\tester\.hermes\hermes-agent\venv\Scripts\hermes.exe"),
|
|
));
|
|
}
|
|
|
|
#[test]
|
|
fn update_marker_guard_writes_then_removes_on_drop() {
|
|
let dir = unique_tmp_dir("marker-guard");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
{
|
|
let _g = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
|
|
assert!(marker.exists(), "marker must exist while the guard is held");
|
|
let body = std::fs::read_to_string(&marker).unwrap();
|
|
let pid_line = body.lines().next().unwrap();
|
|
assert_eq!(
|
|
pid_line.trim().parse::<u32>().unwrap(),
|
|
std::process::id(),
|
|
"marker records our pid so the desktop can probe liveness"
|
|
);
|
|
assert_eq!(body.lines().count(), 2, "marker is pid + started_at lines");
|
|
}
|
|
|
|
assert!(
|
|
!marker.exists(),
|
|
"Drop must remove the marker on every exit path (incl. early return / panic unwind)"
|
|
);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn update_marker_guard_drop_is_quiet_when_already_gone() {
|
|
let dir = unique_tmp_dir("marker-guard-gone");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
|
|
// Simulate an external cleanup (e.g. the desktop pruned a marker it
|
|
// judged stale) before our guard drops — Drop must not panic.
|
|
std::fs::remove_file(&marker).unwrap();
|
|
drop(guard);
|
|
|
|
assert!(!marker.exists());
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
/// Spawn a short-lived sibling process whose pid stands in for a foreign
|
|
/// updater. Same-process double-acquire no longer models contention: since
|
|
/// #74761 `acquire` treats our own pid as adoptable (desktop pre-writes it),
|
|
/// so a second acquire in *this* process would succeed.
|
|
fn spawn_foreign_holder() -> std::process::Child {
|
|
#[cfg(windows)]
|
|
{
|
|
std::process::Command::new("timeout")
|
|
.args(["/t", "30", "/nobreak"])
|
|
.stdout(std::process::Stdio::null())
|
|
.stderr(std::process::Stdio::null())
|
|
.spawn()
|
|
.expect("spawn foreign marker holder")
|
|
}
|
|
#[cfg(not(windows))]
|
|
{
|
|
std::process::Command::new("sleep")
|
|
.arg("30")
|
|
.stdout(std::process::Stdio::null())
|
|
.stderr(std::process::Stdio::null())
|
|
.spawn()
|
|
.expect("spawn foreign marker holder")
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn acquire_refuses_while_a_live_updater_owns_the_marker() {
|
|
let dir = unique_tmp_dir("marker-contended");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
// A live *foreign* updater holds it. We must NOT clobber the marker and
|
|
// run concurrently over the same checkout — that race is what let a
|
|
// dashboard `hermes update` and install-mode bootstrap mutate one tree
|
|
// at once. Own-pid markers are adoptable (#74761), so the foreign pid
|
|
// must be a real sibling process.
|
|
let mut foreign = spawn_foreign_holder();
|
|
let foreign_pid = foreign.id();
|
|
let started_at = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0);
|
|
std::fs::write(&marker, format!("{foreign_pid}\n{started_at}")).unwrap();
|
|
|
|
let owner = UpdateMarkerGuard::acquire(marker.clone())
|
|
.err()
|
|
.expect("acquire must be refused while a foreign updater is live");
|
|
assert_eq!(owner.pid, foreign_pid);
|
|
|
|
// The refused guard must not delete the live owner's marker.
|
|
assert!(marker.exists(), "refused acquire must leave the marker intact");
|
|
let _ = foreign.kill();
|
|
let _ = foreign.wait();
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn acquire_adopts_a_marker_prewritten_with_our_own_pid() {
|
|
// #74761: desktop writeUpdateMarker(hermesHome, child.pid) races ahead
|
|
// of UpdateMarkerGuard::acquire. The marker names US; refusing it made
|
|
// every in-app desktop update loop forever. Adopt it without resetting
|
|
// the holder age, so a wedged updater still reaches the stale ceiling.
|
|
let dir = unique_tmp_dir("marker-own-pid");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
let started_at = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0)
|
|
.saturating_sub(2);
|
|
std::fs::write(&marker, format!("{}\n{started_at}", std::process::id())).unwrap();
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone()).unwrap_or_else(|owner| {
|
|
panic!(
|
|
"own-pid pre-write must be adoptable, got foreign owner pid={}",
|
|
owner.pid
|
|
)
|
|
});
|
|
assert!(marker.exists(), "adopted guard must own the marker");
|
|
let body = std::fs::read_to_string(&marker).unwrap();
|
|
assert_eq!(
|
|
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
|
|
std::process::id(),
|
|
"acquire keeps the adopted marker owner"
|
|
);
|
|
assert_eq!(
|
|
body.lines().nth(1).unwrap().trim().parse::<u64>().unwrap(),
|
|
started_at,
|
|
"adopting an own-pid marker must preserve its original holder age"
|
|
);
|
|
drop(guard);
|
|
assert!(
|
|
!marker.exists(),
|
|
"Drop must still clear the marker we adopted"
|
|
);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
// ---- exit-2 self-marker heal (#75788) --------------------------------
|
|
// The deadlock: the updater holds the marker with its own PID; a stale
|
|
// checkout's `hermes update` reads it as a live foreign update and exits
|
|
// 2; the generic retry deliberately skips exit 2 — so the refusal loops
|
|
// forever. These tests pin the heal decision's full contract. On
|
|
// merge-base product code (no heal) the decision function does not exist
|
|
// and the refusal is terminal — the A/B run proves that.
|
|
|
|
#[test]
|
|
fn self_owned_marker_plus_exit_2_heals() {
|
|
let dir = unique_tmp_dir("heal-self-owned");
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
std::fs::write(&marker, format!("{}\n123\n", std::process::id())).unwrap();
|
|
|
|
assert!(
|
|
should_heal_self_marker_refusal(Some(UPDATE_EXIT_CONCURRENT), &marker),
|
|
"a child refusing over OUR marker is the #75788 deadlock — must heal"
|
|
);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn foreign_owned_marker_never_heals() {
|
|
let dir = unique_tmp_dir("heal-foreign");
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
// A live sibling process stands in for a genuinely concurrent updater.
|
|
let mut foreign = spawn_foreign_holder();
|
|
std::fs::write(&marker, format!("{}\n123\n", foreign.id())).unwrap();
|
|
|
|
assert!(
|
|
!should_heal_self_marker_refusal(Some(UPDATE_EXIT_CONCURRENT), &marker),
|
|
"a foreign owner is a REAL concurrent update — the refusal must stand"
|
|
);
|
|
let _ = foreign.kill();
|
|
let _ = foreign.wait();
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn missing_or_garbage_marker_never_heals() {
|
|
let dir = unique_tmp_dir("heal-garbage");
|
|
let missing = dir.join("never-written");
|
|
assert!(
|
|
!should_heal_self_marker_refusal(Some(UPDATE_EXIT_CONCURRENT), &missing),
|
|
"no marker on disk = the child refused over something else entirely"
|
|
);
|
|
|
|
let garbage = dir.join(".hermes-update-in-progress");
|
|
std::fs::write(&garbage, "not-a-pid\n123\n").unwrap();
|
|
assert!(
|
|
!should_heal_self_marker_refusal(Some(UPDATE_EXIT_CONCURRENT), &garbage),
|
|
"an unparseable marker must not be treated as ours"
|
|
);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn non_exit_2_outcomes_never_heal() {
|
|
let dir = unique_tmp_dir("heal-wrong-exit");
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
std::fs::write(&marker, format!("{}\n123\n", std::process::id())).unwrap();
|
|
|
|
for code in [Some(0), Some(1), Some(3), None] {
|
|
assert!(
|
|
!should_heal_self_marker_refusal(code, &marker),
|
|
"heal is exit-2-only; exit {code:?} must keep its normal path"
|
|
);
|
|
}
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn heal_end_to_end_marker_lifecycle() {
|
|
// The full deadlock-and-heal sequence with a REAL marker guard, as
|
|
// run_update executes it: acquire (marker written with our pid) →
|
|
// child exits 2 refusing our own claim → heal decision fires →
|
|
// complete() drops the claim → the retry's precondition (no marker,
|
|
// or a marker the child can now claim) holds.
|
|
let dir = unique_tmp_dir("heal-e2e");
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
|
|
assert!(marker.exists(), "updater holds the marker during the child run");
|
|
|
|
// Stale child refused over our claim:
|
|
assert!(should_heal_self_marker_refusal(
|
|
Some(UPDATE_EXIT_CONCURRENT),
|
|
&marker
|
|
));
|
|
|
|
// The heal drops the claim exactly as run_update does:
|
|
guard.complete();
|
|
assert!(
|
|
!marker.exists(),
|
|
"claim dropped — the one retry now runs with the marker absent"
|
|
);
|
|
|
|
// And with the marker gone the heal can never fire twice (the retry's
|
|
// own exit 2, e.g. a genuinely still-running Hermes, stays terminal).
|
|
assert!(!should_heal_self_marker_refusal(
|
|
Some(UPDATE_EXIT_CONCURRENT),
|
|
&marker
|
|
));
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn acquire_reclaims_a_marker_owned_by_a_dead_pid() {
|
|
let dir = unique_tmp_dir("marker-dead-pid");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
// pid 1 exists everywhere, so fabricate a dead one: a very large pid
|
|
// that no live process owns. A crashed updater must never wedge every
|
|
// future update.
|
|
let started_at = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0);
|
|
std::fs::write(&marker, format!("4294967294\n{started_at}")).unwrap();
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("a dead owner must not block acquisition"));
|
|
let body = std::fs::read_to_string(&marker).unwrap();
|
|
assert_eq!(
|
|
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
|
|
std::process::id(),
|
|
"reclaiming rewrites the marker with our pid"
|
|
);
|
|
drop(guard);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn acquire_reclaims_a_marker_past_the_age_ceiling() {
|
|
let dir = unique_tmp_dir("marker-stale-age");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
// Our own (live) pid, but started well past the ceiling: a wedged
|
|
// updater must not hold the lock forever.
|
|
let long_ago = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_secs())
|
|
.unwrap_or(0)
|
|
.saturating_sub(UPDATE_MARKER_MAX_AGE_SECS + 60);
|
|
std::fs::write(&marker, format!("{}\n{long_ago}", std::process::id())).unwrap();
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("a marker past the ceiling must be reclaimable"));
|
|
drop(guard);
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn completed_update_releases_marker_before_guard_drop() {
|
|
let dir = unique_tmp_dir("marker-complete");
|
|
std::fs::create_dir_all(&dir).unwrap();
|
|
let marker = dir.join(".hermes-update-in-progress");
|
|
|
|
let guard = UpdateMarkerGuard::acquire(marker.clone())
|
|
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
|
|
guard.complete();
|
|
|
|
assert!(
|
|
!marker.exists(),
|
|
"a successful update must unblock desktop startup before relaunch/exit"
|
|
);
|
|
drop(guard);
|
|
assert!(!marker.exists(), "Drop stays idempotent after completion");
|
|
let _ = std::fs::remove_dir_all(&dir);
|
|
}
|
|
|
|
#[test]
|
|
fn parses_update_branch_from_space_or_equals_args() {
|
|
assert_eq!(
|
|
update_branch_from_args(["--update", "--branch", "bb/test"]),
|
|
Some("bb/test".to_string())
|
|
);
|
|
assert_eq!(
|
|
update_branch_from_args(["--update", "--branch=main"]),
|
|
Some("main".to_string())
|
|
);
|
|
assert_eq!(update_branch_from_args(["--update"]), None);
|
|
}
|
|
|
|
#[test]
|
|
fn update_manifest_leads_with_handoff_and_gates_install() {
|
|
let base = update_stages(false);
|
|
assert_eq!(
|
|
base.first().map(|s| s.name.as_str()),
|
|
Some("handoff"),
|
|
"the lock-wait must surface as the first visible step"
|
|
);
|
|
assert!(
|
|
base.iter().any(|s| s.name == "update") && base.iter().any(|s| s.name == "rebuild"),
|
|
"update + rebuild remain distinct stages"
|
|
);
|
|
assert!(
|
|
base.iter().all(|s| s.name != "install"),
|
|
"no app-swap stage unless an install target was passed"
|
|
);
|
|
|
|
let with_install = update_stages(true);
|
|
assert_eq!(
|
|
with_install.last().map(|s| s.name.as_str()),
|
|
Some("install"),
|
|
"the macOS app-swap is the final stage when present"
|
|
);
|
|
assert_eq!(
|
|
with_install.len(),
|
|
base.len() + 1,
|
|
"include_install adds exactly one stage"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn rebuild_retries_only_on_failure() {
|
|
assert!(!rebuild_needs_retry(Some(0)), "a clean rebuild must not retry");
|
|
assert!(rebuild_needs_retry(Some(1)), "a failed rebuild retries once");
|
|
assert!(
|
|
rebuild_needs_retry(None),
|
|
"a killed/signalled rebuild (no exit code) retries once"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn parses_only_app_targets() {
|
|
assert_eq!(
|
|
target_app_from_args(["--update", "--target-app", "/Applications/Hermes.app"]),
|
|
Some(PathBuf::from("/Applications/Hermes.app"))
|
|
);
|
|
assert_eq!(target_app_from_args(["--target-app", "/tmp/not-an-app"]), None);
|
|
}
|
|
|
|
// Helpers for the swap tests: make a throwaway dir tree we can rename.
|
|
fn unique_tmp_dir(tag: &str) -> PathBuf {
|
|
let base = std::env::temp_dir().join(format!(
|
|
"hermes-swap-test-{tag}-{}-{}",
|
|
std::process::id(),
|
|
std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.unwrap()
|
|
.as_nanos()
|
|
));
|
|
std::fs::create_dir_all(&base).unwrap();
|
|
base
|
|
}
|
|
|
|
fn write_marker(dir: &Path, contents: &str) {
|
|
std::fs::create_dir_all(dir).unwrap();
|
|
std::fs::write(dir.join("marker.txt"), contents).unwrap();
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn swap_installs_new_bundle_and_cleans_up() {
|
|
let base = unique_tmp_dir("ok");
|
|
let target = base.join("Hermes.app");
|
|
let tmp = base.join("Hermes.app.hermes-update-new");
|
|
let old = base.join("Hermes.app.hermes-update-old");
|
|
write_marker(&target, "OLD");
|
|
write_marker(&tmp, "NEW");
|
|
|
|
swap_in_new_bundle(&tmp, &target, &old).await.unwrap();
|
|
|
|
// New bundle is now at target; staging + backup dirs are gone.
|
|
assert_eq!(
|
|
std::fs::read_to_string(target.join("marker.txt")).unwrap(),
|
|
"NEW"
|
|
);
|
|
assert!(!tmp.exists(), "staged copy should be cleaned up");
|
|
assert!(!old.exists(), "backup should be cleaned up on success");
|
|
let _ = std::fs::remove_dir_all(&base);
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn swap_failure_never_leaves_target_missing() {
|
|
// Regression guard for the catastrophic path: the move-aside of the
|
|
// existing app fails AND the staged bundle can't be installed. The
|
|
// buggy version deleted `target` when move-aside failed and then
|
|
// skipped rollback, bricking the install. The fixed version must leave
|
|
// the original app intact on disk.
|
|
//
|
|
// Trigger both failures deterministically:
|
|
// - `old` is a NON-EMPTY dir -> rename(target, old) fails
|
|
// - `tmp` does not exist -> rename(tmp, target) fails
|
|
let base = unique_tmp_dir("fail");
|
|
let target = base.join("Hermes.app");
|
|
let tmp = base.join("Hermes.app.hermes-update-new"); // intentionally absent
|
|
let old = base.join("Hermes.app.hermes-update-old");
|
|
write_marker(&target, "OLD");
|
|
write_marker(&old, "OCCUPIED"); // non-empty => rename(target,old) fails
|
|
|
|
let result = swap_in_new_bundle(&tmp, &target, &old).await;
|
|
|
|
assert!(result.is_err(), "swap should fail when neither move can complete");
|
|
assert!(target.exists(), "original app must NOT be deleted on failure");
|
|
assert_eq!(
|
|
std::fs::read_to_string(target.join("marker.txt")).unwrap(),
|
|
"OLD",
|
|
"original app contents must be intact after a failed swap"
|
|
);
|
|
let _ = std::fs::remove_dir_all(&base);
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn swap_rolls_back_when_install_step_fails() {
|
|
// Move-aside succeeds but installing the staged bundle fails (tmp
|
|
// absent). The original must be rolled back from `old` to `target`.
|
|
let base = unique_tmp_dir("rollback");
|
|
let target = base.join("Hermes.app");
|
|
let tmp = base.join("Hermes.app.hermes-update-new"); // absent
|
|
let old = base.join("Hermes.app.hermes-update-old");
|
|
write_marker(&target, "OLD");
|
|
|
|
let result = swap_in_new_bundle(&tmp, &target, &old).await;
|
|
|
|
assert!(result.is_err());
|
|
assert!(target.exists(), "original must be restored after failed install");
|
|
assert_eq!(
|
|
std::fs::read_to_string(target.join("marker.txt")).unwrap(),
|
|
"OLD"
|
|
);
|
|
assert!(!old.exists(), "backup should be rolled back, not left behind");
|
|
let _ = std::fs::remove_dir_all(&base);
|
|
}
|
|
}
|