"""Gateway fleet restart + post-update verification for ``hermes update``. Split out of ``hermes_cli/update_cmd.py``; every name is re-imported there so ``hermes_cli.update_cmd.`` keeps resolving/monkeypatching. Origin helpers are imported lazily inside each function (no import cycle; test patches stay effective). """ import logging from contextlib import suppress import os import subprocess import sys import time as _time from dataclasses import dataclass from pathlib import Path from hermes_cli.update_cmd_common import _best_effort # Log-record parity with the origin module. logger = logging.getLogger("hermes_cli.update_cmd") def _write_gateway_update_exit_code(ok: bool) -> None: from hermes_cli.update_cmd import get_hermes_home path = get_hermes_home() / ".update_exit_code" with suppress(OSError): path.write_text("0" if ok else "1", encoding="utf-8") # Under HERMES_HOME (not next to the venv): records the fleet-restart obligation # after a pull advanced HEAD; cleared only when the restart completes or nothing ran. _FLEET_RESTART_PENDING_NAME = "fleet_restart_pending" def _fleet_restart_pending_marker_path() -> Path: """HERMES_HOME breadcrumb for a pull that has not yet restarted the fleet.""" from hermes_cli.update_cmd import get_hermes_home return get_hermes_home() / _FLEET_RESTART_PENDING_NAME def _write_fleet_restart_pending_marker(*, expected_sha: str = "") -> None: """Drop the pull→restart obligation breadcrumb. Never raises.""" from hermes_cli.update_cmd import _m path = _fleet_restart_pending_marker_path() if _m()._pytest_owns_live_checkout(path.parent): logger.debug("Skipping fleet-restart-pending marker under pytest (live checkout)") return try: lines = [f"started={_time.time()}", f"pid={os.getpid()}"] if expected_sha: lines.append(f"expected_sha={expected_sha}") path.write_text("\n".join(lines) + "\n", encoding="utf-8") except OSError as exc: logger.debug("Could not write fleet-restart-pending marker: %s", exc) def _clear_fleet_restart_pending_marker() -> None: """Remove the pull→restart obligation breadcrumb. Never raises.""" from hermes_cli.update_cmd import _m _m()._clear_marker_file(_fleet_restart_pending_marker_path(), label="fleet-restart-pending") def _current_checkout_sha() -> str | None: """Current on-disk checkout HEAD, or None if it cannot be resolved.""" from hermes_cli.update_cmd import _capture_head_sha, _m try: from hermes_cli.build_info import get_code_identity sha = (get_code_identity(refresh=True) or {}).get("sha") return str(sha) if sha else None except Exception: return _capture_head_sha(["git"], _m().PROJECT_ROOT) def _receipt_looks_unfinished(receipt: dict) -> bool: """True when *receipt* is from an update that did not finish cleanly.""" if receipt.get("stop_reason"): return True exit_code = receipt.get("exit_code") if exit_code not in (0, None): return True outcome = receipt.get("outcome") if outcome in ("failed", "partial", "running"): return True gateway_restart = receipt.get("gateway_restart") if isinstance(gateway_restart, dict) and gateway_restart.get("incomplete"): return True return False def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: """True when ``update_receipts/latest.json`` records a runtime SHA skew. Prefer the post-restart ``fleet`` matrix. ``plan.runtimes[].code_sha`` is captured *before* the pull, so a finished update's plan always looks stale and must not retrigger a restart; consult it only for an unfinished receipt. """ from hermes_cli.update_cmd import _current_checkout_sha try: from hermes_cli.update_receipt import read_latest_receipt receipt = read_latest_receipt() except Exception: receipt = None if not isinstance(receipt, dict): return False if not expected_sha: expected_sha = _current_checkout_sha() if not expected_sha: return False def _sha_mismatch(code_sha) -> bool: return bool(code_sha) and str(code_sha) != str(expected_sha) fleet = receipt.get("fleet") if isinstance(fleet, list) and fleet: for entry in fleet: if not isinstance(entry, dict): continue if entry.get("state") == "stale": return True if _sha_mismatch(entry.get("code_sha")): return True return False if not _receipt_looks_unfinished(receipt): return False plan = receipt.get("plan") if not isinstance(plan, dict): return False for runtime in plan.get("runtimes") or []: if isinstance(runtime, dict) and _sha_mismatch(runtime.get("code_sha")): return True return False def _pending_fleet_restart_needed() -> bool: """True when a prior pull still owes the fleet a restart.""" with suppress(OSError): if _fleet_restart_pending_marker_path().is_file(): return True return _receipt_reports_stale_runtime() def _warn_pending_fleet_restart(*, startup: bool = False) -> None: """Print the specific interrupted-update fleet-restart warning.""" stream = sys.stderr if startup else sys.stdout print( "⚠ A previous `hermes update` pulled new code but did not " "restart running gateways.", file=stream, ) print(" Gateways may still be serving pre-update modules (mixed sys.modules).", file=stream) if startup: print(" Run `hermes update` or `hermes gateway restart`.", file=stream) def _warn_pending_fleet_restart_on_startup() -> None: """Cheap CLI-startup hint. Never restarts; never raises.""" with suppress(Exception): if not _pending_fleet_restart_needed(): return _warn_pending_fleet_restart(startup=True) def _restart_systemd_gateway_units_best_effort(failed: list) -> None: """Best-effort ``systemctl restart`` of every hermes-gateway/serve unit.""" for scope, scope_cmd in ( ("user", ["systemctl", "--user"]), ("system", ["systemctl"]), ): try: result = _systemctl( scope_cmd + ["list-units", "hermes-gateway*", "hermes-serve*", "--plain", "--no-legend", "--no-pager"], timeout=10, ) except (FileNotFoundError, subprocess.TimeoutExpired): continue if result.returncode != 0: continue def process_unit(svc_name: str, _scope=scope, _cmd=scope_cmd) -> None: restart_cmd = list(_cmd) + ["--no-ask-password", "restart", svc_name] if ( _scope == "system" and hasattr(os, "geteuid") and os.geteuid() != 0 # windows-footgun: ok — systemd path, Linux-only ): restart_cmd = ["sudo", "-n"] + restart_cmd _systemctl(restart_cmd, timeout=30) def on_timeout(svc_name: str, exc: subprocess.TimeoutExpired) -> None: failed.append(svc_name) _for_each_systemd_gateway_unit( result.stdout, process_unit=process_unit, on_unit_timeout=on_timeout, ) def _run_pending_fleet_restart() -> bool: """Catch-up restart for gateways left on pre-update code. Never raises. True when the restart completed or nothing was running; False if incomplete. """ from hermes_cli.update_cmd import _m print("→ Restarting gateways left on pre-update code...") with suppress(Exception): _m()._purge_stale_hermes_modules() try: from hermes_cli.gateway import ( find_gateway_pids, is_macos, is_windows, kill_gateway_processes, supports_systemd_services, _wait_for_gateway_exit, ) except Exception as exc: _warn_gateway_restart_phase_aborted(exc, None) return False try: pids = list(find_gateway_pids(all_profiles=True)) except Exception as exc: logger.debug("Pending fleet restart: gateway probe failed: %s", exc) pids = None if pids == []: print(" ✓ No running gateways — nothing to restart.") return True failed: list = [] try: if supports_systemd_services(): _restart_systemd_gateway_units_best_effort(failed) if is_macos(): restarted: list = [] try: _restart_macos_launchd_gateways(restarted, failed, 45.0) except Exception as exc: logger.debug("Pending fleet restart: launchd failed: %s", exc) failed.append("launchd") if is_windows(): try: from hermes_cli import gateway_windows if gateway_windows.is_installed(): gateway_windows.restart() except Exception as exc: logger.debug("Pending fleet restart: Windows failed: %s", exc) failed.append("windows-gateway") leftover: list = [] try: leftover = list(find_gateway_pids(all_profiles=True)) except Exception: leftover = list(pids or []) if leftover: with _best_effort('Pending fleet restart: PID stop failed: %s'): kill_gateway_processes(all_profiles=True) _wait_for_gateway_exit(timeout=5.0, force_after=None) if failed: _warn_incomplete_gateway_fleet_restart(failed) return False print(" ✓ Pending fleet restart completed.") return True except Exception as exc: surviving = None try: surviving = list(find_gateway_pids(all_profiles=True)) except Exception: surviving = pids _warn_gateway_restart_phase_aborted(exc, surviving) return False def _apply_pending_fleet_restart_catchup() -> None: """On an already-up-to-date ``hermes update``, finish a skipped restart. No-op when nothing is pending; exits 1 on incomplete catch-up so automation does not treat the fleet as healthy. """ from hermes_cli.update_cmd import _run_pending_fleet_restart if not _pending_fleet_restart_needed(): return print() _warn_pending_fleet_restart() print("→ Running the pending fleet restart...") if _run_pending_fleet_restart(): _clear_fleet_restart_pending_marker() return print(" ⚠ Fleet restart incomplete. Recover with: hermes gateway restart") sys.exit(1) def _systemctl(cmd: list, *, timeout: float): """Run a systemctl (or sudo systemctl) invocation, capturing utf-8 text with a timeout.""" return subprocess.run( cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, ) def _systemctl_reset_and_restart(manage_cmd: list, svc_name: str): """``reset-failed`` then ``restart``: a unit parked in failed state by systemd's own auto-restart can wedge a plain ``restart`` against RestartSec backoff and stay dead.""" _systemctl(manage_cmd + ["reset-failed", svc_name], timeout=10) return _systemctl(manage_cmd + ["restart", svc_name], timeout=15) def _for_each_systemd_gateway_unit( list_units_stdout: str, *, process_unit, on_unit_timeout, ) -> None: """Process each hermes-gateway*/hermes-serve* unit from ``systemctl list-units``. ``TimeoutExpired`` from ``process_unit`` is isolated per unit via ``on_unit_timeout`` so one wedged systemctl call cannot abort the rest of the fleet. """ for line in (list_units_stdout or "").strip().splitlines(): parts = line.split() if not parts: continue unit = parts[0] if not unit.endswith(".service"): continue # Name gate against stray lines. Exact base unit or hyphenated profile family # only: ``startswith("hermes-serve")`` would accept ``hermes-server.service``. if not ( unit == "hermes-gateway.service" or unit.startswith("hermes-gateway-") or unit == "hermes-serve.service" or unit.startswith("hermes-serve-") ): continue svc_name = unit.removesuffix(".service") try: process_unit(svc_name) except subprocess.TimeoutExpired as exc: on_unit_timeout(svc_name, exc) def _service_unit_supports_graceful_sigusr1_restart(svc_name: str) -> bool: """Whether *svc_name* wires SIGUSR1 to a graceful drain-then-restart. Only ``hermes-gateway*`` runs ``gateway/run.py`` (the handler); SIGUSR1 would just kill ``hermes-serve*`` and burn the drain budget, so those go straight to the blunt restart. Same exact/hyphenated shape as ``_for_each_systemd_gateway_unit`` so a near-prefix unit like ``hermes-gatewayd`` is never signalled. """ return svc_name == "hermes-gateway" or svc_name.startswith("hermes-gateway-") def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None: """Print an explicit incomplete-update warning for unrestarted units.""" from hermes_cli.gateway import is_macos if not failed_units: return # Preserve discovery order while de-duplicating. seen = set() ordered = [] for name in failed_units: if name in seen: continue seen.add(name) ordered.append(name) print() print("⚠ Update incomplete — some units were not restarted:") for name in ordered: print(f" - {name}") if is_macos(): # A label lands here when launchd wasn't supervising a live process after # the restart — likely deregistered, which `launchctl kickstart` can't revive. print(" Listed services may be deregistered from launchd, or still") print(" running pre-update code (mixed sys.modules). Recover with:") print(" hermes gateway status") print(" launchctl list | grep