From 096826bf7ded2170eafe6a0781af22c807fba6c2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:55:50 -0700 Subject: [PATCH] refactor(update): split gateway fleet restart/verify into update_cmd_fleet.py --- hermes_cli/update_cmd.py | 1789 +------------------------------ hermes_cli/update_cmd_fleet.py | 1796 ++++++++++++++++++++++++++++++++ 2 files changed, 1834 insertions(+), 1751 deletions(-) create mode 100644 hermes_cli/update_cmd_fleet.py diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index f2073b3191..bacb283bd2 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -73,6 +73,44 @@ from hermes_cli.update_cmd_windows import ( # noqa: F401 (re-exported; tests p _wait_for_windows_update_gateway_exit, _write_update_planned_stop_marker, ) +from hermes_cli.update_cmd_fleet import ( # noqa: F401 (re-exported; tests patch hermes_cli.update_cmd.) + _FLEET_RESTART_PENDING_NAME, + _FRESH_RESTART_SUPERVISORS, + _GatewayRestartOutcome, + _apply_pending_fleet_restart_catchup, + _clear_fleet_restart_pending_marker, + _current_checkout_sha, + _drain_or_signal_gateway_for_update, + _fleet_probe_expected_runtimes, + _fleet_restart_pending_marker_path, + _for_each_systemd_gateway_unit, + _gateway_recovery_partition, + _gateway_service_matches_profile, + _pending_fleet_restart_needed, + _receipt_looks_unfinished, + _receipt_reports_stale_runtime, + _resolve_manage_cmd, + _restart_gateway_fleet_after_update, + _restart_launchd_gateway_after_update, + _restart_macos_launchd_gateways, + _restart_phase_failure_is_incomplete, + _restart_systemd_gateway_units, + _restart_systemd_gateway_units_best_effort, + _run_pending_fleet_restart, + _service_restart_sec, + _service_unit_supports_graceful_sigusr1_restart, + _surviving_gateway_pids_after_failed_restart, + _systemctl, + _systemctl_reset_and_restart, + _verify_fleet_after_update, + _wait_for_service_active, + _warn_gateway_restart_phase_aborted, + _warn_incomplete_gateway_fleet_restart, + _warn_pending_fleet_restart, + _warn_pending_fleet_restart_on_startup, + _write_fleet_restart_pending_marker, + _write_gateway_update_exit_code, +) logger = logging.getLogger(__name__) @@ -1927,14 +1965,6 @@ def _print_update_summary( return desktop_build_ok and sqlite_runtime_ok -def _write_gateway_update_exit_code(ok: bool) -> None: - path = get_hermes_home() / ".update_exit_code" - try: - path.write_text("0" if ok else "1", encoding="utf-8") - except OSError: - pass - - def _restore_state_db_from_snapshot(state_path: Path, snap_state: Path) -> bool: """Replace *state_path* with the snapshot image at *snap_state*. @@ -3226,286 +3256,6 @@ def _write_lazy_refresh_incomplete_marker() -> None: _write_marker_file(_m()._lazy_refresh_marker_path(), label="lazy-refresh-incomplete") -# Lives under HERMES_HOME (not next to the venv). Unlike the venv-repair -# markers, this records the fleet-restart obligation after a pull advanced -# HEAD (#95294); cleared only when the restart completes or nothing was running. -_FLEET_RESTART_PENDING_NAME = "fleet_restart_pending" - - -def _fleet_restart_pending_marker_path() -> Path: - """HERMES_HOME breadcrumb for a pull that has not yet restarted the fleet.""" - return get_hermes_home() / _FLEET_RESTART_PENDING_NAME - - -def _write_fleet_restart_pending_marker(*, expected_sha: str = "") -> None: - """Drop the pull→restart obligation breadcrumb. Never raises.""" - path = _fleet_restart_pending_marker_path() - if _m()._pytest_owns_live_checkout(path.parent): - logger.debug("Skipping fleet-restart-pending marker under pytest (live checkout)") - return - try: - lines = [f"started={_time.time()}", f"pid={os.getpid()}"] - if expected_sha: - lines.append(f"expected_sha={expected_sha}") - path.write_text("\n".join(lines) + "\n", encoding="utf-8") - except OSError as exc: - logger.debug("Could not write fleet-restart-pending marker: %s", exc) - - -def _clear_fleet_restart_pending_marker() -> None: - """Remove the pull→restart obligation breadcrumb. Never raises.""" - _m()._clear_marker_file( - _fleet_restart_pending_marker_path(), label="fleet-restart-pending" - ) - - -def _current_checkout_sha() -> str | None: - """Current on-disk checkout HEAD, or None if it cannot be resolved.""" - try: - from hermes_cli.build_info import get_code_identity - - sha = (get_code_identity(refresh=True) or {}).get("sha") - return str(sha) if sha else None - except Exception: - return _capture_head_sha(["git"], _m().PROJECT_ROOT) - - -def _receipt_looks_unfinished(receipt: dict) -> bool: - """True when *receipt* is from an update that did not finish cleanly.""" - if receipt.get("stop_reason"): - return True - exit_code = receipt.get("exit_code") - if exit_code not in (0, None): - return True - outcome = receipt.get("outcome") - if outcome in ("failed", "partial", "running"): - return True - gateway_restart = receipt.get("gateway_restart") - if isinstance(gateway_restart, dict) and gateway_restart.get("incomplete"): - return True - return False - - -def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: - """True when ``update_receipts/latest.json`` records a runtime SHA skew. - - Prefer the post-restart ``fleet`` matrix. ``plan.runtimes[].code_sha`` is - captured *before* the pull, so a finished update's plan always shows stale - SHAs and must not retrigger a restart; consult it only for an unfinished - receipt (#95294). - """ - try: - from hermes_cli.update_receipt import read_latest_receipt - - receipt = read_latest_receipt() - except Exception: - receipt = None - if not isinstance(receipt, dict): - return False - if not expected_sha: - expected_sha = _current_checkout_sha() - if not expected_sha: - return False - - def _sha_mismatch(code_sha) -> bool: - return bool(code_sha) and str(code_sha) != str(expected_sha) - - fleet = receipt.get("fleet") - if isinstance(fleet, list) and fleet: - for entry in fleet: - if not isinstance(entry, dict): - continue - if entry.get("state") == "stale": - return True - if _sha_mismatch(entry.get("code_sha")): - return True - return False - - if not _receipt_looks_unfinished(receipt): - return False - plan = receipt.get("plan") - if not isinstance(plan, dict): - return False - for runtime in plan.get("runtimes") or []: - if isinstance(runtime, dict) and _sha_mismatch(runtime.get("code_sha")): - return True - return False - - -def _pending_fleet_restart_needed() -> bool: - """True when a prior pull still owes the fleet a restart (#95294).""" - try: - if _fleet_restart_pending_marker_path().is_file(): - return True - except OSError: - pass - return _receipt_reports_stale_runtime() - - -def _warn_pending_fleet_restart(*, startup: bool = False) -> None: - """Print the specific interrupted-update fleet-restart warning.""" - stream = sys.stderr if startup else sys.stdout - print( - "⚠ A previous `hermes update` pulled new code but did not " - "restart running gateways.", - file=stream, - ) - print( - " Gateways may still be serving pre-update modules (mixed sys.modules).", - file=stream, - ) - if startup: - print( - " Run `hermes update` or `hermes gateway restart`.", - file=stream, - ) - - -def _warn_pending_fleet_restart_on_startup() -> None: - """Cheap CLI-startup hint. Never restarts; never raises.""" - try: - if not _pending_fleet_restart_needed(): - return - _warn_pending_fleet_restart(startup=True) - except Exception: - pass - - -def _restart_systemd_gateway_units_best_effort(failed: list) -> None: - """Best-effort ``systemctl restart`` of every hermes-gateway/serve unit.""" - for scope, scope_cmd in ( - ("user", ["systemctl", "--user"]), - ("system", ["systemctl"]), - ): - try: - result = _systemctl( - scope_cmd + ["list-units", "hermes-gateway*", "hermes-serve*", - "--plain", "--no-legend", "--no-pager"], - timeout=10, - ) - except (FileNotFoundError, subprocess.TimeoutExpired): - continue - if result.returncode != 0: - continue - - def process_unit(svc_name: str, _scope=scope, _cmd=scope_cmd) -> None: - restart_cmd = list(_cmd) + ["--no-ask-password", "restart", svc_name] - if ( - _scope == "system" - and hasattr(os, "geteuid") - and os.geteuid() != 0 # windows-footgun: ok — systemd path, Linux-only - ): - restart_cmd = ["sudo", "-n"] + restart_cmd - _systemctl(restart_cmd, timeout=30) - - def on_timeout(svc_name: str, exc: subprocess.TimeoutExpired) -> None: - failed.append(svc_name) - - _for_each_systemd_gateway_unit( - result.stdout, - process_unit=process_unit, - on_unit_timeout=on_timeout, - ) - - -def _run_pending_fleet_restart() -> bool: - """Catch-up restart for gateways left on pre-update code (#95294). - - Returns True when restart completed or no services were running. - Returns False if restart was incomplete. Never raises. - """ - print("→ Restarting gateways left on pre-update code...") - try: - _m()._purge_stale_hermes_modules() - except Exception: - pass - try: - from hermes_cli.gateway import ( - find_gateway_pids, - is_macos, - is_windows, - kill_gateway_processes, - supports_systemd_services, - _wait_for_gateway_exit, - ) - except Exception as exc: - _warn_gateway_restart_phase_aborted(exc, None) - return False - - try: - pids = list(find_gateway_pids(all_profiles=True)) - except Exception as exc: - logger.debug("Pending fleet restart: gateway probe failed: %s", exc) - pids = None - - if pids == []: - print(" ✓ No running gateways — nothing to restart.") - return True - - failed: list = [] - try: - if supports_systemd_services(): - _restart_systemd_gateway_units_best_effort(failed) - if is_macos(): - restarted: list = [] - try: - _restart_macos_launchd_gateways(restarted, failed, 45.0) - except Exception as exc: - logger.debug("Pending fleet restart: launchd failed: %s", exc) - failed.append("launchd") - if is_windows(): - try: - from hermes_cli import gateway_windows - - if gateway_windows.is_installed(): - gateway_windows.restart() - except Exception as exc: - logger.debug("Pending fleet restart: Windows failed: %s", exc) - failed.append("windows-gateway") - leftover: list = [] - try: - leftover = list(find_gateway_pids(all_profiles=True)) - except Exception: - leftover = list(pids or []) - if leftover: - try: - kill_gateway_processes(all_profiles=True) - _wait_for_gateway_exit(timeout=5.0, force_after=None) - except Exception as exc: - logger.debug("Pending fleet restart: PID stop failed: %s", exc) - if failed: - _warn_incomplete_gateway_fleet_restart(failed) - return False - print(" ✓ Pending fleet restart completed.") - return True - except Exception as exc: - surviving = None - try: - surviving = list(find_gateway_pids(all_profiles=True)) - except Exception: - surviving = pids - _warn_gateway_restart_phase_aborted(exc, surviving) - return False - - -def _apply_pending_fleet_restart_catchup() -> None: - """On an already-up-to-date ``hermes update``, finish a skipped restart. - - No-op when nothing is pending. Exits 1 when the catch-up restart is - incomplete so automation does not treat the fleet as healthy. - """ - if not _pending_fleet_restart_needed(): - return - print() - _warn_pending_fleet_restart() - print("→ Running the pending fleet restart...") - if _run_pending_fleet_restart(): - _clear_fleet_restart_pending_marker() - return - print(" ⚠ Fleet restart incomplete. Recover with: hermes gateway restart") - sys.exit(1) - - def _format_concurrent_instances_message( matches: list[tuple[int, str]], scripts_dir: Path ) -> str: @@ -4969,317 +4719,6 @@ def _defer_update_for_self_lock(loaded: list[str]) -> None: _m()._write_update_incomplete_marker() -def _systemctl(cmd: list, *, timeout: float): - """Run a systemctl (or sudo systemctl) invocation, capturing utf-8 text with a timeout.""" - return subprocess.run( - cmd, - capture_output=True, - text=True, encoding="utf-8", errors="replace", - timeout=timeout, - ) - - -def _systemctl_reset_and_restart(manage_cmd: list, svc_name: str): - """``reset-failed`` then ``restart`` a unit. Always clear failed state first: if - systemd's own auto-restart attempts already parked the unit in a failed state, - a plain ``restart`` can wedge against the RestartSec backoff and leave it dead.""" - _systemctl(manage_cmd + ["reset-failed", svc_name], timeout=10) - return _systemctl(manage_cmd + ["restart", svc_name], timeout=15) - - -def _for_each_systemd_gateway_unit( - list_units_stdout: str, - *, - process_unit, - on_unit_timeout, -) -> None: - """Process each ``hermes-gateway*.service``/``hermes-serve*.service`` unit - from ``systemctl list-units``. - - ``subprocess.TimeoutExpired`` raised by ``process_unit`` is isolated to - that unit via ``on_unit_timeout`` so one wedged systemctl call cannot - abort the rest of the fleet (#68523). - """ - for line in (list_units_stdout or "").strip().splitlines(): - parts = line.split() - if not parts: - continue - unit = parts[0] - if not unit.endswith(".service"): - continue - # list-units is already pattern-filtered, but keep the name gate so a - # stray line cannot enter the restart path. Require the exact base unit - # or hyphenated profile family: ``startswith("hermes-serve")`` would - # also accept the unrelated ``hermes-server.service`` (#83595). - if not ( - unit == "hermes-gateway.service" - or unit.startswith("hermes-gateway-") - or unit == "hermes-serve.service" - or unit.startswith("hermes-serve-") - ): - continue - svc_name = unit.removesuffix(".service") - try: - process_unit(svc_name) - except subprocess.TimeoutExpired as exc: - on_unit_timeout(svc_name, exc) - -def _service_unit_supports_graceful_sigusr1_restart(svc_name: str) -> bool: - """Whether *svc_name* wires SIGUSR1 to a graceful drain-then-restart. - - Only ``hermes-gateway*`` units run ``gateway/run.py`` (the SIGUSR1 - handler). ``hermes-serve*`` units (#83438) don't: SIGUSR1 would just - terminate them and burn the full drain budget, so they go straight to the - blunt ``systemctl restart`` path. - - Same strict exact/hyphenated shape as the unit-name gate in - ``_for_each_systemd_gateway_unit``, so a near-prefix unit like - ``hermes-gatewayd`` can't be sent a SIGUSR1 it doesn't handle. - """ - return svc_name == "hermes-gateway" or svc_name.startswith("hermes-gateway-") - - -def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None: - """Print an explicit incomplete-update warning for unrestarted units.""" - from hermes_cli.gateway import is_macos - - if not failed_units: - return - # Preserve discovery order while de-duplicating. - seen = set() - ordered = [] - for name in failed_units: - if name in seen: - continue - seen.add(name) - ordered.append(name) - print() - print("⚠ Update incomplete — some units were not restarted:") - for name in ordered: - print(f" - {name}") - if is_macos(): - # A launchd label lands here when launchd was not supervising a live - # process after the restart (#88848) — very likely deregistered, which - # `launchctl kickstart` cannot revive. - print(" Listed services may be deregistered from launchd, or still") - print(" running pre-update code (mixed sys.modules). Recover with:") - print(" hermes gateway status") - print(" launchctl list | grep