"""Gateway subcommand for hermes CLI. Handles: hermes gateway [run|start|stop|restart|status|install|uninstall|setup] """ import asyncio import contextlib from hermes_cli.cli_output import line_input import json import logging import os import shlex import shutil import signal import socket import subprocess import sys import textwrap import time from dataclasses import dataclass from pathlib import Path # UV's bundled Python ships a minimal PATH; ensure launchctl/systemctl are discoverable. if os.name == "posix": _sys_dirs = {"/bin", "/usr/bin", "/usr/sbin", "/sbin"} _path_dirs = set(os.environ.get("PATH", "").split(os.pathsep)) _missing = _sys_dirs - _path_dirs if _missing: os.environ["PATH"] = os.environ.get("PATH", "") + os.pathsep + os.pathsep.join(sorted(_missing)) PROJECT_ROOT = Path(__file__).parent.parent.resolve() from gateway.config import coerce_systemd_watchdog_seconds, load_gateway_config from gateway.status import terminate_pid from gateway.restart import ( DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT, EXTERNAL_GATEWAY_SUPERVISOR_ENV, GATEWAY_FATAL_CONFIG_EXIT_CODE, GATEWAY_SERVICE_RESTART_EXIT_CODE, is_gateway_supervisor_process, parse_cron_drain_timeout, parse_restart_after_turn_timeout, parse_restart_drain_timeout, resolve_restart_exit_wait_budget, resolve_systemd_timeout_stop_sec, ) from hermes_cli.config import ( get_env_value, get_hermes_home, is_managed, managed_error, read_raw_config, save_env_value, write_platform_config_field, ) # display_hermes_home is imported lazily: hermes_constants may be a cached pre-update version. from hermes_cli.setup import ( print_header, print_info, print_success, print_warning, print_error, prompt, prompt_choice, prompt_yes_no, ) from hermes_cli.colors import Colors, color logger = logging.getLogger(__name__) # Shared ``subprocess.run`` kwargs for text-mode probes (stdout/stderr captured, decode-tolerant). _CAPTURE_TEXT = dict(capture_output=True, text=True, encoding="utf-8", errors="replace") # ============================================================================= # Process Management (for manual gateway runs) # ============================================================================= @dataclass(frozen=True) class GatewayRuntimeSnapshot: manager: str service_installed: bool = False service_running: bool = False gateway_pids: tuple[int, ...] = () service_scope: str | None = None @property def running(self) -> bool: return self.service_running or bool(self.gateway_pids) @property def has_process_service_mismatch(self) -> bool: return self.service_installed and self.running and not self.service_running @dataclass(frozen=True) class ProfileGatewayProcess: profile: str path: Path pid: int create_time: float = 0.0 @dataclass(frozen=True) class WindowsGatewayService: """A real Windows service supervising a profile gateway process tree.""" name: str profile: str service_pid: int gateway_pid: int descendant_pids: frozenset[int] descendant_identities: tuple[tuple[int, float], ...] service_create_time: float = 0.0 gateway_create_time: float = 0.0 def _get_service_pids(all_profiles: bool = False) -> set: """Return PIDs managed by systemd/launchd gateway services (excluded from stale-process sweeps). Relies on the service manager committing the new PID before the restart command returns. Default scope covers only the current profile's unit/label; ``all_profiles`` widens to the whole ``hermes-gateway*`` / ``ai.hermes.gateway*`` fleet so the update path and orphan reaper never misclassify a sibling profile's service gateway as a manual process and kill it. """ pids: set = set() # --- systemd (Linux): user and system scopes --- if supports_systemd_services(): pattern = "hermes-gateway*" if all_profiles else get_service_name() for scope_args in [["systemctl", "--user"], ["systemctl"]]: try: result = subprocess.run( scope_args + ["list-units", pattern, "--plain", "--no-legend", "--no-pager"], timeout=5, **_CAPTURE_TEXT, ) for line in result.stdout.strip().splitlines(): parts = line.split() if not parts or not parts[0].endswith(".service"): continue svc = parts[0] try: show = subprocess.run( scope_args + ["show", svc, "--property=MainPID", "--value"], timeout=5, **_CAPTURE_TEXT, ) pid = int(show.stdout.strip()) if pid > 0: pids.add(pid) except (ValueError, subprocess.TimeoutExpired): pass except (FileNotFoundError, subprocess.TimeoutExpired): pass # --- launchd (macOS) --- if is_macos(): labels = {get_launchd_label()} if all_profiles: # Whole fleet, mirroring the systemd ``hermes-gateway*`` glob above. labels.update(launchd_gateway_labels_for_install()) for label in sorted(labels): try: _domain, pid = _locate_launchd_gateway_service(label) except subprocess.TimeoutExpired: continue if pid is not None and pid > 0: pids.add(pid) if all_profiles: # Prefix scan also catches ai.hermes.gateway* agents the label derivation can't map # (renamed profiles, other installs). Over-inclusion is safe: PIDs are only protected. try: result = subprocess.run(["launchctl", "list"], timeout=5, **_CAPTURE_TEXT) if result.returncode == 0: for line in result.stdout.strip().splitlines(): parts = line.split() if len(parts) >= 3 and parts[-1].startswith("ai.hermes.gateway"): try: pid = int(parts[0]) if pid > 0: pids.add(pid) except ValueError: pass except (FileNotFoundError, subprocess.TimeoutExpired): pass return pids def _get_parent_pid(pid: int) -> int | None: """Return the parent PID for ``pid``, or ``None``. psutil first (works on Windows, where ``ps`` doesn't).""" if pid <= 1: return None try: import psutil # type: ignore return psutil.Process(pid).ppid() or None except ImportError: pass except Exception: return None # ps fallback, POSIX only: Git Bash's ps.exe would flash a console from the windowless backend. if is_windows(): return None if not shutil.which("ps"): return None try: result = subprocess.run(["ps", "-o", "ppid=", "-p", str(pid)], timeout=5, **_CAPTURE_TEXT) except (FileNotFoundError, subprocess.TimeoutExpired): return None raw = result.stdout.strip() if result.returncode != 0 or not raw: return None try: parent_pid = int(raw.splitlines()[-1].strip()) except ValueError: return None return parent_pid if parent_pid > 0 else None def _is_pid_ancestor_of_current_process(target_pid: int) -> bool: """Return True when ``target_pid`` is this process or one of its ancestors.""" if target_pid <= 0: return False pid = os.getpid() seen: set[int] = set() while pid and pid not in seen: if pid == target_pid: return True seen.add(pid) pid = _get_parent_pid(pid) or 0 return False def _request_gateway_self_restart(pid: int) -> bool: """Ask a running gateway ancestor to restart itself asynchronously.""" if not hasattr(signal, "SIGUSR1"): return False if not _is_pid_ancestor_of_current_process(pid): return False try: os.kill(pid, signal.SIGUSR1) # windows-footgun: ok — POSIX signal, guarded by hasattr(signal, 'SIGUSR1') above except (ProcessLookupError, PermissionError, OSError): return False return True def _graceful_restart_via_sigusr1(pid: int, drain_timeout: float) -> bool: """Send SIGUSR1 (drain-aware restart) to a gateway PID and wait for it to exit. gateway/run.py maps SIGUSR1 to ``request_restart(via_service=True)``: refuse new turns, wait for in-flight work, ``stop()``, exit; systemd/launchd then relaunch. ``drain_timeout`` must cover the after-turn wait plus the drain — pass ``resolve_restart_exit_wait_budget(...)``. Returns False if the signal couldn't be sent or the process outlived the timeout. """ if not hasattr(signal, "SIGUSR1"): return False if pid <= 0: return False try: os.kill(pid, signal.SIGUSR1) # windows-footgun: ok — POSIX signal, guarded by hasattr(signal, 'SIGUSR1') above except ProcessLookupError: return True except (PermissionError, OSError): return False return _wait_for_pid_exit(pid, max(drain_timeout, 1.0)) def _wait_for_pid_exit(pid: int, timeout: float) -> bool: """Wait up to ``timeout``s for ``pid`` to exit; True once gone, False on timeout. ``launchctl bootstrap`` fails with EIO while the previous instance is still draining, so teardown callers must wait for the real exit before re-bootstrapping. """ if pid <= 0: return True import time as _time # ``os.kill(pid, 0)`` hard-kills on Windows (TerminateProcess); use _pid_exists instead. from gateway.status import _pid_exists deadline = _time.monotonic() + max(timeout, 0.0) while True: if not _pid_exists(pid): return True if _time.monotonic() >= deadline: return False _time.sleep(0.5) # --- Wedged-gateway detection + bounded escalation --------------------------- # # A gateway whose asyncio loop is stalled cannot handle SIGTERM/SIGUSR1, so the drain wait burns # its full budget and `hermes update` can deadlock. Two witnesses classify the loop BEFORE any # drain wait: the heartbeat file ``state/gateway.heartbeat`` (rewritten every 30s, but on a thread # — so freshness/staleness alone is not proof) and the loop-tick socket # ``state/gateway.loop-tick..sock``, answered by the loop itself; the payload records whether # the socket is armed (``loop_tick_socket``). # # - ``alive`` — socket answered, or file fresh and not contradicted. Normal graceful drain, # which honours the in-flight cron drain floor. # - ``wedged`` — heartbeat is this PID's, stale past several beats, AND the armed socket stays # silent across ``tick_strikes`` consecutive misses. Only then may callers # escalate via ``_escalate_wedged_gateway``; one silent probe is never authority. # - ``unknown`` — no/unreadable heartbeat, PID mismatch, or witness conflict. Treated as alive: # never escalate on ambiguity. # # Legacy payloads (no ``loop_tick_socket`` flag) wrote on-loop, so staleness alone remains proof. GATEWAY_LOOP_ALIVE = "alive" GATEWAY_LOOP_WEDGED = "wedged" GATEWAY_LOOP_UNKNOWN = "unknown" # 3 missed 30s beats (gateway.shutdown_watchdog.DEFAULT_HEARTBEAT_INTERVAL_S): decisive, not one slow write. DEFAULT_LOOP_LIVENESS_STALE_AFTER_S = 90.0 # Sentinel for "the producer never wrote the witness flag" (legacy payload). _LOOP_TICK_ABSENT = object() def _probe_loop_tick_socket(pid: int, home: Path | None, timeout: float = 1.0) -> bool | None: """Ping the loop-tick witness socket: True answered, False node present but silent, None no node (not evidence).""" try: from gateway.shutdown_watchdog import get_loop_tick_socket_path path = get_loop_tick_socket_path(home, pid) if not path.is_socket(): return None except Exception: return None return _ping_loop_tick_witness(socket.AF_UNIX, str(path), timeout) def _ping_loop_tick_witness(family: int, address, timeout: float) -> bool: """Connect to a loop-tick witness and expect one byte ``"1"``; False on refusal/timeout/any error.""" sock = None try: sock = socket.socket(family, socket.SOCK_STREAM) sock.settimeout(max(float(timeout), 0.0)) sock.connect(address) return sock.recv(1) == b"1" except Exception: return False finally: if sock is not None: with contextlib.suppress(Exception): sock.close() def _probe_loop_tick_tcp(port: int, timeout: float = 1.0) -> bool | None: """TCP-loopback variant of the tick probe for Windows (no AF_UNIX in asyncio); same semantics, None on invalid port.""" try: port_num = int(port) if port_num <= 0 or port_num > 65535: return None except (TypeError, ValueError): return None return _ping_loop_tick_witness(socket.AF_INET, ("127.0.0.1", port_num), timeout) def _probe_loop_tick_socket_sustained( pid: int, home: Path | None, *, timeout: float = 1.0, strikes: int = 3, gap_s: float = 0.2, tcp_port: int | None = None, ) -> bool | None: """Probe the tick socket up to ``strikes`` times, ``gap_s`` apart, until a reply. One silent probe is not destructive evidence (a transient synchronous stall can outlast one recv timeout). True: some attempt answered. False: a node stayed silent the whole window. None: no socket node on some attempt (vanished / legacy producer) — not evidence. """ total = max(int(strikes), 0) for attempt in range(total): if tcp_port is not None: result = _probe_loop_tick_tcp(tcp_port, timeout=timeout) else: result = _probe_loop_tick_socket(pid, home, timeout=timeout) if result is True: return True if result is None: # No node: ambiguity, never a wedge — absence is not a miss. return None if attempt < total - 1 and gap_s > 0: time.sleep(gap_s) return False def probe_gateway_loop_liveness( pid: int, *, stale_after: float = DEFAULT_LOOP_LIVENESS_STALE_AFTER_S, home: Path | None = None, tick_timeout: float = 1.0, tick_strikes: int = 3, tick_gap_s: float = 0.2, ) -> str: """Classify a gateway PID's event loop as alive / wedged / unknown (see block comment above). A stale heartbeat is ``wedged`` only when the payload declares the tick socket armed AND the socket stays silent across ``tick_strikes`` consecutive misses; any answer is ``alive``; any conflict or ambiguity is ``unknown`` so callers keep the graceful-drain path. """ try: stale_budget = max(float(stale_after), 0.0) except (TypeError, ValueError): stale_budget = DEFAULT_LOOP_LIVENESS_STALE_AFTER_S try: from gateway.shutdown_watchdog import get_loop_heartbeat_path path = get_loop_heartbeat_path(home) mtime = path.stat().st_mtime payload = json.loads(path.read_text(encoding="utf-8")) heartbeat_pid = int(payload.get("pid", 0)) except Exception: return GATEWAY_LOOP_UNKNOWN if heartbeat_pid <= 0 or int(pid) <= 0 or heartbeat_pid != int(pid): # Heartbeat is not this process's (old version, starting up, stale file): not evidence. return GATEWAY_LOOP_UNKNOWN # TCP loopback witness (Windows) takes priority when published; else the AF_UNIX socket. tcp_port = payload.get("loop_tick_tcp_port") try: tcp_port_int = int(tcp_port) if tcp_port is not None else None except (TypeError, ValueError): tcp_port_int = None if tcp_port_int is not None and tcp_port_int > 0: witness = _probe_loop_tick_tcp(tcp_port_int, timeout=tick_timeout) tick_armed = True else: witness = _probe_loop_tick_socket(pid, home, timeout=tick_timeout) tick_armed = payload.get("loop_tick_socket", _LOOP_TICK_ABSENT) if witness is True: # Loop answered: a stale file is a stalled write, not a wedge. return GATEWAY_LOOP_ALIVE age = time.time() - mtime if age <= stale_budget: if witness is False: # Fresh file but silent loop: an off-loop write can land after the loop froze. return GATEWAY_LOOP_UNKNOWN return GATEWAY_LOOP_ALIVE # Stale past the budget; the verdict depends on what the producer promised about its witness. if tick_armed is _LOOP_TICK_ABSENT: # Legacy on-loop writer: staleness proves the loop stopped scheduling. return GATEWAY_LOOP_WEDGED if tick_armed is not True: # Witness could not be armed (bind failed); off-loop write means staleness is not proof. return GATEWAY_LOOP_UNKNOWN if witness is False: # First miss. The probe above is miss #1, so ``tick_strikes - 1`` more attempts follow. sustained = _probe_loop_tick_socket_sustained( pid, home, timeout=tick_timeout, strikes=tick_strikes - 1, gap_s=tick_gap_s, tcp_port=tcp_port_int, ) if sustained is False: return GATEWAY_LOOP_WEDGED if sustained is True: # Transient stall, not a wedge. return GATEWAY_LOOP_ALIVE # Witness vanished mid-window: ambiguity — never kill on it. return GATEWAY_LOOP_UNKNOWN # Armed but unreachable socket: ambiguity — never kill on it. return GATEWAY_LOOP_UNKNOWN def _escalate_wedged_gateway(pid: int, *, term_grace: float = 5.0, kill_wait: float = 5.0) -> bool: """Bounded stop (SIGTERM, ``term_grace``, SIGKILL, ``kill_wait``) for a provably dead loop. Callers MUST have classified the gateway ``GATEWAY_LOOP_WEDGED`` first: escalating a merely busy gateway bypasses the cron drain floor and SIGKILLs live work. True once the PID is gone. """ from gateway.status import get_process_start_time expected_start_time = get_process_start_time(pid) try: terminate_pid(pid, force=False) except (ProcessLookupError, PermissionError, OSError): return _wait_for_pid_exit(pid, 1.0) if _wait_for_pid_exit(pid, max(float(term_grace), 0.0)): return True try: terminate_pid(pid, force=True, expected_start_time=expected_start_time) print(f"⚠ Gateway PID {pid} unresponsive to SIGTERM; sent SIGKILL") except (ProcessLookupError, PermissionError, OSError): pass return _wait_for_pid_exit(pid, max(float(kill_wait), 0.0)) def _get_ancestor_pids() -> set[int]: """PIDs of this process and its ancestors, so scans never count the invoking ``hermes`` CLI as a gateway.""" ancestors: set[int] = set() pid = os.getpid() for _ in range(64): ancestors.add(pid) parent = _get_parent_pid(pid) if parent is None or parent <= 0 or parent in ancestors: break pid = parent return ancestors def _append_unique_pid(pids: list[int], pid: int | None, exclude_pids: set[int]) -> None: if pid is None or pid <= 0: return if pid == os.getpid() or pid in exclude_pids or pid in pids: return pids.append(pid) def _scan_gateway_pids( exclude_pids: set[int], all_profiles: bool = False, include_restart_managers: bool = False, ) -> list[int]: """Best-effort process-table scan for gateway PIDs (backs up a stale/missing PID file; ``--all`` sweeps).""" exclude_pids = exclude_pids | _get_ancestor_pids() pids: list[int] = [] # Strict matcher shared with gateway.status: requires a real ``gateway run`` argv, so # ``gateway status``/``dashboard`` siblings and ``python -m tui_gateway`` don't match. from gateway.status import ( looks_like_gateway_command_line, looks_like_gateway_runtime_command_line, ) current_home = str(get_hermes_home().resolve()) # Forward slashes on both sides of the HERMES_HOME= match (mirrors gateway.status). current_home_lc = current_home.lower().replace("\\", "/") current_profile_arg = _profile_arg(current_home) current_profile_name = (current_profile_arg.split()[-1] if current_profile_arg else "") current_profile_name_lc = current_profile_name.lower() def _matches_current_profile(command: str) -> bool: command_lc = command.lower().replace("\\", "/") if current_profile_name: return ( f"--profile {current_profile_name_lc}" in command_lc or f"-p {current_profile_name_lc}" in command_lc or f"hermes_home={current_home_lc}" in command_lc ) # Default profile: accept unless argv advertises another profile. HERMES_HOME may come via # env (invisible to wmic/CIM), so only a non-matching explicit HERMES_HOME= disqualifies. if "--profile " in command_lc or " -p " in command_lc: return False return not ( "hermes_home=" in command_lc and f"hermes_home={current_home_lc}" not in command_lc ) def _matches_gateway_runtime(command: str) -> bool: if looks_like_gateway_command_line(command): return True return include_restart_managers and looks_like_gateway_runtime_command_line(command) def _consider(pid: int, command: str) -> None: if _matches_gateway_runtime(command) and ( all_profiles or _matches_current_profile(command) ): _append_unique_pid(pids, pid, exclude_pids) try: if is_windows(): listing = _windows_process_listing() if listing is None: return [] for pid, command in _iter_windows_list_processes(listing): _consider(pid, command) else: # /proc first (Docker without procps), then `ps -Aww`. _found_via_proc = False if os.path.isdir("/proc"): try: my_pid = os.getpid() for entry in os.listdir("/proc"): if not entry.isdigit(): continue pid = int(entry) if pid == my_pid or pid in exclude_pids: continue try: with open(f"/proc/{pid}/cmdline", "rb") as _f: cmdline = _f.read().decode("utf-8", errors="replace") _consider(pid, cmdline.replace("\x00", " ")) except (OSError, PermissionError): continue _found_via_proc = True except Exception: pass if not _found_via_proc: # ``-Aww`` not ``-A eww``: BSD/macOS ps rejects ``e``; ``-ww`` = unlimited width. result = subprocess.run(["ps", "-Aww", "-o", "pid=,command="], timeout=10, **_CAPTURE_TEXT) if result.returncode != 0: return [] for line in result.stdout.split("\n"): parsed = _parse_ps_line(line) if parsed is not None: _consider(*parsed) except (OSError, subprocess.TimeoutExpired): return [] # Windows: a venv ``pythonw.exe`` is a launcher stub that spawns the base Python with the same # command line, so each gateway yields two matched PIDs. Drop a matched PID that parents another. if is_windows() and len(pids) > 1: pids = _filter_venv_launcher_stubs(pids) return pids def _parse_ps_line(line: str) -> tuple[int, str] | None: """``(pid, command)`` from one ``ps -o pid=,command=`` line; also accepts ``ps aux`` rows.""" stripped = line.strip() if not stripped or "grep" in stripped: return None parts = stripped.split(None, 1) if len(parts) == 2: with contextlib.suppress(ValueError): return int(parts[0]), parts[1] aux_parts = stripped.split() if len(aux_parts) > 10 and aux_parts[1].isdigit(): return int(aux_parts[1]), " ".join(aux_parts[10:]) return None def _iter_windows_list_processes(listing: str): """Yield ``(pid, command_line)`` from wmic/CIM ``/FORMAT:LIST`` output.""" current_cmd = "" for line in listing.split("\n"): line = line.strip() if line.startswith("CommandLine="): current_cmd = line[len("CommandLine=") :] elif line.startswith("ProcessId="): with contextlib.suppress(ValueError): yield int(line[len("ProcessId=") :]), current_cmd current_cmd = "" def _windows_process_listing() -> str | None: """``CommandLine=``/``ProcessId=`` LIST output for every Windows process, or None. wmic when present, else Get-CimInstance emitting the same shape. ``bounded_probe_run``, NOT ``subprocess.run(timeout=...)``: on Windows run()'s post-timeout cleanup joins pipe readers unbounded and a conhost.exe holding duplicated handles wedges the caller forever; it also hides the console window this windowless pythonw backend would otherwise flash. """ from hermes_cli._subprocess_compat import bounded_probe_run wmic_path = shutil.which("wmic") result = None if wmic_path is not None: result = bounded_probe_run( [wmic_path, "process", "get", "ProcessId,CommandLine", "/FORMAT:LIST"], timeout=10, errors="ignore", ) if result is None or result.returncode != 0 or not (result.stdout or ""): powershell = shutil.which("powershell") or shutil.which("pwsh") if powershell is None: return None ps_cmd = ( "Get-CimInstance Win32_Process | " "ForEach-Object { " " 'CommandLine=' + ($_.CommandLine -replace \"`r`n\",' ' -replace \"`n\",' '); " " 'ProcessId=' + $_.ProcessId; " " '' " "}" ) result = bounded_probe_run( [powershell, "-NoProfile", "-Command", ps_cmd], timeout=15, errors="ignore", ) if result is None: return None if result.returncode != 0 or result.stdout is None: return None return result.stdout def _filter_venv_launcher_stubs(pids: list[int]) -> list[int]: """Drop venv-launcher ``pythonw.exe`` stubs that parent another matched PID (see ``_scan_gateway_pids``).""" try: import psutil # type: ignore except ImportError: return pids pid_set = set(pids) parent_of: dict[int, int | None] = {} for pid in pids: try: parent_of[pid] = psutil.Process(pid).ppid() except (psutil.NoSuchProcess, psutil.AccessDenied): parent_of[pid] = None drop: set[int] = set() for pid, ppid in parent_of.items(): if ppid is not None and ppid in pid_set: drop.add(ppid) return [p for p in pids if p not in drop] def find_gateway_pids(exclude_pids: set | None = None, all_profiles: bool = False) -> list: """Find running gateway PIDs for the current profile, or every profile with ``all_profiles`` (``hermes update``).""" _exclude = set(exclude_pids or set()) pids: list[int] = [] if not all_profiles: try: from gateway.status import get_running_pid _append_unique_pid(pids, get_running_pid(), _exclude) except Exception: pass for pid in _get_service_pids(all_profiles=all_profiles): _append_unique_pid(pids, pid, _exclude) try: include_restart_managers = not supports_systemd_services() except Exception: include_restart_managers = False for pid in _scan_gateway_pids( _exclude, all_profiles=all_profiles, include_restart_managers=include_restart_managers, ): _append_unique_pid(pids, pid, _exclude) return pids def find_profile_gateway_processes( exclude_pids: set | None = None, *, strict: bool = False, ) -> list[ProfileGatewayProcess]: """Return running gateway PIDs mapped to Hermes profiles via PID files.""" _exclude = set(exclude_pids or set()) processes: list[ProfileGatewayProcess] = [] try: from gateway.status import get_running_pid, get_running_pid_identity_strict from hermes_cli.profiles import list_profiles except Exception: if strict: raise return processes seen: set[int] = set() try: profiles = list_profiles() except Exception: if strict: raise return processes for profile in profiles: try: if strict: identity = get_running_pid_identity_strict(profile.path / "gateway.pid") pid = identity[0] if identity else None create_time = identity[1] if identity else 0.0 else: pid = get_running_pid(profile.path / "gateway.pid", cleanup_stale=False) create_time = 0.0 except Exception as exc: if strict: raise RuntimeError(f"Could not inspect gateway PID for profile {profile.name}") from exc continue if pid is None or pid <= 0 or pid in _exclude or pid in seen: continue seen.add(pid) processes.append( ProfileGatewayProcess(profile=profile.name, path=profile.path, pid=pid, create_time=create_time) ) return processes def find_windows_gateway_services( *, psutil_module=None, profile_processes: list[ProfileGatewayProcess] | None = None, ) -> list[WindowsGatewayService]: """Find profile gateways supervised by real Windows services. Service-logon processes may hide their command lines, so identity comes from Hermes's own PID file plus a parent chain ending at a running SCM service PID. The whole service subtree is returned so the Desktop preflight exempts exactly what the updater stops through the SCM. """ if sys.platform != "win32": return [] try: if psutil_module is None: import psutil as psutil_module # type: ignore[no-redef] # noqa: PLC0415 if profile_processes is None: profile_processes = find_profile_gateway_processes(strict=True) service_names_by_pid: dict[int, set[str]] = {} indeterminate_services_by_pid: dict[int, list[tuple[str, object]]] = {} for service in psutil_module.win_service_iter(): try: if all(callable(getattr(service, field, None)) for field in ("name", "status", "pid")): service_name = str(service.name() or "") service_status = service.status() service_pid = int(service.pid() or 0) else: data = service.as_dict() service_name = str(data.get("name") or "") service_status = data.get("status") service_pid = int(data.get("pid") or 0) except FileNotFoundError: # Deleted between enumeration and inspection. continue except Exception as exc: raise RuntimeError("SCM service inspection failed") from exc if not service_name: raise RuntimeError("SCM service has an empty name") if service_status == "stopped": continue if service_status != "running": if service_pid > 0: indeterminate_services_by_pid.setdefault(service_pid, []).append( (service_name, service_status) ) continue if service_pid <= 0: raise RuntimeError(f"Running SCM service {service_name} has no valid process ID") service_names_by_pid.setdefault(service_pid, set()).add(service_name) except Exception as exc: raise RuntimeError("SCM service enumeration failed") from exc found: dict[str, WindowsGatewayService] = {} for profile_process in profile_processes: try: gateway_process = psutil_module.Process(int(profile_process.pid)) gateway_create_time = float(gateway_process.create_time()) if profile_process.create_time <= 0 or abs( gateway_create_time - profile_process.create_time ) > 0.001: raise RuntimeError("Gateway process identity changed during SCM discovery") ancestor_pids = [int(parent.pid) for parent in gateway_process.parents()] for pid in ancestor_pids: indeterminate_services = indeterminate_services_by_pid.get(pid, []) if indeterminate_services: service_name, service_status = indeterminate_services[0] raise RuntimeError( f"SCM service {service_name} has indeterminate status: " f"{service_status}" ) shared_service_pids = [ pid for pid in ancestor_pids if len(service_names_by_pid.get(pid, set())) > 1 ] if shared_service_pids: raise RuntimeError( "Gateway ownership is ambiguous under shared SCM host PID(s): " + ", ".join(str(pid) for pid in shared_service_pids) ) service_pid = next( (pid for pid in ancestor_pids if len(service_names_by_pid.get(pid, set())) == 1), None, ) if service_pid is None: continue service_name = next(iter(service_names_by_pid[service_pid])) service_process = psutil_module.Process(service_pid) service_create_time = float(service_process.create_time()) descendant_processes = service_process.children(recursive=True) descendants = frozenset(int(child.pid) for child in descendant_processes) if int(profile_process.pid) not in descendants: continue descendant_identities = tuple( sorted((int(child.pid), float(child.create_time())) for child in descendant_processes) ) found[service_name] = WindowsGatewayService( name=service_name, profile=str(profile_process.profile), service_pid=service_pid, gateway_pid=int(profile_process.pid), descendant_pids=descendants, descendant_identities=descendant_identities, service_create_time=service_create_time, gateway_create_time=gateway_create_time, ) except RuntimeError: raise except Exception as exc: raise RuntimeError( "Could not determine SCM ownership for gateway profile " f"{profile_process.profile}" ) from exc return [found[name] for name in sorted(found)] def _gateway_run_args_for_profile(profile: str) -> list[str]: args = [get_python_path(), "-m", "hermes_cli.main"] if profile != "default": args.extend(["--profile", profile]) args.extend(["gateway", "run", "--replace"]) return args def _capture_gateway_argv(pid: int) -> list[str] | None: """Live argv of a running gateway (snapshotted before update kills so unmapped gateways can respawn). ``None`` if psutil is unavailable, the process is gone/denied, or the argv isn't a gateway command. """ if pid <= 1: return None try: import psutil # type: ignore except ImportError: return None try: argv = list(psutil.Process(pid).cmdline() or []) except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess): return None except Exception: return None if not argv: return None # Never respawn an unrelated process the scan happened to report. try: from gateway.status import looks_like_gateway_command_line if not looks_like_gateway_command_line(" ".join(argv)): return None except Exception: pass return argv def _prepare_profile_gateway_update_restart(profile: str, pid: int) -> str | None: """Choose who relaunches a profile gateway after ``hermes update``. ``--external-supervisor`` gateways must exit back to their manager (a detached watcher would race its replacement). Otherwise arm the profile-derived detached watcher, falling back to replaying the captured command line. """ argv = _capture_gateway_argv(pid) if argv and "--external-supervisor" in argv: return "external-supervisor" if launch_detached_profile_gateway_restart(profile, pid): return "detached" if argv and launch_detached_gateway_restart_by_cmdline(pid, list(argv)): return "detached-cmdline" return None def launch_detached_gateway_restart_by_cmdline(old_pid: int, run_argv: list[str]) -> bool: """Relaunch a gateway with no profile→PID-file mapping by replaying its captured argv after exit.""" if old_pid <= 0 or not run_argv: return False return _spawn_gateway_restart_watcher(old_pid, list(run_argv)) def launch_detached_profile_gateway_restart(profile: str, old_pid: int) -> bool: """Relaunch a manually-run profile gateway after its current PID exits.""" if old_pid <= 0: return False return _spawn_gateway_restart_watcher(old_pid, _gateway_run_args_for_profile(profile)) def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool: """Spawn the detached watcher that respawns ``run_argv`` once ``old_pid`` exits.""" if old_pid <= 0 or not run_argv: return False # Both watcher and respawned gateway need platform-appropriate detach: POSIX # ``start_new_session=True`` (setsid); on Windows that flag does NOT detach (the watcher would # die with the CLI console), so ``windows_detach_popen_kwargs()`` supplies the creationflags. from hermes_cli._subprocess_compat import ( windows_detach_flags_without_breakaway, windows_detach_popen_kwargs, ) # Windows: normalize the interpreter and capture a stable cwd + env overlay (HERMES_HOME, # VIRTUAL_ENV, PYTHONPATH) so the respawn doesn't depend on the watcher's cwd. No-op on POSIX. respawn_cwd = "" respawn_env_overlay: dict[str, str] = {} if sys.platform == "win32": try: from hermes_cli.gateway_windows import (windowless_gateway_restart_spec) run_argv, respawn_cwd, respawn_env_overlay = ( windowless_gateway_restart_spec(list(run_argv)) ) except Exception: # Fall back to the original argv: a visible window beats a failed respawn. respawn_cwd = "" respawn_env_overlay = {} # Embedded as JSON literals in the watcher source (no extra argv plumbing). respawn_cwd_literal = json.dumps(respawn_cwd) respawn_env_literal = json.dumps(respawn_env_overlay) watcher = textwrap.dedent( """ import os import subprocess import sys import time from hermes_cli._subprocess_compat import ( _WINDOWS_GATEWAY_BREAKAWAY_ENV, windows_detach_flags, windows_detach_flags_without_breakaway, ) pid = int(sys.argv[1]) cmd = sys.argv[2:] _respawn_cwd = {respawn_cwd_literal} _respawn_env_overlay = {respawn_env_literal} deadline = time.monotonic() + 120 while time.monotonic() < deadline: # ``os.kill(pid, 0)`` is not a no-op on Windows — use the # cross-platform existence check. from gateway.status import _pid_exists if not _pid_exists(pid): break time.sleep(0.2) # Route stray stdout/stderr from the respawned gateway to the same # sidecar log _spawn_detached uses. DEVNULL here meant a gateway # killed moments after respawn (e.g. parent Job Object teardown when # breakaway is denied, #48820 4th repro) left ZERO trace anywhere — # no gateway.log line, no exit-diag record, nothing. Best-effort: # fall back to DEVNULL when the log dir is unavailable. _stdio_target = subprocess.DEVNULL _stdio_fh = None try: from hermes_cli.config import get_hermes_home from pathlib import Path _log_dir = Path(get_hermes_home()) / "logs" _log_dir.mkdir(parents=True, exist_ok=True) _stdio_fh = open(_log_dir / "gateway-stdio.log", "ab", buffering=0) _stdio_target = _stdio_fh except Exception: pass # Platform-appropriate detach for the respawned gateway. On POSIX # start_new_session=True maps to os.setsid; on Windows we need # explicit creationflags because start_new_session is a no-op there. # CREATE_BREAKAWAY_FROM_JOB is critical: the watcher itself may have # been spawned inside a job object (Electron/Tauri parent), and # without breakaway the respawned gateway would die when that job # tears down. See _subprocess_compat.windows_detach_flags(). _popen_kwargs = {{ "stdout": _stdio_target, "stderr": _stdio_target, }} # Anchor the respawned gateway at the stable working dir and overlay # the env (VIRTUAL_ENV / PYTHONPATH / HERMES_HOME) the windowless # base interpreter needs to import hermes_cli. Empty on POSIX, where # the venv python resolves imports without help. if _respawn_cwd: _popen_kwargs["cwd"] = _respawn_cwd _base_env = {{**os.environ, **_respawn_env_overlay}} try: if sys.platform == "win32": try: _popen_kwargs["creationflags"] = windows_detach_flags() # Stamp the breakaway state exactly like the canonical # gateway_windows._spawn_detached, so the respawned # gateway's exit-diag / lifecycle records show whether it # escaped the parent Job Object (#48820 4th repro: # without the stamp, a job-teardown kill was # indistinguishable from any other silent death). _popen_kwargs["env"] = {{ **_base_env, _WINDOWS_GATEWAY_BREAKAWAY_ENV: "1", }} subprocess.Popen(cmd, **_popen_kwargs) except OSError: # CREATE_BREAKAWAY_FROM_JOB can be rejected with # ERROR_ACCESS_DENIED when the parent's job object refuses # breakaway. Retry without it — DETACHED_PROCESS et al. # alone are enough in most setups. Mirrors the canonical # fallback in gateway_windows._spawn_detached. _popen_kwargs["creationflags"] = ( windows_detach_flags_without_breakaway() ) _popen_kwargs["env"] = {{ **_base_env, _WINDOWS_GATEWAY_BREAKAWAY_ENV: "0", }} subprocess.Popen(cmd, **_popen_kwargs) else: if _respawn_env_overlay: _popen_kwargs["env"] = _base_env _popen_kwargs["start_new_session"] = True subprocess.Popen(cmd, **_popen_kwargs) finally: if _stdio_fh is not None: try: _stdio_fh.close() except OSError: pass """ ).strip().format( respawn_cwd_literal=respawn_cwd_literal, respawn_env_literal=respawn_env_literal, ) watcher_argv = [sys.executable, "-c", watcher, str(old_pid), *run_argv] # Same detach for the watcher itself, so closing the terminal doesn't kill it. try: subprocess.Popen( watcher_argv, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, **windows_detach_popen_kwargs(), ) except OSError: # Parent job object rejected CREATE_BREAKAWAY_FROM_JOB; retry without it (Windows only — # ``start_new_session=True`` cannot raise OSError on POSIX). try: fallback_kwargs: dict = ( {"creationflags": windows_detach_flags_without_breakaway()} if sys.platform == "win32" else {"start_new_session": True} ) subprocess.Popen( watcher_argv, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, **fallback_kwargs, ) except OSError: return False return True def _systemd_unit_is_active(system: bool) -> bool: """``systemctl is-active`` == "active" for the installed unit in ``system`` scope, else False.""" if not get_systemd_unit_path(system=system).exists(): return False try: result = _run_systemctl(["is-active", get_service_name()], system=system, timeout=10, **_CAPTURE_TEXT) except (RuntimeError, subprocess.TimeoutExpired): return False return result.stdout.strip() == "active" def _probe_systemd_service_running(system: bool = False) -> tuple[bool, bool]: selected_system = _select_systemd_scope(system) return selected_system, _systemd_unit_is_active(selected_system) def _read_systemd_unit_environment(system: bool = False) -> dict[str, str]: """Parse ``systemctl show -p Environment`` (one line of unquoted space-separated KEY=VALUE pairs).""" body = _systemctl_show(("Environment",), system=system).get("Environment", "") parsed: dict[str, str] = {} for token in body.split(): if "=" in token: key, value = token.split("=", 1) parsed[key] = value return parsed def _systemctl_show(properties: tuple[str, ...], *, system: bool) -> dict[str, str]: """``systemctl show --property a,b`` for the gateway unit as ``{key: value}``; {} on failure.""" try: result = _run_systemctl( ["show", get_service_name(), "--no-pager", "--property", ",".join(properties)], system=_select_systemd_scope(system), timeout=10, **_CAPTURE_TEXT, ) except (RuntimeError, subprocess.TimeoutExpired, OSError): return {} if result.returncode != 0: return {} parsed: dict[str, str] = {} for line in result.stdout.splitlines(): if "=" in line: key, value = line.split("=", 1) parsed[key] = value.strip() return parsed def _hermes_home_from_systemd_unit_file(system: bool = False) -> str | None: """``HERMES_HOME`` from the on-disk unit file — what refresh/compare already read, and reliable under ``sudo``.""" unit_path = get_systemd_unit_path(system=system) if not unit_path.exists(): return None try: text = unit_path.read_text(encoding="utf-8") except OSError: return None for line in text.splitlines(): stripped = line.strip() if not stripped.startswith("Environment="): continue body = stripped[len("Environment=") :].strip().strip('"') if body.startswith("HERMES_HOME="): value = body.split("=", 1)[1].strip().strip('"') return value or None return None def _sync_hermes_home_from_systemd_unit(system: bool) -> None: """For a system-scope unit, adopt its ``HERMES_HOME``. Under ``sudo`` HERMES_HOME is stripped and HOME=/root, so get_hermes_home() would pick the wrong profile; mirroring the unit's value makes runtime-status/PID reads hit the right files. """ if not system: return # On-disk unit first; ``systemctl show`` for units that only exist in the manager. unit_home = (_hermes_home_from_systemd_unit_file(system=True) or "").strip() if not unit_home: unit_home = _read_systemd_unit_environment(system=True).get("HERMES_HOME", "").strip() if not unit_home: return current = os.environ.get("HERMES_HOME", "").strip() if current == unit_home: return os.environ["HERMES_HOME"] = unit_home def _read_systemd_unit_properties( system: bool = False, properties: tuple[str, ...] = ("ActiveState", "SubState", "Result", "ExecMainStatus", "MainPID"), ) -> dict[str, str]: """Return selected ``systemctl show`` properties for the gateway unit.""" return _systemctl_show(properties, system=system) def _systemd_main_pid_from_props(props: dict[str, str]) -> int | None: try: pid = int(props.get("MainPID", "0") or "0") except (TypeError, ValueError): return None return pid if pid > 0 else None def _systemd_main_pid(system: bool = False) -> int | None: return _systemd_main_pid_from_props(_read_systemd_unit_properties(system=system)) def _read_gateway_runtime_status() -> dict | None: try: from gateway.status import read_runtime_status state = read_runtime_status() except Exception: return None return state if isinstance(state, dict) else None def _gateway_runtime_status_for_pid(pid: int | None) -> dict | None: if not pid: return None state = _read_gateway_runtime_status() if not state: return None try: state_pid = int(state.get("pid", 0) or 0) except (TypeError, ValueError): return None return state if state_pid == pid else None def _wait_for_systemd_service_restart( *, system: bool = False, previous_pid: int | None = None, timeout: float | None = None, replacement_observed: list[bool] | None = None, ) -> bool: """Wait for the gateway service to become active after a restart handoff.""" import time svc = get_service_name() scope_label = _service_scope_label(system).capitalize() if timeout is None: timeout = _systemd_restart_wait_timeout(system=system) deadline = time.monotonic() + timeout printed_runtime_wait = False while time.monotonic() < deadline: props = _read_systemd_unit_properties(system=system) active_state = props.get("ActiveState", "") sub_state = props.get("SubState", "") new_pid = None try: from gateway.status import get_running_pid new_pid = get_running_pid() except Exception: new_pid = None if not new_pid: new_pid = _systemd_main_pid_from_props(props) runtime_state = _read_gateway_runtime_status() try: runtime_pid = int((runtime_state or {}).get("pid", 0) or 0) except (TypeError, ValueError): runtime_pid = 0 if ( previous_pid is not None and replacement_observed is not None and not replacement_observed and any( candidate_pid > 0 and candidate_pid != previous_pid for candidate_pid in (new_pid or 0, runtime_pid) ) ): replacement_observed.append(True) if active_state == "active" and new_pid and (previous_pid is None or new_pid != previous_pid): if runtime_pid != new_pid: runtime_state = _gateway_runtime_status_for_pid(new_pid) gateway_state = (runtime_state or {}).get("gateway_state") if gateway_state == "running": print(f"✓ {scope_label} service restarted (PID {new_pid})") return True if gateway_state == "startup_failed": reason = (runtime_state or {}).get("exit_reason") or "startup failed" print( f"⚠ {scope_label} service process restarted (PID {new_pid}), but gateway startup failed: {reason}" ) return False if not printed_runtime_wait: print( f"⏳ {scope_label} service process started (PID {new_pid}); waiting for gateway runtime..." ) printed_runtime_wait = True if active_state == "activating" and sub_state == "auto-restart": time.sleep(1) continue if _systemd_unit_is_start_limited(props): _print_systemd_start_limit_wait(system=system) return False time.sleep(2) print( f"⚠ {scope_label} service did not become active within {int(timeout)}s.\n" f" Check status: {'sudo ' if system else ''}hermes gateway status\n" f" Check logs: journalctl {'--user ' if not system else ''}-u {svc} -l --since '2 min ago'" ) return False def _systemd_restart_wait_timeout(system: bool = False) -> float: """Cover systemd's relaunch delays before applying the runtime wait floor.""" from gateway.shutdown_forensics import parse_systemd_duration_to_us props = _read_systemd_unit_properties(system=system, properties=("RestartUSec", "TimeoutStartUSec")) supervisor_budget = 0.0 for name in ("RestartUSec", "TimeoutStartUSec"): raw = props.get(name, "") duration_us = (int(raw) if raw.isdigit() else parse_systemd_duration_to_us(raw)) if duration_us is not None: supervisor_budget += duration_us / 1_000_000 return 60.0 + supervisor_budget def _systemd_unit_is_start_limited(props: dict[str, str]) -> bool: result = props.get("Result", "").lower() sub_state = props.get("SubState", "").lower() return result == "start-limit-hit" or sub_state == "start-limit-hit" def _systemd_error_indicates_start_limit(exc: subprocess.CalledProcessError) -> bool: parts: list[str] = [] for attr in ("stderr", "stdout", "output"): value = getattr(exc, attr, None) if not value: continue if isinstance(value, bytes): value = value.decode(errors="replace") parts.append(str(value)) text = "\n".join(parts).lower() return ( "start-limit-hit" in text or "start request repeated too quickly" in text or "start-limit" in text ) def _systemd_service_is_start_limited(system: bool = False) -> bool: return _systemd_unit_is_start_limited(_read_systemd_unit_properties(system=system)) def _print_systemd_start_limit_wait(system: bool = False) -> None: svc = get_service_name() scope_label = _service_scope_label(system).capitalize() scope_flag = " --system" if system else "" systemctl_prefix = "systemctl " if system else "systemctl --user " journal_prefix = "journalctl " if system else "journalctl --user " print(f"⏳ {scope_label} service is temporarily rate-limited by systemd.") print(" systemd is refusing another immediate start after repeated exits.") print( f" Wait for the start-limit window to expire, then run: {'sudo ' if system else ''}hermes gateway restart{scope_flag}" ) print(f" Or clear the failed state manually: {systemctl_prefix}reset-failed {svc}") print(f" Check logs: {journal_prefix}-u {svc} -l --since '5 min ago'") def _recover_pending_systemd_restart(system: bool = False, previous_pid: int | None = None) -> bool: """Recover a planned service restart that is stuck in systemd state.""" props = _read_systemd_unit_properties(system=system) if not props: return False try: from gateway.status import read_runtime_status except Exception: return False runtime_state = read_runtime_status() or {} if not runtime_state.get("restart_requested"): return False active_state = props.get("ActiveState", "") sub_state = props.get("SubState", "") exec_main_status = props.get("ExecMainStatus", "") result = props.get("Result", "") if active_state == "activating" and sub_state == "auto-restart": print("⏳ Service restart already pending — waiting for systemd relaunch...") return _wait_for_systemd_service_restart(system=system, previous_pid=previous_pid) if active_state == "failed" and ( exec_main_status == str(GATEWAY_SERVICE_RESTART_EXIT_CODE) or result == "exit-code" ): svc = get_service_name() scope_label = _service_scope_label(system).capitalize() print(f"↻ Clearing failed state for pending {scope_label.lower()} service restart...") _run_systemctl(["reset-failed", svc], system=system, check=False, timeout=30) _run_systemctl(["start", svc], system=system, check=False, timeout=90) return _wait_for_systemd_service_restart(system=system, previous_pid=previous_pid) return False def _parse_launchd_pid_from_list_output(output: str) -> int | None: """PID from ``launchctl list