From 2c7f5d12b5c19aa450ca2920107d44903a8545b7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:38:32 -0700 Subject: [PATCH] =?UTF-8?q?refactor(hclib):=20cron/update/install=20lifecy?= =?UTF-8?q?cle=20=E2=80=94=20cron=20status,=20update=5F*=20receipts=20and?= =?UTF-8?q?=20recovery,=20install=20repair,=20service=20manager?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/_early_recovery.py | 269 +++---- hermes_cli/_install_repair.py | 291 +++---- hermes_cli/_scan_venv_blockers.py | 135 +--- hermes_cli/_startup_fast.py | 156 ++-- hermes_cli/build_info.py | 102 +-- hermes_cli/container_boot.py | 177 +---- hermes_cli/cron.py | 615 +++++++-------- hermes_cli/dep_ensure.py | 109 +-- hermes_cli/gitlock.py | 198 ++--- hermes_cli/gui_uninstall.py | 151 ++-- hermes_cli/heartbeat.py | 156 ++-- hermes_cli/image_provenance.py | 51 +- hermes_cli/install_identity.py | 4 +- hermes_cli/lifecycle.py | 15 +- hermes_cli/linux_desktop_entry.py | 220 ++---- hermes_cli/macos_tcc_anchor.py | 91 +-- hermes_cli/managed_uv.py | 830 +++++++------------- hermes_cli/migrate.py | 4 +- hermes_cli/npm_engine.py | 104 +-- hermes_cli/relaunch.py | 66 +- hermes_cli/service_manager.py | 731 ++++++------------ hermes_cli/uninstall.py | 1008 ++++++++++--------------- hermes_cli/update_abort_recovery.py | 250 ++---- hermes_cli/update_contract.py | 29 +- hermes_cli/update_inventory.py | 447 +++++------ hermes_cli/update_lock.py | 114 +-- hermes_cli/update_receipt.py | 235 +++--- hermes_cli/update_restart_recovery.py | 380 +++------- hermes_cli/worktree_cmd.py | 127 ++-- hermes_cli/worktree_gc.py | 264 +++---- 30 files changed, 2586 insertions(+), 4743 deletions(-) diff --git a/hermes_cli/_early_recovery.py b/hermes_cli/_early_recovery.py index 6bae52b973..50b30acd25 100644 --- a/hermes_cli/_early_recovery.py +++ b/hermes_cli/_early_recovery.py @@ -1,24 +1,11 @@ """Dependency-light venv recovery that runs BEFORE hermes_cli.main's imports. -The ``hermes`` console entry point is ``hermes_cli.main:main``. Importing -``hermes_cli.main`` pulls in third-party packages at module level (``dotenv`` -via ``hermes_cli.env_loader``, ``yaml`` via ``hermes_cli.config``, ...). In -the exact failure state the update-recovery markers exist for — a failed lazy -backend refresh or interrupted core install that wiped a core package's -import files (#57828) — a normal launch crashes *while importing main.py*, -before ``_recover_from_interrupted_install()`` can run. The marker system is -unreachable precisely when it is needed most. +This module is deliberately **stdlib-only** so importing it can never fail on a corrupted venv. +``hermes_cli.main`` imports and calls :func:`recover_if_needed` at the very top of its module body, +before any third-party import. -This module is deliberately **stdlib-only** so importing it can never fail on -a corrupted venv. ``hermes_cli.main`` imports and calls -:func:`recover_if_needed` at the very top of its module body, before any -third-party import. - -Scope: this early pass only repairs enough for ``hermes_cli.main`` to become -importable again (force-reinstall of the known-fragile core packages, using -the pins from pyproject.toml). It NEVER clears the recovery markers — the -full, confirmed marker lifecycle stays with ``_recover_from_interrupted_install()`` -in main.py, which runs right after import succeeds. +Scope: this early pass only repairs enough for ``hermes_cli.main`` to become importable again +(force-reinstall of the known-fragile core packages, using the pins from pyproject.toml). """ from __future__ import annotations @@ -83,16 +70,9 @@ def restore_quarantined_shims( ) -> list[tuple[Path, Path]]: """Rename quarantined shims back, retrying a lock instead of giving up. - ``moved`` holds ``(original, quarantined)`` pairs. Returns the pairs that - could NOT be restored, and prints an actionable recovery command for each. - - A pair is not a failure when ``original`` already exists or ``quarantined`` - has gone: the installer wrote a fresh shim, or a concurrent sweep won the - race. Both are silent, so two processes sweeping the same orphan cannot - produce a spurious error. - - Messages go to stderr by default -- the startup sweep runs on EVERY hermes - invocation, and ``hermes acp`` speaks JSON-RPC on stdout. + A pair is not a failure when ``original`` already exists or ``quarantined`` has gone: the + installer wrote a fresh shim, or a concurrent sweep won the race. Both are silent, so two + processes sweeping the same orphan cannot produce a spurious error. """ if stream is None: stream = sys.stderr @@ -153,12 +133,67 @@ def _project_root() -> Path: return Path(__file__).resolve().parent.parent +def _load_pyproject_project(root: Path) -> dict | None: + """``[project]`` table of ``root/pyproject.toml``; ``None`` when missing/unreadable. + + Stdlib-only (tomllib) — shared by every pyproject lookup in the repair path so ``packaging`` + is never needed in the broken-venv state. + """ + pyproject = root / "pyproject.toml" + if not pyproject.is_file(): + return None + try: + import tomllib + + with open(pyproject, "rb") as f: + project = tomllib.load(f).get("project", {}) + except Exception: + return None + return project if isinstance(project, dict) else None + + +def _read_marker_attempts(marker_path: Path) -> int: + """Attempt counter from a marker's opportunistic JSON body; corrupt/missing → 0.""" + try: + raw = marker_path.read_text(encoding="utf-8", errors="replace").strip() + except OSError: + return 0 + if not raw: + return 0 + try: + import json + + return int(json.loads(raw).get("attempts", 0)) + except (ValueError, AttributeError): + return 0 + + +def _run_ensurepip(root: Path) -> None: + """Best-effort pip bootstrap — a killed install can leave the venv with no pip module at all.""" + try: + subprocess.run( + [sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"], + cwd=root, + capture_output=True, + ) + except Exception: + pass + + +def _report_failed_install(result) -> bool: + """Replay the captured tail of a failed installer run to stderr; always ``False``.""" + tail = (result.stderr or result.stdout or "")[-2000:] + if tail: + print(tail, file=sys.stderr) + return False + + def _pid_is_running(pid: int) -> bool: """Best-effort stdlib-only process liveness probe. - ``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process - handle API there. An access-denied result is conservatively live: racing - an elevated updater is worse than postponing recovery for one launch. + ``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process handle API there. An + access-denied result counts as live: racing an elevated updater is worse than postponing + recovery for one launch. """ if pid <= 0: return False @@ -217,20 +252,13 @@ def _marker_owner_is_live(marker: Path) -> bool: def _pinned_specs(packages: list[str], project_root: Path) -> list[str]: """Map bare package names to their pinned specs from pyproject.toml. - Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may - itself be broken in the failure state this module exists for). Unknown - packages fall back to their bare name. + Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may itself be broken in + the failure state this module exists for). Unknown packages fall back to their bare name. """ - pyproject = project_root / "pyproject.toml" - if not pyproject.is_file(): - return packages - try: - import tomllib - - with open(pyproject, "rb") as f: - raw_deps = tomllib.load(f).get("project", {}).get("dependencies", []) or [] - except Exception: + project = _load_pyproject_project(project_root) + if project is None: return packages + raw_deps = project.get("dependencies", []) or [] name_to_spec: dict[str, str] = {} for spec in raw_deps: @@ -249,12 +277,9 @@ def _pinned_specs(packages: list[str], project_root: Path) -> list[str]: def _certifi_bundle_broken() -> bool: """True when certifi imports but its ``cacert.pem`` is missing/corrupt. - A brew Python upgrade or an interrupted venv rebuild can leave certifi's - distribution metadata (and even the module) intact while the bundled - ``cacert.pem`` is gone or a dangling symlink — every TLS connection then - fails with an opaque ``Could not find a suitable TLS CA certificate - bundle`` from deep inside httpx/requests (#29866). An attribute probe - alone passes in that state, so validate the bundle path itself. + A brew Python upgrade or interrupted venv rebuild can leave certifi's metadata intact while + ``cacert.pem`` is gone or a dangling symlink; every TLS connection then fails opaquely deep in + httpx/requests. An attribute probe passes in that state, so validate the bundle path itself. """ try: import certifi @@ -271,12 +296,9 @@ def _certifi_bundle_broken() -> bool: def _probe_broken_packages() -> list[str]: """Import-probe the fragile core packages in THIS process. - Returns repair package names (deduped, probe order) for modules that fail - to import or lack their sentinel attribute. Failed imports leave nothing - in ``sys.modules``, so a post-repair retry in the same process works. - - certifi additionally gets a bundle-file check: the module can import - cleanly while ``cacert.pem`` is missing (#29866). + Returns repair package names (deduped, probe order) for modules that fail to import or lack + their sentinel attribute. Failed imports leave nothing in ``sys.modules``, so a post-repair + retry in the same process works. """ broken: list[str] = [] for mod_name, attr in LAZY_REFRESH_IMPORT_PROBES: @@ -296,10 +318,9 @@ def _probe_broken_packages() -> list[str]: def _find_uv_binary() -> str | None: """Locate a ``uv`` binary without importing third-party modules. - uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so - the stdlib ``pip`` fallback below refuses to touch them. In that state - the only sanctioned installer is uv itself, which Hermes already vendors - (``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only. + uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so the stdlib ``pip`` + fallback below refuses to touch them. In that state the only sanctioned installer is uv itself, + which Hermes already vendors (``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only. """ exe = "uv.exe" if sys.platform == "win32" else "uv" candidates = [ @@ -316,40 +337,30 @@ def _find_uv_binary() -> str | None: def _base_interpreter_is_externally_managed() -> bool: """True when ``sys.executable`` is a uv/standalone-builds managed install. - Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their - stdlib (PEP 668), so ``python -m pip install`` aborts with - ``externally-managed-environment``. The early repair must then go - through uv (or explicitly override pip) or the reinstall no-ops and the - venv stays broken (#83569). + Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their stdlib (PEP 668), so + ``python -m pip install`` aborts with ``externally-managed-environment``. The early repair must + then go through uv (or explicitly override pip) or the reinstall no-ops and the venv stays + broken (#83569). """ try: import sysconfig stdlib = Path(sysconfig.get_path("stdlib")) - if (stdlib / "EXTERNALLY-MANAGED").exists(): - return True - # uv 0.5+ moved the marker into a ``_uv_managed`` sentinel dir… - if (stdlib.parent / "EXTERNALLY-MANAGED").exists(): - return True + # uv 0.5+ moved the marker one level up next to a ``_uv_managed`` sentinel dir. + return (stdlib / "EXTERNALLY-MANAGED").exists() or ( + stdlib.parent / "EXTERNALLY-MANAGED" + ).exists() except Exception: - pass - return False + return False def _run_repair_install(specs: list[str], project_root: Path) -> bool: """``uv pip`` (or stdlib ``pip``) force-reinstall of the given specs. - Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout); - output is captured and replayed to stderr only on failure. Never raises. + Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout); output is captured and + replayed to stderr only on failure. Never raises. Two installer paths, in priority order: - - 1. ``uv pip install`` with ``VIRTUAL_ENV`` pointed at the project venv — - required when the base interpreter is uv-managed (Windows git checkouts - install exactly this way: uv's Python declares PEP 668 - ``EXTERNALLY-MANAGED`` and plain ``python -m pip`` refuses to run). - 2. ``sys.executable -m pip`` as before, for self-contained venvs whose - interpreter carries no PEP 668 marker. """ externally_managed = _base_interpreter_is_externally_managed() if externally_managed: @@ -366,15 +377,10 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool: text=True, encoding="utf-8", errors="replace", env=env, ) - if result.returncode == 0: - return True - tail = (result.stderr or result.stdout or "")[-2000:] - if tail: - print(tail, file=sys.stderr) - return False except Exception as exc: print(f" ✗ Early venv repair could not run uv: {exc}", file=sys.stderr) return False + return result.returncode == 0 or _report_failed_install(result) # No uv available: fall through to pip with the PEP 668 override so # the repair at least attempts to fix the venv instead of no-oping. print( @@ -383,14 +389,7 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool: file=sys.stderr, ) - try: - subprocess.run( - [sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"], - cwd=project_root, - capture_output=True, - ) - except Exception: - pass + _run_ensurepip(project_root) pip_cmd = [sys.executable, "-m", "pip", "install", "--force-reinstall"] if externally_managed: pip_cmd.append("--break-system-packages") @@ -405,24 +404,18 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool: except Exception as exc: print(f" ✗ Early venv repair could not run pip: {exc}", file=sys.stderr) return False - if result.returncode != 0: - tail = (result.stderr or result.stdout or "")[-2000:] - if tail: - print(tail, file=sys.stderr) - return False - return True + return result.returncode == 0 or _report_failed_install(result) def _pytest_owns_live_checkout(root: Path) -> bool: - """True when running under pytest AND ``root`` is this module's own - checkout — the one whose venv is executing the suite right now. + """True when running under pytest AND ``root`` is this module's own checkout — the one whose + venv is executing the suite right now. - Lifecycle tests spawn real subprocesses that import ``hermes_cli.main`` - with recovery armed; ``PYTEST_CURRENT_TEST`` rides the inherited env into - those children. Without this guard, a genuinely-broken dev venv gets a - REAL ``ensurepip`` + ``pip install --force-reinstall`` from inside a - running test suite. Tests that sandbox ``project_root`` to a tmp_path are - unaffected (same posture as ``managed_scope._under_pytest``).""" + Lifecycle tests spawn real subprocesses that import ``hermes_cli.main`` with recovery armed and + inherit ``PYTEST_CURRENT_TEST``. Without this guard a broken dev venv would get a REAL ensurepip + + ``pip install --force-reinstall`` from inside a running suite. Tests sandboxing + ``project_root`` to a tmp_path are unaffected. + """ return ( "PYTEST_CURRENT_TEST" in os.environ and root == Path(__file__).resolve().parent.parent @@ -435,14 +428,10 @@ def recover_if_needed( ) -> None: """Repair wiped core packages so ``hermes_cli.main`` can import at all. - Fast path (no marker present) is two ``lstat`` calls. Only acts when a - recovery marker from a prior ``hermes update`` exists AND an import probe - confirms a core package is actually broken. Markers are intentionally - NOT cleared here — ``_recover_from_interrupted_install()`` in main.py owns - the confirmed marker lifecycle and runs immediately after import succeeds. + Fast path (no marker present) is two ``lstat`` calls. Only acts when a recovery marker from a + prior ``hermes update`` exists AND an import probe confirms a core package is actually broken. - Never raises: on any failure the import of main.py proceeds and surfaces - the real error. + Never raises: on any failure the import of main.py proceeds and surfaces the real error. """ global _UPDATE_RETRY_RECOVERED @@ -496,20 +485,8 @@ def recover_if_needed( # Single-flight: share main.py's recovery lock so an early repair # never races a concurrent full recovery into the same shared venv. - lock_path = root / ".update-incomplete.lock" - try: - fd = os.open(lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY) - os.write(fd, f"{os.getpid()}\n".encode()) - os.close(fd) - except FileExistsError: - try: - if time.time() - lock_path.stat().st_mtime > 3600: - lock_path.unlink() - except OSError: - pass + if not _claim_recovery_lock(root): return - except OSError: - pass # read-only fs / perms — proceed unlocked, install surfaces it try: specs = _pinned_specs(broken, root) @@ -531,10 +508,7 @@ def recover_if_needed( file=sys.stderr, ) finally: - try: - lock_path.unlink() - except OSError: - pass + _release_recovery_lock(root) except Exception: # Never block launch — the import of main.py will surface the truth. pass @@ -581,21 +555,11 @@ def _release_recovery_lock(root: Path) -> None: def _complete_pending_core_install(root: Path, core_marker: Path) -> bool: """Run the pending core install BEFORE main.py can import native modules. - ``recover_if_needed`` invokes this when ``.update-incomplete`` exists — - a prior ``hermes update`` (or the self-lock preflight, #83569) left the - dependency sync deliberately unfinished. Completing it here matters on - Windows: the deferral exists precisely because the process that wrote the - marker had a native venv extension mapped; this process, running before - ``hermes_cli.main``'s third-party imports, maps nothing yet, so the - installer can replace ``.pyd`` files without hitting the lock. + ``recover_if_needed`` invokes this when ``.update-incomplete`` exists — a prior ``hermes + update`` (or the self-lock preflight, #83569) left the dependency sync deliberately unfinished. - Marker lifecycle: cleared on success; kept (attempts counter bumped) on - failure for the next launch or main.py's post-import recovery. An - attempts ceiling caps automatic retries so a persistent installer - failure does not block every launch (``hermes acp`` included). - - Never raises: any failure leaves the marker for the post-import path and - returns ``False``. Returns ``True`` only after the install succeeds. + Never raises: any failure leaves the marker for the post-import path and returns ``False``. + Returns ``True`` only after the install succeeds. """ try: from hermes_cli import _install_repair as ir @@ -603,18 +567,7 @@ def _complete_pending_core_install(root: Path, core_marker: Path) -> bool: # Retry backoff: read current attempts before claiming the lock so a # persistently-failing install stops hammering early. After the # increment the counter reflects THIS attempt. - attempts = 0 - try: - raw = core_marker.read_text(encoding="utf-8", errors="replace").strip() - if raw: - import json as _json - - try: - attempts = int(_json.loads(raw).get("attempts", 0)) - except (ValueError, AttributeError): - attempts = 0 - except OSError: - attempts = 0 + attempts = _read_marker_attempts(core_marker) if attempts >= _EARLY_CORE_INSTALL_MAX_ATTEMPTS: print( diff --git a/hermes_cli/_install_repair.py b/hermes_cli/_install_repair.py index 36e030c412..61a5879533 100644 --- a/hermes_cli/_install_repair.py +++ b/hermes_cli/_install_repair.py @@ -1,22 +1,13 @@ """Dependency install execution shared between early recovery and full recovery. -Both callers need to run the same core ``.[all]`` reinstall: +- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE ``hermes_cli.main``'s +third-party imports, so it can complete a pending update while no native extension is mapped yet +(#83569). - ``hermes_cli.main._recover_core_update_marker_locked`` — the historical post-import +recovery path. -- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE - ``hermes_cli.main``'s third-party imports, so it can complete a pending - update while no native extension is mapped yet (#83569). -- ``hermes_cli.main._recover_core_update_marker_locked`` — the historical - post-import recovery path. Kept as a fallback for installs the early pass - could not complete (marker left in place on failure). - -This module is deliberately **stdlib-only** so importing it can never fail in -the corrupted-venv state it exists to repair. ``hermes_cli.main`` imports -``managed_uv``, ``hermes_constants``, and friends only in its late path; the -early path must not. Where the late path uses ``managed_uv.ensure_uv`` to -bootstrap uv if missing, the early path uses the stdlib -:func:`hermes_cli._early_recovery._find_uv_binary` lookup and falls back to -plain pip when uv is absent — a degraded but working installer (the late -recovery will bootstrap uv on the next launch if it ever matters). +This module is deliberately **stdlib-only** so importing it can never fail in the corrupted-venv +state it exists to repair. ``hermes_cli.main`` imports ``managed_uv``, ``hermes_constants``, and +friends only in its late path; the early path must not. """ from __future__ import annotations @@ -43,10 +34,7 @@ def _is_termux_env(env: dict | None = None) -> bool: """Stdlib Termux probe (hermes_cli.main's version lives behind imports).""" env = env if env is not None else os.environ try: - if env.get("TERMUX_VERSION"): - return True - prefix = env.get("PREFIX", "") - return "com.termux" in prefix + return bool(env.get("TERMUX_VERSION")) or "com.termux" in env.get("PREFIX", "") except Exception: return False @@ -55,11 +43,9 @@ def _is_termux_env(env: dict | None = None) -> bool: def _stdout_to_stderr(): """Route fd 1 (and sys.stdout) to stderr for the duration of an install. - ``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child - writing there would corrupt the protocol. Mirrors - ``main.py::_recover_from_interrupted_install``. + ``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child writing there would + corrupt the protocol. Mirrors ``main.py::_recover_from_interrupted_install``. """ - saved_fd = None saved_sys_stdout = sys.stdout try: saved_fd = os.dup(1) @@ -72,24 +58,18 @@ def _stdout_to_stderr(): finally: sys.stdout = saved_sys_stdout if saved_fd is not None: - try: + with contextlib.suppress(OSError): os.dup2(saved_fd, 1) - except OSError: - pass - try: + with contextlib.suppress(OSError): os.close(saved_fd) - except OSError: - pass def _resolve_install_target(root: Path) -> tuple[list[str], dict | None]: """(install_cmd_prefix, env) for the project venv — stdlib uv lookup. - Mirrors ``main.py::_default_venv_install_target`` but without - ``managed_uv``. ``VIRTUAL_ENV`` steers ``uv pip`` at the project venv even - when invoked from the base interpreter (the early-recovery case). - Termux strips leaked interpreter-path env vars so uv resolves the venv - correctly. + Mirrors ``main.py::_default_venv_install_target`` without ``managed_uv``. ``VIRTUAL_ENV`` + steers ``uv pip`` at the project venv even when invoked from the base interpreter (early + recovery). Termux strips leaked interpreter-path env vars so uv resolves the venv correctly. """ uv_bin = _er._find_uv_binary() if uv_bin: @@ -125,17 +105,34 @@ def _venv_scripts_dir(root: Path) -> Path | None: _WINDOWS_BIN_LAUNCHERS = ("hermes", "hermes-acp") +def _launcher_present(target: Path, name: str) -> bool: + return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists() + + +def _launchers_missing(target: Path) -> bool: + return any(not _launcher_present(target, name) for name in _WINDOWS_BIN_LAUNCHERS) + + +def _default_hermes_root() -> Path | None: + """Per-machine anchor for the managed-clone gate — the DEFAULT Hermes root, not + ``get_hermes_home()``: under ``hermes -p `` that returns ``profiles\\``, which + would fail the gate and silently skip the heal for profile users. ``None`` when unresolvable.""" + from hermes_constants import get_default_hermes_root + + try: + return Path(get_default_hermes_root()) + except Exception: + return None + + def _venv_is_relocatable(venv_dir: Path) -> bool: """True when the venv's pyvenv.cfg declares ``relocatable = true``. - uv writes the flag; ``hermes_cli.managed_uv`` builds its replacement - venvs with ``--relocatable`` (they are constructed aside and swapped - into place). A relocatable venv's console-script trampolines embed a - RELATIVE interpreter reference, so a COPY of one placed outside - ``venv\\Scripts`` fails at run time with ``uv trampoline failed to - canonicalize script path``. Non-relocatable venvs (fresh installs) - embed the absolute interpreter path and their trampolines survive - copying. This flag decides which launcher form a PATH dir gets. + uv writes the flag; ``managed_uv`` builds replacement venvs ``--relocatable``. A relocatable + venv's console-script trampolines embed a RELATIVE interpreter reference, so a copy placed + outside ``venv\Scripts`` fails with ``uv trampoline failed to canonicalize script path``; + non-relocatable venvs embed the absolute path and survive copying. This decides which + launcher form a PATH dir gets. """ try: cfg = (Path(venv_dir) / "pyvenv.cfg").read_text( @@ -143,20 +140,18 @@ def _venv_is_relocatable(venv_dir: Path) -> bool: ) except OSError: return False - for line in cfg.splitlines(): - key, _, value = line.partition("=") - if key.strip().lower() == "relocatable" and value.strip().lower() == "true": - return True - return False + return any( + key.strip().lower() == "relocatable" and value.strip().lower() == "true" + for key, _, value in (line.partition("=") for line in cfg.splitlines()) + ) def _normalize_windows_path(value) -> str: """Windows path equality key: backslashes, no trailing separator, lowered. - Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than - ``os.path.normcase`` — that is an identity function on POSIX, and this - comparison must behave Windows-correct even when tests exercise the - Windows branch from another host (same rationale as + Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than ``os.path.normcase`` — + that is an identity function on POSIX, and this comparison must behave Windows-correct even when + tests exercise the Windows branch from another host (same rationale as ``venv_bin_dir(windows=...)``). """ return str(value).replace("/", "\\").rstrip("\\").lower() @@ -165,8 +160,7 @@ def _normalize_windows_path(value) -> str: def _windows_user_path_entries() -> list[str]: """User PATH entries from the registry — the value install.ps1 writes. - Falls back to the process PATH when the registry is unreadable. Only - called on Windows. + Falls back to the process PATH when the registry is unreadable. Only called on Windows. """ try: import winreg @@ -187,48 +181,13 @@ def ensure_windows_bin_launchers( ) -> list[str]: """Re-stage the Windows ``hermes`` launchers when they vanish. - On Windows, ``hermes`` resolves through launchers derived from the venv - console scripts — never ``venv\\Scripts`` itself on PATH, which would - shadow the user's ``python`` (#83797). The canonical launcher home is - the managed binary dir — the default Hermes root's ``bin`` - (``%LOCALAPPDATA%\\hermes\\bin``, next to the managed uv) — which lives - OUTSIDE the git checkout so no git operation can ever touch it. It is - a per-machine dir shared by every profile: ``get_hermes_home()`` would - point inside ``profiles\\`` under ``hermes -p``, so the anchor - here is :func:`hermes_constants.get_default_hermes_root`. + On Windows, ``hermes`` resolves through launchers derived from the venv console scripts — never + ``venv\Scripts`` itself on PATH, which would shadow the user's ``python`` (#83797). - Earlier installer versions staged them at ``\\bin`` instead — - inside the git working tree — where ``hermes update``'s pre-update - autostash (``git stash push --include-untracked``) swept them off disk; - once the desktop updater stopped re-applying stashes (``--keep-stash``) - nothing restored them and ``hermes`` stopped resolving in every new - terminal. That legacy location is re-staged too, during the transition, - for installs whose user PATH still resolves through it. - - The launcher FORM depends on the venv (see :func:`_venv_is_relocatable`): - a normal venv's exe trampoline embeds an absolute interpreter path and - survives copying, so it is copied as ``.exe``; a relocatable - venv's trampoline resolves relative to its own location and a copy - dies with ``uv trampoline failed to canonicalize script path``, so a - ``.cmd`` delegator invoking the in-venv exe by absolute path is - written instead. A name counts as present when EITHER form exists — - exe copies staged before a venv rebuild keep working (they embed the - swapped-in-place venv's absolute path) and are left alone. - - Two targets, two gates, both failing toward inaction: - - - canonical managed binary dir: only when *root* is the managed clone - (``root.parent == get_default_hermes_root()``), so source checkouts - elsewhere never gain launchers; - - legacy ``\\bin``: only when that dir is on the user PATH - (registry value, process PATH as fallback), i.e. the install opted - into the old layout and still resolves through it. - - Writes go through a staging name + ``os.replace`` so concurrent process - starts cannot tear a launcher. Never raises; returns the restored paths. - - *windows* and *user_path_entries* are injectable for tests, same pattern - as ``hermes_constants.venv_bin_dir``. + - canonical managed binary dir: only when *root* is the managed clone (``root.parent == + get_default_hermes_root()``), so source checkouts elsewhere never gain launchers; - legacy + ``\bin``: only when that dir is on the user PATH (registry value, process PATH as + fallback), i.e. """ if windows is None: windows = _is_windows() @@ -237,20 +196,10 @@ def ensure_windows_bin_launchers( root = Path(root) - # Per-machine anchor: the DEFAULT Hermes root, not get_hermes_home() — - # under ``hermes -p `` that returns ``profiles\\``, which - # would fail the managed-clone gate below and silently skip the heal - # for profile users. The launcher dir serves the whole machine. - from hermes_constants import get_default_hermes_root - - try: - home = Path(get_default_hermes_root()) - except Exception: + home = _default_hermes_root() + if home is None: return [] - def _launcher_present(target: Path, name: str) -> bool: - return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists() - targets: list[Path] = [] # Canonical target — gate on the managed-clone shape. This runs at @@ -258,7 +207,7 @@ def ensure_windows_bin_launchers( # override), so the healthy path must stay at a couple of stat calls. if _normalize_windows_path(root.parent) == _normalize_windows_path(home): canonical = home / "bin" - if any(not _launcher_present(canonical, name) for name in _WINDOWS_BIN_LAUNCHERS): + if _launchers_missing(canonical): targets.append(canonical) # Legacy transition target — the pre-migration in-checkout dir. Only @@ -268,7 +217,7 @@ def ensure_windows_bin_launchers( # network shares. An entry stored some other way (8.3 short path, # subst drive) misses the re-stage, which fails safe: no-op. legacy = root / "bin" - if any(not _launcher_present(legacy, name) for name in _WINDOWS_BIN_LAUNCHERS): + if _launchers_missing(legacy): if user_path_entries is None: user_path_entries = _windows_user_path_entries() configured = {_normalize_windows_path(entry) for entry in user_path_entries} @@ -330,8 +279,8 @@ def ensure_windows_bin_launchers( def _read_user_path_raw() -> tuple[list[str], int]: """Raw (unexpanded) user PATH entries + registry value type. - Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them - (same discipline as ``hermes_cli.uninstall``). Only called on Windows. + Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them (same discipline as + ``hermes_cli.uninstall``). Only called on Windows. """ import winreg @@ -394,12 +343,10 @@ def migrate_windows_bin_path( root = Path(root) - # Same per-machine anchor as ensure_windows_bin_launchers (see there). - from hermes_constants import get_default_hermes_root, venv_bin_dir + from hermes_constants import venv_bin_dir - try: - home = Path(get_default_hermes_root()) - except Exception: + home = _default_hermes_root() + if home is None: return False if _normalize_windows_path(root.parent) != _normalize_windows_path(home): return False # not the managed clone — nothing to migrate @@ -457,17 +404,9 @@ def migrate_windows_bin_path( def _load_console_script_names(root: Path) -> list[str]: """``[project.scripts]`` names from pyproject.toml (tomllib, 3.11+).""" + project = _er._load_pyproject_project(root) try: - import tomllib - except ImportError: # pragma: no cover - return [] - pyproject = root / "pyproject.toml" - if not pyproject.is_file(): - return [] - try: - with open(pyproject, "rb") as f: - data = tomllib.load(f) - scripts = data.get("project", {}).get("scripts", {}) or {} + scripts = (project or {}).get("scripts", {}) or {} return [str(name) for name in scripts if name] except Exception: return [] @@ -476,10 +415,9 @@ def _load_console_script_names(root: Path) -> list[str]: class ShimQuarantineError(RuntimeError): """A live shim could not be renamed aside — the venv is contended (#87331). - Raised BEFORE the install command runs. Callers (early-pass recovery, - core-marker recovery) catch it like any install failure: the - update-incomplete marker survives and a later launch retries once the - holder exits — the contended venv is never mutated. + Raised BEFORE the install command runs. Callers (early-pass recovery, core-marker recovery) + catch it like any install failure: the update-incomplete marker survives and a later launch + retries once the holder exits — the contended venv is never mutated. """ def __init__(self, failed_shims: list[str]): @@ -494,14 +432,9 @@ def _quarantine_running_hermes_exe( ) -> list[tuple[Path, Path]]: """Rename live hermes*.exe shims aside so the installer can rewrite them. - Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort: - silently skips anything that cannot be renamed. Returns (original, - quarantined) pairs. stdlib-only — the console-script set comes from - pyproject ``[project.scripts]`` (fallback: the well-known trio). - - ``failed_out``: when provided, names of shims that could not be renamed - are appended so the caller can refuse instead of mutating a contended - venv (#87331 fail-closed). + Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort: silently skips anything + that cannot be renamed. Returns (original, quarantined) pairs. stdlib-only — the console-script + set comes from pyproject ``[project.scripts]`` (fallback: the well-known trio). """ if not _is_windows(): return [] @@ -529,11 +462,9 @@ def _quarantine_running_hermes_exe( def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None: """Put quarantined shims back when the installer did not replace them. - Delegates to the shared helper in the stdlib-only ``_early_recovery`` - module: one retry ladder and one recovery message for every restore site, - instead of the near-identical copies that had already drifted (#75584). - Warnings land on stderr — this module runs in the early-recovery path and - ``hermes acp`` speaks JSON-RPC on stdout. + Delegates to the shared helper in the stdlib-only ``_early_recovery`` module: one retry ladder + and one recovery message for every restore site, instead of the near-identical copies that had + already drifted (#75584). """ _er.restore_quarantined_shims(moved) @@ -541,13 +472,11 @@ def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None: def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None: """Run an install command with quarantine protection for venv shims. - Fail-closed (#87331): when any live shim cannot be renamed aside, the - venv is contended and the installer would die partway on the same locks - — raise :class:`ShimQuarantineError` WITHOUT running it. The caller's - marker-keeping failure handling turns that into "retry next launch". + Fail-closed (#87331): when any live shim cannot be renamed aside, the venv is contended and the + installer would die partway on the same locks — raise :class:`ShimQuarantineError` WITHOUT + running it. The caller's marker-keeping failure handling turns that into "retry next launch". - Raises CalledProcessError on install failure (callers implement the - per-extra fallback ladder). + Raises CalledProcessError on install failure (callers implement the per-extra fallback ladder). """ scripts_dir = _venv_scripts_dir(root) if _is_windows() else None failed: list[str] = [] @@ -574,19 +503,14 @@ def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None: def _load_installable_optional_extras(root: Path, group: str) -> list[str]: """Optional extras referenced by a dependency group (all / termux-all).""" - try: - import tomllib - - with (root / "pyproject.toml").open("rb") as handle: - project = tomllib.load(handle).get("project", {}) - except Exception: + project = _er._load_pyproject_project(root) + if project is None: return [] optional_deps = project.get("optional-dependencies", {}) if not isinstance(optional_deps, dict): return [] - refs = optional_deps.get(group, []) referenced: list[str] = [] - for ref in refs: + for ref in optional_deps.get(group, []): if "[" in ref and "]" in ref: name = ref.split("[", 1)[1].split("]", 1)[0] if name in optional_deps: @@ -597,34 +521,20 @@ def _load_installable_optional_extras(root: Path, group: str) -> list[str]: def run_core_install(root: Path) -> None: """Full core ``.[all]`` editable reinstall — the recovery install. - Equal in behavior to the install half of - ``main.py::_recover_core_update_marker_locked``: + Equal in behavior to the install half of ``main.py::_recover_core_update_marker_locked``: - - bootstrap pip via ensurepip (a killed install can leave the venv with no - pip module at all) - - prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back - to ``python -m pip`` when no uv binary is available - - target ``.[all]`` (or ``.[termux-all]`` on Termux) with the per-extra - fallback ladder when the combined extras resolve fails - - quarantine live ``hermes*.exe`` shims on Windows so they can be replaced - - route ALL install output to stderr (acp/JSON-RPC safety) - - Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env - - Raises ``subprocess.CalledProcessError`` when even the base install fails; - callers own marker lifecycle (clear on success, keep on failure). + - bootstrap pip via ensurepip (a killed install can leave the venv with no pip module at all) - + prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back to ``python -m pip`` + when no uv binary is available - target ``.[all]`` (or ``.[termux-all]`` on Termux) with the + per-extra fallback ladder when the combined extras resolve fails - quarantine live + ``hermes*.exe`` shims on Windows so they can be replaced - route ALL install output to stderr + (acp/JSON-RPC safety) - Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env """ prefix, env = _resolve_install_target(root) group = "termux-all" if _is_termux_env(env) else "all" with _stdout_to_stderr(): - try: - subprocess.run( - [sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"], - cwd=root, - capture_output=True, - ) - except Exception: - pass + _er._run_ensurepip(root) try: _run_install_cmd( @@ -669,24 +579,11 @@ def run_core_install(root: Path) -> None: def bump_marker_attempts(marker_path: Path) -> int: """Increment an attempts counter stored inside the marker file. - The marker's existence is the signal; opportunistic JSON body carries the - retry count so a persistently failing install can back off instead of - reinstall-hammering every launch. Corrupt/missing bodies restart at 1. - Returns the new attempt count. Never raises. + The marker's existence is the signal; opportunistic JSON body carries the retry count so a + persistently failing install can back off instead of reinstall-hammering every launch. + Corrupt/missing bodies restart at 1. Returns the new attempt count. Never raises. """ - attempts = 0 - try: - raw = marker_path.read_text(encoding="utf-8", errors="replace").strip() - if raw: - try: - attempts = int(json.loads(raw).get("attempts", 0)) - except (ValueError, AttributeError): - attempts = 0 - except OSError: - attempts = 0 - attempts += 1 - try: + attempts = _er._read_marker_attempts(marker_path) + 1 + with contextlib.suppress(OSError): marker_path.write_text(json.dumps({"attempts": attempts}), encoding="utf-8") - except OSError: - pass return attempts diff --git a/hermes_cli/_scan_venv_blockers.py b/hermes_cli/_scan_venv_blockers.py index ca8ceab6d8..5f7ee24c5b 100644 --- a/hermes_cli/_scan_venv_blockers.py +++ b/hermes_cli/_scan_venv_blockers.py @@ -1,12 +1,7 @@ """``hermes_cli/_scan_venv_blockers.py`` — Standalone venv-process scan for JSON consumption. -Invoked by the Desktop Electron app:: - - venv\\Scripts\\python.exe -m hermes_cli._scan_venv_blockers - -Exits 0 for valid clear or blocked results. Non-zero exit signals probe -failure (the detector itself crashed, psutil unavailable, etc.). Exactly -one JSON document on stdout; diagnostics on stderr only. +Exits 0 for valid clear or blocked results. Non-zero exit signals probe failure (the detector itself +crashed, psutil unavailable, etc.). Exactly one JSON document on stdout; diagnostics on stderr only. """ from __future__ import annotations @@ -33,10 +28,9 @@ _SENSITIVE_LONG_FLAGS: list[str] = [ def _probe_fail_json(diagnostic: str = "probe failed") -> str: """Return the standard probe-failure JSON document. - ``ok: false`` plus ``probe_failed: true`` means the detector itself could - not run — this is *not* a clear scan. Callers must treat - ``ok is not True`` / non-zero exit as probe failure, never as - ``blocked: false`` "clear" (#83149). + ``ok: false`` plus ``probe_failed: true`` means the detector itself could not run — this is + *not* a clear scan. Callers must treat ``ok is not True`` / non-zero exit as probe failure, + never as ``blocked: false`` "clear" (#83149). """ return json.dumps( { @@ -57,11 +51,7 @@ def _emit_probe_fail(diagnostic: str) -> NoReturn: def _find_flag(text: str, flag: str) -> int: - """Return the index of *flag* when it starts the string or follows a space. - - Returns -1 when not found. This avoids matching ``--token`` inside an - embedded token or path like ``/some--token-thing``. - """ + """Return the index of *flag* when it starts the string or follows a space.""" low = text.lower() fl = flag.lower() pos = 0 @@ -75,11 +65,7 @@ def _find_flag(text: str, flag: str) -> int: def _redact_sensitive_cmdline(cmdline: str) -> str: - """Apply generic secret redaction then long-flag redaction. - - If the generic redactor itself fails, return ``""`` — the PID - and process name still provide actionable diagnostics. - """ + """Apply generic secret redaction then long-flag redaction.""" # Generic pass: the project's shared secret redactor. try: from agent.redact import redact_sensitive_text # noqa: PLC0415 @@ -94,14 +80,11 @@ def _redact_sensitive_cmdline(cmdline: str) -> str: # diagnostics (toolset, port, profile). earliest = len(cmdline) for flag in _SENSITIVE_LONG_FLAGS: - # --flag=value → preserve "--flag=" - idx = _find_flag(cmdline, flag + "=") - if idx != -1 and idx + len(flag) + 1 < earliest: - earliest = idx + len(flag) + 1 - # --flag value → preserve "--flag " - idx = _find_flag(cmdline, flag + " ") - if idx != -1 and idx + len(flag) + 1 < earliest: - earliest = idx + len(flag) + 1 + # --flag=value → preserve "--flag="; --flag value → preserve "--flag " + for suffix in ("=", " "): + idx = _find_flag(cmdline, flag + suffix) + if idx != -1 and idx + len(flag) + 1 < earliest: + earliest = idx + len(flag) + 1 if earliest < len(cmdline): return cmdline[:earliest] + "" @@ -111,9 +94,8 @@ def _redact_sensitive_cmdline(cmdline: str) -> str: def _classify_local_preview_args(args: object) -> dict[str, object]: """Return safe UI metadata for an exact ``python -m http.server`` argv. - The general holder detector intentionally truncates its diagnostic command - line. Reading argv separately preserves a useful directory label without - exposing an unbounded command line to the renderer. + The general holder detector truncates its diagnostic command line; reading argv separately + preserves a useful directory label without exposing an unbounded command line to the renderer. """ if not isinstance(args, (list, tuple)) or not all(isinstance(arg, str) for arg in args): return {} @@ -123,11 +105,10 @@ def _classify_local_preview_args(args: object) -> dict[str, object]: # to an unrelated script and must never authorize termination. if len(args) < 3 or args[1] != "-m" or args[2].lower() != "http.server": return {} - module_index = 1 port = 8000 - if module_index + 2 < len(args): - candidate = args[module_index + 2] + if len(args) > 3: + candidate = args[3] if candidate.isdigit() and 0 < int(candidate) <= 65535: port = int(candidate) @@ -172,9 +153,9 @@ def _terminate_safe_preview( ) -> tuple[bool, str | None]: """Terminate one verified local preview process tree. - A fresh ``psutil.Process`` identity check and exact argv classification occur - immediately before termination. psutil guards mutating Process methods - against PID reuse, avoiding taskkill's stale-PID race. + A fresh ``psutil.Process`` identity check and exact argv classification occur immediately before + termination. psutil guards mutating Process methods against PID reuse, avoiding taskkill's + stale-PID race. """ try: if psutil_module is None: @@ -203,29 +184,13 @@ def _terminate_safe_preview( def _is_pausable_gateway(cmdline: str) -> bool: """Return True when *cmdline* is a gateway process the updater can pause. - A running gateway shows up in the venv-holder scan as one or both halves - of its launcher/worker chain (``venv\\Scripts\\python.exe -m - hermes_cli.main gateway run`` and the uv-side interpreter re-running the - same argv). Reporting those as blockers dead-ends the Desktop update: - the preflight aborts with ``venv-blocked`` *before* spawning - ``hermes-setup``, so the CLI updater's own - ``_pause_windows_gateways_for_update()`` — which exists precisely to - stop these processes (and is always active: ``hermes-setup`` invokes - ``hermes update --yes --gateway``) — never gets the chance to run. + Only gateway invocations are exempted. Anything else running from the venv (an operator's REPL, + a stray script, a ``serve`` backend that survived the desktop's own teardown) has no pause + machinery downstream and must keep blocking the handoff. - Only gateway invocations are exempted. Anything else running from the - venv (an operator's REPL, a stray script, a ``serve`` backend that - survived the desktop's own teardown) has no pause machinery downstream - and must keep blocking the handoff. - - Delegates to ``gateway.status.looks_like_gateway_command_line`` — the - canonical ``gateway run`` matcher (profile-selector aware, shlex - tokenization, ``run``-only) — so this exemption, the pause discovery, - and the updater's guard fallback all share one parser. A hand-rolled - token scan here regressed ``--profile gateway gateway run``: the profile - *value* shadowed the subcommand token. An import failure counts as - not-pausable — the scan then reports the process as a blocker, which is - exactly the pre-exemption behavior. + Delegates to ``gateway.status.looks_like_gateway_command_line`` — the canonical ``gateway run`` + matcher (profile-selector aware, shlex tokenization, ``run``-only) — so this exemption, the + pause discovery, and the updater's guard fallback all share one parser. """ try: from gateway.status import looks_like_gateway_command_line # noqa: PLC0415 @@ -237,33 +202,11 @@ def _is_pausable_gateway(cmdline: str) -> bool: def _is_updater_owned_backend(pid: int, cmdline: str) -> bool: """Return True when *pid* is a Hermes backend the CLI updater can stop. - The gateway exemption above keeps ``gateway run`` holders out of the - blocker list because the updater's own pause machinery stops and resumes - them. ``hermes serve`` / ``hermes dashboard`` backends had no such - deferral, so a leaked serve child (or a Desktop-owned backend the - teardown lost track of) dead-ended the hand-off with ``venv-blocked`` — - or, worse, survived the hand-off and made the shim quarantine fail with - ``os error 32`` (#98336) — even though the updater downstream owns - exactly this case with its ledger rungs (`_ledger_reapable_backend_pids` - reaps dead-spawner orphans; `_ledger_manual_serve_holders` stops manual - serves and relaunches them on their recorded host/port). + The gateway exemption above keeps ``gateway run`` holders out of the blocker list because the + updater's own pause machinery stops and resumes them. - Positive identity only — never name/substring matching (#90778, and the - #99558 identity-guard contract): - - - the argv's parsed SUBCOMMAND (token-based) is ``serve``/``dashboard``; - - the machine spawn ledger has a live-verified ``(pid, create_time)`` - entry for the process with a matching purpose; - - ownership is provable: the recorded spawner is dead or unrecorded - (the updater's rungs stop/relaunch those), or the spawner is an - ancestor of THIS scan — i.e. the Desktop app performing the hand-off, - which exits before the updater runs, turning the backend into exactly - the dead-spawner orphan the ledger rung reaps. - - A backend whose recorded spawner is alive and is NOT this hand-off's - Desktop (a second Desktop window, another supervisor) keeps blocking: - that supervisor would respawn whatever the updater kills. Anything - unprovable → not exempt (fail closed, pre-exemption behavior). + Positive identity only — never name/substring matching (#90778, and the #99558 identity-guard + contract): """ return _updater_owned_backend_entry(pid, cmdline) is not None @@ -271,10 +214,9 @@ def _is_updater_owned_backend(pid: int, cmdline: str) -> bool: def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None: """Ledger entry for a deferred backend, or ``None`` when it must block. - Same decision logic as ``_is_updater_owned_backend`` (which delegates - here); returning the matched ledger entry lets ``main()`` emit sanitized - decision evidence — structured identity fields only, never argv, which - can carry tokens or private endpoints (#98350). + Same decision logic as ``_is_updater_owned_backend`` (which delegates here); returning the + matched ledger entry lets ``main()`` emit sanitized decision evidence — structured identity + fields only, never argv, which can carry tokens or private endpoints (#98350). """ try: from hermes_cli.update_cmd import _hermes_holder_subcommand # noqa: PLC0415 @@ -312,10 +254,9 @@ def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None: def _deferred_backend_evidence(entries: list[dict]) -> list[dict]: """Sanitized decision evidence for deferred serve/dashboard backends. - Structured ledger fields only — pid, purpose, recorded port — never the - command line, which can carry tokens or private endpoints. Lets the - scan result explain *why* a holder disappeared from ``processes`` - without echoing argv (#98350). + Structured ledger fields only — pid, purpose, recorded port — never the command line, which can + carry tokens or private endpoints. Lets the scan result explain *why* a holder disappeared from + ``processes`` without echoing argv (#98350). """ evidence = [] for entry in entries: @@ -331,9 +272,9 @@ def _deferred_backend_evidence(entries: list[dict]) -> list[dict]: def _spawner_is_this_handoff_desktop(entry: dict) -> bool: """True when the entry's live spawner is an ancestor of this scan. - The scan subprocess is spawned by the Desktop app's update preflight, so - the Desktop performing the hand-off is in our ancestor chain. Identity is - verified by ``(pid, create_time)`` — a recycled PID cannot forge the pair. + The scan subprocess is spawned by the Desktop app's update preflight, so the Desktop performing + the hand-off is in our ancestor chain. Identity is verified by ``(pid, create_time)`` — a + recycled PID cannot forge the pair. """ spawner_pid = entry.get("spawner_pid") if not isinstance(spawner_pid, int) or spawner_pid <= 0: diff --git a/hermes_cli/_startup_fast.py b/hermes_cli/_startup_fast.py index d74b421b9c..33410bfca1 100644 --- a/hermes_cli/_startup_fast.py +++ b/hermes_cli/_startup_fast.py @@ -1,26 +1,12 @@ """Pre-import startup fast paths — THE canonical lightweight helpers. -This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import -wall (config, argparse tree, logging, providers). Everything here must stay -**stdlib-only and cheap** (os/sys file probes; no yaml, no hermes_cli.config, -no argparse). A guard test (``test_startup_fast_import_weight``) subprocess- -imports this module and fails if any heavy module sneaks into sys.modules. +This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import wall (config, argparse +tree, logging, providers). Everything here must stay **stdlib-only and cheap** (os/sys file probes; +no yaml, no hermes_cli.config, no argparse). -Why this module exists (the bug class it kills): version-printing kept being -reimplemented as ``*_fast()`` copies at the top of main.py (Termux first, -then globally), each duplicating canonical logic — project-root resolution, -container detection, profile detection. The copies drifted: eb4040242 -changed the canonical output and referenced ``PROJECT_ROOT`` inside the fast -function, which doesn't exist yet on the fast path → the Termux fast path -NameError'd on --version and nobody noticed. One implementation, imported -by both the fast path and the module constants, makes that drift -structurally impossible; the parity guard test would have caught eb4040242 -the day it landed. - -``hermes_cli/config.py``'s ``get_container_exec_info()`` reads the same -``.container-mode`` file; keep the file-format assumptions here and there in -sync (this module deliberately only PROBES existence/typos cheaply and errs -toward the slow path, which then does the authoritative parse). +Why this module exists (the bug class it kills): version-printing kept being reimplemented as +``*_fast()`` copies at the top of main.py (Termux first, then globally), each duplicating canonical +logic — project-root resolution, container detection, profile detection. """ from __future__ import annotations @@ -29,21 +15,23 @@ import os import sys __all__ = [ - "project_root_str", - "ensure_project_root_on_path", - "is_termux_env", - "is_termux_fast_version_argv", - "is_global_fast_version_argv", - "is_container_startup_environment", - "active_profile_may_override_home", - "container_mode_may_be_active", - "read_openai_version", - "read_install_method", - "print_fast_version_info", - "try_fast_version", + "project_root_str", "ensure_project_root_on_path", "is_termux_env", + "is_termux_fast_version_argv", "is_global_fast_version_argv", + "is_container_startup_environment", "active_profile_may_override_home", + "container_mode_may_be_active", "read_openai_version", "read_install_method", + "print_fast_version_info", "try_fast_version", ] +def _read_text(path: str) -> str | None: + """Read a small text file, or None when it is missing/unreadable.""" + try: + with open(path, encoding="utf-8") as handle: + return handle.read() + except (OSError, UnicodeDecodeError): + return None + + def project_root_str() -> str: """Repo root as a str — the single source for main.py's PROJECT_ROOT.""" return os.path.realpath(os.path.join(os.path.dirname(__file__), os.pardir)) @@ -54,10 +42,8 @@ def ensure_project_root_on_path() -> None: project_root = project_root_str() normalized_root = os.path.normcase(os.path.realpath(project_root)) sys.path[:] = [ - entry - for entry in sys.path - if not entry - or os.path.normcase(os.path.realpath(entry)) != normalized_root + entry for entry in sys.path + if not entry or os.path.normcase(os.path.realpath(entry)) != normalized_root ] sys.path.insert(0, project_root) @@ -66,8 +52,7 @@ def is_termux_env() -> bool: """Tiny Termux check for pre-import startup shortcuts.""" prefix = os.environ.get("PREFIX", "") return bool( - os.environ.get("TERMUX_VERSION") - or "com.termux/files/usr" in prefix + os.environ.get("TERMUX_VERSION") or "com.termux/files/usr" in prefix or prefix.startswith("/data/data/com.termux/") ) @@ -76,54 +61,39 @@ def is_termux_fast_version_argv(argv: list[str]) -> bool: return argv in (["--version"], ["-V"]) -def is_global_fast_version_argv(argv: list[str]) -> bool: - return argv in (["--version"], ["-V"]) +is_global_fast_version_argv = is_termux_fast_version_argv def is_container_startup_environment() -> bool: """True when we're already INSIDE a container (fast path is then safe).""" if os.path.exists("/.dockerenv") or os.path.exists("/run/.containerenv"): return True - try: - with open("/proc/1/cgroup", encoding="utf-8") as handle: - cgroup = handle.read() - except OSError: - return False + cgroup = _read_text("/proc/1/cgroup") or "" return "docker" in cgroup or "podman" in cgroup or "/lxc/" in cgroup def active_profile_may_override_home(hermes_root: str) -> bool: """Cheap probe: does an active non-default profile redirect HERMES_HOME?""" - active_profile = os.path.join(hermes_root, "active_profile") - try: - if os.path.exists(active_profile): - with open(active_profile, encoding="utf-8") as handle: - active = handle.read().strip() - return bool(active and active != "default") - except (OSError, UnicodeDecodeError): - pass - return False + active = (_read_text(os.path.join(hermes_root, "active_profile")) or "").strip() + return bool(active and active != "default") + + +def _default_home() -> str: + return os.path.join(os.path.expanduser("~"), ".hermes") def _resolved_home() -> str: - hermes_home = os.environ.get("HERMES_HOME", "").strip() - if hermes_home: - return hermes_home - return os.path.join(os.path.expanduser("~"), ".hermes") + return os.environ.get("HERMES_HOME", "").strip() or _default_home() def container_mode_may_be_active() -> bool: """Conservative probe for NixOS container-mode routing. - False positives are fine (we fall through to the slow path, whose - ``get_container_exec_info()`` does the authoritative check and routes - into the container). False negatives are NOT fine — they'd print the - host's version instead of the container's. Hence: any profile - ambiguity → assume container mode may be active. + False positives are fine (the slow path does the authoritative check). False negatives are NOT — + they'd print the host's version instead of the container's — so any profile ambiguity means "may + be active". """ - if os.environ.get("HERMES_DEV") == "1": - return False - if is_container_startup_environment(): + if os.environ.get("HERMES_DEV") == "1" or is_container_startup_environment(): return False hermes_home = os.environ.get("HERMES_HOME", "").strip() @@ -131,15 +101,12 @@ def container_mode_may_be_active() -> bool: if os.path.exists(os.path.join(hermes_home, ".container-mode")): return True parent_name = os.path.basename(os.path.dirname(os.path.normpath(hermes_home))) - return ( - parent_name != "profiles" - and active_profile_may_override_home(hermes_home) - ) + return parent_name != "profiles" and active_profile_may_override_home(hermes_home) - default_home = os.path.join(os.path.expanduser("~"), ".hermes") - if active_profile_may_override_home(default_home): - return True - return os.path.exists(os.path.join(default_home, ".container-mode")) + default_home = _default_home() + return active_profile_may_override_home(default_home) or os.path.exists( + os.path.join(default_home, ".container-mode") + ) def read_openai_version() -> str | None: @@ -165,31 +132,18 @@ def read_openai_version() -> str | None: def read_install_method() -> str | None: """Read the installer's ``.install_method`` stamp, if present. - Only the stamp (step 1 of ``config.detect_install_method``'s resolution - order) — the managed/git/pip fallbacks need heavier imports and stay on - the slow path. On the fast path home ambiguity is already excluded: - ``container_mode_may_be_active()`` bails to the slow path whenever a - non-default profile might redirect HERMES_HOME. + Only the stamp (step 1 of ``config.detect_install_method``'s resolution order) — the + managed/git/pip fallbacks need heavier imports and stay on the slow path. """ - stamp = os.path.join(_resolved_home(), ".install_method") - try: - with open(stamp, encoding="utf-8") as handle: - method = handle.read().strip().lower() - return method or None - except OSError: - return None + method = _read_text(os.path.join(_resolved_home(), ".install_method")) + return (method or "").strip().lower() or None def print_fast_version_info(*, check_updates: bool = True) -> None: """THE canonical ``hermes --version`` output (also used by /version). - The static lines print instantly from stdlib-only probes; everything - heavier (upstream SHA in the version line, authoritative install-method - detection, the update-status check) is lazy-imported AFTER the first - line is already on screen, so perceived latency stays instant while the - output carries the full information that used to require the (removed) - ``hermes version`` subcommand. Every lazy block degrades gracefully — - a broken/heavy import can never take the basic version output down. + Every lazy block degrades gracefully — a broken/heavy import can never take the basic version + output down. """ # Line 1: registry-owned banner label (includes "· upstream " for # git installs). banner.py keeps rich/prompt_toolkit lazy, so this @@ -240,10 +194,7 @@ def print_fast_version_info(*, check_updates: bool = True) -> None: print(f"Update available — run '{recommended_update_command()}'") elif behind and behind > 0: commits_word = "commit" if behind == 1 else "commits" - print( - f"Update available: {behind} {commits_word} behind — " - f"run '{recommended_update_command()}'" - ) + print(f"Update available: {behind} {commits_word} behind — run '{recommended_update_command()}'") elif behind == 0: print("Up to date") except Exception: @@ -253,10 +204,9 @@ def print_fast_version_info(*, check_updates: bool = True) -> None: def try_fast_version(argv: list[str] | None = None) -> bool: """Handle ``hermes --version`` before the heavy import wall. - Only ``--version``/``-V`` (the ``version`` subcommand was removed — - ``--version`` now carries the full output incl. update status), and - never when container mode may need to route the command into the - container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch. + Only ``--version``/``-V`` (the ``version`` subcommand was removed — ``--version`` now carries + the full output incl. update status), and never when container mode may need to route the + command into the container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch. """ if argv is None: argv = sys.argv[1:] @@ -266,9 +216,7 @@ def try_fast_version(argv: list[str] | None = None) -> bool: if is_termux: if not is_termux_fast_version_argv(argv): return False - elif not is_global_fast_version_argv(argv): - return False - elif container_mode_may_be_active(): + elif not is_global_fast_version_argv(argv) or container_mode_may_be_active(): return False print_fast_version_info() diff --git a/hermes_cli/build_info.py b/hermes_cli/build_info.py index e2ae06ba73..57cddc76e0 100644 --- a/hermes_cli/build_info.py +++ b/hermes_cli/build_info.py @@ -1,26 +1,9 @@ -""" -Baked-in build metadata for Hermes Agent. +"""Baked-in build metadata for Hermes Agent. -Source installs report their git revision live via ``git rev-parse`` (see -``hermes_cli/dump.py`` and ``hermes_cli/banner.py``). That doesn't work inside -the published Docker image because ``.dockerignore`` excludes ``.git``, so -those callsites fall back to ``"(unknown)"`` / drop the banner suffix entirely. - -To make ``hermes dump`` and the startup banner identify the exact commit the -image was built from, the Docker build writes the build-time ``$HERMES_GIT_SHA`` -arg into ``/.hermes_build_sha``. This module is the single -read-side helper consumed by both callsites — keeping the lookup in one place -so the file path and missing-file behaviour stay consistent. - -Behaviour: - -- Returns ``None`` when the file is absent. Source installs and dev images - built without the ``HERMES_GIT_SHA`` build-arg fall through to live-git - resolution in the caller, so non-Docker installs are unaffected. -- Returns ``None`` on any IO / decoding error. The build-sha is a nice-to-have - for support triage; nothing in the CLI is allowed to crash because of it. -- Truncates to ``short`` characters (default 8) to match the format used by - ``git rev-parse --short=8`` throughout the codebase. +Source installs report their git revision live via ``git rev-parse`` (see ``hermes_cli/dump.py`` and +``hermes_cli/banner.py``). That doesn't work inside the published Docker image because +``.dockerignore`` excludes ``.git``, so those callsites fall back to ``"(unknown)"`` / drop the +banner suffix entirely. """ from __future__ import annotations @@ -36,22 +19,27 @@ _BUILD_SHA_FILE = Path(__file__).parent.parent / ".hermes_build_sha" _code_identity_cache: Optional[dict] = None +def _read_stripped(path: Path) -> str: + return path.read_text(encoding="utf-8", errors="replace").strip() + + +def _sha_or_none(value: str) -> Optional[str]: + return value if len(value) == 40 else None + + def _resolve_git_head_sha(project_root: Path) -> Optional[str]: """Resolve the checkout's HEAD commit sha by reading .git directly. - Deliberately NOT ``git rev-parse`` in a subprocess: this helper runs - inside library paths (gateway runtime-status writes, update receipts) - where spawning processes is both slow and hostile to tests that mock - ``subprocess.run`` tightly (call-count asserts, sequenced side effects). - Handles regular checkouts, worktrees/submodules (``.git`` file with a - ``gitdir:`` pointer + ``commondir``), loose refs, and packed-refs. + Deliberately NOT ``git rev-parse``: this runs in library paths (runtime-status writes, update + receipts) where spawning is slow and hostile to tests that mock ``subprocess.run`` tightly. + Handles worktrees/submodules (``gitdir:`` + ``commondir``), loose refs, and packed-refs. Returns None on any failure. """ try: git_path = project_root / ".git" if git_path.is_file(): # Worktree/submodule: ".git" is a pointer file. - pointer = git_path.read_text(encoding="utf-8", errors="replace").strip() + pointer = _read_stripped(git_path) if not pointer.startswith("gitdir:"): return None git_dir = Path(pointer[len("gitdir:"):].strip()) @@ -66,33 +54,28 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]: common_dir = git_dir commondir_file = git_dir / "commondir" if commondir_file.is_file(): - rel = commondir_file.read_text(encoding="utf-8", errors="replace").strip() - common = Path(rel) - if not common.is_absolute(): - common = (git_dir / common).resolve() - common_dir = common + common = Path(_read_stripped(commondir_file)) + common_dir = common if common.is_absolute() else (git_dir / common).resolve() - head = (git_dir / "HEAD").read_text(encoding="utf-8", errors="replace").strip() + head = _read_stripped(git_dir / "HEAD") if not head.startswith("ref:"): # Detached HEAD: the file holds the sha itself. - return head if len(head) == 40 else None + return _sha_or_none(head) ref_name = head[len("ref:"):].strip() loose = common_dir / ref_name if loose.is_file(): - sha = loose.read_text(encoding="utf-8", errors="replace").strip() - return sha if len(sha) == 40 else None + return _sha_or_none(_read_stripped(loose)) packed = common_dir / "packed-refs" if packed.is_file(): - for line in packed.read_text(encoding="utf-8", errors="replace").splitlines(): + for line in _read_stripped(packed).splitlines(): line = line.strip() if not line or line.startswith(("#", "^")): continue parts = line.split(" ", 1) if len(parts) == 2 and parts[1].strip() == ref_name: - sha = parts[0].strip() - return sha if len(sha) == 40 else None + return _sha_or_none(parts[0].strip()) except Exception: return None return None @@ -101,34 +84,26 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]: def get_code_identity(refresh: bool = False) -> dict: """Return the running checkout's code identity as a dict. - Shape: ``{"sha": full-or-short sha | None, "short_sha": str | None, - "version": pyproject version | None, "source": "git" | "build-file" | - "unknown"}``. + Resolution order mirrors the banner/dump callsites: live ``git rev-parse`` for source installs, + the baked ``.hermes_build_sha`` for Docker images (no ``.git`` inside the published image), else + unknown. - Resolution order mirrors the banner/dump callsites: live ``git - rev-parse`` for source installs, the baked ``.hermes_build_sha`` for - Docker images (no ``.git`` inside the published image), else unknown. - - Cached per process — code identity cannot change while a process is - running (an updated checkout requires a restart to take effect, which - is exactly the property the fleet version verification relies on). - Never raises; every field degrades to ``None`` independently. + Cached per process — code identity cannot change while a process is running (an updated checkout + requires a restart to take effect, which is exactly the property the fleet version verification + relies on). Never raises; every field degrades to ``None`` independently. """ global _code_identity_cache if _code_identity_cache is not None and not refresh: return dict(_code_identity_cache) - sha: Optional[str] = None - source = "unknown" project_root = Path(__file__).parent.parent - resolved = _resolve_git_head_sha(project_root) - if resolved: - sha = resolved + source = "unknown" + sha = _resolve_git_head_sha(project_root) + if sha: source = "git" - if sha is None: - baked = get_build_sha(short=0) - if baked: - sha = baked + else: + sha = get_build_sha(short=0) + if sha: source = "build-file" version: Optional[str] = None @@ -153,9 +128,8 @@ def get_code_identity(refresh: bool = False) -> dict: def get_build_sha(short: int = 8) -> Optional[str]: """Return the baked-in build SHA, truncated to ``short`` chars, or None. - Reads ``/.hermes_build_sha`` if present. The file is - written by the Dockerfile's ``HERMES_GIT_SHA`` build-arg and contains - the full 40-character commit hash on a single line. + Reads ``/.hermes_build_sha``, written by the Dockerfile's ``HERMES_GIT_SHA`` + build-arg (full 40-char hash on one line). """ try: if not _BUILD_SHA_FILE.is_file(): diff --git a/hermes_cli/container_boot.py b/hermes_cli/container_boot.py index cc5c6ff0c0..442310dfe6 100644 --- a/hermes_cli/container_boot.py +++ b/hermes_cli/container_boot.py @@ -1,21 +1,8 @@ """Container-boot reconciliation of per-profile gateway s6 services. -Service directories under /run/service/ live on **tmpfs** and are wiped -on every container restart. Profile directories under -``$HERMES_HOME/profiles//`` live on the persistent VOLUME, and -each one records its gateway's last state in ``gateway_state.json``. -This module bridges the two: on every container boot, walk the -persistent profiles, recreate the s6 service slots, and auto-start -only those whose last recorded state was ``running``. - -Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the -Dockerfile (Phase 4 Task 4.0). Runs as root after 01-hermes-setup -(the stage2 hook) has chowned the volume and seeded $HERMES_HOME, but -before s6-rc starts user services. - -Without this module, every ``docker restart`` would silently wipe -every per-profile gateway, even though the user's profiles still -exist on disk. +Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the Dockerfile (Phase 4 Task 4.0). +Runs as root after 01-hermes-setup (the stage2 hook) has chowned the volume and seeded $HERMES_HOME, +but before s6-rc starts user services. """ from __future__ import annotations @@ -101,34 +88,10 @@ def reconcile_profile_gateways( ) -> list[ReconcileAction]: """Recreate s6 service registrations for every persistent profile. - Always registers a ``gateway-default`` slot for the root profile - (the implicit profile that lives at the top of ``$HERMES_HOME``, - not under ``profiles/``). The dispatcher in ``hermes_cli.gateway`` - maps an empty profile suffix to ``gateway-default``, so this slot - is what ``hermes gateway start`` (no ``-p``) targets. Without it, - bare ``hermes gateway start`` inside the container would land on - ``s6-svc -u /run/service/gateway-default`` → uncaught - ``CalledProcessError`` → traceback to the user (PR #30136 review). - - The default slot's prior state is read from - ``$HERMES_HOME/gateway_state.json`` (sibling to the profile root, - not under ``profiles/``); stale runtime files there are swept the - same way as for named profiles. - - Args: - hermes_home: The container's HERMES_HOME (typically /opt/data). - Profiles live under ``/profiles//``; - the default profile lives at ```` itself. - scandir: The s6 dynamic scandir (typically /run/service). Service - directories are created at ``/gateway-/``. - dry_run: When True, walk and return the action list without - touching the filesystem. For tests and `--dry-run` debug. - container_argv: Optional container PID 1 argv override. Production - reads ``/proc/1/cmdline``; tests inject it directly. - - Returns: - One :class:`ReconcileAction` per profile, in this order: - ``default`` first, then named profiles in directory order. + Always registers a ``gateway-default`` slot for the root profile (the implicit profile that + lives at the top of ``$HERMES_HOME``, not under ``profiles/``). The dispatcher in + ``hermes_cli.gateway`` maps an empty profile suffix to ``gateway-default``, so this slot is what + ``hermes gateway start`` (no ``-p``) targets. """ actions: list[ReconcileAction] = [] @@ -231,13 +194,10 @@ def _maybe_migrate_legacy_gateway_run_state( ) -> str | None: """Seed root gateway_state for pre-s6 `gateway run` containers. - The tini image let Docker users run the gateway as the container - command (`docker run ... gateway run`). After the s6 migration, - profile gateways are restored from persisted gateway_state.json; a - legacy container with no state file would therefore register the - default service down and never start. Only synthesize state when no - root gateway_state.json exists so explicit stopped/failed states keep - winning across restarts. + The tini image let Docker users run the gateway as the container command (`docker run ... + gateway run`). After the s6 migration, profile gateways are restored from persisted + gateway_state.json; a legacy container with no state file would therefore register the default + service down and never start. """ state_file = hermes_home / "gateway_state.json" if state_file.exists(): @@ -264,13 +224,9 @@ def _maybe_migrate_legacy_gateway_run_state( def _read_container_argv() -> tuple[str, ...]: """Best-effort read of the container's main program argv. - Under s6-overlay v2, PID 1 is ``/init`` and its argv contains the - ``main-wrapper.sh`` path. Under s6-overlay v3, PID 1 is - ``s6-svscan`` and the actual command (``rc.init top main-wrapper.sh - ...``) lives on a different PID. We try PID 1 first (fast path, - covers v2 and pre-s6 images), then fall back to scanning - ``/proc/*/cmdline`` for a process whose argv contains - ``main-wrapper.sh`` (the rc.init-launched PID in v3). + s6-overlay v2: PID 1 is ``/init`` and its argv holds ``main-wrapper.sh``. v3: PID 1 is + ``s6-svscan`` and the real command lives on another PID, so after the PID 1 fast path we scan + ``/proc/*/cmdline`` for a process whose argv contains ``main-wrapper.sh``. """ # Fast path: PID 1 is the command itself (s6-overlay v2 / tini). try: @@ -310,24 +266,10 @@ def _read_container_argv() -> tuple[str, ...]: def _strip_container_argv_prefix(argv: Sequence[str]) -> list[str]: """Strip the s6/wrapper prefix off the container argv, leaving the hermes args. - Two container-command argv shapes are handled: - - * **s6-overlay v2 / tini:** PID 1 argv is - ``/init /opt/hermes/docker/main-wrapper.sh [args...]``. - * **s6-overlay v3:** PID 1 is ``s6-svscan`` and the command lives on the - rc.init-launched process as ``/bin/sh -e - /run/s6/basedir/scripts/rc.init top /opt/hermes/docker/main-wrapper.sh - [args...]`` (see :func:`_read_container_argv`). - - Rather than peel each leading token positionally (which silently breaks - the moment s6 changes its launcher shape again — exactly what happened - in the v2→v3 bump), drop everything up to and including the - ``main-wrapper.sh`` token: that wrapper path is the stable boundary the - image owns, and the subcommand always follows it. Pre-s6 / direct - ``hermes`` invocations carry no wrapper, so fall back to peeling a bare - ``init`` prefix. The wrapper re-execs ``hermes ``, so an - explicit leading ``hermes`` is peeled too. Shared by the legacy-gateway - and dashboard role detectors. + Rather than peel each leading token positionally (which silently breaks the moment s6 changes + its launcher shape again — exactly what happened in the v2→v3 bump), drop everything up to and + including the ``main-wrapper.sh`` token: that wrapper path is the stable boundary the image + owns, and the subcommand always follows it. """ args = list(argv) @@ -365,19 +307,8 @@ def _is_legacy_gateway_run_request(argv: Sequence[str]) -> bool: def _is_dashboard_container(argv: Sequence[str]) -> bool: """Return True when the container's command is the dashboard. - A dashboard-only container (``hermes dashboard ...``) never spawns or - supervises per-profile gateways — that is the gateway container's job. - Reconciling profile gateway s6 slots there is not just wasted work: when - the gateway and dashboard containers share a bind-mounted HERMES_HOME, - both race to ``flock()`` the same ``logs/gateways//lock`` files, - producing "Resource busy" failures and an s6-log restart storm. So the - dashboard container skips reconciliation entirely. - - Detected from PID 1 argv (``/proc/1/cmdline``) rather than an operator - flag: the role is a fact about the container's command, not a tunable, - and a flag can be forgotten in a hand-written compose/k8s manifest — - reintroducing the exact storm this prevents. Mirrors the argv handling - in :func:`_is_legacy_gateway_run_request`. + A dashboard-only container (``hermes dashboard ...``) never spawns or supervises per-profile + gateways — that is the gateway container's job. """ args = _strip_container_argv_prefix(argv) return bool(args) and args[0] == "dashboard" @@ -386,22 +317,12 @@ def _is_dashboard_container(argv: Sequence[str]) -> bool: def _read_desired_state(profile_dir: Path) -> str | None: """Read the persisted gateway desired state for reconciliation. - Newer state files carry ``desired_state``: operator intent written by - s6 lifecycle commands. Older files only carry ``gateway_state``; keep - that as a compatibility fallback so existing running/stopped profiles - preserve their behavior until the next explicit start/stop. + Newer state files carry ``desired_state``: operator intent written by s6 lifecycle commands. + Older files only carry ``gateway_state``; keep that as a compatibility fallback so existing + running/stopped profiles preserve their behavior until the next explicit start/stop. - When falling back to ``gateway_state`` (no explicit ``desired_state``), - a transient running sub-state (``draining``) is normalised to ``running`` - — see ``_TRANSIENT_RUNNING_STATES``. A gateway hard-killed mid-drain - leaves ``draining`` as its last persisted value; without this it would be - treated as a non-autostart state and the gateway would stay DOWN forever. - An explicit ``desired_state`` is always honoured verbatim (it is the - operator's durable intent), so this normalisation only affects the - legacy/transient fallback path. - - Missing or unparseable files count as "no desired state" so we don't - bork the whole reconciliation on a corrupt file. + Missing or unparseable files count as "no desired state" so we don't bork the whole + reconciliation on a corrupt file. """ state_file = profile_dir / "gateway_state.json" if not state_file.exists(): @@ -423,9 +344,9 @@ def _read_desired_state(profile_dir: Path) -> str | None: def _cleanup_stale_runtime_files(profile_dir: Path) -> None: - """Remove gateway.pid and processes.json — they reference PIDs in - the dead container's process namespace and would otherwise confuse - the newly-started gateway's process-mismatch checks.""" + """Remove gateway.pid and processes.json — they reference PIDs in the dead container's process + namespace and would otherwise confuse the newly-started gateway's process-mismatch checks. + """ for name in _STALE_RUNTIME_FILES: (profile_dir / name).unlink(missing_ok=True) @@ -433,10 +354,9 @@ def _cleanup_stale_runtime_files(profile_dir: Path) -> None: def _read_prior_exit_label(profile_dir: Path) -> str: """How the profile's previous gateway life ended (clean/unclean/unknown). - Thin, exception-free wrapper over - :func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont-init runs - in a minimal environment and forensics must never block reconciliation - (NS-608).""" + Thin, exception-free wrapper over :func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont- + init runs in a minimal environment and forensics must never block reconciliation (NS-608). + """ try: from gateway.lifecycle_ledger import read_prior_exit_label return read_prior_exit_label(profile_dir) @@ -447,20 +367,10 @@ def _read_prior_exit_label(profile_dir: Path) -> str: def _register_service(scandir: Path, profile: str, *, start: bool) -> None: """Recreate the s6 service slot for one profile. - Mirrors the rendering in :func:`S6ServiceManager.register_profile_gateway`, - but here we control the start state directly via the ``down`` marker - file (s6-svscan honors it on rescan). Cannot use the manager - directly because the cont-init.d phase runs as root before - s6-svscan starts scanning the dynamic scandir — the manager's - ``s6-svscanctl -a`` call would fail with no control socket. - - Atomicity: build the new layout in a sibling temp directory and - rename it into place via :meth:`Path.replace`. This matches - :meth:`S6ServiceManager.register_profile_gateway` (PR #30136 - review item O4) — even though cont-init.d runs before s6-svscan - starts scanning, an atomic publication keeps the contract uniform - between the two registration paths and protects against a - half-populated dir if the script is interrupted mid-write. + Mirrors ``S6ServiceManager.register_profile_gateway`` but sets start state via the ``down`` + marker directly: cont-init.d runs as root before s6-svscan scans the dynamic scandir, so the + manager's ``s6-svscanctl -a`` would fail with no control socket. Built in a sibling temp dir + and ``Path.replace``d into place so an interrupted write never leaves a half-populated dir. """ import shutil @@ -547,18 +457,13 @@ def _write_reconcile_log( ) -> None: """Append one line per profile to $HERMES_HOME/logs/container-boot.log. - Operators inspect this to debug "why didn't my profile come back - up". Keeping a separate log file (vs. mixing into agent.log) lets - troubleshooters grep for "profile=foo" without wading through - unrelated activity. + Operators inspect this to debug "why didn't my profile come back up". Keeping a separate log + file (vs. mixing into agent.log) lets troubleshooters grep for "profile=foo" without wading + through unrelated activity. - Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES`` - (defaults to 256 KiB ≈ 3000 reconcile lines), the current file - is renamed to ``container-boot.log.1`` (replacing any previous - rotation) before the new entries are appended. This gives long- - lived containers a soft cap of ~512 KiB across the two files - without pulling in logrotate or s6-log machinery just for this - one append-only file (PR #30136 review item O3). + Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES`` (defaults to 256 KiB ≈ 3000 reconcile + lines), the current file is renamed to ``container-boot.log.1`` (replacing any previous + rotation) before the new entries are appended. """ import time log_dir = hermes_home / "logs" diff --git a/hermes_cli/cron.py b/hermes_cli/cron.py index cc19e94c6b..104f80b470 100644 --- a/hermes_cli/cron.py +++ b/hermes_cli/cron.py @@ -1,9 +1,4 @@ -""" -Cron subcommand for hermes CLI. - -Handles standalone cron management commands like list, create, edit, -pause/resume/run/remove, status, and tick. -""" +"""Cron subcommand for hermes CLI.""" import json import re @@ -52,9 +47,9 @@ def _cron_api(**kwargs): def _active_cron_provider_name() -> str: """Name of the resolved cron scheduler provider ('builtin', 'chronos', …). - Best-effort + offline (``resolve_cron_scheduler`` reads config and the - provider's ``is_available()`` contract forbids network). Returns 'builtin' - on any failure so callers fall back to the historical ticker-based checks. + Best-effort + offline (``resolve_cron_scheduler`` reads config and the provider's + ``is_available()`` contract forbids network). Returns 'builtin' on any failure so callers fall + back to the historical ticker-based checks. """ try: from cron.scheduler_provider import resolve_cron_scheduler @@ -67,13 +62,9 @@ def _active_cron_provider_name() -> str: def _builtin_gateway_liveness() -> Optional[bool]: """Tri-state liveness of the builtin cron scheduler's trigger. - Single source of truth shared by the CLI (``_warn_if_gateway_not_running``) - and the ``cronjob`` model tool (#87033): the builtin ticker only runs - inside the gateway process, so a scheduled job with no live gateway can - never fire. Non-builtin providers (e.g. Chronos) fire through their own - machinery and are deliberately exempt — a missing gateway process means - nothing for them, so they report active. ``None`` = probe failed; callers - must not claim either way. + Single source of truth shared by the CLI (``_warn_if_gateway_not_running``) and the ``cronjob`` + model tool (#87033): the builtin ticker only runs inside the gateway process, so a scheduled job + with no live gateway can never fire. Non-builtin providers (e.g. """ try: if _active_cron_provider_name() != "builtin": @@ -110,17 +101,9 @@ def _builtin_gateway_liveness() -> Optional[bool]: def _warn_if_gateway_not_running() -> None: """Warn that scheduled jobs won't fire unless the gateway is running. - The cron ticker only runs inside the gateway (``_start_cron_ticker`` in - gateway/run.py); there is no standalone cron daemon. Without a running - gateway, ``next_run_at`` passes but jobs never fire and ``last_run_at`` - stays null — the most common cron support report (#51038). Surfacing this - at create/list time, when the user is right there, prevents it. - - An external provider (e.g. Chronos) fires jobs via a NAS-mediated webhook, - NOT the in-process ticker, so a momentarily-absent gateway process does not - mean jobs won't fire — the warning would be a false alarm. Stay quiet for - any non-builtin provider; the gateway-process heuristic only speaks to the - built-in ticker's trigger. + The cron ticker only runs inside the gateway (``_start_cron_ticker`` in gateway/run.py); there + is no standalone cron daemon. Without a running gateway, ``next_run_at`` passes but jobs never + fire and ``last_run_at`` stays null — the most common cron support report (#51038). """ # _builtin_gateway_liveness never raises (it maps probe failures to None), # so no guard is needed here — False is the only warn-worthy state. @@ -157,10 +140,8 @@ def _format_lateness(seconds: float) -> str: def _dispatch_display(dispatch: dict) -> Optional[str]: """One-line scheduled-vs-actual dispatch summary for a job (#99879). - Returns None when the stamp is malformed. On-time dispatches render a - dim confirmation; late/catch-up dispatches render loudly so a run that - fired 30–150 min after gateway downtime no longer looks like an - ordinary on-time success. + None when the stamp is malformed. On-time dispatches render dim; late/catch-up dispatches render + loudly so a run fired long after gateway downtime doesn't look like an ordinary success. """ if not isinstance(dispatch, dict): return None @@ -180,9 +161,18 @@ def _dispatch_display(dispatch: dict) -> Optional[str]: ) +def _print_banner(title: str) -> None: + """Boxed cyan section header shared by ``cron list`` and ``cron incidents``.""" + print() + print(color("┌" + "─" * 73 + "┐", Colors.CYAN)) + print(color("│" + " " * 25 + title.ljust(48) + "│", Colors.CYAN)) + print(color("└" + "─" * 73 + "┘", Colors.CYAN)) + print() + + def cron_list(show_all: bool = False): """List all scheduled jobs.""" - from cron.jobs import list_jobs + from cron.jobs import effective_job_state, list_jobs jobs = list_jobs(include_disabled=show_all) @@ -191,13 +181,7 @@ def cron_list(show_all: bool = False): print(color("Create one with 'hermes cron create ...' or the /cron command in chat.", Colors.DIM)) return - print() - print(color("┌─────────────────────────────────────────────────────────────────────────┐", Colors.CYAN)) - print(color("│ Scheduled Jobs │", Colors.CYAN)) - print(color("└─────────────────────────────────────────────────────────────────────────┘", Colors.CYAN)) - print() - - from cron.jobs import effective_job_state + _print_banner("Scheduled Jobs") for job in jobs: job_id = job.get("id", "?") @@ -369,10 +353,9 @@ _INCIDENT_STATE_COLORS = { def cron_incidents(args) -> int: """List or acknowledge durable cron failure incidents. - ``hermes cron incidents [--state ]`` lists incidents (the stored error - is redacted and truncated at write time, safe for terminal display); - ``hermes cron incidents ack `` closes one so its failure ping stays - silent until the error signature changes. + ``hermes cron incidents [--state ]`` lists incidents (the stored error is redacted and + truncated at write time, safe for terminal display); ``hermes cron incidents ack `` closes + one so its failure ping stays silent until the error signature changes. """ from cron.incidents import ack_incident, list_incidents @@ -380,27 +363,12 @@ def cron_incidents(args) -> int: if action == "ack": incident_id = getattr(args, "incident_id", None) if not incident_id: - print( - color( - "✗ Incident ID required: hermes cron incidents ack ", - Colors.RED, - ) - ) + print(color("✗ Incident ID required: hermes cron incidents ack ", Colors.RED)) return 1 if ack_incident(incident_id): - print( - color( - f"✓ Incident {incident_id} acknowledged (closed).", - Colors.GREEN, - ) - ) + print(color(f"✓ Incident {incident_id} acknowledged (closed).", Colors.GREEN)) else: - print( - color( - f"Incident {incident_id} not found or already closed.", - Colors.YELLOW, - ) - ) + print(color(f"Incident {incident_id} not found or already closed.", Colors.YELLOW)) return 0 state = getattr(args, "state", None) @@ -411,30 +379,9 @@ def cron_incidents(args) -> int: print(color(f" (filtered by state '{state}')", Colors.DIM)) return 0 - print() - print( - color( - "┌─────────────────────────────────────────────────────────────────────────┐", - Colors.CYAN, - ) - ) - print( - color( - "│ Cron Failure Incidents │", - Colors.CYAN, - ) - ) - print( - color( - "└─────────────────────────────────────────────────────────────────────────┘", - Colors.CYAN, - ) - ) - print() + _print_banner("Cron Failure Incidents") for inc in incidents: - state_display = color( - inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM) - ) + state_display = color(inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM)) print(f" {color(inc['id'], Colors.YELLOW)} {state_display}") print(f" Job: {inc['job_id']}") print(f" Type: {inc.get('failure_type', 'unknown')}") @@ -457,6 +404,90 @@ def cron_incidents(args) -> int: return 0 +_PERMISSION_HINT = ( + " Hint: jobs.json may be owned by another user " + "(e.g. rewritten by a root `docker exec hermes " + "hermes cron ...`). Fix ownership to match the " + "gateway user, and prefer `docker exec -u :`." +) +_FD_EXHAUSTION_HINT = ( + " Hint: the ticker hit file-descriptor exhaustion " + "(EMFILE). The scheduler now retries with backoff and " + "attempts fd reclamation, but if the leak persists, " + "restart the gateway to recover scheduling." +) + + +def _print_ticker_health(pids: list) -> None: + """Report builtin-ticker liveness for a gateway process known to be alive. + + The gateway PROCESS is alive — but the cron ticker THREAD inside it can die silently, or stay + alive while every tick fails. Check both the liveness heartbeat and the last-successful-tick + marker so we don't report "will fire" when the ticker is dead or failing (#32612, #32895). + """ + from cron.jobs import ( + get_ticker_heartbeat_age, + get_ticker_last_error, + get_ticker_success_age, + TICKER_INTERVAL_SECONDS, + ) + from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text + + # Allow ~3 missed ticker iterations (+ a little slack) before declaring + # trouble. Derived from the shared interval constant so this threshold + # tracks the ticker cadence instead of assuming a hardcoded 60s. + STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default + hb_age = get_ticker_heartbeat_age() + ok_age = get_ticker_success_age() + pid_line = f" PID: {', '.join(map(str, pids))}" if pids else None + + def _warn(headline: str) -> None: + print(color(headline, Colors.YELLOW)) + if pid_line: + print(pid_line) + + if hb_age is None: + # No heartbeat file means the ticker thread has never started: gateway + # not in a cron-enabled profile, started moments ago (heartbeat is + # written after startup), or a config issue blocking the ticker. + _warn("⚠ Gateway is running but the cron ticker has not reported a heartbeat.") + print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.") + print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.") + print(" If heartbeat never appears, restart: hermes gateway restart") + elif hb_age > STALE_AFTER: + # No heartbeat at all → the ticker thread is gone. + _warn( + "⚠ Gateway is running but the cron ticker looks STALLED — " + f"no heartbeat for {int(hb_age)}s (expected every ~60s)." + ) + print(" Cron jobs may NOT be firing. Restart: hermes gateway restart") + elif ok_age is not None and ok_age > STALE_AFTER: + # Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a + # long time → ticks are failing every iteration. + _warn( + "⚠ Gateway and cron ticker are running, but no tick has " + f"succeeded in {int(ok_age)}s — ticks may be failing." + ) + last_error = get_ticker_last_error() + if last_error: + # Show WHY ticks fail — e.g. a root-rewritten jobs.json + # (PermissionError) that silently locked out the ticker's uid for + # ~14h in the field (#68483), or fd exhaustion (EMFILE) that used + # to stall the scheduler invisibly (#87644). + print(color(f" Last tick error: {last_error}", Colors.RED)) + if "Permission denied" in last_error: + print(color(_PERMISSION_HINT, Colors.YELLOW)) + elif _cron_is_fd_exhaustion_text(last_error): + print(color(_FD_EXHAUSTION_HINT, Colors.YELLOW)) + print(" Check the gateway log for 'Cron tick error'.") + else: + print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN)) + if pid_line: + print(pid_line) + if hb_age is not None: + print(f" Ticker heartbeat: {int(hb_age)}s ago") + + def cron_status(): """Show cron execution status.""" from cron.jobs import list_jobs @@ -483,131 +514,41 @@ def cron_status(): "due jobs are delivered by an authenticated webhook.)", Colors.DIM, )) - print() - _print_active_jobs_summary(list_jobs(include_disabled=False)) - print() - return - - pids = find_gateway_pids() - gateway_alive_via_lock = False - if not pids: - # Same false-alarm class the cronjob tool fixed (#95947): the pid scan - # can transiently miss a live gateway (just after a restart) while the - # runtime lock — held for exactly the gateway's lifetime — proves the - # ticker's process is alive. Only declare "not running" when both the - # scan AND the lock say so. - try: - from gateway.status import get_running_pid, is_gateway_runtime_lock_active - - if is_gateway_runtime_lock_active(): - gateway_alive_via_lock = True - lock_pid = get_running_pid() - if lock_pid: - pids = [lock_pid] - except Exception: - pass - if pids or gateway_alive_via_lock: - # The gateway PROCESS is alive — but the cron ticker THREAD inside it - # can die silently, or stay alive while every tick fails. Check both - # the liveness heartbeat and the last-successful-tick marker so we - # don't report "will fire" when the ticker is dead or failing - # (#32612, #32895). - from cron.jobs import ( - get_ticker_heartbeat_age, - get_ticker_last_error, - get_ticker_success_age, - TICKER_INTERVAL_SECONDS, - ) - from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text - - # Allow ~3 missed ticker iterations (+ a little slack) before declaring - # trouble. Derived from the shared interval constant so this threshold - # tracks the ticker cadence instead of assuming a hardcoded 60s. - STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default - hb_age = get_ticker_heartbeat_age() - ok_age = get_ticker_success_age() - - if hb_age is None: - # No heartbeat file means the ticker thread has never started. - # This can occur when: - # - Gateway is running but not in a profile with cron enabled, - # - Gateway was started moments ago (heartbeat is written after startup), - # - Or a configuration issue is blocking the ticker from starting at all. - print(color( - "⚠ Gateway is running but the cron ticker has not reported a heartbeat.", - Colors.YELLOW, - )) - if pids: - print(f" PID: {', '.join(map(str, pids))}") - print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.") - print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.") - print(" If heartbeat never appears, restart: hermes gateway restart") - elif hb_age > STALE_AFTER: - # No heartbeat at all → the ticker thread is gone. - print(color( - "⚠ Gateway is running but the cron ticker looks STALLED — " - f"no heartbeat for {int(hb_age)}s (expected every ~60s).", - Colors.YELLOW, - )) - if pids: - print(f" PID: {', '.join(map(str, pids))}") - print(" Cron jobs may NOT be firing. Restart: hermes gateway restart") - elif ok_age is not None and ok_age > STALE_AFTER: - # Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a - # long time → ticks are failing every iteration. - print(color( - "⚠ Gateway and cron ticker are running, but no tick has " - f"succeeded in {int(ok_age)}s — ticks may be failing.", - Colors.YELLOW, - )) - if pids: - print(f" PID: {', '.join(map(str, pids))}") - last_error = get_ticker_last_error() - if last_error: - # Show WHY ticks fail — e.g. a root-rewritten jobs.json - # (PermissionError) that silently locked out the ticker's - # uid for ~14h in the field (#68483), or fd exhaustion - # (EMFILE) that used to stall the scheduler invisibly - # (#87644). - print(color(f" Last tick error: {last_error}", Colors.RED)) - if "Permission denied" in last_error: - print(color( - " Hint: jobs.json may be owned by another user " - "(e.g. rewritten by a root `docker exec hermes " - "hermes cron ...`). Fix ownership to match the " - "gateway user, and prefer `docker exec -u :`.", - Colors.YELLOW, - )) - elif _cron_is_fd_exhaustion_text(last_error): - print(color( - " Hint: the ticker hit file-descriptor exhaustion " - "(EMFILE). The scheduler now retries with backoff and " - "attempts fd reclamation, but if the leak persists, " - "restart the gateway to recover scheduling.", - Colors.YELLOW, - )) - print(" Check the gateway log for 'Cron tick error'.") - else: - print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN)) - if pids: - print(f" PID: {', '.join(map(str, pids))}") - if hb_age is not None: - print(f" Ticker heartbeat: {int(hb_age)}s ago") else: - print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED)) - print() - print(" To enable automatic execution:") - print(" hermes gateway install # Install as a user service") - print(" sudo hermes gateway install --system # Linux servers: boot-time system service") - print(" hermes gateway # Or run in foreground") + pids = find_gateway_pids() + gateway_alive_via_lock = False + if not pids: + # Same false-alarm class the cronjob tool fixed (#95947): the pid scan + # can transiently miss a live gateway (just after a restart) while the + # runtime lock — held for exactly the gateway's lifetime — proves the + # ticker's process is alive. Only declare "not running" when both the + # scan AND the lock say so. + try: + from gateway.status import get_running_pid, is_gateway_runtime_lock_active + + if is_gateway_runtime_lock_active(): + gateway_alive_via_lock = True + lock_pid = get_running_pid() + if lock_pid: + pids = [lock_pid] + except Exception: + pass + if pids or gateway_alive_via_lock: + _print_ticker_health(pids) + else: + print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED)) + print() + print(" To enable automatic execution:") + print(" hermes gateway install # Install as a user service") + print(" sudo hermes gateway install --system # Linux servers: boot-time system service") + print(" hermes gateway # Or run in foreground") print() - _print_active_jobs_summary(list_jobs(include_disabled=False)) - print() + def _print_active_jobs_summary(jobs) -> None: """Print the ' active job(s)' + next-run line shared by every status path (built-in ticker AND external provider).""" @@ -648,9 +589,9 @@ def _print_active_jobs_summary(jobs) -> None: def _scripts_dir_for_cron() -> Path: """Return the scripts directory used by cron jobs. - Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call - so tests and profile-aware callers that monkeypatch cron storage inspect the - same Hermes home the jobs were loaded from. + Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call so tests and + profile-aware callers that monkeypatch cron storage inspect the same Hermes home the jobs were + loaded from. """ from cron.jobs import CRON_DIR @@ -778,42 +719,29 @@ def cron_doctor() -> int: return 1 -def cron_create(args): - # The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on - # every job-creation path (this CLI subcommand AND the agent's `cronjob` - # model tool, which calls create_job directly). When it blocks, create_job - # raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and - # returns it as result["error"], and the `if not result.get("success")` - # branch below prints it in red and exits 1 — same UX as before. - result = _cron_api( - action="create", - schedule=args.schedule, - prompt=args.prompt, - name=getattr(args, "name", None), - deliver=getattr(args, "deliver", None), - failure_deliver=getattr(args, "failure_deliver", None), - repeat=getattr(args, "repeat", None), - skill=getattr(args, "skill", None), - skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)), - script=getattr(args, "script", None), - workdir=getattr(args, "workdir", None), - model=getattr(args, "model", None), - provider=getattr(args, "model_provider", None), - no_agent=getattr(args, "no_agent", False) or None, - monitor_script=getattr(args, "monitor_script", None), - monitor_url=getattr(args, "monitor_url", None), - continuity=getattr(args, "continuity", None), - reasoning_effort=getattr(args, "reasoning_effort", None), - ) - if not result.get("success"): - print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED)) - return 1 - print(color(f"Created job: {result['job_id']}", Colors.GREEN)) - print(f" Name: {result['name']}") - print(f" Schedule: {result['schedule']}") - if result.get("skills"): - print(f" Skills: {', '.join(result['skills'])}") - job_data = result.get("job", {}) +_JOB_ARG_FIELDS = ( + ("name", "name"), + ("deliver", "deliver"), + ("failure_deliver", "failure_deliver"), + ("repeat", "repeat"), + ("script", "script"), + ("workdir", "workdir"), + ("model", "model"), + ("provider", "model_provider"), + ("monitor_script", "monitor_script"), + ("monitor_url", "monitor_url"), + ("continuity", "continuity"), + ("reasoning_effort", "reasoning_effort"), +) + + +def _job_api_kwargs(args) -> Dict[str, Any]: + """Collect the create/update kwargs shared by ``cron create`` and ``cron edit``.""" + return {api_key: getattr(args, attr, None) for api_key, attr in _JOB_ARG_FIELDS} + + +def _print_job_details(job_data: Dict[str, Any]) -> None: + """Print the optional Script/Monitor/Mode/Continuity/Workdir lines of a job record.""" if job_data.get("script"): print(f" Script: {job_data['script']}") if job_data.get("monitor_script"): @@ -826,6 +754,33 @@ def cron_create(args): print(" Continuity: on (each run sees the previous run's output)") if job_data.get("workdir"): print(f" Workdir: {job_data['workdir']}") + + +def cron_create(args): + # The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on + # every job-creation path (this CLI subcommand AND the agent's `cronjob` + # model tool, which calls create_job directly). When it blocks, create_job + # raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and + # returns it as result["error"], and the `if not result.get("success")` + # branch below prints it in red and exits 1 — same UX as before. + result = _cron_api( + action="create", + schedule=args.schedule, + prompt=args.prompt, + skill=getattr(args, "skill", None), + skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)), + no_agent=getattr(args, "no_agent", False) or None, + **_job_api_kwargs(args), + ) + if not result.get("success"): + print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED)) + return 1 + print(color(f"Created job: {result['job_id']}", Colors.GREEN)) + print(f" Name: {result['name']}") + print(f" Schedule: {result['schedule']}") + if result.get("skills"): + print(f" Skills: {', '.join(result['skills'])}") + _print_job_details(result.get("job", {})) print(f" Next run: {result['next_run_at']}") _warn_if_gateway_not_running() return 0 @@ -866,20 +821,9 @@ def cron_edit(args): job_id=args.job_id, schedule=getattr(args, "schedule", None), prompt=getattr(args, "prompt", None), - name=getattr(args, "name", None), - deliver=getattr(args, "deliver", None), - failure_deliver=getattr(args, "failure_deliver", None), - repeat=getattr(args, "repeat", None), skills=final_skills, - script=getattr(args, "script", None), - workdir=getattr(args, "workdir", None), - model=getattr(args, "model", None), - provider=getattr(args, "model_provider", None), no_agent=getattr(args, "no_agent", None), - monitor_script=getattr(args, "monitor_script", None), - monitor_url=getattr(args, "monitor_url", None), - continuity=getattr(args, "continuity", None), - reasoning_effort=getattr(args, "reasoning_effort", None), + **_job_api_kwargs(args), ) if not result.get("success"): print(color(f"Failed to update job: {result.get('error', 'unknown error')}", Colors.RED)) @@ -893,23 +837,12 @@ def cron_edit(args): print(f" Skills: {', '.join(updated['skills'])}") else: print(" Skills: none") - if updated.get("script"): - print(f" Script: {updated['script']}") - if updated.get("monitor_script"): - print(f" Monitor: {updated['monitor_script']} (agent runs only on output change)") - if updated.get("monitor_url"): - print(f" Monitor: {updated['monitor_url']} (agent runs only on output change)") - if updated.get("no_agent"): - print(" Mode: no-agent (script stdout delivered directly)") - if updated.get("continuity"): - print(" Continuity: on (each run sees the previous run's output)") - if updated.get("workdir"): - print(f" Workdir: {updated['workdir']}") + _print_job_details(updated) return 0 def _job_action(action: str, job_id: str, success_verb: str) -> int: - _stateless_reset = None + _stateless_token = None if action == "run": # One-shot CLI: this process exits as soon as the command returns, so # a background-dispatched run (daemon thread of THIS process) would be @@ -925,16 +858,13 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int: from gateway.session_context import _SESSION_ASYNC_DELIVERY _stateless_token = _SESSION_ASYNC_DELIVERY.set(False) - - def _stateless_reset() -> None: - _SESSION_ASYNC_DELIVERY.reset(_stateless_token) except Exception: - _stateless_reset = None + _stateless_token = None try: result = _cron_api(action=action, job_id=job_id) finally: - if _stateless_reset is not None: - _stateless_reset() + if _stateless_token is not None: + _SESSION_ASYNC_DELIVERY.reset(_stateless_token) if not result.get("success"): print(color(f"Failed to {action} job: {result.get('error', 'unknown error')}", Colors.RED)) return 1 @@ -970,14 +900,17 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int: def cron_resume(args) -> int: """Resume a paused job or explicitly re-arm a completed one-shot.""" - if bool(getattr(args, "run_at", None)) == bool(getattr(args, "run_now", False)): - if getattr(args, "run_at", None) or getattr(args, "run_now", False): + run_at = getattr(args, "run_at", None) + run_now = getattr(args, "run_now", False) + if bool(run_at) == bool(run_now): + if run_at or run_now: print(color("Use exactly one of --at or --run-now.", Colors.RED)) return 1 return _job_action("resume", args.job_id, "Resumed") from cron.jobs import AmbiguousJobReference, _hermes_now, rearm_oneshot - run_at = _hermes_now().isoformat() if args.run_now else args.run_at + if run_now: + run_at = _hermes_now().isoformat() try: job = rearm_oneshot(args.job_id, run_at) except (AmbiguousJobReference, ValueError) as exc: @@ -994,10 +927,9 @@ def cron_resume(args) -> int: def cron_notepad(args) -> int: """Handle ``hermes cron notepad [get|set|delete|list]``. - The per-job durable KV scratchpad (``cron/notepad.py``). This CLI is the - write path — a running cron agent updates its own notepad by invoking - these commands via its terminal tool; the scheduler injects non-empty - notepads into the job prompt on each run. + This CLI is the write path for the per-job durable KV scratchpad (``cron/notepad.py``): a + running cron agent updates its own notepad via its terminal tool, and the scheduler injects + non-empty notepads into the job prompt on each run. """ from cron import notepad @@ -1011,30 +943,21 @@ def cron_notepad(args) -> int: return 1 try: - if action == "set": - if key is None or value is None: - print(color("Usage: hermes cron notepad set ", Colors.RED)) + if action in ("set", "get", "delete"): + usage_args = "set " if action == "set" else f"{action} " + if key is None or (action == "set" and value is None): + print(color(f"Usage: hermes cron notepad {usage_args}", Colors.RED)) return 1 - notepad.set_note(job_id, key, value) - print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN)) - return 0 - - if action == "get": - if key is None: - print(color("Usage: hermes cron notepad get ", Colors.RED)) - return 1 - stored = notepad.get_note(job_id, key) - if stored is None: - print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW)) - return 1 - print(stored) - return 0 - - if action == "delete": - if key is None: - print(color("Usage: hermes cron notepad delete ", Colors.RED)) - return 1 - if notepad.delete_note(job_id, key): + if action == "set": + notepad.set_note(job_id, key, value) + print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN)) + return 0 + if action == "get": + stored = notepad.get_note(job_id, key) + if stored is not None: + print(stored) + return 0 + elif notepad.delete_note(job_id, key): print(color(f"Deleted notepad key '{key}' for job {job_id}.", Colors.GREEN)) return 0 print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW)) @@ -1054,52 +977,54 @@ def cron_notepad(args) -> int: return 1 +def _cron_list_cmd(args) -> int: + cron_list(getattr(args, "all", False)) + return 0 + + +def _cron_status_cmd(args) -> int: + cron_status() + return 0 + + +def _cron_runs_cmd(args) -> int: + cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20)) + return 0 + + +def _cron_remove_cmd(args) -> int: + return _job_action("remove", args.job_id, "Removed") + + +# Subcommand -> handler. Late-bound lambdas so module-level monkeypatching of +# the underlying functions keeps working. +_CRON_SUBCOMMANDS = { + "list": _cron_list_cmd, + "status": _cron_status_cmd, + "doctor": lambda args: cron_doctor(), + "tick": lambda args: cron_tick(), + "runs": _cron_runs_cmd, + "history": _cron_runs_cmd, + "incidents": lambda args: cron_incidents(args), + "notepad": lambda args: cron_notepad(args), + "create": lambda args: cron_create(args), + "add": lambda args: cron_create(args), + "edit": lambda args: cron_edit(args), + "pause": lambda args: _job_action("pause", args.job_id, "Paused"), + "resume": lambda args: cron_resume(args), + "run": lambda args: _job_action("run", args.job_id, "Triggered"), + "remove": _cron_remove_cmd, + "rm": _cron_remove_cmd, + "delete": _cron_remove_cmd, +} + + def cron_command(args): """Handle cron subcommands.""" subcmd = getattr(args, 'cron_command', None) - - if subcmd is None or subcmd == "list": - show_all = getattr(args, 'all', False) - cron_list(show_all) - return 0 - - if subcmd == "status": - cron_status() - return 0 - - if subcmd == "doctor": - return cron_doctor() - - if subcmd == "tick": - return cron_tick() - - if subcmd in {"runs", "history"}: - cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20)) - return 0 - - if subcmd == "incidents": - return cron_incidents(args) - - if subcmd == "notepad": - return cron_notepad(args) - - if subcmd in {"create", "add"}: - return cron_create(args) - - if subcmd == "edit": - return cron_edit(args) - - if subcmd == "pause": - return _job_action("pause", args.job_id, "Paused") - - if subcmd == "resume": - return cron_resume(args) - - if subcmd == "run": - return _job_action("run", args.job_id, "Triggered") - - if subcmd in {"remove", "rm", "delete"}: - return _job_action("remove", args.job_id, "Removed") + handler = _CRON_SUBCOMMANDS.get("list" if subcmd is None else subcmd) + if handler is not None: + return handler(args) print(f"Unknown cron command: {subcmd}") print("Usage: hermes cron [list|create|edit|pause|resume|run|remove|status|runs|doctor|tick]") diff --git a/hermes_cli/dep_ensure.py b/hermes_cli/dep_ensure.py index a3fbf51bae..bf5549e7fa 100644 --- a/hermes_cli/dep_ensure.py +++ b/hermes_cli/dep_ensure.py @@ -1,17 +1,13 @@ """Lazy dependency bootstrapper for non-Python runtime deps. -Detection and prompting live here in Python — not in install.sh — because: - 1. shutil.which() works on every platform; install.sh needs bash. - 2. Detection is instant; spawning bash for a "is node installed?" check is waste. - 3. Python controls the UX (rich prompts, non-interactive fallback, TTY detection). +Detection and prompting live here in Python — not in install.sh — because: 1. shutil.which() works +on every platform; install.sh needs bash. 2. Detection is instant; spawning bash for a "is node +installed?" check is waste. 3. Python controls the UX (rich prompts, non-interactive fallback, TTY +detection). -install.sh is still the *installation* backend because it has 1900 lines of -battle-tested OS detection and package-manager logic (apt/brew/pacman/dnf/ -zypper/Termux/…). Reimplementing that in Python would be huge duplication. - -Deps that degrade gracefully (ripgrep → grep fallback, ffmpeg → skip conversion) -don't need ensure_dependency wired in — only hard-fail sites do (TUI needs node, -browser tool needs agent-browser). +install.sh is still the *installation* backend because it has 1900 lines of battle-tested OS +detection and package-manager logic (apt/brew/pacman/dnf/ zypper/Termux/…). Reimplementing that in +Python would be huge duplication. """ from __future__ import annotations @@ -50,21 +46,18 @@ _DEP_DESCRIPTIONS = { def _has_system_browser() -> bool: - if _IS_WINDOWS: - names = ("chrome", "msedge", "chromium") - else: - names = ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome") - for name in names: - if shutil.which(name): - return True - return False + names = ( + ("chrome", "msedge", "chromium") if _IS_WINDOWS + else ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome") + ) + return any(shutil.which(name) for name in names) def _has_npx_agent_browser() -> bool: - """agent-browser resolves lazily via npx on the default install (#43564), - invisible to the PATH/managed-dir probes above. Mirror - tools.browser_tool.check_browser_requirements's Termux carve-out so this - check can't diverge from what browser tools actually find.""" + """agent-browser resolves lazily via npx on the default install (#43564), invisible to the + PATH/managed-dir probes above. Mirror tools.browser_tool.check_browser_requirements's Termux + carve-out so this check can't diverge from what browser tools actually find. + """ try: from tools.browser_tool import ( _find_agent_browser, @@ -74,9 +67,9 @@ def _has_npx_agent_browser() -> bool: browser_cmd = _find_agent_browser(validate=False) except Exception: return False - if not _is_npx_agent_browser_sentinel(browser_cmd): - return False - return not _requires_real_termux_browser_install(browser_cmd) + return _is_npx_agent_browser_sentinel(browser_cmd) and not _requires_real_termux_browser_install( + browser_cmd + ) def _has_hermes_agent_browser() -> bool: @@ -94,41 +87,23 @@ def _has_hermes_agent_browser() -> bool: def _find_install_script( - package_dir: Path | None = None, - repo_root: Path | None = None, + package_dir: Path | None = None, repo_root: Path | None = None ) -> tuple[Path | None, str | None]: - """Locate the install script — bundled in wheel or in git checkout. - - On Windows, prefers install.ps1; on POSIX, prefers install.sh. - Returns a (path, shell) tuple, or (None, None) if neither is found. - """ - if package_dir is None: - package_dir = Path(__file__).parent - if repo_root is None: - repo_root = package_dir.parent - + """Locate the install script — bundled in wheel or in git checkout.""" + package_dir = package_dir or Path(__file__).parent + repo_root = repo_root or package_dir.parent + candidates = [("install.sh", "bash"), ("install.ps1", "powershell")] if _IS_WINDOWS: - preferred = ("install.ps1", "powershell") - fallback = ("install.sh", "bash") - else: - preferred = ("install.sh", "bash") - fallback = ("install.ps1", "powershell") - - for script_name, shell in (preferred, fallback): - bundled = package_dir / "scripts" / script_name - if bundled.is_file(): - return bundled, shell - repo = repo_root / "scripts" / script_name - if repo.is_file(): - return repo, shell - + candidates.reverse() + for script_name, shell in candidates: + for base in (package_dir, repo_root): + script = base / "scripts" / script_name + if script.is_file(): + return script, shell return None, None -def ensure_dependency( - dep: str, - interactive: bool = True, -) -> bool: +def ensure_dependency(dep: str, interactive: bool = True) -> bool: """Ensure a non-Python dependency is available. Returns True if available.""" check = _DEP_CHECKS.get(dep) if check is None: @@ -138,15 +113,14 @@ def ensure_dependency( return True script, shell = _find_install_script() + desc = _DEP_DESCRIPTIONS.get(dep, dep) if script is None: if interactive: - desc = _DEP_DESCRIPTIONS.get(dep, dep) print(f" {desc} is not installed and no install script was found.") print(f" Install {dep} manually and try again.") return False if interactive and sys.stdin.isatty(): - desc = _DEP_DESCRIPTIONS.get(dep, dep) try: reply = input(f"{desc} is not installed. Install now? [Y/n] ").strip().lower() except (EOFError, KeyboardInterrupt): @@ -162,24 +136,13 @@ def ensure_dependency( print(" PowerShell not found. Install PowerShell or run install.ps1 manually.") return False cmd = [ - ps_bin, - "-ExecutionPolicy", "Bypass", - "-File", str(script), - "-Ensure", dep, - "-HermesHome", str(get_hermes_home()), + ps_bin, "-ExecutionPolicy", "Bypass", "-File", str(script), + "-Ensure", dep, "-HermesHome", str(get_hermes_home()), ] else: cmd = ["bash", str(script), "--ensure", dep] run_env = hermes_subprocess_env(inherit_credentials=False) run_env["IS_INTERACTIVE"] = "false" - result = subprocess.run( - cmd, - env=run_env, - ) - if result.returncode != 0: - return False - - if check: - return check() - return True + result = subprocess.run(cmd, env=run_env) + return result.returncode == 0 and check() diff --git a/hermes_cli/gitlock.py b/hermes_cli/gitlock.py index 120fd925af..f6246e6fa4 100644 --- a/hermes_cli/gitlock.py +++ b/hermes_cli/gitlock.py @@ -1,28 +1,9 @@ """Stale git lock-file recovery for update/check paths. -A crashed or killed ``git fetch`` on a shallow clone can leave -``.git/shallow.lock`` behind. Every later fetch then fails with:: +A crashed or killed ``git fetch`` on a shallow clone can leave ``.git/shallow.lock`` behind. Every +later fetch then fails with:: - fatal: Unable to create '/path/.git/shallow.lock': File exists. - -This wedges ``hermes update --check`` (hard failure) and silently degrades the -passive banner check in :mod:`hermes_cli.banner` (the fetch is swallowed, the -stale refs are compared, and the user can be told an update is available when -the checkout already contains the remote tip). Git does not self-heal these -lock files — they persist until a human removes them. - -This module provides two small, defensive helpers used by the update paths: - -* :func:`clear_stale_git_locks` — remove abandoned ``.git`` lock files (with - an age + git-process guard so a live fetch is never yanked). -* :func:`clear_stale_tmp_packs` — remove aborted-fetch ``tmp_pack_*`` / - ``tmp_idx_*`` debris from ``.git/objects/pack``. On flaky lines every - timed-out fetch leaves one behind; unchecked they accumulated to 6 GB / - hundreds of files over 9 days and eventually corrupted the pack directory - outright, permanently wedging the update check (#93732). -* :func:`is_ancestor_of_head` — ask whether a remote tip is already contained - in HEAD. Used by the shallow-clone update check to avoid reporting a false - "update available" when local cherry-picks sit on top of the remote tip. +fatal: Unable to create '/path/.git/shallow.lock': File exists. """ from __future__ import annotations @@ -32,7 +13,7 @@ import os import subprocess import time from pathlib import Path -from typing import List, Optional +from typing import Callable, Iterable, List, Optional logger = logging.getLogger(__name__) @@ -49,13 +30,23 @@ STALE_LOCK_MIN_AGE_SECONDS = 10 * 60 # process guard in :func:`clear_stale_git_locks`. LOCK_NAMES = ("shallow.lock", "index.lock", "HEAD.lock", "MERGE_HEAD.lock") +# Aborted-fetch pack debris younger than this is presumed live (a fetch may +# be writing it right now) and is never removed. A healthy fetch completes in +# minutes; the same 10-minute bar the lock sweep uses is comfortably safe. +STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS + +# Temp-file prefixes git writes into .git/objects/pack during a transfer and +# renames away on success. Anything left with these names after a fetch died +# is garbage by definition — git itself never reuses or cleans them. +_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_") + def _git_proc_running() -> bool: """True when a ``git`` process is currently running. - The conservative answer on any platform we can't probe: if we can't tell, - treat a lock as possibly-live and don't remove it. This is the safety - check that stops us from yanking a lock a real fetch is holding. + The conservative answer on any platform we can't probe: if we can't tell, treat a lock as + possibly-live and don't remove it. This is the safety check that stops us from yanking a lock a + real fetch is holding. """ try: if os.name == "nt": @@ -73,50 +64,54 @@ def _git_proc_running() -> bool: return False +def _sweep_stale( + directory: Path, + candidates: Callable[[], Iterable[Path]], + *, + min_age_seconds: Optional[int], + default_age: int, + skip_msg: str, + log_removed: Callable[[Path, int], None], +) -> List[str]: + """Shared guard + age-floor sweep. Never raises; skips anything it cannot stat/unlink.""" + if not directory.is_dir(): + return [] + if _git_proc_running(): + logger.debug(skip_msg) + return [] + cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else default_age) + removed: List[str] = [] + for entry in candidates(): + try: + if entry.is_file(): + st = entry.stat() + if st.st_mtime < cutoff: + entry.unlink() + removed.append(str(entry)) + log_removed(entry, st.st_size) + except OSError: + logger.debug("Could not clear %s (skipping)", entry, exc_info=True) + return removed + + def clear_stale_git_locks(repo_root: Path, *, min_age_seconds: Optional[int] = None) -> List[str]: """Remove abandoned ``.git`` lock files under ``repo_root``. A lock is removed only when BOTH conditions hold: - * it is older than :data:`STALE_LOCK_MIN_AGE_SECONDS` (default), and - * no ``git`` process is currently running. - - Returns the list of removed lock file paths. Never raises: a lock we - cannot stat or unlink is skipped (a concurrently-held lock may have just - been created between our age check and the unlink — the process guard - makes that window vanishingly small, and skipping is always safe). + Returns the list of removed lock file paths. Never raises: a lock we cannot stat or unlink is + skipped (a concurrently-held lock may have just been created between our age check and the + unlink — the process guard makes that window vanishingly small, and skipping is always safe). """ git_dir = Path(repo_root) / ".git" - if not git_dir.is_dir(): - return [] - - if _git_proc_running(): - logger.debug("git process running; skipping stale-lock sweep") - return [] - - cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else STALE_LOCK_MIN_AGE_SECONDS) - removed: List[str] = [] - for name in LOCK_NAMES: - lock_path = git_dir / name - try: - if lock_path.is_file() and lock_path.stat().st_mtime < cutoff: - lock_path.unlink() - removed.append(str(lock_path)) - logger.info("Removed stale git lock %s", lock_path) - except OSError: - logger.debug("Could not clear %s (skipping)", lock_path, exc_info=True) - return removed - - -# Aborted-fetch pack debris younger than this is presumed live (a fetch may -# be writing it right now) and is never removed. A healthy fetch completes in -# minutes; the same 10-minute bar the lock sweep uses is comfortably safe. -STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS - -# Temp-file prefixes git writes into .git/objects/pack during a transfer and -# renames away on success. Anything left with these names after a fetch died -# is garbage by definition — git itself never reuses or cleans them. -_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_") + return _sweep_stale( + git_dir, + lambda: [git_dir / name for name in LOCK_NAMES], + min_age_seconds=min_age_seconds, + default_age=STALE_LOCK_MIN_AGE_SECONDS, + skip_msg="git process running; skipping stale-lock sweep", + log_removed=lambda p, _size: logger.info("Removed stale git lock %s", p), + ) def clear_stale_tmp_packs( @@ -124,69 +119,28 @@ def clear_stale_tmp_packs( ) -> List[str]: """Remove aborted-fetch temp pack files under ``.git/objects/pack``. - Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped - connection) leaves a ``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file - behind, and git never cleans them up. On a flaky line the banner's - background update check produces several per day; observed in the wild - at hundreds of files / 6 GB after 9 days, after which the pack directory - corrupted outright and every fetch failed permanently (#93732). + Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped connection) leaves a + ``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file behind, and git never cleans them up. - Same safety contract as :func:`clear_stale_git_locks`: only files older - than the age floor, never while a git process is running, never raises. - Returns the removed paths. + Same safety contract as :func:`clear_stale_git_locks`: only files older than the age floor, + never while a git process is running, never raises. Returns the removed paths. """ pack_dir = Path(repo_root) / ".git" / "objects" / "pack" - if not pack_dir.is_dir(): - return [] - if _git_proc_running(): - logger.debug("git process running; skipping tmp-pack sweep") - return [] - - cutoff = time.time() - ( - min_age_seconds if min_age_seconds is not None else STALE_TMP_PACK_MIN_AGE_SECONDS - ) - removed: List[str] = [] - try: - entries = list(pack_dir.iterdir()) - except OSError: - return [] - for entry in entries: - name = entry.name - if not name.startswith(_TMP_PACK_PREFIXES): - continue + def _candidates(): try: - if entry.is_file() and entry.stat().st_mtime < cutoff: - size = entry.stat().st_size - entry.unlink() - removed.append(str(entry)) - logger.info( - "Removed aborted-fetch pack debris %s (%d bytes)", entry, size - ) + entries = list(pack_dir.iterdir()) except OSError: - logger.debug("Could not clear %s (skipping)", entry, exc_info=True) - return removed + return [] + return [e for e in entries if e.name.startswith(_TMP_PACK_PREFIXES)] - -def is_ancestor_of_head(repo_root: Path, rev: str) -> bool: - """True when ``rev`` is an ancestor of (or equal to) HEAD. - - Wraps ``git merge-base --is-ancestor HEAD``. This is the correct - question for update checks: a local cherry-pick on top of the remote tip - makes HEAD *different* from ``origin/main`` but still *contains* it, so - the answer to "is there an update?" is no. - - Returns False on any probe failure (missing rev, shallow boundary, git - error) — callers treat that as "can't prove contained", which is the - conservative direction for an update check. - """ - try: - result = subprocess.run( - ["git", "merge-base", "--is-ancestor", rev, "HEAD"], - cwd=str(repo_root), - capture_output=True, text=True, timeout=10, - ) - return result.returncode == 0 - except Exception: - logger.debug("merge-base --is-ancestor probe failed for %s", rev, exc_info=True) - return False + return _sweep_stale( + pack_dir, + _candidates, + min_age_seconds=min_age_seconds, + default_age=STALE_TMP_PACK_MIN_AGE_SECONDS, + skip_msg="git process running; skipping tmp-pack sweep", + log_removed=lambda p, size: logger.info( + "Removed aborted-fetch pack debris %s (%d bytes)", p, size + ), + ) diff --git a/hermes_cli/gui_uninstall.py b/hermes_cli/gui_uninstall.py index 90c8d5f549..52b8c1e608 100644 --- a/hermes_cli/gui_uninstall.py +++ b/hermes_cli/gui_uninstall.py @@ -1,38 +1,7 @@ -""" -Hermes Desktop (Chat GUI) uninstaller. +"""Hermes Desktop (Chat GUI) uninstaller. -The desktop GUI ships in two shapes and this module knows how to find and -remove the artifacts of both, on Linux, macOS, and Windows, WITHOUT touching -the Python agent or the user's config/data: - - 1. Source-built GUI (``hermes desktop`` / ``hermes gui``) - Built inside the agent checkout under ``$HERMES_HOME/hermes-agent/``: - - ``apps/desktop/dist`` (compiled renderer) - - ``apps/desktop/release`` (electron-builder unpacked app + installers) - - ``apps/desktop/node_modules`` and the workspace-root ``node_modules`` - (Electron itself, ~200MB) — only removed on a GUI uninstall because - the agent does not need them. - - ``$HERMES_HOME/desktop-build-stamp.json`` (the build freshness stamp) - - 2. Packaged distributable (DMG / NSIS / AppImage / deb / rpm) - Installed by the OS to a standard application location and carrying its - own bundled Electron + a per-user Electron ``userData`` directory: - - macOS: ``/Applications/Hermes.app`` or ``~/Applications/Hermes.app`` - - Windows: ``%LOCALAPPDATA%\\Programs\\Hermes`` (NSIS per-user) - - Linux: ``~/.local/share/applications`` .desktop entry + AppImage - -In both shapes the Electron runtime keeps a ``userData`` directory keyed on -the app name ("Hermes"), separate from ``$HERMES_HOME``: - - macOS: ``~/Library/Application Support/Hermes`` - - Windows: ``%APPDATA%\\Hermes`` - - Linux: ``$XDG_CONFIG_HOME/Hermes`` (default ``~/.config/Hermes``) - -This holds the desktop's own ``connection.json`` / ``updates.json`` and -Chromium cache — pure GUI state, safe to remove on a GUI uninstall. - -The functions here are deliberately import-light and side-effect-free at -import time so the Electron main process can shell out to -``hermes uninstall --gui`` (and friends) without paying for the full CLI. +This holds the desktop's own ``connection.json`` / ``updates.json`` and Chromium cache — pure GUI +state, safe to remove on a GUI uninstall. """ import os @@ -57,6 +26,12 @@ def log_warn(msg: str): print(f"{color('⚠', Colors.YELLOW)} {msg}") +def _env_dir(var: str, fallback: Path) -> Path: + """``Path($var)`` when the env var is set, else *fallback*.""" + value = os.environ.get(var) + return Path(value) if value else fallback + + # --------------------------------------------------------------------------- # Discovery # --------------------------------------------------------------------------- @@ -70,29 +45,24 @@ def _agent_root(hermes_home: Path) -> Path: def desktop_userdata_dir() -> Path: """Return the Electron ``userData`` directory for the desktop app. - Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes" - on each platform. This is GUI-only state (connection.json, updates.json, - Chromium cache) and never holds agent config or sessions. + Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes" on each platform. This + is GUI-only state (connection.json, updates.json, Chromium cache) and never holds agent config + or sessions. """ home = Path.home() if sys.platform == "darwin": return home / "Library" / "Application Support" / "Hermes" if sys.platform == "win32": - appdata = os.environ.get("APPDATA") - base = Path(appdata) if appdata else (home / "AppData" / "Roaming") - return base / "Hermes" + return _env_dir("APPDATA", home / "AppData" / "Roaming") / "Hermes" # Linux / other POSIX — XDG config home. - xdg = os.environ.get("XDG_CONFIG_HOME") - base = Path(xdg) if xdg else (home / ".config") - return base / "Hermes" + return _env_dir("XDG_CONFIG_HOME", home / ".config") / "Hermes" def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]": """GUI build artifacts produced by ``hermes desktop`` inside the checkout. - These are removable on a GUI uninstall without harming the agent: the - Python agent runs from ``hermes-agent/`` source + ``venv/`` and never - needs the Electron build output or node_modules. + These are removable on a GUI uninstall without harming the agent: the Python agent runs from + ``hermes-agent/`` source + ``venv/`` and never needs the Electron build output or node_modules. """ agent_root = _agent_root(hermes_home) desktop_dir = agent_root / "apps" / "desktop" @@ -111,9 +81,9 @@ def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]": def packaged_gui_app_paths() -> "list[Path]": """Standard install locations of the packaged desktop distributable. - Returns every candidate for the current OS; the caller filters to those - that actually exist. We never glob system-wide — only the well-known - electron-builder output locations for the "Hermes" product. + Returns every candidate for the current OS; the caller filters to those that actually exist. We + never glob system-wide — only the well-known electron-builder output locations for the "Hermes" + product. """ home = Path.home() paths: list[Path] = [] @@ -123,8 +93,7 @@ def packaged_gui_app_paths() -> "list[Path]": home / "Applications" / "Hermes.app", ] elif sys.platform == "win32": - local = os.environ.get("LOCALAPPDATA") - local_base = Path(local) if local else (home / "AppData" / "Local") + local_base = _env_dir("LOCALAPPDATA", home / "AppData" / "Local") paths += [ # NSIS per-user install (perMachine=false → Programs\Hermes). local_base / "Programs" / "Hermes", @@ -144,8 +113,7 @@ def packaged_gui_app_paths() -> "list[Path]": # ``uninstall_gui``. from hermes_cli.linux_desktop_entry import desktop_entry_path - data = os.environ.get("XDG_DATA_HOME") - data_base = Path(data) if data else (home / ".local" / "share") + data_base = _env_dir("XDG_DATA_HOME", home / ".local" / "share") paths += [ # The launcher entry `hermes desktop` installs. Its icon is # also copied into the hicolor tree (see @@ -166,52 +134,38 @@ def packaged_gui_app_paths() -> "list[Path]": def agent_is_installed(hermes_home: Path) -> bool: """Return True when a usable Python agent install exists under HERMES_HOME. - Used by the desktop UI to decide which uninstall options to offer: if the - agent isn't present (a future "lite" GUI-only client), the "remove agent" - options are hidden. + Used by the desktop UI to decide which uninstall options to offer: if the agent isn't present (a + future "lite" GUI-only client), the "remove agent" options are hidden. """ agent_root = _agent_root(hermes_home) # A real install has the package source + a venv. Either signal alone is # enough — a source checkout without a venv is still "the agent is here". - if (agent_root / "hermes_cli").is_dir(): - return True - if (agent_root / "venv").is_dir() or (agent_root / ".venv").is_dir(): - return True - return False + return any((agent_root / sub).is_dir() for sub in ("hermes_cli", "venv", ".venv")) def gui_is_installed(hermes_home: Path) -> bool: """Return True when any desktop GUI artifact exists (built or packaged).""" - for p in source_built_gui_artifacts(hermes_home): - if p.exists(): - return True - for p in packaged_gui_app_paths(): - if p.exists(): - return True - if desktop_userdata_dir().exists(): - return True - return False + return any( + p.exists() + for p in (*source_built_gui_artifacts(hermes_home), *packaged_gui_app_paths(), desktop_userdata_dir()) + ) def gui_install_summary(hermes_home: "Path | None" = None) -> dict: """Structured snapshot of what's installed, for the desktop UI to render. - Returns JSON-serializable primitives so the Electron main process can - forward it to the renderer via IPC (paths as strings, booleans for the - high-level questions the UI gates options on). + Returns JSON-serializable primitives (paths as strings, booleans for the questions the UI gates + on) so the Electron main process can forward it to the renderer via IPC. """ home: Path = hermes_home if hermes_home is not None else get_hermes_home() - - source_artifacts = [p for p in source_built_gui_artifacts(home) if p.exists()] - packaged = [p for p in packaged_gui_app_paths() if p.exists()] userdata = desktop_userdata_dir() return { "hermes_home": str(home), "agent_installed": agent_is_installed(home), "gui_installed": gui_is_installed(home), - "source_built_artifacts": [str(p) for p in source_artifacts], - "packaged_app_paths": [str(p) for p in packaged], + "source_built_artifacts": [str(p) for p in source_built_gui_artifacts(home) if p.exists()], + "packaged_app_paths": [str(p) for p in packaged_gui_app_paths() if p.exists()], "userdata_dir": str(userdata), "userdata_exists": userdata.exists(), "platform": sys.platform, @@ -242,45 +196,36 @@ def uninstall_gui( ) -> "list[Path]": """Remove the desktop GUI's artifacts, leaving the agent + user data intact. - Removes: - - source-built GUI artifacts (dist/release/node_modules/build-stamp) - - the packaged app bundle / install dir (best-effort; deb/rpm need the - system package manager and are reported, not force-removed) - - the Electron ``userData`` directory (unless ``remove_userdata=False``) - - Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any - config / sessions / .env under ``$HERMES_HOME``. - - Returns the list of paths actually removed. + Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any config / sessions / + .env under ``$HERMES_HOME``. """ home: Path = hermes_home if hermes_home is not None else get_hermes_home() removed: list[Path] = [] + def _remove_existing(paths) -> bool: + """Remove every existing path; True when at least one existed.""" + found = False + for path in paths: + if path.exists(): + found = True + if _remove_path(path): + log_success(f"Removed {path}") + removed.append(path) + return found + log_info("Removing built GUI artifacts (renderer, release, node_modules)...") - for path in source_built_gui_artifacts(home): - if path.exists() and _remove_path(path): - log_success(f"Removed {path}") - removed.append(path) + _remove_existing(source_built_gui_artifacts(home)) log_info("Removing installed desktop app...") - found_packaged = False - for path in packaged_gui_app_paths(): - if path.exists(): - found_packaged = True - if _remove_path(path): - log_success(f"Removed {path}") - removed.append(path) - if not found_packaged: + if not _remove_existing(packaged_gui_app_paths()): log_info("No packaged desktop app found in standard locations") if remove_userdata: userdata = desktop_userdata_dir() if userdata.exists(): log_info("Removing desktop app data (Electron userData)...") - if _remove_path(userdata): - log_success(f"Removed {userdata}") - removed.append(userdata) + _remove_existing([userdata]) if not removed: log_info("No desktop GUI artifacts found to remove") diff --git a/hermes_cli/heartbeat.py b/hermes_cli/heartbeat.py index 11df286869..fdb1d4bb4e 100644 --- a/hermes_cli/heartbeat.py +++ b/hermes_cli/heartbeat.py @@ -1,29 +1,12 @@ """Session heartbeats — recurring re-entry prompts for the current session. -A heartbeat is one user-owned recurring instruction bound to a session -(`/heartbeat every 10m Check the deployment and report meaningful changes`). -When due AND the session is idle, the prompt is injected as a normal user -turn — same mechanism as a /goal continuation, so message-role alternation -and prompt caching are untouched. If the agent is busy at the due moment, -the tick coalesces: it fires once when the session next goes idle, never -stacking a backlog. +This is deliberately session-scoped and in-process (CLI process or gateway process must be running) +— the durable cross-process scheduling surface remains ``hermes cron`` / the ``cronjob`` tool, which +runs in isolated sessions. -This is deliberately session-scoped and in-process (CLI process or gateway -process must be running) — the durable cross-process scheduling surface -remains ``hermes cron`` / the ``cronjob`` tool, which runs in isolated -sessions. A heartbeat is for "keep re-entering THIS conversation", the -cron system is for "run this job on a schedule". Distinct by design. - -State is persisted in SessionDB ``state_meta`` keyed by -``heartbeat:`` so ``/resume`` picks it up. - -Invariants (mirrors goals.py): -- Injection is a plain user message. No system-prompt mutation, no toolset - swap — prompt caching stays intact. -- A real user message always wins: heartbeats only fire into an idle - session with an empty input queue. -- Failures are contained: any DB/import error degrades to "no heartbeat", - never to a crashed input loop. +Invariants (mirrors goals.py): - Injection is a plain user message. No system-prompt mutation, no +toolset swap — prompt caching stays intact. - A real user message always wins: heartbeats only fire +into an idle session with an empty input queue. """ from __future__ import annotations @@ -32,8 +15,8 @@ import json import logging import re import time -from dataclasses import dataclass, asdict -from typing import Any, Dict, Optional +from dataclasses import asdict, dataclass +from typing import Any, Optional logger = logging.getLogger(__name__) @@ -68,32 +51,22 @@ _UNIT_SECONDS = { def parse_interval(text: str) -> Optional[int]: """Parse ``10m`` / ``every 2h`` / ``every 90 minutes`` into seconds. - Returns None when the text is not an interval. Values below - ``MIN_INTERVAL_SECONDS`` are rejected (returns -1 so callers can - distinguish "not an interval" from "too small"). + Returns None when the text is not an interval; values below ``MIN_INTERVAL_SECONDS`` return + -1 so callers can distinguish "not an interval" from "too small". """ - if not text: - return None - m = _INTERVAL_RE.match(text) + m = _INTERVAL_RE.match(text) if text else None if not m: return None - value = float(m.group(1)) - unit = m.group(2).lower() - seconds = int(value * _UNIT_SECONDS[unit]) - if seconds < MIN_INTERVAL_SECONDS: - return -1 - return seconds + seconds = int(float(m.group(1)) * _UNIT_SECONDS[m.group(2).lower()]) + return -1 if seconds < MIN_INTERVAL_SECONDS else seconds def format_interval(seconds: int) -> str: """Human-readable interval (``600`` → ``10m``).""" seconds = int(seconds) - if seconds % 86400 == 0: - return f"{seconds // 86400}d" - if seconds % 3600 == 0: - return f"{seconds // 3600}h" - if seconds % 60 == 0: - return f"{seconds // 60}m" + for unit, suffix in ((86400, "d"), (3600, "h"), (60, "m")): + if seconds % unit == 0: + return f"{seconds // unit}{suffix}" return f"{seconds}s" @@ -114,14 +87,11 @@ class HeartbeatState: @classmethod def from_json(cls, raw: str) -> "HeartbeatState": data = json.loads(raw) - return cls( - prompt=str(data.get("prompt") or ""), - interval_seconds=int(data.get("interval_seconds", 0) or 0), - status=str(data.get("status") or "active"), - created_at=float(data.get("created_at", 0.0) or 0.0), - last_fired_at=float(data.get("last_fired_at", 0.0) or 0.0), - fire_count=int(data.get("fire_count", 0) or 0), - ) + # Falsy/missing values fall back to the default before type coercion. + return cls(**{ + name: coerce(data.get(name) or default) + for name, (coerce, default) in _STATE_FIELDS.items() + }) def is_due(self, now: Optional[float] = None) -> bool: if self.status != "active" or not self.prompt or self.interval_seconds <= 0: @@ -137,9 +107,18 @@ class HeartbeatState: ) -# ────────────────────────────────────────────────────────────────────── +# field -> (coercer, default used when the stored value is missing/falsy) +_STATE_FIELDS = { + "prompt": (str, ""), + "interval_seconds": (int, 0), + "status": (str, "active"), + "created_at": (float, 0.0), + "last_fired_at": (float, 0.0), + "fire_count": (int, 0), +} + + # Persistence (SessionDB state_meta) — same pattern as goals.py -# ────────────────────────────────────────────────────────────────────── def _meta_key(session_id: str) -> str: @@ -159,9 +138,7 @@ def _get_session_db() -> Optional[Any]: def load_heartbeat(session_id: str) -> Optional[HeartbeatState]: - if not session_id: - return None - db = _get_session_db() + db = _get_session_db() if session_id else None if db is None: return None try: @@ -194,18 +171,15 @@ def save_heartbeat(session_id: str, state: HeartbeatState) -> None: logger.debug("HeartbeatManager: set_meta failed: %s", exc) -# ────────────────────────────────────────────────────────────────────── # Manager — the surface CLI + gateway talk to -# ────────────────────────────────────────────────────────────────────── class HeartbeatManager: """Per-session heartbeat state + due-tick decisions. - Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll - cadence while the session is idle; a non-None return is the user-role - message to inject. Firing is recorded immediately so a slow turn can't - double-fire. + Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll cadence while the session + is idle; a non-None return is the user-role message to inject. Firing is recorded immediately so + a slow turn can't double-fire. """ def __init__(self, session_id: str): @@ -232,9 +206,8 @@ class HeartbeatManager: anchor = s.last_fired_at or s.created_at next_in = max(0, int(anchor + s.interval_seconds - time.time())) return f"♥ Heartbeat (every {every}, next in ~{next_in}s{fired}): {s.prompt}" - if s.status == "paused": - return f"⏸ Heartbeat (paused, every {every}{fired}): {s.prompt}" - return f"Heartbeat ({s.status}, every {every}{fired}): {s.prompt}" + icon = "⏸ " if s.status == "paused" else "" + return f"{icon}Heartbeat ({s.status}, every {every}{fired}): {s.prompt}" # --- mutation ----------------------------------------------------- @@ -246,36 +219,31 @@ class HeartbeatManager: if interval_seconds < MIN_INTERVAL_SECONDS: raise ValueError(f"interval must be at least {MIN_INTERVAL_SECONDS}s") state = HeartbeatState( - prompt=prompt, - interval_seconds=interval_seconds, - status="active", - created_at=time.time(), + prompt=prompt, interval_seconds=interval_seconds, status="active", created_at=time.time() ) self._state = state save_heartbeat(self.session_id, state) return state - def pause(self) -> Optional[HeartbeatState]: + def _set_status(self, status: str, *, reanchor: bool = False) -> Optional[HeartbeatState]: if not self._state: return None - self._state.status = "paused" + self._state.status = status + if reanchor: + self._state.last_fired_at = time.time() save_heartbeat(self.session_id, self._state) return self._state + def pause(self) -> Optional[HeartbeatState]: + return self._set_status("paused") + def resume(self) -> Optional[HeartbeatState]: - if not self._state: - return None - self._state.status = "active" # Re-anchor so resuming doesn't instantly fire a stale tick. - self._state.last_fired_at = time.time() - save_heartbeat(self.session_id, self._state) - return self._state + return self._set_status("active", reanchor=True) def clear(self) -> bool: - if self._state is None: + if self._set_status("cleared") is None: return False - self._state.status = "cleared" - save_heartbeat(self.session_id, self._state) self._state = None return True @@ -284,10 +252,9 @@ class HeartbeatManager: def due_prompt(self, now: Optional[float] = None) -> Optional[str]: """Return the injection prompt if the heartbeat is due, else None. - Records the fire immediately (before the turn runs) so overlapping - polls or a long turn can never double-fire the same tick. Missed - ticks coalesce into one — the anchor resets to NOW, not to the - theoretical schedule. + Records the fire immediately (before the turn runs) so overlapping polls or a long turn can + never double-fire the same tick. Missed ticks coalesce into one — the anchor resets to NOW, + not to the theoretical schedule. """ s = self._state if s is None or not s.is_due(now): @@ -301,16 +268,14 @@ class HeartbeatManager: def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bool: """Carry a heartbeat across a compression session rotation. - Same shape as ``goals.migrate_goal_to_session`` — copy to the child, - archive the parent row, never raise. + Same shape as ``goals.migrate_goal_to_session`` — copy to the child, archive the parent row, + never raise. """ if not old_session_id or not new_session_id or old_session_id == new_session_id: return False try: state = load_heartbeat(old_session_id) - if state is None: - return False - if load_heartbeat(new_session_id) is not None: + if state is None or load_heartbeat(new_session_id) is not None: return False save_heartbeat(new_session_id, state) state.status = "cleared" @@ -322,14 +287,7 @@ def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bo __all__ = [ - "HeartbeatState", - "HeartbeatManager", - "parse_interval", - "format_interval", - "load_heartbeat", - "save_heartbeat", - "migrate_heartbeat_to_session", - "HEARTBEAT_PROMPT_TEMPLATE", - "MIN_INTERVAL_SECONDS", - "POLL_SECONDS", + "HeartbeatState", "HeartbeatManager", "parse_interval", "format_interval", + "load_heartbeat", "save_heartbeat", "migrate_heartbeat_to_session", + "HEARTBEAT_PROMPT_TEMPLATE", "MIN_INTERVAL_SECONDS", "POLL_SECONDS", ] diff --git a/hermes_cli/image_provenance.py b/hermes_cli/image_provenance.py index 188182e9fb..17399bfcef 100644 --- a/hermes_cli/image_provenance.py +++ b/hermes_cli/image_provenance.py @@ -1,14 +1,12 @@ """Image-authored deployment provenance for immutable Hermes runtimes. -The published image bakes ``/etc/hermes/image-provenance.json`` outside both -``$HERMES_HOME`` and the mutable checkout. A bind-mounted checkout (including -``.git``) therefore cannot hide the build fact, and environment or config -values cannot forge it. +The published image bakes ``/etc/hermes/image-provenance.json`` outside both ``$HERMES_HOME`` and +the mutable checkout. A bind-mounted checkout (including ``.git``) therefore cannot hide the build +fact, and environment or config values cannot forge it. -Absence preserves every pre-existing source/package install path. Presence -fails closed: an unreadable, non-regular, or malformed marker still means the -runtime is image-managed; it is an integrity defect, never permission to -mutate the image in place. +Absence preserves every pre-existing source/package install path. Presence fails closed: an +unreadable, non-regular, or malformed marker still means the runtime is image-managed; it is an +integrity defect, never permission to mutate the image in place. """ from __future__ import annotations @@ -55,20 +53,22 @@ def _invalid(path: Path, reason: str) -> ImageProvenance: ) +def _optional_string(payload: dict, name: str) -> Optional[str]: + value = payload.get(name) + if value is None: + return None + if not isinstance(value, str): + raise TypeError(name) + return value.strip() or None + + def read_image_provenance( marker_path: Optional[Path] = None, ) -> Optional[ImageProvenance]: """Read the baked marker without consulting environment or config. - ``None`` has one precise meaning: ``lstat`` proved that no marker exists. - Every other filesystem or validation failure returns an invalid - :class:`ImageProvenance`, so callers refuse image mutation closed. In - particular, ``lstat`` makes a dangling symlink visibly *present* and the - regular-file check rejects symlinks, directories, and device nodes. - - ``marker_path`` is a dependency-injection seam for tests and alternate - image builders. Normal callers always use the image-owned absolute path. - This function never raises. + ``marker_path`` is a dependency-injection seam for tests and alternate image builders. Normal + callers always use the image-owned absolute path. This function never raises. """ path = IMAGE_PROVENANCE_PATH @@ -110,19 +110,8 @@ def read_image_provenance( if not isinstance(manager, str) or not manager.strip(): return _invalid(path, "missing_manager") - def _optional_string(name: str) -> Optional[str]: - value = payload.get(name) - if value is None: - return None - if not isinstance(value, str): - raise TypeError(name) - value = value.strip() - return value or None - try: - image = _optional_string("image") - version = _optional_string("version") - revision = _optional_string("revision") + optional = {name: _optional_string(payload, name) for name in ("image", "version", "revision")} except TypeError as exc: return _invalid(path, f"invalid_{exc.args[0]}") @@ -130,8 +119,6 @@ def read_image_provenance( schema=IMAGE_PROVENANCE_SCHEMA, deployment_kind="image", manager=manager.strip(), - image=image, - version=version, - revision=revision, marker_path=str(path), + **optional, ) diff --git a/hermes_cli/install_identity.py b/hermes_cli/install_identity.py index 264f827ff2..a1e0502d90 100644 --- a/hermes_cli/install_identity.py +++ b/hermes_cli/install_identity.py @@ -74,8 +74,8 @@ def _fsync_directory(path: Path) -> None: def read_or_create_install_id(root: Path | None = None) -> Optional[str]: """Read or atomically mint the opaque id for the physical install. - ``None`` means the id could neither be read nor persisted. Returning an - ephemeral id would violate the authority and connection-registry contract. + ``None`` means the id could neither be read nor persisted. Returning an ephemeral id would + violate the authority and connection-registry contract. """ root = get_default_hermes_root() if root is None else root path = root / _INSTALL_ID_FILENAME diff --git a/hermes_cli/lifecycle.py b/hermes_cli/lifecycle.py index 350e3880af..cce43eeb9d 100644 --- a/hermes_cli/lifecycle.py +++ b/hermes_cli/lifecycle.py @@ -8,8 +8,7 @@ from typing import Any, List logger = logging.getLogger(__name__) -def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]: - """Notify first-party observers, then invoke compatibility plugin hooks.""" +def _observe(hook_name: str, **kwargs: Any) -> None: try: from hermes_cli.observability import observe_lifecycle @@ -17,6 +16,11 @@ def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]: except Exception: logger.warning("Built-in observability hook failed", exc_info=True) + +def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]: + """Notify first-party observers, then invoke compatibility plugin hooks.""" + _observe(hook_name, **kwargs) + from hermes_cli import plugins return plugins.invoke_hook(hook_name, **kwargs) @@ -39,12 +43,7 @@ def has_hook(hook_name: str) -> bool: def finalize_session(**kwargs: Any) -> List[Any]: """Notify observers and hard-close one core-owned Relay conversation.""" - try: - from hermes_cli.observability import observe_lifecycle - - observe_lifecycle("on_session_finalize", **kwargs) - except Exception: - logger.warning("Built-in observability hook failed", exc_info=True) + _observe("on_session_finalize", **kwargs) session_id = str(kwargs.get("session_id") or "") if session_id: diff --git a/hermes_cli/linux_desktop_entry.py b/hermes_cli/linux_desktop_entry.py index e33675590e..6cfab8697c 100644 --- a/hermes_cli/linux_desktop_entry.py +++ b/hermes_cli/linux_desktop_entry.py @@ -1,26 +1,10 @@ """Install and remove the Linux desktop entry (``hermes.desktop``). -``hermes desktop`` builds and launches the Electron app. On Linux, a -freshly-built app has no launcher presence: no menu item, no icon. This -module writes the XDG desktop entry that gives it one. -``hermes uninstall --gui`` removes the entry again. - Two values must be absolute for the entry to work: - - ``Exec`` — the launcher runs without shell ``PATH`` customizations, so - a bare ``hermes desktop`` fails when hermes lives in ``~/.local/bin`` - or a venv. Resolve the real binary and write its full path. - - ``Icon`` — an unqualified icon name needs an indexed icon theme. The - spec allows an absolute path instead, so point at the app icon in the - checkout. Do not copy the icon: ``Exec`` already depends on that tree. - -Cache refresh is best-effort and tool-gated: ``update-desktop-database`` -for the freedesktop menu cache, ``gtk-update-icon-cache`` for the user -hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5`` for Plasma. Run -each tool only when it exists. A missing tool is not an error. - -Import-light and side-effect-free at import time: the uninstaller uses -this without loading the full CLI. +Cache refresh is best-effort and tool-gated: ``update-desktop-database`` for the freedesktop menu +cache, ``gtk-update-icon-cache`` for the user hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5`` +for Plasma. Run each tool only when it exists. A missing tool is not an error. """ from __future__ import annotations @@ -60,23 +44,12 @@ def icon_path(project_root: Path) -> Path: def _running_interpreter() -> str: - """The venv-semantic interpreter path for the persisted ``Exec=`` line. - ``sys.executable`` inside a venv is commonly a SYMLINK into a shared - base-interpreter tree (uv, pyenv, conda). ``Path.resolve()`` follows it - out of the venv, and CPython discovers ``pyvenv.cfg`` from the - *lexical* argv[0] — so a dereferenced path boots without the venv's - site-packages and dies on the first third-party import (#90292, one - level up; identified in #80547's review and confirmed on real Zorin/uv - hardware in this PR's review). - - Keep the lexical path only when it actually is venv-semantic (a - ``pyvenv.cfg`` sits at or above it in the tree); otherwise the - dereferenced absolute path is the more durable form (survives the - symlink being re-pointed or its parent moving). - - Idea credit: the lexical-preservation rule was independently proposed - in #92516/#94115/#94544 and by nosliwhtes' review of this PR; the - pyvenv.cfg-detection refinement here keeps both properties. + """The venv-semantic interpreter path for the persisted ``Exec=`` line. ``sys.executable`` inside a + venv is commonly a SYMLINK into a shared base-interpreter tree (uv, pyenv, conda). + ``Path.resolve()`` follows it out of the venv, and CPython discovers ``pyvenv.cfg`` from the + *lexical* argv[0] — so a dereferenced path boots without the venv's site-packages and dies on + the first third-party import (#90292, one level up; identified in #80547's review and confirmed + on real Zorin/uv hardware in this PR's review). """ lexical = os.path.abspath(sys.executable) path = Path(lexical) @@ -92,18 +65,10 @@ _probe_cache: "dict[str, bool]" = {} def _can_import_hermes_cli(interpreter: Path) -> bool: """Whether *interpreter* can import ``hermes_cli.main`` unaided. - Runs the import in a subprocess under ``-I`` (isolated mode: no - user site, no PYTHONPATH inheritance, no cwd on ``sys.path``) from - a neutral cwd, so the answer matches what a cold desktop - environment would get — a checkout cwd or an inherited - ``PYTHONPATH`` cannot produce a false positive. Bounded by a - timeout so a hung interpreter cannot stall entry generation. - - Result is cached per interpreter path for the process lifetime, so - a desktop launch pays the subprocess cost at most once. - - Probe design per @nosliwhtes' isolated-mode capability check - (#92122 lineage, commit 4150501f641). + Runs the import in a subprocess under ``-I`` (isolated mode: no user site, no PYTHONPATH + inheritance, no cwd on ``sys.path``) from a neutral cwd, so the answer matches what a cold + desktop environment would get — a checkout cwd or an inherited ``PYTHONPATH`` cannot produce a + false positive. """ key = str(interpreter) cached = _probe_cache.get(key) @@ -134,9 +99,9 @@ def _can_import_hermes_cli(interpreter: Path) -> bool: def _running_interpreter_fallback() -> str: """The interpreter to persist when the candidate fails the import probe. - The RUNNING interpreter by definition has ``hermes_cli`` importable - (this module is executing), so the module-form entry under it is the - safe landing when every candidate path failed the capability check. + The RUNNING interpreter by definition has ``hermes_cli`` importable (this module is executing), + so the module-form entry under it is the safe landing when every candidate path failed the + capability check. """ return os.path.abspath(sys.executable) @@ -144,22 +109,11 @@ def _running_interpreter_fallback() -> str: def resolve_exec_command(project_root: Optional[Path] = None) -> str: """Build the absolute ``Exec=`` command line for ``hermes desktop``. - Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes - runs as a module with no launcher installed, use the current - interpreter, also absolute. + Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes runs as a module with no + launcher installed, use the current interpreter, also absolute. - The persisted entry must be launch-context independent: whatever - process writes it, the next launch must read and rewrite the same - bytes. ``resolve_hermes_bin()`` prefers ``sys.argv[0]``, which differs - per launch path (wrapper, repo script, ``python -m``), so for this - one caller an argv[0] that points inside the checkout is not a - durable installed launcher — skip it and resolve from PATH instead. - Otherwise a broken entry keeps regenerating itself (the repo-script - form pins a mutable uv interpreter path; the ``python -m`` form - persists a bare `` desktop`` that no DE can run). - - ``project_root`` pins which checkout counts as "internal"; defaults to - the running checkout. + The persisted entry must be launch-context independent: whatever process writes it, the next + launch must read and rewrite the same bytes. """ from hermes_cli.relaunch import resolve_hermes_bin @@ -209,17 +163,11 @@ def _resolve_hermes_bin_for_desktop_entry( ) -> Optional[str]: """Resolve the launcher binary for the persisted ``.desktop`` entry. - Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one - desktop-entry-specific rule: an ``argv[0]`` that points inside this - checkout is a launch-context artifact (the repo ``hermes`` script the - wrapper execs with, or an interpreter binary surfaced by programmatic - relaunch paths), not a durable installed launcher. Persisting it makes - the entry a function of however the previous launch happened — the - bootstrap loop behind #90492's incomplete fix. Skip argv[0]/relative - candidates in that case and fall through to PATH, where the shell - installer's wrapper lives. - - ``resolve_fn`` is injectable for tests. + Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one rule: an ``argv[0]`` inside + this checkout is a launch-context artifact, not a durable installed launcher — persisting it + makes the entry depend on how the previous launch happened (a bootstrap loop). Skip + argv[0]/relative candidates then and fall through to PATH, where the shell installer's + wrapper lives. ``resolve_fn`` is injectable for tests. """ if resolve_fn is None: from hermes_cli.relaunch import resolve_hermes_bin as resolve_fn @@ -271,12 +219,9 @@ def _resolve_hermes_bin_for_desktop_entry( def _is_interpreter(candidate: Path) -> bool: """A python interpreter binary (``bin/python*``), not a launcher. - Strict basename match — accepts ``python``, ``python3``, - ``python3.11``, ``python2.7``; rejects lookalikes such as - ``python3-config``, ``pythonw``, and anything else merely - *containing* "python". Regex approach proposed independently in - #94051; kept here with the parent-dir guard so a script named - ``python`` outside a bin/Scripts tree is not misclassified. + Strict basename match — accepts ``python``, ``python3``, ``python3.11``; rejects lookalikes + such as ``python3-config`` and ``pythonw``. The parent-dir guard keeps a script named + ``python`` outside a bin/Scripts tree from being misclassified. """ import re @@ -353,13 +298,10 @@ def _resolve_hermes_bin_for_desktop_entry( def _wrapper_shebang_safe(wrapper: Path) -> bool: """Whether an executable wrapper can actually run in the DE context. - A wrapper whose own shebang escapes the venv (``#!/usr/bin/env - python3`` or a bare interpreter name) would die exactly like the - broken entry this module exists to fix — the checkout reference in - its body does not save it. Native binaries and shell launchers are - safe by construction (they exec the right interpreter themselves). - A python-shebang wrapper is safe only when its interpreter resolves - to the RUNNING venv's interpreter directory. + A wrapper whose shebang escapes the venv (``#!/usr/bin/env python3`` or a bare interpreter + name) would die exactly like the broken entry this module fixes. Native binaries and shell + launchers are safe by construction; a python-shebang wrapper is safe only when its + interpreter resolves to the RUNNING venv's interpreter directory. """ try: with open(wrapper, "rb") as fh: @@ -406,20 +348,10 @@ def _wrapper_shebang_safe(wrapper: Path) -> bool: def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool: """Whether a candidate launcher script actually launches THIS checkout. - Expects the LEXICAL checkout root (the caller keeps it un-resolved): - the installer writes ``$INSTALL_DIR`` lexically into the shim, so on a - symlinked home the shim text and the resolved root would never match. - Both lexical and resolved forms of the root are tried regardless, to + Expects the LEXICAL checkout root (the caller keeps it un-resolved): the installer writes + ``$INSTALL_DIR`` lexically into the shim, so on a symlinked home the shim text and the resolved + root would never match. Both lexical and resolved forms of the root are tried regardless, to tolerate either caller convention. - - The installer's shim is a small bash script that execs - ``/venv/bin/python /hermes``; a venv console - script carries the venv interpreter in its shebang. Either way, a - text launcher belonging to this installation references the - checkout path (or its venv) somewhere in its first few KB. A - binary launcher (PyInstaller & friends) cannot be inspected that - way — accept it, since binary installs are self-contained and the - external-primary-first rule has already had its say. """ try: head = wrapper.read_bytes()[:4096] @@ -467,9 +399,8 @@ def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool: def _known_wrapper_candidates(): """Durable installed-launcher locations, most likely first. - Mirrors the installer's ``get_command_link_dir()`` layouts: user - (``~/.local/bin``), root FHS (``/usr/local/bin``), and Termux - (``$PREFIX/bin``). The wrapper is always named ``hermes``. + Mirrors the installer's ``get_command_link_dir()`` layouts: user (``~/.local/bin``), root FHS + (``/usr/local/bin``), and Termux (``$PREFIX/bin``). The wrapper is always named ``hermes``. """ candidates = [] home = Path.home() @@ -485,9 +416,9 @@ def _known_wrapper_candidates(): def _project_root() -> Path: """This file lives at ``/hermes_cli/linux_desktop_entry.py``. - Lexical (no .resolve()): callers feed this into shim-text matching - where the installer's lexically-written $INSTALL_DIR must be able to - match; symlinked homes would break a resolved comparison. + Lexical (no .resolve()): callers feed this into shim-text matching where the installer's + lexically-written $INSTALL_DIR must be able to match; symlinked homes would break a resolved + comparison. """ return Path(os.path.abspath(__file__)).parent.parent @@ -515,30 +446,15 @@ def _needs_interpreter(bin_path: Path) -> bool: def _shebang_escapes_running_env(shebang: str) -> bool: """Whether a python shebang resolves OUTSIDE the running interpreter's env. - Tokenizes the shebang (interpreter path plus any flags) and compares - PATH COMPONENTS, never substrings: ``/bin-extra/python`` is not - inside ``/bin`` even though it starts with it (sibling-directory - confusion; independently surfaced in nosliwhtes' #92122 hardening + Tokenizes the shebang (interpreter path plus any flags) and compares PATH COMPONENTS, never + substrings: ``/bin-extra/python`` is not inside ``/bin`` even though it starts with + it (sibling-directory confusion; independently surfaced in nosliwhtes' #92122 hardening ``b96427d0`` — reimplemented here with two extensions). - Extensions over the parent-equality form: - - * ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env`` - resolves through PATH, which in the DE's cold environment is not the - interactive PATH that installed the venv — the parent-equality form - could be fooled when the resolved ``env`` binary happens to sit in - the same directory tree. - * Flags after the interpreter (``-S``, ``-E``...) are stripped before - comparing, so a legitimate ``#!/bin/python -S`` is not - misclassified by comparing against the flag token. - - The comparison uses the LEXICAL interpreter directory (abspath, not - resolve()): on uv venvs the resolved parent is the base interpreter's - dir, which makes a valid ``.venv/bin/python`` shebang look foreign - (#94443 review case 1). Both sides use the SAME case operation - (``.lower()``): interpreter paths legitimately carry uppercase (conda - env names, usernames, uv's ephemeral build dirs) and an asymmetric - compare would flag the venv's own console script as foreign. + * ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env`` resolves through PATH, + which in the DE's cold environment is not the interactive PATH that installed the venv — the + parent-equality form could be fooled when the resolved ``env`` binary happens to sit in the same + directory tree. """ tokens = shebang[2:].strip().split() if not tokens: @@ -560,11 +476,7 @@ def _shebang_escapes_running_env(shebang: str) -> bool: def _quote_exec_arg(arg: str) -> str: - """Quote one ``Exec`` argument per the desktop entry spec. - - Reserved characters require double quotes. Inside the quotes, escape - a backslash and a double quote with a backslash. - """ + """Quote one ``Exec`` argument per the desktop entry spec.""" if not any(c in arg for c in " \t\n\"'\\><~|&;$*?#()`"): return arg escaped = arg.replace("\\", "\\\\").replace('"', '\\"') @@ -588,16 +500,12 @@ def render_desktop_entry(exec_command: str, icon: str) -> str: def refresh_desktop_databases(applications_dir: Path) -> "list[str]": - """Reindex the menu caches. Run each tool only when it exists. - - Return the names of the tools that ran (for logging and tests). - """ + """Reindex the menu caches. Run each tool only when it exists.""" ran: list[str] = [] update_db = shutil.which("update-desktop-database") - if update_db: - if _run_quiet([update_db, str(applications_dir)]): - ran.append("update-desktop-database") + if update_db and _run_quiet([update_db, str(applications_dir)]): + ran.append("update-desktop-database") # Plasma 6 first, then Plasma 5. Only one of them is ever installed. for tool in ("kbuildsycoca6", "kbuildsycoca5"): @@ -664,8 +572,8 @@ def _hicolor_icon_dest(subdir: str) -> Path: def _remove_stale_scalable_icon() -> bool: """Drop a leftover PNG from ``scalable/`` (the pre-fix install path). - Return True when a file was removed so the caller can refresh the - icon cache. A missing file is not an error. + Return True when a file was removed so the caller can refresh the icon cache. A missing file is + not an error. """ stale = _hicolor_icon_dest("scalable") try: @@ -690,9 +598,9 @@ def _refresh_hicolor_cache() -> None: def _resized_hicolor_pngs(raw: bytes) -> Optional[dict[str, bytes]]: """Lanczos-resize *raw* to each panel size. ``None`` when it will not decode. - Pillow is a core dep but this module stays import-light: the import is - local so the uninstaller does not pay it. A truncated/fake PNG (tests, - interrupted copy) returns None and the caller falls back to a copy. + Pillow is a core dep but this module stays import-light: the import is local so the uninstaller + does not pay it. A truncated/fake PNG (tests, interrupted copy) returns None and the caller + falls back to a copy. """ try: from PIL import Image @@ -728,15 +636,9 @@ def _write_hicolor_pngs(files: dict[str, bytes]) -> bool: def _install_icon_to_hicolor(icon: Path) -> bool: """Install the app icon into the user's hicolor icon theme tree. - The freedesktop icon lookup finds an installed ``apps/hermes.png`` - by the unqualified name ``hermes``, so the entry can reference the - icon without an absolute checkout path. Raster PNGs go to indexed - fixed-size dirs, never ``scalable`` (SVG-only). When the source - decodes, it is Lanczos-resized to 24/32/48/256 so Cinnamon's panel - does not nearest-neighbor a 1024px PNG. Undecodable bytes fall back - to a copy into one indexed dir. Idempotent via content-compare; - OSError caught internally (False) — the caller then falls back to - the absolute path. + The freedesktop icon lookup finds an installed ``apps/hermes.png`` by the unqualified name + ``hermes``, so the entry can reference the icon without an absolute checkout path. Raster PNGs + go to indexed fixed-size dirs, never ``scalable`` (SVG-only). """ try: raw = icon.read_bytes() @@ -762,8 +664,8 @@ def _install_icon_to_hicolor(icon: Path) -> bool: def install_desktop_entry(project_root: Path) -> Optional[Path]: """Write (or refresh) the Hermes desktop entry. Return its path. - Return ``None`` on non-Linux platforms or when the write fails. This - is a convenience, never a reason to fail a launch. + Return ``None`` on non-Linux platforms or when the write fails. This is a convenience, never a + reason to fail a launch. """ if not is_supported(): return None diff --git a/hermes_cli/macos_tcc_anchor.py b/hermes_cli/macos_tcc_anchor.py index 3bceddb7ec..cc08a113ca 100644 --- a/hermes_cli/macos_tcc_anchor.py +++ b/hermes_cli/macos_tcc_anchor.py @@ -1,40 +1,11 @@ """Stable macOS TCC anchor for the uv-managed Python interpreter (#95596). -Re-land of the interpreter anchor reverted in #95563. macOS keys TCC grants -to the resolved absolute path of the client binary. Hermes' interpreter is -managed by uv and lives at a versioned store path; every patch bump orphans -every prior grant (#85345). +1. Aliases are materialized as real-file copies of the anchor, never symlinks. 2. If the store ships +``libpython*``, it is hardlinked into ``venv/lib/`` (copy if the store is on another device). +Existing ``LC_RPATH`` already points at ``@executable_path/../lib`` — no rewrite. 3. -The first landing copied the interpreter into ``venv/bin/python`` but left -two holes that bricked real Macs: - -* Dynamically-linked builds look up ``libpython`` via - ``@executable_path/../lib``. That resolved into ``venv/lib/``, which had - no dylib — every hermes command, including update/doctor, died in dyld - (#95425). -* Alias names (``python3``, ``python3.N``) were re-pointed at the copy as - *symlinks*. Invoking the copied interpreter through a symlink makes - CPython getpath lose the venv prefix on affected python-build-standalone - builds — startup dies with ``ModuleNotFoundError: encodings`` and the - stdlib resolves to the build-time ``/install`` prefix (#95541). Console - scripts exec ``python3``, so the entire CLI surface died. - -This re-land keeps the copy + identifier-pinned signature (TCC attribution -stays on the stable venv path) and closes both holes: - -1. Aliases are materialized as real-file copies of the anchor, never - symlinks. -2. If the store ships ``libpython*``, it is hardlinked into ``venv/lib/`` - (copy if the store is on another device). Existing ``LC_RPATH`` already - points at ``@executable_path/../lib`` — no rewrite. -3. A pre-install boot gate actually launches the staged copy and demands - ``import encodings`` plus ``sys.prefix == ``. Failure rolls the - staging file back and leaves the live interpreter untouched (a surplus - provisioned dylib in ``venv/lib/`` may remain — harmless), so a bad - anchor can never brick update/doctor again. - -All functions are no-ops on non-macOS and for interpreters that are not -uv-managed. Best-effort: never raises to callers. +All functions are no-ops on non-macOS and for interpreters that are not uv-managed. Best-effort: +never raises to callers. """ from __future__ import annotations @@ -68,9 +39,9 @@ class _BootGateFailed(Exception): def _marker_value(source_file: Path) -> str: - """Canonical marker value: fully resolved so symlinked spellings of the - same store binary (``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``) - compare equal.""" + """Canonical marker value: fully resolved so symlinked spellings of the same store binary + (``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``) compare equal. + """ return os.path.realpath(str(source_file)) @@ -167,9 +138,9 @@ def _anchor_marker(venv_bin: Path) -> Path: def _write_marker(venv_bin: Path, source_file: Path) -> None: """Write the anchor marker atomically via the shared helper. - A concurrent ensure (update + doctor --fix) must never observe a - partially-written marker: a torn read would compare unequal and trigger - a spurious reinstall, and ``write_text`` alone is not atomic. + A concurrent ensure (update + doctor --fix) must never observe a partially-written marker: a + torn read would compare unequal and trigger a spurious reinstall, and ``write_text`` alone is + not atomic. """ atomic_write_text( _anchor_marker(venv_bin), @@ -188,8 +159,8 @@ def _provision_libpython( ) -> None: """Hardlink (else copy) store ``libpython*`` into ``venv/lib/``. - Provision-if-present: a surplus hardlink on a statically-linked build is - free; a missed detection is the only way #95425 returns. + Provision-if-present: a surplus hardlink on a statically-linked build is free; a missed + detection is the only way #95425 returns. """ src_lib = _store_root(source_file) / "lib" if not src_lib.is_dir(): @@ -222,10 +193,9 @@ def _provision_libpython( def _copy_alias(venv_bin: Path, name: str, anchor: Path) -> bool: """Materialize *name* as a real-file copy of *anchor* (atomic rename). - Returns False (and warns) on failure: a leftover alias *symlink* to the - anchor is the exact #95541 crash shape, so callers must know when the - alias set is incomplete. The staging name is unique (mkstemp) so a - concurrent ensure (update + doctor --fix) cannot promote a truncated + Returns False (and warns) on failure: a leftover alias *symlink* to the anchor is the exact + #95541 crash shape, so callers must know when the alias set is incomplete. The staging name is + unique (mkstemp) so a concurrent ensure (update + doctor --fix) cannot promote a truncated interim copy. """ tmp_path: Path | None = None @@ -278,13 +248,10 @@ def _materialize_aliases( def _passes_boot_gate(staged: Path, venv_dir: Path) -> bool: """Launch *staged* and demand encodings + the venv prefix. - The probe runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited - ``PYTHONHOME`` papers over exactly the prefix-resolution failure the gate - exists to catch. ``OSError`` is split by errno — ``ENOENT``/``ENOEXEC`` - (fixture binaries, foreign-arch images) means the binary cannot run here - at all, so the symlinked venv was equally dead and installing cannot make - things worse: skip. Anything else (notably ``EACCES`` after our own - chmod) is a broken install about to go live: refuse. + Runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited PYTHONHOME papers over the very + prefix failure this gate exists to catch. ``ENOENT``/``ENOEXEC`` (fixture binaries, foreign + arch) mean the binary can't run here at all, so the symlinked venv was equally dead: skip. + Anything else (notably ``EACCES`` after our own chmod) is a broken install going live: refuse. """ env = { k: v @@ -368,10 +335,9 @@ def _install_anchor(venv_dir: Path, source_file: Path) -> None: def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None: """Pin a dylib-complete interpreter anchor for macOS TCC (#95596). - No-op (returns None) on non-macOS, when no venv interpreter exists, or - when the interpreter is not uv-managed. Idempotent. Best-effort — - returns None (and logs) if the copy or boot-gate fails; callers must - never depend on success. + No-op (returns None) on non-macOS, when no venv interpreter exists, or when the interpreter is + not uv-managed. Idempotent. Best-effort — returns None (and logs) if the copy or boot-gate + fails; callers must never depend on success. """ if not is_macos(): return None @@ -411,14 +377,11 @@ def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None: def tcc_anchor_state(project_root: Path | None = None) -> tuple[str, str]: - """Report the anchor state for ``hermes doctor``. + """Report the anchor state for ``hermes doctor`` as ``(status, detail)``. - Returns ``(status, detail)`` with status one of: - - - ``"skip"`` — not applicable (non-macOS, no venv, or not uv-managed) - - ``"active"`` — venv interpreter is pinned at a stable real-file anchor - - ``"stale"`` — pinned but the interpreter changed since the last copy - - ``"missing"`` — uv-managed interpreter with no stable anchor installed + ``skip`` = not applicable (non-macOS, no venv, not uv-managed); ``active`` = pinned at a + stable real-file anchor; ``stale`` = pinned but the interpreter changed since the last copy; + ``missing`` = uv-managed interpreter with no anchor installed. """ if not is_macos(): return "skip", "not macOS" diff --git a/hermes_cli/managed_uv.py b/hermes_cli/managed_uv.py index 704e74435c..15bf6c0798 100644 --- a/hermes_cli/managed_uv.py +++ b/hermes_cli/managed_uv.py @@ -1,20 +1,8 @@ """Hermes-managed uv and Python runtime repair. -Hermes owns its own uv binary at ``$HERMES_HOME/bin/uv`` (or ``uv.exe`` on -Windows). Every code path that needs uv resolves it from that single location. -If the binary is missing, ``ensure_uv()`` bootstraps it via the official -standalone installer with ``UV_UNMANAGED_INSTALL`` / ``UV_INSTALL_DIR`` pointed -at ``$HERMES_HOME/bin`` so the installer writes directly there — no PATH -probing, no conda guards, no multi-location resolution chains. - -The Python backing the install is different: it is shared by every Hermes -profile because the checkout's ``venv`` is shared. Runtime repair therefore -uses an install-scoped store under ``/.hermes-runtime/python``. A -vulnerable interpreter is never reinstalled in place. We provision a new -immutable Python generation, build and smoke-test a relocatable sibling venv, -then cut over with same-filesystem renames. The old venv remains available for -synchronous rollback and is parked for cleanup after the updating process -releases it. +The Python backing the install is different: it is shared by every Hermes profile because the +checkout's ``venv`` is shared. Runtime repair therefore uses an install-scoped store under +``/.hermes-runtime/python``. A vulnerable interpreter is never reinstalled in place. """ from __future__ import annotations @@ -35,7 +23,11 @@ from pathlib import Path from typing import Callable, Optional from hermes_constants import get_hermes_home -from hermes_cli.sqlite_runtime import SQLiteRuntimeInfo, probe_sqlite_runtime +from hermes_cli.sqlite_runtime import ( + SQLiteRuntimeInfo, + isolated_interpreter_env, + probe_sqlite_runtime, +) logger = logging.getLogger(__name__) @@ -52,23 +44,15 @@ _MACOS_MANAGED_PYTHON_IDENTIFIER = "com.nousresearch.hermes.managed-python" def managed_uv_path() -> Path: - """Return the path where Hermes keeps *its* uv binary. + """Return the path where Hermes keeps *its* uv binary (``$HERMES_HOME/bin/uv[.exe]``). - ``$HERMES_HOME/bin/uv`` on POSIX, ``$HERMES_HOME\\bin\\uv.exe`` on - Windows. The directory may not exist yet — callers should use - ``ensure_uv()`` to bootstrap it. + The directory may not exist yet -- callers should use ``ensure_uv()`` to bootstrap it. """ - home = get_hermes_home() - if platform.system() == "Windows": - return home / "bin" / "uv.exe" - return home / "bin" / "uv" + return get_hermes_home() / "bin" / ("uv.exe" if platform.system() == "Windows" else "uv") def resolve_uv() -> Optional[str]: - """Return the managed uv path if it exists, else ``None``. - - No side effects — pure lookup. - """ + """Return the managed uv path if it exists, else ``None``.""" p = managed_uv_path() if p.is_file() and os.access(p, os.X_OK): return str(p) @@ -120,15 +104,14 @@ def managed_python_env( def _macos_sign_managed_python(python: Path) -> bool: """Give a newly downloaded managed Python a stable macOS code identity. - python-build-standalone binaries are ad-hoc signed, which leaves macOS - TCC with a cdhash-only identity that changes whenever Hermes provisions a - new runtime generation. An identifier-pinned designated requirement - gives those generations a stable identity even when no Developer ID + python-build-standalone binaries are ad-hoc signed, which leaves macOS TCC with a cdhash-only + identity that changes whenever Hermes provisions a new runtime generation. An identifier-pinned + designated requirement gives those generations a stable identity even when no Developer ID certificate is available locally. - Signing is deliberately best effort. Runtime repair exists to remove a - security vulnerability, so an unavailable or incompatible ``codesign`` - must not prevent the fixed interpreter from being installed. + Signing is deliberately best effort. Runtime repair exists to remove a security vulnerability, + so an unavailable or incompatible ``codesign`` must not prevent the fixed interpreter from being + installed. """ if platform.system() != "Darwin": return False @@ -145,45 +128,29 @@ def _macos_sign_managed_python(python: Path) -> bool: f'"{_MACOS_MANAGED_PYTHON_IDENTIFIER}"' ) try: - signed = subprocess.run( - [ - codesign, - "--force", - "--deep", - "--sign", - "-", - "--timestamp=none", - "--identifier", - _MACOS_MANAGED_PYTHON_IDENTIFIER, - "--requirements", - requirement, - str(python), - ], - check=False, - capture_output=True, - text=True, - ) - if signed.returncode != 0: - logger.warning( + steps = ( + ( + [ + codesign, "--force", "--deep", "--sign", "-", "--timestamp=none", + "--identifier", _MACOS_MANAGED_PYTHON_IDENTIFIER, + "--requirements", requirement, str(python), + ], "could not stably sign managed Python %s: %s", - python, - (signed.stderr or signed.stdout or "codesign failed").strip(), - ) - return False - - verified = subprocess.run( - [codesign, "--verify", "--deep", "--strict", str(python)], - check=False, - capture_output=True, - text=True, - ) - if verified.returncode != 0: - logger.warning( + "codesign failed", + ), + ( + [codesign, "--verify", "--deep", "--strict", str(python)], "macOS signature verification failed for managed Python %s: %s", - python, - (verified.stderr or verified.stdout or "verification failed").strip(), - ) - return False + "verification failed", + ), + ) + for cmd, warning, fallback in steps: + result = subprocess.run(cmd, check=False, capture_output=True, text=True) + if result.returncode != 0: + logger.warning( + warning, python, (result.stderr or result.stdout or fallback).strip() + ) + return False return True except Exception as exc: logger.warning("could not sign managed Python %s: %s", python, exc) @@ -230,27 +197,8 @@ def _report_runtime_repair_failure(repair: RuntimeRepairResult) -> None: class _UvResult(str): """``ensure_uv()`` return value that survives an update boundary. - ``ensure_uv()``'s arity has flipped between a single path string and a - ``(path, fresh_bootstrap)`` tuple across releases. ``hermes update`` runs - the call site from the *old*, already-imported ``hermes_cli.main`` against - this *freshly pulled* module, so the two can disagree on how many values - ``ensure_uv()`` returns. An install parked on a 2-tuple release runs - ``uv_bin, fresh_bootstrap = ensure_uv()`` against the single-value module - and crashes the first update: the returned path is a plain ``str``, which is - itself iterable, so the 2-target unpack walks its characters and raises - ``ValueError: too many values to unpack (expected 2)`` (and on the failure - path the ``None`` return raises ``TypeError: cannot unpack non-iterable - NoneType``). This wrapper answers to both conventions: - - uv_bin = ensure_uv() # behaves as the path str ("" when absent) - uv_bin, fresh = ensure_uv() # unpacks as (path|None, fresh_bootstrap) - - Missing uv is the empty string (falsy) instead of ``None`` so legacy - 2-target call sites can still unpack a failure without raising, while - ``if not uv_bin`` keeps working for single-value callers. - - POSIX only. This wrapper is **never** returned on Windows — see - ``ensure_uv()`` for why the ``__iter__`` override is unsafe there. + POSIX only. This wrapper is **never** returned on Windows — see ``ensure_uv()`` for why the + ``__iter__`` override is unsafe there. """ fresh_bootstrap: bool @@ -291,56 +239,56 @@ def _ensure_uv_path( # Verify result = resolve_uv() if result: - version = subprocess.run( - [result, "--version"], - capture_output=True, - text=True, encoding='utf-8', errors='replace', - check=False, - ).stdout.strip() - print(f" ✓ Managed uv installed ({version})") + print(f" ✓ Managed uv installed ({_uv_version(result)})") # Compatibility boundary: an older, already-imported updater calls the # freshly pulled ``ensure_uv()`` after bootstrapping uv. Repair here so # that first update can migrate a vulnerable runtime without requiring # a second ``hermes update``. - try: - repair = repair_vulnerable_runtime(result) - if repair_observer is not None: - repair_observer(repair) - if repair.status == "failed": - _report_runtime_repair_failure(repair) - except Exception as exc: - logger.warning("Managed Python runtime repair failed: %s", exc) + _run_runtime_repair(result, repair_observer) else: print(" ✗ Managed uv install appeared to succeed but binary not found") return result +def _uv_version(uv_bin: str) -> str: + return subprocess.run( + [uv_bin, "--version"], + capture_output=True, text=True, encoding='utf-8', errors='replace', check=False, + ).stdout.strip() + + +def _run_runtime_repair( + uv_bin: str, + repair_observer: Callable[[RuntimeRepairResult], None] | None, + *, + print_skip: bool = False, +) -> None: + """Run the vulnerable-runtime repair hook; never raises (repair is non-fatal).""" + try: + repair = repair_vulnerable_runtime(uv_bin) + if repair_observer is not None: + repair_observer(repair) + if repair.status == "failed": + _report_runtime_repair_failure(repair) + except Exception as exc: + logger.warning("Managed Python runtime repair failed: %s", exc) + if print_skip: + print(f" ⚠ Managed Python runtime repair skipped: {exc}") + + def ensure_uv( *, repair_observer: Callable[[RuntimeRepairResult], None] | None = None, ): """Return the managed uv path, installing it first if necessary. - On **POSIX** the result is a :class:`_UvResult` (a ``str`` subclass) that is - both usable directly as the path *and* unpackable as - ``(path, fresh_bootstrap)`` for older call sites parked on a 2-tuple - release — see :class:`_UvResult` for the update-boundary rationale. + On **POSIX** the result is a :class:`_UvResult` (a ``str`` subclass) that is both usable + directly as the path *and* unpackable as ``(path, fresh_bootstrap)`` for older call sites parked + on a 2-tuple release — see :class:`_UvResult` for the update-boundary rationale. - On **Windows** we deliberately return a plain ``str``/``None`` instead. - ``subprocess`` there serializes the argv via ``subprocess.list2cmdline``, - which iterates every entry *as a string* (``for c in arg``). The dependency - installer passes uv straight into the command list (``[uv_bin, "pip", ...]``), - so a ``_UvResult`` — whose ``__iter__`` yields ``(path, fresh_bootstrap)`` - rather than characters — would inject the bool into the command line and - crash the install with ``TypeError: sequence item 1: expected str instance, - bool found``. A plain ``str`` matches the historical Windows contract and is - subprocess-safe. (A single value cannot satisfy both 2-target unpacking and - Windows char-iteration: both use the iterator protocol, with contradictory - results.) - - On failure the result is falsy — never raises — so callers can fall back to - pip gracefully. ``repair_observer``, when provided, receives the runtime - repair result produced after a fresh uv bootstrap. + On failure the result is falsy — never raises — so callers can fall back to pip gracefully. + ``repair_observer``, when provided, receives the runtime repair result produced after a fresh uv + bootstrap. """ result = _ensure_uv_path(repair_observer=repair_observer) if platform.system() == "Windows": @@ -350,19 +298,21 @@ def ensure_uv( return _UvResult(result) +def _uv_self_update_stamp() -> Path: + from hermes_constants import get_hermes_home + + return get_hermes_home() / "cache" / ".uv_self_update_stamp" + + def _uv_self_update_is_fresh(now: float | None = None) -> bool: """Return True when ``uv self update`` ran recently enough to skip. - uv releases roughly weekly while many users run ``hermes update`` daily; - re-running a blocking network self-update on every invocation is waste - and, offline, an unbounded hang risk. A stamp file under HERMES_HOME - caches the last successful self-update time. + uv releases roughly weekly while many users run ``hermes update`` daily; a blocking network + self-update on every run is waste and, offline, an unbounded hang risk. A stamp file under + HERMES_HOME caches the last successful self-update time. """ try: - from hermes_constants import get_hermes_home - - stamp = get_hermes_home() / "cache" / ".uv_self_update_stamp" - age = (now if now is not None else time.time()) - stamp.stat().st_mtime + age = (now if now is not None else time.time()) - _uv_self_update_stamp().stat().st_mtime return 0 <= age < UV_SELF_UPDATE_INTERVAL_SECONDS except Exception: return False @@ -370,9 +320,7 @@ def _uv_self_update_is_fresh(now: float | None = None) -> bool: def _touch_uv_self_update_stamp() -> None: try: - from hermes_constants import get_hermes_home - - stamp = get_hermes_home() / "cache" / ".uv_self_update_stamp" + stamp = _uv_self_update_stamp() stamp.parent.mkdir(parents=True, exist_ok=True) stamp.touch() except OSError: @@ -393,15 +341,14 @@ def update_managed_uv( ) -> Optional[str]: """Run ``uv self update`` on the managed uv binary. - Call this during ``hermes update`` so the managed copy stays current. - Returns the managed path when uv is available and ``None`` otherwise. - A self-update failure is non-fatal because the old version still works. - ``repair_observer``, when provided, receives the runtime repair result. + Call this during ``hermes update`` so the managed copy stays current. Returns the managed path + when uv is available and ``None`` otherwise. ``repair_observer``, when provided, receives the + runtime repair result. The network self-update is skipped when it succeeded within the last - ``UV_SELF_UPDATE_INTERVAL_SECONDS`` (7 days) unless ``force=True``; the - vulnerable-runtime repair probe below ALWAYS runs — CVE-driven runtime - repair must never be gated behind the freshness stamp. + ``UV_SELF_UPDATE_INTERVAL_SECONDS`` (7 days) unless ``force=True``; the vulnerable-runtime + repair probe below ALWAYS runs — CVE-driven runtime repair must never be gated behind the + freshness stamp. """ existing = resolve_uv() if not existing: @@ -422,13 +369,7 @@ def update_managed_uv( result = None if result is not None and result.returncode == 0: _touch_uv_self_update_stamp() - version = subprocess.run( - [existing, "--version"], - capture_output=True, - text=True, encoding='utf-8', errors='replace', - check=False, - ).stdout.strip() - print(f" ✓ Managed uv updated ({version})") + print(f" ✓ Managed uv updated ({_uv_version(existing)})") elif result is not None: # Non-fatal — old uv still works fine. logger.debug( @@ -438,18 +379,10 @@ def update_managed_uv( # Keep this hook inside the long-standing API. During an update, main.py is # already imported from the old checkout, then ``git pull`` replaces this # module on disk before the updater imports it. Calling the repair here is - # what makes the migration happen on that first update. - try: - repair = repair_vulnerable_runtime(existing) - if repair_observer is not None: - repair_observer(repair) - if repair.status == "failed": - _report_runtime_repair_failure(repair) - except Exception as exc: - # Runtime refresh is deliberately non-fatal. The live venv was not - # touched unless a fully prepared candidate reached cutover. - logger.warning("Managed Python runtime repair failed: %s", exc) - print(f" ⚠ Managed Python runtime repair skipped: {exc}") + # what makes the migration happen on that first update. Runtime refresh is + # deliberately non-fatal: the live venv was not touched unless a fully + # prepared candidate reached cutover. + _run_runtime_repair(existing, repair_observer, print_skip=True) return existing @@ -461,19 +394,8 @@ def update_managed_uv( def _reload_hermes_constants(): """Re-execute ``hermes_constants`` from disk and return the fresh module. - ``hermes update`` imports ``hermes_constants`` from the OLD checkout, - ``git pull`` then replaces that file, and this freshly-pulled module runs - its lazy imports against the module object Python already cached in - ``sys.modules`` — the pre-upgrade one. A symbol added by the update is - absent there while the file named in the resulting ``ImportError`` plainly - contains it, which is what made this read as a contradiction: - - cannot import name 'venv_python_path' from 'hermes_constants' - (~/.hermes/hermes-agent/hermes_constants.py) - - Reloading picks up the definitions actually on disk, so callers keep using - the shared helper instead of hand-rolling a second copy of its logic. Same - update-boundary class as the ``ensure_uv()`` arity skew on :class:`_UvResult`. + cannot import name 'venv_python_path' from 'hermes_constants' (~/.hermes/hermes- + agent/hermes_constants.py) """ import hermes_constants @@ -481,12 +403,11 @@ def _reload_hermes_constants(): def _venv_python(venv_dir: Path) -> Path: - windows = platform.system() == "Windows" try: from hermes_constants import venv_python_path except ImportError: venv_python_path = _reload_hermes_constants().venv_python_path - return venv_python_path(venv_dir, windows=windows) + return venv_python_path(venv_dir, windows=platform.system() == "Windows") def _remove_tree(path: Path, *, boundary: Path) -> None: @@ -509,13 +430,9 @@ def _make_world_traversable(path: Path) -> None: def _runtime_request(info: SQLiteRuntimeInfo) -> str: """Pin the candidate to the current CPython minor line (e.g. ``3.11``). - Requesting the exact patch can never repair some installs: for a given - patch, python-build-standalone may have no artifact with fixed SQLite at - all (e.g. every published 3.11.14 build links SQLite 3.50.4; the fix - only exists from 3.11.15). A newer patch on the same minor is what - ``uv python install`` would resolve for a fresh install, stays inside - ``requires-python``, and the locked ``uv sync`` + import smoke tests gate - compatibility before any cutover. + Requesting the exact patch can never repair some installs: for a given patch, python-build- + standalone may have no artifact with fixed SQLite at all (e.g. every published 3.11.14 build + links SQLite 3.50.4; the fix only exists from 3.11.15). """ return ".".join(str(part) for part in info.python_version[:2]) @@ -531,19 +448,13 @@ def _list_available_patches( ) -> list[tuple[int, int, int]]: """Return known patch versions for ``minor`` (e.g. "3.11"), newest first. - Queries ``uv python list --all-versions`` rather than trusting the bare - minor-line request to resolve to the newest patch (issue #71250: on some - hosts/uv versions, the resolved candidate for a bare "3.11" request can - be an older cached/indexed patch that still links a vulnerable SQLite, - even when a newer non-vulnerable patch is available). Returns [] on any - failure (network, parse) -- callers fall back to the original bare-minor + Returns [] on any failure (network, parse) -- callers fall back to the original bare-minor request in that case, preserving prior behavior. """ try: result = subprocess.run( [ - uv_bin, "python", "list", minor, - "--all-versions", "--only-downloads", + uv_bin, "python", "list", minor, "--all-versions", "--only-downloads", "--output-format", "json", "--no-config", ], cwd=cwd, @@ -591,84 +502,47 @@ def _attempt_install_generation( allow_minor_upgrade: bool = False, tried_versions: set[tuple[int, int, int]] | None = None, ) -> tuple[Path, Path, SQLiteRuntimeInfo] | None: - """One install+probe attempt for a specific version request (bare minor - like "3.11", or an explicit patch like "3.11.15"). Each attempt gets its - own generation directory so a rejected candidate's files are fully - cleaned up before the next attempt, matching --reinstall semantics. - Returns None (and cleans up) on any failure, including a vulnerable - or off-line candidate. - - When *tried_versions* is given, the probed candidate's version is - recorded in it so callers looping over explicit patches can skip a - version a bare-minor request already resolved to (and rejected) -- - retrying it explicitly would spend a full download+install+probe+delete - cycle to reach a certain rejection. + """One install+probe attempt for a specific version request (bare minor like "3.11", or an explicit + patch like "3.11.15"). Each attempt gets its own generation directory so a rejected candidate's + files are fully cleaned up before the next attempt, matching --reinstall semantics. Returns None + (and cleans up) on any failure, including a vulnerable or off-line candidate. """ token = f"{int(time.time())}-{os.getpid()}-{uuid.uuid4().hex[:8]}" generation = python_root / f"generation-{token}" generation.mkdir(parents=True, exist_ok=False) _make_world_traversable(generation) - env = managed_python_env(project_root, install_dir=generation) - install = subprocess.run( - [ - uv_bin, - "python", - "install", - request, - "--reinstall", - "--no-bin", - "--no-registry", - "--no-config", - ], - cwd=project_root, - env=env, - capture_output=True, - text=True, - check=False, - ) - if install.returncode != 0: - logger.warning( - "private Python install failed for %s (rc=%d): %s", - request, - install.returncode, - (install.stderr or install.stdout or "").strip(), - ) + def reject(msg: str, *args) -> None: + logger.warning(msg, *args) _remove_tree(generation, boundary=python_root) return None + env = managed_python_env(project_root, install_dir=generation) + run = dict(cwd=project_root, env=env, capture_output=True, text=True, check=False) + install = subprocess.run( + [uv_bin, "python", "install", request, "--reinstall", "--no-bin", "--no-registry", "--no-config"], + **run, + ) + if install.returncode != 0: + return reject( + "private Python install failed for %s (rc=%d): %s", + request, install.returncode, (install.stderr or install.stdout or "").strip(), + ) + found = subprocess.run( - [ - uv_bin, - "python", - "find", - request, - "--managed-python", - "--no-config", - ], - cwd=project_root, - env=env, - capture_output=True, - text=True, - check=False, + [uv_bin, "python", "find", request, "--managed-python", "--no-config"], **run ) if found.returncode != 0 or not found.stdout.strip(): - logger.warning( + return reject( "private Python lookup failed for %s (rc=%d): %s", - request, - found.returncode, - (found.stderr or "").strip(), + request, found.returncode, (found.stderr or "").strip(), ) - _remove_tree(generation, boundary=python_root) - return None python = Path(found.stdout.strip().splitlines()[-1]) try: python.resolve().relative_to(generation.resolve()) except (OSError, ValueError): - logger.warning("uv resolved Python outside the Hermes generation: %s", python) - _remove_tree(generation, boundary=python_root) - return None + return reject("uv resolved Python outside the Hermes generation: %s", python) # Do this before the candidate is probed or promoted. On macOS, the # stable identifier prevents each immutable generation from looking like @@ -678,43 +552,76 @@ def _attempt_install_generation( candidate = probe_sqlite_runtime(python) if candidate is None: - logger.warning("could not probe candidate Python runtime: %s", python) - _remove_tree(generation, boundary=python_root) - return None + return reject("could not probe candidate Python runtime: %s", python) if tried_versions is not None: tried_versions.add(candidate.python_version[:3]) if allow_minor_upgrade: # When falling forward to a higher minor line (e.g. 3.11 → 3.12), # only reject downgrades — allow the minor to differ. if candidate.python_version < current.python_version: - logger.warning( + return reject( "candidate Python downgraded from %s: %s", - ".".join(str(p) for p in current.python_version), - candidate.python_version, + ".".join(str(p) for p in current.python_version), candidate.python_version, ) - _remove_tree(generation, boundary=python_root) - return None elif candidate.python_version[:2] != current.python_version[:2] or ( candidate.python_version < current.python_version ): - logger.warning( + return reject( "candidate Python drifted off the %s minor line or downgraded: %s", - ".".join(str(p) for p in current.python_version[:2]), - candidate.python_version, + ".".join(str(p) for p in current.python_version[:2]), candidate.python_version, ) - _remove_tree(generation, boundary=python_root) - return None if candidate.wal_reset_vulnerable: - logger.warning( + return reject( "candidate Python still links vulnerable SQLite %s (%s)", - candidate.sqlite_version_string, - candidate.sqlite_source_id, + candidate.sqlite_version_string, candidate.sqlite_source_id, ) - _remove_tree(generation, boundary=python_root) - return None return generation, python, candidate +def _retry_explicit_patches( + uv_bin: str, + request: str, + *, + project_root: Path, + python_root: Path, + current: SQLiteRuntimeInfo, + tried: set[tuple[int, int, int]], + allow_minor_upgrade: bool = False, + skip_at_or_below: tuple[int, int, int] | None = None, +) -> tuple[Path, Path, SQLiteRuntimeInfo] | None: + """Retry ``request``'s minor line with explicit patch versions, newest-first, at most + ``_MAX_PATCH_RETRIES`` attempts, skipping versions already in ``tried`` (retrying one explicitly + would spend a full download+install+probe+delete cycle to reach a certain rejection). + + ``skip_at_or_below`` also skips patches at or below that version: only NEWER patches can carry + the SQLite fix, and the downgrade guard rejects the rest anyway. This matters on a uv whose + download catalog is stale: in #71250 the newest indexed 3.11 was 3.11.14, exactly the installed + version, so without this skip the loop burned all five retries walking backwards. + """ + env_for_list = managed_python_env(project_root, install_dir=python_root) + patches = _list_available_patches(uv_bin, request, cwd=project_root, env=env_for_list) + attempts = 0 + for version_tuple in patches: + if attempts >= _MAX_PATCH_RETRIES: + break + if version_tuple in tried: + continue + if skip_at_or_below is not None and version_tuple <= skip_at_or_below: + continue + tried.add(version_tuple) + explicit = ".".join(str(p) for p in version_tuple) + print(f" → Retrying with explicit patch {explicit}...") + attempts += 1 + result = _attempt_install_generation( + uv_bin, explicit, project_root=project_root, + python_root=python_root, current=current, + allow_minor_upgrade=allow_minor_upgrade, + ) + if result is not None: + return result + return None + + def _install_safe_python_generation( uv_bin: str, *, @@ -725,15 +632,12 @@ def _install_safe_python_generation( python_root = managed_python_install_dir(project_root) _make_world_traversable(runtime_root) _make_world_traversable(python_root) + common = dict(project_root=project_root, python_root=python_root, current=current) request = _runtime_request(current) print(f" → Provisioning a private Python {request} runtime with fixed SQLite...") tried_versions = {current.python_version[:3]} - result = _attempt_install_generation( - uv_bin, request, project_root=project_root, - python_root=python_root, current=current, - tried_versions=tried_versions, - ) + result = _attempt_install_generation(uv_bin, request, tried_versions=tried_versions, **common) if result is not None: return result @@ -744,37 +648,12 @@ def _install_safe_python_generation( # case where the default resolution for a bare request picks an older # cached/indexed patch even though a newer, non-vulnerable one is # available (issue #71250). - env_for_list = managed_python_env(project_root, install_dir=python_root) - patches = _list_available_patches( - uv_bin, request, cwd=project_root, env=env_for_list + result = _retry_explicit_patches( + uv_bin, request, tried=tried_versions, + skip_at_or_below=current.python_version[:3], **common, ) - attempts = 0 - for version_tuple in patches: - if attempts >= _MAX_PATCH_RETRIES: - break - if version_tuple in tried_versions: - continue - # Only NEWER patches can carry the SQLite fix. A patch at or below the - # installed one is either the version we already know is vulnerable or - # an older build that cannot contain a later fix, and the downgrade - # guard in _attempt_install_generation rejects it anyway -- so trying - # it spends a full download+install+probe+delete cycle to reach a - # certain rejection. This matters on a uv whose download catalog is - # stale: in #71250 the newest indexed 3.11 was 3.11.14, exactly the - # installed version, so without this skip the loop burned all five - # retries walking backwards (3.11.13 -> 3.11.9) before failing. - if version_tuple <= current.python_version[:3]: - continue - tried_versions.add(version_tuple) - explicit_request = ".".join(str(p) for p in version_tuple) - print(f" → Retrying with explicit patch {explicit_request}...") - attempts += 1 - result = _attempt_install_generation( - uv_bin, explicit_request, project_root=project_root, - python_root=python_root, current=current, - ) - if result is not None: - return result + if result is not None: + return result # All patches on the current minor line are vulnerable or rejected. # Fall forward to the next supported minor (e.g. 3.11 → 3.12) so the @@ -791,38 +670,15 @@ def _install_safe_python_generation( f"trying {next_request} as fallback..." ) result = _attempt_install_generation( - uv_bin, next_request, project_root=project_root, - python_root=python_root, current=current, - allow_minor_upgrade=True, - tried_versions=fb_tried, + uv_bin, next_request, allow_minor_upgrade=True, tried_versions=fb_tried, **common ) if result is not None: return result - # Also try explicit patches on this minor line, skipping whatever - # version the bare request above already resolved to (retrying it - # explicitly would spend a full download+install+probe+delete cycle - # to reach a certain rejection). - env_for_list = managed_python_env(project_root, install_dir=python_root) - fb_patches = _list_available_patches( - uv_bin, next_request, cwd=project_root, env=env_for_list + result = _retry_explicit_patches( + uv_bin, next_request, tried=fb_tried, allow_minor_upgrade=True, **common ) - fb_attempts = 0 - for version_tuple in fb_patches: - if fb_attempts >= _MAX_PATCH_RETRIES: - break - if version_tuple in fb_tried: - continue - fb_tried.add(version_tuple) - explicit = ".".join(str(p) for p in version_tuple) - print(f" → Retrying with explicit patch {explicit}...") - fb_attempts += 1 - result = _attempt_install_generation( - uv_bin, explicit, project_root=project_root, - python_root=python_root, current=current, - allow_minor_upgrade=True, - ) - if result is not None: - return result + if result is not None: + return result return None @@ -843,22 +699,11 @@ def _smoke_candidate_venv(venv_dir: Path) -> tuple[bool, str, SQLiteRuntimeInfo "import dotenv, fastapi, openai, prompt_toolkit, pydantic, rich, uvicorn, yaml\n" "import hermes_state\n" ) - env = dict(os.environ) - for key in ( - "CONDA_DEFAULT_ENV", - "CONDA_PREFIX", - "PYTHONHOME", - "PYTHONPATH", - "UV_PROJECT_ENVIRONMENT", - "UV_PYTHON", - "VIRTUAL_ENV", - ): - env.pop(key, None) try: result = subprocess.run( [str(python), "-I", "-c", check], cwd=venv_dir.parent, - env=env, + env=isolated_interpreter_env(), capture_output=True, text=True, timeout=90, @@ -883,10 +728,7 @@ def _stage_candidate_venv( runtime_root = project_root / _RUNTIME_DIR_NAME token = f"{int(time.time())}-{os.getpid()}-{uuid.uuid4().hex[:8]}" candidate = runtime_root / f"venv-candidate-{token}" - env = managed_python_env( - project_root, - install_dir=generation, - ) + env = managed_python_env(project_root, install_dir=generation) env.update({ "UV_PROJECT_ENVIRONMENT": str(candidate), "UV_PYTHON": str(python), @@ -894,18 +736,16 @@ def _stage_candidate_venv( "VIRTUAL_ENV": str(candidate), }) + def reject(msg: str, *args) -> None: + logger.warning(msg, *args) + _remove_tree(candidate, boundary=runtime_root) + return None + print(" → Building a relocatable replacement environment...") created = subprocess.run( [ - uv_bin, - "venv", - str(candidate), - "--python", - str(python), - "--managed-python", - "--no-python-downloads", - "--relocatable", - "--no-config", + uv_bin, "venv", str(candidate), "--python", str(python), + "--managed-python", "--no-python-downloads", "--relocatable", "--no-config", ], cwd=project_root, env=env, @@ -914,51 +754,33 @@ def _stage_candidate_venv( check=False, ) if created.returncode != 0: - logger.warning( + return reject( "candidate venv creation failed (rc=%d): %s", - created.returncode, - (created.stderr or created.stdout or "").strip(), + created.returncode, (created.stderr or created.stdout or "").strip(), ) - _remove_tree(candidate, boundary=runtime_root) - return None if not (project_root / "uv.lock").is_file(): - logger.warning("candidate dependency sync refused: uv.lock is missing") - _remove_tree(candidate, boundary=runtime_root) - return None + return reject("candidate dependency sync refused: uv.lock is missing") # Locked sync must see project [tool.uv] exclude-newer; --no-config / # UV_NO_CONFIG drops it and uv 0.12+ refuses --locked. sync_env = dict(env) sync_env.pop("UV_NO_CONFIG", None) synced = subprocess.run( - [ - uv_bin, - "sync", - "--extra", - "all", - "--locked", - "--python", - str(_venv_python(candidate)), - ], + [uv_bin, "sync", "--extra", "all", "--locked", "--python", str(_venv_python(candidate))], cwd=project_root, env=sync_env, check=False, ) if synced.returncode != 0: - logger.warning("candidate dependency sync failed (rc=%d)", synced.returncode) - _remove_tree(candidate, boundary=runtime_root) - return None + return reject("candidate dependency sync failed (rc=%d)", synced.returncode) healthy, detail, _ = _smoke_candidate_venv(candidate) if not healthy: - logger.warning("candidate venv smoke failed: %s", detail) - _remove_tree(candidate, boundary=runtime_root) - return None + return reject("candidate venv smoke failed: %s", detail) return candidate def _rename_with_retry(source: Path, destination: Path) -> None: - last_error: OSError | None = None for delay in (0.0, 0.1, 0.25, 0.5, 1.0): if delay: time.sleep(delay) @@ -967,8 +789,7 @@ def _rename_with_retry(source: Path, destination: Path) -> None: return except OSError as exc: last_error = exc - if last_error is not None: - raise last_error + raise last_error def _cut_over_candidate( @@ -995,19 +816,11 @@ def _cut_over_candidate( try: _rename_with_retry(backup, live) except OSError as rollback_error: - return ( - False, - backup, - None, + return False, backup, None, ( "could not promote the replacement venv " - f"({promote_error}); rollback failed ({rollback_error})", + f"({promote_error}); rollback failed ({rollback_error})" ) - return ( - False, - None, - None, - f"could not promote the replacement venv: {promote_error}", - ) + return False, None, None, f"could not promote the replacement venv: {promote_error}" try: healthy, detail, info = _smoke_candidate_venv(live) @@ -1020,12 +833,9 @@ def _cut_over_candidate( _rename_with_retry(live, rejected) _rename_with_retry(backup, live) except OSError as exc: - return ( - False, - backup, - info, + return False, backup, info, ( "post-cutover smoke failed " - f"({detail}); rollback failed ({exc}); rejected venv: {rejected}", + f"({detail}); rollback failed ({exc}); rejected venv: {rejected}" ) _remove_tree(rejected, boundary=runtime_root) return False, None, info, f"post-cutover smoke failed: {detail}" @@ -1054,34 +864,31 @@ def _acquire_repair_lock(runtime_root: Path) -> _RepairLock | None: return None try: - if os.name == "nt": - import msvcrt - - if os.fstat(fd).st_size == 0: - os.write(fd, b"\0") - os.lseek(fd, 0, os.SEEK_SET) - msvcrt.locking(fd, msvcrt.LK_NBLCK, 1) - else: - import fcntl - - fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + _flock(fd, acquire=True) except (ImportError, OSError): os.close(fd) return None return _RepairLock(path=path, fd=fd) +def _flock(fd: int, *, acquire: bool) -> None: + """Non-blocking exclusive lock (or unlock) on *fd*, portable across msvcrt/fcntl.""" + if os.name == "nt": + import msvcrt + + if acquire and os.fstat(fd).st_size == 0: + os.write(fd, b"\0") + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_NBLCK if acquire else msvcrt.LK_UNLCK, 1) + else: + import fcntl + + fcntl.flock(fd, (fcntl.LOCK_EX | fcntl.LOCK_NB) if acquire else fcntl.LOCK_UN) + + def _release_repair_lock(lock: _RepairLock) -> None: try: - if os.name == "nt": - import msvcrt - - os.lseek(lock.fd, 0, os.SEEK_SET) - msvcrt.locking(lock.fd, msvcrt.LK_UNLCK, 1) - else: - import fcntl - - fcntl.flock(lock.fd, fcntl.LOCK_UN) + _flock(lock.fd, acquire=False) except (ImportError, OSError): pass finally: @@ -1111,27 +918,18 @@ def _windows_runtime_holders() -> tuple[bool, str]: def _windows_runtime_self_lock(live: Path) -> tuple[bool, str]: """Detect the one holder the generic scan above is blind to: THIS process. - ``_detect_venv_python_processes`` excludes the calling process and its - ancestors on purpose — a CLI ``hermes update`` itself runs from the - venv python — which is correct for the dependency-sync path, where only - a *loaded* ``.pyd`` image blocks the rewrite and a fresh child process - dodges it. For the whole-venv park rename that exemption is fatal: - Windows keeps the image of any executable a running process was started - from mapped until that process exits, so a directory containing the - updater's own ``python.exe`` (or a waiting ``hermes.exe`` launcher - ancestor) can never be renamed from inside the updater. The retry loop - in ``_cut_over_candidate`` cannot help against that — the lock is - structural, not transient (#93032). - - No-op off Windows: POSIX renames work fine while this process maps - files from the renamed tree (open FDs and mmaps keep inodes alive). + ``_detect_venv_python_processes`` excludes the calling process and its ancestors on purpose — a + CLI ``hermes update`` itself runs from the venv python — which is correct for the dependency- + sync path, where only a *loaded* ``.pyd`` image blocks the rewrite and a fresh child process + dodges it. """ if platform.system() != "Windows": return False, "" try: - live_res = str(live.resolve()).lower().rstrip(os.sep) + os.sep + live_res = str(live.resolve()) except OSError: - live_res = str(live).lower().rstrip(os.sep) + os.sep + live_res = str(live) + live_res = live_res.lower().rstrip(os.sep) + os.sep def _under_live(path_value: str | None) -> bool: if not path_value: @@ -1142,10 +940,7 @@ def _windows_runtime_self_lock(live: Path) -> tuple[bool, str]: resolved = str(path_value).lower() return resolved.startswith(live_res) - try: - exe = sys.executable - except Exception: - exe = None + exe = sys.executable if _under_live(exe): return True, ( f"the updater itself runs from the live venv it must replace " @@ -1158,11 +953,7 @@ def _windows_runtime_self_lock(live: Path) -> tuple[bool, str]: try: import psutil - try: - parents = psutil.Process().parents() - except Exception: - parents = [] - for anc in parents: + for anc in psutil.Process().parents(): try: anc_exe = anc.exe() except Exception: @@ -1183,46 +974,27 @@ def _uv_version_string(uv_bin: str) -> str: try: result = subprocess.run( [uv_bin, "--version"], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=15, + capture_output=True, text=True, encoding="utf-8", errors="replace", + check=False, timeout=15, ) except Exception: return "" - if result.returncode != 0: - return "" - return (result.stdout or "").strip() + return (result.stdout or "").strip() if result.returncode == 0 else "" def _refresh_managed_uv_catalog(uv_bin: str) -> bool: """Re-bootstrap the managed uv binary to refresh its Python catalog. - The managed uv is installed with ``UV_UNMANAGED_INSTALL``, which disables - ``uv self update`` by design — so its embedded python-build-standalone - download catalog stays frozen at bootstrap age. python-build-standalone - re-releases existing CPython patch versions with newer SQLite (e.g. the - 3.11.15 build was re-cut with SQLite 3.53.x), so a stale catalog can make - every provisioning attempt resolve to a vulnerable build even though a - fixed build of the SAME patch version exists (issue #72093). The - patch-retry loop cannot recover from that: the fixed build carries no - newer version number to retry with. - - Re-running the official installer is the only supported refresh path for - unmanaged installs. Only the Hermes-managed binary is ever refreshed; - a caller-supplied foreign uv path is left alone. - - Returns ``True`` when the binary's version actually changed — i.e. a - provisioning retry can now see a different catalog. ``False`` means a - retry would resolve identically and is not worth the download cycle. + Re-running the official installer is the only supported refresh path for unmanaged installs. + Only the Hermes-managed binary is ever refreshed; a caller-supplied foreign uv path is left + alone. """ managed = managed_uv_path() try: - if Path(uv_bin).resolve() != managed.resolve(): - return False + is_managed = Path(uv_bin).resolve() == managed.resolve() except OSError: + is_managed = False + if not is_managed: return False before = _uv_version_string(uv_bin) try: @@ -1231,32 +1003,18 @@ def _refresh_managed_uv_catalog(uv_bin: str) -> bool: logger.warning("managed uv refresh failed: %s", exc) return False after = _uv_version_string(uv_bin) - if not after: - return False - return after != before + return bool(after) and after != before def _default_live_venv(root: Path) -> Path: """Return the venv that runtime repair should target for *root*. - Managed installs create ``/venv``, but uv-default and dev - checkouts use ``/.venv``. Historically only ``venv`` was - probed, so a ``.venv`` install linking a vulnerable SQLite returned - ``not-applicable`` on every ``hermes update`` and stayed on - journal_mode=DELETE forever — even though the WAL fallback warning - promises that ``hermes update`` repairs the runtime (issue class: - 2,600x slower ``state.db`` appends under DELETE). - - ``venv`` wins when it holds an interpreter (managed layout takes - precedence); otherwise fall back to ``.venv`` when that one does. - When neither has an interpreter, return the ``venv`` path so the - caller's existing ``not-applicable`` handling fires unchanged. + ``venv`` wins when it holds an interpreter (managed layout takes precedence); otherwise fall + back to ``.venv`` when that one does. When neither has an interpreter, return the ``venv`` path + so the caller's existing ``not-applicable`` handling fires unchanged. """ - primary = root / _VENV_NAME - if _venv_python(primary).is_file(): - return primary - fallback = root / _ALT_VENV_NAME - if _venv_python(fallback).is_file(): + primary, fallback = root / _VENV_NAME, root / _ALT_VENV_NAME + if not _venv_python(primary).is_file() and _venv_python(fallback).is_file(): return fallback return primary @@ -1270,18 +1028,12 @@ def _sweep_stale_runtime_backups( ) -> None: """Remove leftover ``venv.stale.runtime-*`` backups next to *live*. - A successful runtime repair parks the previous venv as - ``.stale.runtime-``; historically nothing ever reclaimed - those, so each repair leaked a full venv (~1 GB) at the project root - forever (issue #73109). On POSIX, deleting the tree is safe even while - an older process still maps files from it — open FDs and mmaps keep - their inodes alive; the directory entry is what goes away. + On POSIX, deleting the tree is safe even while an older process still maps files from it — open + FDs and mmaps keep their inodes alive; the directory entry is what goes away. - ``min_age_seconds`` guards against racing a concurrent repair in - another process: a backup parked seconds ago may still be that - repair's rollback path, so only clearly-old markers are swept. - ``keep`` exempts the backup the current repair just created. - Best-effort: never raises. + ``min_age_seconds`` guards against racing a concurrent repair in another process: a backup + parked seconds ago may still be that repair's rollback path, so only clearly-old markers are + swept. ``keep`` exempts the backup the current repair just created. Best-effort: never raises. """ try: candidates = list(live.parent.glob(f"{live.name}.stale.runtime-*")) @@ -1308,8 +1060,8 @@ def repair_vulnerable_runtime( ) -> RuntimeRepairResult: """Replace a vulnerable install venv without mutating it in place. - Every failure before cutover leaves the live venv untouched. Rename or - post-cutover smoke failures restore the parked venv synchronously. + Every failure before cutover leaves the live venv untouched. Rename or post-cutover smoke + failures restore the parked venv synchronously. """ root = Path(project_root) if project_root is not None else _PROJECT_ROOT live = Path(venv_dir) if venv_dir is not None else _default_live_venv(root) @@ -1337,14 +1089,15 @@ def repair_vulnerable_runtime( sqlite_after=current.sqlite_version_string, ) + def deferred(detail: str) -> RuntimeRepairResult: + return RuntimeRepairResult( + "skipped", detail, sqlite_before=current.sqlite_version_string + ) + blocked, detail = _windows_runtime_holders() if blocked: print(f" ⚠ SQLite runtime repair deferred: {detail}") - return RuntimeRepairResult( - "skipped", - detail, - sqlite_before=current.sqlite_version_string, - ) + return deferred(detail) self_locked, self_detail = _windows_runtime_self_lock(live) if self_locked: @@ -1368,25 +1121,15 @@ def repair_vulnerable_runtime( " Sessions stay protected meanwhile: Hermes keeps databases " "out of WAL mode on this SQLite build." ) - return RuntimeRepairResult( - "skipped", - self_detail, - sqlite_before=current.sqlite_version_string, - ) + return deferred(self_detail) runtime_root = root / _RUNTIME_DIR_NAME lock = _acquire_repair_lock(runtime_root) if lock is None: detail = "another runtime repair is already in progress" print(f" ⚠ SQLite runtime repair deferred: {detail}") - return RuntimeRepairResult( - "skipped", - detail, - sqlite_before=current.sqlite_version_string, - ) + return deferred(detail) - generation: Path | None = None - candidate: Path | None = None try: # Re-probe under the install-scoped lock: another updater may have # completed the repair while this process was entering the path. @@ -1409,19 +1152,18 @@ def repair_vulnerable_runtime( project_root=root, current=current, ) - if provisioned is None: - # Likely a stale managed-uv catalog: python-build-standalone - # re-releases the same patch versions with fixed SQLite, but a - # frozen catalog keeps resolving the old vulnerable build and the - # patch-retry loop has no newer number to try (issue #72093). - # Refresh the managed binary and retry once. - if _refresh_managed_uv_catalog(uv_bin): - print(" → Managed uv refreshed; retrying provisioning...") - provisioned = _install_safe_python_generation( - uv_bin, - project_root=root, - current=current, - ) + # Likely a stale managed-uv catalog: python-build-standalone + # re-releases the same patch versions with fixed SQLite, but a + # frozen catalog keeps resolving the old vulnerable build and the + # patch-retry loop has no newer number to try (issue #72093). + # Refresh the managed binary and retry once. + if provisioned is None and _refresh_managed_uv_catalog(uv_bin): + print(" → Managed uv refreshed; retrying provisioning...") + provisioned = _install_safe_python_generation( + uv_bin, + project_root=root, + current=current, + ) if provisioned is None: return RuntimeRepairResult( "failed", @@ -1458,9 +1200,7 @@ def repair_vulnerable_runtime( "failed", cutover_detail, sqlite_before=current.sqlite_version_string, - sqlite_after=( - final_info.sqlite_version_string if final_info is not None else "" - ), + sqlite_after=final_info.sqlite_version_string if final_info is not None else "", backup_venv=backup, ) @@ -1493,9 +1233,8 @@ def repair_vulnerable_runtime( def _install_uv(target: Path) -> None: """Bootstrap uv into *target* using the official standalone installer. - Uses ``UV_UNMANAGED_INSTALL`` (POSIX) or ``UV_INSTALL_DIR`` (Windows) - so the astral installer writes the binary directly into - ``$HERMES_HOME/bin/`` instead of ``~/.local/bin/``. + Sets ``UV_UNMANAGED_INSTALL`` (POSIX) / ``UV_INSTALL_DIR`` (Windows) so the installer writes + into ``$HERMES_HOME/bin/`` instead of ``~/.local/bin/``. """ system = platform.system() env = { @@ -1507,10 +1246,7 @@ def _install_uv(target: Path) -> None: "UV_INSTALL_DIR": str(target.parent), } - if system == "Windows": - _install_uv_windows(env) - else: - _install_uv_posix(env) + (_install_uv_windows if system == "Windows" else _install_uv_posix)(env) def _install_uv_posix(env: dict[str, str]) -> None: diff --git a/hermes_cli/migrate.py b/hermes_cli/migrate.py index 0c947f632e..81afa057c5 100644 --- a/hermes_cli/migrate.py +++ b/hermes_cli/migrate.py @@ -1,7 +1,7 @@ """CLI handlers for ``hermes migrate ...``. -Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply) -rewrites references to xAI models retired on May 15, 2026. +Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply) rewrites references to +xAI models retired on May 15, 2026. """ from __future__ import annotations diff --git a/hermes_cli/npm_engine.py b/hermes_cli/npm_engine.py index b26afe6a7e..ac04266b1e 100644 --- a/hermes_cli/npm_engine.py +++ b/hermes_cli/npm_engine.py @@ -1,27 +1,12 @@ """Recover from npm ``EBADENGINE`` failures by upgrading a managed npm. -The repo's ``.npmrc`` sets ``engine-strict=true`` and the root ``package.json`` -pins an ``engines.npm`` range, so an npm outside that range aborts every -``npm ci`` / ``npm install`` we run inside the checkout:: +Rather than predicting the failure (which would mean a semver range matcher and an ``npm --version`` +probe before work that usually succeeds), we react to it: npm states the required range in the +error, so the recovery reads the constraint straight out of the output it just produced. - npm error code EBADENGINE - npm error notsup Required: {"node":">=26.0.0","npm":">=12.0.0"} - npm error notsup Actual: {"npm":"10.9.8","node":"v22.23.1"} - -Rather than predicting the failure (which would mean a semver range matcher and -an ``npm --version`` probe before work that usually succeeds), we react to it: -npm states the required range in the error, so the recovery reads the -constraint straight out of the output it just produced. - -Scope of the repair is deliberately narrow. Hermes only upgrades an npm that -lives inside its **own** managed Node tree (``$HERMES_HOME/node``), installing -in place with ``--prefix`` so ``bin/npm`` keeps resolving to the upgraded -``lib/node_modules/npm``. A system / nvm / brew / Nix npm belongs to the user -and their other projects; Hermes never modifies those. When the failing npm is -one of those foreign installs, Hermes instead provisions its own managed Node -tree (the same tree a fresh install creates), upgrades *that* npm into range, -and hands the caller the managed npm to retry with — leaving the user's -toolchain untouched. +Scope of the repair is deliberately narrow. Hermes only upgrades an npm that lives inside its +**own** managed Node tree (``$HERMES_HOME/node``), installing in place with ``--prefix`` so +``bin/npm`` keeps resolving to the upgraded ``lib/node_modules/npm``. """ from __future__ import annotations @@ -61,14 +46,12 @@ _UPGRADE_TIMEOUT = 300 def is_ebadengine(output: str) -> bool: """Return True when *output* is an npm engine-compatibility failure.""" - if not output: - return False - return "EBADENGINE" in output or "Unsupported engine" in output + return bool(output) and ("EBADENGINE" in output or "Unsupported engine" in output) -def _iter_required_blocks(output: str) -> list[dict]: +def _iter_json_blocks(pattern: re.Pattern[str], output: str) -> list[dict]: blocks: list[dict] = [] - for match in _REQUIRED_RE.finditer(output or ""): + for match in pattern.finditer(output or ""): try: parsed = json.loads(match.group(1)) except ValueError: @@ -78,17 +61,19 @@ def _iter_required_blocks(output: str) -> list[dict]: return blocks +def _iter_required_blocks(output: str) -> list[dict]: + return _iter_json_blocks(_REQUIRED_RE, output) + + def required_npm_range(output: str) -> str | None: """Return the ``engines.npm`` range npm demanded in *output*. - Returns ``None`` when the output has no engine failure, or when the - failure is about Node rather than npm — upgrading npm cannot fix a Node - version mismatch, so the caller must not try. + Returns ``None`` when the output has no engine failure, or when the failure is about Node rather + than npm — upgrading npm cannot fix a Node version mismatch, so the caller must not try. - When several packages report conflicting npm ranges the repo's own root - constraint is preferred (it is the one we control); otherwise the first - range wins, since any of them is a strict improvement over an npm that - satisfies none. + When several packages report conflicting npm ranges the repo's own root constraint is preferred + (it is the one we control); otherwise the first range wins, since any of them is a strict + improvement over an npm that satisfies none. """ if not is_ebadengine(output): return None @@ -109,12 +94,8 @@ def required_npm_range(output: str) -> str | None: def actual_npm_version(output: str) -> str | None: """Return the npm version npm reported as ``Actual`` in *output*.""" - for match in _ACTUAL_RE.finditer(output or ""): - try: - parsed = json.loads(match.group(1)) - except ValueError: - continue - if isinstance(parsed, dict) and parsed.get("npm"): + for parsed in _iter_json_blocks(_ACTUAL_RE, output): + if parsed.get("npm"): return str(parsed["npm"]).strip() return None @@ -137,10 +118,9 @@ def managed_npm_prefix(npm: str | os.PathLike[str] | None) -> Path | None: """Return the Hermes-managed Node root *npm* lives in, else ``None``. Symlinks are resolved first: an install links ``~/.local/bin/npm`` at - ``$HERMES_HOME/node/bin/npm``, which itself links into - ``lib/node_modules/npm/bin/npm-cli.js``. Every one of those spellings is - the managed npm and must be recognised as such, or the repair silently - declines to fix the very install it owns. + ``$HERMES_HOME/node/bin/npm``, which itself links into ``lib/node_modules/npm/bin/npm-cli.js``. + Every one of those spellings is the managed npm and must be recognised as such, or the repair + silently declines to fix the very install it owns. """ if not npm: return None @@ -175,10 +155,10 @@ def upgrade_managed_npm( ) -> bool: """Upgrade the managed npm at *npm* in place to satisfy *npm_range*. - ``--prefix`` targets the managed tree explicitly: a managed install writes - ``prefix=~/.local`` into ``$HERMES_HOME/node/etc/npmrc`` so that global - installs land on PATH, and without the override the "upgrade" would install - a second npm somewhere else while the managed one stayed stale. + ``--prefix`` targets the managed tree explicitly: a managed install writes ``prefix=~/.local`` + into ``$HERMES_HOME/node/etc/npmrc`` so that global installs land on PATH, and without the + override the "upgrade" would install a second npm somewhere else while the managed one stayed + stale. """ if not quiet: print( @@ -274,13 +254,10 @@ def _print_manual_fix(npm: str, npm_range: str, actual: str | None) -> None: def _provision_managed_npm(npm_range: str | None, *, quiet: bool = False) -> str | None: """Provision a Hermes-managed Node tree and return a satisfying npm. - Installs the managed tree under ``$HERMES_HOME/node`` (reusing a healthy - one when present), then upgrades its bundled npm to *npm_range* — a fresh - Node LTS bundles an npm that may itself be outside the repo's range, so - without the upgrade the caller's single retry would fail the same way. - Falls back to the checkout's own ``engines.npm`` when npm did not state a - range (a Node-only mismatch), so the managed npm ends up in range either - way. Returns the managed npm path, or ``None`` when provisioning failed. + Installs (or reuses) the tree under ``$HERMES_HOME/node``, then upgrades its bundled npm to + *npm_range*: a fresh Node LTS may bundle an npm outside the repo's range, so without the + upgrade the caller's single retry would fail the same way. Falls back to the checkout's + ``engines.npm`` when no range was stated. Returns the npm path, or ``None`` on failure. """ if not quiet: print( @@ -314,18 +291,9 @@ def maybe_repair_npm_engine( ) -> str | None: """Repair an ``EBADENGINE`` failure, never touching a foreign toolchain. - *output* is the combined stdout/stderr of the npm command that just failed. - Returns the npm executable the caller should retry its command with — - the same *npm* after an in-place upgrade of a Hermes-managed install, or - a freshly provisioned managed npm when the failing npm belongs to the - user (system / nvm / brew / Nix installs are never modified). Returns - ``None`` when no repair happened — not an engine failure, a Node mismatch - a managed npm upgrade cannot fix, or a failed upgrade/bootstrap — leaving - the original failure to stand. - - The returned value is truthy exactly when the caller should retry once, - so ``if maybe_repair_npm_engine(...)`` call sites keep working; they just - must run the retry with the returned path. + The returned value is truthy exactly when the caller should retry once, so ``if + maybe_repair_npm_engine(...)`` call sites keep working; they just must run the retry with the + returned path. """ if not npm or not is_ebadengine(output): return None @@ -336,9 +304,7 @@ def maybe_repair_npm_engine( if prefix is not None: # Hermes owns this npm — upgrade it in place. Only an npm-range # failure is fixable this way; a Node mismatch needs a Node upgrade. - if not npm_range: - return None - if upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet): + if npm_range and upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet): return npm return None diff --git a/hermes_cli/relaunch.py b/hermes_cli/relaunch.py index a5a8431fbe..31963f4fac 100644 --- a/hermes_cli/relaunch.py +++ b/hermes_cli/relaunch.py @@ -1,11 +1,8 @@ -""" -Unified self-relaunch for Hermes CLI. +"""Unified self-relaunch for Hermes CLI. -Preserves critical flags (--tui, --dev, --profile, --model, etc.) across -process replacement so that ``hermes sessions browse`` or post-setup relaunch -doesn't silently drop the user's UI mode or other preferences. - -Also works when ``hermes`` is not on PATH (e.g. ``nix run`` or ``python -m``). +Preserves critical flags (--tui, --dev, --profile, --model, etc.) across process replacement so that +``hermes sessions browse`` or post-setup relaunch doesn't silently drop the user's UI mode or other +preferences. """ import os @@ -20,12 +17,10 @@ from hermes_cli._parser import ( def _build_inherited_flag_table() -> list[tuple[str, bool]]: - """Build the ``(option_string, takes_value)`` table of flags that must - survive a self-relaunch, by introspecting the real parser used by - ``hermes`` itself. + """Build the ``(option_string, takes_value)`` table of flags that must survive a self-relaunch. - A flag participates if its argparse Action carries - ``inherit_on_relaunch = True`` — set by ``_parser._inherited_flag``. + Introspects the real ``hermes`` parser: a flag participates iff its argparse Action carries + ``inherit_on_relaunch = True`` (set by ``_parser._inherited_flag``). """ parser, _subparsers, chat_parser = build_top_level_parser() @@ -80,20 +75,8 @@ def _extract_inherited_flags(argv: Sequence[str]) -> list[str]: def resolve_hermes_bin() -> Optional[str]: """Find the hermes entry point. - Priority: - 1. ``sys.argv[0]`` if it resolves to a real executable. - 2. ``shutil.which("hermes")`` on PATH. - 3. ``None`` → caller should fall back to ``python -m hermes_cli.main``. - - Windows note: ``os.access(path, os.X_OK)`` returns True for ``.py`` and - ``.pyc`` files on Windows (the OS treats anything listed in PATHEXT as - executable, and Python files are often registered there). But - ``subprocess.run([script.py, ...])`` can't actually execute a .py - directly — CreateProcessW needs a real .exe, not a script associated - with the Python launcher. On Windows we therefore skip the argv[0] - fast-path when it points at a .py file and fall through to either - ``hermes.exe`` on PATH or the ``sys.executable -m hermes_cli.main`` - fallback. + Priority: 1. ``sys.argv[0]`` if it resolves to a real executable. 2. ``shutil.which("hermes")`` + on PATH. 3. ``None`` → caller should fall back to ``python -m hermes_cli.main``. """ argv0 = sys.argv[0] _is_windows = sys.platform == "win32" @@ -127,15 +110,7 @@ def build_relaunch_argv( preserve_inherited: bool = True, original_argv: Optional[Sequence[str]] = None, ) -> list[str]: - """Construct an argv list for replacing the current process with hermes. - - Args: - extra_args: Arguments to append (e.g. ``["--resume", id]``). - preserve_inherited: Whether to carry over UI / behaviour flags - tagged with ``inherit_on_relaunch`` in the parser. - original_argv: The original argv to scan for flags (defaults to - ``sys.argv[1:]``). - """ + """Construct an argv list for replacing the current process with hermes.""" bin_path = resolve_hermes_bin() if bin_path: @@ -160,23 +135,12 @@ def relaunch( ) -> None: """Replace the current process with a fresh hermes invocation. - On POSIX we use ``os.execvp`` which replaces the running process with - the new one in place — same PID, no double-fork. That's what the - relaunch contract wants: "run hermes again as if the user had typed - the new argv". + On POSIX we use ``os.execvp`` which replaces the running process with the new one in place — + same PID, no double-fork. That's what the relaunch contract wants: "run hermes again as if the + user had typed the new argv". - Windows has no native exec semantics — ``os.execvp`` on Windows - *emulates* exec by spawning the child and exiting the parent, but - only works when the target is a real Win32 executable. Our target - is usually ``hermes.exe`` (a Python console-script shim that wraps - ``python -m hermes_cli.main``) or a ``.cmd`` batch file, and both - raise ``OSError(8, "Exec format error")`` on Windows' execvp. - - The Windows-correct pattern is: spawn the child with ``subprocess.run`` - (which routes through ``cmd.exe`` via ``shell=False`` + PATHEXT resolution), - wait for it to exit, then propagate its exit code via ``sys.exit``. - That's functionally equivalent — the user sees "hermes exited, then - new hermes started" — just with two PIDs in play instead of one. + Windows has no native exec semantics — ``os.execvp`` on Windows *emulates* exec by spawning the + child and exiting the parent, but only works when the target is a real Win32 executable. """ new_argv = build_relaunch_argv( extra_args, preserve_inherited=preserve_inherited, original_argv=original_argv diff --git a/hermes_cli/service_manager.py b/hermes_cli/service_manager.py index a7194d99c6..ab0c8ba5c1 100644 --- a/hermes_cli/service_manager.py +++ b/hermes_cli/service_manager.py @@ -1,22 +1,13 @@ -"""Abstract service manager interface. - -Wraps the existing systemd (Linux host), launchd (macOS host), Windows -Scheduled Task (native Windows host), and s6 (container) backends behind -a common Protocol. Only the s6 backend supports runtime registration -(for per-profile gateways) — host backends raise NotImplementedError -from those methods, and callers MUST check supports_runtime_registration() -before invoking them. - -Host-side call sites (setup wizard, uninstall, status) continue to use -the existing module-level functions in hermes_cli.gateway and -hermes_cli.gateway_windows directly. This protocol is a thin facade -used by new code that needs to be backend-agnostic — specifically the -profile create/delete hooks (Phase 4) and the s6 dispatch path in -``hermes gateway start/stop/restart`` when running inside a container. -""" +"""Abstract service manager interface.""" from __future__ import annotations +import json +import os import re +import shlex +import shutil +import subprocess +import time from pathlib import Path from typing import Literal, Protocol, runtime_checkable @@ -33,10 +24,9 @@ _MAX_PROFILE_LEN = 251 # s6-svscan default name_max def validate_profile_name(name: str) -> None: """Raise ValueError if ``name`` is not usable as a profile name. - Profile names are used as s6 service directory names, so they must - match a conservative subset of filesystem-safe characters. Reject - empty strings, uppercase, paths-traversal sequences, and anything - longer than s6's default ``name_max``. + Profile names are used as s6 service directory names, so they must match a conservative subset + of filesystem-safe characters. Reject empty strings, uppercase, paths-traversal sequences, and + anything longer than s6's default ``name_max``. """ if not name: raise ValueError("profile name must not be empty") @@ -54,12 +44,9 @@ def validate_profile_name(name: str) -> None: class ServiceManager(Protocol): """Abstract interface for init-system-specific service operations. - Lifecycle methods (start / stop / restart / is_running) are - implemented by every backend. Runtime registration - (register_profile_gateway / unregister_profile_gateway / - list_profile_gateways) is implemented only by the s6 backend — - callers MUST check ``supports_runtime_registration()`` before - invoking the registration methods. + Lifecycle methods (start/stop/restart/is_running) exist on every backend. Runtime registration + (register/unregister/list_profile_gateways) is s6-only — callers MUST check + ``supports_runtime_registration()`` before invoking it. """ kind: ServiceManagerKind @@ -86,18 +73,9 @@ class ServiceManager(Protocol): def detect_service_manager() -> ServiceManagerKind: """Detect which service manager is available in this environment. - Returns: - "s6" — s6-svscan is PID 1 (s6-overlay image; Docker, Podman, or a - Fly Firecracker microVM) - "windows" — native Windows host - "launchd" — macOS host - "systemd" — Linux host with a working user/system bus - "none" — anything else (Termux, sandbox shells, etc.) - - This function does NOT replace ``supports_systemd_services()`` — - host call sites continue to use that. It exists for new backend- - agnostic code (profile create/delete hooks, the s6 dispatch path - in ``hermes gateway start/stop/restart``). + Returns "s6" (s6-svscan is PID 1), "windows", "launchd", "systemd" (working bus) or "none" + (Termux, sandboxes). Does NOT replace ``supports_systemd_services()`` for host call sites; it + exists for backend-agnostic code (profile hooks, the s6 dispatch in ``hermes gateway``). """ # Imports deferred so importing this module doesn't drag in the # whole gateway dependency graph for callers that only need the @@ -129,26 +107,10 @@ def detect_service_manager() -> ServiceManagerKind: def _s6_running() -> bool: """True when s6-svscan is running as PID 1 in this container. - Detection has to work for **both** root and the unprivileged hermes - user (UID 10000). The obvious probe — ``Path('/proc/1/exe').resolve()`` - — only works as root: for any other UID, the symlink at - ``/proc/1/exe`` is unreadable and ``resolve()`` silently returns the - path unchanged, so the resolved name is the literal ``"exe"`` and - detection always fails. Since every Hermes runtime call inside the - container drops to hermes via ``s6-setuidgid``, that silent failure - made the entire service-manager runtime-registration path inert in - production (PR #30136 review). - - Probe instead via: - * ``/proc/1/comm`` — world-readable, contains the process comm - (``s6-svscan`` when s6-overlay is PID 1). - * ``/run/s6/basedir`` — s6-overlay-specific directory created by - stage1. World-readable. More specific than ``/run/s6`` (which - other tools occasionally create). - - Both signals are required; either alone could false-positive - (e.g. a container with the s6 binaries installed but a different - init, or an unrelated process named ``s6-svscan``). + Must work for root AND the unprivileged hermes user: ``/proc/1/exe`` is unreadable for other + UIDs and ``resolve()`` silently yields the literal ``exe``, which made runtime registration + inert in production. Probe the world-readable ``/proc/1/comm`` and ``/run/s6/basedir`` instead; + both are required since either alone can false-positive. """ try: comm = Path("/proc/1/comm").read_text(encoding="utf-8").strip() @@ -172,8 +134,36 @@ def _s6_running() -> bool: # --------------------------------------------------------------------------- -class _RegistrationUnsupportedMixin: - """Mixin for host backends that don't support runtime registration.""" +class _HostServiceManager: + """Table-driven host backend: ``start``/``stop``/``restart`` resolve to + ``<_fn_prefix>`` on the ``hermes_cli.<_backend>`` submodule at call time + (lazy import; tests monkeypatch the submodule or its functions). Runtime + registration is unsupported on every host backend. + """ + + kind: ServiceManagerKind + _backend: str + _fn_prefix: str = "" + + def _backend_module(self): + import importlib + + import hermes_cli + + importlib.import_module(f"hermes_cli.{self._backend}") + return getattr(hermes_cli, self._backend) + + def _call(self, op: str) -> None: + getattr(self._backend_module(), f"{self._fn_prefix}{op}")() + + def start(self, name: str) -> None: + self._call("start") + + def stop(self, name: str) -> None: + self._call("stop") + + def restart(self, name: str) -> None: + self._call("restart") def supports_runtime_registration(self) -> bool: return False @@ -200,69 +190,44 @@ class _RegistrationUnsupportedMixin: return [] -class SystemdServiceManager(_RegistrationUnsupportedMixin): +class SystemdServiceManager(_HostServiceManager): """Thin wrapper around the ``systemd_*`` functions in hermes_cli.gateway. - Existing host call sites continue to use those functions directly; - this wrapper exists for new code that needs to be backend-agnostic - (the Phase 4 profile create/delete hooks). + Existing host call sites keep using those functions directly; this wrapper exists for + backend-agnostic code such as the profile create/delete hooks. """ kind: ServiceManagerKind = "systemd" - - def start(self, name: str) -> None: - from hermes_cli.gateway import systemd_start - systemd_start() - - def stop(self, name: str) -> None: - from hermes_cli.gateway import systemd_stop - systemd_stop() - - def restart(self, name: str) -> None: - from hermes_cli.gateway import systemd_restart - systemd_restart() + _backend = "gateway" + _fn_prefix = "systemd_" def is_running(self, name: str) -> bool: - from hermes_cli.gateway import _probe_systemd_service_running - _, running = _probe_systemd_service_running() + _, running = self._backend_module()._probe_systemd_service_running() return running -class LaunchdServiceManager(_RegistrationUnsupportedMixin): +class LaunchdServiceManager(_HostServiceManager): """Thin wrapper around the ``launchd_*`` functions in hermes_cli.gateway.""" kind: ServiceManagerKind = "launchd" - - def start(self, name: str) -> None: - from hermes_cli.gateway import launchd_start - launchd_start() - - def stop(self, name: str) -> None: - from hermes_cli.gateway import launchd_stop - launchd_stop() - - def restart(self, name: str) -> None: - from hermes_cli.gateway import launchd_restart - launchd_restart() + _backend = "gateway" + _fn_prefix = "launchd_" def is_running(self, name: str) -> bool: - from hermes_cli.gateway import _probe_launchd_service_running - return _probe_launchd_service_running() + return self._backend_module()._probe_launchd_service_running() -class WindowsServiceManager(_RegistrationUnsupportedMixin): - """Thin wrapper around ``hermes_cli.gateway_windows`` (Scheduled Task / - Startup-folder fallback). +class WindowsServiceManager(_HostServiceManager): + """Thin wrapper around ``hermes_cli.gateway_windows`` (Scheduled Task / Startup-folder + fallback). - The native Windows backend uses a Scheduled Task rather than a true - init-system service, but for protocol purposes the lifecycle is the - same: start / stop / restart / is_running. ``install`` accepts a - handful of Windows-specific kwargs (start_now, start_on_login, - elevated_handoff) that are passed straight through — non-Windows - callers should never invoke ``install`` on this wrapper. + A Scheduled Task is not a true init service, but the lifecycle protocol is the same. ``install`` + takes Windows-specific kwargs (start_now, start_on_login, elevated_handoff) passed straight + through — non-Windows callers must never call ``install`` here. """ kind: ServiceManagerKind = "windows" + _backend = "gateway_windows" def install( self, @@ -272,50 +237,26 @@ class WindowsServiceManager(_RegistrationUnsupportedMixin): start_on_login: bool | None = None, elevated_handoff: bool = False, ) -> None: - from hermes_cli import gateway_windows - gateway_windows.install( + self._backend_module().install( force=force, start_now=start_now, start_on_login=start_on_login, elevated_handoff=elevated_handoff, ) - def start(self, name: str) -> None: - from hermes_cli import gateway_windows - gateway_windows.start() - - def stop(self, name: str) -> None: - from hermes_cli import gateway_windows - gateway_windows.stop() - - def restart(self, name: str) -> None: - from hermes_cli import gateway_windows - gateway_windows.restart() - def is_running(self, name: str) -> bool: - from hermes_cli import gateway_windows from hermes_cli.gateway import find_gateway_pids - if not gateway_windows.is_installed(): + if not self._backend_module().is_installed(): return False return bool(find_gateway_pids()) def get_service_manager() -> ServiceManager: - """Return the ServiceManager instance for the current environment. - - Raises: - RuntimeError: when no supported backend is available. - """ - kind = detect_service_manager() - if kind == "systemd": - return SystemdServiceManager() - if kind == "launchd": - return LaunchdServiceManager() - if kind == "windows": - return WindowsServiceManager() - if kind == "s6": - return S6ServiceManager() - raise RuntimeError("no supported service manager detected") + """Return the ServiceManager instance for the current environment.""" + cls = _MANAGER_CLASSES.get(detect_service_manager()) + if cls is None: + raise RuntimeError("no supported service manager detected") + return cls() # --------------------------------------------------------------------------- @@ -335,41 +276,34 @@ S6_DYNAMIC_SCANDIR = Path("/run/service") S6_SERVICE_PREFIX = "gateway-" +def _profile_from_service(name: str) -> str: + """Strip the ``gateway-`` prefix back off (matches what the user typed via ``-p``).""" + return name[len(S6_SERVICE_PREFIX):] if name.startswith(S6_SERVICE_PREFIX) else name + + def _profile_dir_for_gateway_service(name: str) -> Path: """Resolve ``gateway-`` to its persistent profile directory. - s6 lifecycle commands may be invoked from any active profile, including - ``gateway stop --all``. Do not write the caller's HERMES_HOME blindly; - derive the shared profile root from the current HERMES_HOME and map the - service suffix to either the root default profile or + s6 lifecycle commands may be invoked from any active profile, including ``gateway stop --all``. + Do not write the caller's HERMES_HOME blindly; derive the shared profile root from the current + HERMES_HOME and map the service suffix to either the root default profile or ``/profiles/``. """ - import os - - profile = name[len(S6_SERVICE_PREFIX):] if name.startswith(S6_SERVICE_PREFIX) else name + profile = _profile_from_service(name) validate_profile_name(profile) hermes_home = Path(os.environ.get("HERMES_HOME", "/opt/data")) - if hermes_home.parent.name == "profiles": - root = hermes_home.parent.parent - else: - root = hermes_home + root = hermes_home.parent.parent if hermes_home.parent.name == "profiles" else hermes_home return root if profile == "default" else root / "profiles" / profile def _write_gateway_desired_state(name: str, desired_state: str) -> None: """Persist durable s6 gateway intent next to runtime status. - ``gateway_state`` remains the volatile runtime field written by the - gateway process. ``desired_state`` records the operator's start/stop - intent so container-boot reconciliation can restore the correct s6 - want-up/want-down state after pod recreation even if the previous runtime - state was transient (draining, startup_failed, etc.). The write is - best-effort: a failed persistence attempt must not prevent immediate s6 - lifecycle control. + ``gateway_state`` stays the volatile field written by the gateway; ``desired_state`` records the + operator's start/stop intent so boot reconciliation can restore s6 want-up/down after pod + recreation even if the last runtime state was transient. Best-effort: a failed write must not + block immediate s6 lifecycle control. """ - import json - import time - profile_dir = _profile_dir_for_gateway_service(name) state_file = profile_dir / "gateway_state.json" try: @@ -405,6 +339,15 @@ def _write_gateway_desired_state(name: str, desired_state: str) -> None: _S6_BIN_DIR = "/command" +def _s6_run(cmd: str, *args: str, timeout: float = 5, check: bool = False): + """Run an s6 binary by absolute path with the shared capture/decode settings.""" + return subprocess.run( + [f"{_S6_BIN_DIR}/{cmd}", *args], + check=check, capture_output=True, text=True, encoding='utf-8', errors='replace', + timeout=timeout, + ) + + # UID/GID of the in-image ``hermes`` user. Hardcoded to match what # ``stage2-hook.sh`` enforces (the runtime invariant — see also # tests/docker/test_uid_remap.py). The container starts s6-supervise @@ -413,116 +356,59 @@ _HERMES_UID = 10000 _HERMES_GID = 10000 +def _chown_hermes(path: Path) -> None: + try: + os.chown(path, _HERMES_UID, _HERMES_GID) + except PermissionError: + # Running as the hermes user already — directory is hermes- + # owned by default. The chown is a no-op in that case, so + # swallowing this keeps both root and unprivileged callers + # on one code path. + pass + + def _seed_supervise_skeleton(svc_dir: Path) -> None: - """Pre-create the ``supervise/`` and top-level ``event/`` skeleton - inside a service directory, owned by the hermes user. + """Pre-create the ``supervise/`` and top-level ``event/`` skeleton inside a service directory, + owned by the hermes user. - Why this exists - --------------- - When s6-supervise spawns a service it tries to ``mkdir`` two - directories: ``/event`` and ``/supervise``, both with mode - ``0700``. It also ``mkfifo``s ``/supervise/control`` with mode - ``0600``. Because s6-supervise runs as PID 1's effective UID (root) - these dirs end up root-owned mode 0700, and an unprivileged client - (the ``hermes`` user — UID 10000 — running every Hermes runtime - operation via ``s6-setuidgid``) gets ``EACCES`` on any ``s6-svc``, - ``s6-svstat``, or ``s6-svwait`` invocation against the slot. - - The PR #30136 review surfaced this as a real product gap: the - entire S6ServiceManager lifecycle (``register/start/stop/unregister - _profile_gateway``) was inert in production because every operation - is dispatched as the hermes user. - - Why this works - -------------- - Reading s6's source (src/supervision/s6-supervise.c::trymkdir + - control_init): the ``mkdir`` and ``mkfifo`` calls both treat - ``EEXIST`` as success. If the directory is already present, the - chown/chmod fix-up that would normally make event/ ``03730 - root:root`` is **skipped** entirely — s6-supervise just opens the - pre-existing FIFOs and proceeds. So if we lay the skeleton down - with hermes ownership before triggering ``s6-svscanctl -a``, - s6-supervise inherits our layout and never touches it. - - Layout produced - --------------- - ``svc_dir/`` hermes:hermes, 0755 (parent must already exist) - ``svc_dir/event/`` hermes:hermes, 03730 (setgid + g+rwx + sticky) - ``svc_dir/supervise/`` hermes:hermes, 0755 - ``svc_dir/supervise/event/`` hermes:hermes, 03730 - ``svc_dir/supervise/control`` hermes:hermes, 0660 (FIFO) - - The ``death_tally``, ``lock``, and ``status`` regular files end up - written by s6-supervise itself (as root), but those land mode 0644 — - world-readable — and ``s6-svstat`` only needs read access, so the - hermes user reads them fine. - - If ``svc_dir/log/`` is present (the canonical s6 logger pattern — - one s6-supervise instance per service, plus a second for its - logger), the same skeleton is seeded under ``log/`` as well: - ``log/event/``, ``log/supervise/``, ``log/supervise/event/``, - ``log/supervise/control``. Without this, unregister teardown - would EACCES on the logger's supervise dir even after the parent - slot's supervise/ was hermes-owned. - - Idempotency - ----------- - Safe to call against a directory where the skeleton already exists. - Existing entries are left untouched (the helper doesn't try to - re-chown / re-chmod live FIFOs that s6-supervise may have already - opened). - - Reference - --------- - Discussed at length on the skarnet `skaware` mailing list in 2020 - (``_); see also - just-containers/s6-overlay#130. The pre-creation pattern was - historically called out as forward-compatibility-fragile, but the - EEXIST handling in s6-supervise has been stable since 2015 — it's - the same pattern ``s6-svperms`` and ``fix-attrs.d`` rely on. + s6-supervise runs as root and creates ``event/``/``supervise/`` mode 0700 plus the control FIFO + 0600, so the unprivileged hermes user gets EACCES on every ``s6-svc``/``s6-svstat``. s6 treats + EEXIST as success and skips its chown/chmod fix-up, so laying the skeleton down hermes-owned + before ``s6-svscanctl -a`` makes s6-supervise inherit it. The same skeleton is seeded under + ``log/`` when present, else unregister teardown EACCESes on the logger. Idempotent: existing + entries (possibly live FIFOs) are left untouched. """ - import os def _mkdir_owned(path: Path, mode: int) -> None: if path.exists(): return path.mkdir(parents=False, exist_ok=False) path.chmod(mode) - try: - os.chown(path, _HERMES_UID, _HERMES_GID) - except PermissionError: - # Running as the hermes user already — directory is hermes- - # owned by default. The chown is a no-op in that case, so - # swallowing this keeps both root and unprivileged callers - # on one code path. - pass + _chown_hermes(path) - # Top-level event/ dir (this is the s6-svlisten1 event-subscription - # dir at the service root, distinct from supervise/event/). - _mkdir_owned(svc_dir / "event", 0o3730) - - # supervise/ dir + its inner event/ dir. - supervise = svc_dir / "supervise" - _mkdir_owned(supervise, 0o755) - _mkdir_owned(supervise / "event", 0o3730) - - # supervise/control FIFO. Same EEXIST-safe pattern: if it's already - # there (s6-supervise has already started against this slot), leave - # it alone. The explicit chmod after mkfifo is required because - # mkfifo honors the process umask, which can strip group-write - # (e.g. the default 0022 on most dev hosts → 0o660 becomes 0o640). - # The container runs with umask 0 inside s6-overlay's stage2, but - # being defensive here keeps the helper consistent under any - # invocation context. - control = supervise / "control" - if not control.exists(): - os.mkfifo(control, 0o660) - control.chmod(0o660) - try: - os.chown(control, _HERMES_UID, _HERMES_GID) - except PermissionError: - pass + def _seed(root: Path) -> None: + # Top-level event/ dir (this is the s6-svlisten1 event-subscription + # dir at the service root, distinct from supervise/event/). + _mkdir_owned(root / "event", 0o3730) + # supervise/ dir + its inner event/ dir. + supervise = root / "supervise" + _mkdir_owned(supervise, 0o755) + _mkdir_owned(supervise / "event", 0o3730) + # supervise/control FIFO. Same EEXIST-safe pattern: if it's already + # there (s6-supervise has already started against this slot), leave + # it alone. The explicit chmod after mkfifo is required because + # mkfifo honors the process umask, which can strip group-write + # (e.g. the default 0022 on most dev hosts → 0o660 becomes 0o640). + # The container runs with umask 0 inside s6-overlay's stage2, but + # being defensive here keeps the helper consistent under any + # invocation context. + control = supervise / "control" + if not control.exists(): + os.mkfifo(control, 0o660) + control.chmod(0o660) + _chown_hermes(control) + _seed(svc_dir) # If a log/ subdir is present (the canonical s6 logger pattern — # see servicedir(7)), it gets its own s6-supervise instance and # needs the same skeleton. Without this, unregister teardown @@ -530,26 +416,14 @@ def _seed_supervise_skeleton(svc_dir: Path) -> None: # when the parent slot's supervise/ is hermes-owned. log_dir = svc_dir / "log" if log_dir.is_dir(): - _mkdir_owned(log_dir / "event", 0o3730) - log_supervise = log_dir / "supervise" - _mkdir_owned(log_supervise, 0o755) - _mkdir_owned(log_supervise / "event", 0o3730) - log_control = log_supervise / "control" - if not log_control.exists(): - os.mkfifo(log_control, 0o660) - log_control.chmod(0o660) - try: - os.chown(log_control, _HERMES_UID, _HERMES_GID) - except PermissionError: - pass + _seed(log_dir) class S6Error(RuntimeError): """Base error for S6ServiceManager lifecycle failures. - Concrete subclasses carry the slot name (and, where useful, the - underlying subprocess output) so the CLI can render an actionable - message instead of leaking a raw ``CalledProcessError`` traceback. + Subclasses carry the slot name (and subprocess output where useful) so the CLI can render an + actionable message instead of a raw ``CalledProcessError`` traceback. """ def __init__(self, message: str, *, service: str | None = None) -> None: @@ -560,10 +434,8 @@ class S6Error(RuntimeError): class GatewayNotRegisteredError(S6Error): """Raised when a lifecycle method targets a slot that doesn't exist. - Most commonly: ``hermes -p typo gateway start`` when no profile - ``typo`` exists. Carries the unprefixed profile name (not the - full ``gateway-`` service-dir name) so callers can phrase - a user-facing message like "no such gateway 'typo'". + Carries the unprefixed profile name (not ``gateway-``) so callers can phrase "no such + gateway 'typo'". """ def __init__(self, profile: str) -> None: @@ -577,11 +449,9 @@ class GatewayNotRegisteredError(S6Error): class S6CommandError(S6Error): - """Raised when an s6 command fails for a reason other than a - missing slot — e.g. permission denied on the supervise control - FIFO, or s6-svc returning a non-zero exit for an unexpected - reason. Carries the stderr from the failing command so callers - can surface it. + """Raised when an s6 command fails for a reason other than a missing slot — e.g. permission denied + on the supervise control FIFO, or s6-svc returning a non-zero exit for an unexpected reason. + Carries the stderr from the failing command so callers can surface it. """ def __init__( @@ -601,9 +471,8 @@ class S6CommandError(S6Error): class S6ServiceManager: """Per-profile gateway supervision via s6-overlay. - Only handles runtime-registered services under - ``S6_DYNAMIC_SCANDIR``. Static services (main-hermes, dashboard) - are managed by s6-rc at image-build time and are out of scope. + Only handles runtime-registered services under ``S6_DYNAMIC_SCANDIR``. Static services (main- + hermes, dashboard) are managed by s6-rc at image-build time and are out of scope. """ kind: ServiceManagerKind = "s6" @@ -617,9 +486,6 @@ class S6ServiceManager: validate_profile_name(profile) return self.scandir / f"{S6_SERVICE_PREFIX}{profile}" - def _service_name(self, profile: str) -> str: - return f"{S6_SERVICE_PREFIX}{profile}" - @staticmethod def _render_run_script( profile: str, @@ -627,45 +493,14 @@ class S6ServiceManager: ) -> str: """Generate the run script for a profile-gateway s6 service. - The script: - 1. Sources HERMES_HOME (and any extra env) via with-contenv — - so e.g. ``-e HERMES_HOME=/data/hermes`` is honored at run - time, not Python-substituted at registration time (OQ8-C). - 2. Resets ``HOME`` to ``/opt/data`` before the privilege drop - so with-contenv's root HOME does not leak into the - unprivileged gateway process. - 3. Activates the bundled venv. - 4. Drops to the hermes user and exec's - ``hermes -p gateway run`` (or just ``hermes - gateway run`` for the default profile — see below). - - Special case: ``profile == "default"`` emits ``hermes gateway - run`` with **no** ``-p`` flag. This is the sentinel for "the - root HERMES_HOME profile" (the implicit profile that exists at - the top of $HERMES_HOME, not under profiles/). It must be - spelled this way because ``_profile_suffix()`` returns the - empty string for the root profile, and the dispatcher in - ``hermes_cli.gateway`` maps that empty string to the - ``gateway-default`` service slot. Passing ``-p default`` here - would instead look up ``$HERMES_HOME/profiles/default/`` — a - completely different (and almost always nonexistent) profile. - - Port selection: the gateway binds the port resolved by - ``gateway/config.py`` from the profile's own environment — - ``API_SERVER_PORT`` (or ``platforms.api_server.extra.port`` in - that profile's ``config.yaml``), defaulting to 8642. There is - no ``[gateway] port`` key and no Python-side allocator: because - each supervised profile gateway loads its own ``HERMES_HOME``, - two profiles that both leave the port unset will both try to - bind 8642 — give each profile a distinct ``API_SERVER_PORT`` in - its ``.env``. Previously this method took a ``port`` parameter - that was passed in but never substituted into the rendered - script (carried for "API parity" with a deterministic SHA-256 - allocator in ``hermes_cli.profiles._allocate_gateway_port``). - PR #30136 review item I5 retired both the allocator and the - parameter because they were dead code through the entire stack. + The script sources HERMES_HOME via with-contenv (honored at run time, not baked in at + registration), resets ``HOME`` before the privilege drop so root's HOME does not leak, + activates the venv, and drops to hermes. ``profile == "default"`` emits NO ``-p`` flag: it + is the sentinel for the root HERMES_HOME profile, and ``-p default`` would look up + ``profiles/default/`` instead. Port comes from the profile's own env (``API_SERVER_PORT``, + default 8642); two profiles that both leave it unset will collide, so give each a distinct + port. """ - import shlex lines = [ "#!/command/with-contenv sh", "# shellcheck shell=sh", @@ -717,15 +552,9 @@ class S6ServiceManager: def _render_finish_script() -> str: """Generate the finish script for a profile-gateway s6 service. - When the gateway exits with EX_CONFIG (78) — a fatal - configuration error such as a token collision or no messaging - platforms — we tell s6-supervise to stop restarting by exiting - 125 (permanent failure). A clean exit 0 is an intentional stop, - not a crash: restarting after it turns any normal gateway exit - into a reconnect loop (the ashriel-discord storm in #76435 — - 1,000+ connections and a provider token reset). Only non-zero, - non-78 exits (genuine crashes) let s6 restart normally. - See #51228, #76435. + Exit 78 (EX_CONFIG, fatal config error) makes the script exit 125 so s6 stops restarting. A + clean exit 0 is an intentional stop, not a crash — restarting after it turns any normal exit + into a reconnect storm. Only non-zero, non-78 exits let s6 restart normally. """ from gateway.restart import GATEWAY_FATAL_CONFIG_EXIT_CODE @@ -749,46 +578,11 @@ class S6ServiceManager: def _render_log_run(profile: str) -> str: """Generate the log/run script for a profile-gateway service. - OQ8-C: persist to ``${HERMES_HOME}/logs/gateways//``. - CRITICAL: the HERMES_HOME path is sourced from the runtime env - via with-contenv — NOT Python-substituted at registration time - — so a container started with ``-e HERMES_HOME=/data/hermes`` - gets its logs under /data/hermes/logs/..., not the build-time - default. + Output routing — the script is two action directives, applied per line, in order: - Output routing — the script is two action directives, applied - per line, in order: - - 1. ``1`` (forward to stdout) — propagates the line up the - s6-supervise pipeline to /init's stdout, which is the - container's stdout, which is ``docker logs``. Without - this, supervised stdout would be terminated inside - s6-log and never reach the container's log stream; - users would have to ``docker exec`` and ``tail`` the - file just to see startup banners. (Python's ``logging`` - module defaults to stderr, which s6-supervise leaves - unfiltered — so warnings/errors already reach docker - logs. This change is specifically about the rich-console - banner output and other plain stdout writes.) - 2. ``T `` — also write a timestamped copy to the - rotated log directory (``current`` + archived ``@*.s`` - files). This is what ``hermes logs`` reads and what - persists across container restarts via the volume mount. - - ``T`` is non-sticky: it only prefixes lines for the next - action directive. We deliberately put ``T`` between ``1`` - and the log dir (not before ``1``) so: - - * ``docker logs`` shows raw lines — Python's logging - formatter has its own timestamps, and ``docker logs - --timestamps`` adds a third layer when desired. No - double-stamping in the most common reading path. - * The persisted file gets s6-log's own ISO 8601 timestamp - so even output that lacked a Python-logger timestamp - (rich banners, third-party libs' raw prints) is - correlatable in ``current``. + ``T`` is non-sticky: it only prefixes lines for the next action directive. We deliberately + put ``T`` between ``1`` and the log dir (not before ``1``) so: """ - import shlex prof = shlex.quote(profile) return ( f"#!/command/with-contenv sh\n" @@ -819,41 +613,17 @@ class S6ServiceManager: def _run_svc(self, action_flag: str, action_label: str, name: str) -> None: """Shared lifecycle dispatch for start / stop / restart. - Translates the two failure modes operators care about into - named errors: - - * ``GatewayNotRegisteredError`` — the service directory at - ``//`` doesn't exist. ``s6-svc`` would - exit non-zero with a fairly opaque message; we pre-empt - it with a clear "no such gateway 'X'" tied to the profile - name (without the ``gateway-`` prefix). - * ``S6CommandError`` — anything else (EACCES on the - supervise control FIFO, timeout, etc.). Carries the - subprocess return code and stderr so callers can render - them inline. - - ``action_flag`` is the ``s6-svc`` flag (``-u`` / ``-d`` / - ``-t``); ``action_label`` is the human verb (``start`` / - ``stop`` / ``restart``) used in error messages. + Pre-empts a missing service dir with ``GatewayNotRegisteredError`` (clear "no such gateway" + instead of s6-svc's opaque failure) and wraps anything else (EACCES on the control FIFO, + timeout) in ``S6CommandError`` carrying return code and stderr. ``action_flag`` is the + ``s6-svc`` flag, ``action_label`` the human verb used in messages. """ - import subprocess - service_dir = self.scandir / name if not service_dir.is_dir(): - # Strip the gateway- prefix back off so the message - # matches what the user typed on the CLI (``-p ``). - profile = ( - name[len(S6_SERVICE_PREFIX):] - if name.startswith(S6_SERVICE_PREFIX) - else name - ) - raise GatewayNotRegisteredError(profile) + raise GatewayNotRegisteredError(_profile_from_service(name)) try: - subprocess.run( - [f"{_S6_BIN_DIR}/s6-svc", action_flag, str(service_dir)], - check=True, capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - ) + _s6_run("s6-svc", action_flag, str(service_dir), check=True) except subprocess.CalledProcessError as exc: raise S6CommandError( service=name, @@ -863,32 +633,18 @@ class S6ServiceManager: ) from exc def start(self, name: str) -> None: - """Bring up a registered service (``s6-svc -u``). - - Raises: - GatewayNotRegisteredError: no service directory for ``name``. - S6CommandError: s6-svc exited non-zero for any other reason - (permission denied on the supervise FIFO, timeout, etc.). - """ + """Bring up a registered service (``s6-svc -u``).""" self._run_svc("-u", "start", name) _write_gateway_desired_state(name, "running") def _supervised_pid(self, name: str) -> int | None: """Return the PID of the supervised gateway process, or None. - Parses ``s6-svstat`` output (``up (pid NNNN) ...``). Used to - mark an operator-initiated stop with the planned-stop marker so - the gateway's shutdown handler classifies the incoming SIGTERM - as intentional rather than an unexpected kill (issue #42675). - Best-effort: any parse/exec failure returns None. + Parses ``s6-svstat`` output. Used to write the planned-stop marker before an operator stop + so the gateway classifies the SIGTERM as intentional. Best-effort: any failure returns None. """ - import subprocess - try: - result = subprocess.run( - [f"{_S6_BIN_DIR}/s6-svstat", str(self.scandir / name)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - ) + result = _s6_run("s6-svstat", str(self.scandir / name)) except (OSError, subprocess.SubprocessError): return None if result.returncode != 0: @@ -899,20 +655,9 @@ class S6ServiceManager: def stop(self, name: str) -> None: """Bring down a registered service (``s6-svc -d``). - Writes a planned-stop marker naming the supervised gateway PID - BEFORE sending the down command, so the gateway's shutdown - handler recognises this SIGTERM as an operator-initiated stop - and persists ``gateway_state=stopped`` (respecting the explicit - intent). Without the marker, an intentional ``hermes gateway - stop`` is indistinguishable from the container/s6 SIGTERM sent on - ``docker restart``; the latter must NOT persist ``stopped`` or - container_boot refuses to auto-start on the next boot (#42675). - The marker write is best-effort — a failure only means the stop - is treated as signal-initiated, which is the safe fallback. - - Raises: - GatewayNotRegisteredError: no service directory for ``name``. - S6CommandError: s6-svc exited non-zero for any other reason. + Writes a planned-stop marker naming the supervised gateway PID BEFORE sending the down + command, so the gateway's shutdown handler recognises this SIGTERM as an operator-initiated + stop and persists ``gateway_state=stopped`` (respecting the explicit intent). """ pid = self._supervised_pid(name) if pid is not None: @@ -926,22 +671,13 @@ class S6ServiceManager: _write_gateway_desired_state(name, "stopped") def restart(self, name: str) -> None: - """Restart a registered service (``s6-svc -t`` = SIGTERM). - - Raises: - GatewayNotRegisteredError: no service directory for ``name``. - S6CommandError: s6-svc exited non-zero for any other reason. - """ + """Restart a registered service (``s6-svc -t`` = SIGTERM).""" self._run_svc("-t", "restart", name) _write_gateway_desired_state(name, "running") def is_running(self, name: str) -> bool: """True iff ``s6-svstat`` reports the service as up.""" - import subprocess - result = subprocess.run( - [f"{_S6_BIN_DIR}/s6-svstat", str(self.scandir / name)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - ) + result = _s6_run("s6-svstat", str(self.scandir / name)) return result.returncode == 0 and "up " in result.stdout # -- runtime registration --------------------------------------------- @@ -958,20 +694,11 @@ class S6ServiceManager: ) -> None: """Create the s6 service directory for a profile gateway. - Triggers ``s6-svscanctl -a`` so s6-svscan picks the new directory - up immediately. When *start_now* is ``True`` (the default) the - service starts immediately; when ``False`` a ``down`` marker file - is written so s6-supervise leaves the service stopped until the - user explicitly runs ``hermes -p gateway start``. - - Raises: - ValueError: if the profile name is invalid or the service - directory already exists. - RuntimeError: if ``s6-svscanctl`` fails. + Triggers ``s6-svscanctl -a`` so s6-svscan picks the directory up immediately. With + ``start_now=False`` a ``down`` marker is written so the service stays stopped until an + explicit ``gateway start``. Raises ValueError on an invalid name or existing directory, + RuntimeError if ``s6-svscanctl`` fails. """ - import shutil - import subprocess - svc_dir = self._service_dir(profile) if svc_dir.exists(): raise ValueError( @@ -1000,24 +727,17 @@ class S6ServiceManager: shutil.rmtree(tmp_dir, ignore_errors=True) tmp_dir.mkdir(parents=True) + def _write_script(path: Path, text: str) -> None: + path.write_text(text, encoding="utf-8") + path.chmod(0o755) + try: (tmp_dir / "type").write_text("longrun\n", encoding="utf-8") - - run_script = self._render_run_script(profile, extra_env or {}) - run_path = tmp_dir / "run" - run_path.write_text(run_script, encoding="utf-8") - run_path.chmod(0o755) - - finish_path = tmp_dir / "finish" - finish_path.write_text(self._render_finish_script(), encoding="utf-8") - finish_path.chmod(0o755) - + _write_script(tmp_dir / "run", self._render_run_script(profile, extra_env or {})) + _write_script(tmp_dir / "finish", self._render_finish_script()) # Persistent log rotation (OQ8-C). - log_subdir = tmp_dir / "log" - log_subdir.mkdir() - log_run = log_subdir / "run" - log_run.write_text(self._render_log_run(profile), encoding="utf-8") - log_run.chmod(0o755) + (tmp_dir / "log").mkdir() + _write_script(tmp_dir / "log" / "run", self._render_log_run(profile)) # Pre-create the supervise/ skeleton with hermes ownership # BEFORE we publish the slot. s6-supervise will EEXIST our @@ -1041,10 +761,7 @@ class S6ServiceManager: raise # Trigger rescan so s6-svscan picks up the new service. - result = subprocess.run( - [f"{_S6_BIN_DIR}/s6-svscanctl", "-a", str(self.scandir)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - ) + result = _s6_run("s6-svscanctl", "-a", str(self.scandir)) if result.returncode != 0: # Clean up: rescan failed, leave the directory in place would # be confusing (no supervisor watching it). @@ -1056,39 +773,22 @@ class S6ServiceManager: def unregister_profile_gateway(self, profile: str) -> None: """Stop the profile gateway service and remove its directory. - Idempotent: absent services are a no-op. Best-effort stop + - wait-for-down before removal so the running gateway process - gets a chance to shut down cleanly before its service dir + Idempotent: absent services are a no-op. Best-effort stop + wait-for-down before removal so + the running gateway process gets a chance to shut down cleanly before its service dir disappears. - Teardown ordering matters: ``s6-svscanctl -an`` is fired - **before** ``rmtree`` so s6-svscan reaps the supervise child - process (releasing its handle on ``supervise/lock`` and the - regular files inside the supervise dir), giving us a clean - directory to remove. Without the reap-first ordering, the - rmtree races s6-supervise on a set of root-owned files inside - the supervise dir and the dir is left half-removed. + Teardown ordering matters: ``s6-svscanctl -an`` is fired **before** ``rmtree`` so s6-svscan + reaps the supervise child process (releasing its handle on ``supervise/lock`` and the + regular files inside the supervise dir), giving us a clean directory to remove. """ - import shutil - import subprocess - import time - svc_dir = self._service_dir(profile) if not svc_dir.exists(): return # Stop the service (best effort — service may already be down). - subprocess.run( - [f"{_S6_BIN_DIR}/s6-svc", "-d", str(svc_dir)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - check=False, - ) + _s6_run("s6-svc", "-d", str(svc_dir)) # Wait for it to actually go down (up to 10s). - subprocess.run( - [f"{_S6_BIN_DIR}/s6-svwait", "-D", "-t", "10000", str(svc_dir)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=15, - check=False, - ) + _s6_run("s6-svwait", "-D", "-t", "10000", str(svc_dir), timeout=15) # Reap the supervise child FIRST: -n tells s6-svscan to drop # any supervise processes whose service dir is gone (which @@ -1096,11 +796,7 @@ class S6ServiceManager: # releases the file handles s6-supervise holds against the # supervise/lock + supervise/status + supervise/death_tally # files inside the slot, so the upcoming rmtree doesn't race. - subprocess.run( - [f"{_S6_BIN_DIR}/s6-svscanctl", "-an", str(self.scandir)], - capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, - check=False, - ) + _s6_run("s6-svscanctl", "-an", str(self.scandir)) # Give s6-svscan a moment to reap. There's no synchronous # "scan completed" handshake — the -a/-n trigger just sets a # flag s6-svscan reads on its next loop iteration. 200ms is @@ -1118,20 +814,21 @@ class S6ServiceManager: shutil.rmtree(svc_dir, ignore_errors=True) def list_profile_gateways(self) -> list[str]: - """Return the profile names of all currently-registered gateway services. - - Filters the scandir to entries that match the ``gateway-`` prefix. - Other services (e.g. ``s6-linux-init-shutdownd``) are ignored. - """ + """Return the profile names of all currently-registered gateway services.""" if not self.scandir.exists(): return [] - profiles: list[str] = [] - for entry in self.scandir.iterdir(): - if entry.name.startswith("."): - continue - if not entry.is_dir(): - continue - if not entry.name.startswith(S6_SERVICE_PREFIX): - continue - profiles.append(entry.name[len(S6_SERVICE_PREFIX):]) - return profiles + return [ + entry.name[len(S6_SERVICE_PREFIX):] + for entry in self.scandir.iterdir() + if not entry.name.startswith(".") + and entry.is_dir() + and entry.name.startswith(S6_SERVICE_PREFIX) + ] + + +_MANAGER_CLASSES: dict[str, type] = { + "systemd": SystemdServiceManager, + "launchd": LaunchdServiceManager, + "windows": WindowsServiceManager, + "s6": S6ServiceManager, +} diff --git a/hermes_cli/uninstall.py b/hermes_cli/uninstall.py index fbced3bab7..7586b439e8 100644 --- a/hermes_cli/uninstall.py +++ b/hermes_cli/uninstall.py @@ -1,10 +1,4 @@ -""" -Hermes Agent Uninstaller. - -Provides options for: -- Full uninstall: Remove everything including configs and data -- Keep data: Remove code but keep ~/.hermes/ (configs, sessions, logs) -""" +"""Hermes Agent Uninstaller.""" import os import shutil @@ -16,134 +10,141 @@ from hermes_constants import get_hermes_home from hermes_cli.colors import Colors, color -def log_info(msg: str): - print(f"{color('→', Colors.CYAN)} {msg}") +def _logger(mark: str, col: str): + return lambda msg: print(f"{color(mark, col)} {msg}") -def log_success(msg: str): - print(f"{color('✓', Colors.GREEN)} {msg}") -def log_warn(msg: str): - print(f"{color('⚠', Colors.YELLOW)} {msg}") +log_info = _logger("→", Colors.CYAN) +log_success = _logger("✓", Colors.GREEN) +log_warn = _logger("⚠", Colors.YELLOW) + + +def _print_box(middle: str, col: str) -> None: + """Print the 3-line framed heading used by the uninstall screens.""" + print(color("┌─────────────────────────────────────────────────────────┐", col, Colors.BOLD)) + print(color(middle, col, Colors.BOLD)) + print(color("└─────────────────────────────────────────────────────────┘", col, Colors.BOLD)) + + +def _prompt(text: str): + """``input(text).strip().lower()``; None (after printing "Cancelled.") on Ctrl-C/EOF.""" + try: + return input(text).strip().lower() + except (KeyboardInterrupt, EOFError): + print() + print("Cancelled.") + return None + + +def _cancelled() -> None: + print() + print("Uninstall cancelled.") + + +def _confirm_yes(text: str) -> bool: + """Ask the user to type ``yes``; False (after the cancel line, unless Ctrl-C/EOF) otherwise.""" + confirm = _prompt(f"Type '{color('yes', Colors.YELLOW)}' {text}: ") + if confirm == "yes": + return True + if confirm is not None: + _cancelled() + return False + + +def _remove_each(candidates, remove) -> list: + """Run ``remove(path)`` for each candidate; collect those it reports removed (truthy). + + Failures are downgraded to the shared ``Could not remove : `` warning. + """ + removed = [] + for path in candidates: + try: + if remove(path): + removed.append(path) + except Exception as e: + log_warn(f"Could not remove {path}: {e}") + return removed + def get_project_root() -> Path: """Get the project installation directory.""" return Path(__file__).parent.parent.resolve() -def find_shell_configs() -> list: - """Find shell configuration files that might have PATH entries.""" - home = Path.home() - configs = [] - - candidates = [ - home / ".bashrc", - home / ".bash_profile", - home / ".profile", - home / ".zshrc", - home / ".zprofile", - ] - - for config in candidates: - if config.exists(): - configs.append(config) - - return configs +_SHELL_RC_NAMES = (".bashrc", ".bash_profile", ".profile", ".zshrc", ".zprofile") + + +def _strip_hermes_path_lines(content: str) -> str: + """Drop the ``# Hermes Agent`` marker (+ its PATH line) and any hermes PATH line; squash blank runs.""" + new_lines = [] + skip_next = False + for line in content.split('\n'): + if '# Hermes Agent' in line or '# hermes-agent' in line: + skip_next = True + continue + if skip_next and ('hermes' in line.lower() and 'PATH' in line): + skip_next = False + continue + skip_next = False + if 'hermes' in line.lower() and ('PATH=' in line or 'path=' in line.lower()): + continue + new_lines.append(line) + new_content = '\n'.join(new_lines) + while '\n\n\n' in new_content: + new_content = new_content.replace('\n\n\n', '\n\n') + return new_content def remove_path_from_shell_configs(): """Remove Hermes PATH entries from shell configuration files.""" - configs = find_shell_configs() + home = Path.home() removed_from = [] - - for config_path in configs: + for config_path in (c for c in (home / n for n in _SHELL_RC_NAMES) if c.exists()): try: content = config_path.read_text(encoding="utf-8") - original_content = content - - # Remove lines containing hermes-agent or hermes PATH entries - new_lines = [] - skip_next = False - - for line in content.split('\n'): - # Skip the "# Hermes Agent" comment and following line - if '# Hermes Agent' in line or '# hermes-agent' in line: - skip_next = True - continue - if skip_next and ('hermes' in line.lower() and 'PATH' in line): - skip_next = False - continue - skip_next = False - - # Remove any PATH line containing hermes - if 'hermes' in line.lower() and ('PATH=' in line or 'path=' in line.lower()): - continue - - new_lines.append(line) - - new_content = '\n'.join(new_lines) - - # Clean up multiple blank lines - while '\n\n\n' in new_content: - new_content = new_content.replace('\n\n\n', '\n\n') - - if new_content != original_content: + new_content = _strip_hermes_path_lines(content) + if new_content != content: from utils import atomic_write_text - # This is the user's own shell rc, not a Hermes-owned file, and - # nothing in this function backs it up. A bare write_text() - # truncates it before the new content lands, so a crash or - # SIGINT mid-write leaves the user with an empty or truncated - # ~/.zshrc -- and the enclosing `except Exception` downgrades - # that to a warning, so the next login just starts a bare - # shell. atomic_replace also resolves a symlinked rc file, so a - # dotfiles-repo setup keeps the symlink instead of having it - # replaced by a regular file. preserve_mode keeps the rc's - # permission bits (normally 0644) and owner (sudo-run - # uninstalls) instead of mkstemp's 0600/root. + # This is the user's own shell rc, not a Hermes-owned file, and nothing here backs + # it up. A bare write_text() truncates it before the new content lands, so a crash + # or SIGINT mid-write leaves an empty/truncated ~/.zshrc -- and the enclosing + # `except Exception` downgrades that to a warning, so the next login just starts a + # bare shell. atomic_replace also resolves a symlinked rc file (dotfiles repos keep + # the symlink). preserve_mode keeps the rc's permission bits (normally 0644) and + # owner (sudo-run uninstalls) instead of mkstemp's 0600/root. atomic_write_text(config_path, new_content, preserve_mode=True) removed_from.append(config_path) - except Exception as e: log_warn(f"Could not update {config_path}: {e}") - return removed_from def remove_wrapper_script(): """Remove the hermes wrapper script if it exists.""" - wrapper_paths = [ - Path.home() / ".local" / "bin" / "hermes", - Path.home() / ".local" / "bin" / "hermes-acp", - Path.home() / ".local" / "bin" / "hermes-agent", - Path("/usr/local/bin/hermes"), - Path("/usr/local/bin/hermes-acp"), - Path("/usr/local/bin/hermes-agent"), - ] - - removed = [] - for wrapper in wrapper_paths: - if wrapper.exists(): - try: - # Check if it's our wrapper (contains hermes_cli reference) - content = wrapper.read_text(encoding="utf-8") - if 'hermes_cli' in content or 'hermes-agent' in content: - wrapper.unlink() - removed.append(wrapper) - except Exception as e: - log_warn(f"Could not remove {wrapper}: {e}") - - return removed + def _unlink_ours(wrapper: Path) -> bool: + # Check if it's our wrapper (contains hermes_cli reference) + content = wrapper.read_text(encoding="utf-8") + if 'hermes_cli' in content or 'hermes-agent' in content: + wrapper.unlink() + return True + return False + + candidates = ( + bin_dir / name + for bin_dir in (Path.home() / ".local" / "bin", Path("/usr/local/bin")) + for name in ("hermes", "hermes-acp", "hermes-agent") + ) + return _remove_each((w for w in candidates if w.exists()), _unlink_ours) def _node_symlink_candidate_dirs() -> "list[Path]": """Directories where the installer may have placed node/npm/npx symlinks.""" dirs: list[Path] = [Path.home() / ".local" / "bin"] - # Root FHS installs put links in /usr/local/bin. - if sys.platform == "linux": + if sys.platform == "linux": # root FHS installs put links in /usr/local/bin dirs.append(Path("/usr/local/bin")) - # Termux installs put links in $PREFIX/bin. prefix = os.environ.get("PREFIX", "") - if prefix and "com.termux" in prefix: + if "com.termux" in prefix: # Termux installs put links in $PREFIX/bin dirs.append(Path(prefix) / "bin") return dirs @@ -151,58 +152,41 @@ def _node_symlink_candidate_dirs() -> "list[Path]": def remove_node_symlinks(hermes_home: Path) -> list: """Remove the node/npm/npx symlinks the installer placed on PATH. - The POSIX installer (``scripts/install.sh`` / ``scripts/lib/node-bootstrap.sh``) - symlinks node/npm/npx into the same directory as the ``hermes`` command: + - ``/usr/local/bin/`` on root FHS installs (Linux, uid 0) - ``$PREFIX/bin/`` on Termux - + ``~/.local/bin/`` otherwise (the common non-root case) - - ``/usr/local/bin/`` on root FHS installs (Linux, uid 0) - - ``$PREFIX/bin/`` on Termux - - ``~/.local/bin/`` otherwise (the common non-root case) - - We check all candidate directories so that uninstall works regardless of - how the install was done (e.g. a root FHS install that placed links in - ``/usr/local/bin``, or an older install that used ``~/.local/bin`` before - the FHS fix). Only symlinks that resolve into this Hermes home's ``node`` - directory are removed — links the user has repointed elsewhere (nvm, fnm, - etc.) are left untouched. + We check all candidate directories so that uninstall works regardless of how the install was + done (e.g. a root FHS install that placed links in ``/usr/local/bin``, or an older install that + used ``~/.local/bin`` before the FHS fix). """ node_dir = (hermes_home / "node").resolve() - removed = [] - for name in ("node", "npm", "npx"): - for bin_dir in _node_symlink_candidate_dirs(): - link = bin_dir / name - try: - # Only act on symlinks — never delete a real binary the user put here. - if not link.is_symlink(): - continue + def _unlink_ours(link: Path) -> bool: + # Only act on symlinks — never delete a real binary the user put here. + if not link.is_symlink(): + return False + # Resolve the link target and confirm it points into our node dir. os.readlink + manual + # join handles broken (dangling) links too; Path.resolve() on a dangling link still + # returns the target path. + target = Path(os.readlink(link)) + if not target.is_absolute(): + target = (link.parent / target) + target = target.resolve() + if target == node_dir or node_dir in target.parents: + link.unlink() + return True + return False - # Resolve the link target and confirm it points into our node dir. - # os.readlink + manual join handles broken (dangling) links too; - # Path.resolve() on a dangling link still returns the target path. - target = Path(os.readlink(link)) - if not target.is_absolute(): - target = (link.parent / target) - target = target.resolve() - - if target == node_dir or node_dir in target.parents: - link.unlink() - removed.append(link) - except Exception as e: - log_warn(f"Could not remove {link}: {e}") - - return removed + candidates = (bin_dir / name for name in ("node", "npm", "npx") for bin_dir in _node_symlink_candidate_dirs()) + return _remove_each(candidates, _unlink_ours) def uninstall_gateway_service(): - """Stop and uninstall the gateway service (systemd, launchd, Windows - Scheduled Task / Startup folder) and kill any standalone gateway processes. + """Stop/uninstall the gateway service on every platform and kill standalone gateways. - Delegates to the gateway module which handles: - - Linux: user + system systemd services (with proper DBUS env setup) - - macOS: launchd plists - - Windows: Scheduled Task + Startup-folder fallback, via ``gateway_windows`` - - All platforms: standalone ``hermes gateway run`` processes - - Termux/Android: skips systemd (no systemd on Android), still kills standalone processes + Delegates to the gateway module: systemd user+system units (Linux), launchd (macOS), + Scheduled Task + Startup folder (Windows), plus standalone ``hermes gateway run`` processes. + Termux/Android has no systemd, so only the process kill applies there. """ import platform stopped_something = False @@ -210,100 +194,106 @@ def uninstall_gateway_service(): # 1. Kill any standalone gateway processes (all platforms, including Termux) try: from hermes_cli.gateway import kill_gateway_processes, find_gateway_pids - pids = find_gateway_pids() - if pids: - killed = kill_gateway_processes() - if killed: - log_success(f"Killed {killed} running gateway process(es)") - stopped_something = True + killed = kill_gateway_processes() if find_gateway_pids() else 0 + if killed: + log_success(f"Killed {killed} running gateway process(es)") + stopped_something = True except Exception as e: log_warn(f"Could not check for gateway processes: {e}") - system = platform.system() - # Termux/Android has no systemd and no launchd — nothing left to do. - prefix = os.getenv("PREFIX", "") - is_termux = bool(os.getenv("TERMUX_VERSION") or "com.termux/files/usr" in prefix) - if is_termux: + if os.getenv("TERMUX_VERSION") or "com.termux/files/usr" in os.getenv("PREFIX", ""): return stopped_something - # 2. Linux: uninstall systemd services (both user and system scopes) - if system == "Linux": + # 2. Per-platform service removal (systemd / launchd / Scheduled Task). + step = _GATEWAY_SERVICE_REMOVERS.get(platform.system()) + if step is not None: + remover, warn_label = step try: - from hermes_cli.gateway import ( - get_systemd_unit_path, - get_service_name, - _systemctl_cmd, - ) - svc_name = get_service_name() - - for is_system in (False, True): - unit_path = get_systemd_unit_path(system=is_system) - if not unit_path.exists(): - continue - - scope = "system" if is_system else "user" - try: - if is_system and os.geteuid() != 0: # windows-footgun: ok — Linux systemd uninstall path, guarded by `if system == "Linux"` above - log_warn(f"System gateway service exists at {unit_path} " - f"but needs sudo to remove") - continue - - cmd = _systemctl_cmd(is_system) - subprocess.run(cmd + ["stop", svc_name], - capture_output=True, check=False) - subprocess.run(cmd + ["disable", svc_name], - capture_output=True, check=False) - unit_path.unlink() - subprocess.run(cmd + ["daemon-reload"], - capture_output=True, check=False) - log_success(f"Removed {scope} gateway service ({unit_path})") - stopped_something = True - except Exception as e: - log_warn(f"Could not remove {scope} gateway service: {e}") + stopped_something = remover() or stopped_something except Exception as e: - log_warn(f"Could not check systemd gateway services: {e}") - - # 3. macOS: uninstall launchd plist - elif system == "Darwin": - try: - from hermes_cli.gateway import get_launchd_plist_path - plist_path = get_launchd_plist_path() - if plist_path.exists(): - subprocess.run(["launchctl", "unload", str(plist_path)], - capture_output=True, check=False) - plist_path.unlink() - log_success(f"Removed macOS gateway service ({plist_path})") - stopped_something = True - except Exception as e: - log_warn(f"Could not remove launchd gateway service: {e}") - - # 4. Windows: uninstall Scheduled Task + Startup-folder entry. The - # gateway_windows module already knows how to locate and remove both - # code paths (schtasks /Delete + .cmd unlink) and how to stop any - # running detached pythonw gateway process. We call into it so the - # uninstall logic stays in exactly one place. - elif system == "Windows": - try: - from hermes_cli import gateway_windows - if gateway_windows.is_installed() or gateway_windows.is_task_registered() \ - or gateway_windows.is_startup_entry_installed(): - try: - gateway_windows.stop() - except Exception as e: - log_warn(f"Could not stop Windows gateway cleanly: {e}") - try: - gateway_windows.uninstall() - log_success("Removed Windows gateway (Scheduled Task + Startup entry)") - stopped_something = True - except Exception as e: - log_warn(f"Could not fully uninstall Windows gateway: {e}") - except Exception as e: - log_warn(f"Could not check Windows gateway service: {e}") + log_warn(f"{warn_label}: {e}") return stopped_something +def _remove_systemd_gateway() -> bool: + """Linux: uninstall systemd services (both user and system scopes).""" + from hermes_cli.gateway import ( + get_systemd_unit_path, + get_service_name, + _systemctl_cmd, + ) + svc_name = get_service_name() + removed_any = False + + for is_system, scope in ((False, "user"), (True, "system")): + unit_path = get_systemd_unit_path(system=is_system) + if not unit_path.exists(): + continue + try: + if is_system and os.geteuid() != 0: # windows-footgun: ok — Linux-only systemd path (dispatched on platform.system()) + log_warn(f"System gateway service exists at {unit_path} but needs sudo to remove") + continue + + cmd = _systemctl_cmd(is_system) + for verb in ("stop", "disable"): + subprocess.run(cmd + [verb, svc_name], capture_output=True, check=False) + unit_path.unlink() + subprocess.run(cmd + ["daemon-reload"], capture_output=True, check=False) + log_success(f"Removed {scope} gateway service ({unit_path})") + removed_any = True + except Exception as e: + log_warn(f"Could not remove {scope} gateway service: {e}") + return removed_any + + +def _remove_launchd_gateway() -> bool: + """macOS: uninstall launchd plist.""" + from hermes_cli.gateway import get_launchd_plist_path + plist_path = get_launchd_plist_path() + if not plist_path.exists(): + return False + subprocess.run(["launchctl", "unload", str(plist_path)], + capture_output=True, check=False) + plist_path.unlink() + log_success(f"Removed macOS gateway service ({plist_path})") + return True + + +def _remove_windows_gateway() -> bool: + """Windows: uninstall Scheduled Task + Startup-folder entry. + + The gateway_windows module already knows how to locate and remove both code paths (schtasks + /Delete + .cmd unlink) and how to stop any running detached pythonw gateway process. We call + into it so the uninstall logic stays in exactly one place. + """ + from hermes_cli import gateway_windows + if not any(probe() for probe in ( + gateway_windows.is_installed, gateway_windows.is_task_registered, gateway_windows.is_startup_entry_installed, + )): + return False + try: + gateway_windows.stop() + except Exception as e: + log_warn(f"Could not stop Windows gateway cleanly: {e}") + try: + gateway_windows.uninstall() + log_success("Removed Windows gateway (Scheduled Task + Startup entry)") + return True + except Exception as e: + log_warn(f"Could not fully uninstall Windows gateway: {e}") + return False + + +# platform.system() -> (remover, warning label when the remover itself blows up) +_GATEWAY_SERVICE_REMOVERS = { + "Linux": (_remove_systemd_gateway, "Could not check systemd gateway services"), + "Darwin": (_remove_launchd_gateway, "Could not remove launchd gateway service"), + "Windows": (_remove_windows_gateway, "Could not check Windows gateway service"), +} + + # ============================================================================ # Windows-specific uninstall helpers # ============================================================================ @@ -339,92 +329,77 @@ def uninstall_gateway_service(): def _hermes_path_markers(hermes_home: Path, *, include_managed_bin: bool = False) -> list[str]: """Path-entry substrings that identify Hermes-owned User-PATH entries. - ``include_managed_bin`` adds the managed binary dir (``\\bin``, - holding the hermes launchers and the managed uv) — only wanted when - that dir is about to be deleted (full uninstall from the default root), - so a keep-data uninstall leaves the still-working managed uv resolvable. + ``include_managed_bin`` adds the managed binary dir (``\bin``, holding the hermes + launchers and the managed uv) — only wanted when that dir is about to be deleted (full uninstall + from the default root), so a keep-data uninstall leaves the still-working managed uv resolvable. """ root = str(hermes_home).rstrip("\\/") - # Match on prefix so sub-entries (git\cmd, git\bin, git\usr\bin, node, etc.) - # all get swept. Also match the bare hermes-agent install dir. - markers = [root + "\\hermes-agent", root + "\\git", root + "\\node", root + "\\venv"] - if include_managed_bin: - markers.append(root + "\\bin") - # Also match if HERMES_HOME was customised to somewhere else — find-and-nuke - # any entry whose path component contains "hermes". We don't want to catch - # unrelated entries like "chermes-foo" or "ephermeral", so we look for - # backslash-hermes as a word-ish boundary. - return markers + # Match on prefix so sub-entries (git\cmd, git\bin, git\usr\bin, node, etc.) all get swept. + subs = ("hermes-agent", "git", "node", "venv") + (("bin",) if include_managed_bin else ()) + return [f"{root}\\{sub}" for sub in subs] def remove_path_from_windows_registry(hermes_home: Path, *, include_managed_bin: bool = False) -> list[str]: """Strip Hermes-owned entries from User-scope PATH in the registry. - Returns the list of removed path entries. Operates on HKCU\\Environment, - same key the installer wrote to via ``[Environment]::SetEnvironmentVariable``. - - ``include_managed_bin`` adds ``\\bin`` (the managed binary - dir holding the hermes launchers and the managed uv) to the sweep. Only - pass it when that dir is actually being deleted — full uninstall from - the default root — so a keep-data uninstall leaves the still-working + ``include_managed_bin`` adds ``\bin`` (the managed binary dir holding the hermes + launchers and the managed uv) to the sweep. Only pass it when that dir is actually being deleted + — full uninstall from the default root — so a keep-data uninstall leaves the still-working managed uv resolvable. """ - try: - import winreg - except ImportError: - return [] # not on Windows, nothing to do + markers = tuple(m.lower() for m in _hermes_path_markers(hermes_home, include_managed_bin=include_managed_bin)) - removed: list[str] = [] - key_path = "Environment" - try: - with winreg.OpenKey(winreg.HKEY_CURRENT_USER, key_path, 0, - winreg.KEY_READ | winreg.KEY_WRITE) as key: - try: - path_value, path_type = winreg.QueryValueEx(key, "Path") - except FileNotFoundError: - return [] - # Preserve REG_EXPAND_SZ vs REG_SZ so unexpanded %VARS% survive. - entries = [e for e in path_value.split(";") if e] - markers = _hermes_path_markers(hermes_home, include_managed_bin=include_managed_bin) - kept: list[str] = [] - for entry in entries: - entry_norm = entry.rstrip("\\/") - matched = any(entry_norm.lower().startswith(m.lower()) for m in markers) - if matched: - removed.append(entry) - else: - kept.append(entry) - if removed: - new_value = ";".join(kept) - winreg.SetValueEx(key, "Path", 0, path_type, new_value) - except OSError as e: - log_warn(f"Could not edit User PATH in registry: {e}") - return removed + def edit(winreg, key, removed): + try: + path_value, path_type = winreg.QueryValueEx(key, "Path") + except FileNotFoundError: + return + # Preserve REG_EXPAND_SZ vs REG_SZ so unexpanded %VARS% survive. + kept: list[str] = [] + for entry in (e for e in path_value.split(";") if e): + is_ours = entry.rstrip("\\/").lower().startswith(markers) + (removed if is_ours else kept).append(entry) + if removed: + winreg.SetValueEx(key, "Path", 0, path_type, ";".join(kept)) + + return _edit_user_environment(edit, warn_label="Could not edit User PATH in registry") def remove_hermes_env_vars_windows() -> list[str]: """Delete HERMES_HOME and HERMES_GIT_BASH_PATH from User-scope env vars.""" + + def edit(winreg, key, removed): + for name in ("HERMES_HOME", "HERMES_GIT_BASH_PATH"): + try: + winreg.QueryValueEx(key, name) + except FileNotFoundError: + continue + try: + winreg.DeleteValue(key, name) + removed.append(name) + except OSError as e: + log_warn(f"Could not delete {name} from User env: {e}") + + return _edit_user_environment(edit, warn_label="Could not open User Environment key") + + +def _edit_user_environment(edit, *, warn_label: str) -> list[str]: + """Open HKCU\\Environment read/write and run ``edit(winreg, key, removed)``. + + Returns whatever ``edit`` appended to ``removed`` — even when a later registry call raised, so + callers report exactly what was touched. ``[]`` off-Windows (no ``winreg``). + """ + removed: list[str] = [] try: import winreg except ImportError: - return [] - - removed: list[str] = [] + return removed # not on Windows, nothing to do try: with winreg.OpenKey(winreg.HKEY_CURRENT_USER, "Environment", 0, winreg.KEY_READ | winreg.KEY_WRITE) as key: - for name in ("HERMES_HOME", "HERMES_GIT_BASH_PATH"): - try: - winreg.QueryValueEx(key, name) - except FileNotFoundError: - continue - try: - winreg.DeleteValue(key, name) - removed.append(name) - except OSError as e: - log_warn(f"Could not delete {name} from User env: {e}") + edit(winreg, key, removed) except OSError as e: - log_warn(f"Could not open User Environment key: {e}") + log_warn(f"{warn_label}: {e}") return removed @@ -432,35 +407,21 @@ def remove_portable_tooling_windows(hermes_home: Path) -> list[Path]: """Delete PortableGit and Node installs the Windows installer created under ``%LOCALAPPDATA%\\hermes\\``. Only called on full uninstall; they're isolated from any system Git / Node so they cannot break other tools.""" - removed: list[Path] = [] - for sub in ("git", "node", "gateway-service"): - target = hermes_home / sub - if target.exists(): - try: - shutil.rmtree(target, ignore_errors=False) - removed.append(target) - except Exception as e: - log_warn(f"Could not remove {target}: {e}") - return removed + def _rmtree(target: Path) -> bool: + shutil.rmtree(target, ignore_errors=False) + return True + + targets = (hermes_home / sub for sub in ("git", "node", "gateway-service")) + return _remove_each((t for t in targets if t.exists()), _rmtree) def remove_windows_bin_launchers(*, windows: bool | None = None) -> list[Path]: - """Delete the ``hermes`` launchers install.ps1 staged in the managed - binary dir (the default Hermes root's ``bin``, next to the managed uv). + """Delete the ``hermes`` launchers install.ps1 staged in the managed ``bin`` dir. - Every uninstall mode deletes the code checkout, so the launchers — - which invoke ``\\venv\\Scripts`` — would otherwise dangle: - ``hermes`` in a new terminal resolves to a launcher whose target is - gone and errors, which reads worse than command-not-found. The managed - uv (uv*.exe) in the same dir is left for keep-data reinstalls. - - A launcher that IS this process's own trampoline is mandatory-locked - against deletion but not rename (same fact - ``_install_repair._quarantine_running_hermes_exe`` relies on), so - deletion falls back to renaming it aside with a non-executable suffix. - - *windows* is an injectable platform verdict for tests (same pattern as - ``_install_repair.ensure_windows_bin_launchers``). + Every uninstall mode deletes the checkout, so launchers pointing at ``/venv`` + would dangle and error, which reads worse than command-not-found; the managed uv is kept for + keep-data reinstalls. A launcher that is this process's own trampoline is locked against + deletion but not rename, so it is renamed aside with a non-executable suffix instead. """ if windows is None: windows = _is_windows() @@ -477,27 +438,18 @@ def remove_windows_bin_launchers(*, windows: bool | None = None) -> list[Path]: log_warn(f"Could not locate the managed binary dir: {e}") return [] - removed: list[Path] = [] - for name in _WINDOWS_BIN_LAUNCHERS: - for suffix in (".exe", ".cmd"): - launcher = bin_dir / f"{name}{suffix}" - if not launcher.exists(): - continue - try: - launcher.unlink() - removed.append(launcher) - except OSError: - aside = launcher.with_name(f"{launcher.name}.uninstalled.{os.getpid()}") - try: - os.rename(launcher, aside) - removed.append(launcher) - except OSError as e: - log_warn(f"Could not remove {launcher}: {e}") - return removed + def _unlink_or_rename_aside(launcher: Path) -> bool: + try: + launcher.unlink() + except OSError: + os.rename(launcher, launcher.with_name(f"{launcher.name}.uninstalled.{os.getpid()}")) + return True + + candidates = (bin_dir / f"{name}{suffix}" for name in _WINDOWS_BIN_LAUNCHERS for suffix in (".exe", ".cmd")) + return _remove_each((p for p in candidates if p.exists()), _unlink_or_rename_aside) def _is_windows() -> bool: - import sys return sys.platform == "win32" @@ -511,9 +463,9 @@ def _is_default_hermes_home(hermes_home: Path) -> bool: def _discover_named_profiles(): - """Return a list of ``ProfileInfo`` for every non-default profile, or ``[]`` - if profile support is unavailable or nothing is installed beyond the - default root.""" + """Return a list of ``ProfileInfo`` for every non-default profile, or ``[]`` if profile support is + unavailable or nothing is installed beyond the default root. + """ try: from hermes_cli.profiles import list_profiles except Exception: @@ -526,23 +478,18 @@ def _discover_named_profiles(): def _uninstall_profile(profile) -> None: - """Fully uninstall a single named profile: stop its gateway service, - remove its alias wrapper, and wipe its HERMES_HOME directory. + """Fully uninstall a named profile: stop its gateway, remove its alias, wipe its home. - We shell out to ``hermes -p gateway stop|uninstall`` because - service names, unit paths, and plist paths are all derived from the - current HERMES_HOME and can't be easily switched in-process. + Shells out to ``hermes -p gateway stop|uninstall`` because service names and unit/ + plist paths derive from the current HERMES_HOME and can't easily be switched in-process. """ - import sys as _sys name = profile.name - profile_home = profile.path - log_info(f"Uninstalling profile '{name}'...") # 1. Stop and remove this profile's gateway service. # Use `python -m hermes_cli.main` so we don't depend on a `hermes` # wrapper that may be half-removed mid-uninstall. - hermes_invocation = [_sys.executable, "-m", "hermes_cli.main", "--profile", name] + hermes_invocation = [sys.executable, "-m", "hermes_cli.main", "--profile", name] for subcmd in ("stop", "uninstall"): try: subprocess.run( @@ -567,21 +514,15 @@ def _uninstall_profile(profile) -> None: log_warn(f" Could not remove alias {alias_path}: {e}") # 3. Wipe the profile's HERMES_HOME directory. - try: - if profile_home.exists(): - shutil.rmtree(profile_home) - log_success(f" Removed {profile_home}") - except Exception as e: - log_warn(f" Could not remove {profile_home}: {e}") + _rmtree_step(profile.path, indent=" ", fully=False) def run_gui_uninstall(args): """GUI-only uninstall: remove the Chat GUI, leave the agent + data intact. - Mirrors ``hermes uninstall --gui``. Removes the desktop app's built - artifacts, the packaged app bundle (best-effort), and the Electron - userData dir — nothing under ``$HERMES_HOME`` config/sessions/.env, and - never the Python agent or its venv. + Mirrors ``hermes uninstall --gui``. Removes the desktop app's built artifacts, the packaged app + bundle (best-effort), and the Electron userData dir — nothing under ``$HERMES_HOME`` + config/sessions/.env, and never the Python agent or its venv. """ from hermes_cli.gui_uninstall import ( agent_is_installed, @@ -594,9 +535,7 @@ def run_gui_uninstall(args): skip_confirm = bool(getattr(args, "yes", False)) print() - print(color("┌─────────────────────────────────────────────────────────┐", Colors.MAGENTA, Colors.BOLD)) - print(color("│ ⚕ Hermes Chat GUI Uninstaller │", Colors.MAGENTA, Colors.BOLD)) - print(color("└─────────────────────────────────────────────────────────┘", Colors.MAGENTA, Colors.BOLD)) + _print_box("│ ⚕ Hermes Chat GUI Uninstaller │", Colors.MAGENTA) print() if not summary["gui_installed"]: @@ -607,9 +546,7 @@ def run_gui_uninstall(args): print(color("This removes the Chat GUI only. The Hermes agent stays installed.", Colors.CYAN)) print() print(color("Will remove:", Colors.YELLOW, Colors.BOLD)) - for p in summary["source_built_artifacts"]: - print(f" • {p}") - for p in summary["packaged_app_paths"]: + for p in (*summary["source_built_artifacts"], *summary["packaged_app_paths"]): print(f" • {p}") if summary["userdata_exists"]: print(f" • {summary['userdata_dir']} (desktop app data)") @@ -620,17 +557,8 @@ def run_gui_uninstall(args): print(f" • Your config, sessions, and secrets under {hermes_home}") print() - if not skip_confirm: - try: - confirm = input(f"Type '{color('yes', Colors.YELLOW)}' to remove the Chat GUI: ").strip().lower() - except (KeyboardInterrupt, EOFError): - print() - print("Cancelled.") - return - if confirm != "yes": - print() - print("Uninstall cancelled.") - return + if not skip_confirm and not _confirm_yes("to remove the Chat GUI"): + return print() print(color("Uninstalling Chat GUI...", Colors.CYAN, Colors.BOLD)) @@ -638,9 +566,7 @@ def run_gui_uninstall(args): uninstall_gui(hermes_home) print() - print(color("┌─────────────────────────────────────────────────────────┐", Colors.GREEN, Colors.BOLD)) - print(color("│ ✓ Chat GUI Uninstalled! │", Colors.GREEN, Colors.BOLD)) - print(color("└─────────────────────────────────────────────────────────┘", Colors.GREEN, Colors.BOLD)) + _print_box("│ ✓ Chat GUI Uninstalled! │", Colors.GREEN) print() print("The Hermes agent is still installed. Run 'hermes' to use the CLI,") print("or 'hermes uninstall' to remove the agent too.") @@ -648,12 +574,10 @@ def run_gui_uninstall(args): def run_uninstall(args): - """ - Run the uninstall process. - - Options: - - Full uninstall: removes code + ~/.hermes/ (configs, data, logs) - - Keep data: removes code but keeps ~/.hermes/ for future reinstall + """Run the uninstall process. + + Full uninstall removes code and ``~/.hermes/``; keep-data removes only the code so the + configs, data and logs survive a reinstall. """ project_root = get_project_root() hermes_home = get_hermes_home() @@ -678,22 +602,18 @@ def run_uninstall(args): # an unattended run, so it stays opt-in to the interactive flow. This is # the path the desktop app's detached cleanup script uses for its # lite/full modes. - skip_confirm = bool(getattr(args, "yes", False)) - if skip_confirm: - full_uninstall = bool(getattr(args, "full", False)) + if bool(getattr(args, "yes", False)): _perform_uninstall( project_root=project_root, hermes_home=hermes_home, - full_uninstall=full_uninstall, + full_uninstall=bool(getattr(args, "full", False)), remove_profiles=False, named_profiles=named_profiles, ) return print() - print(color("┌─────────────────────────────────────────────────────────┐", Colors.MAGENTA, Colors.BOLD)) - print(color("│ ⚕ Hermes Agent Uninstaller │", Colors.MAGENTA, Colors.BOLD)) - print(color("└─────────────────────────────────────────────────────────┘", Colors.MAGENTA, Colors.BOLD)) + _print_box("│ ⚕ Hermes Agent Uninstaller │", Colors.MAGENTA) print() # Show what will be affected @@ -707,10 +627,9 @@ def run_uninstall(args): if named_profiles: print(color("Other profiles detected:", Colors.CYAN, Colors.BOLD)) for p in named_profiles: - running = " (gateway running)" if getattr(p, "gateway_running", False) else "" - print(f" • {p.name}{running}: {p.path}") + print(f" • {p.name}{' (gateway running)' if getattr(p, 'gateway_running', False) else ''}: {p.path}") print() - + # Ask for confirmation print(color("Uninstall Options:", Colors.YELLOW, Colors.BOLD)) print() @@ -723,16 +642,12 @@ def run_uninstall(args): print(" 3) " + color("Cancel", Colors.CYAN) + " - Don't uninstall") print() - try: - choice = input(color("Select option [1/2/3]: ", Colors.BOLD)).strip() - except (KeyboardInterrupt, EOFError): - print() - print("Cancelled.") + choice = _prompt(color("Select option [1/2/3]: ", Colors.BOLD)) + if choice is None: return - if choice == "3" or choice.lower() in {"c", "cancel", "q", "quit", "n", "no"}: - print() - print("Uninstall cancelled.") + if choice in {"3", "c", "cancel", "q", "quit", "n", "no"}: + _cancelled() return full_uninstall = (choice == "2") @@ -742,20 +657,15 @@ def run_uninstall(args): # their alias wrappers, and wiping their HERMES_HOME dirs. Otherwise # those leave zombie services and data behind. remove_profiles = False + n_profiles = len(named_profiles) + profile_names = ", ".join(p.name for p in named_profiles) if full_uninstall and named_profiles: print() print(color("Other profiles will NOT be removed by default.", Colors.YELLOW)) - print(f"Found {len(named_profiles)} named profile(s): " + - ", ".join(p.name for p in named_profiles)) + print(f"Found {n_profiles} named profile(s): {profile_names}") print() - try: - resp = input(color( - f"Also stop and remove these {len(named_profiles)} profile(s)? [y/N]: ", - Colors.BOLD - )).strip().lower() - except (KeyboardInterrupt, EOFError): - print() - print("Cancelled.") + resp = _prompt(color(f"Also stop and remove these {n_profiles} profile(s)? [y/N]: ", Colors.BOLD)) + if resp is None: return remove_profiles = resp in {"y", "yes"} @@ -765,25 +675,12 @@ def run_uninstall(args): print(color("⚠️ WARNING: This will permanently delete ALL Hermes data!", Colors.RED, Colors.BOLD)) print(color(" Including: configs, API keys, sessions, scheduled jobs, logs", Colors.RED)) if remove_profiles: - print(color( - f" Plus {len(named_profiles)} profile(s): " + - ", ".join(p.name for p in named_profiles), - Colors.RED - )) + print(color(f" Plus {n_profiles} profile(s): {profile_names}", Colors.RED)) else: print("This will remove the Hermes code but keep your configuration and data.") print() - try: - confirm = input(f"Type '{color('yes', Colors.YELLOW)}' to confirm: ").strip().lower() - except (KeyboardInterrupt, EOFError): - print() - print("Cancelled.") - return - - if confirm != "yes": - print() - print("Uninstall cancelled.") + if not _confirm_yes("to confirm"): return _perform_uninstall( @@ -806,19 +703,41 @@ def _print_uninstall_dry_run(*, project_root: Path, hermes_home: Path, full_unin print(" • Hermes wrapper scripts and Hermes-managed node/npm/npx symlinks") print(" • Desktop Chat GUI artifacts") print(f" • Code checkout: {project_root}") - if full_uninstall: - print(f" • Hermes config/data: {hermes_home}") - if _is_default_hermes_home(hermes_home): - profiles = _discover_named_profiles() - if profiles: - print(" • Named profiles (interactive uninstall asks before removing):") - for prof in profiles: - print(f" - {prof.name}: {prof.path}") - else: + if not full_uninstall: print(f" • Keep Hermes config/data: {hermes_home}") + else: + print(f" • Hermes config/data: {hermes_home}") + profiles = _discover_named_profiles() if _is_default_hermes_home(hermes_home) else [] + if profiles: + print(" • Named profiles (interactive uninstall asks before removing):") + for prof in profiles: + print(f" - {prof.name}: {prof.path}") print() +def _remove_step(label: str, remove, success_fmt: str, none_msg: str) -> None: + """Announce ``label``, run ``remove()``, then log one success line per removed item (or ``none_msg``).""" + log_info(label) + removed = remove() + if removed: + for item in removed: + log_success(success_fmt.format(item)) + else: + log_info(none_msg) + + +def _rmtree_step(path: Path, *, indent: str = "", fully: bool = True) -> None: + """Best-effort ``rmtree`` with the shared success/warning lines.""" + try: + if path.exists(): + shutil.rmtree(path) + log_success(f"{indent}Removed {path}") + except Exception as e: + log_warn(f"{indent}Could not {'fully ' if fully else ''}remove {path}: {e}") + if fully: + log_info("You may need to manually remove it") + + def _perform_uninstall( *, project_root: Path, @@ -827,13 +746,11 @@ def _perform_uninstall( remove_profiles: bool, named_profiles: list, ) -> None: - """Execute the uninstall steps. Shared by the interactive and ``--yes`` - paths so the destructive sequence lives in exactly one place. + """Execute the uninstall steps; shared by the interactive and ``--yes`` paths. - Steps: stop gateway → strip PATH (rc files + Windows registry) → remove the - ``hermes`` wrapper + node symlinks → remove the desktop Chat GUI artifacts → - delete the code checkout → (Windows) remove PortableGit/Node → optionally - wipe ``$HERMES_HOME`` data and named profiles on full uninstall. + Order: stop gateway -> strip PATH (rc files + Windows registry) -> remove wrapper and node + symlinks -> remove desktop Chat GUI artifacts -> delete the checkout -> (Windows) remove + PortableGit/Node -> optionally wipe ``$HERMES_HOME`` and named profiles. """ print() print(color("Uninstalling...", Colors.CYAN, Colors.BOLD)) @@ -844,170 +761,91 @@ def _perform_uninstall( if not uninstall_gateway_service(): log_info("No gateway service or processes found") - # 2. Remove PATH entries from shell configs (POSIX) AND from the Windows - # User-scope registry. Both helpers no-op on the wrong platform so we - # can safely call them unconditionally. - log_info("Removing PATH entries from shell configs...") - removed_configs = remove_path_from_shell_configs() - if removed_configs: - for config in removed_configs: - log_success(f"Updated {config}") - else: - log_info("No PATH entries found to remove in shell rc files") + # 2-3b. PATH entries (POSIX rc files, then the Windows User-scope registry), wrapper script, + # Windows launchers, node/npm/npx symlinks. Windows notes: hermes_home is %VAR%-expanded so + # marker matching runs against fully resolved paths (install.ps1 writes literal + # C:\Users\\AppData\Local\hermes\git\cmd, not %LOCALAPPDATA%); the managed binary dir + # (hermes\bin: launchers + managed uv) leaves the PATH only when the full wipe below is + # about to delete it, so keep-data mode keeps the still-working uv resolvable; the launchers + # themselves always go — both modes delete the checkout, so a surviving launcher would + # dangle and error, worse than command-not-found. Symlinks are removed only when they still + # point into this Hermes home's node dir (never clobber nvm / user-managed Node). + windows = _is_windows() + sweep_managed_bin = windows and full_uninstall and _is_default_hermes_home(hermes_home) + for on_this_platform, label, remove, success_fmt, none_msg in ( + (True, "Removing PATH entries from shell configs...", + remove_path_from_shell_configs, "Updated {}", "No PATH entries found to remove in shell rc files"), + (windows, "Removing PATH entries from Windows User environment...", + lambda: remove_path_from_windows_registry( + Path(os.path.expandvars(str(hermes_home))), include_managed_bin=sweep_managed_bin), + "Removed from User PATH: {}", "No Hermes-owned PATH entries in User environment"), + (windows, "Removing HERMES_HOME / HERMES_GIT_BASH_PATH User env vars...", + remove_hermes_env_vars_windows, "Removed User env var: {}", "No Hermes-set User env vars to remove"), + (True, "Removing hermes command...", remove_wrapper_script, "Removed {}", "No wrapper script found"), + (windows, "Removing Windows hermes launchers...", + remove_windows_bin_launchers, "Removed {}", "No Windows hermes launchers found"), + (True, "Removing Hermes-managed node/npm/npx symlinks...", + lambda: remove_node_symlinks(hermes_home), "Removed {}", "No Hermes-managed node/npm/npx symlinks found"), + ): + if on_this_platform: + _remove_step(label, remove, success_fmt, none_msg) - if _is_windows(): - log_info("Removing PATH entries from Windows User environment...") - # Expand %LOCALAPPDATA% etc. in hermes_home so the marker matching is - # against fully resolved paths — installer writes literal strings - # like C:\Users\\AppData\Local\hermes\git\cmd, not %LOCALAPPDATA%. - # The managed binary dir (hermes\bin: launchers + managed uv) leaves - # the PATH only when the full wipe below is about to delete it; - # keep-data mode keeps the dir and the still-working uv resolvable. - sweep_managed_bin = full_uninstall and _is_default_hermes_home(hermes_home) - removed_path_entries = remove_path_from_windows_registry( - Path(os.path.expandvars(str(hermes_home))), - include_managed_bin=sweep_managed_bin, - ) - if removed_path_entries: - for entry in removed_path_entries: - log_success(f"Removed from User PATH: {entry}") - else: - log_info("No Hermes-owned PATH entries in User environment") - - log_info("Removing HERMES_HOME / HERMES_GIT_BASH_PATH User env vars...") - removed_env = remove_hermes_env_vars_windows() - if removed_env: - for name in removed_env: - log_success(f"Removed User env var: {name}") - else: - log_info("No Hermes-set User env vars to remove") - - # 3. Remove wrapper script - log_info("Removing hermes command...") - removed_wrappers = remove_wrapper_script() - if removed_wrappers: - for wrapper in removed_wrappers: - log_success(f"Removed {wrapper}") - else: - log_info("No wrapper script found") - - # 3a. Remove the Windows launchers from the managed binary dir. Both - # modes delete the code checkout below, so a surviving launcher - # would dangle — `hermes` in a new terminal would resolve and then - # error on its missing venv target, worse than command-not-found. - if _is_windows(): - log_info("Removing Windows hermes launchers...") - removed_launchers = remove_windows_bin_launchers() - if removed_launchers: - for launcher in removed_launchers: - log_success(f"Removed {launcher}") - else: - log_info("No Windows hermes launchers found") - - # 3b. Remove node/npm/npx symlinks the installer left in ~/.local/bin - # (only when they still point into this Hermes home's node dir, so we - # never clobber an existing nvm / user-managed Node). - log_info("Removing Hermes-managed node/npm/npx symlinks...") - removed_node_links = remove_node_symlinks(hermes_home) - if removed_node_links: - for link in removed_node_links: - log_success(f"Removed {link}") - else: - log_info("No Hermes-managed node/npm/npx symlinks found") - - # 3c. Remove the desktop Chat GUI's artifacts too (built renderer/release, - # node_modules, the packaged app bundle, and the Electron userData - # dir). Both the "keep data" and "full" CLI flows remove the agent - # code, so the GUI — which is just another consumer of the same - # checkout — should go with it. uninstall_gui() never touches config / - # sessions / .env, so it's safe in keep-data mode; on full uninstall the - # step-5 rmtree(hermes_home) would sweep the in-tree artifacts anyway, - # but the packaged app + Electron userData live OUTSIDE HERMES_HOME and - # must be cleaned explicitly here. + # 3c. Desktop Chat GUI artifacts (built renderer/release, node_modules, packaged app bundle, + # Electron userData). Both CLI flows remove the agent code, so the GUI — just another + # consumer of the same checkout — goes with it. uninstall_gui() never touches config / + # sessions / .env, so it's safe in keep-data mode; the packaged app + Electron userData + # live OUTSIDE HERMES_HOME, so the full-uninstall rmtree below would not reach them. log_info("Removing desktop Chat GUI artifacts...") try: from hermes_cli.gui_uninstall import uninstall_gui - gui_removed = uninstall_gui(hermes_home) - if not gui_removed: + if not uninstall_gui(hermes_home): log_info("No desktop GUI artifacts found") except Exception as e: log_warn(f"Could not remove desktop GUI artifacts: {e}") - # 4. Remove installation directory (code) + # 4. Remove installation directory (code) — we may be running from inside it. log_info("Removing installation directory...") - - # Check if we're running from within the install dir - # We need to be careful here - try: - if project_root.exists(): - # If the install is inside ~/.hermes/, just remove the hermes-agent subdir - if hermes_home in project_root.parents or project_root.parent == hermes_home: - shutil.rmtree(project_root) - log_success(f"Removed {project_root}") - else: - # Installation is somewhere else entirely - shutil.rmtree(project_root) - log_success(f"Removed {project_root}") - except Exception as e: - log_warn(f"Could not fully remove {project_root}: {e}") - log_info("You may need to manually remove it") + _rmtree_step(project_root) - # 4b. Remove Windows-only installer artifacts that are NOT user data: - # PortableGit, bundled Node, gateway-service dir. Installer put them - # under HERMES_HOME but they're install tooling, not config — safe to - # remove even in "keep data" mode. If we're doing a full uninstall - # the step-5 rmtree(hermes_home) would sweep them anyway; calling - # this helper there is a no-op since they'll already be gone. - if _is_windows(): - log_info("Removing Windows installer artifacts (PortableGit, Node, gateway-service)...") - removed_artifacts = remove_portable_tooling_windows(hermes_home) - if removed_artifacts: - for path in removed_artifacts: - log_success(f"Removed {path}") - else: - log_info("No Windows installer artifacts to remove") + # 4b. Windows-only installer artifacts that are NOT user data (PortableGit, bundled Node, + # gateway-service dir): install tooling under HERMES_HOME, safe to remove in keep-data mode. + if windows: + _remove_step( + "Removing Windows installer artifacts (PortableGit, Node, gateway-service)...", + lambda: remove_portable_tooling_windows(hermes_home), "Removed {}", + "No Windows installer artifacts to remove", + ) # 5. Optionally remove ~/.hermes/ data directory (and named profiles) if full_uninstall: - # 5a. Stop and remove each named profile's gateway service and - # alias wrapper. The profile HERMES_HOME dirs live under - # ``/profiles//`` and will be swept away by the - # rmtree below, but services + alias scripts live OUTSIDE the - # default root and have to be cleaned up explicitly. - if remove_profiles and named_profiles: + # 5a. Stop and remove each named profile's gateway service and alias wrapper. The profile + # HERMES_HOME dirs live under ``/profiles//`` and are swept by the rmtree + # below, but services + alias scripts live OUTSIDE the default root. + if remove_profiles: for prof in named_profiles: _uninstall_profile(prof) log_info("Removing configuration and data...") - try: - if hermes_home.exists(): - shutil.rmtree(hermes_home) - log_success(f"Removed {hermes_home}") - except Exception as e: - log_warn(f"Could not fully remove {hermes_home}: {e}") - log_info("You may need to manually remove it") + _rmtree_step(hermes_home) else: log_info(f"Keeping configuration and data in {hermes_home}") - - # Done + print() - print(color("┌─────────────────────────────────────────────────────────┐", Colors.GREEN, Colors.BOLD)) - print(color("│ ✓ Uninstall Complete! │", Colors.GREEN, Colors.BOLD)) - print(color("└─────────────────────────────────────────────────────────┘", Colors.GREEN, Colors.BOLD)) + _print_box("│ ✓ Uninstall Complete! │", Colors.GREEN) print() - + if not full_uninstall: print(color("Your configuration and data have been preserved:", Colors.CYAN)) print(f" {hermes_home}/") print() print("To reinstall later with your existing settings:") - if _is_windows(): + if windows: print(color(" iex (irm https://hermes-agent.nousresearch.com/install.ps1)", Colors.DIM)) else: print(color(" curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash", Colors.DIM)) print() - if _is_windows(): + if windows: print(color("Open a new terminal (PowerShell / Windows Terminal) to pick up", Colors.YELLOW)) print(color("the updated User PATH and environment variables.", Colors.YELLOW)) else: @@ -1031,16 +869,9 @@ class _UninstallArgs: def main(argv=None) -> int: """Module entrypoint: ``python -m hermes_cli.uninstall --mode ``. - Exists so the desktop app can run the uninstall under a Python interpreter - OUTSIDE the venv being deleted. On Windows, ``lite``/``full`` rmtree the - venv that contains the running ``python.exe`` — and a running .exe is - mandatory-locked, so doing that from the venv's own interpreter half-fails. - The desktop launches this with the system Python + ``PYTHONPATH=`` - so ``import hermes_cli`` resolves from source while the venv is torn down. - - This module imports only stdlib + ``hermes_constants`` + ``hermes_cli.colors`` - (and lazily ``hermes_cli.gui_uninstall``), so it runs fine under a bare - system Python with no site-packages from the venv. + This module imports only stdlib + ``hermes_constants`` + ``hermes_cli.colors`` (and lazily + ``hermes_cli.gui_uninstall``), so it runs fine under a bare system Python with no site-packages + from the venv. """ import argparse @@ -1051,13 +882,8 @@ def main(argv=None) -> int: required=True, help="gui = Chat GUI only; lite = GUI + agent, keep data; full = everything", ) - ns = parser.parse_args(argv) - args = _UninstallArgs(mode=ns.mode) - - if args.gui: - run_gui_uninstall(args) - else: - run_uninstall(args) + args = _UninstallArgs(mode=parser.parse_args(argv).mode) + (run_gui_uninstall if args.gui else run_uninstall)(args) return 0 diff --git a/hermes_cli/update_abort_recovery.py b/hermes_cli/update_abort_recovery.py index a4219db82d..243efe10ba 100644 --- a/hermes_cli/update_abort_recovery.py +++ b/hermes_cli/update_abort_recovery.py @@ -1,18 +1,10 @@ """Fresh-process recovery after the update's in-process restart phase aborts. -``hermes update`` performs its fleet restart in the interpreter that started -before ``git pull``. When that phase raises — the module graph it is running -no longer matches the checkout on disk — this module owns everything that -happens next: which runtimes a clean child may relaunch, what counts as proof -that a relaunch actually replaced the old generation, which pre-update -processes are still alive, and whether any of that adds up to a recovery that -may call itself complete (#92145). +``hermes update`` performs its fleet restart in the interpreter that started before ``git pull``. -It is deliberately a separate owner from ``update_cmd``: the abort path has -its own vocabulary (``verified`` / ``relaunch_attempted`` / ``failed``, serve -units, survivors) and its own fail-closed contract, and the update monolith -must not grow another authority surface (review on #96235). The child process -itself lives in :mod:`hermes_cli.update_restart_recovery`. +It is deliberately a separate owner from ``update_cmd``: the abort path has its own vocabulary +(``verified`` / ``relaunch_attempted`` / ``failed``, serve units, survivors) and its own fail-closed +contract, and the update monolith must not grow another authority surface (review on #96235). """ from __future__ import annotations @@ -35,18 +27,14 @@ def _serve_unit_recovery_available() -> bool: def _surviving_pre_update_serve_runtimes(plan) -> list[dict]: """Pre-update serve/dashboard runtimes that are STILL the same process. - Identity is the process incarnation ``(pid, create_time)``, never the PID - alone. ``ledger_entries()`` re-verifies that pair on every read and prunes - anything that is gone, so a live entry is a real process — but the numeric - PID can be reused, and a serve that was restarted correctly can come back - on the same number. Comparing PIDs alone would then report the successor - as the pre-update survivor and keep the update incomplete forever - (#92145 review). + Identity is the process incarnation ``(pid, create_time)``, never the PID alone. + ``ledger_entries()`` re-verifies that pair on every read and prunes anything that is gone, so a + live entry is a real process — but the numeric PID can be reused, and a serve that was restarted + correctly can come back on the same number. - Anything still here after the recovery pass is a live runtime on the - pre-update code generation, which is precisely the unsafe state #92145 - reports. Fail closed on missing evidence: an unreadable ledger, or an - incarnation neither side can produce, counts the runtime as surviving. + Anything still here after the recovery pass is a live runtime on the pre-update code generation, + which is precisely the unsafe state #92145 reports. Fail closed on missing evidence: an + unreadable ledger, or an incarnation neither side can produce, counts the runtime as surviving. """ planned: dict[int, dict] = {} try: @@ -57,13 +45,12 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]: if not isinstance(pid, int) or pid <= 0: continue detail = getattr(runtime, "detail", None) - created = detail.get("create_time") if isinstance(detail, dict) else None planned[pid] = { "pid": pid, "kind": str(getattr(runtime, "kind", "")), "profile": str(getattr(runtime, "profile", "")), "supervisor": str(getattr(runtime, "supervisor", "")), - "_create_time": created if isinstance(created, (int, float)) else None, + "_create_time": _numeric(detail.get("create_time") if isinstance(detail, dict) else None), } except Exception as exc: logger.debug("Could not read planned serve runtimes: %s", exc) @@ -73,39 +60,36 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]: try: from hermes_cli.process_identity import ledger_entries - live: dict[int, float | None] = {} - for entry in ledger_entries(): - if entry.get("purpose") not in ("serve", "dashboard"): - continue - pid = entry.get("pid") - if not isinstance(pid, int): - continue - created = entry.get("create_time") - live[pid] = created if isinstance(created, (int, float)) else None + live: dict[int, float | None] = { + entry["pid"]: _numeric(entry.get("create_time")) + for entry in ledger_entries() + if entry.get("purpose") in ("serve", "dashboard") and isinstance(entry.get("pid"), int) + } except Exception as exc: logger.debug("Serve/dashboard survivor probe failed: %s", exc) - return sorted( - (_without_incarnation(row) for row in planned.values()), - key=lambda row: row["pid"], - ) + live = None survivors = [] for pid, row in planned.items(): - if pid not in live: - continue - planned_created = row["_create_time"] - live_created = live[pid] - if ( - planned_created is not None - and live_created is not None - and abs(float(live_created) - float(planned_created)) >= 2.0 - ): - # Same number, different process: the pre-update runtime is gone - # and something new registered under its PID. Not a survivor. - continue + if live is not None: + if pid not in live: + continue + planned_created, live_created = row["_create_time"], live[pid] + if ( + planned_created is not None + and live_created is not None + and abs(float(live_created) - float(planned_created)) >= 2.0 + ): + # Same number, different process: the pre-update runtime is gone + # and something new registered under its PID. Not a survivor. + continue survivors.append(_without_incarnation(row)) return sorted(survivors, key=lambda row: row["pid"]) +def _numeric(value): + return value if isinstance(value, (int, float)) else None + + def _without_incarnation(row: dict) -> dict: """The operator-facing survivor row (incarnation is a matching key only).""" return {key: value for key, value in row.items() if key != "_create_time"} @@ -114,13 +98,8 @@ def _without_incarnation(row: dict) -> dict: def _qualified_serve_skips(skip_units) -> list[dict]: """Scope-qualify the units the aborted phase already settled. - ``restarted_scoped_units`` records ``/`` because - ``hermes-serve.service`` can exist in BOTH the user and the system - manager, and those are two different processes. Handing the fresh child a - bare unit name would make one settled scope suppress recovery of the other - — the stale one would never be restarted and nothing downstream would say - so. Entries that carry no scope (a payload built by a pre-update - interpreter) are forwarded without one and read as scope-agnostic there. + ``restarted_scoped_units`` records ``/`` because ``hermes-serve.service`` can exist + in BOTH the user and the system manager, and those are two different processes. """ rows: list[dict] = [] for entry in sorted(skip_units or ()): @@ -141,74 +120,35 @@ def _recover_gateway_restart_after_abort( ) -> dict[str, list]: """Retry supervised gateway restarts from a clean Python process. - ``hermes update`` normally performs the fleet restart in the interpreter - that started before ``git pull``. If that phase raises while importing the - new tree, a warning alone leaves the old gateway alive against new files on - disk. The recovery boundary launches the existing per-profile - ``gateway restart`` command through a new interpreter, preserving its - platform-specific drain and service-manager logic without inheriting the - stale ``sys.modules`` graph. + ``hermes update`` normally performs the fleet restart in the interpreter that started before + ``git pull``. - Only profiles classified as supervisor-owned by the pre-update inventory - are handed off. A manual gateway must remain running and be reported for - explicit operator action rather than being killed without a relaunch - authority; serve/dashboard runtimes from the spawn ledger are likewise - recorded as skipped with a reason instead of vanishing from the pass. - The returned protocol is persisted in the update receipt so operators can - distinguish a spawn failure from a per-profile failure. - - The same child additionally restarts active ``hermes-serve*`` systemd - units (#92145). ``hermes serve`` hosts ``tui_gateway.server`` and is - restarted by the in-process phase alongside the gateway units, but no - per-profile ``gateway restart`` command reaches it — so an abort used to - leave it holding the pre-pull module graph with nothing left to notice. - - ``skip_units`` names the units the aborted phase already settled, as - ``/``. The scope is part of the identity, not decoration: - ``hermes-serve.service`` can exist in both the user and the system - manager as two different processes, and an unqualified token would let a - settled one suppress recovery of a stale one (review on #96235). - - Outcome honesty: ``verified`` means the fresh child independently observed - the profile's systemd unit active after the relaunch. A zero exit from - ``gateway restart`` alone is NOT observed proof that the new code - generation is serving, so those outcomes are reported as - ``relaunch_attempted`` and never claim supervisor coverage. + Only profiles classified as supervisor-owned by the pre-update inventory are handed off. """ from hermes_cli.update_cmd import _gateway_recovery_partition - candidates, skipped = _gateway_recovery_partition( - plan, skip_profiles=skip_profiles - ) + candidates, skipped = _gateway_recovery_partition(plan, skip_profiles=skip_profiles) profiles = sorted(candidates) recover_serve = _serve_unit_recovery_available() _empty_serve: dict[str, list] = {"verified": [], "failed": []} - if not profiles and not recover_serve: + + def _result(requested, verified, relaunch_attempted, failed, serve_units=None) -> dict[str, list]: return { - "requested": [], - "verified": [], - "relaunch_attempted": [], - "failed": [], + "requested": requested, + "verified": verified, + "relaunch_attempted": relaunch_attempted, + "failed": failed, "skipped": skipped, - "serve_units": dict(_empty_serve), + "serve_units": dict(_empty_serve) if serve_units is None else serve_units, } + if not profiles and not recover_serve: + return _result([], [], [], []) + def _all_failed() -> dict[str, list]: - return { - "requested": profiles, - "verified": [], - "relaunch_attempted": [], - "failed": profiles, - "skipped": skipped, - "serve_units": dict(_empty_serve), - } + return _result(profiles, [], [], profiles) - command = [ - sys.executable, - "-m", - "hermes_cli.update_restart_recovery", - "--stdin", - ] + command = [sys.executable, "-m", "hermes_cli.update_restart_recovery", "--stdin"] env = os.environ.copy() env["HERMES_UPDATE_RESTART_RECOVERY"] = "1" for marker in ("_HERMES_GATEWAY", "HERMES_GATEWAY", "HERMES_GATEWAY_MODE"): @@ -224,15 +164,7 @@ def _recover_gateway_restart_after_abort( if not systemd_run: logger.warning("Cannot isolate fresh gateway recovery from the gateway cgroup") return _all_failed() - command = [ - systemd_run, - "--user", - "--scope", - "--quiet", - "--collect", - "--", - *command, - ] + command = [systemd_run, "--user", "--scope", "--quiet", "--collect", "--", *command] kwargs = { "input": json.dumps( @@ -318,47 +250,26 @@ def _recover_gateway_restart_after_abort( logger.warning("Fresh gateway restart recovery returned incomplete profiles") return _all_failed() - if verified: - print( - " ✓ Restarted supervised gateway(s) in a fresh process" - " (systemd-verified active): " + ", ".join(sorted(verified)) - ) - if relaunch_attempted: - print( - " ⚠ Relaunch attempted in a fresh process but not" - " supervisor-verified (check these gateways manually): " - + ", ".join(sorted(relaunch_attempted)) - ) - if serve_units["verified"]: - print( - " ✓ Restarted serve unit(s) in a fresh process" - " (new main PID observed): " - + ", ".join(serve_units["verified"]) - ) - if serve_units["failed"]: - print( - " ⚠ Could not verify a replacement for serve unit(s): " - + ", ".join(serve_units["failed"]) - ) - return { - "requested": profiles, - "verified": sorted(verified), - "relaunch_attempted": sorted(relaunch_attempted), - "failed": sorted(failed), - "skipped": skipped, - "serve_units": serve_units, - } + verified, relaunch_attempted, failed = sorted(verified), sorted(relaunch_attempted), sorted(failed) + for names, text in ( + (verified, " ✓ Restarted supervised gateway(s) in a fresh process (systemd-verified active): "), + (relaunch_attempted, " ⚠ Relaunch attempted in a fresh process but not" + " supervisor-verified (check these gateways manually): "), + (serve_units["verified"], " ✓ Restarted serve unit(s) in a fresh process (new main PID observed): "), + (serve_units["failed"], " ⚠ Could not verify a replacement for serve unit(s): "), + ): + if names: + print(text + ", ".join(names)) + return _result(profiles, verified, relaunch_attempted, failed, serve_units) def _warn_stale_serve_runtimes(rows) -> None: """Name the serve/dashboard processes that survived on pre-update code. - The original #92145 report is a user watching every chat turn fail with an - ``ImportError`` for a symbol that imports fine on disk, with nothing in the - terminal naming the responsible process. ``hermes serve`` hosts - ``tui_gateway.server``; when its unit was never restarted it keeps the - pre-pull ``sys.modules`` graph and there is no gateway row anywhere that - reveals it. Print the PIDs and the exact command that fixes it. + ``hermes serve`` hosts ``tui_gateway.server``; if its unit was never restarted it keeps the + pre-pull ``sys.modules`` graph, every chat turn fails with an ``ImportError`` for a symbol + that imports fine on disk, and no gateway row reveals the culprit. Print the PIDs and the + exact command that fixes it. """ if not rows: return @@ -388,28 +299,11 @@ def _abort_recovery_is_complete( ) -> bool: """May a fresh-process recovery clear the incomplete flag? - Only when EVERY inventoried runtime family is accounted for. The gateway - leg alone is not enough (#92145): the post-update read-back - (``collect_fleet_versions``) is gateway-only — it reads each profile's - ``gateway_state.json`` / control socket — so a ``hermes serve`` still - holding the pre-update ``sys.modules`` graph is invisible to both the - recovery pass and the verification that follows it. Clearing the flag on - gateway coverage alone is exactly how an update reported success while - every chat turn kept failing with an ``ImportError`` for a symbol that - imports fine on disk. + Only when EVERY inventoried runtime family is accounted for. - Fail closed on each leg: - - * every planned gateway profile is covered, with nothing failed and - nothing merely ``relaunch_attempted`` (rc 0 is not observed coverage); - * no ``hermes-serve*`` unit failed to produce a verified replacement; and - * no serve/dashboard process from the pre-update inventory is still the - same process (``stale_runtime_rows``). - - ``planned_gateway_profiles`` being empty is deliberately NOT completeness: - with no gateway leg to prove, the caller's own fail-closed contract - (``_restart_phase_failure_is_incomplete``) plus the stale-runtime rows - decide the outcome. + ``planned_gateway_profiles`` being empty is deliberately NOT completeness: with no gateway leg + to prove, the caller's own fail-closed contract (``_restart_phase_failure_is_incomplete``) plus + the stale-runtime rows decide the outcome. """ if not planned_gateway_profiles: return False diff --git a/hermes_cli/update_contract.py b/hermes_cli/update_contract.py index cf6f4e2f75..764b09d2a1 100644 --- a/hermes_cli/update_contract.py +++ b/hermes_cli/update_contract.py @@ -1,21 +1,8 @@ """Image-managed install refusal contract (#91277 Phase 3). -One shared admission gate for every surface that can start an in-place -``hermes update`` mutation (CLI apply, CLI --check, dashboard update -endpoint). The decision layers: - -1. **Baked provenance marker** (``/etc/hermes/image-provenance.json``, - written by the image build — see :mod:`hermes_cli.image_provenance`): - authoritative ground truth that this filesystem came from an immutable - image. Fail-closed: a present-but-malformed marker still refuses. -2. **Filesystem heuristics** (``detect_install_method()``): the pre-existing - docker/nix/apt detection, kept as the fallback for images built before - the marker existed and for package-managed installs that have no image - marker at all. - -A refusal prints the real update command for the deployment kind, records a -``refused`` receipt (so fleet tooling sees "this install cannot self-update, -use " instead of a silent non-update), and exits 2 on CLI surfaces. +A refusal prints the real update command for the deployment kind, records a ``refused`` receipt (so +fleet tooling sees "this install cannot self-update, use " instead of a silent non-update), +and exits 2 on CLI surfaces. """ from __future__ import annotations @@ -40,9 +27,8 @@ class UpdateRefusal: def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]: """Return an :class:`UpdateRefusal` when in-place update must not run. - ``None`` means the install is eligible for in-place update (git checkout - or unknown-but-mutable). Never raises; on any internal error it falls - back to the heuristic layer only. + ``None`` means the install is eligible for in-place update (git checkout or unknown-but- + mutable). Never raises; on any internal error it falls back to the heuristic layer only. """ # Layer 1: baked provenance marker — authoritative when present. try: @@ -116,9 +102,8 @@ def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]: def record_refusal_receipt(refusal: UpdateRefusal) -> None: """Write a minimal ``refused`` receipt for a blocked update attempt. - Gives fleet tooling a durable record that an update was ATTEMPTED and - refused ("not updatable in place, use ") instead of a silent - nothing. Best-effort; never raises. + Gives fleet tooling a durable record that an update was ATTEMPTED and refused ("not updatable in + place, use ") instead of a silent nothing. Best-effort; never raises. """ try: from hermes_cli.update_receipt import ( diff --git a/hermes_cli/update_inventory.py b/hermes_cli/update_inventory.py index 1e2528e86c..5fe8321a6c 100644 --- a/hermes_cli/update_inventory.py +++ b/hermes_cli/update_inventory.py @@ -1,38 +1,19 @@ """Runtime inventory + update plan for the fleet-update pipeline (#91277 Phase 2). -One read-only pass that answers, BEFORE any mutation: what Hermes runtimes -are running on this machine, how is each one deployed, which of them will -this update touch, and how will each be restarted? +One read-only pass that answers, BEFORE any mutation: what Hermes runtimes are running on this +machine, how is each one deployed, which of them will this update touch, and how will each be +restarted? -This is the "plan" phase of the transactional deployment model (#88683): - - plan → snapshot → apply → restart-per-kind → verify → report - -The module is deliberately side-effect free — every collector is a probe -over primitives that already exist (`find_profile_gateway_processes`, -`_get_service_pids`, `gateway_state.json` code stamps from #91283, -`detect_install_method`) — so `hermes update --plan` can run on a live -fleet with zero risk, and the update receipt can embed the inventory -without changing update behavior. - -Deployment kinds (the concept most fleet-update bugs were missing): - - git — source checkout; updatable in place via `hermes update` - docker — published image; NOT updatable in place (pull + recreate) - nix/apt — package-manager owned; updatable via the manager only - unknown — no marker; treated as in-place updatable (legacy default) - -Supervisors (how a runtime is restarted after code changes): - - systemd / launchd — restart via the service manager (fleet-wide) - desktop — Desktop app supervises `hermes serve`; it respawns - manual — plain process; SIGTERM + watcher/manual relaunch +The module is deliberately side-effect free — every collector is a probe over primitives that +already exist (`find_profile_gateway_processes`, `_get_service_pids`, `gateway_state.json` code +stamps from #91283, `detect_install_method`) — so `hermes update --plan` can run on a live fleet +with zero risk, and the update receipt can embed the inventory without changing update behavior. """ from __future__ import annotations import logging -import os +from contextlib import contextmanager from dataclasses import dataclass, field, asdict from pathlib import Path from typing import Any, Optional @@ -70,17 +51,10 @@ class UpdatePlan: runtimes: list = field(default_factory=list) # list[RuntimeRecord] def to_dict(self) -> dict[str, Any]: - payload = asdict(self) - payload["runtimes"] = [ - r.to_dict() if isinstance(r, RuntimeRecord) else r - for r in self.runtimes - ] - return payload + return asdict(self) # recursive: RuntimeRecord entries become dicts, dict entries are copied -def _detect_supervisor_for_pid( - pid: int, service_pids: set, windows_service_pids: set | None = None -) -> str: +def _detect_supervisor_for_pid(pid: int, service_pids: set, windows_service_pids: set | None = None) -> str: """Classify how a live gateway PID is supervised.""" if windows_service_pids and pid in windows_service_pids: # SCM-supervised Windows gateway (WinSW/NSSM/sc.exe create): the @@ -102,56 +76,99 @@ def _detect_supervisor_for_pid( return "manual" +_RESTART_MECHANISMS = { + "systemd": "systemd", + "launchd": "launchd", + "desktop": "desktop", + "windows-service": "windows-service", + "manual-serve": "respawn-argv", +} + +_MECHANISM_DESCRIPTIONS = { + "systemd": "systemctl restart (drain-first SIGUSR1 when supported)", + "launchd": "launchctl kickstart -k (drain-first, per-label domain)", + "desktop": "Desktop app respawns its serve backend", + "windows-service": "sc.exe stop before venv mutation, sc.exe start after update", + "respawn-argv": "stop before code swap, relaunch with recorded launch args", +} + + def _restart_mechanism(supervisor: str, profile: str) -> str: """Machine-readable restart mechanism id for a runtime. - THE policy table (#91277 Phase 2): restart execution consumes these ids - via :func:`match_runtime_outcomes` / the update's restart phase, and the - receipt records per-runtime outcomes against them. Display strings are - derived by :func:`describe_restart_mechanism` — never the other way - around. + THE policy table (#91277 Phase 2): restart execution consumes these ids via + :func:`match_runtime_outcomes` / the update's restart phase, and the receipt records per-runtime + outcomes against them. Display strings are derived by :func:`describe_restart_mechanism` — never + the other way around. """ - if supervisor == "systemd": - return "systemd" - if supervisor == "launchd": - return "launchd" - if supervisor == "desktop": - return "desktop" - if supervisor == "windows-service": - return "windows-service" - if supervisor == "manual-serve": - return "respawn-argv" - return "manual" + return _RESTART_MECHANISMS.get(supervisor, "manual") def describe_restart_mechanism(mechanism: str, profile: str) -> str: """Human-readable description of a restart mechanism id.""" - if mechanism == "systemd": - return "systemctl restart (drain-first SIGUSR1 when supported)" - if mechanism == "launchd": - return "launchctl kickstart -k (drain-first, per-label domain)" - if mechanism == "desktop": - return "Desktop app respawns its serve backend" - if mechanism == "windows-service": - return "sc.exe stop before venv mutation, sc.exe start after update" - if mechanism == "respawn-argv": - return "stop before code swap, relaunch with recorded launch args" + described = _MECHANISM_DESCRIPTIONS.get(mechanism) + if described is not None: + return described if profile != "default": return f"hermes -p {profile} gateway restart" return "hermes gateway restart" +def _runtime(kind: str, profile: str, pid: Optional[int], supervisor: str, **extra: Any) -> RuntimeRecord: + """A :class:`RuntimeRecord` with ``restart_via`` derived from its supervisor.""" + return RuntimeRecord( + kind=kind, + profile=profile, + pid=pid, + supervisor=supervisor, + restart_via=_restart_mechanism(supervisor, profile), + **extra, + ) + + +def _int_or_none(value: Any) -> Optional[int]: + try: + return int(value) + except (TypeError, ValueError): + return None + + +def _gateway_record( + profile: str, + pid: int, + supervisor: str, + code_sha: Any = None, + code_version: Any = None, +) -> RuntimeRecord: + return _runtime( + "gateway", + profile, + pid, + supervisor, + code_sha=str(code_sha) if code_sha else None, + code_version=code_version, + ) + + +@contextmanager +def _probe(label: str): + """Run one inventory collector; a failure is logged at debug and yields fewer rows, never an exception.""" + try: + yield + except Exception as exc: + logger.debug("%s failed: %s", label, exc) + + def collect_runtime_inventory() -> UpdatePlan: """Build the pre-update plan. Read-only; never raises. - Every collector degrades independently — a probe failure yields fewer - rows, not an exception. The result is embeddable in the update receipt - and printable via :func:`print_update_plan`. + Every collector degrades independently — a probe failure yields fewer rows, not an exception. + The result is embeddable in the update receipt and printable via :func:`print_update_plan`. """ plan = UpdatePlan() # --- install shape / deployment kind --------------------------------- - try: + with _probe("Install-method probe"): from hermes_cli.config import ( detect_install_method, get_managed_system, @@ -169,7 +186,7 @@ def collect_runtime_inventory() -> UpdatePlan: # container can look like `git` to the heuristics while the running # filesystem is actually an immutable image. Fail-closed: an invalid # marker still flips the plan to not-updatable. - try: + with _probe("Image provenance probe"): from hermes_cli.image_provenance import read_image_provenance provenance = read_image_provenance() @@ -177,25 +194,19 @@ def collect_runtime_inventory() -> UpdatePlan: plan.updatable_in_place = False if provenance.valid and provenance.manager: plan.install_method = provenance.manager - except Exception as exc: - logger.debug("Image provenance probe failed: %s", exc) plan.update_mechanism = recommended_update_command_for_method(method) - except Exception as exc: - logger.debug("Install-method probe failed: %s", exc) # --- expected code identity (pre-pull) -------------------------------- - try: + with _probe("Code-identity probe"): from hermes_cli.build_info import get_code_identity identity = get_code_identity(refresh=True) plan.expected_sha = identity.get("sha") plan.expected_version = identity.get("version") - except Exception as exc: - logger.debug("Code-identity probe failed: %s", exc) # --- profiles ---------------------------------------------------------- profile_homes: list[tuple[str, Path]] = [] - try: + with _probe("Profile enumeration"): from hermes_cli.profiles import ( _get_default_hermes_home, _get_profiles_root, @@ -208,24 +219,16 @@ def collect_runtime_inventory() -> UpdatePlan: root = _get_profiles_root() if root.is_dir(): for entry in sorted(root.iterdir()): - if ( - entry.is_dir() - and entry.name != "default" - and _PROFILE_ID_RE.match(entry.name) - ): + if entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name): profile_homes.append((entry.name, entry)) plan.profiles = [name for name, _ in profile_homes] - except Exception as exc: - logger.debug("Profile enumeration failed: %s", exc) # --- service-managed PIDs (fleet-wide) --------------------------------- service_pids: set = set() - try: + with _probe("Service-PID probe"): from hermes_cli.gateway import _get_service_pids service_pids = _get_service_pids(all_profiles=True) or set() - except Exception as exc: - logger.debug("Service-PID probe failed: %s", exc) # --- SCM-supervised gateway PIDs (Windows) ------------------------------ # find_windows_gateway_services() maps validated gateway PIDs through @@ -234,19 +237,14 @@ def collect_runtime_inventory() -> UpdatePlan: # `sc.exe start`, so the plan must carry the matching mechanism id for # the #91277 Phase 2 reconciliation and the fleet check. windows_service_pids: set = set() - try: + with _probe("Windows SCM service-ownership probe"): from hermes_cli.gateway import find_windows_gateway_services - windows_service_pids = { - int(service.gateway_pid) - for service in find_windows_gateway_services() - } - except Exception as exc: - logger.debug("Windows SCM service-ownership probe failed: %s", exc) + windows_service_pids = {int(service.gateway_pid) for service in find_windows_gateway_services()} # --- per-profile gateways (PID files + runtime status stamps) ---------- seen_pids: set[int] = set() - try: + with _probe("Gateway-state inventory"): from gateway.status import _pid_exists, read_runtime_status for profile, home in profile_homes: @@ -260,91 +258,54 @@ def collect_runtime_inventory() -> UpdatePlan: identity = identify_gateway(home) except Exception: identity = None - if identity: - try: - sock_pid = int(identity.get("pid")) - except (TypeError, ValueError): - sock_pid = None - if sock_pid is not None: - if sock_pid in seen_pids: - # One multiplex gateway can answer identify for - # several profile homes — one runtime record per - # process, not per home. - continue - seen_pids.add(sock_pid) - declared = identity.get("supervisor") - supervisor = ( - str(declared) - if declared - else _detect_supervisor_for_pid( - sock_pid, service_pids, windows_service_pids - ) - ) - sock_sha = identity.get("code_sha") - plan.runtimes.append( - RuntimeRecord( - kind="gateway", - profile=profile, - pid=sock_pid, - supervisor=supervisor, - code_sha=str(sock_sha) if sock_sha else None, - code_version=identity.get("code_version"), - restart_via=_restart_mechanism(supervisor, profile), - ) - ) + sock_pid = _int_or_none(identity.get("pid")) if identity else None + if sock_pid is not None: + if sock_pid in seen_pids: + # One multiplex gateway can answer identify for + # several profile homes — one runtime record per + # process, not per home. continue - record = read_runtime_status(home / "gateway_state.json") - pid: Optional[int] = None - code_sha = code_version = None - if record: - try: - pid = int(record.get("pid")) - except (TypeError, ValueError): - pid = None - code_sha = record.get("code_sha") - code_version = record.get("code_version") + seen_pids.add(sock_pid) + declared = identity.get("supervisor") + supervisor = ( + str(declared) + if declared + else _detect_supervisor_for_pid(sock_pid, service_pids, windows_service_pids) + ) + plan.runtimes.append( + _gateway_record( + profile, sock_pid, supervisor, identity.get("code_sha"), identity.get("code_version") + ) + ) + continue + record = read_runtime_status(home / "gateway_state.json") or {} + pid = _int_or_none(record.get("pid")) if pid is None or not _pid_exists(pid): continue seen_pids.add(pid) - supervisor = _detect_supervisor_for_pid( - pid, service_pids, windows_service_pids - ) plan.runtimes.append( - RuntimeRecord( - kind="gateway", - profile=profile, - pid=pid, - supervisor=supervisor, - code_sha=str(code_sha) if code_sha else None, - code_version=code_version, - restart_via=_restart_mechanism(supervisor, profile), + _gateway_record( + profile, + pid, + _detect_supervisor_for_pid(pid, service_pids, windows_service_pids), + record.get("code_sha"), + record.get("code_version"), ) ) - except Exception as exc: - logger.debug("Gateway-state inventory failed: %s", exc) # PID-file mapped gateways not covered by a runtime-status record - try: + with _probe("PID-file gateway inventory"): from hermes_cli.gateway import find_profile_gateway_processes for proc in find_profile_gateway_processes(): if proc.pid in seen_pids: continue seen_pids.add(proc.pid) - supervisor = _detect_supervisor_for_pid( - proc.pid, service_pids, windows_service_pids - ) plan.runtimes.append( - RuntimeRecord( - kind="gateway", - profile=proc.profile, - pid=proc.pid, - supervisor=supervisor, - restart_via=_restart_mechanism(supervisor, proc.profile), + _gateway_record( + proc.profile, proc.pid, _detect_supervisor_for_pid(proc.pid, service_pids, windows_service_pids) ) ) - except Exception as exc: - logger.debug("PID-file gateway inventory failed: %s", exc) # Serve/dashboard backends from the spawn ledger (#63206). These are the # runtimes the gateway collectors above can never see: a manually @@ -355,7 +316,7 @@ def collect_runtime_inventory() -> UpdatePlan: # fabricates a row. Desktop-supervised backends are classified by their # recorded spawner still being alive — those restart via the Desktop's # own respawn, not ours. - try: + with _probe("Serve/dashboard ledger inventory"): from hermes_cli.process_identity import ledger_entries, spawner_is_dead for entry in ledger_entries(): @@ -366,16 +327,12 @@ def collect_runtime_inventory() -> UpdatePlan: if not isinstance(pid, int) or pid in seen_pids: continue seen_pids.add(pid) - has_live_spawner = spawner_is_dead(entry) is False - supervisor = "desktop" if has_live_spawner else "manual-serve" - profile = str(entry.get("profile") or "default") plan.runtimes.append( - RuntimeRecord( - kind=str(purpose), - profile=profile, - pid=pid, - supervisor=supervisor, - restart_via=_restart_mechanism(supervisor, profile), + _runtime( + str(purpose), + str(entry.get("profile") or "default"), + pid, + "desktop" if spawner_is_dead(entry) is False else "manual-serve", detail={ "argv": entry.get("argv") or "", "host": entry.get("host") or "", @@ -388,8 +345,6 @@ def collect_runtime_inventory() -> UpdatePlan: }, ) ) - except Exception as exc: - logger.debug("Serve/dashboard ledger inventory failed: %s", exc) return plan @@ -415,14 +370,8 @@ def print_update_plan(plan: UpdatePlan) -> None: print(f" Running services to restart ({len(plan.runtimes)}):") for runtime in plan.runtimes: sha = f" @ {runtime.code_sha[:8]}" if runtime.code_sha else "" - print( - f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid}" - f" — {runtime.supervisor}{sha}" - ) - print( - " restart: " - f"{describe_restart_mechanism(runtime.restart_via, runtime.profile)}" - ) + print(f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid} — {runtime.supervisor}{sha}") + print(f" restart: {describe_restart_mechanism(runtime.restart_via, runtime.profile)}") _SERVE_KINDS = ("serve", "dashboard") @@ -431,11 +380,8 @@ _SERVE_KINDS = ("serve", "dashboard") def _serve_unit_matches_profile(profile: str, unit: object) -> bool: """Does *unit* name a ``hermes-serve*``/``hermes-dashboard*`` unit for *profile*? - Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's - ``hermes-gateway*`` names never cover them (#100479). Exact names only — - ``work`` must not claim ``hermes-serve-workbench`` — and a scope prefix - (``user/hermes-serve``) is tolerated because the restart phase records - scope-qualified identities in some lists. + Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's ``hermes-gateway*`` names + never cover them (#100479). """ name = str(unit).removesuffix(".service") if "/" in name: @@ -480,32 +426,13 @@ def match_runtime_outcomes( ) -> list[dict[str, Any]]: """Reconcile the plan's runtimes against what the restart phase DID. - #91277 Phase 2 (restart via declared mechanism): the platform restart - branches each re-discover their own targets, so a runtime the plan saw - can be missed entirely with no signal. This cross-checks every planned - runtime against the phase's bookkeeping and returns one outcome row per - runtime:: - - {"kind", "profile", "pid", "mechanism", "outcome"} - - outcome: ``restarted`` (service restarted / profile relaunched / - handed to external supervisor), ``stopped`` (pid killed, watcher or - operator relaunches), ``failed`` (in the phase's failed/stale list) or - ``unaccounted`` — the plan saw it and NO bookkeeping mentions it: the - blind-spot tripwire (same philosophy as the fleet matrix's DOWN row). - Never raises; on any probe error returns what it has. - - Serve/dashboard runtimes are reconciled in their OWN vocabulary - (#100479): a ``hermes-serve*``/``hermes-dashboard*`` unit, a killed - PID, or — when the caller passes ``stale_serve_pids`` (the - ``(pid, create_time)``-verified survivor probe, - :func:`hermes_cli.update_abort_recovery._surviving_pre_update_serve_runtimes`) - — liveness: a pre-update serve whose incarnation is gone was replaced - (unit restart, dashboard cleanup respawn, Desktop respawn) and counts as - ``restarted``; one still alive is ``unaccounted``. They never borrow the - gateway's outcome: ``relaunched_profiles`` and ``hermes-gateway*`` name a - different process that shares the profile, nothing more. Without the - probe result, an untouched serve stays ``unaccounted`` (fail closed). + The platform restart branches each re-discover their own targets, so a runtime the plan saw can + be missed with no signal. Returns one ``{kind, profile, pid, mechanism, outcome}`` row per + planned runtime; outcome is ``restarted``, ``stopped``, ``failed`` or ``unaccounted`` (no + bookkeeping mentions it — the blind-spot tripwire). Never raises. Serve/dashboard runtimes are + reconciled in their OWN vocabulary and never borrow the gateway's outcome: with + ``stale_serve_pids`` a pre-update serve whose incarnation is gone counts as ``restarted``, one + still alive is ``unaccounted``; without the probe an untouched serve stays ``unaccounted``. """ outcomes: list[dict[str, Any]] = [] try: @@ -514,69 +441,51 @@ def match_runtime_outcomes( relaunched = set(relaunched_profiles or []) external = set(externally_supervised_profiles or []) killed = {int(p) for p in (killed_pids or set())} - stale_serves = ( - {int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None - ) + stale_serves = {int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None - for runtime in plan.runtimes: - r = runtime if isinstance(runtime, RuntimeRecord) else None - if r is None: - continue - if r.kind in _SERVE_KINDS: - outcomes.append( - { - "kind": r.kind, - "profile": r.profile, - "pid": r.pid, - "mechanism": r.restart_via, - "outcome": _serve_runtime_outcome( - r, - killed=killed, - failed_set=failed_set, - restarted_set=restarted_set, - stale_serves=stale_serves, - ), - } - ) - continue - outcome = "unaccounted" + def _gateway_names(r: RuntimeRecord, names: set) -> bool: # The bare "hermes-gateway" unit name is gateway-specific: a # serve/dashboard runtime that merely shares the default # profile is a different process the gateway restart never # touched, and must not borrow its outcome (#100479). - if r.profile in relaunched or r.profile in external: + return any( + r.profile in name + or ( + r.kind == "gateway" + and r.profile == "default" + and "hermes-gateway" in name + ) + for name in names + ) + + for r in plan.runtimes: + if not isinstance(r, RuntimeRecord): + continue + if r.kind in _SERVE_KINDS: + outcome = _serve_runtime_outcome( + r, + killed=killed, + failed_set=failed_set, + restarted_set=restarted_set, + stale_serves=stale_serves, + ) + elif r.profile in relaunched or r.profile in external: outcome = "restarted" elif r.pid is not None and r.pid in killed: outcome = "stopped" - elif any( - r.profile in unit - or ( - r.kind == "gateway" - and r.profile == "default" - and "hermes-gateway" in unit - ) - for unit in failed_set - ): + elif _gateway_names(r, failed_set): outcome = "failed" - elif any( - r.profile in svc - or ( - r.kind == "gateway" - and r.profile == "default" - and "hermes-gateway" in svc - ) - for svc in restarted_set - ): + elif _gateway_names(r, restarted_set): outcome = "restarted" - outcomes.append( - { - "kind": r.kind, - "profile": r.profile, - "pid": r.pid, - "mechanism": r.restart_via, - "outcome": outcome, - } - ) + else: + outcome = "unaccounted" + outcomes.append({ + "kind": r.kind, + "profile": r.profile, + "pid": r.pid, + "mechanism": r.restart_via, + "outcome": outcome, + }) except Exception as exc: logger.debug("Runtime-outcome reconciliation failed: %s", exc) return outcomes @@ -585,9 +494,8 @@ def match_runtime_outcomes( def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool: """Print a loud warning for runtimes the restart phase never touched. - Returns True when at least one planned runtime is unaccounted — the - caller escalates exactly like a STALE/DOWN fleet row (exit 1): a runtime - the plan promised to restart, silently missed, is the class this phase + Returns True when at least one planned runtime is unaccounted; the caller escalates like a + STALE/DOWN fleet row (exit 1) — a promised restart silently missed is the class this phase exists to kill. """ missed = [o for o in outcomes if o.get("outcome") == "unaccounted"] @@ -596,10 +504,7 @@ def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool: print() print(" ⚠ Planned runtimes the restart phase never touched:") for o in missed: - print( - f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']}" - f" — planned mechanism: {o['mechanism']}" - ) + print(f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']} — planned mechanism: {o['mechanism']}") print(" Restart them manually, then verify:") if any(o.get("kind") not in _SERVE_KINDS for o in missed): print(" hermes gateway restart # active profile") diff --git a/hermes_cli/update_lock.py b/hermes_cli/update_lock.py index 964638df05..fbc43063a0 100644 --- a/hermes_cli/update_lock.py +++ b/hermes_cli/update_lock.py @@ -1,51 +1,12 @@ """Cross-process mutual exclusion for in-flight Hermes updates. -Three different surfaces can start an update of the same install tree: +Until now only the Tauri updater published an "update in progress" marker (``UpdateMarkerGuard`` in +``apps/bootstrap-installer/src-tauri/src/update.rs``), and only the Electron desktop consumed it +(``electron/update-marker.ts``, to gate local backend startup). -* ``hermes update`` from a terminal, -* the dashboard's Update button (``POST /api/hermes/update`` → - ``_spawn_hermes_action(["update"])``, detached), -* the desktop's Update button, which hands off to the Tauri - ``hermes-setup --update`` and, on its failure screen, to install-mode - bootstrap (``install.ps1`` / ``install.sh``). - -Until now only the Tauri updater published an "update in progress" marker -(``UpdateMarkerGuard`` in ``apps/bootstrap-installer/src-tauri/src/update.rs``), -and only the Electron desktop consumed it (``electron/update-marker.ts``, to -gate local backend startup). Nothing stopped two *updaters* from running at -once — so a dashboard-spawned ``hermes update`` and an installer-driven -``git checkout`` could mutate the same checkout concurrently, rewriting source -under a live interpreter and leaving the tree half-updated. - -This module makes that same marker the single lock for **all** update -entrypoints instead of adding a fourth mechanism. Format and location are -unchanged and remain byte-compatible with the Rust and Electron readers: - - /.hermes-update-in-progress body: "\\n" - -A marker only counts as a live update when its pid is alive AND it is younger -than :data:`UPDATE_MARKER_MAX_AGE_MS` — mirroring ``readLiveUpdateMarker`` so a -crashed updater self-heals instead of wedging every future update. A stale -marker is removed on read by whoever notices it first. - -One layering wrinkle: the Tauri updater holds this marker for its WHOLE run and -then spawns ``hermes update`` as a child stage. Without a handoff the child -sees its own parent's live marker and refuses — the GUI update deadlocks -against itself on every attempt ("Hermes is still running", retry forever). -Two mechanisms recognize the orchestrating parent, and either suffices: - -* The updater exports :data:`HANDOFF_PID_ENV` naming its own pid, and - ``acquire`` treats a live holder matching that pid as the lock we are - already running under. The env var alone grants nothing: the pid must also - be the live marker owner, so a stale or forged value cannot bypass the lock. -* A live holder that is a *process ancestor* of ours is likewise our own - orchestrator. This is the load-bearing path for the fleet: the staged - ``hermes-setup`` binary under ``~/.hermes`` is only refreshed by a full - installer run (``copy_self_to_hermes_home`` deliberately no-ops during - ``--update``), so every desktop whose staged updater predates the - HANDOFF_PID_ENV export runs an old parent against a new child. Without the - ancestry check those users get exit 2 ("Hermes is still running") on every - GUI update forever, with no Hermes process actually running. +This module makes that same marker the single lock for **all** update entrypoints instead of adding +a fourth mechanism. Format and location are unchanged and remain byte-compatible with the Rust and +Electron readers: """ from __future__ import annotations @@ -85,10 +46,10 @@ UPDATE_EXIT_CONCURRENT = 2 def update_marker_path() -> Path: """Path of the shared update marker. - Uses the *process* Hermes home (never the context-local profile override): - the Rust updater resolves ``$HERMES_HOME`` or the platform default, and the - desktop pins that same value into the updater's env. A profile-scoped path - here would put the lock somewhere the other two owners never look. + Uses the *process* Hermes home (never the context-local profile override): the Rust updater + resolves ``$HERMES_HOME`` or the platform default, and the desktop pins that same value into the + updater's env. A profile-scoped path here would put the lock somewhere the other two owners + never look. """ from hermes_constants import get_process_hermes_home @@ -98,15 +59,12 @@ def update_marker_path() -> Path: def _pid_alive(pid: int) -> bool: """True when a process with ``pid`` currently exists. - Delegates to :func:`gateway.status._pid_exists`, the project's existing - no-kill probe. Do NOT hand-roll this with ``os.kill(pid, 0)``: on Windows - that is not a no-op — CPython routes ``sig=0`` to - ``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console - process group (bpo-14484). A liveness check that killed the updater it was - asking about would be a spectacular way to fix a concurrency bug. + Delegates to :func:`gateway.status._pid_exists`, the project's existing no-kill probe. Do NOT + hand-roll this with ``os.kill(pid, 0)``: on Windows that is not a no-op — CPython routes + ``sig=0`` to ``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console process + group (bpo-14484). - Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge - the lock forever. + Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge the lock forever. """ if pid <= 0: return False @@ -124,8 +82,8 @@ def _pid_alive(pid: int) -> bool: def _handoff_pid() -> int | None: """Pid of the orchestrating updater that spawned us, if any. - Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent — - a broken handoff must fall back to the normal refusal, never crash. + Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent — a broken handoff must fall + back to the normal refusal, never crash. """ raw = os.environ.get(HANDOFF_PID_ENV, "").strip() if not raw: @@ -140,14 +98,12 @@ def _handoff_pid() -> int | None: def _is_ancestor_pid(pid: int) -> bool: """True when ``pid`` is a live ancestor (parent chain) of this process. - The orchestrating updater spawns ``hermes update`` as a (grand)child, so a - live marker owned by one of our ancestors can only be the claim we are - already running under — an unrelated concurrent updater is never in our - parent chain. This heals the fleet of staged ``hermes-setup`` binaries - that predate the HANDOFF_PID_ENV export and can never send it. + The orchestrating updater spawns ``hermes update`` as a (grand)child, so a live marker owned by + one of our ancestors can only be the claim we are already running under — an unrelated + concurrent updater is never in our parent chain. - Never includes our own pid, and any failure counts as "not an ancestor": - an unprovable ancestry must fall back to the normal refusal. + Never includes our own pid, and any failure counts as "not an ancestor": an unprovable ancestry + must fall back to the normal refusal. """ if pid <= 0: return False @@ -171,10 +127,9 @@ class UpdateHolder: def read_live_update(*, path: Path | None = None) -> UpdateHolder | None: """Return the live update holding the lock, or ``None``. - Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent, - unreadable, malformed, dead-pid, and past-the-ceiling all mean "no live - update", and a stale marker file is deleted so it can't strand future runs. - Never raises. + Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent, unreadable, + malformed, dead-pid, and past-the-ceiling all mean "no live update", and a stale marker file is + deleted so it can't strand future runs. Never raises. """ marker = path or update_marker_path() try: @@ -220,11 +175,10 @@ def describe_holder(holder: UpdateHolder) -> str: class UpdateLock: """Context manager owning the shared update marker for this process. - ``acquired`` is False when another live update already holds it — callers - decide whether that's a hard refusal (CLI/dashboard) or a wait. Releasing - only removes the marker when *we* still own it, so a marker rewritten by a - handoff partner (the Tauri updater overwrites it with its own pid) is never - deleted out from under its new owner. + ``acquired`` is False when another live update holds it; callers decide between hard refusal + (CLI/dashboard) and waiting. Release only removes the marker when *we* still own it, so a marker + rewritten by a handoff partner (the Tauri updater writes its own pid) is never deleted from + under its new owner. """ def __init__(self, *, path: Path | None = None) -> None: @@ -235,12 +189,10 @@ class UpdateLock: def acquire(self) -> bool: """Claim the lock. Returns False (and sets ``holder``) if it's taken. - A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is a - process ancestor of ours — is our own orchestrating parent (the Tauri - updater spawning `hermes update` as a stage): we run under ITS claim - rather than refusing or re-writing the marker, and ``release`` leaves - the parent's marker untouched. The ancestry path exists because staged - updaters older than the HANDOFF_PID_ENV export never send the env var. + A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is an ancestor of ours — is our + own orchestrating parent (e.g. the Tauri updater staging ``hermes update``): run under ITS + claim and leave its marker untouched on release. The ancestry path covers staged updaters + older than the env-var export. """ existing = read_live_update(path=self.path) if existing is not None: diff --git a/hermes_cli/update_receipt.py b/hermes_cli/update_receipt.py index 11d8d21840..e8e81bfded 100644 --- a/hermes_cli/update_receipt.py +++ b/hermes_cli/update_receipt.py @@ -1,29 +1,10 @@ """Structured update receipts + post-update fleet version verification. -Phase 1 of the fleet-update reliability plan (#91277): the updater must -*prove* its outcome instead of assuming it. +Phase 1 of the fleet-update reliability plan (#91277): the updater must *prove* its outcome instead +of assuming it. -Two additive capabilities, both designed so a failure inside them can never -break an update (every public entry point is exception-swallowing): - -1. **Update receipt** — a machine-readable JSON record of what one - ``hermes update`` run discovered, did, skipped (and why), written to - ``/logs/update_receipts/``. Silent-failure classes this - makes visible: #88848 (helper died after "success" printed), #74973 - (restart silently skipped), #85753 (restart phase never ran), #81193 - (desktop shows failure for a successful update). - -2. **Fleet version verification** — after the restart phase, read every - profile's ``gateway_state.json``, compare each live gateway's stamped - ``code_sha`` (written by ``gateway/status.py`` on every runtime-status - write) against the freshly-updated checkout's HEAD, and print a fleet - version matrix. Mixed-version fleets (#88654, #69754, #77553, #56717) - become a loud, actionable report instead of a latent state. - -Deployment-kind awareness (docker/image-managed installs) rides on -``hermes_cli.build_info.get_code_identity()``: an image build reports -``source="build-file"`` and the receipt records that the install is not -in-place updatable. +Two additive capabilities, both designed so a failure inside them can never break an update (every +public entry point is exception-swallowing): """ from __future__ import annotations @@ -53,6 +34,18 @@ def _utc_now_iso() -> str: return datetime.now(timezone.utc).isoformat() +def _str_records(entries: Any, keys: tuple[str, ...], *, pid: bool = False) -> list[dict[str, Any]]: + """Dict entries reduced to stringified ``keys`` (plus an int ``pid`` first when requested).""" + records = [] + for entry in entries: + if not isinstance(entry, dict): + continue + record: dict[str, Any] = {"pid": int(entry.get("pid", 0) or 0)} if pid else {} + record.update({key: str(entry.get(key, "")) for key in keys}) + records.append(record) + return records + + class UpdateReceipt: """Collects the observable facts of one ``hermes update`` run.""" @@ -122,16 +115,10 @@ class UpdateReceipt: key: [str(profile) for profile in fresh_recovery.get(key, [])] for key in ("requested", "verified", "relaunch_attempted", "failed") } - persisted["skipped"] = [ - { - "profile": str(entry.get("profile", "")), - "kind": str(entry.get("kind", "")), - "supervisor": str(entry.get("supervisor", "")), - "reason": str(entry.get("reason", "")), - } - for entry in fresh_recovery.get("skipped", []) - if isinstance(entry, dict) - ] + persisted["skipped"] = _str_records( + fresh_recovery.get("skipped", []), + ("profile", "kind", "supervisor", "reason"), + ) # Serve/dashboard coverage (#92145). ``hermes serve`` hosts # tui_gateway and is not a gateway profile, so neither the # per-profile buckets above nor the fleet-version matrix can @@ -143,16 +130,11 @@ class UpdateReceipt: key: [str(unit) for unit in (serve_units.get(key) or [])] for key in ("verified", "failed") } - persisted["stale_runtimes"] = [ - { - "pid": int(entry.get("pid", 0) or 0), - "kind": str(entry.get("kind", "")), - "profile": str(entry.get("profile", "")), - "supervisor": str(entry.get("supervisor", "")), - } - for entry in fresh_recovery.get("stale_runtimes", []) - if isinstance(entry, dict) - ] + persisted["stale_runtimes"] = _str_records( + fresh_recovery.get("stale_runtimes", []), + ("kind", "profile", "supervisor"), + pid=True, + ) result["fresh_recovery"] = persisted self.data["gateway_restart"] = result @@ -183,31 +165,28 @@ def begin_update_receipt() -> None: _current = None -def record_step(name: str, ok: bool, detail: str = "") -> None: - """Record one update step outcome. No-op when no receipt is active.""" +def _record(method: str, what: str, *args: Any, **kwargs: Any) -> None: + """Invoke ``method`` on the active receipt; no-op when none, never raises.""" try: if _current is not None: - _current.step(name, ok, detail) + getattr(_current, method)(*args, **kwargs) except Exception as exc: # pragma: no cover - defensive - logger.debug("Could not record update step %s: %s", name, exc) + logger.debug("Could not record %s: %s", what, exc) + + +def record_step(name: str, ok: bool, detail: str = "") -> None: + """Record one update step outcome. No-op when no receipt is active.""" + _record("step", f"update step {name}", name, ok, detail) def record_skip(name: str, reason: str) -> None: """Record a skipped step WITH the reason it was skipped.""" - try: - if _current is not None: - _current.skip(name, reason) - except Exception as exc: # pragma: no cover - defensive - logger.debug("Could not record update skip %s: %s", name, exc) + _record("skip", f"update skip {name}", name, reason) def record_gateway_restart(**kwargs: Any) -> None: """Record the gateway restart phase outcome (see UpdateReceipt).""" - try: - if _current is not None: - _current.gateway_restart_result(**kwargs) - except Exception as exc: # pragma: no cover - defensive - logger.debug("Could not record gateway restart result: %s", exc) + _record("gateway_restart_result", "gateway restart result", **kwargs) def finalize_update_receipt( @@ -215,10 +194,9 @@ def finalize_update_receipt( ) -> Optional[Path]: """Finalize + persist the receipt. Returns the written path or None. - ``outcome`` is one of ``success`` / ``partial`` / ``failed`` / - ``refused``. Exactly-once by construction: the module singleton is - popped first, so a second call (e.g. the command-boundary safety net - after an inner path already finalized) is a no-op returning None. + ``outcome`` is one of ``success`` / ``partial`` / ``failed`` / ``refused``. Exactly-once by + construction: the module singleton is popped first, so a second call (e.g. the command-boundary + safety net after an inner path already finalized) is a no-op returning None. """ global _current receipt = _current @@ -235,15 +213,11 @@ def finalize_update_receipt( directory.mkdir(parents=True, exist_ok=True) stamp = time.strftime("%Y%m%d_%H%M%S") path = directory / f"update_{stamp}_{os.getpid()}.json" - path.write_text( - json.dumps(receipt.data, indent=2, default=str), encoding="utf-8" - ) + body = json.dumps(receipt.data, indent=2, default=str) + path.write_text(body, encoding="utf-8") # Stable pointer for the dashboard/desktop: latest receipt. - latest = directory / "latest.json" try: - latest.write_text( - json.dumps(receipt.data, indent=2, default=str), encoding="utf-8" - ) + (directory / "latest.json").write_text(body, encoding="utf-8") except OSError: pass _prune_old_receipts(directory) @@ -258,19 +232,12 @@ def finalize_pending_update_receipt( ) -> Optional[Path]: """Command-boundary safety net: persist a still-open receipt, if any. - ``hermes update`` has many early-termination paths (Windows - concurrent-instance preflight, venv-holder refusal, head-pinned no-op, - fetch failure — all ``sys.exit``) that predate the inner finalize - call sites. Any receipt still open when the update COMMAND unwinds is - finalized here so every post-begin run leaves a record — the - refused/failed runs are exactly the ones a receipt matters most for - (review on #91283). No-op when no receipt is open (the inner paths - already finalized — exactly-once via the popped singleton) or when - recording was never started. Never raises. - - Outcome mapping: exit 0/None → ``success`` (a path that completed - without an explicit inner finalize), exit 2 → ``refused`` (the - updater's preflight-refusal convention), anything else → ``failed``. + ``hermes update`` has many early ``sys.exit`` paths (preflight refusals, venv-holder + refusal, fetch failure) predating the inner finalize calls; any receipt still open when the + command unwinds is finalized here so refused/failed runs — where a receipt matters most — + leave a record. No-op when nothing is open (inner paths finalize exactly-once via the popped + singleton). Never raises. Exit 0/None → ``success``, exit 2 → ``refused`` (preflight + convention), else → ``failed``. """ if _current is None: return None @@ -280,12 +247,11 @@ def finalize_pending_update_receipt( outcome = "refused" else: outcome = "failed" - try: - receipt = _current - if receipt is not None and exit_code is not None: - receipt.data["exit_code"] = int(exit_code) - except Exception: - pass + if exit_code is not None: + try: + _current.data["exit_code"] = int(exit_code) + except Exception: + pass return finalize_update_receipt(outcome, stop_reason=stop_reason) @@ -321,35 +287,31 @@ def read_latest_receipt() -> Optional[dict[str, Any]]: # Fleet version verification # --------------------------------------------------------------------------- +def _sha_state(code_sha: Any, expected_sha: Any) -> str: + if not code_sha or not expected_sha: + return "unknown" + return "current" if str(code_sha) == str(expected_sha) else "stale" + + +def _fleet_row(profile: str, pid: int, code_sha: Any, code_version: Any, expected_sha: Any) -> dict[str, Any]: + return { + "profile": profile, + "pid": pid, + "code_sha": str(code_sha) if code_sha else None, + "code_version": code_version, + "state": _sha_state(code_sha, expected_sha), + } + + def collect_fleet_versions( *, pre_restart_pids: Optional[list[int]] = None ) -> list[dict[str, Any]]: """Snapshot every profile's gateway code identity vs. the current tree. - Returns one entry per profile home that has a ``gateway_state.json`` - describing a gateway that is live — or that SHOULD be live:: - - {"profile": str, "pid": int, "code_sha": str|None, - "code_version": str|None, "state": "current"|"stale"|"unknown"|"down"} - - ``stale`` — gateway stamped a code_sha that differs from the updated - checkout's HEAD (it is still serving pre-update modules). - ``unknown`` — gateway predates the code-identity stamp (started before - this feature landed) or identity could not be resolved. - ``down`` — the gateway was ALIVE when this update started - (``pre_restart_pids``), its runtime status still says - running, but the PID is dead and no successor rewrote the - record: the restart phase stopped it and nothing came - back. Without this row a killed-and-never-replaced gateway - produced NO entry at all and the matrix passed silently - (Phase-1 verification gap, #88848/#74973 class). - - Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — - a stale state file from a long-dead gateway (machine reboot, manual - kill weeks ago) must NOT fail every future update. Callers that don't - have a pre-restart snapshot (``None``/empty) get the historical - behavior: dead PIDs are skipped. - Never raises; a probe failure yields an empty list. + Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a + long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update. + Callers that don't have a pre-restart snapshot (``None``/empty) get the historical behavior: + dead PIDs are skipped. """ # Runtime-status states that mean "this record does not describe a # gateway that should be running now" — no down row for these. @@ -399,23 +361,12 @@ def collect_fleet_versions( except (TypeError, ValueError): pid = None if pid is not None: - code_sha = identity.get("code_sha") - if not code_sha or not expected_sha: - state = "unknown" - elif str(code_sha) == str(expected_sha): - state = "current" - else: - state = "stale" - results.append( - { - "profile": profile, - "pid": pid, - "code_sha": str(code_sha) if code_sha else None, - "code_version": identity.get("code_version"), - "state": state, - "source": "socket", - } + row = _fleet_row( + profile, pid, identity.get("code_sha"), + identity.get("code_version"), expected_sha, ) + row["source"] = "socket" + results.append(row) continue status_path = home / "gateway_state.json" record = read_runtime_status(status_path) @@ -458,21 +409,11 @@ def collect_fleet_versions( } ) continue - code_sha = record.get("code_sha") - if not code_sha or not expected_sha: - state = "unknown" - elif str(code_sha) == str(expected_sha): - state = "current" - else: - state = "stale" results.append( - { - "profile": profile, - "pid": pid, - "code_sha": str(code_sha) if code_sha else None, - "code_version": record.get("code_version"), - "state": state, - } + _fleet_row( + profile, pid, record.get("code_sha"), + record.get("code_version"), expected_sha, + ) ) except Exception as exc: logger.debug("Fleet version probe failed: %s", exc) @@ -482,13 +423,11 @@ def collect_fleet_versions( def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool: """Print the post-update fleet version matrix. - Returns True when at least one gateway is provably stale (still - serving pre-update code) OR provably down (was running, killed by the - restart phase, nothing came back), so the caller can escalate. - ``unknown`` entries are reported but do NOT fail the update: gateways - started before the code-identity stamp existed have no sha to compare, - and failing on them would turn this feature's own rollout into a - false-positive storm. + Returns True when at least one gateway is provably stale (still serving pre-update code) OR + provably down (killed by the restart phase, nothing came back), so the caller can escalate. + ``unknown`` entries are reported but do NOT fail the update: gateways started before the + code-identity stamp existed have no sha to compare, and failing them would be a false- + positive storm. """ if not fleet: return False diff --git a/hermes_cli/update_restart_recovery.py b/hermes_cli/update_restart_recovery.py index ebab536f64..3e001f9640 100644 --- a/hermes_cli/update_restart_recovery.py +++ b/hermes_cli/update_restart_recovery.py @@ -70,17 +70,35 @@ _UNIT_SETTLE_ATTEMPTS = 10 _UNIT_SETTLE_DELAY = 1.0 -def _profile_command(profile: str) -> list[str]: - """Build a parameterized restart command for exactly one profile.""" - return [ - sys.executable, - "-m", - "hermes_cli.main", - "-p", - profile, - "gateway", - "restart", - ] +def _run_quiet( + run: Callable[..., Any], + argv: list[str], + *, + timeout: int, + **extra: Any, +) -> Any | None: + """Run ``argv`` capturing text output; ``None`` when it errors or times out.""" + try: + return run( + argv, + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + check=False, + timeout=timeout, + **extra, + ) + except (OSError, subprocess.TimeoutExpired): + return None + + +def _stdout(result: Any) -> str: + return (getattr(result, "stdout", "") or "").strip() + + +def _succeeded(result: Any) -> bool: + return result is not None and getattr(result, "returncode", 1) == 0 def _child_environment() -> dict[str, str]: @@ -98,16 +116,7 @@ def _run_profile_restart( run: Callable[..., Any], ) -> bool: """Run one profile restart without inheriting the updater's process state.""" - kwargs: dict[str, Any] = { - "stdin": subprocess.DEVNULL, - "capture_output": True, - "text": True, - "encoding": "utf-8", - "errors": "replace", - "check": False, - "timeout": _PROFILE_RESTART_TIMEOUT, - "env": _child_environment(), - } + kwargs: dict[str, Any] = {"stdin": subprocess.DEVNULL, "env": _child_environment()} if os.name == "nt": kwargs["creationflags"] = ( getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0) @@ -115,58 +124,31 @@ def _run_profile_restart( ) else: kwargs["start_new_session"] = True - - try: - result = run(_profile_command(profile), **kwargs) - except (OSError, subprocess.TimeoutExpired): - return False - return getattr(result, "returncode", 1) == 0 + argv = [sys.executable, "-m", "hermes_cli.main", "-p", profile, "gateway", "restart"] + return _succeeded(_run_quiet(run, argv, timeout=_PROFILE_RESTART_TIMEOUT, **kwargs)) def _systemd_unit_candidates(profile: str) -> tuple[str, ...]: """Unit names the existing systemd gateway lifecycle produces per profile.""" if profile == "default": - return ( - "hermes-gateway.service", - "gateway.service", - "gateway-default.service", - ) - return ( - f"hermes-gateway-{profile}.service", - f"gateway-{profile}.service", - ) + return ("hermes-gateway.service", "gateway.service", "gateway-default.service") + return (f"hermes-gateway-{profile}.service", f"gateway-{profile}.service") def _systemd_verified_active(profile: str, *, run: Callable[..., Any]) -> bool: """Return True only when systemd itself reports the profile's unit active. - This is the observation that separates ``verified`` from - ``relaunch_attempted``. Any failure here (no ``systemctl``, probe error, - unit not ``active``) means we could NOT verify — never that the restart - failed. + This is the observation that separates ``verified`` from ``relaunch_attempted``. Any failure + here (no ``systemctl``, probe error, unit not ``active``) means we could NOT verify — never that + the restart failed. """ systemctl = shutil.which("systemctl") if not systemctl: return False - for unit in _systemd_unit_candidates(profile): - try: - result = run( - [systemctl, "--user", "is-active", unit], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=_VERIFY_TIMEOUT, - ) - except (OSError, subprocess.TimeoutExpired): - continue - if ( - getattr(result, "returncode", 1) == 0 - and (getattr(result, "stdout", "") or "").strip() == "active" - ): - return True - return False + return any( + _unit_is_active([systemctl, "--user"], unit, run=run, require_rc0=True) + for unit in _systemd_unit_candidates(profile) + ) def restart_profiles( @@ -177,21 +159,13 @@ def restart_profiles( ) -> dict[str, list[str]]: """Restart the supplied profiles and return per-profile terminal results. - The caller supplies only profiles whose inventory identified a service - supervisor. Manual gateways are intentionally excluded before this module - is called: killing one without a relaunch authority would turn stale code - into an outage. + The caller supplies only profiles whose inventory identified a service supervisor. - A profile only lands in ``verified`` when its supervisor is systemd and - ``systemctl --user is-active`` independently confirms the unit after the - relaunch command succeeded. Every other zero-exit relaunch is reported as - ``relaunch_attempted`` — the code cannot observe supervisor coverage for - those paths and must not claim it. + A profile only lands in ``verified`` when its supervisor is systemd and ``systemctl --user is- + active`` independently confirms the unit after the relaunch command succeeded. """ supervisors = supervisors or {} - normalized = sorted( - {profile for profile in profiles if isinstance(profile, str) and profile} - ) + normalized = sorted({profile for profile in profiles if isinstance(profile, str) and profile}) verified: list[str] = [] relaunch_attempted: list[str] = [] failed: list[str] = [] @@ -199,31 +173,23 @@ def restart_profiles( if not _run_profile_restart(profile, run=run): failed.append(profile) continue - if supervisors.get(profile) == "systemd" and _systemd_verified_active( - profile, run=run - ): + if supervisors.get(profile) == "systemd" and _systemd_verified_active(profile, run=run): verified.append(profile) else: relaunch_attempted.append(profile) - return { - "verified": verified, - "relaunch_attempted": relaunch_attempted, - "failed": failed, - } + return {"verified": verified, "relaunch_attempted": relaunch_attempted, "failed": failed} def _systemctl_scopes() -> list[tuple[str, list[str]]]: """``systemctl`` invocations for the user and system scopes, or nothing. - Mirrors the scope pair the in-process restart phase walks. ``systemctl`` - is resolved through ``shutil.which`` so this module never has to import - any Hermes platform helper — importing the freshly pulled tree is exactly - what aborted the phase that called us. + Mirrors the scope pair the in-process restart phase walks. ``systemctl`` is resolved through + ``shutil.which`` so this module never has to import any Hermes platform helper — importing the + freshly pulled tree is exactly what aborted the phase that called us. - Each scope is returned with its label because ``hermes-serve.service`` in - the user manager and ``hermes-serve.service`` in the system manager are - two different processes. Every identity this module produces or consumes - stays qualified by that label; the bare unit name is never the key. + Each scope is returned with its label because ``hermes-serve.service`` in the user manager and + ``hermes-serve.service`` in the system manager are two different processes. Every identity this + module produces or consumes stays qualified by that label; the bare unit name is never the key. """ systemctl = shutil.which("systemctl") if not systemctl or sys.platform != "linux": @@ -233,83 +199,43 @@ def _systemctl_scopes() -> list[tuple[str, list[str]]]: def _listed_serve_units(scope: list[str], *, run: Callable[..., Any]) -> list[str]: """Serve units systemd knows about in one scope, validated by name.""" - try: - result = run( - scope - + [ - "list-units", - _SERVE_UNIT_PATTERN, - "--plain", - "--no-legend", - "--no-pager", - ], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=_VERIFY_TIMEOUT, - ) - except (OSError, subprocess.TimeoutExpired): + result = _run_quiet( + run, + scope + ["list-units", _SERVE_UNIT_PATTERN, "--plain", "--no-legend", "--no-pager"], + timeout=_VERIFY_TIMEOUT, + ) + if result is None: return [] + # The glob is a systemd pattern, not a name gate: `hermes-serve*` also + # matches the unrelated `hermes-server.service`. Require the exact + # base unit or the hyphenated profile family, same shape as the + # in-process phase's own name gate. units: list[str] = [] for line in (getattr(result, "stdout", "") or "").splitlines(): parts = line.split() - if not parts: - continue - # The glob is a systemd pattern, not a name gate: `hermes-serve*` also - # matches the unrelated `hermes-server.service`. Require the exact - # base unit or the hyphenated profile family, same shape as the - # in-process phase's own name gate. - if _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units: + if parts and _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units: units.append(parts[0]) return units -def _unit_property( - scope: list[str], unit: str, prop: str, *, run: Callable[..., Any] -) -> str | None: - """One ``systemctl show`` property, or ``None`` when it cannot be read.""" - try: - result = run( - scope + ["show", unit, f"--property={prop}", "--value"], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=_VERIFY_TIMEOUT, - ) - except (OSError, subprocess.TimeoutExpired): - return None - if getattr(result, "returncode", 1) != 0: - return None - return (getattr(result, "stdout", "") or "").strip() - - def _unit_main_pid(scope: list[str], unit: str, *, run: Callable[..., Any]) -> int: - """The unit's ``MainPID``; ``0`` when absent or unreadable.""" - raw = _unit_property(scope, unit, "MainPID", run=run) + """The unit's ``MainPID`` via ``systemctl show``; ``0`` when absent or unreadable.""" + result = _run_quiet( + run, scope + ["show", unit, "--property=MainPID", "--value"], timeout=_VERIFY_TIMEOUT + ) try: - return int(raw or 0) + return int(_stdout(result) or 0) if _succeeded(result) else 0 except ValueError: return 0 -def _unit_is_active(scope: list[str], unit: str, *, run: Callable[..., Any]) -> bool: - try: - result = run( - scope + ["is-active", unit], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=_VERIFY_TIMEOUT, - ) - except (OSError, subprocess.TimeoutExpired): +def _unit_is_active( + scope: list[str], unit: str, *, run: Callable[..., Any], require_rc0: bool = False +) -> bool: + result = _run_quiet(run, scope + ["is-active", unit], timeout=_VERIFY_TIMEOUT) + if result is None or (require_rc0 and not _succeeded(result)): return False - return (getattr(result, "stdout", "") or "").strip() == "active" + return _stdout(result) == "active" def _serve_unit_replaced( @@ -322,11 +248,9 @@ def _serve_unit_replaced( ) -> bool: """Did the unit come back on a NEW main process? - ``restart`` returning 0 is not evidence: the whole point of #92145 is that - a live process can keep serving the pre-update generation while every - status command reports success. A changed ``MainPID`` on an ``active`` - unit is the observation that the old interpreter — and its stale - ``sys.modules`` — is gone. + ``restart`` returning 0 is not evidence: a live process can keep serving the pre-update + generation while every status command reports success. A changed ``MainPID`` on an ``active`` + unit is the observation that the old interpreter and its stale ``sys.modules`` are gone. """ for attempt in range(_UNIT_SETTLE_ATTEMPTS): if attempt: @@ -339,9 +263,16 @@ def _serve_unit_replaced( return False -def _qualified(scope_label: str, base: str) -> str: - """The only identity this module reports for a serve unit.""" - return f"{scope_label}/{base}" +def _split_skip_entry(entry: Any) -> tuple[str, str]: + """``(scope_label, base_unit)`` of a skip entry in either the mapping or ``scope/unit`` shape.""" + if isinstance(entry, Mapping): + scope_label = str(entry.get("scope") or "") + unit = str(entry.get("unit") or "") + else: + scope_label, sep, unit = str(entry).partition("/") + if not sep: + scope_label, unit = "", scope_label + return scope_label, unit.removesuffix(".service") def _normalized_skips( @@ -349,26 +280,13 @@ def _normalized_skips( ) -> tuple[set[tuple[str, str]], set[str]]: """Split already-settled units into scope-qualified and legacy entries. - Qualified entries (``{"scope": "user", "unit": "hermes-serve"}`` or the - equivalent ``"user/hermes-serve"`` string) suppress exactly one process. - - A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH - of two same-named processes was settled. It is honoured across both scopes - because that is all the information it contains — the caller in this tree - always sends the qualified shape, and this branch exists only for a payload - written by a pre-update interpreter that had no scope to send. + A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH of two same-named + processes was settled. """ qualified: set[tuple[str, str]] = set() legacy: set[str] = set() for entry in skip_units or (): - if isinstance(entry, Mapping): - scope_label = str(entry.get("scope") or "") - base = str(entry.get("unit") or "").removesuffix(".service") - else: - scope_label, sep, unit = str(entry).partition("/") - if not sep: - scope_label, unit = "", scope_label - base = unit.removesuffix(".service") + scope_label, base = _split_skip_entry(entry) if not base: continue if scope_label in _SCOPE_LABELS: @@ -386,28 +304,9 @@ def restart_serve_units( ) -> dict[str, list[str]]: """Restart every active ``hermes-serve*`` systemd unit from this process. - ``hermes serve`` hosts ``tui_gateway.server`` and is restarted by the - in-process phase alongside the gateway units, but it is not a gateway - profile: no ``gateway restart`` command reaches it. When the phase aborts - part-way — systemd lists ``hermes-gateway.service`` before - ``hermes-serve.service``, so the gateway is typically already done — the - serve unit is the one left holding generation-N modules over a - generation-N+1 checkout. - - Units are enumerated from systemd, never from the update inventory. That - keeps the relaunch authority requirement structural: a manually launched - or Desktop-owned ``hermes serve`` owns no unit and therefore cannot be - touched here. - - Every identity in and out of this function is scope-qualified. The user - manager and the system manager can each own a ``hermes-serve.service``, - and they are different processes: projecting the scope away would let one - already-settled unit suppress recovery of the other, and would let one - scope's success describe the other's outcome. - - Returns ``{"verified": [...], "failed": [...]}`` whose entries are - ``/`` (no ``.service`` suffix), e.g. - ``user/hermes-serve``. + Units are enumerated from systemd, never from the update inventory. That keeps the relaunch + authority requirement structural: a manually launched or Desktop-owned ``hermes serve`` owns no + unit and therefore cannot be touched here. """ skipped_qualified, skipped_legacy = _normalized_skips(skip_units) # (scope, base unit) -> replaced? A unit name can exist in BOTH the user @@ -433,41 +332,28 @@ def restart_serve_units( # remove. outcomes[target] = False continue - try: - result = run( - scope + ["--no-ask-password", "restart", unit], - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", - check=False, - timeout=_UNIT_RESTART_TIMEOUT, - ) - except (OSError, subprocess.TimeoutExpired): - outcomes[target] = False - continue - if getattr(result, "returncode", 1) != 0: + result = _run_quiet( + run, scope + ["--no-ask-password", "restart", unit], timeout=_UNIT_RESTART_TIMEOUT + ) + if not _succeeded(result): # Includes the unprivileged system-scope case. We do not probe # for sudo here: an unverifiable unit must read as failed so # the update stays explicitly incomplete. outcomes[target] = False continue - outcomes[target] = _serve_unit_replaced( - scope, unit, previous_pid, run=run, sleep=sleep - ) + outcomes[target] = _serve_unit_replaced(scope, unit, previous_pid, run=run, sleep=sleep) + # ``/`` is the only identity this module reports for a serve unit. return { - "verified": sorted( - _qualified(*target) for target, ok in outcomes.items() if ok - ), - "failed": sorted( - _qualified(*target) for target, ok in outcomes.items() if not ok - ), + "verified": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if ok), + "failed": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if not ok), } def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: payload = json.load(stream) - profiles = payload.get("profiles") if isinstance(payload, dict) else None + if not isinstance(payload, dict): + payload = {} + profiles = payload.get("profiles") if not isinstance(profiles, list): raise ValueError("recovery payload must contain a profiles list") if any( @@ -475,7 +361,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: for profile in profiles ): raise ValueError("recovery profiles contain an invalid profile id") - raw_supervisors = payload.get("supervisors") if isinstance(payload, dict) else None + raw_supervisors = payload.get("supervisors") supervisors: dict[str, str] = {} if raw_supervisors is not None: if not isinstance(raw_supervisors, dict) or any( @@ -487,7 +373,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: ): raise ValueError("recovery supervisors map is invalid") supervisors = dict(raw_supervisors) - raw_serve = payload.get("serve_units") if isinstance(payload, dict) else None + raw_serve = payload.get("serve_units") recover_serve = False skip_units: list[str] = [] if raw_serve is not None: @@ -495,9 +381,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: raise ValueError("recovery serve_units block is invalid") recover_serve = bool(raw_serve.get("recover")) raw_skip = raw_serve.get("skip") or [] - if not isinstance(raw_skip, list) or any( - not isinstance(entry, (str, dict)) for entry in raw_skip - ): + if not isinstance(raw_skip, list) or any(not isinstance(entry, (str, dict)) for entry in raw_skip): raise ValueError("recovery serve_units skip list is invalid") for entry in raw_skip: # A skip entry names one already-settled process. The qualified @@ -507,14 +391,11 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: # interpreter can still send; it is kept, and read as # scope-agnostic by `restart_serve_units`. if isinstance(entry, dict): - scope_label = entry.get("scope") - unit = entry.get("unit") - if not isinstance(unit, str): + if not isinstance(entry.get("unit"), str): raise ValueError("recovery serve_units skip list is invalid") - if not isinstance(scope_label, str): - scope_label = "" - else: - scope_label, _, unit = entry.rpartition("/") + if not isinstance(entry.get("scope"), str): + entry = {"unit": entry["unit"]} + scope_label, base = _split_skip_entry(entry) # Only the shapes systemd can actually produce for this family; a # skip entry is a name filter, never a command argument. An # unrecognized scope drops the entry rather than raising: dropping @@ -523,22 +404,15 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]: # process. if scope_label and scope_label not in _SCOPE_LABELS: continue - if not _UNIT_RE.fullmatch( - unit if unit.endswith(".service") else f"{unit}.service" - ): + if not _UNIT_RE.fullmatch(f"{base}.service"): continue - base = unit.removesuffix(".service") skip_units.append(f"{scope_label}/{base}" if scope_label else base) return profiles, supervisors, recover_serve, skip_units def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--stdin", - action="store_true", - help=argparse.SUPPRESS, - ) + parser.add_argument("--stdin", action="store_true", help=argparse.SUPPRESS) args = parser.parse_args(argv) if not args.stdin: parser.error("this command is an internal update-recovery entry point") @@ -547,28 +421,20 @@ def main(argv: list[str] | None = None) -> int: profiles, supervisors, recover_serve, skip_units = _parse_payload(sys.stdin) result = restart_profiles(profiles, supervisors=supervisors) result["serve_units"] = ( - restart_serve_units(skip_units=skip_units) - if recover_serve - else {"verified": [], "failed": []} + restart_serve_units(skip_units=skip_units) if recover_serve else {"verified": [], "failed": []} ) except (ValueError, json.JSONDecodeError) as exc: - print( - json.dumps( - { - "error": str(exc), - "verified": [], - "relaunch_attempted": [], - "failed": [], - "serve_units": {"verified": [], "failed": []}, - } - ) - ) + print(json.dumps({ + "error": str(exc), + "verified": [], + "relaunch_attempted": [], + "failed": [], + "serve_units": {"verified": [], "failed": []}, + })) return 2 print(json.dumps(result, sort_keys=True)) - if result["failed"] or result["serve_units"]["failed"]: - return 1 - return 0 + return 1 if result["failed"] or result["serve_units"]["failed"] else 0 if __name__ == "__main__": diff --git a/hermes_cli/worktree_cmd.py b/hermes_cli/worktree_cmd.py index d615fe9fc3..55e269dc11 100644 --- a/hermes_cli/worktree_cmd.py +++ b/hermes_cli/worktree_cmd.py @@ -1,12 +1,8 @@ """``hermes worktree`` — audit and reclaim accumulated git worktrees/branches. -Attended counterpart of the silent startup pruner (see -``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage: - - hermes worktree list # audit: verdict + reason per tree - hermes worktree prune # reap safe trees + merged branches - hermes worktree prune --dry-run # show the plan, change nothing - hermes worktree prune --trees-only / --branches-only +hermes worktree list # audit: verdict + reason per tree hermes worktree prune # reap safe trees + +merged branches hermes worktree prune --dry-run # show the plan, change nothing hermes worktree +prune --trees-only / --branches-only """ from __future__ import annotations @@ -23,9 +19,54 @@ def _repo_root() -> Optional[str]: def _fmt_size(size_mb: Optional[int]) -> str: if size_mb is None: return "?" - if size_mb >= 1024: - return f"{size_mb / 1024:.1f}G" - return f"{size_mb}M" + return f"{size_mb / 1024:.1f}G" if size_mb >= 1024 else f"{size_mb}M" + + +def _list(worktree_gc, repo_root: str, args) -> int: + records = worktree_gc.audit_worktrees(repo_root) + if not records: + print("No worktrees under .worktrees/ — nothing to reclaim.") + return 0 + total_mb = sum(r.size_mb or 0 for r in records) + reapable_mb = sum(r.size_mb or 0 for r in records if r.verdict.startswith("reap")) + print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON") + for r in sorted(records, key=lambda x: -(x.size_mb or 0)): + print(f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} {r.verdict:13} {r.reason}") + print( + f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — " + f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`." + ) + deletable = [b for b in worktree_gc.audit_branches(repo_root) if b.verdict == "delete"] + if deletable: + print(f"{len(deletable)} local branch(es) fully merged/patch-equivalent upstream would also be deleted.") + return 0 + + +def _prune(worktree_gc, repo_root: str, args) -> int: + dry_run = bool(getattr(args, "dry_run", False)) + actions: list = [] + if not getattr(args, "branches_only", False): + tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False) + actions += worktree_gc.reclaim_worktrees(repo_root, dry_run=dry_run, records=tree_records) + kept = [r for r in tree_records if r.verdict == "keep" + and "kanban" not in r.reason and "in use" not in r.reason] + if kept: + print(f"Preserved {len(kept)} tree(s) with real work:") + for r in kept: + print(f" {r.name}: {r.reason}") + if not getattr(args, "trees_only", False): + actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run) + + if actions: + for line in actions: + print(f" {line}") + print(f"{len(actions)} action(s) {'planned' if dry_run else 'done'}.") + else: + print("Nothing to reclaim — all trees/branches carry real work or are in use.") + return 0 + + +_ACTIONS = {"list": _list, "prune": _prune} def cmd_worktree(args) -> int: @@ -35,65 +76,9 @@ def cmd_worktree(args) -> int: if not repo_root: print("Not inside a git repository (or pass --repo ).") return 1 - action = getattr(args, "worktree_action", None) or "list" - - if action == "list": - records = worktree_gc.audit_worktrees(repo_root) - if not records: - print("No worktrees under .worktrees/ — nothing to reclaim.") - return 0 - total_mb = sum(r.size_mb or 0 for r in records) - reapable_mb = sum( - r.size_mb or 0 for r in records if r.verdict.startswith("reap") - ) - print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON") - for r in sorted(records, key=lambda x: -(x.size_mb or 0)): - print( - f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} " - f"{r.verdict:13} {r.reason}" - ) - print( - f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — " - f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`." - ) - branch_records = worktree_gc.audit_branches(repo_root) - deletable = [b for b in branch_records if b.verdict == "delete"] - if deletable: - print( - f"{len(deletable)} local branch(es) fully merged/patch-equivalent " - f"upstream would also be deleted." - ) - return 0 - - if action == "prune": - dry_run = bool(getattr(args, "dry_run", False)) - trees_only = bool(getattr(args, "trees_only", False)) - branches_only = bool(getattr(args, "branches_only", False)) - - actions: list = [] - if not branches_only: - tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False) - actions += worktree_gc.reclaim_worktrees( - repo_root, dry_run=dry_run, records=tree_records - ) - kept = [r for r in tree_records if r.verdict == "keep" - and "kanban" not in r.reason and "in use" not in r.reason] - if kept: - print(f"Preserved {len(kept)} tree(s) with real work:") - for r in kept: - print(f" {r.name}: {r.reason}") - if not trees_only: - actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run) - - if actions: - for line in actions: - print(f" {line}") - verb = "planned" if dry_run else "done" - print(f"{len(actions)} action(s) {verb}.") - else: - print("Nothing to reclaim — all trees/branches carry real work or are in use.") - return 0 - - print(f"Unknown worktree action: {action}") - return 1 + handler = _ACTIONS.get(action) + if handler is None: + print(f"Unknown worktree action: {action}") + return 1 + return handler(worktree_gc, repo_root, args) diff --git a/hermes_cli/worktree_gc.py b/hermes_cli/worktree_gc.py index 13fb195b9f..c2b4e21e4f 100644 --- a/hermes_cli/worktree_gc.py +++ b/hermes_cli/worktree_gc.py @@ -1,35 +1,11 @@ """On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``). -The startup pruner in ``cli._prune_stale_worktrees`` is deliberately -conservative and silent: it runs before the banner on every ``hermes -w`` -launch, so it only reaps clean, fully-merged scratch trees past an age tier -and preserves everything else. That policy is correct for an unattended -startup path — but it means real installs accumulate two kinds of debris the -startup pass can never touch: +The startup pruner in ``cli._prune_stale_worktrees`` is deliberately conservative and silent: it +runs before the banner on every ``hermes -w`` launch, so it only reaps clean, fully-merged scratch +trees past an age tier and preserves everything else. -- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts, - logs) on an otherwise merged branch — preserved forever by the dirty guard. -- **Orphaned local branches** beyond the two auto-generated prefixes the - startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port - branches, feature branches whose PRs merged months ago. Multi-agent boxes - reach hundreds. - -This module is the *attended* counterpart: an explicit, loud, dry-run-first -reclaim the user invokes, so it can be more thorough while staying just as -safe. Invariants shared with the startup pruner (never violated here either): - -- tracked modifications are NEVER deleted, at any age, in any mode; -- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence - decides "unique"; shallow repos are deepened bloblessly first so the - verdict is trustworthy); -- live-locked trees (owning pid alive) are never touched; -- a branch is deleted only after its worktree removal succeeded — a failed - removal must not orphan reachable commits; -- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/`` - before its tree is reaped, never destroyed. - -Classification primitives are imported from ``cli`` so the two paths can -never drift apart on what "dirty", "unpushed", or "merged" means. +- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts, logs) on an otherwise +merged branch — preserved forever by the dirty guard. """ from __future__ import annotations @@ -79,11 +55,10 @@ class BranchRecord: def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess: """Run git, translating timeouts into a nonzero returncode. - Every verdict in this module fails safe toward "keep" on a nonzero - returncode, so a hung/slow git call (large repos make ``git cherry`` - genuinely slow) must degrade to keep — never crash the whole audit - (live-verified failure on a 746MB .git: TimeoutExpired escaped and - aborted the branch audit mid-list). + Every verdict in this module fails safe toward "keep" on a nonzero returncode, so a hung/slow + git call (large repos make ``git cherry`` genuinely slow) must degrade to keep — never crash the + whole audit (live-verified failure on a 746MB .git: TimeoutExpired escaped and aborted the + branch audit mid-list). """ try: return subprocess.run( @@ -98,42 +73,30 @@ def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess ) -def _tree_size_mb(path: Path) -> Optional[int]: +def _tree_size_mb(path: Path, timeout: int = 30) -> Optional[int]: """Cheap directory size via ``du -sm`` — best-effort, None on failure.""" try: result = subprocess.run( - ["du", "-sm", str(path)], - capture_output=True, text=True, encoding="utf-8", - errors="replace", timeout=30, + ["du", "-sm", str(path)], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout ) - if result.returncode == 0 and result.stdout.strip(): - return int(result.stdout.split()[0]) + return int(result.stdout.split()[0]) if result.returncode == 0 and result.stdout.strip() else None except Exception: - pass - return None + return None def _dirty_split(path: str) -> tuple[bool, List[str]]: """Return (has_tracked_modifications, untracked_paths). - ``git status --porcelain`` counts untracked scratch equally with real - edits; the reclaim policy treats them very differently (tracked = real - work, untracked = archivable scratch), so split here. + ``git status --porcelain`` counts untracked scratch the same as real edits, but the reclaim + policy treats them differently (tracked = real work, untracked = archivable scratch). """ try: result = _git(["status", "--porcelain"], cwd=path, timeout=10) if result.returncode != 0: return True, [] # fail safe: treat as real work - tracked = False - untracked: List[str] = [] - for line in result.stdout.splitlines(): - if not line.strip(): - continue - if line.startswith("??"): - untracked.append(line[3:].strip()) - else: - tracked = True - return tracked, untracked + lines = [line for line in result.stdout.splitlines() if line.strip()] + untracked = [line[3:].strip() for line in lines if line.startswith("??")] + return len(untracked) != len(lines), untracked except Exception: return True, [] @@ -141,15 +104,11 @@ def _dirty_split(path: str) -> tuple[bool, List[str]]: def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]: """Copy untracked files out of a doomed tree. Returns the archive dir. - Never destroys: on any copy failure the caller must treat the tree as - keep. Costs almost nothing and removes the "did I just delete - something?" question. + Never destroys: on any copy failure the caller must treat the tree as keep. Costs almost nothing + and removes the "did I just delete something?" question. """ stamp = time.strftime("%Y%m%d-%H%M%S") - dest = ( - Path.home() / ".hermes" / "archive" / "worktree-prune" - / f"{tree.name}-{stamp}" - ) + dest = Path.home() / ".hermes" / "archive" / "worktree-prune" / f"{tree.name}-{stamp}" try: for rel in untracked: src = tree / rel @@ -167,6 +126,34 @@ def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]: return None +def _classify_tree(_cli, repo_root: str, entry: Path, merge_cache, remote_heads) -> tuple[str, str, List[str]]: + """Return (verdict, reason, untracked) for one tree under ``.worktrees/``.""" + path = str(entry) + if _KANBAN_RE.match(entry.name): + return "keep", "kanban task tree (owned by kanban gc)", [] + if _cli._worktree_lock_is_live(repo_root, path, timeout=5) == "live": + return "keep", "in use by a running hermes session", [] + tracked_dirty, untracked = _dirty_split(path) + if tracked_dirty: + return "keep", "uncommitted tracked changes (real work)", [] + archive_note = f"{len(untracked)} untracked file(s) will be archived" + if _cli._worktree_has_unpushed_commits(path, timeout=5) and not _cli._worktree_commits_all_merged_upstream( + path, timeout=30, cache=merge_cache, max_ahead=_MAX_CHERRY_AHEAD + ): + # Pushed-branch tier: single-branch fetch refspecs (the managed-install default) leave + # pushed PR branches with no refs/remotes/* entry, so `git log HEAD --not --remotes` + # reads them as unpushed forever. A local head that EXACTLY matches the remote branch + # has nothing origin lacks — the checkout is redundant; only the branch ref stays. + if not _cli._worktree_branch_pushed_exact(path, remote_heads, timeout=10): + return "keep", "unpushed commits not found upstream", [] + if untracked: + return "reap-keep-branch", f"pushed to origin (open-PR lane); branch kept; {archive_note}", untracked + return "reap-keep-branch", "pushed to origin (open-PR lane); branch kept", [] + if untracked: + return "reap-archive", f"merged/pushed; {archive_note}", untracked + return "reap", "clean and fully merged/pushed", [] + + def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]: """Classify every tree under ``.worktrees/`` without mutating anything.""" import cli as _cli # lazy: cli.py is heavy @@ -194,75 +181,26 @@ def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeReco age_days = (now - entry.stat().st_mtime) / 86400.0 except Exception: continue - size_mb = _tree_size_mb(entry) if with_sizes else None - try: - branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5) - branch = branch_result.stdout.strip() + branch = _git(["branch", "--show-current"], cwd=str(entry), timeout=5).stdout.strip() except Exception: branch = "" - def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None): - records.append(TreeRecord( - name=entry.name, path=str(entry), branch=branch, - age_days=age_days, size_mb=size_mb, - verdict=verdict, reason=reason, - untracked=untracked or [], - )) - - if _KANBAN_RE.match(entry.name): - rec("keep", "kanban task tree (owned by kanban gc)") - continue - - lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5) - if lock_state == "live": - rec("keep", "in use by a running hermes session") - continue - - tracked_dirty, untracked = _dirty_split(str(entry)) - if tracked_dirty: - rec("keep", "uncommitted tracked changes (real work)") - continue - - if _cli._worktree_has_unpushed_commits(str(entry), timeout=5): - merged = _cli._worktree_commits_all_merged_upstream( - str(entry), timeout=30, cache=merge_cache, - max_ahead=_MAX_CHERRY_AHEAD, - ) - if not merged: - # Pushed-branch tier: single-branch fetch refspecs (the - # managed-install default) leave pushed PR branches with no - # refs/remotes/* entry, so `git log HEAD --not --remotes` - # reads them as unpushed forever. A local head that EXACTLY - # matches the remote branch has nothing origin lacks — the - # checkout is redundant; only the branch ref stays. - if _cli._worktree_branch_pushed_exact( - str(entry), remote_heads, timeout=10 - ): - if untracked: - rec("reap-keep-branch", - f"pushed to origin (open-PR lane); branch kept; " - f"{len(untracked)} untracked file(s) will be archived", - untracked) - else: - rec("reap-keep-branch", - "pushed to origin (open-PR lane); branch kept") - continue - rec("keep", "unpushed commits not found upstream") - continue - - if untracked: - rec("reap-archive", - f"merged/pushed; {len(untracked)} untracked file(s) will be archived", - untracked) - else: - rec("reap", "clean and fully merged/pushed") + verdict, reason, untracked = _classify_tree(_cli, repo_root, entry, merge_cache, remote_heads) + records.append(TreeRecord( + name=entry.name, path=str(entry), branch=branch, + age_days=age_days, size_mb=_tree_size_mb(entry) if with_sizes else None, + verdict=verdict, reason=reason, untracked=untracked, + )) if len(merge_cache) != cache_size_before: _cli._save_worktree_merge_cache(merge_cache) return records +_REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"} + + def reclaim_worktrees( repo_root: str, *, @@ -271,14 +209,13 @@ def reclaim_worktrees( ) -> List[str]: """Remove every reap-verdict tree from a frozen audit list. - Operates ONLY on the provided (or freshly computed) audit records — never - re-globs inside the destructive loop, so trees created by concurrent - sessions after the audit are out of scope by construction. + Operates ONLY on the provided (or freshly computed) audit records — never re-globs inside the + destructive loop, so trees created by concurrent sessions after the audit are out of scope by + construction. """ if records is None: records = audit_worktrees(repo_root, with_sizes=False) actions: List[str] = [] - _REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"} for record in records: if record.verdict not in _REAP_VERDICTS: continue @@ -301,19 +238,12 @@ def reclaim_worktrees( pass try: - remove_result = _git( - ["worktree", "remove", record.path, "--force"], - cwd=repo_root, timeout=30, - ) + remove_result = _git(["worktree", "remove", record.path, "--force"], cwd=repo_root, timeout=30) if remove_result.returncode != 0: - actions.append( - f"failed to remove {record.name}: {remove_result.stderr.strip()}" - ) + actions.append(f"failed to remove {record.name}: {remove_result.stderr.strip()}") continue if record.verdict == "reap-keep-branch": - actions.append( - f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)" - ) + actions.append(f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)") continue if record.branch and record.branch not in _PROTECTED_BRANCHES: _git(["branch", "-D", record.branch], cwd=repo_root, timeout=10) @@ -330,44 +260,39 @@ def reclaim_worktrees( def audit_branches(repo_root: str) -> List[BranchRecord]: - """Classify local branches: safe to delete when their content is on - upstream (fully merged OR every commit patch-equivalent via ``git - cherry``) and they are not checked out anywhere. + """Classify local branches: deletable when their content is on upstream and not checked out. - Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``) - to EVERY local branch, because deletion is gated on content reachability - rather than name: a branch whose commits are all upstream loses nothing - when its ref goes. Branch names checked out in any worktree, protected - names, and branches with unique commits are kept. + "On upstream" means fully merged OR every commit patch-equivalent via ``git cherry``. Applies + to EVERY local branch, not just the startup pass's prefix list, because the gate is content + reachability rather than name. Checked-out, protected, and unique-commit branches are kept. """ import cli as _cli if _cli._repo_is_shallow(repo_root): _cli._deepen_shallow_repo(repo_root) - upstream = None - for candidate in ("origin/HEAD", "origin/main", "origin/master"): - probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5) - if probe.returncode == 0: - upstream = candidate - break + def _lines(result) -> List[str]: + return [b.strip() for b in result.stdout.splitlines() if b.strip()] + + upstream = next( + (c for c in ("origin/HEAD", "origin/main", "origin/master") + if _git(["rev-parse", "--verify", "--quiet", c], cwd=repo_root, timeout=5).returncode == 0), + None, + ) if upstream is None: return [] result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10) if result.returncode != 0: return [] - branches = [b.strip() for b in result.stdout.splitlines() if b.strip()] + branches = _lines(result) - active: set = set() wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10) - for line in wt.stdout.splitlines(): - if line.startswith("branch refs/heads/"): - active.add(line.split("branch refs/heads/", 1)[-1].strip()) - - merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"], - cwd=repo_root, timeout=15) - merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()} + active = { + line.removeprefix("branch refs/heads/").strip() + for line in wt.stdout.splitlines() if line.startswith("branch refs/heads/") + } + merged = set(_lines(_git(["branch", "--merged", upstream, "--format=%(refname:short)"], cwd=repo_root, timeout=15))) def _classify_branch(branch: str) -> BranchRecord: if branch in _PROTECTED_BRANCHES or branch in active: @@ -389,7 +314,7 @@ def audit_branches(repo_root: str) -> List[BranchRecord]: cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30) if cherry.returncode != 0: return BranchRecord(branch, "keep", "could not verify (git cherry failed)") - lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()] + lines = _lines(cherry) if lines and all(ln.startswith("-") for ln in lines): return BranchRecord(branch, "delete", "all commits patch-equivalent upstream") unique = sum(1 for ln in lines if ln.startswith("+")) @@ -429,10 +354,10 @@ def reclaim_branches( actions.append(f"would delete branch {record.name} ({record.reason})") continue result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10) - if result.returncode == 0: - actions.append(f"deleted branch {record.name}") - else: - actions.append(f"failed to delete {record.name}: {result.stderr.strip()}") + actions.append( + f"deleted branch {record.name}" if result.returncode == 0 + else f"failed to delete {record.name}: {result.stderr.strip()}" + ) return actions @@ -446,15 +371,4 @@ def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]: count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir()) except Exception: return 0, None - size_mb: Optional[int] = None - try: - result = subprocess.run( - ["du", "-sm", str(worktrees_dir)], - capture_output=True, text=True, encoding="utf-8", - errors="replace", timeout=20, - ) - if result.returncode == 0 and result.stdout.strip(): - size_mb = int(result.stdout.split()[0]) - except Exception: - pass - return count, size_mb + return count, _tree_size_mb(worktrees_dir, timeout=20)