refactor(hclib): cron/update/install lifecycle — cron status, update_* receipts and recovery, install repair, service manager

This commit is contained in:
Teknium
2026-09-02 14:38:32 -07:00
parent 3ffd44acd3
commit 2c7f5d12b5
30 changed files with 2586 additions and 4743 deletions
+111 -158
View File
@@ -1,24 +1,11 @@
"""Dependency-light venv recovery that runs BEFORE hermes_cli.main's imports.
The ``hermes`` console entry point is ``hermes_cli.main:main``. Importing
``hermes_cli.main`` pulls in third-party packages at module level (``dotenv``
via ``hermes_cli.env_loader``, ``yaml`` via ``hermes_cli.config``, ...). In
the exact failure state the update-recovery markers exist for — a failed lazy
backend refresh or interrupted core install that wiped a core package's
import files (#57828) — a normal launch crashes *while importing main.py*,
before ``_recover_from_interrupted_install()`` can run. The marker system is
unreachable precisely when it is needed most.
This module is deliberately **stdlib-only** so importing it can never fail on a corrupted venv.
``hermes_cli.main`` imports and calls :func:`recover_if_needed` at the very top of its module body,
before any third-party import.
This module is deliberately **stdlib-only** so importing it can never fail on
a corrupted venv. ``hermes_cli.main`` imports and calls
:func:`recover_if_needed` at the very top of its module body, before any
third-party import.
Scope: this early pass only repairs enough for ``hermes_cli.main`` to become
importable again (force-reinstall of the known-fragile core packages, using
the pins from pyproject.toml). It NEVER clears the recovery markers — the
full, confirmed marker lifecycle stays with ``_recover_from_interrupted_install()``
in main.py, which runs right after import succeeds.
Scope: this early pass only repairs enough for ``hermes_cli.main`` to become importable again
(force-reinstall of the known-fragile core packages, using the pins from pyproject.toml).
"""
from __future__ import annotations
@@ -83,16 +70,9 @@ def restore_quarantined_shims(
) -> list[tuple[Path, Path]]:
"""Rename quarantined shims back, retrying a lock instead of giving up.
``moved`` holds ``(original, quarantined)`` pairs. Returns the pairs that
could NOT be restored, and prints an actionable recovery command for each.
A pair is not a failure when ``original`` already exists or ``quarantined``
has gone: the installer wrote a fresh shim, or a concurrent sweep won the
race. Both are silent, so two processes sweeping the same orphan cannot
produce a spurious error.
Messages go to stderr by default -- the startup sweep runs on EVERY hermes
invocation, and ``hermes acp`` speaks JSON-RPC on stdout.
A pair is not a failure when ``original`` already exists or ``quarantined`` has gone: the
installer wrote a fresh shim, or a concurrent sweep won the race. Both are silent, so two
processes sweeping the same orphan cannot produce a spurious error.
"""
if stream is None:
stream = sys.stderr
@@ -153,12 +133,67 @@ def _project_root() -> Path:
return Path(__file__).resolve().parent.parent
def _load_pyproject_project(root: Path) -> dict | None:
"""``[project]`` table of ``root/pyproject.toml``; ``None`` when missing/unreadable.
Stdlib-only (tomllib) — shared by every pyproject lookup in the repair path so ``packaging``
is never needed in the broken-venv state.
"""
pyproject = root / "pyproject.toml"
if not pyproject.is_file():
return None
try:
import tomllib
with open(pyproject, "rb") as f:
project = tomllib.load(f).get("project", {})
except Exception:
return None
return project if isinstance(project, dict) else None
def _read_marker_attempts(marker_path: Path) -> int:
"""Attempt counter from a marker's opportunistic JSON body; corrupt/missing → 0."""
try:
raw = marker_path.read_text(encoding="utf-8", errors="replace").strip()
except OSError:
return 0
if not raw:
return 0
try:
import json
return int(json.loads(raw).get("attempts", 0))
except (ValueError, AttributeError):
return 0
def _run_ensurepip(root: Path) -> None:
"""Best-effort pip bootstrap — a killed install can leave the venv with no pip module at all."""
try:
subprocess.run(
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
cwd=root,
capture_output=True,
)
except Exception:
pass
def _report_failed_install(result) -> bool:
"""Replay the captured tail of a failed installer run to stderr; always ``False``."""
tail = (result.stderr or result.stdout or "")[-2000:]
if tail:
print(tail, file=sys.stderr)
return False
def _pid_is_running(pid: int) -> bool:
"""Best-effort stdlib-only process liveness probe.
``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process
handle API there. An access-denied result is conservatively live: racing
an elevated updater is worse than postponing recovery for one launch.
``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process handle API there. An
access-denied result counts as live: racing an elevated updater is worse than postponing
recovery for one launch.
"""
if pid <= 0:
return False
@@ -217,20 +252,13 @@ def _marker_owner_is_live(marker: Path) -> bool:
def _pinned_specs(packages: list[str], project_root: Path) -> list[str]:
"""Map bare package names to their pinned specs from pyproject.toml.
Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may
itself be broken in the failure state this module exists for). Unknown
packages fall back to their bare name.
Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may itself be broken in
the failure state this module exists for). Unknown packages fall back to their bare name.
"""
pyproject = project_root / "pyproject.toml"
if not pyproject.is_file():
return packages
try:
import tomllib
with open(pyproject, "rb") as f:
raw_deps = tomllib.load(f).get("project", {}).get("dependencies", []) or []
except Exception:
project = _load_pyproject_project(project_root)
if project is None:
return packages
raw_deps = project.get("dependencies", []) or []
name_to_spec: dict[str, str] = {}
for spec in raw_deps:
@@ -249,12 +277,9 @@ def _pinned_specs(packages: list[str], project_root: Path) -> list[str]:
def _certifi_bundle_broken() -> bool:
"""True when certifi imports but its ``cacert.pem`` is missing/corrupt.
A brew Python upgrade or an interrupted venv rebuild can leave certifi's
distribution metadata (and even the module) intact while the bundled
``cacert.pem`` is gone or a dangling symlink — every TLS connection then
fails with an opaque ``Could not find a suitable TLS CA certificate
bundle`` from deep inside httpx/requests (#29866). An attribute probe
alone passes in that state, so validate the bundle path itself.
A brew Python upgrade or interrupted venv rebuild can leave certifi's metadata intact while
``cacert.pem`` is gone or a dangling symlink; every TLS connection then fails opaquely deep in
httpx/requests. An attribute probe passes in that state, so validate the bundle path itself.
"""
try:
import certifi
@@ -271,12 +296,9 @@ def _certifi_bundle_broken() -> bool:
def _probe_broken_packages() -> list[str]:
"""Import-probe the fragile core packages in THIS process.
Returns repair package names (deduped, probe order) for modules that fail
to import or lack their sentinel attribute. Failed imports leave nothing
in ``sys.modules``, so a post-repair retry in the same process works.
certifi additionally gets a bundle-file check: the module can import
cleanly while ``cacert.pem`` is missing (#29866).
Returns repair package names (deduped, probe order) for modules that fail to import or lack
their sentinel attribute. Failed imports leave nothing in ``sys.modules``, so a post-repair
retry in the same process works.
"""
broken: list[str] = []
for mod_name, attr in LAZY_REFRESH_IMPORT_PROBES:
@@ -296,10 +318,9 @@ def _probe_broken_packages() -> list[str]:
def _find_uv_binary() -> str | None:
"""Locate a ``uv`` binary without importing third-party modules.
uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so
the stdlib ``pip`` fallback below refuses to touch them. In that state
the only sanctioned installer is uv itself, which Hermes already vendors
(``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only.
uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so the stdlib ``pip``
fallback below refuses to touch them. In that state the only sanctioned installer is uv itself,
which Hermes already vendors (``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only.
"""
exe = "uv.exe" if sys.platform == "win32" else "uv"
candidates = [
@@ -316,40 +337,30 @@ def _find_uv_binary() -> str | None:
def _base_interpreter_is_externally_managed() -> bool:
"""True when ``sys.executable`` is a uv/standalone-builds managed install.
Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their
stdlib (PEP 668), so ``python -m pip install`` aborts with
``externally-managed-environment``. The early repair must then go
through uv (or explicitly override pip) or the reinstall no-ops and the
venv stays broken (#83569).
Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their stdlib (PEP 668), so
``python -m pip install`` aborts with ``externally-managed-environment``. The early repair must
then go through uv (or explicitly override pip) or the reinstall no-ops and the venv stays
broken (#83569).
"""
try:
import sysconfig
stdlib = Path(sysconfig.get_path("stdlib"))
if (stdlib / "EXTERNALLY-MANAGED").exists():
return True
# uv 0.5+ moved the marker into a ``_uv_managed`` sentinel dir…
if (stdlib.parent / "EXTERNALLY-MANAGED").exists():
return True
# uv 0.5+ moved the marker one level up next to a ``_uv_managed`` sentinel dir.
return (stdlib / "EXTERNALLY-MANAGED").exists() or (
stdlib.parent / "EXTERNALLY-MANAGED"
).exists()
except Exception:
pass
return False
return False
def _run_repair_install(specs: list[str], project_root: Path) -> bool:
"""``uv pip`` (or stdlib ``pip``) force-reinstall of the given specs.
Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout);
output is captured and replayed to stderr only on failure. Never raises.
Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout); output is captured and
replayed to stderr only on failure. Never raises.
Two installer paths, in priority order:
1. ``uv pip install`` with ``VIRTUAL_ENV`` pointed at the project venv —
required when the base interpreter is uv-managed (Windows git checkouts
install exactly this way: uv's Python declares PEP 668
``EXTERNALLY-MANAGED`` and plain ``python -m pip`` refuses to run).
2. ``sys.executable -m pip`` as before, for self-contained venvs whose
interpreter carries no PEP 668 marker.
"""
externally_managed = _base_interpreter_is_externally_managed()
if externally_managed:
@@ -366,15 +377,10 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
text=True, encoding="utf-8", errors="replace",
env=env,
)
if result.returncode == 0:
return True
tail = (result.stderr or result.stdout or "")[-2000:]
if tail:
print(tail, file=sys.stderr)
return False
except Exception as exc:
print(f" ✗ Early venv repair could not run uv: {exc}", file=sys.stderr)
return False
return result.returncode == 0 or _report_failed_install(result)
# No uv available: fall through to pip with the PEP 668 override so
# the repair at least attempts to fix the venv instead of no-oping.
print(
@@ -383,14 +389,7 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
file=sys.stderr,
)
try:
subprocess.run(
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
cwd=project_root,
capture_output=True,
)
except Exception:
pass
_run_ensurepip(project_root)
pip_cmd = [sys.executable, "-m", "pip", "install", "--force-reinstall"]
if externally_managed:
pip_cmd.append("--break-system-packages")
@@ -405,24 +404,18 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
except Exception as exc:
print(f" ✗ Early venv repair could not run pip: {exc}", file=sys.stderr)
return False
if result.returncode != 0:
tail = (result.stderr or result.stdout or "")[-2000:]
if tail:
print(tail, file=sys.stderr)
return False
return True
return result.returncode == 0 or _report_failed_install(result)
def _pytest_owns_live_checkout(root: Path) -> bool:
"""True when running under pytest AND ``root`` is this module's own
checkout — the one whose venv is executing the suite right now.
"""True when running under pytest AND ``root`` is this module's own checkout — the one whose
venv is executing the suite right now.
Lifecycle tests spawn real subprocesses that import ``hermes_cli.main``
with recovery armed; ``PYTEST_CURRENT_TEST`` rides the inherited env into
those children. Without this guard, a genuinely-broken dev venv gets a
REAL ``ensurepip`` + ``pip install --force-reinstall`` from inside a
running test suite. Tests that sandbox ``project_root`` to a tmp_path are
unaffected (same posture as ``managed_scope._under_pytest``)."""
Lifecycle tests spawn real subprocesses that import ``hermes_cli.main`` with recovery armed and
inherit ``PYTEST_CURRENT_TEST``. Without this guard a broken dev venv would get a REAL ensurepip
+ ``pip install --force-reinstall`` from inside a running suite. Tests sandboxing
``project_root`` to a tmp_path are unaffected.
"""
return (
"PYTEST_CURRENT_TEST" in os.environ
and root == Path(__file__).resolve().parent.parent
@@ -435,14 +428,10 @@ def recover_if_needed(
) -> None:
"""Repair wiped core packages so ``hermes_cli.main`` can import at all.
Fast path (no marker present) is two ``lstat`` calls. Only acts when a
recovery marker from a prior ``hermes update`` exists AND an import probe
confirms a core package is actually broken. Markers are intentionally
NOT cleared here — ``_recover_from_interrupted_install()`` in main.py owns
the confirmed marker lifecycle and runs immediately after import succeeds.
Fast path (no marker present) is two ``lstat`` calls. Only acts when a recovery marker from a
prior ``hermes update`` exists AND an import probe confirms a core package is actually broken.
Never raises: on any failure the import of main.py proceeds and surfaces
the real error.
Never raises: on any failure the import of main.py proceeds and surfaces the real error.
"""
global _UPDATE_RETRY_RECOVERED
@@ -496,20 +485,8 @@ def recover_if_needed(
# Single-flight: share main.py's recovery lock so an early repair
# never races a concurrent full recovery into the same shared venv.
lock_path = root / ".update-incomplete.lock"
try:
fd = os.open(lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
os.write(fd, f"{os.getpid()}\n".encode())
os.close(fd)
except FileExistsError:
try:
if time.time() - lock_path.stat().st_mtime > 3600:
lock_path.unlink()
except OSError:
pass
if not _claim_recovery_lock(root):
return
except OSError:
pass # read-only fs / perms — proceed unlocked, install surfaces it
try:
specs = _pinned_specs(broken, root)
@@ -531,10 +508,7 @@ def recover_if_needed(
file=sys.stderr,
)
finally:
try:
lock_path.unlink()
except OSError:
pass
_release_recovery_lock(root)
except Exception:
# Never block launch — the import of main.py will surface the truth.
pass
@@ -581,21 +555,11 @@ def _release_recovery_lock(root: Path) -> None:
def _complete_pending_core_install(root: Path, core_marker: Path) -> bool:
"""Run the pending core install BEFORE main.py can import native modules.
``recover_if_needed`` invokes this when ``.update-incomplete`` exists —
a prior ``hermes update`` (or the self-lock preflight, #83569) left the
dependency sync deliberately unfinished. Completing it here matters on
Windows: the deferral exists precisely because the process that wrote the
marker had a native venv extension mapped; this process, running before
``hermes_cli.main``'s third-party imports, maps nothing yet, so the
installer can replace ``.pyd`` files without hitting the lock.
``recover_if_needed`` invokes this when ``.update-incomplete`` exists — a prior ``hermes
update`` (or the self-lock preflight, #83569) left the dependency sync deliberately unfinished.
Marker lifecycle: cleared on success; kept (attempts counter bumped) on
failure for the next launch or main.py's post-import recovery. An
attempts ceiling caps automatic retries so a persistent installer
failure does not block every launch (``hermes acp`` included).
Never raises: any failure leaves the marker for the post-import path and
returns ``False``. Returns ``True`` only after the install succeeds.
Never raises: any failure leaves the marker for the post-import path and returns ``False``.
Returns ``True`` only after the install succeeds.
"""
try:
from hermes_cli import _install_repair as ir
@@ -603,18 +567,7 @@ def _complete_pending_core_install(root: Path, core_marker: Path) -> bool:
# Retry backoff: read current attempts before claiming the lock so a
# persistently-failing install stops hammering early. After the
# increment the counter reflects THIS attempt.
attempts = 0
try:
raw = core_marker.read_text(encoding="utf-8", errors="replace").strip()
if raw:
import json as _json
try:
attempts = int(_json.loads(raw).get("attempts", 0))
except (ValueError, AttributeError):
attempts = 0
except OSError:
attempts = 0
attempts = _read_marker_attempts(core_marker)
if attempts >= _EARLY_CORE_INSTALL_MAX_ATTEMPTS:
print(
+94 -197
View File
@@ -1,22 +1,13 @@
"""Dependency install execution shared between early recovery and full recovery.
Both callers need to run the same core ``.[all]`` reinstall:
- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE ``hermes_cli.main``'s
third-party imports, so it can complete a pending update while no native extension is mapped yet
(#83569). - ``hermes_cli.main._recover_core_update_marker_locked`` — the historical post-import
recovery path.
- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE
``hermes_cli.main``'s third-party imports, so it can complete a pending
update while no native extension is mapped yet (#83569).
- ``hermes_cli.main._recover_core_update_marker_locked`` — the historical
post-import recovery path. Kept as a fallback for installs the early pass
could not complete (marker left in place on failure).
This module is deliberately **stdlib-only** so importing it can never fail in
the corrupted-venv state it exists to repair. ``hermes_cli.main`` imports
``managed_uv``, ``hermes_constants``, and friends only in its late path; the
early path must not. Where the late path uses ``managed_uv.ensure_uv`` to
bootstrap uv if missing, the early path uses the stdlib
:func:`hermes_cli._early_recovery._find_uv_binary` lookup and falls back to
plain pip when uv is absent — a degraded but working installer (the late
recovery will bootstrap uv on the next launch if it ever matters).
This module is deliberately **stdlib-only** so importing it can never fail in the corrupted-venv
state it exists to repair. ``hermes_cli.main`` imports ``managed_uv``, ``hermes_constants``, and
friends only in its late path; the early path must not.
"""
from __future__ import annotations
@@ -43,10 +34,7 @@ def _is_termux_env(env: dict | None = None) -> bool:
"""Stdlib Termux probe (hermes_cli.main's version lives behind imports)."""
env = env if env is not None else os.environ
try:
if env.get("TERMUX_VERSION"):
return True
prefix = env.get("PREFIX", "")
return "com.termux" in prefix
return bool(env.get("TERMUX_VERSION")) or "com.termux" in env.get("PREFIX", "")
except Exception:
return False
@@ -55,11 +43,9 @@ def _is_termux_env(env: dict | None = None) -> bool:
def _stdout_to_stderr():
"""Route fd 1 (and sys.stdout) to stderr for the duration of an install.
``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child
writing there would corrupt the protocol. Mirrors
``main.py::_recover_from_interrupted_install``.
``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child writing there would
corrupt the protocol. Mirrors ``main.py::_recover_from_interrupted_install``.
"""
saved_fd = None
saved_sys_stdout = sys.stdout
try:
saved_fd = os.dup(1)
@@ -72,24 +58,18 @@ def _stdout_to_stderr():
finally:
sys.stdout = saved_sys_stdout
if saved_fd is not None:
try:
with contextlib.suppress(OSError):
os.dup2(saved_fd, 1)
except OSError:
pass
try:
with contextlib.suppress(OSError):
os.close(saved_fd)
except OSError:
pass
def _resolve_install_target(root: Path) -> tuple[list[str], dict | None]:
"""(install_cmd_prefix, env) for the project venv — stdlib uv lookup.
Mirrors ``main.py::_default_venv_install_target`` but without
``managed_uv``. ``VIRTUAL_ENV`` steers ``uv pip`` at the project venv even
when invoked from the base interpreter (the early-recovery case).
Termux strips leaked interpreter-path env vars so uv resolves the venv
correctly.
Mirrors ``main.py::_default_venv_install_target`` without ``managed_uv``. ``VIRTUAL_ENV``
steers ``uv pip`` at the project venv even when invoked from the base interpreter (early
recovery). Termux strips leaked interpreter-path env vars so uv resolves the venv correctly.
"""
uv_bin = _er._find_uv_binary()
if uv_bin:
@@ -125,17 +105,34 @@ def _venv_scripts_dir(root: Path) -> Path | None:
_WINDOWS_BIN_LAUNCHERS = ("hermes", "hermes-acp")
def _launcher_present(target: Path, name: str) -> bool:
return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists()
def _launchers_missing(target: Path) -> bool:
return any(not _launcher_present(target, name) for name in _WINDOWS_BIN_LAUNCHERS)
def _default_hermes_root() -> Path | None:
"""Per-machine anchor for the managed-clone gate — the DEFAULT Hermes root, not
``get_hermes_home()``: under ``hermes -p <name>`` that returns ``profiles\\<name>``, which
would fail the gate and silently skip the heal for profile users. ``None`` when unresolvable."""
from hermes_constants import get_default_hermes_root
try:
return Path(get_default_hermes_root())
except Exception:
return None
def _venv_is_relocatable(venv_dir: Path) -> bool:
"""True when the venv's pyvenv.cfg declares ``relocatable = true``.
uv writes the flag; ``hermes_cli.managed_uv`` builds its replacement
venvs with ``--relocatable`` (they are constructed aside and swapped
into place). A relocatable venv's console-script trampolines embed a
RELATIVE interpreter reference, so a COPY of one placed outside
``venv\\Scripts`` fails at run time with ``uv trampoline failed to
canonicalize script path``. Non-relocatable venvs (fresh installs)
embed the absolute interpreter path and their trampolines survive
copying. This flag decides which launcher form a PATH dir gets.
uv writes the flag; ``managed_uv`` builds replacement venvs ``--relocatable``. A relocatable
venv's console-script trampolines embed a RELATIVE interpreter reference, so a copy placed
outside ``venv\Scripts`` fails with ``uv trampoline failed to canonicalize script path``;
non-relocatable venvs embed the absolute path and survive copying. This decides which
launcher form a PATH dir gets.
"""
try:
cfg = (Path(venv_dir) / "pyvenv.cfg").read_text(
@@ -143,20 +140,18 @@ def _venv_is_relocatable(venv_dir: Path) -> bool:
)
except OSError:
return False
for line in cfg.splitlines():
key, _, value = line.partition("=")
if key.strip().lower() == "relocatable" and value.strip().lower() == "true":
return True
return False
return any(
key.strip().lower() == "relocatable" and value.strip().lower() == "true"
for key, _, value in (line.partition("=") for line in cfg.splitlines())
)
def _normalize_windows_path(value) -> str:
"""Windows path equality key: backslashes, no trailing separator, lowered.
Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than
``os.path.normcase`` — that is an identity function on POSIX, and this
comparison must behave Windows-correct even when tests exercise the
Windows branch from another host (same rationale as
Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than ``os.path.normcase`` —
that is an identity function on POSIX, and this comparison must behave Windows-correct even when
tests exercise the Windows branch from another host (same rationale as
``venv_bin_dir(windows=...)``).
"""
return str(value).replace("/", "\\").rstrip("\\").lower()
@@ -165,8 +160,7 @@ def _normalize_windows_path(value) -> str:
def _windows_user_path_entries() -> list[str]:
"""User PATH entries from the registry — the value install.ps1 writes.
Falls back to the process PATH when the registry is unreadable. Only
called on Windows.
Falls back to the process PATH when the registry is unreadable. Only called on Windows.
"""
try:
import winreg
@@ -187,48 +181,13 @@ def ensure_windows_bin_launchers(
) -> list[str]:
"""Re-stage the Windows ``hermes`` launchers when they vanish.
On Windows, ``hermes`` resolves through launchers derived from the venv
console scripts — never ``venv\\Scripts`` itself on PATH, which would
shadow the user's ``python`` (#83797). The canonical launcher home is
the managed binary dir — the default Hermes root's ``bin``
(``%LOCALAPPDATA%\\hermes\\bin``, next to the managed uv) — which lives
OUTSIDE the git checkout so no git operation can ever touch it. It is
a per-machine dir shared by every profile: ``get_hermes_home()`` would
point inside ``profiles\\<name>`` under ``hermes -p``, so the anchor
here is :func:`hermes_constants.get_default_hermes_root`.
On Windows, ``hermes`` resolves through launchers derived from the venv console scripts — never
``venv\Scripts`` itself on PATH, which would shadow the user's ``python`` (#83797).
Earlier installer versions staged them at ``<checkout>\\bin`` instead —
inside the git working tree — where ``hermes update``'s pre-update
autostash (``git stash push --include-untracked``) swept them off disk;
once the desktop updater stopped re-applying stashes (``--keep-stash``)
nothing restored them and ``hermes`` stopped resolving in every new
terminal. That legacy location is re-staged too, during the transition,
for installs whose user PATH still resolves through it.
The launcher FORM depends on the venv (see :func:`_venv_is_relocatable`):
a normal venv's exe trampoline embeds an absolute interpreter path and
survives copying, so it is copied as ``<name>.exe``; a relocatable
venv's trampoline resolves relative to its own location and a copy
dies with ``uv trampoline failed to canonicalize script path``, so a
``<name>.cmd`` delegator invoking the in-venv exe by absolute path is
written instead. A name counts as present when EITHER form exists —
exe copies staged before a venv rebuild keep working (they embed the
swapped-in-place venv's absolute path) and are left alone.
Two targets, two gates, both failing toward inaction:
- canonical managed binary dir: only when *root* is the managed clone
(``root.parent == get_default_hermes_root()``), so source checkouts
elsewhere never gain launchers;
- legacy ``<root>\\bin``: only when that dir is on the user PATH
(registry value, process PATH as fallback), i.e. the install opted
into the old layout and still resolves through it.
Writes go through a staging name + ``os.replace`` so concurrent process
starts cannot tear a launcher. Never raises; returns the restored paths.
*windows* and *user_path_entries* are injectable for tests, same pattern
as ``hermes_constants.venv_bin_dir``.
- canonical managed binary dir: only when *root* is the managed clone (``root.parent ==
get_default_hermes_root()``), so source checkouts elsewhere never gain launchers; - legacy
``<root>\bin``: only when that dir is on the user PATH (registry value, process PATH as
fallback), i.e.
"""
if windows is None:
windows = _is_windows()
@@ -237,20 +196,10 @@ def ensure_windows_bin_launchers(
root = Path(root)
# Per-machine anchor: the DEFAULT Hermes root, not get_hermes_home() —
# under ``hermes -p <name>`` that returns ``profiles\\<name>``, which
# would fail the managed-clone gate below and silently skip the heal
# for profile users. The launcher dir serves the whole machine.
from hermes_constants import get_default_hermes_root
try:
home = Path(get_default_hermes_root())
except Exception:
home = _default_hermes_root()
if home is None:
return []
def _launcher_present(target: Path, name: str) -> bool:
return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists()
targets: list[Path] = []
# Canonical target — gate on the managed-clone shape. This runs at
@@ -258,7 +207,7 @@ def ensure_windows_bin_launchers(
# override), so the healthy path must stay at a couple of stat calls.
if _normalize_windows_path(root.parent) == _normalize_windows_path(home):
canonical = home / "bin"
if any(not _launcher_present(canonical, name) for name in _WINDOWS_BIN_LAUNCHERS):
if _launchers_missing(canonical):
targets.append(canonical)
# Legacy transition target — the pre-migration in-checkout dir. Only
@@ -268,7 +217,7 @@ def ensure_windows_bin_launchers(
# network shares. An entry stored some other way (8.3 short path,
# subst drive) misses the re-stage, which fails safe: no-op.
legacy = root / "bin"
if any(not _launcher_present(legacy, name) for name in _WINDOWS_BIN_LAUNCHERS):
if _launchers_missing(legacy):
if user_path_entries is None:
user_path_entries = _windows_user_path_entries()
configured = {_normalize_windows_path(entry) for entry in user_path_entries}
@@ -330,8 +279,8 @@ def ensure_windows_bin_launchers(
def _read_user_path_raw() -> tuple[list[str], int]:
"""Raw (unexpanded) user PATH entries + registry value type.
Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them
(same discipline as ``hermes_cli.uninstall``). Only called on Windows.
Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them (same discipline as
``hermes_cli.uninstall``). Only called on Windows.
"""
import winreg
@@ -394,12 +343,10 @@ def migrate_windows_bin_path(
root = Path(root)
# Same per-machine anchor as ensure_windows_bin_launchers (see there).
from hermes_constants import get_default_hermes_root, venv_bin_dir
from hermes_constants import venv_bin_dir
try:
home = Path(get_default_hermes_root())
except Exception:
home = _default_hermes_root()
if home is None:
return False
if _normalize_windows_path(root.parent) != _normalize_windows_path(home):
return False # not the managed clone — nothing to migrate
@@ -457,17 +404,9 @@ def migrate_windows_bin_path(
def _load_console_script_names(root: Path) -> list[str]:
"""``[project.scripts]`` names from pyproject.toml (tomllib, 3.11+)."""
project = _er._load_pyproject_project(root)
try:
import tomllib
except ImportError: # pragma: no cover
return []
pyproject = root / "pyproject.toml"
if not pyproject.is_file():
return []
try:
with open(pyproject, "rb") as f:
data = tomllib.load(f)
scripts = data.get("project", {}).get("scripts", {}) or {}
scripts = (project or {}).get("scripts", {}) or {}
return [str(name) for name in scripts if name]
except Exception:
return []
@@ -476,10 +415,9 @@ def _load_console_script_names(root: Path) -> list[str]:
class ShimQuarantineError(RuntimeError):
"""A live shim could not be renamed aside — the venv is contended (#87331).
Raised BEFORE the install command runs. Callers (early-pass recovery,
core-marker recovery) catch it like any install failure: the
update-incomplete marker survives and a later launch retries once the
holder exits — the contended venv is never mutated.
Raised BEFORE the install command runs. Callers (early-pass recovery, core-marker recovery)
catch it like any install failure: the update-incomplete marker survives and a later launch
retries once the holder exits — the contended venv is never mutated.
"""
def __init__(self, failed_shims: list[str]):
@@ -494,14 +432,9 @@ def _quarantine_running_hermes_exe(
) -> list[tuple[Path, Path]]:
"""Rename live hermes*.exe shims aside so the installer can rewrite them.
Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort:
silently skips anything that cannot be renamed. Returns (original,
quarantined) pairs. stdlib-only — the console-script set comes from
pyproject ``[project.scripts]`` (fallback: the well-known trio).
``failed_out``: when provided, names of shims that could not be renamed
are appended so the caller can refuse instead of mutating a contended
venv (#87331 fail-closed).
Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort: silently skips anything
that cannot be renamed. Returns (original, quarantined) pairs. stdlib-only — the console-script
set comes from pyproject ``[project.scripts]`` (fallback: the well-known trio).
"""
if not _is_windows():
return []
@@ -529,11 +462,9 @@ def _quarantine_running_hermes_exe(
def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None:
"""Put quarantined shims back when the installer did not replace them.
Delegates to the shared helper in the stdlib-only ``_early_recovery``
module: one retry ladder and one recovery message for every restore site,
instead of the near-identical copies that had already drifted (#75584).
Warnings land on stderr — this module runs in the early-recovery path and
``hermes acp`` speaks JSON-RPC on stdout.
Delegates to the shared helper in the stdlib-only ``_early_recovery`` module: one retry ladder
and one recovery message for every restore site, instead of the near-identical copies that had
already drifted (#75584).
"""
_er.restore_quarantined_shims(moved)
@@ -541,13 +472,11 @@ def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None:
def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None:
"""Run an install command with quarantine protection for venv shims.
Fail-closed (#87331): when any live shim cannot be renamed aside, the
venv is contended and the installer would die partway on the same locks
— raise :class:`ShimQuarantineError` WITHOUT running it. The caller's
marker-keeping failure handling turns that into "retry next launch".
Fail-closed (#87331): when any live shim cannot be renamed aside, the venv is contended and the
installer would die partway on the same locks — raise :class:`ShimQuarantineError` WITHOUT
running it. The caller's marker-keeping failure handling turns that into "retry next launch".
Raises CalledProcessError on install failure (callers implement the
per-extra fallback ladder).
Raises CalledProcessError on install failure (callers implement the per-extra fallback ladder).
"""
scripts_dir = _venv_scripts_dir(root) if _is_windows() else None
failed: list[str] = []
@@ -574,19 +503,14 @@ def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None:
def _load_installable_optional_extras(root: Path, group: str) -> list[str]:
"""Optional extras referenced by a dependency group (all / termux-all)."""
try:
import tomllib
with (root / "pyproject.toml").open("rb") as handle:
project = tomllib.load(handle).get("project", {})
except Exception:
project = _er._load_pyproject_project(root)
if project is None:
return []
optional_deps = project.get("optional-dependencies", {})
if not isinstance(optional_deps, dict):
return []
refs = optional_deps.get(group, [])
referenced: list[str] = []
for ref in refs:
for ref in optional_deps.get(group, []):
if "[" in ref and "]" in ref:
name = ref.split("[", 1)[1].split("]", 1)[0]
if name in optional_deps:
@@ -597,34 +521,20 @@ def _load_installable_optional_extras(root: Path, group: str) -> list[str]:
def run_core_install(root: Path) -> None:
"""Full core ``.[all]`` editable reinstall — the recovery install.
Equal in behavior to the install half of
``main.py::_recover_core_update_marker_locked``:
Equal in behavior to the install half of ``main.py::_recover_core_update_marker_locked``:
- bootstrap pip via ensurepip (a killed install can leave the venv with no
pip module at all)
- prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back
to ``python -m pip`` when no uv binary is available
- target ``.[all]`` (or ``.[termux-all]`` on Termux) with the per-extra
fallback ladder when the combined extras resolve fails
- quarantine live ``hermes*.exe`` shims on Windows so they can be replaced
- route ALL install output to stderr (acp/JSON-RPC safety)
- Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env
Raises ``subprocess.CalledProcessError`` when even the base install fails;
callers own marker lifecycle (clear on success, keep on failure).
- bootstrap pip via ensurepip (a killed install can leave the venv with no pip module at all) -
prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back to ``python -m pip``
when no uv binary is available - target ``.[all]`` (or ``.[termux-all]`` on Termux) with the
per-extra fallback ladder when the combined extras resolve fails - quarantine live
``hermes*.exe`` shims on Windows so they can be replaced - route ALL install output to stderr
(acp/JSON-RPC safety) - Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env
"""
prefix, env = _resolve_install_target(root)
group = "termux-all" if _is_termux_env(env) else "all"
with _stdout_to_stderr():
try:
subprocess.run(
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
cwd=root,
capture_output=True,
)
except Exception:
pass
_er._run_ensurepip(root)
try:
_run_install_cmd(
@@ -669,24 +579,11 @@ def run_core_install(root: Path) -> None:
def bump_marker_attempts(marker_path: Path) -> int:
"""Increment an attempts counter stored inside the marker file.
The marker's existence is the signal; opportunistic JSON body carries the
retry count so a persistently failing install can back off instead of
reinstall-hammering every launch. Corrupt/missing bodies restart at 1.
Returns the new attempt count. Never raises.
The marker's existence is the signal; opportunistic JSON body carries the retry count so a
persistently failing install can back off instead of reinstall-hammering every launch.
Corrupt/missing bodies restart at 1. Returns the new attempt count. Never raises.
"""
attempts = 0
try:
raw = marker_path.read_text(encoding="utf-8", errors="replace").strip()
if raw:
try:
attempts = int(json.loads(raw).get("attempts", 0))
except (ValueError, AttributeError):
attempts = 0
except OSError:
attempts = 0
attempts += 1
try:
attempts = _er._read_marker_attempts(marker_path) + 1
with contextlib.suppress(OSError):
marker_path.write_text(json.dumps({"attempts": attempts}), encoding="utf-8")
except OSError:
pass
return attempts
+38 -97
View File
@@ -1,12 +1,7 @@
"""``hermes_cli/_scan_venv_blockers.py`` — Standalone venv-process scan for JSON consumption.
Invoked by the Desktop Electron app::
venv\\Scripts\\python.exe -m hermes_cli._scan_venv_blockers
Exits 0 for valid clear or blocked results. Non-zero exit signals probe
failure (the detector itself crashed, psutil unavailable, etc.). Exactly
one JSON document on stdout; diagnostics on stderr only.
Exits 0 for valid clear or blocked results. Non-zero exit signals probe failure (the detector itself
crashed, psutil unavailable, etc.). Exactly one JSON document on stdout; diagnostics on stderr only.
"""
from __future__ import annotations
@@ -33,10 +28,9 @@ _SENSITIVE_LONG_FLAGS: list[str] = [
def _probe_fail_json(diagnostic: str = "probe failed") -> str:
"""Return the standard probe-failure JSON document.
``ok: false`` plus ``probe_failed: true`` means the detector itself could
not run — this is *not* a clear scan. Callers must treat
``ok is not True`` / non-zero exit as probe failure, never as
``blocked: false`` "clear" (#83149).
``ok: false`` plus ``probe_failed: true`` means the detector itself could not run — this is
*not* a clear scan. Callers must treat ``ok is not True`` / non-zero exit as probe failure,
never as ``blocked: false`` "clear" (#83149).
"""
return json.dumps(
{
@@ -57,11 +51,7 @@ def _emit_probe_fail(diagnostic: str) -> NoReturn:
def _find_flag(text: str, flag: str) -> int:
"""Return the index of *flag* when it starts the string or follows a space.
Returns -1 when not found. This avoids matching ``--token`` inside an
embedded token or path like ``/some--token-thing``.
"""
"""Return the index of *flag* when it starts the string or follows a space."""
low = text.lower()
fl = flag.lower()
pos = 0
@@ -75,11 +65,7 @@ def _find_flag(text: str, flag: str) -> int:
def _redact_sensitive_cmdline(cmdline: str) -> str:
"""Apply generic secret redaction then long-flag redaction.
If the generic redactor itself fails, return ``"<redacted>"`` — the PID
and process name still provide actionable diagnostics.
"""
"""Apply generic secret redaction then long-flag redaction."""
# Generic pass: the project's shared secret redactor.
try:
from agent.redact import redact_sensitive_text # noqa: PLC0415
@@ -94,14 +80,11 @@ def _redact_sensitive_cmdline(cmdline: str) -> str:
# diagnostics (toolset, port, profile).
earliest = len(cmdline)
for flag in _SENSITIVE_LONG_FLAGS:
# --flag=value → preserve "--flag="
idx = _find_flag(cmdline, flag + "=")
if idx != -1 and idx + len(flag) + 1 < earliest:
earliest = idx + len(flag) + 1
# --flag value → preserve "--flag "
idx = _find_flag(cmdline, flag + " ")
if idx != -1 and idx + len(flag) + 1 < earliest:
earliest = idx + len(flag) + 1
# --flag=value → preserve "--flag="; --flag value → preserve "--flag "
for suffix in ("=", " "):
idx = _find_flag(cmdline, flag + suffix)
if idx != -1 and idx + len(flag) + 1 < earliest:
earliest = idx + len(flag) + 1
if earliest < len(cmdline):
return cmdline[:earliest] + "<redacted>"
@@ -111,9 +94,8 @@ def _redact_sensitive_cmdline(cmdline: str) -> str:
def _classify_local_preview_args(args: object) -> dict[str, object]:
"""Return safe UI metadata for an exact ``python -m http.server`` argv.
The general holder detector intentionally truncates its diagnostic command
line. Reading argv separately preserves a useful directory label without
exposing an unbounded command line to the renderer.
The general holder detector truncates its diagnostic command line; reading argv separately
preserves a useful directory label without exposing an unbounded command line to the renderer.
"""
if not isinstance(args, (list, tuple)) or not all(isinstance(arg, str) for arg in args):
return {}
@@ -123,11 +105,10 @@ def _classify_local_preview_args(args: object) -> dict[str, object]:
# to an unrelated script and must never authorize termination.
if len(args) < 3 or args[1] != "-m" or args[2].lower() != "http.server":
return {}
module_index = 1
port = 8000
if module_index + 2 < len(args):
candidate = args[module_index + 2]
if len(args) > 3:
candidate = args[3]
if candidate.isdigit() and 0 < int(candidate) <= 65535:
port = int(candidate)
@@ -172,9 +153,9 @@ def _terminate_safe_preview(
) -> tuple[bool, str | None]:
"""Terminate one verified local preview process tree.
A fresh ``psutil.Process`` identity check and exact argv classification occur
immediately before termination. psutil guards mutating Process methods
against PID reuse, avoiding taskkill's stale-PID race.
A fresh ``psutil.Process`` identity check and exact argv classification occur immediately before
termination. psutil guards mutating Process methods against PID reuse, avoiding taskkill's
stale-PID race.
"""
try:
if psutil_module is None:
@@ -203,29 +184,13 @@ def _terminate_safe_preview(
def _is_pausable_gateway(cmdline: str) -> bool:
"""Return True when *cmdline* is a gateway process the updater can pause.
A running gateway shows up in the venv-holder scan as one or both halves
of its launcher/worker chain (``venv\\Scripts\\python.exe -m
hermes_cli.main gateway run`` and the uv-side interpreter re-running the
same argv). Reporting those as blockers dead-ends the Desktop update:
the preflight aborts with ``venv-blocked`` *before* spawning
``hermes-setup``, so the CLI updater's own
``_pause_windows_gateways_for_update()`` — which exists precisely to
stop these processes (and is always active: ``hermes-setup`` invokes
``hermes update --yes --gateway``) — never gets the chance to run.
Only gateway invocations are exempted. Anything else running from the venv (an operator's REPL,
a stray script, a ``serve`` backend that survived the desktop's own teardown) has no pause
machinery downstream and must keep blocking the handoff.
Only gateway invocations are exempted. Anything else running from the
venv (an operator's REPL, a stray script, a ``serve`` backend that
survived the desktop's own teardown) has no pause machinery downstream
and must keep blocking the handoff.
Delegates to ``gateway.status.looks_like_gateway_command_line`` — the
canonical ``gateway run`` matcher (profile-selector aware, shlex
tokenization, ``run``-only) — so this exemption, the pause discovery,
and the updater's guard fallback all share one parser. A hand-rolled
token scan here regressed ``--profile gateway gateway run``: the profile
*value* shadowed the subcommand token. An import failure counts as
not-pausable — the scan then reports the process as a blocker, which is
exactly the pre-exemption behavior.
Delegates to ``gateway.status.looks_like_gateway_command_line`` — the canonical ``gateway run``
matcher (profile-selector aware, shlex tokenization, ``run``-only) — so this exemption, the
pause discovery, and the updater's guard fallback all share one parser.
"""
try:
from gateway.status import looks_like_gateway_command_line # noqa: PLC0415
@@ -237,33 +202,11 @@ def _is_pausable_gateway(cmdline: str) -> bool:
def _is_updater_owned_backend(pid: int, cmdline: str) -> bool:
"""Return True when *pid* is a Hermes backend the CLI updater can stop.
The gateway exemption above keeps ``gateway run`` holders out of the
blocker list because the updater's own pause machinery stops and resumes
them. ``hermes serve`` / ``hermes dashboard`` backends had no such
deferral, so a leaked serve child (or a Desktop-owned backend the
teardown lost track of) dead-ended the hand-off with ``venv-blocked`` —
or, worse, survived the hand-off and made the shim quarantine fail with
``os error 32`` (#98336) — even though the updater downstream owns
exactly this case with its ledger rungs (`_ledger_reapable_backend_pids`
reaps dead-spawner orphans; `_ledger_manual_serve_holders` stops manual
serves and relaunches them on their recorded host/port).
The gateway exemption above keeps ``gateway run`` holders out of the blocker list because the
updater's own pause machinery stops and resumes them.
Positive identity only — never name/substring matching (#90778, and the
#99558 identity-guard contract):
- the argv's parsed SUBCOMMAND (token-based) is ``serve``/``dashboard``;
- the machine spawn ledger has a live-verified ``(pid, create_time)``
entry for the process with a matching purpose;
- ownership is provable: the recorded spawner is dead or unrecorded
(the updater's rungs stop/relaunch those), or the spawner is an
ancestor of THIS scan — i.e. the Desktop app performing the hand-off,
which exits before the updater runs, turning the backend into exactly
the dead-spawner orphan the ledger rung reaps.
A backend whose recorded spawner is alive and is NOT this hand-off's
Desktop (a second Desktop window, another supervisor) keeps blocking:
that supervisor would respawn whatever the updater kills. Anything
unprovable → not exempt (fail closed, pre-exemption behavior).
Positive identity only — never name/substring matching (#90778, and the #99558 identity-guard
contract):
"""
return _updater_owned_backend_entry(pid, cmdline) is not None
@@ -271,10 +214,9 @@ def _is_updater_owned_backend(pid: int, cmdline: str) -> bool:
def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None:
"""Ledger entry for a deferred backend, or ``None`` when it must block.
Same decision logic as ``_is_updater_owned_backend`` (which delegates
here); returning the matched ledger entry lets ``main()`` emit sanitized
decision evidence — structured identity fields only, never argv, which
can carry tokens or private endpoints (#98350).
Same decision logic as ``_is_updater_owned_backend`` (which delegates here); returning the
matched ledger entry lets ``main()`` emit sanitized decision evidence — structured identity
fields only, never argv, which can carry tokens or private endpoints (#98350).
"""
try:
from hermes_cli.update_cmd import _hermes_holder_subcommand # noqa: PLC0415
@@ -312,10 +254,9 @@ def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None:
def _deferred_backend_evidence(entries: list[dict]) -> list[dict]:
"""Sanitized decision evidence for deferred serve/dashboard backends.
Structured ledger fields only — pid, purpose, recorded port — never the
command line, which can carry tokens or private endpoints. Lets the
scan result explain *why* a holder disappeared from ``processes``
without echoing argv (#98350).
Structured ledger fields only — pid, purpose, recorded port — never the command line, which can
carry tokens or private endpoints. Lets the scan result explain *why* a holder disappeared from
``processes`` without echoing argv (#98350).
"""
evidence = []
for entry in entries:
@@ -331,9 +272,9 @@ def _deferred_backend_evidence(entries: list[dict]) -> list[dict]:
def _spawner_is_this_handoff_desktop(entry: dict) -> bool:
"""True when the entry's live spawner is an ancestor of this scan.
The scan subprocess is spawned by the Desktop app's update preflight, so
the Desktop performing the hand-off is in our ancestor chain. Identity is
verified by ``(pid, create_time)`` — a recycled PID cannot forge the pair.
The scan subprocess is spawned by the Desktop app's update preflight, so the Desktop performing
the hand-off is in our ancestor chain. Identity is verified by ``(pid, create_time)`` — a
recycled PID cannot forge the pair.
"""
spawner_pid = entry.get("spawner_pid")
if not isinstance(spawner_pid, int) or spawner_pid <= 0:
+52 -104
View File
@@ -1,26 +1,12 @@
"""Pre-import startup fast paths — THE canonical lightweight helpers.
This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import
wall (config, argparse tree, logging, providers). Everything here must stay
**stdlib-only and cheap** (os/sys file probes; no yaml, no hermes_cli.config,
no argparse). A guard test (``test_startup_fast_import_weight``) subprocess-
imports this module and fails if any heavy module sneaks into sys.modules.
This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import wall (config, argparse
tree, logging, providers). Everything here must stay **stdlib-only and cheap** (os/sys file probes;
no yaml, no hermes_cli.config, no argparse).
Why this module exists (the bug class it kills): version-printing kept being
reimplemented as ``*_fast()`` copies at the top of main.py (Termux first,
then globally), each duplicating canonical logic — project-root resolution,
container detection, profile detection. The copies drifted: eb4040242
changed the canonical output and referenced ``PROJECT_ROOT`` inside the fast
function, which doesn't exist yet on the fast path → the Termux fast path
NameError'd on --version and nobody noticed. One implementation, imported
by both the fast path and the module constants, makes that drift
structurally impossible; the parity guard test would have caught eb4040242
the day it landed.
``hermes_cli/config.py``'s ``get_container_exec_info()`` reads the same
``.container-mode`` file; keep the file-format assumptions here and there in
sync (this module deliberately only PROBES existence/typos cheaply and errs
toward the slow path, which then does the authoritative parse).
Why this module exists (the bug class it kills): version-printing kept being reimplemented as
``*_fast()`` copies at the top of main.py (Termux first, then globally), each duplicating canonical
logic — project-root resolution, container detection, profile detection.
"""
from __future__ import annotations
@@ -29,21 +15,23 @@ import os
import sys
__all__ = [
"project_root_str",
"ensure_project_root_on_path",
"is_termux_env",
"is_termux_fast_version_argv",
"is_global_fast_version_argv",
"is_container_startup_environment",
"active_profile_may_override_home",
"container_mode_may_be_active",
"read_openai_version",
"read_install_method",
"print_fast_version_info",
"try_fast_version",
"project_root_str", "ensure_project_root_on_path", "is_termux_env",
"is_termux_fast_version_argv", "is_global_fast_version_argv",
"is_container_startup_environment", "active_profile_may_override_home",
"container_mode_may_be_active", "read_openai_version", "read_install_method",
"print_fast_version_info", "try_fast_version",
]
def _read_text(path: str) -> str | None:
"""Read a small text file, or None when it is missing/unreadable."""
try:
with open(path, encoding="utf-8") as handle:
return handle.read()
except (OSError, UnicodeDecodeError):
return None
def project_root_str() -> str:
"""Repo root as a str — the single source for main.py's PROJECT_ROOT."""
return os.path.realpath(os.path.join(os.path.dirname(__file__), os.pardir))
@@ -54,10 +42,8 @@ def ensure_project_root_on_path() -> None:
project_root = project_root_str()
normalized_root = os.path.normcase(os.path.realpath(project_root))
sys.path[:] = [
entry
for entry in sys.path
if not entry
or os.path.normcase(os.path.realpath(entry)) != normalized_root
entry for entry in sys.path
if not entry or os.path.normcase(os.path.realpath(entry)) != normalized_root
]
sys.path.insert(0, project_root)
@@ -66,8 +52,7 @@ def is_termux_env() -> bool:
"""Tiny Termux check for pre-import startup shortcuts."""
prefix = os.environ.get("PREFIX", "")
return bool(
os.environ.get("TERMUX_VERSION")
or "com.termux/files/usr" in prefix
os.environ.get("TERMUX_VERSION") or "com.termux/files/usr" in prefix
or prefix.startswith("/data/data/com.termux/")
)
@@ -76,54 +61,39 @@ def is_termux_fast_version_argv(argv: list[str]) -> bool:
return argv in (["--version"], ["-V"])
def is_global_fast_version_argv(argv: list[str]) -> bool:
return argv in (["--version"], ["-V"])
is_global_fast_version_argv = is_termux_fast_version_argv
def is_container_startup_environment() -> bool:
"""True when we're already INSIDE a container (fast path is then safe)."""
if os.path.exists("/.dockerenv") or os.path.exists("/run/.containerenv"):
return True
try:
with open("/proc/1/cgroup", encoding="utf-8") as handle:
cgroup = handle.read()
except OSError:
return False
cgroup = _read_text("/proc/1/cgroup") or ""
return "docker" in cgroup or "podman" in cgroup or "/lxc/" in cgroup
def active_profile_may_override_home(hermes_root: str) -> bool:
"""Cheap probe: does an active non-default profile redirect HERMES_HOME?"""
active_profile = os.path.join(hermes_root, "active_profile")
try:
if os.path.exists(active_profile):
with open(active_profile, encoding="utf-8") as handle:
active = handle.read().strip()
return bool(active and active != "default")
except (OSError, UnicodeDecodeError):
pass
return False
active = (_read_text(os.path.join(hermes_root, "active_profile")) or "").strip()
return bool(active and active != "default")
def _default_home() -> str:
return os.path.join(os.path.expanduser("~"), ".hermes")
def _resolved_home() -> str:
hermes_home = os.environ.get("HERMES_HOME", "").strip()
if hermes_home:
return hermes_home
return os.path.join(os.path.expanduser("~"), ".hermes")
return os.environ.get("HERMES_HOME", "").strip() or _default_home()
def container_mode_may_be_active() -> bool:
"""Conservative probe for NixOS container-mode routing.
False positives are fine (we fall through to the slow path, whose
``get_container_exec_info()`` does the authoritative check and routes
into the container). False negatives are NOT fine — they'd print the
host's version instead of the container's. Hence: any profile
ambiguity → assume container mode may be active.
False positives are fine (the slow path does the authoritative check). False negatives are NOT —
they'd print the host's version instead of the container's — so any profile ambiguity means "may
be active".
"""
if os.environ.get("HERMES_DEV") == "1":
return False
if is_container_startup_environment():
if os.environ.get("HERMES_DEV") == "1" or is_container_startup_environment():
return False
hermes_home = os.environ.get("HERMES_HOME", "").strip()
@@ -131,15 +101,12 @@ def container_mode_may_be_active() -> bool:
if os.path.exists(os.path.join(hermes_home, ".container-mode")):
return True
parent_name = os.path.basename(os.path.dirname(os.path.normpath(hermes_home)))
return (
parent_name != "profiles"
and active_profile_may_override_home(hermes_home)
)
return parent_name != "profiles" and active_profile_may_override_home(hermes_home)
default_home = os.path.join(os.path.expanduser("~"), ".hermes")
if active_profile_may_override_home(default_home):
return True
return os.path.exists(os.path.join(default_home, ".container-mode"))
default_home = _default_home()
return active_profile_may_override_home(default_home) or os.path.exists(
os.path.join(default_home, ".container-mode")
)
def read_openai_version() -> str | None:
@@ -165,31 +132,18 @@ def read_openai_version() -> str | None:
def read_install_method() -> str | None:
"""Read the installer's ``.install_method`` stamp, if present.
Only the stamp (step 1 of ``config.detect_install_method``'s resolution
order) — the managed/git/pip fallbacks need heavier imports and stay on
the slow path. On the fast path home ambiguity is already excluded:
``container_mode_may_be_active()`` bails to the slow path whenever a
non-default profile might redirect HERMES_HOME.
Only the stamp (step 1 of ``config.detect_install_method``'s resolution order) — the
managed/git/pip fallbacks need heavier imports and stay on the slow path.
"""
stamp = os.path.join(_resolved_home(), ".install_method")
try:
with open(stamp, encoding="utf-8") as handle:
method = handle.read().strip().lower()
return method or None
except OSError:
return None
method = _read_text(os.path.join(_resolved_home(), ".install_method"))
return (method or "").strip().lower() or None
def print_fast_version_info(*, check_updates: bool = True) -> None:
"""THE canonical ``hermes --version`` output (also used by /version).
The static lines print instantly from stdlib-only probes; everything
heavier (upstream SHA in the version line, authoritative install-method
detection, the update-status check) is lazy-imported AFTER the first
line is already on screen, so perceived latency stays instant while the
output carries the full information that used to require the (removed)
``hermes version`` subcommand. Every lazy block degrades gracefully —
a broken/heavy import can never take the basic version output down.
Every lazy block degrades gracefully — a broken/heavy import can never take the basic version
output down.
"""
# Line 1: registry-owned banner label (includes "· upstream <sha>" for
# git installs). banner.py keeps rich/prompt_toolkit lazy, so this
@@ -240,10 +194,7 @@ def print_fast_version_info(*, check_updates: bool = True) -> None:
print(f"Update available — run '{recommended_update_command()}'")
elif behind and behind > 0:
commits_word = "commit" if behind == 1 else "commits"
print(
f"Update available: {behind} {commits_word} behind — "
f"run '{recommended_update_command()}'"
)
print(f"Update available: {behind} {commits_word} behind — run '{recommended_update_command()}'")
elif behind == 0:
print("Up to date")
except Exception:
@@ -253,10 +204,9 @@ def print_fast_version_info(*, check_updates: bool = True) -> None:
def try_fast_version(argv: list[str] | None = None) -> bool:
"""Handle ``hermes --version`` before the heavy import wall.
Only ``--version``/``-V`` (the ``version`` subcommand was removed —
``--version`` now carries the full output incl. update status), and
never when container mode may need to route the command into the
container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch.
Only ``--version``/``-V`` (the ``version`` subcommand was removed — ``--version`` now carries
the full output incl. update status), and never when container mode may need to route the
command into the container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch.
"""
if argv is None:
argv = sys.argv[1:]
@@ -266,9 +216,7 @@ def try_fast_version(argv: list[str] | None = None) -> bool:
if is_termux:
if not is_termux_fast_version_argv(argv):
return False
elif not is_global_fast_version_argv(argv):
return False
elif container_mode_may_be_active():
elif not is_global_fast_version_argv(argv) or container_mode_may_be_active():
return False
print_fast_version_info()
+38 -64
View File
@@ -1,26 +1,9 @@
"""
Baked-in build metadata for Hermes Agent.
"""Baked-in build metadata for Hermes Agent.
Source installs report their git revision live via ``git rev-parse`` (see
``hermes_cli/dump.py`` and ``hermes_cli/banner.py``). That doesn't work inside
the published Docker image because ``.dockerignore`` excludes ``.git``, so
those callsites fall back to ``"(unknown)"`` / drop the banner suffix entirely.
To make ``hermes dump`` and the startup banner identify the exact commit the
image was built from, the Docker build writes the build-time ``$HERMES_GIT_SHA``
arg into ``<project_root>/.hermes_build_sha``. This module is the single
read-side helper consumed by both callsites — keeping the lookup in one place
so the file path and missing-file behaviour stay consistent.
Behaviour:
- Returns ``None`` when the file is absent. Source installs and dev images
built without the ``HERMES_GIT_SHA`` build-arg fall through to live-git
resolution in the caller, so non-Docker installs are unaffected.
- Returns ``None`` on any IO / decoding error. The build-sha is a nice-to-have
for support triage; nothing in the CLI is allowed to crash because of it.
- Truncates to ``short`` characters (default 8) to match the format used by
``git rev-parse --short=8`` throughout the codebase.
Source installs report their git revision live via ``git rev-parse`` (see ``hermes_cli/dump.py`` and
``hermes_cli/banner.py``). That doesn't work inside the published Docker image because
``.dockerignore`` excludes ``.git``, so those callsites fall back to ``"(unknown)"`` / drop the
banner suffix entirely.
"""
from __future__ import annotations
@@ -36,22 +19,27 @@ _BUILD_SHA_FILE = Path(__file__).parent.parent / ".hermes_build_sha"
_code_identity_cache: Optional[dict] = None
def _read_stripped(path: Path) -> str:
return path.read_text(encoding="utf-8", errors="replace").strip()
def _sha_or_none(value: str) -> Optional[str]:
return value if len(value) == 40 else None
def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
"""Resolve the checkout's HEAD commit sha by reading .git directly.
Deliberately NOT ``git rev-parse`` in a subprocess: this helper runs
inside library paths (gateway runtime-status writes, update receipts)
where spawning processes is both slow and hostile to tests that mock
``subprocess.run`` tightly (call-count asserts, sequenced side effects).
Handles regular checkouts, worktrees/submodules (``.git`` file with a
``gitdir:`` pointer + ``commondir``), loose refs, and packed-refs.
Deliberately NOT ``git rev-parse``: this runs in library paths (runtime-status writes, update
receipts) where spawning is slow and hostile to tests that mock ``subprocess.run`` tightly.
Handles worktrees/submodules (``gitdir:`` + ``commondir``), loose refs, and packed-refs.
Returns None on any failure.
"""
try:
git_path = project_root / ".git"
if git_path.is_file():
# Worktree/submodule: ".git" is a pointer file.
pointer = git_path.read_text(encoding="utf-8", errors="replace").strip()
pointer = _read_stripped(git_path)
if not pointer.startswith("gitdir:"):
return None
git_dir = Path(pointer[len("gitdir:"):].strip())
@@ -66,33 +54,28 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
common_dir = git_dir
commondir_file = git_dir / "commondir"
if commondir_file.is_file():
rel = commondir_file.read_text(encoding="utf-8", errors="replace").strip()
common = Path(rel)
if not common.is_absolute():
common = (git_dir / common).resolve()
common_dir = common
common = Path(_read_stripped(commondir_file))
common_dir = common if common.is_absolute() else (git_dir / common).resolve()
head = (git_dir / "HEAD").read_text(encoding="utf-8", errors="replace").strip()
head = _read_stripped(git_dir / "HEAD")
if not head.startswith("ref:"):
# Detached HEAD: the file holds the sha itself.
return head if len(head) == 40 else None
return _sha_or_none(head)
ref_name = head[len("ref:"):].strip()
loose = common_dir / ref_name
if loose.is_file():
sha = loose.read_text(encoding="utf-8", errors="replace").strip()
return sha if len(sha) == 40 else None
return _sha_or_none(_read_stripped(loose))
packed = common_dir / "packed-refs"
if packed.is_file():
for line in packed.read_text(encoding="utf-8", errors="replace").splitlines():
for line in _read_stripped(packed).splitlines():
line = line.strip()
if not line or line.startswith(("#", "^")):
continue
parts = line.split(" ", 1)
if len(parts) == 2 and parts[1].strip() == ref_name:
sha = parts[0].strip()
return sha if len(sha) == 40 else None
return _sha_or_none(parts[0].strip())
except Exception:
return None
return None
@@ -101,34 +84,26 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
def get_code_identity(refresh: bool = False) -> dict:
"""Return the running checkout's code identity as a dict.
Shape: ``{"sha": full-or-short sha | None, "short_sha": str | None,
"version": pyproject version | None, "source": "git" | "build-file" |
"unknown"}``.
Resolution order mirrors the banner/dump callsites: live ``git rev-parse`` for source installs,
the baked ``.hermes_build_sha`` for Docker images (no ``.git`` inside the published image), else
unknown.
Resolution order mirrors the banner/dump callsites: live ``git
rev-parse`` for source installs, the baked ``.hermes_build_sha`` for
Docker images (no ``.git`` inside the published image), else unknown.
Cached per process — code identity cannot change while a process is
running (an updated checkout requires a restart to take effect, which
is exactly the property the fleet version verification relies on).
Never raises; every field degrades to ``None`` independently.
Cached per process — code identity cannot change while a process is running (an updated checkout
requires a restart to take effect, which is exactly the property the fleet version verification
relies on). Never raises; every field degrades to ``None`` independently.
"""
global _code_identity_cache
if _code_identity_cache is not None and not refresh:
return dict(_code_identity_cache)
sha: Optional[str] = None
source = "unknown"
project_root = Path(__file__).parent.parent
resolved = _resolve_git_head_sha(project_root)
if resolved:
sha = resolved
source = "unknown"
sha = _resolve_git_head_sha(project_root)
if sha:
source = "git"
if sha is None:
baked = get_build_sha(short=0)
if baked:
sha = baked
else:
sha = get_build_sha(short=0)
if sha:
source = "build-file"
version: Optional[str] = None
@@ -153,9 +128,8 @@ def get_code_identity(refresh: bool = False) -> dict:
def get_build_sha(short: int = 8) -> Optional[str]:
"""Return the baked-in build SHA, truncated to ``short`` chars, or None.
Reads ``<project_root>/.hermes_build_sha`` if present. The file is
written by the Dockerfile's ``HERMES_GIT_SHA`` build-arg and contains
the full 40-character commit hash on a single line.
Reads ``<project_root>/.hermes_build_sha``, written by the Dockerfile's ``HERMES_GIT_SHA``
build-arg (full 40-char hash on one line).
"""
try:
if not _BUILD_SHA_FILE.is_file():
+41 -136
View File
@@ -1,21 +1,8 @@
"""Container-boot reconciliation of per-profile gateway s6 services.
Service directories under /run/service/ live on **tmpfs** and are wiped
on every container restart. Profile directories under
``$HERMES_HOME/profiles/<name>/`` live on the persistent VOLUME, and
each one records its gateway's last state in ``gateway_state.json``.
This module bridges the two: on every container boot, walk the
persistent profiles, recreate the s6 service slots, and auto-start
only those whose last recorded state was ``running``.
Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the
Dockerfile (Phase 4 Task 4.0). Runs as root after 01-hermes-setup
(the stage2 hook) has chowned the volume and seeded $HERMES_HOME, but
before s6-rc starts user services.
Without this module, every ``docker restart`` would silently wipe
every per-profile gateway, even though the user's profiles still
exist on disk.
Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the Dockerfile (Phase 4 Task 4.0).
Runs as root after 01-hermes-setup (the stage2 hook) has chowned the volume and seeded $HERMES_HOME,
but before s6-rc starts user services.
"""
from __future__ import annotations
@@ -101,34 +88,10 @@ def reconcile_profile_gateways(
) -> list[ReconcileAction]:
"""Recreate s6 service registrations for every persistent profile.
Always registers a ``gateway-default`` slot for the root profile
(the implicit profile that lives at the top of ``$HERMES_HOME``,
not under ``profiles/``). The dispatcher in ``hermes_cli.gateway``
maps an empty profile suffix to ``gateway-default``, so this slot
is what ``hermes gateway start`` (no ``-p``) targets. Without it,
bare ``hermes gateway start`` inside the container would land on
``s6-svc -u /run/service/gateway-default`` → uncaught
``CalledProcessError`` → traceback to the user (PR #30136 review).
The default slot's prior state is read from
``$HERMES_HOME/gateway_state.json`` (sibling to the profile root,
not under ``profiles/``); stale runtime files there are swept the
same way as for named profiles.
Args:
hermes_home: The container's HERMES_HOME (typically /opt/data).
Profiles live under ``<hermes_home>/profiles/<name>/``;
the default profile lives at ``<hermes_home>`` itself.
scandir: The s6 dynamic scandir (typically /run/service). Service
directories are created at ``<scandir>/gateway-<profile>/``.
dry_run: When True, walk and return the action list without
touching the filesystem. For tests and `--dry-run` debug.
container_argv: Optional container PID 1 argv override. Production
reads ``/proc/1/cmdline``; tests inject it directly.
Returns:
One :class:`ReconcileAction` per profile, in this order:
``default`` first, then named profiles in directory order.
Always registers a ``gateway-default`` slot for the root profile (the implicit profile that
lives at the top of ``$HERMES_HOME``, not under ``profiles/``). The dispatcher in
``hermes_cli.gateway`` maps an empty profile suffix to ``gateway-default``, so this slot is what
``hermes gateway start`` (no ``-p``) targets.
"""
actions: list[ReconcileAction] = []
@@ -231,13 +194,10 @@ def _maybe_migrate_legacy_gateway_run_state(
) -> str | None:
"""Seed root gateway_state for pre-s6 `gateway run` containers.
The tini image let Docker users run the gateway as the container
command (`docker run ... gateway run`). After the s6 migration,
profile gateways are restored from persisted gateway_state.json; a
legacy container with no state file would therefore register the
default service down and never start. Only synthesize state when no
root gateway_state.json exists so explicit stopped/failed states keep
winning across restarts.
The tini image let Docker users run the gateway as the container command (`docker run ...
gateway run`). After the s6 migration, profile gateways are restored from persisted
gateway_state.json; a legacy container with no state file would therefore register the default
service down and never start.
"""
state_file = hermes_home / "gateway_state.json"
if state_file.exists():
@@ -264,13 +224,9 @@ def _maybe_migrate_legacy_gateway_run_state(
def _read_container_argv() -> tuple[str, ...]:
"""Best-effort read of the container's main program argv.
Under s6-overlay v2, PID 1 is ``/init`` and its argv contains the
``main-wrapper.sh`` path. Under s6-overlay v3, PID 1 is
``s6-svscan`` and the actual command (``rc.init top main-wrapper.sh
...``) lives on a different PID. We try PID 1 first (fast path,
covers v2 and pre-s6 images), then fall back to scanning
``/proc/*/cmdline`` for a process whose argv contains
``main-wrapper.sh`` (the rc.init-launched PID in v3).
s6-overlay v2: PID 1 is ``/init`` and its argv holds ``main-wrapper.sh``. v3: PID 1 is
``s6-svscan`` and the real command lives on another PID, so after the PID 1 fast path we scan
``/proc/*/cmdline`` for a process whose argv contains ``main-wrapper.sh``.
"""
# Fast path: PID 1 is the command itself (s6-overlay v2 / tini).
try:
@@ -310,24 +266,10 @@ def _read_container_argv() -> tuple[str, ...]:
def _strip_container_argv_prefix(argv: Sequence[str]) -> list[str]:
"""Strip the s6/wrapper prefix off the container argv, leaving the hermes args.
Two container-command argv shapes are handled:
* **s6-overlay v2 / tini:** PID 1 argv is
``/init /opt/hermes/docker/main-wrapper.sh <subcommand> [args...]``.
* **s6-overlay v3:** PID 1 is ``s6-svscan`` and the command lives on the
rc.init-launched process as ``/bin/sh -e
/run/s6/basedir/scripts/rc.init top /opt/hermes/docker/main-wrapper.sh
<subcommand> [args...]`` (see :func:`_read_container_argv`).
Rather than peel each leading token positionally (which silently breaks
the moment s6 changes its launcher shape again — exactly what happened
in the v2→v3 bump), drop everything up to and including the
``main-wrapper.sh`` token: that wrapper path is the stable boundary the
image owns, and the subcommand always follows it. Pre-s6 / direct
``hermes`` invocations carry no wrapper, so fall back to peeling a bare
``init`` prefix. The wrapper re-execs ``hermes <subcommand>``, so an
explicit leading ``hermes`` is peeled too. Shared by the legacy-gateway
and dashboard role detectors.
Rather than peel each leading token positionally (which silently breaks the moment s6 changes
its launcher shape again — exactly what happened in the v2→v3 bump), drop everything up to and
including the ``main-wrapper.sh`` token: that wrapper path is the stable boundary the image
owns, and the subcommand always follows it.
"""
args = list(argv)
@@ -365,19 +307,8 @@ def _is_legacy_gateway_run_request(argv: Sequence[str]) -> bool:
def _is_dashboard_container(argv: Sequence[str]) -> bool:
"""Return True when the container's command is the dashboard.
A dashboard-only container (``hermes dashboard ...``) never spawns or
supervises per-profile gateways — that is the gateway container's job.
Reconciling profile gateway s6 slots there is not just wasted work: when
the gateway and dashboard containers share a bind-mounted HERMES_HOME,
both race to ``flock()`` the same ``logs/gateways/<profile>/lock`` files,
producing "Resource busy" failures and an s6-log restart storm. So the
dashboard container skips reconciliation entirely.
Detected from PID 1 argv (``/proc/1/cmdline``) rather than an operator
flag: the role is a fact about the container's command, not a tunable,
and a flag can be forgotten in a hand-written compose/k8s manifest —
reintroducing the exact storm this prevents. Mirrors the argv handling
in :func:`_is_legacy_gateway_run_request`.
A dashboard-only container (``hermes dashboard ...``) never spawns or supervises per-profile
gateways — that is the gateway container's job.
"""
args = _strip_container_argv_prefix(argv)
return bool(args) and args[0] == "dashboard"
@@ -386,22 +317,12 @@ def _is_dashboard_container(argv: Sequence[str]) -> bool:
def _read_desired_state(profile_dir: Path) -> str | None:
"""Read the persisted gateway desired state for reconciliation.
Newer state files carry ``desired_state``: operator intent written by
s6 lifecycle commands. Older files only carry ``gateway_state``; keep
that as a compatibility fallback so existing running/stopped profiles
preserve their behavior until the next explicit start/stop.
Newer state files carry ``desired_state``: operator intent written by s6 lifecycle commands.
Older files only carry ``gateway_state``; keep that as a compatibility fallback so existing
running/stopped profiles preserve their behavior until the next explicit start/stop.
When falling back to ``gateway_state`` (no explicit ``desired_state``),
a transient running sub-state (``draining``) is normalised to ``running``
— see ``_TRANSIENT_RUNNING_STATES``. A gateway hard-killed mid-drain
leaves ``draining`` as its last persisted value; without this it would be
treated as a non-autostart state and the gateway would stay DOWN forever.
An explicit ``desired_state`` is always honoured verbatim (it is the
operator's durable intent), so this normalisation only affects the
legacy/transient fallback path.
Missing or unparseable files count as "no desired state" so we don't
bork the whole reconciliation on a corrupt file.
Missing or unparseable files count as "no desired state" so we don't bork the whole
reconciliation on a corrupt file.
"""
state_file = profile_dir / "gateway_state.json"
if not state_file.exists():
@@ -423,9 +344,9 @@ def _read_desired_state(profile_dir: Path) -> str | None:
def _cleanup_stale_runtime_files(profile_dir: Path) -> None:
"""Remove gateway.pid and processes.json — they reference PIDs in
the dead container's process namespace and would otherwise confuse
the newly-started gateway's process-mismatch checks."""
"""Remove gateway.pid and processes.json — they reference PIDs in the dead container's process
namespace and would otherwise confuse the newly-started gateway's process-mismatch checks.
"""
for name in _STALE_RUNTIME_FILES:
(profile_dir / name).unlink(missing_ok=True)
@@ -433,10 +354,9 @@ def _cleanup_stale_runtime_files(profile_dir: Path) -> None:
def _read_prior_exit_label(profile_dir: Path) -> str:
"""How the profile's previous gateway life ended (clean/unclean/unknown).
Thin, exception-free wrapper over
:func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont-init runs
in a minimal environment and forensics must never block reconciliation
(NS-608)."""
Thin, exception-free wrapper over :func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont-
init runs in a minimal environment and forensics must never block reconciliation (NS-608).
"""
try:
from gateway.lifecycle_ledger import read_prior_exit_label
return read_prior_exit_label(profile_dir)
@@ -447,20 +367,10 @@ def _read_prior_exit_label(profile_dir: Path) -> str:
def _register_service(scandir: Path, profile: str, *, start: bool) -> None:
"""Recreate the s6 service slot for one profile.
Mirrors the rendering in :func:`S6ServiceManager.register_profile_gateway`,
but here we control the start state directly via the ``down`` marker
file (s6-svscan honors it on rescan). Cannot use the manager
directly because the cont-init.d phase runs as root before
s6-svscan starts scanning the dynamic scandir — the manager's
``s6-svscanctl -a`` call would fail with no control socket.
Atomicity: build the new layout in a sibling temp directory and
rename it into place via :meth:`Path.replace`. This matches
:meth:`S6ServiceManager.register_profile_gateway` (PR #30136
review item O4) — even though cont-init.d runs before s6-svscan
starts scanning, an atomic publication keeps the contract uniform
between the two registration paths and protects against a
half-populated dir if the script is interrupted mid-write.
Mirrors ``S6ServiceManager.register_profile_gateway`` but sets start state via the ``down``
marker directly: cont-init.d runs as root before s6-svscan scans the dynamic scandir, so the
manager's ``s6-svscanctl -a`` would fail with no control socket. Built in a sibling temp dir
and ``Path.replace``d into place so an interrupted write never leaves a half-populated dir.
"""
import shutil
@@ -547,18 +457,13 @@ def _write_reconcile_log(
) -> None:
"""Append one line per profile to $HERMES_HOME/logs/container-boot.log.
Operators inspect this to debug "why didn't my profile come back
up". Keeping a separate log file (vs. mixing into agent.log) lets
troubleshooters grep for "profile=foo" without wading through
unrelated activity.
Operators inspect this to debug "why didn't my profile come back up". Keeping a separate log
file (vs. mixing into agent.log) lets troubleshooters grep for "profile=foo" without wading
through unrelated activity.
Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES``
(defaults to 256 KiB ≈ 3000 reconcile lines), the current file
is renamed to ``container-boot.log.1`` (replacing any previous
rotation) before the new entries are appended. This gives long-
lived containers a soft cap of ~512 KiB across the two files
without pulling in logrotate or s6-log machinery just for this
one append-only file (PR #30136 review item O3).
Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES`` (defaults to 256 KiB ≈ 3000 reconcile
lines), the current file is renamed to ``container-boot.log.1`` (replacing any previous
rotation) before the new entries are appended.
"""
import time
log_dir = hermes_home / "logs"
+270 -345
View File
@@ -1,9 +1,4 @@
"""
Cron subcommand for hermes CLI.
Handles standalone cron management commands like list, create, edit,
pause/resume/run/remove, status, and tick.
"""
"""Cron subcommand for hermes CLI."""
import json
import re
@@ -52,9 +47,9 @@ def _cron_api(**kwargs):
def _active_cron_provider_name() -> str:
"""Name of the resolved cron scheduler provider ('builtin', 'chronos', …).
Best-effort + offline (``resolve_cron_scheduler`` reads config and the
provider's ``is_available()`` contract forbids network). Returns 'builtin'
on any failure so callers fall back to the historical ticker-based checks.
Best-effort + offline (``resolve_cron_scheduler`` reads config and the provider's
``is_available()`` contract forbids network). Returns 'builtin' on any failure so callers fall
back to the historical ticker-based checks.
"""
try:
from cron.scheduler_provider import resolve_cron_scheduler
@@ -67,13 +62,9 @@ def _active_cron_provider_name() -> str:
def _builtin_gateway_liveness() -> Optional[bool]:
"""Tri-state liveness of the builtin cron scheduler's trigger.
Single source of truth shared by the CLI (``_warn_if_gateway_not_running``)
and the ``cronjob`` model tool (#87033): the builtin ticker only runs
inside the gateway process, so a scheduled job with no live gateway can
never fire. Non-builtin providers (e.g. Chronos) fire through their own
machinery and are deliberately exempt — a missing gateway process means
nothing for them, so they report active. ``None`` = probe failed; callers
must not claim either way.
Single source of truth shared by the CLI (``_warn_if_gateway_not_running``) and the ``cronjob``
model tool (#87033): the builtin ticker only runs inside the gateway process, so a scheduled job
with no live gateway can never fire. Non-builtin providers (e.g.
"""
try:
if _active_cron_provider_name() != "builtin":
@@ -110,17 +101,9 @@ def _builtin_gateway_liveness() -> Optional[bool]:
def _warn_if_gateway_not_running() -> None:
"""Warn that scheduled jobs won't fire unless the gateway is running.
The cron ticker only runs inside the gateway (``_start_cron_ticker`` in
gateway/run.py); there is no standalone cron daemon. Without a running
gateway, ``next_run_at`` passes but jobs never fire and ``last_run_at``
stays null — the most common cron support report (#51038). Surfacing this
at create/list time, when the user is right there, prevents it.
An external provider (e.g. Chronos) fires jobs via a NAS-mediated webhook,
NOT the in-process ticker, so a momentarily-absent gateway process does not
mean jobs won't fire — the warning would be a false alarm. Stay quiet for
any non-builtin provider; the gateway-process heuristic only speaks to the
built-in ticker's trigger.
The cron ticker only runs inside the gateway (``_start_cron_ticker`` in gateway/run.py); there
is no standalone cron daemon. Without a running gateway, ``next_run_at`` passes but jobs never
fire and ``last_run_at`` stays null — the most common cron support report (#51038).
"""
# _builtin_gateway_liveness never raises (it maps probe failures to None),
# so no guard is needed here — False is the only warn-worthy state.
@@ -157,10 +140,8 @@ def _format_lateness(seconds: float) -> str:
def _dispatch_display(dispatch: dict) -> Optional[str]:
"""One-line scheduled-vs-actual dispatch summary for a job (#99879).
Returns None when the stamp is malformed. On-time dispatches render a
dim confirmation; late/catch-up dispatches render loudly so a run that
fired 30–150 min after gateway downtime no longer looks like an
ordinary on-time success.
None when the stamp is malformed. On-time dispatches render dim; late/catch-up dispatches render
loudly so a run fired long after gateway downtime doesn't look like an ordinary success.
"""
if not isinstance(dispatch, dict):
return None
@@ -180,9 +161,18 @@ def _dispatch_display(dispatch: dict) -> Optional[str]:
)
def _print_banner(title: str) -> None:
"""Boxed cyan section header shared by ``cron list`` and ``cron incidents``."""
print()
print(color("┌" + "─" * 73 + "┐", Colors.CYAN))
print(color("│" + " " * 25 + title.ljust(48) + "│", Colors.CYAN))
print(color("└" + "─" * 73 + "┘", Colors.CYAN))
print()
def cron_list(show_all: bool = False):
"""List all scheduled jobs."""
from cron.jobs import list_jobs
from cron.jobs import effective_job_state, list_jobs
jobs = list_jobs(include_disabled=show_all)
@@ -191,13 +181,7 @@ def cron_list(show_all: bool = False):
print(color("Create one with 'hermes cron create ...' or the /cron command in chat.", Colors.DIM))
return
print()
print(color("┌─────────────────────────────────────────────────────────────────────────┐", Colors.CYAN))
print(color("│ Scheduled Jobs │", Colors.CYAN))
print(color("└─────────────────────────────────────────────────────────────────────────┘", Colors.CYAN))
print()
from cron.jobs import effective_job_state
_print_banner("Scheduled Jobs")
for job in jobs:
job_id = job.get("id", "?")
@@ -369,10 +353,9 @@ _INCIDENT_STATE_COLORS = {
def cron_incidents(args) -> int:
"""List or acknowledge durable cron failure incidents.
``hermes cron incidents [--state <s>]`` lists incidents (the stored error
is redacted and truncated at write time, safe for terminal display);
``hermes cron incidents ack <id>`` closes one so its failure ping stays
silent until the error signature changes.
``hermes cron incidents [--state <s>]`` lists incidents (the stored error is redacted and
truncated at write time, safe for terminal display); ``hermes cron incidents ack <id>`` closes
one so its failure ping stays silent until the error signature changes.
"""
from cron.incidents import ack_incident, list_incidents
@@ -380,27 +363,12 @@ def cron_incidents(args) -> int:
if action == "ack":
incident_id = getattr(args, "incident_id", None)
if not incident_id:
print(
color(
"✗ Incident ID required: hermes cron incidents ack <incident_id>",
Colors.RED,
)
)
print(color("✗ Incident ID required: hermes cron incidents ack <incident_id>", Colors.RED))
return 1
if ack_incident(incident_id):
print(
color(
f"✓ Incident {incident_id} acknowledged (closed).",
Colors.GREEN,
)
)
print(color(f"✓ Incident {incident_id} acknowledged (closed).", Colors.GREEN))
else:
print(
color(
f"Incident {incident_id} not found or already closed.",
Colors.YELLOW,
)
)
print(color(f"Incident {incident_id} not found or already closed.", Colors.YELLOW))
return 0
state = getattr(args, "state", None)
@@ -411,30 +379,9 @@ def cron_incidents(args) -> int:
print(color(f" (filtered by state '{state}')", Colors.DIM))
return 0
print()
print(
color(
"┌─────────────────────────────────────────────────────────────────────────┐",
Colors.CYAN,
)
)
print(
color(
"│ Cron Failure Incidents │",
Colors.CYAN,
)
)
print(
color(
"└─────────────────────────────────────────────────────────────────────────┘",
Colors.CYAN,
)
)
print()
_print_banner("Cron Failure Incidents")
for inc in incidents:
state_display = color(
inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM)
)
state_display = color(inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM))
print(f" {color(inc['id'], Colors.YELLOW)} {state_display}")
print(f" Job: {inc['job_id']}")
print(f" Type: {inc.get('failure_type', 'unknown')}")
@@ -457,6 +404,90 @@ def cron_incidents(args) -> int:
return 0
_PERMISSION_HINT = (
" Hint: jobs.json may be owned by another user "
"(e.g. rewritten by a root `docker exec hermes "
"hermes cron ...`). Fix ownership to match the "
"gateway user, and prefer `docker exec -u <uid>:<gid>`."
)
_FD_EXHAUSTION_HINT = (
" Hint: the ticker hit file-descriptor exhaustion "
"(EMFILE). The scheduler now retries with backoff and "
"attempts fd reclamation, but if the leak persists, "
"restart the gateway to recover scheduling."
)
def _print_ticker_health(pids: list) -> None:
"""Report builtin-ticker liveness for a gateway process known to be alive.
The gateway PROCESS is alive — but the cron ticker THREAD inside it can die silently, or stay
alive while every tick fails. Check both the liveness heartbeat and the last-successful-tick
marker so we don't report "will fire" when the ticker is dead or failing (#32612, #32895).
"""
from cron.jobs import (
get_ticker_heartbeat_age,
get_ticker_last_error,
get_ticker_success_age,
TICKER_INTERVAL_SECONDS,
)
from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text
# Allow ~3 missed ticker iterations (+ a little slack) before declaring
# trouble. Derived from the shared interval constant so this threshold
# tracks the ticker cadence instead of assuming a hardcoded 60s.
STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default
hb_age = get_ticker_heartbeat_age()
ok_age = get_ticker_success_age()
pid_line = f" PID: {', '.join(map(str, pids))}" if pids else None
def _warn(headline: str) -> None:
print(color(headline, Colors.YELLOW))
if pid_line:
print(pid_line)
if hb_age is None:
# No heartbeat file means the ticker thread has never started: gateway
# not in a cron-enabled profile, started moments ago (heartbeat is
# written after startup), or a config issue blocking the ticker.
_warn("⚠ Gateway is running but the cron ticker has not reported a heartbeat.")
print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.")
print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.")
print(" If heartbeat never appears, restart: hermes gateway restart")
elif hb_age > STALE_AFTER:
# No heartbeat at all → the ticker thread is gone.
_warn(
"⚠ Gateway is running but the cron ticker looks STALLED — "
f"no heartbeat for {int(hb_age)}s (expected every ~60s)."
)
print(" Cron jobs may NOT be firing. Restart: hermes gateway restart")
elif ok_age is not None and ok_age > STALE_AFTER:
# Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a
# long time → ticks are failing every iteration.
_warn(
"⚠ Gateway and cron ticker are running, but no tick has "
f"succeeded in {int(ok_age)}s — ticks may be failing."
)
last_error = get_ticker_last_error()
if last_error:
# Show WHY ticks fail — e.g. a root-rewritten jobs.json
# (PermissionError) that silently locked out the ticker's uid for
# ~14h in the field (#68483), or fd exhaustion (EMFILE) that used
# to stall the scheduler invisibly (#87644).
print(color(f" Last tick error: {last_error}", Colors.RED))
if "Permission denied" in last_error:
print(color(_PERMISSION_HINT, Colors.YELLOW))
elif _cron_is_fd_exhaustion_text(last_error):
print(color(_FD_EXHAUSTION_HINT, Colors.YELLOW))
print(" Check the gateway log for 'Cron tick error'.")
else:
print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN))
if pid_line:
print(pid_line)
if hb_age is not None:
print(f" Ticker heartbeat: {int(hb_age)}s ago")
def cron_status():
"""Show cron execution status."""
from cron.jobs import list_jobs
@@ -483,131 +514,41 @@ def cron_status():
"due jobs are delivered by an authenticated webhook.)",
Colors.DIM,
))
print()
_print_active_jobs_summary(list_jobs(include_disabled=False))
print()
return
pids = find_gateway_pids()
gateway_alive_via_lock = False
if not pids:
# Same false-alarm class the cronjob tool fixed (#95947): the pid scan
# can transiently miss a live gateway (just after a restart) while the
# runtime lock — held for exactly the gateway's lifetime — proves the
# ticker's process is alive. Only declare "not running" when both the
# scan AND the lock say so.
try:
from gateway.status import get_running_pid, is_gateway_runtime_lock_active
if is_gateway_runtime_lock_active():
gateway_alive_via_lock = True
lock_pid = get_running_pid()
if lock_pid:
pids = [lock_pid]
except Exception:
pass
if pids or gateway_alive_via_lock:
# The gateway PROCESS is alive — but the cron ticker THREAD inside it
# can die silently, or stay alive while every tick fails. Check both
# the liveness heartbeat and the last-successful-tick marker so we
# don't report "will fire" when the ticker is dead or failing
# (#32612, #32895).
from cron.jobs import (
get_ticker_heartbeat_age,
get_ticker_last_error,
get_ticker_success_age,
TICKER_INTERVAL_SECONDS,
)
from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text
# Allow ~3 missed ticker iterations (+ a little slack) before declaring
# trouble. Derived from the shared interval constant so this threshold
# tracks the ticker cadence instead of assuming a hardcoded 60s.
STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default
hb_age = get_ticker_heartbeat_age()
ok_age = get_ticker_success_age()
if hb_age is None:
# No heartbeat file means the ticker thread has never started.
# This can occur when:
# - Gateway is running but not in a profile with cron enabled,
# - Gateway was started moments ago (heartbeat is written after startup),
# - Or a configuration issue is blocking the ticker from starting at all.
print(color(
"⚠ Gateway is running but the cron ticker has not reported a heartbeat.",
Colors.YELLOW,
))
if pids:
print(f" PID: {', '.join(map(str, pids))}")
print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.")
print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.")
print(" If heartbeat never appears, restart: hermes gateway restart")
elif hb_age > STALE_AFTER:
# No heartbeat at all → the ticker thread is gone.
print(color(
"⚠ Gateway is running but the cron ticker looks STALLED — "
f"no heartbeat for {int(hb_age)}s (expected every ~60s).",
Colors.YELLOW,
))
if pids:
print(f" PID: {', '.join(map(str, pids))}")
print(" Cron jobs may NOT be firing. Restart: hermes gateway restart")
elif ok_age is not None and ok_age > STALE_AFTER:
# Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a
# long time → ticks are failing every iteration.
print(color(
"⚠ Gateway and cron ticker are running, but no tick has "
f"succeeded in {int(ok_age)}s — ticks may be failing.",
Colors.YELLOW,
))
if pids:
print(f" PID: {', '.join(map(str, pids))}")
last_error = get_ticker_last_error()
if last_error:
# Show WHY ticks fail — e.g. a root-rewritten jobs.json
# (PermissionError) that silently locked out the ticker's
# uid for ~14h in the field (#68483), or fd exhaustion
# (EMFILE) that used to stall the scheduler invisibly
# (#87644).
print(color(f" Last tick error: {last_error}", Colors.RED))
if "Permission denied" in last_error:
print(color(
" Hint: jobs.json may be owned by another user "
"(e.g. rewritten by a root `docker exec hermes "
"hermes cron ...`). Fix ownership to match the "
"gateway user, and prefer `docker exec -u <uid>:<gid>`.",
Colors.YELLOW,
))
elif _cron_is_fd_exhaustion_text(last_error):
print(color(
" Hint: the ticker hit file-descriptor exhaustion "
"(EMFILE). The scheduler now retries with backoff and "
"attempts fd reclamation, but if the leak persists, "
"restart the gateway to recover scheduling.",
Colors.YELLOW,
))
print(" Check the gateway log for 'Cron tick error'.")
else:
print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN))
if pids:
print(f" PID: {', '.join(map(str, pids))}")
if hb_age is not None:
print(f" Ticker heartbeat: {int(hb_age)}s ago")
else:
print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED))
print()
print(" To enable automatic execution:")
print(" hermes gateway install # Install as a user service")
print(" sudo hermes gateway install --system # Linux servers: boot-time system service")
print(" hermes gateway # Or run in foreground")
pids = find_gateway_pids()
gateway_alive_via_lock = False
if not pids:
# Same false-alarm class the cronjob tool fixed (#95947): the pid scan
# can transiently miss a live gateway (just after a restart) while the
# runtime lock — held for exactly the gateway's lifetime — proves the
# ticker's process is alive. Only declare "not running" when both the
# scan AND the lock say so.
try:
from gateway.status import get_running_pid, is_gateway_runtime_lock_active
if is_gateway_runtime_lock_active():
gateway_alive_via_lock = True
lock_pid = get_running_pid()
if lock_pid:
pids = [lock_pid]
except Exception:
pass
if pids or gateway_alive_via_lock:
_print_ticker_health(pids)
else:
print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED))
print()
print(" To enable automatic execution:")
print(" hermes gateway install # Install as a user service")
print(" sudo hermes gateway install --system # Linux servers: boot-time system service")
print(" hermes gateway # Or run in foreground")
print()
_print_active_jobs_summary(list_jobs(include_disabled=False))
print()
def _print_active_jobs_summary(jobs) -> None:
"""Print the '<N> active job(s)' + next-run line shared by every status
path (built-in ticker AND external provider)."""
@@ -648,9 +589,9 @@ def _print_active_jobs_summary(jobs) -> None:
def _scripts_dir_for_cron() -> Path:
"""Return the scripts directory used by cron jobs.
Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call
so tests and profile-aware callers that monkeypatch cron storage inspect the
same Hermes home the jobs were loaded from.
Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call so tests and
profile-aware callers that monkeypatch cron storage inspect the same Hermes home the jobs were
loaded from.
"""
from cron.jobs import CRON_DIR
@@ -778,42 +719,29 @@ def cron_doctor() -> int:
return 1
def cron_create(args):
# The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on
# every job-creation path (this CLI subcommand AND the agent's `cronjob`
# model tool, which calls create_job directly). When it blocks, create_job
# raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and
# returns it as result["error"], and the `if not result.get("success")`
# branch below prints it in red and exits 1 — same UX as before.
result = _cron_api(
action="create",
schedule=args.schedule,
prompt=args.prompt,
name=getattr(args, "name", None),
deliver=getattr(args, "deliver", None),
failure_deliver=getattr(args, "failure_deliver", None),
repeat=getattr(args, "repeat", None),
skill=getattr(args, "skill", None),
skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)),
script=getattr(args, "script", None),
workdir=getattr(args, "workdir", None),
model=getattr(args, "model", None),
provider=getattr(args, "model_provider", None),
no_agent=getattr(args, "no_agent", False) or None,
monitor_script=getattr(args, "monitor_script", None),
monitor_url=getattr(args, "monitor_url", None),
continuity=getattr(args, "continuity", None),
reasoning_effort=getattr(args, "reasoning_effort", None),
)
if not result.get("success"):
print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED))
return 1
print(color(f"Created job: {result['job_id']}", Colors.GREEN))
print(f" Name: {result['name']}")
print(f" Schedule: {result['schedule']}")
if result.get("skills"):
print(f" Skills: {', '.join(result['skills'])}")
job_data = result.get("job", {})
_JOB_ARG_FIELDS = (
("name", "name"),
("deliver", "deliver"),
("failure_deliver", "failure_deliver"),
("repeat", "repeat"),
("script", "script"),
("workdir", "workdir"),
("model", "model"),
("provider", "model_provider"),
("monitor_script", "monitor_script"),
("monitor_url", "monitor_url"),
("continuity", "continuity"),
("reasoning_effort", "reasoning_effort"),
)
def _job_api_kwargs(args) -> Dict[str, Any]:
"""Collect the create/update kwargs shared by ``cron create`` and ``cron edit``."""
return {api_key: getattr(args, attr, None) for api_key, attr in _JOB_ARG_FIELDS}
def _print_job_details(job_data: Dict[str, Any]) -> None:
"""Print the optional Script/Monitor/Mode/Continuity/Workdir lines of a job record."""
if job_data.get("script"):
print(f" Script: {job_data['script']}")
if job_data.get("monitor_script"):
@@ -826,6 +754,33 @@ def cron_create(args):
print(" Continuity: on (each run sees the previous run's output)")
if job_data.get("workdir"):
print(f" Workdir: {job_data['workdir']}")
def cron_create(args):
# The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on
# every job-creation path (this CLI subcommand AND the agent's `cronjob`
# model tool, which calls create_job directly). When it blocks, create_job
# raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and
# returns it as result["error"], and the `if not result.get("success")`
# branch below prints it in red and exits 1 — same UX as before.
result = _cron_api(
action="create",
schedule=args.schedule,
prompt=args.prompt,
skill=getattr(args, "skill", None),
skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)),
no_agent=getattr(args, "no_agent", False) or None,
**_job_api_kwargs(args),
)
if not result.get("success"):
print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED))
return 1
print(color(f"Created job: {result['job_id']}", Colors.GREEN))
print(f" Name: {result['name']}")
print(f" Schedule: {result['schedule']}")
if result.get("skills"):
print(f" Skills: {', '.join(result['skills'])}")
_print_job_details(result.get("job", {}))
print(f" Next run: {result['next_run_at']}")
_warn_if_gateway_not_running()
return 0
@@ -866,20 +821,9 @@ def cron_edit(args):
job_id=args.job_id,
schedule=getattr(args, "schedule", None),
prompt=getattr(args, "prompt", None),
name=getattr(args, "name", None),
deliver=getattr(args, "deliver", None),
failure_deliver=getattr(args, "failure_deliver", None),
repeat=getattr(args, "repeat", None),
skills=final_skills,
script=getattr(args, "script", None),
workdir=getattr(args, "workdir", None),
model=getattr(args, "model", None),
provider=getattr(args, "model_provider", None),
no_agent=getattr(args, "no_agent", None),
monitor_script=getattr(args, "monitor_script", None),
monitor_url=getattr(args, "monitor_url", None),
continuity=getattr(args, "continuity", None),
reasoning_effort=getattr(args, "reasoning_effort", None),
**_job_api_kwargs(args),
)
if not result.get("success"):
print(color(f"Failed to update job: {result.get('error', 'unknown error')}", Colors.RED))
@@ -893,23 +837,12 @@ def cron_edit(args):
print(f" Skills: {', '.join(updated['skills'])}")
else:
print(" Skills: none")
if updated.get("script"):
print(f" Script: {updated['script']}")
if updated.get("monitor_script"):
print(f" Monitor: {updated['monitor_script']} (agent runs only on output change)")
if updated.get("monitor_url"):
print(f" Monitor: {updated['monitor_url']} (agent runs only on output change)")
if updated.get("no_agent"):
print(" Mode: no-agent (script stdout delivered directly)")
if updated.get("continuity"):
print(" Continuity: on (each run sees the previous run's output)")
if updated.get("workdir"):
print(f" Workdir: {updated['workdir']}")
_print_job_details(updated)
return 0
def _job_action(action: str, job_id: str, success_verb: str) -> int:
_stateless_reset = None
_stateless_token = None
if action == "run":
# One-shot CLI: this process exits as soon as the command returns, so
# a background-dispatched run (daemon thread of THIS process) would be
@@ -925,16 +858,13 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int:
from gateway.session_context import _SESSION_ASYNC_DELIVERY
_stateless_token = _SESSION_ASYNC_DELIVERY.set(False)
def _stateless_reset() -> None:
_SESSION_ASYNC_DELIVERY.reset(_stateless_token)
except Exception:
_stateless_reset = None
_stateless_token = None
try:
result = _cron_api(action=action, job_id=job_id)
finally:
if _stateless_reset is not None:
_stateless_reset()
if _stateless_token is not None:
_SESSION_ASYNC_DELIVERY.reset(_stateless_token)
if not result.get("success"):
print(color(f"Failed to {action} job: {result.get('error', 'unknown error')}", Colors.RED))
return 1
@@ -970,14 +900,17 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int:
def cron_resume(args) -> int:
"""Resume a paused job or explicitly re-arm a completed one-shot."""
if bool(getattr(args, "run_at", None)) == bool(getattr(args, "run_now", False)):
if getattr(args, "run_at", None) or getattr(args, "run_now", False):
run_at = getattr(args, "run_at", None)
run_now = getattr(args, "run_now", False)
if bool(run_at) == bool(run_now):
if run_at or run_now:
print(color("Use exactly one of --at or --run-now.", Colors.RED))
return 1
return _job_action("resume", args.job_id, "Resumed")
from cron.jobs import AmbiguousJobReference, _hermes_now, rearm_oneshot
run_at = _hermes_now().isoformat() if args.run_now else args.run_at
if run_now:
run_at = _hermes_now().isoformat()
try:
job = rearm_oneshot(args.job_id, run_at)
except (AmbiguousJobReference, ValueError) as exc:
@@ -994,10 +927,9 @@ def cron_resume(args) -> int:
def cron_notepad(args) -> int:
"""Handle ``hermes cron notepad <job_id> [get|set|delete|list]``.
The per-job durable KV scratchpad (``cron/notepad.py``). This CLI is the
write path — a running cron agent updates its own notepad by invoking
these commands via its terminal tool; the scheduler injects non-empty
notepads into the job prompt on each run.
This CLI is the write path for the per-job durable KV scratchpad (``cron/notepad.py``): a
running cron agent updates its own notepad via its terminal tool, and the scheduler injects
non-empty notepads into the job prompt on each run.
"""
from cron import notepad
@@ -1011,30 +943,21 @@ def cron_notepad(args) -> int:
return 1
try:
if action == "set":
if key is None or value is None:
print(color("Usage: hermes cron notepad <job_id> set <key> <value>", Colors.RED))
if action in ("set", "get", "delete"):
usage_args = "set <key> <value>" if action == "set" else f"{action} <key>"
if key is None or (action == "set" and value is None):
print(color(f"Usage: hermes cron notepad <job_id> {usage_args}", Colors.RED))
return 1
notepad.set_note(job_id, key, value)
print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN))
return 0
if action == "get":
if key is None:
print(color("Usage: hermes cron notepad <job_id> get <key>", Colors.RED))
return 1
stored = notepad.get_note(job_id, key)
if stored is None:
print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW))
return 1
print(stored)
return 0
if action == "delete":
if key is None:
print(color("Usage: hermes cron notepad <job_id> delete <key>", Colors.RED))
return 1
if notepad.delete_note(job_id, key):
if action == "set":
notepad.set_note(job_id, key, value)
print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN))
return 0
if action == "get":
stored = notepad.get_note(job_id, key)
if stored is not None:
print(stored)
return 0
elif notepad.delete_note(job_id, key):
print(color(f"Deleted notepad key '{key}' for job {job_id}.", Colors.GREEN))
return 0
print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW))
@@ -1054,52 +977,54 @@ def cron_notepad(args) -> int:
return 1
def _cron_list_cmd(args) -> int:
cron_list(getattr(args, "all", False))
return 0
def _cron_status_cmd(args) -> int:
cron_status()
return 0
def _cron_runs_cmd(args) -> int:
cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20))
return 0
def _cron_remove_cmd(args) -> int:
return _job_action("remove", args.job_id, "Removed")
# Subcommand -> handler. Late-bound lambdas so module-level monkeypatching of
# the underlying functions keeps working.
_CRON_SUBCOMMANDS = {
"list": _cron_list_cmd,
"status": _cron_status_cmd,
"doctor": lambda args: cron_doctor(),
"tick": lambda args: cron_tick(),
"runs": _cron_runs_cmd,
"history": _cron_runs_cmd,
"incidents": lambda args: cron_incidents(args),
"notepad": lambda args: cron_notepad(args),
"create": lambda args: cron_create(args),
"add": lambda args: cron_create(args),
"edit": lambda args: cron_edit(args),
"pause": lambda args: _job_action("pause", args.job_id, "Paused"),
"resume": lambda args: cron_resume(args),
"run": lambda args: _job_action("run", args.job_id, "Triggered"),
"remove": _cron_remove_cmd,
"rm": _cron_remove_cmd,
"delete": _cron_remove_cmd,
}
def cron_command(args):
"""Handle cron subcommands."""
subcmd = getattr(args, 'cron_command', None)
if subcmd is None or subcmd == "list":
show_all = getattr(args, 'all', False)
cron_list(show_all)
return 0
if subcmd == "status":
cron_status()
return 0
if subcmd == "doctor":
return cron_doctor()
if subcmd == "tick":
return cron_tick()
if subcmd in {"runs", "history"}:
cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20))
return 0
if subcmd == "incidents":
return cron_incidents(args)
if subcmd == "notepad":
return cron_notepad(args)
if subcmd in {"create", "add"}:
return cron_create(args)
if subcmd == "edit":
return cron_edit(args)
if subcmd == "pause":
return _job_action("pause", args.job_id, "Paused")
if subcmd == "resume":
return cron_resume(args)
if subcmd == "run":
return _job_action("run", args.job_id, "Triggered")
if subcmd in {"remove", "rm", "delete"}:
return _job_action("remove", args.job_id, "Removed")
handler = _CRON_SUBCOMMANDS.get("list" if subcmd is None else subcmd)
if handler is not None:
return handler(args)
print(f"Unknown cron command: {subcmd}")
print("Usage: hermes cron [list|create|edit|pause|resume|run|remove|status|runs|doctor|tick]")
+36 -73
View File
@@ -1,17 +1,13 @@
"""Lazy dependency bootstrapper for non-Python runtime deps.
Detection and prompting live here in Python — not in install.sh — because:
1. shutil.which() works on every platform; install.sh needs bash.
2. Detection is instant; spawning bash for a "is node installed?" check is waste.
3. Python controls the UX (rich prompts, non-interactive fallback, TTY detection).
Detection and prompting live here in Python — not in install.sh — because: 1. shutil.which() works
on every platform; install.sh needs bash. 2. Detection is instant; spawning bash for a "is node
installed?" check is waste. 3. Python controls the UX (rich prompts, non-interactive fallback, TTY
detection).
install.sh is still the *installation* backend because it has 1900 lines of
battle-tested OS detection and package-manager logic (apt/brew/pacman/dnf/
zypper/Termux/…). Reimplementing that in Python would be huge duplication.
Deps that degrade gracefully (ripgrep → grep fallback, ffmpeg → skip conversion)
don't need ensure_dependency wired in — only hard-fail sites do (TUI needs node,
browser tool needs agent-browser).
install.sh is still the *installation* backend because it has 1900 lines of battle-tested OS
detection and package-manager logic (apt/brew/pacman/dnf/ zypper/Termux/…). Reimplementing that in
Python would be huge duplication.
"""
from __future__ import annotations
@@ -50,21 +46,18 @@ _DEP_DESCRIPTIONS = {
def _has_system_browser() -> bool:
if _IS_WINDOWS:
names = ("chrome", "msedge", "chromium")
else:
names = ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome")
for name in names:
if shutil.which(name):
return True
return False
names = (
("chrome", "msedge", "chromium") if _IS_WINDOWS
else ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome")
)
return any(shutil.which(name) for name in names)
def _has_npx_agent_browser() -> bool:
"""agent-browser resolves lazily via npx on the default install (#43564),
invisible to the PATH/managed-dir probes above. Mirror
tools.browser_tool.check_browser_requirements's Termux carve-out so this
check can't diverge from what browser tools actually find."""
"""agent-browser resolves lazily via npx on the default install (#43564), invisible to the
PATH/managed-dir probes above. Mirror tools.browser_tool.check_browser_requirements's Termux
carve-out so this check can't diverge from what browser tools actually find.
"""
try:
from tools.browser_tool import (
_find_agent_browser,
@@ -74,9 +67,9 @@ def _has_npx_agent_browser() -> bool:
browser_cmd = _find_agent_browser(validate=False)
except Exception:
return False
if not _is_npx_agent_browser_sentinel(browser_cmd):
return False
return not _requires_real_termux_browser_install(browser_cmd)
return _is_npx_agent_browser_sentinel(browser_cmd) and not _requires_real_termux_browser_install(
browser_cmd
)
def _has_hermes_agent_browser() -> bool:
@@ -94,41 +87,23 @@ def _has_hermes_agent_browser() -> bool:
def _find_install_script(
package_dir: Path | None = None,
repo_root: Path | None = None,
package_dir: Path | None = None, repo_root: Path | None = None
) -> tuple[Path | None, str | None]:
"""Locate the install script — bundled in wheel or in git checkout.
On Windows, prefers install.ps1; on POSIX, prefers install.sh.
Returns a (path, shell) tuple, or (None, None) if neither is found.
"""
if package_dir is None:
package_dir = Path(__file__).parent
if repo_root is None:
repo_root = package_dir.parent
"""Locate the install script — bundled in wheel or in git checkout."""
package_dir = package_dir or Path(__file__).parent
repo_root = repo_root or package_dir.parent
candidates = [("install.sh", "bash"), ("install.ps1", "powershell")]
if _IS_WINDOWS:
preferred = ("install.ps1", "powershell")
fallback = ("install.sh", "bash")
else:
preferred = ("install.sh", "bash")
fallback = ("install.ps1", "powershell")
for script_name, shell in (preferred, fallback):
bundled = package_dir / "scripts" / script_name
if bundled.is_file():
return bundled, shell
repo = repo_root / "scripts" / script_name
if repo.is_file():
return repo, shell
candidates.reverse()
for script_name, shell in candidates:
for base in (package_dir, repo_root):
script = base / "scripts" / script_name
if script.is_file():
return script, shell
return None, None
def ensure_dependency(
dep: str,
interactive: bool = True,
) -> bool:
def ensure_dependency(dep: str, interactive: bool = True) -> bool:
"""Ensure a non-Python dependency is available. Returns True if available."""
check = _DEP_CHECKS.get(dep)
if check is None:
@@ -138,15 +113,14 @@ def ensure_dependency(
return True
script, shell = _find_install_script()
desc = _DEP_DESCRIPTIONS.get(dep, dep)
if script is None:
if interactive:
desc = _DEP_DESCRIPTIONS.get(dep, dep)
print(f" {desc} is not installed and no install script was found.")
print(f" Install {dep} manually and try again.")
return False
if interactive and sys.stdin.isatty():
desc = _DEP_DESCRIPTIONS.get(dep, dep)
try:
reply = input(f"{desc} is not installed. Install now? [Y/n] ").strip().lower()
except (EOFError, KeyboardInterrupt):
@@ -162,24 +136,13 @@ def ensure_dependency(
print(" PowerShell not found. Install PowerShell or run install.ps1 manually.")
return False
cmd = [
ps_bin,
"-ExecutionPolicy", "Bypass",
"-File", str(script),
"-Ensure", dep,
"-HermesHome", str(get_hermes_home()),
ps_bin, "-ExecutionPolicy", "Bypass", "-File", str(script),
"-Ensure", dep, "-HermesHome", str(get_hermes_home()),
]
else:
cmd = ["bash", str(script), "--ensure", dep]
run_env = hermes_subprocess_env(inherit_credentials=False)
run_env["IS_INTERACTIVE"] = "false"
result = subprocess.run(
cmd,
env=run_env,
)
if result.returncode != 0:
return False
if check:
return check()
return True
result = subprocess.run(cmd, env=run_env)
return result.returncode == 0 and check()
+76 -122
View File
@@ -1,28 +1,9 @@
"""Stale git lock-file recovery for update/check paths.
A crashed or killed ``git fetch`` on a shallow clone can leave
``.git/shallow.lock`` behind. Every later fetch then fails with::
A crashed or killed ``git fetch`` on a shallow clone can leave ``.git/shallow.lock`` behind. Every
later fetch then fails with::
fatal: Unable to create '/path/.git/shallow.lock': File exists.
This wedges ``hermes update --check`` (hard failure) and silently degrades the
passive banner check in :mod:`hermes_cli.banner` (the fetch is swallowed, the
stale refs are compared, and the user can be told an update is available when
the checkout already contains the remote tip). Git does not self-heal these
lock files — they persist until a human removes them.
This module provides two small, defensive helpers used by the update paths:
* :func:`clear_stale_git_locks` — remove abandoned ``.git`` lock files (with
an age + git-process guard so a live fetch is never yanked).
* :func:`clear_stale_tmp_packs` — remove aborted-fetch ``tmp_pack_*`` /
``tmp_idx_*`` debris from ``.git/objects/pack``. On flaky lines every
timed-out fetch leaves one behind; unchecked they accumulated to 6 GB /
hundreds of files over 9 days and eventually corrupted the pack directory
outright, permanently wedging the update check (#93732).
* :func:`is_ancestor_of_head` — ask whether a remote tip is already contained
in HEAD. Used by the shallow-clone update check to avoid reporting a false
"update available" when local cherry-picks sit on top of the remote tip.
fatal: Unable to create '/path/.git/shallow.lock': File exists.
"""
from __future__ import annotations
@@ -32,7 +13,7 @@ import os
import subprocess
import time
from pathlib import Path
from typing import List, Optional
from typing import Callable, Iterable, List, Optional
logger = logging.getLogger(__name__)
@@ -49,13 +30,23 @@ STALE_LOCK_MIN_AGE_SECONDS = 10 * 60
# process guard in :func:`clear_stale_git_locks`.
LOCK_NAMES = ("shallow.lock", "index.lock", "HEAD.lock", "MERGE_HEAD.lock")
# Aborted-fetch pack debris younger than this is presumed live (a fetch may
# be writing it right now) and is never removed. A healthy fetch completes in
# minutes; the same 10-minute bar the lock sweep uses is comfortably safe.
STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS
# Temp-file prefixes git writes into .git/objects/pack during a transfer and
# renames away on success. Anything left with these names after a fetch died
# is garbage by definition — git itself never reuses or cleans them.
_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_")
def _git_proc_running() -> bool:
"""True when a ``git`` process is currently running.
The conservative answer on any platform we can't probe: if we can't tell,
treat a lock as possibly-live and don't remove it. This is the safety
check that stops us from yanking a lock a real fetch is holding.
The conservative answer on any platform we can't probe: if we can't tell, treat a lock as
possibly-live and don't remove it. This is the safety check that stops us from yanking a lock a
real fetch is holding.
"""
try:
if os.name == "nt":
@@ -73,50 +64,54 @@ def _git_proc_running() -> bool:
return False
def _sweep_stale(
directory: Path,
candidates: Callable[[], Iterable[Path]],
*,
min_age_seconds: Optional[int],
default_age: int,
skip_msg: str,
log_removed: Callable[[Path, int], None],
) -> List[str]:
"""Shared guard + age-floor sweep. Never raises; skips anything it cannot stat/unlink."""
if not directory.is_dir():
return []
if _git_proc_running():
logger.debug(skip_msg)
return []
cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else default_age)
removed: List[str] = []
for entry in candidates():
try:
if entry.is_file():
st = entry.stat()
if st.st_mtime < cutoff:
entry.unlink()
removed.append(str(entry))
log_removed(entry, st.st_size)
except OSError:
logger.debug("Could not clear %s (skipping)", entry, exc_info=True)
return removed
def clear_stale_git_locks(repo_root: Path, *, min_age_seconds: Optional[int] = None) -> List[str]:
"""Remove abandoned ``.git`` lock files under ``repo_root``.
A lock is removed only when BOTH conditions hold:
* it is older than :data:`STALE_LOCK_MIN_AGE_SECONDS` (default), and
* no ``git`` process is currently running.
Returns the list of removed lock file paths. Never raises: a lock we
cannot stat or unlink is skipped (a concurrently-held lock may have just
been created between our age check and the unlink — the process guard
makes that window vanishingly small, and skipping is always safe).
Returns the list of removed lock file paths. Never raises: a lock we cannot stat or unlink is
skipped (a concurrently-held lock may have just been created between our age check and the
unlink — the process guard makes that window vanishingly small, and skipping is always safe).
"""
git_dir = Path(repo_root) / ".git"
if not git_dir.is_dir():
return []
if _git_proc_running():
logger.debug("git process running; skipping stale-lock sweep")
return []
cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else STALE_LOCK_MIN_AGE_SECONDS)
removed: List[str] = []
for name in LOCK_NAMES:
lock_path = git_dir / name
try:
if lock_path.is_file() and lock_path.stat().st_mtime < cutoff:
lock_path.unlink()
removed.append(str(lock_path))
logger.info("Removed stale git lock %s", lock_path)
except OSError:
logger.debug("Could not clear %s (skipping)", lock_path, exc_info=True)
return removed
# Aborted-fetch pack debris younger than this is presumed live (a fetch may
# be writing it right now) and is never removed. A healthy fetch completes in
# minutes; the same 10-minute bar the lock sweep uses is comfortably safe.
STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS
# Temp-file prefixes git writes into .git/objects/pack during a transfer and
# renames away on success. Anything left with these names after a fetch died
# is garbage by definition — git itself never reuses or cleans them.
_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_")
return _sweep_stale(
git_dir,
lambda: [git_dir / name for name in LOCK_NAMES],
min_age_seconds=min_age_seconds,
default_age=STALE_LOCK_MIN_AGE_SECONDS,
skip_msg="git process running; skipping stale-lock sweep",
log_removed=lambda p, _size: logger.info("Removed stale git lock %s", p),
)
def clear_stale_tmp_packs(
@@ -124,69 +119,28 @@ def clear_stale_tmp_packs(
) -> List[str]:
"""Remove aborted-fetch temp pack files under ``.git/objects/pack``.
Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped
connection) leaves a ``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file
behind, and git never cleans them up. On a flaky line the banner's
background update check produces several per day; observed in the wild
at hundreds of files / 6 GB after 9 days, after which the pack directory
corrupted outright and every fetch failed permanently (#93732).
Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped connection) leaves a
``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file behind, and git never cleans them up.
Same safety contract as :func:`clear_stale_git_locks`: only files older
than the age floor, never while a git process is running, never raises.
Returns the removed paths.
Same safety contract as :func:`clear_stale_git_locks`: only files older than the age floor,
never while a git process is running, never raises. Returns the removed paths.
"""
pack_dir = Path(repo_root) / ".git" / "objects" / "pack"
if not pack_dir.is_dir():
return []
if _git_proc_running():
logger.debug("git process running; skipping tmp-pack sweep")
return []
cutoff = time.time() - (
min_age_seconds if min_age_seconds is not None else STALE_TMP_PACK_MIN_AGE_SECONDS
)
removed: List[str] = []
try:
entries = list(pack_dir.iterdir())
except OSError:
return []
for entry in entries:
name = entry.name
if not name.startswith(_TMP_PACK_PREFIXES):
continue
def _candidates():
try:
if entry.is_file() and entry.stat().st_mtime < cutoff:
size = entry.stat().st_size
entry.unlink()
removed.append(str(entry))
logger.info(
"Removed aborted-fetch pack debris %s (%d bytes)", entry, size
)
entries = list(pack_dir.iterdir())
except OSError:
logger.debug("Could not clear %s (skipping)", entry, exc_info=True)
return removed
return []
return [e for e in entries if e.name.startswith(_TMP_PACK_PREFIXES)]
def is_ancestor_of_head(repo_root: Path, rev: str) -> bool:
"""True when ``rev`` is an ancestor of (or equal to) HEAD.
Wraps ``git merge-base --is-ancestor <rev> HEAD``. This is the correct
question for update checks: a local cherry-pick on top of the remote tip
makes HEAD *different* from ``origin/main`` but still *contains* it, so
the answer to "is there an update?" is no.
Returns False on any probe failure (missing rev, shallow boundary, git
error) — callers treat that as "can't prove contained", which is the
conservative direction for an update check.
"""
try:
result = subprocess.run(
["git", "merge-base", "--is-ancestor", rev, "HEAD"],
cwd=str(repo_root),
capture_output=True, text=True, timeout=10,
)
return result.returncode == 0
except Exception:
logger.debug("merge-base --is-ancestor probe failed for %s", rev, exc_info=True)
return False
return _sweep_stale(
pack_dir,
_candidates,
min_age_seconds=min_age_seconds,
default_age=STALE_TMP_PACK_MIN_AGE_SECONDS,
skip_msg="git process running; skipping tmp-pack sweep",
log_removed=lambda p, size: logger.info(
"Removed aborted-fetch pack debris %s (%d bytes)", p, size
),
)
+48 -103
View File
@@ -1,38 +1,7 @@
"""
Hermes Desktop (Chat GUI) uninstaller.
"""Hermes Desktop (Chat GUI) uninstaller.
The desktop GUI ships in two shapes and this module knows how to find and
remove the artifacts of both, on Linux, macOS, and Windows, WITHOUT touching
the Python agent or the user's config/data:
1. Source-built GUI (``hermes desktop`` / ``hermes gui``)
Built inside the agent checkout under ``$HERMES_HOME/hermes-agent/``:
- ``apps/desktop/dist`` (compiled renderer)
- ``apps/desktop/release`` (electron-builder unpacked app + installers)
- ``apps/desktop/node_modules`` and the workspace-root ``node_modules``
(Electron itself, ~200MB) — only removed on a GUI uninstall because
the agent does not need them.
- ``$HERMES_HOME/desktop-build-stamp.json`` (the build freshness stamp)
2. Packaged distributable (DMG / NSIS / AppImage / deb / rpm)
Installed by the OS to a standard application location and carrying its
own bundled Electron + a per-user Electron ``userData`` directory:
- macOS: ``/Applications/Hermes.app`` or ``~/Applications/Hermes.app``
- Windows: ``%LOCALAPPDATA%\\Programs\\Hermes`` (NSIS per-user)
- Linux: ``~/.local/share/applications`` .desktop entry + AppImage
In both shapes the Electron runtime keeps a ``userData`` directory keyed on
the app name ("Hermes"), separate from ``$HERMES_HOME``:
- macOS: ``~/Library/Application Support/Hermes``
- Windows: ``%APPDATA%\\Hermes``
- Linux: ``$XDG_CONFIG_HOME/Hermes`` (default ``~/.config/Hermes``)
This holds the desktop's own ``connection.json`` / ``updates.json`` and
Chromium cache — pure GUI state, safe to remove on a GUI uninstall.
The functions here are deliberately import-light and side-effect-free at
import time so the Electron main process can shell out to
``hermes uninstall --gui`` (and friends) without paying for the full CLI.
This holds the desktop's own ``connection.json`` / ``updates.json`` and Chromium cache — pure GUI
state, safe to remove on a GUI uninstall.
"""
import os
@@ -57,6 +26,12 @@ def log_warn(msg: str):
print(f"{color('⚠', Colors.YELLOW)} {msg}")
def _env_dir(var: str, fallback: Path) -> Path:
"""``Path($var)`` when the env var is set, else *fallback*."""
value = os.environ.get(var)
return Path(value) if value else fallback
# ---------------------------------------------------------------------------
# Discovery
# ---------------------------------------------------------------------------
@@ -70,29 +45,24 @@ def _agent_root(hermes_home: Path) -> Path:
def desktop_userdata_dir() -> Path:
"""Return the Electron ``userData`` directory for the desktop app.
Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes"
on each platform. This is GUI-only state (connection.json, updates.json,
Chromium cache) and never holds agent config or sessions.
Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes" on each platform. This
is GUI-only state (connection.json, updates.json, Chromium cache) and never holds agent config
or sessions.
"""
home = Path.home()
if sys.platform == "darwin":
return home / "Library" / "Application Support" / "Hermes"
if sys.platform == "win32":
appdata = os.environ.get("APPDATA")
base = Path(appdata) if appdata else (home / "AppData" / "Roaming")
return base / "Hermes"
return _env_dir("APPDATA", home / "AppData" / "Roaming") / "Hermes"
# Linux / other POSIX — XDG config home.
xdg = os.environ.get("XDG_CONFIG_HOME")
base = Path(xdg) if xdg else (home / ".config")
return base / "Hermes"
return _env_dir("XDG_CONFIG_HOME", home / ".config") / "Hermes"
def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]":
"""GUI build artifacts produced by ``hermes desktop`` inside the checkout.
These are removable on a GUI uninstall without harming the agent: the
Python agent runs from ``hermes-agent/`` source + ``venv/`` and never
needs the Electron build output or node_modules.
These are removable on a GUI uninstall without harming the agent: the Python agent runs from
``hermes-agent/`` source + ``venv/`` and never needs the Electron build output or node_modules.
"""
agent_root = _agent_root(hermes_home)
desktop_dir = agent_root / "apps" / "desktop"
@@ -111,9 +81,9 @@ def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]":
def packaged_gui_app_paths() -> "list[Path]":
"""Standard install locations of the packaged desktop distributable.
Returns every candidate for the current OS; the caller filters to those
that actually exist. We never glob system-wide — only the well-known
electron-builder output locations for the "Hermes" product.
Returns every candidate for the current OS; the caller filters to those that actually exist. We
never glob system-wide — only the well-known electron-builder output locations for the "Hermes"
product.
"""
home = Path.home()
paths: list[Path] = []
@@ -123,8 +93,7 @@ def packaged_gui_app_paths() -> "list[Path]":
home / "Applications" / "Hermes.app",
]
elif sys.platform == "win32":
local = os.environ.get("LOCALAPPDATA")
local_base = Path(local) if local else (home / "AppData" / "Local")
local_base = _env_dir("LOCALAPPDATA", home / "AppData" / "Local")
paths += [
# NSIS per-user install (perMachine=false → Programs\Hermes).
local_base / "Programs" / "Hermes",
@@ -144,8 +113,7 @@ def packaged_gui_app_paths() -> "list[Path]":
# ``uninstall_gui``.
from hermes_cli.linux_desktop_entry import desktop_entry_path
data = os.environ.get("XDG_DATA_HOME")
data_base = Path(data) if data else (home / ".local" / "share")
data_base = _env_dir("XDG_DATA_HOME", home / ".local" / "share")
paths += [
# The launcher entry `hermes desktop` installs. Its icon is
# also copied into the hicolor tree (see
@@ -166,52 +134,38 @@ def packaged_gui_app_paths() -> "list[Path]":
def agent_is_installed(hermes_home: Path) -> bool:
"""Return True when a usable Python agent install exists under HERMES_HOME.
Used by the desktop UI to decide which uninstall options to offer: if the
agent isn't present (a future "lite" GUI-only client), the "remove agent"
options are hidden.
Used by the desktop UI to decide which uninstall options to offer: if the agent isn't present (a
future "lite" GUI-only client), the "remove agent" options are hidden.
"""
agent_root = _agent_root(hermes_home)
# A real install has the package source + a venv. Either signal alone is
# enough — a source checkout without a venv is still "the agent is here".
if (agent_root / "hermes_cli").is_dir():
return True
if (agent_root / "venv").is_dir() or (agent_root / ".venv").is_dir():
return True
return False
return any((agent_root / sub).is_dir() for sub in ("hermes_cli", "venv", ".venv"))
def gui_is_installed(hermes_home: Path) -> bool:
"""Return True when any desktop GUI artifact exists (built or packaged)."""
for p in source_built_gui_artifacts(hermes_home):
if p.exists():
return True
for p in packaged_gui_app_paths():
if p.exists():
return True
if desktop_userdata_dir().exists():
return True
return False
return any(
p.exists()
for p in (*source_built_gui_artifacts(hermes_home), *packaged_gui_app_paths(), desktop_userdata_dir())
)
def gui_install_summary(hermes_home: "Path | None" = None) -> dict:
"""Structured snapshot of what's installed, for the desktop UI to render.
Returns JSON-serializable primitives so the Electron main process can
forward it to the renderer via IPC (paths as strings, booleans for the
high-level questions the UI gates options on).
Returns JSON-serializable primitives (paths as strings, booleans for the questions the UI gates
on) so the Electron main process can forward it to the renderer via IPC.
"""
home: Path = hermes_home if hermes_home is not None else get_hermes_home()
source_artifacts = [p for p in source_built_gui_artifacts(home) if p.exists()]
packaged = [p for p in packaged_gui_app_paths() if p.exists()]
userdata = desktop_userdata_dir()
return {
"hermes_home": str(home),
"agent_installed": agent_is_installed(home),
"gui_installed": gui_is_installed(home),
"source_built_artifacts": [str(p) for p in source_artifacts],
"packaged_app_paths": [str(p) for p in packaged],
"source_built_artifacts": [str(p) for p in source_built_gui_artifacts(home) if p.exists()],
"packaged_app_paths": [str(p) for p in packaged_gui_app_paths() if p.exists()],
"userdata_dir": str(userdata),
"userdata_exists": userdata.exists(),
"platform": sys.platform,
@@ -242,45 +196,36 @@ def uninstall_gui(
) -> "list[Path]":
"""Remove the desktop GUI's artifacts, leaving the agent + user data intact.
Removes:
- source-built GUI artifacts (dist/release/node_modules/build-stamp)
- the packaged app bundle / install dir (best-effort; deb/rpm need the
system package manager and are reported, not force-removed)
- the Electron ``userData`` directory (unless ``remove_userdata=False``)
Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any
config / sessions / .env under ``$HERMES_HOME``.
Returns the list of paths actually removed.
Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any config / sessions /
.env under ``$HERMES_HOME``.
"""
home: Path = hermes_home if hermes_home is not None else get_hermes_home()
removed: list[Path] = []
def _remove_existing(paths) -> bool:
"""Remove every existing path; True when at least one existed."""
found = False
for path in paths:
if path.exists():
found = True
if _remove_path(path):
log_success(f"Removed {path}")
removed.append(path)
return found
log_info("Removing built GUI artifacts (renderer, release, node_modules)...")
for path in source_built_gui_artifacts(home):
if path.exists() and _remove_path(path):
log_success(f"Removed {path}")
removed.append(path)
_remove_existing(source_built_gui_artifacts(home))
log_info("Removing installed desktop app...")
found_packaged = False
for path in packaged_gui_app_paths():
if path.exists():
found_packaged = True
if _remove_path(path):
log_success(f"Removed {path}")
removed.append(path)
if not found_packaged:
if not _remove_existing(packaged_gui_app_paths()):
log_info("No packaged desktop app found in standard locations")
if remove_userdata:
userdata = desktop_userdata_dir()
if userdata.exists():
log_info("Removing desktop app data (Electron userData)...")
if _remove_path(userdata):
log_success(f"Removed {userdata}")
removed.append(userdata)
_remove_existing([userdata])
if not removed:
log_info("No desktop GUI artifacts found to remove")
+57 -99
View File
@@ -1,29 +1,12 @@
"""Session heartbeats — recurring re-entry prompts for the current session.
A heartbeat is one user-owned recurring instruction bound to a session
(`/heartbeat every 10m Check the deployment and report meaningful changes`).
When due AND the session is idle, the prompt is injected as a normal user
turn — same mechanism as a /goal continuation, so message-role alternation
and prompt caching are untouched. If the agent is busy at the due moment,
the tick coalesces: it fires once when the session next goes idle, never
stacking a backlog.
This is deliberately session-scoped and in-process (CLI process or gateway process must be running)
— the durable cross-process scheduling surface remains ``hermes cron`` / the ``cronjob`` tool, which
runs in isolated sessions.
This is deliberately session-scoped and in-process (CLI process or gateway
process must be running) — the durable cross-process scheduling surface
remains ``hermes cron`` / the ``cronjob`` tool, which runs in isolated
sessions. A heartbeat is for "keep re-entering THIS conversation", the
cron system is for "run this job on a schedule". Distinct by design.
State is persisted in SessionDB ``state_meta`` keyed by
``heartbeat:<session_id>`` so ``/resume`` picks it up.
Invariants (mirrors goals.py):
- Injection is a plain user message. No system-prompt mutation, no toolset
swap — prompt caching stays intact.
- A real user message always wins: heartbeats only fire into an idle
session with an empty input queue.
- Failures are contained: any DB/import error degrades to "no heartbeat",
never to a crashed input loop.
Invariants (mirrors goals.py): - Injection is a plain user message. No system-prompt mutation, no
toolset swap — prompt caching stays intact. - A real user message always wins: heartbeats only fire
into an idle session with an empty input queue.
"""
from __future__ import annotations
@@ -32,8 +15,8 @@ import json
import logging
import re
import time
from dataclasses import dataclass, asdict
from typing import Any, Dict, Optional
from dataclasses import asdict, dataclass
from typing import Any, Optional
logger = logging.getLogger(__name__)
@@ -68,32 +51,22 @@ _UNIT_SECONDS = {
def parse_interval(text: str) -> Optional[int]:
"""Parse ``10m`` / ``every 2h`` / ``every 90 minutes`` into seconds.
Returns None when the text is not an interval. Values below
``MIN_INTERVAL_SECONDS`` are rejected (returns -1 so callers can
distinguish "not an interval" from "too small").
Returns None when the text is not an interval; values below ``MIN_INTERVAL_SECONDS`` return
-1 so callers can distinguish "not an interval" from "too small".
"""
if not text:
return None
m = _INTERVAL_RE.match(text)
m = _INTERVAL_RE.match(text) if text else None
if not m:
return None
value = float(m.group(1))
unit = m.group(2).lower()
seconds = int(value * _UNIT_SECONDS[unit])
if seconds < MIN_INTERVAL_SECONDS:
return -1
return seconds
seconds = int(float(m.group(1)) * _UNIT_SECONDS[m.group(2).lower()])
return -1 if seconds < MIN_INTERVAL_SECONDS else seconds
def format_interval(seconds: int) -> str:
"""Human-readable interval (``600`` → ``10m``)."""
seconds = int(seconds)
if seconds % 86400 == 0:
return f"{seconds // 86400}d"
if seconds % 3600 == 0:
return f"{seconds // 3600}h"
if seconds % 60 == 0:
return f"{seconds // 60}m"
for unit, suffix in ((86400, "d"), (3600, "h"), (60, "m")):
if seconds % unit == 0:
return f"{seconds // unit}{suffix}"
return f"{seconds}s"
@@ -114,14 +87,11 @@ class HeartbeatState:
@classmethod
def from_json(cls, raw: str) -> "HeartbeatState":
data = json.loads(raw)
return cls(
prompt=str(data.get("prompt") or ""),
interval_seconds=int(data.get("interval_seconds", 0) or 0),
status=str(data.get("status") or "active"),
created_at=float(data.get("created_at", 0.0) or 0.0),
last_fired_at=float(data.get("last_fired_at", 0.0) or 0.0),
fire_count=int(data.get("fire_count", 0) or 0),
)
# Falsy/missing values fall back to the default before type coercion.
return cls(**{
name: coerce(data.get(name) or default)
for name, (coerce, default) in _STATE_FIELDS.items()
})
def is_due(self, now: Optional[float] = None) -> bool:
if self.status != "active" or not self.prompt or self.interval_seconds <= 0:
@@ -137,9 +107,18 @@ class HeartbeatState:
)
# ──────────────────────────────────────────────────────────────────────
# field -> (coercer, default used when the stored value is missing/falsy)
_STATE_FIELDS = {
"prompt": (str, ""),
"interval_seconds": (int, 0),
"status": (str, "active"),
"created_at": (float, 0.0),
"last_fired_at": (float, 0.0),
"fire_count": (int, 0),
}
# Persistence (SessionDB state_meta) — same pattern as goals.py
# ──────────────────────────────────────────────────────────────────────
def _meta_key(session_id: str) -> str:
@@ -159,9 +138,7 @@ def _get_session_db() -> Optional[Any]:
def load_heartbeat(session_id: str) -> Optional[HeartbeatState]:
if not session_id:
return None
db = _get_session_db()
db = _get_session_db() if session_id else None
if db is None:
return None
try:
@@ -194,18 +171,15 @@ def save_heartbeat(session_id: str, state: HeartbeatState) -> None:
logger.debug("HeartbeatManager: set_meta failed: %s", exc)
# ──────────────────────────────────────────────────────────────────────
# Manager — the surface CLI + gateway talk to
# ──────────────────────────────────────────────────────────────────────
class HeartbeatManager:
"""Per-session heartbeat state + due-tick decisions.
Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll
cadence while the session is idle; a non-None return is the user-role
message to inject. Firing is recorded immediately so a slow turn can't
double-fire.
Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll cadence while the session
is idle; a non-None return is the user-role message to inject. Firing is recorded immediately so
a slow turn can't double-fire.
"""
def __init__(self, session_id: str):
@@ -232,9 +206,8 @@ class HeartbeatManager:
anchor = s.last_fired_at or s.created_at
next_in = max(0, int(anchor + s.interval_seconds - time.time()))
return f"♥ Heartbeat (every {every}, next in ~{next_in}s{fired}): {s.prompt}"
if s.status == "paused":
return f"⏸ Heartbeat (paused, every {every}{fired}): {s.prompt}"
return f"Heartbeat ({s.status}, every {every}{fired}): {s.prompt}"
icon = "⏸ " if s.status == "paused" else ""
return f"{icon}Heartbeat ({s.status}, every {every}{fired}): {s.prompt}"
# --- mutation -----------------------------------------------------
@@ -246,36 +219,31 @@ class HeartbeatManager:
if interval_seconds < MIN_INTERVAL_SECONDS:
raise ValueError(f"interval must be at least {MIN_INTERVAL_SECONDS}s")
state = HeartbeatState(
prompt=prompt,
interval_seconds=interval_seconds,
status="active",
created_at=time.time(),
prompt=prompt, interval_seconds=interval_seconds, status="active", created_at=time.time()
)
self._state = state
save_heartbeat(self.session_id, state)
return state
def pause(self) -> Optional[HeartbeatState]:
def _set_status(self, status: str, *, reanchor: bool = False) -> Optional[HeartbeatState]:
if not self._state:
return None
self._state.status = "paused"
self._state.status = status
if reanchor:
self._state.last_fired_at = time.time()
save_heartbeat(self.session_id, self._state)
return self._state
def pause(self) -> Optional[HeartbeatState]:
return self._set_status("paused")
def resume(self) -> Optional[HeartbeatState]:
if not self._state:
return None
self._state.status = "active"
# Re-anchor so resuming doesn't instantly fire a stale tick.
self._state.last_fired_at = time.time()
save_heartbeat(self.session_id, self._state)
return self._state
return self._set_status("active", reanchor=True)
def clear(self) -> bool:
if self._state is None:
if self._set_status("cleared") is None:
return False
self._state.status = "cleared"
save_heartbeat(self.session_id, self._state)
self._state = None
return True
@@ -284,10 +252,9 @@ class HeartbeatManager:
def due_prompt(self, now: Optional[float] = None) -> Optional[str]:
"""Return the injection prompt if the heartbeat is due, else None.
Records the fire immediately (before the turn runs) so overlapping
polls or a long turn can never double-fire the same tick. Missed
ticks coalesce into one — the anchor resets to NOW, not to the
theoretical schedule.
Records the fire immediately (before the turn runs) so overlapping polls or a long turn can
never double-fire the same tick. Missed ticks coalesce into one — the anchor resets to NOW,
not to the theoretical schedule.
"""
s = self._state
if s is None or not s.is_due(now):
@@ -301,16 +268,14 @@ class HeartbeatManager:
def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bool:
"""Carry a heartbeat across a compression session rotation.
Same shape as ``goals.migrate_goal_to_session`` — copy to the child,
archive the parent row, never raise.
Same shape as ``goals.migrate_goal_to_session`` — copy to the child, archive the parent row,
never raise.
"""
if not old_session_id or not new_session_id or old_session_id == new_session_id:
return False
try:
state = load_heartbeat(old_session_id)
if state is None:
return False
if load_heartbeat(new_session_id) is not None:
if state is None or load_heartbeat(new_session_id) is not None:
return False
save_heartbeat(new_session_id, state)
state.status = "cleared"
@@ -322,14 +287,7 @@ def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bo
__all__ = [
"HeartbeatState",
"HeartbeatManager",
"parse_interval",
"format_interval",
"load_heartbeat",
"save_heartbeat",
"migrate_heartbeat_to_session",
"HEARTBEAT_PROMPT_TEMPLATE",
"MIN_INTERVAL_SECONDS",
"POLL_SECONDS",
"HeartbeatState", "HeartbeatManager", "parse_interval", "format_interval",
"load_heartbeat", "save_heartbeat", "migrate_heartbeat_to_session",
"HEARTBEAT_PROMPT_TEMPLATE", "MIN_INTERVAL_SECONDS", "POLL_SECONDS",
]
+19 -32
View File
@@ -1,14 +1,12 @@
"""Image-authored deployment provenance for immutable Hermes runtimes.
The published image bakes ``/etc/hermes/image-provenance.json`` outside both
``$HERMES_HOME`` and the mutable checkout. A bind-mounted checkout (including
``.git``) therefore cannot hide the build fact, and environment or config
values cannot forge it.
The published image bakes ``/etc/hermes/image-provenance.json`` outside both ``$HERMES_HOME`` and
the mutable checkout. A bind-mounted checkout (including ``.git``) therefore cannot hide the build
fact, and environment or config values cannot forge it.
Absence preserves every pre-existing source/package install path. Presence
fails closed: an unreadable, non-regular, or malformed marker still means the
runtime is image-managed; it is an integrity defect, never permission to
mutate the image in place.
Absence preserves every pre-existing source/package install path. Presence fails closed: an
unreadable, non-regular, or malformed marker still means the runtime is image-managed; it is an
integrity defect, never permission to mutate the image in place.
"""
from __future__ import annotations
@@ -55,20 +53,22 @@ def _invalid(path: Path, reason: str) -> ImageProvenance:
)
def _optional_string(payload: dict, name: str) -> Optional[str]:
value = payload.get(name)
if value is None:
return None
if not isinstance(value, str):
raise TypeError(name)
return value.strip() or None
def read_image_provenance(
marker_path: Optional[Path] = None,
) -> Optional[ImageProvenance]:
"""Read the baked marker without consulting environment or config.
``None`` has one precise meaning: ``lstat`` proved that no marker exists.
Every other filesystem or validation failure returns an invalid
:class:`ImageProvenance`, so callers refuse image mutation closed. In
particular, ``lstat`` makes a dangling symlink visibly *present* and the
regular-file check rejects symlinks, directories, and device nodes.
``marker_path`` is a dependency-injection seam for tests and alternate
image builders. Normal callers always use the image-owned absolute path.
This function never raises.
``marker_path`` is a dependency-injection seam for tests and alternate image builders. Normal
callers always use the image-owned absolute path. This function never raises.
"""
path = IMAGE_PROVENANCE_PATH
@@ -110,19 +110,8 @@ def read_image_provenance(
if not isinstance(manager, str) or not manager.strip():
return _invalid(path, "missing_manager")
def _optional_string(name: str) -> Optional[str]:
value = payload.get(name)
if value is None:
return None
if not isinstance(value, str):
raise TypeError(name)
value = value.strip()
return value or None
try:
image = _optional_string("image")
version = _optional_string("version")
revision = _optional_string("revision")
optional = {name: _optional_string(payload, name) for name in ("image", "version", "revision")}
except TypeError as exc:
return _invalid(path, f"invalid_{exc.args[0]}")
@@ -130,8 +119,6 @@ def read_image_provenance(
schema=IMAGE_PROVENANCE_SCHEMA,
deployment_kind="image",
manager=manager.strip(),
image=image,
version=version,
revision=revision,
marker_path=str(path),
**optional,
)
+2 -2
View File
@@ -74,8 +74,8 @@ def _fsync_directory(path: Path) -> None:
def read_or_create_install_id(root: Path | None = None) -> Optional[str]:
"""Read or atomically mint the opaque id for the physical install.
``None`` means the id could neither be read nor persisted. Returning an
ephemeral id would violate the authority and connection-registry contract.
``None`` means the id could neither be read nor persisted. Returning an ephemeral id would
violate the authority and connection-registry contract.
"""
root = get_default_hermes_root() if root is None else root
path = root / _INSTALL_ID_FILENAME
+7 -8
View File
@@ -8,8 +8,7 @@ from typing import Any, List
logger = logging.getLogger(__name__)
def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
"""Notify first-party observers, then invoke compatibility plugin hooks."""
def _observe(hook_name: str, **kwargs: Any) -> None:
try:
from hermes_cli.observability import observe_lifecycle
@@ -17,6 +16,11 @@ def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
except Exception:
logger.warning("Built-in observability hook failed", exc_info=True)
def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
"""Notify first-party observers, then invoke compatibility plugin hooks."""
_observe(hook_name, **kwargs)
from hermes_cli import plugins
return plugins.invoke_hook(hook_name, **kwargs)
@@ -39,12 +43,7 @@ def has_hook(hook_name: str) -> bool:
def finalize_session(**kwargs: Any) -> List[Any]:
"""Notify observers and hard-close one core-owned Relay conversation."""
try:
from hermes_cli.observability import observe_lifecycle
observe_lifecycle("on_session_finalize", **kwargs)
except Exception:
logger.warning("Built-in observability hook failed", exc_info=True)
_observe("on_session_finalize", **kwargs)
session_id = str(kwargs.get("session_id") or "")
if session_id:
+61 -159
View File
@@ -1,26 +1,10 @@
"""Install and remove the Linux desktop entry (``hermes.desktop``).
``hermes desktop`` builds and launches the Electron app. On Linux, a
freshly-built app has no launcher presence: no menu item, no icon. This
module writes the XDG desktop entry that gives it one.
``hermes uninstall --gui`` removes the entry again.
Two values must be absolute for the entry to work:
- ``Exec`` — the launcher runs without shell ``PATH`` customizations, so
a bare ``hermes desktop`` fails when hermes lives in ``~/.local/bin``
or a venv. Resolve the real binary and write its full path.
- ``Icon`` — an unqualified icon name needs an indexed icon theme. The
spec allows an absolute path instead, so point at the app icon in the
checkout. Do not copy the icon: ``Exec`` already depends on that tree.
Cache refresh is best-effort and tool-gated: ``update-desktop-database``
for the freedesktop menu cache, ``gtk-update-icon-cache`` for the user
hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5`` for Plasma. Run
each tool only when it exists. A missing tool is not an error.
Import-light and side-effect-free at import time: the uninstaller uses
this without loading the full CLI.
Cache refresh is best-effort and tool-gated: ``update-desktop-database`` for the freedesktop menu
cache, ``gtk-update-icon-cache`` for the user hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5``
for Plasma. Run each tool only when it exists. A missing tool is not an error.
"""
from __future__ import annotations
@@ -60,23 +44,12 @@ def icon_path(project_root: Path) -> Path:
def _running_interpreter() -> str:
"""The venv-semantic interpreter path for the persisted ``Exec=`` line.
``sys.executable`` inside a venv is commonly a SYMLINK into a shared
base-interpreter tree (uv, pyenv, conda). ``Path.resolve()`` follows it
out of the venv, and CPython discovers ``pyvenv.cfg`` from the
*lexical* argv[0] — so a dereferenced path boots without the venv's
site-packages and dies on the first third-party import (#90292, one
level up; identified in #80547's review and confirmed on real Zorin/uv
hardware in this PR's review).
Keep the lexical path only when it actually is venv-semantic (a
``pyvenv.cfg`` sits at or above it in the tree); otherwise the
dereferenced absolute path is the more durable form (survives the
symlink being re-pointed or its parent moving).
Idea credit: the lexical-preservation rule was independently proposed
in #92516/#94115/#94544 and by nosliwhtes' review of this PR; the
pyvenv.cfg-detection refinement here keeps both properties.
"""The venv-semantic interpreter path for the persisted ``Exec=`` line. ``sys.executable`` inside a
venv is commonly a SYMLINK into a shared base-interpreter tree (uv, pyenv, conda).
``Path.resolve()`` follows it out of the venv, and CPython discovers ``pyvenv.cfg`` from the
*lexical* argv[0] — so a dereferenced path boots without the venv's site-packages and dies on
the first third-party import (#90292, one level up; identified in #80547's review and confirmed
on real Zorin/uv hardware in this PR's review).
"""
lexical = os.path.abspath(sys.executable)
path = Path(lexical)
@@ -92,18 +65,10 @@ _probe_cache: "dict[str, bool]" = {}
def _can_import_hermes_cli(interpreter: Path) -> bool:
"""Whether *interpreter* can import ``hermes_cli.main`` unaided.
Runs the import in a subprocess under ``-I`` (isolated mode: no
user site, no PYTHONPATH inheritance, no cwd on ``sys.path``) from
a neutral cwd, so the answer matches what a cold desktop
environment would get — a checkout cwd or an inherited
``PYTHONPATH`` cannot produce a false positive. Bounded by a
timeout so a hung interpreter cannot stall entry generation.
Result is cached per interpreter path for the process lifetime, so
a desktop launch pays the subprocess cost at most once.
Probe design per @nosliwhtes' isolated-mode capability check
(#92122 lineage, commit 4150501f641).
Runs the import in a subprocess under ``-I`` (isolated mode: no user site, no PYTHONPATH
inheritance, no cwd on ``sys.path``) from a neutral cwd, so the answer matches what a cold
desktop environment would get — a checkout cwd or an inherited ``PYTHONPATH`` cannot produce a
false positive.
"""
key = str(interpreter)
cached = _probe_cache.get(key)
@@ -134,9 +99,9 @@ def _can_import_hermes_cli(interpreter: Path) -> bool:
def _running_interpreter_fallback() -> str:
"""The interpreter to persist when the candidate fails the import probe.
The RUNNING interpreter by definition has ``hermes_cli`` importable
(this module is executing), so the module-form entry under it is the
safe landing when every candidate path failed the capability check.
The RUNNING interpreter by definition has ``hermes_cli`` importable (this module is executing),
so the module-form entry under it is the safe landing when every candidate path failed the
capability check.
"""
return os.path.abspath(sys.executable)
@@ -144,22 +109,11 @@ def _running_interpreter_fallback() -> str:
def resolve_exec_command(project_root: Optional[Path] = None) -> str:
"""Build the absolute ``Exec=`` command line for ``hermes desktop``.
Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes
runs as a module with no launcher installed, use the current
interpreter, also absolute.
Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes runs as a module with no
launcher installed, use the current interpreter, also absolute.
The persisted entry must be launch-context independent: whatever
process writes it, the next launch must read and rewrite the same
bytes. ``resolve_hermes_bin()`` prefers ``sys.argv[0]``, which differs
per launch path (wrapper, repo script, ``python -m``), so for this
one caller an argv[0] that points inside the checkout is not a
durable installed launcher — skip it and resolve from PATH instead.
Otherwise a broken entry keeps regenerating itself (the repo-script
form pins a mutable uv interpreter path; the ``python -m`` form
persists a bare ``<python> desktop`` that no DE can run).
``project_root`` pins which checkout counts as "internal"; defaults to
the running checkout.
The persisted entry must be launch-context independent: whatever process writes it, the next
launch must read and rewrite the same bytes.
"""
from hermes_cli.relaunch import resolve_hermes_bin
@@ -209,17 +163,11 @@ def _resolve_hermes_bin_for_desktop_entry(
) -> Optional[str]:
"""Resolve the launcher binary for the persisted ``.desktop`` entry.
Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one
desktop-entry-specific rule: an ``argv[0]`` that points inside this
checkout is a launch-context artifact (the repo ``hermes`` script the
wrapper execs with, or an interpreter binary surfaced by programmatic
relaunch paths), not a durable installed launcher. Persisting it makes
the entry a function of however the previous launch happened — the
bootstrap loop behind #90492's incomplete fix. Skip argv[0]/relative
candidates in that case and fall through to PATH, where the shell
installer's wrapper lives.
``resolve_fn`` is injectable for tests.
Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one rule: an ``argv[0]`` inside
this checkout is a launch-context artifact, not a durable installed launcher — persisting it
makes the entry depend on how the previous launch happened (a bootstrap loop). Skip
argv[0]/relative candidates then and fall through to PATH, where the shell installer's
wrapper lives. ``resolve_fn`` is injectable for tests.
"""
if resolve_fn is None:
from hermes_cli.relaunch import resolve_hermes_bin as resolve_fn
@@ -271,12 +219,9 @@ def _resolve_hermes_bin_for_desktop_entry(
def _is_interpreter(candidate: Path) -> bool:
"""A python interpreter binary (``bin/python*``), not a launcher.
Strict basename match — accepts ``python``, ``python3``,
``python3.11``, ``python2.7``; rejects lookalikes such as
``python3-config``, ``pythonw``, and anything else merely
*containing* "python". Regex approach proposed independently in
#94051; kept here with the parent-dir guard so a script named
``python`` outside a bin/Scripts tree is not misclassified.
Strict basename match — accepts ``python``, ``python3``, ``python3.11``; rejects lookalikes
such as ``python3-config`` and ``pythonw``. The parent-dir guard keeps a script named
``python`` outside a bin/Scripts tree from being misclassified.
"""
import re
@@ -353,13 +298,10 @@ def _resolve_hermes_bin_for_desktop_entry(
def _wrapper_shebang_safe(wrapper: Path) -> bool:
"""Whether an executable wrapper can actually run in the DE context.
A wrapper whose own shebang escapes the venv (``#!/usr/bin/env
python3`` or a bare interpreter name) would die exactly like the
broken entry this module exists to fix — the checkout reference in
its body does not save it. Native binaries and shell launchers are
safe by construction (they exec the right interpreter themselves).
A python-shebang wrapper is safe only when its interpreter resolves
to the RUNNING venv's interpreter directory.
A wrapper whose shebang escapes the venv (``#!/usr/bin/env python3`` or a bare interpreter
name) would die exactly like the broken entry this module fixes. Native binaries and shell
launchers are safe by construction; a python-shebang wrapper is safe only when its
interpreter resolves to the RUNNING venv's interpreter directory.
"""
try:
with open(wrapper, "rb") as fh:
@@ -406,20 +348,10 @@ def _wrapper_shebang_safe(wrapper: Path) -> bool:
def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool:
"""Whether a candidate launcher script actually launches THIS checkout.
Expects the LEXICAL checkout root (the caller keeps it un-resolved):
the installer writes ``$INSTALL_DIR`` lexically into the shim, so on a
symlinked home the shim text and the resolved root would never match.
Both lexical and resolved forms of the root are tried regardless, to
Expects the LEXICAL checkout root (the caller keeps it un-resolved): the installer writes
``$INSTALL_DIR`` lexically into the shim, so on a symlinked home the shim text and the resolved
root would never match. Both lexical and resolved forms of the root are tried regardless, to
tolerate either caller convention.
The installer's shim is a small bash script that execs
``<checkout>/venv/bin/python <checkout>/hermes``; a venv console
script carries the venv interpreter in its shebang. Either way, a
text launcher belonging to this installation references the
checkout path (or its venv) somewhere in its first few KB. A
binary launcher (PyInstaller & friends) cannot be inspected that
way — accept it, since binary installs are self-contained and the
external-primary-first rule has already had its say.
"""
try:
head = wrapper.read_bytes()[:4096]
@@ -467,9 +399,8 @@ def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool:
def _known_wrapper_candidates():
"""Durable installed-launcher locations, most likely first.
Mirrors the installer's ``get_command_link_dir()`` layouts: user
(``~/.local/bin``), root FHS (``/usr/local/bin``), and Termux
(``$PREFIX/bin``). The wrapper is always named ``hermes``.
Mirrors the installer's ``get_command_link_dir()`` layouts: user (``~/.local/bin``), root FHS
(``/usr/local/bin``), and Termux (``$PREFIX/bin``). The wrapper is always named ``hermes``.
"""
candidates = []
home = Path.home()
@@ -485,9 +416,9 @@ def _known_wrapper_candidates():
def _project_root() -> Path:
"""This file lives at ``<checkout>/hermes_cli/linux_desktop_entry.py``.
Lexical (no .resolve()): callers feed this into shim-text matching
where the installer's lexically-written $INSTALL_DIR must be able to
match; symlinked homes would break a resolved comparison.
Lexical (no .resolve()): callers feed this into shim-text matching where the installer's
lexically-written $INSTALL_DIR must be able to match; symlinked homes would break a resolved
comparison.
"""
return Path(os.path.abspath(__file__)).parent.parent
@@ -515,30 +446,15 @@ def _needs_interpreter(bin_path: Path) -> bool:
def _shebang_escapes_running_env(shebang: str) -> bool:
"""Whether a python shebang resolves OUTSIDE the running interpreter's env.
Tokenizes the shebang (interpreter path plus any flags) and compares
PATH COMPONENTS, never substrings: ``<venv>/bin-extra/python`` is not
inside ``<venv>/bin`` even though it starts with it (sibling-directory
confusion; independently surfaced in nosliwhtes' #92122 hardening
Tokenizes the shebang (interpreter path plus any flags) and compares PATH COMPONENTS, never
substrings: ``<venv>/bin-extra/python`` is not inside ``<venv>/bin`` even though it starts with
it (sibling-directory confusion; independently surfaced in nosliwhtes' #92122 hardening
``b96427d0`` — reimplemented here with two extensions).
Extensions over the parent-equality form:
* ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env``
resolves through PATH, which in the DE's cold environment is not the
interactive PATH that installed the venv — the parent-equality form
could be fooled when the resolved ``env`` binary happens to sit in
the same directory tree.
* Flags after the interpreter (``-S``, ``-E``...) are stripped before
comparing, so a legitimate ``#!<venv>/bin/python -S`` is not
misclassified by comparing against the flag token.
The comparison uses the LEXICAL interpreter directory (abspath, not
resolve()): on uv venvs the resolved parent is the base interpreter's
dir, which makes a valid ``.venv/bin/python`` shebang look foreign
(#94443 review case 1). Both sides use the SAME case operation
(``.lower()``): interpreter paths legitimately carry uppercase (conda
env names, usernames, uv's ephemeral build dirs) and an asymmetric
compare would flag the venv's own console script as foreign.
* ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env`` resolves through PATH,
which in the DE's cold environment is not the interactive PATH that installed the venv — the
parent-equality form could be fooled when the resolved ``env`` binary happens to sit in the same
directory tree.
"""
tokens = shebang[2:].strip().split()
if not tokens:
@@ -560,11 +476,7 @@ def _shebang_escapes_running_env(shebang: str) -> bool:
def _quote_exec_arg(arg: str) -> str:
"""Quote one ``Exec`` argument per the desktop entry spec.
Reserved characters require double quotes. Inside the quotes, escape
a backslash and a double quote with a backslash.
"""
"""Quote one ``Exec`` argument per the desktop entry spec."""
if not any(c in arg for c in " \t\n\"'\\><~|&;$*?#()`"):
return arg
escaped = arg.replace("\\", "\\\\").replace('"', '\\"')
@@ -588,16 +500,12 @@ def render_desktop_entry(exec_command: str, icon: str) -> str:
def refresh_desktop_databases(applications_dir: Path) -> "list[str]":
"""Reindex the menu caches. Run each tool only when it exists.
Return the names of the tools that ran (for logging and tests).
"""
"""Reindex the menu caches. Run each tool only when it exists."""
ran: list[str] = []
update_db = shutil.which("update-desktop-database")
if update_db:
if _run_quiet([update_db, str(applications_dir)]):
ran.append("update-desktop-database")
if update_db and _run_quiet([update_db, str(applications_dir)]):
ran.append("update-desktop-database")
# Plasma 6 first, then Plasma 5. Only one of them is ever installed.
for tool in ("kbuildsycoca6", "kbuildsycoca5"):
@@ -664,8 +572,8 @@ def _hicolor_icon_dest(subdir: str) -> Path:
def _remove_stale_scalable_icon() -> bool:
"""Drop a leftover PNG from ``scalable/`` (the pre-fix install path).
Return True when a file was removed so the caller can refresh the
icon cache. A missing file is not an error.
Return True when a file was removed so the caller can refresh the icon cache. A missing file is
not an error.
"""
stale = _hicolor_icon_dest("scalable")
try:
@@ -690,9 +598,9 @@ def _refresh_hicolor_cache() -> None:
def _resized_hicolor_pngs(raw: bytes) -> Optional[dict[str, bytes]]:
"""Lanczos-resize *raw* to each panel size. ``None`` when it will not decode.
Pillow is a core dep but this module stays import-light: the import is
local so the uninstaller does not pay it. A truncated/fake PNG (tests,
interrupted copy) returns None and the caller falls back to a copy.
Pillow is a core dep but this module stays import-light: the import is local so the uninstaller
does not pay it. A truncated/fake PNG (tests, interrupted copy) returns None and the caller
falls back to a copy.
"""
try:
from PIL import Image
@@ -728,15 +636,9 @@ def _write_hicolor_pngs(files: dict[str, bytes]) -> bool:
def _install_icon_to_hicolor(icon: Path) -> bool:
"""Install the app icon into the user's hicolor icon theme tree.
The freedesktop icon lookup finds an installed ``apps/hermes.png``
by the unqualified name ``hermes``, so the entry can reference the
icon without an absolute checkout path. Raster PNGs go to indexed
fixed-size dirs, never ``scalable`` (SVG-only). When the source
decodes, it is Lanczos-resized to 24/32/48/256 so Cinnamon's panel
does not nearest-neighbor a 1024px PNG. Undecodable bytes fall back
to a copy into one indexed dir. Idempotent via content-compare;
OSError caught internally (False) — the caller then falls back to
the absolute path.
The freedesktop icon lookup finds an installed ``apps/hermes.png`` by the unqualified name
``hermes``, so the entry can reference the icon without an absolute checkout path. Raster PNGs
go to indexed fixed-size dirs, never ``scalable`` (SVG-only).
"""
try:
raw = icon.read_bytes()
@@ -762,8 +664,8 @@ def _install_icon_to_hicolor(icon: Path) -> bool:
def install_desktop_entry(project_root: Path) -> Optional[Path]:
"""Write (or refresh) the Hermes desktop entry. Return its path.
Return ``None`` on non-Linux platforms or when the write fails. This
is a convenience, never a reason to fail a launch.
Return ``None`` on non-Linux platforms or when the write fails. This is a convenience, never a
reason to fail a launch.
"""
if not is_supported():
return None
+27 -64
View File
@@ -1,40 +1,11 @@
"""Stable macOS TCC anchor for the uv-managed Python interpreter (#95596).
Re-land of the interpreter anchor reverted in #95563. macOS keys TCC grants
to the resolved absolute path of the client binary. Hermes' interpreter is
managed by uv and lives at a versioned store path; every patch bump orphans
every prior grant (#85345).
1. Aliases are materialized as real-file copies of the anchor, never symlinks. 2. If the store ships
``libpython*``, it is hardlinked into ``venv/lib/`` (copy if the store is on another device).
Existing ``LC_RPATH`` already points at ``@executable_path/../lib`` — no rewrite. 3.
The first landing copied the interpreter into ``venv/bin/python`` but left
two holes that bricked real Macs:
* Dynamically-linked builds look up ``libpython`` via
``@executable_path/../lib``. That resolved into ``venv/lib/``, which had
no dylib — every hermes command, including update/doctor, died in dyld
(#95425).
* Alias names (``python3``, ``python3.N``) were re-pointed at the copy as
*symlinks*. Invoking the copied interpreter through a symlink makes
CPython getpath lose the venv prefix on affected python-build-standalone
builds — startup dies with ``ModuleNotFoundError: encodings`` and the
stdlib resolves to the build-time ``/install`` prefix (#95541). Console
scripts exec ``python3``, so the entire CLI surface died.
This re-land keeps the copy + identifier-pinned signature (TCC attribution
stays on the stable venv path) and closes both holes:
1. Aliases are materialized as real-file copies of the anchor, never
symlinks.
2. If the store ships ``libpython*``, it is hardlinked into ``venv/lib/``
(copy if the store is on another device). Existing ``LC_RPATH`` already
points at ``@executable_path/../lib`` — no rewrite.
3. A pre-install boot gate actually launches the staged copy and demands
``import encodings`` plus ``sys.prefix == <venv>``. Failure rolls the
staging file back and leaves the live interpreter untouched (a surplus
provisioned dylib in ``venv/lib/`` may remain — harmless), so a bad
anchor can never brick update/doctor again.
All functions are no-ops on non-macOS and for interpreters that are not
uv-managed. Best-effort: never raises to callers.
All functions are no-ops on non-macOS and for interpreters that are not uv-managed. Best-effort:
never raises to callers.
"""
from __future__ import annotations
@@ -68,9 +39,9 @@ class _BootGateFailed(Exception):
def _marker_value(source_file: Path) -> str:
"""Canonical marker value: fully resolved so symlinked spellings of the
same store binary (``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``)
compare equal."""
"""Canonical marker value: fully resolved so symlinked spellings of the same store binary
(``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``) compare equal.
"""
return os.path.realpath(str(source_file))
@@ -167,9 +138,9 @@ def _anchor_marker(venv_bin: Path) -> Path:
def _write_marker(venv_bin: Path, source_file: Path) -> None:
"""Write the anchor marker atomically via the shared helper.
A concurrent ensure (update + doctor --fix) must never observe a
partially-written marker: a torn read would compare unequal and trigger
a spurious reinstall, and ``write_text`` alone is not atomic.
A concurrent ensure (update + doctor --fix) must never observe a partially-written marker: a
torn read would compare unequal and trigger a spurious reinstall, and ``write_text`` alone is
not atomic.
"""
atomic_write_text(
_anchor_marker(venv_bin),
@@ -188,8 +159,8 @@ def _provision_libpython(
) -> None:
"""Hardlink (else copy) store ``libpython*`` into ``venv/lib/``.
Provision-if-present: a surplus hardlink on a statically-linked build is
free; a missed detection is the only way #95425 returns.
Provision-if-present: a surplus hardlink on a statically-linked build is free; a missed
detection is the only way #95425 returns.
"""
src_lib = _store_root(source_file) / "lib"
if not src_lib.is_dir():
@@ -222,10 +193,9 @@ def _provision_libpython(
def _copy_alias(venv_bin: Path, name: str, anchor: Path) -> bool:
"""Materialize *name* as a real-file copy of *anchor* (atomic rename).
Returns False (and warns) on failure: a leftover alias *symlink* to the
anchor is the exact #95541 crash shape, so callers must know when the
alias set is incomplete. The staging name is unique (mkstemp) so a
concurrent ensure (update + doctor --fix) cannot promote a truncated
Returns False (and warns) on failure: a leftover alias *symlink* to the anchor is the exact
#95541 crash shape, so callers must know when the alias set is incomplete. The staging name is
unique (mkstemp) so a concurrent ensure (update + doctor --fix) cannot promote a truncated
interim copy.
"""
tmp_path: Path | None = None
@@ -278,13 +248,10 @@ def _materialize_aliases(
def _passes_boot_gate(staged: Path, venv_dir: Path) -> bool:
"""Launch *staged* and demand encodings + the venv prefix.
The probe runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited
``PYTHONHOME`` papers over exactly the prefix-resolution failure the gate
exists to catch. ``OSError`` is split by errno — ``ENOENT``/``ENOEXEC``
(fixture binaries, foreign-arch images) means the binary cannot run here
at all, so the symlinked venv was equally dead and installing cannot make
things worse: skip. Anything else (notably ``EACCES`` after our own
chmod) is a broken install about to go live: refuse.
Runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited PYTHONHOME papers over the very
prefix failure this gate exists to catch. ``ENOENT``/``ENOEXEC`` (fixture binaries, foreign
arch) mean the binary can't run here at all, so the symlinked venv was equally dead: skip.
Anything else (notably ``EACCES`` after our own chmod) is a broken install going live: refuse.
"""
env = {
k: v
@@ -368,10 +335,9 @@ def _install_anchor(venv_dir: Path, source_file: Path) -> None:
def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None:
"""Pin a dylib-complete interpreter anchor for macOS TCC (#95596).
No-op (returns None) on non-macOS, when no venv interpreter exists, or
when the interpreter is not uv-managed. Idempotent. Best-effort —
returns None (and logs) if the copy or boot-gate fails; callers must
never depend on success.
No-op (returns None) on non-macOS, when no venv interpreter exists, or when the interpreter is
not uv-managed. Idempotent. Best-effort — returns None (and logs) if the copy or boot-gate
fails; callers must never depend on success.
"""
if not is_macos():
return None
@@ -411,14 +377,11 @@ def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None:
def tcc_anchor_state(project_root: Path | None = None) -> tuple[str, str]:
"""Report the anchor state for ``hermes doctor``.
"""Report the anchor state for ``hermes doctor`` as ``(status, detail)``.
Returns ``(status, detail)`` with status one of:
- ``"skip"`` — not applicable (non-macOS, no venv, or not uv-managed)
- ``"active"`` — venv interpreter is pinned at a stable real-file anchor
- ``"stale"`` — pinned but the interpreter changed since the last copy
- ``"missing"`` — uv-managed interpreter with no stable anchor installed
``skip`` = not applicable (non-macOS, no venv, not uv-managed); ``active`` = pinned at a
stable real-file anchor; ``stale`` = pinned but the interpreter changed since the last copy;
``missing`` = uv-managed interpreter with no anchor installed.
"""
if not is_macos():
return "skip", "not macOS"
+283 -547
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,7 +1,7 @@
"""CLI handlers for ``hermes migrate ...``.
Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply)
rewrites references to xAI models retired on May 15, 2026.
Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply) rewrites references to
xAI models retired on May 15, 2026.
"""
from __future__ import annotations
+35 -69
View File
@@ -1,27 +1,12 @@
"""Recover from npm ``EBADENGINE`` failures by upgrading a managed npm.
The repo's ``.npmrc`` sets ``engine-strict=true`` and the root ``package.json``
pins an ``engines.npm`` range, so an npm outside that range aborts every
``npm ci`` / ``npm install`` we run inside the checkout::
Rather than predicting the failure (which would mean a semver range matcher and an ``npm --version``
probe before work that usually succeeds), we react to it: npm states the required range in the
error, so the recovery reads the constraint straight out of the output it just produced.
npm error code EBADENGINE
npm error notsup Required: {"node":">=26.0.0","npm":">=12.0.0"}
npm error notsup Actual: {"npm":"10.9.8","node":"v22.23.1"}
Rather than predicting the failure (which would mean a semver range matcher and
an ``npm --version`` probe before work that usually succeeds), we react to it:
npm states the required range in the error, so the recovery reads the
constraint straight out of the output it just produced.
Scope of the repair is deliberately narrow. Hermes only upgrades an npm that
lives inside its **own** managed Node tree (``$HERMES_HOME/node``), installing
in place with ``--prefix`` so ``bin/npm`` keeps resolving to the upgraded
``lib/node_modules/npm``. A system / nvm / brew / Nix npm belongs to the user
and their other projects; Hermes never modifies those. When the failing npm is
one of those foreign installs, Hermes instead provisions its own managed Node
tree (the same tree a fresh install creates), upgrades *that* npm into range,
and hands the caller the managed npm to retry with — leaving the user's
toolchain untouched.
Scope of the repair is deliberately narrow. Hermes only upgrades an npm that lives inside its
**own** managed Node tree (``$HERMES_HOME/node``), installing in place with ``--prefix`` so
``bin/npm`` keeps resolving to the upgraded ``lib/node_modules/npm``.
"""
from __future__ import annotations
@@ -61,14 +46,12 @@ _UPGRADE_TIMEOUT = 300
def is_ebadengine(output: str) -> bool:
"""Return True when *output* is an npm engine-compatibility failure."""
if not output:
return False
return "EBADENGINE" in output or "Unsupported engine" in output
return bool(output) and ("EBADENGINE" in output or "Unsupported engine" in output)
def _iter_required_blocks(output: str) -> list[dict]:
def _iter_json_blocks(pattern: re.Pattern[str], output: str) -> list[dict]:
blocks: list[dict] = []
for match in _REQUIRED_RE.finditer(output or ""):
for match in pattern.finditer(output or ""):
try:
parsed = json.loads(match.group(1))
except ValueError:
@@ -78,17 +61,19 @@ def _iter_required_blocks(output: str) -> list[dict]:
return blocks
def _iter_required_blocks(output: str) -> list[dict]:
return _iter_json_blocks(_REQUIRED_RE, output)
def required_npm_range(output: str) -> str | None:
"""Return the ``engines.npm`` range npm demanded in *output*.
Returns ``None`` when the output has no engine failure, or when the
failure is about Node rather than npm — upgrading npm cannot fix a Node
version mismatch, so the caller must not try.
Returns ``None`` when the output has no engine failure, or when the failure is about Node rather
than npm — upgrading npm cannot fix a Node version mismatch, so the caller must not try.
When several packages report conflicting npm ranges the repo's own root
constraint is preferred (it is the one we control); otherwise the first
range wins, since any of them is a strict improvement over an npm that
satisfies none.
When several packages report conflicting npm ranges the repo's own root constraint is preferred
(it is the one we control); otherwise the first range wins, since any of them is a strict
improvement over an npm that satisfies none.
"""
if not is_ebadengine(output):
return None
@@ -109,12 +94,8 @@ def required_npm_range(output: str) -> str | None:
def actual_npm_version(output: str) -> str | None:
"""Return the npm version npm reported as ``Actual`` in *output*."""
for match in _ACTUAL_RE.finditer(output or ""):
try:
parsed = json.loads(match.group(1))
except ValueError:
continue
if isinstance(parsed, dict) and parsed.get("npm"):
for parsed in _iter_json_blocks(_ACTUAL_RE, output):
if parsed.get("npm"):
return str(parsed["npm"]).strip()
return None
@@ -137,10 +118,9 @@ def managed_npm_prefix(npm: str | os.PathLike[str] | None) -> Path | None:
"""Return the Hermes-managed Node root *npm* lives in, else ``None``.
Symlinks are resolved first: an install links ``~/.local/bin/npm`` at
``$HERMES_HOME/node/bin/npm``, which itself links into
``lib/node_modules/npm/bin/npm-cli.js``. Every one of those spellings is
the managed npm and must be recognised as such, or the repair silently
declines to fix the very install it owns.
``$HERMES_HOME/node/bin/npm``, which itself links into ``lib/node_modules/npm/bin/npm-cli.js``.
Every one of those spellings is the managed npm and must be recognised as such, or the repair
silently declines to fix the very install it owns.
"""
if not npm:
return None
@@ -175,10 +155,10 @@ def upgrade_managed_npm(
) -> bool:
"""Upgrade the managed npm at *npm* in place to satisfy *npm_range*.
``--prefix`` targets the managed tree explicitly: a managed install writes
``prefix=~/.local`` into ``$HERMES_HOME/node/etc/npmrc`` so that global
installs land on PATH, and without the override the "upgrade" would install
a second npm somewhere else while the managed one stayed stale.
``--prefix`` targets the managed tree explicitly: a managed install writes ``prefix=~/.local``
into ``$HERMES_HOME/node/etc/npmrc`` so that global installs land on PATH, and without the
override the "upgrade" would install a second npm somewhere else while the managed one stayed
stale.
"""
if not quiet:
print(
@@ -274,13 +254,10 @@ def _print_manual_fix(npm: str, npm_range: str, actual: str | None) -> None:
def _provision_managed_npm(npm_range: str | None, *, quiet: bool = False) -> str | None:
"""Provision a Hermes-managed Node tree and return a satisfying npm.
Installs the managed tree under ``$HERMES_HOME/node`` (reusing a healthy
one when present), then upgrades its bundled npm to *npm_range* — a fresh
Node LTS bundles an npm that may itself be outside the repo's range, so
without the upgrade the caller's single retry would fail the same way.
Falls back to the checkout's own ``engines.npm`` when npm did not state a
range (a Node-only mismatch), so the managed npm ends up in range either
way. Returns the managed npm path, or ``None`` when provisioning failed.
Installs (or reuses) the tree under ``$HERMES_HOME/node``, then upgrades its bundled npm to
*npm_range*: a fresh Node LTS may bundle an npm outside the repo's range, so without the
upgrade the caller's single retry would fail the same way. Falls back to the checkout's
``engines.npm`` when no range was stated. Returns the npm path, or ``None`` on failure.
"""
if not quiet:
print(
@@ -314,18 +291,9 @@ def maybe_repair_npm_engine(
) -> str | None:
"""Repair an ``EBADENGINE`` failure, never touching a foreign toolchain.
*output* is the combined stdout/stderr of the npm command that just failed.
Returns the npm executable the caller should retry its command with —
the same *npm* after an in-place upgrade of a Hermes-managed install, or
a freshly provisioned managed npm when the failing npm belongs to the
user (system / nvm / brew / Nix installs are never modified). Returns
``None`` when no repair happened — not an engine failure, a Node mismatch
a managed npm upgrade cannot fix, or a failed upgrade/bootstrap — leaving
the original failure to stand.
The returned value is truthy exactly when the caller should retry once,
so ``if maybe_repair_npm_engine(...)`` call sites keep working; they just
must run the retry with the returned path.
The returned value is truthy exactly when the caller should retry once, so ``if
maybe_repair_npm_engine(...)`` call sites keep working; they just must run the retry with the
returned path.
"""
if not npm or not is_ebadengine(output):
return None
@@ -336,9 +304,7 @@ def maybe_repair_npm_engine(
if prefix is not None:
# Hermes owns this npm — upgrade it in place. Only an npm-range
# failure is fixable this way; a Node mismatch needs a Node upgrade.
if not npm_range:
return None
if upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet):
if npm_range and upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet):
return npm
return None
+15 -51
View File
@@ -1,11 +1,8 @@
"""
Unified self-relaunch for Hermes CLI.
"""Unified self-relaunch for Hermes CLI.
Preserves critical flags (--tui, --dev, --profile, --model, etc.) across
process replacement so that ``hermes sessions browse`` or post-setup relaunch
doesn't silently drop the user's UI mode or other preferences.
Also works when ``hermes`` is not on PATH (e.g. ``nix run`` or ``python -m``).
Preserves critical flags (--tui, --dev, --profile, --model, etc.) across process replacement so that
``hermes sessions browse`` or post-setup relaunch doesn't silently drop the user's UI mode or other
preferences.
"""
import os
@@ -20,12 +17,10 @@ from hermes_cli._parser import (
def _build_inherited_flag_table() -> list[tuple[str, bool]]:
"""Build the ``(option_string, takes_value)`` table of flags that must
survive a self-relaunch, by introspecting the real parser used by
``hermes`` itself.
"""Build the ``(option_string, takes_value)`` table of flags that must survive a self-relaunch.
A flag participates if its argparse Action carries
``inherit_on_relaunch = True`` — set by ``_parser._inherited_flag``.
Introspects the real ``hermes`` parser: a flag participates iff its argparse Action carries
``inherit_on_relaunch = True`` (set by ``_parser._inherited_flag``).
"""
parser, _subparsers, chat_parser = build_top_level_parser()
@@ -80,20 +75,8 @@ def _extract_inherited_flags(argv: Sequence[str]) -> list[str]:
def resolve_hermes_bin() -> Optional[str]:
"""Find the hermes entry point.
Priority:
1. ``sys.argv[0]`` if it resolves to a real executable.
2. ``shutil.which("hermes")`` on PATH.
3. ``None`` → caller should fall back to ``python -m hermes_cli.main``.
Windows note: ``os.access(path, os.X_OK)`` returns True for ``.py`` and
``.pyc`` files on Windows (the OS treats anything listed in PATHEXT as
executable, and Python files are often registered there). But
``subprocess.run([script.py, ...])`` can't actually execute a .py
directly — CreateProcessW needs a real .exe, not a script associated
with the Python launcher. On Windows we therefore skip the argv[0]
fast-path when it points at a .py file and fall through to either
``hermes.exe`` on PATH or the ``sys.executable -m hermes_cli.main``
fallback.
Priority: 1. ``sys.argv[0]`` if it resolves to a real executable. 2. ``shutil.which("hermes")``
on PATH. 3. ``None`` → caller should fall back to ``python -m hermes_cli.main``.
"""
argv0 = sys.argv[0]
_is_windows = sys.platform == "win32"
@@ -127,15 +110,7 @@ def build_relaunch_argv(
preserve_inherited: bool = True,
original_argv: Optional[Sequence[str]] = None,
) -> list[str]:
"""Construct an argv list for replacing the current process with hermes.
Args:
extra_args: Arguments to append (e.g. ``["--resume", id]``).
preserve_inherited: Whether to carry over UI / behaviour flags
tagged with ``inherit_on_relaunch`` in the parser.
original_argv: The original argv to scan for flags (defaults to
``sys.argv[1:]``).
"""
"""Construct an argv list for replacing the current process with hermes."""
bin_path = resolve_hermes_bin()
if bin_path:
@@ -160,23 +135,12 @@ def relaunch(
) -> None:
"""Replace the current process with a fresh hermes invocation.
On POSIX we use ``os.execvp`` which replaces the running process with
the new one in place — same PID, no double-fork. That's what the
relaunch contract wants: "run hermes again as if the user had typed
the new argv".
On POSIX we use ``os.execvp`` which replaces the running process with the new one in place —
same PID, no double-fork. That's what the relaunch contract wants: "run hermes again as if the
user had typed the new argv".
Windows has no native exec semantics — ``os.execvp`` on Windows
*emulates* exec by spawning the child and exiting the parent, but
only works when the target is a real Win32 executable. Our target
is usually ``hermes.exe`` (a Python console-script shim that wraps
``python -m hermes_cli.main``) or a ``.cmd`` batch file, and both
raise ``OSError(8, "Exec format error")`` on Windows' execvp.
The Windows-correct pattern is: spawn the child with ``subprocess.run``
(which routes through ``cmd.exe`` via ``shell=False`` + PATHEXT resolution),
wait for it to exit, then propagate its exit code via ``sys.exit``.
That's functionally equivalent — the user sees "hermes exited, then
new hermes started" — just with two PIDs in play instead of one.
Windows has no native exec semantics — ``os.execvp`` on Windows *emulates* exec by spawning the
child and exiting the parent, but only works when the target is a real Win32 executable.
"""
new_argv = build_relaunch_argv(
extra_args, preserve_inherited=preserve_inherited, original_argv=original_argv
File diff suppressed because it is too large Load Diff
+417 -591
View File
File diff suppressed because it is too large Load Diff
+72 -178
View File
@@ -1,18 +1,10 @@
"""Fresh-process recovery after the update's in-process restart phase aborts.
``hermes update`` performs its fleet restart in the interpreter that started
before ``git pull``. When that phase raises — the module graph it is running
no longer matches the checkout on disk — this module owns everything that
happens next: which runtimes a clean child may relaunch, what counts as proof
that a relaunch actually replaced the old generation, which pre-update
processes are still alive, and whether any of that adds up to a recovery that
may call itself complete (#92145).
``hermes update`` performs its fleet restart in the interpreter that started before ``git pull``.
It is deliberately a separate owner from ``update_cmd``: the abort path has
its own vocabulary (``verified`` / ``relaunch_attempted`` / ``failed``, serve
units, survivors) and its own fail-closed contract, and the update monolith
must not grow another authority surface (review on #96235). The child process
itself lives in :mod:`hermes_cli.update_restart_recovery`.
It is deliberately a separate owner from ``update_cmd``: the abort path has its own vocabulary
(``verified`` / ``relaunch_attempted`` / ``failed``, serve units, survivors) and its own fail-closed
contract, and the update monolith must not grow another authority surface (review on #96235).
"""
from __future__ import annotations
@@ -35,18 +27,14 @@ def _serve_unit_recovery_available() -> bool:
def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
"""Pre-update serve/dashboard runtimes that are STILL the same process.
Identity is the process incarnation ``(pid, create_time)``, never the PID
alone. ``ledger_entries()`` re-verifies that pair on every read and prunes
anything that is gone, so a live entry is a real process — but the numeric
PID can be reused, and a serve that was restarted correctly can come back
on the same number. Comparing PIDs alone would then report the successor
as the pre-update survivor and keep the update incomplete forever
(#92145 review).
Identity is the process incarnation ``(pid, create_time)``, never the PID alone.
``ledger_entries()`` re-verifies that pair on every read and prunes anything that is gone, so a
live entry is a real process — but the numeric PID can be reused, and a serve that was restarted
correctly can come back on the same number.
Anything still here after the recovery pass is a live runtime on the
pre-update code generation, which is precisely the unsafe state #92145
reports. Fail closed on missing evidence: an unreadable ledger, or an
incarnation neither side can produce, counts the runtime as surviving.
Anything still here after the recovery pass is a live runtime on the pre-update code generation,
which is precisely the unsafe state #92145 reports. Fail closed on missing evidence: an
unreadable ledger, or an incarnation neither side can produce, counts the runtime as surviving.
"""
planned: dict[int, dict] = {}
try:
@@ -57,13 +45,12 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
if not isinstance(pid, int) or pid <= 0:
continue
detail = getattr(runtime, "detail", None)
created = detail.get("create_time") if isinstance(detail, dict) else None
planned[pid] = {
"pid": pid,
"kind": str(getattr(runtime, "kind", "")),
"profile": str(getattr(runtime, "profile", "")),
"supervisor": str(getattr(runtime, "supervisor", "")),
"_create_time": created if isinstance(created, (int, float)) else None,
"_create_time": _numeric(detail.get("create_time") if isinstance(detail, dict) else None),
}
except Exception as exc:
logger.debug("Could not read planned serve runtimes: %s", exc)
@@ -73,39 +60,36 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
try:
from hermes_cli.process_identity import ledger_entries
live: dict[int, float | None] = {}
for entry in ledger_entries():
if entry.get("purpose") not in ("serve", "dashboard"):
continue
pid = entry.get("pid")
if not isinstance(pid, int):
continue
created = entry.get("create_time")
live[pid] = created if isinstance(created, (int, float)) else None
live: dict[int, float | None] = {
entry["pid"]: _numeric(entry.get("create_time"))
for entry in ledger_entries()
if entry.get("purpose") in ("serve", "dashboard") and isinstance(entry.get("pid"), int)
}
except Exception as exc:
logger.debug("Serve/dashboard survivor probe failed: %s", exc)
return sorted(
(_without_incarnation(row) for row in planned.values()),
key=lambda row: row["pid"],
)
live = None
survivors = []
for pid, row in planned.items():
if pid not in live:
continue
planned_created = row["_create_time"]
live_created = live[pid]
if (
planned_created is not None
and live_created is not None
and abs(float(live_created) - float(planned_created)) >= 2.0
):
# Same number, different process: the pre-update runtime is gone
# and something new registered under its PID. Not a survivor.
continue
if live is not None:
if pid not in live:
continue
planned_created, live_created = row["_create_time"], live[pid]
if (
planned_created is not None
and live_created is not None
and abs(float(live_created) - float(planned_created)) >= 2.0
):
# Same number, different process: the pre-update runtime is gone
# and something new registered under its PID. Not a survivor.
continue
survivors.append(_without_incarnation(row))
return sorted(survivors, key=lambda row: row["pid"])
def _numeric(value):
return value if isinstance(value, (int, float)) else None
def _without_incarnation(row: dict) -> dict:
"""The operator-facing survivor row (incarnation is a matching key only)."""
return {key: value for key, value in row.items() if key != "_create_time"}
@@ -114,13 +98,8 @@ def _without_incarnation(row: dict) -> dict:
def _qualified_serve_skips(skip_units) -> list[dict]:
"""Scope-qualify the units the aborted phase already settled.
``restarted_scoped_units`` records ``<scope>/<unit>`` because
``hermes-serve.service`` can exist in BOTH the user and the system
manager, and those are two different processes. Handing the fresh child a
bare unit name would make one settled scope suppress recovery of the other
— the stale one would never be restarted and nothing downstream would say
so. Entries that carry no scope (a payload built by a pre-update
interpreter) are forwarded without one and read as scope-agnostic there.
``restarted_scoped_units`` records ``<scope>/<unit>`` because ``hermes-serve.service`` can exist
in BOTH the user and the system manager, and those are two different processes.
"""
rows: list[dict] = []
for entry in sorted(skip_units or ()):
@@ -141,74 +120,35 @@ def _recover_gateway_restart_after_abort(
) -> dict[str, list]:
"""Retry supervised gateway restarts from a clean Python process.
``hermes update`` normally performs the fleet restart in the interpreter
that started before ``git pull``. If that phase raises while importing the
new tree, a warning alone leaves the old gateway alive against new files on
disk. The recovery boundary launches the existing per-profile
``gateway restart`` command through a new interpreter, preserving its
platform-specific drain and service-manager logic without inheriting the
stale ``sys.modules`` graph.
``hermes update`` normally performs the fleet restart in the interpreter that started before
``git pull``.
Only profiles classified as supervisor-owned by the pre-update inventory
are handed off. A manual gateway must remain running and be reported for
explicit operator action rather than being killed without a relaunch
authority; serve/dashboard runtimes from the spawn ledger are likewise
recorded as skipped with a reason instead of vanishing from the pass.
The returned protocol is persisted in the update receipt so operators can
distinguish a spawn failure from a per-profile failure.
The same child additionally restarts active ``hermes-serve*`` systemd
units (#92145). ``hermes serve`` hosts ``tui_gateway.server`` and is
restarted by the in-process phase alongside the gateway units, but no
per-profile ``gateway restart`` command reaches it — so an abort used to
leave it holding the pre-pull module graph with nothing left to notice.
``skip_units`` names the units the aborted phase already settled, as
``<scope>/<unit>``. The scope is part of the identity, not decoration:
``hermes-serve.service`` can exist in both the user and the system
manager as two different processes, and an unqualified token would let a
settled one suppress recovery of a stale one (review on #96235).
Outcome honesty: ``verified`` means the fresh child independently observed
the profile's systemd unit active after the relaunch. A zero exit from
``gateway restart`` alone is NOT observed proof that the new code
generation is serving, so those outcomes are reported as
``relaunch_attempted`` and never claim supervisor coverage.
Only profiles classified as supervisor-owned by the pre-update inventory are handed off.
"""
from hermes_cli.update_cmd import _gateway_recovery_partition
candidates, skipped = _gateway_recovery_partition(
plan, skip_profiles=skip_profiles
)
candidates, skipped = _gateway_recovery_partition(plan, skip_profiles=skip_profiles)
profiles = sorted(candidates)
recover_serve = _serve_unit_recovery_available()
_empty_serve: dict[str, list] = {"verified": [], "failed": []}
if not profiles and not recover_serve:
def _result(requested, verified, relaunch_attempted, failed, serve_units=None) -> dict[str, list]:
return {
"requested": [],
"verified": [],
"relaunch_attempted": [],
"failed": [],
"requested": requested,
"verified": verified,
"relaunch_attempted": relaunch_attempted,
"failed": failed,
"skipped": skipped,
"serve_units": dict(_empty_serve),
"serve_units": dict(_empty_serve) if serve_units is None else serve_units,
}
if not profiles and not recover_serve:
return _result([], [], [], [])
def _all_failed() -> dict[str, list]:
return {
"requested": profiles,
"verified": [],
"relaunch_attempted": [],
"failed": profiles,
"skipped": skipped,
"serve_units": dict(_empty_serve),
}
return _result(profiles, [], [], profiles)
command = [
sys.executable,
"-m",
"hermes_cli.update_restart_recovery",
"--stdin",
]
command = [sys.executable, "-m", "hermes_cli.update_restart_recovery", "--stdin"]
env = os.environ.copy()
env["HERMES_UPDATE_RESTART_RECOVERY"] = "1"
for marker in ("_HERMES_GATEWAY", "HERMES_GATEWAY", "HERMES_GATEWAY_MODE"):
@@ -224,15 +164,7 @@ def _recover_gateway_restart_after_abort(
if not systemd_run:
logger.warning("Cannot isolate fresh gateway recovery from the gateway cgroup")
return _all_failed()
command = [
systemd_run,
"--user",
"--scope",
"--quiet",
"--collect",
"--",
*command,
]
command = [systemd_run, "--user", "--scope", "--quiet", "--collect", "--", *command]
kwargs = {
"input": json.dumps(
@@ -318,47 +250,26 @@ def _recover_gateway_restart_after_abort(
logger.warning("Fresh gateway restart recovery returned incomplete profiles")
return _all_failed()
if verified:
print(
" ✓ Restarted supervised gateway(s) in a fresh process"
" (systemd-verified active): " + ", ".join(sorted(verified))
)
if relaunch_attempted:
print(
" ⚠ Relaunch attempted in a fresh process but not"
" supervisor-verified (check these gateways manually): "
+ ", ".join(sorted(relaunch_attempted))
)
if serve_units["verified"]:
print(
" ✓ Restarted serve unit(s) in a fresh process"
" (new main PID observed): "
+ ", ".join(serve_units["verified"])
)
if serve_units["failed"]:
print(
" ⚠ Could not verify a replacement for serve unit(s): "
+ ", ".join(serve_units["failed"])
)
return {
"requested": profiles,
"verified": sorted(verified),
"relaunch_attempted": sorted(relaunch_attempted),
"failed": sorted(failed),
"skipped": skipped,
"serve_units": serve_units,
}
verified, relaunch_attempted, failed = sorted(verified), sorted(relaunch_attempted), sorted(failed)
for names, text in (
(verified, " ✓ Restarted supervised gateway(s) in a fresh process (systemd-verified active): "),
(relaunch_attempted, " ⚠ Relaunch attempted in a fresh process but not"
" supervisor-verified (check these gateways manually): "),
(serve_units["verified"], " ✓ Restarted serve unit(s) in a fresh process (new main PID observed): "),
(serve_units["failed"], " ⚠ Could not verify a replacement for serve unit(s): "),
):
if names:
print(text + ", ".join(names))
return _result(profiles, verified, relaunch_attempted, failed, serve_units)
def _warn_stale_serve_runtimes(rows) -> None:
"""Name the serve/dashboard processes that survived on pre-update code.
The original #92145 report is a user watching every chat turn fail with an
``ImportError`` for a symbol that imports fine on disk, with nothing in the
terminal naming the responsible process. ``hermes serve`` hosts
``tui_gateway.server``; when its unit was never restarted it keeps the
pre-pull ``sys.modules`` graph and there is no gateway row anywhere that
reveals it. Print the PIDs and the exact command that fixes it.
``hermes serve`` hosts ``tui_gateway.server``; if its unit was never restarted it keeps the
pre-pull ``sys.modules`` graph, every chat turn fails with an ``ImportError`` for a symbol
that imports fine on disk, and no gateway row reveals the culprit. Print the PIDs and the
exact command that fixes it.
"""
if not rows:
return
@@ -388,28 +299,11 @@ def _abort_recovery_is_complete(
) -> bool:
"""May a fresh-process recovery clear the incomplete flag?
Only when EVERY inventoried runtime family is accounted for. The gateway
leg alone is not enough (#92145): the post-update read-back
(``collect_fleet_versions``) is gateway-only — it reads each profile's
``gateway_state.json`` / control socket — so a ``hermes serve`` still
holding the pre-update ``sys.modules`` graph is invisible to both the
recovery pass and the verification that follows it. Clearing the flag on
gateway coverage alone is exactly how an update reported success while
every chat turn kept failing with an ``ImportError`` for a symbol that
imports fine on disk.
Only when EVERY inventoried runtime family is accounted for.
Fail closed on each leg:
* every planned gateway profile is covered, with nothing failed and
nothing merely ``relaunch_attempted`` (rc 0 is not observed coverage);
* no ``hermes-serve*`` unit failed to produce a verified replacement; and
* no serve/dashboard process from the pre-update inventory is still the
same process (``stale_runtime_rows``).
``planned_gateway_profiles`` being empty is deliberately NOT completeness:
with no gateway leg to prove, the caller's own fail-closed contract
(``_restart_phase_failure_is_incomplete``) plus the stale-runtime rows
decide the outcome.
``planned_gateway_profiles`` being empty is deliberately NOT completeness: with no gateway leg
to prove, the caller's own fail-closed contract (``_restart_phase_failure_is_incomplete``) plus
the stale-runtime rows decide the outcome.
"""
if not planned_gateway_profiles:
return False
+7 -22
View File
@@ -1,21 +1,8 @@
"""Image-managed install refusal contract (#91277 Phase 3).
One shared admission gate for every surface that can start an in-place
``hermes update`` mutation (CLI apply, CLI --check, dashboard update
endpoint). The decision layers:
1. **Baked provenance marker** (``/etc/hermes/image-provenance.json``,
written by the image build — see :mod:`hermes_cli.image_provenance`):
authoritative ground truth that this filesystem came from an immutable
image. Fail-closed: a present-but-malformed marker still refuses.
2. **Filesystem heuristics** (``detect_install_method()``): the pre-existing
docker/nix/apt detection, kept as the fallback for images built before
the marker existed and for package-managed installs that have no image
marker at all.
A refusal prints the real update command for the deployment kind, records a
``refused`` receipt (so fleet tooling sees "this install cannot self-update,
use <command>" instead of a silent non-update), and exits 2 on CLI surfaces.
A refusal prints the real update command for the deployment kind, records a ``refused`` receipt (so
fleet tooling sees "this install cannot self-update, use <command>" instead of a silent non-update),
and exits 2 on CLI surfaces.
"""
from __future__ import annotations
@@ -40,9 +27,8 @@ class UpdateRefusal:
def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]:
"""Return an :class:`UpdateRefusal` when in-place update must not run.
``None`` means the install is eligible for in-place update (git checkout
or unknown-but-mutable). Never raises; on any internal error it falls
back to the heuristic layer only.
``None`` means the install is eligible for in-place update (git checkout or unknown-but-
mutable). Never raises; on any internal error it falls back to the heuristic layer only.
"""
# Layer 1: baked provenance marker — authoritative when present.
try:
@@ -116,9 +102,8 @@ def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]:
def record_refusal_receipt(refusal: UpdateRefusal) -> None:
"""Write a minimal ``refused`` receipt for a blocked update attempt.
Gives fleet tooling a durable record that an update was ATTEMPTED and
refused ("not updatable in place, use <command>") instead of a silent
nothing. Best-effort; never raises.
Gives fleet tooling a durable record that an update was ATTEMPTED and refused ("not updatable in
place, use <command>") instead of a silent nothing. Best-effort; never raises.
"""
try:
from hermes_cli.update_receipt import (
+176 -271
View File
@@ -1,38 +1,19 @@
"""Runtime inventory + update plan for the fleet-update pipeline (#91277 Phase 2).
One read-only pass that answers, BEFORE any mutation: what Hermes runtimes
are running on this machine, how is each one deployed, which of them will
this update touch, and how will each be restarted?
One read-only pass that answers, BEFORE any mutation: what Hermes runtimes are running on this
machine, how is each one deployed, which of them will this update touch, and how will each be
restarted?
This is the "plan" phase of the transactional deployment model (#88683):
plan → snapshot → apply → restart-per-kind → verify → report
The module is deliberately side-effect free — every collector is a probe
over primitives that already exist (`find_profile_gateway_processes`,
`_get_service_pids`, `gateway_state.json` code stamps from #91283,
`detect_install_method`) — so `hermes update --plan` can run on a live
fleet with zero risk, and the update receipt can embed the inventory
without changing update behavior.
Deployment kinds (the concept most fleet-update bugs were missing):
git — source checkout; updatable in place via `hermes update`
docker — published image; NOT updatable in place (pull + recreate)
nix/apt — package-manager owned; updatable via the manager only
unknown — no marker; treated as in-place updatable (legacy default)
Supervisors (how a runtime is restarted after code changes):
systemd / launchd — restart via the service manager (fleet-wide)
desktop — Desktop app supervises `hermes serve`; it respawns
manual — plain process; SIGTERM + watcher/manual relaunch
The module is deliberately side-effect free — every collector is a probe over primitives that
already exist (`find_profile_gateway_processes`, `_get_service_pids`, `gateway_state.json` code
stamps from #91283, `detect_install_method`) — so `hermes update --plan` can run on a live fleet
with zero risk, and the update receipt can embed the inventory without changing update behavior.
"""
from __future__ import annotations
import logging
import os
from contextlib import contextmanager
from dataclasses import dataclass, field, asdict
from pathlib import Path
from typing import Any, Optional
@@ -70,17 +51,10 @@ class UpdatePlan:
runtimes: list = field(default_factory=list) # list[RuntimeRecord]
def to_dict(self) -> dict[str, Any]:
payload = asdict(self)
payload["runtimes"] = [
r.to_dict() if isinstance(r, RuntimeRecord) else r
for r in self.runtimes
]
return payload
return asdict(self) # recursive: RuntimeRecord entries become dicts, dict entries are copied
def _detect_supervisor_for_pid(
pid: int, service_pids: set, windows_service_pids: set | None = None
) -> str:
def _detect_supervisor_for_pid(pid: int, service_pids: set, windows_service_pids: set | None = None) -> str:
"""Classify how a live gateway PID is supervised."""
if windows_service_pids and pid in windows_service_pids:
# SCM-supervised Windows gateway (WinSW/NSSM/sc.exe create): the
@@ -102,56 +76,99 @@ def _detect_supervisor_for_pid(
return "manual"
_RESTART_MECHANISMS = {
"systemd": "systemd",
"launchd": "launchd",
"desktop": "desktop",
"windows-service": "windows-service",
"manual-serve": "respawn-argv",
}
_MECHANISM_DESCRIPTIONS = {
"systemd": "systemctl restart (drain-first SIGUSR1 when supported)",
"launchd": "launchctl kickstart -k (drain-first, per-label domain)",
"desktop": "Desktop app respawns its serve backend",
"windows-service": "sc.exe stop before venv mutation, sc.exe start after update",
"respawn-argv": "stop before code swap, relaunch with recorded launch args",
}
def _restart_mechanism(supervisor: str, profile: str) -> str:
"""Machine-readable restart mechanism id for a runtime.
THE policy table (#91277 Phase 2): restart execution consumes these ids
via :func:`match_runtime_outcomes` / the update's restart phase, and the
receipt records per-runtime outcomes against them. Display strings are
derived by :func:`describe_restart_mechanism` — never the other way
around.
THE policy table (#91277 Phase 2): restart execution consumes these ids via
:func:`match_runtime_outcomes` / the update's restart phase, and the receipt records per-runtime
outcomes against them. Display strings are derived by :func:`describe_restart_mechanism` — never
the other way around.
"""
if supervisor == "systemd":
return "systemd"
if supervisor == "launchd":
return "launchd"
if supervisor == "desktop":
return "desktop"
if supervisor == "windows-service":
return "windows-service"
if supervisor == "manual-serve":
return "respawn-argv"
return "manual"
return _RESTART_MECHANISMS.get(supervisor, "manual")
def describe_restart_mechanism(mechanism: str, profile: str) -> str:
"""Human-readable description of a restart mechanism id."""
if mechanism == "systemd":
return "systemctl restart (drain-first SIGUSR1 when supported)"
if mechanism == "launchd":
return "launchctl kickstart -k (drain-first, per-label domain)"
if mechanism == "desktop":
return "Desktop app respawns its serve backend"
if mechanism == "windows-service":
return "sc.exe stop before venv mutation, sc.exe start after update"
if mechanism == "respawn-argv":
return "stop before code swap, relaunch with recorded launch args"
described = _MECHANISM_DESCRIPTIONS.get(mechanism)
if described is not None:
return described
if profile != "default":
return f"hermes -p {profile} gateway restart"
return "hermes gateway restart"
def _runtime(kind: str, profile: str, pid: Optional[int], supervisor: str, **extra: Any) -> RuntimeRecord:
"""A :class:`RuntimeRecord` with ``restart_via`` derived from its supervisor."""
return RuntimeRecord(
kind=kind,
profile=profile,
pid=pid,
supervisor=supervisor,
restart_via=_restart_mechanism(supervisor, profile),
**extra,
)
def _int_or_none(value: Any) -> Optional[int]:
try:
return int(value)
except (TypeError, ValueError):
return None
def _gateway_record(
profile: str,
pid: int,
supervisor: str,
code_sha: Any = None,
code_version: Any = None,
) -> RuntimeRecord:
return _runtime(
"gateway",
profile,
pid,
supervisor,
code_sha=str(code_sha) if code_sha else None,
code_version=code_version,
)
@contextmanager
def _probe(label: str):
"""Run one inventory collector; a failure is logged at debug and yields fewer rows, never an exception."""
try:
yield
except Exception as exc:
logger.debug("%s failed: %s", label, exc)
def collect_runtime_inventory() -> UpdatePlan:
"""Build the pre-update plan. Read-only; never raises.
Every collector degrades independently — a probe failure yields fewer
rows, not an exception. The result is embeddable in the update receipt
and printable via :func:`print_update_plan`.
Every collector degrades independently — a probe failure yields fewer rows, not an exception.
The result is embeddable in the update receipt and printable via :func:`print_update_plan`.
"""
plan = UpdatePlan()
# --- install shape / deployment kind ---------------------------------
try:
with _probe("Install-method probe"):
from hermes_cli.config import (
detect_install_method,
get_managed_system,
@@ -169,7 +186,7 @@ def collect_runtime_inventory() -> UpdatePlan:
# container can look like `git` to the heuristics while the running
# filesystem is actually an immutable image. Fail-closed: an invalid
# marker still flips the plan to not-updatable.
try:
with _probe("Image provenance probe"):
from hermes_cli.image_provenance import read_image_provenance
provenance = read_image_provenance()
@@ -177,25 +194,19 @@ def collect_runtime_inventory() -> UpdatePlan:
plan.updatable_in_place = False
if provenance.valid and provenance.manager:
plan.install_method = provenance.manager
except Exception as exc:
logger.debug("Image provenance probe failed: %s", exc)
plan.update_mechanism = recommended_update_command_for_method(method)
except Exception as exc:
logger.debug("Install-method probe failed: %s", exc)
# --- expected code identity (pre-pull) --------------------------------
try:
with _probe("Code-identity probe"):
from hermes_cli.build_info import get_code_identity
identity = get_code_identity(refresh=True)
plan.expected_sha = identity.get("sha")
plan.expected_version = identity.get("version")
except Exception as exc:
logger.debug("Code-identity probe failed: %s", exc)
# --- profiles ----------------------------------------------------------
profile_homes: list[tuple[str, Path]] = []
try:
with _probe("Profile enumeration"):
from hermes_cli.profiles import (
_get_default_hermes_home,
_get_profiles_root,
@@ -208,24 +219,16 @@ def collect_runtime_inventory() -> UpdatePlan:
root = _get_profiles_root()
if root.is_dir():
for entry in sorted(root.iterdir()):
if (
entry.is_dir()
and entry.name != "default"
and _PROFILE_ID_RE.match(entry.name)
):
if entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name):
profile_homes.append((entry.name, entry))
plan.profiles = [name for name, _ in profile_homes]
except Exception as exc:
logger.debug("Profile enumeration failed: %s", exc)
# --- service-managed PIDs (fleet-wide) ---------------------------------
service_pids: set = set()
try:
with _probe("Service-PID probe"):
from hermes_cli.gateway import _get_service_pids
service_pids = _get_service_pids(all_profiles=True) or set()
except Exception as exc:
logger.debug("Service-PID probe failed: %s", exc)
# --- SCM-supervised gateway PIDs (Windows) ------------------------------
# find_windows_gateway_services() maps validated gateway PIDs through
@@ -234,19 +237,14 @@ def collect_runtime_inventory() -> UpdatePlan:
# `sc.exe start`, so the plan must carry the matching mechanism id for
# the #91277 Phase 2 reconciliation and the fleet check.
windows_service_pids: set = set()
try:
with _probe("Windows SCM service-ownership probe"):
from hermes_cli.gateway import find_windows_gateway_services
windows_service_pids = {
int(service.gateway_pid)
for service in find_windows_gateway_services()
}
except Exception as exc:
logger.debug("Windows SCM service-ownership probe failed: %s", exc)
windows_service_pids = {int(service.gateway_pid) for service in find_windows_gateway_services()}
# --- per-profile gateways (PID files + runtime status stamps) ----------
seen_pids: set[int] = set()
try:
with _probe("Gateway-state inventory"):
from gateway.status import _pid_exists, read_runtime_status
for profile, home in profile_homes:
@@ -260,91 +258,54 @@ def collect_runtime_inventory() -> UpdatePlan:
identity = identify_gateway(home)
except Exception:
identity = None
if identity:
try:
sock_pid = int(identity.get("pid"))
except (TypeError, ValueError):
sock_pid = None
if sock_pid is not None:
if sock_pid in seen_pids:
# One multiplex gateway can answer identify for
# several profile homes — one runtime record per
# process, not per home.
continue
seen_pids.add(sock_pid)
declared = identity.get("supervisor")
supervisor = (
str(declared)
if declared
else _detect_supervisor_for_pid(
sock_pid, service_pids, windows_service_pids
)
)
sock_sha = identity.get("code_sha")
plan.runtimes.append(
RuntimeRecord(
kind="gateway",
profile=profile,
pid=sock_pid,
supervisor=supervisor,
code_sha=str(sock_sha) if sock_sha else None,
code_version=identity.get("code_version"),
restart_via=_restart_mechanism(supervisor, profile),
)
)
sock_pid = _int_or_none(identity.get("pid")) if identity else None
if sock_pid is not None:
if sock_pid in seen_pids:
# One multiplex gateway can answer identify for
# several profile homes — one runtime record per
# process, not per home.
continue
record = read_runtime_status(home / "gateway_state.json")
pid: Optional[int] = None
code_sha = code_version = None
if record:
try:
pid = int(record.get("pid"))
except (TypeError, ValueError):
pid = None
code_sha = record.get("code_sha")
code_version = record.get("code_version")
seen_pids.add(sock_pid)
declared = identity.get("supervisor")
supervisor = (
str(declared)
if declared
else _detect_supervisor_for_pid(sock_pid, service_pids, windows_service_pids)
)
plan.runtimes.append(
_gateway_record(
profile, sock_pid, supervisor, identity.get("code_sha"), identity.get("code_version")
)
)
continue
record = read_runtime_status(home / "gateway_state.json") or {}
pid = _int_or_none(record.get("pid"))
if pid is None or not _pid_exists(pid):
continue
seen_pids.add(pid)
supervisor = _detect_supervisor_for_pid(
pid, service_pids, windows_service_pids
)
plan.runtimes.append(
RuntimeRecord(
kind="gateway",
profile=profile,
pid=pid,
supervisor=supervisor,
code_sha=str(code_sha) if code_sha else None,
code_version=code_version,
restart_via=_restart_mechanism(supervisor, profile),
_gateway_record(
profile,
pid,
_detect_supervisor_for_pid(pid, service_pids, windows_service_pids),
record.get("code_sha"),
record.get("code_version"),
)
)
except Exception as exc:
logger.debug("Gateway-state inventory failed: %s", exc)
# PID-file mapped gateways not covered by a runtime-status record
try:
with _probe("PID-file gateway inventory"):
from hermes_cli.gateway import find_profile_gateway_processes
for proc in find_profile_gateway_processes():
if proc.pid in seen_pids:
continue
seen_pids.add(proc.pid)
supervisor = _detect_supervisor_for_pid(
proc.pid, service_pids, windows_service_pids
)
plan.runtimes.append(
RuntimeRecord(
kind="gateway",
profile=proc.profile,
pid=proc.pid,
supervisor=supervisor,
restart_via=_restart_mechanism(supervisor, proc.profile),
_gateway_record(
proc.profile, proc.pid, _detect_supervisor_for_pid(proc.pid, service_pids, windows_service_pids)
)
)
except Exception as exc:
logger.debug("PID-file gateway inventory failed: %s", exc)
# Serve/dashboard backends from the spawn ledger (#63206). These are the
# runtimes the gateway collectors above can never see: a manually
@@ -355,7 +316,7 @@ def collect_runtime_inventory() -> UpdatePlan:
# fabricates a row. Desktop-supervised backends are classified by their
# recorded spawner still being alive — those restart via the Desktop's
# own respawn, not ours.
try:
with _probe("Serve/dashboard ledger inventory"):
from hermes_cli.process_identity import ledger_entries, spawner_is_dead
for entry in ledger_entries():
@@ -366,16 +327,12 @@ def collect_runtime_inventory() -> UpdatePlan:
if not isinstance(pid, int) or pid in seen_pids:
continue
seen_pids.add(pid)
has_live_spawner = spawner_is_dead(entry) is False
supervisor = "desktop" if has_live_spawner else "manual-serve"
profile = str(entry.get("profile") or "default")
plan.runtimes.append(
RuntimeRecord(
kind=str(purpose),
profile=profile,
pid=pid,
supervisor=supervisor,
restart_via=_restart_mechanism(supervisor, profile),
_runtime(
str(purpose),
str(entry.get("profile") or "default"),
pid,
"desktop" if spawner_is_dead(entry) is False else "manual-serve",
detail={
"argv": entry.get("argv") or "",
"host": entry.get("host") or "",
@@ -388,8 +345,6 @@ def collect_runtime_inventory() -> UpdatePlan:
},
)
)
except Exception as exc:
logger.debug("Serve/dashboard ledger inventory failed: %s", exc)
return plan
@@ -415,14 +370,8 @@ def print_update_plan(plan: UpdatePlan) -> None:
print(f" Running services to restart ({len(plan.runtimes)}):")
for runtime in plan.runtimes:
sha = f" @ {runtime.code_sha[:8]}" if runtime.code_sha else ""
print(
f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid}"
f" — {runtime.supervisor}{sha}"
)
print(
" restart: "
f"{describe_restart_mechanism(runtime.restart_via, runtime.profile)}"
)
print(f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid} — {runtime.supervisor}{sha}")
print(f" restart: {describe_restart_mechanism(runtime.restart_via, runtime.profile)}")
_SERVE_KINDS = ("serve", "dashboard")
@@ -431,11 +380,8 @@ _SERVE_KINDS = ("serve", "dashboard")
def _serve_unit_matches_profile(profile: str, unit: object) -> bool:
"""Does *unit* name a ``hermes-serve*``/``hermes-dashboard*`` unit for *profile*?
Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's
``hermes-gateway*`` names never cover them (#100479). Exact names only —
``work`` must not claim ``hermes-serve-workbench`` — and a scope prefix
(``user/hermes-serve``) is tolerated because the restart phase records
scope-qualified identities in some lists.
Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's ``hermes-gateway*`` names
never cover them (#100479).
"""
name = str(unit).removesuffix(".service")
if "/" in name:
@@ -480,32 +426,13 @@ def match_runtime_outcomes(
) -> list[dict[str, Any]]:
"""Reconcile the plan's runtimes against what the restart phase DID.
#91277 Phase 2 (restart via declared mechanism): the platform restart
branches each re-discover their own targets, so a runtime the plan saw
can be missed entirely with no signal. This cross-checks every planned
runtime against the phase's bookkeeping and returns one outcome row per
runtime::
{"kind", "profile", "pid", "mechanism", "outcome"}
outcome: ``restarted`` (service restarted / profile relaunched /
handed to external supervisor), ``stopped`` (pid killed, watcher or
operator relaunches), ``failed`` (in the phase's failed/stale list) or
``unaccounted`` — the plan saw it and NO bookkeeping mentions it: the
blind-spot tripwire (same philosophy as the fleet matrix's DOWN row).
Never raises; on any probe error returns what it has.
Serve/dashboard runtimes are reconciled in their OWN vocabulary
(#100479): a ``hermes-serve*``/``hermes-dashboard*`` unit, a killed
PID, or — when the caller passes ``stale_serve_pids`` (the
``(pid, create_time)``-verified survivor probe,
:func:`hermes_cli.update_abort_recovery._surviving_pre_update_serve_runtimes`)
— liveness: a pre-update serve whose incarnation is gone was replaced
(unit restart, dashboard cleanup respawn, Desktop respawn) and counts as
``restarted``; one still alive is ``unaccounted``. They never borrow the
gateway's outcome: ``relaunched_profiles`` and ``hermes-gateway*`` name a
different process that shares the profile, nothing more. Without the
probe result, an untouched serve stays ``unaccounted`` (fail closed).
The platform restart branches each re-discover their own targets, so a runtime the plan saw can
be missed with no signal. Returns one ``{kind, profile, pid, mechanism, outcome}`` row per
planned runtime; outcome is ``restarted``, ``stopped``, ``failed`` or ``unaccounted`` (no
bookkeeping mentions it — the blind-spot tripwire). Never raises. Serve/dashboard runtimes are
reconciled in their OWN vocabulary and never borrow the gateway's outcome: with
``stale_serve_pids`` a pre-update serve whose incarnation is gone counts as ``restarted``, one
still alive is ``unaccounted``; without the probe an untouched serve stays ``unaccounted``.
"""
outcomes: list[dict[str, Any]] = []
try:
@@ -514,69 +441,51 @@ def match_runtime_outcomes(
relaunched = set(relaunched_profiles or [])
external = set(externally_supervised_profiles or [])
killed = {int(p) for p in (killed_pids or set())}
stale_serves = (
{int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None
)
stale_serves = {int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None
for runtime in plan.runtimes:
r = runtime if isinstance(runtime, RuntimeRecord) else None
if r is None:
continue
if r.kind in _SERVE_KINDS:
outcomes.append(
{
"kind": r.kind,
"profile": r.profile,
"pid": r.pid,
"mechanism": r.restart_via,
"outcome": _serve_runtime_outcome(
r,
killed=killed,
failed_set=failed_set,
restarted_set=restarted_set,
stale_serves=stale_serves,
),
}
)
continue
outcome = "unaccounted"
def _gateway_names(r: RuntimeRecord, names: set) -> bool:
# The bare "hermes-gateway" unit name is gateway-specific: a
# serve/dashboard runtime that merely shares the default
# profile is a different process the gateway restart never
# touched, and must not borrow its outcome (#100479).
if r.profile in relaunched or r.profile in external:
return any(
r.profile in name
or (
r.kind == "gateway"
and r.profile == "default"
and "hermes-gateway" in name
)
for name in names
)
for r in plan.runtimes:
if not isinstance(r, RuntimeRecord):
continue
if r.kind in _SERVE_KINDS:
outcome = _serve_runtime_outcome(
r,
killed=killed,
failed_set=failed_set,
restarted_set=restarted_set,
stale_serves=stale_serves,
)
elif r.profile in relaunched or r.profile in external:
outcome = "restarted"
elif r.pid is not None and r.pid in killed:
outcome = "stopped"
elif any(
r.profile in unit
or (
r.kind == "gateway"
and r.profile == "default"
and "hermes-gateway" in unit
)
for unit in failed_set
):
elif _gateway_names(r, failed_set):
outcome = "failed"
elif any(
r.profile in svc
or (
r.kind == "gateway"
and r.profile == "default"
and "hermes-gateway" in svc
)
for svc in restarted_set
):
elif _gateway_names(r, restarted_set):
outcome = "restarted"
outcomes.append(
{
"kind": r.kind,
"profile": r.profile,
"pid": r.pid,
"mechanism": r.restart_via,
"outcome": outcome,
}
)
else:
outcome = "unaccounted"
outcomes.append({
"kind": r.kind,
"profile": r.profile,
"pid": r.pid,
"mechanism": r.restart_via,
"outcome": outcome,
})
except Exception as exc:
logger.debug("Runtime-outcome reconciliation failed: %s", exc)
return outcomes
@@ -585,9 +494,8 @@ def match_runtime_outcomes(
def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool:
"""Print a loud warning for runtimes the restart phase never touched.
Returns True when at least one planned runtime is unaccounted — the
caller escalates exactly like a STALE/DOWN fleet row (exit 1): a runtime
the plan promised to restart, silently missed, is the class this phase
Returns True when at least one planned runtime is unaccounted; the caller escalates like a
STALE/DOWN fleet row (exit 1) — a promised restart silently missed is the class this phase
exists to kill.
"""
missed = [o for o in outcomes if o.get("outcome") == "unaccounted"]
@@ -596,10 +504,7 @@ def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool:
print()
print(" ⚠ Planned runtimes the restart phase never touched:")
for o in missed:
print(
f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']}"
f" — planned mechanism: {o['mechanism']}"
)
print(f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']} — planned mechanism: {o['mechanism']}")
print(" Restart them manually, then verify:")
if any(o.get("kind") not in _SERVE_KINDS for o in missed):
print(" hermes gateway restart # active profile")
+33 -81
View File
@@ -1,51 +1,12 @@
"""Cross-process mutual exclusion for in-flight Hermes updates.
Three different surfaces can start an update of the same install tree:
Until now only the Tauri updater published an "update in progress" marker (``UpdateMarkerGuard`` in
``apps/bootstrap-installer/src-tauri/src/update.rs``), and only the Electron desktop consumed it
(``electron/update-marker.ts``, to gate local backend startup).
* ``hermes update`` from a terminal,
* the dashboard's Update button (``POST /api/hermes/update`` →
``_spawn_hermes_action(["update"])``, detached),
* the desktop's Update button, which hands off to the Tauri
``hermes-setup --update`` and, on its failure screen, to install-mode
bootstrap (``install.ps1`` / ``install.sh``).
Until now only the Tauri updater published an "update in progress" marker
(``UpdateMarkerGuard`` in ``apps/bootstrap-installer/src-tauri/src/update.rs``),
and only the Electron desktop consumed it (``electron/update-marker.ts``, to
gate local backend startup). Nothing stopped two *updaters* from running at
once — so a dashboard-spawned ``hermes update`` and an installer-driven
``git checkout`` could mutate the same checkout concurrently, rewriting source
under a live interpreter and leaving the tree half-updated.
This module makes that same marker the single lock for **all** update
entrypoints instead of adding a fourth mechanism. Format and location are
unchanged and remain byte-compatible with the Rust and Electron readers:
<HERMES_HOME>/.hermes-update-in-progress body: "<pid>\\n<started_at_unix>"
A marker only counts as a live update when its pid is alive AND it is younger
than :data:`UPDATE_MARKER_MAX_AGE_MS` — mirroring ``readLiveUpdateMarker`` so a
crashed updater self-heals instead of wedging every future update. A stale
marker is removed on read by whoever notices it first.
One layering wrinkle: the Tauri updater holds this marker for its WHOLE run and
then spawns ``hermes update`` as a child stage. Without a handoff the child
sees its own parent's live marker and refuses — the GUI update deadlocks
against itself on every attempt ("Hermes is still running", retry forever).
Two mechanisms recognize the orchestrating parent, and either suffices:
* The updater exports :data:`HANDOFF_PID_ENV` naming its own pid, and
``acquire`` treats a live holder matching that pid as the lock we are
already running under. The env var alone grants nothing: the pid must also
be the live marker owner, so a stale or forged value cannot bypass the lock.
* A live holder that is a *process ancestor* of ours is likewise our own
orchestrator. This is the load-bearing path for the fleet: the staged
``hermes-setup`` binary under ``~/.hermes`` is only refreshed by a full
installer run (``copy_self_to_hermes_home`` deliberately no-ops during
``--update``), so every desktop whose staged updater predates the
HANDOFF_PID_ENV export runs an old parent against a new child. Without the
ancestry check those users get exit 2 ("Hermes is still running") on every
GUI update forever, with no Hermes process actually running.
This module makes that same marker the single lock for **all** update entrypoints instead of adding
a fourth mechanism. Format and location are unchanged and remain byte-compatible with the Rust and
Electron readers:
"""
from __future__ import annotations
@@ -85,10 +46,10 @@ UPDATE_EXIT_CONCURRENT = 2
def update_marker_path() -> Path:
"""Path of the shared update marker.
Uses the *process* Hermes home (never the context-local profile override):
the Rust updater resolves ``$HERMES_HOME`` or the platform default, and the
desktop pins that same value into the updater's env. A profile-scoped path
here would put the lock somewhere the other two owners never look.
Uses the *process* Hermes home (never the context-local profile override): the Rust updater
resolves ``$HERMES_HOME`` or the platform default, and the desktop pins that same value into the
updater's env. A profile-scoped path here would put the lock somewhere the other two owners
never look.
"""
from hermes_constants import get_process_hermes_home
@@ -98,15 +59,12 @@ def update_marker_path() -> Path:
def _pid_alive(pid: int) -> bool:
"""True when a process with ``pid`` currently exists.
Delegates to :func:`gateway.status._pid_exists`, the project's existing
no-kill probe. Do NOT hand-roll this with ``os.kill(pid, 0)``: on Windows
that is not a no-op — CPython routes ``sig=0`` to
``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console
process group (bpo-14484). A liveness check that killed the updater it was
asking about would be a spectacular way to fix a concurrency bug.
Delegates to :func:`gateway.status._pid_exists`, the project's existing no-kill probe. Do NOT
hand-roll this with ``os.kill(pid, 0)``: on Windows that is not a no-op — CPython routes
``sig=0`` to ``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console process
group (bpo-14484).
Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge
the lock forever.
Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge the lock forever.
"""
if pid <= 0:
return False
@@ -124,8 +82,8 @@ def _pid_alive(pid: int) -> bool:
def _handoff_pid() -> int | None:
"""Pid of the orchestrating updater that spawned us, if any.
Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent —
a broken handoff must fall back to the normal refusal, never crash.
Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent — a broken handoff must fall
back to the normal refusal, never crash.
"""
raw = os.environ.get(HANDOFF_PID_ENV, "").strip()
if not raw:
@@ -140,14 +98,12 @@ def _handoff_pid() -> int | None:
def _is_ancestor_pid(pid: int) -> bool:
"""True when ``pid`` is a live ancestor (parent chain) of this process.
The orchestrating updater spawns ``hermes update`` as a (grand)child, so a
live marker owned by one of our ancestors can only be the claim we are
already running under — an unrelated concurrent updater is never in our
parent chain. This heals the fleet of staged ``hermes-setup`` binaries
that predate the HANDOFF_PID_ENV export and can never send it.
The orchestrating updater spawns ``hermes update`` as a (grand)child, so a live marker owned by
one of our ancestors can only be the claim we are already running under — an unrelated
concurrent updater is never in our parent chain.
Never includes our own pid, and any failure counts as "not an ancestor":
an unprovable ancestry must fall back to the normal refusal.
Never includes our own pid, and any failure counts as "not an ancestor": an unprovable ancestry
must fall back to the normal refusal.
"""
if pid <= 0:
return False
@@ -171,10 +127,9 @@ class UpdateHolder:
def read_live_update(*, path: Path | None = None) -> UpdateHolder | None:
"""Return the live update holding the lock, or ``None``.
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent,
unreadable, malformed, dead-pid, and past-the-ceiling all mean "no live
update", and a stale marker file is deleted so it can't strand future runs.
Never raises.
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent, unreadable,
malformed, dead-pid, and past-the-ceiling all mean "no live update", and a stale marker file is
deleted so it can't strand future runs. Never raises.
"""
marker = path or update_marker_path()
try:
@@ -220,11 +175,10 @@ def describe_holder(holder: UpdateHolder) -> str:
class UpdateLock:
"""Context manager owning the shared update marker for this process.
``acquired`` is False when another live update already holds it — callers
decide whether that's a hard refusal (CLI/dashboard) or a wait. Releasing
only removes the marker when *we* still own it, so a marker rewritten by a
handoff partner (the Tauri updater overwrites it with its own pid) is never
deleted out from under its new owner.
``acquired`` is False when another live update holds it; callers decide between hard refusal
(CLI/dashboard) and waiting. Release only removes the marker when *we* still own it, so a marker
rewritten by a handoff partner (the Tauri updater writes its own pid) is never deleted from
under its new owner.
"""
def __init__(self, *, path: Path | None = None) -> None:
@@ -235,12 +189,10 @@ class UpdateLock:
def acquire(self) -> bool:
"""Claim the lock. Returns False (and sets ``holder``) if it's taken.
A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is a
process ancestor of ours — is our own orchestrating parent (the Tauri
updater spawning `hermes update` as a stage): we run under ITS claim
rather than refusing or re-writing the marker, and ``release`` leaves
the parent's marker untouched. The ancestry path exists because staged
updaters older than the HANDOFF_PID_ENV export never send the env var.
A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is an ancestor of ours — is our
own orchestrating parent (e.g. the Tauri updater staging ``hermes update``): run under ITS
claim and leave its marker untouched on release. The ancestry path covers staged updaters
older than the env-var export.
"""
existing = read_live_update(path=self.path)
if existing is not None:
+87 -148
View File
@@ -1,29 +1,10 @@
"""Structured update receipts + post-update fleet version verification.
Phase 1 of the fleet-update reliability plan (#91277): the updater must
*prove* its outcome instead of assuming it.
Phase 1 of the fleet-update reliability plan (#91277): the updater must *prove* its outcome instead
of assuming it.
Two additive capabilities, both designed so a failure inside them can never
break an update (every public entry point is exception-swallowing):
1. **Update receipt** — a machine-readable JSON record of what one
``hermes update`` run discovered, did, skipped (and why), written to
``<HERMES_HOME>/logs/update_receipts/``. Silent-failure classes this
makes visible: #88848 (helper died after "success" printed), #74973
(restart silently skipped), #85753 (restart phase never ran), #81193
(desktop shows failure for a successful update).
2. **Fleet version verification** — after the restart phase, read every
profile's ``gateway_state.json``, compare each live gateway's stamped
``code_sha`` (written by ``gateway/status.py`` on every runtime-status
write) against the freshly-updated checkout's HEAD, and print a fleet
version matrix. Mixed-version fleets (#88654, #69754, #77553, #56717)
become a loud, actionable report instead of a latent state.
Deployment-kind awareness (docker/image-managed installs) rides on
``hermes_cli.build_info.get_code_identity()``: an image build reports
``source="build-file"`` and the receipt records that the install is not
in-place updatable.
Two additive capabilities, both designed so a failure inside them can never break an update (every
public entry point is exception-swallowing):
"""
from __future__ import annotations
@@ -53,6 +34,18 @@ def _utc_now_iso() -> str:
return datetime.now(timezone.utc).isoformat()
def _str_records(entries: Any, keys: tuple[str, ...], *, pid: bool = False) -> list[dict[str, Any]]:
"""Dict entries reduced to stringified ``keys`` (plus an int ``pid`` first when requested)."""
records = []
for entry in entries:
if not isinstance(entry, dict):
continue
record: dict[str, Any] = {"pid": int(entry.get("pid", 0) or 0)} if pid else {}
record.update({key: str(entry.get(key, "")) for key in keys})
records.append(record)
return records
class UpdateReceipt:
"""Collects the observable facts of one ``hermes update`` run."""
@@ -122,16 +115,10 @@ class UpdateReceipt:
key: [str(profile) for profile in fresh_recovery.get(key, [])]
for key in ("requested", "verified", "relaunch_attempted", "failed")
}
persisted["skipped"] = [
{
"profile": str(entry.get("profile", "")),
"kind": str(entry.get("kind", "")),
"supervisor": str(entry.get("supervisor", "")),
"reason": str(entry.get("reason", "")),
}
for entry in fresh_recovery.get("skipped", [])
if isinstance(entry, dict)
]
persisted["skipped"] = _str_records(
fresh_recovery.get("skipped", []),
("profile", "kind", "supervisor", "reason"),
)
# Serve/dashboard coverage (#92145). ``hermes serve`` hosts
# tui_gateway and is not a gateway profile, so neither the
# per-profile buckets above nor the fleet-version matrix can
@@ -143,16 +130,11 @@ class UpdateReceipt:
key: [str(unit) for unit in (serve_units.get(key) or [])]
for key in ("verified", "failed")
}
persisted["stale_runtimes"] = [
{
"pid": int(entry.get("pid", 0) or 0),
"kind": str(entry.get("kind", "")),
"profile": str(entry.get("profile", "")),
"supervisor": str(entry.get("supervisor", "")),
}
for entry in fresh_recovery.get("stale_runtimes", [])
if isinstance(entry, dict)
]
persisted["stale_runtimes"] = _str_records(
fresh_recovery.get("stale_runtimes", []),
("kind", "profile", "supervisor"),
pid=True,
)
result["fresh_recovery"] = persisted
self.data["gateway_restart"] = result
@@ -183,31 +165,28 @@ def begin_update_receipt() -> None:
_current = None
def record_step(name: str, ok: bool, detail: str = "") -> None:
"""Record one update step outcome. No-op when no receipt is active."""
def _record(method: str, what: str, *args: Any, **kwargs: Any) -> None:
"""Invoke ``method`` on the active receipt; no-op when none, never raises."""
try:
if _current is not None:
_current.step(name, ok, detail)
getattr(_current, method)(*args, **kwargs)
except Exception as exc: # pragma: no cover - defensive
logger.debug("Could not record update step %s: %s", name, exc)
logger.debug("Could not record %s: %s", what, exc)
def record_step(name: str, ok: bool, detail: str = "") -> None:
"""Record one update step outcome. No-op when no receipt is active."""
_record("step", f"update step {name}", name, ok, detail)
def record_skip(name: str, reason: str) -> None:
"""Record a skipped step WITH the reason it was skipped."""
try:
if _current is not None:
_current.skip(name, reason)
except Exception as exc: # pragma: no cover - defensive
logger.debug("Could not record update skip %s: %s", name, exc)
_record("skip", f"update skip {name}", name, reason)
def record_gateway_restart(**kwargs: Any) -> None:
"""Record the gateway restart phase outcome (see UpdateReceipt)."""
try:
if _current is not None:
_current.gateway_restart_result(**kwargs)
except Exception as exc: # pragma: no cover - defensive
logger.debug("Could not record gateway restart result: %s", exc)
_record("gateway_restart_result", "gateway restart result", **kwargs)
def finalize_update_receipt(
@@ -215,10 +194,9 @@ def finalize_update_receipt(
) -> Optional[Path]:
"""Finalize + persist the receipt. Returns the written path or None.
``outcome`` is one of ``success`` / ``partial`` / ``failed`` /
``refused``. Exactly-once by construction: the module singleton is
popped first, so a second call (e.g. the command-boundary safety net
after an inner path already finalized) is a no-op returning None.
``outcome`` is one of ``success`` / ``partial`` / ``failed`` / ``refused``. Exactly-once by
construction: the module singleton is popped first, so a second call (e.g. the command-boundary
safety net after an inner path already finalized) is a no-op returning None.
"""
global _current
receipt = _current
@@ -235,15 +213,11 @@ def finalize_update_receipt(
directory.mkdir(parents=True, exist_ok=True)
stamp = time.strftime("%Y%m%d_%H%M%S")
path = directory / f"update_{stamp}_{os.getpid()}.json"
path.write_text(
json.dumps(receipt.data, indent=2, default=str), encoding="utf-8"
)
body = json.dumps(receipt.data, indent=2, default=str)
path.write_text(body, encoding="utf-8")
# Stable pointer for the dashboard/desktop: latest receipt.
latest = directory / "latest.json"
try:
latest.write_text(
json.dumps(receipt.data, indent=2, default=str), encoding="utf-8"
)
(directory / "latest.json").write_text(body, encoding="utf-8")
except OSError:
pass
_prune_old_receipts(directory)
@@ -258,19 +232,12 @@ def finalize_pending_update_receipt(
) -> Optional[Path]:
"""Command-boundary safety net: persist a still-open receipt, if any.
``hermes update`` has many early-termination paths (Windows
concurrent-instance preflight, venv-holder refusal, head-pinned no-op,
fetch failure — all ``sys.exit``) that predate the inner finalize
call sites. Any receipt still open when the update COMMAND unwinds is
finalized here so every post-begin run leaves a record — the
refused/failed runs are exactly the ones a receipt matters most for
(review on #91283). No-op when no receipt is open (the inner paths
already finalized — exactly-once via the popped singleton) or when
recording was never started. Never raises.
Outcome mapping: exit 0/None → ``success`` (a path that completed
without an explicit inner finalize), exit 2 → ``refused`` (the
updater's preflight-refusal convention), anything else → ``failed``.
``hermes update`` has many early ``sys.exit`` paths (preflight refusals, venv-holder
refusal, fetch failure) predating the inner finalize calls; any receipt still open when the
command unwinds is finalized here so refused/failed runs — where a receipt matters most —
leave a record. No-op when nothing is open (inner paths finalize exactly-once via the popped
singleton). Never raises. Exit 0/None → ``success``, exit 2 → ``refused`` (preflight
convention), else → ``failed``.
"""
if _current is None:
return None
@@ -280,12 +247,11 @@ def finalize_pending_update_receipt(
outcome = "refused"
else:
outcome = "failed"
try:
receipt = _current
if receipt is not None and exit_code is not None:
receipt.data["exit_code"] = int(exit_code)
except Exception:
pass
if exit_code is not None:
try:
_current.data["exit_code"] = int(exit_code)
except Exception:
pass
return finalize_update_receipt(outcome, stop_reason=stop_reason)
@@ -321,35 +287,31 @@ def read_latest_receipt() -> Optional[dict[str, Any]]:
# Fleet version verification
# ---------------------------------------------------------------------------
def _sha_state(code_sha: Any, expected_sha: Any) -> str:
if not code_sha or not expected_sha:
return "unknown"
return "current" if str(code_sha) == str(expected_sha) else "stale"
def _fleet_row(profile: str, pid: int, code_sha: Any, code_version: Any, expected_sha: Any) -> dict[str, Any]:
return {
"profile": profile,
"pid": pid,
"code_sha": str(code_sha) if code_sha else None,
"code_version": code_version,
"state": _sha_state(code_sha, expected_sha),
}
def collect_fleet_versions(
*, pre_restart_pids: Optional[list[int]] = None
) -> list[dict[str, Any]]:
"""Snapshot every profile's gateway code identity vs. the current tree.
Returns one entry per profile home that has a ``gateway_state.json``
describing a gateway that is live — or that SHOULD be live::
{"profile": str, "pid": int, "code_sha": str|None,
"code_version": str|None, "state": "current"|"stale"|"unknown"|"down"}
``stale`` — gateway stamped a code_sha that differs from the updated
checkout's HEAD (it is still serving pre-update modules).
``unknown`` — gateway predates the code-identity stamp (started before
this feature landed) or identity could not be resolved.
``down`` — the gateway was ALIVE when this update started
(``pre_restart_pids``), its runtime status still says
running, but the PID is dead and no successor rewrote the
record: the restart phase stopped it and nothing came
back. Without this row a killed-and-never-replaced gateway
produced NO entry at all and the matrix passed silently
(Phase-1 verification gap, #88848/#74973 class).
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` —
a stale state file from a long-dead gateway (machine reboot, manual
kill weeks ago) must NOT fail every future update. Callers that don't
have a pre-restart snapshot (``None``/empty) get the historical
behavior: dead PIDs are skipped.
Never raises; a probe failure yields an empty list.
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a
long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update.
Callers that don't have a pre-restart snapshot (``None``/empty) get the historical behavior:
dead PIDs are skipped.
"""
# Runtime-status states that mean "this record does not describe a
# gateway that should be running now" — no down row for these.
@@ -399,23 +361,12 @@ def collect_fleet_versions(
except (TypeError, ValueError):
pid = None
if pid is not None:
code_sha = identity.get("code_sha")
if not code_sha or not expected_sha:
state = "unknown"
elif str(code_sha) == str(expected_sha):
state = "current"
else:
state = "stale"
results.append(
{
"profile": profile,
"pid": pid,
"code_sha": str(code_sha) if code_sha else None,
"code_version": identity.get("code_version"),
"state": state,
"source": "socket",
}
row = _fleet_row(
profile, pid, identity.get("code_sha"),
identity.get("code_version"), expected_sha,
)
row["source"] = "socket"
results.append(row)
continue
status_path = home / "gateway_state.json"
record = read_runtime_status(status_path)
@@ -458,21 +409,11 @@ def collect_fleet_versions(
}
)
continue
code_sha = record.get("code_sha")
if not code_sha or not expected_sha:
state = "unknown"
elif str(code_sha) == str(expected_sha):
state = "current"
else:
state = "stale"
results.append(
{
"profile": profile,
"pid": pid,
"code_sha": str(code_sha) if code_sha else None,
"code_version": record.get("code_version"),
"state": state,
}
_fleet_row(
profile, pid, record.get("code_sha"),
record.get("code_version"), expected_sha,
)
)
except Exception as exc:
logger.debug("Fleet version probe failed: %s", exc)
@@ -482,13 +423,11 @@ def collect_fleet_versions(
def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
"""Print the post-update fleet version matrix.
Returns True when at least one gateway is provably stale (still
serving pre-update code) OR provably down (was running, killed by the
restart phase, nothing came back), so the caller can escalate.
``unknown`` entries are reported but do NOT fail the update: gateways
started before the code-identity stamp existed have no sha to compare,
and failing on them would turn this feature's own rollout into a
false-positive storm.
Returns True when at least one gateway is provably stale (still serving pre-update code) OR
provably down (killed by the restart phase, nothing came back), so the caller can escalate.
``unknown`` entries are reported but do NOT fail the update: gateways started before the
code-identity stamp existed have no sha to compare, and failing them would be a false-
positive storm.
"""
if not fleet:
return False
+123 -257
View File
@@ -70,17 +70,35 @@ _UNIT_SETTLE_ATTEMPTS = 10
_UNIT_SETTLE_DELAY = 1.0
def _profile_command(profile: str) -> list[str]:
"""Build a parameterized restart command for exactly one profile."""
return [
sys.executable,
"-m",
"hermes_cli.main",
"-p",
profile,
"gateway",
"restart",
]
def _run_quiet(
run: Callable[..., Any],
argv: list[str],
*,
timeout: int,
**extra: Any,
) -> Any | None:
"""Run ``argv`` capturing text output; ``None`` when it errors or times out."""
try:
return run(
argv,
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=timeout,
**extra,
)
except (OSError, subprocess.TimeoutExpired):
return None
def _stdout(result: Any) -> str:
return (getattr(result, "stdout", "") or "").strip()
def _succeeded(result: Any) -> bool:
return result is not None and getattr(result, "returncode", 1) == 0
def _child_environment() -> dict[str, str]:
@@ -98,16 +116,7 @@ def _run_profile_restart(
run: Callable[..., Any],
) -> bool:
"""Run one profile restart without inheriting the updater's process state."""
kwargs: dict[str, Any] = {
"stdin": subprocess.DEVNULL,
"capture_output": True,
"text": True,
"encoding": "utf-8",
"errors": "replace",
"check": False,
"timeout": _PROFILE_RESTART_TIMEOUT,
"env": _child_environment(),
}
kwargs: dict[str, Any] = {"stdin": subprocess.DEVNULL, "env": _child_environment()}
if os.name == "nt":
kwargs["creationflags"] = (
getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
@@ -115,58 +124,31 @@ def _run_profile_restart(
)
else:
kwargs["start_new_session"] = True
try:
result = run(_profile_command(profile), **kwargs)
except (OSError, subprocess.TimeoutExpired):
return False
return getattr(result, "returncode", 1) == 0
argv = [sys.executable, "-m", "hermes_cli.main", "-p", profile, "gateway", "restart"]
return _succeeded(_run_quiet(run, argv, timeout=_PROFILE_RESTART_TIMEOUT, **kwargs))
def _systemd_unit_candidates(profile: str) -> tuple[str, ...]:
"""Unit names the existing systemd gateway lifecycle produces per profile."""
if profile == "default":
return (
"hermes-gateway.service",
"gateway.service",
"gateway-default.service",
)
return (
f"hermes-gateway-{profile}.service",
f"gateway-{profile}.service",
)
return ("hermes-gateway.service", "gateway.service", "gateway-default.service")
return (f"hermes-gateway-{profile}.service", f"gateway-{profile}.service")
def _systemd_verified_active(profile: str, *, run: Callable[..., Any]) -> bool:
"""Return True only when systemd itself reports the profile's unit active.
This is the observation that separates ``verified`` from
``relaunch_attempted``. Any failure here (no ``systemctl``, probe error,
unit not ``active``) means we could NOT verify — never that the restart
failed.
This is the observation that separates ``verified`` from ``relaunch_attempted``. Any failure
here (no ``systemctl``, probe error, unit not ``active``) means we could NOT verify — never that
the restart failed.
"""
systemctl = shutil.which("systemctl")
if not systemctl:
return False
for unit in _systemd_unit_candidates(profile):
try:
result = run(
[systemctl, "--user", "is-active", unit],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=_VERIFY_TIMEOUT,
)
except (OSError, subprocess.TimeoutExpired):
continue
if (
getattr(result, "returncode", 1) == 0
and (getattr(result, "stdout", "") or "").strip() == "active"
):
return True
return False
return any(
_unit_is_active([systemctl, "--user"], unit, run=run, require_rc0=True)
for unit in _systemd_unit_candidates(profile)
)
def restart_profiles(
@@ -177,21 +159,13 @@ def restart_profiles(
) -> dict[str, list[str]]:
"""Restart the supplied profiles and return per-profile terminal results.
The caller supplies only profiles whose inventory identified a service
supervisor. Manual gateways are intentionally excluded before this module
is called: killing one without a relaunch authority would turn stale code
into an outage.
The caller supplies only profiles whose inventory identified a service supervisor.
A profile only lands in ``verified`` when its supervisor is systemd and
``systemctl --user is-active`` independently confirms the unit after the
relaunch command succeeded. Every other zero-exit relaunch is reported as
``relaunch_attempted`` — the code cannot observe supervisor coverage for
those paths and must not claim it.
A profile only lands in ``verified`` when its supervisor is systemd and ``systemctl --user is-
active`` independently confirms the unit after the relaunch command succeeded.
"""
supervisors = supervisors or {}
normalized = sorted(
{profile for profile in profiles if isinstance(profile, str) and profile}
)
normalized = sorted({profile for profile in profiles if isinstance(profile, str) and profile})
verified: list[str] = []
relaunch_attempted: list[str] = []
failed: list[str] = []
@@ -199,31 +173,23 @@ def restart_profiles(
if not _run_profile_restart(profile, run=run):
failed.append(profile)
continue
if supervisors.get(profile) == "systemd" and _systemd_verified_active(
profile, run=run
):
if supervisors.get(profile) == "systemd" and _systemd_verified_active(profile, run=run):
verified.append(profile)
else:
relaunch_attempted.append(profile)
return {
"verified": verified,
"relaunch_attempted": relaunch_attempted,
"failed": failed,
}
return {"verified": verified, "relaunch_attempted": relaunch_attempted, "failed": failed}
def _systemctl_scopes() -> list[tuple[str, list[str]]]:
"""``systemctl`` invocations for the user and system scopes, or nothing.
Mirrors the scope pair the in-process restart phase walks. ``systemctl``
is resolved through ``shutil.which`` so this module never has to import
any Hermes platform helper — importing the freshly pulled tree is exactly
what aborted the phase that called us.
Mirrors the scope pair the in-process restart phase walks. ``systemctl`` is resolved through
``shutil.which`` so this module never has to import any Hermes platform helper — importing the
freshly pulled tree is exactly what aborted the phase that called us.
Each scope is returned with its label because ``hermes-serve.service`` in
the user manager and ``hermes-serve.service`` in the system manager are
two different processes. Every identity this module produces or consumes
stays qualified by that label; the bare unit name is never the key.
Each scope is returned with its label because ``hermes-serve.service`` in the user manager and
``hermes-serve.service`` in the system manager are two different processes. Every identity this
module produces or consumes stays qualified by that label; the bare unit name is never the key.
"""
systemctl = shutil.which("systemctl")
if not systemctl or sys.platform != "linux":
@@ -233,83 +199,43 @@ def _systemctl_scopes() -> list[tuple[str, list[str]]]:
def _listed_serve_units(scope: list[str], *, run: Callable[..., Any]) -> list[str]:
"""Serve units systemd knows about in one scope, validated by name."""
try:
result = run(
scope
+ [
"list-units",
_SERVE_UNIT_PATTERN,
"--plain",
"--no-legend",
"--no-pager",
],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=_VERIFY_TIMEOUT,
)
except (OSError, subprocess.TimeoutExpired):
result = _run_quiet(
run,
scope + ["list-units", _SERVE_UNIT_PATTERN, "--plain", "--no-legend", "--no-pager"],
timeout=_VERIFY_TIMEOUT,
)
if result is None:
return []
# The glob is a systemd pattern, not a name gate: `hermes-serve*` also
# matches the unrelated `hermes-server.service`. Require the exact
# base unit or the hyphenated profile family, same shape as the
# in-process phase's own name gate.
units: list[str] = []
for line in (getattr(result, "stdout", "") or "").splitlines():
parts = line.split()
if not parts:
continue
# The glob is a systemd pattern, not a name gate: `hermes-serve*` also
# matches the unrelated `hermes-server.service`. Require the exact
# base unit or the hyphenated profile family, same shape as the
# in-process phase's own name gate.
if _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units:
if parts and _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units:
units.append(parts[0])
return units
def _unit_property(
scope: list[str], unit: str, prop: str, *, run: Callable[..., Any]
) -> str | None:
"""One ``systemctl show`` property, or ``None`` when it cannot be read."""
try:
result = run(
scope + ["show", unit, f"--property={prop}", "--value"],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=_VERIFY_TIMEOUT,
)
except (OSError, subprocess.TimeoutExpired):
return None
if getattr(result, "returncode", 1) != 0:
return None
return (getattr(result, "stdout", "") or "").strip()
def _unit_main_pid(scope: list[str], unit: str, *, run: Callable[..., Any]) -> int:
"""The unit's ``MainPID``; ``0`` when absent or unreadable."""
raw = _unit_property(scope, unit, "MainPID", run=run)
"""The unit's ``MainPID`` via ``systemctl show``; ``0`` when absent or unreadable."""
result = _run_quiet(
run, scope + ["show", unit, "--property=MainPID", "--value"], timeout=_VERIFY_TIMEOUT
)
try:
return int(raw or 0)
return int(_stdout(result) or 0) if _succeeded(result) else 0
except ValueError:
return 0
def _unit_is_active(scope: list[str], unit: str, *, run: Callable[..., Any]) -> bool:
try:
result = run(
scope + ["is-active", unit],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=_VERIFY_TIMEOUT,
)
except (OSError, subprocess.TimeoutExpired):
def _unit_is_active(
scope: list[str], unit: str, *, run: Callable[..., Any], require_rc0: bool = False
) -> bool:
result = _run_quiet(run, scope + ["is-active", unit], timeout=_VERIFY_TIMEOUT)
if result is None or (require_rc0 and not _succeeded(result)):
return False
return (getattr(result, "stdout", "") or "").strip() == "active"
return _stdout(result) == "active"
def _serve_unit_replaced(
@@ -322,11 +248,9 @@ def _serve_unit_replaced(
) -> bool:
"""Did the unit come back on a NEW main process?
``restart`` returning 0 is not evidence: the whole point of #92145 is that
a live process can keep serving the pre-update generation while every
status command reports success. A changed ``MainPID`` on an ``active``
unit is the observation that the old interpreter — and its stale
``sys.modules`` — is gone.
``restart`` returning 0 is not evidence: a live process can keep serving the pre-update
generation while every status command reports success. A changed ``MainPID`` on an ``active``
unit is the observation that the old interpreter and its stale ``sys.modules`` are gone.
"""
for attempt in range(_UNIT_SETTLE_ATTEMPTS):
if attempt:
@@ -339,9 +263,16 @@ def _serve_unit_replaced(
return False
def _qualified(scope_label: str, base: str) -> str:
"""The only identity this module reports for a serve unit."""
return f"{scope_label}/{base}"
def _split_skip_entry(entry: Any) -> tuple[str, str]:
"""``(scope_label, base_unit)`` of a skip entry in either the mapping or ``scope/unit`` shape."""
if isinstance(entry, Mapping):
scope_label = str(entry.get("scope") or "")
unit = str(entry.get("unit") or "")
else:
scope_label, sep, unit = str(entry).partition("/")
if not sep:
scope_label, unit = "", scope_label
return scope_label, unit.removesuffix(".service")
def _normalized_skips(
@@ -349,26 +280,13 @@ def _normalized_skips(
) -> tuple[set[tuple[str, str]], set[str]]:
"""Split already-settled units into scope-qualified and legacy entries.
Qualified entries (``{"scope": "user", "unit": "hermes-serve"}`` or the
equivalent ``"user/hermes-serve"`` string) suppress exactly one process.
A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH
of two same-named processes was settled. It is honoured across both scopes
because that is all the information it contains — the caller in this tree
always sends the qualified shape, and this branch exists only for a payload
written by a pre-update interpreter that had no scope to send.
A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH of two same-named
processes was settled.
"""
qualified: set[tuple[str, str]] = set()
legacy: set[str] = set()
for entry in skip_units or ():
if isinstance(entry, Mapping):
scope_label = str(entry.get("scope") or "")
base = str(entry.get("unit") or "").removesuffix(".service")
else:
scope_label, sep, unit = str(entry).partition("/")
if not sep:
scope_label, unit = "", scope_label
base = unit.removesuffix(".service")
scope_label, base = _split_skip_entry(entry)
if not base:
continue
if scope_label in _SCOPE_LABELS:
@@ -386,28 +304,9 @@ def restart_serve_units(
) -> dict[str, list[str]]:
"""Restart every active ``hermes-serve*`` systemd unit from this process.
``hermes serve`` hosts ``tui_gateway.server`` and is restarted by the
in-process phase alongside the gateway units, but it is not a gateway
profile: no ``gateway restart`` command reaches it. When the phase aborts
part-way — systemd lists ``hermes-gateway.service`` before
``hermes-serve.service``, so the gateway is typically already done — the
serve unit is the one left holding generation-N modules over a
generation-N+1 checkout.
Units are enumerated from systemd, never from the update inventory. That
keeps the relaunch authority requirement structural: a manually launched
or Desktop-owned ``hermes serve`` owns no unit and therefore cannot be
touched here.
Every identity in and out of this function is scope-qualified. The user
manager and the system manager can each own a ``hermes-serve.service``,
and they are different processes: projecting the scope away would let one
already-settled unit suppress recovery of the other, and would let one
scope's success describe the other's outcome.
Returns ``{"verified": [...], "failed": [...]}`` whose entries are
``<scope>/<base unit>`` (no ``.service`` suffix), e.g.
``user/hermes-serve``.
Units are enumerated from systemd, never from the update inventory. That keeps the relaunch
authority requirement structural: a manually launched or Desktop-owned ``hermes serve`` owns no
unit and therefore cannot be touched here.
"""
skipped_qualified, skipped_legacy = _normalized_skips(skip_units)
# (scope, base unit) -> replaced? A unit name can exist in BOTH the user
@@ -433,41 +332,28 @@ def restart_serve_units(
# remove.
outcomes[target] = False
continue
try:
result = run(
scope + ["--no-ask-password", "restart", unit],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
check=False,
timeout=_UNIT_RESTART_TIMEOUT,
)
except (OSError, subprocess.TimeoutExpired):
outcomes[target] = False
continue
if getattr(result, "returncode", 1) != 0:
result = _run_quiet(
run, scope + ["--no-ask-password", "restart", unit], timeout=_UNIT_RESTART_TIMEOUT
)
if not _succeeded(result):
# Includes the unprivileged system-scope case. We do not probe
# for sudo here: an unverifiable unit must read as failed so
# the update stays explicitly incomplete.
outcomes[target] = False
continue
outcomes[target] = _serve_unit_replaced(
scope, unit, previous_pid, run=run, sleep=sleep
)
outcomes[target] = _serve_unit_replaced(scope, unit, previous_pid, run=run, sleep=sleep)
# ``<scope>/<unit>`` is the only identity this module reports for a serve unit.
return {
"verified": sorted(
_qualified(*target) for target, ok in outcomes.items() if ok
),
"failed": sorted(
_qualified(*target) for target, ok in outcomes.items() if not ok
),
"verified": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if ok),
"failed": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if not ok),
}
def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
payload = json.load(stream)
profiles = payload.get("profiles") if isinstance(payload, dict) else None
if not isinstance(payload, dict):
payload = {}
profiles = payload.get("profiles")
if not isinstance(profiles, list):
raise ValueError("recovery payload must contain a profiles list")
if any(
@@ -475,7 +361,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
for profile in profiles
):
raise ValueError("recovery profiles contain an invalid profile id")
raw_supervisors = payload.get("supervisors") if isinstance(payload, dict) else None
raw_supervisors = payload.get("supervisors")
supervisors: dict[str, str] = {}
if raw_supervisors is not None:
if not isinstance(raw_supervisors, dict) or any(
@@ -487,7 +373,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
):
raise ValueError("recovery supervisors map is invalid")
supervisors = dict(raw_supervisors)
raw_serve = payload.get("serve_units") if isinstance(payload, dict) else None
raw_serve = payload.get("serve_units")
recover_serve = False
skip_units: list[str] = []
if raw_serve is not None:
@@ -495,9 +381,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
raise ValueError("recovery serve_units block is invalid")
recover_serve = bool(raw_serve.get("recover"))
raw_skip = raw_serve.get("skip") or []
if not isinstance(raw_skip, list) or any(
not isinstance(entry, (str, dict)) for entry in raw_skip
):
if not isinstance(raw_skip, list) or any(not isinstance(entry, (str, dict)) for entry in raw_skip):
raise ValueError("recovery serve_units skip list is invalid")
for entry in raw_skip:
# A skip entry names one already-settled process. The qualified
@@ -507,14 +391,11 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
# interpreter can still send; it is kept, and read as
# scope-agnostic by `restart_serve_units`.
if isinstance(entry, dict):
scope_label = entry.get("scope")
unit = entry.get("unit")
if not isinstance(unit, str):
if not isinstance(entry.get("unit"), str):
raise ValueError("recovery serve_units skip list is invalid")
if not isinstance(scope_label, str):
scope_label = ""
else:
scope_label, _, unit = entry.rpartition("/")
if not isinstance(entry.get("scope"), str):
entry = {"unit": entry["unit"]}
scope_label, base = _split_skip_entry(entry)
# Only the shapes systemd can actually produce for this family; a
# skip entry is a name filter, never a command argument. An
# unrecognized scope drops the entry rather than raising: dropping
@@ -523,22 +404,15 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
# process.
if scope_label and scope_label not in _SCOPE_LABELS:
continue
if not _UNIT_RE.fullmatch(
unit if unit.endswith(".service") else f"{unit}.service"
):
if not _UNIT_RE.fullmatch(f"{base}.service"):
continue
base = unit.removesuffix(".service")
skip_units.append(f"{scope_label}/{base}" if scope_label else base)
return profiles, supervisors, recover_serve, skip_units
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--stdin",
action="store_true",
help=argparse.SUPPRESS,
)
parser.add_argument("--stdin", action="store_true", help=argparse.SUPPRESS)
args = parser.parse_args(argv)
if not args.stdin:
parser.error("this command is an internal update-recovery entry point")
@@ -547,28 +421,20 @@ def main(argv: list[str] | None = None) -> int:
profiles, supervisors, recover_serve, skip_units = _parse_payload(sys.stdin)
result = restart_profiles(profiles, supervisors=supervisors)
result["serve_units"] = (
restart_serve_units(skip_units=skip_units)
if recover_serve
else {"verified": [], "failed": []}
restart_serve_units(skip_units=skip_units) if recover_serve else {"verified": [], "failed": []}
)
except (ValueError, json.JSONDecodeError) as exc:
print(
json.dumps(
{
"error": str(exc),
"verified": [],
"relaunch_attempted": [],
"failed": [],
"serve_units": {"verified": [], "failed": []},
}
)
)
print(json.dumps({
"error": str(exc),
"verified": [],
"relaunch_attempted": [],
"failed": [],
"serve_units": {"verified": [], "failed": []},
}))
return 2
print(json.dumps(result, sort_keys=True))
if result["failed"] or result["serve_units"]["failed"]:
return 1
return 0
return 1 if result["failed"] or result["serve_units"]["failed"] else 0
if __name__ == "__main__":
+56 -71
View File
@@ -1,12 +1,8 @@
"""``hermes worktree`` — audit and reclaim accumulated git worktrees/branches.
Attended counterpart of the silent startup pruner (see
``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage:
hermes worktree list # audit: verdict + reason per tree
hermes worktree prune # reap safe trees + merged branches
hermes worktree prune --dry-run # show the plan, change nothing
hermes worktree prune --trees-only / --branches-only
hermes worktree list # audit: verdict + reason per tree hermes worktree prune # reap safe trees +
merged branches hermes worktree prune --dry-run # show the plan, change nothing hermes worktree
prune --trees-only / --branches-only
"""
from __future__ import annotations
@@ -23,9 +19,54 @@ def _repo_root() -> Optional[str]:
def _fmt_size(size_mb: Optional[int]) -> str:
if size_mb is None:
return "?"
if size_mb >= 1024:
return f"{size_mb / 1024:.1f}G"
return f"{size_mb}M"
return f"{size_mb / 1024:.1f}G" if size_mb >= 1024 else f"{size_mb}M"
def _list(worktree_gc, repo_root: str, args) -> int:
records = worktree_gc.audit_worktrees(repo_root)
if not records:
print("No worktrees under .worktrees/ — nothing to reclaim.")
return 0
total_mb = sum(r.size_mb or 0 for r in records)
reapable_mb = sum(r.size_mb or 0 for r in records if r.verdict.startswith("reap"))
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
print(f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} {r.verdict:13} {r.reason}")
print(
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
)
deletable = [b for b in worktree_gc.audit_branches(repo_root) if b.verdict == "delete"]
if deletable:
print(f"{len(deletable)} local branch(es) fully merged/patch-equivalent upstream would also be deleted.")
return 0
def _prune(worktree_gc, repo_root: str, args) -> int:
dry_run = bool(getattr(args, "dry_run", False))
actions: list = []
if not getattr(args, "branches_only", False):
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
actions += worktree_gc.reclaim_worktrees(repo_root, dry_run=dry_run, records=tree_records)
kept = [r for r in tree_records if r.verdict == "keep"
and "kanban" not in r.reason and "in use" not in r.reason]
if kept:
print(f"Preserved {len(kept)} tree(s) with real work:")
for r in kept:
print(f" {r.name}: {r.reason}")
if not getattr(args, "trees_only", False):
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
if actions:
for line in actions:
print(f" {line}")
print(f"{len(actions)} action(s) {'planned' if dry_run else 'done'}.")
else:
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
return 0
_ACTIONS = {"list": _list, "prune": _prune}
def cmd_worktree(args) -> int:
@@ -35,65 +76,9 @@ def cmd_worktree(args) -> int:
if not repo_root:
print("Not inside a git repository (or pass --repo <path>).")
return 1
action = getattr(args, "worktree_action", None) or "list"
if action == "list":
records = worktree_gc.audit_worktrees(repo_root)
if not records:
print("No worktrees under .worktrees/ — nothing to reclaim.")
return 0
total_mb = sum(r.size_mb or 0 for r in records)
reapable_mb = sum(
r.size_mb or 0 for r in records if r.verdict.startswith("reap")
)
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
print(
f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} "
f"{r.verdict:13} {r.reason}"
)
print(
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
)
branch_records = worktree_gc.audit_branches(repo_root)
deletable = [b for b in branch_records if b.verdict == "delete"]
if deletable:
print(
f"{len(deletable)} local branch(es) fully merged/patch-equivalent "
f"upstream would also be deleted."
)
return 0
if action == "prune":
dry_run = bool(getattr(args, "dry_run", False))
trees_only = bool(getattr(args, "trees_only", False))
branches_only = bool(getattr(args, "branches_only", False))
actions: list = []
if not branches_only:
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
actions += worktree_gc.reclaim_worktrees(
repo_root, dry_run=dry_run, records=tree_records
)
kept = [r for r in tree_records if r.verdict == "keep"
and "kanban" not in r.reason and "in use" not in r.reason]
if kept:
print(f"Preserved {len(kept)} tree(s) with real work:")
for r in kept:
print(f" {r.name}: {r.reason}")
if not trees_only:
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
if actions:
for line in actions:
print(f" {line}")
verb = "planned" if dry_run else "done"
print(f"{len(actions)} action(s) {verb}.")
else:
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
return 0
print(f"Unknown worktree action: {action}")
return 1
handler = _ACTIONS.get(action)
if handler is None:
print(f"Unknown worktree action: {action}")
return 1
return handler(worktree_gc, repo_root, args)
+89 -175
View File
@@ -1,35 +1,11 @@
"""On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``).
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately
conservative and silent: it runs before the banner on every ``hermes -w``
launch, so it only reaps clean, fully-merged scratch trees past an age tier
and preserves everything else. That policy is correct for an unattended
startup path — but it means real installs accumulate two kinds of debris the
startup pass can never touch:
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately conservative and silent: it
runs before the banner on every ``hermes -w`` launch, so it only reaps clean, fully-merged scratch
trees past an age tier and preserves everything else.
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts,
logs) on an otherwise merged branch — preserved forever by the dirty guard.
- **Orphaned local branches** beyond the two auto-generated prefixes the
startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port
branches, feature branches whose PRs merged months ago. Multi-agent boxes
reach hundreds.
This module is the *attended* counterpart: an explicit, loud, dry-run-first
reclaim the user invokes, so it can be more thorough while staying just as
safe. Invariants shared with the startup pruner (never violated here either):
- tracked modifications are NEVER deleted, at any age, in any mode;
- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence
decides "unique"; shallow repos are deepened bloblessly first so the
verdict is trustworthy);
- live-locked trees (owning pid alive) are never touched;
- a branch is deleted only after its worktree removal succeeded — a failed
removal must not orphan reachable commits;
- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/``
before its tree is reaped, never destroyed.
Classification primitives are imported from ``cli`` so the two paths can
never drift apart on what "dirty", "unpushed", or "merged" means.
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts, logs) on an otherwise
merged branch — preserved forever by the dirty guard.
"""
from __future__ import annotations
@@ -79,11 +55,10 @@ class BranchRecord:
def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess:
"""Run git, translating timeouts into a nonzero returncode.
Every verdict in this module fails safe toward "keep" on a nonzero
returncode, so a hung/slow git call (large repos make ``git cherry``
genuinely slow) must degrade to keep — never crash the whole audit
(live-verified failure on a 746MB .git: TimeoutExpired escaped and
aborted the branch audit mid-list).
Every verdict in this module fails safe toward "keep" on a nonzero returncode, so a hung/slow
git call (large repos make ``git cherry`` genuinely slow) must degrade to keep — never crash the
whole audit (live-verified failure on a 746MB .git: TimeoutExpired escaped and aborted the
branch audit mid-list).
"""
try:
return subprocess.run(
@@ -98,42 +73,30 @@ def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess
)
def _tree_size_mb(path: Path) -> Optional[int]:
def _tree_size_mb(path: Path, timeout: int = 30) -> Optional[int]:
"""Cheap directory size via ``du -sm`` — best-effort, None on failure."""
try:
result = subprocess.run(
["du", "-sm", str(path)],
capture_output=True, text=True, encoding="utf-8",
errors="replace", timeout=30,
["du", "-sm", str(path)], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout
)
if result.returncode == 0 and result.stdout.strip():
return int(result.stdout.split()[0])
return int(result.stdout.split()[0]) if result.returncode == 0 and result.stdout.strip() else None
except Exception:
pass
return None
return None
def _dirty_split(path: str) -> tuple[bool, List[str]]:
"""Return (has_tracked_modifications, untracked_paths).
``git status --porcelain`` counts untracked scratch equally with real
edits; the reclaim policy treats them very differently (tracked = real
work, untracked = archivable scratch), so split here.
``git status --porcelain`` counts untracked scratch the same as real edits, but the reclaim
policy treats them differently (tracked = real work, untracked = archivable scratch).
"""
try:
result = _git(["status", "--porcelain"], cwd=path, timeout=10)
if result.returncode != 0:
return True, [] # fail safe: treat as real work
tracked = False
untracked: List[str] = []
for line in result.stdout.splitlines():
if not line.strip():
continue
if line.startswith("??"):
untracked.append(line[3:].strip())
else:
tracked = True
return tracked, untracked
lines = [line for line in result.stdout.splitlines() if line.strip()]
untracked = [line[3:].strip() for line in lines if line.startswith("??")]
return len(untracked) != len(lines), untracked
except Exception:
return True, []
@@ -141,15 +104,11 @@ def _dirty_split(path: str) -> tuple[bool, List[str]]:
def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
"""Copy untracked files out of a doomed tree. Returns the archive dir.
Never destroys: on any copy failure the caller must treat the tree as
keep. Costs almost nothing and removes the "did I just delete
something?" question.
Never destroys: on any copy failure the caller must treat the tree as keep. Costs almost nothing
and removes the "did I just delete something?" question.
"""
stamp = time.strftime("%Y%m%d-%H%M%S")
dest = (
Path.home() / ".hermes" / "archive" / "worktree-prune"
/ f"{tree.name}-{stamp}"
)
dest = Path.home() / ".hermes" / "archive" / "worktree-prune" / f"{tree.name}-{stamp}"
try:
for rel in untracked:
src = tree / rel
@@ -167,6 +126,34 @@ def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
return None
def _classify_tree(_cli, repo_root: str, entry: Path, merge_cache, remote_heads) -> tuple[str, str, List[str]]:
"""Return (verdict, reason, untracked) for one tree under ``.worktrees/``."""
path = str(entry)
if _KANBAN_RE.match(entry.name):
return "keep", "kanban task tree (owned by kanban gc)", []
if _cli._worktree_lock_is_live(repo_root, path, timeout=5) == "live":
return "keep", "in use by a running hermes session", []
tracked_dirty, untracked = _dirty_split(path)
if tracked_dirty:
return "keep", "uncommitted tracked changes (real work)", []
archive_note = f"{len(untracked)} untracked file(s) will be archived"
if _cli._worktree_has_unpushed_commits(path, timeout=5) and not _cli._worktree_commits_all_merged_upstream(
path, timeout=30, cache=merge_cache, max_ahead=_MAX_CHERRY_AHEAD
):
# Pushed-branch tier: single-branch fetch refspecs (the managed-install default) leave
# pushed PR branches with no refs/remotes/* entry, so `git log HEAD --not --remotes`
# reads them as unpushed forever. A local head that EXACTLY matches the remote branch
# has nothing origin lacks — the checkout is redundant; only the branch ref stays.
if not _cli._worktree_branch_pushed_exact(path, remote_heads, timeout=10):
return "keep", "unpushed commits not found upstream", []
if untracked:
return "reap-keep-branch", f"pushed to origin (open-PR lane); branch kept; {archive_note}", untracked
return "reap-keep-branch", "pushed to origin (open-PR lane); branch kept", []
if untracked:
return "reap-archive", f"merged/pushed; {archive_note}", untracked
return "reap", "clean and fully merged/pushed", []
def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]:
"""Classify every tree under ``.worktrees/`` without mutating anything."""
import cli as _cli # lazy: cli.py is heavy
@@ -194,75 +181,26 @@ def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeReco
age_days = (now - entry.stat().st_mtime) / 86400.0
except Exception:
continue
size_mb = _tree_size_mb(entry) if with_sizes else None
try:
branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5)
branch = branch_result.stdout.strip()
branch = _git(["branch", "--show-current"], cwd=str(entry), timeout=5).stdout.strip()
except Exception:
branch = ""
def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None):
records.append(TreeRecord(
name=entry.name, path=str(entry), branch=branch,
age_days=age_days, size_mb=size_mb,
verdict=verdict, reason=reason,
untracked=untracked or [],
))
if _KANBAN_RE.match(entry.name):
rec("keep", "kanban task tree (owned by kanban gc)")
continue
lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5)
if lock_state == "live":
rec("keep", "in use by a running hermes session")
continue
tracked_dirty, untracked = _dirty_split(str(entry))
if tracked_dirty:
rec("keep", "uncommitted tracked changes (real work)")
continue
if _cli._worktree_has_unpushed_commits(str(entry), timeout=5):
merged = _cli._worktree_commits_all_merged_upstream(
str(entry), timeout=30, cache=merge_cache,
max_ahead=_MAX_CHERRY_AHEAD,
)
if not merged:
# Pushed-branch tier: single-branch fetch refspecs (the
# managed-install default) leave pushed PR branches with no
# refs/remotes/* entry, so `git log HEAD --not --remotes`
# reads them as unpushed forever. A local head that EXACTLY
# matches the remote branch has nothing origin lacks — the
# checkout is redundant; only the branch ref stays.
if _cli._worktree_branch_pushed_exact(
str(entry), remote_heads, timeout=10
):
if untracked:
rec("reap-keep-branch",
f"pushed to origin (open-PR lane); branch kept; "
f"{len(untracked)} untracked file(s) will be archived",
untracked)
else:
rec("reap-keep-branch",
"pushed to origin (open-PR lane); branch kept")
continue
rec("keep", "unpushed commits not found upstream")
continue
if untracked:
rec("reap-archive",
f"merged/pushed; {len(untracked)} untracked file(s) will be archived",
untracked)
else:
rec("reap", "clean and fully merged/pushed")
verdict, reason, untracked = _classify_tree(_cli, repo_root, entry, merge_cache, remote_heads)
records.append(TreeRecord(
name=entry.name, path=str(entry), branch=branch,
age_days=age_days, size_mb=_tree_size_mb(entry) if with_sizes else None,
verdict=verdict, reason=reason, untracked=untracked,
))
if len(merge_cache) != cache_size_before:
_cli._save_worktree_merge_cache(merge_cache)
return records
_REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"}
def reclaim_worktrees(
repo_root: str,
*,
@@ -271,14 +209,13 @@ def reclaim_worktrees(
) -> List[str]:
"""Remove every reap-verdict tree from a frozen audit list.
Operates ONLY on the provided (or freshly computed) audit records — never
re-globs inside the destructive loop, so trees created by concurrent
sessions after the audit are out of scope by construction.
Operates ONLY on the provided (or freshly computed) audit records — never re-globs inside the
destructive loop, so trees created by concurrent sessions after the audit are out of scope by
construction.
"""
if records is None:
records = audit_worktrees(repo_root, with_sizes=False)
actions: List[str] = []
_REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"}
for record in records:
if record.verdict not in _REAP_VERDICTS:
continue
@@ -301,19 +238,12 @@ def reclaim_worktrees(
pass
try:
remove_result = _git(
["worktree", "remove", record.path, "--force"],
cwd=repo_root, timeout=30,
)
remove_result = _git(["worktree", "remove", record.path, "--force"], cwd=repo_root, timeout=30)
if remove_result.returncode != 0:
actions.append(
f"failed to remove {record.name}: {remove_result.stderr.strip()}"
)
actions.append(f"failed to remove {record.name}: {remove_result.stderr.strip()}")
continue
if record.verdict == "reap-keep-branch":
actions.append(
f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)"
)
actions.append(f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)")
continue
if record.branch and record.branch not in _PROTECTED_BRANCHES:
_git(["branch", "-D", record.branch], cwd=repo_root, timeout=10)
@@ -330,44 +260,39 @@ def reclaim_worktrees(
def audit_branches(repo_root: str) -> List[BranchRecord]:
"""Classify local branches: safe to delete when their content is on
upstream (fully merged OR every commit patch-equivalent via ``git
cherry``) and they are not checked out anywhere.
"""Classify local branches: deletable when their content is on upstream and not checked out.
Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``)
to EVERY local branch, because deletion is gated on content reachability
rather than name: a branch whose commits are all upstream loses nothing
when its ref goes. Branch names checked out in any worktree, protected
names, and branches with unique commits are kept.
"On upstream" means fully merged OR every commit patch-equivalent via ``git cherry``. Applies
to EVERY local branch, not just the startup pass's prefix list, because the gate is content
reachability rather than name. Checked-out, protected, and unique-commit branches are kept.
"""
import cli as _cli
if _cli._repo_is_shallow(repo_root):
_cli._deepen_shallow_repo(repo_root)
upstream = None
for candidate in ("origin/HEAD", "origin/main", "origin/master"):
probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5)
if probe.returncode == 0:
upstream = candidate
break
def _lines(result) -> List[str]:
return [b.strip() for b in result.stdout.splitlines() if b.strip()]
upstream = next(
(c for c in ("origin/HEAD", "origin/main", "origin/master")
if _git(["rev-parse", "--verify", "--quiet", c], cwd=repo_root, timeout=5).returncode == 0),
None,
)
if upstream is None:
return []
result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10)
if result.returncode != 0:
return []
branches = [b.strip() for b in result.stdout.splitlines() if b.strip()]
branches = _lines(result)
active: set = set()
wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10)
for line in wt.stdout.splitlines():
if line.startswith("branch refs/heads/"):
active.add(line.split("branch refs/heads/", 1)[-1].strip())
merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"],
cwd=repo_root, timeout=15)
merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()}
active = {
line.removeprefix("branch refs/heads/").strip()
for line in wt.stdout.splitlines() if line.startswith("branch refs/heads/")
}
merged = set(_lines(_git(["branch", "--merged", upstream, "--format=%(refname:short)"], cwd=repo_root, timeout=15)))
def _classify_branch(branch: str) -> BranchRecord:
if branch in _PROTECTED_BRANCHES or branch in active:
@@ -389,7 +314,7 @@ def audit_branches(repo_root: str) -> List[BranchRecord]:
cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30)
if cherry.returncode != 0:
return BranchRecord(branch, "keep", "could not verify (git cherry failed)")
lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()]
lines = _lines(cherry)
if lines and all(ln.startswith("-") for ln in lines):
return BranchRecord(branch, "delete", "all commits patch-equivalent upstream")
unique = sum(1 for ln in lines if ln.startswith("+"))
@@ -429,10 +354,10 @@ def reclaim_branches(
actions.append(f"would delete branch {record.name} ({record.reason})")
continue
result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10)
if result.returncode == 0:
actions.append(f"deleted branch {record.name}")
else:
actions.append(f"failed to delete {record.name}: {result.stderr.strip()}")
actions.append(
f"deleted branch {record.name}" if result.returncode == 0
else f"failed to delete {record.name}: {result.stderr.strip()}"
)
return actions
@@ -446,15 +371,4 @@ def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]:
count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir())
except Exception:
return 0, None
size_mb: Optional[int] = None
try:
result = subprocess.run(
["du", "-sm", str(worktrees_dir)],
capture_output=True, text=True, encoding="utf-8",
errors="replace", timeout=20,
)
if result.returncode == 0 and result.stdout.strip():
size_mb = int(result.stdout.split()[0])
except Exception:
pass
return count, size_mb
return count, _tree_size_mb(worktrees_dir, timeout=20)