refactor(hclib): cron/update/install lifecycle — cron status, update_* receipts and recovery, install repair, service manager
This commit is contained in:
+111
-158
@@ -1,24 +1,11 @@
|
||||
"""Dependency-light venv recovery that runs BEFORE hermes_cli.main's imports.
|
||||
|
||||
The ``hermes`` console entry point is ``hermes_cli.main:main``. Importing
|
||||
``hermes_cli.main`` pulls in third-party packages at module level (``dotenv``
|
||||
via ``hermes_cli.env_loader``, ``yaml`` via ``hermes_cli.config``, ...). In
|
||||
the exact failure state the update-recovery markers exist for — a failed lazy
|
||||
backend refresh or interrupted core install that wiped a core package's
|
||||
import files (#57828) — a normal launch crashes *while importing main.py*,
|
||||
before ``_recover_from_interrupted_install()`` can run. The marker system is
|
||||
unreachable precisely when it is needed most.
|
||||
This module is deliberately **stdlib-only** so importing it can never fail on a corrupted venv.
|
||||
``hermes_cli.main`` imports and calls :func:`recover_if_needed` at the very top of its module body,
|
||||
before any third-party import.
|
||||
|
||||
This module is deliberately **stdlib-only** so importing it can never fail on
|
||||
a corrupted venv. ``hermes_cli.main`` imports and calls
|
||||
:func:`recover_if_needed` at the very top of its module body, before any
|
||||
third-party import.
|
||||
|
||||
Scope: this early pass only repairs enough for ``hermes_cli.main`` to become
|
||||
importable again (force-reinstall of the known-fragile core packages, using
|
||||
the pins from pyproject.toml). It NEVER clears the recovery markers — the
|
||||
full, confirmed marker lifecycle stays with ``_recover_from_interrupted_install()``
|
||||
in main.py, which runs right after import succeeds.
|
||||
Scope: this early pass only repairs enough for ``hermes_cli.main`` to become importable again
|
||||
(force-reinstall of the known-fragile core packages, using the pins from pyproject.toml).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -83,16 +70,9 @@ def restore_quarantined_shims(
|
||||
) -> list[tuple[Path, Path]]:
|
||||
"""Rename quarantined shims back, retrying a lock instead of giving up.
|
||||
|
||||
``moved`` holds ``(original, quarantined)`` pairs. Returns the pairs that
|
||||
could NOT be restored, and prints an actionable recovery command for each.
|
||||
|
||||
A pair is not a failure when ``original`` already exists or ``quarantined``
|
||||
has gone: the installer wrote a fresh shim, or a concurrent sweep won the
|
||||
race. Both are silent, so two processes sweeping the same orphan cannot
|
||||
produce a spurious error.
|
||||
|
||||
Messages go to stderr by default -- the startup sweep runs on EVERY hermes
|
||||
invocation, and ``hermes acp`` speaks JSON-RPC on stdout.
|
||||
A pair is not a failure when ``original`` already exists or ``quarantined`` has gone: the
|
||||
installer wrote a fresh shim, or a concurrent sweep won the race. Both are silent, so two
|
||||
processes sweeping the same orphan cannot produce a spurious error.
|
||||
"""
|
||||
if stream is None:
|
||||
stream = sys.stderr
|
||||
@@ -153,12 +133,67 @@ def _project_root() -> Path:
|
||||
return Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def _load_pyproject_project(root: Path) -> dict | None:
|
||||
"""``[project]`` table of ``root/pyproject.toml``; ``None`` when missing/unreadable.
|
||||
|
||||
Stdlib-only (tomllib) — shared by every pyproject lookup in the repair path so ``packaging``
|
||||
is never needed in the broken-venv state.
|
||||
"""
|
||||
pyproject = root / "pyproject.toml"
|
||||
if not pyproject.is_file():
|
||||
return None
|
||||
try:
|
||||
import tomllib
|
||||
|
||||
with open(pyproject, "rb") as f:
|
||||
project = tomllib.load(f).get("project", {})
|
||||
except Exception:
|
||||
return None
|
||||
return project if isinstance(project, dict) else None
|
||||
|
||||
|
||||
def _read_marker_attempts(marker_path: Path) -> int:
|
||||
"""Attempt counter from a marker's opportunistic JSON body; corrupt/missing → 0."""
|
||||
try:
|
||||
raw = marker_path.read_text(encoding="utf-8", errors="replace").strip()
|
||||
except OSError:
|
||||
return 0
|
||||
if not raw:
|
||||
return 0
|
||||
try:
|
||||
import json
|
||||
|
||||
return int(json.loads(raw).get("attempts", 0))
|
||||
except (ValueError, AttributeError):
|
||||
return 0
|
||||
|
||||
|
||||
def _run_ensurepip(root: Path) -> None:
|
||||
"""Best-effort pip bootstrap — a killed install can leave the venv with no pip module at all."""
|
||||
try:
|
||||
subprocess.run(
|
||||
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
|
||||
cwd=root,
|
||||
capture_output=True,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _report_failed_install(result) -> bool:
|
||||
"""Replay the captured tail of a failed installer run to stderr; always ``False``."""
|
||||
tail = (result.stderr or result.stdout or "")[-2000:]
|
||||
if tail:
|
||||
print(tail, file=sys.stderr)
|
||||
return False
|
||||
|
||||
|
||||
def _pid_is_running(pid: int) -> bool:
|
||||
"""Best-effort stdlib-only process liveness probe.
|
||||
|
||||
``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process
|
||||
handle API there. An access-denied result is conservatively live: racing
|
||||
an elevated updater is worse than postponing recovery for one launch.
|
||||
``os.kill(pid, 0)`` is not a no-op on Windows, so use the Win32 process handle API there. An
|
||||
access-denied result counts as live: racing an elevated updater is worse than postponing
|
||||
recovery for one launch.
|
||||
"""
|
||||
if pid <= 0:
|
||||
return False
|
||||
@@ -217,20 +252,13 @@ def _marker_owner_is_live(marker: Path) -> bool:
|
||||
def _pinned_specs(packages: list[str], project_root: Path) -> list[str]:
|
||||
"""Map bare package names to their pinned specs from pyproject.toml.
|
||||
|
||||
Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may
|
||||
itself be broken in the failure state this module exists for). Unknown
|
||||
packages fall back to their bare name.
|
||||
Stdlib-only (tomllib + naive requirement-head parsing — ``packaging`` may itself be broken in
|
||||
the failure state this module exists for). Unknown packages fall back to their bare name.
|
||||
"""
|
||||
pyproject = project_root / "pyproject.toml"
|
||||
if not pyproject.is_file():
|
||||
return packages
|
||||
try:
|
||||
import tomllib
|
||||
|
||||
with open(pyproject, "rb") as f:
|
||||
raw_deps = tomllib.load(f).get("project", {}).get("dependencies", []) or []
|
||||
except Exception:
|
||||
project = _load_pyproject_project(project_root)
|
||||
if project is None:
|
||||
return packages
|
||||
raw_deps = project.get("dependencies", []) or []
|
||||
|
||||
name_to_spec: dict[str, str] = {}
|
||||
for spec in raw_deps:
|
||||
@@ -249,12 +277,9 @@ def _pinned_specs(packages: list[str], project_root: Path) -> list[str]:
|
||||
def _certifi_bundle_broken() -> bool:
|
||||
"""True when certifi imports but its ``cacert.pem`` is missing/corrupt.
|
||||
|
||||
A brew Python upgrade or an interrupted venv rebuild can leave certifi's
|
||||
distribution metadata (and even the module) intact while the bundled
|
||||
``cacert.pem`` is gone or a dangling symlink — every TLS connection then
|
||||
fails with an opaque ``Could not find a suitable TLS CA certificate
|
||||
bundle`` from deep inside httpx/requests (#29866). An attribute probe
|
||||
alone passes in that state, so validate the bundle path itself.
|
||||
A brew Python upgrade or interrupted venv rebuild can leave certifi's metadata intact while
|
||||
``cacert.pem`` is gone or a dangling symlink; every TLS connection then fails opaquely deep in
|
||||
httpx/requests. An attribute probe passes in that state, so validate the bundle path itself.
|
||||
"""
|
||||
try:
|
||||
import certifi
|
||||
@@ -271,12 +296,9 @@ def _certifi_bundle_broken() -> bool:
|
||||
def _probe_broken_packages() -> list[str]:
|
||||
"""Import-probe the fragile core packages in THIS process.
|
||||
|
||||
Returns repair package names (deduped, probe order) for modules that fail
|
||||
to import or lack their sentinel attribute. Failed imports leave nothing
|
||||
in ``sys.modules``, so a post-repair retry in the same process works.
|
||||
|
||||
certifi additionally gets a bundle-file check: the module can import
|
||||
cleanly while ``cacert.pem`` is missing (#29866).
|
||||
Returns repair package names (deduped, probe order) for modules that fail to import or lack
|
||||
their sentinel attribute. Failed imports leave nothing in ``sys.modules``, so a post-repair
|
||||
retry in the same process works.
|
||||
"""
|
||||
broken: list[str] = []
|
||||
for mod_name, attr in LAZY_REFRESH_IMPORT_PROBES:
|
||||
@@ -296,10 +318,9 @@ def _probe_broken_packages() -> list[str]:
|
||||
def _find_uv_binary() -> str | None:
|
||||
"""Locate a ``uv`` binary without importing third-party modules.
|
||||
|
||||
uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so
|
||||
the stdlib ``pip`` fallback below refuses to touch them. In that state
|
||||
the only sanctioned installer is uv itself, which Hermes already vendors
|
||||
(``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only.
|
||||
uv-managed base interpreters carry an ``EXTERNALLY-MANAGED`` marker, so the stdlib ``pip``
|
||||
fallback below refuses to touch them. In that state the only sanctioned installer is uv itself,
|
||||
which Hermes already vendors (``~/.hermes/bin/uv.exe``) or the user has on PATH. Stdlib-only.
|
||||
"""
|
||||
exe = "uv.exe" if sys.platform == "win32" else "uv"
|
||||
candidates = [
|
||||
@@ -316,40 +337,30 @@ def _find_uv_binary() -> str | None:
|
||||
def _base_interpreter_is_externally_managed() -> bool:
|
||||
"""True when ``sys.executable`` is a uv/standalone-builds managed install.
|
||||
|
||||
Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their
|
||||
stdlib (PEP 668), so ``python -m pip install`` aborts with
|
||||
``externally-managed-environment``. The early repair must then go
|
||||
through uv (or explicitly override pip) or the reinstall no-ops and the
|
||||
venv stays broken (#83569).
|
||||
Those interpreters ship an ``EXTERNALLY-MANAGED`` marker next to their stdlib (PEP 668), so
|
||||
``python -m pip install`` aborts with ``externally-managed-environment``. The early repair must
|
||||
then go through uv (or explicitly override pip) or the reinstall no-ops and the venv stays
|
||||
broken (#83569).
|
||||
"""
|
||||
try:
|
||||
import sysconfig
|
||||
|
||||
stdlib = Path(sysconfig.get_path("stdlib"))
|
||||
if (stdlib / "EXTERNALLY-MANAGED").exists():
|
||||
return True
|
||||
# uv 0.5+ moved the marker into a ``_uv_managed`` sentinel dir…
|
||||
if (stdlib.parent / "EXTERNALLY-MANAGED").exists():
|
||||
return True
|
||||
# uv 0.5+ moved the marker one level up next to a ``_uv_managed`` sentinel dir.
|
||||
return (stdlib / "EXTERNALLY-MANAGED").exists() or (
|
||||
stdlib.parent / "EXTERNALLY-MANAGED"
|
||||
).exists()
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
return False
|
||||
|
||||
|
||||
def _run_repair_install(specs: list[str], project_root: Path) -> bool:
|
||||
"""``uv pip`` (or stdlib ``pip``) force-reinstall of the given specs.
|
||||
|
||||
Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout);
|
||||
output is captured and replayed to stderr only on failure. Never raises.
|
||||
Streams nothing to stdout (``hermes acp`` speaks JSON-RPC on stdout); output is captured and
|
||||
replayed to stderr only on failure. Never raises.
|
||||
|
||||
Two installer paths, in priority order:
|
||||
|
||||
1. ``uv pip install`` with ``VIRTUAL_ENV`` pointed at the project venv —
|
||||
required when the base interpreter is uv-managed (Windows git checkouts
|
||||
install exactly this way: uv's Python declares PEP 668
|
||||
``EXTERNALLY-MANAGED`` and plain ``python -m pip`` refuses to run).
|
||||
2. ``sys.executable -m pip`` as before, for self-contained venvs whose
|
||||
interpreter carries no PEP 668 marker.
|
||||
"""
|
||||
externally_managed = _base_interpreter_is_externally_managed()
|
||||
if externally_managed:
|
||||
@@ -366,15 +377,10 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
|
||||
text=True, encoding="utf-8", errors="replace",
|
||||
env=env,
|
||||
)
|
||||
if result.returncode == 0:
|
||||
return True
|
||||
tail = (result.stderr or result.stdout or "")[-2000:]
|
||||
if tail:
|
||||
print(tail, file=sys.stderr)
|
||||
return False
|
||||
except Exception as exc:
|
||||
print(f" ✗ Early venv repair could not run uv: {exc}", file=sys.stderr)
|
||||
return False
|
||||
return result.returncode == 0 or _report_failed_install(result)
|
||||
# No uv available: fall through to pip with the PEP 668 override so
|
||||
# the repair at least attempts to fix the venv instead of no-oping.
|
||||
print(
|
||||
@@ -383,14 +389,7 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
try:
|
||||
subprocess.run(
|
||||
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
|
||||
cwd=project_root,
|
||||
capture_output=True,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
_run_ensurepip(project_root)
|
||||
pip_cmd = [sys.executable, "-m", "pip", "install", "--force-reinstall"]
|
||||
if externally_managed:
|
||||
pip_cmd.append("--break-system-packages")
|
||||
@@ -405,24 +404,18 @@ def _run_repair_install(specs: list[str], project_root: Path) -> bool:
|
||||
except Exception as exc:
|
||||
print(f" ✗ Early venv repair could not run pip: {exc}", file=sys.stderr)
|
||||
return False
|
||||
if result.returncode != 0:
|
||||
tail = (result.stderr or result.stdout or "")[-2000:]
|
||||
if tail:
|
||||
print(tail, file=sys.stderr)
|
||||
return False
|
||||
return True
|
||||
return result.returncode == 0 or _report_failed_install(result)
|
||||
|
||||
|
||||
def _pytest_owns_live_checkout(root: Path) -> bool:
|
||||
"""True when running under pytest AND ``root`` is this module's own
|
||||
checkout — the one whose venv is executing the suite right now.
|
||||
"""True when running under pytest AND ``root`` is this module's own checkout — the one whose
|
||||
venv is executing the suite right now.
|
||||
|
||||
Lifecycle tests spawn real subprocesses that import ``hermes_cli.main``
|
||||
with recovery armed; ``PYTEST_CURRENT_TEST`` rides the inherited env into
|
||||
those children. Without this guard, a genuinely-broken dev venv gets a
|
||||
REAL ``ensurepip`` + ``pip install --force-reinstall`` from inside a
|
||||
running test suite. Tests that sandbox ``project_root`` to a tmp_path are
|
||||
unaffected (same posture as ``managed_scope._under_pytest``)."""
|
||||
Lifecycle tests spawn real subprocesses that import ``hermes_cli.main`` with recovery armed and
|
||||
inherit ``PYTEST_CURRENT_TEST``. Without this guard a broken dev venv would get a REAL ensurepip
|
||||
+ ``pip install --force-reinstall`` from inside a running suite. Tests sandboxing
|
||||
``project_root`` to a tmp_path are unaffected.
|
||||
"""
|
||||
return (
|
||||
"PYTEST_CURRENT_TEST" in os.environ
|
||||
and root == Path(__file__).resolve().parent.parent
|
||||
@@ -435,14 +428,10 @@ def recover_if_needed(
|
||||
) -> None:
|
||||
"""Repair wiped core packages so ``hermes_cli.main`` can import at all.
|
||||
|
||||
Fast path (no marker present) is two ``lstat`` calls. Only acts when a
|
||||
recovery marker from a prior ``hermes update`` exists AND an import probe
|
||||
confirms a core package is actually broken. Markers are intentionally
|
||||
NOT cleared here — ``_recover_from_interrupted_install()`` in main.py owns
|
||||
the confirmed marker lifecycle and runs immediately after import succeeds.
|
||||
Fast path (no marker present) is two ``lstat`` calls. Only acts when a recovery marker from a
|
||||
prior ``hermes update`` exists AND an import probe confirms a core package is actually broken.
|
||||
|
||||
Never raises: on any failure the import of main.py proceeds and surfaces
|
||||
the real error.
|
||||
Never raises: on any failure the import of main.py proceeds and surfaces the real error.
|
||||
"""
|
||||
global _UPDATE_RETRY_RECOVERED
|
||||
|
||||
@@ -496,20 +485,8 @@ def recover_if_needed(
|
||||
|
||||
# Single-flight: share main.py's recovery lock so an early repair
|
||||
# never races a concurrent full recovery into the same shared venv.
|
||||
lock_path = root / ".update-incomplete.lock"
|
||||
try:
|
||||
fd = os.open(lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
||||
os.write(fd, f"{os.getpid()}\n".encode())
|
||||
os.close(fd)
|
||||
except FileExistsError:
|
||||
try:
|
||||
if time.time() - lock_path.stat().st_mtime > 3600:
|
||||
lock_path.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
if not _claim_recovery_lock(root):
|
||||
return
|
||||
except OSError:
|
||||
pass # read-only fs / perms — proceed unlocked, install surfaces it
|
||||
|
||||
try:
|
||||
specs = _pinned_specs(broken, root)
|
||||
@@ -531,10 +508,7 @@ def recover_if_needed(
|
||||
file=sys.stderr,
|
||||
)
|
||||
finally:
|
||||
try:
|
||||
lock_path.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
_release_recovery_lock(root)
|
||||
except Exception:
|
||||
# Never block launch — the import of main.py will surface the truth.
|
||||
pass
|
||||
@@ -581,21 +555,11 @@ def _release_recovery_lock(root: Path) -> None:
|
||||
def _complete_pending_core_install(root: Path, core_marker: Path) -> bool:
|
||||
"""Run the pending core install BEFORE main.py can import native modules.
|
||||
|
||||
``recover_if_needed`` invokes this when ``.update-incomplete`` exists —
|
||||
a prior ``hermes update`` (or the self-lock preflight, #83569) left the
|
||||
dependency sync deliberately unfinished. Completing it here matters on
|
||||
Windows: the deferral exists precisely because the process that wrote the
|
||||
marker had a native venv extension mapped; this process, running before
|
||||
``hermes_cli.main``'s third-party imports, maps nothing yet, so the
|
||||
installer can replace ``.pyd`` files without hitting the lock.
|
||||
``recover_if_needed`` invokes this when ``.update-incomplete`` exists — a prior ``hermes
|
||||
update`` (or the self-lock preflight, #83569) left the dependency sync deliberately unfinished.
|
||||
|
||||
Marker lifecycle: cleared on success; kept (attempts counter bumped) on
|
||||
failure for the next launch or main.py's post-import recovery. An
|
||||
attempts ceiling caps automatic retries so a persistent installer
|
||||
failure does not block every launch (``hermes acp`` included).
|
||||
|
||||
Never raises: any failure leaves the marker for the post-import path and
|
||||
returns ``False``. Returns ``True`` only after the install succeeds.
|
||||
Never raises: any failure leaves the marker for the post-import path and returns ``False``.
|
||||
Returns ``True`` only after the install succeeds.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli import _install_repair as ir
|
||||
@@ -603,18 +567,7 @@ def _complete_pending_core_install(root: Path, core_marker: Path) -> bool:
|
||||
# Retry backoff: read current attempts before claiming the lock so a
|
||||
# persistently-failing install stops hammering early. After the
|
||||
# increment the counter reflects THIS attempt.
|
||||
attempts = 0
|
||||
try:
|
||||
raw = core_marker.read_text(encoding="utf-8", errors="replace").strip()
|
||||
if raw:
|
||||
import json as _json
|
||||
|
||||
try:
|
||||
attempts = int(_json.loads(raw).get("attempts", 0))
|
||||
except (ValueError, AttributeError):
|
||||
attempts = 0
|
||||
except OSError:
|
||||
attempts = 0
|
||||
attempts = _read_marker_attempts(core_marker)
|
||||
|
||||
if attempts >= _EARLY_CORE_INSTALL_MAX_ATTEMPTS:
|
||||
print(
|
||||
|
||||
+94
-197
@@ -1,22 +1,13 @@
|
||||
"""Dependency install execution shared between early recovery and full recovery.
|
||||
|
||||
Both callers need to run the same core ``.[all]`` reinstall:
|
||||
- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE ``hermes_cli.main``'s
|
||||
third-party imports, so it can complete a pending update while no native extension is mapped yet
|
||||
(#83569). - ``hermes_cli.main._recover_core_update_marker_locked`` — the historical post-import
|
||||
recovery path.
|
||||
|
||||
- ``hermes_cli._early_recovery.recover_if_needed`` — stdlib-only, runs BEFORE
|
||||
``hermes_cli.main``'s third-party imports, so it can complete a pending
|
||||
update while no native extension is mapped yet (#83569).
|
||||
- ``hermes_cli.main._recover_core_update_marker_locked`` — the historical
|
||||
post-import recovery path. Kept as a fallback for installs the early pass
|
||||
could not complete (marker left in place on failure).
|
||||
|
||||
This module is deliberately **stdlib-only** so importing it can never fail in
|
||||
the corrupted-venv state it exists to repair. ``hermes_cli.main`` imports
|
||||
``managed_uv``, ``hermes_constants``, and friends only in its late path; the
|
||||
early path must not. Where the late path uses ``managed_uv.ensure_uv`` to
|
||||
bootstrap uv if missing, the early path uses the stdlib
|
||||
:func:`hermes_cli._early_recovery._find_uv_binary` lookup and falls back to
|
||||
plain pip when uv is absent — a degraded but working installer (the late
|
||||
recovery will bootstrap uv on the next launch if it ever matters).
|
||||
This module is deliberately **stdlib-only** so importing it can never fail in the corrupted-venv
|
||||
state it exists to repair. ``hermes_cli.main`` imports ``managed_uv``, ``hermes_constants``, and
|
||||
friends only in its late path; the early path must not.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -43,10 +34,7 @@ def _is_termux_env(env: dict | None = None) -> bool:
|
||||
"""Stdlib Termux probe (hermes_cli.main's version lives behind imports)."""
|
||||
env = env if env is not None else os.environ
|
||||
try:
|
||||
if env.get("TERMUX_VERSION"):
|
||||
return True
|
||||
prefix = env.get("PREFIX", "")
|
||||
return "com.termux" in prefix
|
||||
return bool(env.get("TERMUX_VERSION")) or "com.termux" in env.get("PREFIX", "")
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@@ -55,11 +43,9 @@ def _is_termux_env(env: dict | None = None) -> bool:
|
||||
def _stdout_to_stderr():
|
||||
"""Route fd 1 (and sys.stdout) to stderr for the duration of an install.
|
||||
|
||||
``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child
|
||||
writing there would corrupt the protocol. Mirrors
|
||||
``main.py::_recover_from_interrupted_install``.
|
||||
``hermes acp`` speaks JSON-RPC on stdout; an inherited-fd install child writing there would
|
||||
corrupt the protocol. Mirrors ``main.py::_recover_from_interrupted_install``.
|
||||
"""
|
||||
saved_fd = None
|
||||
saved_sys_stdout = sys.stdout
|
||||
try:
|
||||
saved_fd = os.dup(1)
|
||||
@@ -72,24 +58,18 @@ def _stdout_to_stderr():
|
||||
finally:
|
||||
sys.stdout = saved_sys_stdout
|
||||
if saved_fd is not None:
|
||||
try:
|
||||
with contextlib.suppress(OSError):
|
||||
os.dup2(saved_fd, 1)
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
with contextlib.suppress(OSError):
|
||||
os.close(saved_fd)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_install_target(root: Path) -> tuple[list[str], dict | None]:
|
||||
"""(install_cmd_prefix, env) for the project venv — stdlib uv lookup.
|
||||
|
||||
Mirrors ``main.py::_default_venv_install_target`` but without
|
||||
``managed_uv``. ``VIRTUAL_ENV`` steers ``uv pip`` at the project venv even
|
||||
when invoked from the base interpreter (the early-recovery case).
|
||||
Termux strips leaked interpreter-path env vars so uv resolves the venv
|
||||
correctly.
|
||||
Mirrors ``main.py::_default_venv_install_target`` without ``managed_uv``. ``VIRTUAL_ENV``
|
||||
steers ``uv pip`` at the project venv even when invoked from the base interpreter (early
|
||||
recovery). Termux strips leaked interpreter-path env vars so uv resolves the venv correctly.
|
||||
"""
|
||||
uv_bin = _er._find_uv_binary()
|
||||
if uv_bin:
|
||||
@@ -125,17 +105,34 @@ def _venv_scripts_dir(root: Path) -> Path | None:
|
||||
_WINDOWS_BIN_LAUNCHERS = ("hermes", "hermes-acp")
|
||||
|
||||
|
||||
def _launcher_present(target: Path, name: str) -> bool:
|
||||
return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists()
|
||||
|
||||
|
||||
def _launchers_missing(target: Path) -> bool:
|
||||
return any(not _launcher_present(target, name) for name in _WINDOWS_BIN_LAUNCHERS)
|
||||
|
||||
|
||||
def _default_hermes_root() -> Path | None:
|
||||
"""Per-machine anchor for the managed-clone gate — the DEFAULT Hermes root, not
|
||||
``get_hermes_home()``: under ``hermes -p <name>`` that returns ``profiles\\<name>``, which
|
||||
would fail the gate and silently skip the heal for profile users. ``None`` when unresolvable."""
|
||||
from hermes_constants import get_default_hermes_root
|
||||
|
||||
try:
|
||||
return Path(get_default_hermes_root())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _venv_is_relocatable(venv_dir: Path) -> bool:
|
||||
"""True when the venv's pyvenv.cfg declares ``relocatable = true``.
|
||||
|
||||
uv writes the flag; ``hermes_cli.managed_uv`` builds its replacement
|
||||
venvs with ``--relocatable`` (they are constructed aside and swapped
|
||||
into place). A relocatable venv's console-script trampolines embed a
|
||||
RELATIVE interpreter reference, so a COPY of one placed outside
|
||||
``venv\\Scripts`` fails at run time with ``uv trampoline failed to
|
||||
canonicalize script path``. Non-relocatable venvs (fresh installs)
|
||||
embed the absolute interpreter path and their trampolines survive
|
||||
copying. This flag decides which launcher form a PATH dir gets.
|
||||
uv writes the flag; ``managed_uv`` builds replacement venvs ``--relocatable``. A relocatable
|
||||
venv's console-script trampolines embed a RELATIVE interpreter reference, so a copy placed
|
||||
outside ``venv\Scripts`` fails with ``uv trampoline failed to canonicalize script path``;
|
||||
non-relocatable venvs embed the absolute path and survive copying. This decides which
|
||||
launcher form a PATH dir gets.
|
||||
"""
|
||||
try:
|
||||
cfg = (Path(venv_dir) / "pyvenv.cfg").read_text(
|
||||
@@ -143,20 +140,18 @@ def _venv_is_relocatable(venv_dir: Path) -> bool:
|
||||
)
|
||||
except OSError:
|
||||
return False
|
||||
for line in cfg.splitlines():
|
||||
key, _, value = line.partition("=")
|
||||
if key.strip().lower() == "relocatable" and value.strip().lower() == "true":
|
||||
return True
|
||||
return False
|
||||
return any(
|
||||
key.strip().lower() == "relocatable" and value.strip().lower() == "true"
|
||||
for key, _, value in (line.partition("=") for line in cfg.splitlines())
|
||||
)
|
||||
|
||||
|
||||
def _normalize_windows_path(value) -> str:
|
||||
"""Windows path equality key: backslashes, no trailing separator, lowered.
|
||||
|
||||
Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than
|
||||
``os.path.normcase`` — that is an identity function on POSIX, and this
|
||||
comparison must behave Windows-correct even when tests exercise the
|
||||
Windows branch from another host (same rationale as
|
||||
Lowercase via ``.lower()`` (what ``ntpath.normcase`` does) rather than ``os.path.normcase`` —
|
||||
that is an identity function on POSIX, and this comparison must behave Windows-correct even when
|
||||
tests exercise the Windows branch from another host (same rationale as
|
||||
``venv_bin_dir(windows=...)``).
|
||||
"""
|
||||
return str(value).replace("/", "\\").rstrip("\\").lower()
|
||||
@@ -165,8 +160,7 @@ def _normalize_windows_path(value) -> str:
|
||||
def _windows_user_path_entries() -> list[str]:
|
||||
"""User PATH entries from the registry — the value install.ps1 writes.
|
||||
|
||||
Falls back to the process PATH when the registry is unreadable. Only
|
||||
called on Windows.
|
||||
Falls back to the process PATH when the registry is unreadable. Only called on Windows.
|
||||
"""
|
||||
try:
|
||||
import winreg
|
||||
@@ -187,48 +181,13 @@ def ensure_windows_bin_launchers(
|
||||
) -> list[str]:
|
||||
"""Re-stage the Windows ``hermes`` launchers when they vanish.
|
||||
|
||||
On Windows, ``hermes`` resolves through launchers derived from the venv
|
||||
console scripts — never ``venv\\Scripts`` itself on PATH, which would
|
||||
shadow the user's ``python`` (#83797). The canonical launcher home is
|
||||
the managed binary dir — the default Hermes root's ``bin``
|
||||
(``%LOCALAPPDATA%\\hermes\\bin``, next to the managed uv) — which lives
|
||||
OUTSIDE the git checkout so no git operation can ever touch it. It is
|
||||
a per-machine dir shared by every profile: ``get_hermes_home()`` would
|
||||
point inside ``profiles\\<name>`` under ``hermes -p``, so the anchor
|
||||
here is :func:`hermes_constants.get_default_hermes_root`.
|
||||
On Windows, ``hermes`` resolves through launchers derived from the venv console scripts — never
|
||||
``venv\Scripts`` itself on PATH, which would shadow the user's ``python`` (#83797).
|
||||
|
||||
Earlier installer versions staged them at ``<checkout>\\bin`` instead —
|
||||
inside the git working tree — where ``hermes update``'s pre-update
|
||||
autostash (``git stash push --include-untracked``) swept them off disk;
|
||||
once the desktop updater stopped re-applying stashes (``--keep-stash``)
|
||||
nothing restored them and ``hermes`` stopped resolving in every new
|
||||
terminal. That legacy location is re-staged too, during the transition,
|
||||
for installs whose user PATH still resolves through it.
|
||||
|
||||
The launcher FORM depends on the venv (see :func:`_venv_is_relocatable`):
|
||||
a normal venv's exe trampoline embeds an absolute interpreter path and
|
||||
survives copying, so it is copied as ``<name>.exe``; a relocatable
|
||||
venv's trampoline resolves relative to its own location and a copy
|
||||
dies with ``uv trampoline failed to canonicalize script path``, so a
|
||||
``<name>.cmd`` delegator invoking the in-venv exe by absolute path is
|
||||
written instead. A name counts as present when EITHER form exists —
|
||||
exe copies staged before a venv rebuild keep working (they embed the
|
||||
swapped-in-place venv's absolute path) and are left alone.
|
||||
|
||||
Two targets, two gates, both failing toward inaction:
|
||||
|
||||
- canonical managed binary dir: only when *root* is the managed clone
|
||||
(``root.parent == get_default_hermes_root()``), so source checkouts
|
||||
elsewhere never gain launchers;
|
||||
- legacy ``<root>\\bin``: only when that dir is on the user PATH
|
||||
(registry value, process PATH as fallback), i.e. the install opted
|
||||
into the old layout and still resolves through it.
|
||||
|
||||
Writes go through a staging name + ``os.replace`` so concurrent process
|
||||
starts cannot tear a launcher. Never raises; returns the restored paths.
|
||||
|
||||
*windows* and *user_path_entries* are injectable for tests, same pattern
|
||||
as ``hermes_constants.venv_bin_dir``.
|
||||
- canonical managed binary dir: only when *root* is the managed clone (``root.parent ==
|
||||
get_default_hermes_root()``), so source checkouts elsewhere never gain launchers; - legacy
|
||||
``<root>\bin``: only when that dir is on the user PATH (registry value, process PATH as
|
||||
fallback), i.e.
|
||||
"""
|
||||
if windows is None:
|
||||
windows = _is_windows()
|
||||
@@ -237,20 +196,10 @@ def ensure_windows_bin_launchers(
|
||||
|
||||
root = Path(root)
|
||||
|
||||
# Per-machine anchor: the DEFAULT Hermes root, not get_hermes_home() —
|
||||
# under ``hermes -p <name>`` that returns ``profiles\\<name>``, which
|
||||
# would fail the managed-clone gate below and silently skip the heal
|
||||
# for profile users. The launcher dir serves the whole machine.
|
||||
from hermes_constants import get_default_hermes_root
|
||||
|
||||
try:
|
||||
home = Path(get_default_hermes_root())
|
||||
except Exception:
|
||||
home = _default_hermes_root()
|
||||
if home is None:
|
||||
return []
|
||||
|
||||
def _launcher_present(target: Path, name: str) -> bool:
|
||||
return (target / f"{name}.exe").exists() or (target / f"{name}.cmd").exists()
|
||||
|
||||
targets: list[Path] = []
|
||||
|
||||
# Canonical target — gate on the managed-clone shape. This runs at
|
||||
@@ -258,7 +207,7 @@ def ensure_windows_bin_launchers(
|
||||
# override), so the healthy path must stay at a couple of stat calls.
|
||||
if _normalize_windows_path(root.parent) == _normalize_windows_path(home):
|
||||
canonical = home / "bin"
|
||||
if any(not _launcher_present(canonical, name) for name in _WINDOWS_BIN_LAUNCHERS):
|
||||
if _launchers_missing(canonical):
|
||||
targets.append(canonical)
|
||||
|
||||
# Legacy transition target — the pre-migration in-checkout dir. Only
|
||||
@@ -268,7 +217,7 @@ def ensure_windows_bin_launchers(
|
||||
# network shares. An entry stored some other way (8.3 short path,
|
||||
# subst drive) misses the re-stage, which fails safe: no-op.
|
||||
legacy = root / "bin"
|
||||
if any(not _launcher_present(legacy, name) for name in _WINDOWS_BIN_LAUNCHERS):
|
||||
if _launchers_missing(legacy):
|
||||
if user_path_entries is None:
|
||||
user_path_entries = _windows_user_path_entries()
|
||||
configured = {_normalize_windows_path(entry) for entry in user_path_entries}
|
||||
@@ -330,8 +279,8 @@ def ensure_windows_bin_launchers(
|
||||
def _read_user_path_raw() -> tuple[list[str], int]:
|
||||
"""Raw (unexpanded) user PATH entries + registry value type.
|
||||
|
||||
Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them
|
||||
(same discipline as ``hermes_cli.uninstall``). Only called on Windows.
|
||||
Raw so a rewrite preserves ``%VARS%`` exactly as the user stored them (same discipline as
|
||||
``hermes_cli.uninstall``). Only called on Windows.
|
||||
"""
|
||||
import winreg
|
||||
|
||||
@@ -394,12 +343,10 @@ def migrate_windows_bin_path(
|
||||
|
||||
root = Path(root)
|
||||
|
||||
# Same per-machine anchor as ensure_windows_bin_launchers (see there).
|
||||
from hermes_constants import get_default_hermes_root, venv_bin_dir
|
||||
from hermes_constants import venv_bin_dir
|
||||
|
||||
try:
|
||||
home = Path(get_default_hermes_root())
|
||||
except Exception:
|
||||
home = _default_hermes_root()
|
||||
if home is None:
|
||||
return False
|
||||
if _normalize_windows_path(root.parent) != _normalize_windows_path(home):
|
||||
return False # not the managed clone — nothing to migrate
|
||||
@@ -457,17 +404,9 @@ def migrate_windows_bin_path(
|
||||
|
||||
def _load_console_script_names(root: Path) -> list[str]:
|
||||
"""``[project.scripts]`` names from pyproject.toml (tomllib, 3.11+)."""
|
||||
project = _er._load_pyproject_project(root)
|
||||
try:
|
||||
import tomllib
|
||||
except ImportError: # pragma: no cover
|
||||
return []
|
||||
pyproject = root / "pyproject.toml"
|
||||
if not pyproject.is_file():
|
||||
return []
|
||||
try:
|
||||
with open(pyproject, "rb") as f:
|
||||
data = tomllib.load(f)
|
||||
scripts = data.get("project", {}).get("scripts", {}) or {}
|
||||
scripts = (project or {}).get("scripts", {}) or {}
|
||||
return [str(name) for name in scripts if name]
|
||||
except Exception:
|
||||
return []
|
||||
@@ -476,10 +415,9 @@ def _load_console_script_names(root: Path) -> list[str]:
|
||||
class ShimQuarantineError(RuntimeError):
|
||||
"""A live shim could not be renamed aside — the venv is contended (#87331).
|
||||
|
||||
Raised BEFORE the install command runs. Callers (early-pass recovery,
|
||||
core-marker recovery) catch it like any install failure: the
|
||||
update-incomplete marker survives and a later launch retries once the
|
||||
holder exits — the contended venv is never mutated.
|
||||
Raised BEFORE the install command runs. Callers (early-pass recovery, core-marker recovery)
|
||||
catch it like any install failure: the update-incomplete marker survives and a later launch
|
||||
retries once the holder exits — the contended venv is never mutated.
|
||||
"""
|
||||
|
||||
def __init__(self, failed_shims: list[str]):
|
||||
@@ -494,14 +432,9 @@ def _quarantine_running_hermes_exe(
|
||||
) -> list[tuple[Path, Path]]:
|
||||
"""Rename live hermes*.exe shims aside so the installer can rewrite them.
|
||||
|
||||
Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort:
|
||||
silently skips anything that cannot be renamed. Returns (original,
|
||||
quarantined) pairs. stdlib-only — the console-script set comes from
|
||||
pyproject ``[project.scripts]`` (fallback: the well-known trio).
|
||||
|
||||
``failed_out``: when provided, names of shims that could not be renamed
|
||||
are appended so the caller can refuse instead of mutating a contended
|
||||
venv (#87331 fail-closed).
|
||||
Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort: silently skips anything
|
||||
that cannot be renamed. Returns (original, quarantined) pairs. stdlib-only — the console-script
|
||||
set comes from pyproject ``[project.scripts]`` (fallback: the well-known trio).
|
||||
"""
|
||||
if not _is_windows():
|
||||
return []
|
||||
@@ -529,11 +462,9 @@ def _quarantine_running_hermes_exe(
|
||||
def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None:
|
||||
"""Put quarantined shims back when the installer did not replace them.
|
||||
|
||||
Delegates to the shared helper in the stdlib-only ``_early_recovery``
|
||||
module: one retry ladder and one recovery message for every restore site,
|
||||
instead of the near-identical copies that had already drifted (#75584).
|
||||
Warnings land on stderr — this module runs in the early-recovery path and
|
||||
``hermes acp`` speaks JSON-RPC on stdout.
|
||||
Delegates to the shared helper in the stdlib-only ``_early_recovery`` module: one retry ladder
|
||||
and one recovery message for every restore site, instead of the near-identical copies that had
|
||||
already drifted (#75584).
|
||||
"""
|
||||
_er.restore_quarantined_shims(moved)
|
||||
|
||||
@@ -541,13 +472,11 @@ def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None:
|
||||
def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None:
|
||||
"""Run an install command with quarantine protection for venv shims.
|
||||
|
||||
Fail-closed (#87331): when any live shim cannot be renamed aside, the
|
||||
venv is contended and the installer would die partway on the same locks
|
||||
— raise :class:`ShimQuarantineError` WITHOUT running it. The caller's
|
||||
marker-keeping failure handling turns that into "retry next launch".
|
||||
Fail-closed (#87331): when any live shim cannot be renamed aside, the venv is contended and the
|
||||
installer would die partway on the same locks — raise :class:`ShimQuarantineError` WITHOUT
|
||||
running it. The caller's marker-keeping failure handling turns that into "retry next launch".
|
||||
|
||||
Raises CalledProcessError on install failure (callers implement the
|
||||
per-extra fallback ladder).
|
||||
Raises CalledProcessError on install failure (callers implement the per-extra fallback ladder).
|
||||
"""
|
||||
scripts_dir = _venv_scripts_dir(root) if _is_windows() else None
|
||||
failed: list[str] = []
|
||||
@@ -574,19 +503,14 @@ def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None:
|
||||
|
||||
def _load_installable_optional_extras(root: Path, group: str) -> list[str]:
|
||||
"""Optional extras referenced by a dependency group (all / termux-all)."""
|
||||
try:
|
||||
import tomllib
|
||||
|
||||
with (root / "pyproject.toml").open("rb") as handle:
|
||||
project = tomllib.load(handle).get("project", {})
|
||||
except Exception:
|
||||
project = _er._load_pyproject_project(root)
|
||||
if project is None:
|
||||
return []
|
||||
optional_deps = project.get("optional-dependencies", {})
|
||||
if not isinstance(optional_deps, dict):
|
||||
return []
|
||||
refs = optional_deps.get(group, [])
|
||||
referenced: list[str] = []
|
||||
for ref in refs:
|
||||
for ref in optional_deps.get(group, []):
|
||||
if "[" in ref and "]" in ref:
|
||||
name = ref.split("[", 1)[1].split("]", 1)[0]
|
||||
if name in optional_deps:
|
||||
@@ -597,34 +521,20 @@ def _load_installable_optional_extras(root: Path, group: str) -> list[str]:
|
||||
def run_core_install(root: Path) -> None:
|
||||
"""Full core ``.[all]`` editable reinstall — the recovery install.
|
||||
|
||||
Equal in behavior to the install half of
|
||||
``main.py::_recover_core_update_marker_locked``:
|
||||
Equal in behavior to the install half of ``main.py::_recover_core_update_marker_locked``:
|
||||
|
||||
- bootstrap pip via ensurepip (a killed install can leave the venv with no
|
||||
pip module at all)
|
||||
- prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back
|
||||
to ``python -m pip`` when no uv binary is available
|
||||
- target ``.[all]`` (or ``.[termux-all]`` on Termux) with the per-extra
|
||||
fallback ladder when the combined extras resolve fails
|
||||
- quarantine live ``hermes*.exe`` shims on Windows so they can be replaced
|
||||
- route ALL install output to stderr (acp/JSON-RPC safety)
|
||||
- Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env
|
||||
|
||||
Raises ``subprocess.CalledProcessError`` when even the base install fails;
|
||||
callers own marker lifecycle (clear on success, keep on failure).
|
||||
- bootstrap pip via ensurepip (a killed install can leave the venv with no pip module at all) -
|
||||
prefer ``uv pip`` with VIRTUAL_ENV pointed at the project venv; fall back to ``python -m pip``
|
||||
when no uv binary is available - target ``.[all]`` (or ``.[termux-all]`` on Termux) with the
|
||||
per-extra fallback ladder when the combined extras resolve fails - quarantine live
|
||||
``hermes*.exe`` shims on Windows so they can be replaced - route ALL install output to stderr
|
||||
(acp/JSON-RPC safety) - Termux strips leaked PYTHONPATH/PYTHONHOME from the uv env
|
||||
"""
|
||||
prefix, env = _resolve_install_target(root)
|
||||
group = "termux-all" if _is_termux_env(env) else "all"
|
||||
|
||||
with _stdout_to_stderr():
|
||||
try:
|
||||
subprocess.run(
|
||||
[sys.executable, "-m", "ensurepip", "--upgrade", "--default-pip"],
|
||||
cwd=root,
|
||||
capture_output=True,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
_er._run_ensurepip(root)
|
||||
|
||||
try:
|
||||
_run_install_cmd(
|
||||
@@ -669,24 +579,11 @@ def run_core_install(root: Path) -> None:
|
||||
def bump_marker_attempts(marker_path: Path) -> int:
|
||||
"""Increment an attempts counter stored inside the marker file.
|
||||
|
||||
The marker's existence is the signal; opportunistic JSON body carries the
|
||||
retry count so a persistently failing install can back off instead of
|
||||
reinstall-hammering every launch. Corrupt/missing bodies restart at 1.
|
||||
Returns the new attempt count. Never raises.
|
||||
The marker's existence is the signal; opportunistic JSON body carries the retry count so a
|
||||
persistently failing install can back off instead of reinstall-hammering every launch.
|
||||
Corrupt/missing bodies restart at 1. Returns the new attempt count. Never raises.
|
||||
"""
|
||||
attempts = 0
|
||||
try:
|
||||
raw = marker_path.read_text(encoding="utf-8", errors="replace").strip()
|
||||
if raw:
|
||||
try:
|
||||
attempts = int(json.loads(raw).get("attempts", 0))
|
||||
except (ValueError, AttributeError):
|
||||
attempts = 0
|
||||
except OSError:
|
||||
attempts = 0
|
||||
attempts += 1
|
||||
try:
|
||||
attempts = _er._read_marker_attempts(marker_path) + 1
|
||||
with contextlib.suppress(OSError):
|
||||
marker_path.write_text(json.dumps({"attempts": attempts}), encoding="utf-8")
|
||||
except OSError:
|
||||
pass
|
||||
return attempts
|
||||
|
||||
@@ -1,12 +1,7 @@
|
||||
"""``hermes_cli/_scan_venv_blockers.py`` — Standalone venv-process scan for JSON consumption.
|
||||
|
||||
Invoked by the Desktop Electron app::
|
||||
|
||||
venv\\Scripts\\python.exe -m hermes_cli._scan_venv_blockers
|
||||
|
||||
Exits 0 for valid clear or blocked results. Non-zero exit signals probe
|
||||
failure (the detector itself crashed, psutil unavailable, etc.). Exactly
|
||||
one JSON document on stdout; diagnostics on stderr only.
|
||||
Exits 0 for valid clear or blocked results. Non-zero exit signals probe failure (the detector itself
|
||||
crashed, psutil unavailable, etc.). Exactly one JSON document on stdout; diagnostics on stderr only.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -33,10 +28,9 @@ _SENSITIVE_LONG_FLAGS: list[str] = [
|
||||
def _probe_fail_json(diagnostic: str = "probe failed") -> str:
|
||||
"""Return the standard probe-failure JSON document.
|
||||
|
||||
``ok: false`` plus ``probe_failed: true`` means the detector itself could
|
||||
not run — this is *not* a clear scan. Callers must treat
|
||||
``ok is not True`` / non-zero exit as probe failure, never as
|
||||
``blocked: false`` "clear" (#83149).
|
||||
``ok: false`` plus ``probe_failed: true`` means the detector itself could not run — this is
|
||||
*not* a clear scan. Callers must treat ``ok is not True`` / non-zero exit as probe failure,
|
||||
never as ``blocked: false`` "clear" (#83149).
|
||||
"""
|
||||
return json.dumps(
|
||||
{
|
||||
@@ -57,11 +51,7 @@ def _emit_probe_fail(diagnostic: str) -> NoReturn:
|
||||
|
||||
|
||||
def _find_flag(text: str, flag: str) -> int:
|
||||
"""Return the index of *flag* when it starts the string or follows a space.
|
||||
|
||||
Returns -1 when not found. This avoids matching ``--token`` inside an
|
||||
embedded token or path like ``/some--token-thing``.
|
||||
"""
|
||||
"""Return the index of *flag* when it starts the string or follows a space."""
|
||||
low = text.lower()
|
||||
fl = flag.lower()
|
||||
pos = 0
|
||||
@@ -75,11 +65,7 @@ def _find_flag(text: str, flag: str) -> int:
|
||||
|
||||
|
||||
def _redact_sensitive_cmdline(cmdline: str) -> str:
|
||||
"""Apply generic secret redaction then long-flag redaction.
|
||||
|
||||
If the generic redactor itself fails, return ``"<redacted>"`` — the PID
|
||||
and process name still provide actionable diagnostics.
|
||||
"""
|
||||
"""Apply generic secret redaction then long-flag redaction."""
|
||||
# Generic pass: the project's shared secret redactor.
|
||||
try:
|
||||
from agent.redact import redact_sensitive_text # noqa: PLC0415
|
||||
@@ -94,14 +80,11 @@ def _redact_sensitive_cmdline(cmdline: str) -> str:
|
||||
# diagnostics (toolset, port, profile).
|
||||
earliest = len(cmdline)
|
||||
for flag in _SENSITIVE_LONG_FLAGS:
|
||||
# --flag=value → preserve "--flag="
|
||||
idx = _find_flag(cmdline, flag + "=")
|
||||
if idx != -1 and idx + len(flag) + 1 < earliest:
|
||||
earliest = idx + len(flag) + 1
|
||||
# --flag value → preserve "--flag "
|
||||
idx = _find_flag(cmdline, flag + " ")
|
||||
if idx != -1 and idx + len(flag) + 1 < earliest:
|
||||
earliest = idx + len(flag) + 1
|
||||
# --flag=value → preserve "--flag="; --flag value → preserve "--flag "
|
||||
for suffix in ("=", " "):
|
||||
idx = _find_flag(cmdline, flag + suffix)
|
||||
if idx != -1 and idx + len(flag) + 1 < earliest:
|
||||
earliest = idx + len(flag) + 1
|
||||
|
||||
if earliest < len(cmdline):
|
||||
return cmdline[:earliest] + "<redacted>"
|
||||
@@ -111,9 +94,8 @@ def _redact_sensitive_cmdline(cmdline: str) -> str:
|
||||
def _classify_local_preview_args(args: object) -> dict[str, object]:
|
||||
"""Return safe UI metadata for an exact ``python -m http.server`` argv.
|
||||
|
||||
The general holder detector intentionally truncates its diagnostic command
|
||||
line. Reading argv separately preserves a useful directory label without
|
||||
exposing an unbounded command line to the renderer.
|
||||
The general holder detector truncates its diagnostic command line; reading argv separately
|
||||
preserves a useful directory label without exposing an unbounded command line to the renderer.
|
||||
"""
|
||||
if not isinstance(args, (list, tuple)) or not all(isinstance(arg, str) for arg in args):
|
||||
return {}
|
||||
@@ -123,11 +105,10 @@ def _classify_local_preview_args(args: object) -> dict[str, object]:
|
||||
# to an unrelated script and must never authorize termination.
|
||||
if len(args) < 3 or args[1] != "-m" or args[2].lower() != "http.server":
|
||||
return {}
|
||||
module_index = 1
|
||||
|
||||
port = 8000
|
||||
if module_index + 2 < len(args):
|
||||
candidate = args[module_index + 2]
|
||||
if len(args) > 3:
|
||||
candidate = args[3]
|
||||
if candidate.isdigit() and 0 < int(candidate) <= 65535:
|
||||
port = int(candidate)
|
||||
|
||||
@@ -172,9 +153,9 @@ def _terminate_safe_preview(
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Terminate one verified local preview process tree.
|
||||
|
||||
A fresh ``psutil.Process`` identity check and exact argv classification occur
|
||||
immediately before termination. psutil guards mutating Process methods
|
||||
against PID reuse, avoiding taskkill's stale-PID race.
|
||||
A fresh ``psutil.Process`` identity check and exact argv classification occur immediately before
|
||||
termination. psutil guards mutating Process methods against PID reuse, avoiding taskkill's
|
||||
stale-PID race.
|
||||
"""
|
||||
try:
|
||||
if psutil_module is None:
|
||||
@@ -203,29 +184,13 @@ def _terminate_safe_preview(
|
||||
def _is_pausable_gateway(cmdline: str) -> bool:
|
||||
"""Return True when *cmdline* is a gateway process the updater can pause.
|
||||
|
||||
A running gateway shows up in the venv-holder scan as one or both halves
|
||||
of its launcher/worker chain (``venv\\Scripts\\python.exe -m
|
||||
hermes_cli.main gateway run`` and the uv-side interpreter re-running the
|
||||
same argv). Reporting those as blockers dead-ends the Desktop update:
|
||||
the preflight aborts with ``venv-blocked`` *before* spawning
|
||||
``hermes-setup``, so the CLI updater's own
|
||||
``_pause_windows_gateways_for_update()`` — which exists precisely to
|
||||
stop these processes (and is always active: ``hermes-setup`` invokes
|
||||
``hermes update --yes --gateway``) — never gets the chance to run.
|
||||
Only gateway invocations are exempted. Anything else running from the venv (an operator's REPL,
|
||||
a stray script, a ``serve`` backend that survived the desktop's own teardown) has no pause
|
||||
machinery downstream and must keep blocking the handoff.
|
||||
|
||||
Only gateway invocations are exempted. Anything else running from the
|
||||
venv (an operator's REPL, a stray script, a ``serve`` backend that
|
||||
survived the desktop's own teardown) has no pause machinery downstream
|
||||
and must keep blocking the handoff.
|
||||
|
||||
Delegates to ``gateway.status.looks_like_gateway_command_line`` — the
|
||||
canonical ``gateway run`` matcher (profile-selector aware, shlex
|
||||
tokenization, ``run``-only) — so this exemption, the pause discovery,
|
||||
and the updater's guard fallback all share one parser. A hand-rolled
|
||||
token scan here regressed ``--profile gateway gateway run``: the profile
|
||||
*value* shadowed the subcommand token. An import failure counts as
|
||||
not-pausable — the scan then reports the process as a blocker, which is
|
||||
exactly the pre-exemption behavior.
|
||||
Delegates to ``gateway.status.looks_like_gateway_command_line`` — the canonical ``gateway run``
|
||||
matcher (profile-selector aware, shlex tokenization, ``run``-only) — so this exemption, the
|
||||
pause discovery, and the updater's guard fallback all share one parser.
|
||||
"""
|
||||
try:
|
||||
from gateway.status import looks_like_gateway_command_line # noqa: PLC0415
|
||||
@@ -237,33 +202,11 @@ def _is_pausable_gateway(cmdline: str) -> bool:
|
||||
def _is_updater_owned_backend(pid: int, cmdline: str) -> bool:
|
||||
"""Return True when *pid* is a Hermes backend the CLI updater can stop.
|
||||
|
||||
The gateway exemption above keeps ``gateway run`` holders out of the
|
||||
blocker list because the updater's own pause machinery stops and resumes
|
||||
them. ``hermes serve`` / ``hermes dashboard`` backends had no such
|
||||
deferral, so a leaked serve child (or a Desktop-owned backend the
|
||||
teardown lost track of) dead-ended the hand-off with ``venv-blocked`` —
|
||||
or, worse, survived the hand-off and made the shim quarantine fail with
|
||||
``os error 32`` (#98336) — even though the updater downstream owns
|
||||
exactly this case with its ledger rungs (`_ledger_reapable_backend_pids`
|
||||
reaps dead-spawner orphans; `_ledger_manual_serve_holders` stops manual
|
||||
serves and relaunches them on their recorded host/port).
|
||||
The gateway exemption above keeps ``gateway run`` holders out of the blocker list because the
|
||||
updater's own pause machinery stops and resumes them.
|
||||
|
||||
Positive identity only — never name/substring matching (#90778, and the
|
||||
#99558 identity-guard contract):
|
||||
|
||||
- the argv's parsed SUBCOMMAND (token-based) is ``serve``/``dashboard``;
|
||||
- the machine spawn ledger has a live-verified ``(pid, create_time)``
|
||||
entry for the process with a matching purpose;
|
||||
- ownership is provable: the recorded spawner is dead or unrecorded
|
||||
(the updater's rungs stop/relaunch those), or the spawner is an
|
||||
ancestor of THIS scan — i.e. the Desktop app performing the hand-off,
|
||||
which exits before the updater runs, turning the backend into exactly
|
||||
the dead-spawner orphan the ledger rung reaps.
|
||||
|
||||
A backend whose recorded spawner is alive and is NOT this hand-off's
|
||||
Desktop (a second Desktop window, another supervisor) keeps blocking:
|
||||
that supervisor would respawn whatever the updater kills. Anything
|
||||
unprovable → not exempt (fail closed, pre-exemption behavior).
|
||||
Positive identity only — never name/substring matching (#90778, and the #99558 identity-guard
|
||||
contract):
|
||||
"""
|
||||
return _updater_owned_backend_entry(pid, cmdline) is not None
|
||||
|
||||
@@ -271,10 +214,9 @@ def _is_updater_owned_backend(pid: int, cmdline: str) -> bool:
|
||||
def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None:
|
||||
"""Ledger entry for a deferred backend, or ``None`` when it must block.
|
||||
|
||||
Same decision logic as ``_is_updater_owned_backend`` (which delegates
|
||||
here); returning the matched ledger entry lets ``main()`` emit sanitized
|
||||
decision evidence — structured identity fields only, never argv, which
|
||||
can carry tokens or private endpoints (#98350).
|
||||
Same decision logic as ``_is_updater_owned_backend`` (which delegates here); returning the
|
||||
matched ledger entry lets ``main()`` emit sanitized decision evidence — structured identity
|
||||
fields only, never argv, which can carry tokens or private endpoints (#98350).
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.update_cmd import _hermes_holder_subcommand # noqa: PLC0415
|
||||
@@ -312,10 +254,9 @@ def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None:
|
||||
def _deferred_backend_evidence(entries: list[dict]) -> list[dict]:
|
||||
"""Sanitized decision evidence for deferred serve/dashboard backends.
|
||||
|
||||
Structured ledger fields only — pid, purpose, recorded port — never the
|
||||
command line, which can carry tokens or private endpoints. Lets the
|
||||
scan result explain *why* a holder disappeared from ``processes``
|
||||
without echoing argv (#98350).
|
||||
Structured ledger fields only — pid, purpose, recorded port — never the command line, which can
|
||||
carry tokens or private endpoints. Lets the scan result explain *why* a holder disappeared from
|
||||
``processes`` without echoing argv (#98350).
|
||||
"""
|
||||
evidence = []
|
||||
for entry in entries:
|
||||
@@ -331,9 +272,9 @@ def _deferred_backend_evidence(entries: list[dict]) -> list[dict]:
|
||||
def _spawner_is_this_handoff_desktop(entry: dict) -> bool:
|
||||
"""True when the entry's live spawner is an ancestor of this scan.
|
||||
|
||||
The scan subprocess is spawned by the Desktop app's update preflight, so
|
||||
the Desktop performing the hand-off is in our ancestor chain. Identity is
|
||||
verified by ``(pid, create_time)`` — a recycled PID cannot forge the pair.
|
||||
The scan subprocess is spawned by the Desktop app's update preflight, so the Desktop performing
|
||||
the hand-off is in our ancestor chain. Identity is verified by ``(pid, create_time)`` — a
|
||||
recycled PID cannot forge the pair.
|
||||
"""
|
||||
spawner_pid = entry.get("spawner_pid")
|
||||
if not isinstance(spawner_pid, int) or spawner_pid <= 0:
|
||||
|
||||
+52
-104
@@ -1,26 +1,12 @@
|
||||
"""Pre-import startup fast paths — THE canonical lightweight helpers.
|
||||
|
||||
This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import
|
||||
wall (config, argparse tree, logging, providers). Everything here must stay
|
||||
**stdlib-only and cheap** (os/sys file probes; no yaml, no hermes_cli.config,
|
||||
no argparse). A guard test (``test_startup_fast_import_weight``) subprocess-
|
||||
imports this module and fails if any heavy module sneaks into sys.modules.
|
||||
This module is imported by ``hermes_cli/main.py`` BEFORE its heavy import wall (config, argparse
|
||||
tree, logging, providers). Everything here must stay **stdlib-only and cheap** (os/sys file probes;
|
||||
no yaml, no hermes_cli.config, no argparse).
|
||||
|
||||
Why this module exists (the bug class it kills): version-printing kept being
|
||||
reimplemented as ``*_fast()`` copies at the top of main.py (Termux first,
|
||||
then globally), each duplicating canonical logic — project-root resolution,
|
||||
container detection, profile detection. The copies drifted: eb4040242
|
||||
changed the canonical output and referenced ``PROJECT_ROOT`` inside the fast
|
||||
function, which doesn't exist yet on the fast path → the Termux fast path
|
||||
NameError'd on --version and nobody noticed. One implementation, imported
|
||||
by both the fast path and the module constants, makes that drift
|
||||
structurally impossible; the parity guard test would have caught eb4040242
|
||||
the day it landed.
|
||||
|
||||
``hermes_cli/config.py``'s ``get_container_exec_info()`` reads the same
|
||||
``.container-mode`` file; keep the file-format assumptions here and there in
|
||||
sync (this module deliberately only PROBES existence/typos cheaply and errs
|
||||
toward the slow path, which then does the authoritative parse).
|
||||
Why this module exists (the bug class it kills): version-printing kept being reimplemented as
|
||||
``*_fast()`` copies at the top of main.py (Termux first, then globally), each duplicating canonical
|
||||
logic — project-root resolution, container detection, profile detection.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,21 +15,23 @@ import os
|
||||
import sys
|
||||
|
||||
__all__ = [
|
||||
"project_root_str",
|
||||
"ensure_project_root_on_path",
|
||||
"is_termux_env",
|
||||
"is_termux_fast_version_argv",
|
||||
"is_global_fast_version_argv",
|
||||
"is_container_startup_environment",
|
||||
"active_profile_may_override_home",
|
||||
"container_mode_may_be_active",
|
||||
"read_openai_version",
|
||||
"read_install_method",
|
||||
"print_fast_version_info",
|
||||
"try_fast_version",
|
||||
"project_root_str", "ensure_project_root_on_path", "is_termux_env",
|
||||
"is_termux_fast_version_argv", "is_global_fast_version_argv",
|
||||
"is_container_startup_environment", "active_profile_may_override_home",
|
||||
"container_mode_may_be_active", "read_openai_version", "read_install_method",
|
||||
"print_fast_version_info", "try_fast_version",
|
||||
]
|
||||
|
||||
|
||||
def _read_text(path: str) -> str | None:
|
||||
"""Read a small text file, or None when it is missing/unreadable."""
|
||||
try:
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
return handle.read()
|
||||
except (OSError, UnicodeDecodeError):
|
||||
return None
|
||||
|
||||
|
||||
def project_root_str() -> str:
|
||||
"""Repo root as a str — the single source for main.py's PROJECT_ROOT."""
|
||||
return os.path.realpath(os.path.join(os.path.dirname(__file__), os.pardir))
|
||||
@@ -54,10 +42,8 @@ def ensure_project_root_on_path() -> None:
|
||||
project_root = project_root_str()
|
||||
normalized_root = os.path.normcase(os.path.realpath(project_root))
|
||||
sys.path[:] = [
|
||||
entry
|
||||
for entry in sys.path
|
||||
if not entry
|
||||
or os.path.normcase(os.path.realpath(entry)) != normalized_root
|
||||
entry for entry in sys.path
|
||||
if not entry or os.path.normcase(os.path.realpath(entry)) != normalized_root
|
||||
]
|
||||
sys.path.insert(0, project_root)
|
||||
|
||||
@@ -66,8 +52,7 @@ def is_termux_env() -> bool:
|
||||
"""Tiny Termux check for pre-import startup shortcuts."""
|
||||
prefix = os.environ.get("PREFIX", "")
|
||||
return bool(
|
||||
os.environ.get("TERMUX_VERSION")
|
||||
or "com.termux/files/usr" in prefix
|
||||
os.environ.get("TERMUX_VERSION") or "com.termux/files/usr" in prefix
|
||||
or prefix.startswith("/data/data/com.termux/")
|
||||
)
|
||||
|
||||
@@ -76,54 +61,39 @@ def is_termux_fast_version_argv(argv: list[str]) -> bool:
|
||||
return argv in (["--version"], ["-V"])
|
||||
|
||||
|
||||
def is_global_fast_version_argv(argv: list[str]) -> bool:
|
||||
return argv in (["--version"], ["-V"])
|
||||
is_global_fast_version_argv = is_termux_fast_version_argv
|
||||
|
||||
|
||||
def is_container_startup_environment() -> bool:
|
||||
"""True when we're already INSIDE a container (fast path is then safe)."""
|
||||
if os.path.exists("/.dockerenv") or os.path.exists("/run/.containerenv"):
|
||||
return True
|
||||
try:
|
||||
with open("/proc/1/cgroup", encoding="utf-8") as handle:
|
||||
cgroup = handle.read()
|
||||
except OSError:
|
||||
return False
|
||||
cgroup = _read_text("/proc/1/cgroup") or ""
|
||||
return "docker" in cgroup or "podman" in cgroup or "/lxc/" in cgroup
|
||||
|
||||
|
||||
def active_profile_may_override_home(hermes_root: str) -> bool:
|
||||
"""Cheap probe: does an active non-default profile redirect HERMES_HOME?"""
|
||||
active_profile = os.path.join(hermes_root, "active_profile")
|
||||
try:
|
||||
if os.path.exists(active_profile):
|
||||
with open(active_profile, encoding="utf-8") as handle:
|
||||
active = handle.read().strip()
|
||||
return bool(active and active != "default")
|
||||
except (OSError, UnicodeDecodeError):
|
||||
pass
|
||||
return False
|
||||
active = (_read_text(os.path.join(hermes_root, "active_profile")) or "").strip()
|
||||
return bool(active and active != "default")
|
||||
|
||||
|
||||
def _default_home() -> str:
|
||||
return os.path.join(os.path.expanduser("~"), ".hermes")
|
||||
|
||||
|
||||
def _resolved_home() -> str:
|
||||
hermes_home = os.environ.get("HERMES_HOME", "").strip()
|
||||
if hermes_home:
|
||||
return hermes_home
|
||||
return os.path.join(os.path.expanduser("~"), ".hermes")
|
||||
return os.environ.get("HERMES_HOME", "").strip() or _default_home()
|
||||
|
||||
|
||||
def container_mode_may_be_active() -> bool:
|
||||
"""Conservative probe for NixOS container-mode routing.
|
||||
|
||||
False positives are fine (we fall through to the slow path, whose
|
||||
``get_container_exec_info()`` does the authoritative check and routes
|
||||
into the container). False negatives are NOT fine — they'd print the
|
||||
host's version instead of the container's. Hence: any profile
|
||||
ambiguity → assume container mode may be active.
|
||||
False positives are fine (the slow path does the authoritative check). False negatives are NOT —
|
||||
they'd print the host's version instead of the container's — so any profile ambiguity means "may
|
||||
be active".
|
||||
"""
|
||||
if os.environ.get("HERMES_DEV") == "1":
|
||||
return False
|
||||
if is_container_startup_environment():
|
||||
if os.environ.get("HERMES_DEV") == "1" or is_container_startup_environment():
|
||||
return False
|
||||
|
||||
hermes_home = os.environ.get("HERMES_HOME", "").strip()
|
||||
@@ -131,15 +101,12 @@ def container_mode_may_be_active() -> bool:
|
||||
if os.path.exists(os.path.join(hermes_home, ".container-mode")):
|
||||
return True
|
||||
parent_name = os.path.basename(os.path.dirname(os.path.normpath(hermes_home)))
|
||||
return (
|
||||
parent_name != "profiles"
|
||||
and active_profile_may_override_home(hermes_home)
|
||||
)
|
||||
return parent_name != "profiles" and active_profile_may_override_home(hermes_home)
|
||||
|
||||
default_home = os.path.join(os.path.expanduser("~"), ".hermes")
|
||||
if active_profile_may_override_home(default_home):
|
||||
return True
|
||||
return os.path.exists(os.path.join(default_home, ".container-mode"))
|
||||
default_home = _default_home()
|
||||
return active_profile_may_override_home(default_home) or os.path.exists(
|
||||
os.path.join(default_home, ".container-mode")
|
||||
)
|
||||
|
||||
|
||||
def read_openai_version() -> str | None:
|
||||
@@ -165,31 +132,18 @@ def read_openai_version() -> str | None:
|
||||
def read_install_method() -> str | None:
|
||||
"""Read the installer's ``.install_method`` stamp, if present.
|
||||
|
||||
Only the stamp (step 1 of ``config.detect_install_method``'s resolution
|
||||
order) — the managed/git/pip fallbacks need heavier imports and stay on
|
||||
the slow path. On the fast path home ambiguity is already excluded:
|
||||
``container_mode_may_be_active()`` bails to the slow path whenever a
|
||||
non-default profile might redirect HERMES_HOME.
|
||||
Only the stamp (step 1 of ``config.detect_install_method``'s resolution order) — the
|
||||
managed/git/pip fallbacks need heavier imports and stay on the slow path.
|
||||
"""
|
||||
stamp = os.path.join(_resolved_home(), ".install_method")
|
||||
try:
|
||||
with open(stamp, encoding="utf-8") as handle:
|
||||
method = handle.read().strip().lower()
|
||||
return method or None
|
||||
except OSError:
|
||||
return None
|
||||
method = _read_text(os.path.join(_resolved_home(), ".install_method"))
|
||||
return (method or "").strip().lower() or None
|
||||
|
||||
|
||||
def print_fast_version_info(*, check_updates: bool = True) -> None:
|
||||
"""THE canonical ``hermes --version`` output (also used by /version).
|
||||
|
||||
The static lines print instantly from stdlib-only probes; everything
|
||||
heavier (upstream SHA in the version line, authoritative install-method
|
||||
detection, the update-status check) is lazy-imported AFTER the first
|
||||
line is already on screen, so perceived latency stays instant while the
|
||||
output carries the full information that used to require the (removed)
|
||||
``hermes version`` subcommand. Every lazy block degrades gracefully —
|
||||
a broken/heavy import can never take the basic version output down.
|
||||
Every lazy block degrades gracefully — a broken/heavy import can never take the basic version
|
||||
output down.
|
||||
"""
|
||||
# Line 1: registry-owned banner label (includes "· upstream <sha>" for
|
||||
# git installs). banner.py keeps rich/prompt_toolkit lazy, so this
|
||||
@@ -240,10 +194,7 @@ def print_fast_version_info(*, check_updates: bool = True) -> None:
|
||||
print(f"Update available — run '{recommended_update_command()}'")
|
||||
elif behind and behind > 0:
|
||||
commits_word = "commit" if behind == 1 else "commits"
|
||||
print(
|
||||
f"Update available: {behind} {commits_word} behind — "
|
||||
f"run '{recommended_update_command()}'"
|
||||
)
|
||||
print(f"Update available: {behind} {commits_word} behind — run '{recommended_update_command()}'")
|
||||
elif behind == 0:
|
||||
print("Up to date")
|
||||
except Exception:
|
||||
@@ -253,10 +204,9 @@ def print_fast_version_info(*, check_updates: bool = True) -> None:
|
||||
def try_fast_version(argv: list[str] | None = None) -> bool:
|
||||
"""Handle ``hermes --version`` before the heavy import wall.
|
||||
|
||||
Only ``--version``/``-V`` (the ``version`` subcommand was removed —
|
||||
``--version`` now carries the full output incl. update status), and
|
||||
never when container mode may need to route the command into the
|
||||
container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch.
|
||||
Only ``--version``/``-V`` (the ``version`` subcommand was removed — ``--version`` now carries
|
||||
the full output incl. update status), and never when container mode may need to route the
|
||||
command into the container. Termux keeps the HERMES_TERMUX_DISABLE_FAST_CLI escape hatch.
|
||||
"""
|
||||
if argv is None:
|
||||
argv = sys.argv[1:]
|
||||
@@ -266,9 +216,7 @@ def try_fast_version(argv: list[str] | None = None) -> bool:
|
||||
if is_termux:
|
||||
if not is_termux_fast_version_argv(argv):
|
||||
return False
|
||||
elif not is_global_fast_version_argv(argv):
|
||||
return False
|
||||
elif container_mode_may_be_active():
|
||||
elif not is_global_fast_version_argv(argv) or container_mode_may_be_active():
|
||||
return False
|
||||
|
||||
print_fast_version_info()
|
||||
|
||||
+38
-64
@@ -1,26 +1,9 @@
|
||||
"""
|
||||
Baked-in build metadata for Hermes Agent.
|
||||
"""Baked-in build metadata for Hermes Agent.
|
||||
|
||||
Source installs report their git revision live via ``git rev-parse`` (see
|
||||
``hermes_cli/dump.py`` and ``hermes_cli/banner.py``). That doesn't work inside
|
||||
the published Docker image because ``.dockerignore`` excludes ``.git``, so
|
||||
those callsites fall back to ``"(unknown)"`` / drop the banner suffix entirely.
|
||||
|
||||
To make ``hermes dump`` and the startup banner identify the exact commit the
|
||||
image was built from, the Docker build writes the build-time ``$HERMES_GIT_SHA``
|
||||
arg into ``<project_root>/.hermes_build_sha``. This module is the single
|
||||
read-side helper consumed by both callsites — keeping the lookup in one place
|
||||
so the file path and missing-file behaviour stay consistent.
|
||||
|
||||
Behaviour:
|
||||
|
||||
- Returns ``None`` when the file is absent. Source installs and dev images
|
||||
built without the ``HERMES_GIT_SHA`` build-arg fall through to live-git
|
||||
resolution in the caller, so non-Docker installs are unaffected.
|
||||
- Returns ``None`` on any IO / decoding error. The build-sha is a nice-to-have
|
||||
for support triage; nothing in the CLI is allowed to crash because of it.
|
||||
- Truncates to ``short`` characters (default 8) to match the format used by
|
||||
``git rev-parse --short=8`` throughout the codebase.
|
||||
Source installs report their git revision live via ``git rev-parse`` (see ``hermes_cli/dump.py`` and
|
||||
``hermes_cli/banner.py``). That doesn't work inside the published Docker image because
|
||||
``.dockerignore`` excludes ``.git``, so those callsites fall back to ``"(unknown)"`` / drop the
|
||||
banner suffix entirely.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -36,22 +19,27 @@ _BUILD_SHA_FILE = Path(__file__).parent.parent / ".hermes_build_sha"
|
||||
_code_identity_cache: Optional[dict] = None
|
||||
|
||||
|
||||
def _read_stripped(path: Path) -> str:
|
||||
return path.read_text(encoding="utf-8", errors="replace").strip()
|
||||
|
||||
|
||||
def _sha_or_none(value: str) -> Optional[str]:
|
||||
return value if len(value) == 40 else None
|
||||
|
||||
|
||||
def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
|
||||
"""Resolve the checkout's HEAD commit sha by reading .git directly.
|
||||
|
||||
Deliberately NOT ``git rev-parse`` in a subprocess: this helper runs
|
||||
inside library paths (gateway runtime-status writes, update receipts)
|
||||
where spawning processes is both slow and hostile to tests that mock
|
||||
``subprocess.run`` tightly (call-count asserts, sequenced side effects).
|
||||
Handles regular checkouts, worktrees/submodules (``.git`` file with a
|
||||
``gitdir:`` pointer + ``commondir``), loose refs, and packed-refs.
|
||||
Deliberately NOT ``git rev-parse``: this runs in library paths (runtime-status writes, update
|
||||
receipts) where spawning is slow and hostile to tests that mock ``subprocess.run`` tightly.
|
||||
Handles worktrees/submodules (``gitdir:`` + ``commondir``), loose refs, and packed-refs.
|
||||
Returns None on any failure.
|
||||
"""
|
||||
try:
|
||||
git_path = project_root / ".git"
|
||||
if git_path.is_file():
|
||||
# Worktree/submodule: ".git" is a pointer file.
|
||||
pointer = git_path.read_text(encoding="utf-8", errors="replace").strip()
|
||||
pointer = _read_stripped(git_path)
|
||||
if not pointer.startswith("gitdir:"):
|
||||
return None
|
||||
git_dir = Path(pointer[len("gitdir:"):].strip())
|
||||
@@ -66,33 +54,28 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
|
||||
common_dir = git_dir
|
||||
commondir_file = git_dir / "commondir"
|
||||
if commondir_file.is_file():
|
||||
rel = commondir_file.read_text(encoding="utf-8", errors="replace").strip()
|
||||
common = Path(rel)
|
||||
if not common.is_absolute():
|
||||
common = (git_dir / common).resolve()
|
||||
common_dir = common
|
||||
common = Path(_read_stripped(commondir_file))
|
||||
common_dir = common if common.is_absolute() else (git_dir / common).resolve()
|
||||
|
||||
head = (git_dir / "HEAD").read_text(encoding="utf-8", errors="replace").strip()
|
||||
head = _read_stripped(git_dir / "HEAD")
|
||||
if not head.startswith("ref:"):
|
||||
# Detached HEAD: the file holds the sha itself.
|
||||
return head if len(head) == 40 else None
|
||||
return _sha_or_none(head)
|
||||
ref_name = head[len("ref:"):].strip()
|
||||
|
||||
loose = common_dir / ref_name
|
||||
if loose.is_file():
|
||||
sha = loose.read_text(encoding="utf-8", errors="replace").strip()
|
||||
return sha if len(sha) == 40 else None
|
||||
return _sha_or_none(_read_stripped(loose))
|
||||
|
||||
packed = common_dir / "packed-refs"
|
||||
if packed.is_file():
|
||||
for line in packed.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
for line in _read_stripped(packed).splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith(("#", "^")):
|
||||
continue
|
||||
parts = line.split(" ", 1)
|
||||
if len(parts) == 2 and parts[1].strip() == ref_name:
|
||||
sha = parts[0].strip()
|
||||
return sha if len(sha) == 40 else None
|
||||
return _sha_or_none(parts[0].strip())
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
@@ -101,34 +84,26 @@ def _resolve_git_head_sha(project_root: Path) -> Optional[str]:
|
||||
def get_code_identity(refresh: bool = False) -> dict:
|
||||
"""Return the running checkout's code identity as a dict.
|
||||
|
||||
Shape: ``{"sha": full-or-short sha | None, "short_sha": str | None,
|
||||
"version": pyproject version | None, "source": "git" | "build-file" |
|
||||
"unknown"}``.
|
||||
Resolution order mirrors the banner/dump callsites: live ``git rev-parse`` for source installs,
|
||||
the baked ``.hermes_build_sha`` for Docker images (no ``.git`` inside the published image), else
|
||||
unknown.
|
||||
|
||||
Resolution order mirrors the banner/dump callsites: live ``git
|
||||
rev-parse`` for source installs, the baked ``.hermes_build_sha`` for
|
||||
Docker images (no ``.git`` inside the published image), else unknown.
|
||||
|
||||
Cached per process — code identity cannot change while a process is
|
||||
running (an updated checkout requires a restart to take effect, which
|
||||
is exactly the property the fleet version verification relies on).
|
||||
Never raises; every field degrades to ``None`` independently.
|
||||
Cached per process — code identity cannot change while a process is running (an updated checkout
|
||||
requires a restart to take effect, which is exactly the property the fleet version verification
|
||||
relies on). Never raises; every field degrades to ``None`` independently.
|
||||
"""
|
||||
global _code_identity_cache
|
||||
if _code_identity_cache is not None and not refresh:
|
||||
return dict(_code_identity_cache)
|
||||
|
||||
sha: Optional[str] = None
|
||||
source = "unknown"
|
||||
project_root = Path(__file__).parent.parent
|
||||
resolved = _resolve_git_head_sha(project_root)
|
||||
if resolved:
|
||||
sha = resolved
|
||||
source = "unknown"
|
||||
sha = _resolve_git_head_sha(project_root)
|
||||
if sha:
|
||||
source = "git"
|
||||
if sha is None:
|
||||
baked = get_build_sha(short=0)
|
||||
if baked:
|
||||
sha = baked
|
||||
else:
|
||||
sha = get_build_sha(short=0)
|
||||
if sha:
|
||||
source = "build-file"
|
||||
|
||||
version: Optional[str] = None
|
||||
@@ -153,9 +128,8 @@ def get_code_identity(refresh: bool = False) -> dict:
|
||||
def get_build_sha(short: int = 8) -> Optional[str]:
|
||||
"""Return the baked-in build SHA, truncated to ``short`` chars, or None.
|
||||
|
||||
Reads ``<project_root>/.hermes_build_sha`` if present. The file is
|
||||
written by the Dockerfile's ``HERMES_GIT_SHA`` build-arg and contains
|
||||
the full 40-character commit hash on a single line.
|
||||
Reads ``<project_root>/.hermes_build_sha``, written by the Dockerfile's ``HERMES_GIT_SHA``
|
||||
build-arg (full 40-char hash on one line).
|
||||
"""
|
||||
try:
|
||||
if not _BUILD_SHA_FILE.is_file():
|
||||
|
||||
+41
-136
@@ -1,21 +1,8 @@
|
||||
"""Container-boot reconciliation of per-profile gateway s6 services.
|
||||
|
||||
Service directories under /run/service/ live on **tmpfs** and are wiped
|
||||
on every container restart. Profile directories under
|
||||
``$HERMES_HOME/profiles/<name>/`` live on the persistent VOLUME, and
|
||||
each one records its gateway's last state in ``gateway_state.json``.
|
||||
This module bridges the two: on every container boot, walk the
|
||||
persistent profiles, recreate the s6 service slots, and auto-start
|
||||
only those whose last recorded state was ``running``.
|
||||
|
||||
Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the
|
||||
Dockerfile (Phase 4 Task 4.0). Runs as root after 01-hermes-setup
|
||||
(the stage2 hook) has chowned the volume and seeded $HERMES_HOME, but
|
||||
before s6-rc starts user services.
|
||||
|
||||
Without this module, every ``docker restart`` would silently wipe
|
||||
every per-profile gateway, even though the user's profiles still
|
||||
exist on disk.
|
||||
Wired into the image as /etc/cont-init.d/02-reconcile-profiles by the Dockerfile (Phase 4 Task 4.0).
|
||||
Runs as root after 01-hermes-setup (the stage2 hook) has chowned the volume and seeded $HERMES_HOME,
|
||||
but before s6-rc starts user services.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -101,34 +88,10 @@ def reconcile_profile_gateways(
|
||||
) -> list[ReconcileAction]:
|
||||
"""Recreate s6 service registrations for every persistent profile.
|
||||
|
||||
Always registers a ``gateway-default`` slot for the root profile
|
||||
(the implicit profile that lives at the top of ``$HERMES_HOME``,
|
||||
not under ``profiles/``). The dispatcher in ``hermes_cli.gateway``
|
||||
maps an empty profile suffix to ``gateway-default``, so this slot
|
||||
is what ``hermes gateway start`` (no ``-p``) targets. Without it,
|
||||
bare ``hermes gateway start`` inside the container would land on
|
||||
``s6-svc -u /run/service/gateway-default`` → uncaught
|
||||
``CalledProcessError`` → traceback to the user (PR #30136 review).
|
||||
|
||||
The default slot's prior state is read from
|
||||
``$HERMES_HOME/gateway_state.json`` (sibling to the profile root,
|
||||
not under ``profiles/``); stale runtime files there are swept the
|
||||
same way as for named profiles.
|
||||
|
||||
Args:
|
||||
hermes_home: The container's HERMES_HOME (typically /opt/data).
|
||||
Profiles live under ``<hermes_home>/profiles/<name>/``;
|
||||
the default profile lives at ``<hermes_home>`` itself.
|
||||
scandir: The s6 dynamic scandir (typically /run/service). Service
|
||||
directories are created at ``<scandir>/gateway-<profile>/``.
|
||||
dry_run: When True, walk and return the action list without
|
||||
touching the filesystem. For tests and `--dry-run` debug.
|
||||
container_argv: Optional container PID 1 argv override. Production
|
||||
reads ``/proc/1/cmdline``; tests inject it directly.
|
||||
|
||||
Returns:
|
||||
One :class:`ReconcileAction` per profile, in this order:
|
||||
``default`` first, then named profiles in directory order.
|
||||
Always registers a ``gateway-default`` slot for the root profile (the implicit profile that
|
||||
lives at the top of ``$HERMES_HOME``, not under ``profiles/``). The dispatcher in
|
||||
``hermes_cli.gateway`` maps an empty profile suffix to ``gateway-default``, so this slot is what
|
||||
``hermes gateway start`` (no ``-p``) targets.
|
||||
"""
|
||||
actions: list[ReconcileAction] = []
|
||||
|
||||
@@ -231,13 +194,10 @@ def _maybe_migrate_legacy_gateway_run_state(
|
||||
) -> str | None:
|
||||
"""Seed root gateway_state for pre-s6 `gateway run` containers.
|
||||
|
||||
The tini image let Docker users run the gateway as the container
|
||||
command (`docker run ... gateway run`). After the s6 migration,
|
||||
profile gateways are restored from persisted gateway_state.json; a
|
||||
legacy container with no state file would therefore register the
|
||||
default service down and never start. Only synthesize state when no
|
||||
root gateway_state.json exists so explicit stopped/failed states keep
|
||||
winning across restarts.
|
||||
The tini image let Docker users run the gateway as the container command (`docker run ...
|
||||
gateway run`). After the s6 migration, profile gateways are restored from persisted
|
||||
gateway_state.json; a legacy container with no state file would therefore register the default
|
||||
service down and never start.
|
||||
"""
|
||||
state_file = hermes_home / "gateway_state.json"
|
||||
if state_file.exists():
|
||||
@@ -264,13 +224,9 @@ def _maybe_migrate_legacy_gateway_run_state(
|
||||
def _read_container_argv() -> tuple[str, ...]:
|
||||
"""Best-effort read of the container's main program argv.
|
||||
|
||||
Under s6-overlay v2, PID 1 is ``/init`` and its argv contains the
|
||||
``main-wrapper.sh`` path. Under s6-overlay v3, PID 1 is
|
||||
``s6-svscan`` and the actual command (``rc.init top main-wrapper.sh
|
||||
...``) lives on a different PID. We try PID 1 first (fast path,
|
||||
covers v2 and pre-s6 images), then fall back to scanning
|
||||
``/proc/*/cmdline`` for a process whose argv contains
|
||||
``main-wrapper.sh`` (the rc.init-launched PID in v3).
|
||||
s6-overlay v2: PID 1 is ``/init`` and its argv holds ``main-wrapper.sh``. v3: PID 1 is
|
||||
``s6-svscan`` and the real command lives on another PID, so after the PID 1 fast path we scan
|
||||
``/proc/*/cmdline`` for a process whose argv contains ``main-wrapper.sh``.
|
||||
"""
|
||||
# Fast path: PID 1 is the command itself (s6-overlay v2 / tini).
|
||||
try:
|
||||
@@ -310,24 +266,10 @@ def _read_container_argv() -> tuple[str, ...]:
|
||||
def _strip_container_argv_prefix(argv: Sequence[str]) -> list[str]:
|
||||
"""Strip the s6/wrapper prefix off the container argv, leaving the hermes args.
|
||||
|
||||
Two container-command argv shapes are handled:
|
||||
|
||||
* **s6-overlay v2 / tini:** PID 1 argv is
|
||||
``/init /opt/hermes/docker/main-wrapper.sh <subcommand> [args...]``.
|
||||
* **s6-overlay v3:** PID 1 is ``s6-svscan`` and the command lives on the
|
||||
rc.init-launched process as ``/bin/sh -e
|
||||
/run/s6/basedir/scripts/rc.init top /opt/hermes/docker/main-wrapper.sh
|
||||
<subcommand> [args...]`` (see :func:`_read_container_argv`).
|
||||
|
||||
Rather than peel each leading token positionally (which silently breaks
|
||||
the moment s6 changes its launcher shape again — exactly what happened
|
||||
in the v2→v3 bump), drop everything up to and including the
|
||||
``main-wrapper.sh`` token: that wrapper path is the stable boundary the
|
||||
image owns, and the subcommand always follows it. Pre-s6 / direct
|
||||
``hermes`` invocations carry no wrapper, so fall back to peeling a bare
|
||||
``init`` prefix. The wrapper re-execs ``hermes <subcommand>``, so an
|
||||
explicit leading ``hermes`` is peeled too. Shared by the legacy-gateway
|
||||
and dashboard role detectors.
|
||||
Rather than peel each leading token positionally (which silently breaks the moment s6 changes
|
||||
its launcher shape again — exactly what happened in the v2→v3 bump), drop everything up to and
|
||||
including the ``main-wrapper.sh`` token: that wrapper path is the stable boundary the image
|
||||
owns, and the subcommand always follows it.
|
||||
"""
|
||||
args = list(argv)
|
||||
|
||||
@@ -365,19 +307,8 @@ def _is_legacy_gateway_run_request(argv: Sequence[str]) -> bool:
|
||||
def _is_dashboard_container(argv: Sequence[str]) -> bool:
|
||||
"""Return True when the container's command is the dashboard.
|
||||
|
||||
A dashboard-only container (``hermes dashboard ...``) never spawns or
|
||||
supervises per-profile gateways — that is the gateway container's job.
|
||||
Reconciling profile gateway s6 slots there is not just wasted work: when
|
||||
the gateway and dashboard containers share a bind-mounted HERMES_HOME,
|
||||
both race to ``flock()`` the same ``logs/gateways/<profile>/lock`` files,
|
||||
producing "Resource busy" failures and an s6-log restart storm. So the
|
||||
dashboard container skips reconciliation entirely.
|
||||
|
||||
Detected from PID 1 argv (``/proc/1/cmdline``) rather than an operator
|
||||
flag: the role is a fact about the container's command, not a tunable,
|
||||
and a flag can be forgotten in a hand-written compose/k8s manifest —
|
||||
reintroducing the exact storm this prevents. Mirrors the argv handling
|
||||
in :func:`_is_legacy_gateway_run_request`.
|
||||
A dashboard-only container (``hermes dashboard ...``) never spawns or supervises per-profile
|
||||
gateways — that is the gateway container's job.
|
||||
"""
|
||||
args = _strip_container_argv_prefix(argv)
|
||||
return bool(args) and args[0] == "dashboard"
|
||||
@@ -386,22 +317,12 @@ def _is_dashboard_container(argv: Sequence[str]) -> bool:
|
||||
def _read_desired_state(profile_dir: Path) -> str | None:
|
||||
"""Read the persisted gateway desired state for reconciliation.
|
||||
|
||||
Newer state files carry ``desired_state``: operator intent written by
|
||||
s6 lifecycle commands. Older files only carry ``gateway_state``; keep
|
||||
that as a compatibility fallback so existing running/stopped profiles
|
||||
preserve their behavior until the next explicit start/stop.
|
||||
Newer state files carry ``desired_state``: operator intent written by s6 lifecycle commands.
|
||||
Older files only carry ``gateway_state``; keep that as a compatibility fallback so existing
|
||||
running/stopped profiles preserve their behavior until the next explicit start/stop.
|
||||
|
||||
When falling back to ``gateway_state`` (no explicit ``desired_state``),
|
||||
a transient running sub-state (``draining``) is normalised to ``running``
|
||||
— see ``_TRANSIENT_RUNNING_STATES``. A gateway hard-killed mid-drain
|
||||
leaves ``draining`` as its last persisted value; without this it would be
|
||||
treated as a non-autostart state and the gateway would stay DOWN forever.
|
||||
An explicit ``desired_state`` is always honoured verbatim (it is the
|
||||
operator's durable intent), so this normalisation only affects the
|
||||
legacy/transient fallback path.
|
||||
|
||||
Missing or unparseable files count as "no desired state" so we don't
|
||||
bork the whole reconciliation on a corrupt file.
|
||||
Missing or unparseable files count as "no desired state" so we don't bork the whole
|
||||
reconciliation on a corrupt file.
|
||||
"""
|
||||
state_file = profile_dir / "gateway_state.json"
|
||||
if not state_file.exists():
|
||||
@@ -423,9 +344,9 @@ def _read_desired_state(profile_dir: Path) -> str | None:
|
||||
|
||||
|
||||
def _cleanup_stale_runtime_files(profile_dir: Path) -> None:
|
||||
"""Remove gateway.pid and processes.json — they reference PIDs in
|
||||
the dead container's process namespace and would otherwise confuse
|
||||
the newly-started gateway's process-mismatch checks."""
|
||||
"""Remove gateway.pid and processes.json — they reference PIDs in the dead container's process
|
||||
namespace and would otherwise confuse the newly-started gateway's process-mismatch checks.
|
||||
"""
|
||||
for name in _STALE_RUNTIME_FILES:
|
||||
(profile_dir / name).unlink(missing_ok=True)
|
||||
|
||||
@@ -433,10 +354,9 @@ def _cleanup_stale_runtime_files(profile_dir: Path) -> None:
|
||||
def _read_prior_exit_label(profile_dir: Path) -> str:
|
||||
"""How the profile's previous gateway life ended (clean/unclean/unknown).
|
||||
|
||||
Thin, exception-free wrapper over
|
||||
:func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont-init runs
|
||||
in a minimal environment and forensics must never block reconciliation
|
||||
(NS-608)."""
|
||||
Thin, exception-free wrapper over :func:`gateway.lifecycle_ledger.read_prior_exit_label` — cont-
|
||||
init runs in a minimal environment and forensics must never block reconciliation (NS-608).
|
||||
"""
|
||||
try:
|
||||
from gateway.lifecycle_ledger import read_prior_exit_label
|
||||
return read_prior_exit_label(profile_dir)
|
||||
@@ -447,20 +367,10 @@ def _read_prior_exit_label(profile_dir: Path) -> str:
|
||||
def _register_service(scandir: Path, profile: str, *, start: bool) -> None:
|
||||
"""Recreate the s6 service slot for one profile.
|
||||
|
||||
Mirrors the rendering in :func:`S6ServiceManager.register_profile_gateway`,
|
||||
but here we control the start state directly via the ``down`` marker
|
||||
file (s6-svscan honors it on rescan). Cannot use the manager
|
||||
directly because the cont-init.d phase runs as root before
|
||||
s6-svscan starts scanning the dynamic scandir — the manager's
|
||||
``s6-svscanctl -a`` call would fail with no control socket.
|
||||
|
||||
Atomicity: build the new layout in a sibling temp directory and
|
||||
rename it into place via :meth:`Path.replace`. This matches
|
||||
:meth:`S6ServiceManager.register_profile_gateway` (PR #30136
|
||||
review item O4) — even though cont-init.d runs before s6-svscan
|
||||
starts scanning, an atomic publication keeps the contract uniform
|
||||
between the two registration paths and protects against a
|
||||
half-populated dir if the script is interrupted mid-write.
|
||||
Mirrors ``S6ServiceManager.register_profile_gateway`` but sets start state via the ``down``
|
||||
marker directly: cont-init.d runs as root before s6-svscan scans the dynamic scandir, so the
|
||||
manager's ``s6-svscanctl -a`` would fail with no control socket. Built in a sibling temp dir
|
||||
and ``Path.replace``d into place so an interrupted write never leaves a half-populated dir.
|
||||
"""
|
||||
import shutil
|
||||
|
||||
@@ -547,18 +457,13 @@ def _write_reconcile_log(
|
||||
) -> None:
|
||||
"""Append one line per profile to $HERMES_HOME/logs/container-boot.log.
|
||||
|
||||
Operators inspect this to debug "why didn't my profile come back
|
||||
up". Keeping a separate log file (vs. mixing into agent.log) lets
|
||||
troubleshooters grep for "profile=foo" without wading through
|
||||
unrelated activity.
|
||||
Operators inspect this to debug "why didn't my profile come back up". Keeping a separate log
|
||||
file (vs. mixing into agent.log) lets troubleshooters grep for "profile=foo" without wading
|
||||
through unrelated activity.
|
||||
|
||||
Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES``
|
||||
(defaults to 256 KiB ≈ 3000 reconcile lines), the current file
|
||||
is renamed to ``container-boot.log.1`` (replacing any previous
|
||||
rotation) before the new entries are appended. This gives long-
|
||||
lived containers a soft cap of ~512 KiB across the two files
|
||||
without pulling in logrotate or s6-log machinery just for this
|
||||
one append-only file (PR #30136 review item O3).
|
||||
Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES`` (defaults to 256 KiB ≈ 3000 reconcile
|
||||
lines), the current file is renamed to ``container-boot.log.1`` (replacing any previous
|
||||
rotation) before the new entries are appended.
|
||||
"""
|
||||
import time
|
||||
log_dir = hermes_home / "logs"
|
||||
|
||||
+270
-345
@@ -1,9 +1,4 @@
|
||||
"""
|
||||
Cron subcommand for hermes CLI.
|
||||
|
||||
Handles standalone cron management commands like list, create, edit,
|
||||
pause/resume/run/remove, status, and tick.
|
||||
"""
|
||||
"""Cron subcommand for hermes CLI."""
|
||||
|
||||
import json
|
||||
import re
|
||||
@@ -52,9 +47,9 @@ def _cron_api(**kwargs):
|
||||
def _active_cron_provider_name() -> str:
|
||||
"""Name of the resolved cron scheduler provider ('builtin', 'chronos', …).
|
||||
|
||||
Best-effort + offline (``resolve_cron_scheduler`` reads config and the
|
||||
provider's ``is_available()`` contract forbids network). Returns 'builtin'
|
||||
on any failure so callers fall back to the historical ticker-based checks.
|
||||
Best-effort + offline (``resolve_cron_scheduler`` reads config and the provider's
|
||||
``is_available()`` contract forbids network). Returns 'builtin' on any failure so callers fall
|
||||
back to the historical ticker-based checks.
|
||||
"""
|
||||
try:
|
||||
from cron.scheduler_provider import resolve_cron_scheduler
|
||||
@@ -67,13 +62,9 @@ def _active_cron_provider_name() -> str:
|
||||
def _builtin_gateway_liveness() -> Optional[bool]:
|
||||
"""Tri-state liveness of the builtin cron scheduler's trigger.
|
||||
|
||||
Single source of truth shared by the CLI (``_warn_if_gateway_not_running``)
|
||||
and the ``cronjob`` model tool (#87033): the builtin ticker only runs
|
||||
inside the gateway process, so a scheduled job with no live gateway can
|
||||
never fire. Non-builtin providers (e.g. Chronos) fire through their own
|
||||
machinery and are deliberately exempt — a missing gateway process means
|
||||
nothing for them, so they report active. ``None`` = probe failed; callers
|
||||
must not claim either way.
|
||||
Single source of truth shared by the CLI (``_warn_if_gateway_not_running``) and the ``cronjob``
|
||||
model tool (#87033): the builtin ticker only runs inside the gateway process, so a scheduled job
|
||||
with no live gateway can never fire. Non-builtin providers (e.g.
|
||||
"""
|
||||
try:
|
||||
if _active_cron_provider_name() != "builtin":
|
||||
@@ -110,17 +101,9 @@ def _builtin_gateway_liveness() -> Optional[bool]:
|
||||
def _warn_if_gateway_not_running() -> None:
|
||||
"""Warn that scheduled jobs won't fire unless the gateway is running.
|
||||
|
||||
The cron ticker only runs inside the gateway (``_start_cron_ticker`` in
|
||||
gateway/run.py); there is no standalone cron daemon. Without a running
|
||||
gateway, ``next_run_at`` passes but jobs never fire and ``last_run_at``
|
||||
stays null — the most common cron support report (#51038). Surfacing this
|
||||
at create/list time, when the user is right there, prevents it.
|
||||
|
||||
An external provider (e.g. Chronos) fires jobs via a NAS-mediated webhook,
|
||||
NOT the in-process ticker, so a momentarily-absent gateway process does not
|
||||
mean jobs won't fire — the warning would be a false alarm. Stay quiet for
|
||||
any non-builtin provider; the gateway-process heuristic only speaks to the
|
||||
built-in ticker's trigger.
|
||||
The cron ticker only runs inside the gateway (``_start_cron_ticker`` in gateway/run.py); there
|
||||
is no standalone cron daemon. Without a running gateway, ``next_run_at`` passes but jobs never
|
||||
fire and ``last_run_at`` stays null — the most common cron support report (#51038).
|
||||
"""
|
||||
# _builtin_gateway_liveness never raises (it maps probe failures to None),
|
||||
# so no guard is needed here — False is the only warn-worthy state.
|
||||
@@ -157,10 +140,8 @@ def _format_lateness(seconds: float) -> str:
|
||||
def _dispatch_display(dispatch: dict) -> Optional[str]:
|
||||
"""One-line scheduled-vs-actual dispatch summary for a job (#99879).
|
||||
|
||||
Returns None when the stamp is malformed. On-time dispatches render a
|
||||
dim confirmation; late/catch-up dispatches render loudly so a run that
|
||||
fired 30–150 min after gateway downtime no longer looks like an
|
||||
ordinary on-time success.
|
||||
None when the stamp is malformed. On-time dispatches render dim; late/catch-up dispatches render
|
||||
loudly so a run fired long after gateway downtime doesn't look like an ordinary success.
|
||||
"""
|
||||
if not isinstance(dispatch, dict):
|
||||
return None
|
||||
@@ -180,9 +161,18 @@ def _dispatch_display(dispatch: dict) -> Optional[str]:
|
||||
)
|
||||
|
||||
|
||||
def _print_banner(title: str) -> None:
|
||||
"""Boxed cyan section header shared by ``cron list`` and ``cron incidents``."""
|
||||
print()
|
||||
print(color("┌" + "─" * 73 + "┐", Colors.CYAN))
|
||||
print(color("│" + " " * 25 + title.ljust(48) + "│", Colors.CYAN))
|
||||
print(color("└" + "─" * 73 + "┘", Colors.CYAN))
|
||||
print()
|
||||
|
||||
|
||||
def cron_list(show_all: bool = False):
|
||||
"""List all scheduled jobs."""
|
||||
from cron.jobs import list_jobs
|
||||
from cron.jobs import effective_job_state, list_jobs
|
||||
|
||||
jobs = list_jobs(include_disabled=show_all)
|
||||
|
||||
@@ -191,13 +181,7 @@ def cron_list(show_all: bool = False):
|
||||
print(color("Create one with 'hermes cron create ...' or the /cron command in chat.", Colors.DIM))
|
||||
return
|
||||
|
||||
print()
|
||||
print(color("┌─────────────────────────────────────────────────────────────────────────┐", Colors.CYAN))
|
||||
print(color("│ Scheduled Jobs │", Colors.CYAN))
|
||||
print(color("└─────────────────────────────────────────────────────────────────────────┘", Colors.CYAN))
|
||||
print()
|
||||
|
||||
from cron.jobs import effective_job_state
|
||||
_print_banner("Scheduled Jobs")
|
||||
|
||||
for job in jobs:
|
||||
job_id = job.get("id", "?")
|
||||
@@ -369,10 +353,9 @@ _INCIDENT_STATE_COLORS = {
|
||||
def cron_incidents(args) -> int:
|
||||
"""List or acknowledge durable cron failure incidents.
|
||||
|
||||
``hermes cron incidents [--state <s>]`` lists incidents (the stored error
|
||||
is redacted and truncated at write time, safe for terminal display);
|
||||
``hermes cron incidents ack <id>`` closes one so its failure ping stays
|
||||
silent until the error signature changes.
|
||||
``hermes cron incidents [--state <s>]`` lists incidents (the stored error is redacted and
|
||||
truncated at write time, safe for terminal display); ``hermes cron incidents ack <id>`` closes
|
||||
one so its failure ping stays silent until the error signature changes.
|
||||
"""
|
||||
from cron.incidents import ack_incident, list_incidents
|
||||
|
||||
@@ -380,27 +363,12 @@ def cron_incidents(args) -> int:
|
||||
if action == "ack":
|
||||
incident_id = getattr(args, "incident_id", None)
|
||||
if not incident_id:
|
||||
print(
|
||||
color(
|
||||
"✗ Incident ID required: hermes cron incidents ack <incident_id>",
|
||||
Colors.RED,
|
||||
)
|
||||
)
|
||||
print(color("✗ Incident ID required: hermes cron incidents ack <incident_id>", Colors.RED))
|
||||
return 1
|
||||
if ack_incident(incident_id):
|
||||
print(
|
||||
color(
|
||||
f"✓ Incident {incident_id} acknowledged (closed).",
|
||||
Colors.GREEN,
|
||||
)
|
||||
)
|
||||
print(color(f"✓ Incident {incident_id} acknowledged (closed).", Colors.GREEN))
|
||||
else:
|
||||
print(
|
||||
color(
|
||||
f"Incident {incident_id} not found or already closed.",
|
||||
Colors.YELLOW,
|
||||
)
|
||||
)
|
||||
print(color(f"Incident {incident_id} not found or already closed.", Colors.YELLOW))
|
||||
return 0
|
||||
|
||||
state = getattr(args, "state", None)
|
||||
@@ -411,30 +379,9 @@ def cron_incidents(args) -> int:
|
||||
print(color(f" (filtered by state '{state}')", Colors.DIM))
|
||||
return 0
|
||||
|
||||
print()
|
||||
print(
|
||||
color(
|
||||
"┌─────────────────────────────────────────────────────────────────────────┐",
|
||||
Colors.CYAN,
|
||||
)
|
||||
)
|
||||
print(
|
||||
color(
|
||||
"│ Cron Failure Incidents │",
|
||||
Colors.CYAN,
|
||||
)
|
||||
)
|
||||
print(
|
||||
color(
|
||||
"└─────────────────────────────────────────────────────────────────────────┘",
|
||||
Colors.CYAN,
|
||||
)
|
||||
)
|
||||
print()
|
||||
_print_banner("Cron Failure Incidents")
|
||||
for inc in incidents:
|
||||
state_display = color(
|
||||
inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM)
|
||||
)
|
||||
state_display = color(inc["state"], _INCIDENT_STATE_COLORS.get(inc["state"], Colors.DIM))
|
||||
print(f" {color(inc['id'], Colors.YELLOW)} {state_display}")
|
||||
print(f" Job: {inc['job_id']}")
|
||||
print(f" Type: {inc.get('failure_type', 'unknown')}")
|
||||
@@ -457,6 +404,90 @@ def cron_incidents(args) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
_PERMISSION_HINT = (
|
||||
" Hint: jobs.json may be owned by another user "
|
||||
"(e.g. rewritten by a root `docker exec hermes "
|
||||
"hermes cron ...`). Fix ownership to match the "
|
||||
"gateway user, and prefer `docker exec -u <uid>:<gid>`."
|
||||
)
|
||||
_FD_EXHAUSTION_HINT = (
|
||||
" Hint: the ticker hit file-descriptor exhaustion "
|
||||
"(EMFILE). The scheduler now retries with backoff and "
|
||||
"attempts fd reclamation, but if the leak persists, "
|
||||
"restart the gateway to recover scheduling."
|
||||
)
|
||||
|
||||
|
||||
def _print_ticker_health(pids: list) -> None:
|
||||
"""Report builtin-ticker liveness for a gateway process known to be alive.
|
||||
|
||||
The gateway PROCESS is alive — but the cron ticker THREAD inside it can die silently, or stay
|
||||
alive while every tick fails. Check both the liveness heartbeat and the last-successful-tick
|
||||
marker so we don't report "will fire" when the ticker is dead or failing (#32612, #32895).
|
||||
"""
|
||||
from cron.jobs import (
|
||||
get_ticker_heartbeat_age,
|
||||
get_ticker_last_error,
|
||||
get_ticker_success_age,
|
||||
TICKER_INTERVAL_SECONDS,
|
||||
)
|
||||
from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text
|
||||
|
||||
# Allow ~3 missed ticker iterations (+ a little slack) before declaring
|
||||
# trouble. Derived from the shared interval constant so this threshold
|
||||
# tracks the ticker cadence instead of assuming a hardcoded 60s.
|
||||
STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default
|
||||
hb_age = get_ticker_heartbeat_age()
|
||||
ok_age = get_ticker_success_age()
|
||||
pid_line = f" PID: {', '.join(map(str, pids))}" if pids else None
|
||||
|
||||
def _warn(headline: str) -> None:
|
||||
print(color(headline, Colors.YELLOW))
|
||||
if pid_line:
|
||||
print(pid_line)
|
||||
|
||||
if hb_age is None:
|
||||
# No heartbeat file means the ticker thread has never started: gateway
|
||||
# not in a cron-enabled profile, started moments ago (heartbeat is
|
||||
# written after startup), or a config issue blocking the ticker.
|
||||
_warn("⚠ Gateway is running but the cron ticker has not reported a heartbeat.")
|
||||
print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.")
|
||||
print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.")
|
||||
print(" If heartbeat never appears, restart: hermes gateway restart")
|
||||
elif hb_age > STALE_AFTER:
|
||||
# No heartbeat at all → the ticker thread is gone.
|
||||
_warn(
|
||||
"⚠ Gateway is running but the cron ticker looks STALLED — "
|
||||
f"no heartbeat for {int(hb_age)}s (expected every ~60s)."
|
||||
)
|
||||
print(" Cron jobs may NOT be firing. Restart: hermes gateway restart")
|
||||
elif ok_age is not None and ok_age > STALE_AFTER:
|
||||
# Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a
|
||||
# long time → ticks are failing every iteration.
|
||||
_warn(
|
||||
"⚠ Gateway and cron ticker are running, but no tick has "
|
||||
f"succeeded in {int(ok_age)}s — ticks may be failing."
|
||||
)
|
||||
last_error = get_ticker_last_error()
|
||||
if last_error:
|
||||
# Show WHY ticks fail — e.g. a root-rewritten jobs.json
|
||||
# (PermissionError) that silently locked out the ticker's uid for
|
||||
# ~14h in the field (#68483), or fd exhaustion (EMFILE) that used
|
||||
# to stall the scheduler invisibly (#87644).
|
||||
print(color(f" Last tick error: {last_error}", Colors.RED))
|
||||
if "Permission denied" in last_error:
|
||||
print(color(_PERMISSION_HINT, Colors.YELLOW))
|
||||
elif _cron_is_fd_exhaustion_text(last_error):
|
||||
print(color(_FD_EXHAUSTION_HINT, Colors.YELLOW))
|
||||
print(" Check the gateway log for 'Cron tick error'.")
|
||||
else:
|
||||
print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN))
|
||||
if pid_line:
|
||||
print(pid_line)
|
||||
if hb_age is not None:
|
||||
print(f" Ticker heartbeat: {int(hb_age)}s ago")
|
||||
|
||||
|
||||
def cron_status():
|
||||
"""Show cron execution status."""
|
||||
from cron.jobs import list_jobs
|
||||
@@ -483,131 +514,41 @@ def cron_status():
|
||||
"due jobs are delivered by an authenticated webhook.)",
|
||||
Colors.DIM,
|
||||
))
|
||||
print()
|
||||
_print_active_jobs_summary(list_jobs(include_disabled=False))
|
||||
print()
|
||||
return
|
||||
|
||||
pids = find_gateway_pids()
|
||||
gateway_alive_via_lock = False
|
||||
if not pids:
|
||||
# Same false-alarm class the cronjob tool fixed (#95947): the pid scan
|
||||
# can transiently miss a live gateway (just after a restart) while the
|
||||
# runtime lock — held for exactly the gateway's lifetime — proves the
|
||||
# ticker's process is alive. Only declare "not running" when both the
|
||||
# scan AND the lock say so.
|
||||
try:
|
||||
from gateway.status import get_running_pid, is_gateway_runtime_lock_active
|
||||
|
||||
if is_gateway_runtime_lock_active():
|
||||
gateway_alive_via_lock = True
|
||||
lock_pid = get_running_pid()
|
||||
if lock_pid:
|
||||
pids = [lock_pid]
|
||||
except Exception:
|
||||
pass
|
||||
if pids or gateway_alive_via_lock:
|
||||
# The gateway PROCESS is alive — but the cron ticker THREAD inside it
|
||||
# can die silently, or stay alive while every tick fails. Check both
|
||||
# the liveness heartbeat and the last-successful-tick marker so we
|
||||
# don't report "will fire" when the ticker is dead or failing
|
||||
# (#32612, #32895).
|
||||
from cron.jobs import (
|
||||
get_ticker_heartbeat_age,
|
||||
get_ticker_last_error,
|
||||
get_ticker_success_age,
|
||||
TICKER_INTERVAL_SECONDS,
|
||||
)
|
||||
from cron.scheduler import _is_fd_exhaustion_text as _cron_is_fd_exhaustion_text
|
||||
|
||||
# Allow ~3 missed ticker iterations (+ a little slack) before declaring
|
||||
# trouble. Derived from the shared interval constant so this threshold
|
||||
# tracks the ticker cadence instead of assuming a hardcoded 60s.
|
||||
STALE_AFTER = TICKER_INTERVAL_SECONDS * 3 + 20 # = 200s at the 60s default
|
||||
hb_age = get_ticker_heartbeat_age()
|
||||
ok_age = get_ticker_success_age()
|
||||
|
||||
if hb_age is None:
|
||||
# No heartbeat file means the ticker thread has never started.
|
||||
# This can occur when:
|
||||
# - Gateway is running but not in a profile with cron enabled,
|
||||
# - Gateway was started moments ago (heartbeat is written after startup),
|
||||
# - Or a configuration issue is blocking the ticker from starting at all.
|
||||
print(color(
|
||||
"⚠ Gateway is running but the cron ticker has not reported a heartbeat.",
|
||||
Colors.YELLOW,
|
||||
))
|
||||
if pids:
|
||||
print(f" PID: {', '.join(map(str, pids))}")
|
||||
print(" Cron jobs will NOT fire until the ticker writes its first heartbeat.")
|
||||
print(" If the gateway just started, wait ~60s and re-run `hermes cron status`.")
|
||||
print(" If heartbeat never appears, restart: hermes gateway restart")
|
||||
elif hb_age > STALE_AFTER:
|
||||
# No heartbeat at all → the ticker thread is gone.
|
||||
print(color(
|
||||
"⚠ Gateway is running but the cron ticker looks STALLED — "
|
||||
f"no heartbeat for {int(hb_age)}s (expected every ~60s).",
|
||||
Colors.YELLOW,
|
||||
))
|
||||
if pids:
|
||||
print(f" PID: {', '.join(map(str, pids))}")
|
||||
print(" Cron jobs may NOT be firing. Restart: hermes gateway restart")
|
||||
elif ok_age is not None and ok_age > STALE_AFTER:
|
||||
# Loop is alive (fresh heartbeat) but no tick has SUCCEEDED in a
|
||||
# long time → ticks are failing every iteration.
|
||||
print(color(
|
||||
"⚠ Gateway and cron ticker are running, but no tick has "
|
||||
f"succeeded in {int(ok_age)}s — ticks may be failing.",
|
||||
Colors.YELLOW,
|
||||
))
|
||||
if pids:
|
||||
print(f" PID: {', '.join(map(str, pids))}")
|
||||
last_error = get_ticker_last_error()
|
||||
if last_error:
|
||||
# Show WHY ticks fail — e.g. a root-rewritten jobs.json
|
||||
# (PermissionError) that silently locked out the ticker's
|
||||
# uid for ~14h in the field (#68483), or fd exhaustion
|
||||
# (EMFILE) that used to stall the scheduler invisibly
|
||||
# (#87644).
|
||||
print(color(f" Last tick error: {last_error}", Colors.RED))
|
||||
if "Permission denied" in last_error:
|
||||
print(color(
|
||||
" Hint: jobs.json may be owned by another user "
|
||||
"(e.g. rewritten by a root `docker exec hermes "
|
||||
"hermes cron ...`). Fix ownership to match the "
|
||||
"gateway user, and prefer `docker exec -u <uid>:<gid>`.",
|
||||
Colors.YELLOW,
|
||||
))
|
||||
elif _cron_is_fd_exhaustion_text(last_error):
|
||||
print(color(
|
||||
" Hint: the ticker hit file-descriptor exhaustion "
|
||||
"(EMFILE). The scheduler now retries with backoff and "
|
||||
"attempts fd reclamation, but if the leak persists, "
|
||||
"restart the gateway to recover scheduling.",
|
||||
Colors.YELLOW,
|
||||
))
|
||||
print(" Check the gateway log for 'Cron tick error'.")
|
||||
else:
|
||||
print(color("✓ Gateway is running — cron jobs will fire automatically", Colors.GREEN))
|
||||
if pids:
|
||||
print(f" PID: {', '.join(map(str, pids))}")
|
||||
if hb_age is not None:
|
||||
print(f" Ticker heartbeat: {int(hb_age)}s ago")
|
||||
else:
|
||||
print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED))
|
||||
print()
|
||||
print(" To enable automatic execution:")
|
||||
print(" hermes gateway install # Install as a user service")
|
||||
print(" sudo hermes gateway install --system # Linux servers: boot-time system service")
|
||||
print(" hermes gateway # Or run in foreground")
|
||||
pids = find_gateway_pids()
|
||||
gateway_alive_via_lock = False
|
||||
if not pids:
|
||||
# Same false-alarm class the cronjob tool fixed (#95947): the pid scan
|
||||
# can transiently miss a live gateway (just after a restart) while the
|
||||
# runtime lock — held for exactly the gateway's lifetime — proves the
|
||||
# ticker's process is alive. Only declare "not running" when both the
|
||||
# scan AND the lock say so.
|
||||
try:
|
||||
from gateway.status import get_running_pid, is_gateway_runtime_lock_active
|
||||
|
||||
if is_gateway_runtime_lock_active():
|
||||
gateway_alive_via_lock = True
|
||||
lock_pid = get_running_pid()
|
||||
if lock_pid:
|
||||
pids = [lock_pid]
|
||||
except Exception:
|
||||
pass
|
||||
if pids or gateway_alive_via_lock:
|
||||
_print_ticker_health(pids)
|
||||
else:
|
||||
print(color("✗ Gateway is not running — cron jobs will NOT fire", Colors.RED))
|
||||
print()
|
||||
print(" To enable automatic execution:")
|
||||
print(" hermes gateway install # Install as a user service")
|
||||
print(" sudo hermes gateway install --system # Linux servers: boot-time system service")
|
||||
print(" hermes gateway # Or run in foreground")
|
||||
|
||||
print()
|
||||
|
||||
_print_active_jobs_summary(list_jobs(include_disabled=False))
|
||||
|
||||
print()
|
||||
|
||||
|
||||
|
||||
def _print_active_jobs_summary(jobs) -> None:
|
||||
"""Print the '<N> active job(s)' + next-run line shared by every status
|
||||
path (built-in ticker AND external provider)."""
|
||||
@@ -648,9 +589,9 @@ def _print_active_jobs_summary(jobs) -> None:
|
||||
def _scripts_dir_for_cron() -> Path:
|
||||
"""Return the scripts directory used by cron jobs.
|
||||
|
||||
Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call
|
||||
so tests and profile-aware callers that monkeypatch cron storage inspect the
|
||||
same Hermes home the jobs were loaded from.
|
||||
Prefer ``cron.jobs.CRON_DIR.parent`` over a fresh ``get_hermes_home()`` call so tests and
|
||||
profile-aware callers that monkeypatch cron storage inspect the same Hermes home the jobs were
|
||||
loaded from.
|
||||
"""
|
||||
from cron.jobs import CRON_DIR
|
||||
|
||||
@@ -778,42 +719,29 @@ def cron_doctor() -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def cron_create(args):
|
||||
# The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on
|
||||
# every job-creation path (this CLI subcommand AND the agent's `cronjob`
|
||||
# model tool, which calls create_job directly). When it blocks, create_job
|
||||
# raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and
|
||||
# returns it as result["error"], and the `if not result.get("success")`
|
||||
# branch below prints it in red and exits 1 — same UX as before.
|
||||
result = _cron_api(
|
||||
action="create",
|
||||
schedule=args.schedule,
|
||||
prompt=args.prompt,
|
||||
name=getattr(args, "name", None),
|
||||
deliver=getattr(args, "deliver", None),
|
||||
failure_deliver=getattr(args, "failure_deliver", None),
|
||||
repeat=getattr(args, "repeat", None),
|
||||
skill=getattr(args, "skill", None),
|
||||
skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)),
|
||||
script=getattr(args, "script", None),
|
||||
workdir=getattr(args, "workdir", None),
|
||||
model=getattr(args, "model", None),
|
||||
provider=getattr(args, "model_provider", None),
|
||||
no_agent=getattr(args, "no_agent", False) or None,
|
||||
monitor_script=getattr(args, "monitor_script", None),
|
||||
monitor_url=getattr(args, "monitor_url", None),
|
||||
continuity=getattr(args, "continuity", None),
|
||||
reasoning_effort=getattr(args, "reasoning_effort", None),
|
||||
)
|
||||
if not result.get("success"):
|
||||
print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED))
|
||||
return 1
|
||||
print(color(f"Created job: {result['job_id']}", Colors.GREEN))
|
||||
print(f" Name: {result['name']}")
|
||||
print(f" Schedule: {result['schedule']}")
|
||||
if result.get("skills"):
|
||||
print(f" Skills: {', '.join(result['skills'])}")
|
||||
job_data = result.get("job", {})
|
||||
_JOB_ARG_FIELDS = (
|
||||
("name", "name"),
|
||||
("deliver", "deliver"),
|
||||
("failure_deliver", "failure_deliver"),
|
||||
("repeat", "repeat"),
|
||||
("script", "script"),
|
||||
("workdir", "workdir"),
|
||||
("model", "model"),
|
||||
("provider", "model_provider"),
|
||||
("monitor_script", "monitor_script"),
|
||||
("monitor_url", "monitor_url"),
|
||||
("continuity", "continuity"),
|
||||
("reasoning_effort", "reasoning_effort"),
|
||||
)
|
||||
|
||||
|
||||
def _job_api_kwargs(args) -> Dict[str, Any]:
|
||||
"""Collect the create/update kwargs shared by ``cron create`` and ``cron edit``."""
|
||||
return {api_key: getattr(args, attr, None) for api_key, attr in _JOB_ARG_FIELDS}
|
||||
|
||||
|
||||
def _print_job_details(job_data: Dict[str, Any]) -> None:
|
||||
"""Print the optional Script/Monitor/Mode/Continuity/Workdir lines of a job record."""
|
||||
if job_data.get("script"):
|
||||
print(f" Script: {job_data['script']}")
|
||||
if job_data.get("monitor_script"):
|
||||
@@ -826,6 +754,33 @@ def cron_create(args):
|
||||
print(" Continuity: on (each run sees the previous run's output)")
|
||||
if job_data.get("workdir"):
|
||||
print(f" Workdir: {job_data['workdir']}")
|
||||
|
||||
|
||||
def cron_create(args):
|
||||
# The gateway-lifecycle guard lives in cron.jobs.create_job so it fires on
|
||||
# every job-creation path (this CLI subcommand AND the agent's `cronjob`
|
||||
# model tool, which calls create_job directly). When it blocks, create_job
|
||||
# raises GatewayLifecycleBlocked, the `cronjob` tool wrapper catches it and
|
||||
# returns it as result["error"], and the `if not result.get("success")`
|
||||
# branch below prints it in red and exits 1 — same UX as before.
|
||||
result = _cron_api(
|
||||
action="create",
|
||||
schedule=args.schedule,
|
||||
prompt=args.prompt,
|
||||
skill=getattr(args, "skill", None),
|
||||
skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)),
|
||||
no_agent=getattr(args, "no_agent", False) or None,
|
||||
**_job_api_kwargs(args),
|
||||
)
|
||||
if not result.get("success"):
|
||||
print(color(f"Failed to create job: {result.get('error', 'unknown error')}", Colors.RED))
|
||||
return 1
|
||||
print(color(f"Created job: {result['job_id']}", Colors.GREEN))
|
||||
print(f" Name: {result['name']}")
|
||||
print(f" Schedule: {result['schedule']}")
|
||||
if result.get("skills"):
|
||||
print(f" Skills: {', '.join(result['skills'])}")
|
||||
_print_job_details(result.get("job", {}))
|
||||
print(f" Next run: {result['next_run_at']}")
|
||||
_warn_if_gateway_not_running()
|
||||
return 0
|
||||
@@ -866,20 +821,9 @@ def cron_edit(args):
|
||||
job_id=args.job_id,
|
||||
schedule=getattr(args, "schedule", None),
|
||||
prompt=getattr(args, "prompt", None),
|
||||
name=getattr(args, "name", None),
|
||||
deliver=getattr(args, "deliver", None),
|
||||
failure_deliver=getattr(args, "failure_deliver", None),
|
||||
repeat=getattr(args, "repeat", None),
|
||||
skills=final_skills,
|
||||
script=getattr(args, "script", None),
|
||||
workdir=getattr(args, "workdir", None),
|
||||
model=getattr(args, "model", None),
|
||||
provider=getattr(args, "model_provider", None),
|
||||
no_agent=getattr(args, "no_agent", None),
|
||||
monitor_script=getattr(args, "monitor_script", None),
|
||||
monitor_url=getattr(args, "monitor_url", None),
|
||||
continuity=getattr(args, "continuity", None),
|
||||
reasoning_effort=getattr(args, "reasoning_effort", None),
|
||||
**_job_api_kwargs(args),
|
||||
)
|
||||
if not result.get("success"):
|
||||
print(color(f"Failed to update job: {result.get('error', 'unknown error')}", Colors.RED))
|
||||
@@ -893,23 +837,12 @@ def cron_edit(args):
|
||||
print(f" Skills: {', '.join(updated['skills'])}")
|
||||
else:
|
||||
print(" Skills: none")
|
||||
if updated.get("script"):
|
||||
print(f" Script: {updated['script']}")
|
||||
if updated.get("monitor_script"):
|
||||
print(f" Monitor: {updated['monitor_script']} (agent runs only on output change)")
|
||||
if updated.get("monitor_url"):
|
||||
print(f" Monitor: {updated['monitor_url']} (agent runs only on output change)")
|
||||
if updated.get("no_agent"):
|
||||
print(" Mode: no-agent (script stdout delivered directly)")
|
||||
if updated.get("continuity"):
|
||||
print(" Continuity: on (each run sees the previous run's output)")
|
||||
if updated.get("workdir"):
|
||||
print(f" Workdir: {updated['workdir']}")
|
||||
_print_job_details(updated)
|
||||
return 0
|
||||
|
||||
|
||||
def _job_action(action: str, job_id: str, success_verb: str) -> int:
|
||||
_stateless_reset = None
|
||||
_stateless_token = None
|
||||
if action == "run":
|
||||
# One-shot CLI: this process exits as soon as the command returns, so
|
||||
# a background-dispatched run (daemon thread of THIS process) would be
|
||||
@@ -925,16 +858,13 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int:
|
||||
from gateway.session_context import _SESSION_ASYNC_DELIVERY
|
||||
|
||||
_stateless_token = _SESSION_ASYNC_DELIVERY.set(False)
|
||||
|
||||
def _stateless_reset() -> None:
|
||||
_SESSION_ASYNC_DELIVERY.reset(_stateless_token)
|
||||
except Exception:
|
||||
_stateless_reset = None
|
||||
_stateless_token = None
|
||||
try:
|
||||
result = _cron_api(action=action, job_id=job_id)
|
||||
finally:
|
||||
if _stateless_reset is not None:
|
||||
_stateless_reset()
|
||||
if _stateless_token is not None:
|
||||
_SESSION_ASYNC_DELIVERY.reset(_stateless_token)
|
||||
if not result.get("success"):
|
||||
print(color(f"Failed to {action} job: {result.get('error', 'unknown error')}", Colors.RED))
|
||||
return 1
|
||||
@@ -970,14 +900,17 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int:
|
||||
|
||||
def cron_resume(args) -> int:
|
||||
"""Resume a paused job or explicitly re-arm a completed one-shot."""
|
||||
if bool(getattr(args, "run_at", None)) == bool(getattr(args, "run_now", False)):
|
||||
if getattr(args, "run_at", None) or getattr(args, "run_now", False):
|
||||
run_at = getattr(args, "run_at", None)
|
||||
run_now = getattr(args, "run_now", False)
|
||||
if bool(run_at) == bool(run_now):
|
||||
if run_at or run_now:
|
||||
print(color("Use exactly one of --at or --run-now.", Colors.RED))
|
||||
return 1
|
||||
return _job_action("resume", args.job_id, "Resumed")
|
||||
from cron.jobs import AmbiguousJobReference, _hermes_now, rearm_oneshot
|
||||
|
||||
run_at = _hermes_now().isoformat() if args.run_now else args.run_at
|
||||
if run_now:
|
||||
run_at = _hermes_now().isoformat()
|
||||
try:
|
||||
job = rearm_oneshot(args.job_id, run_at)
|
||||
except (AmbiguousJobReference, ValueError) as exc:
|
||||
@@ -994,10 +927,9 @@ def cron_resume(args) -> int:
|
||||
def cron_notepad(args) -> int:
|
||||
"""Handle ``hermes cron notepad <job_id> [get|set|delete|list]``.
|
||||
|
||||
The per-job durable KV scratchpad (``cron/notepad.py``). This CLI is the
|
||||
write path — a running cron agent updates its own notepad by invoking
|
||||
these commands via its terminal tool; the scheduler injects non-empty
|
||||
notepads into the job prompt on each run.
|
||||
This CLI is the write path for the per-job durable KV scratchpad (``cron/notepad.py``): a
|
||||
running cron agent updates its own notepad via its terminal tool, and the scheduler injects
|
||||
non-empty notepads into the job prompt on each run.
|
||||
"""
|
||||
from cron import notepad
|
||||
|
||||
@@ -1011,30 +943,21 @@ def cron_notepad(args) -> int:
|
||||
return 1
|
||||
|
||||
try:
|
||||
if action == "set":
|
||||
if key is None or value is None:
|
||||
print(color("Usage: hermes cron notepad <job_id> set <key> <value>", Colors.RED))
|
||||
if action in ("set", "get", "delete"):
|
||||
usage_args = "set <key> <value>" if action == "set" else f"{action} <key>"
|
||||
if key is None or (action == "set" and value is None):
|
||||
print(color(f"Usage: hermes cron notepad <job_id> {usage_args}", Colors.RED))
|
||||
return 1
|
||||
notepad.set_note(job_id, key, value)
|
||||
print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN))
|
||||
return 0
|
||||
|
||||
if action == "get":
|
||||
if key is None:
|
||||
print(color("Usage: hermes cron notepad <job_id> get <key>", Colors.RED))
|
||||
return 1
|
||||
stored = notepad.get_note(job_id, key)
|
||||
if stored is None:
|
||||
print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW))
|
||||
return 1
|
||||
print(stored)
|
||||
return 0
|
||||
|
||||
if action == "delete":
|
||||
if key is None:
|
||||
print(color("Usage: hermes cron notepad <job_id> delete <key>", Colors.RED))
|
||||
return 1
|
||||
if notepad.delete_note(job_id, key):
|
||||
if action == "set":
|
||||
notepad.set_note(job_id, key, value)
|
||||
print(color(f"Set notepad key '{key}' for job {job_id}.", Colors.GREEN))
|
||||
return 0
|
||||
if action == "get":
|
||||
stored = notepad.get_note(job_id, key)
|
||||
if stored is not None:
|
||||
print(stored)
|
||||
return 0
|
||||
elif notepad.delete_note(job_id, key):
|
||||
print(color(f"Deleted notepad key '{key}' for job {job_id}.", Colors.GREEN))
|
||||
return 0
|
||||
print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW))
|
||||
@@ -1054,52 +977,54 @@ def cron_notepad(args) -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def _cron_list_cmd(args) -> int:
|
||||
cron_list(getattr(args, "all", False))
|
||||
return 0
|
||||
|
||||
|
||||
def _cron_status_cmd(args) -> int:
|
||||
cron_status()
|
||||
return 0
|
||||
|
||||
|
||||
def _cron_runs_cmd(args) -> int:
|
||||
cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20))
|
||||
return 0
|
||||
|
||||
|
||||
def _cron_remove_cmd(args) -> int:
|
||||
return _job_action("remove", args.job_id, "Removed")
|
||||
|
||||
|
||||
# Subcommand -> handler. Late-bound lambdas so module-level monkeypatching of
|
||||
# the underlying functions keeps working.
|
||||
_CRON_SUBCOMMANDS = {
|
||||
"list": _cron_list_cmd,
|
||||
"status": _cron_status_cmd,
|
||||
"doctor": lambda args: cron_doctor(),
|
||||
"tick": lambda args: cron_tick(),
|
||||
"runs": _cron_runs_cmd,
|
||||
"history": _cron_runs_cmd,
|
||||
"incidents": lambda args: cron_incidents(args),
|
||||
"notepad": lambda args: cron_notepad(args),
|
||||
"create": lambda args: cron_create(args),
|
||||
"add": lambda args: cron_create(args),
|
||||
"edit": lambda args: cron_edit(args),
|
||||
"pause": lambda args: _job_action("pause", args.job_id, "Paused"),
|
||||
"resume": lambda args: cron_resume(args),
|
||||
"run": lambda args: _job_action("run", args.job_id, "Triggered"),
|
||||
"remove": _cron_remove_cmd,
|
||||
"rm": _cron_remove_cmd,
|
||||
"delete": _cron_remove_cmd,
|
||||
}
|
||||
|
||||
|
||||
def cron_command(args):
|
||||
"""Handle cron subcommands."""
|
||||
subcmd = getattr(args, 'cron_command', None)
|
||||
|
||||
if subcmd is None or subcmd == "list":
|
||||
show_all = getattr(args, 'all', False)
|
||||
cron_list(show_all)
|
||||
return 0
|
||||
|
||||
if subcmd == "status":
|
||||
cron_status()
|
||||
return 0
|
||||
|
||||
if subcmd == "doctor":
|
||||
return cron_doctor()
|
||||
|
||||
if subcmd == "tick":
|
||||
return cron_tick()
|
||||
|
||||
if subcmd in {"runs", "history"}:
|
||||
cron_runs(getattr(args, "job_id", None), getattr(args, "limit", 20))
|
||||
return 0
|
||||
|
||||
if subcmd == "incidents":
|
||||
return cron_incidents(args)
|
||||
|
||||
if subcmd == "notepad":
|
||||
return cron_notepad(args)
|
||||
|
||||
if subcmd in {"create", "add"}:
|
||||
return cron_create(args)
|
||||
|
||||
if subcmd == "edit":
|
||||
return cron_edit(args)
|
||||
|
||||
if subcmd == "pause":
|
||||
return _job_action("pause", args.job_id, "Paused")
|
||||
|
||||
if subcmd == "resume":
|
||||
return cron_resume(args)
|
||||
|
||||
if subcmd == "run":
|
||||
return _job_action("run", args.job_id, "Triggered")
|
||||
|
||||
if subcmd in {"remove", "rm", "delete"}:
|
||||
return _job_action("remove", args.job_id, "Removed")
|
||||
handler = _CRON_SUBCOMMANDS.get("list" if subcmd is None else subcmd)
|
||||
if handler is not None:
|
||||
return handler(args)
|
||||
|
||||
print(f"Unknown cron command: {subcmd}")
|
||||
print("Usage: hermes cron [list|create|edit|pause|resume|run|remove|status|runs|doctor|tick]")
|
||||
|
||||
+36
-73
@@ -1,17 +1,13 @@
|
||||
"""Lazy dependency bootstrapper for non-Python runtime deps.
|
||||
|
||||
Detection and prompting live here in Python — not in install.sh — because:
|
||||
1. shutil.which() works on every platform; install.sh needs bash.
|
||||
2. Detection is instant; spawning bash for a "is node installed?" check is waste.
|
||||
3. Python controls the UX (rich prompts, non-interactive fallback, TTY detection).
|
||||
Detection and prompting live here in Python — not in install.sh — because: 1. shutil.which() works
|
||||
on every platform; install.sh needs bash. 2. Detection is instant; spawning bash for a "is node
|
||||
installed?" check is waste. 3. Python controls the UX (rich prompts, non-interactive fallback, TTY
|
||||
detection).
|
||||
|
||||
install.sh is still the *installation* backend because it has 1900 lines of
|
||||
battle-tested OS detection and package-manager logic (apt/brew/pacman/dnf/
|
||||
zypper/Termux/…). Reimplementing that in Python would be huge duplication.
|
||||
|
||||
Deps that degrade gracefully (ripgrep → grep fallback, ffmpeg → skip conversion)
|
||||
don't need ensure_dependency wired in — only hard-fail sites do (TUI needs node,
|
||||
browser tool needs agent-browser).
|
||||
install.sh is still the *installation* backend because it has 1900 lines of battle-tested OS
|
||||
detection and package-manager logic (apt/brew/pacman/dnf/ zypper/Termux/…). Reimplementing that in
|
||||
Python would be huge duplication.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -50,21 +46,18 @@ _DEP_DESCRIPTIONS = {
|
||||
|
||||
|
||||
def _has_system_browser() -> bool:
|
||||
if _IS_WINDOWS:
|
||||
names = ("chrome", "msedge", "chromium")
|
||||
else:
|
||||
names = ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome")
|
||||
for name in names:
|
||||
if shutil.which(name):
|
||||
return True
|
||||
return False
|
||||
names = (
|
||||
("chrome", "msedge", "chromium") if _IS_WINDOWS
|
||||
else ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome")
|
||||
)
|
||||
return any(shutil.which(name) for name in names)
|
||||
|
||||
|
||||
def _has_npx_agent_browser() -> bool:
|
||||
"""agent-browser resolves lazily via npx on the default install (#43564),
|
||||
invisible to the PATH/managed-dir probes above. Mirror
|
||||
tools.browser_tool.check_browser_requirements's Termux carve-out so this
|
||||
check can't diverge from what browser tools actually find."""
|
||||
"""agent-browser resolves lazily via npx on the default install (#43564), invisible to the
|
||||
PATH/managed-dir probes above. Mirror tools.browser_tool.check_browser_requirements's Termux
|
||||
carve-out so this check can't diverge from what browser tools actually find.
|
||||
"""
|
||||
try:
|
||||
from tools.browser_tool import (
|
||||
_find_agent_browser,
|
||||
@@ -74,9 +67,9 @@ def _has_npx_agent_browser() -> bool:
|
||||
browser_cmd = _find_agent_browser(validate=False)
|
||||
except Exception:
|
||||
return False
|
||||
if not _is_npx_agent_browser_sentinel(browser_cmd):
|
||||
return False
|
||||
return not _requires_real_termux_browser_install(browser_cmd)
|
||||
return _is_npx_agent_browser_sentinel(browser_cmd) and not _requires_real_termux_browser_install(
|
||||
browser_cmd
|
||||
)
|
||||
|
||||
|
||||
def _has_hermes_agent_browser() -> bool:
|
||||
@@ -94,41 +87,23 @@ def _has_hermes_agent_browser() -> bool:
|
||||
|
||||
|
||||
def _find_install_script(
|
||||
package_dir: Path | None = None,
|
||||
repo_root: Path | None = None,
|
||||
package_dir: Path | None = None, repo_root: Path | None = None
|
||||
) -> tuple[Path | None, str | None]:
|
||||
"""Locate the install script — bundled in wheel or in git checkout.
|
||||
|
||||
On Windows, prefers install.ps1; on POSIX, prefers install.sh.
|
||||
Returns a (path, shell) tuple, or (None, None) if neither is found.
|
||||
"""
|
||||
if package_dir is None:
|
||||
package_dir = Path(__file__).parent
|
||||
if repo_root is None:
|
||||
repo_root = package_dir.parent
|
||||
|
||||
"""Locate the install script — bundled in wheel or in git checkout."""
|
||||
package_dir = package_dir or Path(__file__).parent
|
||||
repo_root = repo_root or package_dir.parent
|
||||
candidates = [("install.sh", "bash"), ("install.ps1", "powershell")]
|
||||
if _IS_WINDOWS:
|
||||
preferred = ("install.ps1", "powershell")
|
||||
fallback = ("install.sh", "bash")
|
||||
else:
|
||||
preferred = ("install.sh", "bash")
|
||||
fallback = ("install.ps1", "powershell")
|
||||
|
||||
for script_name, shell in (preferred, fallback):
|
||||
bundled = package_dir / "scripts" / script_name
|
||||
if bundled.is_file():
|
||||
return bundled, shell
|
||||
repo = repo_root / "scripts" / script_name
|
||||
if repo.is_file():
|
||||
return repo, shell
|
||||
|
||||
candidates.reverse()
|
||||
for script_name, shell in candidates:
|
||||
for base in (package_dir, repo_root):
|
||||
script = base / "scripts" / script_name
|
||||
if script.is_file():
|
||||
return script, shell
|
||||
return None, None
|
||||
|
||||
|
||||
def ensure_dependency(
|
||||
dep: str,
|
||||
interactive: bool = True,
|
||||
) -> bool:
|
||||
def ensure_dependency(dep: str, interactive: bool = True) -> bool:
|
||||
"""Ensure a non-Python dependency is available. Returns True if available."""
|
||||
check = _DEP_CHECKS.get(dep)
|
||||
if check is None:
|
||||
@@ -138,15 +113,14 @@ def ensure_dependency(
|
||||
return True
|
||||
|
||||
script, shell = _find_install_script()
|
||||
desc = _DEP_DESCRIPTIONS.get(dep, dep)
|
||||
if script is None:
|
||||
if interactive:
|
||||
desc = _DEP_DESCRIPTIONS.get(dep, dep)
|
||||
print(f" {desc} is not installed and no install script was found.")
|
||||
print(f" Install {dep} manually and try again.")
|
||||
return False
|
||||
|
||||
if interactive and sys.stdin.isatty():
|
||||
desc = _DEP_DESCRIPTIONS.get(dep, dep)
|
||||
try:
|
||||
reply = input(f"{desc} is not installed. Install now? [Y/n] ").strip().lower()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
@@ -162,24 +136,13 @@ def ensure_dependency(
|
||||
print(" PowerShell not found. Install PowerShell or run install.ps1 manually.")
|
||||
return False
|
||||
cmd = [
|
||||
ps_bin,
|
||||
"-ExecutionPolicy", "Bypass",
|
||||
"-File", str(script),
|
||||
"-Ensure", dep,
|
||||
"-HermesHome", str(get_hermes_home()),
|
||||
ps_bin, "-ExecutionPolicy", "Bypass", "-File", str(script),
|
||||
"-Ensure", dep, "-HermesHome", str(get_hermes_home()),
|
||||
]
|
||||
else:
|
||||
cmd = ["bash", str(script), "--ensure", dep]
|
||||
|
||||
run_env = hermes_subprocess_env(inherit_credentials=False)
|
||||
run_env["IS_INTERACTIVE"] = "false"
|
||||
result = subprocess.run(
|
||||
cmd,
|
||||
env=run_env,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return False
|
||||
|
||||
if check:
|
||||
return check()
|
||||
return True
|
||||
result = subprocess.run(cmd, env=run_env)
|
||||
return result.returncode == 0 and check()
|
||||
|
||||
+76
-122
@@ -1,28 +1,9 @@
|
||||
"""Stale git lock-file recovery for update/check paths.
|
||||
|
||||
A crashed or killed ``git fetch`` on a shallow clone can leave
|
||||
``.git/shallow.lock`` behind. Every later fetch then fails with::
|
||||
A crashed or killed ``git fetch`` on a shallow clone can leave ``.git/shallow.lock`` behind. Every
|
||||
later fetch then fails with::
|
||||
|
||||
fatal: Unable to create '/path/.git/shallow.lock': File exists.
|
||||
|
||||
This wedges ``hermes update --check`` (hard failure) and silently degrades the
|
||||
passive banner check in :mod:`hermes_cli.banner` (the fetch is swallowed, the
|
||||
stale refs are compared, and the user can be told an update is available when
|
||||
the checkout already contains the remote tip). Git does not self-heal these
|
||||
lock files — they persist until a human removes them.
|
||||
|
||||
This module provides two small, defensive helpers used by the update paths:
|
||||
|
||||
* :func:`clear_stale_git_locks` — remove abandoned ``.git`` lock files (with
|
||||
an age + git-process guard so a live fetch is never yanked).
|
||||
* :func:`clear_stale_tmp_packs` — remove aborted-fetch ``tmp_pack_*`` /
|
||||
``tmp_idx_*`` debris from ``.git/objects/pack``. On flaky lines every
|
||||
timed-out fetch leaves one behind; unchecked they accumulated to 6 GB /
|
||||
hundreds of files over 9 days and eventually corrupted the pack directory
|
||||
outright, permanently wedging the update check (#93732).
|
||||
* :func:`is_ancestor_of_head` — ask whether a remote tip is already contained
|
||||
in HEAD. Used by the shallow-clone update check to avoid reporting a false
|
||||
"update available" when local cherry-picks sit on top of the remote tip.
|
||||
fatal: Unable to create '/path/.git/shallow.lock': File exists.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -32,7 +13,7 @@ import os
|
||||
import subprocess
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
from typing import Callable, Iterable, List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -49,13 +30,23 @@ STALE_LOCK_MIN_AGE_SECONDS = 10 * 60
|
||||
# process guard in :func:`clear_stale_git_locks`.
|
||||
LOCK_NAMES = ("shallow.lock", "index.lock", "HEAD.lock", "MERGE_HEAD.lock")
|
||||
|
||||
# Aborted-fetch pack debris younger than this is presumed live (a fetch may
|
||||
# be writing it right now) and is never removed. A healthy fetch completes in
|
||||
# minutes; the same 10-minute bar the lock sweep uses is comfortably safe.
|
||||
STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS
|
||||
|
||||
# Temp-file prefixes git writes into .git/objects/pack during a transfer and
|
||||
# renames away on success. Anything left with these names after a fetch died
|
||||
# is garbage by definition — git itself never reuses or cleans them.
|
||||
_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_")
|
||||
|
||||
|
||||
def _git_proc_running() -> bool:
|
||||
"""True when a ``git`` process is currently running.
|
||||
|
||||
The conservative answer on any platform we can't probe: if we can't tell,
|
||||
treat a lock as possibly-live and don't remove it. This is the safety
|
||||
check that stops us from yanking a lock a real fetch is holding.
|
||||
The conservative answer on any platform we can't probe: if we can't tell, treat a lock as
|
||||
possibly-live and don't remove it. This is the safety check that stops us from yanking a lock a
|
||||
real fetch is holding.
|
||||
"""
|
||||
try:
|
||||
if os.name == "nt":
|
||||
@@ -73,50 +64,54 @@ def _git_proc_running() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _sweep_stale(
|
||||
directory: Path,
|
||||
candidates: Callable[[], Iterable[Path]],
|
||||
*,
|
||||
min_age_seconds: Optional[int],
|
||||
default_age: int,
|
||||
skip_msg: str,
|
||||
log_removed: Callable[[Path, int], None],
|
||||
) -> List[str]:
|
||||
"""Shared guard + age-floor sweep. Never raises; skips anything it cannot stat/unlink."""
|
||||
if not directory.is_dir():
|
||||
return []
|
||||
if _git_proc_running():
|
||||
logger.debug(skip_msg)
|
||||
return []
|
||||
cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else default_age)
|
||||
removed: List[str] = []
|
||||
for entry in candidates():
|
||||
try:
|
||||
if entry.is_file():
|
||||
st = entry.stat()
|
||||
if st.st_mtime < cutoff:
|
||||
entry.unlink()
|
||||
removed.append(str(entry))
|
||||
log_removed(entry, st.st_size)
|
||||
except OSError:
|
||||
logger.debug("Could not clear %s (skipping)", entry, exc_info=True)
|
||||
return removed
|
||||
|
||||
|
||||
def clear_stale_git_locks(repo_root: Path, *, min_age_seconds: Optional[int] = None) -> List[str]:
|
||||
"""Remove abandoned ``.git`` lock files under ``repo_root``.
|
||||
|
||||
A lock is removed only when BOTH conditions hold:
|
||||
|
||||
* it is older than :data:`STALE_LOCK_MIN_AGE_SECONDS` (default), and
|
||||
* no ``git`` process is currently running.
|
||||
|
||||
Returns the list of removed lock file paths. Never raises: a lock we
|
||||
cannot stat or unlink is skipped (a concurrently-held lock may have just
|
||||
been created between our age check and the unlink — the process guard
|
||||
makes that window vanishingly small, and skipping is always safe).
|
||||
Returns the list of removed lock file paths. Never raises: a lock we cannot stat or unlink is
|
||||
skipped (a concurrently-held lock may have just been created between our age check and the
|
||||
unlink — the process guard makes that window vanishingly small, and skipping is always safe).
|
||||
"""
|
||||
git_dir = Path(repo_root) / ".git"
|
||||
if not git_dir.is_dir():
|
||||
return []
|
||||
|
||||
if _git_proc_running():
|
||||
logger.debug("git process running; skipping stale-lock sweep")
|
||||
return []
|
||||
|
||||
cutoff = time.time() - (min_age_seconds if min_age_seconds is not None else STALE_LOCK_MIN_AGE_SECONDS)
|
||||
removed: List[str] = []
|
||||
for name in LOCK_NAMES:
|
||||
lock_path = git_dir / name
|
||||
try:
|
||||
if lock_path.is_file() and lock_path.stat().st_mtime < cutoff:
|
||||
lock_path.unlink()
|
||||
removed.append(str(lock_path))
|
||||
logger.info("Removed stale git lock %s", lock_path)
|
||||
except OSError:
|
||||
logger.debug("Could not clear %s (skipping)", lock_path, exc_info=True)
|
||||
return removed
|
||||
|
||||
|
||||
# Aborted-fetch pack debris younger than this is presumed live (a fetch may
|
||||
# be writing it right now) and is never removed. A healthy fetch completes in
|
||||
# minutes; the same 10-minute bar the lock sweep uses is comfortably safe.
|
||||
STALE_TMP_PACK_MIN_AGE_SECONDS = STALE_LOCK_MIN_AGE_SECONDS
|
||||
|
||||
# Temp-file prefixes git writes into .git/objects/pack during a transfer and
|
||||
# renames away on success. Anything left with these names after a fetch died
|
||||
# is garbage by definition — git itself never reuses or cleans them.
|
||||
_TMP_PACK_PREFIXES = ("tmp_pack_", "tmp_idx_", "tmp_rev_", "tmp_mtimes_")
|
||||
return _sweep_stale(
|
||||
git_dir,
|
||||
lambda: [git_dir / name for name in LOCK_NAMES],
|
||||
min_age_seconds=min_age_seconds,
|
||||
default_age=STALE_LOCK_MIN_AGE_SECONDS,
|
||||
skip_msg="git process running; skipping stale-lock sweep",
|
||||
log_removed=lambda p, _size: logger.info("Removed stale git lock %s", p),
|
||||
)
|
||||
|
||||
|
||||
def clear_stale_tmp_packs(
|
||||
@@ -124,69 +119,28 @@ def clear_stale_tmp_packs(
|
||||
) -> List[str]:
|
||||
"""Remove aborted-fetch temp pack files under ``.git/objects/pack``.
|
||||
|
||||
Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped
|
||||
connection) leaves a ``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file
|
||||
behind, and git never cleans them up. On a flaky line the banner's
|
||||
background update check produces several per day; observed in the wild
|
||||
at hundreds of files / 6 GB after 9 days, after which the pack directory
|
||||
corrupted outright and every fetch failed permanently (#93732).
|
||||
Every ``git fetch`` that dies mid-transfer (timeout, HTTP 429, dropped connection) leaves a
|
||||
``tmp_pack_*`` (and sometimes ``tmp_idx_*``) file behind, and git never cleans them up.
|
||||
|
||||
Same safety contract as :func:`clear_stale_git_locks`: only files older
|
||||
than the age floor, never while a git process is running, never raises.
|
||||
Returns the removed paths.
|
||||
Same safety contract as :func:`clear_stale_git_locks`: only files older than the age floor,
|
||||
never while a git process is running, never raises. Returns the removed paths.
|
||||
"""
|
||||
pack_dir = Path(repo_root) / ".git" / "objects" / "pack"
|
||||
if not pack_dir.is_dir():
|
||||
return []
|
||||
|
||||
if _git_proc_running():
|
||||
logger.debug("git process running; skipping tmp-pack sweep")
|
||||
return []
|
||||
|
||||
cutoff = time.time() - (
|
||||
min_age_seconds if min_age_seconds is not None else STALE_TMP_PACK_MIN_AGE_SECONDS
|
||||
)
|
||||
removed: List[str] = []
|
||||
try:
|
||||
entries = list(pack_dir.iterdir())
|
||||
except OSError:
|
||||
return []
|
||||
for entry in entries:
|
||||
name = entry.name
|
||||
if not name.startswith(_TMP_PACK_PREFIXES):
|
||||
continue
|
||||
def _candidates():
|
||||
try:
|
||||
if entry.is_file() and entry.stat().st_mtime < cutoff:
|
||||
size = entry.stat().st_size
|
||||
entry.unlink()
|
||||
removed.append(str(entry))
|
||||
logger.info(
|
||||
"Removed aborted-fetch pack debris %s (%d bytes)", entry, size
|
||||
)
|
||||
entries = list(pack_dir.iterdir())
|
||||
except OSError:
|
||||
logger.debug("Could not clear %s (skipping)", entry, exc_info=True)
|
||||
return removed
|
||||
return []
|
||||
return [e for e in entries if e.name.startswith(_TMP_PACK_PREFIXES)]
|
||||
|
||||
|
||||
def is_ancestor_of_head(repo_root: Path, rev: str) -> bool:
|
||||
"""True when ``rev`` is an ancestor of (or equal to) HEAD.
|
||||
|
||||
Wraps ``git merge-base --is-ancestor <rev> HEAD``. This is the correct
|
||||
question for update checks: a local cherry-pick on top of the remote tip
|
||||
makes HEAD *different* from ``origin/main`` but still *contains* it, so
|
||||
the answer to "is there an update?" is no.
|
||||
|
||||
Returns False on any probe failure (missing rev, shallow boundary, git
|
||||
error) — callers treat that as "can't prove contained", which is the
|
||||
conservative direction for an update check.
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["git", "merge-base", "--is-ancestor", rev, "HEAD"],
|
||||
cwd=str(repo_root),
|
||||
capture_output=True, text=True, timeout=10,
|
||||
)
|
||||
return result.returncode == 0
|
||||
except Exception:
|
||||
logger.debug("merge-base --is-ancestor probe failed for %s", rev, exc_info=True)
|
||||
return False
|
||||
return _sweep_stale(
|
||||
pack_dir,
|
||||
_candidates,
|
||||
min_age_seconds=min_age_seconds,
|
||||
default_age=STALE_TMP_PACK_MIN_AGE_SECONDS,
|
||||
skip_msg="git process running; skipping tmp-pack sweep",
|
||||
log_removed=lambda p, size: logger.info(
|
||||
"Removed aborted-fetch pack debris %s (%d bytes)", p, size
|
||||
),
|
||||
)
|
||||
|
||||
+48
-103
@@ -1,38 +1,7 @@
|
||||
"""
|
||||
Hermes Desktop (Chat GUI) uninstaller.
|
||||
"""Hermes Desktop (Chat GUI) uninstaller.
|
||||
|
||||
The desktop GUI ships in two shapes and this module knows how to find and
|
||||
remove the artifacts of both, on Linux, macOS, and Windows, WITHOUT touching
|
||||
the Python agent or the user's config/data:
|
||||
|
||||
1. Source-built GUI (``hermes desktop`` / ``hermes gui``)
|
||||
Built inside the agent checkout under ``$HERMES_HOME/hermes-agent/``:
|
||||
- ``apps/desktop/dist`` (compiled renderer)
|
||||
- ``apps/desktop/release`` (electron-builder unpacked app + installers)
|
||||
- ``apps/desktop/node_modules`` and the workspace-root ``node_modules``
|
||||
(Electron itself, ~200MB) — only removed on a GUI uninstall because
|
||||
the agent does not need them.
|
||||
- ``$HERMES_HOME/desktop-build-stamp.json`` (the build freshness stamp)
|
||||
|
||||
2. Packaged distributable (DMG / NSIS / AppImage / deb / rpm)
|
||||
Installed by the OS to a standard application location and carrying its
|
||||
own bundled Electron + a per-user Electron ``userData`` directory:
|
||||
- macOS: ``/Applications/Hermes.app`` or ``~/Applications/Hermes.app``
|
||||
- Windows: ``%LOCALAPPDATA%\\Programs\\Hermes`` (NSIS per-user)
|
||||
- Linux: ``~/.local/share/applications`` .desktop entry + AppImage
|
||||
|
||||
In both shapes the Electron runtime keeps a ``userData`` directory keyed on
|
||||
the app name ("Hermes"), separate from ``$HERMES_HOME``:
|
||||
- macOS: ``~/Library/Application Support/Hermes``
|
||||
- Windows: ``%APPDATA%\\Hermes``
|
||||
- Linux: ``$XDG_CONFIG_HOME/Hermes`` (default ``~/.config/Hermes``)
|
||||
|
||||
This holds the desktop's own ``connection.json`` / ``updates.json`` and
|
||||
Chromium cache — pure GUI state, safe to remove on a GUI uninstall.
|
||||
|
||||
The functions here are deliberately import-light and side-effect-free at
|
||||
import time so the Electron main process can shell out to
|
||||
``hermes uninstall --gui`` (and friends) without paying for the full CLI.
|
||||
This holds the desktop's own ``connection.json`` / ``updates.json`` and Chromium cache — pure GUI
|
||||
state, safe to remove on a GUI uninstall.
|
||||
"""
|
||||
|
||||
import os
|
||||
@@ -57,6 +26,12 @@ def log_warn(msg: str):
|
||||
print(f"{color('⚠', Colors.YELLOW)} {msg}")
|
||||
|
||||
|
||||
def _env_dir(var: str, fallback: Path) -> Path:
|
||||
"""``Path($var)`` when the env var is set, else *fallback*."""
|
||||
value = os.environ.get(var)
|
||||
return Path(value) if value else fallback
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Discovery
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -70,29 +45,24 @@ def _agent_root(hermes_home: Path) -> Path:
|
||||
def desktop_userdata_dir() -> Path:
|
||||
"""Return the Electron ``userData`` directory for the desktop app.
|
||||
|
||||
Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes"
|
||||
on each platform. This is GUI-only state (connection.json, updates.json,
|
||||
Chromium cache) and never holds agent config or sessions.
|
||||
Mirrors Electron's ``app.getPath('userData')`` for an app named "Hermes" on each platform. This
|
||||
is GUI-only state (connection.json, updates.json, Chromium cache) and never holds agent config
|
||||
or sessions.
|
||||
"""
|
||||
home = Path.home()
|
||||
if sys.platform == "darwin":
|
||||
return home / "Library" / "Application Support" / "Hermes"
|
||||
if sys.platform == "win32":
|
||||
appdata = os.environ.get("APPDATA")
|
||||
base = Path(appdata) if appdata else (home / "AppData" / "Roaming")
|
||||
return base / "Hermes"
|
||||
return _env_dir("APPDATA", home / "AppData" / "Roaming") / "Hermes"
|
||||
# Linux / other POSIX — XDG config home.
|
||||
xdg = os.environ.get("XDG_CONFIG_HOME")
|
||||
base = Path(xdg) if xdg else (home / ".config")
|
||||
return base / "Hermes"
|
||||
return _env_dir("XDG_CONFIG_HOME", home / ".config") / "Hermes"
|
||||
|
||||
|
||||
def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]":
|
||||
"""GUI build artifacts produced by ``hermes desktop`` inside the checkout.
|
||||
|
||||
These are removable on a GUI uninstall without harming the agent: the
|
||||
Python agent runs from ``hermes-agent/`` source + ``venv/`` and never
|
||||
needs the Electron build output or node_modules.
|
||||
These are removable on a GUI uninstall without harming the agent: the Python agent runs from
|
||||
``hermes-agent/`` source + ``venv/`` and never needs the Electron build output or node_modules.
|
||||
"""
|
||||
agent_root = _agent_root(hermes_home)
|
||||
desktop_dir = agent_root / "apps" / "desktop"
|
||||
@@ -111,9 +81,9 @@ def source_built_gui_artifacts(hermes_home: Path) -> "list[Path]":
|
||||
def packaged_gui_app_paths() -> "list[Path]":
|
||||
"""Standard install locations of the packaged desktop distributable.
|
||||
|
||||
Returns every candidate for the current OS; the caller filters to those
|
||||
that actually exist. We never glob system-wide — only the well-known
|
||||
electron-builder output locations for the "Hermes" product.
|
||||
Returns every candidate for the current OS; the caller filters to those that actually exist. We
|
||||
never glob system-wide — only the well-known electron-builder output locations for the "Hermes"
|
||||
product.
|
||||
"""
|
||||
home = Path.home()
|
||||
paths: list[Path] = []
|
||||
@@ -123,8 +93,7 @@ def packaged_gui_app_paths() -> "list[Path]":
|
||||
home / "Applications" / "Hermes.app",
|
||||
]
|
||||
elif sys.platform == "win32":
|
||||
local = os.environ.get("LOCALAPPDATA")
|
||||
local_base = Path(local) if local else (home / "AppData" / "Local")
|
||||
local_base = _env_dir("LOCALAPPDATA", home / "AppData" / "Local")
|
||||
paths += [
|
||||
# NSIS per-user install (perMachine=false → Programs\Hermes).
|
||||
local_base / "Programs" / "Hermes",
|
||||
@@ -144,8 +113,7 @@ def packaged_gui_app_paths() -> "list[Path]":
|
||||
# ``uninstall_gui``.
|
||||
from hermes_cli.linux_desktop_entry import desktop_entry_path
|
||||
|
||||
data = os.environ.get("XDG_DATA_HOME")
|
||||
data_base = Path(data) if data else (home / ".local" / "share")
|
||||
data_base = _env_dir("XDG_DATA_HOME", home / ".local" / "share")
|
||||
paths += [
|
||||
# The launcher entry `hermes desktop` installs. Its icon is
|
||||
# also copied into the hicolor tree (see
|
||||
@@ -166,52 +134,38 @@ def packaged_gui_app_paths() -> "list[Path]":
|
||||
def agent_is_installed(hermes_home: Path) -> bool:
|
||||
"""Return True when a usable Python agent install exists under HERMES_HOME.
|
||||
|
||||
Used by the desktop UI to decide which uninstall options to offer: if the
|
||||
agent isn't present (a future "lite" GUI-only client), the "remove agent"
|
||||
options are hidden.
|
||||
Used by the desktop UI to decide which uninstall options to offer: if the agent isn't present (a
|
||||
future "lite" GUI-only client), the "remove agent" options are hidden.
|
||||
"""
|
||||
agent_root = _agent_root(hermes_home)
|
||||
# A real install has the package source + a venv. Either signal alone is
|
||||
# enough — a source checkout without a venv is still "the agent is here".
|
||||
if (agent_root / "hermes_cli").is_dir():
|
||||
return True
|
||||
if (agent_root / "venv").is_dir() or (agent_root / ".venv").is_dir():
|
||||
return True
|
||||
return False
|
||||
return any((agent_root / sub).is_dir() for sub in ("hermes_cli", "venv", ".venv"))
|
||||
|
||||
|
||||
def gui_is_installed(hermes_home: Path) -> bool:
|
||||
"""Return True when any desktop GUI artifact exists (built or packaged)."""
|
||||
for p in source_built_gui_artifacts(hermes_home):
|
||||
if p.exists():
|
||||
return True
|
||||
for p in packaged_gui_app_paths():
|
||||
if p.exists():
|
||||
return True
|
||||
if desktop_userdata_dir().exists():
|
||||
return True
|
||||
return False
|
||||
return any(
|
||||
p.exists()
|
||||
for p in (*source_built_gui_artifacts(hermes_home), *packaged_gui_app_paths(), desktop_userdata_dir())
|
||||
)
|
||||
|
||||
|
||||
def gui_install_summary(hermes_home: "Path | None" = None) -> dict:
|
||||
"""Structured snapshot of what's installed, for the desktop UI to render.
|
||||
|
||||
Returns JSON-serializable primitives so the Electron main process can
|
||||
forward it to the renderer via IPC (paths as strings, booleans for the
|
||||
high-level questions the UI gates options on).
|
||||
Returns JSON-serializable primitives (paths as strings, booleans for the questions the UI gates
|
||||
on) so the Electron main process can forward it to the renderer via IPC.
|
||||
"""
|
||||
home: Path = hermes_home if hermes_home is not None else get_hermes_home()
|
||||
|
||||
source_artifacts = [p for p in source_built_gui_artifacts(home) if p.exists()]
|
||||
packaged = [p for p in packaged_gui_app_paths() if p.exists()]
|
||||
userdata = desktop_userdata_dir()
|
||||
|
||||
return {
|
||||
"hermes_home": str(home),
|
||||
"agent_installed": agent_is_installed(home),
|
||||
"gui_installed": gui_is_installed(home),
|
||||
"source_built_artifacts": [str(p) for p in source_artifacts],
|
||||
"packaged_app_paths": [str(p) for p in packaged],
|
||||
"source_built_artifacts": [str(p) for p in source_built_gui_artifacts(home) if p.exists()],
|
||||
"packaged_app_paths": [str(p) for p in packaged_gui_app_paths() if p.exists()],
|
||||
"userdata_dir": str(userdata),
|
||||
"userdata_exists": userdata.exists(),
|
||||
"platform": sys.platform,
|
||||
@@ -242,45 +196,36 @@ def uninstall_gui(
|
||||
) -> "list[Path]":
|
||||
"""Remove the desktop GUI's artifacts, leaving the agent + user data intact.
|
||||
|
||||
Removes:
|
||||
- source-built GUI artifacts (dist/release/node_modules/build-stamp)
|
||||
- the packaged app bundle / install dir (best-effort; deb/rpm need the
|
||||
system package manager and are reported, not force-removed)
|
||||
- the Electron ``userData`` directory (unless ``remove_userdata=False``)
|
||||
|
||||
Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any
|
||||
config / sessions / .env under ``$HERMES_HOME``.
|
||||
|
||||
Returns the list of paths actually removed.
|
||||
Never touches ``hermes-agent/hermes_cli`` (agent source), ``venv/``, or any config / sessions /
|
||||
.env under ``$HERMES_HOME``.
|
||||
"""
|
||||
home: Path = hermes_home if hermes_home is not None else get_hermes_home()
|
||||
|
||||
removed: list[Path] = []
|
||||
|
||||
def _remove_existing(paths) -> bool:
|
||||
"""Remove every existing path; True when at least one existed."""
|
||||
found = False
|
||||
for path in paths:
|
||||
if path.exists():
|
||||
found = True
|
||||
if _remove_path(path):
|
||||
log_success(f"Removed {path}")
|
||||
removed.append(path)
|
||||
return found
|
||||
|
||||
log_info("Removing built GUI artifacts (renderer, release, node_modules)...")
|
||||
for path in source_built_gui_artifacts(home):
|
||||
if path.exists() and _remove_path(path):
|
||||
log_success(f"Removed {path}")
|
||||
removed.append(path)
|
||||
_remove_existing(source_built_gui_artifacts(home))
|
||||
|
||||
log_info("Removing installed desktop app...")
|
||||
found_packaged = False
|
||||
for path in packaged_gui_app_paths():
|
||||
if path.exists():
|
||||
found_packaged = True
|
||||
if _remove_path(path):
|
||||
log_success(f"Removed {path}")
|
||||
removed.append(path)
|
||||
if not found_packaged:
|
||||
if not _remove_existing(packaged_gui_app_paths()):
|
||||
log_info("No packaged desktop app found in standard locations")
|
||||
|
||||
if remove_userdata:
|
||||
userdata = desktop_userdata_dir()
|
||||
if userdata.exists():
|
||||
log_info("Removing desktop app data (Electron userData)...")
|
||||
if _remove_path(userdata):
|
||||
log_success(f"Removed {userdata}")
|
||||
removed.append(userdata)
|
||||
_remove_existing([userdata])
|
||||
|
||||
if not removed:
|
||||
log_info("No desktop GUI artifacts found to remove")
|
||||
|
||||
+57
-99
@@ -1,29 +1,12 @@
|
||||
"""Session heartbeats — recurring re-entry prompts for the current session.
|
||||
|
||||
A heartbeat is one user-owned recurring instruction bound to a session
|
||||
(`/heartbeat every 10m Check the deployment and report meaningful changes`).
|
||||
When due AND the session is idle, the prompt is injected as a normal user
|
||||
turn — same mechanism as a /goal continuation, so message-role alternation
|
||||
and prompt caching are untouched. If the agent is busy at the due moment,
|
||||
the tick coalesces: it fires once when the session next goes idle, never
|
||||
stacking a backlog.
|
||||
This is deliberately session-scoped and in-process (CLI process or gateway process must be running)
|
||||
— the durable cross-process scheduling surface remains ``hermes cron`` / the ``cronjob`` tool, which
|
||||
runs in isolated sessions.
|
||||
|
||||
This is deliberately session-scoped and in-process (CLI process or gateway
|
||||
process must be running) — the durable cross-process scheduling surface
|
||||
remains ``hermes cron`` / the ``cronjob`` tool, which runs in isolated
|
||||
sessions. A heartbeat is for "keep re-entering THIS conversation", the
|
||||
cron system is for "run this job on a schedule". Distinct by design.
|
||||
|
||||
State is persisted in SessionDB ``state_meta`` keyed by
|
||||
``heartbeat:<session_id>`` so ``/resume`` picks it up.
|
||||
|
||||
Invariants (mirrors goals.py):
|
||||
- Injection is a plain user message. No system-prompt mutation, no toolset
|
||||
swap — prompt caching stays intact.
|
||||
- A real user message always wins: heartbeats only fire into an idle
|
||||
session with an empty input queue.
|
||||
- Failures are contained: any DB/import error degrades to "no heartbeat",
|
||||
never to a crashed input loop.
|
||||
Invariants (mirrors goals.py): - Injection is a plain user message. No system-prompt mutation, no
|
||||
toolset swap — prompt caching stays intact. - A real user message always wins: heartbeats only fire
|
||||
into an idle session with an empty input queue.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -32,8 +15,8 @@ import json
|
||||
import logging
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass, asdict
|
||||
from typing import Any, Dict, Optional
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -68,32 +51,22 @@ _UNIT_SECONDS = {
|
||||
def parse_interval(text: str) -> Optional[int]:
|
||||
"""Parse ``10m`` / ``every 2h`` / ``every 90 minutes`` into seconds.
|
||||
|
||||
Returns None when the text is not an interval. Values below
|
||||
``MIN_INTERVAL_SECONDS`` are rejected (returns -1 so callers can
|
||||
distinguish "not an interval" from "too small").
|
||||
Returns None when the text is not an interval; values below ``MIN_INTERVAL_SECONDS`` return
|
||||
-1 so callers can distinguish "not an interval" from "too small".
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
m = _INTERVAL_RE.match(text)
|
||||
m = _INTERVAL_RE.match(text) if text else None
|
||||
if not m:
|
||||
return None
|
||||
value = float(m.group(1))
|
||||
unit = m.group(2).lower()
|
||||
seconds = int(value * _UNIT_SECONDS[unit])
|
||||
if seconds < MIN_INTERVAL_SECONDS:
|
||||
return -1
|
||||
return seconds
|
||||
seconds = int(float(m.group(1)) * _UNIT_SECONDS[m.group(2).lower()])
|
||||
return -1 if seconds < MIN_INTERVAL_SECONDS else seconds
|
||||
|
||||
|
||||
def format_interval(seconds: int) -> str:
|
||||
"""Human-readable interval (``600`` → ``10m``)."""
|
||||
seconds = int(seconds)
|
||||
if seconds % 86400 == 0:
|
||||
return f"{seconds // 86400}d"
|
||||
if seconds % 3600 == 0:
|
||||
return f"{seconds // 3600}h"
|
||||
if seconds % 60 == 0:
|
||||
return f"{seconds // 60}m"
|
||||
for unit, suffix in ((86400, "d"), (3600, "h"), (60, "m")):
|
||||
if seconds % unit == 0:
|
||||
return f"{seconds // unit}{suffix}"
|
||||
return f"{seconds}s"
|
||||
|
||||
|
||||
@@ -114,14 +87,11 @@ class HeartbeatState:
|
||||
@classmethod
|
||||
def from_json(cls, raw: str) -> "HeartbeatState":
|
||||
data = json.loads(raw)
|
||||
return cls(
|
||||
prompt=str(data.get("prompt") or ""),
|
||||
interval_seconds=int(data.get("interval_seconds", 0) or 0),
|
||||
status=str(data.get("status") or "active"),
|
||||
created_at=float(data.get("created_at", 0.0) or 0.0),
|
||||
last_fired_at=float(data.get("last_fired_at", 0.0) or 0.0),
|
||||
fire_count=int(data.get("fire_count", 0) or 0),
|
||||
)
|
||||
# Falsy/missing values fall back to the default before type coercion.
|
||||
return cls(**{
|
||||
name: coerce(data.get(name) or default)
|
||||
for name, (coerce, default) in _STATE_FIELDS.items()
|
||||
})
|
||||
|
||||
def is_due(self, now: Optional[float] = None) -> bool:
|
||||
if self.status != "active" or not self.prompt or self.interval_seconds <= 0:
|
||||
@@ -137,9 +107,18 @@ class HeartbeatState:
|
||||
)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# field -> (coercer, default used when the stored value is missing/falsy)
|
||||
_STATE_FIELDS = {
|
||||
"prompt": (str, ""),
|
||||
"interval_seconds": (int, 0),
|
||||
"status": (str, "active"),
|
||||
"created_at": (float, 0.0),
|
||||
"last_fired_at": (float, 0.0),
|
||||
"fire_count": (int, 0),
|
||||
}
|
||||
|
||||
|
||||
# Persistence (SessionDB state_meta) — same pattern as goals.py
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _meta_key(session_id: str) -> str:
|
||||
@@ -159,9 +138,7 @@ def _get_session_db() -> Optional[Any]:
|
||||
|
||||
|
||||
def load_heartbeat(session_id: str) -> Optional[HeartbeatState]:
|
||||
if not session_id:
|
||||
return None
|
||||
db = _get_session_db()
|
||||
db = _get_session_db() if session_id else None
|
||||
if db is None:
|
||||
return None
|
||||
try:
|
||||
@@ -194,18 +171,15 @@ def save_heartbeat(session_id: str, state: HeartbeatState) -> None:
|
||||
logger.debug("HeartbeatManager: set_meta failed: %s", exc)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# Manager — the surface CLI + gateway talk to
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class HeartbeatManager:
|
||||
"""Per-session heartbeat state + due-tick decisions.
|
||||
|
||||
Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll
|
||||
cadence while the session is idle; a non-None return is the user-role
|
||||
message to inject. Firing is recorded immediately so a slow turn can't
|
||||
double-fire.
|
||||
Drivers (CLI thread / gateway task) call :meth:`due_prompt` on a poll cadence while the session
|
||||
is idle; a non-None return is the user-role message to inject. Firing is recorded immediately so
|
||||
a slow turn can't double-fire.
|
||||
"""
|
||||
|
||||
def __init__(self, session_id: str):
|
||||
@@ -232,9 +206,8 @@ class HeartbeatManager:
|
||||
anchor = s.last_fired_at or s.created_at
|
||||
next_in = max(0, int(anchor + s.interval_seconds - time.time()))
|
||||
return f"♥ Heartbeat (every {every}, next in ~{next_in}s{fired}): {s.prompt}"
|
||||
if s.status == "paused":
|
||||
return f"⏸ Heartbeat (paused, every {every}{fired}): {s.prompt}"
|
||||
return f"Heartbeat ({s.status}, every {every}{fired}): {s.prompt}"
|
||||
icon = "⏸ " if s.status == "paused" else ""
|
||||
return f"{icon}Heartbeat ({s.status}, every {every}{fired}): {s.prompt}"
|
||||
|
||||
# --- mutation -----------------------------------------------------
|
||||
|
||||
@@ -246,36 +219,31 @@ class HeartbeatManager:
|
||||
if interval_seconds < MIN_INTERVAL_SECONDS:
|
||||
raise ValueError(f"interval must be at least {MIN_INTERVAL_SECONDS}s")
|
||||
state = HeartbeatState(
|
||||
prompt=prompt,
|
||||
interval_seconds=interval_seconds,
|
||||
status="active",
|
||||
created_at=time.time(),
|
||||
prompt=prompt, interval_seconds=interval_seconds, status="active", created_at=time.time()
|
||||
)
|
||||
self._state = state
|
||||
save_heartbeat(self.session_id, state)
|
||||
return state
|
||||
|
||||
def pause(self) -> Optional[HeartbeatState]:
|
||||
def _set_status(self, status: str, *, reanchor: bool = False) -> Optional[HeartbeatState]:
|
||||
if not self._state:
|
||||
return None
|
||||
self._state.status = "paused"
|
||||
self._state.status = status
|
||||
if reanchor:
|
||||
self._state.last_fired_at = time.time()
|
||||
save_heartbeat(self.session_id, self._state)
|
||||
return self._state
|
||||
|
||||
def pause(self) -> Optional[HeartbeatState]:
|
||||
return self._set_status("paused")
|
||||
|
||||
def resume(self) -> Optional[HeartbeatState]:
|
||||
if not self._state:
|
||||
return None
|
||||
self._state.status = "active"
|
||||
# Re-anchor so resuming doesn't instantly fire a stale tick.
|
||||
self._state.last_fired_at = time.time()
|
||||
save_heartbeat(self.session_id, self._state)
|
||||
return self._state
|
||||
return self._set_status("active", reanchor=True)
|
||||
|
||||
def clear(self) -> bool:
|
||||
if self._state is None:
|
||||
if self._set_status("cleared") is None:
|
||||
return False
|
||||
self._state.status = "cleared"
|
||||
save_heartbeat(self.session_id, self._state)
|
||||
self._state = None
|
||||
return True
|
||||
|
||||
@@ -284,10 +252,9 @@ class HeartbeatManager:
|
||||
def due_prompt(self, now: Optional[float] = None) -> Optional[str]:
|
||||
"""Return the injection prompt if the heartbeat is due, else None.
|
||||
|
||||
Records the fire immediately (before the turn runs) so overlapping
|
||||
polls or a long turn can never double-fire the same tick. Missed
|
||||
ticks coalesce into one — the anchor resets to NOW, not to the
|
||||
theoretical schedule.
|
||||
Records the fire immediately (before the turn runs) so overlapping polls or a long turn can
|
||||
never double-fire the same tick. Missed ticks coalesce into one — the anchor resets to NOW,
|
||||
not to the theoretical schedule.
|
||||
"""
|
||||
s = self._state
|
||||
if s is None or not s.is_due(now):
|
||||
@@ -301,16 +268,14 @@ class HeartbeatManager:
|
||||
def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bool:
|
||||
"""Carry a heartbeat across a compression session rotation.
|
||||
|
||||
Same shape as ``goals.migrate_goal_to_session`` — copy to the child,
|
||||
archive the parent row, never raise.
|
||||
Same shape as ``goals.migrate_goal_to_session`` — copy to the child, archive the parent row,
|
||||
never raise.
|
||||
"""
|
||||
if not old_session_id or not new_session_id or old_session_id == new_session_id:
|
||||
return False
|
||||
try:
|
||||
state = load_heartbeat(old_session_id)
|
||||
if state is None:
|
||||
return False
|
||||
if load_heartbeat(new_session_id) is not None:
|
||||
if state is None or load_heartbeat(new_session_id) is not None:
|
||||
return False
|
||||
save_heartbeat(new_session_id, state)
|
||||
state.status = "cleared"
|
||||
@@ -322,14 +287,7 @@ def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bo
|
||||
|
||||
|
||||
__all__ = [
|
||||
"HeartbeatState",
|
||||
"HeartbeatManager",
|
||||
"parse_interval",
|
||||
"format_interval",
|
||||
"load_heartbeat",
|
||||
"save_heartbeat",
|
||||
"migrate_heartbeat_to_session",
|
||||
"HEARTBEAT_PROMPT_TEMPLATE",
|
||||
"MIN_INTERVAL_SECONDS",
|
||||
"POLL_SECONDS",
|
||||
"HeartbeatState", "HeartbeatManager", "parse_interval", "format_interval",
|
||||
"load_heartbeat", "save_heartbeat", "migrate_heartbeat_to_session",
|
||||
"HEARTBEAT_PROMPT_TEMPLATE", "MIN_INTERVAL_SECONDS", "POLL_SECONDS",
|
||||
]
|
||||
|
||||
@@ -1,14 +1,12 @@
|
||||
"""Image-authored deployment provenance for immutable Hermes runtimes.
|
||||
|
||||
The published image bakes ``/etc/hermes/image-provenance.json`` outside both
|
||||
``$HERMES_HOME`` and the mutable checkout. A bind-mounted checkout (including
|
||||
``.git``) therefore cannot hide the build fact, and environment or config
|
||||
values cannot forge it.
|
||||
The published image bakes ``/etc/hermes/image-provenance.json`` outside both ``$HERMES_HOME`` and
|
||||
the mutable checkout. A bind-mounted checkout (including ``.git``) therefore cannot hide the build
|
||||
fact, and environment or config values cannot forge it.
|
||||
|
||||
Absence preserves every pre-existing source/package install path. Presence
|
||||
fails closed: an unreadable, non-regular, or malformed marker still means the
|
||||
runtime is image-managed; it is an integrity defect, never permission to
|
||||
mutate the image in place.
|
||||
Absence preserves every pre-existing source/package install path. Presence fails closed: an
|
||||
unreadable, non-regular, or malformed marker still means the runtime is image-managed; it is an
|
||||
integrity defect, never permission to mutate the image in place.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -55,20 +53,22 @@ def _invalid(path: Path, reason: str) -> ImageProvenance:
|
||||
)
|
||||
|
||||
|
||||
def _optional_string(payload: dict, name: str) -> Optional[str]:
|
||||
value = payload.get(name)
|
||||
if value is None:
|
||||
return None
|
||||
if not isinstance(value, str):
|
||||
raise TypeError(name)
|
||||
return value.strip() or None
|
||||
|
||||
|
||||
def read_image_provenance(
|
||||
marker_path: Optional[Path] = None,
|
||||
) -> Optional[ImageProvenance]:
|
||||
"""Read the baked marker without consulting environment or config.
|
||||
|
||||
``None`` has one precise meaning: ``lstat`` proved that no marker exists.
|
||||
Every other filesystem or validation failure returns an invalid
|
||||
:class:`ImageProvenance`, so callers refuse image mutation closed. In
|
||||
particular, ``lstat`` makes a dangling symlink visibly *present* and the
|
||||
regular-file check rejects symlinks, directories, and device nodes.
|
||||
|
||||
``marker_path`` is a dependency-injection seam for tests and alternate
|
||||
image builders. Normal callers always use the image-owned absolute path.
|
||||
This function never raises.
|
||||
``marker_path`` is a dependency-injection seam for tests and alternate image builders. Normal
|
||||
callers always use the image-owned absolute path. This function never raises.
|
||||
"""
|
||||
|
||||
path = IMAGE_PROVENANCE_PATH
|
||||
@@ -110,19 +110,8 @@ def read_image_provenance(
|
||||
if not isinstance(manager, str) or not manager.strip():
|
||||
return _invalid(path, "missing_manager")
|
||||
|
||||
def _optional_string(name: str) -> Optional[str]:
|
||||
value = payload.get(name)
|
||||
if value is None:
|
||||
return None
|
||||
if not isinstance(value, str):
|
||||
raise TypeError(name)
|
||||
value = value.strip()
|
||||
return value or None
|
||||
|
||||
try:
|
||||
image = _optional_string("image")
|
||||
version = _optional_string("version")
|
||||
revision = _optional_string("revision")
|
||||
optional = {name: _optional_string(payload, name) for name in ("image", "version", "revision")}
|
||||
except TypeError as exc:
|
||||
return _invalid(path, f"invalid_{exc.args[0]}")
|
||||
|
||||
@@ -130,8 +119,6 @@ def read_image_provenance(
|
||||
schema=IMAGE_PROVENANCE_SCHEMA,
|
||||
deployment_kind="image",
|
||||
manager=manager.strip(),
|
||||
image=image,
|
||||
version=version,
|
||||
revision=revision,
|
||||
marker_path=str(path),
|
||||
**optional,
|
||||
)
|
||||
|
||||
@@ -74,8 +74,8 @@ def _fsync_directory(path: Path) -> None:
|
||||
def read_or_create_install_id(root: Path | None = None) -> Optional[str]:
|
||||
"""Read or atomically mint the opaque id for the physical install.
|
||||
|
||||
``None`` means the id could neither be read nor persisted. Returning an
|
||||
ephemeral id would violate the authority and connection-registry contract.
|
||||
``None`` means the id could neither be read nor persisted. Returning an ephemeral id would
|
||||
violate the authority and connection-registry contract.
|
||||
"""
|
||||
root = get_default_hermes_root() if root is None else root
|
||||
path = root / _INSTALL_ID_FILENAME
|
||||
|
||||
@@ -8,8 +8,7 @@ from typing import Any, List
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
|
||||
"""Notify first-party observers, then invoke compatibility plugin hooks."""
|
||||
def _observe(hook_name: str, **kwargs: Any) -> None:
|
||||
try:
|
||||
from hermes_cli.observability import observe_lifecycle
|
||||
|
||||
@@ -17,6 +16,11 @@ def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
|
||||
except Exception:
|
||||
logger.warning("Built-in observability hook failed", exc_info=True)
|
||||
|
||||
|
||||
def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]:
|
||||
"""Notify first-party observers, then invoke compatibility plugin hooks."""
|
||||
_observe(hook_name, **kwargs)
|
||||
|
||||
from hermes_cli import plugins
|
||||
|
||||
return plugins.invoke_hook(hook_name, **kwargs)
|
||||
@@ -39,12 +43,7 @@ def has_hook(hook_name: str) -> bool:
|
||||
|
||||
def finalize_session(**kwargs: Any) -> List[Any]:
|
||||
"""Notify observers and hard-close one core-owned Relay conversation."""
|
||||
try:
|
||||
from hermes_cli.observability import observe_lifecycle
|
||||
|
||||
observe_lifecycle("on_session_finalize", **kwargs)
|
||||
except Exception:
|
||||
logger.warning("Built-in observability hook failed", exc_info=True)
|
||||
_observe("on_session_finalize", **kwargs)
|
||||
|
||||
session_id = str(kwargs.get("session_id") or "")
|
||||
if session_id:
|
||||
|
||||
@@ -1,26 +1,10 @@
|
||||
"""Install and remove the Linux desktop entry (``hermes.desktop``).
|
||||
|
||||
``hermes desktop`` builds and launches the Electron app. On Linux, a
|
||||
freshly-built app has no launcher presence: no menu item, no icon. This
|
||||
module writes the XDG desktop entry that gives it one.
|
||||
``hermes uninstall --gui`` removes the entry again.
|
||||
|
||||
Two values must be absolute for the entry to work:
|
||||
|
||||
- ``Exec`` — the launcher runs without shell ``PATH`` customizations, so
|
||||
a bare ``hermes desktop`` fails when hermes lives in ``~/.local/bin``
|
||||
or a venv. Resolve the real binary and write its full path.
|
||||
- ``Icon`` — an unqualified icon name needs an indexed icon theme. The
|
||||
spec allows an absolute path instead, so point at the app icon in the
|
||||
checkout. Do not copy the icon: ``Exec`` already depends on that tree.
|
||||
|
||||
Cache refresh is best-effort and tool-gated: ``update-desktop-database``
|
||||
for the freedesktop menu cache, ``gtk-update-icon-cache`` for the user
|
||||
hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5`` for Plasma. Run
|
||||
each tool only when it exists. A missing tool is not an error.
|
||||
|
||||
Import-light and side-effect-free at import time: the uninstaller uses
|
||||
this without loading the full CLI.
|
||||
Cache refresh is best-effort and tool-gated: ``update-desktop-database`` for the freedesktop menu
|
||||
cache, ``gtk-update-icon-cache`` for the user hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5``
|
||||
for Plasma. Run each tool only when it exists. A missing tool is not an error.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -60,23 +44,12 @@ def icon_path(project_root: Path) -> Path:
|
||||
|
||||
|
||||
def _running_interpreter() -> str:
|
||||
"""The venv-semantic interpreter path for the persisted ``Exec=`` line.
|
||||
``sys.executable`` inside a venv is commonly a SYMLINK into a shared
|
||||
base-interpreter tree (uv, pyenv, conda). ``Path.resolve()`` follows it
|
||||
out of the venv, and CPython discovers ``pyvenv.cfg`` from the
|
||||
*lexical* argv[0] — so a dereferenced path boots without the venv's
|
||||
site-packages and dies on the first third-party import (#90292, one
|
||||
level up; identified in #80547's review and confirmed on real Zorin/uv
|
||||
hardware in this PR's review).
|
||||
|
||||
Keep the lexical path only when it actually is venv-semantic (a
|
||||
``pyvenv.cfg`` sits at or above it in the tree); otherwise the
|
||||
dereferenced absolute path is the more durable form (survives the
|
||||
symlink being re-pointed or its parent moving).
|
||||
|
||||
Idea credit: the lexical-preservation rule was independently proposed
|
||||
in #92516/#94115/#94544 and by nosliwhtes' review of this PR; the
|
||||
pyvenv.cfg-detection refinement here keeps both properties.
|
||||
"""The venv-semantic interpreter path for the persisted ``Exec=`` line. ``sys.executable`` inside a
|
||||
venv is commonly a SYMLINK into a shared base-interpreter tree (uv, pyenv, conda).
|
||||
``Path.resolve()`` follows it out of the venv, and CPython discovers ``pyvenv.cfg`` from the
|
||||
*lexical* argv[0] — so a dereferenced path boots without the venv's site-packages and dies on
|
||||
the first third-party import (#90292, one level up; identified in #80547's review and confirmed
|
||||
on real Zorin/uv hardware in this PR's review).
|
||||
"""
|
||||
lexical = os.path.abspath(sys.executable)
|
||||
path = Path(lexical)
|
||||
@@ -92,18 +65,10 @@ _probe_cache: "dict[str, bool]" = {}
|
||||
def _can_import_hermes_cli(interpreter: Path) -> bool:
|
||||
"""Whether *interpreter* can import ``hermes_cli.main`` unaided.
|
||||
|
||||
Runs the import in a subprocess under ``-I`` (isolated mode: no
|
||||
user site, no PYTHONPATH inheritance, no cwd on ``sys.path``) from
|
||||
a neutral cwd, so the answer matches what a cold desktop
|
||||
environment would get — a checkout cwd or an inherited
|
||||
``PYTHONPATH`` cannot produce a false positive. Bounded by a
|
||||
timeout so a hung interpreter cannot stall entry generation.
|
||||
|
||||
Result is cached per interpreter path for the process lifetime, so
|
||||
a desktop launch pays the subprocess cost at most once.
|
||||
|
||||
Probe design per @nosliwhtes' isolated-mode capability check
|
||||
(#92122 lineage, commit 4150501f641).
|
||||
Runs the import in a subprocess under ``-I`` (isolated mode: no user site, no PYTHONPATH
|
||||
inheritance, no cwd on ``sys.path``) from a neutral cwd, so the answer matches what a cold
|
||||
desktop environment would get — a checkout cwd or an inherited ``PYTHONPATH`` cannot produce a
|
||||
false positive.
|
||||
"""
|
||||
key = str(interpreter)
|
||||
cached = _probe_cache.get(key)
|
||||
@@ -134,9 +99,9 @@ def _can_import_hermes_cli(interpreter: Path) -> bool:
|
||||
def _running_interpreter_fallback() -> str:
|
||||
"""The interpreter to persist when the candidate fails the import probe.
|
||||
|
||||
The RUNNING interpreter by definition has ``hermes_cli`` importable
|
||||
(this module is executing), so the module-form entry under it is the
|
||||
safe landing when every candidate path failed the capability check.
|
||||
The RUNNING interpreter by definition has ``hermes_cli`` importable (this module is executing),
|
||||
so the module-form entry under it is the safe landing when every candidate path failed the
|
||||
capability check.
|
||||
"""
|
||||
return os.path.abspath(sys.executable)
|
||||
|
||||
@@ -144,22 +109,11 @@ def _running_interpreter_fallback() -> str:
|
||||
def resolve_exec_command(project_root: Optional[Path] = None) -> str:
|
||||
"""Build the absolute ``Exec=`` command line for ``hermes desktop``.
|
||||
|
||||
Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes
|
||||
runs as a module with no launcher installed, use the current
|
||||
interpreter, also absolute.
|
||||
Prefer the real ``hermes`` executable (argv[0] or PATH). When Hermes runs as a module with no
|
||||
launcher installed, use the current interpreter, also absolute.
|
||||
|
||||
The persisted entry must be launch-context independent: whatever
|
||||
process writes it, the next launch must read and rewrite the same
|
||||
bytes. ``resolve_hermes_bin()`` prefers ``sys.argv[0]``, which differs
|
||||
per launch path (wrapper, repo script, ``python -m``), so for this
|
||||
one caller an argv[0] that points inside the checkout is not a
|
||||
durable installed launcher — skip it and resolve from PATH instead.
|
||||
Otherwise a broken entry keeps regenerating itself (the repo-script
|
||||
form pins a mutable uv interpreter path; the ``python -m`` form
|
||||
persists a bare ``<python> desktop`` that no DE can run).
|
||||
|
||||
``project_root`` pins which checkout counts as "internal"; defaults to
|
||||
the running checkout.
|
||||
The persisted entry must be launch-context independent: whatever process writes it, the next
|
||||
launch must read and rewrite the same bytes.
|
||||
"""
|
||||
from hermes_cli.relaunch import resolve_hermes_bin
|
||||
|
||||
@@ -209,17 +163,11 @@ def _resolve_hermes_bin_for_desktop_entry(
|
||||
) -> Optional[str]:
|
||||
"""Resolve the launcher binary for the persisted ``.desktop`` entry.
|
||||
|
||||
Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one
|
||||
desktop-entry-specific rule: an ``argv[0]`` that points inside this
|
||||
checkout is a launch-context artifact (the repo ``hermes`` script the
|
||||
wrapper execs with, or an interpreter binary surfaced by programmatic
|
||||
relaunch paths), not a durable installed launcher. Persisting it makes
|
||||
the entry a function of however the previous launch happened — the
|
||||
bootstrap loop behind #90492's incomplete fix. Skip argv[0]/relative
|
||||
candidates in that case and fall through to PATH, where the shell
|
||||
installer's wrapper lives.
|
||||
|
||||
``resolve_fn`` is injectable for tests.
|
||||
Wraps :func:`hermes_cli.relaunch.resolve_hermes_bin` with one rule: an ``argv[0]`` inside
|
||||
this checkout is a launch-context artifact, not a durable installed launcher — persisting it
|
||||
makes the entry depend on how the previous launch happened (a bootstrap loop). Skip
|
||||
argv[0]/relative candidates then and fall through to PATH, where the shell installer's
|
||||
wrapper lives. ``resolve_fn`` is injectable for tests.
|
||||
"""
|
||||
if resolve_fn is None:
|
||||
from hermes_cli.relaunch import resolve_hermes_bin as resolve_fn
|
||||
@@ -271,12 +219,9 @@ def _resolve_hermes_bin_for_desktop_entry(
|
||||
def _is_interpreter(candidate: Path) -> bool:
|
||||
"""A python interpreter binary (``bin/python*``), not a launcher.
|
||||
|
||||
Strict basename match — accepts ``python``, ``python3``,
|
||||
``python3.11``, ``python2.7``; rejects lookalikes such as
|
||||
``python3-config``, ``pythonw``, and anything else merely
|
||||
*containing* "python". Regex approach proposed independently in
|
||||
#94051; kept here with the parent-dir guard so a script named
|
||||
``python`` outside a bin/Scripts tree is not misclassified.
|
||||
Strict basename match — accepts ``python``, ``python3``, ``python3.11``; rejects lookalikes
|
||||
such as ``python3-config`` and ``pythonw``. The parent-dir guard keeps a script named
|
||||
``python`` outside a bin/Scripts tree from being misclassified.
|
||||
"""
|
||||
import re
|
||||
|
||||
@@ -353,13 +298,10 @@ def _resolve_hermes_bin_for_desktop_entry(
|
||||
def _wrapper_shebang_safe(wrapper: Path) -> bool:
|
||||
"""Whether an executable wrapper can actually run in the DE context.
|
||||
|
||||
A wrapper whose own shebang escapes the venv (``#!/usr/bin/env
|
||||
python3`` or a bare interpreter name) would die exactly like the
|
||||
broken entry this module exists to fix — the checkout reference in
|
||||
its body does not save it. Native binaries and shell launchers are
|
||||
safe by construction (they exec the right interpreter themselves).
|
||||
A python-shebang wrapper is safe only when its interpreter resolves
|
||||
to the RUNNING venv's interpreter directory.
|
||||
A wrapper whose shebang escapes the venv (``#!/usr/bin/env python3`` or a bare interpreter
|
||||
name) would die exactly like the broken entry this module fixes. Native binaries and shell
|
||||
launchers are safe by construction; a python-shebang wrapper is safe only when its
|
||||
interpreter resolves to the RUNNING venv's interpreter directory.
|
||||
"""
|
||||
try:
|
||||
with open(wrapper, "rb") as fh:
|
||||
@@ -406,20 +348,10 @@ def _wrapper_shebang_safe(wrapper: Path) -> bool:
|
||||
def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool:
|
||||
"""Whether a candidate launcher script actually launches THIS checkout.
|
||||
|
||||
Expects the LEXICAL checkout root (the caller keeps it un-resolved):
|
||||
the installer writes ``$INSTALL_DIR`` lexically into the shim, so on a
|
||||
symlinked home the shim text and the resolved root would never match.
|
||||
Both lexical and resolved forms of the root are tried regardless, to
|
||||
Expects the LEXICAL checkout root (the caller keeps it un-resolved): the installer writes
|
||||
``$INSTALL_DIR`` lexically into the shim, so on a symlinked home the shim text and the resolved
|
||||
root would never match. Both lexical and resolved forms of the root are tried regardless, to
|
||||
tolerate either caller convention.
|
||||
|
||||
The installer's shim is a small bash script that execs
|
||||
``<checkout>/venv/bin/python <checkout>/hermes``; a venv console
|
||||
script carries the venv interpreter in its shebang. Either way, a
|
||||
text launcher belonging to this installation references the
|
||||
checkout path (or its venv) somewhere in its first few KB. A
|
||||
binary launcher (PyInstaller & friends) cannot be inspected that
|
||||
way — accept it, since binary installs are self-contained and the
|
||||
external-primary-first rule has already had its say.
|
||||
"""
|
||||
try:
|
||||
head = wrapper.read_bytes()[:4096]
|
||||
@@ -467,9 +399,8 @@ def _wrapper_targets_checkout(wrapper: Path, checkout_root: Path) -> bool:
|
||||
def _known_wrapper_candidates():
|
||||
"""Durable installed-launcher locations, most likely first.
|
||||
|
||||
Mirrors the installer's ``get_command_link_dir()`` layouts: user
|
||||
(``~/.local/bin``), root FHS (``/usr/local/bin``), and Termux
|
||||
(``$PREFIX/bin``). The wrapper is always named ``hermes``.
|
||||
Mirrors the installer's ``get_command_link_dir()`` layouts: user (``~/.local/bin``), root FHS
|
||||
(``/usr/local/bin``), and Termux (``$PREFIX/bin``). The wrapper is always named ``hermes``.
|
||||
"""
|
||||
candidates = []
|
||||
home = Path.home()
|
||||
@@ -485,9 +416,9 @@ def _known_wrapper_candidates():
|
||||
def _project_root() -> Path:
|
||||
"""This file lives at ``<checkout>/hermes_cli/linux_desktop_entry.py``.
|
||||
|
||||
Lexical (no .resolve()): callers feed this into shim-text matching
|
||||
where the installer's lexically-written $INSTALL_DIR must be able to
|
||||
match; symlinked homes would break a resolved comparison.
|
||||
Lexical (no .resolve()): callers feed this into shim-text matching where the installer's
|
||||
lexically-written $INSTALL_DIR must be able to match; symlinked homes would break a resolved
|
||||
comparison.
|
||||
"""
|
||||
return Path(os.path.abspath(__file__)).parent.parent
|
||||
|
||||
@@ -515,30 +446,15 @@ def _needs_interpreter(bin_path: Path) -> bool:
|
||||
def _shebang_escapes_running_env(shebang: str) -> bool:
|
||||
"""Whether a python shebang resolves OUTSIDE the running interpreter's env.
|
||||
|
||||
Tokenizes the shebang (interpreter path plus any flags) and compares
|
||||
PATH COMPONENTS, never substrings: ``<venv>/bin-extra/python`` is not
|
||||
inside ``<venv>/bin`` even though it starts with it (sibling-directory
|
||||
confusion; independently surfaced in nosliwhtes' #92122 hardening
|
||||
Tokenizes the shebang (interpreter path plus any flags) and compares PATH COMPONENTS, never
|
||||
substrings: ``<venv>/bin-extra/python`` is not inside ``<venv>/bin`` even though it starts with
|
||||
it (sibling-directory confusion; independently surfaced in nosliwhtes' #92122 hardening
|
||||
``b96427d0`` — reimplemented here with two extensions).
|
||||
|
||||
Extensions over the parent-equality form:
|
||||
|
||||
* ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env``
|
||||
resolves through PATH, which in the DE's cold environment is not the
|
||||
interactive PATH that installed the venv — the parent-equality form
|
||||
could be fooled when the resolved ``env`` binary happens to sit in
|
||||
the same directory tree.
|
||||
* Flags after the interpreter (``-S``, ``-E``...) are stripped before
|
||||
comparing, so a legitimate ``#!<venv>/bin/python -S`` is not
|
||||
misclassified by comparing against the flag token.
|
||||
|
||||
The comparison uses the LEXICAL interpreter directory (abspath, not
|
||||
resolve()): on uv venvs the resolved parent is the base interpreter's
|
||||
dir, which makes a valid ``.venv/bin/python`` shebang look foreign
|
||||
(#94443 review case 1). Both sides use the SAME case operation
|
||||
(``.lower()``): interpreter paths legitimately carry uppercase (conda
|
||||
env names, usernames, uv's ephemeral build dirs) and an asymmetric
|
||||
compare would flag the venv's own console script as foreign.
|
||||
* ``env`` shebangs (``#!/usr/bin/env python3``) ALWAYS escape: ``env`` resolves through PATH,
|
||||
which in the DE's cold environment is not the interactive PATH that installed the venv — the
|
||||
parent-equality form could be fooled when the resolved ``env`` binary happens to sit in the same
|
||||
directory tree.
|
||||
"""
|
||||
tokens = shebang[2:].strip().split()
|
||||
if not tokens:
|
||||
@@ -560,11 +476,7 @@ def _shebang_escapes_running_env(shebang: str) -> bool:
|
||||
|
||||
|
||||
def _quote_exec_arg(arg: str) -> str:
|
||||
"""Quote one ``Exec`` argument per the desktop entry spec.
|
||||
|
||||
Reserved characters require double quotes. Inside the quotes, escape
|
||||
a backslash and a double quote with a backslash.
|
||||
"""
|
||||
"""Quote one ``Exec`` argument per the desktop entry spec."""
|
||||
if not any(c in arg for c in " \t\n\"'\\><~|&;$*?#()`"):
|
||||
return arg
|
||||
escaped = arg.replace("\\", "\\\\").replace('"', '\\"')
|
||||
@@ -588,16 +500,12 @@ def render_desktop_entry(exec_command: str, icon: str) -> str:
|
||||
|
||||
|
||||
def refresh_desktop_databases(applications_dir: Path) -> "list[str]":
|
||||
"""Reindex the menu caches. Run each tool only when it exists.
|
||||
|
||||
Return the names of the tools that ran (for logging and tests).
|
||||
"""
|
||||
"""Reindex the menu caches. Run each tool only when it exists."""
|
||||
ran: list[str] = []
|
||||
|
||||
update_db = shutil.which("update-desktop-database")
|
||||
if update_db:
|
||||
if _run_quiet([update_db, str(applications_dir)]):
|
||||
ran.append("update-desktop-database")
|
||||
if update_db and _run_quiet([update_db, str(applications_dir)]):
|
||||
ran.append("update-desktop-database")
|
||||
|
||||
# Plasma 6 first, then Plasma 5. Only one of them is ever installed.
|
||||
for tool in ("kbuildsycoca6", "kbuildsycoca5"):
|
||||
@@ -664,8 +572,8 @@ def _hicolor_icon_dest(subdir: str) -> Path:
|
||||
def _remove_stale_scalable_icon() -> bool:
|
||||
"""Drop a leftover PNG from ``scalable/`` (the pre-fix install path).
|
||||
|
||||
Return True when a file was removed so the caller can refresh the
|
||||
icon cache. A missing file is not an error.
|
||||
Return True when a file was removed so the caller can refresh the icon cache. A missing file is
|
||||
not an error.
|
||||
"""
|
||||
stale = _hicolor_icon_dest("scalable")
|
||||
try:
|
||||
@@ -690,9 +598,9 @@ def _refresh_hicolor_cache() -> None:
|
||||
def _resized_hicolor_pngs(raw: bytes) -> Optional[dict[str, bytes]]:
|
||||
"""Lanczos-resize *raw* to each panel size. ``None`` when it will not decode.
|
||||
|
||||
Pillow is a core dep but this module stays import-light: the import is
|
||||
local so the uninstaller does not pay it. A truncated/fake PNG (tests,
|
||||
interrupted copy) returns None and the caller falls back to a copy.
|
||||
Pillow is a core dep but this module stays import-light: the import is local so the uninstaller
|
||||
does not pay it. A truncated/fake PNG (tests, interrupted copy) returns None and the caller
|
||||
falls back to a copy.
|
||||
"""
|
||||
try:
|
||||
from PIL import Image
|
||||
@@ -728,15 +636,9 @@ def _write_hicolor_pngs(files: dict[str, bytes]) -> bool:
|
||||
def _install_icon_to_hicolor(icon: Path) -> bool:
|
||||
"""Install the app icon into the user's hicolor icon theme tree.
|
||||
|
||||
The freedesktop icon lookup finds an installed ``apps/hermes.png``
|
||||
by the unqualified name ``hermes``, so the entry can reference the
|
||||
icon without an absolute checkout path. Raster PNGs go to indexed
|
||||
fixed-size dirs, never ``scalable`` (SVG-only). When the source
|
||||
decodes, it is Lanczos-resized to 24/32/48/256 so Cinnamon's panel
|
||||
does not nearest-neighbor a 1024px PNG. Undecodable bytes fall back
|
||||
to a copy into one indexed dir. Idempotent via content-compare;
|
||||
OSError caught internally (False) — the caller then falls back to
|
||||
the absolute path.
|
||||
The freedesktop icon lookup finds an installed ``apps/hermes.png`` by the unqualified name
|
||||
``hermes``, so the entry can reference the icon without an absolute checkout path. Raster PNGs
|
||||
go to indexed fixed-size dirs, never ``scalable`` (SVG-only).
|
||||
"""
|
||||
try:
|
||||
raw = icon.read_bytes()
|
||||
@@ -762,8 +664,8 @@ def _install_icon_to_hicolor(icon: Path) -> bool:
|
||||
def install_desktop_entry(project_root: Path) -> Optional[Path]:
|
||||
"""Write (or refresh) the Hermes desktop entry. Return its path.
|
||||
|
||||
Return ``None`` on non-Linux platforms or when the write fails. This
|
||||
is a convenience, never a reason to fail a launch.
|
||||
Return ``None`` on non-Linux platforms or when the write fails. This is a convenience, never a
|
||||
reason to fail a launch.
|
||||
"""
|
||||
if not is_supported():
|
||||
return None
|
||||
|
||||
@@ -1,40 +1,11 @@
|
||||
"""Stable macOS TCC anchor for the uv-managed Python interpreter (#95596).
|
||||
|
||||
Re-land of the interpreter anchor reverted in #95563. macOS keys TCC grants
|
||||
to the resolved absolute path of the client binary. Hermes' interpreter is
|
||||
managed by uv and lives at a versioned store path; every patch bump orphans
|
||||
every prior grant (#85345).
|
||||
1. Aliases are materialized as real-file copies of the anchor, never symlinks. 2. If the store ships
|
||||
``libpython*``, it is hardlinked into ``venv/lib/`` (copy if the store is on another device).
|
||||
Existing ``LC_RPATH`` already points at ``@executable_path/../lib`` — no rewrite. 3.
|
||||
|
||||
The first landing copied the interpreter into ``venv/bin/python`` but left
|
||||
two holes that bricked real Macs:
|
||||
|
||||
* Dynamically-linked builds look up ``libpython`` via
|
||||
``@executable_path/../lib``. That resolved into ``venv/lib/``, which had
|
||||
no dylib — every hermes command, including update/doctor, died in dyld
|
||||
(#95425).
|
||||
* Alias names (``python3``, ``python3.N``) were re-pointed at the copy as
|
||||
*symlinks*. Invoking the copied interpreter through a symlink makes
|
||||
CPython getpath lose the venv prefix on affected python-build-standalone
|
||||
builds — startup dies with ``ModuleNotFoundError: encodings`` and the
|
||||
stdlib resolves to the build-time ``/install`` prefix (#95541). Console
|
||||
scripts exec ``python3``, so the entire CLI surface died.
|
||||
|
||||
This re-land keeps the copy + identifier-pinned signature (TCC attribution
|
||||
stays on the stable venv path) and closes both holes:
|
||||
|
||||
1. Aliases are materialized as real-file copies of the anchor, never
|
||||
symlinks.
|
||||
2. If the store ships ``libpython*``, it is hardlinked into ``venv/lib/``
|
||||
(copy if the store is on another device). Existing ``LC_RPATH`` already
|
||||
points at ``@executable_path/../lib`` — no rewrite.
|
||||
3. A pre-install boot gate actually launches the staged copy and demands
|
||||
``import encodings`` plus ``sys.prefix == <venv>``. Failure rolls the
|
||||
staging file back and leaves the live interpreter untouched (a surplus
|
||||
provisioned dylib in ``venv/lib/`` may remain — harmless), so a bad
|
||||
anchor can never brick update/doctor again.
|
||||
|
||||
All functions are no-ops on non-macOS and for interpreters that are not
|
||||
uv-managed. Best-effort: never raises to callers.
|
||||
All functions are no-ops on non-macOS and for interpreters that are not uv-managed. Best-effort:
|
||||
never raises to callers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -68,9 +39,9 @@ class _BootGateFailed(Exception):
|
||||
|
||||
|
||||
def _marker_value(source_file: Path) -> str:
|
||||
"""Canonical marker value: fully resolved so symlinked spellings of the
|
||||
same store binary (``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``)
|
||||
compare equal."""
|
||||
"""Canonical marker value: fully resolved so symlinked spellings of the same store binary
|
||||
(``cpython-3.11-macos-*`` → ``cpython-3.11.15-macos-*``) compare equal.
|
||||
"""
|
||||
return os.path.realpath(str(source_file))
|
||||
|
||||
|
||||
@@ -167,9 +138,9 @@ def _anchor_marker(venv_bin: Path) -> Path:
|
||||
def _write_marker(venv_bin: Path, source_file: Path) -> None:
|
||||
"""Write the anchor marker atomically via the shared helper.
|
||||
|
||||
A concurrent ensure (update + doctor --fix) must never observe a
|
||||
partially-written marker: a torn read would compare unequal and trigger
|
||||
a spurious reinstall, and ``write_text`` alone is not atomic.
|
||||
A concurrent ensure (update + doctor --fix) must never observe a partially-written marker: a
|
||||
torn read would compare unequal and trigger a spurious reinstall, and ``write_text`` alone is
|
||||
not atomic.
|
||||
"""
|
||||
atomic_write_text(
|
||||
_anchor_marker(venv_bin),
|
||||
@@ -188,8 +159,8 @@ def _provision_libpython(
|
||||
) -> None:
|
||||
"""Hardlink (else copy) store ``libpython*`` into ``venv/lib/``.
|
||||
|
||||
Provision-if-present: a surplus hardlink on a statically-linked build is
|
||||
free; a missed detection is the only way #95425 returns.
|
||||
Provision-if-present: a surplus hardlink on a statically-linked build is free; a missed
|
||||
detection is the only way #95425 returns.
|
||||
"""
|
||||
src_lib = _store_root(source_file) / "lib"
|
||||
if not src_lib.is_dir():
|
||||
@@ -222,10 +193,9 @@ def _provision_libpython(
|
||||
def _copy_alias(venv_bin: Path, name: str, anchor: Path) -> bool:
|
||||
"""Materialize *name* as a real-file copy of *anchor* (atomic rename).
|
||||
|
||||
Returns False (and warns) on failure: a leftover alias *symlink* to the
|
||||
anchor is the exact #95541 crash shape, so callers must know when the
|
||||
alias set is incomplete. The staging name is unique (mkstemp) so a
|
||||
concurrent ensure (update + doctor --fix) cannot promote a truncated
|
||||
Returns False (and warns) on failure: a leftover alias *symlink* to the anchor is the exact
|
||||
#95541 crash shape, so callers must know when the alias set is incomplete. The staging name is
|
||||
unique (mkstemp) so a concurrent ensure (update + doctor --fix) cannot promote a truncated
|
||||
interim copy.
|
||||
"""
|
||||
tmp_path: Path | None = None
|
||||
@@ -278,13 +248,10 @@ def _materialize_aliases(
|
||||
def _passes_boot_gate(staged: Path, venv_dir: Path) -> bool:
|
||||
"""Launch *staged* and demand encodings + the venv prefix.
|
||||
|
||||
The probe runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited
|
||||
``PYTHONHOME`` papers over exactly the prefix-resolution failure the gate
|
||||
exists to catch. ``OSError`` is split by errno — ``ENOENT``/``ENOEXEC``
|
||||
(fixture binaries, foreign-arch images) means the binary cannot run here
|
||||
at all, so the symlinked venv was equally dead and installing cannot make
|
||||
things worse: skip. Anything else (notably ``EACCES`` after our own
|
||||
chmod) is a broken install about to go live: refuse.
|
||||
Runs with ``PYTHONHOME``/``PYTHONPATH`` scrubbed: an inherited PYTHONHOME papers over the very
|
||||
prefix failure this gate exists to catch. ``ENOENT``/``ENOEXEC`` (fixture binaries, foreign
|
||||
arch) mean the binary can't run here at all, so the symlinked venv was equally dead: skip.
|
||||
Anything else (notably ``EACCES`` after our own chmod) is a broken install going live: refuse.
|
||||
"""
|
||||
env = {
|
||||
k: v
|
||||
@@ -368,10 +335,9 @@ def _install_anchor(venv_dir: Path, source_file: Path) -> None:
|
||||
def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None:
|
||||
"""Pin a dylib-complete interpreter anchor for macOS TCC (#95596).
|
||||
|
||||
No-op (returns None) on non-macOS, when no venv interpreter exists, or
|
||||
when the interpreter is not uv-managed. Idempotent. Best-effort —
|
||||
returns None (and logs) if the copy or boot-gate fails; callers must
|
||||
never depend on success.
|
||||
No-op (returns None) on non-macOS, when no venv interpreter exists, or when the interpreter is
|
||||
not uv-managed. Idempotent. Best-effort — returns None (and logs) if the copy or boot-gate
|
||||
fails; callers must never depend on success.
|
||||
"""
|
||||
if not is_macos():
|
||||
return None
|
||||
@@ -411,14 +377,11 @@ def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None:
|
||||
|
||||
|
||||
def tcc_anchor_state(project_root: Path | None = None) -> tuple[str, str]:
|
||||
"""Report the anchor state for ``hermes doctor``.
|
||||
"""Report the anchor state for ``hermes doctor`` as ``(status, detail)``.
|
||||
|
||||
Returns ``(status, detail)`` with status one of:
|
||||
|
||||
- ``"skip"`` — not applicable (non-macOS, no venv, or not uv-managed)
|
||||
- ``"active"`` — venv interpreter is pinned at a stable real-file anchor
|
||||
- ``"stale"`` — pinned but the interpreter changed since the last copy
|
||||
- ``"missing"`` — uv-managed interpreter with no stable anchor installed
|
||||
``skip`` = not applicable (non-macOS, no venv, not uv-managed); ``active`` = pinned at a
|
||||
stable real-file anchor; ``stale`` = pinned but the interpreter changed since the last copy;
|
||||
``missing`` = uv-managed interpreter with no anchor installed.
|
||||
"""
|
||||
if not is_macos():
|
||||
return "skip", "not macOS"
|
||||
|
||||
+283
-547
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
||||
"""CLI handlers for ``hermes migrate ...``.
|
||||
|
||||
Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply)
|
||||
rewrites references to xAI models retired on May 15, 2026.
|
||||
Currently exposes only ``hermes migrate xai`` — diagnoses and (with --apply) rewrites references to
|
||||
xAI models retired on May 15, 2026.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
+35
-69
@@ -1,27 +1,12 @@
|
||||
"""Recover from npm ``EBADENGINE`` failures by upgrading a managed npm.
|
||||
|
||||
The repo's ``.npmrc`` sets ``engine-strict=true`` and the root ``package.json``
|
||||
pins an ``engines.npm`` range, so an npm outside that range aborts every
|
||||
``npm ci`` / ``npm install`` we run inside the checkout::
|
||||
Rather than predicting the failure (which would mean a semver range matcher and an ``npm --version``
|
||||
probe before work that usually succeeds), we react to it: npm states the required range in the
|
||||
error, so the recovery reads the constraint straight out of the output it just produced.
|
||||
|
||||
npm error code EBADENGINE
|
||||
npm error notsup Required: {"node":">=26.0.0","npm":">=12.0.0"}
|
||||
npm error notsup Actual: {"npm":"10.9.8","node":"v22.23.1"}
|
||||
|
||||
Rather than predicting the failure (which would mean a semver range matcher and
|
||||
an ``npm --version`` probe before work that usually succeeds), we react to it:
|
||||
npm states the required range in the error, so the recovery reads the
|
||||
constraint straight out of the output it just produced.
|
||||
|
||||
Scope of the repair is deliberately narrow. Hermes only upgrades an npm that
|
||||
lives inside its **own** managed Node tree (``$HERMES_HOME/node``), installing
|
||||
in place with ``--prefix`` so ``bin/npm`` keeps resolving to the upgraded
|
||||
``lib/node_modules/npm``. A system / nvm / brew / Nix npm belongs to the user
|
||||
and their other projects; Hermes never modifies those. When the failing npm is
|
||||
one of those foreign installs, Hermes instead provisions its own managed Node
|
||||
tree (the same tree a fresh install creates), upgrades *that* npm into range,
|
||||
and hands the caller the managed npm to retry with — leaving the user's
|
||||
toolchain untouched.
|
||||
Scope of the repair is deliberately narrow. Hermes only upgrades an npm that lives inside its
|
||||
**own** managed Node tree (``$HERMES_HOME/node``), installing in place with ``--prefix`` so
|
||||
``bin/npm`` keeps resolving to the upgraded ``lib/node_modules/npm``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -61,14 +46,12 @@ _UPGRADE_TIMEOUT = 300
|
||||
|
||||
def is_ebadengine(output: str) -> bool:
|
||||
"""Return True when *output* is an npm engine-compatibility failure."""
|
||||
if not output:
|
||||
return False
|
||||
return "EBADENGINE" in output or "Unsupported engine" in output
|
||||
return bool(output) and ("EBADENGINE" in output or "Unsupported engine" in output)
|
||||
|
||||
|
||||
def _iter_required_blocks(output: str) -> list[dict]:
|
||||
def _iter_json_blocks(pattern: re.Pattern[str], output: str) -> list[dict]:
|
||||
blocks: list[dict] = []
|
||||
for match in _REQUIRED_RE.finditer(output or ""):
|
||||
for match in pattern.finditer(output or ""):
|
||||
try:
|
||||
parsed = json.loads(match.group(1))
|
||||
except ValueError:
|
||||
@@ -78,17 +61,19 @@ def _iter_required_blocks(output: str) -> list[dict]:
|
||||
return blocks
|
||||
|
||||
|
||||
def _iter_required_blocks(output: str) -> list[dict]:
|
||||
return _iter_json_blocks(_REQUIRED_RE, output)
|
||||
|
||||
|
||||
def required_npm_range(output: str) -> str | None:
|
||||
"""Return the ``engines.npm`` range npm demanded in *output*.
|
||||
|
||||
Returns ``None`` when the output has no engine failure, or when the
|
||||
failure is about Node rather than npm — upgrading npm cannot fix a Node
|
||||
version mismatch, so the caller must not try.
|
||||
Returns ``None`` when the output has no engine failure, or when the failure is about Node rather
|
||||
than npm — upgrading npm cannot fix a Node version mismatch, so the caller must not try.
|
||||
|
||||
When several packages report conflicting npm ranges the repo's own root
|
||||
constraint is preferred (it is the one we control); otherwise the first
|
||||
range wins, since any of them is a strict improvement over an npm that
|
||||
satisfies none.
|
||||
When several packages report conflicting npm ranges the repo's own root constraint is preferred
|
||||
(it is the one we control); otherwise the first range wins, since any of them is a strict
|
||||
improvement over an npm that satisfies none.
|
||||
"""
|
||||
if not is_ebadengine(output):
|
||||
return None
|
||||
@@ -109,12 +94,8 @@ def required_npm_range(output: str) -> str | None:
|
||||
|
||||
def actual_npm_version(output: str) -> str | None:
|
||||
"""Return the npm version npm reported as ``Actual`` in *output*."""
|
||||
for match in _ACTUAL_RE.finditer(output or ""):
|
||||
try:
|
||||
parsed = json.loads(match.group(1))
|
||||
except ValueError:
|
||||
continue
|
||||
if isinstance(parsed, dict) and parsed.get("npm"):
|
||||
for parsed in _iter_json_blocks(_ACTUAL_RE, output):
|
||||
if parsed.get("npm"):
|
||||
return str(parsed["npm"]).strip()
|
||||
return None
|
||||
|
||||
@@ -137,10 +118,9 @@ def managed_npm_prefix(npm: str | os.PathLike[str] | None) -> Path | None:
|
||||
"""Return the Hermes-managed Node root *npm* lives in, else ``None``.
|
||||
|
||||
Symlinks are resolved first: an install links ``~/.local/bin/npm`` at
|
||||
``$HERMES_HOME/node/bin/npm``, which itself links into
|
||||
``lib/node_modules/npm/bin/npm-cli.js``. Every one of those spellings is
|
||||
the managed npm and must be recognised as such, or the repair silently
|
||||
declines to fix the very install it owns.
|
||||
``$HERMES_HOME/node/bin/npm``, which itself links into ``lib/node_modules/npm/bin/npm-cli.js``.
|
||||
Every one of those spellings is the managed npm and must be recognised as such, or the repair
|
||||
silently declines to fix the very install it owns.
|
||||
"""
|
||||
if not npm:
|
||||
return None
|
||||
@@ -175,10 +155,10 @@ def upgrade_managed_npm(
|
||||
) -> bool:
|
||||
"""Upgrade the managed npm at *npm* in place to satisfy *npm_range*.
|
||||
|
||||
``--prefix`` targets the managed tree explicitly: a managed install writes
|
||||
``prefix=~/.local`` into ``$HERMES_HOME/node/etc/npmrc`` so that global
|
||||
installs land on PATH, and without the override the "upgrade" would install
|
||||
a second npm somewhere else while the managed one stayed stale.
|
||||
``--prefix`` targets the managed tree explicitly: a managed install writes ``prefix=~/.local``
|
||||
into ``$HERMES_HOME/node/etc/npmrc`` so that global installs land on PATH, and without the
|
||||
override the "upgrade" would install a second npm somewhere else while the managed one stayed
|
||||
stale.
|
||||
"""
|
||||
if not quiet:
|
||||
print(
|
||||
@@ -274,13 +254,10 @@ def _print_manual_fix(npm: str, npm_range: str, actual: str | None) -> None:
|
||||
def _provision_managed_npm(npm_range: str | None, *, quiet: bool = False) -> str | None:
|
||||
"""Provision a Hermes-managed Node tree and return a satisfying npm.
|
||||
|
||||
Installs the managed tree under ``$HERMES_HOME/node`` (reusing a healthy
|
||||
one when present), then upgrades its bundled npm to *npm_range* — a fresh
|
||||
Node LTS bundles an npm that may itself be outside the repo's range, so
|
||||
without the upgrade the caller's single retry would fail the same way.
|
||||
Falls back to the checkout's own ``engines.npm`` when npm did not state a
|
||||
range (a Node-only mismatch), so the managed npm ends up in range either
|
||||
way. Returns the managed npm path, or ``None`` when provisioning failed.
|
||||
Installs (or reuses) the tree under ``$HERMES_HOME/node``, then upgrades its bundled npm to
|
||||
*npm_range*: a fresh Node LTS may bundle an npm outside the repo's range, so without the
|
||||
upgrade the caller's single retry would fail the same way. Falls back to the checkout's
|
||||
``engines.npm`` when no range was stated. Returns the npm path, or ``None`` on failure.
|
||||
"""
|
||||
if not quiet:
|
||||
print(
|
||||
@@ -314,18 +291,9 @@ def maybe_repair_npm_engine(
|
||||
) -> str | None:
|
||||
"""Repair an ``EBADENGINE`` failure, never touching a foreign toolchain.
|
||||
|
||||
*output* is the combined stdout/stderr of the npm command that just failed.
|
||||
Returns the npm executable the caller should retry its command with —
|
||||
the same *npm* after an in-place upgrade of a Hermes-managed install, or
|
||||
a freshly provisioned managed npm when the failing npm belongs to the
|
||||
user (system / nvm / brew / Nix installs are never modified). Returns
|
||||
``None`` when no repair happened — not an engine failure, a Node mismatch
|
||||
a managed npm upgrade cannot fix, or a failed upgrade/bootstrap — leaving
|
||||
the original failure to stand.
|
||||
|
||||
The returned value is truthy exactly when the caller should retry once,
|
||||
so ``if maybe_repair_npm_engine(...)`` call sites keep working; they just
|
||||
must run the retry with the returned path.
|
||||
The returned value is truthy exactly when the caller should retry once, so ``if
|
||||
maybe_repair_npm_engine(...)`` call sites keep working; they just must run the retry with the
|
||||
returned path.
|
||||
"""
|
||||
if not npm or not is_ebadengine(output):
|
||||
return None
|
||||
@@ -336,9 +304,7 @@ def maybe_repair_npm_engine(
|
||||
if prefix is not None:
|
||||
# Hermes owns this npm — upgrade it in place. Only an npm-range
|
||||
# failure is fixable this way; a Node mismatch needs a Node upgrade.
|
||||
if not npm_range:
|
||||
return None
|
||||
if upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet):
|
||||
if npm_range and upgrade_managed_npm(npm, npm_range, prefix=prefix, quiet=quiet):
|
||||
return npm
|
||||
return None
|
||||
|
||||
|
||||
+15
-51
@@ -1,11 +1,8 @@
|
||||
"""
|
||||
Unified self-relaunch for Hermes CLI.
|
||||
"""Unified self-relaunch for Hermes CLI.
|
||||
|
||||
Preserves critical flags (--tui, --dev, --profile, --model, etc.) across
|
||||
process replacement so that ``hermes sessions browse`` or post-setup relaunch
|
||||
doesn't silently drop the user's UI mode or other preferences.
|
||||
|
||||
Also works when ``hermes`` is not on PATH (e.g. ``nix run`` or ``python -m``).
|
||||
Preserves critical flags (--tui, --dev, --profile, --model, etc.) across process replacement so that
|
||||
``hermes sessions browse`` or post-setup relaunch doesn't silently drop the user's UI mode or other
|
||||
preferences.
|
||||
"""
|
||||
|
||||
import os
|
||||
@@ -20,12 +17,10 @@ from hermes_cli._parser import (
|
||||
|
||||
|
||||
def _build_inherited_flag_table() -> list[tuple[str, bool]]:
|
||||
"""Build the ``(option_string, takes_value)`` table of flags that must
|
||||
survive a self-relaunch, by introspecting the real parser used by
|
||||
``hermes`` itself.
|
||||
"""Build the ``(option_string, takes_value)`` table of flags that must survive a self-relaunch.
|
||||
|
||||
A flag participates if its argparse Action carries
|
||||
``inherit_on_relaunch = True`` — set by ``_parser._inherited_flag``.
|
||||
Introspects the real ``hermes`` parser: a flag participates iff its argparse Action carries
|
||||
``inherit_on_relaunch = True`` (set by ``_parser._inherited_flag``).
|
||||
"""
|
||||
parser, _subparsers, chat_parser = build_top_level_parser()
|
||||
|
||||
@@ -80,20 +75,8 @@ def _extract_inherited_flags(argv: Sequence[str]) -> list[str]:
|
||||
def resolve_hermes_bin() -> Optional[str]:
|
||||
"""Find the hermes entry point.
|
||||
|
||||
Priority:
|
||||
1. ``sys.argv[0]`` if it resolves to a real executable.
|
||||
2. ``shutil.which("hermes")`` on PATH.
|
||||
3. ``None`` → caller should fall back to ``python -m hermes_cli.main``.
|
||||
|
||||
Windows note: ``os.access(path, os.X_OK)`` returns True for ``.py`` and
|
||||
``.pyc`` files on Windows (the OS treats anything listed in PATHEXT as
|
||||
executable, and Python files are often registered there). But
|
||||
``subprocess.run([script.py, ...])`` can't actually execute a .py
|
||||
directly — CreateProcessW needs a real .exe, not a script associated
|
||||
with the Python launcher. On Windows we therefore skip the argv[0]
|
||||
fast-path when it points at a .py file and fall through to either
|
||||
``hermes.exe`` on PATH or the ``sys.executable -m hermes_cli.main``
|
||||
fallback.
|
||||
Priority: 1. ``sys.argv[0]`` if it resolves to a real executable. 2. ``shutil.which("hermes")``
|
||||
on PATH. 3. ``None`` → caller should fall back to ``python -m hermes_cli.main``.
|
||||
"""
|
||||
argv0 = sys.argv[0]
|
||||
_is_windows = sys.platform == "win32"
|
||||
@@ -127,15 +110,7 @@ def build_relaunch_argv(
|
||||
preserve_inherited: bool = True,
|
||||
original_argv: Optional[Sequence[str]] = None,
|
||||
) -> list[str]:
|
||||
"""Construct an argv list for replacing the current process with hermes.
|
||||
|
||||
Args:
|
||||
extra_args: Arguments to append (e.g. ``["--resume", id]``).
|
||||
preserve_inherited: Whether to carry over UI / behaviour flags
|
||||
tagged with ``inherit_on_relaunch`` in the parser.
|
||||
original_argv: The original argv to scan for flags (defaults to
|
||||
``sys.argv[1:]``).
|
||||
"""
|
||||
"""Construct an argv list for replacing the current process with hermes."""
|
||||
bin_path = resolve_hermes_bin()
|
||||
|
||||
if bin_path:
|
||||
@@ -160,23 +135,12 @@ def relaunch(
|
||||
) -> None:
|
||||
"""Replace the current process with a fresh hermes invocation.
|
||||
|
||||
On POSIX we use ``os.execvp`` which replaces the running process with
|
||||
the new one in place — same PID, no double-fork. That's what the
|
||||
relaunch contract wants: "run hermes again as if the user had typed
|
||||
the new argv".
|
||||
On POSIX we use ``os.execvp`` which replaces the running process with the new one in place —
|
||||
same PID, no double-fork. That's what the relaunch contract wants: "run hermes again as if the
|
||||
user had typed the new argv".
|
||||
|
||||
Windows has no native exec semantics — ``os.execvp`` on Windows
|
||||
*emulates* exec by spawning the child and exiting the parent, but
|
||||
only works when the target is a real Win32 executable. Our target
|
||||
is usually ``hermes.exe`` (a Python console-script shim that wraps
|
||||
``python -m hermes_cli.main``) or a ``.cmd`` batch file, and both
|
||||
raise ``OSError(8, "Exec format error")`` on Windows' execvp.
|
||||
|
||||
The Windows-correct pattern is: spawn the child with ``subprocess.run``
|
||||
(which routes through ``cmd.exe`` via ``shell=False`` + PATHEXT resolution),
|
||||
wait for it to exit, then propagate its exit code via ``sys.exit``.
|
||||
That's functionally equivalent — the user sees "hermes exited, then
|
||||
new hermes started" — just with two PIDs in play instead of one.
|
||||
Windows has no native exec semantics — ``os.execvp`` on Windows *emulates* exec by spawning the
|
||||
child and exiting the parent, but only works when the target is a real Win32 executable.
|
||||
"""
|
||||
new_argv = build_relaunch_argv(
|
||||
extra_args, preserve_inherited=preserve_inherited, original_argv=original_argv
|
||||
|
||||
+214
-517
File diff suppressed because it is too large
Load Diff
+417
-591
File diff suppressed because it is too large
Load Diff
@@ -1,18 +1,10 @@
|
||||
"""Fresh-process recovery after the update's in-process restart phase aborts.
|
||||
|
||||
``hermes update`` performs its fleet restart in the interpreter that started
|
||||
before ``git pull``. When that phase raises — the module graph it is running
|
||||
no longer matches the checkout on disk — this module owns everything that
|
||||
happens next: which runtimes a clean child may relaunch, what counts as proof
|
||||
that a relaunch actually replaced the old generation, which pre-update
|
||||
processes are still alive, and whether any of that adds up to a recovery that
|
||||
may call itself complete (#92145).
|
||||
``hermes update`` performs its fleet restart in the interpreter that started before ``git pull``.
|
||||
|
||||
It is deliberately a separate owner from ``update_cmd``: the abort path has
|
||||
its own vocabulary (``verified`` / ``relaunch_attempted`` / ``failed``, serve
|
||||
units, survivors) and its own fail-closed contract, and the update monolith
|
||||
must not grow another authority surface (review on #96235). The child process
|
||||
itself lives in :mod:`hermes_cli.update_restart_recovery`.
|
||||
It is deliberately a separate owner from ``update_cmd``: the abort path has its own vocabulary
|
||||
(``verified`` / ``relaunch_attempted`` / ``failed``, serve units, survivors) and its own fail-closed
|
||||
contract, and the update monolith must not grow another authority surface (review on #96235).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -35,18 +27,14 @@ def _serve_unit_recovery_available() -> bool:
|
||||
def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
|
||||
"""Pre-update serve/dashboard runtimes that are STILL the same process.
|
||||
|
||||
Identity is the process incarnation ``(pid, create_time)``, never the PID
|
||||
alone. ``ledger_entries()`` re-verifies that pair on every read and prunes
|
||||
anything that is gone, so a live entry is a real process — but the numeric
|
||||
PID can be reused, and a serve that was restarted correctly can come back
|
||||
on the same number. Comparing PIDs alone would then report the successor
|
||||
as the pre-update survivor and keep the update incomplete forever
|
||||
(#92145 review).
|
||||
Identity is the process incarnation ``(pid, create_time)``, never the PID alone.
|
||||
``ledger_entries()`` re-verifies that pair on every read and prunes anything that is gone, so a
|
||||
live entry is a real process — but the numeric PID can be reused, and a serve that was restarted
|
||||
correctly can come back on the same number.
|
||||
|
||||
Anything still here after the recovery pass is a live runtime on the
|
||||
pre-update code generation, which is precisely the unsafe state #92145
|
||||
reports. Fail closed on missing evidence: an unreadable ledger, or an
|
||||
incarnation neither side can produce, counts the runtime as surviving.
|
||||
Anything still here after the recovery pass is a live runtime on the pre-update code generation,
|
||||
which is precisely the unsafe state #92145 reports. Fail closed on missing evidence: an
|
||||
unreadable ledger, or an incarnation neither side can produce, counts the runtime as surviving.
|
||||
"""
|
||||
planned: dict[int, dict] = {}
|
||||
try:
|
||||
@@ -57,13 +45,12 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
|
||||
if not isinstance(pid, int) or pid <= 0:
|
||||
continue
|
||||
detail = getattr(runtime, "detail", None)
|
||||
created = detail.get("create_time") if isinstance(detail, dict) else None
|
||||
planned[pid] = {
|
||||
"pid": pid,
|
||||
"kind": str(getattr(runtime, "kind", "")),
|
||||
"profile": str(getattr(runtime, "profile", "")),
|
||||
"supervisor": str(getattr(runtime, "supervisor", "")),
|
||||
"_create_time": created if isinstance(created, (int, float)) else None,
|
||||
"_create_time": _numeric(detail.get("create_time") if isinstance(detail, dict) else None),
|
||||
}
|
||||
except Exception as exc:
|
||||
logger.debug("Could not read planned serve runtimes: %s", exc)
|
||||
@@ -73,39 +60,36 @@ def _surviving_pre_update_serve_runtimes(plan) -> list[dict]:
|
||||
try:
|
||||
from hermes_cli.process_identity import ledger_entries
|
||||
|
||||
live: dict[int, float | None] = {}
|
||||
for entry in ledger_entries():
|
||||
if entry.get("purpose") not in ("serve", "dashboard"):
|
||||
continue
|
||||
pid = entry.get("pid")
|
||||
if not isinstance(pid, int):
|
||||
continue
|
||||
created = entry.get("create_time")
|
||||
live[pid] = created if isinstance(created, (int, float)) else None
|
||||
live: dict[int, float | None] = {
|
||||
entry["pid"]: _numeric(entry.get("create_time"))
|
||||
for entry in ledger_entries()
|
||||
if entry.get("purpose") in ("serve", "dashboard") and isinstance(entry.get("pid"), int)
|
||||
}
|
||||
except Exception as exc:
|
||||
logger.debug("Serve/dashboard survivor probe failed: %s", exc)
|
||||
return sorted(
|
||||
(_without_incarnation(row) for row in planned.values()),
|
||||
key=lambda row: row["pid"],
|
||||
)
|
||||
live = None
|
||||
survivors = []
|
||||
for pid, row in planned.items():
|
||||
if pid not in live:
|
||||
continue
|
||||
planned_created = row["_create_time"]
|
||||
live_created = live[pid]
|
||||
if (
|
||||
planned_created is not None
|
||||
and live_created is not None
|
||||
and abs(float(live_created) - float(planned_created)) >= 2.0
|
||||
):
|
||||
# Same number, different process: the pre-update runtime is gone
|
||||
# and something new registered under its PID. Not a survivor.
|
||||
continue
|
||||
if live is not None:
|
||||
if pid not in live:
|
||||
continue
|
||||
planned_created, live_created = row["_create_time"], live[pid]
|
||||
if (
|
||||
planned_created is not None
|
||||
and live_created is not None
|
||||
and abs(float(live_created) - float(planned_created)) >= 2.0
|
||||
):
|
||||
# Same number, different process: the pre-update runtime is gone
|
||||
# and something new registered under its PID. Not a survivor.
|
||||
continue
|
||||
survivors.append(_without_incarnation(row))
|
||||
return sorted(survivors, key=lambda row: row["pid"])
|
||||
|
||||
|
||||
def _numeric(value):
|
||||
return value if isinstance(value, (int, float)) else None
|
||||
|
||||
|
||||
def _without_incarnation(row: dict) -> dict:
|
||||
"""The operator-facing survivor row (incarnation is a matching key only)."""
|
||||
return {key: value for key, value in row.items() if key != "_create_time"}
|
||||
@@ -114,13 +98,8 @@ def _without_incarnation(row: dict) -> dict:
|
||||
def _qualified_serve_skips(skip_units) -> list[dict]:
|
||||
"""Scope-qualify the units the aborted phase already settled.
|
||||
|
||||
``restarted_scoped_units`` records ``<scope>/<unit>`` because
|
||||
``hermes-serve.service`` can exist in BOTH the user and the system
|
||||
manager, and those are two different processes. Handing the fresh child a
|
||||
bare unit name would make one settled scope suppress recovery of the other
|
||||
— the stale one would never be restarted and nothing downstream would say
|
||||
so. Entries that carry no scope (a payload built by a pre-update
|
||||
interpreter) are forwarded without one and read as scope-agnostic there.
|
||||
``restarted_scoped_units`` records ``<scope>/<unit>`` because ``hermes-serve.service`` can exist
|
||||
in BOTH the user and the system manager, and those are two different processes.
|
||||
"""
|
||||
rows: list[dict] = []
|
||||
for entry in sorted(skip_units or ()):
|
||||
@@ -141,74 +120,35 @@ def _recover_gateway_restart_after_abort(
|
||||
) -> dict[str, list]:
|
||||
"""Retry supervised gateway restarts from a clean Python process.
|
||||
|
||||
``hermes update`` normally performs the fleet restart in the interpreter
|
||||
that started before ``git pull``. If that phase raises while importing the
|
||||
new tree, a warning alone leaves the old gateway alive against new files on
|
||||
disk. The recovery boundary launches the existing per-profile
|
||||
``gateway restart`` command through a new interpreter, preserving its
|
||||
platform-specific drain and service-manager logic without inheriting the
|
||||
stale ``sys.modules`` graph.
|
||||
``hermes update`` normally performs the fleet restart in the interpreter that started before
|
||||
``git pull``.
|
||||
|
||||
Only profiles classified as supervisor-owned by the pre-update inventory
|
||||
are handed off. A manual gateway must remain running and be reported for
|
||||
explicit operator action rather than being killed without a relaunch
|
||||
authority; serve/dashboard runtimes from the spawn ledger are likewise
|
||||
recorded as skipped with a reason instead of vanishing from the pass.
|
||||
The returned protocol is persisted in the update receipt so operators can
|
||||
distinguish a spawn failure from a per-profile failure.
|
||||
|
||||
The same child additionally restarts active ``hermes-serve*`` systemd
|
||||
units (#92145). ``hermes serve`` hosts ``tui_gateway.server`` and is
|
||||
restarted by the in-process phase alongside the gateway units, but no
|
||||
per-profile ``gateway restart`` command reaches it — so an abort used to
|
||||
leave it holding the pre-pull module graph with nothing left to notice.
|
||||
|
||||
``skip_units`` names the units the aborted phase already settled, as
|
||||
``<scope>/<unit>``. The scope is part of the identity, not decoration:
|
||||
``hermes-serve.service`` can exist in both the user and the system
|
||||
manager as two different processes, and an unqualified token would let a
|
||||
settled one suppress recovery of a stale one (review on #96235).
|
||||
|
||||
Outcome honesty: ``verified`` means the fresh child independently observed
|
||||
the profile's systemd unit active after the relaunch. A zero exit from
|
||||
``gateway restart`` alone is NOT observed proof that the new code
|
||||
generation is serving, so those outcomes are reported as
|
||||
``relaunch_attempted`` and never claim supervisor coverage.
|
||||
Only profiles classified as supervisor-owned by the pre-update inventory are handed off.
|
||||
"""
|
||||
from hermes_cli.update_cmd import _gateway_recovery_partition
|
||||
|
||||
candidates, skipped = _gateway_recovery_partition(
|
||||
plan, skip_profiles=skip_profiles
|
||||
)
|
||||
candidates, skipped = _gateway_recovery_partition(plan, skip_profiles=skip_profiles)
|
||||
profiles = sorted(candidates)
|
||||
recover_serve = _serve_unit_recovery_available()
|
||||
_empty_serve: dict[str, list] = {"verified": [], "failed": []}
|
||||
if not profiles and not recover_serve:
|
||||
|
||||
def _result(requested, verified, relaunch_attempted, failed, serve_units=None) -> dict[str, list]:
|
||||
return {
|
||||
"requested": [],
|
||||
"verified": [],
|
||||
"relaunch_attempted": [],
|
||||
"failed": [],
|
||||
"requested": requested,
|
||||
"verified": verified,
|
||||
"relaunch_attempted": relaunch_attempted,
|
||||
"failed": failed,
|
||||
"skipped": skipped,
|
||||
"serve_units": dict(_empty_serve),
|
||||
"serve_units": dict(_empty_serve) if serve_units is None else serve_units,
|
||||
}
|
||||
|
||||
if not profiles and not recover_serve:
|
||||
return _result([], [], [], [])
|
||||
|
||||
def _all_failed() -> dict[str, list]:
|
||||
return {
|
||||
"requested": profiles,
|
||||
"verified": [],
|
||||
"relaunch_attempted": [],
|
||||
"failed": profiles,
|
||||
"skipped": skipped,
|
||||
"serve_units": dict(_empty_serve),
|
||||
}
|
||||
return _result(profiles, [], [], profiles)
|
||||
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"hermes_cli.update_restart_recovery",
|
||||
"--stdin",
|
||||
]
|
||||
command = [sys.executable, "-m", "hermes_cli.update_restart_recovery", "--stdin"]
|
||||
env = os.environ.copy()
|
||||
env["HERMES_UPDATE_RESTART_RECOVERY"] = "1"
|
||||
for marker in ("_HERMES_GATEWAY", "HERMES_GATEWAY", "HERMES_GATEWAY_MODE"):
|
||||
@@ -224,15 +164,7 @@ def _recover_gateway_restart_after_abort(
|
||||
if not systemd_run:
|
||||
logger.warning("Cannot isolate fresh gateway recovery from the gateway cgroup")
|
||||
return _all_failed()
|
||||
command = [
|
||||
systemd_run,
|
||||
"--user",
|
||||
"--scope",
|
||||
"--quiet",
|
||||
"--collect",
|
||||
"--",
|
||||
*command,
|
||||
]
|
||||
command = [systemd_run, "--user", "--scope", "--quiet", "--collect", "--", *command]
|
||||
|
||||
kwargs = {
|
||||
"input": json.dumps(
|
||||
@@ -318,47 +250,26 @@ def _recover_gateway_restart_after_abort(
|
||||
logger.warning("Fresh gateway restart recovery returned incomplete profiles")
|
||||
return _all_failed()
|
||||
|
||||
if verified:
|
||||
print(
|
||||
" ✓ Restarted supervised gateway(s) in a fresh process"
|
||||
" (systemd-verified active): " + ", ".join(sorted(verified))
|
||||
)
|
||||
if relaunch_attempted:
|
||||
print(
|
||||
" ⚠ Relaunch attempted in a fresh process but not"
|
||||
" supervisor-verified (check these gateways manually): "
|
||||
+ ", ".join(sorted(relaunch_attempted))
|
||||
)
|
||||
if serve_units["verified"]:
|
||||
print(
|
||||
" ✓ Restarted serve unit(s) in a fresh process"
|
||||
" (new main PID observed): "
|
||||
+ ", ".join(serve_units["verified"])
|
||||
)
|
||||
if serve_units["failed"]:
|
||||
print(
|
||||
" ⚠ Could not verify a replacement for serve unit(s): "
|
||||
+ ", ".join(serve_units["failed"])
|
||||
)
|
||||
return {
|
||||
"requested": profiles,
|
||||
"verified": sorted(verified),
|
||||
"relaunch_attempted": sorted(relaunch_attempted),
|
||||
"failed": sorted(failed),
|
||||
"skipped": skipped,
|
||||
"serve_units": serve_units,
|
||||
}
|
||||
verified, relaunch_attempted, failed = sorted(verified), sorted(relaunch_attempted), sorted(failed)
|
||||
for names, text in (
|
||||
(verified, " ✓ Restarted supervised gateway(s) in a fresh process (systemd-verified active): "),
|
||||
(relaunch_attempted, " ⚠ Relaunch attempted in a fresh process but not"
|
||||
" supervisor-verified (check these gateways manually): "),
|
||||
(serve_units["verified"], " ✓ Restarted serve unit(s) in a fresh process (new main PID observed): "),
|
||||
(serve_units["failed"], " ⚠ Could not verify a replacement for serve unit(s): "),
|
||||
):
|
||||
if names:
|
||||
print(text + ", ".join(names))
|
||||
return _result(profiles, verified, relaunch_attempted, failed, serve_units)
|
||||
|
||||
|
||||
def _warn_stale_serve_runtimes(rows) -> None:
|
||||
"""Name the serve/dashboard processes that survived on pre-update code.
|
||||
|
||||
The original #92145 report is a user watching every chat turn fail with an
|
||||
``ImportError`` for a symbol that imports fine on disk, with nothing in the
|
||||
terminal naming the responsible process. ``hermes serve`` hosts
|
||||
``tui_gateway.server``; when its unit was never restarted it keeps the
|
||||
pre-pull ``sys.modules`` graph and there is no gateway row anywhere that
|
||||
reveals it. Print the PIDs and the exact command that fixes it.
|
||||
``hermes serve`` hosts ``tui_gateway.server``; if its unit was never restarted it keeps the
|
||||
pre-pull ``sys.modules`` graph, every chat turn fails with an ``ImportError`` for a symbol
|
||||
that imports fine on disk, and no gateway row reveals the culprit. Print the PIDs and the
|
||||
exact command that fixes it.
|
||||
"""
|
||||
if not rows:
|
||||
return
|
||||
@@ -388,28 +299,11 @@ def _abort_recovery_is_complete(
|
||||
) -> bool:
|
||||
"""May a fresh-process recovery clear the incomplete flag?
|
||||
|
||||
Only when EVERY inventoried runtime family is accounted for. The gateway
|
||||
leg alone is not enough (#92145): the post-update read-back
|
||||
(``collect_fleet_versions``) is gateway-only — it reads each profile's
|
||||
``gateway_state.json`` / control socket — so a ``hermes serve`` still
|
||||
holding the pre-update ``sys.modules`` graph is invisible to both the
|
||||
recovery pass and the verification that follows it. Clearing the flag on
|
||||
gateway coverage alone is exactly how an update reported success while
|
||||
every chat turn kept failing with an ``ImportError`` for a symbol that
|
||||
imports fine on disk.
|
||||
Only when EVERY inventoried runtime family is accounted for.
|
||||
|
||||
Fail closed on each leg:
|
||||
|
||||
* every planned gateway profile is covered, with nothing failed and
|
||||
nothing merely ``relaunch_attempted`` (rc 0 is not observed coverage);
|
||||
* no ``hermes-serve*`` unit failed to produce a verified replacement; and
|
||||
* no serve/dashboard process from the pre-update inventory is still the
|
||||
same process (``stale_runtime_rows``).
|
||||
|
||||
``planned_gateway_profiles`` being empty is deliberately NOT completeness:
|
||||
with no gateway leg to prove, the caller's own fail-closed contract
|
||||
(``_restart_phase_failure_is_incomplete``) plus the stale-runtime rows
|
||||
decide the outcome.
|
||||
``planned_gateway_profiles`` being empty is deliberately NOT completeness: with no gateway leg
|
||||
to prove, the caller's own fail-closed contract (``_restart_phase_failure_is_incomplete``) plus
|
||||
the stale-runtime rows decide the outcome.
|
||||
"""
|
||||
if not planned_gateway_profiles:
|
||||
return False
|
||||
|
||||
@@ -1,21 +1,8 @@
|
||||
"""Image-managed install refusal contract (#91277 Phase 3).
|
||||
|
||||
One shared admission gate for every surface that can start an in-place
|
||||
``hermes update`` mutation (CLI apply, CLI --check, dashboard update
|
||||
endpoint). The decision layers:
|
||||
|
||||
1. **Baked provenance marker** (``/etc/hermes/image-provenance.json``,
|
||||
written by the image build — see :mod:`hermes_cli.image_provenance`):
|
||||
authoritative ground truth that this filesystem came from an immutable
|
||||
image. Fail-closed: a present-but-malformed marker still refuses.
|
||||
2. **Filesystem heuristics** (``detect_install_method()``): the pre-existing
|
||||
docker/nix/apt detection, kept as the fallback for images built before
|
||||
the marker existed and for package-managed installs that have no image
|
||||
marker at all.
|
||||
|
||||
A refusal prints the real update command for the deployment kind, records a
|
||||
``refused`` receipt (so fleet tooling sees "this install cannot self-update,
|
||||
use <command>" instead of a silent non-update), and exits 2 on CLI surfaces.
|
||||
A refusal prints the real update command for the deployment kind, records a ``refused`` receipt (so
|
||||
fleet tooling sees "this install cannot self-update, use <command>" instead of a silent non-update),
|
||||
and exits 2 on CLI surfaces.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -40,9 +27,8 @@ class UpdateRefusal:
|
||||
def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]:
|
||||
"""Return an :class:`UpdateRefusal` when in-place update must not run.
|
||||
|
||||
``None`` means the install is eligible for in-place update (git checkout
|
||||
or unknown-but-mutable). Never raises; on any internal error it falls
|
||||
back to the heuristic layer only.
|
||||
``None`` means the install is eligible for in-place update (git checkout or unknown-but-
|
||||
mutable). Never raises; on any internal error it falls back to the heuristic layer only.
|
||||
"""
|
||||
# Layer 1: baked provenance marker — authoritative when present.
|
||||
try:
|
||||
@@ -116,9 +102,8 @@ def evaluate_update_admission(project_root: Path) -> Optional[UpdateRefusal]:
|
||||
def record_refusal_receipt(refusal: UpdateRefusal) -> None:
|
||||
"""Write a minimal ``refused`` receipt for a blocked update attempt.
|
||||
|
||||
Gives fleet tooling a durable record that an update was ATTEMPTED and
|
||||
refused ("not updatable in place, use <command>") instead of a silent
|
||||
nothing. Best-effort; never raises.
|
||||
Gives fleet tooling a durable record that an update was ATTEMPTED and refused ("not updatable in
|
||||
place, use <command>") instead of a silent nothing. Best-effort; never raises.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.update_receipt import (
|
||||
|
||||
+176
-271
@@ -1,38 +1,19 @@
|
||||
"""Runtime inventory + update plan for the fleet-update pipeline (#91277 Phase 2).
|
||||
|
||||
One read-only pass that answers, BEFORE any mutation: what Hermes runtimes
|
||||
are running on this machine, how is each one deployed, which of them will
|
||||
this update touch, and how will each be restarted?
|
||||
One read-only pass that answers, BEFORE any mutation: what Hermes runtimes are running on this
|
||||
machine, how is each one deployed, which of them will this update touch, and how will each be
|
||||
restarted?
|
||||
|
||||
This is the "plan" phase of the transactional deployment model (#88683):
|
||||
|
||||
plan → snapshot → apply → restart-per-kind → verify → report
|
||||
|
||||
The module is deliberately side-effect free — every collector is a probe
|
||||
over primitives that already exist (`find_profile_gateway_processes`,
|
||||
`_get_service_pids`, `gateway_state.json` code stamps from #91283,
|
||||
`detect_install_method`) — so `hermes update --plan` can run on a live
|
||||
fleet with zero risk, and the update receipt can embed the inventory
|
||||
without changing update behavior.
|
||||
|
||||
Deployment kinds (the concept most fleet-update bugs were missing):
|
||||
|
||||
git — source checkout; updatable in place via `hermes update`
|
||||
docker — published image; NOT updatable in place (pull + recreate)
|
||||
nix/apt — package-manager owned; updatable via the manager only
|
||||
unknown — no marker; treated as in-place updatable (legacy default)
|
||||
|
||||
Supervisors (how a runtime is restarted after code changes):
|
||||
|
||||
systemd / launchd — restart via the service manager (fleet-wide)
|
||||
desktop — Desktop app supervises `hermes serve`; it respawns
|
||||
manual — plain process; SIGTERM + watcher/manual relaunch
|
||||
The module is deliberately side-effect free — every collector is a probe over primitives that
|
||||
already exist (`find_profile_gateway_processes`, `_get_service_pids`, `gateway_state.json` code
|
||||
stamps from #91283, `detect_install_method`) — so `hermes update --plan` can run on a live fleet
|
||||
with zero risk, and the update receipt can embed the inventory without changing update behavior.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from contextlib import contextmanager
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
@@ -70,17 +51,10 @@ class UpdatePlan:
|
||||
runtimes: list = field(default_factory=list) # list[RuntimeRecord]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
payload = asdict(self)
|
||||
payload["runtimes"] = [
|
||||
r.to_dict() if isinstance(r, RuntimeRecord) else r
|
||||
for r in self.runtimes
|
||||
]
|
||||
return payload
|
||||
return asdict(self) # recursive: RuntimeRecord entries become dicts, dict entries are copied
|
||||
|
||||
|
||||
def _detect_supervisor_for_pid(
|
||||
pid: int, service_pids: set, windows_service_pids: set | None = None
|
||||
) -> str:
|
||||
def _detect_supervisor_for_pid(pid: int, service_pids: set, windows_service_pids: set | None = None) -> str:
|
||||
"""Classify how a live gateway PID is supervised."""
|
||||
if windows_service_pids and pid in windows_service_pids:
|
||||
# SCM-supervised Windows gateway (WinSW/NSSM/sc.exe create): the
|
||||
@@ -102,56 +76,99 @@ def _detect_supervisor_for_pid(
|
||||
return "manual"
|
||||
|
||||
|
||||
_RESTART_MECHANISMS = {
|
||||
"systemd": "systemd",
|
||||
"launchd": "launchd",
|
||||
"desktop": "desktop",
|
||||
"windows-service": "windows-service",
|
||||
"manual-serve": "respawn-argv",
|
||||
}
|
||||
|
||||
_MECHANISM_DESCRIPTIONS = {
|
||||
"systemd": "systemctl restart (drain-first SIGUSR1 when supported)",
|
||||
"launchd": "launchctl kickstart -k (drain-first, per-label domain)",
|
||||
"desktop": "Desktop app respawns its serve backend",
|
||||
"windows-service": "sc.exe stop before venv mutation, sc.exe start after update",
|
||||
"respawn-argv": "stop before code swap, relaunch with recorded launch args",
|
||||
}
|
||||
|
||||
|
||||
def _restart_mechanism(supervisor: str, profile: str) -> str:
|
||||
"""Machine-readable restart mechanism id for a runtime.
|
||||
|
||||
THE policy table (#91277 Phase 2): restart execution consumes these ids
|
||||
via :func:`match_runtime_outcomes` / the update's restart phase, and the
|
||||
receipt records per-runtime outcomes against them. Display strings are
|
||||
derived by :func:`describe_restart_mechanism` — never the other way
|
||||
around.
|
||||
THE policy table (#91277 Phase 2): restart execution consumes these ids via
|
||||
:func:`match_runtime_outcomes` / the update's restart phase, and the receipt records per-runtime
|
||||
outcomes against them. Display strings are derived by :func:`describe_restart_mechanism` — never
|
||||
the other way around.
|
||||
"""
|
||||
if supervisor == "systemd":
|
||||
return "systemd"
|
||||
if supervisor == "launchd":
|
||||
return "launchd"
|
||||
if supervisor == "desktop":
|
||||
return "desktop"
|
||||
if supervisor == "windows-service":
|
||||
return "windows-service"
|
||||
if supervisor == "manual-serve":
|
||||
return "respawn-argv"
|
||||
return "manual"
|
||||
return _RESTART_MECHANISMS.get(supervisor, "manual")
|
||||
|
||||
|
||||
def describe_restart_mechanism(mechanism: str, profile: str) -> str:
|
||||
"""Human-readable description of a restart mechanism id."""
|
||||
if mechanism == "systemd":
|
||||
return "systemctl restart (drain-first SIGUSR1 when supported)"
|
||||
if mechanism == "launchd":
|
||||
return "launchctl kickstart -k (drain-first, per-label domain)"
|
||||
if mechanism == "desktop":
|
||||
return "Desktop app respawns its serve backend"
|
||||
if mechanism == "windows-service":
|
||||
return "sc.exe stop before venv mutation, sc.exe start after update"
|
||||
if mechanism == "respawn-argv":
|
||||
return "stop before code swap, relaunch with recorded launch args"
|
||||
described = _MECHANISM_DESCRIPTIONS.get(mechanism)
|
||||
if described is not None:
|
||||
return described
|
||||
if profile != "default":
|
||||
return f"hermes -p {profile} gateway restart"
|
||||
return "hermes gateway restart"
|
||||
|
||||
|
||||
def _runtime(kind: str, profile: str, pid: Optional[int], supervisor: str, **extra: Any) -> RuntimeRecord:
|
||||
"""A :class:`RuntimeRecord` with ``restart_via`` derived from its supervisor."""
|
||||
return RuntimeRecord(
|
||||
kind=kind,
|
||||
profile=profile,
|
||||
pid=pid,
|
||||
supervisor=supervisor,
|
||||
restart_via=_restart_mechanism(supervisor, profile),
|
||||
**extra,
|
||||
)
|
||||
|
||||
|
||||
def _int_or_none(value: Any) -> Optional[int]:
|
||||
try:
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _gateway_record(
|
||||
profile: str,
|
||||
pid: int,
|
||||
supervisor: str,
|
||||
code_sha: Any = None,
|
||||
code_version: Any = None,
|
||||
) -> RuntimeRecord:
|
||||
return _runtime(
|
||||
"gateway",
|
||||
profile,
|
||||
pid,
|
||||
supervisor,
|
||||
code_sha=str(code_sha) if code_sha else None,
|
||||
code_version=code_version,
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _probe(label: str):
|
||||
"""Run one inventory collector; a failure is logged at debug and yields fewer rows, never an exception."""
|
||||
try:
|
||||
yield
|
||||
except Exception as exc:
|
||||
logger.debug("%s failed: %s", label, exc)
|
||||
|
||||
|
||||
def collect_runtime_inventory() -> UpdatePlan:
|
||||
"""Build the pre-update plan. Read-only; never raises.
|
||||
|
||||
Every collector degrades independently — a probe failure yields fewer
|
||||
rows, not an exception. The result is embeddable in the update receipt
|
||||
and printable via :func:`print_update_plan`.
|
||||
Every collector degrades independently — a probe failure yields fewer rows, not an exception.
|
||||
The result is embeddable in the update receipt and printable via :func:`print_update_plan`.
|
||||
"""
|
||||
plan = UpdatePlan()
|
||||
|
||||
# --- install shape / deployment kind ---------------------------------
|
||||
try:
|
||||
with _probe("Install-method probe"):
|
||||
from hermes_cli.config import (
|
||||
detect_install_method,
|
||||
get_managed_system,
|
||||
@@ -169,7 +186,7 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
# container can look like `git` to the heuristics while the running
|
||||
# filesystem is actually an immutable image. Fail-closed: an invalid
|
||||
# marker still flips the plan to not-updatable.
|
||||
try:
|
||||
with _probe("Image provenance probe"):
|
||||
from hermes_cli.image_provenance import read_image_provenance
|
||||
|
||||
provenance = read_image_provenance()
|
||||
@@ -177,25 +194,19 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
plan.updatable_in_place = False
|
||||
if provenance.valid and provenance.manager:
|
||||
plan.install_method = provenance.manager
|
||||
except Exception as exc:
|
||||
logger.debug("Image provenance probe failed: %s", exc)
|
||||
plan.update_mechanism = recommended_update_command_for_method(method)
|
||||
except Exception as exc:
|
||||
logger.debug("Install-method probe failed: %s", exc)
|
||||
|
||||
# --- expected code identity (pre-pull) --------------------------------
|
||||
try:
|
||||
with _probe("Code-identity probe"):
|
||||
from hermes_cli.build_info import get_code_identity
|
||||
|
||||
identity = get_code_identity(refresh=True)
|
||||
plan.expected_sha = identity.get("sha")
|
||||
plan.expected_version = identity.get("version")
|
||||
except Exception as exc:
|
||||
logger.debug("Code-identity probe failed: %s", exc)
|
||||
|
||||
# --- profiles ----------------------------------------------------------
|
||||
profile_homes: list[tuple[str, Path]] = []
|
||||
try:
|
||||
with _probe("Profile enumeration"):
|
||||
from hermes_cli.profiles import (
|
||||
_get_default_hermes_home,
|
||||
_get_profiles_root,
|
||||
@@ -208,24 +219,16 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
root = _get_profiles_root()
|
||||
if root.is_dir():
|
||||
for entry in sorted(root.iterdir()):
|
||||
if (
|
||||
entry.is_dir()
|
||||
and entry.name != "default"
|
||||
and _PROFILE_ID_RE.match(entry.name)
|
||||
):
|
||||
if entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name):
|
||||
profile_homes.append((entry.name, entry))
|
||||
plan.profiles = [name for name, _ in profile_homes]
|
||||
except Exception as exc:
|
||||
logger.debug("Profile enumeration failed: %s", exc)
|
||||
|
||||
# --- service-managed PIDs (fleet-wide) ---------------------------------
|
||||
service_pids: set = set()
|
||||
try:
|
||||
with _probe("Service-PID probe"):
|
||||
from hermes_cli.gateway import _get_service_pids
|
||||
|
||||
service_pids = _get_service_pids(all_profiles=True) or set()
|
||||
except Exception as exc:
|
||||
logger.debug("Service-PID probe failed: %s", exc)
|
||||
|
||||
# --- SCM-supervised gateway PIDs (Windows) ------------------------------
|
||||
# find_windows_gateway_services() maps validated gateway PIDs through
|
||||
@@ -234,19 +237,14 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
# `sc.exe start`, so the plan must carry the matching mechanism id for
|
||||
# the #91277 Phase 2 reconciliation and the fleet check.
|
||||
windows_service_pids: set = set()
|
||||
try:
|
||||
with _probe("Windows SCM service-ownership probe"):
|
||||
from hermes_cli.gateway import find_windows_gateway_services
|
||||
|
||||
windows_service_pids = {
|
||||
int(service.gateway_pid)
|
||||
for service in find_windows_gateway_services()
|
||||
}
|
||||
except Exception as exc:
|
||||
logger.debug("Windows SCM service-ownership probe failed: %s", exc)
|
||||
windows_service_pids = {int(service.gateway_pid) for service in find_windows_gateway_services()}
|
||||
|
||||
# --- per-profile gateways (PID files + runtime status stamps) ----------
|
||||
seen_pids: set[int] = set()
|
||||
try:
|
||||
with _probe("Gateway-state inventory"):
|
||||
from gateway.status import _pid_exists, read_runtime_status
|
||||
|
||||
for profile, home in profile_homes:
|
||||
@@ -260,91 +258,54 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
identity = identify_gateway(home)
|
||||
except Exception:
|
||||
identity = None
|
||||
if identity:
|
||||
try:
|
||||
sock_pid = int(identity.get("pid"))
|
||||
except (TypeError, ValueError):
|
||||
sock_pid = None
|
||||
if sock_pid is not None:
|
||||
if sock_pid in seen_pids:
|
||||
# One multiplex gateway can answer identify for
|
||||
# several profile homes — one runtime record per
|
||||
# process, not per home.
|
||||
continue
|
||||
seen_pids.add(sock_pid)
|
||||
declared = identity.get("supervisor")
|
||||
supervisor = (
|
||||
str(declared)
|
||||
if declared
|
||||
else _detect_supervisor_for_pid(
|
||||
sock_pid, service_pids, windows_service_pids
|
||||
)
|
||||
)
|
||||
sock_sha = identity.get("code_sha")
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind="gateway",
|
||||
profile=profile,
|
||||
pid=sock_pid,
|
||||
supervisor=supervisor,
|
||||
code_sha=str(sock_sha) if sock_sha else None,
|
||||
code_version=identity.get("code_version"),
|
||||
restart_via=_restart_mechanism(supervisor, profile),
|
||||
)
|
||||
)
|
||||
sock_pid = _int_or_none(identity.get("pid")) if identity else None
|
||||
if sock_pid is not None:
|
||||
if sock_pid in seen_pids:
|
||||
# One multiplex gateway can answer identify for
|
||||
# several profile homes — one runtime record per
|
||||
# process, not per home.
|
||||
continue
|
||||
record = read_runtime_status(home / "gateway_state.json")
|
||||
pid: Optional[int] = None
|
||||
code_sha = code_version = None
|
||||
if record:
|
||||
try:
|
||||
pid = int(record.get("pid"))
|
||||
except (TypeError, ValueError):
|
||||
pid = None
|
||||
code_sha = record.get("code_sha")
|
||||
code_version = record.get("code_version")
|
||||
seen_pids.add(sock_pid)
|
||||
declared = identity.get("supervisor")
|
||||
supervisor = (
|
||||
str(declared)
|
||||
if declared
|
||||
else _detect_supervisor_for_pid(sock_pid, service_pids, windows_service_pids)
|
||||
)
|
||||
plan.runtimes.append(
|
||||
_gateway_record(
|
||||
profile, sock_pid, supervisor, identity.get("code_sha"), identity.get("code_version")
|
||||
)
|
||||
)
|
||||
continue
|
||||
record = read_runtime_status(home / "gateway_state.json") or {}
|
||||
pid = _int_or_none(record.get("pid"))
|
||||
if pid is None or not _pid_exists(pid):
|
||||
continue
|
||||
seen_pids.add(pid)
|
||||
supervisor = _detect_supervisor_for_pid(
|
||||
pid, service_pids, windows_service_pids
|
||||
)
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind="gateway",
|
||||
profile=profile,
|
||||
pid=pid,
|
||||
supervisor=supervisor,
|
||||
code_sha=str(code_sha) if code_sha else None,
|
||||
code_version=code_version,
|
||||
restart_via=_restart_mechanism(supervisor, profile),
|
||||
_gateway_record(
|
||||
profile,
|
||||
pid,
|
||||
_detect_supervisor_for_pid(pid, service_pids, windows_service_pids),
|
||||
record.get("code_sha"),
|
||||
record.get("code_version"),
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Gateway-state inventory failed: %s", exc)
|
||||
|
||||
# PID-file mapped gateways not covered by a runtime-status record
|
||||
try:
|
||||
with _probe("PID-file gateway inventory"):
|
||||
from hermes_cli.gateway import find_profile_gateway_processes
|
||||
|
||||
for proc in find_profile_gateway_processes():
|
||||
if proc.pid in seen_pids:
|
||||
continue
|
||||
seen_pids.add(proc.pid)
|
||||
supervisor = _detect_supervisor_for_pid(
|
||||
proc.pid, service_pids, windows_service_pids
|
||||
)
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind="gateway",
|
||||
profile=proc.profile,
|
||||
pid=proc.pid,
|
||||
supervisor=supervisor,
|
||||
restart_via=_restart_mechanism(supervisor, proc.profile),
|
||||
_gateway_record(
|
||||
proc.profile, proc.pid, _detect_supervisor_for_pid(proc.pid, service_pids, windows_service_pids)
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("PID-file gateway inventory failed: %s", exc)
|
||||
|
||||
# Serve/dashboard backends from the spawn ledger (#63206). These are the
|
||||
# runtimes the gateway collectors above can never see: a manually
|
||||
@@ -355,7 +316,7 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
# fabricates a row. Desktop-supervised backends are classified by their
|
||||
# recorded spawner still being alive — those restart via the Desktop's
|
||||
# own respawn, not ours.
|
||||
try:
|
||||
with _probe("Serve/dashboard ledger inventory"):
|
||||
from hermes_cli.process_identity import ledger_entries, spawner_is_dead
|
||||
|
||||
for entry in ledger_entries():
|
||||
@@ -366,16 +327,12 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
if not isinstance(pid, int) or pid in seen_pids:
|
||||
continue
|
||||
seen_pids.add(pid)
|
||||
has_live_spawner = spawner_is_dead(entry) is False
|
||||
supervisor = "desktop" if has_live_spawner else "manual-serve"
|
||||
profile = str(entry.get("profile") or "default")
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind=str(purpose),
|
||||
profile=profile,
|
||||
pid=pid,
|
||||
supervisor=supervisor,
|
||||
restart_via=_restart_mechanism(supervisor, profile),
|
||||
_runtime(
|
||||
str(purpose),
|
||||
str(entry.get("profile") or "default"),
|
||||
pid,
|
||||
"desktop" if spawner_is_dead(entry) is False else "manual-serve",
|
||||
detail={
|
||||
"argv": entry.get("argv") or "",
|
||||
"host": entry.get("host") or "",
|
||||
@@ -388,8 +345,6 @@ def collect_runtime_inventory() -> UpdatePlan:
|
||||
},
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Serve/dashboard ledger inventory failed: %s", exc)
|
||||
|
||||
return plan
|
||||
|
||||
@@ -415,14 +370,8 @@ def print_update_plan(plan: UpdatePlan) -> None:
|
||||
print(f" Running services to restart ({len(plan.runtimes)}):")
|
||||
for runtime in plan.runtimes:
|
||||
sha = f" @ {runtime.code_sha[:8]}" if runtime.code_sha else ""
|
||||
print(
|
||||
f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid}"
|
||||
f" — {runtime.supervisor}{sha}"
|
||||
)
|
||||
print(
|
||||
" restart: "
|
||||
f"{describe_restart_mechanism(runtime.restart_via, runtime.profile)}"
|
||||
)
|
||||
print(f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid} — {runtime.supervisor}{sha}")
|
||||
print(f" restart: {describe_restart_mechanism(runtime.restart_via, runtime.profile)}")
|
||||
|
||||
|
||||
_SERVE_KINDS = ("serve", "dashboard")
|
||||
@@ -431,11 +380,8 @@ _SERVE_KINDS = ("serve", "dashboard")
|
||||
def _serve_unit_matches_profile(profile: str, unit: object) -> bool:
|
||||
"""Does *unit* name a ``hermes-serve*``/``hermes-dashboard*`` unit for *profile*?
|
||||
|
||||
Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's
|
||||
``hermes-gateway*`` names never cover them (#100479). Exact names only —
|
||||
``work`` must not claim ``hermes-serve-workbench`` — and a scope prefix
|
||||
(``user/hermes-serve``) is tolerated because the restart phase records
|
||||
scope-qualified identities in some lists.
|
||||
Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's ``hermes-gateway*`` names
|
||||
never cover them (#100479).
|
||||
"""
|
||||
name = str(unit).removesuffix(".service")
|
||||
if "/" in name:
|
||||
@@ -480,32 +426,13 @@ def match_runtime_outcomes(
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Reconcile the plan's runtimes against what the restart phase DID.
|
||||
|
||||
#91277 Phase 2 (restart via declared mechanism): the platform restart
|
||||
branches each re-discover their own targets, so a runtime the plan saw
|
||||
can be missed entirely with no signal. This cross-checks every planned
|
||||
runtime against the phase's bookkeeping and returns one outcome row per
|
||||
runtime::
|
||||
|
||||
{"kind", "profile", "pid", "mechanism", "outcome"}
|
||||
|
||||
outcome: ``restarted`` (service restarted / profile relaunched /
|
||||
handed to external supervisor), ``stopped`` (pid killed, watcher or
|
||||
operator relaunches), ``failed`` (in the phase's failed/stale list) or
|
||||
``unaccounted`` — the plan saw it and NO bookkeeping mentions it: the
|
||||
blind-spot tripwire (same philosophy as the fleet matrix's DOWN row).
|
||||
Never raises; on any probe error returns what it has.
|
||||
|
||||
Serve/dashboard runtimes are reconciled in their OWN vocabulary
|
||||
(#100479): a ``hermes-serve*``/``hermes-dashboard*`` unit, a killed
|
||||
PID, or — when the caller passes ``stale_serve_pids`` (the
|
||||
``(pid, create_time)``-verified survivor probe,
|
||||
:func:`hermes_cli.update_abort_recovery._surviving_pre_update_serve_runtimes`)
|
||||
— liveness: a pre-update serve whose incarnation is gone was replaced
|
||||
(unit restart, dashboard cleanup respawn, Desktop respawn) and counts as
|
||||
``restarted``; one still alive is ``unaccounted``. They never borrow the
|
||||
gateway's outcome: ``relaunched_profiles`` and ``hermes-gateway*`` name a
|
||||
different process that shares the profile, nothing more. Without the
|
||||
probe result, an untouched serve stays ``unaccounted`` (fail closed).
|
||||
The platform restart branches each re-discover their own targets, so a runtime the plan saw can
|
||||
be missed with no signal. Returns one ``{kind, profile, pid, mechanism, outcome}`` row per
|
||||
planned runtime; outcome is ``restarted``, ``stopped``, ``failed`` or ``unaccounted`` (no
|
||||
bookkeeping mentions it — the blind-spot tripwire). Never raises. Serve/dashboard runtimes are
|
||||
reconciled in their OWN vocabulary and never borrow the gateway's outcome: with
|
||||
``stale_serve_pids`` a pre-update serve whose incarnation is gone counts as ``restarted``, one
|
||||
still alive is ``unaccounted``; without the probe an untouched serve stays ``unaccounted``.
|
||||
"""
|
||||
outcomes: list[dict[str, Any]] = []
|
||||
try:
|
||||
@@ -514,69 +441,51 @@ def match_runtime_outcomes(
|
||||
relaunched = set(relaunched_profiles or [])
|
||||
external = set(externally_supervised_profiles or [])
|
||||
killed = {int(p) for p in (killed_pids or set())}
|
||||
stale_serves = (
|
||||
{int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None
|
||||
)
|
||||
stale_serves = {int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None
|
||||
|
||||
for runtime in plan.runtimes:
|
||||
r = runtime if isinstance(runtime, RuntimeRecord) else None
|
||||
if r is None:
|
||||
continue
|
||||
if r.kind in _SERVE_KINDS:
|
||||
outcomes.append(
|
||||
{
|
||||
"kind": r.kind,
|
||||
"profile": r.profile,
|
||||
"pid": r.pid,
|
||||
"mechanism": r.restart_via,
|
||||
"outcome": _serve_runtime_outcome(
|
||||
r,
|
||||
killed=killed,
|
||||
failed_set=failed_set,
|
||||
restarted_set=restarted_set,
|
||||
stale_serves=stale_serves,
|
||||
),
|
||||
}
|
||||
)
|
||||
continue
|
||||
outcome = "unaccounted"
|
||||
def _gateway_names(r: RuntimeRecord, names: set) -> bool:
|
||||
# The bare "hermes-gateway" unit name is gateway-specific: a
|
||||
# serve/dashboard runtime that merely shares the default
|
||||
# profile is a different process the gateway restart never
|
||||
# touched, and must not borrow its outcome (#100479).
|
||||
if r.profile in relaunched or r.profile in external:
|
||||
return any(
|
||||
r.profile in name
|
||||
or (
|
||||
r.kind == "gateway"
|
||||
and r.profile == "default"
|
||||
and "hermes-gateway" in name
|
||||
)
|
||||
for name in names
|
||||
)
|
||||
|
||||
for r in plan.runtimes:
|
||||
if not isinstance(r, RuntimeRecord):
|
||||
continue
|
||||
if r.kind in _SERVE_KINDS:
|
||||
outcome = _serve_runtime_outcome(
|
||||
r,
|
||||
killed=killed,
|
||||
failed_set=failed_set,
|
||||
restarted_set=restarted_set,
|
||||
stale_serves=stale_serves,
|
||||
)
|
||||
elif r.profile in relaunched or r.profile in external:
|
||||
outcome = "restarted"
|
||||
elif r.pid is not None and r.pid in killed:
|
||||
outcome = "stopped"
|
||||
elif any(
|
||||
r.profile in unit
|
||||
or (
|
||||
r.kind == "gateway"
|
||||
and r.profile == "default"
|
||||
and "hermes-gateway" in unit
|
||||
)
|
||||
for unit in failed_set
|
||||
):
|
||||
elif _gateway_names(r, failed_set):
|
||||
outcome = "failed"
|
||||
elif any(
|
||||
r.profile in svc
|
||||
or (
|
||||
r.kind == "gateway"
|
||||
and r.profile == "default"
|
||||
and "hermes-gateway" in svc
|
||||
)
|
||||
for svc in restarted_set
|
||||
):
|
||||
elif _gateway_names(r, restarted_set):
|
||||
outcome = "restarted"
|
||||
outcomes.append(
|
||||
{
|
||||
"kind": r.kind,
|
||||
"profile": r.profile,
|
||||
"pid": r.pid,
|
||||
"mechanism": r.restart_via,
|
||||
"outcome": outcome,
|
||||
}
|
||||
)
|
||||
else:
|
||||
outcome = "unaccounted"
|
||||
outcomes.append({
|
||||
"kind": r.kind,
|
||||
"profile": r.profile,
|
||||
"pid": r.pid,
|
||||
"mechanism": r.restart_via,
|
||||
"outcome": outcome,
|
||||
})
|
||||
except Exception as exc:
|
||||
logger.debug("Runtime-outcome reconciliation failed: %s", exc)
|
||||
return outcomes
|
||||
@@ -585,9 +494,8 @@ def match_runtime_outcomes(
|
||||
def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool:
|
||||
"""Print a loud warning for runtimes the restart phase never touched.
|
||||
|
||||
Returns True when at least one planned runtime is unaccounted — the
|
||||
caller escalates exactly like a STALE/DOWN fleet row (exit 1): a runtime
|
||||
the plan promised to restart, silently missed, is the class this phase
|
||||
Returns True when at least one planned runtime is unaccounted; the caller escalates like a
|
||||
STALE/DOWN fleet row (exit 1) — a promised restart silently missed is the class this phase
|
||||
exists to kill.
|
||||
"""
|
||||
missed = [o for o in outcomes if o.get("outcome") == "unaccounted"]
|
||||
@@ -596,10 +504,7 @@ def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool:
|
||||
print()
|
||||
print(" ⚠ Planned runtimes the restart phase never touched:")
|
||||
for o in missed:
|
||||
print(
|
||||
f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']}"
|
||||
f" — planned mechanism: {o['mechanism']}"
|
||||
)
|
||||
print(f" ✗ {o['kind']} [{o['profile']}] pid {o['pid']} — planned mechanism: {o['mechanism']}")
|
||||
print(" Restart them manually, then verify:")
|
||||
if any(o.get("kind") not in _SERVE_KINDS for o in missed):
|
||||
print(" hermes gateway restart # active profile")
|
||||
|
||||
+33
-81
@@ -1,51 +1,12 @@
|
||||
"""Cross-process mutual exclusion for in-flight Hermes updates.
|
||||
|
||||
Three different surfaces can start an update of the same install tree:
|
||||
Until now only the Tauri updater published an "update in progress" marker (``UpdateMarkerGuard`` in
|
||||
``apps/bootstrap-installer/src-tauri/src/update.rs``), and only the Electron desktop consumed it
|
||||
(``electron/update-marker.ts``, to gate local backend startup).
|
||||
|
||||
* ``hermes update`` from a terminal,
|
||||
* the dashboard's Update button (``POST /api/hermes/update`` →
|
||||
``_spawn_hermes_action(["update"])``, detached),
|
||||
* the desktop's Update button, which hands off to the Tauri
|
||||
``hermes-setup --update`` and, on its failure screen, to install-mode
|
||||
bootstrap (``install.ps1`` / ``install.sh``).
|
||||
|
||||
Until now only the Tauri updater published an "update in progress" marker
|
||||
(``UpdateMarkerGuard`` in ``apps/bootstrap-installer/src-tauri/src/update.rs``),
|
||||
and only the Electron desktop consumed it (``electron/update-marker.ts``, to
|
||||
gate local backend startup). Nothing stopped two *updaters* from running at
|
||||
once — so a dashboard-spawned ``hermes update`` and an installer-driven
|
||||
``git checkout`` could mutate the same checkout concurrently, rewriting source
|
||||
under a live interpreter and leaving the tree half-updated.
|
||||
|
||||
This module makes that same marker the single lock for **all** update
|
||||
entrypoints instead of adding a fourth mechanism. Format and location are
|
||||
unchanged and remain byte-compatible with the Rust and Electron readers:
|
||||
|
||||
<HERMES_HOME>/.hermes-update-in-progress body: "<pid>\\n<started_at_unix>"
|
||||
|
||||
A marker only counts as a live update when its pid is alive AND it is younger
|
||||
than :data:`UPDATE_MARKER_MAX_AGE_MS` — mirroring ``readLiveUpdateMarker`` so a
|
||||
crashed updater self-heals instead of wedging every future update. A stale
|
||||
marker is removed on read by whoever notices it first.
|
||||
|
||||
One layering wrinkle: the Tauri updater holds this marker for its WHOLE run and
|
||||
then spawns ``hermes update`` as a child stage. Without a handoff the child
|
||||
sees its own parent's live marker and refuses — the GUI update deadlocks
|
||||
against itself on every attempt ("Hermes is still running", retry forever).
|
||||
Two mechanisms recognize the orchestrating parent, and either suffices:
|
||||
|
||||
* The updater exports :data:`HANDOFF_PID_ENV` naming its own pid, and
|
||||
``acquire`` treats a live holder matching that pid as the lock we are
|
||||
already running under. The env var alone grants nothing: the pid must also
|
||||
be the live marker owner, so a stale or forged value cannot bypass the lock.
|
||||
* A live holder that is a *process ancestor* of ours is likewise our own
|
||||
orchestrator. This is the load-bearing path for the fleet: the staged
|
||||
``hermes-setup`` binary under ``~/.hermes`` is only refreshed by a full
|
||||
installer run (``copy_self_to_hermes_home`` deliberately no-ops during
|
||||
``--update``), so every desktop whose staged updater predates the
|
||||
HANDOFF_PID_ENV export runs an old parent against a new child. Without the
|
||||
ancestry check those users get exit 2 ("Hermes is still running") on every
|
||||
GUI update forever, with no Hermes process actually running.
|
||||
This module makes that same marker the single lock for **all** update entrypoints instead of adding
|
||||
a fourth mechanism. Format and location are unchanged and remain byte-compatible with the Rust and
|
||||
Electron readers:
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -85,10 +46,10 @@ UPDATE_EXIT_CONCURRENT = 2
|
||||
def update_marker_path() -> Path:
|
||||
"""Path of the shared update marker.
|
||||
|
||||
Uses the *process* Hermes home (never the context-local profile override):
|
||||
the Rust updater resolves ``$HERMES_HOME`` or the platform default, and the
|
||||
desktop pins that same value into the updater's env. A profile-scoped path
|
||||
here would put the lock somewhere the other two owners never look.
|
||||
Uses the *process* Hermes home (never the context-local profile override): the Rust updater
|
||||
resolves ``$HERMES_HOME`` or the platform default, and the desktop pins that same value into the
|
||||
updater's env. A profile-scoped path here would put the lock somewhere the other two owners
|
||||
never look.
|
||||
"""
|
||||
from hermes_constants import get_process_hermes_home
|
||||
|
||||
@@ -98,15 +59,12 @@ def update_marker_path() -> Path:
|
||||
def _pid_alive(pid: int) -> bool:
|
||||
"""True when a process with ``pid`` currently exists.
|
||||
|
||||
Delegates to :func:`gateway.status._pid_exists`, the project's existing
|
||||
no-kill probe. Do NOT hand-roll this with ``os.kill(pid, 0)``: on Windows
|
||||
that is not a no-op — CPython routes ``sig=0`` to
|
||||
``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console
|
||||
process group (bpo-14484). A liveness check that killed the updater it was
|
||||
asking about would be a spectacular way to fix a concurrency bug.
|
||||
Delegates to :func:`gateway.status._pid_exists`, the project's existing no-kill probe. Do NOT
|
||||
hand-roll this with ``os.kill(pid, 0)``: on Windows that is not a no-op — CPython routes
|
||||
``sig=0`` to ``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console process
|
||||
group (bpo-14484).
|
||||
|
||||
Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge
|
||||
the lock forever.
|
||||
Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge the lock forever.
|
||||
"""
|
||||
if pid <= 0:
|
||||
return False
|
||||
@@ -124,8 +82,8 @@ def _pid_alive(pid: int) -> bool:
|
||||
def _handoff_pid() -> int | None:
|
||||
"""Pid of the orchestrating updater that spawned us, if any.
|
||||
|
||||
Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent —
|
||||
a broken handoff must fall back to the normal refusal, never crash.
|
||||
Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent — a broken handoff must fall
|
||||
back to the normal refusal, never crash.
|
||||
"""
|
||||
raw = os.environ.get(HANDOFF_PID_ENV, "").strip()
|
||||
if not raw:
|
||||
@@ -140,14 +98,12 @@ def _handoff_pid() -> int | None:
|
||||
def _is_ancestor_pid(pid: int) -> bool:
|
||||
"""True when ``pid`` is a live ancestor (parent chain) of this process.
|
||||
|
||||
The orchestrating updater spawns ``hermes update`` as a (grand)child, so a
|
||||
live marker owned by one of our ancestors can only be the claim we are
|
||||
already running under — an unrelated concurrent updater is never in our
|
||||
parent chain. This heals the fleet of staged ``hermes-setup`` binaries
|
||||
that predate the HANDOFF_PID_ENV export and can never send it.
|
||||
The orchestrating updater spawns ``hermes update`` as a (grand)child, so a live marker owned by
|
||||
one of our ancestors can only be the claim we are already running under — an unrelated
|
||||
concurrent updater is never in our parent chain.
|
||||
|
||||
Never includes our own pid, and any failure counts as "not an ancestor":
|
||||
an unprovable ancestry must fall back to the normal refusal.
|
||||
Never includes our own pid, and any failure counts as "not an ancestor": an unprovable ancestry
|
||||
must fall back to the normal refusal.
|
||||
"""
|
||||
if pid <= 0:
|
||||
return False
|
||||
@@ -171,10 +127,9 @@ class UpdateHolder:
|
||||
def read_live_update(*, path: Path | None = None) -> UpdateHolder | None:
|
||||
"""Return the live update holding the lock, or ``None``.
|
||||
|
||||
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent,
|
||||
unreadable, malformed, dead-pid, and past-the-ceiling all mean "no live
|
||||
update", and a stale marker file is deleted so it can't strand future runs.
|
||||
Never raises.
|
||||
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent, unreadable,
|
||||
malformed, dead-pid, and past-the-ceiling all mean "no live update", and a stale marker file is
|
||||
deleted so it can't strand future runs. Never raises.
|
||||
"""
|
||||
marker = path or update_marker_path()
|
||||
try:
|
||||
@@ -220,11 +175,10 @@ def describe_holder(holder: UpdateHolder) -> str:
|
||||
class UpdateLock:
|
||||
"""Context manager owning the shared update marker for this process.
|
||||
|
||||
``acquired`` is False when another live update already holds it — callers
|
||||
decide whether that's a hard refusal (CLI/dashboard) or a wait. Releasing
|
||||
only removes the marker when *we* still own it, so a marker rewritten by a
|
||||
handoff partner (the Tauri updater overwrites it with its own pid) is never
|
||||
deleted out from under its new owner.
|
||||
``acquired`` is False when another live update holds it; callers decide between hard refusal
|
||||
(CLI/dashboard) and waiting. Release only removes the marker when *we* still own it, so a marker
|
||||
rewritten by a handoff partner (the Tauri updater writes its own pid) is never deleted from
|
||||
under its new owner.
|
||||
"""
|
||||
|
||||
def __init__(self, *, path: Path | None = None) -> None:
|
||||
@@ -235,12 +189,10 @@ class UpdateLock:
|
||||
def acquire(self) -> bool:
|
||||
"""Claim the lock. Returns False (and sets ``holder``) if it's taken.
|
||||
|
||||
A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is a
|
||||
process ancestor of ours — is our own orchestrating parent (the Tauri
|
||||
updater spawning `hermes update` as a stage): we run under ITS claim
|
||||
rather than refusing or re-writing the marker, and ``release`` leaves
|
||||
the parent's marker untouched. The ancestry path exists because staged
|
||||
updaters older than the HANDOFF_PID_ENV export never send the env var.
|
||||
A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is an ancestor of ours — is our
|
||||
own orchestrating parent (e.g. the Tauri updater staging ``hermes update``): run under ITS
|
||||
claim and leave its marker untouched on release. The ancestry path covers staged updaters
|
||||
older than the env-var export.
|
||||
"""
|
||||
existing = read_live_update(path=self.path)
|
||||
if existing is not None:
|
||||
|
||||
+87
-148
@@ -1,29 +1,10 @@
|
||||
"""Structured update receipts + post-update fleet version verification.
|
||||
|
||||
Phase 1 of the fleet-update reliability plan (#91277): the updater must
|
||||
*prove* its outcome instead of assuming it.
|
||||
Phase 1 of the fleet-update reliability plan (#91277): the updater must *prove* its outcome instead
|
||||
of assuming it.
|
||||
|
||||
Two additive capabilities, both designed so a failure inside them can never
|
||||
break an update (every public entry point is exception-swallowing):
|
||||
|
||||
1. **Update receipt** — a machine-readable JSON record of what one
|
||||
``hermes update`` run discovered, did, skipped (and why), written to
|
||||
``<HERMES_HOME>/logs/update_receipts/``. Silent-failure classes this
|
||||
makes visible: #88848 (helper died after "success" printed), #74973
|
||||
(restart silently skipped), #85753 (restart phase never ran), #81193
|
||||
(desktop shows failure for a successful update).
|
||||
|
||||
2. **Fleet version verification** — after the restart phase, read every
|
||||
profile's ``gateway_state.json``, compare each live gateway's stamped
|
||||
``code_sha`` (written by ``gateway/status.py`` on every runtime-status
|
||||
write) against the freshly-updated checkout's HEAD, and print a fleet
|
||||
version matrix. Mixed-version fleets (#88654, #69754, #77553, #56717)
|
||||
become a loud, actionable report instead of a latent state.
|
||||
|
||||
Deployment-kind awareness (docker/image-managed installs) rides on
|
||||
``hermes_cli.build_info.get_code_identity()``: an image build reports
|
||||
``source="build-file"`` and the receipt records that the install is not
|
||||
in-place updatable.
|
||||
Two additive capabilities, both designed so a failure inside them can never break an update (every
|
||||
public entry point is exception-swallowing):
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -53,6 +34,18 @@ def _utc_now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def _str_records(entries: Any, keys: tuple[str, ...], *, pid: bool = False) -> list[dict[str, Any]]:
|
||||
"""Dict entries reduced to stringified ``keys`` (plus an int ``pid`` first when requested)."""
|
||||
records = []
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
record: dict[str, Any] = {"pid": int(entry.get("pid", 0) or 0)} if pid else {}
|
||||
record.update({key: str(entry.get(key, "")) for key in keys})
|
||||
records.append(record)
|
||||
return records
|
||||
|
||||
|
||||
class UpdateReceipt:
|
||||
"""Collects the observable facts of one ``hermes update`` run."""
|
||||
|
||||
@@ -122,16 +115,10 @@ class UpdateReceipt:
|
||||
key: [str(profile) for profile in fresh_recovery.get(key, [])]
|
||||
for key in ("requested", "verified", "relaunch_attempted", "failed")
|
||||
}
|
||||
persisted["skipped"] = [
|
||||
{
|
||||
"profile": str(entry.get("profile", "")),
|
||||
"kind": str(entry.get("kind", "")),
|
||||
"supervisor": str(entry.get("supervisor", "")),
|
||||
"reason": str(entry.get("reason", "")),
|
||||
}
|
||||
for entry in fresh_recovery.get("skipped", [])
|
||||
if isinstance(entry, dict)
|
||||
]
|
||||
persisted["skipped"] = _str_records(
|
||||
fresh_recovery.get("skipped", []),
|
||||
("profile", "kind", "supervisor", "reason"),
|
||||
)
|
||||
# Serve/dashboard coverage (#92145). ``hermes serve`` hosts
|
||||
# tui_gateway and is not a gateway profile, so neither the
|
||||
# per-profile buckets above nor the fleet-version matrix can
|
||||
@@ -143,16 +130,11 @@ class UpdateReceipt:
|
||||
key: [str(unit) for unit in (serve_units.get(key) or [])]
|
||||
for key in ("verified", "failed")
|
||||
}
|
||||
persisted["stale_runtimes"] = [
|
||||
{
|
||||
"pid": int(entry.get("pid", 0) or 0),
|
||||
"kind": str(entry.get("kind", "")),
|
||||
"profile": str(entry.get("profile", "")),
|
||||
"supervisor": str(entry.get("supervisor", "")),
|
||||
}
|
||||
for entry in fresh_recovery.get("stale_runtimes", [])
|
||||
if isinstance(entry, dict)
|
||||
]
|
||||
persisted["stale_runtimes"] = _str_records(
|
||||
fresh_recovery.get("stale_runtimes", []),
|
||||
("kind", "profile", "supervisor"),
|
||||
pid=True,
|
||||
)
|
||||
result["fresh_recovery"] = persisted
|
||||
self.data["gateway_restart"] = result
|
||||
|
||||
@@ -183,31 +165,28 @@ def begin_update_receipt() -> None:
|
||||
_current = None
|
||||
|
||||
|
||||
def record_step(name: str, ok: bool, detail: str = "") -> None:
|
||||
"""Record one update step outcome. No-op when no receipt is active."""
|
||||
def _record(method: str, what: str, *args: Any, **kwargs: Any) -> None:
|
||||
"""Invoke ``method`` on the active receipt; no-op when none, never raises."""
|
||||
try:
|
||||
if _current is not None:
|
||||
_current.step(name, ok, detail)
|
||||
getattr(_current, method)(*args, **kwargs)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug("Could not record update step %s: %s", name, exc)
|
||||
logger.debug("Could not record %s: %s", what, exc)
|
||||
|
||||
|
||||
def record_step(name: str, ok: bool, detail: str = "") -> None:
|
||||
"""Record one update step outcome. No-op when no receipt is active."""
|
||||
_record("step", f"update step {name}", name, ok, detail)
|
||||
|
||||
|
||||
def record_skip(name: str, reason: str) -> None:
|
||||
"""Record a skipped step WITH the reason it was skipped."""
|
||||
try:
|
||||
if _current is not None:
|
||||
_current.skip(name, reason)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug("Could not record update skip %s: %s", name, exc)
|
||||
_record("skip", f"update skip {name}", name, reason)
|
||||
|
||||
|
||||
def record_gateway_restart(**kwargs: Any) -> None:
|
||||
"""Record the gateway restart phase outcome (see UpdateReceipt)."""
|
||||
try:
|
||||
if _current is not None:
|
||||
_current.gateway_restart_result(**kwargs)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug("Could not record gateway restart result: %s", exc)
|
||||
_record("gateway_restart_result", "gateway restart result", **kwargs)
|
||||
|
||||
|
||||
def finalize_update_receipt(
|
||||
@@ -215,10 +194,9 @@ def finalize_update_receipt(
|
||||
) -> Optional[Path]:
|
||||
"""Finalize + persist the receipt. Returns the written path or None.
|
||||
|
||||
``outcome`` is one of ``success`` / ``partial`` / ``failed`` /
|
||||
``refused``. Exactly-once by construction: the module singleton is
|
||||
popped first, so a second call (e.g. the command-boundary safety net
|
||||
after an inner path already finalized) is a no-op returning None.
|
||||
``outcome`` is one of ``success`` / ``partial`` / ``failed`` / ``refused``. Exactly-once by
|
||||
construction: the module singleton is popped first, so a second call (e.g. the command-boundary
|
||||
safety net after an inner path already finalized) is a no-op returning None.
|
||||
"""
|
||||
global _current
|
||||
receipt = _current
|
||||
@@ -235,15 +213,11 @@ def finalize_update_receipt(
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
stamp = time.strftime("%Y%m%d_%H%M%S")
|
||||
path = directory / f"update_{stamp}_{os.getpid()}.json"
|
||||
path.write_text(
|
||||
json.dumps(receipt.data, indent=2, default=str), encoding="utf-8"
|
||||
)
|
||||
body = json.dumps(receipt.data, indent=2, default=str)
|
||||
path.write_text(body, encoding="utf-8")
|
||||
# Stable pointer for the dashboard/desktop: latest receipt.
|
||||
latest = directory / "latest.json"
|
||||
try:
|
||||
latest.write_text(
|
||||
json.dumps(receipt.data, indent=2, default=str), encoding="utf-8"
|
||||
)
|
||||
(directory / "latest.json").write_text(body, encoding="utf-8")
|
||||
except OSError:
|
||||
pass
|
||||
_prune_old_receipts(directory)
|
||||
@@ -258,19 +232,12 @@ def finalize_pending_update_receipt(
|
||||
) -> Optional[Path]:
|
||||
"""Command-boundary safety net: persist a still-open receipt, if any.
|
||||
|
||||
``hermes update`` has many early-termination paths (Windows
|
||||
concurrent-instance preflight, venv-holder refusal, head-pinned no-op,
|
||||
fetch failure — all ``sys.exit``) that predate the inner finalize
|
||||
call sites. Any receipt still open when the update COMMAND unwinds is
|
||||
finalized here so every post-begin run leaves a record — the
|
||||
refused/failed runs are exactly the ones a receipt matters most for
|
||||
(review on #91283). No-op when no receipt is open (the inner paths
|
||||
already finalized — exactly-once via the popped singleton) or when
|
||||
recording was never started. Never raises.
|
||||
|
||||
Outcome mapping: exit 0/None → ``success`` (a path that completed
|
||||
without an explicit inner finalize), exit 2 → ``refused`` (the
|
||||
updater's preflight-refusal convention), anything else → ``failed``.
|
||||
``hermes update`` has many early ``sys.exit`` paths (preflight refusals, venv-holder
|
||||
refusal, fetch failure) predating the inner finalize calls; any receipt still open when the
|
||||
command unwinds is finalized here so refused/failed runs — where a receipt matters most —
|
||||
leave a record. No-op when nothing is open (inner paths finalize exactly-once via the popped
|
||||
singleton). Never raises. Exit 0/None → ``success``, exit 2 → ``refused`` (preflight
|
||||
convention), else → ``failed``.
|
||||
"""
|
||||
if _current is None:
|
||||
return None
|
||||
@@ -280,12 +247,11 @@ def finalize_pending_update_receipt(
|
||||
outcome = "refused"
|
||||
else:
|
||||
outcome = "failed"
|
||||
try:
|
||||
receipt = _current
|
||||
if receipt is not None and exit_code is not None:
|
||||
receipt.data["exit_code"] = int(exit_code)
|
||||
except Exception:
|
||||
pass
|
||||
if exit_code is not None:
|
||||
try:
|
||||
_current.data["exit_code"] = int(exit_code)
|
||||
except Exception:
|
||||
pass
|
||||
return finalize_update_receipt(outcome, stop_reason=stop_reason)
|
||||
|
||||
|
||||
@@ -321,35 +287,31 @@ def read_latest_receipt() -> Optional[dict[str, Any]]:
|
||||
# Fleet version verification
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _sha_state(code_sha: Any, expected_sha: Any) -> str:
|
||||
if not code_sha or not expected_sha:
|
||||
return "unknown"
|
||||
return "current" if str(code_sha) == str(expected_sha) else "stale"
|
||||
|
||||
|
||||
def _fleet_row(profile: str, pid: int, code_sha: Any, code_version: Any, expected_sha: Any) -> dict[str, Any]:
|
||||
return {
|
||||
"profile": profile,
|
||||
"pid": pid,
|
||||
"code_sha": str(code_sha) if code_sha else None,
|
||||
"code_version": code_version,
|
||||
"state": _sha_state(code_sha, expected_sha),
|
||||
}
|
||||
|
||||
|
||||
def collect_fleet_versions(
|
||||
*, pre_restart_pids: Optional[list[int]] = None
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Snapshot every profile's gateway code identity vs. the current tree.
|
||||
|
||||
Returns one entry per profile home that has a ``gateway_state.json``
|
||||
describing a gateway that is live — or that SHOULD be live::
|
||||
|
||||
{"profile": str, "pid": int, "code_sha": str|None,
|
||||
"code_version": str|None, "state": "current"|"stale"|"unknown"|"down"}
|
||||
|
||||
``stale`` — gateway stamped a code_sha that differs from the updated
|
||||
checkout's HEAD (it is still serving pre-update modules).
|
||||
``unknown`` — gateway predates the code-identity stamp (started before
|
||||
this feature landed) or identity could not be resolved.
|
||||
``down`` — the gateway was ALIVE when this update started
|
||||
(``pre_restart_pids``), its runtime status still says
|
||||
running, but the PID is dead and no successor rewrote the
|
||||
record: the restart phase stopped it and nothing came
|
||||
back. Without this row a killed-and-never-replaced gateway
|
||||
produced NO entry at all and the matrix passed silently
|
||||
(Phase-1 verification gap, #88848/#74973 class).
|
||||
|
||||
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` —
|
||||
a stale state file from a long-dead gateway (machine reboot, manual
|
||||
kill weeks ago) must NOT fail every future update. Callers that don't
|
||||
have a pre-restart snapshot (``None``/empty) get the historical
|
||||
behavior: dead PIDs are skipped.
|
||||
Never raises; a probe failure yields an empty list.
|
||||
Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a
|
||||
long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update.
|
||||
Callers that don't have a pre-restart snapshot (``None``/empty) get the historical behavior:
|
||||
dead PIDs are skipped.
|
||||
"""
|
||||
# Runtime-status states that mean "this record does not describe a
|
||||
# gateway that should be running now" — no down row for these.
|
||||
@@ -399,23 +361,12 @@ def collect_fleet_versions(
|
||||
except (TypeError, ValueError):
|
||||
pid = None
|
||||
if pid is not None:
|
||||
code_sha = identity.get("code_sha")
|
||||
if not code_sha or not expected_sha:
|
||||
state = "unknown"
|
||||
elif str(code_sha) == str(expected_sha):
|
||||
state = "current"
|
||||
else:
|
||||
state = "stale"
|
||||
results.append(
|
||||
{
|
||||
"profile": profile,
|
||||
"pid": pid,
|
||||
"code_sha": str(code_sha) if code_sha else None,
|
||||
"code_version": identity.get("code_version"),
|
||||
"state": state,
|
||||
"source": "socket",
|
||||
}
|
||||
row = _fleet_row(
|
||||
profile, pid, identity.get("code_sha"),
|
||||
identity.get("code_version"), expected_sha,
|
||||
)
|
||||
row["source"] = "socket"
|
||||
results.append(row)
|
||||
continue
|
||||
status_path = home / "gateway_state.json"
|
||||
record = read_runtime_status(status_path)
|
||||
@@ -458,21 +409,11 @@ def collect_fleet_versions(
|
||||
}
|
||||
)
|
||||
continue
|
||||
code_sha = record.get("code_sha")
|
||||
if not code_sha or not expected_sha:
|
||||
state = "unknown"
|
||||
elif str(code_sha) == str(expected_sha):
|
||||
state = "current"
|
||||
else:
|
||||
state = "stale"
|
||||
results.append(
|
||||
{
|
||||
"profile": profile,
|
||||
"pid": pid,
|
||||
"code_sha": str(code_sha) if code_sha else None,
|
||||
"code_version": record.get("code_version"),
|
||||
"state": state,
|
||||
}
|
||||
_fleet_row(
|
||||
profile, pid, record.get("code_sha"),
|
||||
record.get("code_version"), expected_sha,
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Fleet version probe failed: %s", exc)
|
||||
@@ -482,13 +423,11 @@ def collect_fleet_versions(
|
||||
def print_fleet_version_matrix(fleet: list[dict[str, Any]]) -> bool:
|
||||
"""Print the post-update fleet version matrix.
|
||||
|
||||
Returns True when at least one gateway is provably stale (still
|
||||
serving pre-update code) OR provably down (was running, killed by the
|
||||
restart phase, nothing came back), so the caller can escalate.
|
||||
``unknown`` entries are reported but do NOT fail the update: gateways
|
||||
started before the code-identity stamp existed have no sha to compare,
|
||||
and failing on them would turn this feature's own rollout into a
|
||||
false-positive storm.
|
||||
Returns True when at least one gateway is provably stale (still serving pre-update code) OR
|
||||
provably down (killed by the restart phase, nothing came back), so the caller can escalate.
|
||||
``unknown`` entries are reported but do NOT fail the update: gateways started before the
|
||||
code-identity stamp existed have no sha to compare, and failing them would be a false-
|
||||
positive storm.
|
||||
"""
|
||||
if not fleet:
|
||||
return False
|
||||
|
||||
@@ -70,17 +70,35 @@ _UNIT_SETTLE_ATTEMPTS = 10
|
||||
_UNIT_SETTLE_DELAY = 1.0
|
||||
|
||||
|
||||
def _profile_command(profile: str) -> list[str]:
|
||||
"""Build a parameterized restart command for exactly one profile."""
|
||||
return [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"hermes_cli.main",
|
||||
"-p",
|
||||
profile,
|
||||
"gateway",
|
||||
"restart",
|
||||
]
|
||||
def _run_quiet(
|
||||
run: Callable[..., Any],
|
||||
argv: list[str],
|
||||
*,
|
||||
timeout: int,
|
||||
**extra: Any,
|
||||
) -> Any | None:
|
||||
"""Run ``argv`` capturing text output; ``None`` when it errors or times out."""
|
||||
try:
|
||||
return run(
|
||||
argv,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=timeout,
|
||||
**extra,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
|
||||
|
||||
def _stdout(result: Any) -> str:
|
||||
return (getattr(result, "stdout", "") or "").strip()
|
||||
|
||||
|
||||
def _succeeded(result: Any) -> bool:
|
||||
return result is not None and getattr(result, "returncode", 1) == 0
|
||||
|
||||
|
||||
def _child_environment() -> dict[str, str]:
|
||||
@@ -98,16 +116,7 @@ def _run_profile_restart(
|
||||
run: Callable[..., Any],
|
||||
) -> bool:
|
||||
"""Run one profile restart without inheriting the updater's process state."""
|
||||
kwargs: dict[str, Any] = {
|
||||
"stdin": subprocess.DEVNULL,
|
||||
"capture_output": True,
|
||||
"text": True,
|
||||
"encoding": "utf-8",
|
||||
"errors": "replace",
|
||||
"check": False,
|
||||
"timeout": _PROFILE_RESTART_TIMEOUT,
|
||||
"env": _child_environment(),
|
||||
}
|
||||
kwargs: dict[str, Any] = {"stdin": subprocess.DEVNULL, "env": _child_environment()}
|
||||
if os.name == "nt":
|
||||
kwargs["creationflags"] = (
|
||||
getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
|
||||
@@ -115,58 +124,31 @@ def _run_profile_restart(
|
||||
)
|
||||
else:
|
||||
kwargs["start_new_session"] = True
|
||||
|
||||
try:
|
||||
result = run(_profile_command(profile), **kwargs)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return False
|
||||
return getattr(result, "returncode", 1) == 0
|
||||
argv = [sys.executable, "-m", "hermes_cli.main", "-p", profile, "gateway", "restart"]
|
||||
return _succeeded(_run_quiet(run, argv, timeout=_PROFILE_RESTART_TIMEOUT, **kwargs))
|
||||
|
||||
|
||||
def _systemd_unit_candidates(profile: str) -> tuple[str, ...]:
|
||||
"""Unit names the existing systemd gateway lifecycle produces per profile."""
|
||||
if profile == "default":
|
||||
return (
|
||||
"hermes-gateway.service",
|
||||
"gateway.service",
|
||||
"gateway-default.service",
|
||||
)
|
||||
return (
|
||||
f"hermes-gateway-{profile}.service",
|
||||
f"gateway-{profile}.service",
|
||||
)
|
||||
return ("hermes-gateway.service", "gateway.service", "gateway-default.service")
|
||||
return (f"hermes-gateway-{profile}.service", f"gateway-{profile}.service")
|
||||
|
||||
|
||||
def _systemd_verified_active(profile: str, *, run: Callable[..., Any]) -> bool:
|
||||
"""Return True only when systemd itself reports the profile's unit active.
|
||||
|
||||
This is the observation that separates ``verified`` from
|
||||
``relaunch_attempted``. Any failure here (no ``systemctl``, probe error,
|
||||
unit not ``active``) means we could NOT verify — never that the restart
|
||||
failed.
|
||||
This is the observation that separates ``verified`` from ``relaunch_attempted``. Any failure
|
||||
here (no ``systemctl``, probe error, unit not ``active``) means we could NOT verify — never that
|
||||
the restart failed.
|
||||
"""
|
||||
systemctl = shutil.which("systemctl")
|
||||
if not systemctl:
|
||||
return False
|
||||
for unit in _systemd_unit_candidates(profile):
|
||||
try:
|
||||
result = run(
|
||||
[systemctl, "--user", "is-active", unit],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=_VERIFY_TIMEOUT,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
continue
|
||||
if (
|
||||
getattr(result, "returncode", 1) == 0
|
||||
and (getattr(result, "stdout", "") or "").strip() == "active"
|
||||
):
|
||||
return True
|
||||
return False
|
||||
return any(
|
||||
_unit_is_active([systemctl, "--user"], unit, run=run, require_rc0=True)
|
||||
for unit in _systemd_unit_candidates(profile)
|
||||
)
|
||||
|
||||
|
||||
def restart_profiles(
|
||||
@@ -177,21 +159,13 @@ def restart_profiles(
|
||||
) -> dict[str, list[str]]:
|
||||
"""Restart the supplied profiles and return per-profile terminal results.
|
||||
|
||||
The caller supplies only profiles whose inventory identified a service
|
||||
supervisor. Manual gateways are intentionally excluded before this module
|
||||
is called: killing one without a relaunch authority would turn stale code
|
||||
into an outage.
|
||||
The caller supplies only profiles whose inventory identified a service supervisor.
|
||||
|
||||
A profile only lands in ``verified`` when its supervisor is systemd and
|
||||
``systemctl --user is-active`` independently confirms the unit after the
|
||||
relaunch command succeeded. Every other zero-exit relaunch is reported as
|
||||
``relaunch_attempted`` — the code cannot observe supervisor coverage for
|
||||
those paths and must not claim it.
|
||||
A profile only lands in ``verified`` when its supervisor is systemd and ``systemctl --user is-
|
||||
active`` independently confirms the unit after the relaunch command succeeded.
|
||||
"""
|
||||
supervisors = supervisors or {}
|
||||
normalized = sorted(
|
||||
{profile for profile in profiles if isinstance(profile, str) and profile}
|
||||
)
|
||||
normalized = sorted({profile for profile in profiles if isinstance(profile, str) and profile})
|
||||
verified: list[str] = []
|
||||
relaunch_attempted: list[str] = []
|
||||
failed: list[str] = []
|
||||
@@ -199,31 +173,23 @@ def restart_profiles(
|
||||
if not _run_profile_restart(profile, run=run):
|
||||
failed.append(profile)
|
||||
continue
|
||||
if supervisors.get(profile) == "systemd" and _systemd_verified_active(
|
||||
profile, run=run
|
||||
):
|
||||
if supervisors.get(profile) == "systemd" and _systemd_verified_active(profile, run=run):
|
||||
verified.append(profile)
|
||||
else:
|
||||
relaunch_attempted.append(profile)
|
||||
return {
|
||||
"verified": verified,
|
||||
"relaunch_attempted": relaunch_attempted,
|
||||
"failed": failed,
|
||||
}
|
||||
return {"verified": verified, "relaunch_attempted": relaunch_attempted, "failed": failed}
|
||||
|
||||
|
||||
def _systemctl_scopes() -> list[tuple[str, list[str]]]:
|
||||
"""``systemctl`` invocations for the user and system scopes, or nothing.
|
||||
|
||||
Mirrors the scope pair the in-process restart phase walks. ``systemctl``
|
||||
is resolved through ``shutil.which`` so this module never has to import
|
||||
any Hermes platform helper — importing the freshly pulled tree is exactly
|
||||
what aborted the phase that called us.
|
||||
Mirrors the scope pair the in-process restart phase walks. ``systemctl`` is resolved through
|
||||
``shutil.which`` so this module never has to import any Hermes platform helper — importing the
|
||||
freshly pulled tree is exactly what aborted the phase that called us.
|
||||
|
||||
Each scope is returned with its label because ``hermes-serve.service`` in
|
||||
the user manager and ``hermes-serve.service`` in the system manager are
|
||||
two different processes. Every identity this module produces or consumes
|
||||
stays qualified by that label; the bare unit name is never the key.
|
||||
Each scope is returned with its label because ``hermes-serve.service`` in the user manager and
|
||||
``hermes-serve.service`` in the system manager are two different processes. Every identity this
|
||||
module produces or consumes stays qualified by that label; the bare unit name is never the key.
|
||||
"""
|
||||
systemctl = shutil.which("systemctl")
|
||||
if not systemctl or sys.platform != "linux":
|
||||
@@ -233,83 +199,43 @@ def _systemctl_scopes() -> list[tuple[str, list[str]]]:
|
||||
|
||||
def _listed_serve_units(scope: list[str], *, run: Callable[..., Any]) -> list[str]:
|
||||
"""Serve units systemd knows about in one scope, validated by name."""
|
||||
try:
|
||||
result = run(
|
||||
scope
|
||||
+ [
|
||||
"list-units",
|
||||
_SERVE_UNIT_PATTERN,
|
||||
"--plain",
|
||||
"--no-legend",
|
||||
"--no-pager",
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=_VERIFY_TIMEOUT,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
result = _run_quiet(
|
||||
run,
|
||||
scope + ["list-units", _SERVE_UNIT_PATTERN, "--plain", "--no-legend", "--no-pager"],
|
||||
timeout=_VERIFY_TIMEOUT,
|
||||
)
|
||||
if result is None:
|
||||
return []
|
||||
# The glob is a systemd pattern, not a name gate: `hermes-serve*` also
|
||||
# matches the unrelated `hermes-server.service`. Require the exact
|
||||
# base unit or the hyphenated profile family, same shape as the
|
||||
# in-process phase's own name gate.
|
||||
units: list[str] = []
|
||||
for line in (getattr(result, "stdout", "") or "").splitlines():
|
||||
parts = line.split()
|
||||
if not parts:
|
||||
continue
|
||||
# The glob is a systemd pattern, not a name gate: `hermes-serve*` also
|
||||
# matches the unrelated `hermes-server.service`. Require the exact
|
||||
# base unit or the hyphenated profile family, same shape as the
|
||||
# in-process phase's own name gate.
|
||||
if _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units:
|
||||
if parts and _UNIT_RE.fullmatch(parts[0]) and parts[0] not in units:
|
||||
units.append(parts[0])
|
||||
return units
|
||||
|
||||
|
||||
def _unit_property(
|
||||
scope: list[str], unit: str, prop: str, *, run: Callable[..., Any]
|
||||
) -> str | None:
|
||||
"""One ``systemctl show`` property, or ``None`` when it cannot be read."""
|
||||
try:
|
||||
result = run(
|
||||
scope + ["show", unit, f"--property={prop}", "--value"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=_VERIFY_TIMEOUT,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if getattr(result, "returncode", 1) != 0:
|
||||
return None
|
||||
return (getattr(result, "stdout", "") or "").strip()
|
||||
|
||||
|
||||
def _unit_main_pid(scope: list[str], unit: str, *, run: Callable[..., Any]) -> int:
|
||||
"""The unit's ``MainPID``; ``0`` when absent or unreadable."""
|
||||
raw = _unit_property(scope, unit, "MainPID", run=run)
|
||||
"""The unit's ``MainPID`` via ``systemctl show``; ``0`` when absent or unreadable."""
|
||||
result = _run_quiet(
|
||||
run, scope + ["show", unit, "--property=MainPID", "--value"], timeout=_VERIFY_TIMEOUT
|
||||
)
|
||||
try:
|
||||
return int(raw or 0)
|
||||
return int(_stdout(result) or 0) if _succeeded(result) else 0
|
||||
except ValueError:
|
||||
return 0
|
||||
|
||||
|
||||
def _unit_is_active(scope: list[str], unit: str, *, run: Callable[..., Any]) -> bool:
|
||||
try:
|
||||
result = run(
|
||||
scope + ["is-active", unit],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=_VERIFY_TIMEOUT,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
def _unit_is_active(
|
||||
scope: list[str], unit: str, *, run: Callable[..., Any], require_rc0: bool = False
|
||||
) -> bool:
|
||||
result = _run_quiet(run, scope + ["is-active", unit], timeout=_VERIFY_TIMEOUT)
|
||||
if result is None or (require_rc0 and not _succeeded(result)):
|
||||
return False
|
||||
return (getattr(result, "stdout", "") or "").strip() == "active"
|
||||
return _stdout(result) == "active"
|
||||
|
||||
|
||||
def _serve_unit_replaced(
|
||||
@@ -322,11 +248,9 @@ def _serve_unit_replaced(
|
||||
) -> bool:
|
||||
"""Did the unit come back on a NEW main process?
|
||||
|
||||
``restart`` returning 0 is not evidence: the whole point of #92145 is that
|
||||
a live process can keep serving the pre-update generation while every
|
||||
status command reports success. A changed ``MainPID`` on an ``active``
|
||||
unit is the observation that the old interpreter — and its stale
|
||||
``sys.modules`` — is gone.
|
||||
``restart`` returning 0 is not evidence: a live process can keep serving the pre-update
|
||||
generation while every status command reports success. A changed ``MainPID`` on an ``active``
|
||||
unit is the observation that the old interpreter and its stale ``sys.modules`` are gone.
|
||||
"""
|
||||
for attempt in range(_UNIT_SETTLE_ATTEMPTS):
|
||||
if attempt:
|
||||
@@ -339,9 +263,16 @@ def _serve_unit_replaced(
|
||||
return False
|
||||
|
||||
|
||||
def _qualified(scope_label: str, base: str) -> str:
|
||||
"""The only identity this module reports for a serve unit."""
|
||||
return f"{scope_label}/{base}"
|
||||
def _split_skip_entry(entry: Any) -> tuple[str, str]:
|
||||
"""``(scope_label, base_unit)`` of a skip entry in either the mapping or ``scope/unit`` shape."""
|
||||
if isinstance(entry, Mapping):
|
||||
scope_label = str(entry.get("scope") or "")
|
||||
unit = str(entry.get("unit") or "")
|
||||
else:
|
||||
scope_label, sep, unit = str(entry).partition("/")
|
||||
if not sep:
|
||||
scope_label, unit = "", scope_label
|
||||
return scope_label, unit.removesuffix(".service")
|
||||
|
||||
|
||||
def _normalized_skips(
|
||||
@@ -349,26 +280,13 @@ def _normalized_skips(
|
||||
) -> tuple[set[tuple[str, str]], set[str]]:
|
||||
"""Split already-settled units into scope-qualified and legacy entries.
|
||||
|
||||
Qualified entries (``{"scope": "user", "unit": "hermes-serve"}`` or the
|
||||
equivalent ``"user/hermes-serve"`` string) suppress exactly one process.
|
||||
|
||||
A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH
|
||||
of two same-named processes was settled. It is honoured across both scopes
|
||||
because that is all the information it contains — the caller in this tree
|
||||
always sends the qualified shape, and this branch exists only for a payload
|
||||
written by a pre-update interpreter that had no scope to send.
|
||||
A bare ``"hermes-serve"`` carries no scope and therefore cannot say WHICH of two same-named
|
||||
processes was settled.
|
||||
"""
|
||||
qualified: set[tuple[str, str]] = set()
|
||||
legacy: set[str] = set()
|
||||
for entry in skip_units or ():
|
||||
if isinstance(entry, Mapping):
|
||||
scope_label = str(entry.get("scope") or "")
|
||||
base = str(entry.get("unit") or "").removesuffix(".service")
|
||||
else:
|
||||
scope_label, sep, unit = str(entry).partition("/")
|
||||
if not sep:
|
||||
scope_label, unit = "", scope_label
|
||||
base = unit.removesuffix(".service")
|
||||
scope_label, base = _split_skip_entry(entry)
|
||||
if not base:
|
||||
continue
|
||||
if scope_label in _SCOPE_LABELS:
|
||||
@@ -386,28 +304,9 @@ def restart_serve_units(
|
||||
) -> dict[str, list[str]]:
|
||||
"""Restart every active ``hermes-serve*`` systemd unit from this process.
|
||||
|
||||
``hermes serve`` hosts ``tui_gateway.server`` and is restarted by the
|
||||
in-process phase alongside the gateway units, but it is not a gateway
|
||||
profile: no ``gateway restart`` command reaches it. When the phase aborts
|
||||
part-way — systemd lists ``hermes-gateway.service`` before
|
||||
``hermes-serve.service``, so the gateway is typically already done — the
|
||||
serve unit is the one left holding generation-N modules over a
|
||||
generation-N+1 checkout.
|
||||
|
||||
Units are enumerated from systemd, never from the update inventory. That
|
||||
keeps the relaunch authority requirement structural: a manually launched
|
||||
or Desktop-owned ``hermes serve`` owns no unit and therefore cannot be
|
||||
touched here.
|
||||
|
||||
Every identity in and out of this function is scope-qualified. The user
|
||||
manager and the system manager can each own a ``hermes-serve.service``,
|
||||
and they are different processes: projecting the scope away would let one
|
||||
already-settled unit suppress recovery of the other, and would let one
|
||||
scope's success describe the other's outcome.
|
||||
|
||||
Returns ``{"verified": [...], "failed": [...]}`` whose entries are
|
||||
``<scope>/<base unit>`` (no ``.service`` suffix), e.g.
|
||||
``user/hermes-serve``.
|
||||
Units are enumerated from systemd, never from the update inventory. That keeps the relaunch
|
||||
authority requirement structural: a manually launched or Desktop-owned ``hermes serve`` owns no
|
||||
unit and therefore cannot be touched here.
|
||||
"""
|
||||
skipped_qualified, skipped_legacy = _normalized_skips(skip_units)
|
||||
# (scope, base unit) -> replaced? A unit name can exist in BOTH the user
|
||||
@@ -433,41 +332,28 @@ def restart_serve_units(
|
||||
# remove.
|
||||
outcomes[target] = False
|
||||
continue
|
||||
try:
|
||||
result = run(
|
||||
scope + ["--no-ask-password", "restart", unit],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
check=False,
|
||||
timeout=_UNIT_RESTART_TIMEOUT,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
outcomes[target] = False
|
||||
continue
|
||||
if getattr(result, "returncode", 1) != 0:
|
||||
result = _run_quiet(
|
||||
run, scope + ["--no-ask-password", "restart", unit], timeout=_UNIT_RESTART_TIMEOUT
|
||||
)
|
||||
if not _succeeded(result):
|
||||
# Includes the unprivileged system-scope case. We do not probe
|
||||
# for sudo here: an unverifiable unit must read as failed so
|
||||
# the update stays explicitly incomplete.
|
||||
outcomes[target] = False
|
||||
continue
|
||||
outcomes[target] = _serve_unit_replaced(
|
||||
scope, unit, previous_pid, run=run, sleep=sleep
|
||||
)
|
||||
outcomes[target] = _serve_unit_replaced(scope, unit, previous_pid, run=run, sleep=sleep)
|
||||
# ``<scope>/<unit>`` is the only identity this module reports for a serve unit.
|
||||
return {
|
||||
"verified": sorted(
|
||||
_qualified(*target) for target, ok in outcomes.items() if ok
|
||||
),
|
||||
"failed": sorted(
|
||||
_qualified(*target) for target, ok in outcomes.items() if not ok
|
||||
),
|
||||
"verified": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if ok),
|
||||
"failed": sorted(f"{scope}/{base}" for (scope, base), ok in outcomes.items() if not ok),
|
||||
}
|
||||
|
||||
|
||||
def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
payload = json.load(stream)
|
||||
profiles = payload.get("profiles") if isinstance(payload, dict) else None
|
||||
if not isinstance(payload, dict):
|
||||
payload = {}
|
||||
profiles = payload.get("profiles")
|
||||
if not isinstance(profiles, list):
|
||||
raise ValueError("recovery payload must contain a profiles list")
|
||||
if any(
|
||||
@@ -475,7 +361,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
for profile in profiles
|
||||
):
|
||||
raise ValueError("recovery profiles contain an invalid profile id")
|
||||
raw_supervisors = payload.get("supervisors") if isinstance(payload, dict) else None
|
||||
raw_supervisors = payload.get("supervisors")
|
||||
supervisors: dict[str, str] = {}
|
||||
if raw_supervisors is not None:
|
||||
if not isinstance(raw_supervisors, dict) or any(
|
||||
@@ -487,7 +373,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
):
|
||||
raise ValueError("recovery supervisors map is invalid")
|
||||
supervisors = dict(raw_supervisors)
|
||||
raw_serve = payload.get("serve_units") if isinstance(payload, dict) else None
|
||||
raw_serve = payload.get("serve_units")
|
||||
recover_serve = False
|
||||
skip_units: list[str] = []
|
||||
if raw_serve is not None:
|
||||
@@ -495,9 +381,7 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
raise ValueError("recovery serve_units block is invalid")
|
||||
recover_serve = bool(raw_serve.get("recover"))
|
||||
raw_skip = raw_serve.get("skip") or []
|
||||
if not isinstance(raw_skip, list) or any(
|
||||
not isinstance(entry, (str, dict)) for entry in raw_skip
|
||||
):
|
||||
if not isinstance(raw_skip, list) or any(not isinstance(entry, (str, dict)) for entry in raw_skip):
|
||||
raise ValueError("recovery serve_units skip list is invalid")
|
||||
for entry in raw_skip:
|
||||
# A skip entry names one already-settled process. The qualified
|
||||
@@ -507,14 +391,11 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
# interpreter can still send; it is kept, and read as
|
||||
# scope-agnostic by `restart_serve_units`.
|
||||
if isinstance(entry, dict):
|
||||
scope_label = entry.get("scope")
|
||||
unit = entry.get("unit")
|
||||
if not isinstance(unit, str):
|
||||
if not isinstance(entry.get("unit"), str):
|
||||
raise ValueError("recovery serve_units skip list is invalid")
|
||||
if not isinstance(scope_label, str):
|
||||
scope_label = ""
|
||||
else:
|
||||
scope_label, _, unit = entry.rpartition("/")
|
||||
if not isinstance(entry.get("scope"), str):
|
||||
entry = {"unit": entry["unit"]}
|
||||
scope_label, base = _split_skip_entry(entry)
|
||||
# Only the shapes systemd can actually produce for this family; a
|
||||
# skip entry is a name filter, never a command argument. An
|
||||
# unrecognized scope drops the entry rather than raising: dropping
|
||||
@@ -523,22 +404,15 @@ def _parse_payload(stream) -> tuple[list[str], dict[str, str], bool, list[str]]:
|
||||
# process.
|
||||
if scope_label and scope_label not in _SCOPE_LABELS:
|
||||
continue
|
||||
if not _UNIT_RE.fullmatch(
|
||||
unit if unit.endswith(".service") else f"{unit}.service"
|
||||
):
|
||||
if not _UNIT_RE.fullmatch(f"{base}.service"):
|
||||
continue
|
||||
base = unit.removesuffix(".service")
|
||||
skip_units.append(f"{scope_label}/{base}" if scope_label else base)
|
||||
return profiles, supervisors, recover_serve, skip_units
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument(
|
||||
"--stdin",
|
||||
action="store_true",
|
||||
help=argparse.SUPPRESS,
|
||||
)
|
||||
parser.add_argument("--stdin", action="store_true", help=argparse.SUPPRESS)
|
||||
args = parser.parse_args(argv)
|
||||
if not args.stdin:
|
||||
parser.error("this command is an internal update-recovery entry point")
|
||||
@@ -547,28 +421,20 @@ def main(argv: list[str] | None = None) -> int:
|
||||
profiles, supervisors, recover_serve, skip_units = _parse_payload(sys.stdin)
|
||||
result = restart_profiles(profiles, supervisors=supervisors)
|
||||
result["serve_units"] = (
|
||||
restart_serve_units(skip_units=skip_units)
|
||||
if recover_serve
|
||||
else {"verified": [], "failed": []}
|
||||
restart_serve_units(skip_units=skip_units) if recover_serve else {"verified": [], "failed": []}
|
||||
)
|
||||
except (ValueError, json.JSONDecodeError) as exc:
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"error": str(exc),
|
||||
"verified": [],
|
||||
"relaunch_attempted": [],
|
||||
"failed": [],
|
||||
"serve_units": {"verified": [], "failed": []},
|
||||
}
|
||||
)
|
||||
)
|
||||
print(json.dumps({
|
||||
"error": str(exc),
|
||||
"verified": [],
|
||||
"relaunch_attempted": [],
|
||||
"failed": [],
|
||||
"serve_units": {"verified": [], "failed": []},
|
||||
}))
|
||||
return 2
|
||||
|
||||
print(json.dumps(result, sort_keys=True))
|
||||
if result["failed"] or result["serve_units"]["failed"]:
|
||||
return 1
|
||||
return 0
|
||||
return 1 if result["failed"] or result["serve_units"]["failed"] else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+56
-71
@@ -1,12 +1,8 @@
|
||||
"""``hermes worktree`` — audit and reclaim accumulated git worktrees/branches.
|
||||
|
||||
Attended counterpart of the silent startup pruner (see
|
||||
``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage:
|
||||
|
||||
hermes worktree list # audit: verdict + reason per tree
|
||||
hermes worktree prune # reap safe trees + merged branches
|
||||
hermes worktree prune --dry-run # show the plan, change nothing
|
||||
hermes worktree prune --trees-only / --branches-only
|
||||
hermes worktree list # audit: verdict + reason per tree hermes worktree prune # reap safe trees +
|
||||
merged branches hermes worktree prune --dry-run # show the plan, change nothing hermes worktree
|
||||
prune --trees-only / --branches-only
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -23,9 +19,54 @@ def _repo_root() -> Optional[str]:
|
||||
def _fmt_size(size_mb: Optional[int]) -> str:
|
||||
if size_mb is None:
|
||||
return "?"
|
||||
if size_mb >= 1024:
|
||||
return f"{size_mb / 1024:.1f}G"
|
||||
return f"{size_mb}M"
|
||||
return f"{size_mb / 1024:.1f}G" if size_mb >= 1024 else f"{size_mb}M"
|
||||
|
||||
|
||||
def _list(worktree_gc, repo_root: str, args) -> int:
|
||||
records = worktree_gc.audit_worktrees(repo_root)
|
||||
if not records:
|
||||
print("No worktrees under .worktrees/ — nothing to reclaim.")
|
||||
return 0
|
||||
total_mb = sum(r.size_mb or 0 for r in records)
|
||||
reapable_mb = sum(r.size_mb or 0 for r in records if r.verdict.startswith("reap"))
|
||||
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
|
||||
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
|
||||
print(f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} {r.verdict:13} {r.reason}")
|
||||
print(
|
||||
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
|
||||
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
|
||||
)
|
||||
deletable = [b for b in worktree_gc.audit_branches(repo_root) if b.verdict == "delete"]
|
||||
if deletable:
|
||||
print(f"{len(deletable)} local branch(es) fully merged/patch-equivalent upstream would also be deleted.")
|
||||
return 0
|
||||
|
||||
|
||||
def _prune(worktree_gc, repo_root: str, args) -> int:
|
||||
dry_run = bool(getattr(args, "dry_run", False))
|
||||
actions: list = []
|
||||
if not getattr(args, "branches_only", False):
|
||||
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
|
||||
actions += worktree_gc.reclaim_worktrees(repo_root, dry_run=dry_run, records=tree_records)
|
||||
kept = [r for r in tree_records if r.verdict == "keep"
|
||||
and "kanban" not in r.reason and "in use" not in r.reason]
|
||||
if kept:
|
||||
print(f"Preserved {len(kept)} tree(s) with real work:")
|
||||
for r in kept:
|
||||
print(f" {r.name}: {r.reason}")
|
||||
if not getattr(args, "trees_only", False):
|
||||
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
|
||||
|
||||
if actions:
|
||||
for line in actions:
|
||||
print(f" {line}")
|
||||
print(f"{len(actions)} action(s) {'planned' if dry_run else 'done'}.")
|
||||
else:
|
||||
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
|
||||
return 0
|
||||
|
||||
|
||||
_ACTIONS = {"list": _list, "prune": _prune}
|
||||
|
||||
|
||||
def cmd_worktree(args) -> int:
|
||||
@@ -35,65 +76,9 @@ def cmd_worktree(args) -> int:
|
||||
if not repo_root:
|
||||
print("Not inside a git repository (or pass --repo <path>).")
|
||||
return 1
|
||||
|
||||
action = getattr(args, "worktree_action", None) or "list"
|
||||
|
||||
if action == "list":
|
||||
records = worktree_gc.audit_worktrees(repo_root)
|
||||
if not records:
|
||||
print("No worktrees under .worktrees/ — nothing to reclaim.")
|
||||
return 0
|
||||
total_mb = sum(r.size_mb or 0 for r in records)
|
||||
reapable_mb = sum(
|
||||
r.size_mb or 0 for r in records if r.verdict.startswith("reap")
|
||||
)
|
||||
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
|
||||
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
|
||||
print(
|
||||
f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} "
|
||||
f"{r.verdict:13} {r.reason}"
|
||||
)
|
||||
print(
|
||||
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
|
||||
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
|
||||
)
|
||||
branch_records = worktree_gc.audit_branches(repo_root)
|
||||
deletable = [b for b in branch_records if b.verdict == "delete"]
|
||||
if deletable:
|
||||
print(
|
||||
f"{len(deletable)} local branch(es) fully merged/patch-equivalent "
|
||||
f"upstream would also be deleted."
|
||||
)
|
||||
return 0
|
||||
|
||||
if action == "prune":
|
||||
dry_run = bool(getattr(args, "dry_run", False))
|
||||
trees_only = bool(getattr(args, "trees_only", False))
|
||||
branches_only = bool(getattr(args, "branches_only", False))
|
||||
|
||||
actions: list = []
|
||||
if not branches_only:
|
||||
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
|
||||
actions += worktree_gc.reclaim_worktrees(
|
||||
repo_root, dry_run=dry_run, records=tree_records
|
||||
)
|
||||
kept = [r for r in tree_records if r.verdict == "keep"
|
||||
and "kanban" not in r.reason and "in use" not in r.reason]
|
||||
if kept:
|
||||
print(f"Preserved {len(kept)} tree(s) with real work:")
|
||||
for r in kept:
|
||||
print(f" {r.name}: {r.reason}")
|
||||
if not trees_only:
|
||||
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
|
||||
|
||||
if actions:
|
||||
for line in actions:
|
||||
print(f" {line}")
|
||||
verb = "planned" if dry_run else "done"
|
||||
print(f"{len(actions)} action(s) {verb}.")
|
||||
else:
|
||||
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
|
||||
return 0
|
||||
|
||||
print(f"Unknown worktree action: {action}")
|
||||
return 1
|
||||
handler = _ACTIONS.get(action)
|
||||
if handler is None:
|
||||
print(f"Unknown worktree action: {action}")
|
||||
return 1
|
||||
return handler(worktree_gc, repo_root, args)
|
||||
|
||||
+89
-175
@@ -1,35 +1,11 @@
|
||||
"""On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``).
|
||||
|
||||
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately
|
||||
conservative and silent: it runs before the banner on every ``hermes -w``
|
||||
launch, so it only reaps clean, fully-merged scratch trees past an age tier
|
||||
and preserves everything else. That policy is correct for an unattended
|
||||
startup path — but it means real installs accumulate two kinds of debris the
|
||||
startup pass can never touch:
|
||||
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately conservative and silent: it
|
||||
runs before the banner on every ``hermes -w`` launch, so it only reaps clean, fully-merged scratch
|
||||
trees past an age tier and preserves everything else.
|
||||
|
||||
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts,
|
||||
logs) on an otherwise merged branch — preserved forever by the dirty guard.
|
||||
- **Orphaned local branches** beyond the two auto-generated prefixes the
|
||||
startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port
|
||||
branches, feature branches whose PRs merged months ago. Multi-agent boxes
|
||||
reach hundreds.
|
||||
|
||||
This module is the *attended* counterpart: an explicit, loud, dry-run-first
|
||||
reclaim the user invokes, so it can be more thorough while staying just as
|
||||
safe. Invariants shared with the startup pruner (never violated here either):
|
||||
|
||||
- tracked modifications are NEVER deleted, at any age, in any mode;
|
||||
- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence
|
||||
decides "unique"; shallow repos are deepened bloblessly first so the
|
||||
verdict is trustworthy);
|
||||
- live-locked trees (owning pid alive) are never touched;
|
||||
- a branch is deleted only after its worktree removal succeeded — a failed
|
||||
removal must not orphan reachable commits;
|
||||
- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/``
|
||||
before its tree is reaped, never destroyed.
|
||||
|
||||
Classification primitives are imported from ``cli`` so the two paths can
|
||||
never drift apart on what "dirty", "unpushed", or "merged" means.
|
||||
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts, logs) on an otherwise
|
||||
merged branch — preserved forever by the dirty guard.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -79,11 +55,10 @@ class BranchRecord:
|
||||
def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess:
|
||||
"""Run git, translating timeouts into a nonzero returncode.
|
||||
|
||||
Every verdict in this module fails safe toward "keep" on a nonzero
|
||||
returncode, so a hung/slow git call (large repos make ``git cherry``
|
||||
genuinely slow) must degrade to keep — never crash the whole audit
|
||||
(live-verified failure on a 746MB .git: TimeoutExpired escaped and
|
||||
aborted the branch audit mid-list).
|
||||
Every verdict in this module fails safe toward "keep" on a nonzero returncode, so a hung/slow
|
||||
git call (large repos make ``git cherry`` genuinely slow) must degrade to keep — never crash the
|
||||
whole audit (live-verified failure on a 746MB .git: TimeoutExpired escaped and aborted the
|
||||
branch audit mid-list).
|
||||
"""
|
||||
try:
|
||||
return subprocess.run(
|
||||
@@ -98,42 +73,30 @@ def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess
|
||||
)
|
||||
|
||||
|
||||
def _tree_size_mb(path: Path) -> Optional[int]:
|
||||
def _tree_size_mb(path: Path, timeout: int = 30) -> Optional[int]:
|
||||
"""Cheap directory size via ``du -sm`` — best-effort, None on failure."""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["du", "-sm", str(path)],
|
||||
capture_output=True, text=True, encoding="utf-8",
|
||||
errors="replace", timeout=30,
|
||||
["du", "-sm", str(path)], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
return int(result.stdout.split()[0])
|
||||
return int(result.stdout.split()[0]) if result.returncode == 0 and result.stdout.strip() else None
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _dirty_split(path: str) -> tuple[bool, List[str]]:
|
||||
"""Return (has_tracked_modifications, untracked_paths).
|
||||
|
||||
``git status --porcelain`` counts untracked scratch equally with real
|
||||
edits; the reclaim policy treats them very differently (tracked = real
|
||||
work, untracked = archivable scratch), so split here.
|
||||
``git status --porcelain`` counts untracked scratch the same as real edits, but the reclaim
|
||||
policy treats them differently (tracked = real work, untracked = archivable scratch).
|
||||
"""
|
||||
try:
|
||||
result = _git(["status", "--porcelain"], cwd=path, timeout=10)
|
||||
if result.returncode != 0:
|
||||
return True, [] # fail safe: treat as real work
|
||||
tracked = False
|
||||
untracked: List[str] = []
|
||||
for line in result.stdout.splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
if line.startswith("??"):
|
||||
untracked.append(line[3:].strip())
|
||||
else:
|
||||
tracked = True
|
||||
return tracked, untracked
|
||||
lines = [line for line in result.stdout.splitlines() if line.strip()]
|
||||
untracked = [line[3:].strip() for line in lines if line.startswith("??")]
|
||||
return len(untracked) != len(lines), untracked
|
||||
except Exception:
|
||||
return True, []
|
||||
|
||||
@@ -141,15 +104,11 @@ def _dirty_split(path: str) -> tuple[bool, List[str]]:
|
||||
def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
|
||||
"""Copy untracked files out of a doomed tree. Returns the archive dir.
|
||||
|
||||
Never destroys: on any copy failure the caller must treat the tree as
|
||||
keep. Costs almost nothing and removes the "did I just delete
|
||||
something?" question.
|
||||
Never destroys: on any copy failure the caller must treat the tree as keep. Costs almost nothing
|
||||
and removes the "did I just delete something?" question.
|
||||
"""
|
||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||
dest = (
|
||||
Path.home() / ".hermes" / "archive" / "worktree-prune"
|
||||
/ f"{tree.name}-{stamp}"
|
||||
)
|
||||
dest = Path.home() / ".hermes" / "archive" / "worktree-prune" / f"{tree.name}-{stamp}"
|
||||
try:
|
||||
for rel in untracked:
|
||||
src = tree / rel
|
||||
@@ -167,6 +126,34 @@ def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
|
||||
return None
|
||||
|
||||
|
||||
def _classify_tree(_cli, repo_root: str, entry: Path, merge_cache, remote_heads) -> tuple[str, str, List[str]]:
|
||||
"""Return (verdict, reason, untracked) for one tree under ``.worktrees/``."""
|
||||
path = str(entry)
|
||||
if _KANBAN_RE.match(entry.name):
|
||||
return "keep", "kanban task tree (owned by kanban gc)", []
|
||||
if _cli._worktree_lock_is_live(repo_root, path, timeout=5) == "live":
|
||||
return "keep", "in use by a running hermes session", []
|
||||
tracked_dirty, untracked = _dirty_split(path)
|
||||
if tracked_dirty:
|
||||
return "keep", "uncommitted tracked changes (real work)", []
|
||||
archive_note = f"{len(untracked)} untracked file(s) will be archived"
|
||||
if _cli._worktree_has_unpushed_commits(path, timeout=5) and not _cli._worktree_commits_all_merged_upstream(
|
||||
path, timeout=30, cache=merge_cache, max_ahead=_MAX_CHERRY_AHEAD
|
||||
):
|
||||
# Pushed-branch tier: single-branch fetch refspecs (the managed-install default) leave
|
||||
# pushed PR branches with no refs/remotes/* entry, so `git log HEAD --not --remotes`
|
||||
# reads them as unpushed forever. A local head that EXACTLY matches the remote branch
|
||||
# has nothing origin lacks — the checkout is redundant; only the branch ref stays.
|
||||
if not _cli._worktree_branch_pushed_exact(path, remote_heads, timeout=10):
|
||||
return "keep", "unpushed commits not found upstream", []
|
||||
if untracked:
|
||||
return "reap-keep-branch", f"pushed to origin (open-PR lane); branch kept; {archive_note}", untracked
|
||||
return "reap-keep-branch", "pushed to origin (open-PR lane); branch kept", []
|
||||
if untracked:
|
||||
return "reap-archive", f"merged/pushed; {archive_note}", untracked
|
||||
return "reap", "clean and fully merged/pushed", []
|
||||
|
||||
|
||||
def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]:
|
||||
"""Classify every tree under ``.worktrees/`` without mutating anything."""
|
||||
import cli as _cli # lazy: cli.py is heavy
|
||||
@@ -194,75 +181,26 @@ def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeReco
|
||||
age_days = (now - entry.stat().st_mtime) / 86400.0
|
||||
except Exception:
|
||||
continue
|
||||
size_mb = _tree_size_mb(entry) if with_sizes else None
|
||||
|
||||
try:
|
||||
branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5)
|
||||
branch = branch_result.stdout.strip()
|
||||
branch = _git(["branch", "--show-current"], cwd=str(entry), timeout=5).stdout.strip()
|
||||
except Exception:
|
||||
branch = ""
|
||||
|
||||
def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None):
|
||||
records.append(TreeRecord(
|
||||
name=entry.name, path=str(entry), branch=branch,
|
||||
age_days=age_days, size_mb=size_mb,
|
||||
verdict=verdict, reason=reason,
|
||||
untracked=untracked or [],
|
||||
))
|
||||
|
||||
if _KANBAN_RE.match(entry.name):
|
||||
rec("keep", "kanban task tree (owned by kanban gc)")
|
||||
continue
|
||||
|
||||
lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5)
|
||||
if lock_state == "live":
|
||||
rec("keep", "in use by a running hermes session")
|
||||
continue
|
||||
|
||||
tracked_dirty, untracked = _dirty_split(str(entry))
|
||||
if tracked_dirty:
|
||||
rec("keep", "uncommitted tracked changes (real work)")
|
||||
continue
|
||||
|
||||
if _cli._worktree_has_unpushed_commits(str(entry), timeout=5):
|
||||
merged = _cli._worktree_commits_all_merged_upstream(
|
||||
str(entry), timeout=30, cache=merge_cache,
|
||||
max_ahead=_MAX_CHERRY_AHEAD,
|
||||
)
|
||||
if not merged:
|
||||
# Pushed-branch tier: single-branch fetch refspecs (the
|
||||
# managed-install default) leave pushed PR branches with no
|
||||
# refs/remotes/* entry, so `git log HEAD --not --remotes`
|
||||
# reads them as unpushed forever. A local head that EXACTLY
|
||||
# matches the remote branch has nothing origin lacks — the
|
||||
# checkout is redundant; only the branch ref stays.
|
||||
if _cli._worktree_branch_pushed_exact(
|
||||
str(entry), remote_heads, timeout=10
|
||||
):
|
||||
if untracked:
|
||||
rec("reap-keep-branch",
|
||||
f"pushed to origin (open-PR lane); branch kept; "
|
||||
f"{len(untracked)} untracked file(s) will be archived",
|
||||
untracked)
|
||||
else:
|
||||
rec("reap-keep-branch",
|
||||
"pushed to origin (open-PR lane); branch kept")
|
||||
continue
|
||||
rec("keep", "unpushed commits not found upstream")
|
||||
continue
|
||||
|
||||
if untracked:
|
||||
rec("reap-archive",
|
||||
f"merged/pushed; {len(untracked)} untracked file(s) will be archived",
|
||||
untracked)
|
||||
else:
|
||||
rec("reap", "clean and fully merged/pushed")
|
||||
verdict, reason, untracked = _classify_tree(_cli, repo_root, entry, merge_cache, remote_heads)
|
||||
records.append(TreeRecord(
|
||||
name=entry.name, path=str(entry), branch=branch,
|
||||
age_days=age_days, size_mb=_tree_size_mb(entry) if with_sizes else None,
|
||||
verdict=verdict, reason=reason, untracked=untracked,
|
||||
))
|
||||
|
||||
if len(merge_cache) != cache_size_before:
|
||||
_cli._save_worktree_merge_cache(merge_cache)
|
||||
return records
|
||||
|
||||
|
||||
_REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"}
|
||||
|
||||
|
||||
def reclaim_worktrees(
|
||||
repo_root: str,
|
||||
*,
|
||||
@@ -271,14 +209,13 @@ def reclaim_worktrees(
|
||||
) -> List[str]:
|
||||
"""Remove every reap-verdict tree from a frozen audit list.
|
||||
|
||||
Operates ONLY on the provided (or freshly computed) audit records — never
|
||||
re-globs inside the destructive loop, so trees created by concurrent
|
||||
sessions after the audit are out of scope by construction.
|
||||
Operates ONLY on the provided (or freshly computed) audit records — never re-globs inside the
|
||||
destructive loop, so trees created by concurrent sessions after the audit are out of scope by
|
||||
construction.
|
||||
"""
|
||||
if records is None:
|
||||
records = audit_worktrees(repo_root, with_sizes=False)
|
||||
actions: List[str] = []
|
||||
_REAP_VERDICTS = {"reap", "reap-archive", "reap-keep-branch"}
|
||||
for record in records:
|
||||
if record.verdict not in _REAP_VERDICTS:
|
||||
continue
|
||||
@@ -301,19 +238,12 @@ def reclaim_worktrees(
|
||||
pass
|
||||
|
||||
try:
|
||||
remove_result = _git(
|
||||
["worktree", "remove", record.path, "--force"],
|
||||
cwd=repo_root, timeout=30,
|
||||
)
|
||||
remove_result = _git(["worktree", "remove", record.path, "--force"], cwd=repo_root, timeout=30)
|
||||
if remove_result.returncode != 0:
|
||||
actions.append(
|
||||
f"failed to remove {record.name}: {remove_result.stderr.strip()}"
|
||||
)
|
||||
actions.append(f"failed to remove {record.name}: {remove_result.stderr.strip()}")
|
||||
continue
|
||||
if record.verdict == "reap-keep-branch":
|
||||
actions.append(
|
||||
f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)"
|
||||
)
|
||||
actions.append(f"removed {record.name} (branch {record.branch} kept — pushed open-PR lane)")
|
||||
continue
|
||||
if record.branch and record.branch not in _PROTECTED_BRANCHES:
|
||||
_git(["branch", "-D", record.branch], cwd=repo_root, timeout=10)
|
||||
@@ -330,44 +260,39 @@ def reclaim_worktrees(
|
||||
|
||||
|
||||
def audit_branches(repo_root: str) -> List[BranchRecord]:
|
||||
"""Classify local branches: safe to delete when their content is on
|
||||
upstream (fully merged OR every commit patch-equivalent via ``git
|
||||
cherry``) and they are not checked out anywhere.
|
||||
"""Classify local branches: deletable when their content is on upstream and not checked out.
|
||||
|
||||
Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``)
|
||||
to EVERY local branch, because deletion is gated on content reachability
|
||||
rather than name: a branch whose commits are all upstream loses nothing
|
||||
when its ref goes. Branch names checked out in any worktree, protected
|
||||
names, and branches with unique commits are kept.
|
||||
"On upstream" means fully merged OR every commit patch-equivalent via ``git cherry``. Applies
|
||||
to EVERY local branch, not just the startup pass's prefix list, because the gate is content
|
||||
reachability rather than name. Checked-out, protected, and unique-commit branches are kept.
|
||||
"""
|
||||
import cli as _cli
|
||||
|
||||
if _cli._repo_is_shallow(repo_root):
|
||||
_cli._deepen_shallow_repo(repo_root)
|
||||
|
||||
upstream = None
|
||||
for candidate in ("origin/HEAD", "origin/main", "origin/master"):
|
||||
probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5)
|
||||
if probe.returncode == 0:
|
||||
upstream = candidate
|
||||
break
|
||||
def _lines(result) -> List[str]:
|
||||
return [b.strip() for b in result.stdout.splitlines() if b.strip()]
|
||||
|
||||
upstream = next(
|
||||
(c for c in ("origin/HEAD", "origin/main", "origin/master")
|
||||
if _git(["rev-parse", "--verify", "--quiet", c], cwd=repo_root, timeout=5).returncode == 0),
|
||||
None,
|
||||
)
|
||||
if upstream is None:
|
||||
return []
|
||||
|
||||
result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10)
|
||||
if result.returncode != 0:
|
||||
return []
|
||||
branches = [b.strip() for b in result.stdout.splitlines() if b.strip()]
|
||||
branches = _lines(result)
|
||||
|
||||
active: set = set()
|
||||
wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10)
|
||||
for line in wt.stdout.splitlines():
|
||||
if line.startswith("branch refs/heads/"):
|
||||
active.add(line.split("branch refs/heads/", 1)[-1].strip())
|
||||
|
||||
merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"],
|
||||
cwd=repo_root, timeout=15)
|
||||
merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()}
|
||||
active = {
|
||||
line.removeprefix("branch refs/heads/").strip()
|
||||
for line in wt.stdout.splitlines() if line.startswith("branch refs/heads/")
|
||||
}
|
||||
merged = set(_lines(_git(["branch", "--merged", upstream, "--format=%(refname:short)"], cwd=repo_root, timeout=15)))
|
||||
|
||||
def _classify_branch(branch: str) -> BranchRecord:
|
||||
if branch in _PROTECTED_BRANCHES or branch in active:
|
||||
@@ -389,7 +314,7 @@ def audit_branches(repo_root: str) -> List[BranchRecord]:
|
||||
cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30)
|
||||
if cherry.returncode != 0:
|
||||
return BranchRecord(branch, "keep", "could not verify (git cherry failed)")
|
||||
lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()]
|
||||
lines = _lines(cherry)
|
||||
if lines and all(ln.startswith("-") for ln in lines):
|
||||
return BranchRecord(branch, "delete", "all commits patch-equivalent upstream")
|
||||
unique = sum(1 for ln in lines if ln.startswith("+"))
|
||||
@@ -429,10 +354,10 @@ def reclaim_branches(
|
||||
actions.append(f"would delete branch {record.name} ({record.reason})")
|
||||
continue
|
||||
result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10)
|
||||
if result.returncode == 0:
|
||||
actions.append(f"deleted branch {record.name}")
|
||||
else:
|
||||
actions.append(f"failed to delete {record.name}: {result.stderr.strip()}")
|
||||
actions.append(
|
||||
f"deleted branch {record.name}" if result.returncode == 0
|
||||
else f"failed to delete {record.name}: {result.stderr.strip()}"
|
||||
)
|
||||
return actions
|
||||
|
||||
|
||||
@@ -446,15 +371,4 @@ def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]:
|
||||
count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir())
|
||||
except Exception:
|
||||
return 0, None
|
||||
size_mb: Optional[int] = None
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["du", "-sm", str(worktrees_dir)],
|
||||
capture_output=True, text=True, encoding="utf-8",
|
||||
errors="replace", timeout=20,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
size_mb = int(result.stdout.split()[0])
|
||||
except Exception:
|
||||
pass
|
||||
return count, size_mb
|
||||
return count, _tree_size_mb(worktrees_dir, timeout=20)
|
||||
|
||||
Reference in New Issue
Block a user