fix(cron): degrade gracefully when systemd user scopes are unavailable

A systemd-supervised gateway (INVOCATION_ID set) with no user D-Bus
session (containers, minimal LXCs, supervisors without linger) fails
EVERY scheduled job at dispatch: restart_safe_gateway_child_argv()
raises, run_one_job() records a failure, and the only symptom is
silently skipped executions (a missed nightly backup, dead watchdogs,
no alert).

Cron now degrades to a direct external subprocess with a
once-per-process warning instead of raising, unless
cron.require_restart_safe_scope=true (config.yaml, default false)
restores fail-closed. Degraded jobs keep process separation and the
full #101940 ownership handoff - only cgroup isolation is lost, so a
mid-job gateway restart kills the worker and the execution ledger
records exactly that.

The dispatch is a GatewayChildDispatch NamedTuple (in_process /
scoped / degraded) so the degraded case can never collapse into the
"not managed, stay in-process" sentinel - the failure mode that would
recreate the restart-interruption edge #101940 closed.

Kanban stays fail-closed (require_restart_safe_scope=True at its call
sites): its workers are long-lived agentic runs, so the degrade policy
is limited to bounded cron jobs in this PR.

Addresses the #102431 review: the env-var flag became a config key per
AGENTS.md (no new HERMES_* non-secret vars), Kanban keeps fail-closed
instead of updating its tests to a degraded contract, main's
enable-linger remedy message is preserved, and the degrade warning
fires once per process.
This commit is contained in:
Paul Robertson
2026-09-06 18:49:40 +12:00
committed by kshitij
parent 9a60a7f316
commit 560b6d2e81
6 changed files with 379 additions and 46 deletions
+99 -20
View File
@@ -27,7 +27,7 @@ _IS_LINUX = platform.system() == "Linux"
from tools.environments.local import _find_shell, _resolve_safe_cwd, _sanitize_subprocess_env
from hermes_cli._subprocess_compat import windows_hide_flags
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Literal, NamedTuple, Optional
from hermes_cli.config import get_hermes_home
@@ -291,36 +291,115 @@ def _build_systemd_scope_argv(shell_argv: List[str], unit_suffix: str) -> List[s
return _systemd_scope_argv(binary, f"hermes-worker-{unit_suffix}", *shell_argv)
# --- restart-safe gateway child dispatch --------------------------------------
# A systemd-supervised gateway restart kills every process in the service cgroup, so
# children that must survive it are launched outside that cgroup via a transient user
# scope (systemd-run --user --scope) when one can be created. Hosts with no user
# systemd session at all (containers, LXCs without linger) cannot create scopes; the
# callers set policy explicitly through ``require_restart_safe_scope`` (cron reads
# ``cron.require_restart_safe_scope`` from config.yaml; kanban always requires a scope).
_scope_degraded_warned = False
def _warn_scope_degraded_once(unit_suffix: str) -> None:
"""Emit the degrade warning once per process.
The scope-availability verdict is cached, so without this guard the warning
would fire on every dispatch (every cron fire on a bus-less host).
"""
global _scope_degraded_warned
if _scope_degraded_warned:
return
_scope_degraded_warned = True
logger.warning(
"%s: systemd-run --user --scope is unavailable (no user D-Bus session at "
"/run/user/%d/bus); dispatching the gateway child as a direct external "
"subprocess without restart-safe cgroup isolation. The job still runs "
"outside the gateway process, but will be killed if the gateway restarts "
"mid-job. Remediate with `sudo loginctl enable-linger <gateway-user>` (plus "
"XDG_RUNTIME_DIR/DBUS_SESSION_BUS_ADDRESS in the service unit), or set "
"cron.require_restart_safe_scope=true in config.yaml to fail closed instead.",
unit_suffix, os.getuid(),
)
class GatewayChildDispatch(NamedTuple):
"""How a managed-gateway child should be launched.
Three mutually exclusive topologies - the whole point of this type is
that (1) and (3) must never collapse into the same value:
- ``"in_process"`` - not a managed systemd gateway (standalone process,
non-systemd supervisor, non-Linux host). The caller keeps its existing
in-process path. ``argv is command`` holds, preserving the historical
passthrough contract.
- ``"scoped"`` - managed gateway with a working user bus. ``argv`` is the
``systemd-run --user --scope`` wrapper; the caller launches it as an
external worker with the #101940 ownership handoff.
- ``"degraded"`` - managed gateway WITHOUT a user bus. ``argv`` is the
direct command, but the caller MUST still launch it as an external
subprocess (same ownership handoff as ``"scoped"``) - never fall back
to the in-process path, which would recreate the restart-interruption
edge #101940 closed. Isolation is lost but process separation is kept.
"""
mode: Literal["in_process", "scoped", "degraded"]
argv: List[str]
reason: str = ""
def restart_safe_gateway_child_argv(
command: List[str], *, unit_suffix: str
) -> List[str]:
command: List[str], *, unit_suffix: str, require_restart_safe_scope: bool,
) -> GatewayChildDispatch:
"""Place a managed-systemd gateway child outside the gateway cgroup.
Returns a :class:`GatewayChildDispatch` distinguishing three topologies -
never the bare command list, so callers cannot mistake a degraded dispatch
for "not managed, stay in-process".
Children that must survive an intentional gateway restart cannot rely on
``start_new_session`` alone: systemd still kills every process in the
service cgroup. In that topology, require a transient user scope and fail
closed if it cannot be established. Standalone processes, non-systemd
supervisors, and non-Linux hosts retain the direct command.
service cgroup. In that topology, prefer a transient user scope.
When a user systemd session is genuinely absent (containers, minimal LXCs,
macOS-style supervisors) the scope cannot be created, but hard failing takes
down every scheduled job on the host - a silent cron outage with no
operator-visible symptom beyond skipped executions (the #101940 durability
contract covers the restart case, not the never-had-a-bus case). Callers
therefore state their policy explicitly: ``require_restart_safe_scope=True``
raises when no scope can be established (fail-closed, the kanban contract);
``False`` degrades to a direct external subprocess with a once-per-process
warning (the cron default behind ``cron.require_restart_safe_scope``).
Standalone processes, non-systemd supervisors, and non-Linux hosts return
``mode == "in_process"`` - the caller keeps its existing in-process path.
"""
if not _IS_LINUX:
return command
return GatewayChildDispatch("in_process", command)
if not _is_supervised_gateway_process() or not os.environ.get("INVOCATION_ID"):
return command
return GatewayChildDispatch("in_process", command)
if not _systemd_run_user_scope_available():
# Stored as the cron execution's error and shown on the job row: name the remedy.
raise RuntimeError(
"cannot create restart-safe systemd scope for gateway child: "
"systemd-run --user --scope is unavailable (usually no reachable user D-Bus session at "
f"/run/user/{os.getuid()}/bus). On a system-level service install, run " # windows-footgun: ok — behind the _IS_LINUX return above
"`sudo loginctl enable-linger <gateway-user>` and restart the gateway."
)
if require_restart_safe_scope:
# Stored as the cron execution's error and shown on the job row: name the remedy.
raise RuntimeError(
"cannot create restart-safe systemd scope for gateway child: "
"systemd-run --user --scope is unavailable (usually no reachable user D-Bus session at "
f"/run/user/{os.getuid()}/bus). On a system-level service install, run " # windows-footgun: ok — behind the _IS_LINUX return above
"`sudo loginctl enable-linger <gateway-user>` and restart the gateway."
)
_warn_scope_degraded_once(unit_suffix)
return GatewayChildDispatch("degraded", command, "no-user-bus")
scoped = _build_systemd_scope_argv(command, unit_suffix=unit_suffix)
if scoped == command:
raise RuntimeError(
"cannot create restart-safe systemd scope for gateway child: "
"systemd-run disappeared after the availability probe"
)
return scoped
if require_restart_safe_scope:
raise RuntimeError(
"cannot create restart-safe systemd scope for gateway child: "
"systemd-run disappeared after the availability probe"
)
_warn_scope_degraded_once(unit_suffix)
return GatewayChildDispatch("degraded", command, "scope-binary-vanished")
return GatewayChildDispatch("scoped", scoped)
def _stop_systemd_unit(unit_name: str) -> bool: