fix(cron): degrade gracefully when systemd user scopes are unavailable
A systemd-supervised gateway (INVOCATION_ID set) with no user D-Bus session (containers, minimal LXCs, supervisors without linger) fails EVERY scheduled job at dispatch: restart_safe_gateway_child_argv() raises, run_one_job() records a failure, and the only symptom is silently skipped executions (a missed nightly backup, dead watchdogs, no alert). Cron now degrades to a direct external subprocess with a once-per-process warning instead of raising, unless cron.require_restart_safe_scope=true (config.yaml, default false) restores fail-closed. Degraded jobs keep process separation and the full #101940 ownership handoff - only cgroup isolation is lost, so a mid-job gateway restart kills the worker and the execution ledger records exactly that. The dispatch is a GatewayChildDispatch NamedTuple (in_process / scoped / degraded) so the degraded case can never collapse into the "not managed, stay in-process" sentinel - the failure mode that would recreate the restart-interruption edge #101940 closed. Kanban stays fail-closed (require_restart_safe_scope=True at its call sites): its workers are long-lived agentic runs, so the degrade policy is limited to bounded cron jobs in this PR. Addresses the #102431 review: the env-var flag became a config key per AGENTS.md (no new HERMES_* non-secret vars), Kanban keeps fail-closed instead of updating its tests to a degraded contract, main's enable-linger remedy message is preserved, and the degrade warning fires once per process.
This commit is contained in:
+99
-20
@@ -27,7 +27,7 @@ _IS_LINUX = platform.system() == "Linux"
|
||||
from tools.environments.local import _find_shell, _resolve_safe_cwd, _sanitize_subprocess_env
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Literal, NamedTuple, Optional
|
||||
|
||||
from hermes_cli.config import get_hermes_home
|
||||
|
||||
@@ -291,36 +291,115 @@ def _build_systemd_scope_argv(shell_argv: List[str], unit_suffix: str) -> List[s
|
||||
return _systemd_scope_argv(binary, f"hermes-worker-{unit_suffix}", *shell_argv)
|
||||
|
||||
|
||||
# --- restart-safe gateway child dispatch --------------------------------------
|
||||
# A systemd-supervised gateway restart kills every process in the service cgroup, so
|
||||
# children that must survive it are launched outside that cgroup via a transient user
|
||||
# scope (systemd-run --user --scope) when one can be created. Hosts with no user
|
||||
# systemd session at all (containers, LXCs without linger) cannot create scopes; the
|
||||
# callers set policy explicitly through ``require_restart_safe_scope`` (cron reads
|
||||
# ``cron.require_restart_safe_scope`` from config.yaml; kanban always requires a scope).
|
||||
|
||||
_scope_degraded_warned = False
|
||||
|
||||
|
||||
def _warn_scope_degraded_once(unit_suffix: str) -> None:
|
||||
"""Emit the degrade warning once per process.
|
||||
|
||||
The scope-availability verdict is cached, so without this guard the warning
|
||||
would fire on every dispatch (every cron fire on a bus-less host).
|
||||
"""
|
||||
global _scope_degraded_warned
|
||||
if _scope_degraded_warned:
|
||||
return
|
||||
_scope_degraded_warned = True
|
||||
logger.warning(
|
||||
"%s: systemd-run --user --scope is unavailable (no user D-Bus session at "
|
||||
"/run/user/%d/bus); dispatching the gateway child as a direct external "
|
||||
"subprocess without restart-safe cgroup isolation. The job still runs "
|
||||
"outside the gateway process, but will be killed if the gateway restarts "
|
||||
"mid-job. Remediate with `sudo loginctl enable-linger <gateway-user>` (plus "
|
||||
"XDG_RUNTIME_DIR/DBUS_SESSION_BUS_ADDRESS in the service unit), or set "
|
||||
"cron.require_restart_safe_scope=true in config.yaml to fail closed instead.",
|
||||
unit_suffix, os.getuid(),
|
||||
)
|
||||
|
||||
|
||||
class GatewayChildDispatch(NamedTuple):
|
||||
"""How a managed-gateway child should be launched.
|
||||
|
||||
Three mutually exclusive topologies - the whole point of this type is
|
||||
that (1) and (3) must never collapse into the same value:
|
||||
|
||||
- ``"in_process"`` - not a managed systemd gateway (standalone process,
|
||||
non-systemd supervisor, non-Linux host). The caller keeps its existing
|
||||
in-process path. ``argv is command`` holds, preserving the historical
|
||||
passthrough contract.
|
||||
- ``"scoped"`` - managed gateway with a working user bus. ``argv`` is the
|
||||
``systemd-run --user --scope`` wrapper; the caller launches it as an
|
||||
external worker with the #101940 ownership handoff.
|
||||
- ``"degraded"`` - managed gateway WITHOUT a user bus. ``argv`` is the
|
||||
direct command, but the caller MUST still launch it as an external
|
||||
subprocess (same ownership handoff as ``"scoped"``) - never fall back
|
||||
to the in-process path, which would recreate the restart-interruption
|
||||
edge #101940 closed. Isolation is lost but process separation is kept.
|
||||
"""
|
||||
|
||||
mode: Literal["in_process", "scoped", "degraded"]
|
||||
argv: List[str]
|
||||
reason: str = ""
|
||||
|
||||
|
||||
def restart_safe_gateway_child_argv(
|
||||
command: List[str], *, unit_suffix: str
|
||||
) -> List[str]:
|
||||
command: List[str], *, unit_suffix: str, require_restart_safe_scope: bool,
|
||||
) -> GatewayChildDispatch:
|
||||
"""Place a managed-systemd gateway child outside the gateway cgroup.
|
||||
|
||||
Returns a :class:`GatewayChildDispatch` distinguishing three topologies -
|
||||
never the bare command list, so callers cannot mistake a degraded dispatch
|
||||
for "not managed, stay in-process".
|
||||
|
||||
Children that must survive an intentional gateway restart cannot rely on
|
||||
``start_new_session`` alone: systemd still kills every process in the
|
||||
service cgroup. In that topology, require a transient user scope and fail
|
||||
closed if it cannot be established. Standalone processes, non-systemd
|
||||
supervisors, and non-Linux hosts retain the direct command.
|
||||
service cgroup. In that topology, prefer a transient user scope.
|
||||
|
||||
When a user systemd session is genuinely absent (containers, minimal LXCs,
|
||||
macOS-style supervisors) the scope cannot be created, but hard failing takes
|
||||
down every scheduled job on the host - a silent cron outage with no
|
||||
operator-visible symptom beyond skipped executions (the #101940 durability
|
||||
contract covers the restart case, not the never-had-a-bus case). Callers
|
||||
therefore state their policy explicitly: ``require_restart_safe_scope=True``
|
||||
raises when no scope can be established (fail-closed, the kanban contract);
|
||||
``False`` degrades to a direct external subprocess with a once-per-process
|
||||
warning (the cron default behind ``cron.require_restart_safe_scope``).
|
||||
|
||||
Standalone processes, non-systemd supervisors, and non-Linux hosts return
|
||||
``mode == "in_process"`` - the caller keeps its existing in-process path.
|
||||
"""
|
||||
if not _IS_LINUX:
|
||||
return command
|
||||
return GatewayChildDispatch("in_process", command)
|
||||
if not _is_supervised_gateway_process() or not os.environ.get("INVOCATION_ID"):
|
||||
return command
|
||||
return GatewayChildDispatch("in_process", command)
|
||||
if not _systemd_run_user_scope_available():
|
||||
# Stored as the cron execution's error and shown on the job row: name the remedy.
|
||||
raise RuntimeError(
|
||||
"cannot create restart-safe systemd scope for gateway child: "
|
||||
"systemd-run --user --scope is unavailable (usually no reachable user D-Bus session at "
|
||||
f"/run/user/{os.getuid()}/bus). On a system-level service install, run " # windows-footgun: ok — behind the _IS_LINUX return above
|
||||
"`sudo loginctl enable-linger <gateway-user>` and restart the gateway."
|
||||
)
|
||||
if require_restart_safe_scope:
|
||||
# Stored as the cron execution's error and shown on the job row: name the remedy.
|
||||
raise RuntimeError(
|
||||
"cannot create restart-safe systemd scope for gateway child: "
|
||||
"systemd-run --user --scope is unavailable (usually no reachable user D-Bus session at "
|
||||
f"/run/user/{os.getuid()}/bus). On a system-level service install, run " # windows-footgun: ok — behind the _IS_LINUX return above
|
||||
"`sudo loginctl enable-linger <gateway-user>` and restart the gateway."
|
||||
)
|
||||
_warn_scope_degraded_once(unit_suffix)
|
||||
return GatewayChildDispatch("degraded", command, "no-user-bus")
|
||||
scoped = _build_systemd_scope_argv(command, unit_suffix=unit_suffix)
|
||||
if scoped == command:
|
||||
raise RuntimeError(
|
||||
"cannot create restart-safe systemd scope for gateway child: "
|
||||
"systemd-run disappeared after the availability probe"
|
||||
)
|
||||
return scoped
|
||||
if require_restart_safe_scope:
|
||||
raise RuntimeError(
|
||||
"cannot create restart-safe systemd scope for gateway child: "
|
||||
"systemd-run disappeared after the availability probe"
|
||||
)
|
||||
_warn_scope_degraded_once(unit_suffix)
|
||||
return GatewayChildDispatch("degraded", command, "scope-binary-vanished")
|
||||
return GatewayChildDispatch("scoped", scoped)
|
||||
|
||||
|
||||
def _stop_systemd_unit(unit_name: str) -> bool:
|
||||
|
||||
Reference in New Issue
Block a user