diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 361441073d..1e588e600b 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -2064,6 +2064,26 @@ def _ensure_user_systemd_env() -> None: os.environ["DBUS_SESSION_BUS_ADDRESS"] = f"unix:path={bus_path}" +def _adopt_user_bus_when_started_by_systemd() -> None: + """Point a systemd-launched gateway at its own user bus before it boots (#104893). + + A *system*-level unit (``/etc/systemd/system``, ``User=``) is exec'd with + neither ``XDG_RUNTIME_DIR`` nor ``DBUS_SESSION_BUS_ADDRESS``, and a process environment + is fixed at exec time — so ``systemd-run --user`` fails for the whole lifetime of that + gateway even once ``/run/user//bus`` is up. Every restart-safe worker crosses that + seam (``tools.process_registry.restart_safe_gateway_child_argv``) and it fails closed by + design, so cron and Kanban dispatch died at every fire on headless service installs. + ``_ensure_user_systemd_env`` derives both values from our own uid and adopts them only + when the runtime dir is really ours and the socket really exists, so a host with no user + manager keeps its honest "unavailable" verdict instead of a fabricated bus address. + Doing it here rather than at the dispatch seam matters: worker environments are snapshots + of ``os.environ`` taken at different points, so the adoption has to precede all of them. + """ + if os.name != "posix" or not os.environ.get("INVOCATION_ID"): + return + _ensure_user_systemd_env() + + def _wait_for_user_dbus_socket(timeout: float = 3.0) -> bool: """Poll up to ``timeout`` s for a user systemd control socket (user@.service takes a moment after enable-linger).""" deadline = time.monotonic() + timeout @@ -4530,6 +4550,8 @@ def run_gateway(verbose: int = 0, quiet: bool = False, replace: bool = False, fo if _absorb: _absorb_windows_console_controls() + _adopt_user_bus_when_started_by_systemd() + # Refresh the systemd unit on every boot so restart settings stay current even after an # exit-code-75 respawn (stale-code or /restart), which bypasses `hermes gateway restart`. if supports_systemd_services(): diff --git a/tools/process_registry.py b/tools/process_registry.py index 9580eb2a54..b51b0404c8 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -235,9 +235,15 @@ def restart_safe_gateway_child_argv( if not _is_supervised_gateway_process() or not os.environ.get("INVOCATION_ID"): return command if not _systemd_run_user_scope_available(): + # This text is what the operator actually sees: it is stored as the cron + # execution's error and printed on the job row, so it names the remedy + # rather than only the symptom. raise RuntimeError( "cannot create restart-safe systemd scope for gateway child: " - "systemd-run --user --scope is unavailable" + "systemd-run --user --scope is unavailable — this gateway has no reachable " + f"user D-Bus session (expected /run/user/{os.getuid()}/bus). On a system-level " # windows-footgun: ok — behind the _IS_LINUX return above + "service install, run `sudo loginctl enable-linger ` and restart " + "the gateway." ) scoped = _build_systemd_scope_argv(command, unit_suffix=unit_suffix) if scoped == command: