From 802f0f97adfbf3708888f9d21ba85ca80e0cec87 Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Mon, 7 Sep 2026 18:55:10 +0700 Subject: [PATCH] fix(gateway): adopt the user D-Bus session when systemd starts the gateway MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A system-level unit (/etc/systemd/system, User=) is exec'd with neither XDG_RUNTIME_DIR nor DBUS_SESSION_BUS_ADDRESS, and a process environment is fixed at exec time. 'systemd-run --user --scope' therefore fails for the whole lifetime of that gateway even after the user manager is up and /run/user//bus is reachable. That is the seam every restart-safe worker crosses (restart_safe_gateway_child_argv), and it fails closed by design — so on headless systemd installs every agent-driven cron job and every Kanban dispatch died at launch, ~26ms in, with nothing but 'error' on the job row. _ensure_user_systemd_env() already derives both values from our own uid and adopts them only when the runtime dir is really ours and the socket really exists; it was just wired exclusively to the systemctl management paths, never to the gateway's own boot. Call it from run_gateway() — the single in-process boot every entry point goes through — so the adoption precedes every worker-environment snapshot (cron builds its env after the scope check, Kanban before it, so fixing this at the dispatch seam would only fix one of them). The fail-closed posture is unchanged: with no user manager at all the probe still reports unavailable and dispatch still refuses. That refusal now names the remedy in the message the operator actually reads (it is stored as the cron execution's error), instead of only the symptom. Fixes #104893 --- hermes_cli/gateway.py | 22 ++++++++++++++++++++++ tools/process_registry.py | 8 +++++++- 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 361441073d..1e588e600b 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -2064,6 +2064,26 @@ def _ensure_user_systemd_env() -> None: os.environ["DBUS_SESSION_BUS_ADDRESS"] = f"unix:path={bus_path}" +def _adopt_user_bus_when_started_by_systemd() -> None: + """Point a systemd-launched gateway at its own user bus before it boots (#104893). + + A *system*-level unit (``/etc/systemd/system``, ``User=``) is exec'd with + neither ``XDG_RUNTIME_DIR`` nor ``DBUS_SESSION_BUS_ADDRESS``, and a process environment + is fixed at exec time — so ``systemd-run --user`` fails for the whole lifetime of that + gateway even once ``/run/user//bus`` is up. Every restart-safe worker crosses that + seam (``tools.process_registry.restart_safe_gateway_child_argv``) and it fails closed by + design, so cron and Kanban dispatch died at every fire on headless service installs. + ``_ensure_user_systemd_env`` derives both values from our own uid and adopts them only + when the runtime dir is really ours and the socket really exists, so a host with no user + manager keeps its honest "unavailable" verdict instead of a fabricated bus address. + Doing it here rather than at the dispatch seam matters: worker environments are snapshots + of ``os.environ`` taken at different points, so the adoption has to precede all of them. + """ + if os.name != "posix" or not os.environ.get("INVOCATION_ID"): + return + _ensure_user_systemd_env() + + def _wait_for_user_dbus_socket(timeout: float = 3.0) -> bool: """Poll up to ``timeout`` s for a user systemd control socket (user@.service takes a moment after enable-linger).""" deadline = time.monotonic() + timeout @@ -4530,6 +4550,8 @@ def run_gateway(verbose: int = 0, quiet: bool = False, replace: bool = False, fo if _absorb: _absorb_windows_console_controls() + _adopt_user_bus_when_started_by_systemd() + # Refresh the systemd unit on every boot so restart settings stay current even after an # exit-code-75 respawn (stale-code or /restart), which bypasses `hermes gateway restart`. if supports_systemd_services(): diff --git a/tools/process_registry.py b/tools/process_registry.py index 9580eb2a54..b51b0404c8 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -235,9 +235,15 @@ def restart_safe_gateway_child_argv( if not _is_supervised_gateway_process() or not os.environ.get("INVOCATION_ID"): return command if not _systemd_run_user_scope_available(): + # This text is what the operator actually sees: it is stored as the cron + # execution's error and printed on the job row, so it names the remedy + # rather than only the symptom. raise RuntimeError( "cannot create restart-safe systemd scope for gateway child: " - "systemd-run --user --scope is unavailable" + "systemd-run --user --scope is unavailable — this gateway has no reachable " + f"user D-Bus session (expected /run/user/{os.getuid()}/bus). On a system-level " # windows-footgun: ok — behind the _IS_LINUX return above + "service install, run `sudo loginctl enable-linger ` and restart " + "the gateway." ) scoped = _build_systemd_scope_argv(command, unit_suffix=unit_suffix) if scoped == command: