From ec30eefec9f5335a5f7dc8bf5da8c5e0d6751f68 Mon Sep 17 00:00:00 2001 From: Zaar Hai Date: Mon, 14 Sep 2026 13:08:57 +0700 Subject: [PATCH] fix(nix): linger the service uid so cron can create its worker scope MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The NixOS module produces the one topology the restart-safe cron worker cannot run in: a system service (`User=hermes`, `isSystemUser = true`) whose uid has no systemd user manager. `run_one_job` hands every fire to `_launch_external_cron_worker` → `restart_safe_gateway_child_argv`, which requires a transient `systemd-run --user --scope` and fails closed when it cannot get one: Restart-safe cron worker dispatch failed: cannot create restart-safe systemd scope for gateway child: systemd-run --user --scope is unavailable (usually no reachable user D-Bus session at /run/user/995/bus). On a system-level service install, run `sudo loginctl enable-linger ` and restart the gateway. So on a stock `services.hermes-agent.enable = true` host, no cron job runs at all — every fire is recorded as a failed execution before an agent starts, and the occurrence is consumed, so a weekly job does not retry for another week. The error names the remedy, but nothing in the module applies it, and the terminal-worker path with the same cgroup-isolation goal degrades gracefully instead (`process_registry.py`: "worker shares the gateway cgroup"), so nothing else on the host looks wrong. Three parts, because linger alone leaves a startup race: - `linger = lib.mkDefault true` on the created user: the declarative `loginctl enable-linger`, which gives the uid a user manager and with it /run/user//bus. mkDefault, so an operator can still refuse it. - after/wants linger-users.service, the unit that runs enable-linger. - a bounded preStart wait for the socket. logind starts user@.service asynchronously, and `run_gateway` resolves XDG_RUNTIME_DIR / DBUS_SESSION_BUS_ADDRESS exactly once at startup (802f0f9), so a bus that appears after ExecStart is one the process never sees for its lifetime. Non-fatal after 10s: a gateway without cron beats no gateway. All three are gated on `lingerEnabled`, read back off `config` rather than assumed: with `createUser = false` the operator owns the user, nothing here knows whether they lingered it, and the unit must not block on a bus that may never arrive. Test: nix/checks.nix `cron-worker-user-scope` relates the unit's `User=` to that user's `linger`, requires the ordering and the wait when it lingers, and requires neither when the module does not own the user. Red on the parent commit for the first three arms. Needs nixpkgs >= 25.05 for `users.manageLingering`. --- nix/checks.nix | 43 +++++++++++++++++++++++++++++++++++++++ nix/nixosModules.nix | 48 ++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 89 insertions(+), 2 deletions(-) diff --git a/nix/checks.nix b/nix/checks.nix index 227f57d5e5..b47fe19c3d 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -810,6 +810,49 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) '' ); + # ── The user manager that restart-safe cron workers need ───────── + # The cron scheduler launches every job in a transient `systemd-run + # --user --scope`, so that a gateway restart cannot kill a running job, + # and it fails the fire closed when the scope cannot be created. A + # system service has no user manager — and so no /run/user//bus — + # unless the uid lingers. That makes `linger` load-bearing for cron on + # this module, not cosmetic: without it every job dies before an agent + # starts. This check proves three properties: the uid the gateway runs + # as lingers, a lingering gateway does not start before that uid's bus + # exists, and a gateway whose uid does not linger waits for nothing. + cron-worker-user-scope = + let + configOf = settings: (evalNixosModule ({ enable = true; } // settings)).config; + + managed = configOf { }; + gateway = managed.systemd.services.hermes-agent; + gatewayUser = gateway.serviceConfig.User; + + # The operator declares the user themselves. Nothing here knows + # whether they lingered it, so nothing may assume a bus. + unmanaged = (configOf { createUser = false; }).systemd.services.hermes-agent; + + failures = + lib.optional (!((managed.users.users.${gatewayUser}.linger or false) == true)) + "the uid the gateway runs as (${gatewayUser}) must linger: `systemd-run --user --scope` has no user manager to ask without it, and cron dispatch fails closed" + ++ lib.optional (!lib.elem "linger-users.service" gateway.after) + "a lingering gateway must be ordered after linger-users.service, the unit that runs `loginctl enable-linger`" + ++ lib.optional (!lib.hasInfix "/run/user" gateway.preStart) + "a lingering gateway must wait for its user bus before ExecStart: the bus environment is resolved once at startup, so a bus that appears later is one this process never sees" + ++ lib.optional (lib.hasInfix "/run/user" unmanaged.preStart) + "a gateway whose uid is not known to linger must not block on a user bus that may never arrive"; + in + pkgs.runCommand "hermes-cron-worker-user-scope" { } ( + if failures != [ ] then + throw "cron worker user scope check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: the gateway uid lingers and the unit waits for its user bus" + mkdir -p $out + echo "ok" > $out/result + '' + ); + # ── How .env is built ──────────────────────────────────────────── # This check runs the real script that both modules use to build # $HERMES_HOME/.env. The important property is that a second run diff --git a/nix/nixosModules.nix b/nix/nixosModules.nix index 0a3182d640..52ba773776 100644 --- a/nix/nixosModules.nix +++ b/nix/nixosModules.nix @@ -240,6 +240,13 @@ unitPath = common.processPath { inherit pkgs cfg; }; + # Whether the service uid gets a systemd user manager, and with it the + # user bus that restart-safe cron workers need (see the `linger` attribute + # under createUser). Read back off `config` rather than assumed, so an + # operator who declares the user themselves — or who turns the default off + # — does not pay for a bus that will never arrive. + lingerEnabled = ((config.users.users.${cfg.user} or { }).linger or false) == true; + in { options.services.hermes-agent = @@ -350,6 +357,15 @@ home = cfg.stateDir; createHome = true; shell = pkgs.bashInteractive; + + # The cron scheduler launches every job in a transient `systemd-run + # --user --scope` so a gateway restart cannot kill a running job, + # and that needs a systemd user manager for this uid. A system + # service has no /run/user/ unless the uid lingers, and the + # dispatch fails closed — without this, no cron job runs at all. + # + # Needs nixpkgs >= 25.05 (users.manageLingering). + linger = lib.mkDefault true; }; }) @@ -549,14 +565,42 @@ systemd.services.hermes-agent = { description = "Hermes Agent Gateway"; wantedBy = [ "multi-user.target" ]; - after = [ "network-online.target" ]; - wants = [ "network-online.target" ]; + # linger-users.service is the unit that runs `loginctl + # enable-linger` for a declared `users.users..linger`. + after = [ + "network-online.target" + ] + ++ lib.optional lingerEnabled "linger-users.service"; + wants = [ + "network-online.target" + ] + ++ lib.optional lingerEnabled "linger-users.service"; # cfg.environment and cfg.environmentFiles are written to # $HERMES_HOME/.env by the activation script. load_hermes_dotenv() # reads them at Python startup — no systemd EnvironmentFile needed. environment = commonUnitEnvironment; + # Wait for the user bus that `systemd-run --user --scope` connects + # to. Ordering after linger-users.service is not enough on its own: + # `loginctl enable-linger` returns before logind has finished + # starting user@.service, and run_gateway() resolves + # XDG_RUNTIME_DIR / DBUS_SESSION_BUS_ADDRESS exactly once at + # startup — so a bus that appears after ExecStart is a bus this + # process never sees, for its whole lifetime. + # + # Bounded and non-fatal: a gateway without cron beats no gateway. + preStart = lib.mkIf lingerEnabled '' + for _ in $(seq 1 50); do + [ -S "/run/user/$(id -u)/bus" ] && break + sleep 0.2 + done + if [ ! -S "/run/user/$(id -u)/bus" ]; then + echo "hermes-agent: no user bus at /run/user/$(id -u)/bus after 10s;" \ + "restart-safe cron dispatch will fail for the life of this process" >&2 + fi + ''; + serviceConfig = commonServiceConfig // { ExecStart = lib.escapeShellArgs (common.gatewayArgv cfg); };