Files
hermes-agent/gateway/restart.py
T
joaomarcos 45bb486b26 fix(gateway): give in-flight cron work its own drain floor
`agent.restart_drain_timeout` defaults to 0 and governed every class of
in-flight work at once. That default is deliberate for chat turns: the
gateway announces the restart to the user and pre-marks the session
resume_pending, so interrupting one is cheap and recoverable.

A cron run has neither property. Nobody is waiting on it, it is written
to jobs.json as a permanent failure, and a recurring job simply skips to
its next schedule. Sharing the chat budget meant `_drain_active_agents()`
short-circuited on `timeout <= 0` before entering the wait loop, so the
drain reported `drain took 0.00s, timed_out=True, cron_at_start=1,
cron_now=1` — it detected the job and killed it anyway.

Cron work now drains on its own deadline, `agent.cron_drain_timeout`
(default 30s, 0 opts out). The floor is clamped to the shutdown-watchdog
leash minus a teardown reserve, so the longer wait can never consume the
post-drain cleanup window: being SIGKILLed mid-cleanup would leave the
job wedged at `last_status=running`, strictly worse than the bug. Being
bounded also means a cron-triggered restart cannot deadlock on itself.

The `timeout <= 0` special case is gone — an expired deadline expresses
the legacy "interrupt immediately" behaviour, so `timed_out` is always
computed from real state instead of asserted up front. The drain-timeout
warning now reports the elapsed wait rather than the configured budget,
which is what made "timed out after 0.0s" so confusing in the report.

Chat-only shutdowns are unchanged: `restart_drain_timeout: 0` still
interrupts chat turns immediately.

Relates to #82161 (complements #82195, which removes the `hermes update`
self-deadlock that triggered the reported instance).
2026-08-14 21:47:16 -07:00

197 lines
7.5 KiB
Python

"""Shared gateway restart constants and supervisor detection helpers."""
import os
from collections.abc import Mapping
from hermes_cli.config import DEFAULT_CONFIG
# EX_TEMPFAIL from sysexits.h — used to ask the service manager to restart
# the gateway after a graceful drain/reload path completes.
GATEWAY_SERVICE_RESTART_EXIT_CODE = 75
# EX_CONFIG from sysexits.h — fatal configuration error (e.g. token
# collision, no messaging platforms). The s6 finish script translates
# this into exit 125 (permanent failure) so the supervisor stops
# restarting the gateway. See #51228.
GATEWAY_FATAL_CONFIG_EXIT_CODE = 78
# Set by ``hermes gateway run --external-supervisor``. Unlike systemd's
# INVOCATION_ID and launchd's XPC_SERVICE_NAME, this survives wrappers that
# intentionally replace the child environment (for example ``sudo env -i``).
EXTERNAL_GATEWAY_SUPERVISOR_ENV = "HERMES_GATEWAY_EXTERNAL_SUPERVISOR"
DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT = float(
DEFAULT_CONFIG["agent"]["restart_drain_timeout"]
)
# In-band restart (``/restart``, SIGUSR1, self-restart from a child CLI)
# waits for active turns to finish *before* ``stop()`` begins. Distinct
# from ``restart_drain_timeout``, which is the force-interrupt budget
# once ``stop()`` is running (and must stay short under systemd
# TimeoutStopSec). See #77184.
DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT = float(
DEFAULT_CONFIG["agent"]["restart_after_turn_timeout"]
)
# Cron-only floor under the ``stop()`` drain. ``restart_drain_timeout``
# defaults to 0 because interrupting a *chat* turn is cheap and recoverable:
# the user is told the gateway is restarting and the session is pre-marked
# resume_pending. An interrupted *cron* run has neither property — nobody is
# waiting on it, it lands in jobs.json as a permanent failure, and a recurring
# job just waits for its next schedule — so a zero-second drain silently
# destroys work. See #82161.
DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT = float(
DEFAULT_CONFIG["agent"]["cron_drain_timeout"]
)
# Seconds of the shutdown watchdog leash held back for the work that still has
# to happen after the drain returns: interrupt agents, kill tool subprocesses,
# mark in-flight jobs interrupted, disconnect adapters. Waiting for cron past
# that point trades a job that is killed *and recorded* for one that is
# SIGKILLed mid-write and stays wedged at ``last_status=running`` forever.
CRON_DRAIN_CLEANUP_RESERVE_S = 10.0
def is_gateway_supervisor_process(
environ: Mapping[str, str] | None = None,
) -> bool:
"""Return whether this gateway process is owned by a supervisor."""
env = os.environ if environ is None else environ
if env.get("INVOCATION_ID"):
return True
if env.get("HERMES_S6_SUPERVISED_CHILD"):
return True
xpc_service = env.get("XPC_SERVICE_NAME", "")
if xpc_service and xpc_service != "0":
return True
return str(env.get(EXTERNAL_GATEWAY_SUPERVISOR_ENV, "")).strip().lower() in {
"1",
"true",
"yes",
"on",
}
def is_container_restart_context() -> bool:
"""Return whether the gateway is running inside a container for restart
routing purposes (Docker/Podman ⇒ the detached setsid path dies with the
cgroup; exit-75 service restart is the only viable path).
Extracted from the inline probe in the /restart handler so tests can mock
container detection hermetically — a real ``/.dockerenv`` on a
containerized CI runner otherwise flips the routing under the test.
"""
return os.path.exists("/.dockerenv") or os.path.exists("/run/.containerenv")
def parse_restart_drain_timeout(raw: object) -> float:
"""Parse a configured drain timeout, falling back to the shared default."""
try:
value = float(raw) if str(raw or "").strip() else DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT
except (TypeError, ValueError):
return DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT
return max(0.0, value)
def parse_restart_after_turn_timeout(raw: object) -> float:
"""Parse the after-turn wait cap for in-band restart, falling back to default.
``0`` is a deliberate disable (legacy immediate drain) and must not fall
through to the default — unlike empty/missing input.
"""
if raw is None:
return DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT
if isinstance(raw, str) and not raw.strip():
return DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT
try:
value = float(raw)
except (TypeError, ValueError):
return DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT
return max(0.0, value)
def parse_cron_drain_timeout(raw: object) -> float:
"""Parse the cron-only drain floor, falling back to the shared default.
``0`` is a deliberate opt-out — cron work is then interrupted on the same
budget as chat work, the pre-#82161 behaviour — and must not fall through
to the default, unlike empty/missing input.
"""
if raw is None:
return DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT
if isinstance(raw, str) and not raw.strip():
return DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT
try:
value = float(raw)
except (TypeError, ValueError):
return DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT
return max(0.0, value)
def resolve_cron_drain_budget(
drain_timeout: float,
cron_drain_timeout: float,
*,
watchdog_delay: float,
elapsed: float = 0.0,
cleanup_reserve_s: float = CRON_DRAIN_CLEANUP_RESERVE_S,
) -> float:
"""Seconds the shutdown drain may spend waiting on in-flight cron work.
The configured floor is clamped to what this process can actually honour.
The shutdown watchdog hard-exits at ``watchdog_delay`` and the service
manager's ``TimeoutStopSec`` is sized from the same drain timeout, so
waiting past that leash (minus ``cleanup_reserve_s`` for the teardown that
follows the drain) would swap a cleanly-interrupted job for a SIGKILL that
leaves it wedged mid-run — strictly worse than the bug being fixed.
Never returns less than ``drain_timeout``: the cron floor only ever
extends the wait, so an operator who deliberately configured a long
``restart_drain_timeout`` keeps it.
"""
def _seconds(value: object, fallback: float = 0.0) -> float:
try:
return max(float(value), 0.0) # type: ignore[arg-type]
except (TypeError, ValueError):
return fallback
drain = _seconds(drain_timeout)
floor = _seconds(cron_drain_timeout)
if floor <= 0.0:
return drain
ceiling = (
_seconds(watchdog_delay)
- _seconds(elapsed)
- _seconds(cleanup_reserve_s, CRON_DRAIN_CLEANUP_RESERVE_S)
)
return max(drain, min(floor, ceiling))
def resolve_restart_exit_wait_budget(
drain_timeout: float,
after_turn_timeout: float,
*,
headroom: float = 15.0,
) -> float:
"""Seconds a CLI should wait for the gateway PID to exit after SIGUSR1.
In-band restart may defer ``stop()`` until active turns finish
(``after_turn_timeout``) and then spend up to ``drain_timeout`` inside
``stop()``. Callers that fall back to a hard kill on wait expiry must
cover both phases or they reintroduce #77184.
"""
try:
drain = max(float(drain_timeout), 0.0)
except (TypeError, ValueError):
drain = 0.0
try:
after_turn = max(float(after_turn_timeout), 0.0)
except (TypeError, ValueError):
after_turn = 0.0
try:
margin = max(float(headroom), 0.0)
except (TypeError, ValueError):
margin = 0.0
return drain + after_turn + margin