fix(serve): Desktop-over-SSH isolated backends retire themselves when no client is connected (#101626)

serve --isolated is detached on purpose (setsid/nohup, PPID 1) so it survives the SSH
channel closing, and every teardown path lived on the client. A laptop that sleeps
mid-session (dark wake reconnects the tunnel, spawns a backend, sleeps again) therefore
left a new backend behind every cycle, each one an extra writer on state.db — the
multi-writer source behind the WAL corruption incidents.

Only for backends started with --ssh-session-token-file: an ASGI wrapper counts accepted
WebSocket sessions on every dashboard route without touching the handlers; a watchdog
requests a graceful uvicorn exit (WAL checkpoint, exit 0) once no client has been
connected for dashboard.ssh_isolated_idle_grace_s (default 15 min) AND no agent turn is
running. An unreadable turn state fails closed (the backend stays up). Loopback normally
disables the WS ping, but across a tunnel the local socket stays healthy while the far
end sleeps, so these backends keep a slow ping (60s / 10 min) that a GIL-holding turn
cannot trip.

Design, client-count/turn-probe/fail-closed shape and the tunnel-ping rationale from
#101678 by @StanleyStetson; this is the slim redo on the decomposed web server (no
exclusive home lock / handover protocol — the newcomer never needs to evict an idle
incumbent once idle incumbents exit on their own).
This commit is contained in:
StanleyStetson
2026-09-05 19:45:26 +05:30
committed by kshitij
parent 5284057770
commit fca7fdd35d
2 changed files with 166 additions and 5 deletions
+28 -5
View File
@@ -1114,7 +1114,7 @@ def _configure_auth_gate(
)
def _build_uvicorn_server(host: str, port: int):
def _build_uvicorn_server(host: str, port: int, *, ssh_isolated: bool = False):
"""Build the uvicorn ``Config`` + ``Server`` for this bind (reads ``app.state.auth_required``).
uvicorn.Server is driven directly (not uvicorn.run) so startup is split from
@@ -1145,8 +1145,22 @@ def _build_uvicorn_server(host: str, port: int):
except (TypeError, ValueError):
return default
# A Desktop-owned SSH-isolated backend is loopback on the SERVER, but the client sits at the far
# end of a tunnel: the local socket stays healthy while the laptop sleeps, so only a slow WS ping
# notices the half-open tunnel (#101626). Its client count is tracked at the ASGI boundary so
# the idle watchdog can retire the backend once nothing is connected.
served_app = app
ping_interval, ping_timeout = (None, None) if _is_loopback else (
_ws_ping_setting("ws_ping_interval"), _ws_ping_setting("ws_ping_timeout"))
if ssh_isolated:
from hermes_cli.web_server_idle_exit import (
TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S, IdleClientTracker, wrap_asgi_with_ws_tracking)
app.state.ssh_isolated_clients = IdleClientTracker()
served_app = wrap_asgi_with_ws_tracking(app, app.state.ssh_isolated_clients)
ping_interval, ping_timeout = TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S
config = uvicorn.Config(
app, host=host, port=port, log_level="warning",
served_app, host=host, port=port, log_level="warning",
# Off by default so _ws_client_is_allowed sees the real peer, not
# X-Forwarded-For. Gated mode runs behind a TLS terminator and needs
# X-Forwarded-Proto for cookie Secure flags.
@@ -1154,8 +1168,8 @@ def _build_uvicorn_server(host: str, port: int):
# Loopback-only unless the operator trusts a bounded upstream proxy, so
# spoofed X-Forwarded-* from arbitrary callers is never honoured.
forwarded_allow_ips=_dashboard_forwarded_allow_ips(_dash_cfg),
ws_ping_interval=None if _is_loopback else _ws_ping_setting("ws_ping_interval"),
ws_ping_timeout=None if _is_loopback else _ws_ping_setting("ws_ping_timeout"),
ws_ping_interval=ping_interval,
ws_ping_timeout=ping_timeout,
ws_max_size=_DESKTOP_ATTACHMENT_WS_MAX_BYTES,
)
return config, uvicorn.Server(config)
@@ -1207,6 +1221,15 @@ def _on_server_started(
# No-op for standalone `hermes serve` (no HERMES_PARENT_PID).
_start_parent_death_watchdog()
# SSH-isolated backends are detached from any parent on purpose (#91668); their liveness signal
# is "does a client still hold a WebSocket" (#101626).
if getattr(app.state, "ssh_isolated_clients", None) is not None:
from hermes_cli.web_server_idle_exit import DEFAULT_IDLE_GRACE_S, start_idle_watchdog
try:
grace = float((load_config().get("dashboard") or {}).get("ssh_isolated_idle_grace_s", DEFAULT_IDLE_GRACE_S))
except (TypeError, ValueError):
grace = DEFAULT_IDLE_GRACE_S
start_idle_watchdog(server, app.state.ssh_isolated_clients, grace_s=grace)
actual_port = _read_bound_port(server, fallback=port)
app.state.bound_port = actual_port
@@ -1372,7 +1395,7 @@ def start_server(
# GHSA-ppp5-vxwm-4cf7).
app.state.bound_host = host
config, server = _build_uvicorn_server(host, port)
config, server = _build_uvicorn_server(host, port, ssh_isolated=bool(ssh_session_token))
# Flush-on-kill guard (#94724): chaining SIGTERM/SIGINT handlers persist
# in-memory transcripts to state.db before shutdown. Installed BEFORE
+138
View File
@@ -0,0 +1,138 @@
"""Idle-exit for Desktop-owned ``hermes serve --isolated`` backends reached over SSH (#101626).
That backend is deliberately detached (``setsid``/``nohup``, PPID 1) so it survives the SSH channel
closing, and every teardown path lives on the CLIENT. A laptop that sleeps mid-session (dark wake
reconnects the tunnel, spawns a backend, sleeps again) therefore leaves a backend behind every
cycle — each one an extra writer on ``state.db``. The server needs its own liveness signal.
Two pieces, both scoped to the SSH-isolated case (a session token was handed over via
``--ssh-session-token-file``):
* An ASGI wrapper counts accepted WebSocket connections (every dashboard WS route: /api/ws,
/api/pty, /api/console, /api/pub, /api/events, /api/audio/speak-stream) without touching the
handlers. When the count has been zero for the grace window and no agent turn is running, the
watchdog asks uvicorn to exit gracefully (WAL checkpoint, exit 0). An indeterminate turn probe
fails closed: the backend stays up.
* Loopback normally disables uvicorn's WS ping (a dead local client sends FIN/RST). Across an SSH
tunnel the local socket is healthy while the far end is asleep, so pings are the only way to notice
a half-open tunnel; the isolated backend keeps a slow ping with a long timeout so a GIL-holding
turn cannot trip it.
Design and the client-count/turn-probe/fail-closed shape are from #101678 by @StanleyStetson; this
is the slim redo on the decomposed web server.
"""
from __future__ import annotations
import logging
import threading
import time
from typing import Callable, Optional
_log = logging.getLogger(__name__)
DEFAULT_IDLE_GRACE_S = 900.0
# Slow enough that a long GIL-holding turn (minutes) cannot trip it, fast enough that a sleeping
# laptop's half-open tunnel is noticed well inside the idle grace window.
TUNNEL_WS_PING_INTERVAL_S = 60.0
TUNNEL_WS_PING_TIMEOUT_S = 600.0
class IdleClientTracker:
"""Live accepted-WebSocket count plus the moment the last client left."""
def __init__(self, now: Callable[[], float] = time.monotonic) -> None:
self._now = now
self._lock = threading.Lock()
self._live = 0
self._last_client_at = now()
def on_open(self) -> None:
with self._lock:
self._live += 1
self._last_client_at = self._now()
def on_close(self) -> None:
with self._lock:
self._live = max(0, self._live - 1)
self._last_client_at = self._now()
def live_count(self) -> int:
with self._lock:
return self._live
def idle_for(self) -> float:
with self._lock:
return 0.0 if self._live else self._now() - self._last_client_at
def wrap_asgi_with_ws_tracking(app, tracker: IdleClientTracker):
"""Count WebSocket sessions at the ASGI boundary: open on the ``websocket.accept`` send, close
when the scope ends. Handlers stay untouched, so a new WS route is tracked automatically."""
async def _app(scope, receive, send):
if scope.get("type") != "websocket":
return await app(scope, receive, send)
accepted = False
async def _send(message):
nonlocal accepted
if message.get("type") == "websocket.accept" and not accepted:
accepted = True
tracker.on_open()
await send(message)
try:
await app(scope, receive, _send)
finally:
if accepted:
tracker.on_close()
return _app
_probe_failure_logged = False
def turn_in_flight() -> Optional[bool]:
"""True/False from the gateway's running-session table; None when it cannot be read. The table
lives on ``tui_gateway.server`` (the voice mixin's helper is bound into that namespace). None
keeps the backend alive forever, so the cause is logged once — a silent never-exits would be
the original bug with a new face."""
global _probe_failure_logged
try:
import tui_gateway.server as gateway
with gateway._sessions_lock:
return any(s.get("running") for s in gateway._sessions.values())
except Exception:
if not _probe_failure_logged:
_probe_failure_logged = True
_log.warning("idle-exit turn probe unavailable; this backend will not self-retire", exc_info=True)
return None
def should_exit_idle(tracker: IdleClientTracker, grace_s: float,
probe: Callable[[], Optional[bool]] = turn_in_flight) -> bool:
"""Exit only when no client has been connected for ``grace_s`` AND no turn is provably running.
A probe that cannot answer keeps the process (fail closed)."""
return tracker.idle_for() >= grace_s and probe() is False # idle_for() is 0 while a client is connected
def start_idle_watchdog(server, tracker: IdleClientTracker, *, grace_s: float = DEFAULT_IDLE_GRACE_S,
poll_s: float = 15.0, probe: Callable[[], Optional[bool]] = turn_in_flight) -> threading.Thread:
"""Daemon thread that sets ``server.should_exit`` once :func:`should_exit_idle` holds."""
poll_s = min(poll_s, max(0.5, grace_s / 4))
def _loop() -> None:
while not getattr(server, "should_exit", False):
if should_exit_idle(tracker, grace_s, probe):
_log.warning("SSH-isolated backend idle for %.0fs with no client and no running turn; exiting.",
tracker.idle_for())
server.should_exit = True
return
time.sleep(poll_s)
thread = threading.Thread(target=_loop, daemon=True, name="ssh-isolated-idle-watchdog")
thread.start()
return thread