Files
hermes-agent/gateway/scale_to_zero.py
T
Teknium 3c7069bdcb refactor(gateway): dead-code removal, helper unification, defensive-layer collapse and rationale-preserving comment compaction across 40 modules
authz_mixin, browser_control_broker, delivery, delivery_ledger, display_config, drain_control,
hosted_room_links/peer/policy_checkpoint, hosted_rooms, platform_registry, relay/__init__,
relay/ws_transport, run.py and slash_commands.py (comments), session_context, session_state,
streaming_tts_consumer, turn_lease and small modules.

- HostedRoomPolicyCheckpoint._apply_event -> per-kind handler table
- WebSocketRelayTransport._handle_frame -> frame-handler table
- GatewayAuthorizationMixin: unified adapter setting/flag/extra readers
- dead symbols removed (verified zero references): RoomLinkProbe/select_room_link,
  relay_bot_username, is_restart_loop_tripped, debug_rows, DeadTargetRegistry.all_dead,
  BrowserControlBroker.detach_owner/_prune_tickets, StreamingTTSConsumer.started/_enqueue_done/
  _iter_stream_chunks/_next_stream_chunk, RecoverableHandleCache.status_for, _auth_env,
  _copy_default_catalog, _parse_timestamp_prefix, _present_* helpers, _send_result_error_kind,
  _truthy_env, SessionFieldView/TurnLeaseTokenView dunder shims, and their orphaned tests.
- lost WHY/invariant text from the earlier compaction restored compactly (541 hunks audited)
2026-09-02 13:30:50 -07:00

259 lines
10 KiB
Python

"""Scale-to-zero idle detection + dormant-quiesce for the gateway.
Gateway-side behaviour layer over the relay scale-to-zero primitives: it owns
the *decision* to go idle, drives the relay transport's ``go_dormant()``, then
SUSPENDS the machine through the local Fly Machines API socket. Wake stays
platform-side (autostart-on-wakeUrl).
Why the gateway self-suspends instead of relying on Fly ``autostop:"suspend"``:
Fly Proxy judges idle only on INBOUND proxied connections — it cannot see an
in-flight agent turn (outbound-only LLM traffic) and no longer counts open
outbound sockets, so it would suspend mid-job or BEFORE ``go_dormant()`` flipped
the relay destination (buffered-event black hole). Owning the call means it only
fires after the idle predicate holds AND the dormant quiesce completed.
Design constraints:
- Per-instance enable is gated SOLELY by the NAS "Labs" toggle, carried as
the ``HERMES_SCALE_TO_ZERO`` env stamp — not a config key;
``scale_to_zero.idle_timeout_minutes`` IS config.yaml.
- Arm only when messaging is relay-only or absent AND a wakeUrl is registered
AND the flag is set.
- Idle = no in-flight work AND no inbound for N min AND no live background work.
- Quiesce uses ``go_dormant()`` (socket closed, supervisor preserved), NEVER
the stop/restart drain or ``disconnect()``. The process stays alive.
- ``mark_resume_pending`` is deliberately NOT called: suspend preserves RAM.
The pure helpers take plain inputs so they unit-test without a live gateway.
"""
from __future__ import annotations
import json
import logging
import os
import socket
import time
from pathlib import Path
from typing import Any, Iterable, Optional
logger = logging.getLogger(__name__)
# Env flag stamped by NAS when the scaleToZero Labs toggle is on. Truthy values only.
SCALE_TO_ZERO_ENV = "HERMES_SCALE_TO_ZERO"
# Fly-injected machine identity; both must be present for self_suspend_available().
FLY_APP_NAME_ENV = "FLY_APP_NAME"
FLY_MACHINE_ID_ENV = "FLY_MACHINE_ID"
# Local flaps (Fly Machines API) unix socket. POST .../suspend snapshots RAM and
# suspends THIS machine (https://fly.io/docs/reference/suspend-resume/).
FLY_API_SOCKET = "/.fly/api"
# config.yaml default (behavioural setting -> config, not env). Short is safe
# because real work always blocks the suspend and resume is sub-second; longer
# windows just bill idle RAM. Raise per-instance via
# gateway.scale_to_zero.idle_timeout_minutes.
DEFAULT_IDLE_TIMEOUT_MINUTES = 2
_TRUTHY = {"1", "true", "yes", "on"}
# Dashboard-client liveness marker. The dashboard process (tui_gateway/ws.py, a
# DIFFERENT process on hosted instances) touches this on every /api/ws connect
# and inbound frame (clients ping every 15s). The gateway folds the mtime into
# its inbound clock so an open client holds the box awake like a chat message
# does (and gets the same idle_timeout grace after it disconnects) — otherwise
# the box suspends under the client, whose reconnect re-pokes
# the wake URL and the instance flaps every ~60s. Deliberately NO staleness
# cutoff: the mtime is a real inbound timestamp and is_idle already decides
# whether it is recent enough.
DASHBOARD_CLIENT_HEARTBEAT_REL = os.path.join("state", "dashboard_clients.heartbeat")
def _env_str(env: Optional[dict], key: str) -> str:
return str((os.environ if env is None else env).get(key, "")).strip()
def scale_to_zero_enabled(environ: Optional[dict] = None) -> bool:
"""Whether the Labs toggle stamp is set. Absent/blank/falsey -> disabled."""
return _env_str(environ, SCALE_TO_ZERO_ENV).lower() in _TRUTHY
def parse_idle_timeout_seconds(
cfg_value: Any, default_minutes: int = DEFAULT_IDLE_TIMEOUT_MINUTES
) -> float:
"""Coerce ``scale_to_zero.idle_timeout_minutes`` to seconds.
Any non-numeric / non-positive value degrades to the default (never <= 0:
that would make the gateway go dormant instantly).
"""
try:
minutes = float(cfg_value)
except (TypeError, ValueError):
minutes = float(default_minutes)
if minutes <= 0:
minutes = float(default_minutes)
return minutes * 60.0
def messaging_is_relay_only_or_absent(platforms: Iterable[Any]) -> bool:
"""True iff the only connected platform is RELAY, or there is none.
A directly-connected platform holds a live socket and cannot scale to zero.
Compared by ``.value``/name so this module stays enum-import-free.
"""
names = {_platform_name(p) for p in platforms}
names.discard("relay")
return not names
def _platform_name(platform: Any) -> str:
return str(getattr(platform, "value", platform)).strip().lower()
def should_arm(
*,
enabled: bool,
relay_only_or_absent: bool,
wake_url: Optional[str],
) -> bool:
"""Start the idle watcher only if ALL hold: flag on, relay-only/absent messaging,
wakeUrl registered (a suspended instance with no wake target is a black hole).
Any unmet -> the watcher never starts (no idle timer, no dormancy), so a
non-opted instance behaves exactly as before."""
return bool(enabled) and bool(relay_only_or_absent) and bool(wake_url)
def is_idle(
*,
active_work_count: int,
seconds_since_last_inbound: float,
idle_timeout_seconds: float,
has_live_background_work: bool,
) -> bool:
"""The pure idle predicate: no counted active work, no inbound within the
timeout window, and no live background work (backgrounded delegate_task /
kanban / bg terminal) — suspending mid-flight would lose it.
``active_work_count`` is the BROAD aggregate (agent turns + cron + API runs),
not just agents — passing only ``len(_running_agents)`` reopens the
mid-cron-job suspend hole. Callers that cannot read a work source must fail
AWAKE (pass a positive sentinel), never fail to 0.
"""
if active_work_count > 0 or has_live_background_work:
return False
return seconds_since_last_inbound >= idle_timeout_seconds
def dashboard_client_heartbeat_path(hermes_home: Optional[os.PathLike | str] = None):
"""Path of the dashboard-client liveness marker under HERMES_HOME."""
if hermes_home is None:
from hermes_constants import get_hermes_home
hermes_home = get_hermes_home()
return Path(hermes_home) / DASHBOARD_CLIENT_HEARTBEAT_REL
def touch_dashboard_client_heartbeat(path: Optional[os.PathLike | str] = None) -> bool:
"""Mark "a dashboard client is attached right now". Best-effort, never raises."""
try:
p = dashboard_client_heartbeat_path() if path is None else path
os.makedirs(os.path.dirname(p), exist_ok=True)
with open(p, "a", encoding="utf-8"):
pass
os.utime(p, None)
return True
except Exception: # noqa: BLE001 - liveness garnish must never break the WS
logger.debug("scale-to-zero: dashboard heartbeat touch failed", exc_info=True)
return False
def dashboard_client_last_seen(
path: Optional[os.PathLike | str] = None,
*,
now: Optional[float] = None,
) -> Optional[float]:
"""Epoch seconds a dashboard client last sent a WS frame, or None if never.
Missing marker -> None (steady state when nobody has the dashboard open —
NOT fail-awake, or no instance would ever sleep). Unreadable marker ->
``now`` (fail-awake, same rule as the work counters in ``is_idle``).
"""
current = time.time() if now is None else now
p = dashboard_client_heartbeat_path() if path is None else path
try:
# Clamp to now: an NTP step-back can leave the mtime in the future,
# which would push idle out by the step size for no reason.
return min(os.stat(p).st_mtime, current)
except FileNotFoundError:
return None
except OSError:
return current
def self_suspend_available(environ: Optional[dict] = None) -> bool:
"""True iff Fly machine identity is present AND the local Machines API socket exists.
Off-Fly (local dev, other clouds, tests) the watcher skips the quiesce:
the platform owns the freeze, so the gateway stays connected until it lands.
"""
return bool(
_env_str(environ, FLY_APP_NAME_ENV)
and _env_str(environ, FLY_MACHINE_ID_ENV)
and os.path.exists(FLY_API_SOCKET)
)
def suspend_self(
environ: Optional[dict] = None,
*,
socket_path: str = FLY_API_SOCKET,
timeout: float = 10.0,
) -> bool:
"""POST /v1/apps/{app}/machines/{id}/suspend on the local flaps socket.
No token needed — the socket is the credential. Returns True when flaps
accepted (2xx); on success the kernel freezes this process shortly after,
so treat as fire-and-forget. Never raises: a failed suspend just leaves the
machine running (fail-awake, never fail-frozen). stdlib-only on purpose —
a plain unix-socket HTTP/1.1 request, no async plumbing to freeze mid-await.
"""
app = _env_str(environ, FLY_APP_NAME_ENV)
machine_id = _env_str(environ, FLY_MACHINE_ID_ENV)
if not app or not machine_id:
logger.warning("scale-to-zero: suspend_self called without Fly machine identity")
return False
request = (
f"POST /v1/apps/{app}/machines/{machine_id}/suspend HTTP/1.1\r\n"
"Host: flaps\r\n"
"Content-Length: 0\r\n"
"Connection: close\r\n"
"\r\n"
)
try:
with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as sock:
sock.settimeout(timeout)
sock.connect(socket_path)
sock.sendall(request.encode("ascii"))
response = b""
while len(response) < 65536:
chunk = sock.recv(4096)
if not chunk:
break
response += chunk
except OSError as exc:
logger.warning("scale-to-zero: flaps suspend request failed: %s", exc)
return False
status_line = response.split(b"\r\n", 1)[0].decode("ascii", "replace")
parts = status_line.split()
ok = len(parts) >= 2 and parts[1].isdigit() and 200 <= int(parts[1]) < 300
if ok:
logger.info("scale-to-zero: machine suspend accepted by flaps (%s)", status_line)
else:
body = response.split(b"\r\n\r\n", 1)[-1][:500].decode("utf-8", "replace")
logger.warning(
"scale-to-zero: flaps suspend rejected: %s %s",
status_line,
json.dumps(body)[:500],
)
return ok