3c7069bdcb
authz_mixin, browser_control_broker, delivery, delivery_ledger, display_config, drain_control, hosted_room_links/peer/policy_checkpoint, hosted_rooms, platform_registry, relay/__init__, relay/ws_transport, run.py and slash_commands.py (comments), session_context, session_state, streaming_tts_consumer, turn_lease and small modules. - HostedRoomPolicyCheckpoint._apply_event -> per-kind handler table - WebSocketRelayTransport._handle_frame -> frame-handler table - GatewayAuthorizationMixin: unified adapter setting/flag/extra readers - dead symbols removed (verified zero references): RoomLinkProbe/select_room_link, relay_bot_username, is_restart_loop_tripped, debug_rows, DeadTargetRegistry.all_dead, BrowserControlBroker.detach_owner/_prune_tickets, StreamingTTSConsumer.started/_enqueue_done/ _iter_stream_chunks/_next_stream_chunk, RecoverableHandleCache.status_for, _auth_env, _copy_default_catalog, _parse_timestamp_prefix, _present_* helpers, _send_result_error_kind, _truthy_env, SessionFieldView/TurnLeaseTokenView dunder shims, and their orphaned tests. - lost WHY/invariant text from the earlier compaction restored compactly (541 hunks audited)
259 lines
10 KiB
Python
259 lines
10 KiB
Python
"""Scale-to-zero idle detection + dormant-quiesce for the gateway.
|
|
|
|
Gateway-side behaviour layer over the relay scale-to-zero primitives: it owns
|
|
the *decision* to go idle, drives the relay transport's ``go_dormant()``, then
|
|
SUSPENDS the machine through the local Fly Machines API socket. Wake stays
|
|
platform-side (autostart-on-wakeUrl).
|
|
|
|
Why the gateway self-suspends instead of relying on Fly ``autostop:"suspend"``:
|
|
Fly Proxy judges idle only on INBOUND proxied connections — it cannot see an
|
|
in-flight agent turn (outbound-only LLM traffic) and no longer counts open
|
|
outbound sockets, so it would suspend mid-job or BEFORE ``go_dormant()`` flipped
|
|
the relay destination (buffered-event black hole). Owning the call means it only
|
|
fires after the idle predicate holds AND the dormant quiesce completed.
|
|
|
|
Design constraints:
|
|
- Per-instance enable is gated SOLELY by the NAS "Labs" toggle, carried as
|
|
the ``HERMES_SCALE_TO_ZERO`` env stamp — not a config key;
|
|
``scale_to_zero.idle_timeout_minutes`` IS config.yaml.
|
|
- Arm only when messaging is relay-only or absent AND a wakeUrl is registered
|
|
AND the flag is set.
|
|
- Idle = no in-flight work AND no inbound for N min AND no live background work.
|
|
- Quiesce uses ``go_dormant()`` (socket closed, supervisor preserved), NEVER
|
|
the stop/restart drain or ``disconnect()``. The process stays alive.
|
|
- ``mark_resume_pending`` is deliberately NOT called: suspend preserves RAM.
|
|
|
|
The pure helpers take plain inputs so they unit-test without a live gateway.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import socket
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any, Iterable, Optional
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Env flag stamped by NAS when the scaleToZero Labs toggle is on. Truthy values only.
|
|
SCALE_TO_ZERO_ENV = "HERMES_SCALE_TO_ZERO"
|
|
|
|
# Fly-injected machine identity; both must be present for self_suspend_available().
|
|
FLY_APP_NAME_ENV = "FLY_APP_NAME"
|
|
FLY_MACHINE_ID_ENV = "FLY_MACHINE_ID"
|
|
|
|
# Local flaps (Fly Machines API) unix socket. POST .../suspend snapshots RAM and
|
|
# suspends THIS machine (https://fly.io/docs/reference/suspend-resume/).
|
|
FLY_API_SOCKET = "/.fly/api"
|
|
|
|
# config.yaml default (behavioural setting -> config, not env). Short is safe
|
|
# because real work always blocks the suspend and resume is sub-second; longer
|
|
# windows just bill idle RAM. Raise per-instance via
|
|
# gateway.scale_to_zero.idle_timeout_minutes.
|
|
DEFAULT_IDLE_TIMEOUT_MINUTES = 2
|
|
|
|
_TRUTHY = {"1", "true", "yes", "on"}
|
|
|
|
# Dashboard-client liveness marker. The dashboard process (tui_gateway/ws.py, a
|
|
# DIFFERENT process on hosted instances) touches this on every /api/ws connect
|
|
# and inbound frame (clients ping every 15s). The gateway folds the mtime into
|
|
# its inbound clock so an open client holds the box awake like a chat message
|
|
# does (and gets the same idle_timeout grace after it disconnects) — otherwise
|
|
# the box suspends under the client, whose reconnect re-pokes
|
|
# the wake URL and the instance flaps every ~60s. Deliberately NO staleness
|
|
# cutoff: the mtime is a real inbound timestamp and is_idle already decides
|
|
# whether it is recent enough.
|
|
DASHBOARD_CLIENT_HEARTBEAT_REL = os.path.join("state", "dashboard_clients.heartbeat")
|
|
|
|
|
|
def _env_str(env: Optional[dict], key: str) -> str:
|
|
return str((os.environ if env is None else env).get(key, "")).strip()
|
|
|
|
|
|
def scale_to_zero_enabled(environ: Optional[dict] = None) -> bool:
|
|
"""Whether the Labs toggle stamp is set. Absent/blank/falsey -> disabled."""
|
|
return _env_str(environ, SCALE_TO_ZERO_ENV).lower() in _TRUTHY
|
|
|
|
|
|
def parse_idle_timeout_seconds(
|
|
cfg_value: Any, default_minutes: int = DEFAULT_IDLE_TIMEOUT_MINUTES
|
|
) -> float:
|
|
"""Coerce ``scale_to_zero.idle_timeout_minutes`` to seconds.
|
|
|
|
Any non-numeric / non-positive value degrades to the default (never <= 0:
|
|
that would make the gateway go dormant instantly).
|
|
"""
|
|
try:
|
|
minutes = float(cfg_value)
|
|
except (TypeError, ValueError):
|
|
minutes = float(default_minutes)
|
|
if minutes <= 0:
|
|
minutes = float(default_minutes)
|
|
return minutes * 60.0
|
|
|
|
|
|
def messaging_is_relay_only_or_absent(platforms: Iterable[Any]) -> bool:
|
|
"""True iff the only connected platform is RELAY, or there is none.
|
|
|
|
A directly-connected platform holds a live socket and cannot scale to zero.
|
|
Compared by ``.value``/name so this module stays enum-import-free.
|
|
"""
|
|
names = {_platform_name(p) for p in platforms}
|
|
names.discard("relay")
|
|
return not names
|
|
|
|
|
|
def _platform_name(platform: Any) -> str:
|
|
return str(getattr(platform, "value", platform)).strip().lower()
|
|
|
|
|
|
def should_arm(
|
|
*,
|
|
enabled: bool,
|
|
relay_only_or_absent: bool,
|
|
wake_url: Optional[str],
|
|
) -> bool:
|
|
"""Start the idle watcher only if ALL hold: flag on, relay-only/absent messaging,
|
|
wakeUrl registered (a suspended instance with no wake target is a black hole).
|
|
Any unmet -> the watcher never starts (no idle timer, no dormancy), so a
|
|
non-opted instance behaves exactly as before."""
|
|
return bool(enabled) and bool(relay_only_or_absent) and bool(wake_url)
|
|
|
|
|
|
def is_idle(
|
|
*,
|
|
active_work_count: int,
|
|
seconds_since_last_inbound: float,
|
|
idle_timeout_seconds: float,
|
|
has_live_background_work: bool,
|
|
) -> bool:
|
|
"""The pure idle predicate: no counted active work, no inbound within the
|
|
timeout window, and no live background work (backgrounded delegate_task /
|
|
kanban / bg terminal) — suspending mid-flight would lose it.
|
|
|
|
``active_work_count`` is the BROAD aggregate (agent turns + cron + API runs),
|
|
not just agents — passing only ``len(_running_agents)`` reopens the
|
|
mid-cron-job suspend hole. Callers that cannot read a work source must fail
|
|
AWAKE (pass a positive sentinel), never fail to 0.
|
|
"""
|
|
if active_work_count > 0 or has_live_background_work:
|
|
return False
|
|
return seconds_since_last_inbound >= idle_timeout_seconds
|
|
|
|
|
|
def dashboard_client_heartbeat_path(hermes_home: Optional[os.PathLike | str] = None):
|
|
"""Path of the dashboard-client liveness marker under HERMES_HOME."""
|
|
if hermes_home is None:
|
|
from hermes_constants import get_hermes_home
|
|
|
|
hermes_home = get_hermes_home()
|
|
return Path(hermes_home) / DASHBOARD_CLIENT_HEARTBEAT_REL
|
|
|
|
|
|
def touch_dashboard_client_heartbeat(path: Optional[os.PathLike | str] = None) -> bool:
|
|
"""Mark "a dashboard client is attached right now". Best-effort, never raises."""
|
|
try:
|
|
p = dashboard_client_heartbeat_path() if path is None else path
|
|
os.makedirs(os.path.dirname(p), exist_ok=True)
|
|
with open(p, "a", encoding="utf-8"):
|
|
pass
|
|
os.utime(p, None)
|
|
return True
|
|
except Exception: # noqa: BLE001 - liveness garnish must never break the WS
|
|
logger.debug("scale-to-zero: dashboard heartbeat touch failed", exc_info=True)
|
|
return False
|
|
|
|
|
|
def dashboard_client_last_seen(
|
|
path: Optional[os.PathLike | str] = None,
|
|
*,
|
|
now: Optional[float] = None,
|
|
) -> Optional[float]:
|
|
"""Epoch seconds a dashboard client last sent a WS frame, or None if never.
|
|
|
|
Missing marker -> None (steady state when nobody has the dashboard open —
|
|
NOT fail-awake, or no instance would ever sleep). Unreadable marker ->
|
|
``now`` (fail-awake, same rule as the work counters in ``is_idle``).
|
|
"""
|
|
current = time.time() if now is None else now
|
|
p = dashboard_client_heartbeat_path() if path is None else path
|
|
try:
|
|
# Clamp to now: an NTP step-back can leave the mtime in the future,
|
|
# which would push idle out by the step size for no reason.
|
|
return min(os.stat(p).st_mtime, current)
|
|
except FileNotFoundError:
|
|
return None
|
|
except OSError:
|
|
return current
|
|
|
|
|
|
def self_suspend_available(environ: Optional[dict] = None) -> bool:
|
|
"""True iff Fly machine identity is present AND the local Machines API socket exists.
|
|
|
|
Off-Fly (local dev, other clouds, tests) the watcher skips the quiesce:
|
|
the platform owns the freeze, so the gateway stays connected until it lands.
|
|
"""
|
|
return bool(
|
|
_env_str(environ, FLY_APP_NAME_ENV)
|
|
and _env_str(environ, FLY_MACHINE_ID_ENV)
|
|
and os.path.exists(FLY_API_SOCKET)
|
|
)
|
|
|
|
|
|
def suspend_self(
|
|
environ: Optional[dict] = None,
|
|
*,
|
|
socket_path: str = FLY_API_SOCKET,
|
|
timeout: float = 10.0,
|
|
) -> bool:
|
|
"""POST /v1/apps/{app}/machines/{id}/suspend on the local flaps socket.
|
|
|
|
No token needed — the socket is the credential. Returns True when flaps
|
|
accepted (2xx); on success the kernel freezes this process shortly after,
|
|
so treat as fire-and-forget. Never raises: a failed suspend just leaves the
|
|
machine running (fail-awake, never fail-frozen). stdlib-only on purpose —
|
|
a plain unix-socket HTTP/1.1 request, no async plumbing to freeze mid-await.
|
|
"""
|
|
app = _env_str(environ, FLY_APP_NAME_ENV)
|
|
machine_id = _env_str(environ, FLY_MACHINE_ID_ENV)
|
|
if not app or not machine_id:
|
|
logger.warning("scale-to-zero: suspend_self called without Fly machine identity")
|
|
return False
|
|
request = (
|
|
f"POST /v1/apps/{app}/machines/{machine_id}/suspend HTTP/1.1\r\n"
|
|
"Host: flaps\r\n"
|
|
"Content-Length: 0\r\n"
|
|
"Connection: close\r\n"
|
|
"\r\n"
|
|
)
|
|
try:
|
|
with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as sock:
|
|
sock.settimeout(timeout)
|
|
sock.connect(socket_path)
|
|
sock.sendall(request.encode("ascii"))
|
|
response = b""
|
|
while len(response) < 65536:
|
|
chunk = sock.recv(4096)
|
|
if not chunk:
|
|
break
|
|
response += chunk
|
|
except OSError as exc:
|
|
logger.warning("scale-to-zero: flaps suspend request failed: %s", exc)
|
|
return False
|
|
status_line = response.split(b"\r\n", 1)[0].decode("ascii", "replace")
|
|
parts = status_line.split()
|
|
ok = len(parts) >= 2 and parts[1].isdigit() and 200 <= int(parts[1]) < 300
|
|
if ok:
|
|
logger.info("scale-to-zero: machine suspend accepted by flaps (%s)", status_line)
|
|
else:
|
|
body = response.split(b"\r\n\r\n", 1)[-1][:500].decode("utf-8", "replace")
|
|
logger.warning(
|
|
"scale-to-zero: flaps suspend rejected: %s %s",
|
|
status_line,
|
|
json.dumps(body)[:500],
|
|
)
|
|
return ok
|