diff --git a/gateway/browser_control_artifacts.py b/gateway/browser_control_artifacts.py index 683957646e..433091549f 100644 --- a/gateway/browser_control_artifacts.py +++ b/gateway/browser_control_artifacts.py @@ -60,6 +60,7 @@ import time from dataclasses import dataclass from pathlib import Path from typing import Any, Callable, Optional +import contextlib logger = logging.getLogger(__name__) @@ -338,10 +339,8 @@ class ArtifactStore: except Exception: with self._lock: self._entries.pop(artifact_id, None) - try: + with contextlib.suppress(Exception): temp.unlink(missing_ok=True) - except Exception: - pass raise return receipt @@ -397,10 +396,8 @@ class ArtifactStore: for artifact_id, entry in list(self._entries.items()): if entry.receipt.expires_at <= now: self._entries.pop(artifact_id, None) - try: + with contextlib.suppress(OSError): entry.path.unlink(missing_ok=True) - except OSError: - pass removed += 1 for temp in self._root.glob(f"*{_TEMP_SUFFIX}"): try: @@ -435,10 +432,8 @@ class ArtifactStore: raise ArtifactNotFound(f"unknown artifact {artifact_id!r}") if entry.receipt.expires_at <= now: self._entries.pop(artifact_id, None) - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass raise ArtifactExpired(f"artifact {artifact_id!r} expired") if entry.receipt.scope_key != scope_key: raise ArtifactScopeMismatch( @@ -450,10 +445,8 @@ class ArtifactStore: for artifact_id, entry in list(self._entries.items()): if entry.receipt.expires_at <= now: self._entries.pop(artifact_id, None) - try: + with contextlib.suppress(OSError): entry.path.unlink(missing_ok=True) - except OSError: - pass def _artifact_path(self, artifact_id: str) -> Path: """Resolve a minted id strictly inside the controlled root.""" diff --git a/gateway/config.py b/gateway/config.py index 91f8479b66..2e9c9fc603 100644 --- a/gateway/config.py +++ b/gateway/config.py @@ -25,6 +25,7 @@ from gateway.shutdown_watchdog import ( DEFAULT_LOOP_WATCHDOG_TIMEOUT_S, ) from utils import is_truthy_value +import contextlib logger = logging.getLogger(__name__) @@ -2062,10 +2063,8 @@ def _env_int_extra(extra: Dict[str, Any], key: str, env: str) -> None: """Set an int extra from env; a non-integer value is silently ignored.""" raw = _getenv_str(env) if raw: - try: + with contextlib.suppress(ValueError): extra[key] = int(raw) - except ValueError: - pass def _env_home_channel( @@ -2627,10 +2626,8 @@ def _apply_env_overrides(config: GatewayConfig) -> None: for env, attr in (("SESSION_IDLE_MINUTES", "idle_minutes"), ("SESSION_RESET_HOUR", "at_hour")): raw = getenv(env) if raw: - try: + with contextlib.suppress(ValueError): setattr(config.default_reset_policy, attr, int(raw)) - except ValueError: - pass _enable_plugin_platforms_from_env(config) diff --git a/gateway/drain_control.py b/gateway/drain_control.py index 32d6c7e1d3..e1e323368a 100644 --- a/gateway/drain_control.py +++ b/gateway/drain_control.py @@ -42,6 +42,7 @@ from typing import Any, Optional from hermes_constants import get_hermes_home from utils import atomic_json_write +import contextlib _log = logging.getLogger(__name__) @@ -71,10 +72,8 @@ def current_instantiation_epoch() -> str: which disables the epoch check downstream — never fail-closed. """ boot_id = "" - try: + with contextlib.suppress(OSError): boot_id = Path("/proc/sys/kernel/random/boot_id").read_text(encoding="utf-8").strip() - except OSError: - pass pid1_start = "" try: diff --git a/gateway/hosted_room_links.py b/gateway/hosted_room_links.py index 83277cc61f..98841b0421 100644 --- a/gateway/hosted_room_links.py +++ b/gateway/hosted_room_links.py @@ -21,6 +21,7 @@ from gateway.hosted_room_peer import ( TransportSecurity, validate_room_link_url, ) +import contextlib MAX_LINKS = 512 @@ -164,10 +165,8 @@ def save_room_link(db_path: Path | str, link: StoredRoomLink) -> None: db_path, record=link.as_record(), max_links=MAX_LINKS ) if os.name == "posix": - try: + with contextlib.suppress(OSError): Path(db_path).chmod(0o600) - except OSError: - pass def mark_room_link_status( diff --git a/gateway/kanban_watchers.py b/gateway/kanban_watchers.py index 6ff4edc77b..46221c1316 100644 --- a/gateway/kanban_watchers.py +++ b/gateway/kanban_watchers.py @@ -21,6 +21,7 @@ from pathlib import Path from typing import Any, Callable, Optional from agent.i18n import t +import contextlib # Match the logger run.py uses (logging.getLogger(__name__) where __name__ == # "gateway.run") so extracted log records keep their original logger name. @@ -164,10 +165,8 @@ def _release_singleton_lock(handle) -> None: _release_file_lock(handle) except Exception: pass - try: + with contextlib.suppress(Exception): handle.close() - except Exception: - pass def _wake_scope_id(adapter: Any, sub: dict) -> Optional[str]: @@ -341,7 +340,7 @@ class GatewayKanbanWatchersMixin: ) active_platforms = { getattr(platform, "value", str(platform)).lower() - for platform in self.adapters.keys() + for platform in self.adapters } # Widen to every platform any secondary profile has live, # not just the default profile's. This is only a coarse @@ -359,7 +358,7 @@ class GatewayKanbanWatchersMixin: for _profile_adapter_map in getattr(self, "_profile_adapters", {}).values(): active_platforms.update( getattr(platform, "value", str(platform)).lower() - for platform in _profile_adapter_map.keys() + for platform in _profile_adapter_map ) if not active_platforms: logger.debug("kanban notifier: no connected adapters; skipping tick") @@ -1606,10 +1605,8 @@ class GatewayKanbanWatchersMixin: return None finally: if conn is not None: - try: + with contextlib.suppress(Exception): conn.close() - except Exception: - pass def _tick_once() -> "list[tuple[str, Optional[object]]]": """Run one dispatch_once per board. Returns (slug, result) pairs. @@ -1664,10 +1661,8 @@ class GatewayKanbanWatchersMixin: continue finally: if conn is not None: - try: + with contextlib.suppress(Exception): conn.close() - except Exception: - pass return False # Auto-decompose: turn fresh triage tasks into ready workgraphs diff --git a/gateway/memory_monitor.py b/gateway/memory_monitor.py index bacbbba34e..272eb6e771 100644 --- a/gateway/memory_monitor.py +++ b/gateway/memory_monitor.py @@ -37,6 +37,7 @@ import sys import threading import time from typing import Optional +import contextlib logger = logging.getLogger(__name__) @@ -205,10 +206,8 @@ def stop_memory_monitoring(timeout: float = 2.0) -> None: return # Final snapshot before teardown so "last RSS" is always in the log. - try: + with contextlib.suppress(Exception): log_memory_usage(prefix="shutdown") - except Exception: - pass _stop_event.set() thread = _monitor_thread @@ -216,10 +215,8 @@ def stop_memory_monitoring(timeout: float = 2.0) -> None: _stop_event = None # Join outside the lock so a stuck log call can't deadlock shutdown. - try: + with contextlib.suppress(Exception): thread.join(timeout=timeout) - except Exception: - pass logger.info("[MEMORY] Periodic memory monitoring stopped") diff --git a/gateway/pairing.py b/gateway/pairing.py index 7e7b4cb52f..693fb917d2 100644 --- a/gateway/pairing.py +++ b/gateway/pairing.py @@ -39,6 +39,7 @@ from hermes_constants import ( get_hermes_home, ) from utils import atomic_replace +import contextlib logger = logging.getLogger(__name__) @@ -304,20 +305,16 @@ def _sync_live_adapter_allowlist_remove(platform: str, user_id: str) -> None: if _adapter_platform_name(adapter) != platform_name: continue if hasattr(adapter, "_allow_from"): - try: + with contextlib.suppress(Exception): adapter._allow_from = _purge_allowlist_entries( set(adapter._allow_from or ()), platform_name, user_id ) - except Exception: - pass extra = getattr(getattr(adapter, "config", None), "extra", None) if isinstance(extra, dict) and "allow_from" in extra: - try: + with contextlib.suppress(Exception): extra["allow_from"] = _purge_allowlist_entries( extra.get("allow_from"), platform_name, user_id ) - except Exception: - pass def _sync_allowlist_remove(platform: str, user_id: str) -> None: @@ -426,10 +423,8 @@ def _secure_write(path: Path, data: str) -> None: except OSError: pass # Windows doesn't support chmod the same way except BaseException: - try: + with contextlib.suppress(OSError): os.unlink(tmp_path) - except OSError: - pass raise diff --git a/gateway/relay/adapter.py b/gateway/relay/adapter.py index a42bef5d83..f3b4091a62 100644 --- a/gateway/relay/adapter.py +++ b/gateway/relay/adapter.py @@ -338,9 +338,7 @@ class RelayAdapter(BasePlatformAdapter): platform = self._platform_by_chat.get(str(chat_id)) if platform is None: platform = getattr(desc, "platform", None) - if self._slack_unfurl_hints(platform): - return False - return True + return not self._slack_unfurl_hints(platform) def prefers_fresh_final_streaming( self, diff --git a/gateway/relay/media.py b/gateway/relay/media.py index 2ba218fd0a..986e743e81 100644 --- a/gateway/relay/media.py +++ b/gateway/relay/media.py @@ -36,7 +36,6 @@ import mimetypes import os import tempfile import urllib.error -import urllib.parse import urllib.request from pathlib import Path from typing import Optional diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index 416f5cbdc4..3af90f8e53 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -29,6 +29,7 @@ from gateway.platforms.base import MessageEvent, MessageType from gateway.session import SessionSource from gateway.relay.descriptor import CapabilityDescriptor from gateway.relay.transport import InboundHandler +import contextlib logger = logging.getLogger(__name__) @@ -77,10 +78,8 @@ def _env_disconnect_budget_s() -> float: budget = 5.0 raw = os.getenv("HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT", "").strip() if raw: - try: + with contextlib.suppress(ValueError): budget = max(0.0, float(raw)) - except ValueError: - pass return budget diff --git a/gateway/restart_loop_guard.py b/gateway/restart_loop_guard.py index 78ed396800..433836845a 100644 --- a/gateway/restart_loop_guard.py +++ b/gateway/restart_loop_guard.py @@ -29,6 +29,7 @@ import time from typing import List, Optional from hermes_constants import get_hermes_home +import contextlib logger = logging.getLogger("gateway.run") @@ -119,10 +120,8 @@ def record_restart_interrupted_boot( def clear() -> None: """Remove the persisted boot log (used on clean shutdown / by tests).""" - try: + with contextlib.suppress(OSError): _state_path().unlink(missing_ok=True) - except OSError: - pass def check_and_record( diff --git a/gateway/run.py b/gateway/run.py index 1656a40cfd..d05b1d49ec 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -1,16 +1,7 @@ -""" -Gateway runner - entry point for messaging platform integrations. +"""Gateway runner - entry point for messaging platform integrations. -This module provides: -- start_gateway(): Start all configured platform adapters -- GatewayRunner: Main class managing the gateway lifecycle - -Usage: - # Start the gateway - python -m gateway.run - - # Or from CLI - python cli.py --gateway +Provides ``start_gateway()`` (start all configured adapters) and ``GatewayRunner`` (lifecycle). +Run via ``python -m gateway.run`` or ``python cli.py --gateway``. """ # IMPORTANT: hermes_bootstrap must be the very first import — UTF-8 stdio @@ -18,10 +9,8 @@ Usage: try: import hermes_bootstrap # noqa: F401 except ModuleNotFoundError: - # Graceful fallback when hermes_bootstrap isn't registered in the venv - # yet — happens during partial ``hermes update`` where git-reset landed - # new code but ``uv pip install -e .`` didn't finish. Missing bootstrap - # means UTF-8 stdio setup is skipped on Windows; POSIX is unaffected. + # Partial ``hermes update`` (git reset landed, ``uv pip install -e .`` didn't) leaves the + # bootstrap unregistered; without it Windows skips UTF-8 stdio setup, POSIX is unaffected. pass import asyncio @@ -70,35 +59,29 @@ from agent.turn_context import ( from hermes_cli.config import _is_ssh_remote_tilde_cwd, cfg_get from hermes_cli.fallback_config import get_fallback_chain -# --- Agent cache tuning --------------------------------------------------- -# Bounds the per-session AIAgent cache to prevent unbounded growth in long-lived gateways (each -# AIAgent holds LLM clients, tool schemas, memory providers, etc.). LRU order + idle TTL eviction -# are enforced from _enforce_agent_cache_cap() and _session_expiry_watcher() below. +# Per-session AIAgent cache bounds (each agent holds LLM clients, tool schemas, memory providers); +# LRU cap + idle TTL eviction are enforced by _enforce_agent_cache_cap()/_session_expiry_watcher(). _AGENT_CACHE_MAX_SIZE = 128 _AGENT_CACHE_IDLE_TTL_SECS = 3600.0 # evict agents idle for >1h _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT = 30.0 -# Telegram cold polling now proves one real getUpdates round trip before connect -# returns. Leave enough outer budget for initialize/deleteWebhook/start_polling -# wall deadlines plus readiness; other platforms retain the 30s isolation bound. +# Telegram connect proves a real getUpdates round trip, so its budget must cover the +# initialize/deleteWebhook/start_polling wall deadlines plus readiness; others keep the 30s bound. _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT = 180.0 -# Cold-start cap for Telegram: the initial connect awaited before the gateway reaches `running` must -# not spend the full 180s budget — an unreachable Telegram would hold EVERY platform's serving state -# hostage for the whole window. +# Telegram's initial connect (awaited before the gateway reaches `running`) must not spend the full +# 180s: an unreachable Telegram would hold EVERY platform's serving state hostage. _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT = 45.0 _ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT = 5.0 -# End reasons that mean the USER deliberately closed this thread of work (/new -> session_reset / -# new_session, an explicit exit, or a /switch). Shared by _classify_completion_target and -# _resolve_async_delegation_session so they can never disagree: a reason the classifier calls -# "deliver" that the resolver drops would be acked at adapter acceptance and then silently lost. +# End reasons meaning the USER deliberately closed this thread (/new, explicit exit, /switch). Shared +# by _classify_completion_target and _resolve_async_delegation_session so they can never disagree: +# a reason the classifier "delivers" but the resolver drops would be acked and then silently lost. _USER_BOUNDARY_END_REASONS = ( "session_reset", "user_exit", "session_switch", "new_session", ) -# Round-2 #2: upper bound on a single stall-notify adapter.send so a wedged -# transport cannot block the session-stall watcher pass (notify-only path; -# on timeout the latch stays clear and the next tick retries). +# Bound on a single stall-notify adapter.send so a wedged transport cannot block the stall watcher +# pass; on timeout the latch stays clear and the next tick retries. _STALL_NOTIFY_SEND_TIMEOUT_SECONDS = 15.0 _GATEWAY_PROXY_SSE_BUFFER_MAX_CHARS = 16 * 1024 * 1024 _TELEGRAM_COMMAND_MENTION_RE = re.compile(r"(?)\s+\d[\d,]*\s+messages,\s+retrying" r"|compressed\s+~[\d,]+\s+(?:→|->)\s+~[\d,]+\s+tokens,\s+retrying" @@ -142,17 +123,12 @@ _TELEGRAM_NOISY_STATUS_RE = re.compile( _HYGIENE_COOLDOWN_LADDER_MULTIPLIERS = (1, 3, 9) -# Absolute ceiling on an escalated hygiene cooldown, mirroring _RECONNECT_BACKOFF_CAP above: with an -# operator-raised base the multiplier ladder alone would reach 9h (base 3600 -> 32400s), which is -# indistinguishable from "compaction silently switched off". +# Ceiling on an escalated hygiene cooldown (cf. _RECONNECT_BACKOFF_CAP): with an operator-raised +# base the ladder alone reaches 9h, indistinguishable from "compaction silently switched off". _HYGIENE_COOLDOWN_MAX_SECONDS = 3600.0 -# Flat retry-after recorded when a hygiene compression is ABANDONED because -# the turn-hold budget expired while the summary was still streaming (not a -# failure — the compressor was healthy, we just could not keep holding the -# user's turn). Spaces out re-attempts so sustained traffic does not spawn, -# hold, and cancel a fresh compressor on every single turn; deliberately -# outside the failure-streak ladder above. +# Flat retry-after when hygiene compression is ABANDONED by turn-hold expiry (not a failure, so +# outside the streak ladder); keeps sustained traffic from spawn/hold/cancelling one every turn. _HYGIENE_TURNHOLD_RETRY_SECONDS = 60.0 @@ -163,14 +139,9 @@ def _hygiene_cooldown_for_failure( ) -> float: """Bump the hygiene failure streak and return the escalated cooldown. - This is a MULTIPLIER ladder (x1, x3, x9) over the operator's configured - ``hygiene_failure_cooldown_seconds``, clamped to ``_HYGIENE_COOLDOWN_MAX_SECONDS``, so a - tuned base is preserved as rung 1. - - Needed because the in-agent ladder (``ContextCompressor.record_timeout_failure``) keys off an - in-memory streak that ``bind_session_state`` zeroes; hygiene builds a FRESH ``AIAgent`` per - run, so from here that streak is always 0. The streak is instead mirrored to SQLite by - rotation-stable ``session_key`` so it outlives the per-run agent and gateway restarts. + Multiplier ladder (x1, x3, x9) over the configured base, clamped to the max, so a tuned base + stays rung 1. Hygiene's per-run ``AIAgent`` is fresh (in-memory streak always 0), so the streak + lives in SQLite keyed by rotation-stable ``session_key``. """ streak = 1 state = None @@ -203,8 +174,7 @@ def _hygiene_cooldown_for_failure( def _reset_hygiene_failure_streak(gateway, session_key: str) -> None: """Clear the hygiene failure streak after a compression that reduced context. - Peeks rather than get-or-creates: writing a 0 that is already 0 must not - materialise a ``_sessions`` entry (those are never evicted). + Peeks, never get-or-creates: a no-op 0 write must not create a never-evicted ``_sessions`` row. """ try: state = gateway._peek_session_state(session_key) @@ -232,17 +202,12 @@ def hygiene_compaction_recovered( approx_tokens: int, new_tokens: int, ) -> bool: - """True when a hygiene run actually recovered the session. + """True when a hygiene run actually recovered the session (extracted to be unit testable). - Extracted from ``_handle_message_with_agent`` so the decision is unit testable: it previously - lived inline in a ~2000-line async method, and the only way to pin it was a source-reading - test — which AGENTS.md bans outright, naming this file. - - "Recovered" requires all three: the compressor did not abort; the transcript was actually - rewritten (rotated or compacted in place — the no-op path reuses pre-compression counts, so - numbers alone would read it as success); and the request materially shrank per - :func:`compression_made_progress` (a bare ``<`` would miss row-count wins and count the - 30-50% estimate noise as one). + Requires all three: the compressor did not abort; the transcript was actually rewritten (the + no-op path reuses pre-compression counts, so numbers alone read as success); and the request + materially shrank per :func:`compression_made_progress` (a bare ``<`` misses row-count wins + and counts 30-50% estimate noise as one). """ if aborted: return False @@ -292,24 +257,18 @@ async def run_codex_hygiene_compaction( ) -> str: """Session hygiene for ``codex_app_server`` sessions. - On this runtime the model's real context is the app-server's server-side thread; the local - transcript is a never-replayed mirror. So rewriting the mirror shrinks nothing (permanent - no-op) and evicting the cached live agent destroys the only real context (next turn starts an - EMPTY thread — abrupt amnesia). Hygiene must therefore compact the LIVE agent's thread via - ``thread/compact/start`` and KEEP that agent cached: never build a detached compressor, never - evict here. Modes ``native``/``off`` return without touching thread or transcript and must - never fall back to the local compressor. - - Returns an outcome tag for logging/tests: ``compacted``, - ``skipped:`` or ``failed:``. + The real context is the server-side thread; the local transcript is a never-replayed mirror, so + rewriting it shrinks nothing and evicting the live agent starts the next turn on an EMPTY thread. + So: compact the LIVE agent via ``thread/compact/start``, keep it cached, never build a detached + compressor. ``native``/``off`` skip without falling back to the local compressor. + Returns ``compacted``, ``skipped:`` or ``failed:``. """ mode = str(auto_mode or "native").lower() if mode not in {"native", "hermes", "off"}: mode = "native" if mode != "hermes": - # native: the app-server compacts on its own schedule; off: the operator disabled Hermes- - # initiated automatic compaction. A local transcript fallback is wrong in EVERY mode (it - # cannot shrink the thread), so both are a clean skip — without the detached path's eviction. + # native = app-server compacts itself; off = operator disabled it. A local transcript + # fallback cannot shrink the thread in any mode, so both skip cleanly with no eviction. return f"skipped:mode={mode}" agent = None @@ -326,9 +285,8 @@ async def run_codex_hygiene_compaction( entry = None agent = entry[0] if isinstance(entry, tuple) and entry else entry if agent is None or agent is _AGENT_PENDING_SENTINEL: - # No live agent → no live thread → nothing real to compact. The - # mirror-only rewrite the detached path would perform is exactly the - # no-op this function exists to remove, so skip honestly instead. + # No live agent → no live thread; the detached path's mirror-only rewrite would be the + # exact no-op this function exists to remove, so skip honestly. return "skipped:no-cached-agent" if getattr(agent, "_codex_session", None) is None: return "skipped:no-live-thread" @@ -360,9 +318,8 @@ async def run_codex_hygiene_compaction( timeout=max(float(timeout_seconds), 1.0), ) except asyncio.TimeoutError: - # The executor thread keeps running (compact_thread has its own RPC - # timeouts); brake per-turn retries so a wedged app-server does not - # re-trigger a compaction attempt on every message. + # The executor thread keeps running (compact_thread has its own RPC timeouts); brake + # retries so a wedged app-server does not re-trigger compaction on every message. if failure_cooldown_seconds >= 0: _record_hygiene_cooldown( gateway, @@ -388,15 +345,12 @@ async def run_codex_hygiene_compaction( count_after = getattr(compressor, "compression_count", 0) if count_after > count_before: - # A native compaction boundary was recorded on the live agent - # (thread compacted server-side; transcript intentionally NOT - # rewritten — state.db records the boundary, the mirror stays - # intact and the agent stays cached). + # Native boundary recorded: thread compacted server-side, transcript intentionally NOT + # rewritten (state.db holds the boundary, mirror intact, agent stays cached). _reset_hygiene_failure_streak(gateway, session_key) return "compacted" - # compress_context returned without recording a boundary: an internal - # skip (its own failure cooldown) or a compaction error — the codex - # route already persisted its own failure cooldown in that case. + # No boundary recorded: an internal skip or a compaction error; either way the codex route + # already persisted its own failure cooldown. return "failed:no-boundary" def hygiene_wait_should_extend( @@ -409,9 +363,8 @@ def hygiene_wait_should_extend( ) -> bool: """Whether the hygiene host should keep waiting for a slow summary. - A cancelled commit fence cannot produce a commit (#96953): extending the - wait up to the 600s ceiling only queues inbound messages behind a doomed - attempt. Stop extending immediately so the turn can continue. + A cancelled commit fence cannot produce a commit: extending the wait only queues inbound + messages behind a doomed attempt, so stop extending immediately. """ if fence_cancelled: return False @@ -426,11 +379,8 @@ def _record_hygiene_cooldown( ) -> None: """Persist a session-hygiene compression-failure cooldown to the state DB. - Uses the same ``compression_failure_cooldown_until`` column and - ``record_compression_failure_cooldown`` method that the in-conversation compression path - (``agent/context_compressor.py``) already uses, so the cooldown survives gateway restarts. - ``error`` is forwarded because the recorder writes ``compression_failure_error`` - UNCONDITIONALLY — omitting it clobbers any reason the in-conversation path recorded to NULL. + Shares the in-conversation path's column/recorder so it survives restarts. ``error`` must be + forwarded: the recorder writes ``compression_failure_error`` UNCONDITIONALLY (else NULL clobber). """ import time as _time session_db = getattr(gateway, "_session_db", None) @@ -449,8 +399,8 @@ def _record_hygiene_cooldown( def _status_template_to_regex(template: str) -> str: """Compile a compression status template constant into a regex source. - Literal text is escaped verbatim (the constants ARE the wording, so drift cannot silently - diverge from this matcher); each ``{field}`` placeholder becomes a numeric-ish pattern. + Literal text is escaped verbatim so wording drift cannot silently diverge from the matcher; + each ``{field}`` placeholder becomes a numeric-ish pattern. """ parts = re.split(r"\{[^{}]*\}", template) return r"[\d,]+".join(re.escape(part) for part in parts) @@ -478,12 +428,9 @@ _COMPRESSION_PROGRESS_STATUS_RE = re.compile( def _gateway_compression_progress_notices_enabled() -> bool: - """True when the user opted into routine compression progress notices. + """True when ``compression.progress_notices`` is on (default False: chat is silent by design). - Reads ``compression.progress_notices`` from the gateway's raw YAML config. Default False — - routine compression stays silent-by-design on chat platforms unless explicitly enabled. Read - live (mtime-cached) so a config edit on a running gateway takes effect on the next status. - Fail-closed: any config read error keeps the silent default. + Read live (mtime-cached) so a config edit applies at the next status; fail-closed on read error. """ try: config = _load_gateway_config() @@ -499,9 +446,8 @@ def _gateway_compression_progress_notices_enabled() -> bool: pass return False -# Surfaces that consume gateway text programmatically (CLI/TUI "local" diagnostics, API JSON, -# webhook payloads) and therefore must keep RAW status/error text. Widens #28533's Telegram-only -# filter to all chat gateways. Fail-closed: unknown/empty platform -> chat. +# Surfaces that consume gateway text programmatically (local diagnostics, API JSON, webhooks) and +# so must keep RAW status/error text. Fail-closed: unknown/empty platform -> chat. _GATEWAY_RAW_TEXT_PLATFORMS = frozenset( {"local", "api_server", "webhook", "msgraph_webhook"} ) @@ -584,8 +530,7 @@ _GATEWAY_SECRET_PATTERNS = ( def _ensure_windows_gateway_venv_imports() -> None: """Make detached Windows gateway runs see the Hermes venv packages. - Patch the live process before MCP discovery so tool injection does not depend on every - launcher preserving PYTHONPATH perfectly. + Patched before MCP discovery so tool injection does not depend on launchers preserving PYTHONPATH. """ if sys.platform != "win32": return @@ -654,11 +599,9 @@ def _interim_metadata( ) -> Dict[str, Any]: """Mark a mid-turn status/advisory send as NOT the turn-final. - Stream-is-the-message adapters (relay Slack native streaming) intercept the first unmarked - send to an armed (chat, turn) key and seal the live stream with its content. Every gateway- - side send that can fire mid-turn (heartbeats, inactivity warnings, approval fallbacks, - background-review notices) MUST carry this marker or it seals the user's answer stream with - status text. The marker is gateway-internal; adapters strip it before the wire. + Stream-is-the-message adapters seal the live stream with the first unmarked send to an armed + (chat, turn) key, so every mid-turn gateway send MUST carry this marker or it seals the user's + answer stream with status text. Gateway-internal; adapters strip it before the wire. """ merged = dict(metadata or {}) merged["_interim_send"] = True @@ -671,12 +614,10 @@ def _seed_hygiene_system_prompt( ) -> bool: """Keep gateway hygiene from rebuilding a live session's system prompt. - The hygiene helper runs outside the live session's fully initialized prompt environment - (hygiene-only platform marker, no platform context files, memory provider only when required). - Compression is allowed to persist a system prompt, so letting that helper rebuild one would - strip external provider blocks from the live session. Seed the exact persisted prompt - instead. When no usable prompt can be restored, seed an empty cache entry; the real turn - rebuilds either form with its fully initialized providers. + The hygiene helper lacks the live session's fully initialized prompt environment, and + compression may persist a system prompt, so a rebuilt one would strip external provider blocks. + Seed the exact persisted prompt, or an empty cache entry when none is usable; the real turn + rebuilds either with fully initialized providers. """ stored_prompt = "" if isinstance(session_row, dict): @@ -689,11 +630,9 @@ def _seed_hygiene_system_prompt( def _is_transient_network_error(exc: BaseException) -> bool: - """Return True for transient network errors safe to log + swallow. + """True for transient network errors safe to log + swallow (the next poll recovers; never crash). - These are by definition transient — the next poll cycle or user action recovers — so they - must never crash the process. Walk the exception cause chain so wrapped errors (e.g. PTB's - ``NetworkError`` wrapping ``httpx.ConnectError``) are still classified. + Walks the cause chain so wrapped errors (PTB ``NetworkError`` over ``httpx.ConnectError``) match. """ seen: set[int] = set() cur: Optional[BaseException] = exc @@ -729,11 +668,9 @@ def _is_transient_network_error(exc: BaseException) -> bool: def _gateway_loop_exception_handler( loop: "asyncio.AbstractEventLoop", context: Dict[str, Any] ) -> None: - """Loop-level safety net for transient network errors. + """Loop-level safety net for transient network errors (installed once by ``start_gateway``). - Installed once during :func:`start_gateway`. Logs at WARNING with full traceback so the - originating call site stays diagnosable; non-transient errors are forwarded to the default - loop handler so real bugs still surface. + Logs WARNING with traceback; non-transient errors go to the default handler so real bugs surface. """ exc = context.get("exception") if exc is not None and _is_transient_network_error(exc): @@ -759,12 +696,9 @@ def _gateway_loop_exception_handler( def _redact_gateway_user_facing_secrets(text: str) -> str: """Secret redaction before text can leave the gateway. - Delegates to ``agent.redact.redact_sensitive_text`` (the same redactor used for logs, tool - output and approval prompts) so the chat path masks the full credential set, not a divergent - subset; ``force=True`` redacts even when ``security.redact_secrets`` is off. - The narrow ``_GATEWAY_SECRET_PATTERNS`` set runs as a belt-and-suspenders second pass so - nothing the gateway historically caught can regress, and so redaction still degrades - gracefully if the import ever fails. + Delegates to the shared ``redact_sensitive_text`` (full credential set) with ``force=True`` so + it holds even when ``security.redact_secrets`` is off; ``_GATEWAY_SECRET_PATTERNS`` is a second + pass so redaction degrades gracefully if that import fails. """ redacted = str(text or "") try: @@ -783,10 +717,8 @@ def _redact_gateway_user_facing_secrets(text: str) -> str: def _redact_approval_command(cmd: "str | None") -> str: """Redact credentials from a command before it goes into an approval prompt. - Tirith's *findings* are already redacted, but the gateway approval prompt is built from the - raw command string, so a credential-shaped value Tirith flagged would otherwise be echoed - verbatim to the chat platform. ``force=True`` so the prompt is redacted even when - ``security.redact_secrets`` is off. Module-level so the wiring is unit-testable. + The prompt is built from the raw command, so a Tirith-flagged credential would otherwise echo + verbatim to chat; ``force=True`` holds even when ``security.redact_secrets`` is off. """ from agent.redact import redact_sensitive_text @@ -871,12 +803,10 @@ _GATEWAY_PROVIDER_ERROR_SHAPE_RE = re.compile( def _looks_like_gateway_provider_error(text: str) -> bool: - """True when text is infrastructure/provider failure, not normal content. + """True when text is a provider failure envelope, not normal content. - Two heuristics combined so the rewrite only fires on actual provider error envelopes, not on - assistant prose that happens to mention an HTTP status code: the text is short (real - envelopes are 1-3 lines) AND the error marker appears at the start of the message, not - buried mid-paragraph in an explanation. + Text must be short (envelopes are 1-3 lines) AND start with the marker, so prose that merely + mentions a status code does not match. """ if not text: return False @@ -889,22 +819,16 @@ def _looks_like_gateway_provider_error(text: str) -> bool: def _sanitize_gateway_final_response(platform: Any, text: str) -> str: - """Sanitize final gateway replies before sending them to chat surfaces. - - Every human-facing chat surface (Telegram, WhatsApp, Discord, Slack, Signal, Matrix, plugin - platforms, etc.) should receive concise, safe provider failure categories with secrets - redacted instead of raw HTTP bodies, request IDs, leaked credentials, or policy text. + """Sanitize final gateway replies for chat surfaces: concise, secret-redacted provider failure + categories instead of raw HTTP bodies, request IDs, leaked credentials, or policy text. """ if not text: return text if _gateway_surface_passes_raw_text(platform): return text - # Lone UTF-16 surrogates (U+D800–U+DFFF) in model output crash chat surfaces downstream: - # Telegram's ``utf16_len`` length check and Signal formatting both ``.encode()`` the reply and - # raise UnicodeEncodeError before any send. Stored history is already sanitized upstream; this - # boundary is the last line of defense for legacy/plugin delivery paths that hand us raw text. - # Raw-text/programmatic surfaces above keep passthrough — their JSON consumers escape safely. + # Lone UTF-16 surrogates make Telegram/Signal ``.encode()`` raise before any send. Last line of + # defense for legacy/plugin paths; the raw-text surfaces above pass through (JSON escapes safely). from agent.message_sanitization import _sanitize_surrogates text = _sanitize_surrogates(str(text)) @@ -923,8 +847,7 @@ def _sanitize_gateway_final_response(platform: Any, text: str) -> str: def _prepare_gateway_status_message(platform: Any, event_type: str, message: str) -> Optional[str]: """Filter/sanitize agent status callbacks before platform delivery. - Local/CLI sessions keep the raw diagnostic stream. Messaging gateway - surfaces should not receive transient auxiliary/compression chatter. + Local/CLI keep the raw diagnostic stream; messaging surfaces drop transient aux/compression noise. """ text = str(message or "").strip() if not text: @@ -934,10 +857,9 @@ def _prepare_gateway_status_message(platform: Any, event_type: str, message: str text = _redact_gateway_user_facing_secrets(text) if _TELEGRAM_NOISY_STATUS_RE.search(text): - # Opt-in #52995: `compression.progress_notices: true` lets ROUTINE compression progress - # statuses through to chat platforms. Membership is derived from the template constants, so - # non-compression noise (aux failures, retry chatter) stays suppressed even when the gate is - # open. Default False keeps the silent-by-design behavior byte-identical. + # Opt-in `compression.progress_notices` lets ROUTINE compression progress through; membership + # comes from the template constants, so other noise (aux failures, retry chatter) stays + # suppressed even when the gate is open. if not ( _gateway_compression_progress_notices_enabled() and _COMPRESSION_PROGRESS_STATUS_RE.search(text) @@ -949,23 +871,17 @@ def _prepare_gateway_status_message(platform: Any, event_type: str, message: str def render_notice_line(notice) -> str: - """Render an AgentNotice to a single plaintext line for messaging platforms. + """Render an AgentNotice to a single plaintext line (messaging has no status bar: one-shot push). - Messaging has no persistent status bar (unlike the TUI), so a notice is a one-shot standalone - push. The notice policy already bakes the level glyph (⚠ / • / ✕ / ✓) into the text, and the - TUI + CLI REPL render that text verbatim — so we emit it as-is here too; prepending a per-level - glyph would DOUBLE it. Plaintext only (no markdown) so it renders uniformly across platforms. - Fail-soft: a malformed/empty notice degrades to "" rather than raising in the callback path. + The notice policy already bakes the level glyph into the text — prepending one would DOUBLE it. + Fail-soft: a malformed/empty notice degrades to "" rather than raising. """ return str(getattr(notice, "text", "") or "").strip() async def _send_or_update_status_coro(adapter, chat_id, status_key, content, metadata): - """Route a status message through adapter.send_or_update_status when supported. - - Issue #30045: adapters that implement send_or_update_status (currently - Telegram) edit the previous bubble for the same status_key instead of - appending a new one. Adapters without the method fall back to plain send. + """Route a status through adapter.send_or_update_status when supported (edits the previous + bubble for the same status_key instead of appending); otherwise fall back to plain send. """ sender = getattr(adapter, "send_or_update_status", None) if callable(sender): @@ -976,11 +892,9 @@ async def _send_or_update_status_coro(adapter, chat_id, status_key, content, met def _approval_send_outcome(future, timeout: float) -> str: """Classify an approval prompt send as ``sent`` / ``failed`` / ``ambiguous``. - ``ambiguous`` == the scheduling future timed out; the card may well have posted (late - connector ack), and treating that as failure re-sent the card repeatedly in live testing. - Callers must treat ``ambiguous`` as possibly-delivered: keep the prompt registration alive - and do NOT re-send or fall back — the boundary rule is that only a DEFINITIVE failure (error - result / non-timeout exception / no future) re-asks. Definitive failures log their detail here. + ``ambiguous`` = scheduling future timed out but the card may have posted (late connector ack): + keep the registration alive, do NOT re-send or fall back. Only a DEFINITIVE failure (error + result / non-timeout exception / no future) re-asks; those log their detail here. """ if future is None: logger.warning("Prompt send failed: no scheduling future (loop unavailable)") @@ -1003,14 +917,9 @@ def _approval_send_outcome(future, timeout: float) -> str: def _clarify_send_disposition(fut, *, session_key: str, clarify_mod) -> "str | None": """Decide whether a clarify prompt send aborts the wait, per the boundary rule. - Same physics as the exec-approval card: the scheduling future can hit its deadline while the - clarify card HAS already posted (late connector ack); treating that as definitive cleared the - registration under a rendered card. Only a DEFINITIVE failure tears down the registration and - aborts; ``ambiguous`` keeps it armed and proceeds to the bounded wait (whose response timeout - already covers the truly-lost-card case). - - Returns the abort sentinel string on definitive failure, else ``None`` - (proceed to ``wait_for_response``). + As with exec-approval, the scheduling future can time out while the card HAS posted. Only a + DEFINITIVE failure tears down the registration; ``ambiguous`` stays armed and proceeds to the + bounded wait (its response timeout covers a lost card). Returns the abort sentinel or ``None``. """ outcome = _approval_send_outcome(fut, timeout=15) if outcome == "failed": @@ -1051,10 +960,9 @@ def _resolve_progress_thread_id( ) -> Optional[str]: """Return thread/root ID that progress/status bubbles should target. - ``reply_in_thread=False`` (Slack ``platforms.slack.extra.reply_in_thread``) disables the - synthetic-thread fallback: progress messages must not create a thread the final flat reply - would then inherit. A source.thread_id equal to the event's own message id is the adapter's - synthetic session-keying thread, not a real one — treat it as "no thread" too. + ``reply_in_thread=False`` (Slack) disables the synthetic-thread fallback: progress messages + must not create a thread the final flat reply would inherit. A source.thread_id equal to the + event's own message id is the adapter's synthetic session-keying thread — treat as no thread. """ platform_value = getattr(platform, "value", platform) platform_key = str(platform_value or "").lower() @@ -1096,9 +1004,8 @@ def _resolve_gateway_display_bool( ) -> bool: """Resolve a boolean display setting with optional platform-only opt-in. - Some display features expose assistant scratch text rather than deliberate user-facing - output. For high-noise threaded chat surfaces such as Mattermost, a global opt-in is too - broad: they must be enabled with an explicit display.platforms.. override. + Scratch-text features are too noisy for threaded surfaces (Mattermost) under a global opt-in, + so they require an explicit display.platforms.. override. """ current_platform = _gateway_platform_value(platform or platform_key) platform_only = { @@ -1124,11 +1031,9 @@ def _resolve_gateway_display_bool( def _telegramize_command_mentions(text: str, platform: Any) -> str: - """Rewrite slash-command mentions to Telegram-valid command names. + """Rewrite slash-command mentions to Telegram-valid names (lowercase, digits, underscores only). - Telegram Bot API command names allow only lowercase letters, digits, and - underscores. Keep other platform renderings unchanged, but normalize - Telegram help text so command mentions remain clickable/valid there. + Other platforms' renderings are left unchanged. """ platform_value = getattr(platform, "value", platform) if platform_value != "telegram": @@ -1143,44 +1048,25 @@ def _telegramize_command_mentions(text: str, platform: Any) -> str: return _TELEGRAM_COMMAND_MENTION_RE.sub(_replace, text) -# Only auto-continue interrupted gateway turns while the interruption is fresh. -# Stale tool-tail/resume markers can otherwise revive an unrelated old task -# after a gateway restart when the user's next message starts new work. -# -# The freshness signal is the timestamp of the last transcript row, which -# ``hermes_state.get_messages`` carries on every persisted message. This -# handles the two auto-continue cases uniformly: -# * resume_pending (gateway restart/shutdown watchdog marked the session) -# * tool-tail (last persisted message is a tool result the agent -# never got to reply to) -# In both cases "when did we last do anything on this transcript" is the -# correct freshness question, so one signal replaces two divergent ones. -# -# Default window: 1 hour. This comfortably covers ``agent.gateway_timeout`` -# (30 min default) plus runtime slack — a legitimate long-running turn that -# gets interrupted near its timeout boundary and is resumed shortly after -# is still classified fresh. Override via -# ``config.yaml`` ``agent.gateway_auto_continue_freshness``. +# Auto-continue interrupted turns only while fresh (last transcript row timestamp), else stale +# tool-tail/resume_pending markers revive an unrelated old task after a restart. 1h covers +# ``agent.gateway_timeout`` (30 min) plus slack; override: ``agent.gateway_auto_continue_freshness``. _AUTO_CONTINUE_FRESHNESS_SECS_DEFAULT = 60 * 60 -# Default bound for how long ``_finish_startup_restore`` waits on boot auto-resume turns before -# releasing the inbound gate (see ``_startup_restore_drain_timeout_secs``). Override via -# ``config.yaml`` ``agent.gateway_startup_restore_drain_timeout``. +# How long ``_finish_startup_restore`` waits on boot auto-resume turns before releasing the inbound +# gate. Override: ``agent.gateway_startup_restore_drain_timeout``. _STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT = 30.0 -# Default bound for the boot-time turn-machinery warm-up. Without it a message arriving right after -# boot was served with a skeleton system prompt (cold run_agent/model_tools import graph, no tool -# schemas). The warm-up runs BEFORE the gate opens so the first inbound turn starts with initialized -# machinery; the bound keeps a wedged init from making the gateway permanently unavailable. -# Override via ``agent.gateway_startup_warmup_timeout`` (non-positive disables warm-up). +# Bound on the boot warm-up that runs BEFORE the gate opens (so the first turn is not served a +# skeleton system prompt) — keeps a wedged init from making the gateway permanently unavailable. +# Override: ``agent.gateway_startup_warmup_timeout`` (non-positive disables warm-up). _STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT = 20.0 def _coerce_gateway_timestamp(value: Any) -> Optional[float]: """Best-effort conversion of stored gateway timestamps to epoch seconds. - Missing/unparseable timestamps return None so legacy transcripts keep the historical auto- - continue behaviour instead of being silently dropped. + Missing/unparseable -> None, so legacy transcripts keep auto-continuing instead of being dropped. """ if value is None: return None @@ -1210,65 +1096,38 @@ def _coerce_gateway_timestamp(value: Any) -> Optional[float]: def _auto_continue_freshness_window() -> float: """Return the configured auto-continue freshness window in seconds. - Thin wrapper over the canonical implementation in ``gateway.session`` (shared with the - routing-time zombie gate in ``get_or_create_session``). Falls back to the module default when - unset or malformed; non-positive values disable the freshness gate. Kept here so existing call - sites and test patches importing it from ``gateway.run`` continue to work. + Thin wrapper over ``gateway.session`` kept so ``gateway.run`` imports/test patches keep working. + Falls back to the module default when unset/malformed; non-positive disables the gate. """ from gateway.session import auto_continue_freshness_window return auto_continue_freshness_window() def _startup_restore_drain_timeout_secs() -> float: - """Max seconds ``_finish_startup_restore`` waits on boot auto-resume turns before releasing the - inbound gate and draining the queue. + """Max seconds ``_finish_startup_restore`` waits on boot auto-resume turns before opening the + inbound gate (all inbound is QUEUED until then). Non-positive disables the bound. - While startup restore is in progress the gateway QUEUES every inbound message - (``_queue_startup_restore_event``) instead of processing it, so no channel gets a reply until - the gate opens. This bounds that wait. Duplicate-agent safety does NOT depend on the wait: - ``_schedule_resume_pending_sessions`` claims each session's ``_running_agents`` slot - SYNCHRONOUSLY, so a drained message queues behind a still-running resume turn rather than - spawning a second agent. Non-positive disables the bound. + Duplicate-agent safety does NOT depend on it: ``_schedule_resume_pending_sessions`` claims + ``_running_agents`` SYNCHRONOUSLY, so a drained message queues behind a running resume turn. """ - raw = os.environ.get("HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT") - if raw is None or raw == "": - return float(_STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT) - try: - return float(raw) - except (TypeError, ValueError): - return float(_STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT) + return _float_env("HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT", _STARTUP_RESTORE_DRAIN_TIMEOUT_SECS_DEFAULT) def _startup_warmup_timeout_secs() -> float: - """Max seconds the boot warm-up may hold the inbound gate shut. + """Max seconds the boot warm-up (``_warm_turn_prerequisites``) may hold the inbound gate shut. - ``GatewayRunner._warm_turn_prerequisites`` initializes the agent-side turn machinery BEFORE - ``_finish_startup_restore`` opens the inbound gate, so a message arriving seconds after boot - can no longer be served with a skeleton system prompt (no context tier, no tool schemas). - Bounded so a wedged import/probe can never make the gateway permanently unavailable — on - timeout the gate opens anyway and the warm-up finishes in the background. Non-positive - disables the warm-up. + Bounded so a wedged import/probe cannot wedge the gateway: on timeout the gate opens anyway and + the warm-up finishes in the background. Non-positive disables it. """ - raw = os.environ.get("HERMES_STARTUP_WARMUP_TIMEOUT") - if raw is None or raw == "": - return float(_STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT) - try: - return float(raw) - except (TypeError, ValueError): - return float(_STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT) + return _float_env("HERMES_STARTUP_WARMUP_TIMEOUT", _STARTUP_WARMUP_TIMEOUT_SECS_DEFAULT) def _warm_turn_machinery_sync() -> int: - """Synchronously initialize the turn prerequisites a first turn needs. + """Synchronously initialize first-turn prerequisites (executor thread); returns the schema count. - Runs on an executor thread from ``_warm_turn_prerequisites``. Covers exactly the lazy init - observed inside skeleton turns: the ``run_agent`` import graph (otherwise only pulled in - lazily per request), ``model_tools.get_tool_definitions`` (materializes schemas and primes - the tool-registry ``check_fn`` TTL cache so probes don't run cold inside the first turn), and - the context-file tier. - - Returns the number of tool schemas materialized (logged for - diagnosability). + Covers exactly the lazy init seen in skeleton turns: the ``run_agent`` import graph, + ``get_tool_definitions`` (materializes schemas, primes the ``check_fn`` TTL cache) and the + context-file tier. """ import run_agent # noqa: F401 # heavy import graph, cached in sys.modules import model_tools @@ -1286,8 +1145,7 @@ def _warm_turn_machinery_sync() -> int: def _as_thread_info(info: Any) -> Optional[Tuple[str, str]]: """*info* as a (thread_id, initial_name) pair, or None if it isn't one. - The pair comes back across the relay connector boundary, so its shape is - the connector's word rather than ours. + The pair crosses the relay connector boundary, so its shape is the connector's word, not ours. """ if isinstance(info, tuple) and len(info) == 2 and all(isinstance(x, str) for x in info): return cast(Tuple[str, str], info) @@ -1295,10 +1153,9 @@ def _as_thread_info(info: Any) -> Optional[Tuple[str, str]]: def _float_env(name: str, default: float) -> float: - """Read an env var as float, falling back to ``default`` on typos/empty. + """Read an env var as float; unset/empty/malformed fall back to ``default``. - A misconfigured env var (e.g. ``HERMES_AGENT_TIMEOUT=abc``) must not - crash the gateway or an agent turn. Unset/empty also falls back. + A misconfigured env var (``HERMES_AGENT_TIMEOUT=abc``) must not crash the gateway or a turn. """ raw = os.environ.get(name) if raw is None or raw == "": @@ -1328,11 +1185,9 @@ def _is_fresh_gateway_interruption( now: Optional[float] = None, window_secs: Optional[float] = None, ) -> bool: - """Return True when an interruption marker is fresh enough to auto-continue. + """True when an interruption marker is fresh enough to auto-continue. - Unknown timestamps are treated as fresh for backward compatibility with legacy transcripts - (pre-dating timestamp persistence) and with in-memory test scaffolding that constructs - history entries without timestamps. + Unknown timestamps count as fresh (legacy transcripts, in-memory test scaffolding). """ window = ( float(window_secs) @@ -1356,12 +1211,8 @@ def build_resume_recovery_note( ) -> str: """Build the resume-pending recovery system note for an interrupted turn. - Empty ``message`` means the startup auto-resume turn with no human message attached. - ``interactive`` selects the empty-message guidance: on interactive platforms "report the - restore and ask what next" is right. On non-interactive event platforms (webhook, API server — - adapters with ``interactive_resume = False``) nobody can answer; the resumed turn must instead - complete the interrupted work, or the task is silently abandoned behind a "restored" - acknowledgement that goes nowhere. + Empty ``message`` = startup auto-resume. Interactive platforms report the restore and ask what + next; on non-interactive ones (``interactive_resume = False``) nobody can answer: finish the work. """ reason_phrase = ( "a gateway restart" @@ -1418,11 +1269,8 @@ def _prepare_resume_pending_message( ) -> tuple[str, str]: """Return the recovery message and the user text to persist. - When the original message is empty (synthesized auto-resume turn), persist the note too — - persisting "" left a blank user row that the pre-call sanitizer re-healed on every later call. - When the user sent REAL text while the resume was pending, keep persisting their clean words: - the transcript stays scaffold-free (the model still receives the wrapped note), and a non- - empty row never trips the sanitizer. + Empty original (synthesized auto-resume): persist the note — a "" user row trips the pre-call + sanitizer every call. Real user text: persist clean words so the transcript stays scaffold-free. """ recovery_message = build_resume_recovery_note( reason, message or "", interactive=interactive, @@ -1433,34 +1281,10 @@ def _prepare_resume_pending_message( return recovery_message, persist_message -# Assistant-message fields that must survive transcript replay so multi-turn -# reasoning context, prefix-cache hits, and provider-specific echo -# requirements all behave the same on the gateway as they do in the CLI. -# -# ``reasoning`` and ``reasoning_details`` were the original three preserved -# by PR #2974 (schema v6). ``reasoning_content``, ``codex_reasoning_items``, -# ``codex_message_items``, and ``finish_reason`` were added to the DB later -# but the gateway's replay whitelist was never expanded to match — so any -# pure-text assistant turn (no ``tool_calls``) silently dropped them on -# replay, regressing the CLI-vs-gateway behavioural parity. -# -# Why each field matters on replay: -# * ``reasoning`` / ``reasoning_content``: provider-facing thinking text. -# ``_copy_reasoning_content_for_api`` promotes ``reasoning`` → -# ``reasoning_content`` at send time, but only when the strings happen to -# match. Carrying the original ``reasoning_content`` verbatim avoids -# reconstruction loss for providers that return them as distinct fields -# (DeepSeek/Kimi/Moonshot thinking modes). -# * ``reasoning_details``: opaque structured array (signature, -# encrypted_content) used by OpenRouter/Anthropic to maintain reasoning -# continuity across turns. -# * ``codex_reasoning_items``: encrypted reasoning blobs for the OpenAI -# Codex Responses API. -# * ``codex_message_items``: exact assistant message items with ``phase``. -# OpenAI docs: "preserve and resend phase on all assistant messages — -# dropping it can degrade performance." Required for prefix cache hits. -# * ``finish_reason``: informational; cheap to keep so transcripts replay -# identically across CLI and gateway. +# Assistant fields that must survive transcript replay for CLI parity (reasoning continuity, +# prefix-cache hits, provider echo requirements). ``reasoning``/``reasoning_content``: thinking +# text, unreconstructable (DeepSeek/Kimi/Moonshot). ``reasoning_details``: opaque signatures +# (OpenRouter/Anthropic). ``codex_*_items``: Codex blobs; ``phase`` is resent or caching degrades. _ASSISTANT_REPLAY_FIELDS: tuple[str, ...] = ( "reasoning", "reasoning_content", @@ -1477,23 +1301,15 @@ def _build_replay_entry( msg: Dict[str, Any], preserve_timestamp: bool = False, ) -> Dict[str, Any]: - """Build a replay entry for a non-tool-calling message, preserving the assistant fields the - agent's API builders rely on for multi-turn fidelity. - - Lifted out of the inline ``run_sync`` closure so the field whitelist can be unit-tested in - isolation. Mirrors the ``_ASSISTANT_REPLAY_FIELDS`` contract above. + """Build a replay entry for a non-tool-calling message, preserving ``_ASSISTANT_REPLAY_FIELDS``. ``preserve_timestamp``: only user messages need it (the stale-dangerous-confirmation stripper - reads it); assistant/tool messages are not timestamp-stripped, so we keep dropping it. - Falsy fields are dropped EXCEPT ``reasoning_content``: DeepSeek/Kimi thinking-mode replay - treats "" as a sentinel (upgraded to a space downstream); dropping it can cause HTTP 400. + reads it). Falsy fields are dropped EXCEPT ``reasoning_content``: DeepSeek/Kimi treat "" as a + sentinel; dropping it can 400. """ entry: Dict[str, Any] = {"role": role, "content": content} - # api_content sidecar (persist-what-you-send, prompt-cache stability): forward the exact bytes - # previously sent to the API for this message so the agent's api_messages build can substitute - # them and keep the request prefix byte-stable across turns. Forward ONLY when this pipeline - # did not rewrite the content — resending the sidecar would reintroduce exactly what was - # stripped. Dropping it costs one cache boundary; resending stripped noise is a regression. + # api_content sidecar: forward the exact bytes previously sent so the request prefix stays + # byte-stable — ONLY if this pipeline did not rewrite the content (else we resend what was stripped). _sidecar = msg.get("api_content") if ( role in ("user", "assistant") @@ -1529,9 +1345,8 @@ _CURRENT_ADDRESSED_MESSAGE_HEADER = "[Current addressed message - answer only th def _uses_telegram_observed_group_context(channel_prompt: Optional[str]) -> bool: """Return True for Telegram group turns that may include observed chatter. - Telegram's observe-unmentioned mode persists skipped group chatter so a later @mention can - see it. Those rows must not replay as ordinary user turns: a weak wake word like ``@bot - cambio`` should not make the model treat old unmentioned chatter as pending work. + Observe-unmentioned mode persists skipped group chatter for later @mentions; those rows must + not replay as ordinary user turns or a weak wake word makes old chatter look like pending work. """ return bool(channel_prompt and _TELEGRAM_OBSERVED_CONTEXT_PROMPT_MARKER in channel_prompt) @@ -1552,18 +1367,16 @@ def _csv_or_list_to_set(raw: Any) -> set[str]: def _slack_ignored_channels_from_gateway_config(config: Any) -> set[str]: """Return Slack channels that the generic gateway must never dispatch. - The Slack adapter has the first-line drop, but this runner-level guard is intentionally - duplicated as a fail-safe: if any code path, test hook, or stale adapter bypasses the plugin - adapter, ignored channels still cannot reach auth, pairing, sessions, or the prompt pipeline. + Deliberately duplicates the adapter's first-line drop as a fail-safe: even if a code path or + test hook bypasses the adapter, ignored channels cannot reach auth, pairing or sessions. """ platform_cfg = getattr(config, "platforms", {}).get(Platform.SLACK) raw = None if platform_cfg is not None: raw = getattr(platform_cfg, "extra", {}).get("ignored_channels") if raw is None: - # Top-level ``slack.ignored_channels`` config flows through the - # plugin's YAML→env bridge (SLACK_IGNORED_CHANNELS) rather than - # PlatformConfig.extra — honor it here too (#46925). + # Top-level ``slack.ignored_channels`` reaches us via the plugin's YAML→env bridge + # (SLACK_IGNORED_CHANNELS), not PlatformConfig.extra — honor it here too. raw = os.getenv("SLACK_IGNORED_CHANNELS") or None return _csv_or_list_to_set(raw) @@ -1585,9 +1398,7 @@ def _is_slack_ignored_channel(config: Any, chat_id: Any) -> bool: def _message_timestamps_enabled(user_config: Optional[dict]) -> bool: """True when gateway.message_timestamps.enabled is opted in. - Default OFF: injecting a ``[Tue 2026-04-28 13:40:53 CEST]`` prefix onto every user message - changes what the model sees for all gateway users, so it must be explicitly enabled in - config.yaml under ``gateway.message_timestamps.enabled``. + Default OFF: a timestamp prefix on every user message changes what the model sees. """ if not isinstance(user_config, dict): return False @@ -1609,9 +1420,8 @@ def _build_gateway_agent_history( ) -> tuple[List[Dict[str, Any]], Optional[str]]: """Convert stored gateway transcript rows into agent replay messages. - Keeping that context out of ``conversation_history`` avoids consecutive-user repair merging - it with the live user turn and then hiding the current message behind ``history_offset`` - during persistence. + Keeping that context out of ``conversation_history`` stops consecutive-user repair merging it + with the live turn and hiding the current message behind ``history_offset`` on persistence. """ from hermes_time import get_timezone as _get_msg_tz @@ -1655,9 +1465,8 @@ def _build_gateway_agent_history( clean_msg = {k: v for k, v in msg.items() if k not in {"timestamp", "observed"}} agent_history.append(clean_msg) elif content: - # Strip gateway-injected auto-continue notes that were persisted as part of user - # messages during interrupted turns. Keep the user's real text but never replay the - # recovery instruction itself — that caused infinite re-execution loops. + # Strip persisted auto-continue notes from user messages (interrupted turns): keep the + # user's real text but never replay the recovery instruction — it caused infinite loops. if role == "user": content = _strip_auto_continue_noise(content) if not content: @@ -1676,14 +1485,12 @@ def _build_gateway_agent_history( agent_history = strip_interrupted_tool_tails(agent_history) # Strip a dangling assistant(tool_calls) tail with no tool answers — the signature of a SIGKILL - # mid-tool-call (e.g. the tool itself ran `docker restart`/`kill` and took the gateway down - # before the result was persisted). Without this the model re-issues the unanswered call on - # resume and loops the restart forever. + # mid-tool-call (e.g. the tool ran `docker restart`/`kill` and took the gateway down before the + # result persisted). Else the model re-issues the unanswered call on resume and loops forever. agent_history = strip_dangling_tool_call_tail(agent_history) - # Strip stale dangerous-confirmation text in user messages. A high-risk confirmation phrase - # (e.g. "confirm forced restart") older than the expiry window must not replay, or an unrelated - # follow-up could be read as a fresh confirmation and re-trigger the destructive action. + # Strip expired dangerous-confirmation phrases (e.g. "confirm forced restart") from user text: + # replayed, an unrelated follow-up could read as a fresh confirmation and re-trigger the action. agent_history = strip_stale_dangerous_confirmations( agent_history, now=time.time() ) @@ -1696,20 +1503,13 @@ def _select_cached_agent_history( persisted_history: List[Dict[str, Any]], live_history: Any, ) -> List[Dict[str, Any]]: - """Prefer a cached live transcript only when it is longer and contains at least one real, non- - ephemeral unpersisted row. + """Prefer a cached live transcript only when it is longer and has at least one real, + non-ephemeral unpersisted row; otherwise return ``persisted_history`` unchanged. - Guards the FTS write-corruption case: when message writes fail silently through corrupt FTS - triggers, the next turn reloads a stale/empty ``conversation_history`` from disk even though - the same cached ``AIAgent`` still holds unpersisted real rows in ``_session_messages``. - Replacing those with the shorter persisted copy causes immediate same-session amnesia. - Length alone does not trigger retention. - - Returns ``persisted_history`` unchanged unless the live copy is a longer - list containing at least one real transcript row without the intrinsic - ``_db_persisted`` marker. A longer all-durable list can be an expected - replay-filtering delta (for example, cleanup of an interrupted read-only - tool block). Deliberately unpersisted retry scaffolding is ignored. + Guards the FTS write-corruption case: silent write failures make the next turn reload a stale + ``conversation_history`` while the cached ``AIAgent`` still holds unpersisted real rows; + replacing them causes same-session amnesia. Length alone is not enough: a longer all-durable + list can be an expected replay-filtering delta, and unpersisted retry scaffolding is ignored. """ if isinstance(live_history, list) and len(live_history) > len(persisted_history): from run_agent import _is_ephemeral_scaffolding @@ -1754,9 +1554,8 @@ def _wrap_current_message_with_observed_context(message: Any, observed_context: def _last_transcript_timestamp(history: Optional[List[Dict[str, Any]]]) -> Any: """Return the ``timestamp`` of the last usable transcript row, if any. - Skips metadata-only rows (``session_meta``, system injections) that are dropped before being - handed to the agent. Returns ``None`` when no usable row carries a timestamp — callers should - treat that as "fresh" for backward compatibility. + Skips metadata-only rows dropped before reaching the agent. ``None`` when no usable row has + a timestamp — callers treat that as "fresh" for backward compatibility. """ if not history: return None @@ -1775,9 +1574,8 @@ def _last_transcript_timestamp(history: Optional[List[Dict[str, Any]]]) -> Any: return None -# Tool results can contain literal MEDIA: examples in docs, logs, or other ordinary outputs. Only -# tools that intentionally create deliverable media artifacts should be eligible for automatic -# append when the model omits them from the final gateway reply. +# Tool output may hold literal MEDIA: examples (docs, logs); only tools that intentionally create +# deliverable media are eligible for auto-append when the model omits them from the final reply. _AUTO_APPEND_MEDIA_TOOL_NAMES = { "text_to_speech", "text_to_speech_tool", @@ -1811,11 +1609,9 @@ def _is_auto_continue_noise(content: Any) -> bool: def _strip_auto_continue_noise(content: Any) -> Any: - """Remove persisted gateway auto-continue note prefix from user text. + """Strip one or more leading persisted auto-continue note prefixes from user text. - Older gateway builds prepended the recovery note directly to the user message, so the - transcript row can contain both the synthetic note and the user's real question. Strip one or - more leading synthetic notes while preserving any real text that follows. + A row may hold both the note and the user's real question; the trailing real text is preserved. """ if not _is_auto_continue_noise(content): return content @@ -1827,9 +1623,8 @@ def _strip_auto_continue_noise(content: Any) -> Any: text = text[end + 1 :].lstrip() return text -# Tools in this set return their deliverable artifact as a JSON payload with a local-file path field -# rather than a literal ``MEDIA:`` tag (e.g. image_generate returns ``{"success": true, "image": -# "/abs/path.png"}``). +# Tools whose deliverable is a JSON payload with a local-file path field rather than a literal +# ``MEDIA:`` tag (e.g. image_generate -> ``{"success": true, "image": "/abs/path.png"}``). _JSON_MEDIA_TOOL_PATH_FIELDS = ("host_image", "image", "agent_visible_image") @@ -1844,9 +1639,8 @@ _TOOL_MEDIA_RE = re.compile( ) -# Shared with cron delivery and gateway background tasks — the repair must run on every surface that -# feeds a final response into media extraction. Canonical names live in gateway.media_repair (same -# retirement of private aliases as the agent.replay_cleanup import above). +# Shared with cron delivery and gateway background tasks — the repair must run on every surface +# that feeds a final response into media extraction; canonical names live in gateway.media_repair. from gateway.media_repair import ( # noqa: E402 repair_explicit_computer_use_media_paths, tool_name_by_call_id as _tool_name_by_call_id, @@ -1860,12 +1654,10 @@ def _collect_auto_append_media_tags( ) -> tuple[List[str], bool]: """Collect real media tags from current-turn producer-tool results only. - Two layered guards keep stale/example MEDIA: strings out of the reply: a producer-tool - allowlist (docs/logs/search results contain example MEDIA: strings that must never become - attachments) and current-turn isolation (an earlier turn's tool result must not leak onto a - later text-only reply). If mid-run compression shrank the list below the original history - length the slice boundary is untrustworthy: scan every message and rely on - ``history_media_paths`` for dedup; the allowlist still applies. + Two guards: a producer-tool allowlist (docs/logs/search results contain example MEDIA: strings + that must never become attachments) and current-turn isolation (no leaking an earlier turn's + result). If mid-run compression shrank the list below the original history length the slice + is untrustworthy: scan every message, dedup via ``history_media_paths``. """ history_media_paths = history_media_paths or set() # Only trust the slice boundary when the message list still contains the @@ -1887,9 +1679,8 @@ def _collect_auto_append_media_tags( continue content = str(msg.get("content") or "") tool_name = tool_name_by_call_id.get(call_id) - # JSON-payload tools (image_generate) return a local-file path in a - # known field rather than a MEDIA: tag. Extract it so delivery is - # deterministic even when the model omits the path from its reply. + # JSON-payload tools (image_generate) return a local-file path in a known field, not a + # MEDIA: tag; extract it so delivery is deterministic even if the model omits the path. if tool_name == "image_generate" and "MEDIA:" not in content: try: payload = json.loads(content) @@ -1919,9 +1710,8 @@ def _collect_auto_append_media_tags( def _collect_history_media_paths(agent_history: List[Dict[str, Any]]) -> set: """Collect every media path already delivered in prior assistant/tool output. - Used to dedup auto-appended and model-emitted MEDIA tags so the same file is not re-sent on - later turns. Missing the JSON-payload shape caused #46627; missing the assistant-message - shape caused repeated delivery when the model echoed a previous MEDIA tag. + Used to dedup auto-appended and model-emitted MEDIA tags so a file is not re-sent later; both + the JSON-payload and assistant-message shapes must be covered or delivery repeats. """ paths: set = set() tool_name_by_call_id = _tool_name_by_call_id(agent_history) @@ -1931,9 +1721,8 @@ def _collect_history_media_paths(agent_history: List[Dict[str, Any]]) -> set: path = match.group(1).strip().rstrip('",}') if path: paths.add(path) - # The regex alone misses quoted and spaced paths that the delivery pipeline's extract_media - # grammar accepts — collect through the same extractor so the dedup set sees every path that - # could actually have been delivered. + # The regex alone misses quoted/spaced paths that extract_media accepts — use the same + # extractor so the dedup set sees every path that could actually have been delivered. media_files, _ = BasePlatformAdapter.extract_media(content) paths.update(path for path, _is_voice in media_files) @@ -1971,9 +1760,8 @@ def _collect_history_media_paths(agent_history: List[Dict[str, Any]]) -> set: def _ensure_ssl_certs() -> None: """Set SSL_CERT_FILE if the system doesn't expose CA certs to Python. - Returning just because the variable is present makes every later httpx/OpenAI client - construction fail with FileNotFoundError from ssl.load_verify_locations(). Treat a missing - path as unset and fall back to certifi instead. + A set-but-missing path makes every later httpx/OpenAI client fail in ssl.load_verify_locations(), + so treat it as unset and fall back to certifi. """ configured_cert = os.environ.get("SSL_CERT_FILE") if configured_cert: @@ -2020,9 +1808,8 @@ def _ensure_ssl_certs() -> None: def _home_target_env_var(platform_name: str) -> str: """Return the configured home-target env var for a platform. - Consults built-in ``_HOME_TARGET_ENV_VARS`` first, then the plugin - registry via ``cron.scheduler._resolve_home_env_var``, then falls back - to ``_HOME_CHANNEL`` for unknown names. + Built-in ``_HOME_TARGET_ENV_VARS`` first, then the plugin registry + (``cron.scheduler._resolve_home_env_var``), then ``_HOME_CHANNEL`` for unknown names. """ from cron.scheduler import _resolve_home_env_var @@ -2080,18 +1867,15 @@ load_hermes_dotenv(hermes_home=_hermes_home, project_env=Path(__file__).resolve( def _reload_runtime_env_preserving_config_authority() -> None: """Reload .env for fresh credentials without letting stale .env override config. - Gateways are long-lived, so per-turn code reloads ~/.hermes/.env to pick up rotated keys; - config.yaml stays authoritative for agent budget settings (else a stale HERMES_MAX_ITERATIONS - in .env would replace the startup bridge). In multiplex mode the credential reload is a NO-OP: - secrets come from the per-turn ``set_secret_scope`` isolated mapping, and mutating the - process-global ``os.environ`` here would defeat that isolation and leak the - default profile's keys to every profile's turns and subprocesses. + Long-lived gateways reload ~/.hermes/.env per turn for rotated keys; config.yaml stays + authoritative for budget settings (else stale HERMES_MAX_ITERATIONS wins). NO-OP in multiplex + mode: secrets come from the per-turn ``set_secret_scope`` mapping, and mutating ``os.environ`` + would leak the default profile's keys to every profile. """ from agent.secret_scope import is_multiplex_active if is_multiplex_active(): - # Credentials are resolved from the active profile's secret scope, not - # os.environ. Still honor config.yaml's agent.max_turns bridge below - # using the scoped home, but never reload .env into global env. + # Credentials come from the active profile's secret scope, not os.environ: still honor the + # config.yaml agent.max_turns bridge below (scoped home), but never reload .env globally. _bridge_max_turns_from_config(_hermes_home) return @@ -2115,9 +1899,8 @@ def _bridge_max_turns_from_config(home: "Path") -> None: cfg = _expand_env_vars(cfg) if not isinstance(cfg, dict): cfg = {} - # Managed scope: keep administrator-pinned values authoritative on every turn too. This per- - # turn reload re-bridges config→env, so without the overlay a managed agent.max_turns / - # timezone / redact_secrets would be replaced by the user's value after the first turn. + # Managed scope: the per-turn reload re-bridges config→env, so without the overlay a managed + # agent.max_turns/timezone/redact_secrets would revert to the user's value after one turn. try: from hermes_cli import managed_scope cfg = managed_scope.apply_managed_overlay(cfg) @@ -2129,10 +1912,9 @@ def _bridge_max_turns_from_config(home: "Path") -> None: agent_cfg = cfg.get("agent", {}) if isinstance(agent_cfg, dict) and "max_turns" in agent_cfg: raw = agent_cfg["max_turns"] - # Preserve the raw value's spelling (e.g. "none", "unlimited", "120") so - # resolve_turn_limit() in _current_max_iterations can interpret it. Skip when the YAML value - # is Python None (`null` / bare `key:`): str(None) -> "None" would map to the unlimited - # sentinel instead of preserving "absent = default". + # Preserve the raw spelling ("none", "unlimited", "120") so resolve_turn_limit() in + # _current_max_iterations can interpret it. Skip Python None (`null` / bare `key:`): + # str(None) -> "None" would map to the unlimited sentinel instead of "absent = default". if raw is not None: os.environ["HERMES_MAX_ITERATIONS"] = str(raw) elif "HERMES_MAX_ITERATIONS" in os.environ: @@ -2151,9 +1933,8 @@ def _bridge_max_turns_from_config(home: "Path") -> None: def _current_max_iterations() -> int: """Return the current per-turn iteration budget after runtime env refresh. - Goes through :func:`hermes_cli.config.resolve_turn_limit` so that ``agent.max_turns: none`` / - ``unlimited`` (bridged into ``HERMES_MAX_ITERATIONS`` as a string) resolves to the unlimited - sentinel instead of crashing ``int()``. + Uses ``resolve_turn_limit`` so ``agent.max_turns: none``/``unlimited`` (bridged as a string + into ``HERMES_MAX_ITERATIONS``) yields the unlimited sentinel instead of an ``int()`` crash. """ _reload_runtime_env_preserving_config_authority() from hermes_cli.config import resolve_turn_limit as _resolve_turn_limit @@ -2163,16 +1944,15 @@ def _current_max_iterations() -> int: from contextlib import ( asynccontextmanager as _asynccontextmanager, contextmanager as _contextmanager, +suppress, ) -# Platforms that bind a host TCP port (HTTP/webhook listeners). In a profile multiplexer the default -# profile owns the single shared listener and serves every profile through the /p// URL -# prefix, so a SECONDARY profile enabling one of these is always a misconfiguration. That profile is -# skipped (SecondaryPortBindingConfigError) so one bad profile cannot take down the multiplexer; the -# set lives in gateway.config so the dashboard's pre-write validation enforces the same policy. +# Platforms that bind a host TCP port. In a profile multiplexer the default profile owns the single +# shared listener (serving every profile via the /p// prefix), so a SECONDARY profile +# enabling one is always a misconfiguration and is skipped (SecondaryPortBindingConfigError) rather +# than taking down the multiplexer. Lives in gateway.config so dashboard validation enforces it too. from gateway.config import ( - PORT_BINDING_PLATFORM_VALUES as _PORT_BINDING_PLATFORM_VALUES, platform_binds_port as _platform_binds_port, ) @@ -2180,9 +1960,8 @@ from gateway.config import ( class MultiplexConfigError(RuntimeError): """A profile multiplexer config is invalid. - Distinct from a transient adapter-connect failure: a config error means the - operator must fix config.yaml. Fatal configuration errors propagate to the - startup guard instead of being treated as retryable adapter noise. + Distinct from a transient adapter-connect failure: the operator must fix config.yaml, so it + propagates to the startup guard instead of being treated as retryable adapter noise. """ @@ -2191,13 +1970,11 @@ class SecondaryPortBindingConfigError(MultiplexConfigError): class HygieneTurnHoldExceeded(Exception): - """The hygiene-compression turn-hold budget elapsed while the summary model was still streaming - progress. + """The hygiene-compression turn-hold budget elapsed while the summary model was still streaming. - This is an availability boundary, not a failure: the compressor is healthy, but the current - user turn cannot wait any longer. It must NOT be routed through the idle-timeout failure path - (which stamps AGENT_COMPRESSION_TIMEOUT, sends a "no output" message, and advances the - failure cooldown ladder). + An availability boundary, not a failure: the compressor is healthy but the user turn cannot + wait. Must NOT be routed through the idle-timeout failure path (AGENT_COMPRESSION_TIMEOUT, + "no output" message, failure cooldown ladder). """ @@ -2216,10 +1993,9 @@ def _multiplex_profile_homes(config: object) -> list[tuple[str, "Path"]]: def _enable_multiplex_log_routing(config: object) -> bool: """Route agent.log/errors.log/gateway.log records to their owning profile. - ``setup_logging(mode="gateway")`` binds the queued file handlers to the launch home, so under - ``multiplex_profiles`` every secondary profile's records (emitted inside - ``_profile_runtime_scope``) land in the default profile's log files. Swap in the profile - routers once the served-profile set is known; inert for single-profile gateways. + ``setup_logging(mode="gateway")`` binds the file handlers to the launch home, so under + ``multiplex_profiles`` every secondary profile's records land in the default profile's logs. + Swap in the profile routers once the served-profile set is known; inert for single-profile. """ if not getattr(config, "multiplex_profiles", False): return False @@ -2237,13 +2013,11 @@ def _enable_multiplex_log_routing(config: object) -> bool: def _handoff_watch_scopes(runner: object) -> list: """``(profile_name, home)`` pairs whose ``state.db`` the watcher must poll. - ``/handoff`` writes into the store of the profile the CLI ran under, but the watcher resolves - ``_session_db`` from the active HERMES_HOME. Unscoped, that is always the ROOT store, so a - pending handoff queued by any secondary profile is never seen and the CLI times out with the - gateway plainly alive. ``(None, None)`` = unscoped root poll, always first (legacy behaviour - unchanged); SECONDARY profiles are added, the default profile is not repeated (same state.db). - Defensive on purpose: a raising resolver would be swallowed by the loop handler and silently - disable the watcher. Any failure degrades to the root poll. + ``/handoff`` writes into the store of the profile the CLI ran under, but an unscoped watcher + polls only the ROOT store, so a secondary profile's handoff is never seen and the CLI times out. + ``(None, None)`` = root poll, always first; SECONDARY profiles are added, the default is not + repeated. Defensive: a raising resolver would silently disable the watcher, so failures + degrade to the root poll. """ scopes: list = [(None, None)] try: @@ -2261,11 +2035,9 @@ def _handoff_watch_scopes(runner: object) -> list: async def _reclaim_stale(runner: object) -> None: """Fail handoffs left in ``running`` by a gateway that died mid-dispatch. - Runs once per store at watcher startup. ``running`` is only ever set by the watcher for the - duration of one in-process dispatch, so a row still in that state belongs to a previous - process. It can never reach a terminal state on its own, and ``request_handoff`` refuses a - NEW request while it sits there — the session could never hand off again. Defensive: a - raising reclaim would abort watcher startup. + Runs once per store at watcher startup. ``running`` is only set for one in-process dispatch, + so a row still in it belongs to a dead process, can never reach a terminal state, and blocks + ``request_handoff`` for that session forever. Defensive: a raising reclaim would abort startup. """ session_db = getattr(runner, "_session_db", None) if session_db is None: @@ -2291,8 +2063,7 @@ async def _reclaim_stale(runner: object) -> None: def _terminal_scope_cwd(default: str = "") -> str: """Scope-aware TERMINAL_CWD read for footer/context surfaces. - Only an import failure falls back: an active refusal scope must raise, - not resolve the launch profile's cwd. + Only an import failure falls back: an active refusal scope must raise, not use the launch cwd. """ try: from tools.terminal_scope import terminal_env as _ts_env @@ -2322,15 +2093,12 @@ def _profile_runtime_scope( *, hydrate_secrets: bool = True, ): - """Scope config/skills/memory AND credentials to a profile for one turn. + """Scope config/skills/memory AND credentials to a profile for one turn (multiplexed path only). - Combines the two seams the multiplexer needs: (1) ``set_hermes_home_override`` redirects - ``get_hermes_home()`` to the profile's home — a contextvar, so it propagates into the agent - worker thread via ``copy_context()``; (2) ``set_secret_scope`` installs the profile's ``.env`` - as the authoritative credential source so ``get_secret`` never reads process-global - ``os.environ``. Only used on the multiplexed inbound path; single-profile gateways never enter - this scope. Loading ``.env`` here does NOT mutate ``os.environ``, which is what keeps - subprocesses (MCP, kanban) from inheriting cross-profile secrets. + (1) ``set_hermes_home_override`` redirects ``get_hermes_home()`` — a contextvar, so it reaches + the agent worker thread via ``copy_context()``; (2) ``set_secret_scope`` makes the profile's + ``.env`` the credential source so ``get_secret`` never reads ``os.environ``. Loading ``.env`` + does NOT mutate ``os.environ``, which keeps subprocesses from inheriting cross-profile secrets. """ from hermes_constants import set_hermes_home_override, reset_hermes_home_override from agent.secret_scope import ( @@ -2349,10 +2117,9 @@ def _profile_runtime_scope( secrets = build_profile_secret_scope(Path(profile_home)) secret_token = set_secret_scope(secrets) - # Per-turn terminal scope (third seam of the profile boundary): installs the routed profile's - # COMPLETE terminal policy — never ambient env — via tools.terminal_scope. Without it - # terminal_tool reads process-global TERMINAL_* vars a previous profile's turn may have pinned - # (first-writer-wins backend leak). + # Per-turn terminal scope (third seam of the profile boundary): install the routed profile's + # COMPLETE terminal policy — never ambient env — via tools.terminal_scope, else terminal_tool + # reads process-global TERMINAL_* vars a prior profile's turn pinned (first-writer-wins leak). from tools.terminal_scope import install_and_reset_profile_terminal_scope with install_and_reset_profile_terminal_scope(Path(profile_home)): @@ -2374,11 +2141,10 @@ async def _async_profile_runtime_scope(profile_home: "Path"): def load_gateway_config_for_runner() -> "GatewayConfig": """Load gateway config for the process-level GatewayRunner. - When multiplexing is on, reload under the default/active profile's ``_profile_runtime_scope`` - so platform tokens in that profile's ``.env`` resolve through the secret scope — the same - path secondary profiles use in ``_start_one_profile_adapters``. Unscoped, ``_getenv`` falls - through to ``os.environ``, which often lacks the token once it lives only under - ``profiles//.env``. Off -> identical to ``load_gateway_config()``. + With multiplexing on, reload under the default profile's ``_profile_runtime_scope`` so platform + tokens in that profile's ``.env`` resolve through the secret scope (as secondary profiles do); + unscoped, ``_getenv`` falls through to ``os.environ``, which often lacks a token that lives only + under ``profiles//.env``. Off -> identical to ``load_gateway_config()``. """ cfg = load_gateway_config() if not getattr(cfg, "multiplex_profiles", False): @@ -2401,9 +2167,8 @@ def load_gateway_config_for_runner() -> "GatewayConfig": async def _discover_gateway_mcp_tools(config: object) -> None: """Run startup MCP discovery for every profile this gateway serves. - ``discover_mcp_tools`` reads ``mcp_servers`` from ``get_hermes_home()``'s config, so an - unscoped call only ever connects the launch profile's servers. Single-profile gateways keep - the one unscoped call. + ``discover_mcp_tools`` reads ``mcp_servers`` from ``get_hermes_home()``'s config, so an unscoped + call only connects the launch profile's servers. Single-profile gateways keep the unscoped call. """ from tools.mcp_tool import discover_mcp_tools @@ -2438,12 +2203,11 @@ def _platform_has_bot_credential(platform: "Platform", platform_config: "Platfor api_key = getattr(platform_config, "api_key", None) or "" if isinstance(api_key, str) and api_key.strip(): return True - # Matrix also authenticates by password login (MATRIX_USER_ID + MATRIX_PASSWORD, no - # MATRIX_ACCESS_TOKEN). Those land in ``extra``, so a token-only check would evict a perfectly - # reconnectable config from the retry queue on the first transient failure. Mirror the - # adapter's own gate: homeserver + user_id + password. Read ONLY from extra, never os.getenv: - # build_config() already copies the env vars onto extra, and an env fallback would report "has - # credential" for every Matrix config on the box — including the empty-primary multiplex case. + # Matrix also authenticates by password login (MATRIX_USER_ID + MATRIX_PASSWORD in ``extra``), + # so a token-only check would evict a reconnectable config from the retry queue on the first + # transient failure; mirror the adapter's gate: homeserver + user_id + password. Read ONLY from + # extra (build_config() already copies env vars there) — an env fallback would report "has + # credential" for every Matrix config on the box, including the empty-primary multiplex case. if platform is Platform.MATRIX: extra = getattr(platform_config, "extra", None) or {} if all( @@ -2457,9 +2221,8 @@ def _platform_has_bot_credential(platform: "Platform", platform_config: "Platfor _DOCKER_VOLUME_SPEC_RE = re.compile(r"^(?P.+):(?P/[^:]+?)(?::(?P[^:]+))?$") _DOCKER_MEDIA_OUTPUT_CONTAINER_PATHS = {"/output", "/outputs"} -# This env var is internal bridge plumbing, not a user-facing configuration -# source. Initialize it from the canonical config default after dotenv loading -# so an ambient process/.env value can never control lease safety on its own. +# Internal bridge plumbing, not a user-facing config source: initialize from the canonical config +# default after dotenv loading so an ambient process/.env value can never control lease safety. from hermes_cli.config_defaults import DEFAULT_CONFIG as _DEFAULT_CONFIG os.environ["HERMES_TURN_LEASE_TIMEOUT"] = str( @@ -2471,18 +2234,16 @@ os.environ["HERMES_TURN_LEASE_TIMEOUT"] = str( _config_path = _hermes_home / 'config.yaml' if _config_path.exists(): try: - # Presence-sensitive env bridge: raw read is deliberate — only keys the - # user actually wrote may be bridged (a defaults merge would export the - # whole DEFAULT_CONFIG into the env). Overlay + expansion applied below. + # Presence-sensitive env bridge: raw read is deliberate — only keys the user actually wrote + # may be bridged (a defaults merge would export all DEFAULT_CONFIG); overlay applied below. from hermes_cli.config import _expand_env_vars, read_user_config_raw _cfg = read_user_config_raw(_config_path) # Expand ${ENV_VAR} references before bridging to env vars. _cfg = _expand_env_vars(_cfg) if not isinstance(_cfg, dict): _cfg = {} - # Managed scope: overlay administrator-pinned values BEFORE bridging to env vars, so a - # managed timezone / redact_secrets / max_turns / terminal setting wins over the user's - # value at the env layer too. Fail-open via the helper. + # Managed scope: overlay administrator-pinned values BEFORE bridging so a managed timezone/ + # redact_secrets/max_turns/terminal setting wins at the env layer too; fail-open via helper. try: from hermes_cli import managed_scope _cfg = managed_scope.apply_managed_overlay(_cfg) @@ -2536,15 +2297,13 @@ if _config_path.exists(): for _cfg_key, _env_var in _terminal_env_map.items(): if _cfg_key in _terminal_cfg: _val = _terminal_cfg[_cfg_key] - # Skip cwd placeholder values (".", "auto", "cwd") — the gateway resolves these - # to Path.home() later (line ~255). Writing the raw placeholder here would just - # be noise. Only bridge explicit absolute paths from config.yaml. + # Skip cwd placeholders (".", "auto", "cwd") — the gateway resolves them to + # Path.home() later; only bridge explicit absolute paths from config.yaml. if _cfg_key == "cwd" and str(_val) in {".", "auto", "cwd"}: continue - # Expand shell tilde in local/container cwd so subprocess.Popen never receives a - # literal "~/" which the kernel rejects. SSH cwd is interpreted by the remote - # shell, so preserve "~" for the SSH backend instead of expanding to the Hermes - # host HOME. Shared predicate with terminal_tool so the two sites can't drift. + # Expand "~" in local/container cwd so subprocess.Popen never gets a literal + # "~/" (the kernel rejects it); SSH cwd is interpreted by the remote shell, so + # keep "~" there. Predicate shared with terminal_tool so the sites can't drift. if _cfg_key == "cwd" and isinstance(_val, str): if not _is_ssh_remote_tilde_cwd(_terminal_backend, _val.strip()): _val = os.path.expanduser(_val) @@ -2552,14 +2311,11 @@ if _config_path.exists(): os.environ[_env_var] = json.dumps(_val) else: os.environ[_env_var] = str(_val) - # Compression config is read directly from config.yaml by run_agent.py and - # auxiliary_client.py — no env var bridging needed. Auxiliary model/direct-endpoint - # overrides (vision, approval, plus any plugin-registered auxiliary tasks). + # Compression config is read from config.yaml by run_agent.py/auxiliary_client.py (no env + # bridge). Auxiliary model/endpoint overrides: vision, approval, plugin-registered tasks. _auxiliary_cfg = _cfg.get("auxiliary", {}) if _auxiliary_cfg and isinstance(_auxiliary_cfg, dict): - # Built-in tasks that previously had explicit env-var bridging. - # Kept here as the canonical bridged set; plugin tasks are added - # below via the plugin auxiliary registry. + # Canonical built-in bridged set; plugin tasks are added below via the aux registry. _aux_bridged_keys = {"vision", "approval"} try: from hermes_cli.plugins import get_plugin_auxiliary_tasks @@ -2587,12 +2343,8 @@ if _config_path.exists(): os.environ[f"AUXILIARY_{_upper}_BASE_URL"] = _base_url if _api_key: os.environ[f"AUXILIARY_{_upper}_API_KEY"] = _api_key - # config.yaml is the documented, authoritative source for these - # settings — it unconditionally wins over .env values. Previously - # the guards below read `if X not in os.environ` and let stale - # .env entries (e.g. HERMES_MAX_ITERATIONS=60 written by an old - # `hermes setup` run) silently shadow the user's current config. - # See PR #18413 / the 60-vs-500 max_turns incident. + # config.yaml is authoritative and unconditionally wins over .env; a `not in os.environ` + # guard would let stale .env entries (an old HERMES_MAX_ITERATIONS) shadow current config. _agent_cfg = _cfg.get("agent", {}) if _agent_cfg and isinstance(_agent_cfg, dict): if "max_turns" in _agent_cfg: @@ -2657,9 +2409,8 @@ if _config_path.exists(): os.environ["HERMES_GATEWAY_BUSY_TEXT_MODE"] = str(_display_cfg["busy_text_mode"]) if "busy_ack_enabled" in _display_cfg: os.environ["HERMES_GATEWAY_BUSY_ACK_ENABLED"] = str(_display_cfg["busy_ack_enabled"]) - # This process-level env var is documented as an override for - # service managers, so preserve it when already set. Other display - # bridges stay config-authoritative for backwards compatibility. + # Documented as a service-manager override, so preserve it when already set; other + # display bridges stay config-authoritative for backwards compatibility. if ( "busy_steer_ack_enabled" in _display_cfg and "HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED" not in os.environ @@ -2677,9 +2428,8 @@ if _config_path.exists(): _redact = _security_cfg.get("redact_secrets") if _redact is not None: os.environ["HERMES_REDACT_SECRETS"] = str(_redact).lower() - # Gateway settings (media delivery allowlist + recency trust + strict mode) Delegated to the - # shared bridge so standalone delivery entrypoints (manual `hermes cron run`, ticks without - # the gateway) apply the SAME policy translation — process parity for attachment filtering. + # Media settings (delivery allowlist, recency trust, strict mode) use the shared bridge so + # standalone entrypoints (`hermes cron run`, gateway-less ticks) apply the SAME policy. _gateway_cfg = _cfg.get("gateway", {}) if isinstance(_gateway_cfg, dict): from gateway.media_policy import apply_media_policy_env @@ -2688,9 +2438,8 @@ if _config_path.exists(): _trust_recent_seconds = _gateway_cfg.get("trust_recent_files_seconds") if _trust_recent_seconds is not None: os.environ["HERMES_MEDIA_TRUST_RECENT_SECONDS"] = str(_trust_recent_seconds) - # Bridge gateway.platform_connect_timeout → the internal env var the connect path + - # Discord adapter ready-wait both read. Unlike the config-authoritative bridges above, - # this env var is the manual-override escape hatch: it WINS if already set explicitly. + # Bridge gateway.platform_connect_timeout → the env var the connect path and Discord + # ready-wait read. Unlike the bridges above, it is an escape hatch: WINS if already set. if ( "platform_connect_timeout" in _gateway_cfg and not os.environ.get("HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT", "").strip() @@ -2742,9 +2491,8 @@ os.environ["HERMES_QUIET"] = "1" # tools (e.g. send_message → _gateway_runner_ref) must not flip interactive CLI sessions into ask- # mode, or Dangerous Command prompts become silent pending_approval with no Approve/Deny UI. -# Set terminal working directory for messaging platforms. config.yaml terminal.cwd is the canonical -# source (bridged to TERMINAL_CWD by the config bridge above). MESSAGING_CWD is a backward-compat -# fallback. +# Terminal cwd for messaging platforms: config.yaml terminal.cwd is canonical (bridged to +# TERMINAL_CWD above); MESSAGING_CWD is a backward-compat fallback. from gateway.cwd_placeholder import CWD_PLACEHOLDERS, resolve_placeholder_terminal_cwd _configured_cwd = os.environ.get("TERMINAL_CWD", "") @@ -2849,9 +2597,7 @@ from gateway.restart import ( from gateway.whatsapp_identity import ( canonical_whatsapp_identifier as _canonical_whatsapp_identifier, # noqa: F401 - expand_whatsapp_aliases as _expand_whatsapp_auth_aliases, - normalize_whatsapp_identifier as _normalize_whatsapp_identifier, -) + ) logger = logging.getLogger(__name__) @@ -2906,34 +2652,19 @@ def _own_policy_open_startup_violation(config) -> Optional[str]: return None -# Sentinel placed into _running_agents immediately when a session starts processing, *before* any -# await. Prevents a second message for the same session from bypassing the "already running" guard -# during the async gap between the guard check and actual agent creation. +# Sentinel placed into _running_agents *before* any await when a session starts processing, so a +# second message can't slip past the "already running" guard before the agent actually exists. _AGENT_PENDING_SENTINEL = object() -# Conversation-scoped per-session state registry (legacy contract). -# The state itself now lives in ``SessionState.conversation`` (see -# gateway/session_state.py) and boundaries clear it structurally via -# ``ConversationState.clear()`` — adding a field to ConversationState means -# every boundary picks it up automatically. This tuple is retained for: -# (a) plain-dict conversation-scoped stores not yet folded into -# SessionState (currently ``_pending_model_notes``), which -# _clear_conversation_scope still pops per-key; and -# (b) the public test contract (tests import and iterate this tuple). -# History: boundaries used to each carry a hand-copied pop-list that drifted -# whenever a new dict was added (#48031, #58403, #10702, #35809). -# -# NOT in this list (different lifecycles): -# - _running_agents/_running_agents_ts/_active_session_leases/_busy_ack_ts/ -# _turn_lease_tokens: turn-scoped, owned by _release_running_agent_state -# and the dispatch finally. -# - _session_run_generation: monotonic by design; clearing it would reset -# the counter and break stale-run detection (#28686). -# - _agent_cache: has its own eviction path (_evict_cached_agent) with -# resource cleanup; boundaries call it explicitly. -# - _pending_approvals/_update_prompt_pending/slash-confirm/tool-approval -# state: cleared via _clear_session_boundary_security_state, which -# _clear_conversation_scope calls. +# Conversation-scoped per-session state registry (legacy contract). The state itself lives in +# ``SessionState.conversation`` and boundaries clear it via ``ConversationState.clear()`` (new +# fields are picked up automatically). Retained for (a) plain-dict stores not yet folded into +# SessionState (``_pending_model_notes``), popped per-key by _clear_conversation_scope, and (b) the +# public test contract. NOT in this list (different lifecycles): _running_agents/_running_agents_ts/ +# _active_session_leases/_busy_ack_ts/_turn_lease_tokens (turn-scoped, owned by +# _release_running_agent_state and the dispatch finally); _session_run_generation (monotonic — +# clearing breaks stale-run detection); _agent_cache (own eviction path _evict_cached_agent); +# approval/slash-confirm state (cleared via _clear_session_boundary_security_state). _CONVERSATION_SCOPED_STATE: tuple = ( "_session_model_overrides", "_pending_one_turn_model_restores", @@ -2945,9 +2676,8 @@ _CONVERSATION_SCOPED_STATE: tuple = ( # Stall-watchdog "already notified" latch (#72016). Cleared on /new so a # fresh conversation can warn again if it later stalls with pending inbound. "_session_stall_notified", - # Staged-but-never-consumed sidecar notes (turn aborted between staging - # and run_sync) must not leak into a future conversation's first user - # message — session keys are source-derived and REUSED. + # Sidecar notes staged but never consumed (turn aborted before run_sync) must not leak into a + # future conversation's first user message — session keys are source-derived and REUSED. "_pending_turn_sidecar_notes", ) @@ -2958,9 +2688,8 @@ _UNSET = object() def _resolve_runtime_agent_kwargs() -> dict: """Resolve provider credentials for gateway-created AIAgent instances. - ``resolve_runtime_provider()`` falls through to env var lookups internally for legacy - compatibility, but the gateway does not consult environment variables for behavioral config — - config.yaml is authoritative. + ``resolve_runtime_provider()`` falls through to env vars for legacy compatibility, but the + gateway never consults env vars for behavioral config — config.yaml is authoritative. """ from hermes_cli.runtime_provider import ( resolve_runtime_provider, @@ -2972,9 +2701,8 @@ def _resolve_runtime_agent_kwargs() -> dict: try: runtime = resolve_runtime_provider() except AuthError as auth_exc: - # Distinguish a transient rate-limit/quota cap (credentials are fine, re-auth cannot help) - # from a genuine auth failure (expired/revoked token). Both fall through to the fallback - # chain, but the log message must not mislabel a quota exhaustion as an auth failure. + # Distinguish a rate-limit/quota cap (credentials fine, re-auth can't help) from a real auth + # failure (expired/revoked token): both use the fallback chain; the log must not mislabel. if is_rate_limited_auth_error(auth_exc): logger.warning("Primary provider rate-limited (429): %s — trying fallback", auth_exc) else: @@ -2998,9 +2726,8 @@ def _resolve_runtime_agent_kwargs() -> dict: mt = model_cfg.get("max_tokens") if isinstance(mt, int): max_tokens = mt - # Fall back to a per-provider output cap (custom_providers max_output_tokens) - # only when the documented global model.max_tokens isn't set, so the global - # key always wins. + # Per-provider output cap (custom_providers max_output_tokens) applies only when the documented + # global model.max_tokens is unset, so the global key always wins. if max_tokens is None: _runtime_mot = runtime.get("max_output_tokens") if isinstance(_runtime_mot, int) and _runtime_mot > 0: @@ -3028,10 +2755,9 @@ def _resolve_runtime_agent_kwargs() -> dict: "credential_pool": runtime.get("credential_pool"), "request_overrides": dict(runtime.get("request_overrides") or {}), "max_tokens": max_tokens, - # Per-provider request_overrides (e.g. a custom_providers ``extra_body`` - # carrying ``chat_template_kwargs``) resolved by resolve_runtime_provider(). - # Must flow through to the per-turn route or the provider's configured - # request body never reaches the model on the gateway path. + # Per-provider request_overrides (e.g. custom_providers ``extra_body`` with + # ``chat_template_kwargs``) from resolve_runtime_provider() must reach the per-turn route, + # else the provider's configured request body never reaches the model on the gateway path. "request_overrides": runtime.get("request_overrides"), "capabilities": capabilities, } @@ -3051,9 +2777,8 @@ class _GatewayModelContext: def _resolve_gateway_model_context(model: Optional[str] = None) -> _GatewayModelContext: """Resolve the configured gateway route and its effective context window. - This is the shared non-resident authority for status/session banners and - slash commands. Call it off the event loop: runtime credential resolution - and model metadata may perform blocking work. + Shared authority for status/session banners and slash commands. Call it off the event loop: + credential resolution and model metadata may block. """ from agent.model_metadata import DEFAULT_FALLBACK_CONTEXT, get_model_context_length @@ -3075,10 +2800,8 @@ def _resolve_gateway_model_context(model: Optional[str] = None) -> _GatewayModel configured_model = model_cfg.get("default") or model_cfg.get("model") raw_ctx = model_cfg.get("context_length") if raw_ctx is not None: - try: + with suppress(TypeError, ValueError): config_context_length = int(raw_ctx) - except (TypeError, ValueError): - pass provider = model_cfg.get("provider") or None base_url = model_cfg.get("base_url") or None configured_provider = provider @@ -3213,9 +2936,8 @@ def _try_resolve_fallback_provider() -> dict | None: """Attempt to resolve credentials from the fallback_model/fallback_providers config.""" from hermes_cli.runtime_provider import resolve_runtime_provider try: - # Canonical gateway loader: managed overlay + ${VAR} expansion + - # root-model normalization now reach the fallback chain too (a raw - # read here used to miss administrator-pinned fallback_providers). + # Canonical loader so managed overlay, ${VAR} expansion and root-model normalization + # reach the fallback chain (a raw read misses administrator-pinned fallback_providers). cfg = _load_gateway_runtime_config() fb_list = get_fallback_chain(cfg) if not fb_list: @@ -3229,9 +2951,8 @@ def _try_resolve_fallback_provider() -> dict | None: explicit_base_url=entry.get("base_url"), explicit_api_key=resolve_entry_api_key(entry), ) - # Log the literal `provider` key from config, not the resolved runtime category — an - # Ollama fallback resolves through the OpenAI-compatible path and would otherwise be - # logged as "openrouter", contradicting the operator's config. + # Log the literal config `provider`, not the resolved runtime category: an Ollama + # fallback resolves via the OpenAI-compatible path and would log as "openrouter". logger.info( "Fallback provider resolved: %s model=%s", entry.get("provider") or runtime.get("provider"), @@ -3259,11 +2980,7 @@ def _try_resolve_fallback_provider() -> dict | None: def _event_media_type_at(event, index: int) -> str: - """Return the per-attachment MIME for the attachment at *index*. - - Empty string when the platform didn't populate a per-file MIME for - that slot (some adapters only set a message-level type). - """ + """Per-attachment MIME at *index*; "" when the adapter set only a message-level type.""" media_types = getattr(event, "media_types", None) or [] return media_types[index] if index < len(media_types) else "" @@ -3271,9 +2988,8 @@ def _event_media_type_at(event, index: int) -> str: def _event_media_is_image(event, index: int) -> bool: """True if the attachment at *index* is an image. - Trust the per-attachment MIME when present. Only fall back to the message-level ``PHOTO`` - type when this attachment's MIME is unknown — otherwise a document uploaded alongside an - image gets mis-routed as an image, base64'd into a vision part, and the provider 400s. + Trust the per-attachment MIME; fall back to message-level ``PHOTO`` only when unknown, else a + document uploaded alongside an image is base64'd as vision and the provider 400s. """ mtype = _event_media_type_at(event, index) if mtype: @@ -3309,11 +3025,10 @@ def _event_media_is_video(event, index: int) -> bool: def _build_media_placeholder(event) -> str: - """Build a text placeholder for media-only events so they aren't dropped. + """Text placeholder for media-only events (later replaced by vision enrichment). - When a photo/document is queued during active processing and later dequeued, only .text is - extracted. If the event has no caption, the media would be silently lost. This builds a - placeholder that the vision enrichment pipeline will replace with a real description. + Media queued during active processing is dequeued via .text only, so a caption-less event + would otherwise be lost. """ parts = [] media_urls = getattr(event, "media_urls", None) or [] @@ -3338,10 +3053,8 @@ def _build_document_context_note( ) -> str: """Context note prepended to a user turn when they attach a document. - ``content_inlined=False`` records adapters that cache the file without injecting its content, - so the note tells the agent to read it. Binary documents (PDF, DOCX, XLSX, …) cannot be - inlined as text; the note must tell the agent to *extract* the text itself — earlier "ask the - user what to do with it" wording made the model punt, so attachments looked "unreadable". + ``content_inlined=False`` = adapter cached the file without injecting content, so tell the agent + to read it. Binary docs (PDF, DOCX, …) must say *extract* the text; "ask the user" made it punt. """ if mtype.startswith("text/") and content_inlined: return ( @@ -3420,8 +3133,8 @@ async def _probe_audio_duration(path: str) -> Optional[str]: def _dequeue_pending_event(adapter, session_key: str) -> MessageEvent | None: """Consume and return the full pending event for a session. - Queued follow-ups must preserve their media metadata so they can re-enter the normal - image/STT/document preprocessing path instead of being reduced to a placeholder string. + Queued follow-ups keep their media metadata so they re-enter the normal image/STT/document + preprocessing path instead of collapsing to a placeholder string. """ return adapter.get_pending_message(session_key) @@ -3443,16 +3156,13 @@ def _reap_gateway_turn_processes( ) -> int: """Reap only background processes created by one abandoned turn. - ``task_id`` is session-scoped (task_id == session_id), not turn-scoped, so a *replacement* - turn on the same session can start and spawn its own legitimate process while this reap is - still in flight. ``is_still_current`` (closure over the captured run_generation) lets the - caller bail out instead of killing the newer turn's process; that turn snapshots its own - baseline, so nothing stays permanently unreaped. + ``task_id`` is session-scoped, so a *replacement* turn can spawn its own process mid-reap; + ``is_still_current`` (closure over the captured run_generation) lets the caller bail instead of + killing it. That turn snapshots its own baseline, so nothing stays unreaped. """ if not task_id: - # ProcessSession.task_id defaults to "" for sessionless callers, so a - # blank id would match (and kill) every unrelated empty-task process - # instead of this turn's own. Nothing session-scoped to reap. + # ProcessSession.task_id defaults to "" for sessionless callers; a blank id would match + # (and kill) every unrelated empty-task process. Nothing session-scoped to reap. return 0 if is_still_current is not None: try: @@ -3481,9 +3191,8 @@ def _reap_gateway_turn_processes( source=source, ) except Exception: - # Runs on a detached daemon thread (interrupt and timeout call sites both fire-and-forget - # it) — an uncaught exception here would only surface via threading.excepthook, bypassing - # the app's logger. Swallow and log through the normal channel instead. + # Runs on a detached daemon thread (fire-and-forget from interrupt and timeout paths); an + # uncaught exception would only reach threading.excepthook. Swallow and log normally. logger.warning( "Failed to reap background processes for turn %s (%s)", task_id, @@ -3514,12 +3223,8 @@ _TURN_STACK_DUMP_FRAME_MARKERS = ( def _dump_wedged_turn_stacks(task_id: str) -> None: """Log the stack of every thread that looks like turn work, at reap time. - When the inactivity reaper fires, the model loop is usually long done and the worker thread - is wedged somewhere in post-turn finalization — but the reaper's hard interrupt frees it, so - the blocked frame is gone before anyone can attach a profiler. Dumping BEFORE the interrupt - names the frame. Best-effort and bounded: in-process frame walking only, only threads whose - stack mentions a turn-machinery marker, output capped per thread. Must never raise into the - reaper. + The reaper's hard interrupt frees the wedged worker before a profiler can attach, so dump BEFORE + interrupting. Best-effort, bounded (turn-machinery threads only, capped output), never raises. """ try: frames = sys._current_frames() @@ -3573,9 +3278,8 @@ def _abandon_timed_out_gateway_turn( return False timeout_fired.set() - # Capture the wedged worker's stack BEFORE interrupting it — the - # interrupt frees the blocked frame, destroying the only evidence of - # where the turn was stuck (see _dump_wedged_turn_stacks). + # Capture the wedged worker's stack BEFORE interrupting: the interrupt frees the blocked + # frame, destroying the only evidence of where the turn was stuck. _dump_wedged_turn_stacks(task_id) agent = agent_holder[0] if agent_holder else None @@ -3661,10 +3365,9 @@ def _is_control_interrupt_message(message: Optional[str]) -> bool: def _strip_response_attachments_for_direct_send(response: str, adapter) -> str: """Return the visible text portion of a response before direct send(). - Queued follow-up resends only replay explicit ``MEDIA:`` attachments in this path. Keep bare - local paths and ordinary image URLs visible because the post-stream uploader intentionally - ignores them. Do not apply a broad ``MEDIA:`` regex after ``extract_media()`` — the extractor - deliberately preserves protected code/inline spans and unsupported/unvalidated tags. + Queued follow-up resends replay only explicit ``MEDIA:`` attachments; bare local paths and image + URLs stay visible because the post-stream uploader ignores them. No broad ``MEDIA:`` regex after + ``extract_media()`` — it deliberately preserves protected code spans and unvalidated tags. """ _, cleaned = adapter.extract_media(response) cleaned = cleaned.replace("[[audio_as_voice]]", "").strip() @@ -3675,13 +3378,8 @@ def _strip_response_attachments_for_direct_send(response: str, adapter) -> str: def _skill_slug_from_frontmatter(skill_md: Path) -> tuple[str | None, str | None]: """Derive the /command slug and declared frontmatter name from a SKILL.md. - Matches the normalization in :func:`agent.skill_commands.scan_skill_commands` so the slug is - the string a user types after ``/`` — derived from frontmatter ``name:``, NOT the directory - name (e.g. dir ``stable-diffusion`` vs slug ``stable-diffusion-image-generation``). Using the - directory name silently broke :func:`_check_unavailable_skill` for every drifted skill. - - Returns ``(slug, declared_name)`` or ``(None, None)`` when the file - can't be read or lacks a ``name:`` in its frontmatter. + Matches ``scan_skill_commands``: the slug comes from frontmatter ``name:``, NOT the directory + name. Returns ``(slug, declared_name)`` or ``(None, None)`` if unreadable or lacking ``name:``. """ try: content = skill_md.read_text(encoding="utf-8", errors="replace") @@ -3716,10 +3414,9 @@ def _skill_slug_from_frontmatter(skill_md: Path) -> tuple[str | None, str | None def _check_unavailable_skill(command_name: str) -> str | None: - """Check if a command matches a known-but-inactive skill. + """Match a command to a known-but-inactive skill. - Returns a helpful message if the skill exists but is disabled or only - available as an optional install. Returns None if no match found. + Returns a hint if the skill exists but is disabled or optional-install only; else None. """ # Normalize: command uses hyphens, skill names may use hyphens or underscores normalized = command_name.lower().replace("_", "-") @@ -3794,12 +3491,11 @@ def _gateway_config_home() -> Path: def _load_gateway_config(config_path: "Path | None" = None) -> dict: - """Load and parse a gateway config.yaml, returning {} on any error. + """Load and parse a gateway config.yaml, returning {} on any error (fail-open). - Defaults to the active gateway home (so tests that monkeypatch ``_hermes_home`` still see - their fixture). Callers handling multiplexed profile routes may pass that profile's explicit - config path. Managed scope is overlaid on the result so administrator-pinned values are - honored — neither read_raw_config nor yaml.safe_load carries the managed merge. Fail-open. + Defaults to the active gateway home (``_hermes_home`` monkeypatches apply); multiplexed callers + may pass a profile path. Managed scope is overlaid here because neither read_raw_config nor + yaml.safe_load carries the managed merge, and pinned values must be honored. """ if config_path is None: config_path = _gateway_config_home() / 'config.yaml' @@ -3807,9 +3503,8 @@ def _load_gateway_config(config_path: "Path | None" = None) -> dict: used_canonical = False try: from hermes_cli.config import get_config_path, read_raw_config - # Fast path: if _hermes_home agrees with the canonical config location, reuse the shared - # cache. Otherwise fall through to a direct read (keeps test fixtures with a monkeypatched - # _hermes_home working). + # Fast path: reuse the shared cache when _hermes_home agrees with the canonical config + # location; otherwise fall through to a direct read (monkeypatched _hermes_home in tests). if config_path == get_config_path(): raw = read_raw_config() used_canonical = True @@ -3826,9 +3521,8 @@ def _load_gateway_config(config_path: "Path | None" = None) -> dict: logger.debug("Could not load gateway config from %s", config_path) raw = {} - # Overlay managed scope. read_raw_config() returns the user's raw YAML WITHOUT the managed merge - # (that lives in load_config/_load_config_impl), so the overlay is required on both paths for - # the gateway to honor pinned values. + # read_raw_config() returns raw YAML WITHOUT the managed merge (that lives in load_config), so + # the managed overlay is required on both paths for the gateway to honor pinned values. try: from hermes_cli import managed_scope raw = managed_scope.apply_managed_overlay(raw if isinstance(raw, dict) else {}) @@ -3837,9 +3531,8 @@ def _load_gateway_config(config_path: "Path | None" = None) -> dict: if not isinstance(raw, dict): return {} # Canonicalize model-id aliases (model.name / model.model → model.default) and migrate stale - # root-level provider/base_url into the model section. The gateway bypasses load_config() (raw - # YAML for speed), so its normalization must be replayed here or ``model: {name: }`` would - # resolve to an empty model while the CLI resolves it correctly. Fail-open. + # root-level provider/base_url into the model section. The gateway bypasses load_config(), so + # without this replay ``model: {name: }`` resolves to an empty model. Fail-open. try: from hermes_cli.config import _normalize_root_model_keys raw = _normalize_root_model_keys(raw) @@ -3851,9 +3544,7 @@ def _load_gateway_config(config_path: "Path | None" = None) -> dict: def _checkpoint_agent_kwargs(config: dict | None) -> dict: """Translate gateway checkpoint config into ``AIAgent`` constructor args. - The gateway reads raw YAML instead of ``load_config()``, so checkpoint - defaults must be supplied here. Keep legacy ``checkpoints: true`` configs - working while giving every gateway-created agent the same limits. + Gateway bypasses ``load_config()``, so defaults are here; legacy ``checkpoints: true`` works. """ cp_cfg = config.get("checkpoints", {}) if isinstance(config, dict) else {} if isinstance(cp_cfg, bool): @@ -3880,9 +3571,8 @@ def _checkpoint_agent_kwargs(config: dict | None) -> dict: def _load_gateway_runtime_config() -> dict: """Load gateway config for runtime reads, expanding supported ``${VAR}`` refs. - Build on ``_load_gateway_config()`` rather than calling the canonical loader directly so both - behaviors stay aligned. Expansion failures are intentionally NOT swallowed — silently - returning the unexpanded dict would mask the very bug this helper exists to fix. + Built on ``_load_gateway_config()``. Expansion failures are deliberately NOT swallowed — + returning the unexpanded dict would mask the very bug this helper fixes. """ cfg = _load_gateway_config() if not isinstance(cfg, dict) or not cfg: @@ -3894,10 +3584,10 @@ def _load_gateway_runtime_config() -> dict: def _resolve_gateway_model(config: dict | None = None) -> str: - """Read model from config.yaml — single source of truth. + """Read model from config.yaml (single source of truth). - Without this, temporary AIAgent instances (e.g. /compress) fall back to the hardcoded default - which fails when the active provider is openai-codex. + Otherwise temporary AIAgent instances (e.g. /compress) use the hardcoded default, which fails + when the active provider is openai-codex. """ cfg = config if config is not None else _load_gateway_config() model_cfg = cfg.get("model", {}) @@ -3914,10 +3604,10 @@ def _channel_override_lookup_keys( thread_id: Optional[str] = None, parent_id: Optional[str] = None, ) -> list[str]: - """Ordered, de-duplicated keys for ``channel_overrides`` lookup. + """Ordered, de-duplicated ``channel_overrides`` lookup keys. - Matches ``resolve_channel_prompt`` semantics: exact thread/channel id first, - then parent channel/forum id (Discord threads inherit parent overrides). + Matches ``resolve_channel_prompt``: exact thread/channel id first, then parent channel/forum id + (Discord threads inherit parent overrides). """ keys: list[str] = [] seen: set[str] = set() @@ -3940,10 +3630,9 @@ def _get_channel_override( thread_id: Optional[str] = None, parent_id: Optional[str] = None, ) -> Optional[ChannelOverride]: - """Return per-channel override for this platform/chat_id, or None. + """Per-channel override for this platform/chat_id, or None. - Looks up ``channel_overrides`` by ``chat_id``, then ``thread_id``, then - ``parent_id`` (forum threads / child channels inherit the parent entry). + Looks up ``chat_id``, then ``thread_id``, then ``parent_id`` (child channels inherit parent). """ platforms = getattr(config, "platforms", None) if not platforms: @@ -3962,13 +3651,9 @@ def _get_channel_override( def _resolve_hermes_bin() -> Optional[list[str]]: - """Resolve the Hermes update command as argv parts. + """Resolve the Hermes update command as argv parts, or ``None``. - Tries in order: 1. ``shutil.which("hermes")`` — standard PATH lookup 2. ``sys.executable -m - hermes_cli.main`` — fallback when Hermes is running from a venv/module invocation and the - ``hermes`` shim is not on PATH - - Returns argv parts ready for quoting/joining, or ``None`` if neither works. + Tries ``shutil.which("hermes")``, then ``sys.executable -m hermes_cli.main`` (no shim on PATH). """ import shutil @@ -3988,11 +3673,10 @@ def _resolve_hermes_bin() -> Optional[list[str]]: def _parse_session_key(session_key: str) -> "dict | None": - """Parse a session key into its component parts. + """Parse a session key (``agent:main:{platform}:{chat_type}:{chat_id}[:{extra}...]``). - Session keys follow the format ``agent:main:{platform}:{chat_type}:{chat_id}[:{extra}...]``. - For group/channel sessions the suffix may be a user_id (per-user isolation) rather than a - thread_id, so we leave ``thread_id`` out to avoid mis-routing. + For group/channel sessions the suffix may be a user_id (per-user isolation), not a thread_id, + so ``thread_id`` is left out to avoid mis-routing. """ parts = session_key.split(":") if len(parts) >= 5 and parts[0] == "agent" and parts[1] == "main": @@ -4022,11 +3706,9 @@ def _format_concise_process_notification( output: str, duration_seconds=None, ) -> str: - """One-line "pretty" completion message for the ``concise`` display mode. + """One-line completion message for the ``concise`` display mode. - Success is a single status line; failure appends a short tail of output so - the user can see what went wrong without the full raw dump. The full - output always remains available to the agent via process(log/wait). + Success is one status line; failure appends a short output tail (full output via process(log)). """ ok = exit_code in {0, None} icon = "✅" if ok else "❌" @@ -4064,9 +3746,8 @@ def _format_gateway_process_notification(evt: dict) -> "str | None": if evt_type == "watch_disabled": return f"[IMPORTANT: {evt.get('message', '')}]" - # Overflow events carry their human-readable summary in `message`, - # like watch_disabled — see the shared formatter in - # tools/process_registry.py. + # Overflow events carry their human-readable summary in `message`, like watch_disabled + # (shared formatter in tools/process_registry.py). if evt_type in ("watch_overflow_tripped", "watch_overflow_released"): return f"[IMPORTANT: {evt.get('message', '')}]" @@ -4096,11 +3777,9 @@ def _format_gateway_process_notification(evt: dict) -> "str | None": def _drain_gateway_watch_events(completion_queue) -> "list[dict]": """Drain gateway-owned watch events without spinning on requeued events. - Watch events are handled by the post-turn gateway drain; process completions are owned by - their per-process watcher task and async delegation completions by - ``_async_delegation_watcher``. Requeueing async events inside - ``while not queue.empty()`` would make the loop non-terminating, so detach the current batch - first, then requeue any events this drain does not own after the queue is empty. + Process completions belong to per-process watchers, async delegation completions to + ``_async_delegation_watcher``; requeueing them inside ``while not queue.empty()`` never + terminates, so detach the batch first and requeue foreign events afterwards. """ watch_events: list[dict] = [] requeue: list[dict] = [] @@ -4125,9 +3804,8 @@ def _drain_gateway_watch_events(completion_queue) -> "list[dict]": return watch_events -# Module-level weak reference to the active GatewayRunner instance. -# Used by tools (e.g. send_message) that need to route through a live -# adapter for plugin platforms. Set in GatewayRunner.__init__(). +# Weak ref to the active GatewayRunner (set in GatewayRunner.__init__), used by tools such as +# send_message that must route through a live adapter for plugin platforms. import weakref as _weakref _gateway_runner_ref: _weakref.ref = lambda: None @@ -4140,23 +3818,20 @@ def _normalize_empty_agent_response( ) -> str: """Normalize empty/None agent responses into user-facing messages. - Consolidates the existing ``failed`` handler and adds a catch-all for the case where the - agent did work (api_calls > 0) but returned no text. Also surfaces a retry hint when the - agent never ran (api_calls == 0, not interrupted/failed): the post-/stop silent-drop pattern - where a stale generation token returns an empty result. + Covers ``failed`` plus the case where the agent did work (api_calls > 0) but returned no text, + and surfaces a retry hint when it never ran (api_calls == 0, not interrupted/failed): the + post-/stop silent-drop where a stale generation token returns an empty result. """ if response: return response if agent_result.get("failed"): - # None-safe: the gateway result dict is built with ``'error': holder.get('error')`` and can - # carry an EXPLICIT None, which bypasses dict.get's default and would render "The request - # failed: None". + # None-safe: the result dict is built with ``'error': holder.get('error')`` and can carry an + # EXPLICIT None, bypassing dict.get's default and rendering "The request failed: None". error_detail = agent_result.get("error") or "unknown error" error_str = str(error_detail).lower() - # Session-persistence failures get a dedicated recovery message. Suggesting /reset here - # would be actively harmful: it destroys the user's conversation context and does nothing to - # fix the underlying storage problem (lock contention, disk exhaustion, ...). + # Session-persistence failures get a dedicated recovery message: suggesting /reset would + # destroy the user's context without fixing the storage problem (locks, full disk). failure_reason = str(agent_result.get("failure_reason") or "") if failure_reason.startswith("session_persistence_failed") or ( "session storage" in error_str @@ -4191,11 +3866,9 @@ def _normalize_empty_agent_response( api_calls = int(agent_result.get("api_calls", 0) or 0) if agent_result.get("interrupted"): - # An interrupted run that did work (api_calls > 0) is the drain of a run the user - # deliberately stopped or steered — its silence is intentional, and any queued/interrupting - # message is delivered by the recursive drain inside _run_agent before this result is seen. - # An interrupted run with ZERO api_calls never processed the message at all (killed by an - # interrupt flag left over from a recent /stop) — silence would swallow it, so surface it. + # Interrupted with api_calls > 0 = deliberately stopped/steered; silence is intentional and + # queued messages come via the recursive drain in _run_agent. With ZERO api_calls the + # message was never processed (stale /stop interrupt flag), so surface it. if api_calls == 0: return ( "⚠️ Your message was interrupted before processing started " @@ -4213,9 +3886,8 @@ def _normalize_empty_agent_response( "This may be a transient error — try sending your message again." ) - # api_calls == 0, not failed, not interrupted: the agent never ran for this turn. This is the - # post-/stop generation-race pattern where the gateway would otherwise silently drop the turn - # (response=0 chars) and the user sees no reply at all. + # api_calls == 0, not failed, not interrupted: the agent never ran (post-/stop generation race). + # Without this the gateway silently drops the turn and the user sees no reply. if ( api_calls == 0 and not agent_result.get("interrupted") @@ -4233,11 +3905,9 @@ def _normalize_empty_agent_response( def _is_gateway_hidden_reasoning_incomplete_turn(agent_result: dict) -> bool: """Detect retry-exhausted turns with hidden reasoning but no visible answer. - The conversation loop returns the retry-exhaustion sentinel as BOTH ``final_response`` and - ``error`` ("Codex response remained incomplete after 3 continuation attempts"), so - ``final_response`` being non-empty does not mean the model produced a visible answer. Hidden - only when the sentinel is present and ``final_response`` is empty or merely echoes it — any - genuinely different text means the model DID answer and must be delivered. + The loop returns the retry-exhaustion sentinel as BOTH ``final_response`` and ``error``, so a + non-empty ``final_response`` proves nothing. Hidden only when the sentinel is present and + ``final_response`` is empty or echoes it; any other text is a real answer. """ if not isinstance(agent_result, dict): return False @@ -4253,12 +3923,10 @@ def _is_gateway_hidden_reasoning_incomplete_turn(agent_result: dict) -> bool: def _should_clear_resume_pending_after_turn(agent_result: dict) -> bool: - """Return True only when a gateway turn really completed successfully. + """True only when a gateway turn really completed successfully. - Restart recovery uses ``resume_pending`` as a durable marker for sessions interrupted during - gateway drain. A soft interrupt can surface as a normal-looking result with an empty final - response; clearing the marker then loses the recovery signal and startup auto-resume has - nothing to schedule. + Restart recovery uses ``resume_pending`` as a durable marker; a soft interrupt can look like a + normal result with an empty final response, and clearing the marker then loses the signal. """ if not isinstance(agent_result, dict): return False @@ -4266,9 +3934,7 @@ def _should_clear_resume_pending_after_turn(agent_result: dict) -> bool: return False if agent_result.get("failed") or agent_result.get("partial") or agent_result.get("error"): return False - if agent_result.get("completed") is False: - return False - return True + return agent_result.get("completed") is not False def _preserve_queued_followup_history_offset( @@ -4277,9 +3943,8 @@ def _preserve_queued_followup_history_offset( ) -> dict: """Carry the outer history offset through queued follow-up drains. - Each recursive ``_run_agent()`` call advances ``history_offset`` to the history it received, - so without correction the outermost persistence step sees only the *last* queued turn as - "new" and silently drops earlier turns from the same drain chain. + Each recursive ``_run_agent()`` advances ``history_offset``; uncorrected, the outer persistence + step sees only the *last* queued turn as "new" and drops earlier ones. """ if not isinstance(followup_result, dict): return followup_result @@ -4301,23 +3966,19 @@ def _preserve_queued_followup_history_offset( async def _dispose_unused_adapter(adapter: "BasePlatformAdapter | None") -> None: """Best-effort dispose for an adapter that never made it onto ``self.adapters``. - When the connect call fails — for any of the three reasons (non-retryable error, retryable - error, exception during connect) — the adapter is dropped without ever being installed, so - nothing else will call its ``disconnect()``. Resources opened in ``__init__`` (e.g. - APIServerAdapter's SQLite store, 2 fds) then leak until GC, which is not prompt for - asyncio-bound objects; at the 300s backoff cap that exhausts the fd ulimit in ~12h and the - gateway becomes a zombie. ``adapter`` may be ``None`` (half-constructed / ``_create_adapter`` - returned None). + A failed connect leaves the adapter uninstalled, so nothing else calls ``disconnect()``; + resources opened in ``__init__`` (e.g. SQLite fds) would leak until GC (not prompt for + asyncio-bound objects) and exhaust the fd ulimit over a long retry loop. ``adapter`` may be + ``None`` (half-constructed / ``_create_adapter`` returned None). """ if adapter is None: return try: await adapter.disconnect() except Exception: - # Half-constructed adapters (e.g. APIServerAdapter that crashed during aiohttp app setup) - # can raise from disconnect() on objects that never finished initializing. We must not let - # that escape and abort the watcher loop. ``asyncio.CancelledError`` is a BaseException, so - # this does not swallow cancellation; dispose failures are best-effort by design. + # Half-constructed adapters (e.g. APIServerAdapter that crashed in aiohttp setup) can raise + # from disconnect(); that must not abort the watcher loop. ``asyncio.CancelledError`` is a + # BaseException, so cancellation is not swallowed; dispose failures are best-effort. logger.debug( "Adapter dispose raised on unowned adapter %r", getattr(adapter, "name", type(adapter).__name__), @@ -4329,11 +3990,9 @@ async def _dispose_unused_adapter(adapter: "BasePlatformAdapter | None") -> None # secondary-profile reconnects share this policy — tune in one place). _RECONNECT_BACKOFF_CAP = 300 -# Seconds a platform may sit continuously in the reconnect queue before the watcher flags it -# NEEDS_ATTENTION in runtime status. Retrying never stops (auto-pause was deliberately removed — a -# transient outage must self-heal without operator action); this only makes a long-lived retry -# loop loud. A dead bot token, a revoked Discord intent, or a deterministically crashing sidecar -# all present as "retrying" forever without this signal. 0 disables. +# Seconds a platform may sit continuously in the reconnect queue before it is flagged +# NEEDS_ATTENTION. Retrying never stops (transient outages must self-heal); this only makes a +# permanently-failing loop loud. 0 disables. _RECONNECT_ATTENTION_AFTER_SECONDS = _float_env( "HERMES_RECONNECT_ATTENTION_AFTER_SECONDS", 7200 ) @@ -4345,12 +4004,9 @@ def _reconnect_backoff(attempt: int) -> int: def _reconnect_needs_attention(info: dict, now: float) -> bool: - """Return True when a reconnect-queue entry has been continuously queued long enough to warrant - a NEEDS_ATTENTION signal. + """True when a reconnect-queue entry has waited long enough for NEEDS_ATTENTION. - ``queued_at`` is (re)stamped whenever the platform (re)enters the queue, so a platform that - reconnects successfully and later fails again starts a fresh clock — only *continuous* - failure escalates. + ``queued_at`` is re-stamped on each (re)entry, so only *continuous* failure escalates. """ if _RECONNECT_ATTENTION_AFTER_SECONDS <= 0: return False # escalation disabled @@ -4362,11 +4018,9 @@ def _reconnect_needs_attention(info: dict, now: float) -> bool: class TurnRunner: - """Per-turn collaborator carrying the tool-progress callbacks that used to be nested closures - inside ``GatewayRunner._run_agent_inner``. + """Per-turn collaborator carrying ``GatewayRunner._run_agent_inner``'s tool-progress callbacks. - Module-global references (logger, cfg_get, BasePlatformAdapter, ...) resolve in this same - module exactly as before. + Module-global references (logger, cfg_get, BasePlatformAdapter, ...) resolve in this module. """ def __init__(self, runner: "GatewayRunner", ctx: TurnContext) -> None: @@ -4376,10 +4030,10 @@ class TurnRunner: def progress_callback(self, event_type: str, tool_name: str = None, preview: str = None, args: dict = None, **kwargs): """Callback invoked by agent on tool lifecycle events.""" ctx = self._ctx - # Failed subagent → one clean user-facing notice. Handled FIRST, before every progress-queue - # gate: platforms with tool_progress off must still hear about a dead delegation, or it - # looks like the agent dropped the task. Success/interrupt completions stay quiet; - # only terminal failure statuses render, via the same notice rail as credit warnings. + # Failed subagent → one clean user-facing notice, handled FIRST, before every progress-queue + # gate: platforms with tool_progress off must still hear about a dead delegation. Only + # terminal failure statuses render (same notice rail as credit warnings); success/interrupt + # stay quiet. if event_type == "subagent.complete": _sub_status = kwargs.get("status") try: @@ -4403,9 +4057,8 @@ class TurnRunner: except Exception: logger.debug("subagent failure notice failed", exc_info=True) return - # Live status line (Slack's assistant status): stash the current tool phrase on the adapter; - # the _keep_typing refresh renders it within a couple of seconds. Plain dict write — safe - # from the agent's sync worker thread, no event-loop hop needed. + # Live status line (Slack assistant status): stash the tool phrase on the adapter; the + # _keep_typing refresh renders it. Plain dict write, safe from the sync worker thread. if ( ctx._live_status_adapter is not None and ctx._live_status_mode != "off" @@ -4425,9 +4078,8 @@ class TurnRunner: ctx._live_status_adapter.set_status_text(ctx.source.chat_id, None) except Exception as _ls_err: logger.debug("live status update failed: %s", _ls_err) - # "log" mode: append tool.started lines to the log queue and stay - # silent in chat. Handled before the progress_queue guard because - # log mode runs without a chat progress queue. + # "log" mode: append tool.started lines to the log queue, silent in chat. Handled before + # the progress_queue guard because log mode runs without a chat progress queue. if ctx.log_queue is not None: if event_type == "tool.started" and tool_name and tool_name != "_thinking": ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S") @@ -4438,9 +4090,8 @@ class TurnRunner: if not ctx.progress_queue or not ctx._run_still_current(): return - # First-touch onboarding: the first time a tool takes longer than _LONG_TOOL_THRESHOLD_S - # during a run that's streaming every tool (progress_mode == "all"), append a one-time hint - # suggesting /verbose. The CLI has its own trigger. + # First-touch onboarding: the first time a tool exceeds _LONG_TOOL_THRESHOLD_S while + # streaming every tool (progress_mode == "all"), append a one-time /verbose hint. if event_type == "tool.completed" and not ctx.long_tool_hint_fired[0]: try: duration = kwargs.get("duration") or 0 @@ -4475,9 +4126,8 @@ class TurnRunner: ctx.progress_queue.put(msg) return - # Native task cards consume the authoritative ID-bearing tool_start/tool_complete callbacks - # instead. Do not also enqueue name-correlated text events, which would duplicate cards and - # mispair concurrent calls to the same tool. + # Native task cards consume the ID-bearing tool_start/tool_complete callbacks instead; + # name-correlated text events would duplicate cards and mispair concurrent same-tool calls. if ctx._native_slack_task_cards and event_type in { "tool.started", "tool.completed", @@ -4493,20 +4143,15 @@ class TurnRunner: if event_type not in {"tool.started",}: return - # Never render a progress bubble for the clarify tool: send_clarify IS the user-facing - # rendering, so a bubble is pure duplication — and verbose mode would dump the raw tool-call - # args JSON. Because the progress queue drains on a background task, that raw JSON typically - # lands right underneath the rendered prompt. + # Never render a progress bubble for clarify: send_clarify IS the user-facing rendering, so + # a bubble is duplication, and verbose mode would dump the raw tool-call args JSON, which + # (progress queue drains on a background task) lands right under the rendered prompt. if tool_name == "clarify": return - # Suppress tool-progress bubbles once the user has sent `stop`. - # When the LLM response carries N parallel tool calls, the agent - # fires N "tool.started" events back-to-back before checking for - # interrupts — without this guard, a late `stop` still renders - # all N as 🔍 bubbles, making the interrupt feel ignored. - # (agent lives in run_sync's scope; agent_holder[0] is the shared - # handle across nested scopes — see line ~9607.) + # Suppress tool-progress bubbles once the user sent `stop`: N parallel tool calls fire N + # "tool.started" events before the interrupt check, so a late `stop` would still render + # all N bubbles. (agent_holder[0] is the shared agent handle across nested scopes.) try: _agent_for_interrupt = ctx.agent_holder[0] if ctx.agent_holder else None if _agent_for_interrupt is not None and getattr( @@ -4525,19 +4170,10 @@ class TurnRunner: from agent.display import get_tool_emoji emoji = get_tool_emoji(tool_name, default="⚙️") - # Markdown-capable platforms render a terminal command as a fenced - # code block instead of the compact `terminal: "cmd…"` preview. - # Gated on the adapter's ``supports_code_blocks`` capability so - # plain-text platforms keep the short line. No language tag is - # emitted — Slack mrkdwn renders the tag as a literal first code - # line ("bash"), and a bare fence renders correctly everywhere - # that supports blocks. - # - # Verbose mode shows the FULL command. Non-verbose ("all"/"new") - # modes still wrap in a fence but truncate to a single line capped - # at ``tool_preview_length`` (default 40) so a long or multi-line - # command doesn't render as a huge block — matching the budget the - # non-terminal preview path already applies (#42634). + # Markdown platforms (``supports_code_blocks``) fence terminal commands; plain-text ones + # keep the compact `terminal: "cmd…"` line. No language tag: Slack mrkdwn renders it as a + # literal first code line. Verbose shows the FULL command; "all"/"new" fence but truncate to + # one line capped at ``tool_preview_length`` (default 40), the non-terminal preview budget. _code_block_full = None _code_block_short = None try: @@ -4553,9 +4189,8 @@ class TurnRunner: ): from agent.display import get_tool_preview_max_len _cmd_full = args["command"].rstrip() - # Consecutive terminal calls: drop the repeated - # "💻 terminal" header so back-to-back commands render as - # adjacent code blocks under a single header. + # Consecutive terminal calls drop the repeated "💻 terminal" header so back-to-back + # commands render as adjacent code blocks under one header. _block_header = ( "" if ctx.last_was_terminal_block[0] else f"{emoji} {tool_name}\n" ) @@ -4583,9 +4218,8 @@ class TurnRunner: from agent.display import get_tool_preview_max_len _pl = get_tool_preview_max_len() args_str = json.dumps(args, ensure_ascii=False, default=str) - # When tool_preview_length is 0 (default), don't truncate - # in verbose mode — the user explicitly asked for full - # detail. Platform message-length limits handle the rest. + # tool_preview_length 0 (default) = no truncation in verbose mode; the user asked + # for full detail and platform message-length limits handle the rest. if _pl > 0 and len(args_str) > _pl: args_str = args_str[:_pl - 3] + "..." msg = f"{emoji} {tool_name}({list(args.keys())})\n{args_str}" @@ -4596,11 +4230,8 @@ class TurnRunner: ctx.progress_queue.put(msg) return - # "all" / "new" modes: short preview, respects tool_preview_length - # config (defaults to 40 chars when unset to keep gateway messages - # compact — unlike CLI spinners, these persist as permanent messages). - # Terminal commands on markdown platforms get a single-line capped - # fenced block (built above) instead of the truncated preview. + # "all" / "new" modes: short preview capped by tool_preview_length (default 40; gateway + # messages persist, unlike CLI spinners). Markdown terminal commands use the fence above. if _code_block_short is not None: msg = _code_block_short ctx.last_was_terminal_block[0] = True @@ -4624,9 +4255,8 @@ class TurnRunner: preview = _progress_adapter.format_tool_preview(_prepared_preview) else: preview = _prepared_preview.text - # Friendly labels: render a human-phrased line for built-in tools ("🔍 Searching the web - # for ...") by prefixing the verb onto the preview the callback already computed (so the - # command/url/query is preserved). + # Friendly labels: human-phrased line for built-in tools ("🔍 Searching the web for ...") + # by prefixing the verb onto the computed preview, so the command/url/query is kept. _verb = get_tool_verb(tool_name) if _verb: if verb_drops_preview(tool_name): @@ -4640,9 +4270,8 @@ class TurnRunner: msg = f"{emoji} {tool_name}..." ctx.last_was_terminal_block[0] = False - # Dedup: collapse consecutive identical progress messages. - # Common with execute_code where models iterate with the same - # code (same boilerplate imports → identical previews). + # Dedup consecutive identical progress messages (common with execute_code: same + # boilerplate imports → identical previews). if msg == ctx.last_progress_msg[0]: ctx.repeat_count[0] += 1 # Native-stream-progress routing: dedup updates the last line @@ -4659,9 +4288,8 @@ class TurnRunner: ctx.last_progress_msg[0] = msg ctx.repeat_count[0] = 0 - # Native-stream-progress routing: if the stream consumer is active - # and using native streaming, inject progress directly into the - # stream bubble instead of the separate progress queue. + # If the stream consumer is active with native streaming, inject progress into the stream + # bubble instead of the separate progress queue. _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None if _sc is not None and getattr(_sc, "accepts_tool_progress", False): _sc.on_tool_progress(msg) @@ -4672,8 +4300,7 @@ class TurnRunner: async def _send_native_task_card_progress(self, adapter) -> None: """Drain the progress queue into Slack-native plan/task cards. - On any native failure, falls back to an editable in-thread text message so progress stays - live for the rest of the turn. + On any native failure, fall back to an editable in-thread message so progress stays live. """ ctx = self._ctx tasks: Dict[str, Dict[str, str]] = {} @@ -4732,9 +4359,8 @@ class TurnRunner: task = tasks.get(call_id) if task is None: - # Completion-only events are rare but valid on some - # runtimes. Keep their real ID instead of guessing a - # same-name pending call. + # Completion-only events are rare but valid on some runtimes; keep their real ID + # instead of guessing a same-name pending call. task = { "id": call_id, "title": _compact(tool_name), @@ -4842,12 +4468,8 @@ class TurnRunner: return finally: if hasattr(adapter, "stop_native_task_card_progress"): - # Best-effort: this finally runs on the turn-cleanup path. - # An escaping transport exception here propagated through - # the cleanup awaits (which caught only CancelledError) and - # skipped final-delivery logic (review B7). Adapters now - # return failed SendResults, but defend the seam anyway — - # any adapter, any transport. + # Best-effort on the turn-cleanup path: an escaping transport exception would skip + # final-delivery logic (cleanup awaits catch only CancelledError). try: await adapter.stop_native_task_card_progress( ctx.source.chat_id, @@ -4877,12 +4499,9 @@ class TurnRunner: await self._send_native_task_card_progress(adapter) return - # Skip tool progress for platforms that don't support message - # editing (e.g. iMessage/BlueBubbles) — each progress update - # would become a separate message bubble, which is noisy. - # getattr, not attribute access: duck-typed adapters (test fakes, - # minimal plugin adapters) may not define edit_message at all — - # "missing" means the same thing as "base no-op": can't edit. + # Skip tool progress for platforms that can't edit messages (e.g. iMessage/BlueBubbles): + # each update would be a separate bubble. getattr, not attribute access: duck-typed + # adapters (test fakes, minimal plugins) may lack edit_message — treated as "can't edit". _adapter_edit = getattr(type(adapter), "edit_message", None) if _adapter_edit is None or _adapter_edit is BasePlatformAdapter.edit_message: while not ctx.progress_queue.empty(): @@ -4907,9 +4526,8 @@ class TurnRunner: _raw_progress_limit = int(getattr(adapter, "MAX_MESSAGE_LENGTH", 4000) or 4000) except Exception: _raw_progress_limit = 4000 - # Per-chat resolution (relay adapter fronting N platforms): the cap - # and length unit follow the chat's underlying platform. Native - # adapters return their scalar/property unchanged. + # Per-chat resolution (relay adapter fronting N platforms): cap and length unit follow the + # chat's underlying platform; native adapters return their scalar/property unchanged. if isinstance(adapter, BasePlatformAdapter): try: _raw_progress_limit = int( @@ -4992,10 +4610,8 @@ class TurnRunner: async def _roll_progress_overflow_if_needed() -> bool: """Start fresh editable progress bubbles before a bubble exceeds limit. - Returns True when it delivered/split the current buffer, or when - a transient edit failure left the buffer and message identity - intact for a later retry. In either case the caller should skip - the normal send/edit path for this tick. + Returns True when it delivered/split the buffer or a transient edit failure left it + intact for retry — either way the caller skips the normal send/edit path this tick. """ nonlocal progress_msg_id, progress_lines, can_edit if not progress_lines or not can_edit: @@ -5027,9 +4643,8 @@ class TurnRunner: if result.success and result.message_id: progress_msg_id = result.message_id - # The newest continuation is now the only mutable bubble. Keep - # just its lines so subsequent edits update it instead of - # replaying the full historical transcript into new messages. + # The newest continuation is the only mutable bubble: keep just its lines so later + # edits update it instead of replaying the full transcript into new messages. progress_lines = groups[-1] return True @@ -5065,11 +4680,8 @@ class TurnRunner: progress_lines[-1] = f"{base_msg} (×{count + 1})" msg = progress_lines[-1] if progress_lines else base_msg elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": - # Content bubble just landed on the platform — close off the current tool- - # progress bubble so the next tool starts a fresh bubble below the content. - # Otherwise tool lines keep editing the ORIGINAL progress message above the new - # content and the chat reads out of order. - # Mirrors GatewayStreamConsumer.on_segment_break on the content side. + # Content bubble landed — close the tool-progress bubble so the next tool starts + # fresh below it; else tool edits hit the ORIGINAL message above (out of order). progress_msg_id = None progress_lines = [] ctx.last_progress_msg[0] = None @@ -5086,15 +4698,13 @@ class TurnRunner: await adapter.send_typing(ctx.source.chat_id, metadata=ctx._progress_metadata) continue - # Throttle edits: batch rapid tool updates into fewer API calls to avoid hitting - # Telegram flood control. (grammY auto-retry pattern: proactively rate-limit instead - # of reacting to 429s.) + # Throttle edits: batch rapid tool updates into fewer API calls to avoid Telegram + # flood control (grammY pattern: proactively rate-limit rather than react to 429s). _now = time.monotonic() _remaining = _PROGRESS_EDIT_INTERVAL - (_now - _last_edit_ts) if _remaining > 0: - # Wait out the throttle interval, then loop back to - # drain any additional queued messages before sending - # a single batched edit. + # Wait out the throttle interval, then loop back to drain any further queued + # messages before sending a single batched edit. await asyncio.sleep(_remaining) continue @@ -5107,11 +4717,8 @@ class TurnRunner: result = await _edit_progress_message(progress_msg_id, full_text) if not result.success: _err = (getattr(result, "error", "") or "").lower() - # Transient network errors (ConnectError, timeouts) - # must not permanently disable progress-message - # editing — the next cycle can catch up. Only - # permanent failures (flood control, message not - # found, permissions) should set can_edit = False. + # Transient network errors (ConnectError, timeouts) must not disable editing; + # only permanent failures (flood, not found, permissions) set can_edit = False. if getattr(result, "retryable", False): logger.debug( "[%s] Transient edit failure — keeping can_edit=True", @@ -5183,16 +4790,13 @@ class TurnRunner: progress_lines[-1] = f"{base_msg} (×{count + 1})" await _roll_progress_overflow_if_needed() elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": - # Content-bubble marker during drain: close off - # the current progress bubble and start a fresh - # one for any tool lines that arrived after. + # Content-bubble marker during drain: close the current progress bubble + # and start a fresh one for tool lines that arrived after. await _roll_progress_overflow_if_needed() if can_edit and progress_lines and progress_msg_id: _pending_text = _progress_text(progress_lines) - try: + with suppress(Exception): await _edit_progress_message(progress_msg_id, _pending_text) - except Exception: - pass progress_msg_id = None progress_lines = [] ctx.last_progress_msg[0] = None @@ -5207,10 +4811,8 @@ class TurnRunner: await _roll_progress_overflow_if_needed() if can_edit and progress_lines and progress_msg_id: full_text = _progress_text(progress_lines) - try: + with suppress(Exception): await _edit_progress_message(progress_msg_id, full_text) - except Exception: - pass return except Exception as e: logger.error("Progress message error: %s", e) @@ -5237,10 +4839,9 @@ class TurnRunner: except Exception as _ack_err: logger.debug("voice ack schedule failed: %s", _ack_err) - # ── Slack-native task cards: ID-bearing lifecycle callbacks ── These ride - # agent.tool_start_callback / agent.tool_complete_callback so start/completion events correlate - # by the REAL tool-call id — the name-correlated text events in progress_callback would - # duplicate cards and mispair concurrent calls to the same tool. + # ── Slack-native task cards: ID-bearing lifecycle callbacks ── ride agent.tool_start_callback / + # agent.tool_complete_callback so start/completion correlate by the REAL tool-call id; the + # name-correlated progress_callback text events would duplicate cards and mispair concurrent calls. def native_tool_start_callback(self, call_id, tool_name, args): """Queue an ID-correlated native progress start from the agent thread.""" @@ -5302,9 +4903,8 @@ class TurnRunner: ctx = self._ctx if not ctx._run_still_current(): return - # prev_tools may be list[str] or list[dict] with "name"/"result" - # keys. Normalise to keep "tool_names" backward-compatible for - # user-authored hooks that do ', '.join(tool_names)'. + # prev_tools may be list[str] or list[dict] with "name"/"result" keys. Normalise so + # "tool_names" stays backward-compatible for user hooks that do ', '.join(tool_names). _names: list[str] = [] for _t in (prev_tools or []): if isinstance(_t, dict): @@ -5338,14 +4938,11 @@ class TurnRunner: def _attach_session_title_callback(self, agent, ctx) -> None: """Wire the platform thread-rename lane onto the agent as `_on_session_title`. - The session titler runs inside the turn prologue now (it derives the title from the - user's first message, so it no longer needs the response), which means the callback has - to be attached before the run rather than registered after it. + The titler runs in the turn prologue, so attach before the run, not after it. """ try: - # Gateway auto-title failures must NOT be surfaced as user-visible messages — they are - # not actionable to the end user. Overriding the failure sink here keeps CLI mode on the - # agent's _emit_auxiliary_failure path while the gateway logs at debug. + # Gateway auto-title failures are not user-actionable, so never surface them as messages; + # overriding the failure sink keeps CLI on _emit_auxiliary_failure while gateway logs debug. def _title_failure_cb(task: str, exc: BaseException) -> None: logger.debug( "Gateway auto-title failure suppressed (not user-visible): %s: %s", @@ -5357,9 +4954,8 @@ class TurnRunner: session_id = getattr(agent, "session_id", None) source = ctx.source - # Both lanes below spend a rate-limited platform call per title, so they take the - # model's title and skip the derived one — see TitleCallback. Renaming twice costs - # double and Discord's 2-per-10-min channel budget can burn on the throwaway title. + # Both lanes spend a rate-limited platform call per title, so they use the model's title + # only (TitleCallback); renaming twice burns Discord's 2-per-10-min budget on a throwaway. if self._runner._is_telegram_topic_lane(source): agent._on_session_title = lambda title, title_source: ( title_source == "llm" @@ -5421,32 +5017,21 @@ class TurnRunner: def run_sync(self): ctx = self._ctx - # As a method the turn message lives on the shared TurnContext instead: every rebind writes - # `ctx.message`, so the outer `_run_agent_inner` body observes the updated value exactly as - # it did through the closure cell. + # As a method the turn message lives on the shared TurnContext: every rebind writes + # `ctx.message`, so the outer `_run_agent_inner` body sees the update as via the closure cell. - # session_key is propagated via contextvars in _set_session_env() - # (_SESSION_KEY) and via set_current_session_key() (_approval_session_key) - # below — both concurrency-safe and inherited by tool worker threads. - # We deliberately do NOT write os.environ["HERMES_SESSION_KEY"] here: - # os.environ is process-global, so concurrent gateway sessions (e.g. - # two Discord threads) would clobber each other's value, and a tool - # thread whose contextvar is unset would fall back to os.environ and - # read the wrong session key — misrouting command-approval prompts to - # the wrong thread (#24100). The non-gateway surfaces don't depend on - # this write: CLI and cron bind the session via contextvars - # (set_current_session_key / session context), and only the TUI - # slash-worker *subprocess* exports HERMES_SESSION_KEY (from its own - # --session-key argv, a separate process) — so removing this in-process - # gateway write does not affect any of them. + # session_key propagates via contextvars (_set_session_env / set_current_session_key): + # concurrency-safe and inherited by tool worker threads. Deliberately do NOT write + # os.environ["HERMES_SESSION_KEY"]: it is process-global, so concurrent sessions would clobber + # each other and a tool thread with an unset contextvar would read the wrong key, misrouting + # approvals. Only the TUI slash-worker subprocess exports the env var (from its own argv). # Map platform enum to the platform hint key the agent understands. # Platform.LOCAL ("local") maps to "cli"; others pass through as-is. platform_key = "cli" if ctx.source.platform == Platform.LOCAL else ctx.source.platform.value - # Combine platform context, YAML channel_prompts hint for this chat, - # channel_overrides system_prompt (or global ephemeral), and gateway - # ephemeral prompt from _get_system_prompt_for_channel. + # Combine platform context, YAML channel_prompts hint for this chat, channel_overrides + # system_prompt (or global ephemeral), and the gateway ephemeral prompt. combined_ephemeral = ctx.context_prompt or "" event_channel_prompt = (ctx.channel_prompt or "").strip() if event_channel_prompt: @@ -5501,9 +5086,8 @@ class TurnRunner: from gateway.config import StreamingConfig _scfg = StreamingConfig() - # Per-platform streaming gate: display.platforms..streaming - # can disable streaming for specific platforms even when the global - # streaming config is enabled. + # Per-platform streaming gate: display.platforms..streaming can disable streaming + # for specific platforms even when the global streaming config is enabled. _plat_streaming = ctx.resolve_display_setting( ctx.user_config, platform_key, "streaming" ) @@ -5552,9 +5136,8 @@ class TurnRunner: except Exception as _sc_err: logger.debug("Could not set up stream consumer: %s", _sc_err) - # When text streaming is off but streaming TTS is active, - # install a TTS-only delta callback so the consumer still - # receives LLM deltas for audio synthesis (#60671). + # Text streaming off but streaming TTS active: install a TTS-only delta callback so the + # consumer still receives LLM deltas for audio synthesis. if _stream_delta_cb is None and _stts_consumer_ref is not None: def _stream_delta_cb(text: str) -> None: if ctx._run_still_current(): @@ -5586,21 +5169,18 @@ class TurnRunner: turn_route = self._runner._resolve_turn_agent_config(ctx.message, model, runtime_kwargs) # Per-platform skip_context_files — messaging platforms can opt out of filesystem-heavy - # context-file discovery (SOUL.md, AGENTS.md, .cursorrules) to cut AIAgent construction - # latency. + # context-file discovery (SOUL.md, AGENTS.md, .cursorrules) to cut AIAgent build latency. _platforms_gw_cfg = (ctx.user_config.get("gateway") or {}).get("platforms") or {} - # ``hermes gateway setup`` writes ``gateway.platforms`` as a LIST of enabled platform names - # (e.g. ``- telegram``), not a dict. Treat any non-dict shape as "no per-platform overrides" - # instead of crashing on ``.get()`` for every incoming turn. + # ``hermes gateway setup`` writes ``gateway.platforms`` as a LIST of enabled platform names, + # not a dict; treat any non-dict shape as "no per-platform overrides" rather than crashing. if not isinstance(_platforms_gw_cfg, dict): _platforms_gw_cfg = {} _plat_gw_cfg = _platforms_gw_cfg.get(platform_key) or {} _skip_context = _plat_gw_cfg.get("skip_context_files") skip_context_files = bool(_skip_context) if _skip_context is not None else False - # Check agent cache — reuse the AIAgent from the previous message - # in this session to preserve the frozen system prompt and tool - # schemas for prompt cache hits. + # Agent cache: reuse this session's previous AIAgent to preserve the frozen system prompt + # and tool schemas for prompt cache hits. _sig = self._runner._agent_config_signature( turn_route["model"], turn_route["runtime"], @@ -5616,11 +5196,10 @@ class TurnRunner: _cache_lock = getattr(self._runner, "_agent_cache_lock", None) _cache = getattr(self._runner, "_agent_cache", None) - # Peek at the cached entry's snapshot session_id (if any) so we can check, OUTSIDE the cache - # lock, whether THAT session_id is a DEAD session in state.db. The #54947 rule treats - # "cached sid != current sid" as an intentional switch and reuses the agent, but the #54878 - # self-heal yields the same shape with an agent bound to a DEAD session; reusing it writes - # the routing key back onto the dead sid and loops every message. + # Peek at the cached entry's snapshot session_id so we can check, OUTSIDE the cache lock, + # whether it is a DEAD session in state.db. "cached sid != current sid" normally means an + # intentional switch (reuse the agent), but the routing-key self-heal yields the same shape + # with an agent bound to a DEAD session; reusing it re-binds the dead sid and loops. _peek_cached_sid = None if _cache_lock and _cache is not None: with _cache_lock: @@ -5640,10 +5219,9 @@ class TurnRunner: except Exception: _cached_sid_is_dead = False - # Detect cross-process writes: when another process (e.g. hermes dashboard) appends to the - # same session in the shared SessionDB, the cached agent's in-memory transcript becomes - # stale. Compare current message_count against the count recorded at cache time; on - # mismatch invalidate so a fresh agent re-reads from disk. + # Cross-process write guard: another process (e.g. hermes dashboard) appending to the same + # SessionDB session makes the cached agent's transcript stale. On message_count mismatch vs + # the count recorded at cache time, invalidate so a fresh agent re-reads from disk. _current_msg_count = None if self._runner._session_db is not None and ctx.session_id: try: @@ -5664,30 +5242,26 @@ class TurnRunner: # taken for — used to skip the guard when the active session_id differs. _cached_mc = cached[2] if len(cached) > 2 else None _cached_sid = cached[3] if len(cached) > 3 else None - # If the snapshot belongs to a different session_id (same session_key, different - # conversation), the message_count comparison is meaningless — the counts track - # DIFFERENT DB rows. REUSE the cached agent rather than rebuild and bust the - # prompt cache on every session switch. + # Snapshot from a different session_id (same session_key, other conversation): the + # counts track DIFFERENT DB rows, so the comparison is meaningless. REUSE the cached + # agent rather than rebuild and bust the prompt cache on every session switch. _session_id_mismatch = ( _cached_sid is not None and ctx.session_id is not None and _cached_sid != ctx.session_id ) - # Re-validate the OUTSIDE-lock dead-session peek against the tuple actually read - # under THIS lock — the cache entry could have been replaced between the peek - # and this lock acquisition, and a stale "dead" verdict must never be applied to - # a different (possibly live) cached agent. + # Re-validate the OUTSIDE-lock dead-session peek against the tuple read under THIS + # lock: the entry may have been replaced between peek and acquisition, and a stale + # "dead" verdict must never be applied to a different (possibly live) cached agent. _stale_dead_sid_reuse = ( _session_id_mismatch and _cached_sid_is_dead and _cached_sid == _peek_cached_sid ) if _stale_dead_sid_reuse: - # #54878 x #54947 interaction: the routing key was just self-healed away - # from a session that state.db already marked ended, but the cached AIAgent - # here still belongs to that DEAD session_id. Not a sibling conversation — - # a stale agent; reusing it would write the routing key back onto the dead - # sid and undo the self-heal. Discard and rebuild fresh. + # The routing key was just self-healed away from a session state.db marked + # ended, but this cached AIAgent still belongs to that DEAD session_id. + # Reusing it would re-bind the dead sid and undo the self-heal; rebuild fresh. logger.info( "Agent cache invalidated for session %s: " "cached agent's session_id %s is ended in " @@ -5699,9 +5273,8 @@ class TurnRunner: evicted = self._runner._agent_cache.pop(ctx.session_key, None) _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: - # Same deferred-cleanup rationale as the cross-process branch below: - # don't block the event loop / cache lock on memory-provider shutdown or - # socket teardown. + # Same deferred-cleanup rationale as the cross-process branch below: don't + # block the event loop / cache lock on memory-provider or socket teardown. _xproc_evicted_agent = _ev_agent elif ( not _session_id_mismatch @@ -5720,29 +5293,20 @@ class TurnRunner: evicted = self._runner._agent_cache.pop(ctx.session_key, None) _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: - # Defer cleanup until AFTER the lock is - # released — _cleanup_agent_resources / - # release_clients can block on memory-provider - # shutdown and socket teardown, and running it - # here would stall the gateway event loop while - # _sweep_idle_cached_agents (session-expiry - # watcher) waits on the same lock, blocking - # Discord heartbeats (#52197). The same session - # rebuilds a fresh agent immediately below, so - # use the SOFT release that preserves the - # session's terminal sandbox / browser / bg - # processes for the rebuilt agent to inherit — - # mirrors _evict_cached_agent / idle-sweep. + # Defer cleanup until AFTER the lock is released: release_clients can + # block on memory-provider/socket teardown, stalling the event loop while + # the idle sweeper waits on this lock (blocking Discord heartbeats). The + # session rebuilds a fresh agent below, so use the SOFT release that keeps + # its terminal sandbox / browser / bg processes for the new agent to + # inherit — mirrors _evict_cached_agent / idle-sweep. _xproc_evicted_agent = _ev_agent else: agent = cached[0] # Refresh LRU order so the cap enforcement evicts # truly-oldest entries, not the one we just used. if hasattr(_cache, "move_to_end"): - try: + with suppress(KeyError): _cache.move_to_end(ctx.session_key) - except KeyError: - pass self._runner._init_cached_agent_for_turn(agent, ctx._interrupt_depth) # Refresh agent max_iterations from current config # (cached agent may have been created with old config) @@ -5750,18 +5314,16 @@ class TurnRunner: logger.debug("Reusing cached agent for session %s", ctx.session_key) reused_cached_agent = True - # Lock released — refresh the fallback chain from disk for the reused agent OUTSIDE the - # cache lock (config.yaml read is disk I/O; the idle-sweep watcher contends on this lock and - # stalls Discord heartbeats — same reasoning as #52197). A chain configured after this agent - # was cached must reach the next turn; per-session serialization keeps this safe post-lock. + # Lock released — refresh the reused agent's fallback chain from disk OUTSIDE the cache lock + # (disk I/O under the lock stalls the idle-sweep watcher and Discord heartbeats). A chain + # configured after caching must reach the next turn; per-session serialization keeps it safe. if reused_cached_agent and agent is not None: self._runner._apply_fallback_chain_to_agent( agent, self._runner._refresh_fallback_model(), ) - # Lock released — now schedule cleanup of any cross-process-evicted agent on a daemon thread - # so memory-provider shutdown / socket teardown never blocks the gateway event loop or the - # cache lock the session-expiry watcher needs. + # Lock released — schedule cleanup of any cross-process-evicted agent on a daemon thread so + # memory-provider/socket teardown never blocks the gateway loop or the expiry watcher's lock. if _xproc_evicted_agent is not None: try: threading.Thread( @@ -5773,10 +5335,8 @@ class TurnRunner: except Exception: # Interpreter shutdown or thread-spawn failure — release # inline as a best-effort fallback. - try: + with suppress(Exception): self._runner._release_evicted_agent_soft(_xproc_evicted_agent) - except Exception: - pass if agent is None: # Config changed or first message — create fresh agent @@ -5820,24 +5380,21 @@ class TurnRunner: ) if _cache_lock and _cache is not None: with _cache_lock: - # Record the session_id the snapshot was taken for alongside the message_count, - # so the cross-process guard can skip the (meaningless) count comparison when - # the active session_id later switches under the same session_key. + # Record the snapshot's session_id with message_count so the cross-process guard can + # skip the meaningless count comparison if the active session_id later switches. _cache[ctx.session_key] = ( agent, _sig, _current_msg_count, ctx.session_id, ) self._runner._enforce_agent_cache_cap() logger.debug("Created new agent for session %s (sig=%s)", ctx.session_key, _sig) - # Per-message state — callbacks and reasoning config change every turn and must not be baked - # into the cached agent constructor. The progress callback is ALWAYS attached (never gated - # to None): its body gates each event class itself, and subagent-failure notices must fire - # even with tool_progress/thinking off — a None gate made dead subagents vanish silently. - # _thinking scratch relays on thinking_progress independently of tool_progress. + # Per-message state — callbacks and reasoning config change every turn, so they aren't baked + # into the cached agent. The progress callback is ALWAYS attached (never gated to None): its + # body gates each event class, and subagent-failure notices must fire even with + # tool_progress/thinking off — a None gate made dead subagents vanish silently. agent.tool_progress_callback = ctx.progress_callback - # Compose ID-bearing lifecycle consumers: Discord's one-time voice - # ack and Slack's native task cards both ride the authoritative - # start callback, so neither has to infer identity from tool names. + # Compose ID-bearing lifecycle consumers: Discord's one-time voice ack and Slack's task cards + # both ride the authoritative start callback, so neither infers identity from tool names. _combined_start_cb = ctx.native_tool_start_callback or ctx.voice_ack_callback agent.tool_start_callback = ( _combined_start_cb @@ -5857,11 +5414,9 @@ class TurnRunner: agent.stream_delta_callback = _stream_delta_cb agent.interim_assistant_callback = _interim_assistant_cb if _want_interim_messages else None agent.status_callback = ctx._status_callback_sync - # Credits / out-of-band notices (usage bands, depletion, restored). Fires from the agent's - # sync worker thread, so we hop onto the gateway loop with safe_schedule_threadsafe - same - # pattern as _status_callback_sync. The fired-once latch lives on the cached agent, so a band - # crossing pushes once (no per-turn re-nag). The clear callback is a no-op: a sent platform - # message can't be retracted. + # Credits / out-of-band notices (usage bands, depletion, restored) fire from the agent's sync + # worker thread, so hop onto the gateway loop via safe_schedule_threadsafe. Fired-once latch + # lives on the cached agent (no per-turn re-nag); clear is a no-op — sends can't be retracted. def _notice_callback_sync(notice) -> None: if not ctx._status_adapter or not ctx._run_still_current(): return @@ -5898,10 +5453,9 @@ class TurnRunner: request_overrides.update(turn_request_overrides) agent.request_overrides = request_overrides agent._gateway_turn_request_overrides = turn_request_overrides - # Must-deliver notes for THIS turn ride the current user message (api_content sidecar), - # never the system prompt: staged by _handle_message_with_agent (auto-reset note, first- - # contact intro, voice-channel change). Assigned unconditionally so a reused cached agent - # never replays a stale note. + # Must-deliver notes for THIS turn ride the current user message (api_content sidecar), never + # the system prompt: staged by _handle_message_with_agent (auto-reset, first-contact intro, + # voice-channel change). Assigned unconditionally so a reused agent never replays a stale note. agent._gateway_turn_context_notes = "\n\n".join( self._runner._consume_pending_turn_sidecar_notes(ctx.session_key) ) @@ -5967,12 +5521,10 @@ class TurnRunner: agent.memory_notifications = str(_mem_notif).lower() if _mem_notif else "on" # ------------------------------------------------------------------ - # Shared native-stream boundary close. For platforms with native streaming (e.g. WeCom - # msgtype:"stream"), an interaction that interrupts the stream — a dangerous-command - # approval prompt OR a clarify decision prompt — must finalize the current stream and - # disable native streaming first; otherwise post-interaction output keeps updating the OLD - # bubble above the prompt instead of starting a fresh one below it. Runs on the agent - # thread; the consumer processes the boundary serially via its queue. + # Shared native-stream boundary close: for native-streaming platforms (e.g. WeCom), an + # interrupting interaction (approval or clarify prompt) must finalize the current stream + # and disable native streaming first, or post-interaction output keeps updating the OLD + # bubble above the prompt. Runs on the agent thread; the consumer serializes via its queue. def _close_native_stream_boundary( _reason: str, _placeholder: str | None = None, _reopen: bool = False, ) -> bool: @@ -6007,15 +5559,10 @@ class TurnRunner: return False # ------------------------------------------------------------------ - # Clarify callback: present a clarify prompt and block on a response. - # - # Runs on the agent's worker thread (see clarify_tool's synchronous - # callback contract). Bridges sync→async by scheduling the - # adapter's send_clarify on the gateway event loop, then blocks on - # the clarify primitive's threading.Event with a configurable - # timeout. Returns the user's response string, or a sentinel - # explaining that no response arrived (so the agent can adapt - # rather than hang forever). + # Clarify callback: present a clarify prompt and block on a response. Runs on the agent's + # worker thread (clarify_tool's synchronous contract): schedules the adapter's send_clarify + # on the gateway loop, then blocks on the primitive's threading.Event with a timeout. + # Returns the response string, or a sentinel explaining no response arrived. # ------------------------------------------------------------------ def _clarify_callback_sync(question: str, choices, multi_select: bool = False) -> str: from tools import clarify_gateway as _clarify_mod @@ -6033,28 +5580,23 @@ class TurnRunner: multi_select=bool(multi_select), ) - # For WeCom native streaming: finalize the current stream before showing the clarify - # prompt so the post-answer output opens a fresh bubble below the question instead of - # updating the bubble that preceded it — the "气泡割裂" symptom. Unlike approval, clarify - # passes reopen=True so the continuation re-opens a native stream rather than degrading - # to one-shot send(); if the re-seed fails the consumer degrades to send() automatically. + # WeCom native streaming: finalize the current stream before the clarify prompt so the + # post-answer output opens a fresh bubble below the question ("气泡割裂" otherwise). Unlike + # approval, clarify passes reopen=True so the continuation re-opens a native stream; if + # the re-seed fails the consumer degrades to send() automatically. _close_native_stream_boundary( "Clarify", "💬 等待你的选择...", _reopen=True, ) - # Pause typing — like approval, we don't want a "thinking..." status to obscure the - # prompt or block the user from typing an "Other" response on platforms that disable - # input while typing is active (Slack Assistant API). - try: + # Pause typing — as with approval, a "thinking..." status must not obscure the prompt or + # block an "Other" reply on platforms that disable input while typing (Slack Assistant). + with suppress(Exception): ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) - except Exception: - pass - # Ordering barrier (#clarify-ordering): flush any buffered assistant prose (interim - # commentary / streamed deltas) to the platform BEFORE sending the poll. The poll goes - # out on a separate agent-thread-blocking path and would otherwise render ABOVE its own - # explanation. Best-effort + short timeout: never hang the agent thread if the consumer - # task isn't running. + # Ordering barrier: flush buffered assistant prose to the platform BEFORE sending the + # poll, which goes out on a separate agent-thread-blocking path and would otherwise + # render ABOVE its own explanation. Best-effort + short timeout so the agent thread + # never hangs if the consumer task isn't running. try: _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None _flush = getattr(_sc, "flush_pending_sync", None) @@ -6088,17 +5630,15 @@ class TurnRunner: session_key=ctx.session_key or "", clarify_mod=_clarify_mod, ) - # Only re-arm typing when the user actually answered — the - # undeliverable sentinel and the timeout/cancellation strings - # start with '[' and must pass through untouched. + # Only re-arm typing when the user actually answered — the undeliverable sentinel and the + # timeout/cancellation strings start with '[' and must pass through untouched. if not ( isinstance(_clarify_response, str) and _clarify_response.startswith("[") ): - # User answered. Reopen the typing indicator IMMEDIATELY — don't wait for the LLM's - # first post-answer token (native streaming otherwise re-seeds lazily on the first - # delta: ~48s of dead air). request_reopen_seed is a no-op outside the reopen-pending - # native state, so it is safe to call unconditionally. + # User answered: reopen typing IMMEDIATELY, not on the LLM's first post-answer token + # (native streaming otherwise re-seeds lazily on the first delta: ~48s of dead air). + # request_reopen_seed is a no-op outside the reopen-pending native state; always safe. _sc_reopen = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None if _sc_reopen is not None: try: @@ -6119,49 +5659,37 @@ class TurnRunner: agent.clarify_callback = _clarify_callback_sync - # Show assistant thinking between tool calls — independent of - # tool_progress mode. Mattermost needs an explicit per-platform - # opt-in so global scratch-text display does not leak into threads. + # Show assistant thinking between tool calls — independent of tool_progress mode. Mattermost + # needs an explicit per-platform opt-in so global scratch-text doesn't leak into threads. agent.thinking_progress = ctx._thinking_enabled # Store agent reference for interrupt support ctx.agent_holder[0] = agent - # Wire the platform thread-rename lane onto the agent, because the - # session titler now fires from the turn prologue rather than after - # the response. Titles are pushed here the moment they land. + # Wire the platform thread-rename lane onto the agent: the titler fires from the turn prologue, + # not after the response, so titles are pushed the moment they land. self._attach_session_title_callback(agent, ctx) - # Publish turn ownership for explicit /stop, /new, disconnect, and - # shutdown interrupts. Older session processes are outside this - # baseline and remain alive. + # Publish turn ownership for explicit /stop, /new, disconnect, and shutdown interrupts. + # Older session processes are outside this baseline and remain alive. agent._gateway_turn_process_task_id = ctx.process_task_id agent._gateway_turn_process_baseline = ctx.process_baseline # Capture the full tool definitions for transcript logging ctx.tools_holder[0] = agent.tools if hasattr(agent, 'tools') else None - # Convert history to agent format. - # Two cases: - # 1. Normal path (from transcript): simple {role, content, timestamp} dicts - # - Strip timestamps, keep role+content - # 2. Interrupt path (from agent result["messages"]): full agent messages - # that may include tool_calls, tool_call_id, reasoning, etc. - # - These must be passed through intact so the API sees valid - # assistant→tool sequences (dropping tool_calls causes 500 errors) - # - # Telegram observed group context is handled structurally here: - # observed=True transcript rows are withheld from replayable - # history and attached to the current addressed message as - # API-only context, so persisted history stores only the real - # addressed user turn. + # Convert history to agent format. Transcript path: {role, content, timestamp} dicts — strip + # timestamps. Interrupt path (agent result["messages"]): full agent messages with + # tool_calls/tool_call_id/reasoning — pass through intact so the API sees valid assistant→tool + # sequences (dropping tool_calls causes 500s). Telegram observed group context: observed=True + # rows are withheld from replayable history and attached to the current addressed message as + # API-only context, so persisted history stores only the real addressed user turn. agent_history, observed_group_context = _build_gateway_agent_history( ctx.history, channel_prompt=ctx.channel_prompt, inject_timestamps=_message_timestamps_enabled(ctx.user_config), ) - # FTS write-corruption guard: when message persistence fails silently through corrupt FTS - # triggers, the reloaded transcript above is stale/empty even though the SAME cached agent - # still holds the full live conversation in `_session_messages`. Replacing it with the - # shorter copy causes immediate same-session amnesia. Only for a reused agent bound to - # this exact session_id. + # FTS write-corruption guard: if persistence failed silently via corrupt FTS triggers, the + # reloaded transcript is stale/empty while the SAME cached agent still holds the full live + # conversation in `_session_messages`; replacing it causes same-session amnesia. Only for + # a reused agent bound to this exact session_id. if reused_cached_agent and getattr(agent, "session_id", None) == ctx.session_id: _selected = _select_cached_agent_history( agent_history, getattr(agent, "_session_messages", None) @@ -6173,21 +5701,18 @@ class TurnRunner: "conversation context (possible FTS write corruption)", ctx.session_key, len(agent_history), len(_selected), ) - # The live in-memory history bypassed the _build_gateway_agent_history cleanup - # pipeline above — re-apply the stale-confirmation expiry so a dangerous - # confirmation can't slip through this path either. + # The live in-memory history bypassed the _build_gateway_agent_history cleanup above — + # re-apply the stale-confirmation expiry so a dangerous confirmation can't slip through. agent_history = strip_stale_dangerous_confirmations( _selected, now=time.time() ) - # Collect MEDIA paths already in history so we can exclude them - # from the current turn's extraction. This is compression-safe: - # even if the message list shrinks, we know which paths are old. + # Collect MEDIA paths already in history to exclude them from this turn's extraction. + # Compression-safe: even if the message list shrinks, we know which paths are old. _history_media_paths: set = _collect_history_media_paths(agent_history) - # Register per-session gateway approval callback so dangerous command approval blocks the - # agent thread (mirrors CLI input()). The callback bridges sync→async to send the approval - # request to the user immediately. + # Per-session gateway approval callback: dangerous-command approval blocks the agent thread + # (mirrors CLI input()); the callback bridges sync→async to send the request immediately. from tools.approval import ( register_gateway_notify, reset_current_session_key, @@ -6198,34 +5723,28 @@ class TurnRunner: def _approval_notify_sync(approval_data: dict) -> None: """Send the approval request to the user from the agent thread. - If the adapter supports interactive button-based approvals (e.g. Discord's - ``send_exec_approval``), use that for a richer UX. Otherwise fall back to a plain - text message with ``/approve`` instructions. + Uses the adapter's interactive button approvals (e.g. ``send_exec_approval``) when + available, else a plain text message with ``/approve`` instructions. """ - # Pause the typing indicator while the agent waits for user approval. Critical for - # Slack's Assistant API where assistant_threads_setStatus disables the compose box — the - # user literally cannot type /approve while "is thinking..." is active. The approval send - # auto-clears the Slack status; pausing stops _keep_typing re-setting it. Typing resumes - # in _handle_approve_command/_handle_deny_command. + # Pause typing while awaiting approval: Slack's assistant_threads_setStatus disables the + # compose box, so the user can't type /approve while "is thinking..." shows. The approval + # send auto-clears it; pausing stops _keep_typing re-setting it. Resumed in approve/deny. ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) - # For WeCom native streaming: signal the stream consumer to close the current stream - # before showing the approval prompt. This goes through the consumer's queue for serial - # processing, avoiding race conditions with pending deltas. + # WeCom native streaming: ask the stream consumer to close the current stream before the + # approval prompt — via the consumer's queue, so it serializes with pending deltas. _close_native_stream_boundary("Approval") cmd = approval_data.get("command", "") desc = approval_data.get("description", "dangerous command") - # Redact credentials from the command before displaying it in the approval prompt — - # Tirith's findings are already redacted, but the raw command string still leaks secrets - # to the chat platform. Applied here so BOTH the button-based and plain-text fallback - # paths below use the redacted value. + # Redact credentials from the command before display — Tirith's findings are already + # redacted, but the raw command string still leaks secrets to the chat platform. Done + # here so BOTH the button-based and plain-text fallback paths use the redacted value. cmd = _redact_approval_command(cmd) - # Prefer button-based approval when the adapter supports it. - # Check the *class* for the method, not the instance — avoids - # false positives from MagicMock auto-attribute creation in tests. + # Prefer button-based approval when the adapter supports it. Check the *class*, not the + # instance — avoids false positives from MagicMock auto-attribute creation in tests. if getattr(type(ctx._status_adapter), "send_exec_approval", None) is not None: try: _approval_fut = safe_schedule_threadsafe( @@ -6249,10 +5768,9 @@ class TurnRunner: if _outcome == "sent": return if _outcome == "ambiguous": - # Timeout ≠ failure: the card may have posted with a late ack (slow platform - # API call or transient connector backpressure). The prompt registration stays - # alive so a tap on the rendered card still resolves; re-sending produced - # duplicate cards + an orphaned "/approve: nothing pending". Skip the fallback. + # Timeout ≠ failure: the card may have posted with a late ack (slow API or + # backpressure). The prompt registration stays alive so a tap still resolves; + # re-sending made duplicate cards + orphaned "/approve: nothing pending". Skip. logger.warning( "Button-based approval send timed out — treating " "as possibly-delivered (no re-send; the prompt " @@ -6267,9 +5785,8 @@ class TurnRunner: "Button-based approval failed, falling back to text: %s", _e ) - # Fallback: plain text approval prompt. Use the adapter's typed prefix so Slack/Matrix - # users are told the form they can actually type (`!approve`) — typed "/" is blocked in - # Slack threads and reserved by Matrix clients. + # Fallback: plain-text approval prompt with the adapter's typed prefix (e.g. `!approve`) — + # typed "/" is blocked in Slack threads and reserved by Matrix clients. _p = getattr(ctx._status_adapter, "typed_command_prefix", "/") msg = _format_exec_approval_fallback( cmd, @@ -6299,9 +5816,8 @@ class TurnRunner: except Exception as _e: logger.error("Failed to send approval request: %s", _e) - # Keep real user text separate from API-only recovery guidance. If - # an auto-continue note is prepended below, persist the original - # message so stale guidance never replays as user-authored text. + # Keep real user text separate from API-only recovery guidance: if an auto-continue note is + # prepended below, persist the original so stale guidance never replays as user text. _persist_user_message_override: Optional[Any] = ctx.persist_user_message _persist_user_timestamp_override: Optional[float] = ctx.persist_user_timestamp @@ -6311,14 +5827,11 @@ class TurnRunner: if _msn: ctx.message = _msn + "\n\n" + ctx.message - # Auto-continue: if the loaded history ends with a tool result, the previous agent turn was - # interrupted mid-work (gateway restart, crash, SIGTERM). Prepend a system note so the model - # finishes processing the pending tool results before addressing the user's new message. - # Session-level resume_pending (drain-timeout shutdown) escalates the wording: the last role - # may be anything, so a stronger reason-aware instruction subsumes the tool-tail case. - # Freshness gate: both branches are gated on the age of the last persisted transcript row. - # Read ``history[-1]`` (not agent_history, which stripped ``timestamp`` off tool rows for - # API purity). Rows without a timestamp (legacy) are treated as fresh. + # Auto-continue: history ending with a tool result means the previous turn was cut off + # (restart, crash, SIGTERM) — prepend a system note so the model finishes the pending tool + # results first. Session-level resume_pending (drain-timeout shutdown) uses stronger + # reason-aware wording that subsumes this case. Both gate on the age of ``history[-1]`` (not + # agent_history, which stripped ``timestamp`` off tool rows); rows without one are fresh. _freshness_window = _auto_continue_freshness_window() _interruption_is_fresh = _is_fresh_gateway_interruption( _last_transcript_timestamp(ctx.history), @@ -6332,19 +5845,10 @@ class TurnRunner: except Exception: _resume_entry = None - # resume_pending freshness uses a SECOND signal in addition to the - # transcript clock above. The restart watchdog stamps the session - # with ``last_resume_marked_at`` at interrupt time — that is the - # correct "when were we interrupted" signal. The transcript clock - # (_interruption_is_fresh) can be far older: an active thread you - # return to may have its last persisted row hours back, even though - # the interruption itself just happened. Gating resume_pending on - # the transcript clock alone makes the recovery note silently drop, - # and because the startup auto-resume turn carries empty text - # (_schedule_resume_pending_sessions), the model then receives a - # blank user message and replies with confused "the message came - # through blank" noise. Treat the marker as fresh when - # EITHER signal is fresh so the two freshness checks agree. + # resume_pending freshness also uses the restart watchdog's ``last_resume_marked_at`` (the + # true interruption stamp): the transcript clock (_interruption_is_fresh) can be hours older + # for an active thread, so gating on it alone drops the recovery note — and the startup + # auto-resume turn has empty text, so the model gets a blank user message. Fresh if EITHER is. _resume_mark_is_fresh = False if _resume_entry is not None and getattr(_resume_entry, "resume_pending", False): _resume_mark_is_fresh = _is_fresh_gateway_interruption( @@ -6364,11 +5868,10 @@ class TurnRunner: if _is_resume_pending: _reason = getattr(_resume_entry, "resume_reason", None) or "restart_timeout" - # The empty-message case is the auto-resume startup turn synthesized by - # _schedule_resume_pending_sessions — there is no NEW user message to address. Guidance - # is adapter-aware: interactive platforms report the restore and ask what next; event - # platforms (webhook, API server) continue the work — nobody is present to answer, and - # an acknowledgement would silently abandon the task. + # Empty message = the startup auto-resume turn from _schedule_resume_pending_sessions; + # there is no NEW user message. Interactive platforms report the restore and ask what + # next; event platforms (webhook, API server) continue the work — nobody is present to + # answer, and an acknowledgement would silently abandon the task. _resume_adapter = self._runner._adapter_for_source(ctx.source) _interactive_resume = bool( getattr(_resume_adapter, "interactive_resume", True) @@ -6386,21 +5889,18 @@ class TurnRunner: + ctx.message ) - # Consume one-shot /reload-skills note (if the user ran /reload-skills since their last turn - # in this session). Same queue pattern as CLI: prepend to the NEXT user message, then clear. - # Nothing was written to the transcript out-of-band, so message alternation stays intact. + # Consume one-shot /reload-skills note (same queue pattern as CLI): prepend to the NEXT user + # message, then clear. Nothing hit the transcript out-of-band, so alternation stays intact. _pending_notes = getattr(self._runner, "_pending_skills_reload_notes", None) if _pending_notes and ctx.session_key and ctx.session_key in _pending_notes: _srn = _pending_notes.pop(ctx.session_key, None) if _srn: ctx.message = _srn + "\n\n" + ctx.message - # Safety net: a startup auto-resume event carries empty text and relies on the - # resume_pending branch above to supply the recovery note. If it did not fire (freshness - # signals disagreed, marker cleared between scheduling and dispatch) we must NOT hand the - # model a blank user turn — it replies with confused "message came through blank" noise. - # Restricted to resume_pending sessions so legitimately empty turns (image, no caption, - # wrapped as native content below) are untouched. + # Safety net: a startup auto-resume event carries empty text and relies on the resume_pending + # branch above for the recovery note. If it did not fire (freshness signals disagreed, marker + # cleared before dispatch) we must NOT hand the model a blank user turn. Restricted to + # resume_pending sessions so legitimately empty turns (caption-less image) are untouched. if ( isinstance(ctx.message, str) and not ctx.message.strip() @@ -6477,28 +5977,24 @@ class TurnRunner: if _persist_user_timestamp_override is not None: _conversation_kwargs["persist_user_timestamp"] = _persist_user_timestamp_override # Thread the platform-side inbound message id onto the persisted user turn so a turn - # interrupted by a gateway restart is durably recorded WITH its id — restart drain- - # window recovery dedups against has_platform_message_id, and without this the - # interrupted turn is invisible to that check. Uses the raw inbound id (NOT - # event_message_id, which is the reply anchor). + # interrupted by a restart is recorded WITH its id — drain-window recovery dedups on + # has_platform_message_id. Uses the raw inbound id, NOT event_message_id (reply anchor). if ctx.inbound_message_id is not None: _conversation_kwargs["persist_user_platform_id"] = str(ctx.inbound_message_id) result = agent.run_conversation(_api_run_message, **_conversation_kwargs) finally: unregister_gateway_notify(_approval_session_key) - # Cancel any pending clarify entries so blocked agent - # threads don't hang past the end of the run (interrupt, - # completion, gateway shutdown). Idempotent. + # Cancel any pending clarify entries so blocked agent threads don't hang past the end of + # the run (interrupt, completion, gateway shutdown). Idempotent. try: from tools.clarify_gateway import clear_session as _clear_clarify_session _clear_clarify_session(_approval_session_key) except Exception: pass reset_current_session_key(_approval_session_token) - # Canonicalize an explicitly emitted computer-use screenshot path at the common result - # boundary: the streaming finalizer below and the non-streaming delivery path must see the - # same response; repairing only during later media scanning leaves streaming with the - # model-mangled path and a rejected attachment. + # Canonicalize a model-emitted computer-use screenshot path at the common result boundary: the + # streaming finalizer below and the non-streaming delivery path must see the same response; + # repairing only in later media scanning leaves streaming a mangled path + rejected attachment. if isinstance(result, dict): _result_final = result.get("final_response") if isinstance(_result_final, str): @@ -6510,21 +6006,16 @@ class TurnRunner: ctx.result_holder[0] = result - # Signal the stream consumer that the agent is done. Pass the completed final_response as - # the authoritative finalize payload: it includes post-stream augmentation (verifier footer, - # turn-completion explainer) the consumer's accumulator never saw, so the seal delivers the - # TRUE final and no separate corrective send fires. Failed turns pass nothing — error text - # is delivered by the gateway's normal path, not baked into the stream. + # Signal the stream consumer that the agent is done, passing final_response as the + # authoritative finalize payload: it includes post-stream augmentation (verifier footer, + # explainer) the accumulator never saw, so the seal delivers the TRUE final with no + # corrective send. Failed turns pass nothing — error text goes via the normal path. if _stream_consumer is not None: _final_for_stream = None - # Adopt ONLY a genuinely completed final (review B6): interrupt - # paths return {interrupted: True, completed: False} with a - # DIAGNOSTIC final_response ("Operation interrupted during …") - # and no failed key — adopting that would seal the user's - # streamed partial answer over with the diagnostic AND make - # delivered_final_matches reconcile, suppressing the gateway's - # own error-delivery path. Writers of these shapes: - # agent/conversation_loop.py interrupt/retry-abort returns. + # Adopt ONLY a genuinely completed final: interrupt paths return {interrupted: True, + # completed: False} with a DIAGNOSTIC final_response and no failed key — adopting it + # would seal the streamed partial answer over with the diagnostic AND make + # delivered_final_matches reconcile, suppressing the gateway's own error delivery. if ( isinstance(result, dict) and not result.get("failed") @@ -6535,9 +6026,8 @@ class TurnRunner: if isinstance(_fr, str) and _fr.strip() and _fr != "(empty)": _final_for_stream = _fr if _final_for_stream is not None: - # Duck-type safe: test doubles / older consumers may expose a - # zero-arg finish(). The payload is an optimization, not a - # requirement — fall back to the bare signal. + # Duck-type safe: test doubles / older consumers may expose a zero-arg finish(). The + # payload is an optimization, not a requirement — fall back to the bare signal. try: _stream_consumer.finish(_final_for_stream) except TypeError: @@ -6545,9 +6035,8 @@ class TurnRunner: else: _stream_consumer.finish() - # Signal the streaming-TTS consumer that the agent is done. finish() is called from the - # outer event-loop thread after the executor returns, so early returns from run_sync are - # also finalised. See the outer finally/completion section below. + # Signal the streaming-TTS consumer that the agent is done. finish() runs on the outer + # event-loop thread after the executor returns, so early run_sync returns are also finalised. # Return final response, or a message if something went wrong final_response = result.get("final_response") @@ -6565,17 +6054,13 @@ class TurnRunner: _context_length = getattr(_agent.context_compressor, "context_length", 0) or 0 _resolved_model = getattr(_agent, "model", None) if _agent else None - # Sync session_id immediately after run_conversation(). Compression - # can rotate before a follow-up model call fails; the failure return - # below must still point the gateway at the compressed child. + # Sync session_id right after run_conversation(): compression can rotate before a follow-up + # model call fails, and the failure return below must still point at the compressed child. agent = ctx.agent_holder[0] _session_was_split = False - # In-place compaction (compression.in_place / #38763) compacts the - # transcript WITHOUT rotating the id, so the id-change diff below - # can't detect it. compress_context() sets this rotation-independent - # flag on the agent; the gateway uses it to re-baseline transcript - # handling (history_offset=0 + rewrite the JSONL transcript) the - # same way a split would, even though the session_id is unchanged. + # In-place compaction (compression.in_place) compacts the transcript WITHOUT rotating the id, + # so the id-change diff below can't see it. compress_context() sets this flag on the agent; the + # gateway re-baselines (history_offset=0 + JSONL rewrite) as for a split despite unchanged id. _compacted_in_place = bool(getattr(agent, "_last_compaction_in_place", False)) if agent else False agent_session_id = getattr(agent, 'session_id', ctx.session_id) if agent else ctx.session_id if agent and ctx.session_key and agent_session_id != ctx.session_id: @@ -6616,12 +6101,10 @@ class TurnRunner: ) _session_split_entry_persisted = True - # If this is a Telegram DM and source.thread_id was lost during the session split - # (synthetic / recovered event), restore it from the binding so - # _thread_metadata_for_source produces the correct message_thread_id instead of routing - # to the General thread. Non-fatal (worst case: lands in General). Only after this run - # published its session split — a stale /stop→/new predecessor must not mutate - # routing/binding state for the fresh session. + # Telegram DM whose source.thread_id was lost in the session split (synthetic/recovered + # event): restore it from the binding so _thread_metadata_for_source yields the right + # message_thread_id instead of the General thread (non-fatal). Only after this run + # published its split — a stale /stop→/new predecessor must not mutate routing state. if _session_split_entry_persisted and ( getattr(ctx.source, "platform", None) == Platform.TELEGRAM and getattr(ctx.source, "chat_type", None) == "dm" @@ -6653,10 +6136,9 @@ class TurnRunner: effective_session_id = agent_session_id self._runner._sync_session_model_from_agent(effective_session_id, agent) - # history_offset=0 whenever the agent's message list no longer has the original history - # prefix — i.e. on rotation (split) OR in-place compaction. In both cases the returned - # `messages` is the compacted set, so persist all of it (offset 0) rather than slicing - # past the pre-compaction length (which would drop everything). + # history_offset=0 whenever the agent's message list lost the original history prefix: rotation + # (split) OR in-place compaction. Either way the returned `messages` is the compacted set, so + # persist all of it; slicing past the pre-compaction length would drop everything. _effective_history_offset = ( 0 if (_session_was_split or _compacted_in_place) else len(agent_history) ) @@ -6673,12 +6155,9 @@ class TurnRunner: "messages": result.get("messages", []), "api_calls": result.get("api_calls", 0), "failed": result.get("failed", False), - # Sibling of the non-empty-response return below (#64686): - # the classifier's failure_reason must survive the - # empty-response normalization path too, or downstream - # consumers (TUI billing surface, transient-failure - # persistence) lose the structured reason exactly when - # the run produced no text. + # Sibling of the non-empty-response return below: the classifier's failure_reason + # must survive the empty-response path too, or downstream consumers (TUI billing, + # transient-failure persistence) lose the structured reason when no text was produced. "failure_reason": result.get("failure_reason"), "partial": result.get("partial", False), "completed": result.get("completed"), @@ -6698,14 +6177,11 @@ class TurnRunner: "context_length": _context_length, } - # Scan tool results for MEDIA: tags that need to be delivered as native audio/file - # attachments. The TTS tool embeds MEDIA: tags in its JSON response, but the model's final - # text reply usually doesn't include them; append missing ones so extract_media() delivers - # each file exactly once. Scope to THIS turn only: returned ``messages`` is ``agent_history`` - # + this turn, so slicing at ``len(agent_history)`` keeps a stale MEDIA: path from an - # earlier turn off a later text-only reply. Path dedup against _history_media_paths is the - # secondary guard — and the sole guard on the fallback branch when mid-run compression - # shrank the message list below the original history length. + # Append MEDIA: tags from tool results (e.g. TTS) that the model's final text omits, so + # extract_media() delivers each file once. Scope to THIS turn (slice at ``len(agent_history)``) + # so a stale MEDIA: path from an earlier turn doesn't ride a later text-only reply; dedup + # against _history_media_paths is the secondary guard — and the sole one on the fallback + # branch when mid-run compression shrank the list below the history length. if "MEDIA:" not in final_response: media_tags, has_voice_directive = _collect_auto_append_media_tags( result.get("messages", []), @@ -6724,10 +6200,9 @@ class TurnRunner: unique_tags.insert(0, "[[audio_as_voice]]") final_response = final_response + "\n" + "\n".join(unique_tags) - # Auto-titling runs at TURN START (agent/turn_context.py) from the user's message alone, so - # it no longer waits on final_response — a failed or interrupted turn still gets a titled - # session. Thread-rename callbacks are attached as `_on_session_title` before the run - # (_attach_session_title_callback) because the titler fires from the turn prologue. + # Auto-titling runs at TURN START (agent/turn_context.py) from the user's message alone, so a + # failed/interrupted turn is still titled. Thread-rename callbacks are attached as + # `_on_session_title` before the run because the titler fires from the turn prologue. return { "final_response": final_response, @@ -6747,9 +6222,8 @@ class TurnRunner: ctx.result_holder[0].get("compression_exhausted", False) if ctx.result_holder[0] else False ), - # Soft lock-contention defer (#69870 consumer): distinct from - # compression_exhausted so the gateway never auto-resets a - # session that a concurrent compressor is about to shrink. + # Soft lock-contention defer: distinct from compression_exhausted so the gateway never + # auto-resets a session that a concurrent compressor is about to shrink. "compression_deferred": ( ctx.result_holder[0].get("compression_deferred", False) if ctx.result_holder[0] else False @@ -6765,26 +6239,122 @@ class TurnRunner: "session_id": effective_session_id, "response_previewed": result.get("response_previewed", False), "response_transformed": result.get("response_transformed", False), - # Pass through the agent_persisted flag so the persistence block - # above can correctly determine whether the codex app-server path - # self-persisted (it didn't — see codex_runtime.py). Default - # True preserves the skip-db behaviour for the standard runtime. + # Pass through agent_persisted so the persistence block above can tell whether the codex + # app-server path self-persisted (it didn't — see codex_runtime.py); default True keeps the + # skip-db behaviour for the standard runtime. "agent_persisted": (ctx.result_holder[0].get("agent_persisted", True) if ctx.result_holder[0] else True), } -# Sentinel for "no explicit session DB has been pinned on this runner", so the ``_session_db`` -# property can distinguish "resolve from the active profile scope" from a deliberate -# ``runner._session_db = None`` (which disables DB-backed commands and is how many suites construct -# a bare runner). A plain ``None`` cannot express both. Mirrors ``gateway.session._DB_UNPINNED``. +# Sentinel for "no explicit session DB pinned on this runner", so ``_session_db`` can distinguish +# "resolve from the active profile scope" from a deliberate ``runner._session_db = None`` (disables +# DB-backed commands, as many test suites do). Mirrors ``gateway.session._DB_UNPINNED``. _SESSION_DB_UNPINNED = object() -class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, GatewaySlashCommandsMixin): - """Main gateway controller. +# Agent-facing sidecar note per auto-reset reason (default: idle). +_AUTO_RESET_CONTEXT_NOTES = { + "suspended": "[System note: The user's previous session was stopped and suspended. This is a fresh conversation with no prior context.]", + "daily": "[System note: The user's session was automatically reset by the daily schedule. This is a fresh conversation with no prior context.]", + "resume_pending_expired": "[System note: The previous gateway session could not be recovered after a restart (API recovery timed out). This is a fresh conversation — use /resume to restore history if needed.]", + "idle": "[System note: The user's previous session expired due to inactivity. This is a fresh conversation with no prior context.]", +} - Manages the lifecycle of all platform adapters and routes messages to/from the agent. - """ + +def _auto_reset_reason_text(reset_reason: str, policy) -> str: + """Human-readable cause for the user-facing auto-reset notice.""" + if reset_reason == "suspended": + return "previous session was stopped or interrupted" + if reset_reason == "resume_pending_expired": + return "gateway restart recovery timed out" + if reset_reason == "daily": + return f"daily schedule at {policy.at_hour}:00" + hours = policy.idle_minutes // 60 + mins = policy.idle_minutes % 60 + duration = f"{hours}h" if not mins else f"{hours}h {mins}m" if hours else f"{mins}m" + return f"inactive for {duration}" + + +def _write_runtime_status_quiet(**fields: Any) -> None: + """Best-effort ``gateway_state.json`` write; status persistence must never abort the caller.""" + try: + from gateway.status import write_runtime_status + + write_runtime_status(**fields) + except Exception: + pass + + +def _command_origin_for_source(source: Any) -> Optional[dict]: + """Delivery origin for a shared CLI/gateway command so its job replies to this chat/thread.""" + try: + platform = getattr(source.platform, "value", None) or str(getattr(source, "platform", "") or "") + chat_id = getattr(source, "chat_id", None) + if platform and chat_id: + return { + "platform": platform, + "chat_id": str(chat_id), + "chat_name": getattr(source, "chat_name", None), + "thread_id": getattr(source, "thread_id", None), + } + except Exception: + pass + return None + + +def _builtin_adapter_import(module: str, adapter_name: str, requirement: str): + """Lazy-import ``(adapter_cls, requirements_ok)`` from ``gateway.platforms.``.""" + import importlib + + mod = importlib.import_module(f"gateway.platforms.{module}") + return getattr(mod, adapter_name), getattr(mod, requirement) + + +# Built-in (non-plugin) adapters: platform -> (module, adapter class, requirements probe +# name, warning when the probe fails). Signal additionally validates its config below. +_BUILTIN_ADAPTERS: dict[Platform, tuple[str, str, str, str]] = { + Platform.WHATSAPP_CLOUD: ("whatsapp_cloud", "WhatsAppCloudAdapter", "check_whatsapp_cloud_requirements", + "WhatsApp Cloud: aiohttp/httpx missing — reinstall hermes-agent"), + Platform.SIGNAL: ("signal", "SignalAdapter", "check_signal_requirements", + "Signal: runtime requirements not met"), + Platform.WEIXIN: ("weixin", "WeixinAdapter", "check_weixin_requirements", + "Weixin: aiohttp/cryptography not installed"), + Platform.API_SERVER: ("api_server", "APIServerAdapter", "check_api_server_requirements", + "API Server: aiohttp not installed"), + Platform.WEBHOOK: ("webhook", "WebhookAdapter", "check_webhook_requirements", + "Webhook: aiohttp not installed"), + Platform.MSGRAPH_WEBHOOK: ("msgraph_webhook", "MSGraphWebhookAdapter", "check_msgraph_webhook_requirements", + "MSGraph webhook: aiohttp not installed"), + Platform.BLUEBUBBLES: ("bluebubbles", "BlueBubblesAdapter", "check_bluebubbles_requirements", + "BlueBubbles: aiohttp/httpx missing or BLUEBUBBLES_SERVER_URL/BLUEBUBBLES_PASSWORD not configured"), + Platform.QQBOT: ("qqbot", "QQAdapter", "check_qq_requirements", + "QQBot: aiohttp/httpx missing or QQ_APP_ID/QQ_CLIENT_SECRET not configured"), + Platform.YUANBAO: ("yuanbao", "YuanbaoAdapter", "WEBSOCKETS_AVAILABLE", + "Yuanbao: websockets not installed. Run: pip install websockets"), +} + + +def _instantiate_builtin_adapter(platform: Platform, config: Any) -> Optional[BasePlatformAdapter]: + """Instantiate a core (non-plugin) adapter, or None when its requirements are unmet/unknown.""" + spec = _BUILTIN_ADAPTERS.get(platform) + if spec is None: + return None + module, adapter_name, requirement, warning = spec + adapter_cls, requirements_ok = _builtin_adapter_import(module, adapter_name, requirement) + if not (requirements_ok() if callable(requirements_ok) else requirements_ok): + logger.warning(warning) + return None + if platform == Platform.SIGNAL: + from gateway.platforms.signal import validate_signal_config + + if not validate_signal_config(config): + logger.warning("Signal: SIGNAL_HTTP_URL or SIGNAL_ACCOUNT not configured") + return None + return adapter_cls(config) + + +class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, GatewaySlashCommandsMixin): + """Main gateway controller: manages adapter lifecycles, routes messages to/from the agent.""" # Class-level defaults so partial construction in tests doesn't # blow up on attribute access. @@ -6813,12 +6383,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _startup_warmup_task: Optional[asyncio.Task] = None # ------------------------------------------------------------------ - # Legacy per-session dict adapters. All per-session state lives in - # ``self._sessions`` (Dict[str, SessionState]); these properties expose - # the pre-consolidation dict attributes as LIVE MutableMapping views so - # the extensive test surface (and a few mixin/adapter call sites) that - # read/write ``runner._running_agents`` etc. keeps working unchanged. - # New production code should use ``self._session_state(key)`` directly. + # Legacy per-session dict adapters: all per-session state lives in ``self._sessions`` + # (Dict[str, SessionState]); these properties expose the old dict attrs as LIVE MutableMapping + # views so ``runner._running_agents`` etc. keep working. New code: ``self._session_state(key)``. # ------------------------------------------------------------------ _running_agents = legacy_dict_property("_running_agents") _running_agents_ts = legacy_dict_property("_running_agents_ts") @@ -6886,9 +6453,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew for key, state in self._sessions_map().items() if state.turn.agent is not None ] - # Loop-liveness heartbeat / watchdog handles (#66892, #69089). Class-level - # defaults so partial construction in tests doesn't blow up on access; the - # real values are set in __init__ / start() / stop(). + # Loop-liveness heartbeat / watchdog handles. Class-level defaults so partial construction in + # tests doesn't blow up on access; real values are set in __init__ / start() / stop(). _loop_heartbeat_task: Optional["asyncio.Task"] = None _loop_floor_timer_handle: Optional[Any] = None _loop_liveness_watchdog: Optional[Any] = None @@ -6899,27 +6465,24 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def __init__(self, config: Optional[GatewayConfig] = None): global _gateway_runner_ref - # When multiplex_profiles is on, load under the default profile secret scope so bot tokens - # in that profile's .env resolve the same way secondary profiles do. Explicit config= - # injection (tests) is left untouched. + # With multiplex_profiles on, load under the default profile secret scope so bot tokens in its + # .env resolve as secondary profiles' do; explicit config= injection (tests) is left untouched. self.config = config if config is not None else load_gateway_config_for_runner() # Mark the process as a profile multiplexer when configured. This flips # agent.secret_scope.get_secret() to fail-closed on any unscoped credential read, so a - # missed migration crashes loudly instead of leaking a cross-profile value (Workstream A). + # missed migration crashes loudly instead of leaking a cross-profile value. try: from agent.secret_scope import set_multiplex_active set_multiplex_active(bool(getattr(self.config, "multiplex_profiles", False))) except Exception: logger.debug("could not set multiplex-active flag", exc_info=True) self.adapters: Dict[Platform, BasePlatformAdapter] = {} - # When non-None, SessionDB init failed — the gateway broadcasts a one-time warning to the - # home channel(s) after connecting, so the user knows persistence is broken instead of - # discovering it later via a missing /resume or empty history. + # Non-None means SessionDB init failed — the gateway broadcasts a one-time warning to the home + # channel(s) after connecting so the user learns persistence is broken before /resume fails. self._session_db_init_error: Optional[str] = None # Multi-profile multiplexing: adapters for NON-default profiles live here, keyed by profile - # name then Platform. self.adapters stays the default/active profile's map so the ~93 - # existing self.adapters[...] sites are untouched when multiplexing is off (this dict is - # empty). + # name then Platform. self.adapters stays the default/active profile's map so existing + # self.adapters[...] sites are untouched when multiplexing is off (this dict is then empty). self._profile_adapters: Dict[str, Dict[Platform, BasePlatformAdapter]] = {} self._warn_if_docker_media_delivery_is_risky() _gateway_runner_ref = _weakref.ref(self) @@ -6932,9 +6495,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._show_reasoning = self._load_show_reasoning() self._busy_input_mode = self._load_busy_input_mode() self._busy_text_mode = self._load_busy_text_mode() - # Secondary-profile busy modes are snapshotted during multiplex - # startup. Busy-message handlers consult these maps by routed source - # without rereading config or mutating process-global environment. + # Secondary-profile busy modes, snapshotted at multiplex startup; busy-message handlers consult + # them by routed source without rereading config or mutating process-global environment. self._busy_input_modes_by_profile: Dict[str, str] = {} self._busy_text_modes_by_profile: Dict[str, str] = {} self._restart_drain_timeout = self._load_restart_drain_timeout() @@ -6962,9 +6524,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew key, max_active_age=_bg_max_age_seconds, ), ) - # One enforced loop-side boundary for the synchronous SessionStore. - # Sync helpers keep using ``session_store`` directly; async gateway - # handlers call this facade and await every operation. + # One enforced loop-side boundary for the synchronous SessionStore: sync helpers keep using + # ``session_store`` directly; async gateway handlers call this facade and await every op. self._async_session_store = AsyncSessionStore(self.session_store) self.delivery_router = DeliveryRouter(self.config) self._running = False @@ -6977,32 +6538,26 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._draining = False self._profile_failed_platforms: Dict[str, Dict[Platform, asyncio.Task]] = {} self._systemd_watchdog = None - # External (NAS-driven) drain state — distinct from the shutdown ``_draining`` flag above. - # Set by ``_drain_control_watcher`` when ``.drain_request.json`` is present: gateway_state - # -> draining, NEW turns refused, but the process does NOT exit (quiesce-without-restart). - # It is fully reversible: removing the marker reverts to ``running`` and re-accepts turns. - # ``_draining`` (shutdown) is one-way and ends in process exit. + # External (NAS-driven) drain state, distinct from the shutdown ``_draining`` flag: set by + # ``_drain_control_watcher`` when ``.drain_request.json`` exists — NEW turns refused, but the + # process stays up and removing the marker reverts to ``running``. ``_draining`` is one-way. self._external_drain_active = False self._restart_requested = False - # Set by shutdown_signal_handler when a SIGTERM/SIGINT arrived WITHOUT a planned-stop / - # takeover marker — i.e. an unexpected external signal (container/s6 SIGTERM on `docker - # restart` or image upgrade, OOM-killer, bare `kill`). _stop_impl uses it to decide whether - # to persist gateway_state=stopped: an unexpected signal must NOT persist "stopped", or - # container_boot refuses to auto-start the gateway on the next boot. + # Set by shutdown_signal_handler when SIGTERM/SIGINT arrived WITHOUT a planned-stop/takeover + # marker (container SIGTERM, OOM-killer, bare `kill`); _stop_impl must NOT persist + # gateway_state=stopped for an unexpected signal, or container_boot won't auto-start next boot. self._signal_initiated_shutdown = False self._restart_task_started = False self._restart_detached = False self._restart_via_service = False self._detached_restart_helper_started = False self._restart_command_source: Optional[SessionSource] = None - # Monotonic-ish wall clock of when this GatewayRunner was constructed. - # Used by the /restart redelivery guard to bound the window in which a - # missing dedup marker is treated as a stale redelivery. + # Monotonic-ish wall clock of when this GatewayRunner was constructed. Used by the /restart + # redelivery guard to bound the window where a missing dedup marker means a stale redelivery. self._startup_time: float = time.time() - # Set True at startup when this process booted as the result of a chat-originated /restart - # (i.e. .restart_notify.json existed on boot). One-shot signal consumed by - # _is_stale_restart_redelivery so the marker-missing fallback only suppresses a /restart - # when we KNOW we just came out of a restart cycle — never on a genuine fresh boot. + # True when this process booted from a chat-originated /restart (.restart_notify.json existed + # on boot). One-shot signal consumed by _is_stale_restart_redelivery so the marker-missing + # fallback suppresses a /restart only when we KNOW we just restarted — never on a fresh boot. self._booted_from_restart: bool = False self._stop_task: Optional[asyncio.Task] = None self._restart_task: Optional[asyncio.Task] = None @@ -7019,19 +6574,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # ROUTING KEYS resolve to one session_id (switch_session's many-to-one mapping). The # routing-key guards above cannot see that overlap. self._turn_leases = SessionTurnLeaseRegistry() - # Held turn-lease tokens live on SessionState.turn.lease_token/.lease_generation; the - # generation check means a stale unwind can never free a newer turn's lease. - # SessionState.persistent.pending_command_text holds runner-level queued interrupt text - # (distinct from the adapter-level _pending_messages in gateway/platforms/base.py). - # SessionState.conversation.last_resolved_model: last non-empty model per session, used - # when a fresh config read transiently returns "" (config-cache miss) — otherwise the agent - # is built with model="" and every call fails HTTP 400; "*" holds a process-wide fallback. - # SessionState.conversation.queued_events: /queue overflow — the adapter slot is single; - # /queue needs one full turn per item in FIFO order, promoted one at a time after each - # drain. Cleared on /new and /reset (preserved across /model). The run-generation counter - # on SessionState is monotonic and NEVER reset. - # Session keys already notified for the current stall episode (cleared when pending clears - # / activity resumes / conversation boundary). See gateway.session_stall. + # Turn-lease tokens live on SessionState.turn.lease_token/.lease_generation; the generation + # check means a stale unwind can never free a newer turn's lease. pending_command_text is + # runner-level queued interrupt text (distinct from adapter-level _pending_messages). + # last_resolved_model backs a config read that transiently returns "" (else model="" → every + # call HTTP 400); "*" is the process-wide fallback. queued_events is /queue overflow, promoted + # one per drain FIFO; cleared on /new and /reset, kept across /model. The run-generation + # counter is monotonic and NEVER reset. Stall-notified keys clear when pending clears / + # activity resumes / conversation boundary (gateway.session_stall). self._session_stall_notified: Dict[str, bool] = {} # Startup restore gate: while restart-interrupted sessions are being auto-resumed, real # inbound messages are queued instead of competing with the synthetic resume turns for the @@ -7049,27 +6599,24 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._session_sources: "OrderedDict[str, SessionSource]" = OrderedDict() self._session_sources_max = 512 # Completion delivery is intentionally lifecycle-scoped: it closes duplicate queue/watcher - # races inside one gateway without pretending adapter send + persistence write are - # exactly-once across a crash. Any durable async-delegation replay state remains owned by - # tools.async_delegation, not a parallel gateway ledger. + # races inside one gateway without pretending adapter send + persistence write are exactly-once + # across a crash. Durable async-delegation replay state stays owned by tools.async_delegation. self._completion_delivery_lock = threading.Lock() self._completion_deliveries_inflight: set[tuple[str, str, object]] = set() self._completion_deliveries_delivered: "OrderedDict[tuple[str, str, object], None]" = OrderedDict() self._completion_delivery_retention = 2048 - # Agent-triggered terminal completions from one conversation often land - # in the same scheduler tick. Hold them briefly so the agent receives - # one synthetic turn instead of one turn per process (#70300). + # Agent-triggered terminal completions from one conversation often land in the same scheduler + # tick; hold them briefly so the agent gets one synthetic turn instead of one per process. self._completion_notification_batches: dict[tuple[str, ...], list[tuple[str, dict, asyncio.Future]]] = {} self._completion_notification_batch_tasks: dict[tuple[str, ...], asyncio.Task] = {} self._completion_notification_batch_flush_tasks: set[asyncio.Task] = set() self._completion_notification_batch_window = 0.1 self._completion_notification_batches_stopping = False - # Cache AIAgent instances per session to preserve prompt caching: a fresh AIAgent per message - # rebuilds the system prompt (incl. memory) every turn, breaking the prefix cache (~10x cost - # on Anthropic). Key: session_key, Value: (AIAgent, config_signature_str). OrderedDict so - # _enforce_agent_cache_cap() can evict LRU (move_to_end on hit, popitem(last=False)); - # hard cap _AGENT_CACHE_MAX_SIZE, idle TTL enforced from _session_expiry_watcher(). + # Cache AIAgent instances per session to preserve prompt caching (a fresh agent per message + # rebuilds the system prompt and breaks the prefix cache, ~10x cost on Anthropic). Value: + # (AIAgent, config_signature_str). OrderedDict for LRU eviction in _enforce_agent_cache_cap(); + # hard cap _AGENT_CACHE_MAX_SIZE, idle TTL from _session_expiry_watcher(). import threading as _threading self._agent_cache: "OrderedDict[str, tuple]" = OrderedDict() self._agent_cache_lock = _threading.Lock() @@ -7078,9 +6625,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # per-turn sidecar notes; ephemeral context pin; last-delivered voice-channel context) lives # on SessionState.conversation — see gateway/session_state.py. self._kanban_notifier_profile = self._active_profile_name() - # Launch-time identity of the profile that owns ``self.adapters``; - # ``_authorization_adapter`` compares against this rather than the - # per-turn ``_active_profile_name()`` (see gateway/authz_mixin.py). + # Launch-time identity of the profile that owns ``self.adapters``; ``_authorization_adapter`` + # compares against this rather than the per-turn ``_active_profile_name()``. self._primary_profile_name = self._kanban_notifier_profile # Teams meeting pipeline runtime (bound later when msgraph_webhook adapter exists). self._teams_pipeline_runtime = None @@ -7104,9 +6650,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew import itertools as _itertools self._slash_confirm_counter = _itertools.count(1) - # Persistent Honcho managers keyed by gateway session key. - # This preserves write_frequency="session" semantics across short-lived - # per-message AIAgent instances. + # Persistent Honcho managers keyed by gateway session key: preserves write_frequency="session" + # semantics across short-lived per-message AIAgent instances. # Ensure tirith security scanner is available (downloads if needed) try: @@ -7115,13 +6660,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass # Non-fatal — fail-open at scan time if unavailable - # Startup heads-up (#30882): a gateway in manual approval mode with no - # automated risk assessor (tirith disabled AND no auxiliary.approval - # model) can only gate dangerous commands / execute_code scripts via - # live in-chat approval. With approval routing fixed, those actions now - # fail closed (block) rather than silently auto-running — surface that - # so operators knowingly enable tirith or configure auxiliary.approval - # for unattended gateways. + # Startup heads-up: manual approval mode with no automated risk assessor (tirith disabled AND + # no auxiliary.approval model) can only gate dangerous commands via live in-chat approval, so + # they fail closed on unattended gateways — surface it so operators knowingly enable one. try: from hermes_cli.config import load_config as _load_full_config _appr_cfg = _load_full_config() @@ -7142,11 +6683,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: logger.debug("approvals.mode startup check skipped", exc_info=True) - # Initialize session database for session_search tool support. - # A handle bound here would be pinned to the process root home, but /resume, /title, - # /history and session search run inside _profile_runtime_scope on a multiplexed gateway - # and must see that profile's state.db — so resolve via a property caching one - # AsyncSessionDB per path; priming here keeps startup diagnostics at construction time. + # Session DB for session_search: a property caches one AsyncSessionDB per path (not a handle + # bound here, which would pin the root home — /resume, /title, /history and search run inside + # _profile_runtime_scope under multiplex); priming here keeps startup diagnostics at init. self._session_db_pinned: Any = _SESSION_DB_UNPINNED self._session_db_handles: Dict[Path, Any] = {} self._session_db_handles_lock = threading.Lock() @@ -7159,21 +6698,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: self._open_session_db_for_active_scope(raise_on_error=True) except Exception as e: - # WARNING (not DEBUG) so the failure appears in errors.log — matches cli.py's handling - # of the same init path. Users hitting NFS-mounted HERMES_HOME silently lost /resume, - # /title, /history, /branch, and session search without this. + # WARNING (not DEBUG) so it lands in errors.log, matching cli.py; otherwise an NFS-mounted + # HERMES_HOME silently loses /resume, /title, /history, /branch and session search. logger.warning("SQLite session store not available: %s", e) - # Surface the failure to the user via their home channel(s) once the gateway connects. - # Without this, state.db corruption or NFS/SMB lock failures silently degrade the whole - # gateway — messages flow but nothing is persisted until /resume finds nothing. + # Surface the failure on the user's home channel(s) once connected; otherwise state.db + # corruption or NFS/SMB lock failures silently degrade the gateway (nothing persists). self._session_db_init_error = str(e) - # Opportunistic state.db maintenance: prune ended sessions inactive - # for sessions.retention_days + optional VACUUM. Tracks last-run - # in state_meta so it only actually executes once per - # sessions.min_interval_hours. Gateway is long-lived so blocking - # a few seconds once per day is acceptable; failures are logged - # but never raised. + # Opportunistic state.db maintenance: prune ended sessions past sessions.retention_days + + # optional VACUUM, at most once per sessions.min_interval_hours (last-run in state_meta). + # A few blocking seconds per day is fine for a long-lived gateway; failures log, never raise. if self._session_db is not None: try: from hermes_cli.config import load_config as _load_full_config @@ -7198,18 +6732,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.debug("state.db auto-maintenance skipped: %s", exc) - # Opportunistic shadow-repo cleanup — deletes stale checkpoint repos - # under ~/.hermes/checkpoints/. Opt-in via checkpoints.auto_prune, - # idempotent via .last_prune marker. + # Opportunistic shadow-repo cleanup of stale checkpoint repos under ~/.hermes/checkpoints/; + # opt-in via checkpoints.auto_prune, idempotent via .last_prune marker. try: from hermes_cli.config import load_config as _load_full_config _ckpt_cfg = (_load_full_config().get("checkpoints") or {}) if _ckpt_cfg.get("auto_prune", False): from tools.checkpoint_manager import maybe_auto_prune_checkpoints - # delete_orphans is intentionally never honoured here: a missing workdir at startup - # is ambiguous (deleted project vs. an unmounted external volume / network share / - # VPN not yet up) and this sweep runs unattended. Orphan cleanup happens only via - # the explicit `hermes checkpoints prune` command. + # delete_orphans is never honoured here: this sweep runs unattended and a missing + # workdir at startup is ambiguous (deleted project vs. unmounted volume/share/VPN). + # Orphan cleanup happens only via the explicit `hermes checkpoints prune` command. maybe_auto_prune_checkpoints( retention_days=int(_ckpt_cfg.get("retention_days", 7)), min_interval_hours=int(_ckpt_cfg.get("min_interval_hours", 24)), @@ -7219,10 +6751,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.debug("checkpoint auto-maintenance skipped: %s", exc) - # DM pairing store for code-based user authorization. ``pairing_store`` stays as the - # global/default store for the ``hermes pairing`` CLI and any caller without a profile - # context. ``pairing_stores`` is the per-profile map ``authz_mixin._is_user_authorized`` - # routes checks through (one whitelist per profile in multiplex mode). + # DM pairing store for code-based user authorization. ``pairing_store`` is the global/default + # store (``hermes pairing`` CLI, callers without profile context); ``pairing_stores`` is the + # per-profile map ``authz_mixin._is_user_authorized`` routes through (one whitelist/profile). from gateway.pairing import PairingStore self.pairing_store = PairingStore() self.pairing_stores: Dict[str, "PairingStore"] = {} @@ -7233,26 +6764,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Per-chat voice reply mode: "off" | "voice_only" | "all" self._voice_mode: Dict[str, str] = self._load_voice_modes() - # Recent voice transcripts per (guild,user) for duplicate suppression. - # Protects against the same utterance being emitted twice by the voice - # capture / STT pipeline, which otherwise produces a second delayed reply. + # Recent voice transcripts per (guild,user): the voice capture / STT pipeline can emit the + # same utterance twice, which would otherwise produce a second delayed reply. self._recent_voice_transcripts: Dict[tuple[int, int], List[tuple[float, str]]] = {} # Track background tasks to prevent garbage collection mid-execution self._background_tasks: set = set() - # Event-loop liveness heartbeat (#66892): rewritten every 30s while - # the loop is dispatching. External supervisors use the file mtime / - # updated_at to distinguish "process alive" from "loop frozen". + # Event-loop liveness heartbeat: rewritten every 30s while the loop dispatches; supervisors + # use the file mtime / updated_at to tell "process alive" from "loop frozen". self._gateway_started_at: float = time.time() self._loop_heartbeat_task: Optional[asyncio.Task] = None self._loop_floor_timer_handle = None self._loop_liveness_watchdog = None - # scale-to-zero (Phase 0, F13): gateway-scoped "last inbound seen" clock. There is no such - # clock today (only a per-agent _last_activity_ts), so the idle predicate needs this. - # Stamped in _handle_message (the single inbound chokepoint); seeded to "now" so a fresh - # gateway isn't considered idle from epoch. Read by the scale-to-zero watcher. + # scale-to-zero: gateway-scoped "last inbound seen" clock (only a per-agent _last_activity_ts + # exists otherwise). Stamped in _handle_message (the single inbound chokepoint), seeded to + # "now" so a fresh gateway isn't idle from epoch; read by the scale-to-zero watcher. self._last_inbound_at: float = time.time() # Set after a wake (re-arm cooldown, 0.F) so we don't immediately re-go # dormant before the drained backlog has a chance to update the clock. @@ -7263,17 +6791,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _open_session_db_for_active_scope(self, raise_on_error: bool = False) -> Any: """Return the AsyncSessionDB for the profile scope active on this task. - ``SessionDB()`` resolves ``_default_db_path()`` at call time through the context-local - HERMES_HOME override installed by ``_profile_runtime_scope``, so resolving per access - (not once in ``__init__``) is what lets a multiplexed profile read its own store. - - One ``AsyncSessionDB`` is cached per resolved path, so the wrapper identity is stable per - profile (callers compare and stash it) and two profiles never share a handle. A - construction failure enters bounded backoff (one caller retries after the deadline). - ``raise_on_error=True`` (construction-time priming) propagates the failure after recording - that state so ``__init__`` can set ``_session_db_init_error``. + ``SessionDB()`` resolves its path at call time via the context-local HERMES_HOME override + from ``_profile_runtime_scope``, so resolving per access (not in ``__init__``) lets a + multiplexed profile read its own store. One ``AsyncSessionDB`` is cached per path (stable + identity; profiles never share a handle). Construction failure enters bounded backoff; + ``raise_on_error=True`` (priming) propagates it after recording state for ``__init__``. """ - from hermes_state import AsyncSessionDB, SessionDB, _default_db_path, get_shared_session_db + from hermes_state import AsyncSessionDB, _default_db_path, get_shared_session_db from gateway.session_db_recovery import RecoverableHandleCache path = Path(_default_db_path()) @@ -7288,12 +6812,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._session_db_handle_cache = cache def _open(): - # Borrow the SessionStore's handle for this path rather than opening a second one: both - # caches resolve the SAME path, so the process held two writer connections + two read - # pools per state.db (doubled again per profile). The store owns the handle and sweeps - # it at shutdown; this cache holds only the async wrapper. A borrowed wrapper cannot go - # stale: the store only drops handles in close_all_db_handles() (shutdown), and while - # its open is failing there is nothing to borrow, so nothing is cached here either. + # Borrow the SessionStore's handle for this path rather than opening a second one (both + # caches resolve the SAME path; otherwise two writer connections + two read pools per + # state.db, doubled per profile). The store owns and sweeps the handle at shutdown; this + # cache holds only the async wrapper. It cannot go stale: the store drops handles only in + # close_all_db_handles(), and while its open fails there is nothing to borrow or cache. store = getattr(self, "session_store", None) borrowed = getattr(store, "_db", None) if store is not None else None if borrowed is not None: @@ -7303,9 +6826,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew wrapper.__dict__["_hermes_borrowed_handle"] = True return wrapper if store is not None: - # The store exists and its handle is unavailable (failed open or backoff). Opening - # our own here would resurrect exactly the duplicate this borrows away from, so - # report the same unavailability the store is already reporting. + # Store handle unavailable (failed open/backoff): opening our own would resurrect the + # very duplicate this borrows away, so report the store's own unavailability. raise RuntimeError("SessionStore SQLite handle unavailable") try: return AsyncSessionDB(get_shared_session_db()) @@ -7328,10 +6850,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _session_db(self) -> Any: """The AsyncSessionDB for the active profile scope, or a pinned override. - Assigning ``runner._session_db`` pins that value for every subsequent read — tests rely - on installing fakes or ``None`` this way. Unpinned (the production path), each read - resolves the active scope so a multiplexed profile's slash commands and session search - hit its own store. + Assigning ``runner._session_db`` pins that value for every later read (tests install fakes + or ``None`` this way); unpinned, each read resolves the active profile scope's own store. """ if self._session_db_pinned is not _SESSION_DB_UNPINNED: return self._session_db_pinned @@ -7344,10 +6864,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def close_all_session_db_handles(self) -> None: """Close every per-profile AsyncSessionDB this runner opened. - Handles are drained under the lock and closed outside it; a pinned handle is the pinner's - to close. Wrappers around a handle BORROWED from ``session_store`` are drained but not - closed: the store owns that connection and its own sweep — which runs first in the - shutdown sequence — closes it. + Handles are drained under the lock and closed outside it; a pinned handle is the pinner's to + close. Wrappers BORROWED from ``session_store`` are drained but not closed: the store's own + sweep, which runs first in the shutdown sequence, closes that connection. """ def _close(db) -> None: if getattr(db, "__dict__", {}).get("_hermes_borrowed_handle"): @@ -7366,9 +6885,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _wire_teams_pipeline_runtime(self) -> None: """Bind the Teams meeting pipeline runtime to Graph webhook ingress. - No-op when the msgraph_webhook adapter isn't running or the - teams_pipeline plugin isn't enabled — lets the gateway start cleanly - whether or not the user has opted into the pipeline. + No-op when the msgraph_webhook adapter isn't running or the teams_pipeline plugin is off. """ if Platform.MSGRAPH_WEBHOOK not in self.adapters: return @@ -7396,9 +6913,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _warn_if_docker_media_delivery_is_risky(self) -> None: """Warn when Docker-backed gateways lack an explicit export mount. - MEDIA delivery happens in the gateway process, so paths emitted by the model must be - readable from the host. A container-local path like `/output/report.txt` often exists - only inside Docker, so users need an export mount such as `host-dir:/output`. + MEDIA delivery runs in the gateway process, so model-emitted paths like `/output/report.txt` + must be host-readable — users need an export mount such as `host-dir:/output`. """ if os.getenv("TERMINAL_ENV", "").strip().lower() != "docker": return @@ -7457,10 +6973,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Return a platform-namespaced key for voice mode state. - Under multiplexing the key is additionally namespaced by the profile whose bot speaks in - the chat (``::``); the default profile keeps the historical - ``:`` shape so persisted state stays valid. Two bots in one Discord - channel otherwise share a key and one profile's ``/voice`` flips the other's. + Under multiplexing the key is ``::`` (profile whose bot speaks); + the default profile keeps ``:`` so persisted state stays valid. Otherwise + two bots in one Discord channel share a key and one profile's ``/voice`` flips the other's. """ base = f"{platform.value}:{chat_id}" profile = profile.strip() if isinstance(profile, str) else "" @@ -7472,8 +6987,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Voice-state key for an inbound source, namespaced by its transport owner. Voice mode belongs to the (bot, chat) pair, so the namespace is the profile that OWNS the - receiving adapter (``_adapter_profile_for_source``) — the same profile - ``_sync_voice_mode_state_to_adapter`` uses on reconnect — not the routed runtime profile. + receiving adapter (matching ``_sync_voice_mode_state_to_adapter``), not the routed profile. """ return self._voice_key( source.platform, @@ -7523,45 +7037,40 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except OSError as e: logger.warning("Failed to save voice modes: %s", e) + @staticmethod + def _toggle_adapter_auto_tts_set(adapter, chat_id: str, on: bool, *, add_to: str, clear_from: str) -> None: + """Add/discard ``chat_id`` in the adapter's ``add_to`` set; adding also clears it from ``clear_from``. + + ``/voice off`` and an explicit ``/voice on``/``/voice tts`` are hard overrides of each other.""" + target = getattr(adapter, add_to, None) + if not isinstance(target, set): + return + if on: + target.add(chat_id) + other = getattr(adapter, clear_from, None) + if isinstance(other, set): + other.discard(chat_id) + else: + target.discard(chat_id) + def _set_adapter_auto_tts_disabled(self, adapter, chat_id: str, disabled: bool) -> None: """Update an adapter's in-memory auto-TTS suppression set if present.""" - disabled_chats = getattr(adapter, "_auto_tts_disabled_chats", None) - if not isinstance(disabled_chats, set): - return - if disabled: - disabled_chats.add(chat_id) - # ``/voice off`` also clears any explicit enable — it's a hard override. - enabled_chats = getattr(adapter, "_auto_tts_enabled_chats", None) - if isinstance(enabled_chats, set): - enabled_chats.discard(chat_id) - else: - disabled_chats.discard(chat_id) + self._toggle_adapter_auto_tts_set( + adapter, chat_id, disabled, add_to="_auto_tts_disabled_chats", clear_from="_auto_tts_enabled_chats" + ) def _set_adapter_auto_tts_enabled(self, adapter, chat_id: str, enabled: bool) -> None: - """Update an adapter's per-chat auto-TTS opt-in set if present. - - Used for ``/voice on``/``/voice tts`` where the user explicitly wants - auto-TTS even when ``voice.auto_tts`` is False globally. - """ - enabled_chats = getattr(adapter, "_auto_tts_enabled_chats", None) - if not isinstance(enabled_chats, set): - return - if enabled: - enabled_chats.add(chat_id) - # An explicit opt-in clears any stale /voice off for this chat. - disabled_chats = getattr(adapter, "_auto_tts_disabled_chats", None) - if isinstance(disabled_chats, set): - disabled_chats.discard(chat_id) - else: - enabled_chats.discard(chat_id) + """Update an adapter's per-chat auto-TTS opt-in set (auto-TTS even when ``voice.auto_tts`` is False).""" + self._toggle_adapter_auto_tts_set( + adapter, chat_id, enabled, add_to="_auto_tts_enabled_chats", clear_from="_auto_tts_disabled_chats" + ) def _sync_voice_mode_state_to_adapter(self, adapter) -> None: """Restore persisted /voice state into a live platform adapter. - Populates three fields from config + ``self._voice_mode``: - - ``_auto_tts_default``: global default from ``voice.auto_tts`` - - ``_auto_tts_enabled_chats``: chats with mode ``voice_only``/``all`` - - ``_auto_tts_disabled_chats``: chats with mode ``off`` + Sets ``_auto_tts_default`` (from ``voice.auto_tts``) and, from ``self._voice_mode``, + ``_auto_tts_enabled_chats`` (modes ``voice_only``/``all``) and ``_auto_tts_disabled_chats`` + (mode ``off``). """ platform = getattr(adapter, "platform", None) if not isinstance(platform, Platform): @@ -7630,10 +7139,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _safe_adapter_disconnect(self, adapter, platform) -> None: """Call adapter.disconnect() defensively, swallowing any error. - Used when adapter.connect() failed or raised — the adapter may have allocated partial - resources (aiohttp.ClientSession, poll tasks, child subprocesses) that would otherwise - leak and surface as "Unclosed client session" warnings at process exit. Must tolerate - partial-init state and never raise — callers use it inside error-handling blocks. + For a failed/raised connect(): partial resources (aiohttp.ClientSession, poll tasks, child + subprocesses) would otherwise leak. Must tolerate partial-init state and never raise. """ timeout = self._adapter_disconnect_timeout_secs() try: @@ -7658,13 +7165,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Tear down one adapter on the shutdown path with bounded awaits. - ``cancel_background_tasks()`` and ``disconnect()`` can block indefinitely on half-dead - network state (e.g. a wedged Feishu/Lark WebSocket thread). An unbounded await here stalls - the entire shutdown sequence past systemd's ``TimeoutStopSec``; the resulting SIGKILL - skips ``atexit`` PID-file cleanup, so the next start dies with "PID file race lost". - Each await uses the per-adapter budget (``HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT``); on - timeout the task is cancelled and detached so the loop never hangs even if an adapter - swallows cancellation. Never raises. + ``cancel_background_tasks()`` and ``disconnect()`` can block forever on half-dead network + state (e.g. a wedged WebSocket thread), stalling shutdown past systemd's ``TimeoutStopSec``; + the SIGKILL skips ``atexit`` PID-file cleanup and the next start dies with "PID file race + lost". Each await uses ``HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT``; on timeout the task is + cancelled and detached so a cancellation-swallowing adapter can't hang the loop. Never raises. """ timeout = self._adapter_disconnect_timeout_secs() suffix = f" (profile: {profile})" if profile else "" @@ -7718,10 +7223,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _platform_connect_timeout_secs(self, platform=None, *, initial: bool = False) -> float: """Return the per-platform connect timeout used during startup/retry. - Telegram's full connect budget (180s, raised for #67498 so cold polling can prove - getUpdates readiness) is deliberately NOT spent there: an unreachable Telegram would hold - the whole gateway out of the ``running`` state for the full budget. The cold-start wait is - capped and the platform handed to the reconnect watcher, which retries with the full + Telegram's full 180s connect budget is deliberately NOT spent at cold start: an unreachable + Telegram would hold the gateway out of ``running`` for the whole budget. The cold-start wait + is capped and the platform handed to the reconnect watcher, which retries with the full budget and ``is_reconnect=True`` (preserving the offline update queue). """ raw = os.getenv("HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT", "").strip() @@ -7746,20 +7250,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> bool: """Connect an adapter without allowing one platform to block others. - ``is_reconnect`` is forwarded to ``adapter.connect()`` so adapters can distinguish a cold - first boot (drop any stale server-side queue) from a watcher reconnect after an outage - (preserve the queue so messages sent meanwhile are delivered, not silently dropped). - - ``initial`` selects the capped cold-start budget for platforms whose full connect budget - is too long to spend before the gateway reaches ``running`` (#85993 — Telegram's 180s). + ``is_reconnect`` lets adapters distinguish a cold first boot (drop any stale server-side + queue) from a watcher reconnect (preserve the queue so interim messages aren't dropped). + ``initial`` selects the capped cold-start budget for platforms whose full connect budget is + too long to spend before the gateway reaches ``running`` (Telegram's 180s). """ timeout = self._platform_connect_timeout_secs(platform, initial=initial) if timeout <= 0: return await adapter.connect(is_reconnect=is_reconnect) - # Use the detach-on-timeout pattern instead of plain asyncio.wait_for: asyncio.wait_for - # cancels the overdue task but then waits for it to exit, so a connect() that catches - # CancelledError blocks recovery forever (the watcher never reaches the next retry). Keep - # ownership of the old task through its done callback, but release the runner at the deadline. + # Detach-on-timeout rather than plain asyncio.wait_for: wait_for cancels the overdue task but + # then waits for it to exit, so a connect() that catches CancelledError blocks recovery + # forever (watcher never retries). Keep ownership via its done callback; release at deadline. task = asyncio.ensure_future( adapter.connect(is_reconnect=is_reconnect) ) @@ -7821,9 +7322,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass config = getattr(self, "config", None) - # Mirror SessionStore._resolve_profile_for_key so this fallback path - # produces the same namespace as the primary path: None (legacy - # agent:main) unless multiplexing is on, then the active profile. + # Mirror SessionStore._resolve_profile_for_key so this fallback yields the primary path's + # namespace: None (legacy agent:main) unless multiplexing is on, then the active profile. _profile = None if getattr(config, "multiplex_profiles", False): if source.profile: @@ -7845,9 +7345,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _telegram_topic_profile_name(source: SessionSource) -> str: """Profile namespace for Telegram topic-mode rows. - Prefer the profile already stamped on the routed event (``source.profile``). Do **not** - fall back to the process-global active profile here — under multiplex that can mis- - attribute topic state across bots sharing one ``state.db``. + Use the profile stamped on the routed event (``source.profile``), never the process-global + active profile — under multiplex that mis-attributes topic state across bots sharing state.db. """ name = str(getattr(source, "profile", None) or "").strip() return name if name else "default" @@ -7870,14 +7369,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: logger.debug("Failed to read Telegram topic mode state", exc_info=True) return False - # Only honor a real True from the SessionDB. Any other value - # (including MagicMock instances from test fixtures that didn't - # opt into topic mode) means topic mode is off for this chat. + # Only a real True from the SessionDB enables topic mode; anything else (including MagicMock + # from test fixtures that didn't opt in) means off for this chat. return raw is True - # Telegram's General (pinned top) topic in forum-enabled private chats. - # Bot API behavior varies: some clients omit message_thread_id for - # General, others send "1". Treat both as "root" for lobby/lane purposes. + # Telegram's General (pinned top) topic in forum-enabled private chats: clients variously omit + # message_thread_id or send "1" for it. Treat both as "root" for lobby/lane purposes. _TELEGRAM_GENERAL_TOPIC_IDS = frozenset({"", "1"}) def _is_telegram_topic_root_lobby(self, source: SessionSource) -> bool: @@ -7896,9 +7393,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not self._telegram_topic_mode_enabled(source): return False tid = str(source.thread_id or "") - if not tid or tid in self._TELEGRAM_GENERAL_TOPIC_IDS: - return False - return True + return bool(tid) and tid not in self._TELEGRAM_GENERAL_TOPIC_IDS _TELEGRAM_LOBBY_REMINDER_COOLDOWN_S = 30.0 @@ -7914,12 +7409,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return f"{self._telegram_topic_profile_name(source)}:{chat_id}" def _should_send_telegram_lobby_reminder(self, source: SessionSource) -> bool: - """Rate-limit root-DM lobby reminders to one message per cooldown window. - - A user who forgets multi-session mode is enabled and types several - prompts in the root DM would otherwise get a reminder for every - message. Cap it so the first one lands and the rest stay quiet. - """ + """Rate-limit root-DM lobby reminders to one per cooldown window, not one per prompt typed.""" if not hasattr(self, "_telegram_lobby_reminder_ts"): self._telegram_lobby_reminder_ts = {} key = self._telegram_topic_cooldown_key(source) @@ -7990,11 +7480,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Update the topic binding to point at ``session_entry.session_id``. - Telegram topic lanes persist a (chat_id, thread_id) -> session_id row so reopening a - topic in a fresh process resumes the right Hermes session. When compression rotates - ``session_entry.session_id`` mid-turn the binding goes stale and the next inbound message - reloads the oversized parent instead of the compressed child, retriggering preflight - compression — sometimes in a loop. + Topic lanes persist (chat_id, thread_id) -> session_id so reopening a topic resumes the + right session. When compression rotates the id mid-turn a stale binding reloads the + oversized parent next message, retriggering preflight compression — sometimes in a loop. """ if not self._is_telegram_topic_lane(source): return @@ -8011,12 +7499,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[str]: """Pin DM-topic routing to the user's last-active topic. - Telegram can omit ``message_thread_id`` or surface General (``1``) for some topic-mode DM - replies; in those lobby-shaped cases keep the conversation on the user's most-recent - bound topic. Do not rewrite a non-lobby, previously-unbound thread id: a newly created - Telegram DM topic is also "unknown" until the first inbound message is recorded, and - rewriting it would send that brand-new topic's answer into an older lane. Returns None to - leave the source alone. + Telegram can omit ``message_thread_id`` or surface General (``1``) for topic-mode DM + replies; in those lobby-shaped cases keep the conversation on the user's most-recent bound + topic. Do not rewrite a non-lobby, previously-unbound thread id: a brand-new DM topic is + also "unknown" until its first inbound message is recorded, and rewriting would send its + answer into an older lane. Returns None to leave the source alone. """ if ( source.platform != Platform.TELEGRAM @@ -8029,9 +7516,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew inbound = str(source.thread_id or "") is_lobby = not inbound or inbound in self._TELEGRAM_GENERAL_TOPIC_IDS if not is_lobby: - # A non-lobby, unknown thread_id is most likely the first message in a brand-new - # Telegram DM topic. Preserve it so it can be recorded as a new independent lane below - # instead of hijacking the latest existing topic binding. + # A non-lobby, unknown thread_id is likely the first message of a new Telegram DM topic: + # preserve it to be recorded as a new lane below rather than hijack the latest binding. return None session_db = getattr(self, "_session_db", None) if session_db is None: @@ -8064,15 +7550,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Apply Telegram DM topic recovery to a source for session-key purposes. ``_handle_message_with_agent`` rewrites ``source.thread_id`` via - ``_recover_telegram_topic_thread_id`` *before* deriving the session key for a normal - message turn (a lobby/stripped reply gets pinned to the user's last-active topic). - Session-scoped handlers like ``/model`` deriving their key from the raw ``event.source`` - skip that recovery, so the override lands under a different key than the next turn reads - and is silently dropped on forum topics / after compression splits. - - Returns a recovery-normalized copy when a rewrite applies, otherwise - the original source unchanged. Always derive the override storage key - from the result so storage and read use an identical key. + ``_recover_telegram_topic_thread_id`` *before* deriving the session key, so handlers like + ``/model`` keying off the raw ``event.source`` would store an override under a different key + than the next turn reads (silently dropped on forum topics / after splits). Returns a + recovery-normalized copy when a rewrite applies, else the original; always derive the override + storage key from the result so storage and read match. """ try: recovered = self._recover_telegram_topic_thread_id(source) @@ -8082,6 +7564,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return source return dataclasses.replace(source, thread_id=recovered) + def _resolve_session_key_or_none(self, source, session_key: Optional[str]) -> Optional[str]: + """``session_key`` if given, else the key for ``source`` (None when it cannot be derived).""" + if session_key or source is None: + return session_key + try: + return self._session_key_for_source(source) + except Exception: + return None + def _resolve_session_agent_runtime( self, *, @@ -8094,12 +7585,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew Priority (highest first): session ``/model`` → ``channel_overrides`` → global config/env (``_resolve_gateway_model(user_config)`` and default provider resolution). """ - resolved_session_key = session_key - if not resolved_session_key and source is not None: - try: - resolved_session_key = self._session_key_for_source(source) - except Exception: - resolved_session_key = None + resolved_session_key = self._resolve_session_key_or_none(source, session_key) model = _resolve_gateway_model(user_config) if resolved_session_key: @@ -8199,9 +7685,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew resolved_session_key, model, runtime_kwargs ) - # When the config has no model.default but a provider was resolved (e.g. user ran `hermes - # auth add openai-codex` without `hermes model`), fall back to the provider's first catalog - # model so the API call doesn't fail with "model must be a non-empty string". + # No model.default but a provider resolved (e.g. `hermes auth add openai-codex` without + # `hermes model`): fall back to the provider's first catalog model so the API call has one. if not model and runtime_kwargs.get("provider"): try: from hermes_cli.models import get_default_model_for_provider @@ -8214,14 +7699,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass - # Final safety net (#35314): if resolution still produced an empty - # model — e.g. a transient config-cache miss during a post-interrupt - # recovery turn returned an empty user_config — reuse the last model we - # successfully resolved for this session (or, failing that, the most - # recent one resolved process-wide). Building an agent with model="" - # makes every API call fail HTTP 400 "No models provided" and the - # session goes silent until the user manually re-sends. ``getattr`` - # guards against bare test runners built via ``object.__new__``. + # Final safety net: if resolution still produced an empty model (e.g. a transient config-cache + # miss on a post-interrupt recovery turn), reuse the last model resolved for this session, + # else the most recent process-wide — model="" makes every API call fail HTTP 400 and the + # session goes silent. ``getattr`` guards bare test runners built via ``object.__new__``. if not model: _lr_state = ( self._peek_session_state(resolved_session_key) @@ -8255,10 +7736,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Build the effective model/runtime config for a single turn. Always uses the session's primary model/provider. If `/fast` is enabled and the model - supports Priority Processing / Anthropic fast mode, attach `request_overrides` so the API - call is marked accordingly. Per-provider ``request_overrides`` from - ``resolve_runtime_provider`` (e.g. a ``custom_providers`` ``extra_body``) are preserved and - merged *under* the fast-mode overrides so they still reach the model on the gateway path. + supports it, attach `request_overrides` for priority processing. Per-provider + ``request_overrides`` from ``resolve_runtime_provider`` (e.g. ``custom_providers`` + ``extra_body``) are merged *under* the fast-mode overrides so they still reach the model. """ from hermes_cli.models import resolve_fast_mode_overrides @@ -8318,10 +7798,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Persist the runtime model/provider actually used by a gateway turn. Provider fallback can switch ``agent.model``/``agent.provider`` after the session row was - created. Keep the session DB metadata in sync so session lists, desktop/dashboard details, - and follow-up session tooling report the backend that actually answered the latest turn. - Called from the ``run_sync`` closure (executor thread, off the event loop), so the sync - ``SessionDB`` (``_db``) is used directly rather than awaiting the AsyncSessionDB forwarder. + created; keep the DB metadata in sync so session lists and tooling report the backend that + actually answered. Runs in the ``run_sync`` closure (executor thread), so it uses the sync + ``SessionDB`` (``_db``) directly rather than the AsyncSessionDB forwarder. """ if not session_id or agent is None or self._session_db is None: return @@ -8362,9 +7841,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _handle_reaction_event(self, ctx: Dict[str, Any]) -> None: """Fan a normalised platform reaction event out to the HookRegistry. - The adapter-supplied ``event_name`` ("reaction:added" / "reaction:removed") becomes the - hook event so user hooks subscribe with the same name scheme as the existing ``agent:*`` - family. Errors never block the adapter's event loop — the hook contract is non-blocking. + The adapter-supplied ``event_name`` ("reaction:added"/"reaction:removed") is the hook event, + matching the ``agent:*`` naming scheme. Errors never block the adapter's event loop. """ event_name = str(ctx.get("event_name") or "reaction:added") try: @@ -8375,14 +7853,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _handle_adapter_fatal_error(self, adapter: BasePlatformAdapter) -> None: """React to an adapter failure after startup. - Retryable errors (network blip, DNS) queue the platform for background reconnection - instead of giving up permanently. - - The notification arrives on the failing adapter's own polling task, and the disconnect - inside the handler can cancel that task mid-flight: disconnect()'s current-task guard - misses it because _safe_adapter_disconnect runs the close in a wrapper task. A cancelled - handler dies between the fatal log and the reconnect queue, silently stranding the - platform — so the real work runs in a detached task adapter teardown cannot cancel. + Retryable errors (network blip, DNS) queue the platform for background reconnection. + The notification arrives on the failing adapter's own polling task, and the disconnect in + the handler can cancel that task mid-flight (disconnect()'s current-task guard misses it + because _safe_adapter_disconnect closes in a wrapper task), stranding the platform between + the fatal log and the reconnect queue — so the real work runs in a detached task. """ tasks = getattr(self, "_fatal_handler_tasks", None) if tasks is None: @@ -8399,9 +7874,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _queue_retryable_fatal_platform(self, adapter: BasePlatformAdapter) -> bool: """Queue a retryable fatal adapter for background reconnection. - Returns True when the platform was newly queued. Idempotent if already - queued. Must not await: callers invoke this *before* any disconnect - await so a wedged close cannot strand the platform (#80598). + Returns True when newly queued; idempotent if already queued. Must not await: callers + invoke this *before* any disconnect await so a wedged close cannot strand the platform. """ if not adapter.fatal_error_retryable: return False @@ -8409,12 +7883,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not platform_config: return False if adapter.platform in self._failed_platforms: - # Nothing to enqueue -- but "already queued" is precisely the state in which the watcher - # has had time to die, and the enqueue branch below holds the ONLY call to - # _ensure_reconnect_watcher_running(). _spawn_supervised restarts a crashed watcher only - # _MAX_SUPERVISED_RESTARTS times, then it stays dead; without this backstop a queued - # platform is a silent permanent outage (nothing retries, and the stranded check treats - # a queued platform as safe so the process never restarts either). + # Nothing to enqueue — but "already queued" is exactly when the watcher may have died, + # and the enqueue branch below holds the ONLY _ensure_reconnect_watcher_running() call. + # _spawn_supervised gives up after _MAX_SUPERVISED_RESTARTS; without this backstop a + # queued platform is a silent permanent outage (nothing retries, and the stranded check + # treats a queued platform as safe so the process never restarts either). self._ensure_reconnect_watcher_running() return False self._failed_platforms[adapter.platform] = { @@ -8433,19 +7906,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "%s queued for background reconnection", adapter.platform.value, ) - # Ensure the reconnect watcher is alive — if it died (e.g. from - # exhausting its restart budget), respawn it so queued platforms - # are not permanently stranded (#70344). + # Ensure the reconnect watcher is alive — respawn if it died (e.g. restart budget exhausted) + # so queued platforms are not permanently stranded. self._ensure_reconnect_watcher_running() return True async def _handle_adapter_fatal_error_detached( self, adapter: BasePlatformAdapter ) -> None: - """Run the fatal handler; if the platform still ends up stranded - (not reconnected, not queued, not intentionally disabled), exit the - gateway with failure so the service manager restarts it instead of - leaving a silent partial outage.""" + """Run the fatal handler; if the platform still ends up stranded (not reconnected, not + queued, not intentionally disabled), exit the gateway with failure so the service manager + restarts it instead of leaving a silent partial outage.""" try: # Outer hard deadline: even with queue-before-disconnect, a hang anywhere in the impl # (status write side effects, detach races, etc.) must not leave this task wedged @@ -8454,9 +7925,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if timeout <= 0: await self._handle_adapter_fatal_error_impl(adapter) else: - # Disconnect budget plus a small overhead for queue/status - # bookkeeping. Keep the additive proportional so tests that - # shrink the disconnect timeout still finish promptly. + # Disconnect budget plus a little queue/status bookkeeping overhead; keep the extra + # proportional so tests that shrink the disconnect timeout still finish promptly. outer = timeout + min(2.0, max(0.05, timeout)) completed = await self._await_adapter_cleanup_with_timeout( self._handle_adapter_fatal_error_impl(adapter), @@ -8519,9 +7989,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await self.stop() async def _handle_adapter_fatal_error_impl(self, adapter: BasePlatformAdapter) -> None: - # Snapshot the current owner of this platform slot before doing anything else. Acting on a - # stale notification would overwrite an already-healthy platform's runtime status and - # incorrectly re-queue it for reconnection, so bail out before any of that happens. + # Snapshot this platform slot's current owner first: acting on a stale notification would + # overwrite a healthy platform's runtime status and wrongly re-queue it for reconnection. existing = self.adapters.get(adapter.platform) if existing is not None and existing is not adapter: logger.debug( @@ -8537,9 +8006,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew adapter.fatal_error_code or "unknown", adapter.fatal_error_message or "unknown error", ) - # Phase 7 Unit 7d-B: a relay credential revoked by opt-out is not an error to retry — render - # it as a clean "disabled" state, not red "fatal"/"retrying". (The code is set non- - # retryable, so it also drops out of the reconnect queue below.) + # A relay credential revoked by opt-out is not an error to retry: render a clean "disabled" + # state, not red "fatal"/"retrying" (non-retryable code, so it also leaves the queue below). if adapter.fatal_error_code == "relay_disabled": platform_state = "disabled" elif adapter.fatal_error_retryable: @@ -8554,10 +8022,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) if existing is adapter: - # Claim this adapter for teardown before awaiting disconnect() — a second fatal-error - # notification for the same adapter (e.g. from a concurrent recovery path) would - # otherwise still see itself as "existing" during the await below and disconnect() the - # same object twice. + # Claim this adapter for teardown before awaiting disconnect(): a second fatal-error + # notification for the same adapter (e.g. a concurrent recovery path) would otherwise + # still see itself as "existing" during the await and disconnect() the same object twice. self.adapters.pop(adapter.platform, None) self.delivery_router.adapters = self.adapters @@ -8568,9 +8035,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._queue_retryable_fatal_platform(adapter) if existing is adapter: - # A half-closed transport can wedge an adapter's native close() - # indefinitely. Reuse the shutdown-path timeout so this runtime - # fatal handler always returns to the stay-alive / stranded path. + # A half-closed transport can wedge native close() indefinitely; reuse the shutdown-path + # timeout so this runtime fatal handler always returns to the stay-alive / stranded path. await self._safe_adapter_disconnect(adapter, adapter.platform) if not self.adapters and not self._failed_platforms: @@ -8582,16 +8048,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.error("No connected messaging platforms remain. Shutting down gateway cleanly.") await self.stop() elif not self.adapters and self._failed_platforms: - # All platforms are down and queued for background reconnection. - # Keep the gateway alive so: - # • cron jobs still run - # • the reconnect watcher can recover platforms when the - # underlying problem clears (proxy comes back, user runs - # `hermes whatsapp`, etc.) - # We used to exit-with-failure here to trigger systemd restart, - # but that converted a transient outage into a restart loop and - # killed in-process state every time. The reconnect watcher - # already handles long-running recovery — let it do its job. + # All platforms are down and queued for reconnection. Keep the gateway alive so cron jobs + # still run and the watcher can recover platforms when the problem clears; exiting for a + # systemd restart would turn a transient outage into a state-killing restart loop. logger.warning( "No connected messaging platforms remain, but %d platform(s) " "queued for reconnection — gateway staying alive, watcher will " @@ -8617,14 +8076,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) def _active_cron_job_count(self) -> int: - """Count of cron jobs currently executing, from the cron scheduler's own in-flight tracking - (``cron.scheduler._running_job_ids``). + """Count of cron jobs currently executing (``cron.scheduler._running_job_ids``). - Cron jobs run through a standalone ``AIAgent`` on the scheduler's own thread pool, outside - ``self._running_agents`` — the dict every OTHER active-work check here reads. Without this - the shutdown drain is blind to in-flight cron work and can kill a cron job's tool - subprocess while it is still running. Best-effort: returns 0 if the cron module can't be - imported (e.g. a minimal test double for this class). + Cron jobs run on the scheduler's own thread pool, outside ``self._running_agents`` which + every OTHER active-work check reads; without this the shutdown drain can kill a cron job's + tool subprocess mid-run. Best-effort: returns 0 if the cron module can't be imported. """ try: from cron.scheduler import get_running_job_ids @@ -8635,9 +8091,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _active_api_run_count(self) -> int: """Count API-server work that is outside ``_running_agents``. - The primary API server owns the sole HTTP listener. Secondary multiplex - profiles cannot create an ``api_server`` adapter because it binds a port, - so only the primary registry is a supported source of this work. + Only the primary API server owns the HTTP listener (secondary multiplex profiles cannot + bind a port), so only the primary registry is a source of this work. """ try: adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) @@ -8649,10 +8104,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _interrupt_api_server_runs(self, reason: str) -> int: """Interrupt API-server agents that are not in ``_running_agents``. - Counterpart of ``_active_api_run_count()``: that method folds adapter-owned API work into - the shutdown drain, so this one must reach the same agents when the drain times out. - Duck-typed on the adapter so an older adapter (or minimal test double) without the hook is - skipped rather than raising mid-shutdown. + Counterpart of ``_active_api_run_count()``: must reach the same agents when the drain times + out. Duck-typed so an adapter (or test double) without the hook is skipped, not raised on. """ try: adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) @@ -8728,15 +8181,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # runner/transport. def _scale_to_zero_has_live_background_work(self) -> bool: - """Live background work that must block a suspend (D3/F7). + """Live background work that must block a suspend. Backgrounded delegate_task / kanban / terminal(background=true) are NOT counted by - _running_agent_count(), but suspending mid-flight loses them. Checks tracked tasks + the - process registry + pending completion watchers. PERMANENT supervised watchers (tagged - _hermes_supervised_watcher) are excluded: they live for the whole process — including the - scale-to-zero watcher itself — so counting them would make this True forever and the - gateway could never go dormant. Fly's coarse autostop used to mask this; with the gateway - owning the suspend it became load-bearing. + _running_agent_count() but suspending loses them; checks tracked tasks + process registry + + pending completion watchers. PERMANENT supervised watchers (_hermes_supervised_watcher) are + excluded — they live for the whole process (including the scale-to-zero watcher itself), so + counting them would make this True forever and the gateway could never go dormant. """ if any( not t.done() and not getattr(t, "_hermes_supervised_watcher", False) @@ -8776,13 +8227,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return parse_idle_timeout_seconds(raw) def _restart_loop_guard_config(self) -> tuple: - """Return ``(max_restarts, window_seconds, max_gap_seconds)`` for the auto-resume restart- - loop breaker (#30719, defense-3), read from ``gateway.restart_loop_guard`` in config.yaml - with the module defaults as fallback. + """Return ``(max_restarts, window_seconds, max_gap_seconds)`` for the auto-resume + restart-loop breaker, from ``gateway.restart_loop_guard`` with module defaults as fallback. ``max_restarts <= 0`` disables the breaker. ``max_gap_seconds`` is the longest spacing - between two consecutive restart-interrupted boots that still counts them as the same - loop, so a crash cycle slower than ``window_seconds`` stays visible. + between consecutive restart-interrupted boots that still counts as the same loop, so a + crash cycle slower than ``window_seconds`` stays visible. """ from gateway import restart_loop_guard as _rlg @@ -8808,16 +8258,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return max_restarts, window_seconds, max_gap_seconds def _scale_to_zero_active_messaging_platforms(self) -> list: - """ENABLED platforms that count for the relay-only arm gate (D1/F6). + """ENABLED platforms that count for the relay-only arm gate. - Two filters, both load-bearing: - - enabled only: config.platforms is pre-seeded with disabled placeholders for the full - platform catalog (the F25 bug). - - MESSAGING only: non-messaging surfaces must not disarm scale-to-zero. The api_server is - a loopback listener force-enabled by API_SERVER_KEY (present on every hosted container); - it holds no outbound socket and Chronos fires through it already reset the idle clock. - Counting it silently disarmed the feature on every hosted instance. Mirrors the - non-messaging exclusion used for handoff eligibility in _connect_platforms. + Two load-bearing filters: enabled only (config.platforms is pre-seeded with disabled + placeholders for the whole catalog) and MESSAGING only (the api_server is a loopback listener + force-enabled on every hosted container with no outbound socket; counting it silently + disarmed the feature everywhere). Mirrors the non-messaging exclusion in _connect_platforms. """ if not self.config: return [] @@ -8854,9 +8300,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _log_scale_to_zero_not_armed_reason(self) -> None: """Log why the idle watcher did NOT arm — but only for an OPTED-IN instance. - A non-opted instance (no HERMES_SCALE_TO_ZERO stamp) not arming is the normal case and - must stay silent. When the stamp IS set but the watcher didn't arm, that surprise earns - one INFO line so "why won't it suspend/wake?" is a log grep, not a box-dive. + A non-opted instance (no HERMES_SCALE_TO_ZERO stamp) not arming is normal and stays silent; + with the stamp set, the surprise earns one INFO line so the answer is a log grep. """ from gateway.relay import relay_wake_url from gateway.scale_to_zero import ( @@ -8891,17 +8336,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _scale_to_zero_is_idle(self) -> bool: from gateway.scale_to_zero import is_idle - # The FULL work aggregate, not _running_agent_count(): cron jobs run - # on the scheduler's own thread pool and API-server runs live on the - # adapter — both outside _running_agents (the #60432 blind spot), so - # counting agents alone let a suspend land mid-cron-job. - # - # Fail-AWAKE accounting: the shared shutdown-drain counters - # (_active_cron_job_count/_active_api_run_count) swallow exceptions to - # 0, which is fine for a drain but unsafe for a suspend predicate — a - # transient read failure would make live work look idle and reopen the - # mid-job freeze. Here an unreadable source counts as work (sentinel 1) - # so the machine stays awake until the source is readable again. + # The FULL work aggregate, not _running_agent_count(): cron jobs and API-server runs live + # outside _running_agents, so counting agents alone let a suspend land mid-cron-job. + # Fail-AWAKE accounting: the shutdown-drain counters swallow exceptions to 0, which is fine + # for a drain but unsafe for a suspend predicate (a transient read failure would look idle). + # Here an unreadable source counts as work (sentinel 1) so the machine stays awake. try: from cron.scheduler import get_running_job_ids @@ -8916,11 +8355,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: # noqa: BLE001 - unreadable source => assume busy logger.debug("scale-to-zero: api work count unreadable — staying awake", exc_info=True) api_count = 1 - # An attached dashboard/desktop/TUI client is inbound activity too. It lives in the - # DASHBOARD process, so it reaches us as a file mtime that process refreshes on every WS - # frame (gateway/scale_to_zero.py). Fold it into the inbound clock rather than adding a - # conjunct: the client gets the same idle_timeout grace after disconnect as a chat message, - # and a lingering marker cannot pin the box (an old mtime is outside idle_timeout). + # An attached dashboard/desktop/TUI client is inbound activity too; it lives in the DASHBOARD + # process and reaches us as a file mtime refreshed on every WS frame (gateway/scale_to_zero.py). + # Folded into the inbound clock rather than a conjunct: same idle_timeout grace after + # disconnect as a chat message, and a lingering marker cannot pin the box. last_inbound = self._last_inbound_at try: from gateway.scale_to_zero import dashboard_client_last_seen @@ -8941,10 +8379,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _scale_to_zero_note_real_inbound(self) -> None: """Stamp real inbound and restore lifecycle after a dormant wake. - The watcher marks runtime status `draining` as it quiesces the relay, but dormancy is not - the stop/restart drain path: the process remains alive and should present as running once - real traffic wakes it and re-enters the gateway. Internal completion/replay events - intentionally do not call this helper, so they do not keep an idle gateway awake. + Dormancy marks status `draining` but is not the stop/restart drain: the process stays alive + and should present as running once real traffic wakes it. Internal completion/replay events + deliberately do not call this, so they don't keep an idle gateway awake. """ self._last_inbound_at = time.time() if getattr(self, "_scale_to_zero_cooldown_until", 0.0) > 0: @@ -8965,18 +8402,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _scale_to_zero_watcher(self, interval: float = 30.0) -> None: """Watch for idle, drive the relay dormant, then self-suspend the machine. - Started ONLY when _scale_to_zero_should_arm() (Labs HERMES_SCALE_TO_ZERO stamp + - relay-only/absent messaging + a wakeUrl). On a sustained idle window it runs the DORMANT - sequence: mark status `draining` (does NOT set _running=False), relay adapter.go_dormant() - (supervisor-preserving socket close — NOT disconnect()/the stop path), deliberately NO - mark_resume_pending (suspend preserves RAM), THEN suspend the machine via the local flaps - socket. The gateway owns the suspend because Fly autostop judges idle on INBOUND - connections only: it cannot see an in-flight agent turn or an outbound relay socket, so - autostop:"suspend" would freeze mid-job or before the relay flip; NAS provisions these - machines with autostop:"off". Autostart stays platform-side (wakeUrl poke, supervisor - re-dials, connector drains the backlog). After driving dormant we set a re-arm cooldown so - a wake's drained backlog isn't immediately re-quiesced. Off-Fly (no flaps socket) the - watcher does not quiesce at all: the platform suspends on its own timer. + Armed ONLY via _scale_to_zero_should_arm() (HERMES_SCALE_TO_ZERO stamp + relay-only/absent + messaging + wakeUrl). On sustained idle: mark status `draining` (NOT _running=False), relay + adapter.go_dormant() (supervisor-preserving socket close, NOT disconnect()), NO + mark_resume_pending (suspend preserves RAM), THEN suspend via the local flaps socket. The + gateway owns the suspend because Fly autostop sees only INBOUND connections and would freeze + mid-job or before the relay flip (machines run autostop:"off"); autostart stays platform-side. + A re-arm cooldown keeps a wake's drained backlog from being re-quiesced. Off-Fly (no flaps + socket) the watcher does not quiesce at all. """ await asyncio.sleep(min(interval, 30.0)) # let startup settle while self._running: @@ -8994,11 +8427,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew go_dormant = getattr(adapter, "go_dormant", None) if not callable(go_dormant): continue - # Quiesce only when a suspend can follow it. Off-Fly the platform owns the freeze, - # and go_dormant()'s socket close arms the reconnect supervisor: it re-dials ~1.4s - # later and the drain clears the flip every cooldown. The destination is then - # unflipped when the freeze lands, and inbound is dropped instead of buffered. Stay - # connected and let the connector's orphan detection adopt the destination. + # Quiesce only when a suspend can follow. Off-Fly the platform owns the freeze and + # go_dormant()'s socket close arms the reconnect supervisor (re-dial ~1.4s, unflipped + # at freeze, inbound dropped not buffered); stay connected, orphan detection adopts it. from gateway.scale_to_zero import self_suspend_available if not self_suspend_available(): @@ -9027,14 +8458,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: # noqa: BLE001 - dormancy is best-effort dormant_ok = False logger.debug("scale-to-zero: go_dormant failed", exc_info=True) - # 0.F: after a wake the drained inbound updates _last_inbound_at, - # but give it a window so we don't immediately re-go-dormant on the - # same idle reading before traffic lands. + # After a wake the drained inbound updates _last_inbound_at; give it a window so we + # don't immediately re-go-dormant on the same idle reading before traffic lands. self._scale_to_zero_cooldown_until = time.time() + max(interval, 60.0) - # Self-suspend ONLY after a clean quiesce: the relay flip must be - # set (buffered delivery + wake poke armed) before the freeze, or - # inbound events black-hole while we sleep. Re-check idle one last - # time — inbound may have landed during the quiesce await. + # Self-suspend ONLY after a clean quiesce: the relay flip (buffered delivery + wake + # poke armed) must be set before the freeze, or inbound black-holes while we sleep. + # Re-check idle one last time — inbound may have landed during the quiesce await. if not dormant_ok: continue if not self._scale_to_zero_is_idle(): @@ -9051,10 +8480,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _scale_to_zero_self_suspend(self) -> None: """Suspend this Fly machine via the local flaps socket (fail-awake). - Runs the blocking unix-socket call in a worker thread so the event loop stays live right - up to the kernel freeze. On success the process is frozen shortly after — nothing - meaningful runs until the wake resume. Off-Fly (self_suspend_available() False) this is a - silent no-op. + Blocking unix-socket call runs in a worker thread so the loop stays live until the kernel + freeze; nothing meaningful runs until wake. Off-Fly this is a silent no-op. """ from gateway.scale_to_zero import self_suspend_available, suspend_self @@ -9083,18 +8510,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _queue_during_drain_enabled( self, busy_input_mode: Optional[str] = None ) -> bool: - # Both "queue" and "steer" modes imply the user doesn't want messages - # to be lost during restart — queue them for the newly-spawned gateway - # process to pick up. "interrupt" mode drops them (current behaviour). + # "queue" and "steer" both mean messages must not be lost across restart: queue them for + # the newly-spawned gateway process to pick up. "interrupt" mode drops them. mode = busy_input_mode or self._busy_input_mode return self._restart_requested and mode in {"queue", "steer"} # -------- /queue FIFO helpers -------------------------------------- - # /queue must produce one full agent turn per invocation, in FIFO order, with no merging. The - # adapter's _pending_messages dict is a single "next-up" slot (shared with photo-burst follow- - # ups), so we use it for the head of the queue and an overflow list for the tail. Enqueue fills - # the slot when free, else the overflow; promotion (after each run's drain) moves the next - # overflow item into the slot for the following recursion. Cleared on /new and /reset. + # /queue yields one full agent turn per invocation, FIFO, no merging. _pending_messages is a + # single "next-up" slot (shared with photo-burst follow-ups) holding the head; an overflow list + # holds the tail. Promotion after each run's drain refills the slot. Cleared on /new and /reset. def _enqueue_fifo(self, session_key: str, queued_event: "MessageEvent", adapter: Any) -> None: """Append a /queue event to the FIFO chain for a session.""" @@ -9118,10 +8542,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional["MessageEvent"]: """Promote the next overflow item after the slot was drained. - Called at the drain site after _dequeue_pending_event consumed (or failed to consume) the - slot. With an overflow item: if pending_event is None return the overflow head as the new - pending_event; if the slot is already populated (interrupt follow-up etc.), stage the head - in the slot so the NEXT recursion picks it up. Returns the (possibly updated) pending_event. + If pending_event is None, return the overflow head as the new pending_event; if the slot is + already populated (interrupt follow-up etc.), stage the head there for the NEXT recursion. + Returns the (possibly updated) pending_event. """ _q_state = self._peek_session_state(session_key) overflow = _q_state.conversation.queued_events if _q_state else None @@ -9150,20 +8573,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional["MessageEvent"]: """Pop the oldest orphaned FIFO overflow event for an idle session. - The FIFO overflow (``queued_events``) drains only at the post-turn promotion site inside - the ``_run_agent`` drain. When a busy window ends without that drain running (recursion - exits before promoting, or an exception / interrupt / generation-bump exit), the overflow - entries are silently orphaned: never dispatched, persisted, or logged. - - This rescue runs at the point where a NEW event arrives for a session that is NOT busy - (the idle entry in ``_process_message_priority``). The oldest orphan is returned so the - caller runs it as THIS turn and the next one is staged into the slot so the post-turn - drain continues the chain in arrival order. The caller then enqueues the incoming event - behind the chain via ``_enqueue_fifo``. The returned event is REMOVED from both stores: - leaving it in the slot would make the post-turn ``_dequeue_pending_event`` run it twice. - - Returns the orphaned event to run now, or ``None`` when there is - nothing to rescue (no overflow, slot occupied, or no slot storage). + ``queued_events`` drains only at the post-turn promotion site in ``_run_agent``; if a busy + window ends without that drain (early recursion exit, exception/interrupt/generation-bump), + the overflow is silently orphaned. Called when a NEW event arrives for a NON-busy session: + the oldest orphan is returned to run as THIS turn, the next is staged into the slot so the + chain continues in arrival order, and the caller enqueues the incoming event behind it. The + returned event is REMOVED from both stores, else the post-turn dequeue would run it twice. + Returns ``None`` when there is nothing to rescue (no overflow, slot occupied, or no slot). """ try: _q_state = self._peek_session_state(session_key) @@ -9212,9 +8628,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _clear_goal_pending_continuations(self, session_key: str, adapter: Any) -> int: """Remove queued synthetic /goal continuations for one session. - User-issued /goal pause/clear can race with a continuation already - queued by the judge. Remove only synthetic goal continuations while - preserving normal /queue and user follow-up events. + User /goal pause/clear can race a judge-queued continuation; only synthetic goal + continuations are removed, normal /queue and user follow-up events are preserved. """ removed = 0 pending_slot = getattr(adapter, "_pending_messages", None) if adapter is not None else None @@ -9248,50 +8663,34 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return False def _update_runtime_status(self, gateway_state: Optional[str] = None, exit_reason: Optional[str] = None) -> None: - try: - from gateway.status import write_runtime_status - write_runtime_status( - gateway_state=gateway_state, - exit_reason=exit_reason, - restart_requested=self._restart_requested, - active_agents=self._active_work_count(), - ) - except Exception: - pass + _write_runtime_status_quiet( + gateway_state=gateway_state, + exit_reason=exit_reason, + restart_requested=self._restart_requested, + active_agents=self._active_work_count(), + ) def _persist_active_agents(self) -> None: """Persist the live in-flight agent count to ``gateway_state.json``. - Called at every turn boundary (a running-agent slot is claimed or released) so the - dashboard ``/api/status`` readout reflects in-flight gateway turns in near-real-time - (otherwise the file only moves on lifecycle transitions and reads between them are stale). - Deliberately passes ONLY ``active_agents`` — the other fields stay ``_UNSET`` so - ``write_runtime_status``'s read-merge-write preserves the current lifecycle state; passing - ``gateway_state=None`` here would clobber it. Best-effort: a failed write must never - disrupt a turn. + Called at every turn boundary so the dashboard ``/api/status`` readout is near-real-time + (otherwise the file only moves on lifecycle transitions). Passes ONLY ``active_agents`` — + other fields stay ``_UNSET`` so the read-merge-write preserves lifecycle state; passing + ``gateway_state=None`` would clobber it. Best-effort: a failed write must never disrupt a turn. """ - try: - from gateway.status import write_runtime_status - write_runtime_status(active_agents=self._active_work_count()) - except Exception: - pass + _write_runtime_status_quiet(active_agents=self._active_work_count()) # ------------------------------------------------------------------ - # External drain control (NAS-driven quiesce-without-restart, Phase 2). - # The dashboard's begin/cancel-drain endpoint writes/removes the - # ``.drain_request.json`` marker (gateway/drain_control.py); this watcher - # observes the marker and flips the gateway between accepting and refusing - # NEW turns, WITHOUT exiting the process. Reversible by design (D4a): NAS - # POSTs begin-drain, polls /api/status until active_agents hits 0, proceeds - # with its lifecycle action, then (on cancel/abort) the marker is removed - # and the gateway re-accepts turns. + # External drain control (NAS-driven quiesce-without-restart). The dashboard's + # begin/cancel-drain endpoint writes/removes the ``.drain_request.json`` marker + # (gateway/drain_control.py); this watcher flips the gateway between accepting and refusing + # NEW turns WITHOUT exiting. Reversible: NAS begins drain, polls /api/status until + # active_agents hits 0, acts; on cancel/abort the marker is removed and turns resume. # ------------------------------------------------------------------ def _enter_external_drain(self) -> None: - """Begin external drain: stop accepting new turns, flip state. + """Begin external drain: refuse NEW turns (in-flight ones are NOT interrupted), flip state. - Idempotent — re-entering while already draining is a no-op beyond a - best-effort status re-write. In-flight turns are NOT interrupted (the - whole point is to let them finish); only NEW turns are refused. + Idempotent: re-entry only re-writes status. """ if self._external_drain_active: return @@ -9301,17 +8700,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "new turns; %d in-flight turn(s) will finish. Process stays up.", self._active_work_count(), ) - # Flip the persisted lifecycle state so /api/status.gateway_busy / - # gateway_drainable track the drain. Preserve active_agents (the - # read-merge keeps the live count); only the state changes. + # Flip persisted lifecycle state so /api/status.gateway_busy / gateway_drainable track the + # drain; active_agents is preserved (read-merge keeps the live count), only state changes. self._update_runtime_status("draining") def _exit_external_drain(self) -> None: """Cancel external drain: revert state, re-accept new turns. - Idempotent. Only reverts to ``running`` when we are actually mid-drain - AND not also shutting down (a real shutdown ``_draining`` must win — - never resurrect a stopping gateway to ``running``). + Idempotent. Reverts to ``running`` only when actually mid-drain AND not shutting down — + a real shutdown ``_draining`` must win; never resurrect a stopping gateway. """ if not self._external_drain_active: return @@ -9333,12 +8730,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _drain_control_watcher(self, interval: float = 1.0) -> None: """Background task: reconcile gateway accept-state with the drain marker. - Polls ``.drain_request.json`` (presence-based contract, gateway/drain_control.py): marker - present -> ``_enter_external_drain``, absent -> ``_exit_external_drain``; the 1s cadence - bounds observe-the-marker latency. Reconciles once at startup. A marker stamped with a - PRIOR instantiation epoch (survived a machine restart on the durable volume) is treated as - absent by ``drain_requested`` and NOT honoured. Best-effort: any tick error is logged and - the loop continues (a transient stat() failure must not wedge the gateway). + Polls ``.drain_request.json`` (presence-based) at 1s: present -> enter drain, absent -> exit; + reconciles once at startup. A marker from a PRIOR instantiation epoch (survived a machine + restart) is treated as absent. Best-effort: tick errors are logged and the loop continues. """ from gateway.drain_control import drain_requested @@ -9349,9 +8743,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # pressure one read can stall 30s+ and take every platform heartbeat down. Off-thread it. if await asyncio.to_thread(drain_requested): self._enter_external_drain() - # API and cron work live outside messaging's - # _running_agents map. Refresh the aggregate while an - # external caller polls this reversible drain state. + # API and cron work live outside messaging's _running_agents map; refresh the + # aggregate while an external caller polls this reversible drain state. self._persist_active_agents() else: self._exit_external_drain() @@ -9389,17 +8782,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew pass # ------------------------------------------------------------------ - # Per-platform circuit breaker (pause/resume) — used by the reconnect - # watcher when a retryable failure recurs past a threshold, and by the - # /platform pause|resume slash command for manual control. + # Per-platform circuit breaker (pause/resume): reconnect watcher + /platform pause|resume. # ------------------------------------------------------------------ def _pause_failed_platform(self, platform, *, reason: str = "") -> None: - """Mark a queued platform as paused — keep it in ``_failed_platforms`` but stop the - reconnect watcher from hammering it. + """Mark a queued platform as paused — stays in ``_failed_platforms`` but the reconnect + watcher stops hammering it. - Used by ``/platform pause `` for manual operator intervention. Note: the reconnect - watcher does NOT auto-pause — retryable (network/DNS) failures keep retrying at the - backoff cap indefinitely so a transient outage self-heals without manual intervention. + Manual (``/platform pause ``) only: the watcher never auto-pauses — retryable failures + keep retrying at the backoff cap so a transient outage self-heals. """ info = getattr(self, "_failed_platforms", {}).get(platform) if info is None: @@ -9411,15 +8801,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Push next_retry far enough out that even if "paused" is missed # by a stale code path, the watcher won't fire on it. info["next_retry"] = float("inf") - try: + with suppress(Exception): self._update_platform_runtime_status( platform.value, platform_state="paused", error_code=None, error_message=info["pause_reason"], ) - except Exception: - pass logger.warning( "%s paused after %d consecutive failures (%s) — " "fix the underlying issue then run `/platform resume %s` " @@ -9442,15 +8830,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew info.pop("pause_reason", None) info["attempts"] = 0 info["next_retry"] = time.monotonic() # retry on next watcher tick - try: + with suppress(Exception): self._update_platform_runtime_status( platform.value, platform_state="retrying", error_code=None, error_message=None, ) - except Exception: - pass logger.info("%s resumed — retrying on next watcher tick", platform.value) return True @@ -9458,9 +8844,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _load_prefill_messages() -> List[Dict[str, Any]]: """Load ephemeral prefill messages from config or env var. - Checks HERMES_PREFILL_MESSAGES_FILE env var first, then falls back to the top-level - prefill_messages_file key in ~/.hermes/config.yaml. agent.prefill_messages_file is - accepted as a legacy fallback. Relative paths are resolved from ~/.hermes/. + HERMES_PREFILL_MESSAGES_FILE env wins, then top-level prefill_messages_file in config.yaml, + then legacy agent.prefill_messages_file. Relative paths resolve from ~/.hermes/. """ file_path = os.getenv("HERMES_PREFILL_MESSAGES_FILE", "") if not file_path: @@ -9489,9 +8874,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @staticmethod def _load_ephemeral_system_prompt() -> str: - """Load ephemeral system prompt from config or env var. - - Checks HERMES_EPHEMERAL_SYSTEM_PROMPT env var first, then + """Load ephemeral system prompt: HERMES_EPHEMERAL_SYSTEM_PROMPT env var first, then ``display.personality`` / ``agent.system_prompt`` in config.yaml. """ from hermes_cli.config import resolve_ephemeral_system_prompt_from_config @@ -9513,11 +8896,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Resolve model for this channel: channel_overrides else global default. - Delegates the precedence rule to :func:`hermes_cli.model_switch.resolve_effective_model` - (session override > channel override > global default) — the single owner shared with the - API server, so the two surfaces cannot diverge again (see 7dd00bb47d). This call site has - no session tier: session /model overrides are applied later by - ``_apply_session_model_override`` on the resolved runtime. + Precedence lives in :func:`hermes_cli.model_switch.resolve_effective_model` (shared with the + API server so the surfaces cannot diverge). No session tier here: session /model overrides + are applied later by ``_apply_session_model_override``. """ from hermes_cli.model_switch import resolve_effective_model @@ -9547,12 +8928,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Ephemeral system prompt for this channel/thread. - Uses ``channel_overrides`` when set, else the gateway prompt resolved from the CURRENT - profile's config on every call: callers run inside ``_profile_runtime_scope``, so a routed - multiplex profile gets its own ``display.personality`` / ``agent.system_prompt`` rather - than a boot-time snapshot, and ``/personality`` edits apply on the next turn. - Legacy ``channel_prompts`` are applied separately via ``event.channel_prompt`` in - ``run_sync`` (adapter ``resolve_channel_prompt``), so they are not duplicated here. + ``channel_overrides`` when set, else the gateway prompt resolved from the CURRENT profile's + config on every call (callers run inside ``_profile_runtime_scope``, so routed multiplex + profiles get their own personality/system_prompt and ``/personality`` edits apply next turn). + Legacy ``channel_prompts`` are applied separately via ``event.channel_prompt`` in ``run_sync``. """ config = getattr(self, "config", None) if config: @@ -9571,12 +8950,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _load_reasoning_config(model: str = "") -> dict | None: """Load reasoning effort from config.yaml, respecting per-model overrides. - Thin wrapper over the shared chokepoint :func:`hermes_constants.resolve_reasoning_config` - (per-model override > global ``agent.reasoning_effort``; YAML boolean False = disabled). - - Args: - model: The effective model for the calling session. When empty, - the config's ``model.default`` is used. + Thin wrapper over :func:`hermes_constants.resolve_reasoning_config` (per-model override > + global ``agent.reasoning_effort``; YAML False = disabled). Empty ``model`` uses ``model.default``. """ from hermes_constants import resolve_reasoning_config cfg = _load_gateway_runtime_config() @@ -9586,8 +8961,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _parse_reasoning_command_args(raw_args: str) -> tuple[str, bool]: """Parse `/reasoning` args into `(value, persist_global)`. - `/reasoning ` is session-scoped by default. `--global` may be - supplied in any position to persist the change to config.yaml. + Session-scoped by default; `--global` in any position persists the change to config.yaml. """ import shlex @@ -9617,17 +8991,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> dict | None: """Resolve reasoning effort for a session, honoring session overrides. - Priority: session-scoped ``/reasoning --session`` override > per-model override - (``agent.reasoning_overrides``) > global ``agent.reasoning_effort``. ``model`` should be - the session's *effective* model (session ``/model`` override included) so per-model - overrides track what the session actually runs — when empty, ``model.default`` is used. + Priority: session ``/reasoning --session`` > per-model ``agent.reasoning_overrides`` > global + ``agent.reasoning_effort``. ``model`` must be the session's *effective* model (session + ``/model`` override included); empty uses ``model.default``. """ - resolved_session_key = session_key - if not resolved_session_key and source is not None: - try: - resolved_session_key = self._session_key_for_source(source) - except Exception: - resolved_session_key = None + resolved_session_key = self._resolve_session_key_or_none(source, session_key) if resolved_session_key: _r_state = self._peek_session_state(resolved_session_key) @@ -9643,9 +9011,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Set or clear the session-scoped reasoning override.""" if not session_key: return - # Per-session field write — the old lazy ``self._session_reasoning_overrides - # = {}`` init replaced the WHOLE dict, racing concurrent sessions' - # overrides; a SessionState field reset cannot cross sessions. + # Per-session field write: a lazy ``_session_reasoning_overrides = {}`` init replaced the + # WHOLE dict, racing concurrent sessions; a SessionState field reset cannot cross sessions. self._session_state(session_key).conversation.reasoning_override = ( None if reasoning_config is None else dict(reasoning_config) ) @@ -9657,16 +9024,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[str]: """Resolve the effective service tier for a session. - A session-scoped /fast override wins over the config default. The - override dict stores "priority" or None (explicit normal), so key - presence — not value truthiness — decides whether it applies. + A session-scoped /fast override beats the config default; the override dict stores + "priority" or None (explicit normal), so key presence — not truthiness — decides. """ - resolved_session_key = session_key - if not resolved_session_key and source is not None: - try: - resolved_session_key = self._session_key_for_source(source) - except Exception: - resolved_session_key = None + resolved_session_key = self._resolve_session_key_or_none(source, session_key) if resolved_session_key: _t_state = self._peek_session_state(resolved_session_key) @@ -9692,19 +9053,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not session_key: return # Presence-sensitive: "priority" or None (explicit normal) both count as an override; the - # sentinel means "no override". Old code wholesale-replaced the dict on lazy init (cross- - # session race) — per-session field writes eliminate that class of bug. + # sentinel means "no override". Per-session field write: a lazy dict replace races sessions. self._session_state(session_key).conversation.service_tier_override = ( _SERVICE_TIER_UNSET if clear else service_tier ) @staticmethod def _load_service_tier() -> str | None: - """Load Priority Processing setting from config.yaml. - - Reads agent.service_tier from config.yaml. Accepted values mirror the CLI: - "fast"/"priority"/"on" => "priority", while "normal"/"off" disables it. - Returns None when unset or unsupported. + """Load Priority Processing (agent.service_tier) from config.yaml: "fast"/"priority"/"on" => + "priority"; "normal"/"off" disable; None when unset/unsupported. """ cfg = _load_gateway_runtime_config() raw = str(cfg_get(cfg, "agent", "service_tier", default="") or "").strip() @@ -9745,9 +9102,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _load_busy_text_mode() -> str: """Resolve normal busy TEXT follow-up behavior. - ``busy_input_mode`` is the single source of truth (default ``interrupt``). The legacy - ``busy_text_mode`` knob is honored only when a user explicitly set it, so existing queue - setups keep working; new installs follow ``busy_input_mode``. + ``busy_input_mode`` is the source of truth (default ``interrupt``); legacy ``busy_text_mode`` + is honored only when explicitly set so existing queue setups keep working. """ # Legacy explicit override wins for backward compat. legacy = os.getenv("HERMES_GATEWAY_BUSY_TEXT_MODE", "").strip().lower() @@ -9852,50 +9208,39 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return value @staticmethod - def _load_restart_after_turn_timeout() -> float: - """Load in-band restart wait-for-idle timeout in seconds (#77184).""" - env_raw = os.getenv("HERMES_RESTART_AFTER_TURN_TIMEOUT") + def _load_env_or_agent_cfg_timeout(env_var: str, cfg_key: str, parse, default: float) -> float: + """Env var (non-empty) else ``agent.``; warn once when a supplied value fails to parse. + + ``0`` is a valid value; the parser falls back to ``default`` on garbage.""" + env_raw = os.getenv(env_var) if env_raw is not None and str(env_raw).strip() != "": raw: object = env_raw else: cfg = _load_gateway_runtime_config() - raw = cfg_get(cfg, "agent", "restart_after_turn_timeout", default=None) - value = parse_restart_after_turn_timeout(raw) - # Warn only when the user supplied a non-empty value that failed to - # parse (parser falls back to the default). ``0`` is valid. + raw = cfg_get(cfg, "agent", cfg_key, default=None) + value = parse(raw) if raw is not None and str(raw).strip() != "": try: float(raw) except (TypeError, ValueError): - logger.warning( - "Invalid restart_after_turn_timeout '%s', using default %.0fs", - raw, - DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, - ) + logger.warning("Invalid %s '%s', using default %.0fs", cfg_key, raw, default) return value - @staticmethod - def _load_cron_drain_timeout() -> float: - """Load the cron-only floor under the stop()/drain wait (#82161).""" - env_raw = os.getenv("HERMES_CRON_DRAIN_TIMEOUT") - if env_raw is not None and str(env_raw).strip() != "": - raw: object = env_raw - else: - cfg = _load_gateway_runtime_config() - raw = cfg_get(cfg, "agent", "cron_drain_timeout", default=None) - value = parse_cron_drain_timeout(raw) - # Warn only when the user supplied a non-empty value that failed to - # parse (parser falls back to the default). ``0`` is valid. - if raw is not None and str(raw).strip() != "": - try: - float(raw) - except (TypeError, ValueError): - logger.warning( - "Invalid cron_drain_timeout '%s', using default %.0fs", - raw, - DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, - ) - return value + @classmethod + def _load_restart_after_turn_timeout(cls) -> float: + """Load in-band restart wait-for-idle timeout in seconds.""" + return cls._load_env_or_agent_cfg_timeout( + "HERMES_RESTART_AFTER_TURN_TIMEOUT", "restart_after_turn_timeout", + parse_restart_after_turn_timeout, DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, + ) + + @classmethod + def _load_cron_drain_timeout(cls) -> float: + """Load the cron-only floor under the stop()/drain wait.""" + return cls._load_env_or_agent_cfg_timeout( + "HERMES_CRON_DRAIN_TIMEOUT", "cron_drain_timeout", + parse_cron_drain_timeout, DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, + ) @staticmethod def _load_signal_interrupt_grace_timeout() -> float: @@ -9974,9 +9319,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _load_fallback_model() -> list | None: """Load fallback provider chain from config.yaml. - Returns the merged effective chain from ``fallback_providers`` plus any - legacy ``fallback_model`` entries. ``fallback_providers`` stays first - when both keys are present. + Merges ``fallback_providers`` (kept first) with legacy ``fallback_model`` entries. """ try: # Canonical gateway loader (fail-open): managed overlay + ${VAR} @@ -9992,12 +9335,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _refresh_fallback_model(self) -> list | None: """Re-read fallback_providers from disk for the next agent create/reuse. - Cron already does this per job via ``get_fallback_chain``; the gateway previously froze - ``self._fallback_model`` at process start, so a chain changed after startup never reached - messaging sessions. A TRANSIENT read/parse failure (user mid-edit of config.yaml with a - non-atomic write) keeps the last known-good chain instead of wiping a cached agent's - working fallback for that turn. Only a successful read that genuinely lacks the key clears - the chain. + Lets a chain edited after startup reach messaging sessions (cron already re-reads per job). + A TRANSIENT read/parse failure (user mid-edit, non-atomic write) keeps the last known-good + chain; only a successful read that genuinely lacks the key clears it. """ try: from hermes_cli.config import read_user_config_raw @@ -10035,10 +9375,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _apply_fallback_chain_to_agent(agent: Any, chain: list | None) -> None: """Keep a cached agent's fallback chain aligned with current config. - Skips the rewrite while a cooldown holds the agent on an already-activated fallback - provider — ``restore_primary_runtime`` owns that turn-scoped lifecycle. When primary is - active (or cooldown expired), replace the chain so mid-uptime ``fallback_providers`` edits - take effect without requiring a gateway restart. + Skips the rewrite while a cooldown holds the agent on an activated fallback provider + (``restore_primary_runtime`` owns that lifecycle); otherwise replaces the chain so + mid-uptime ``fallback_providers`` edits apply without a restart. """ if agent is None: return @@ -10054,16 +9393,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew agent._fallback_model = new_chain[0] if new_chain else None if not getattr(agent, "_fallback_activated", False): agent._fallback_index = 0 - # A config edit signals the user changed something — drop the session-scoped unavailability - # memo so re-configured entries (e.g. credentials added mid-uptime for a previously-failing - # provider) get retried instead of staying suppressed for the cached agent's lifetime. - # Only on actual content change, so the per-message no-op refresh keeps the memo's - # rate-limiting benefit. + # A config edit means the user changed something — drop the session-scoped unavailability + # memo so re-configured entries (e.g. credentials added mid-uptime) get retried. Only on real + # content change, so the per-message no-op refresh keeps the memo's rate-limiting benefit. if new_chain != old_chain: unavailable = getattr(agent, "_unavailable_fallback_keys", None) if unavailable: unavailable.clear() + def _running_agent_ids(self) -> set: + """``id()`` of every agent mid-turn — identity-keyed so the lookup is O(1) and independent of + ``AIAgent.__eq__`` (MagicMock overrides it in tests).""" + return { + id(a) + for _, a in self._running_agent_items() + if a is not None and a is not _AGENT_PENDING_SENTINEL + } + def _snapshot_running_agents(self) -> Dict[str, Any]: return { session_key: agent @@ -10117,11 +9463,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "platform": platform, "chat_id": getattr(source, "chat_id", "") or "", "user_id": getattr(source, "user_id", "") or "", - # Writer identity for re-entrancy (#94595): if this - # process leaks a lease for this session (exception path - # skipped release), the next turn re-acquires its own - # entry instead of being fenced out of it forever — - # pruning only reclaims entries whose PROCESS died. + # Writer identity for re-entrancy: if this process leaks a lease for this session + # (exception path skipped release), the next turn re-acquires its own entry rather + # than being fenced out forever — pruning only reclaims entries whose PROCESS died. "live_session_id": str(session_key), }, ) @@ -10131,24 +9475,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @staticmethod def _agent_has_active_subagents(running_agent: Any) -> bool: - """Return True when *running_agent* is currently driving subagents via the ``delegate_task`` - tool. + """Return True when *running_agent* is driving subagents via ``delegate_task``. - ``AIAgent.interrupt()`` cascades through the parent's ``_active_children`` and interrupts - every child synchronously, aborting in-flight subagent work. Demoting - ``busy_input_mode='interrupt'`` to ``queue`` while this is True protects that work from - conversational follow-ups; explicit ``/stop`` (``_interrupt_and_clear_session``) is - untouched. Safe-by-default: returns False on any attribute or lock error so a - missing/broken parent never blocks the existing interrupt path. + ``AIAgent.interrupt()`` cascades through ``_active_children`` and aborts in-flight subagent + work, so callers demote ``busy_input_mode='interrupt'`` to ``queue`` while this is True; + explicit ``/stop`` is untouched. Fail-safe: returns False on any attribute/lock error. """ if running_agent is None or running_agent is _AGENT_PENDING_SENTINEL: return False children = getattr(running_agent, "_active_children", None) - # AIAgent always initialises this as a concrete list (see - # agent/agent_init.py). Reject anything that isn't a real - # collection — this guards against ``MagicMock()._active_children`` - # auto-creating a truthy stub in tests and triggering the demotion - # against an agent that doesn't actually have subagents. + # AIAgent always initialises this as a concrete list. Reject anything that isn't a real + # collection — guards against ``MagicMock()._active_children`` auto-creating a truthy stub + # in tests and triggering the demotion for an agent with no subagents. if not isinstance(children, (list, tuple, set)): return False if not children: @@ -10165,12 +9503,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _session_has_compression_in_flight(self, session_key: str) -> bool: """Return True when a compression lock is held for this session's id. - Context compression is interrupt-protected, but gateway ``interrupt`` busy mode can still - start a follow-up turn against the pre-rotation parent while compression is mid-flight, - producing orphaned compression siblings. Callers demote interrupt to queue when this - returns True. Both blocking sources — the ``session_store`` lock + JSON load, and the - SQLite ``get_compression_lock_holder`` SELECT — are offloaded to a worker thread so a - large state.db never freezes the event loop. + Gateway ``interrupt`` busy mode could start a follow-up against the pre-rotation parent while + compression is mid-flight, producing orphaned compression siblings; callers demote interrupt + to queue when True. Both blocking sources (``session_store`` lock + JSON load, SQLite lock + holder SELECT) run in a worker thread so a large state.db never freezes the event loop. """ session_store = getattr(self, "session_store", None) if not session_key or session_store is None: @@ -10200,9 +9536,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew holder = await asyncio.to_thread( raw_db.get_compression_lock_holder, str(session_id) ) - # Production returns Optional[str]. Reject non-strings so a - # MagicMock auto-attr (or any unexpected truthy) cannot look - # like a held lock and skip hygiene (#96953). + # Production returns Optional[str]. Reject non-strings so a MagicMock auto-attr (or any + # unexpected truthy) cannot look like a held lock and skip hygiene. return isinstance(holder, str) and bool(holder) except (AttributeError, TypeError): return False @@ -10234,10 +9569,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew adapter = self._adapter_for_source(event.source) if not adapter: return - # Route through the FIFO infrastructure shared with ``/queue`` so each follow-up gets its - # own turn in arrival order (the old merge_text=False call silently OVERWROTE the single - # pending slot). Photo bursts still merge into the head slot (album semantics); everything - # else appends to the overflow tail. + # Route through the ``/queue`` FIFO infrastructure so each follow-up gets its own turn in + # arrival order (merge_text=False silently OVERWROTE the single pending slot). Photo bursts + # still merge into the head slot (album semantics); everything else appends to the tail. pending_slot = getattr(adapter, "_pending_messages", None) existing = pending_slot.get(session_key) if isinstance(pending_slot, dict) else None security_metadata_keys = ( @@ -10285,14 +9619,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _prepare_busy_steer_text(self, event: MessageEvent) -> str: """Return steerable text for a busy follow-up, transcribing voice first. - Fresh and queued voice messages reach the normal inbound STT pipeline, but successful - steer messages intentionally bypass that queue. Without preprocessing here, a media-only - voice follow-up has an empty text payload and steer mode silently degrades to queue mode. - Audio file attachments remain files; only voice-message media follows the automatic STT - contract. If transcription fails, preserve any caption and let the steer fallback handle - the event. Routes through ``_transcribe_and_echo_pending_voice`` — the single out-of-band - STT choke point — so STT runs at most once per message (cached on the event) and a later - queue fallback reuses the transcript instead of paying/echoing twice. + Successful steer messages bypass the inbound STT queue, so without this a media-only voice + follow-up has empty text and steer silently degrades to queue mode. Only voice-message media + (not audio file attachments) is transcribed; on failure keep any caption and let the steer + fallback handle it. Goes through ``_transcribe_and_echo_pending_voice`` — the single + out-of-band STT choke point — so STT runs at most once per message (cached on the event). """ text = (event.text or "").strip() if not self._pending_event_audio_paths(event): @@ -10311,10 +9642,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return (enriched_text or text).strip() async def _handle_active_session_busy_message(self, event: MessageEvent, session_key: str) -> bool: - # --- Authorization gate (#17775) --- - # The cold path (_handle_message) checks _is_user_authorized before creating a session. The - # busy path must enforce the same check; otherwise unauthorized users in shared threads - # (Slack/Telegram/Discord) can inject messages into an active session they don't own. + # Authorization gate: the cold path (_handle_message) checks _is_user_authorized before + # creating a session; the busy path must enforce the same check, else unauthorized users in + # shared threads (Slack/Telegram/Discord) inject messages into a session they don't own. if not self._is_user_authorized(event.source): logger.warning( "Dropping message from unauthorized user in active session: " @@ -10356,16 +9686,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return True - # --- Approval response routing (#46866) --- - # When the agent is blocked waiting for a dangerous-command approval, plain-text responses - # like "yes" or "approve" must be routed to the approval handler instead of being - # steered/queued/interrupted — otherwise the reply queues behind a turn that can't start - # until the approval resolves, so it times out and auto-denies (deadlock). Slash forms - # already bypass at the base-adapter guard; this handles bare words (Signal/SMS users type - # "yes"). Gating on has_blocking_approval(session_key) keeps a conversational "yes" from - # triggering a dangerous command when nothing is pending. Reuse the canonical /approve and - # /deny handlers (they resolve the thread, resume typing, return a localized confirmation); - # the busy path does not auto-send that return, so we deliver it ourselves. + # Approval routing: while blocked on a dangerous-command approval, a bare "yes" must reach the + # approval handler, not be steered/queued/interrupted (else it queues behind a turn that can't + # start until the approval resolves -> auto-deny deadlock). Slash forms already bypass at the + # base-adapter guard. Gated on has_blocking_approval so a conversational "yes" never fires a + # command. Reuse the /approve and /deny handlers; the busy path does not auto-send their return. try: from tools.approval import has_blocking_approval if event.allow_gateway_control and has_blocking_approval(session_key): @@ -10385,10 +9710,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _approval_handler = self._handle_approve_command _normalized_args = "session" if _approval_handler is not None: - # Synthesize the canonical "/approve [args]" / "/deny" command text so the slash - # handlers parse modifiers via event.get_command_args(). Always a literal "/" — - # is_command()/get_command_args() only recognize "/", not the per-platform - # display prefix ("!" on Slack/Matrix). + # Synthesize "/approve [args]" / "/deny" so the slash handlers parse modifiers via + # event.get_command_args(). Always a literal "/": is_command()/get_command_args() + # don't recognize per-platform display prefixes ("!" on Slack/Matrix). _verb = "approve" if _approval_handler is self._handle_approve_command else "deny" _synth = f"/{_verb}" if _normalized_args: @@ -10423,13 +9747,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not adapter: return False # let default path handle it - # --- Internal synthetic events must never interrupt/steer --- - # Async-delegation completions (delegate_task(background=true)) and background-process - # completions (terminal notify_on_complete) re-enter the originating session as internal - # MessageEvents. Treating them like user TEXT while busy would let interrupt mode abort the - # active turn and send an "Interrupting" ack — the opposite of the invariant that a - # completion surfaces as a NEW turn only when idle. Plugin events carry untrusted payload - # text, so queue those through the gateway FIFO to keep their security metadata separate. + # Internal synthetic events (async-delegation / background-process completions) must never + # interrupt/steer: treated as user TEXT while busy, interrupt mode would abort the active turn; + # a completion surfaces as a NEW turn only when idle. Plugin events carry untrusted payload + # text, so queue them through the gateway FIFO (security metadata kept apart). if getattr(event, "internal", False) and not event.allow_gateway_control: self._queue_or_replace_pending_event(session_key, event) return True @@ -10447,12 +9768,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): return False - # Steer mode: inject mid-run via running_agent.steer() instead of queueing + interrupting. - # If the agent isn't running yet (sentinel) or lacks steer(), or the payload is empty, fall - # back to queue semantics so nothing is lost. Subagent protection: interrupt() cascades to - # ``_active_children`` and aborts delegate_task work, so demote ``interrupt`` to ``queue`` - # while the parent drives subagents. Explicit /stop and /new go through - # ``_interrupt_and_clear_session`` and still force-cancel everything. + # Steer mode injects mid-run via running_agent.steer(); fall back to queue (nothing lost) if the + # agent isn't running yet (sentinel), lacks steer(), or the payload is empty. interrupt() + # cascades to ``_active_children`` and aborts delegate_task work, so demote ``interrupt`` to + # ``queue`` while the parent drives subagents; explicit /stop and /new still force-cancel all. demoted_for_subagents = ( effective_mode == "interrupt" and self._agent_has_active_subagents(running_agent) @@ -10479,9 +9798,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew redirected = False if effective_mode == "steer": steer_text = await self._prepare_busy_steer_text(event) - # A follow-up qualifies for steering when it is plain text, OR when every attachment is - # STT-eligible voice media whose transcript was just folded into steer_text — otherwise - # a voice note in steer mode silently degrades to queue mode. + # Steerable: plain text, OR every attachment is STT-eligible voice media whose transcript + # was folded into steer_text — else a voice note in steer mode silently degrades to queue. _steer_media_urls = getattr(event, "media_urls", None) or [] _steer_all_voice = bool(_steer_media_urls) and ( len(self._pending_event_audio_paths(event)) == len(_steer_media_urls) @@ -10525,13 +9843,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.warning("Gateway redirect failed for session %s: %s", session_key, exc) redirected = False - # Store the message so it's processed as the next turn after the current run finishes (or is - # interrupted). Skip this for a successful steer — the text already landed inside the run - # and must NOT also be replayed as a next-turn user message. Route through - # _queue_or_replace_pending_event (FIFO) rather than a raw merge_pending_message_event - # (merge_text=True): the raw merge newline-joins consecutive TEXT follow-ups into ONE turn, - # destroying message boundaries. FIFO gives each text its own turn while keeping - # photo-burst / album merge semantics for media. + # Queue as the next turn after the current run ends. Skip after a successful steer — the text + # is already in the run and must NOT replay. Use the FIFO helper, not raw + # merge_pending_message_event (merge_text=True newline-joins consecutive TEXT follow-ups into + # ONE turn); FIFO gives each text its own turn while keeping photo-burst / album merge for media. if not steered and not redirected: self._queue_or_replace_pending_event(session_key, event) @@ -10539,9 +9854,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew is_steer_mode = effective_mode == "steer" is_redirect_mode = effective_mode == "interrupt" and redirected - # If not in queue/steer mode, interrupt the running agent immediately. - # This aborts in-flight tool calls and causes the agent loop to exit - # at the next check point. + # Interrupt mode: abort in-flight tool calls; the agent loop exits at its next check point. if ( effective_mode == "interrupt" and not redirected @@ -10565,17 +9878,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass # don't let interrupt failure block the ack - # Check if busy ack is disabled — skip sending but still process the input. - # Placed before debounce so we don't stamp a "last ack" timestamp that was - # never actually delivered. + # Disabled ack: skip sending, still process input. Checked before debounce so we never stamp a + # "last ack" timestamp for an ack that was not delivered. busy_ack_enabled = os.environ.get("HERMES_GATEWAY_BUSY_ACK_ENABLED", "true").lower() == "true" if not busy_ack_enabled: logger.debug("Busy ack suppressed for session %s", session_key) return True # input still processed, just no ack sent - # Debounce before consulting config-heavy display settings. Rapid - # follow-ups should be processed but should not trigger another config - # read just to discover that no ack will be sent. + # Debounce before the config-heavy display lookup: rapid follow-ups are still processed but + # shouldn't cost a config read just to learn no ack will be sent. _BUSY_ACK_COOLDOWN = 30 now = time.time() last_ack = _busy_state.turn.busy_ack_ts if _busy_state else 0 @@ -10585,9 +9896,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew from gateway.display_config import resolve_display_setting platform_key = _platform_config_key(event.source.platform) - # In steer mode the user's text has already been injected into the active run. Some mobile - # chat setups want that steering to be silent, like STT transcript echo suppression: keep - # the behavior, drop only the confirmation bubble. + # Steer mode already injected the text; some mobile chat setups want silent steering (like STT + # echo suppression) — keep the behavior, drop only the confirmation bubble. if is_steer_mode: steer_ack_env = os.environ.get("HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED") if steer_ack_env is not None: @@ -10607,9 +9917,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._session_state(session_key).turn.busy_ack_ts = now - # Build a status-rich acknowledgment. Mobile chat defaults keep this - # terse; detailed iteration/tool state is still available in logs and - # can be opted in per platform via display.platforms..busy_ack_detail. + # Mobile chat defaults keep the ack terse; iteration/tool detail stays in logs and can be opted + # in per platform via display.platforms..busy_ack_detail. status_parts = [] busy_ack_detail_enabled = bool( resolve_display_setting( @@ -10650,9 +9959,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"I'll adjust using your correction." ) elif is_queue_mode and demoted_for_subagents: - # #30170 — explain the demotion so the user knows their - # follow-up didn't accidentally kill the subagent and - # discovers `/stop` as the explicit escape hatch. + # Explain the demotion: the follow-up didn't kill the subagent; /stop is the escape hatch. message = ( f"⏳ Subagent working{status_detail} — your message is queued for " f"when it finishes (use /stop to cancel everything)." @@ -10673,8 +9980,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"I'll respond to your message shortly." ) - # First-touch onboarding: the very first time a user sends a message while the agent is - # busy, append a one-time hint explaining the queue/interrupt knob. Flag is persisted to + # First-touch onboarding: one-time hint about the queue/interrupt knob; the flag is persisted to # config.yaml so it never fires again on this install. try: from agent.onboarding import ( @@ -10754,11 +10060,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew last_deferred_count = deferred_count last_status_at = now - # Cron jobs run on the scheduler's own thread pool, outside ``self._running_agents`` — fold - # their in-flight count into the same wait/timeout this method already applies to chat - # sessions, or a cron job's tool work gets killed with zero warning the instant it's the - # only active thing running. API-server / desk sessions and detached deferred workers - # have the same structural gap. + # Cron jobs run on the scheduler's pool, outside ``self._running_agents`` — fold their in-flight + # count into this wait, or a cron job's tool work is killed without warning once it's the only + # active thing running. API-server/desk sessions and detached deferred workers share the gap. if ( not self._running_agents and last_cron_count == 0 @@ -10770,13 +10074,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _maybe_update_status(force=True) - # Cron work drains on its own deadline. ``timeout`` - # (``restart_drain_timeout``) defaults to 0 because interrupting a - # chat turn is announced and resumable; a cron run killed mid-flight - # is recorded in jobs.json as a permanent failure nobody is waiting - # on. Sharing one budget meant the default config could report - # ``timed_out=True`` after 0.00s with a cron job in flight and kill - # it — the drain never even entered this loop (#82161). + # Cron drains on its own deadline: ``timeout`` (``restart_drain_timeout``) defaults to 0 since + # an interrupted chat turn is announced and resumable, while a cron run killed mid-flight is a + # permanent failure nobody is waiting on. One shared budget would kill cron after 0.00s. loop = asyncio.get_running_loop() started = loop.time() deadline = started + timeout @@ -10792,9 +10092,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return True return bool(self._active_cron_job_count()) and now < cron_deadline - # Both budgets at 0 leave this loop unentered, which is the legacy "interrupt immediately" - # behaviour — expressed as an expired deadline rather than a special case, so the timed_out - # value below is always computed from real state instead of asserted up front. + # Both budgets at 0 leave this loop unentered ("interrupt immediately") as an expired deadline, + # not a special case, so timed_out below is always computed from real state. while _still_draining(): _maybe_update_status() await asyncio.sleep(0.1) @@ -10816,9 +10115,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Interrupted running agent for session %s during shutdown", session_key) except Exception as e: logger.debug("Failed interrupting agent during shutdown: %s", e) - # API-server / desk turns are adapter-owned and never enter - # _running_agents, so the loop above cannot see them even though - # _drain_active_agents() waited for them (#63529). + # API-server / desk turns are adapter-owned and never enter _running_agents, so the loop above + # cannot see them even though _drain_active_agents() waited for them. interrupted_api = self._interrupt_api_server_runs(reason) if interrupted_api: logger.debug("Interrupted %d api_server run(s) during shutdown", interrupted_api) @@ -10832,14 +10130,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _notify_interrupted_cron_jobs(self, job_ids) -> int: """Tell the owner of each just-interrupted cron job that its run died. - The cron worker cannot do this itself: its thread reaches ``_deliver_result`` after - ``_bounded_adapter_teardown`` has closed the transport, so the notice never leaves the - process and the run's only trace is a line in jobs.json nobody reads. Must therefore run - in the post-interrupt phase while adapters are still connected — the same window - ``_notify_active_sessions_of_shutdown`` uses, which is blind to cron work because cron runs - on the scheduler's own pool rather than ``self._running_agents``. Best-effort by - construction: every failure is swallowed so a wedged adapter can never extend shutdown. - Returns the number of notices sent. + The cron worker can't: its thread reaches ``_deliver_result`` after teardown closed the + transport. Must run post-interrupt while adapters are still connected (the window + ``_notify_active_sessions_of_shutdown`` uses, which is blind to cron work). Best-effort: every + failure is swallowed so a wedged adapter can't extend shutdown. Returns notices sent. """ if not job_ids: return 0 @@ -10857,11 +10151,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew job = get_job(job_id) if not job: continue - # deliver=local jobs — and deliver=origin jobs with no - # resolvable origin (#43014) — resolve to zero targets and - # must stay silent rather than fall back to a home channel. - # Interrupted notices are failure-category engine status, so - # they honor the job's failure_deliver override (NS-788). + # deliver=local jobs, and deliver=origin jobs with no resolvable origin, resolve to zero + # targets and must stay silent rather than fall back to a home channel. Interrupted + # notices are failure-category engine status, so they honor failure_deliver. targets = _resolve_delivery_targets(job, for_failure=True) except Exception as e: logger.debug("Cron interrupt targets unresolved for %s: %s", job_id, e) @@ -10924,9 +10216,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _notify_active_sessions_of_shutdown(self) -> None: """Send shutdown/restart notifications to active chats and home channels. - Called at the very start of stop() — adapters are still connected so - messages can be delivered. Best-effort: individual send failures are - logged and swallowed so they never block the shutdown sequence. + Called at the start of stop() while adapters are connected; send failures never block shutdown. """ active = self._snapshot_running_agents() restart_source = self._restart_command_source if self._restart_requested else None @@ -10972,9 +10262,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew chat_id = _parsed["chat_id"] thread_id = _parsed.get("thread_id") - # Deduplicate only identical delivery targets. Thread/topic-aware - # platforms can share a parent chat while still routing to distinct - # destinations via metadata. + # Dedupe only identical targets: thread/topic platforms share a parent chat yet route to + # distinct destinations via metadata. dedup_key = (platform_str, chat_id, str(thread_id) if thread_id else None) if dedup_key in notified: continue @@ -11038,19 +10327,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Skipping home-channel shutdown notifications for in-chat restart") return - # Suppress ONLY the home-channel broadcast when the drain that is ending - # in this shutdown asked us to be quiet (e.g. a NAS auto-update image - # migration — drain-gated, then the machine is recreated). On the - # always-on Hermes Cloud fleet that broadcast would otherwise fire on - # every routine auto-update, spamming home channels with operator- - # flavoured "gateway shutting down" pings the user doesn't care about. - # The per-active-session interrupt pings above are deliberately NOT - # gated: on a drained shutdown they're empty by construction, and in the - # force-interrupt (deadline-exceeded) case they carry the genuinely - # useful "your task was cut off, message me to resume" hint. The flag is - # only honoured for a CURRENT-epoch marker (drain_notification_suppressed - # reuses the NS-570 staleness check), so an orphaned marker can never - # silence a fresh gateway's legitimate broadcast. + # Suppress ONLY the home-channel broadcast when the drain asked to be quiet (e.g. routine + # auto-update on an always-on fleet). Per-session interrupt pings above are NOT gated: empty by + # construction on a drained shutdown, and useful ("task cut off, message me to resume") on a + # force-interrupt. Honoured only for a CURRENT-epoch marker (staleness check inside + # drain_notification_suppressed), so an orphaned marker can't silence a fresh gateway. try: from gateway.drain_control import drain_notification_suppressed if drain_notification_suppressed(): @@ -11064,10 +10345,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # fail toward the louder, more-visible behaviour. logger.debug("drain_notification_suppressed check failed: %s", e) - # Snapshot adapters up front: adapter.send() can hit a fatal error path that pops the - # adapter from self.adapters (see _handle_fatal elsewhere), which would otherwise trigger - # ``RuntimeError: dictionary changed size during iteration`` — observed in a user report - # during gateway shutdown. + # Snapshot adapters: adapter.send() can hit a fatal path (_handle_fatal) that pops the adapter + # from self.adapters -> ``RuntimeError: dictionary changed size during iteration``. for platform, adapter in list(self.adapters.items()): home = self.config.get_home_channel(platform) if not home or not home.chat_id: @@ -11121,36 +10400,28 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _finalize_shutdown_agents(self, active_agents: Dict[str, Any]) -> None: for agent in active_agents.values(): - # Persist any in-flight transcript to the SQLite session store before teardown. An agent - # force-interrupted by the drain-timeout escalation may never reach finalize_turn (the - # only mid-turn flush to state.db); its tool rounds live only in in-memory - # ``_session_messages``, so the pre-restart turn would vanish from load_transcript() on - # resume. The resume branches already expect a tail that may be a pending tool result. - # The flush is idempotent (identity-tracked in ``_flush_messages_to_session_db``), so - # agents that DID finish gracefully re-flush nothing. + # Persist in-flight transcripts before teardown: a force-interrupted agent may never reach + # finalize_turn (the only mid-turn flush), so its tool rounds would vanish from + # load_transcript() on resume (resume already tolerates a pending-tool-result tail). The + # flush is idempotent (identity-tracked); gracefully finished agents re-flush nothing. try: _flush = getattr(agent, "_flush_messages_to_session_db", None) _session_messages = getattr(agent, "_session_messages", None) if callable(_flush) and isinstance(_session_messages, list) and _session_messages: - # Strip private empty-response retry scaffolding from the tail first, mirroring - # the graceful ``_persist_session`` path, so a resumed turn doesn't replay - # synthetic recovery nudges. + # Strip empty-response retry scaffolding from the tail first (as ``_persist_session`` + # does) so a resumed turn doesn't replay synthetic recovery nudges. _strip = getattr( agent, "_drop_trailing_empty_response_scaffolding", None ) if callable(_strip): - try: + with suppress(Exception): _strip(_session_messages) - except Exception: - pass try: _flush(_session_messages) except Exception as _flush_err: - # The in-memory transcript could not be persisted (e.g. FTS/SQLite index - # corruption). A debug log alone loses the conversation for good at exit, so - # dump the live agent history to an external JSON recovery snapshot an - # operator can salvage after repairing state.db. Non-fatal: shutdown must - # never block on a best-effort backup. + # Transcript could not be persisted (e.g. FTS/SQLite index corruption). A log + # line alone loses the conversation at exit, so dump the live history to an + # external JSON recovery snapshot. Non-fatal: shutdown never blocks on a backup. logger.warning( "Shutdown transcript flush failed (%s); preserving " "%d in-memory message(s) to recovery snapshot", @@ -11164,9 +10435,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception as _e: logger.debug("Shutdown transcript flush failed: %s", _e) - # Off-loop + bounded: finalize_session fans out to plugin on_session_finalize hooks that - # can do arbitrary synchronous work (e.g. an observability plugin serializing a full- - # session trace export). Same class as the memory-provider hang below. + # Off-loop + bounded: plugin on_session_finalize hooks can do arbitrary synchronous work + # (e.g. a full-session trace export) — same hang class as the memory provider below. await self._finalize_session_off_loop( session_id=getattr(agent, "session_id", None), platform="gateway", @@ -11186,9 +10456,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> bool: """Only emit the heartbeat while this task still owns the live run. - Stop once the executor finishes, the agent is gone, or the session key was rebound to a - different live agent (e.g. ``/new`` mid-run) — otherwise a stale ``running: - delegate_task`` heartbeat outlives the run that started it. + Stop once the executor finishes, the agent is gone, or the session key was rebound (e.g. + ``/new`` mid-run) — else a stale ``running: delegate_task`` heartbeat outlives its run. """ if agent is None: return False @@ -11200,12 +10469,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return False return True - # Upper bound on off-loop agent-resource cleanup invoked from coroutines running on the - # gateway's event loop (session-expiry sweep, in-turn cache-hygiene re-eviction). - # _cleanup_agent_resources is synchronous and can block for a long time (agent.close() does - # subprocess teardown; shutdown_memory_provider() may do network/SQLite IO). Inline it wedges - # the loop — bot silent, status heartbeat frozen, SIGTERM unserviced — so it is offloaded to a - # worker thread under this timeout (mirrors the /new reset path). + # Bound for off-loop agent-resource cleanup from event-loop coroutines (expiry sweep, cache-hygiene + # re-eviction). _cleanup_agent_resources is synchronous and can block long (subprocess teardown, + # memory-provider network/SQLite IO); inline it wedges the loop, so it runs in a worker thread. _CLEANUP_TIMEOUT_S = 30.0 def _defer_agent_cleanup_until_future_done( @@ -11217,9 +10483,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Clean up ``agent`` only after its executor future has finished. - A timed-out executor call keeps running in its worker thread. Closing the agent before - that thread exits can tear down clients or providers it is still using, so keep a strong - task reference and wait for the real future before the normal bounded off-loop cleanup. + A timed-out executor call keeps running in its worker thread; closing the agent first can + tear down clients it still uses, so hold a strong task ref and await the real future. """ async def _cleanup_when_done() -> None: @@ -11247,9 +10512,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew tasks.add(task) task.add_done_callback(tasks.discard) - # Bounded budget for one finalize_session() dispatch (plugin on_session_finalize hooks + core - # Relay conversation close). Generous enough for a normal trace-export flush, small enough that - # a wedged plugin can never eat the systemd stop window. + # Budget for one finalize_session() dispatch (plugin on_session_finalize hooks + Relay close): + # enough for a normal trace-export flush, small enough a wedged plugin can't eat the stop window. _FINALIZE_TIMEOUT_S = 10.0 async def _finalize_session_off_loop( @@ -11262,9 +10526,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Run hermes_cli.lifecycle.finalize_session off the event loop, bounded. - Off-loop + ``wait_for`` keeps the loop live; on timeout the worker thread is left to - finish (or leak) on its own and the caller proceeds — mirroring - ``_cleanup_agent_resources_off_loop``. + On timeout the worker thread is left to finish (or leak) and the caller proceeds. """ def _call() -> None: @@ -11304,16 +10566,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Run _cleanup_agent_resources in a worker thread with a bounded wait. - On timeout the await is cancelled and the worker thread is left to finish (or leak) on - its own — the caller proceeds regardless, exactly as the /new reset path does. + On timeout the worker thread is left to finish (or leak) and the caller proceeds, as /new does. """ if agent is None: return if context.startswith("shutdown") or context == "session expiry": - try: + with suppress(Exception): agent._end_session_on_close = False - except Exception: - pass try: await asyncio.wait_for( self._run_in_executor_with_context( @@ -11342,23 +10601,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return try: if hasattr(agent, "shutdown_memory_provider"): - # Drain queued memory writes BEFORE tearing the provider down. The memory manager - # runs per-turn sync + end-of-session extraction on one serialized worker; - # shutdown_memory_provider() -> shutdown_all() gives it only a ~5s drain and cancels - # the rest, so a /reset or session rotation could drop writes already handed off and - # the next session loads stale memory. Give pending work a bounded head start via - # the manager's own barrier first (mirrors the CLI exit path). Best-effort: a flush - # failure must never block teardown. + # Drain queued memory writes BEFORE tearing the provider down: shutdown_all() gives + # the serialized memory worker only ~5s and cancels the rest, so a /reset or rotation + # could drop handed-off writes and the next session loads stale memory. Bounded head + # start via the manager's own barrier (mirrors CLI exit); a failure never blocks teardown. _mm = getattr(agent, "_memory_manager", None) if _mm is not None and hasattr(_mm, "flush_pending"): - try: + with suppress(Exception): _mm.flush_pending(timeout=10) - except Exception: - pass - # Pass the agent's own conversation transcript so memory providers' - # ``on_session_end`` hooks see the real messages instead of the empty default. - # ``_session_messages`` may be absent on agents built via ``object.__new__`` (test - # stubs), so getattr with a ``None`` default keeps the call signature-compatible. + # Pass the real transcript so ``on_session_end`` hooks don't see the empty default. + # ``_session_messages`` may be absent on ``object.__new__`` test stubs, hence getattr. session_messages = getattr(agent, "_session_messages", None) if isinstance(session_messages, list): agent.shutdown_memory_provider(session_messages) @@ -11366,17 +10618,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew agent.shutdown_memory_provider() except Exception: pass - # Close tool resources (terminal sandboxes, browser daemons, - # background processes, httpx clients) to prevent zombie - # process accumulation. + # Close tool resources (sandboxes, browser daemons, background processes, httpx clients). try: if hasattr(agent, "close"): agent.close() except Exception: pass - # Auxiliary async clients (session_search/web/vision/etc.) live in a process-global cache - # and are created inside worker threads. Clean up any entries whose event loop is now dead - # so their httpx transports do not accumulate across gateway turns. + # Auxiliary async clients live in a process-global cache created from worker threads; drop + # entries whose event loop is dead so httpx transports don't accumulate across turns. try: from agent.auxiliary_client import cleanup_stale_async_clients cleanup_stale_async_clients() @@ -11407,17 +10656,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Keep any entries that are still above 0 even if not active now # (they might become active again next restart) - try: + with suppress(Exception): atomic_json_write(path, new_counts, indent=None) - except Exception: - pass def _suspend_stuck_loop_sessions(self) -> int: - """Suspend sessions that have been active across too many restarts. + """Suspend sessions active across too many restarts (load → stuck → restart loop). - Returns the number of sessions suspended. Called on gateway startup - AFTER suspend_recently_active() to catch the stuck-loop pattern: - session loads → agent gets stuck → gateway restarts → repeat. + Runs at startup AFTER suspend_recently_active(). Returns the number suspended. """ import json @@ -11448,26 +10693,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew pass if suspended: - try: + with suppress(Exception): self.session_store._save() - except Exception: - pass # Clear the file — counters start fresh after suspension - try: + with suppress(Exception): path.unlink(missing_ok=True) - except Exception: - pass return suspended async def _clear_restart_failure_count(self, session_key: str) -> None: - """Clear the restart-failure counter for a session that completed OK. - - Called after a successful agent turn to signal the loop is broken. - Offloaded to a thread because the caller (_handle_message_with_agent) - runs on the event loop and atomic_json_write calls os.fsync. - """ + """Clear a completed session's restart-failure counter off-loop (atomic_json_write fsyncs).""" import json path = _hermes_home / self._STUCK_LOOP_FILE @@ -11559,15 +10795,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ).strip() from tools.environments.local import build_subprocess_env watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) - # This watcher is intentionally outside the running gateway. If it - # inherits the gateway marker, `hermes gateway restart` refuses to - # run as a self-restart loop guard and the gateway stays stopped. + # The watcher must not inherit the gateway marker, else `hermes gateway restart` refuses to + # run (self-restart loop guard) and the gateway stays stopped. watcher_env.pop("_HERMES_GATEWAY", None) project_root = Path(__file__).resolve().parent.parent - # The watcher runs sys.executable (console python) under the CREATE_NO_WINDOW detach - # kwargs below: it owns one hidden console, inherited by the `hermes gateway restart` - # child, so nothing flashes. Do NOT swap in GUI-subsystem pythonw.exe — a console-less - # watcher forces every console-subsystem descendant to allocate a visible conhost. + # Console python under CREATE_NO_WINDOW owns one hidden console inherited by the restart + # child, so nothing flashes. Do NOT swap in pythonw.exe — a console-less watcher forces + # every console-subsystem descendant to allocate a visible conhost. watcher_python = sys.executable venv_dir = Path(watcher_env.get("VIRTUAL_ENV") or project_root / "venv") site_packages = venv_dir / "Lib" / "site-packages" @@ -11585,16 +10819,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew str(restart_after_s), *cmd_argv, ] - # The watcher process must itself break away from any job object the - # parent CLI lives in (Electron/Tauri-wrapped Hermes Desktop, Windows - # Terminal, schtasks shells); otherwise it is reaped when the CLI - # exits and the gateway never respawns. windows_detach_popen_kwargs() - # carries CREATE_BREAKAWAY_FROM_JOB, but a restrictive job object - # (no JOB_OBJECT_LIMIT_BREAKAWAY_OK) rejects that bit with - # ERROR_ACCESS_DENIED, surfaced as OSError. Retry once without the - # breakaway bit, preserving argv and the scrubbed watcher_env. - # Mirrors the canonical fallback in - # hermes_cli/gateway_windows.py::_spawn_detached. + # The watcher must break away from any job object the parent CLI lives in (Desktop + # wrappers, Windows Terminal, schtasks), else it is reaped when the CLI exits and the + # gateway never respawns. windows_detach_popen_kwargs() sets CREATE_BREAKAWAY_FROM_JOB, + # but a job without JOB_OBJECT_LIMIT_BREAKAWAY_OK rejects it (ERROR_ACCESS_DENIED as + # OSError); retry once without the bit, preserving argv and the scrubbed watcher_env. try: subprocess.Popen( watcher_argv, @@ -11613,11 +10842,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew creationflags=windows_detach_flags_without_breakaway(), ) except OSError as exc: - # Both spawn attempts failed (a breakaway-denying job object is the common - # cause, but OSError covers others too). Record a minimal, path-safe diagnostic - # and return without crashing the caller: log only the interpreter basename and - # a numeric error code — never argv, env, watcher source, or str(exc) (which can - # carry a full interpreter path). + # Both spawns failed. Log only the interpreter basename and numeric errno — never + # argv, env, watcher source, or str(exc) (may carry a full path) — and return. winerror = getattr(exc, "winerror", None) error_code = winerror if winerror is not None else exc.errno error_field = "winerror" if winerror is not None else "errno" @@ -11637,10 +10863,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"while kill -0 {current_pid} 2>/dev/null && [ $(date +%s) -lt $deadline ]; do sleep 0.2; done; " f"{cmd} gateway restart" ) - # Same marker scrub as the Windows watcher above: this watcher runs `hermes gateway restart` - # from outside the gateway, but it inherits _HERMES_GATEWAY=1 from us, and the CLI's self- - # restart loop guard refuses to run when that marker is set — silently (DEVNULL), so the - # gateway stops and never comes back. + # Same marker scrub as the Windows watcher: an inherited _HERMES_GATEWAY=1 makes the CLI's + # self-restart loop guard refuse silently (DEVNULL), so the gateway stops and never comes back. from tools.environments.local import build_subprocess_env watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) watcher_env.pop("_HERMES_GATEWAY", None) @@ -11665,16 +10889,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _wedged_agent_count(self) -> int: """Count running chat agents already past the inactivity timeout. - A turn whose agent has recorded no activity (no API bytes, no tool progress) for longer - than ``agent.gateway_timeout`` is wedged — the same threshold at which the turn reaper - gives up on it. - - Returns 0 when the inactivity timeout is disabled (``gateway_timeout`` - 0/unset ⇒ the operator opted into unbounded turns; the after-turn cap - still bounds the wait). Cron/API-server work has no per-turn activity - clock and is never counted as wedged. Pending sentinels are brand-new - turns, never wedged. Fail-open per agent: an unreadable activity - summary means "not wedged". + No activity (API bytes, tool progress) for ``agent.gateway_timeout`` = wedged (the turn reaper's + threshold). Returns 0 when the timeout is disabled (the after-turn cap still bounds the wait). + Cron/API-server work has no activity clock and pending sentinels are brand-new, so neither + counts. Fail-open per agent: an unreadable activity summary means "not wedged". """ timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) if timeout <= 0: @@ -11704,20 +10922,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _await_active_work_before_restart(self) -> bool: """Wait for in-flight work to finish before entering ``stop()``. - In-band restart used to call ``stop()`` immediately, which folded the requesting turn - into the drain wait set and force-interrupted it at ``restart_drain_timeout``. Instead we - refuse new turns, wait here for active agents/cron/api work to reach zero, then let - ``stop()`` run against an idle gateway (drain is instant). - - Turns already past the inactivity timeout are excluded from the wait - (``_wedged_agent_count``): restart is usually the *remedy* for a wedged turn, so - deferring it behind one inverts the point of the graceful path. ``stop()``'s drain - interrupts them under ``restart_drain_timeout`` instead. - - Returns True when work drained to zero, False when the safety cap - elapsed with work still active — or when only wedged work remains — - (caller proceeds to ``stop()``, which may then interrupt remaining - runs under ``restart_drain_timeout``). + Calling ``stop()`` immediately would fold the requesting turn into the drain set and + force-interrupt it at ``restart_drain_timeout``; instead refuse new turns, wait for active + agents/cron/api work to reach zero, then ``stop()`` an idle gateway. Wedged turns + (``_wedged_agent_count``) are excluded — restart is the remedy, so ``stop()``'s drain + interrupts them. Returns True when drained to zero, False when the safety cap elapsed or + only wedged work remains (caller proceeds to ``stop()``). """ active = self._active_work_count() if active <= 0: @@ -11749,10 +10959,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew active, timeout, ) - try: + with suppress(Exception): self._update_runtime_status("draining") - except Exception: - pass loop = asyncio.get_running_loop() deadline = loop.time() + timeout @@ -11776,10 +10984,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._wedged_agent_count(), deadline - now, ) - try: + with suppress(Exception): self._update_runtime_status("draining") - except Exception: - pass last_status_at = now await asyncio.sleep(0.1) @@ -11804,16 +11010,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._restart_detached = detached self._restart_via_service = via_service self._restart_task_started = True - # Refuse new turns immediately while in-flight work finishes. - # Keep ``_running`` True so adapters stay connected and the active - # turn can still deliver its final response (#77184). + # Refuse new turns while in-flight work finishes. Keep ``_running`` True so adapters stay + # connected and the active turn can still deliver its final response. self._draining = True async def _run_restart() -> None: await self._await_active_work_before_restart() - # Launch the detached helper only AFTER the after-turn wait. Its deadline is - # drain_timeout+5 and covers stop() teardown — launching earlier would fire `hermes - # gateway restart` while the requesting turn was still running. + # Launch the detached helper only AFTER the after-turn wait: its drain_timeout+5 deadline + # covers stop() teardown; earlier it would fire the restart mid-turn. if detached: try: await self._launch_detached_restart_command() @@ -11822,19 +11026,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await asyncio.sleep(0.05) await self.stop(restart=True, detached_restart=detached, service_restart=via_service) - # _run_restart is a short-lived self-terminating task (calls stop() then returns). Do NOT - # add it to _background_tasks: _stop_impl cancels every entry there, which would cancel - # _run_restart while it awaits _stop_task and propagate CancelledError into _stop_impl, - # preventing _shutdown_event.set() / _exit_code = 75. Keep a strong ref in - # self._restart_task (bare create_task holds only a weak ref, so a pending task can be - # GC'd mid-flight); _stop_impl's cancel loop skips it, as it skips _stop_task. + # Do NOT add _run_restart to _background_tasks: _stop_impl cancels every entry there, which + # would cancel it while awaiting _stop_task and propagate CancelledError into _stop_impl, + # skipping _shutdown_event.set() / _exit_code = 75. Keep a strong ref in self._restart_task. self._restart_task = asyncio.create_task(_run_restart()) return True - # Drain-timeout reasons set by _stop_impl() when a still-running turn is force-interrupted; - # "restart_interrupted" is set by SessionStore.suspend_recently_active() on crash recovery (no - # .clean_shutdown marker). All mean "the agent was mid-turn and we killed it" — eligible for - # startup auto-resume. + # Reasons set by _stop_impl() on force-interrupt; "restart_interrupted" by suspend_recently_active() + # on crash recovery (no .clean_shutdown marker). All mean "killed mid-turn" -> startup auto-resume. _AUTO_RESUME_REASONS = frozenset( {"restart_timeout", "shutdown_timeout", "restart_interrupted"} ) @@ -11847,9 +11046,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Dispatch one synthetic startup resume and wait for its agent turn. - Startup restore needs a stronger boundary: inbound messages must stay queued until the - resumed agent turn itself has finished, otherwise a user message can race the restore - turn immediately after ``handle_message`` returns. + Inbound messages stay queued until the resumed turn finishes, else a user message can race it. """ try: await adapter.handle_message(event) @@ -11858,9 +11055,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if task is not None: await asyncio.shield(task) finally: - # _schedule_resume_pending_sessions pre-claims the runner slot before spawning this - # task. If adapter.handle_message raises before _handle_message takes ownership, release - # that pre-claim; otherwise the real run's normal cleanup owns the slot. + # The runner slot was pre-claimed before this task spawned; release it if handle_message + # raises before _handle_message takes ownership, else the real run's cleanup owns it. _pre_state = self._peek_session_state(session_key) if (_pre_state.turn.agent if _pre_state else None) is _AGENT_PENDING_SENTINEL: self._release_running_agent_state(session_key) @@ -11899,10 +11095,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew continue # Mark this replay so _handle_message does not queue it again while # the restore gate remains closed for any fresh inbound arrivals. - try: + with suppress(Exception): setattr(event, "_hermes_startup_restore_replay", True) - except Exception: - pass await adapter.handle_message(event) drained += 1 return drained @@ -11910,9 +11104,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _start_startup_warmup(self) -> None: """Kick off the boot turn-machinery warm-up in the background. - Called from ``start()`` right after the startup-restore gate closes, so the warm-up - overlaps the (slow, network-bound) platform connects instead of adding boot latency. - ``_finish_startup_restore`` awaits it (bounded) before opening the inbound gate. + Called from ``start()`` right after the startup-restore gate closes so the warm-up overlaps + the network-bound platform connects; ``_finish_startup_restore`` awaits it (bounded). """ timeout = _startup_warmup_timeout_secs() if timeout <= 0: @@ -11923,12 +11116,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) async def _warm_turn_prerequisites(self) -> None: - """Initialize turn machinery off-loop before the gate opens. + """Initialize turn machinery on an executor thread before the gate opens. - Runs ``_warm_turn_machinery_sync`` (run_agent import graph, tool schemas + check_fn probe - cache, context-file tier) on an executor thread so the event loop — platform heartbeats, - connects — stays responsive. Never raises: a failed warm-up degrades to the historical - lazy init; it must not block startup. + Never raises: a failed warm-up degrades to lazy init and must not block startup. """ try: loop = asyncio.get_running_loop() @@ -11950,8 +11140,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Bounded wait for the boot warm-up before the inbound gate opens. On timeout the gate opens anyway (availability outranks prompt completeness for a WEDGED - init — same principle as the bounded restore-drain wait above) and the warm-up continues - in the background; a late failure is still logged. + init) and the warm-up continues in the background; a late failure is still logged. """ task = getattr(self, "_startup_warmup_task", None) if task is None or task.done(): @@ -11979,21 +11168,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _finish_startup_restore(self) -> None: """Wait (BOUNDED) for startup auto-resume, then release + drain inbound. - The wait is bounded by ``_startup_restore_drain_timeout_secs`` so that a single - pathologically long boot-resume turn cannot hold the inbound gate shut for every channel. - On timeout the gate is released and the resume turn(s) finish in the background — NOT - cancelled. Safe because duplicate-agent protection does not depend on the wait: - ``_schedule_resume_pending_sessions`` claims each session's ``_running_agents`` slot - SYNCHRONOUSLY before this gate runs, so a drained inbound message queues behind that slot - instead of spawning a second agent. + Bounded by ``_startup_restore_drain_timeout_secs`` so one pathological boot-resume turn + cannot hold the gate shut for every channel; on timeout the gate opens and resume turns + finish in the background (NOT cancelled). Safe because ``_schedule_resume_pending_sessions`` + claims each ``_running_agents`` slot SYNCHRONOUSLY first, so drained inbound queues behind. """ tasks = list(getattr(self, "_startup_restore_tasks", []) or []) if tasks: timeout = _startup_restore_drain_timeout_secs() if timeout > 0: - # asyncio.wait (unlike wait_for / gather+timeout) does NOT - # cancel the pending tasks on timeout — the slow resume turn - # keeps running in the background instead of being killed. + # asyncio.wait (unlike wait_for / gather+timeout) does NOT cancel pending tasks on + # timeout — the slow resume turn keeps running in the background. done, pending = await asyncio.wait(tasks, timeout=timeout) if pending: logger.warning( @@ -12034,10 +11219,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @staticmethod def _log_background_resume_result(task: "asyncio.Task") -> None: - """Done-callback for a boot-resume turn that outlived the - startup-restore gate. Logs a late failure that would otherwise be - swallowed once the task is discarded from ``_background_tasks``. - Cancellation is expected (shutdown) and is not an error.""" + """Done-callback for a boot-resume turn that outlived the startup-restore gate.""" GatewayRunner._log_late_background_failure( task, "background startup auto-resume task failed after gate release", @@ -12048,10 +11230,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _log_late_background_failure( task: "asyncio.Task", message: str, *, level: int = logging.WARNING ) -> None: - """Shared done-callback body for boot-path tasks that outlive the - startup-restore gate: surface a late failure that would otherwise be - swallowed once the task is discarded from ``_background_tasks``. - Cancellation is expected (shutdown) and is not an error.""" + """Shared done-callback body for boot-path tasks that outlive the startup-restore gate: + surface a late failure otherwise swallowed once the task leaves ``_background_tasks``. + Cancellation (shutdown) is expected, not an error.""" if task.cancelled(): return exc = task.exception() @@ -12069,17 +11250,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Run boot-path sends without letting them pin the inbound restore gate. - ``_send_restart_notification`` and ``_redeliver_pending_obligations`` used to be awaited - inline *before* ``_finish_startup_restore`` released the gate; one Telegram flood-control - sleep froze inbound on every platform for the full ``retry_after``. Uses the same bounded - ``asyncio.wait`` as the resume gate: on timeout return and let the sends finish in the - background. Tasks are not cancelled. - - The ledger claim + ``resume_pending`` clear happen INLINE here, before the send task - exists: they are pure bounded DB work, and deferring them into the send task let a hung - restart notification expire the gate with zero rows claimed — the resume scheduler then - replayed turns already answered in the ledger and the background task redelivered them - too (duplicate delivery + re-paid turn). + Awaiting ``_send_restart_notification`` / ``_redeliver_pending_obligations`` inline before + the gate releases lets one Telegram flood-control sleep freeze inbound on every platform. + Same bounded ``asyncio.wait`` as the resume gate: on timeout return and let the sends finish + in the background (not cancelled). The ledger claim + ``resume_pending`` clear run INLINE + before the send task exists: bounded DB work, and deferring it let a hung notification + expire the gate with zero rows claimed, so answered turns were replayed AND redelivered. """ claimed = await self._claim_pending_obligations() @@ -12128,9 +11304,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> list: """Clear resume flags and return rows safe to redeliver. - Startup recovery preserves its historical best-effort behavior. Runtime reconnect - recovery is stricter: if the session-store write fails, the corresponding response must - not be sent because the same agent turn could otherwise be resumed immediately afterward. + Startup recovery stays best-effort. Runtime reconnect recovery is stricter: if the + session-store write fails the response must not be sent, or the turn could be resumed too. """ sendable = [] for row in claimed: @@ -12154,14 +11329,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _claim_pending_obligations(self) -> list: """Claim recoverable delivery-ledger rows and clear their ``resume_pending`` flags. - Pure DB work — no network sends. Must run INLINE at startup BEFORE - ``_schedule_resume_pending_sessions`` and before the abandonable boot-send task exists: - these sessions already produced their answer (only delivery is owed), so clearing - ``resume_pending`` here stops the resume path from re-running (and re-paying for) the - turn, however long the sends ahead of redelivery take. Crash-ambiguity contract (see - gateway/delivery_ledger.py): rows that were mid-send or previously rejected carry a - visible recovered-reply marker so a possible duplicate is labeled, never silent. - Returns the claimed rows for redelivery. + Pure DB work, no sends. Must run INLINE at startup BEFORE ``_schedule_resume_pending_sessions`` + and before the abandonable boot-send task exists: these sessions already produced their + answer, so clearing ``resume_pending`` here stops the resume path from re-running (and + re-paying for) the turn however long the sends take. Rows that were mid-send or previously + rejected carry a visible recovered-reply marker so a possible duplicate is labeled, never + silent (gateway/delivery_ledger.py). Returns the claimed rows for redelivery. """ try: from gateway.delivery_ledger import ( @@ -12178,9 +11351,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _deliverable_targets = { (getattr(p, "value", str(p)), "default") for p in self.adapters } - # Legacy rows predate adapter_profile. They are unambiguous only in - # a non-multiplexed gateway; fail closed when multiple bot identities - # share the process. + # Legacy rows predate adapter_profile. They are unambiguous only in a non-multiplexed + # gateway; fail closed when multiple bot identities share the process. if not _profile_adapters: _deliverable_targets.update( (getattr(p, "value", str(p)), None) for p in self.adapters @@ -12202,19 +11374,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not claimed: return [] - # Clear resume_pending for EVERY claimed row up front, before any send. Claiming already - # spent one of the row's redelivery attempts — the answer is in the ledger, so the resume - # path must never re-run these turns. + # Clear resume_pending for EVERY claimed row before any send: claiming already spent one + # redelivery attempt and the answer is in the ledger, so the resume path must never re-run. await self._clear_resume_pending_for_claimed_obligations(claimed) return claimed async def _redeliver_claimed_obligations(self, claimed: list) -> int: - """Redeliver final responses for rows already claimed (and - resume-cleared) by :meth:`_claim_pending_obligations`. + """Redeliver final responses for rows claimed by :meth:`_claim_pending_obligations`. - Network half of the split — runs inside the bounded boot-send task, - so a flood-limited send can be abandoned by the restore gate without - reopening the turn-replay window. Returns redeliveries attempted. + Network half of the split: runs inside the bounded boot-send task, so a flood-limited send + can be abandoned by the restore gate without reopening the turn-replay window. Returns count. """ if not claimed: return 0 @@ -12306,11 +11475,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return redelivered async def _redeliver_pending_obligations(self) -> int: - """Claim + redeliver in one call — composition of :meth:`_claim_pending_obligations` and - :meth:`_redeliver_claimed_obligations`. - - Kept as the stable public shape (tests and any external callers drive this name); the - startup path calls the two halves separately so the DB half can run inline before the + """Claim + redeliver in one call (:meth:`_claim_pending_obligations` then + :meth:`_redeliver_claimed_obligations`). Stable public shape for tests/external callers; + the startup path calls the halves separately so the DB half runs inline before the abandonable send task. """ return await self._redeliver_claimed_obligations( @@ -12325,10 +11492,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> int: """Replay one adapter identity's transient failures after reconnect. - The startup sweep cannot claim live-owner rows by design. A platform adapter can - reconnect without the gateway process exiting, however, so ``send_path_degraded`` - responses otherwise remain failed until the next process restart. Claim/clear/send stay - best-effort and reuse the startup redelivery path's attempt and ambiguity contract. + The startup sweep cannot claim live-owner rows, and an adapter can reconnect without the + process exiting, so ``send_path_degraded`` responses would otherwise stay failed until the + next restart. Claim/clear/send are best-effort and reuse the startup redelivery contract. """ try: from gateway.delivery_ledger import ( @@ -12380,15 +11546,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _schedule_resume_pending_sessions(self, platform=None) -> int: """Auto-continue fresh restart-interrupted sessions after startup. - ``resume_pending`` preserves the transcript and ``_is_resume_pending`` in - ``_handle_message_with_agent`` injects the recovery note on the next turn. This method - closes the UX gap by synthesizing that next turn once adapters are back online — the - event text is empty so the existing injection path owns the wording and we never - double up. Sessions whose adapter is not in ``self.adapters`` are skipped silently - (they stay ``resume_pending``; the reconnect watcher calls this again scoped to that - ``platform``). ``platform`` restricts the pass so a reconnecting platform never - re-touches another platform's in-flight recoveries; sessions whose agent is already - running are skipped, so a startup-scheduled session is never resumed twice. + Synthesizes the next turn once adapters are back online; the event text is empty so the + existing ``_is_resume_pending`` injection path owns the recovery wording. Sessions whose + adapter is not in ``self.adapters`` stay ``resume_pending`` for the reconnect watcher, which + re-calls this scoped to that ``platform`` (a reconnecting platform never touches another's + recoveries); sessions with a running agent are skipped so none is resumed twice. """ window = _auto_continue_freshness_window() try: @@ -12462,18 +11624,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) continue - # Claim the session slot *before* spawning the task so that an inbound message arriving + # Claim the session slot *before* spawning the task so an inbound message arriving # between task creation and the task's first await (where _process_message_background - # sets the real sentinel) sees the slot as occupied and queues behind it instead of - # spinning up a duplicate AIAgent. + # sets the real sentinel) sees the slot occupied and queues, not a duplicate AIAgent. _resume_state = self._session_state(entry.session_key) _resume_state.turn.agent = _AGENT_PENDING_SENTINEL _resume_state.turn.started_ts = time.time() self._persist_active_agents() - # Empty-text internal event — the _is_resume_pending branch in - # _handle_message_with_agent prepends the proper reason-aware - # system note before the turn runs. + # Empty-text internal event: the _is_resume_pending branch in _handle_message_with_agent + # prepends the reason-aware system note before the turn runs. event = MessageEvent( text="", message_type=MessageType.TEXT, @@ -12535,8 +11695,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _start_loop_liveness_guards(self, loop: asyncio.AbstractEventLoop) -> None: """Arm the selector floor and out-of-loop watchdog before adapters. - Disabled entirely with ``gateway.loop_watchdog: false`` in config.yaml - (no env override — config-only knob, #69089). + Disabled entirely with ``gateway.loop_watchdog: false`` in config.yaml (config-only knob). """ config = getattr(self, "config", None) if config is not None and not getattr(config, "loop_watchdog", True): @@ -12550,9 +11709,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew watchdog = getattr(self, "_loop_liveness_watchdog", None) if watchdog is None or not watchdog.is_alive(): try: - # getattr defaults cover the config=None / bare-object test - # path; config-loaded values are already validated+clamped - # by GatewayConfig.from_dict, so no re-clamping here. + # getattr defaults cover the config=None / bare-object test path; config-loaded + # values are already validated+clamped by GatewayConfig.from_dict; no re-clamping. interval = getattr( config, "loop_watchdog_probe_interval_s", @@ -12595,9 +11753,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: logger.debug("Failed to cancel gateway loop floor timer", exc_info=True) - # Also disarm the heartbeat writer task itself (ported from #95808): once shutdown starts - # loading the loop, a heartbeat that keeps refreshing the file can make a draining gateway - # look healthy to external probes. + # Also disarm the heartbeat writer task: once shutdown starts loading the loop, a heartbeat + # that keeps refreshing the file makes a draining gateway look healthy to external probes. heartbeat = getattr(self, "_loop_heartbeat_task", None) self._loop_heartbeat_task = None if heartbeat is not None: @@ -12609,9 +11766,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _consume_clean_shutdown_marker(self, marker_path) -> int: """Discard orphan turn markers before consuming a clean-exit receipt. - If either persistence or marker removal fails, startup must fail closed. - Continuing with the old receipt would let a later unclean exit masquerade - as clean and discard genuinely interrupted turns. + If persistence or marker removal fails, startup must fail closed: continuing with the old + receipt would let a later unclean exit masquerade as clean and discard interrupted turns. """ discarded = await self.async_session_store.discard_active_turn_markers() marker_path.unlink() @@ -12677,9 +11833,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _start_loop_heartbeat_task(self) -> None: """Start the loop-liveness heartbeat task, idempotent. - An asyncio task so a frozen loop stops refreshing ``state/gateway.heartbeat``. Cancelled - with the other background tasks during stop(). Best-effort — a liveness probe must never - be able to abort startup. + An asyncio task so a frozen loop stops refreshing ``state/gateway.heartbeat``; cancelled + with the other background tasks in stop(). Best-effort — must never abort startup. """ try: _existing_hb = getattr(self, "_loop_heartbeat_task", None) @@ -12725,10 +11880,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew faulthandler.enable(file=_fh_enable_file, all_threads=True) except Exception: logger.debug("faulthandler.enable() unavailable", exc_info=True) - # Also dump stacks to a rotating file for off-line analysis when the gateway is running - # under a service manager that doesn't capture stderr. faulthandler.register() and SIGUSR2 - # are POSIX-only; skip the signal-triggered file dump on Windows (faulthandler.enable() - # above still covers fatal-error dumps there). + # Also dump stacks to a rotating file for off-line analysis under a service manager that + # doesn't capture stderr. faulthandler.register()/SIGUSR2 are POSIX-only: skip the signal- + # triggered file dump on Windows (faulthandler.enable() above still covers fatal errors). _sigusr2 = getattr(signal, "SIGUSR2", None) if _sigusr2 is not None and hasattr(faulthandler, "register"): try: @@ -12754,14 +11908,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._gateway_loop = None if self._gateway_loop is not None: self._start_loop_liveness_guards(self._gateway_loop) - # The event loop is confirmed live: the startup-liveness - # watchdog's job is done and the loop-liveness watchdog (armed - # just above) takes over from here (OOF-298). Disarm even when - # the loop guards are config-disabled — the startup watchdog - # only covers the pre-loop window, never adapter connects or - # steady-state. Deliberately inside the loop-confirmed branch: - # if the loop somehow isn't live, startup has NOT reached the - # milestone and the watchdog must stay armed. + # Loop confirmed live: the startup-liveness watchdog is done and the loop-liveness + # watchdog (armed above) takes over. Disarm even when loop guards are config-disabled — + # the startup watchdog covers only the pre-loop window. Deliberately inside this branch: + # if the loop isn't live, startup has NOT reached the milestone and it must stay armed. try: from gateway.startup_watchdog import disarm_startup_watchdog @@ -12770,11 +11920,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Startup watchdog disarm failed", exc_info=True) logger.info("Session storage: %s", self.config.sessions_dir) - # Sanity-check that systemd's TimeoutStopSec covers our drain - # window. When the user upgraded hermes-agent without re-running - # ``hermes setup``, their unit file may still encode the old - # default — in which case SIGKILL hits mid-drain and looks like - # a phantom kill in the journal. Best-effort, never raises. + # Sanity-check that systemd's TimeoutStopSec covers our drain window: a unit file from + # before a hermes-agent upgrade (no ``hermes setup`` re-run) may encode the old default, + # so SIGKILL hits mid-drain and looks like a phantom kill in the journal. Never raises. try: from gateway.shutdown_forensics import check_systemd_timing_alignment _alignment = check_systemd_timing_alignment( @@ -12798,9 +11946,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception as _e: logger.debug("check_systemd_timing_alignment failed: %s", _e) - # Log the resolved max_iterations budget so operators can verify the - # config.yaml → env bridge did the right thing at a glance (instead - # of silently running at a stale .env value for weeks). + # Log the resolved max_iterations budget so operators can verify the config.yaml → env + # bridge at a glance (instead of silently running at a stale .env value for weeks). try: _effective_max_iter = int(os.getenv("HERMES_MAX_ITERATIONS", "500")) logger.info( @@ -12810,11 +11957,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception: pass - # Redaction status: ON by default (#17691). Surface a prominent - # warning if an operator has explicitly opted out so they don't - # forget the downgrade is active — the redactor snapshots its - # state at import time, so this log line is the source of truth - # for this process's lifetime. + # Redaction is ON by default; warn prominently when an operator has explicitly opted out so + # the downgrade isn't forgotten. The redactor snapshots its state at import time, so this + # log line is the source of truth for the process lifetime. try: _redact_raw = os.getenv("HERMES_REDACT_SECRETS", "true") _redact_on = _redact_raw.lower() in {"1", "true", "yes", "on"} @@ -12860,7 +12005,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Log any active supply-chain security advisories. Deliberately does NOT block startup or # surface inline to users — only the operator can act (uninstall, rotate credentials). - # See hermes_cli/security_advisories.py. try: from hermes_cli.security_advisories import ( detect_compromised, @@ -12917,9 +12061,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "QQ_ALLOW_ALL_USERS", "YUANBAO_ALLOW_ALL_USERS", ) - # Also pick up plugin-registered platforms — each entry can declare - # its own allowed_users_env / allow_all_env, so the warning stays - # accurate as plugins like IRC come online. + # Also pick up plugin-registered platforms — each entry can declare its own + # allowed_users_env / allow_all_env, so the warning stays accurate as plugins (IRC) arrive. _plugin_allowed_vars: tuple = () _plugin_allow_all_vars: tuple = () try: @@ -12964,11 +12107,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform_value, allow_all_env or "a platform allow-all flag", ) - try: - from gateway.status import write_runtime_status - write_runtime_status(gateway_state="startup_failed", exit_reason=reason) - except Exception: - pass + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) self._request_clean_exit(reason) return True @@ -12983,9 +12122,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "plugin discovery failed at gateway startup", exc_info=True, ) - # Register the generic relay adapter when a connector relay URL is configured - # (GATEWAY_RELAY_URL / gateway.relay_url). No URL -> no-op, so direct/single-tenant - # deployments are unaffected. + # Register the generic relay adapter only if GATEWAY_RELAY_URL / gateway.relay_url is set. + # No URL -> no-op, so direct/single-tenant deployments are unaffected. try: from gateway.relay import ( register_relay_adapter, @@ -13009,11 +12147,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "relay adapter registration failed at gateway startup", exc_info=True, ) - # Register declarative shell hooks from cli-config.yaml. Gateway has no TTY, so consent has - # to come from one of the three opt-in channels (--accept-hooks on launch, - # HERMES_ACCEPT_HOOKS env var, or hooks_auto_accept: true in config.yaml). We pass - # accept_hooks=False and let register_from_config resolve env + config itself. - # Failures are logged but must never block gateway startup. + # Register declarative shell hooks from cli-config.yaml. Gateway has no TTY, so consent must + # come from --accept-hooks, HERMES_ACCEPT_HOOKS, or hooks_auto_accept: true; pass + # accept_hooks=False and let register_from_config resolve env + config. Never blocks startup. try: from hermes_cli.config import load_config from agent.shell_hooks import register_from_config @@ -13042,11 +12178,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.warning("Process checkpoint recovery: %s", e) - # Recover sessions that were active when the gateway last exited. Exact durable turn markers - # cover long-running work; the 120-second recency heuristic remains as an upgrade fallback - # for turns started by older Hermes versions that did not write exact markers. - # SKIP suspension after a clean (graceful) shutdown — the previous process already drained - # active agents, so this must not auto-reset sessions after update/restart. + # Recover sessions active when the gateway last exited. Exact durable turn markers cover + # long-running work; the 120s recency heuristic remains as a fallback for turns from older + # versions without markers. SKIP after a clean shutdown — the previous process already drained. _clean_marker = _hermes_home / ".clean_shutdown" if _clean_marker.exists(): logger.info("Previous gateway exited cleanly — skipping session suspension") @@ -13076,9 +12210,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew fallback, ) - # Stuck-loop detection: if a session has been active across 3+ consecutive restarts, it's - # probably stuck in a loop (the same history keeps causing the agent to hang). Auto-suspend - # it so the user gets a clean slate on the next message. + # Stuck-loop detection: a session active across 3+ consecutive restarts is probably looping + # (its history keeps hanging the agent); auto-suspend so the next message starts clean. try: stuck = self._suspend_stuck_loop_sessions() if stuck: @@ -13086,18 +12219,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.debug("Stuck-loop detection failed: %s", e) - # Serialize startup restore against inbound dispatch. Platform adapters can begin receiving - # messages as soon as they connect, but restart-interrupted sessions are not auto-resumed - # until all startup wiring below completes. Inbound messages queue until the resume pass - # runs and every synthetic resume turn has finished. + # Serialize startup restore against inbound dispatch: adapters can receive messages as soon + # as they connect, but restart-interrupted sessions are not auto-resumed until all startup + # wiring below completes, so inbound queues until every synthetic resume turn has finished. self._startup_restore_in_progress = True self._startup_restore_queue = [] self._startup_restore_tasks = [] - # Fresh-boot readiness: with no resume_pending sessions the gate above opens almost - # immediately, while the agent-side turn machinery (run_agent import graph, tool schemas, - # check_fn probes, context tier) is still cold — a message in that window was served with a - # skeleton system prompt. Warm NOW so the work overlaps the network-bound connects below; - # _finish_startup_restore awaits it (bounded) before opening the gate. + # Fresh-boot readiness: with no resume_pending sessions the gate opens almost immediately + # while the turn machinery is still cold, so a message in that window got a skeleton system + # prompt. Warm NOW to overlap the connects below; _finish_startup_restore awaits it (bounded). self._start_startup_warmup() connected_count = 0 @@ -13107,9 +12237,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _multiplex_on = bool(getattr(self.config, "multiplex_profiles", False)) _multiplex_skipped_platforms: list[Platform] = [] # Initialize and connect each configured platform. connect() calls run concurrently so one - # slow/failing platform (e.g. Telegram behind a dead proxy) cannot delay every other - # platform by a full timeout window; the cheap pre-filter (checks, adapter creation, - # handler wiring) stays serial. Per-platform timeouts/error handling are unchanged. + # slow/failing platform (e.g. Telegram behind a dead proxy) cannot delay the others by a + # full timeout window; the cheap serial pre-filter and per-platform timeouts are unchanged. _pending_connects = [] # (platform, platform_config, adapter) for platform, platform_config in self.config.platforms.items(): if await self._abort_startup_if_shutdown_requested(): @@ -13180,11 +12309,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return (p, adp, p_cfg, "ok" if ok else "failed", None) if _pending_connects: - # Abort-aware concurrent wait (parity with the serial loop's - # between-platforms abort check): a restart/shutdown requested - # while connects are in flight must cancel the still-pending - # connects — no later platform may finish connecting — clean up - # the ones that already completed, and abort startup. + # Abort-aware concurrent wait (parity with the serial loop's between-platforms check): a + # restart/shutdown requested mid-connect must cancel still-pending connects, clean up the + # ones already completed, and abort startup. _task_map: dict = {} for (p, c, a) in _pending_connects: _t = asyncio.ensure_future(_connect_one_startup(p, c, a)) @@ -13199,9 +12326,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _abort_mid_connect = True break if _abort_mid_connect: - # Cancel and fully settle the in-flight connects FIRST, so a - # completed adapter's disconnect cannot unblock a sibling's - # connect() before the sibling is cancelled. + # Cancel and fully settle the in-flight connects FIRST, so a completed adapter's + # disconnect cannot unblock a sibling's connect() before the sibling is cancelled. for _t in _pending_tasks: _t.cancel() await asyncio.gather(*_pending_tasks, return_exceptions=True) @@ -13251,9 +12377,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew continue if outcome == "exception": logger.error("\u2717 %s error: %s", platform.value, exc) - # Same defensive cleanup path for exceptions -- an adapter that - # raised mid-connect may still have a live aiohttp.ClientSession or - # child subprocess. + # Same defensive cleanup path for exceptions -- an adapter that raised mid-connect + # may still have a live aiohttp.ClientSession or child subprocess. await self._safe_adapter_disconnect(adapter, platform) self._update_platform_runtime_status( platform.value, platform_state="retrying", error_code=None, error_message=str(exc), @@ -13286,13 +12411,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # (aiohttp.ClientSession, poll tasks, bridge subprocesses) before giving up. await self._safe_adapter_disconnect(adapter, platform) if adapter.has_fatal_error: - # A live foreign holder of this bot token / identity is - # a single-writer ownership conflict, not a transient - # blip — even though ``_acquire_platform_lock`` emits it - # retryable so a MID-RUN reconnect can recover (#54167). - # At startup route it as non-retryable: with nothing - # connected the gateway exits 78 instead of sitting alive - # and deaf in the retry queue forever (#83183). + # A live foreign holder of this bot token is a single-writer ownership conflict, + # not a blip — ``_acquire_platform_lock`` emits it retryable only so a MID-RUN + # reconnect can recover. At startup route it non-retryable: with nothing connected + # the gateway exits 78 instead of sitting alive and deaf in the retry queue. _retryable = adapter.fatal_error_retryable and not ( is_global_startup_conflict(adapter.fatal_error_code) ) @@ -13345,11 +12467,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # fixes config.yaml rather than running a half-wired gateway. reason = str(e) logger.error("Gateway multiplexer config error: %s", reason) - try: - from gateway.status import write_runtime_status - write_runtime_status(gateway_state="startup_failed", exit_reason=reason) - except Exception: - pass + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE self._request_clean_exit(reason) self._startup_restore_in_progress = False @@ -13361,11 +12479,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # From this point onward every adapter retry is non-evicting. self._platform_lock_takeover_on_start = False - # A platform we skipped on the primary for a missing credential was - # supposed to be picked up by a secondary profile that owns the token. - # If none did, the platform is enabled in config.yaml yet silently - # unserved — surface it loudly so the operator sees a config problem - # instead of a quiet dead channel (#64674 follow-up). + # A platform skipped on the primary for a missing credential should have been picked up by + # a secondary profile owning the token. If none did, it is enabled in config.yaml yet + # silently unserved — surface it loudly instead of leaving a quiet dead channel. for _skipped in _multiplex_skipped_platforms: _served_by_secondary = any( _skipped in _profile_map @@ -13384,23 +12500,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if startup_nonretryable_errors and not startup_retryable_errors: reason = "; ".join(startup_nonretryable_errors) logger.error("Gateway hit a non-retryable startup conflict: %s", reason) - try: - from gateway.status import write_runtime_status - write_runtime_status(gateway_state="startup_failed", exit_reason=reason) - except Exception: - pass + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE self._request_clean_exit(reason) self._startup_restore_in_progress = False return True if startup_nonretryable_errors: - # Mixed failure mode (NS-609): some platforms are fatally misconfigured (e.g. - # WhatsApp enabled but never paired) while others hit merely transient errors (e.g. - # Telegram TimedOut). Exiting with GATEWAY_FATAL_CONFIG_EXIT_CODE is wrong here: - # supervisors honoring the exit-78 contract take the gateway PERMANENTLY down over - # a network blip, others crash-loop. Either way the retryable platforms never get - # their retry. Log the fatal side loudly, then fall through to the degraded/retry - # path: the reconnect watcher recovers the retryable ones; the rest stay fatal-parked. + # Mixed failure mode: some platforms fatally misconfigured (e.g. WhatsApp never + # paired), others merely transient (e.g. Telegram TimedOut). Exiting 78 here would + # let exit-78 supervisors take the gateway PERMANENTLY down over a network blip and + # deny the retryable ones their retry. Log the fatal side loudly, then fall through to + # the degraded/retry path: the watcher recovers the retryable; the rest stay parked. logger.error( "%d platform(s) fatally misconfigured and parked: %s. " "Staying alive so retryable platforms can recover.", @@ -13409,16 +12519,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) if enabled_platform_count > 0: if startup_retryable_errors: - # All enabled platforms hit retryable failures (network - # blip, bridge not paired, npm install timeout, etc.). - # Keep the gateway alive so: - # • cron jobs still run - # • the reconnect watcher gets a chance to recover the - # failing platforms once the underlying problem is - # fixed (e.g. user runs `hermes whatsapp`, fixes - # proxy, etc.) - # Exiting here used to convert a single misconfigured - # platform into an infinite systemd restart loop. + # All enabled platforms hit retryable failures (network blip, bridge not paired, + # npm install timeout...). Keep the gateway alive so cron jobs still run and the + # reconnect watcher can recover the platforms once the cause is fixed; exiting + # here would turn one misconfigured platform into an infinite systemd restart loop. reason = "; ".join(startup_retryable_errors) logger.warning( "Gateway started with no connected platforms — " @@ -13433,11 +12537,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception: pass - # Fall through to the normal "running" state — reconnect watcher takes it from - # here. - # All enabled platforms had no adapter (missing library or credentials). Fleet - # deployments share one config.yaml across nodes holding credentials for only a - # subset of platforms, so degrade gracefully and allow cron jobs to run. + # Fall through to the normal "running" state — reconnect watcher takes it from here. + # All enabled platforms had no adapter (missing library or credentials). Fleet nodes + # share one config.yaml but hold credentials for only a subset of platforms, so + # degrade gracefully and let cron jobs run. logger.warning( "No adapter could be created for any of the %d configured platform(s). " "Check that required dependencies are installed and credentials are set. " @@ -13478,7 +12581,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if hook_count: logger.info("%s hook(s) loaded", hook_count) await self.hooks.emit("gateway:startup", { - "platforms": [p.value for p in self.adapters.keys()], + "platforms": [p.value for p in self.adapters], }) if connected_count > 0: @@ -13505,9 +12608,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): self._schedule_update_notification_watch() - # Give freshly connected platform adapters a brief moment to settle - # before sending restart/startup lifecycle messages. In practice this - # helps Discord thread deliveries right after reconnect. + # Give freshly connected adapters a brief moment to settle before sending restart/startup + # lifecycle messages; in practice this helps Discord thread deliveries after reconnect. if connected_count > 0: await asyncio.sleep(1.0) @@ -13527,11 +12629,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew planned_restart_notification_pending=planned_restart_notification_pending, ) - # Automatically continue fresh sessions that were interrupted by the previous gateway - # restart/shutdown. The resume_pending flag is cleared by the normal successful-turn path, - # so a failed auto-resume remains visible for manual recovery on the next user message. - # Obligation redelivery already ran in _await_startup_boot_sends and cleared resume_pending - # for sessions whose answer is in the ledger — redelivering beats re-running the turn. + # Auto-continue fresh sessions interrupted by the previous restart/shutdown. resume_pending + # is cleared by the normal successful-turn path, so a failed auto-resume stays visible on the + # next user message. _await_startup_boot_sends already cleared sessions answered in the ledger. self._schedule_resume_pending_sessions() await self._finish_startup_restore() @@ -13566,17 +12666,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Keep the /model picker's remote catalogs (curated manifest, OpenRouter live list, Nous # Portal recommendations) warm on disk so a delisted or newly-published model reaches the - # picker within one TTL window (model_catalog.ttl_minutes, default 20) without waiting for a - # cold /model open to trigger the refresh. + # picker within one TTL window (model_catalog.ttl_minutes, default 20) without a cold open. self._spawn_supervised(self._model_catalog_refresh_watcher, "model_catalog_refresh_watcher") # Stall watchdog: pending inbound + stale agent activity → warn user # to /new (does not kill the turn; see agent.session_stall_timeout). self._spawn_supervised(self._session_stall_watcher, "session_stall_watcher") - # Start background kanban notifier — each gateway delivers events for - # subscriptions owned by the profiles whose adapters it hosts, even - # when another gateway owns the single dispatcher. + # Start the kanban notifier — each gateway delivers events for subscriptions owned by the + # profiles whose adapters it hosts, even when another gateway owns the single dispatcher. self._spawn_supervised(self._kanban_notifier_watcher, "kanban_notifier_watcher") # Start background kanban dispatcher — spawns workers for ready tasks. Gated by @@ -13591,22 +12689,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew len(self._failed_platforms), ", ".join(p.value for p in self._failed_platforms), ) - # Track the reconnect watcher task so _ensure_reconnect_watcher_running - # can detect if it dies and respawn it (#70344). Spawned via - # _spawn_supervised (not a bare asyncio.create_task) so an exception - # escaping the watcher's OUTER while-loop -- not just the per-platform - # inner try/except -- is caught, logged, and auto-restarted with - # backoff instead of silently killing the watcher forever. Without - # this, a platform already queued in _failed_platforms when the - # watcher dies stays stranded indefinitely: _ensure_reconnect_watcher_running() - # only gets called from a NEW fatal-error arrival, so if no other - # platform ever fails afterward, nothing ever notices the watcher is - # dead (#71758 -- reported as 17.5h of silent downtime for a platform - # whose transient upstream outage had long since recovered). - # ``on_spawn`` keeps ``_reconnect_watcher_task`` pointed at the CURRENT - # live task even when _spawn_supervised's own backoff respawns it — so - # _ensure_reconnect_watcher_running never mistakes a superseded handle - # for a dead watcher and spawns a duplicate. + # Track the reconnect watcher task so _ensure_reconnect_watcher_running can detect death + # and respawn it. Spawned via _spawn_supervised so an exception escaping the watcher's OUTER + # loop is caught, logged, and restarted with backoff instead of silently killing it (else a + # platform already queued in _failed_platforms stays stranded: the ensure hook only runs on + # a NEW fatal-error arrival). ``on_spawn`` keeps ``_reconnect_watcher_task`` on the CURRENT + # live task across backoff respawns so a superseded handle never looks like a dead watcher. self._spawn_reconnect_watcher() # Start background handoff watcher — picks up CLI sessions marked handoff_state='pending' in @@ -13614,22 +12702,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # synthetic user turn so the agent kicks off the new chat. self._spawn_supervised(self._handoff_watcher, "handoff_watcher") - # Start background async-delegation watcher — drains completion events - # from delegate_task(background=true) subagents and injects each - # result back into its originating session as a new turn, covering the - # idle case where the subagent finishes with no agent turn running. + # Async-delegation watcher: drains delegate_task(background=true) completions and injects + # each result into its originating session as a new turn (covers the idle, no-turn case). self._spawn_supervised(self._async_delegation_watcher, "async_delegation_watcher") - # Start background /loop wakeup watcher — scans persisted loops - # (SessionDB loop:* rows) and injects due wakeup prompts into their - # originating chats while the session is idle. + # /loop wakeup watcher: scans persisted loops (SessionDB loop:* rows) and injects due + # wakeup prompts into their originating chats while the session is idle. self._spawn_supervised(self._loop_wakeup_watcher, "loop_wakeup_watcher") - # Start the scale-to-zero idle watcher ONLY when this instance is opted in (the NAS "Labs" - # HERMES_SCALE_TO_ZERO stamp), messaging is relay-only/absent, and a wakeUrl is registered - # (decisions.md D1/D11/ §3.4(1)). Non-opted instances are unchanged. When armed it drives - # the relay dormant on sustained idle, then suspends the machine via the local flaps socket - # — Fly Proxy autostop is inbound-only and job-blind, so the gateway owns the decision. + # Start the scale-to-zero idle watcher ONLY when opted in (HERMES_SCALE_TO_ZERO stamp), + # messaging is relay-only/absent, and a wakeUrl is registered. When armed it drives the relay + # dormant on sustained idle, then suspends via flaps — Fly autostop is inbound-only, job-blind. try: if self._scale_to_zero_should_arm(): logger.info( @@ -13638,17 +12721,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) self._spawn_supervised(self._scale_to_zero_watcher, "scale_to_zero_watcher") else: - # Surface WHY an OPTED-IN instance didn't arm (a non-opted instance - # not arming is normal — stay silent there). Without this, a failed - # arm is invisible and "why won't it suspend/wake?" needs a box-dive. + # Surface WHY an OPTED-IN instance didn't arm (non-opted not arming is normal — + # stay silent); otherwise a failed arm is invisible and needs a box-dive. self._log_scale_to_zero_not_armed_reason() except Exception: # noqa: BLE001 - arming must never block startup logger.debug("scale-to-zero: arm check failed at startup", exc_info=True) - # Start background drain-control watcher — reconciles the gateway's new-turn accept-state - # with the external ``.drain_request.json`` marker the dashboard begin/cancel-drain endpoint - # writes (Phase 2). A marker left by a prior instantiation (durable-volume restart) is - # ignored via its instantiation epoch; only a current-epoch marker engages drain. + # Drain-control watcher: reconciles the gateway's new-turn accept-state with the external + # ``.drain_request.json`` marker the dashboard begin/cancel-drain endpoint writes. A marker + # from a prior instantiation (durable-volume restart) is ignored via its epoch. self._spawn_supervised(self._drain_control_watcher, "drain_control_watcher") logger.info("Press Ctrl+C to stop") @@ -13656,20 +12737,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return True _MAX_SUPERVISED_RESTARTS = 5 - # A task that ran at least this long before crashing is treated as having been HEALTHY — its - # crash is a fresh, isolated failure rather than part of a rapid crash-loop, so the consecutive- - # restart counter resets to 0. Only crashes within this window of a (re)spawn count toward - # _MAX_SUPERVISED_RESTARTS; otherwise a long-lived daemon whose watcher crashed a few times - # over days would hit the cap and be permanently abandoned (silent loss of reconnect/kanban). + # A task that ran at least this long before crashing is HEALTHY: an isolated crash, not a + # crash-loop; the consecutive-restart counter resets so a long-lived daemon isn't abandoned. _SUPERVISED_HEALTHY_SECS = 300 @staticmethod def _supervised_backoff(attempt: int) -> float: - """Delay before the supervisor's next respawn, in seconds. + """Delay before the supervisor's next respawn, in seconds (capped exponential). - Capped exponential. A method (not an inline expression) so the schedule has one name and - tests can collapse it — exhaustion tests assert crash/give-up/slow-tier ordering and would - take minutes sleeping through the real curve. + A method so tests can collapse the schedule instead of sleeping through the real curve. """ return min(60, 2 ** min(attempt, 6)) @@ -13679,22 +12755,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): """Launch a long-lived background task with task-level supervision. - Covers what a per-iteration try/except cannot: exceptions in the watcher's OUTER loop or - pre-try setup, and task-level death. A bare ``asyncio.create_task`` drops such an exception - on the floor — no log, no restart, the watcher silently gone. Restarts with capped backoff - up to ``_MAX_SUPERVISED_RESTARTS`` rapid failures; the counter resets after any run healthy - for ``_SUPERVISED_HEALTHY_SECS``. - - Each watcher starts in a fresh ``Context``: these are process-level services, and an - inherited delegated-child marker would make the Kanban dispatcher reject its own writes. - - ``on_spawn`` fires on EVERY spawn, including backoff respawns. Callers tracking the live - handle elsewhere (e.g. ``_reconnect_watcher_task``) MUST pass it, or a respawn leaves a - stale handle and ``_ensure_...`` spawns a SECOND concurrent watcher. - - ``on_give_up`` fires with ``name`` when the restart budget is spent and this task will - never be respawned. Supervision being finite is correct; having no owner of the invariant - afterwards is not — only the supervisor knows it is done. + Catches what a per-iteration try/except cannot — exceptions in the OUTER loop or pre-try + setup — which a bare ``asyncio.create_task`` drops silently. Restarts with capped backoff up + to ``_MAX_SUPERVISED_RESTARTS`` rapid failures; the counter resets after a run healthy for + ``_SUPERVISED_HEALTHY_SECS``. Each spawn uses a fresh ``Context``: an inherited + delegated-child marker would make the Kanban dispatcher reject its own writes. + ``on_spawn`` fires on EVERY spawn incl. respawns; callers tracking the handle elsewhere + (e.g. ``_reconnect_watcher_task``) MUST pass it or a respawn leaves a stale handle and a + SECOND watcher. ``on_give_up(name)`` fires when the restart budget is spent. """ if getattr(self, "_background_tasks", None) is None: self._background_tasks = set() @@ -13703,20 +12771,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # uses it to distinguish a rapid crash-loop from a healthy-run-then-crash. _started = time.monotonic() - # Deliberately do NOT pass kwargs to create_task — some test doubles mock it with a narrow - # signature. Calling it from a fresh Context has the same isolation semantics as - # create_task(..., context=Context()) while preserving that compatibility. + # Deliberately no kwargs to create_task (some test doubles mock a narrow signature); calling + # it from a fresh Context gives the same isolation as create_task(..., context=Context()). task = Context().run(lambda: asyncio.create_task(coro_factory())) - # Mark this as a PERMANENT supervised watcher, not transient background WORK: the - # scale-to-zero idle check must ignore process-lifetime watchers or the gateway would - # count itself busy forever and never go dormant. Transient tasks added to - # _background_tasks elsewhere (startup-resume events etc.) stay counted. + # PERMANENT supervised watcher, not transient background WORK: the scale-to-zero idle check + # must ignore process-lifetime watchers or the gateway counts itself busy forever. Transient + # tasks added to _background_tasks elsewhere (startup-resume events etc.) stay counted. task._hermes_supervised_watcher = True # type: ignore[attr-defined] self._background_tasks.add(task) if on_spawn is not None: - # Record the live handle NOW so an external tracker (e.g. - # _reconnect_watcher_task) always points at the current task, not a - # dead one left behind by a prior supervised respawn. + # Record the live handle NOW so an external tracker (e.g. _reconnect_watcher_task) + # points at the current task, not a dead one left by a prior supervised respawn. try: on_spawn(task) except Exception: # pragma: no cover - defensive; a tracker must never kill the spawn @@ -13728,17 +12793,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return exc = t.exception() if exc is None: - # Clean return == deliberate shutdown or a self-disabling watcher - # (e.g. a gated no-op that returns synchronously). Respawning here - # would busy-spin such a watcher — so NEVER restart on clean exit. + # Clean return == deliberate shutdown or a self-disabling watcher (e.g. a gated + # no-op returning at once); respawning would busy-spin it — NEVER restart on it. return logger.error("Supervised task %s died: %r", name, exc, exc_info=exc) if restart and self._running: ran_for = time.monotonic() - _started if ran_for >= self._SUPERVISED_HEALTHY_SECS: - # Ran healthily for a while before crashing — this is a FRESH failure, not part - # of a rapid crash-loop. Reset the consecutive counter so a daemon that crashes - # a handful of times over days is never permanently abandoned. + # Ran healthily before crashing — a FRESH failure, not a rapid crash-loop. Reset + # the counter so a daemon crashing a few times over days is never abandoned. effective_attempt = 0 else: effective_attempt = _attempt @@ -13770,17 +12833,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew restart=restart, _attempt=effective_attempt + 1, on_spawn=on_spawn, - # Must be threaded through the recursion for the - # same reason on_spawn is: the give-up that - # matters is the LAST respawn's, and a callback - # dropped here would leave the exhaustion branch - # with no owner at exactly the moment it needs one. + # Threaded through the recursion like on_spawn: only the LAST respawn's give-up + # matters, and dropping the callback leaves the exhaustion branch with no owner. on_give_up=on_give_up, ) - # The done callback retains the context in which it was - # registered, so isolate the backoff task too; otherwise a - # restart could reintroduce the original caller's turn scope. + # The done callback retains its registration context, so isolate the backoff task + # too; otherwise a restart could reintroduce the original caller's turn scope. respawn_task = Context().run(lambda: asyncio.create_task(_respawn())) self._background_tasks.add(respawn_task) respawn_task.add_done_callback(self._background_tasks.discard) @@ -13793,20 +12852,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Background task that processes pending CLI→gateway session handoffs. - Polls ``state.db`` for sessions in ``handoff_state='pending'`` and, for each one: - atomically claims it (pending → running); resolves the destination platform's home - channel; re-binds that channel's session_key to the CLI session_id via - ``session_store.switch_session`` so the full transcript replays; dispatches a synthetic - ``MessageEvent`` (``internal=True``) through the normal pipeline; marks the row - ``completed`` (or ``failed`` with ``handoff_error``). The CLI polls the terminal state. + Polls ``state.db`` for ``handoff_state='pending'`` rows: claim atomically (pending → + running), re-bind the home channel's session_key to the CLI session_id via + ``switch_session``, dispatch a synthetic ``MessageEvent``, mark ``completed``/``failed``. """ # Initial delay so the gateway is fully connected to its platforms # before we try to dispatch handoffs through them. await asyncio.sleep(5) - # Does this runner's _process_handoff accept the profile argument? - # The real one does; test stand-ins bind a one-parameter callable. - # Probed once, outside the loop. + # Does _process_handoff accept the profile argument? The real one does; test stand-ins bind + # a one-parameter callable. Probed once, outside the loop. try: import inspect as _inspect _process_takes_profile = len( @@ -13815,10 +12870,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: _process_takes_profile = False - # In-flight dispatches, keyed by session id. A handoff runs a FULL agent turn plus delivery - # (far longer than the CLI's 60s wait); processing them inline would make one slow handoff - # block every other profile's poll — a legitimate handoff could then time out purely because - # another profile was ahead of it. Dispatch is fire-and-forget; the poll loop only claims. + # In-flight dispatches keyed by session id. A handoff runs a FULL agent turn plus delivery + # (far longer than the CLI's 60s wait); inline processing would let one slow handoff block + # every other profile's poll and time them out. Fire-and-forget; the poll loop only claims. inflight: Dict[str, "asyncio.Task"] = {} async def _dispatch(row, session_id, session_db, profile_name) -> None: @@ -13848,13 +12902,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _tick(profile_name: Optional[str] = None) -> None: """One poll of the CURRENTLY-SCOPED session store. - Deliberately a closure over ``self`` rather than a method: the watcher's unit tests - bind ``_handoff_watcher`` onto a minimal ``SimpleNamespace`` stand-in that exposes - only ``_session_db``, ``_running`` and ``_process_handoff``. Any ``self.`` - call would raise AttributeError, be swallowed by the loop's ``except Exception``, and - silently turn the watcher into a no-op — touch only what the stand-in provides. - ``profile_name`` (``None`` = root) is threaded to ``_process_handoff`` so delivery - uses that profile's OWN adapter/home channel. + A closure over ``self``, not a method: unit tests bind ``_handoff_watcher`` onto a + ``SimpleNamespace`` exposing only ``_session_db``, ``_running`` and ``_process_handoff``; + any other ``self.`` would raise, be swallowed by the loop, and silently no-op the + watcher. ``profile_name`` (``None`` = root) makes delivery use that profile's OWN adapter. """ session_db = getattr(self, "_session_db", None) if session_db is None: @@ -13867,21 +12918,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not await session_db.claim_handoff(session_id): # Another tick or another gateway already claimed it. continue - # Positional, not keyword: the watcher's existing unit tests bind a stand-in - # ``_process_handoff(row)`` with no second parameter, and a keyword call would - # TypeError into the failure branch — turning a passing suite into a silent no-op - # watcher. Arity is probed above. + # Positional, not keyword: tests bind a one-arg ``_process_handoff(row)`` stand-in and a + # keyword call would TypeError into the failure branch (arity probed above). # INVARIANT (do not weaken): this task is created inside _profile_runtime_scope but - # typically RUNS after the scope exits; it sees the profile's home/secret scope only - # because those seams are ContextVar-based and ensure_future copies the Context. A - # thread-local/global there silently regresses secondary handoffs to primary delivery. + # typically RUNS after it exits; it sees the profile's home/secret scope only because + # those seams are ContextVar-based and ensure_future copies the Context. inflight[session_id] = asyncio.ensure_future( _dispatch(row, session_id, session_db, profile_name) ) - # A row still in 'running' at startup belongs to a gateway that died mid-dispatch. It can - # never reach a terminal state on its own, and request_handoff refuses new requests while it - # sits there — so the session would be permanently unable to hand off again. + # A row still 'running' at startup belongs to a gateway that died mid-dispatch: it can never + # reach a terminal state, and request_handoff refuses new requests while it sits there. for _pname, _phome in _handoff_watch_scopes(self): try: if _phome is None: @@ -13907,9 +12954,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Handoff watcher tick error: %s", exc, exc_info=True) await asyncio.sleep(interval) finally: - # Drain in-flight dispatches before returning. Cancelling them outright would strand - # their rows in 'running'; giving them a bounded grace period lets an almost-done - # handoff record its own terminal state. + # Drain in-flight dispatches before returning: cancelling would strand their rows in + # 'running'; a bounded grace period lets an almost-done handoff record its own state. pending_tasks = [t for t in inflight.values() if not t.done()] if pending_tasks: try: @@ -13923,16 +12969,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _process_handoff( self, row: Dict[str, Any], profile_name: Optional[str] = None, ) -> None: - """Execute one handoff row. + """Execute one handoff row. Raises on failure (caller marks failed). - Raises on failure (caller marks failed). ``profile_name`` (``None`` = root) is the profile - whose store queued this handoff. On a multiplexed gateway it is load-bearing for THREE - things that otherwise silently resolve to the primary profile and deliver through the - wrong bot: ``self.adapters`` holds only the DEFAULT profile's adapters (secondaries live - in ``self._profile_adapters[name]``); ``self.config`` is the primary's, so - ``get_home_channel()`` would return the primary's chat; and the session key must be - namespaced ``agent::...`` to match that profile's adapter or it binds a key - nobody reads. Passing the name beats re-deriving it from the contextvar. + ``profile_name`` (``None`` = root) is the profile whose store queued this handoff. Under + multiplex it is load-bearing: ``self.adapters``/``self.config`` are the primary's (secondaries + live in ``_profile_adapters``), and the session key must be namespaced ``agent::...`` + or it binds a key nobody reads. Passing the name beats re-deriving it from the contextvar. """ from gateway.config import Platform from gateway.session import SessionSource, build_session_key @@ -13949,9 +12991,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except (ValueError, KeyError): raise RuntimeError(f"unknown platform '{platform_name}'") - # Resolve the config + adapter map for the profile that queued this handoff. On a single- - # profile gateway (or a default-profile handoff) both fall back to - # self.config/self.adapters, so behaviour is byte-identical to before. + # Resolve the config + adapter map for the profile that queued this handoff; single-profile + # gateways (or a default-profile handoff) fall back to self.config/self.adapters. handoff_config = self.config handoff_adapters = self.adapters if profile_name and profile_name != "default": @@ -13961,11 +13002,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"profile '{profile_name}' has no live adapters in this gateway" ) handoff_adapters = secondary - # The watcher already entered _profile_runtime_scope for this profile, so a fresh load - # resolves that profile's config.yaml and .env (home channel, tokens) rather than the - # primary's. Fail closed on a load error: falling back to self.config would deliver - # through the right bot to the WRONG chat and report completed. A failed row the CLI - # can retry beats a wrong delivery. + # The watcher already entered _profile_runtime_scope, so a fresh load resolves THIS + # profile's config. Fail closed — self.config would deliver to the WRONG chat. try: handoff_config = load_gateway_config() except Exception as exc: @@ -13999,9 +13037,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew cli_title = row.get("title") or cli_session_id[:8] - # Try to create a fresh thread on the destination so the handoff has its own scrollback. - # Adapter returns None if threading isn't supported (Matrix/WhatsApp/Signal/SMS) or if - # creation failed (no permission, topics-mode off, parent is a DM, etc.). + # Create a fresh thread on the destination so the handoff has its own scrollback. Adapter + # returns None if threading is unsupported (Matrix/WhatsApp/Signal/SMS) or creation failed. thread_name = f"Hermes — {cli_title}" try: new_thread_id = await adapter.create_handoff_thread( @@ -14014,18 +13051,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) new_thread_id = None - # Use the new thread if the adapter created one; otherwise fall - # back to whatever thread (if any) the home channel was configured - # with. effective_thread_id = new_thread_id or ( str(home.thread_id) if home.thread_id else None ) - # Determine chat_type/user_id for the destination source. Telegram private-chat DM topics - # are shaped differently from group/forum threads by the inbound adapter, so a handoff- - # created topic in a positive chat_id must use the same DM-topic source shape as the - # user's next real message — otherwise the synthetic turn binds a `thread` key while - # real replies arrive on a `dm` key. + # Telegram private-chat DM topics are shaped differently from group/forum threads by the + # inbound adapter: a handoff-created topic in a positive chat_id must use the DM-topic source + # shape, or the synthetic turn binds a `thread` key while real replies arrive on a `dm` key. home_chat_id = str(home.chat_id) is_telegram_private_chat = ( platform == Platform.TELEGRAM @@ -14036,17 +13068,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew dest_chat_type = "thread" dest_user_id = "system:handoff" else: - # No thread — assume DM-style for the home channel. For Telegram private-chat topics, - # use the real user id (same as chat_id) so topic-mode checks and binding persistence - # see the same identity as subsequent inbound user messages. + # No thread — assume DM-style. For Telegram private-chat topics use the real user id + # (== chat_id) so topic-mode checks and binding persistence match later inbound turns. dest_chat_type = "dm" dest_user_id = home_chat_id if is_telegram_private_chat else "system:handoff" - # Discord thread destinations must key on the thread's OWN id, not the parent channel's, - # because the Discord adapter builds organic in-thread messages with ``chat_id == thread - # id`` — so ``build_session_key`` yields ``…:thread:{thread}:{thread}``; keying on the - # parent would make the next real reply spawn a fresh session. Discord-specific: Slack and - # Telegram key organic thread messages with ``chat_id == parent_channel``. + # Discord (unlike Slack/Telegram) builds in-thread messages with ``chat_id == thread id``, + # so key on the thread's OWN id; keying on the parent would make the next reply spawn anew. if platform == Platform.DISCORD and dest_chat_type == "thread" and effective_thread_id: dest_chat_id = str(effective_thread_id) else: @@ -14062,17 +13090,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew profile=profile_name, ) - # Compute the gateway's session_key for that destination using the same rules its adapters - # use, so switch_session targets the right entry. Thread destinations key without user_id - # (thread_sessions_per_user defaults False) so the next real message shares this session. + # Build the session_key with the adapters' own rules so switch_session hits the right entry. + # Thread keys omit user_id (thread_sessions_per_user default) so the next message shares it. platform_cfg = handoff_config.platforms.get(platform) extra = platform_cfg.extra if platform_cfg else {} - # Namespace the key to the profile that queued this handoff. Without it, a multiplexed - # gateway builds ``agent:main:...`` here while the profile's own adapter routes real inbound - # messages on ``agent::...`` — the handoff would bind a key nobody reads. The - # resolver is only the fallback for the root case (None when multiplexing is off, keeping - # the old key byte-for-byte). The isinstance check below is load-bearing: a Mock session - # store returns a truthy MagicMock that would be interpolated straight into the key. + # Namespace the key to the queuing profile: a multiplexed gateway would otherwise build + # ``agent:main:...`` while the profile's adapter routes inbound on ``agent::...``. + # The resolver is only the root fallback (None when multiplexing is off; old key unchanged). + # The isinstance check is load-bearing: a Mock store returns a truthy MagicMock. handoff_profile = profile_name if (profile_name and profile_name != "default") else None if handoff_profile is None: try: @@ -14091,14 +13116,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew profile=handoff_profile, ) - # Make sure there's an entry in the session_store for this key. If - # the home channel has never been used, get_or_create_session - # creates one; switch_session then re-points it. + # Ensure a session_store entry exists for this key (get_or_create_session creates one for a + # never-used home channel); switch_session then re-points it. await self.async_session_store.get_or_create_session(dest_source) - # Re-bind the destination key to the CLI session_id. switch_session ends the prior session - # in SQLite and reopens the CLI session under the new key. The CLI's transcript becomes the - # active one for the gateway from this moment on. + # Re-bind the destination key to the CLI session_id: switch_session ends the prior session + # in SQLite and reopens the CLI session under the new key; its transcript is now active. switched = await self.async_session_store.switch_session(session_key, cli_session_id) if switched is None: raise RuntimeError( @@ -14133,19 +13156,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew session_key, ) - # Dispatch through the runner directly. Going through adapter.handle_message would spawn a - # background task and we'd lose synchronous error visibility; calling _handle_message inline - # keeps the success/failure path observable for the watcher. + # Dispatch through the runner directly: adapter.handle_message would spawn a background task + # and lose error visibility; inline _handle_message keeps success/failure observable. response_text = await self._handle_message(synthetic_event) if not response_text: # Streaming may have already delivered the response inline. # Either way, agent ran without raising — count as success. return - # Send the agent's reply to the destination. Route to the new thread if we created one; - # otherwise the configured home channel (which may itself carry a thread_id). Send via the - # resolved transport (not adapter.send) so a relay-fronted logical platform is stamped on - # the outbound frame (send_for_platform). + # Send the reply to the new thread if we created one, else the configured home channel + # (which may carry a thread_id). Use the resolved transport (not adapter.send) so a + # relay-fronted logical platform is stamped on the outbound frame (send_for_platform). send_metadata: Dict[str, Any] = {} if effective_thread_id: send_metadata["thread_id"] = effective_thread_id @@ -14164,11 +13185,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew raise RuntimeError(f"adapter.send failed: {err}") async def _session_expiry_watcher(self, interval: int = 300): - """Background task that finalizes expired sessions. - - For each session whose reset policy has expired, invokes ``on_session_finalize`` hooks, - cleans up the cached AIAgent's tool resources, evicts the cache entry so it can be - garbage-collected, and marks the session so it won't be finalized again. + """Background task that finalizes expired sessions: runs ``on_session_finalize`` hooks, + cleans up the cached agent's tool resources, evicts the cache entry, and marks the session + finalized so it is not finalized again. """ await asyncio.sleep(60) # initial delay — let the gateway fully start _finalize_failures: dict[str, int] = {} # session_id -> consecutive failure count @@ -14206,9 +13225,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: _parts = key.split(":") _platform = _parts[2] if len(_parts) > 2 else "" - # Off-loop + bounded: plugin finalize hooks can - # block arbitrarily (see _finalize_session_off_loop) - # and this watcher runs on the gateway event loop. + # Off-loop + bounded: plugin finalize hooks can block arbitrarily, and + # this watcher runs on the gateway event loop. await self._finalize_session_off_loop( session_id=entry.session_id, platform=_platform, @@ -14216,9 +13234,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception: pass - # Shut down memory provider and close tool resources - # on the cached agent. Idle agents live in - # _agent_cache (not _running_agents), so look there. + # Close the cached agent's memory provider and tool resources. Idle agents + # live in _agent_cache (not _running_agents), so look there. _cached_agent = None _cache_lock = getattr(self, "_agent_cache_lock", None) if _cache_lock is not None: @@ -14234,22 +13251,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await self._cleanup_agent_resources_off_loop( _cached_agent, context="session expiry" ) - # Drop the cache entry so the AIAgent (and its LLM clients, tool schemas, - # memory provider refs) can be garbage-collected. Otherwise the cache grows - # unbounded across the gateway's lifetime. + # Drop the cache entry so the AIAgent (LLM clients, tool schemas, memory + # provider refs) can be GC'd; otherwise the cache grows unbounded. self._evict_cached_agent(key) - # Permanently finalizing this session — one funnel call drops every - # conversation-scoped dict AND the boundary security state (approvals, - # update prompts, slash-confirm) so the dicts don't grow unbounded across - # the gateway's lifetime. Idle agent-cache eviction must NOT do this: that - # session is still alive and a resumed turn rebuilds its agent from these - # overrides. Only finalization, /new and /reset clear them. + # Permanent finalization: one funnel call drops every conversation-scoped + # dict AND boundary security state so they don't grow unbounded. Idle + # agent-cache eviction must NOT do this — that session is still alive and a + # resumed turn rebuilds from these overrides. Only finalize, /new, /reset clear. self._clear_conversation_scope( key, reason="expiry_finalized" ) - # Persist the finalized flag to sessions.json AND state.db (single write- - # path, #9006) — also drops the persisted /model override, since - # finalization is a conversation boundary. + # Persist finalized flag (sessions.json AND state.db, single write-path); + # also drops the /model override — finalization is a conversation boundary. await self.async_session_store.set_expiry_finalized(entry) logger.debug( "Session expiry finalized for %s", @@ -14290,9 +13303,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Session expiry done: %d finalized", _done, ) - # Sweep agents that have been idle beyond the TTL regardless of session reset - # policy. This catches sessions with very long / "never" reset windows, whose cached - # AIAgents would otherwise pin memory for the gateway's entire lifetime. + # Sweep agents idle beyond the TTL regardless of session reset policy: sessions with + # long / "never" reset windows would otherwise pin memory for the gateway's life. try: _idle_evicted = self._sweep_idle_cached_agents() if _idle_evicted: @@ -14303,17 +13315,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as _e: logger.debug("Idle agent sweep failed: %s", _e) - # Neither the LRU cap nor the idle TTL is aware of how much memory a cached - # transcript costs, so a busy gateway keeps every warm session's tool output - # resident until RSS hits the cgroup limit. + # Neither LRU cap nor idle TTL knows what a cached transcript costs in memory, so a + # busy gateway keeps every warm session's tool output resident until the RSS limit. try: self._sweep_agent_cache_under_pressure() except Exception as _e: logger.debug("Agent cache pressure sweep failed: %s", _e) - # Periodically prune stale SessionStore entries. The in-memory dict (and - # sessions.json) would otherwise grow unbounded in gateways serving many rotating - # chats / threads / users over long time windows. + # Prune stale SessionStore entries; the in-memory dict (and sessions.json) would + # otherwise grow unbounded with many rotating chats / threads / users. _last_prune_ts = getattr(self, "_last_session_store_prune_ts", 0.0) _prune_interval = 3600.0 # once per hour if time.time() - _last_prune_ts > _prune_interval: @@ -14365,10 +13375,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew yield adapter def _session_activity_for_stall(self, session_key: str) -> Optional[dict]: - """Return the shared activity snapshot for stall progress (#72039). - - Single progress source: ``AIAgent.get_activity_summary()`` / - ``agent.session_activity``. No turn-start or pending-inbound clocks. + """Return the shared activity snapshot for stall progress: the single source is + ``AIAgent.get_activity_summary()`` / ``agent.session_activity``; no turn-start or + pending-inbound clocks. """ agent = (getattr(self, "_running_agents", None) or {}).get(session_key) if agent is None or agent is _AGENT_PENDING_SENTINEL: @@ -14382,9 +13391,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return summary if isinstance(summary, dict) else None async def _check_session_stalls(self, timeout_seconds: float) -> int: - """Scan pending inbound sessions and notify once per stall episode. - - Returns the number of notifications sent this pass (for tests). + """Scan pending inbound sessions and notify once per stall episode; returns the number of + notifications sent this pass (for tests). """ from gateway.session_stall import ( format_session_stall_notification, @@ -14478,10 +13486,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Cannot deliver; latch to avoid log spam every tick. notified_map[session_key] = True continue - # #76354 review S2: re-read pending state + activity timestamp IMMEDIATELY before - # delivery. The snapshot above ages while earlier candidates await their sends; an - # agent that made progress (or drained its queue) meanwhile must not get a false stall - # notice. Abort and leave the latch un-set so the next tick re-evaluates from scratch. + # Re-read pending state + activity IMMEDIATELY before delivery: the snapshot above ages + # while earlier candidates await sends; an agent that progressed (or drained its queue) + # must not get a false stall notice. Abort, latch un-set, so the next tick re-evaluates. still_pending = ( (getattr(adapter, "_pending_messages", None) or {}).get( session_key @@ -14517,9 +13524,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if source is not None and hasattr(self, "_thread_metadata_for_source") else None ) - # Round-2 #2: bound the send. A wedged adapter transport (network hang, dead - # websocket) must not block the whole watcher pass — sibling candidates in this loop - # would never be evaluated and the watcher itself would stop ticking. + # Bound the send: a wedged adapter transport (network hang, dead websocket) must not + # block the watcher pass — siblings would go unevaluated and the watcher stop. try: result = await asyncio.wait_for( adapter.send( @@ -14563,10 +13569,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return sent async def _model_catalog_refresh_watcher(self) -> None: - """Refresh the /model picker's remote catalogs every TTL window. - - The picker itself only refreshes on a cold or stale open, so a gateway that nobody opens - ``/model`` in keeps serving whatever was cached. + """Refresh the /model picker's remote catalogs every TTL window. The picker itself only + refreshes on a cold/stale open, so if nobody opens ``/model`` the cache never updates. """ from hermes_cli.model_catalog import refresh_catalogs, refresh_interval_seconds @@ -14616,39 +13620,31 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return "default" # ── Kanban board watchers ─────────────────────────────────────────── - # The kanban notifier/dispatcher watcher loops + their helpers live in - # GatewayKanbanWatchersMixin (gateway/kanban_watchers.py). + # Loops + helpers live in GatewayKanbanWatchersMixin (gateway/kanban_watchers.py). - #: Interval of the slow respawn tier that takes over once the reconnect watcher has exhausted - #: its supervised restart budget. Long on purpose: the budget is spent precisely when the - #: watcher is crashing on contact, so the useful cadence is "check back later". A tight loop - #: here would be worse than the outage it is healing. + #: Slow respawn tier interval, used once the reconnect watcher has exhausted its supervised + #: restart budget. Long on purpose: the budget is spent when the watcher is crashing on contact, + #: so the useful cadence is "check back later"; a tight loop would be worse than the outage. _RECONNECT_WATCHER_SLOW_RETRY_SECS = 300 - #: How many slow-tier respawns to attempt while work is still queued. Bounded, not infinite: - #: if half an hour of five-minute retries cannot keep a watcher alive, the fault is not - #: transient and a louder failure is more useful than a quieter one that never stops. + #: Slow-tier respawns to attempt while work is still queued. Bounded: if half an hour of + #: five-minute retries cannot keep a watcher alive, the fault is not transient — fail loudly. _MAX_SLOW_WATCHER_RESPAWNS = 6 def _on_reconnect_watcher_gave_up(self, name: str = "") -> None: """Own the reconnect invariant once supervision has abandoned it. - Invariant: while the gateway is running and ``_failed_platforms`` is non-empty, either a - reconnect watcher is live or a bounded respawn is scheduled. Previously only a *later - fatal error from another platform* noticed a dead watcher — event-coupled recovery that - needs an event which, by construction, may never come (the failed adapter is dropped from - the live map, so nothing is left to emit it; the platform stays queued and the stranded - check treats a queued platform as safe, so the process is never restarted either). - Deliberately NOT done here: requesting a supervisor/process restart when the slow tier is - also exhausted — that is a blast-radius policy decision for a maintainer. Instead a single - loud error names the still-queued platforms for an operator/external supervisor. + Invariant: while running and ``_failed_platforms`` is non-empty, a reconnect watcher is live + or a bounded respawn is scheduled. Event-coupled recovery is not enough: the failed adapter + is dropped from the live map, so no later event may ever arrive to notice a dead watcher. + Deliberately NOT done here: requesting a process restart when the slow tier is exhausted — + a blast-radius policy call; a single loud error names the still-queued platforms instead. """ if not getattr(self, "_running", False): return if not getattr(self, "_failed_platforms", None): - # No queued work depends on the watcher. Letting it stay dead is - # correct -- the enqueue path spawns a fresh one the moment a - # platform is queued again. + # No queued work depends on the watcher; leaving it dead is correct — the enqueue path + # spawns a fresh one the moment a platform is queued again. logger.warning( "Reconnect watcher supervision exhausted with an empty retry " "queue — leaving it down until a platform is queued." @@ -14702,9 +13698,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _spawn_reconnect_watcher(self, *, on_give_up=None): """Single place that knows how to launch the reconnect watcher. - The ``on_spawn`` half is load-bearing: without it the supervisor's own respawn leaves - ``_reconnect_watcher_task`` pointing at a dead handle and ``_ensure_...`` spawns a second - concurrent watcher. + ``on_spawn`` is load-bearing: without it the supervisor's own respawn leaves + ``_reconnect_watcher_task`` at a dead handle and ``_ensure_...`` spawns a second watcher. """ self._reconnect_watcher_task = self._spawn_supervised( self._platform_reconnect_watcher, @@ -14717,12 +13712,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _ensure_reconnect_watcher_running(self) -> None: """Ensure the platform reconnect watcher background task is alive. - If the tracked reconnect watcher task has died (e.g. from exhausting its restart budget, - or a terminal exception that _spawn_supervised could not recover), respawns it so - platforms queued for reconnection are not permanently stranded. Called from - _queue_retryable_fatal_platform on BOTH paths: after a new enqueue, and after a re-fatal - for an already-queued platform — the only case in which the watcher can have been - retrying long enough to exhaust its budget. + Respawns a dead watcher (exhausted restart budget, unrecoverable exception) so queued + platforms are not stranded. Called on BOTH _queue_retryable_fatal_platform paths: the + re-fatal of an already-queued platform is the only case where the budget can be exhausted. """ if not getattr(self, "_running", False): return @@ -14738,12 +13730,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _platform_reconnect_watcher(self) -> None: """Background task that periodically retries connecting failed platforms. - Uses exponential backoff: 30s → 60s → 120s → 240s → 300s (cap). Retryable failures - (network/DNS blips) keep retrying at the backoff cap indefinitely — they self-heal once - connectivity returns, so a transient outage never requires manual intervention. - Non-retryable failures (bad auth) drop out of the queue immediately. The circuit breaker - (``/platform pause``) stays available manually but is never triggered automatically — - auto-pausing left bots silently dead after a transient DNS failure. + Exponential backoff 30s → 300s cap; retryable failures (network/DNS) retry at the cap + indefinitely so transient outages self-heal, non-retryable (bad auth) drop out immediately. + The circuit breaker (``/platform pause``) is manual only — auto-pausing left bots dead. """ await asyncio.sleep(10) # initial delay — let startup finish while self._running: @@ -14763,22 +13752,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return info = self._failed_platforms.get(platform) if info is None: - # Removed concurrently (e.g. a manual /platform resume, or a reconnect that - # succeeded via a different path) between the snapshot above and this lookup. - # Not an error -- just nothing to do for it this pass. + # Removed concurrently (/platform resume, reconnect via another path) between + # the snapshot above and this lookup — not an error, nothing to do this pass. continue # Skip paused platforms entirely — they need explicit # /platform resume to come back. if info.get("paused"): continue - # Long-lived retry-loop escalation (OOF-156): once a platform - # has been continuously queued past the attention threshold, - # flag it NEEDS_ATTENTION in runtime status so owners and - # fleet monitoring see "this is not a blip" — a dead token, - # revoked intent, or crash-looping sidecar otherwise presents - # as ordinary "retrying" forever. Retries continue unchanged: - # this is a signal, NOT a circuit breaker (auto-pause was - # deliberately removed — see this docstring's history). + # Long-lived retry escalation: past the attention threshold flag the platform + # NEEDS_ATTENTION in runtime status so a dead token/revoked intent doesn't look + # like ordinary "retrying" forever. A signal, NOT a circuit breaker — retries continue. if not info.get("attention_flagged") and _reconnect_needs_attention(info, now): info["attention_flagged"] = True queued_for = now - info.get("queued_at", now) @@ -14806,9 +13789,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform_config = info["config"] attempt = info["attempts"] + 1 - # Empty-token primary configs can never reconnect; drop them so - # multiplex setups where a secondary profile owns the bot do - # not spin forever (#64674). + # Empty-token primary configs can never reconnect; drop them so multiplex setups + # where a secondary profile owns the bot do not spin forever. if not _platform_has_bot_credential(platform, platform_config): logger.warning( "Reconnect %s: no bot credential on queued config, " @@ -14845,9 +13827,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew adapter.set_platform_event_handler(self._primary_platform_event_handler()) adapter._busy_text_mode = self._busy_text_mode - # Reconnect after an outage: preserve the platform's - # server-side update queue so messages sent while the bot - # was offline are delivered rather than dropped (#46621). + # Reconnect after outage: keep the platform's server-side update queue so + # messages sent while the bot was offline are delivered rather than dropped. success = await self._connect_adapter_with_timeout( adapter, platform, is_reconnect=True ) @@ -14935,15 +13916,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Same fd-leak concern as the non-retryable branch above: the adapter failed # to connect and is being thrown away. await _dispose_unused_adapter(adapter) - # Retryable failures (network/DNS blips) keep retrying at the backoff cap - # indefinitely — they self-heal once connectivity returns. We do NOT - # auto-pause them: a transient outage must never require a manual - # `/platform resume`. Anything reaching here is retryable by construction. + # Retryable failures (network/DNS blips) retry at the backoff cap forever, + # self-healing when connectivity returns. Never auto-pause them: a transient + # outage must not need `/platform resume`. Everything here is retryable. except Exception as e: if adapter is not None: - # An exception escaping the connect call path (DNS timeout, aiohttp - # server.start() crash, etc.) leaves the adapter in the same unowned state - # as the two branches above. + # An exception escaping connect (DNS timeout, aiohttp server.start() crash, + # etc.) leaves the adapter in the same unowned state as the branches above. await _dispose_unused_adapter(adapter) self._update_platform_runtime_status( platform.value, @@ -14958,8 +13937,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Reconnect %s error: %s, next retry in %ds", platform.value, e, backoff, ) - # A raised exception during reconnect (connect timeout, DNS - # resolution failure, etc.) is inherently transient — keep + # A reconnect exception (connect timeout, DNS failure, ...) is transient; keep # retrying at the backoff cap rather than auto-pausing. # Check every 10 seconds for platforms that need reconnection @@ -15047,14 +14025,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _kill_tool_subprocesses(phase: str) -> list: """Kill tool subprocesses + tear down terminal envs + browsers. - Returns the cron job IDs this phase marked interrupted, so the - caller can notify their owners while adapters are still up - (#82232). Empty list when no cron work was in flight. - - Called twice in shutdown: eagerly after a drain timeout forces agent interrupt - (reclaim bash/sleep children before systemd TimeoutStopSec escalates to SIGKILL), - and as a final catch-all at the end of _stop_impl(). All steps best-effort; - exceptions swallowed so one subsystem cannot block the rest. + Returns the cron job IDs marked interrupted so the caller can notify owners while + adapters are still up. Called twice: eagerly after a drain timeout forces interrupt + (reclaim children before systemd SIGKILLs) and as a final catch-all in _stop_impl(). + Best-effort; exceptions swallowed so one subsystem cannot block the rest. """ try: from tools.process_registry import process_registry @@ -15068,11 +14042,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("process_registry.kill_all (%s) error: %s", phase, _e) _marked_cron_jobs: list = [] try: - # Any cron job still dispatched at this instant just had its tool subprocess - # killed above (kill_all() has no per-job-ID targeting — it's a global sweep). - # Its agent thread may still produce a plausible final response from the - # truncated output; mark the run interrupted so it can never be reported as - # success. No-op when no cron job is in flight. + # kill_all() is a global sweep, so any cron job dispatched right now lost its tool + # subprocess; its agent thread may still emit a plausible response from truncated + # output. Mark the run interrupted so it can never be reported as success. from cron.scheduler import mark_running_jobs_interrupted _interrupted = _marked_cron_jobs = mark_running_jobs_interrupted( f"Gateway shutdown ({phase}) killed the job's tool " @@ -15208,9 +14180,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew timeout = self._restart_drain_timeout - # Pre-mark sessions as resume_pending BEFORE the drain wait. If the process is killed by - # the service manager during the drain, the durable marker is already written so the - # next gateway boot can recover in-flight sessions. + # Pre-mark sessions resume_pending BEFORE the drain wait: if the service manager kills + # the process mid-drain, the durable marker already lets the next boot recover them. _pre_drain_keys: list[str] = [] for _sk, _agent in list(self._running_agents.items()): if _agent is _AGENT_PENDING_SENTINEL: @@ -15227,10 +14198,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _cron_at_start = self._active_cron_job_count() _api_at_start = self._active_api_run_count() _deferred_at_start = _deferred_worker_count() - # In-flight cron work gets its own floor, clamped to the watchdog leash we're already - # running under so the extra wait can never cost us the post-drain cleanup window. - # getattr-guard: shutdown-path tests drive _stop_impl_body from bare doubles that aren't - # GatewayRunner instances, so they don't pick up the class-level default. + # In-flight cron work gets its own floor, clamped to the watchdog leash so the extra + # wait never costs the post-drain cleanup window. getattr-guard: shutdown-path tests + # drive _stop_impl_body from bare doubles (not GatewayRunner) lacking the class default. _cron_drain_cfg = getattr( self, "_cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT ) @@ -15275,9 +14245,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) if not timed_out: - # Drain completed gracefully — all running sessions finished. - # Clear the pre-drain resume_pending markers so sessions that - # completed during the drain window don't carry a stale flag. + # Graceful drain: clear the pre-drain resume_pending markers so sessions that + # finished during the drain window don't carry a stale flag. for _sk in _pre_drain_keys: if _sk not in self._running_agents: try: @@ -15300,27 +14269,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._active_api_run_count(), _deferred_worker_count(), ) - # Mark forcibly-interrupted sessions as resume_pending BEFORE - # interrupting the agents. This preserves each session's - # session_id + transcript so the next message on the same - # session_key auto-resumes from the existing conversation - # instead of getting routed through suspend_recently_active() - # and converted into a fresh session. Terminal escalation - # for genuinely stuck sessions still flows through the - # existing ``.restart_failure_counts`` stuck-loop counter - # (incremented below, threshold 3), which sets - # ``suspended=True`` and overrides resume_pending. + # Mark forcibly-interrupted sessions resume_pending BEFORE interrupting, so the next + # message on the same session_key auto-resumes instead of being converted to a fresh + # session by suspend_recently_active(). Genuinely stuck sessions still escalate via + # ``.restart_failure_counts`` (threshold 3), which sets ``suspended=True`` and wins. # - # Iterate self._running_agents (current) rather than the - # drain-start ``active_agents`` snapshot — the snapshot - # may include sessions that finished gracefully during - # the drain window, and marking those falsely would give - # them a stray restart-interruption system note on their - # next turn even though their previous turn completed - # cleanly. Skip pending sentinels for the same reason - # _interrupt_running_agents() does: their agent hasn't - # started yet, there's nothing to interrupt, and the - # session shouldn't carry a misleading resume flag. + # Iterate self._running_agents (current), not the drain-start snapshot: sessions that + # finished cleanly during the drain would otherwise get a stray interruption note. + # Skip pending sentinels as _interrupt_running_agents() does — nothing has started. _resume_reason = ( "restart_timeout" if self._restart_requested else "shutdown_timeout" ) @@ -15347,11 +14303,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Shutdown phase: allowing %.1fs for interrupted agents to unwind", interrupt_grace_timeout, ) - # Wait on API-server work too. The interrupt is cooperative: - # without this the settle window closes the instant - # _running_agents is empty, and an API turn that was just asked - # to stop gets its tool subprocesses killed below before it can - # unwind — the exact amputation this interrupt exists to avoid. + # Wait on API-server work too: the interrupt is cooperative, and without this the + # settle window closes as soon as _running_agents is empty, so an API turn just asked + # to stop has its tool subprocesses killed below before it can unwind. while ( self._running_agents or self._active_api_run_count() @@ -15360,16 +14314,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._update_runtime_status("draining") await asyncio.sleep(0.1) - # The interrupt above fires exactly once, but work can - # materialize AFTER that one shot: a /v1/runs task admitted - # before the drain populates _active_run_agents only once - # _create_agent returns, and a _running_agents entry claimed - # as _AGENT_PENDING_SENTINEL is promoted to a real agent by - # track_agent() on its own schedule. Either way the settle - # loop waited on work nothing signaled. If any is still live - # at settle-loop exit, re-signal so a late-materializing - # agent gets a cooperative interrupt instead of going - # straight to the tool-subprocess kill. + # The interrupt fires once, but work can materialize AFTER it: a /v1/runs task enters + # _active_run_agents only when _create_agent returns, and a _AGENT_PENDING_SENTINEL + # entry is promoted by track_agent() on its own schedule. Re-signal anything still + # live so it gets a cooperative interrupt instead of a bare tool-subprocess kill. if ( self._running_agents or self._active_api_run_count() @@ -15384,11 +14332,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Re-signaled interrupt for work still live at settle-window exit" ) - # Kill lingering tool subprocesses NOW, before we spend more budget on adapter - # disconnect / session DB close: under systemd (TimeoutStopSec ≈ drain_timeout + - # headroom) deferring risks SIGKILL on the cgroup first, so orphaned bash/sleep - # children die by systemd instead of us. The final catch-all below still runs for - # the graceful path. + # Kill lingering tool subprocesses NOW, before adapter disconnect / DB close: under + # systemd (TimeoutStopSec ≈ drain_timeout + headroom) deferring risks the cgroup + # SIGKILL reaping orphaned children instead of us. The final catch-all still runs. _interrupted_cron_jobs = _kill_tool_subprocesses("post-interrupt") logger.info( "Shutdown phase: post-interrupt tool kill done at +%.2fs", @@ -15427,9 +14373,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _agent = ( _entry[0] if isinstance(_entry, tuple) else _entry ) - # Bounded + off-loop so a wedged memory provider on one - # idle agent can't hang shutdown indefinitely — that path - # is why SIGTERM failed to kill the process (#53175). + # Bounded + off-loop so a wedged memory provider can't hang shutdown forever + # (this path is why SIGTERM once failed to kill the process). await self._cleanup_agent_resources_off_loop( _agent, context="shutdown idle-cache" ) @@ -15474,17 +14419,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self.adapters.clear() for _session_key in list(self._running_agents): self._release_running_agent_state(_session_key) - # Flush pending messages to disk before clearing. When FTS5 corruption prevents message - # persistence, the in-memory pending text is the only surviving copy. Clearing without - # flushing causes permanent data loss. + # Flush pending messages to disk before clearing: under FTS5 corruption the in-memory + # pending text is the only surviving copy; clearing unflushed loses it permanently. try: from gateway.shutdown_flush import flush_pending_to_file flush_pending_to_file(dict(self._pending_messages), reason="shutdown") except Exception: pass - # The FIFO tail lives in SessionState.conversation.queued_events, - # not in the slot dict above — flush it too or every follow-up - # parked in overflow at restart time is lost (#99882). + # The FIFO tail lives in SessionState.conversation.queued_events, not the slot dict + # above — flush it too or every follow-up parked in overflow at restart time is lost. try: from gateway.shutdown_flush import flush_overflow_to_file flush_overflow_to_file( @@ -15510,26 +14453,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._busy_ack_ts.clear() self._shutdown_event.set() - # Global cleanup: kill any remaining tool subprocesses not tied - # to a specific agent (catch-all for zombie prevention). On the - # drain-timeout path we already did this earlier after agent - # interrupt — this second call catches (a) the graceful path - # where drain succeeded without interrupt, and (b) anything - # that got respawned between the earlier call and adapter - # disconnect (defense in depth; safe to call repeatedly). + # Global catch-all subprocess kill (safe to repeat): covers the graceful path and + # anything respawned since the drain-timeout path's post-interrupt kill. _kill_tool_subprocesses("final-cleanup") logger.info( "Shutdown phase: final-cleanup tool kill done at +%.2fs", _phase_elapsed(), ) - # Reap the process-global auxiliary-client cache once at the very - # end of teardown. Per-turn cleanup runs in _cleanup_agent_resources - # for each active agent, but clients bound to worker-thread loops - # that died with their ThreadPoolExecutor (notably cron ticks) only - # get swept here. Without this, long-running gateways accumulate - # async httpx transports until they hit EMFILE on macOS's default - # RLIMIT_NOFILE=256. See #14210. + # Reap the process-global auxiliary-client cache once at the end of teardown. Per-turn + # cleanup misses clients bound to worker-thread loops that died with their executor + # (cron ticks); without this sweep async httpx transports accumulate until EMFILE. try: from agent.auxiliary_client import shutdown_cached_clients shutdown_cached_clients() @@ -15584,11 +14518,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _phase_elapsed(), ) - # Close SQLite session DBs so the WAL write lock is released. Otherwise --replace and - # similar restart flows leave the old connection holding the lock until Python exits, - # so the new gateway gets 'database is locked'. ``self`` holds the DB at - # ``_session_db`` (an AsyncSessionDB facade); unwrap to the sync handle. - # ``session_store`` holds it at ``_db``. + # Close SQLite session DBs so the WAL lock is released; otherwise --replace leaves the old + # connection holding it until exit and the new gateway gets 'database is locked'. + # ``_session_db`` is an AsyncSessionDB facade — unwrap; ``session_store`` holds ``_db``. _self_db = getattr(self, "_session_db", None) _self_db = getattr(_self_db, "_db", _self_db) for _db in (_self_db, getattr(getattr(self, "session_store", None), "_db", None)): @@ -15598,11 +14530,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _db.close() except Exception as _e: logger.debug("SessionDB close error: %s", _e) - # A multiplexed session_store caches one SessionDB per profile - # path (#88532); reading ``_db`` above only resolved the handle - # for the shutdown task's own (root) scope. Sweep the rest so - # secondary profiles' WAL locks are released before --replace - # brings a new gateway up on the same files. + # A multiplexed session_store caches one SessionDB per profile; ``_db`` above only covered + # the root scope. Sweep the rest so secondary WAL locks are released before --replace. _sweep = getattr( getattr(self, "session_store", None), "close_all_db_handles", None ) @@ -15617,9 +14546,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew GatewayRunner.close_all_session_db_handles(self) except Exception as _e: logger.debug("Runner SessionDB handle sweep error: %s", _e) - # Final sweep: close any shared SessionDB instances still held by the process-wide - # registry (in-process tools, cron, mirror, etc. that opened via get_shared_session_db - # but weren't released by the sweeps above). + # Final sweep: close shared SessionDB instances still held by the process-wide registry + # (tools, cron, mirror, etc. opened via get_shared_session_db but not released above). try: from hermes_state import close_shared_session_dbs closed = close_shared_session_dbs() @@ -15636,16 +14564,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew remove_pid_file() release_gateway_runtime_lock() - # Write a clean-shutdown marker so the next startup knows this wasn't a crash. - # suspend_recently_active() only needs to run after unexpected exits. If the drain - # timed out and agents were force-interrupted, their sessions may be half-finished - # (trailing tool response, no final assistant message) — skip the marker so the next - # startup suspends them and users get a clean slate. + # Clean-shutdown marker: suspend_recently_active() need only run after unexpected exits. + # If the drain timed out and agents were force-interrupted, sessions may be half-finished + # — skip the marker so the next startup suspends them. if not timed_out: - try: + with suppress(Exception): (_hermes_home / ".clean_shutdown").touch() - except Exception: - pass else: logger.info( "Skipping .clean_shutdown marker — drain timed out with " @@ -15653,11 +14577,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "active sessions." ) - # Track sessions that were active at shutdown for stuck-loop - # detection (#7536). On each restart, the counter increments - # for sessions that were running. If a session hits the - # threshold (3 consecutive restarts while active), the next - # startup auto-suspends it — breaking the loop. + # Stuck-loop detection: the counter increments for sessions active at each restart; at + # the threshold (3 consecutive) the next startup auto-suspends the session. if active_agents: self._increment_restart_failure_counts(set(active_agents.keys())) @@ -15676,21 +14597,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Failed to write planned restart notification marker: %s", e) if self._restart_requested and self._restart_via_service: - # The service manager is the sole restart owner. Exit 75 paired with - # ``RestartForceExitStatus=75`` asks systemd to replace this process without a - # second helper racing the unit's stop/start job. + # Service manager owns restarts: exit 75 + ``RestartForceExitStatus=75`` has systemd + # replace this process without a second helper racing the unit's stop/start job. self._exit_code = GATEWAY_SERVICE_RESTART_EXIT_CODE self._exit_reason = self._exit_reason or "Gateway restart requested" self._draining = False - # Persist the terminal gateway_state. Default "stopped", but an UNEXPECTED external - # signal (container/s6 SIGTERM on docker restart, OOM-kill, bare kill) persists - # "running" to preserve run-intent: container_boot.py only auto-starts gateways whose - # last state was "running", so writing "stopped" (or leaving "draining") for a routine - # `docker compose up --force-recreate` would leave channels dark until a manual - # restart. Operator-initiated stops write a planned-stop marker BEFORE signalling, so - # they persist "stopped"; a restart also persists "stopped" (the new process brings - # the gateway back up itself). + # Persist terminal gateway_state: "stopped" by default, but "running" on an UNEXPECTED + # external signal (s6 SIGTERM on docker restart, OOM-kill, kill) — container_boot.py + # only auto-starts gateways last seen "running", so "stopped"/"draining" after a routine + # recreate would leave channels dark. Operator stops write a planned-stop marker BEFORE + # signalling and persist "stopped"; a restart also persists "stopped". if getattr(self, "_signal_initiated_shutdown", False) and not self._restart_requested: logger.info( "Gateway stopped by an unexpected signal — persisting " @@ -15713,14 +14630,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _start_secondary_profile_adapters(self) -> int: """Bring up adapters for every non-active profile this gateway serves. - Returns the number of secondary adapters that connected. No-op (returns - 0) unless ``gateway.multiplex_profiles`` is on. - - Each profile's adapters connect under that profile's HERMES_HOME + secret scope, live in - ``self._profile_adapters[profile]``, and get a handler that stamps ``source.profile`` so - the turn resolves that profile's config/skills/credentials. Same-platform credential - collisions (two profiles polling one bot token) are refused here — the only point that - sees every profile's resolved credentials together. + Returns the count of connected secondary adapters; 0 unless ``gateway.multiplex_profiles``. + Each profile's adapters connect under its HERMES_HOME + secret scope, live in + ``self._profile_adapters[profile]``, and get a handler stamping ``source.profile``. Same- + platform credential collisions are refused here — the only point seeing every profile's + resolved credentials together. """ if not getattr(self.config, "multiplex_profiles", False): return 0 @@ -15732,9 +14646,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew active = get_active_profile_name() or "default" connected = 0 - # Resource claim -> profile that owns it. Credential claims prevent two - # profiles polling the same account; listener claims prevent sidecars - # with distinct credentials from binding the same endpoint. + # Resource claim -> owning profile. Credential claims stop two profiles polling the same + # account; listener claims stop sidecars with distinct credentials binding one endpoint. claimed: Dict[tuple, str] = {} for _plat, _ad in self.adapters.items(): fp = self._adapter_credential_fingerprint(_ad) @@ -15743,9 +14656,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew listener_claim = self._adapter_listener_claim(_plat, _ad) if listener_claim is not None: claimed[listener_claim] = active - # A retryable primary still owns its configured credential and listener. - # Reserve both while it is queued so a secondary cannot take the endpoint - # before the reconnect watcher retries the primary adapter. + # A retryable primary still owns its credential and listener; reserve both while queued + # so a secondary cannot take the endpoint before the reconnect watcher retries it. for retry_info in getattr(self, "_failed_platforms", {}).values(): for claim_name in ("credential_claim", "listener_claim"): retry_claim = retry_info.get(claim_name) @@ -15783,9 +14695,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew served = [active] + sorted( name for name, _home in profile_homes if name != active ) - # Per-profile PairingStores so authz_mixin can route pairing checks to the right - # whitelist. The active profile gets a store at its HERMES_HOME; additional served - # profiles resolve from their own profile homes. See gateway.pairing.PairingStore. + # Per-profile PairingStores so authz_mixin routes pairing checks to the right whitelist; + # the active profile's store is at its HERMES_HOME, other served profiles at their own. for name in served: if name and name not in self.pairing_stores: self.pairing_stores[name] = ( @@ -15888,11 +14799,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform.value, ) continue - # Relay and WhatsApp are shared process-level ingress in multiplex mode: one connection - # owned by the active profile, with route-stamped source.profile fanning inbound turns - # out to secondary profiles. The WhatsApp bridge is one authenticated session tied to - # one phone number; a secondary has no credential to bring, so an adapter for it would - # only connect/retry-loop and stall startup for every profile queued behind it. + # Relay and WhatsApp are shared process-level ingress in multiplex mode (one connection + # owned by the active profile, route-stamped source.profile fans out). WhatsApp is one + # session per phone number; a secondary adapter would only retry-loop and stall startup. if ( getattr(self.config, "multiplex_profiles", False) and platform in (Platform.RELAY, Platform.WHATSAPP) @@ -16018,9 +14927,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform: Platform, ) -> None: """Install the profile-scoped handlers shared by startup and reconnect.""" - # Runtime status is process-scoped even while message/config work is - # profile-scoped. Preserve both dimensions in the key so dashboard - # and NAS health aggregation can see which secondary profile failed. + # Runtime status is process-scoped while message/config work is profile-scoped. Keep both + # dimensions in the key so dashboard/NAS health aggregation sees which secondary failed. adapter._runtime_status_platform_key = f"{profile_name}:{platform.value}" adapter.set_message_handler(self._make_profile_message_handler(profile_name)) adapter.set_fatal_error_handler( @@ -16084,9 +14992,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew profile_config = load_gateway_config().platforms.get(platform) if profile_config is None or not profile_config.enabled: return - # Mirrors the startup credential gate (#84079): a - # credential removed from this profile's scope must - # not rebuild an adapter that would fan out turns. + # Mirrors the startup credential gate: a credential removed from this + # profile's scope must not rebuild an adapter that would fan out turns. if not _platform_has_bot_credential(platform, profile_config): logger.info( "Secondary %s reconnect skipped: no bot credential " @@ -16129,9 +15036,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await self._safe_adapter_disconnect(adapter, platform) return - # Shutdown can begin while connect() is in flight. Do not - # republish a newly connected adapter after the registry has - # been drained; release its partial resources instead. + # Shutdown can begin mid-connect(): never republish a newly connected adapter + # after the registry has been drained; release its partial resources instead. if success: await self._safe_adapter_disconnect(adapter, platform) return @@ -16183,12 +15089,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Queue a cold-start reconnect for a secondary adapter. - Startup failure branches run BEFORE ``self._running`` flips True, so the regular - scheduler's ``not self._running`` guard would drop the request and the runner loop would - exit immediately. This bridge parks a task across the rest of startup and hands off to - the regular scheduler once live (its ``_profile_failed_platforms`` slot dedupes); if - shutdown begins first, the request is released. Non-retryable failures are dropped here - exactly as the regular scheduler would. + Startup failures happen BEFORE ``self._running`` flips True, so the regular scheduler's + guard would drop the request. Park a task across startup and hand off to the scheduler once + live (``_profile_failed_platforms`` dedupes); release it if shutdown begins first. + Non-retryable failures are dropped as the regular scheduler would. """ if not getattr(adapter, "fatal_error_retryable", True): return @@ -16228,9 +15132,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform.value, ) return - # Modest poll interval: startup completion has no dedicated event, - # and the reconnect runner's own backoff makes sub-100ms precision - # irrelevant. Bounded so a wedged startup cannot spin the loop. + # Modest poll: startup completion has no dedicated event, and the reconnect runner's own + # backoff makes sub-100ms precision irrelevant. Bounded so a wedged startup cannot spin. while not self._running and not self._shutdown_event.is_set(): await asyncio.sleep(0.1) if self._running and not self._shutdown_event.is_set(): @@ -16239,9 +15142,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew profile_name, platform, adapter ) except Exception: - # The handoff touches live registries; if it raises, the parked task would - # otherwise die as an unretrieved-task exception logged only at GC time. Surface - # it where operators look. + # The handoff touches live registries; if it raises, the parked task dies as an + # unretrieved-task exception logged only at GC. Surface it where operators look. logger.exception( "secondary-startup-reconnect handoff failed " "(profile=%s platform=%s)", @@ -16302,9 +15204,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Remove a failed multiplexed adapter without touching the primary slot. - Secondary adapters are owned by ``_profile_adapters`` rather than ``self.adapters``. The - primary-only fatal handler intentionally ignores them; without this route, a fatal - secondary Discord client stayed live forever after its liveness sampler stopped. + Secondaries live in ``_profile_adapters``, which the primary-only fatal handler ignores; + without this route a fatal secondary Discord client stayed live forever. """ profile_map = getattr(self, "_profile_adapters", {}).get(profile_name) if not isinstance(profile_map, dict) or profile_map.get(platform) is not adapter: @@ -16373,21 +15274,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _make_default_profile_message_handler(self): """Scope primary-adapter messages to their routed multiplex profile. - Profile routes are normally stamped on ``event.source`` before this handler runs. Resolve - the home per event so session lookup and transcript loading use the same profile store as - the later agent run and persistence path. Authorization still belongs to the primary - transport profile (a shared adapter may route to a profile that intentionally has no bot - credential or allowlist), so the transport home is preserved on the live source and the - auth gate never re-checks the sender against the routed profile's secret scope. - Genuinely unrouted events retain the gateway's launch/default home. + Resolve the home per event so session lookup and transcript loading use the same profile + store as the agent run. Authorization stays with the transport profile (a routed profile + may intentionally have no bot credential/allowlist): the transport home is preserved on the + live source and never re-checked against the routed scope. Unrouted events keep the default. """ default_home = Path(get_hermes_home()) async def _handler(event): source = event.source - # In-process only (SessionSource serialization ignores dynamic attrs). - # The route selects agent/session state, not which bot admitted the - # message. Keep those two trust domains separate. + # In-process only (SessionSource serialization ignores dynamic attrs). The route selects + # agent/session state, not which bot admitted the message — separate trust domains. source._authorization_profile_home = default_home if ( not getattr(source, "profile", None) @@ -16398,9 +15295,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: source.profile = self._profile_name_for_source(source) except ProfileRouteRejected: - # NOT write-only: ``_handle_message``'s ingress gate reads this exact marker and - # drops the message fail-closed ("explicit profile route targets an unserved - # profile"). + # NOT write-only: the ``_handle_message`` ingress gate reads this exact marker + # and drops the message fail-closed (explicit route to an unserved profile). source.profile_route_rejected = True profile_home = ( @@ -16471,11 +15367,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> bool: """Authorize under the live transport's profile, not the routed runtime. - A primary adapter may route one chat into another profile's agent/session namespace. That - runtime profile need not (and normally should not) copy the shared bot token or - allowlist. The primary handlers stamp the transport home as an in-process-only attribute - before entering the routed scope; consult it here for the narrow authorization read, - then restore the routed scope for the rest of the turn. + The routed runtime profile need not copy the shared bot token or allowlist. The primary + handlers stamp the transport home as an in-process attribute; read it for authorization + only, then restore the routed scope for the rest of the turn. """ def _check() -> bool: # Preserve the historical one-argument seam used by plugins/tests; @@ -16512,9 +15406,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _adapter_listener_claim(platform: Platform, adapter: Any) -> Optional[tuple]: """Return the exclusive listener resource claimed by an adapter. - Even when two profiles use different project credentials, their sidecars cannot share a - bind and port. Represent that endpoint as a claim so multiplex startup rejects the later - adapter before either ``connect()`` or ``disconnect()`` can disturb the first profile. + Sidecars with different credentials still cannot share a bind+port; expose it as a claim so + multiplex startup rejects the later adapter before connect()/disconnect() disturb the first. """ if getattr(platform, "value", None) != "photon": return None @@ -16532,9 +15425,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _adapter_credential_fingerprint(adapter: Any) -> Optional[str]: """Return a stable, log-safe fingerprint of an adapter's credential. - Used only to detect two profiles claiming the same platform credential. Returns a salted - hash (never the credential itself) of the adapter's primary credential, or None when no - credential is discoverable (in which case we don't attempt conflict detection for it). + Salted hash (never the credential) used to detect two profiles sharing one platform + credential; None when no credential is discoverable (conflict detection is then skipped). """ token = None for attr in ( @@ -16543,15 +15435,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "_token", "api_token", "_bot_token", - # Photon/Spectrum authenticates with project credentials instead - # of a bot token. Including its secret keeps multiplexed profiles - # from spawning competing sidecars for the same account and port. + # Photon/Spectrum authenticates with project credentials, not a bot token; including + # its secret stops multiplexed profiles spawning rival sidecars for one account/port. "_project_secret", - # Feishu/Lark authenticates with an app_id/app_secret pair rather - # than a single token (one active WebSocket connection per app). - # app_id is stable, log-safe, and already used as the adapter's - # _app_lock_identity, so including it lets the multiplex guard - # refuse cloned profiles competing for the same Feishu app. + # Feishu/Lark authenticates with an app_id/app_secret pair (one WebSocket per app). + # app_id is stable, log-safe and already the adapter's _app_lock_identity, so including + # it lets the multiplex guard refuse cloned profiles competing for the same app. "_app_id", # Same class: Teams (client_id/client_secret) and WeCom # (bot_id/secret) authenticate with an app-style id pair too. @@ -16562,12 +15451,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if isinstance(val, str) and val.strip(): token = val.strip() break - # Many adapters (e.g. Discord) store the token on their `config` - # sub-object rather than directly on the adapter. Without this lookup - # those adapters all return None here, the same-token conflict check - # is silently skipped, and every profile's adapter for that platform - # starts polling the same bot token — producing a per-message race - # for which adapter answers. See test_reads_config_token. + # Many adapters (e.g. Discord) store the token on their `config` sub-object. Without this + # lookup they return None, the same-token check is silently skipped, and every profile's + # adapter polls the same bot token — a per-message race over which one answers. if not token: cfg = getattr(adapter, "config", None) if cfg is not None: @@ -16593,9 +15479,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[BasePlatformAdapter]: """Create an adapter and bind it to this gateway runner. - Every lifecycle path — primary/secondary startup and reconnect — goes - through this method. Keep runner binding here so adapters can resolve - inbound profile routes before handlers or ``connect()`` run. + Every lifecycle path (primary/secondary startup, reconnect) uses this method; keep runner + binding here so adapters can resolve inbound profile routes before handlers or connect(). """ adapter = self._instantiate_adapter(platform, config) if adapter is not None: @@ -16609,8 +15494,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[BasePlatformAdapter]: """Instantiate the appropriate adapter for a platform. - Checks the platform_registry first (plugin adapters), then falls - through to the built-in if/elif chain for core platforms. + Checks platform_registry (plugin adapters) first, then the built-in table of core platforms. """ if hasattr(config, "extra") and isinstance(config.extra, dict): config.extra.setdefault( @@ -16641,85 +15525,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Platform registry lookup for '%s' failed: %s", platform.value, e) # Fall through to built-in adapters below - if platform == Platform.WHATSAPP_CLOUD: - from gateway.platforms.whatsapp_cloud import ( - WhatsAppCloudAdapter, - check_whatsapp_cloud_requirements, - ) - if not check_whatsapp_cloud_requirements(): - logger.warning( - "WhatsApp Cloud: aiohttp/httpx missing — reinstall hermes-agent" - ) - return None - return WhatsAppCloudAdapter(config) - - elif platform == Platform.SIGNAL: - from gateway.platforms.signal import ( - SignalAdapter, - check_signal_requirements, - validate_signal_config, - ) - if not check_signal_requirements(): - logger.warning("Signal: runtime requirements not met") - return None - if not validate_signal_config(config): - logger.warning("Signal: SIGNAL_HTTP_URL or SIGNAL_ACCOUNT not configured") - return None - return SignalAdapter(config) - - elif platform == Platform.WEIXIN: - from gateway.platforms.weixin import WeixinAdapter, check_weixin_requirements - if not check_weixin_requirements(): - logger.warning("Weixin: aiohttp/cryptography not installed") - return None - return WeixinAdapter(config) - - elif platform == Platform.API_SERVER: - from gateway.platforms.api_server import APIServerAdapter, check_api_server_requirements - if not check_api_server_requirements(): - logger.warning("API Server: aiohttp not installed") - return None - return APIServerAdapter(config) - - elif platform == Platform.WEBHOOK: - from gateway.platforms.webhook import WebhookAdapter, check_webhook_requirements - if not check_webhook_requirements(): - logger.warning("Webhook: aiohttp not installed") - return None - return WebhookAdapter(config) - - elif platform == Platform.MSGRAPH_WEBHOOK: - from gateway.platforms.msgraph_webhook import ( - MSGraphWebhookAdapter, - check_msgraph_webhook_requirements, - ) - if not check_msgraph_webhook_requirements(): - logger.warning("MSGraph webhook: aiohttp not installed") - return None - return MSGraphWebhookAdapter(config) - - elif platform == Platform.BLUEBUBBLES: - from gateway.platforms.bluebubbles import BlueBubblesAdapter, check_bluebubbles_requirements - if not check_bluebubbles_requirements(): - logger.warning("BlueBubbles: aiohttp/httpx missing or BLUEBUBBLES_SERVER_URL/BLUEBUBBLES_PASSWORD not configured") - return None - return BlueBubblesAdapter(config) - - elif platform == Platform.QQBOT: - from gateway.platforms.qqbot import QQAdapter, check_qq_requirements - if not check_qq_requirements(): - logger.warning("QQBot: aiohttp/httpx missing or QQ_APP_ID/QQ_CLIENT_SECRET not configured") - return None - return QQAdapter(config) - - elif platform == Platform.YUANBAO: - from gateway.platforms.yuanbao import YuanbaoAdapter, WEBSOCKETS_AVAILABLE - if not WEBSOCKETS_AVAILABLE: - logger.warning("Yuanbao: websockets not installed. Run: pip install websockets") - return None - return YuanbaoAdapter(config) - - return None + return _instantiate_builtin_adapter(platform, config) def _make_adapter_auth_check( self, @@ -16728,17 +15534,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Callable[[str, Optional[str], Optional[str]], bool]: """Build a platform-bound auth callback for adapter use. - Adapters that fetch external context (e.g. Slack ``conversations.replies``) call this via - ``BasePlatformAdapter._is_sender_authorized`` to mark non-allowlisted senders as - unverified in LLM context (indirect prompt-injection mitigation). The returned callback - delegates to :meth:`_is_user_authorized` so the full auth chain — platform allowlists, - group allowlists, pairing store, allow-all flags — stays the single source of truth. - ``profile_name`` binds the callback to a secondary adapter's own profile so its - ``SessionSource`` resolves that secret scope. For the shared primary adapter under - multiplex (``profile_name`` None) the callback mirrors the inbound path: the - ``profile_routes`` match is stamped on the source so the routed profile's pairing store - is consulted, while allowlist reads stay under the transport home — without this a - caller approved only in the routed profile's pairing store was denied. + Adapters fetching external context (e.g. Slack ``conversations.replies``) use it via + ``_is_sender_authorized`` to mark non-allowlisted senders unverified (prompt-injection + mitigation). Delegates to :meth:`_is_user_authorized` so the full auth chain stays the single + source of truth. ``profile_name`` binds a secondary adapter to its own secret scope; for the + shared primary (None) the ``profile_routes`` match is stamped on the source so the routed + profile's pairing store is consulted while allowlist reads stay under the transport home. """ multiplex = bool(getattr(self.config, "multiplex_profiles", False)) transport_home = ( @@ -16838,10 +15639,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[SessionEntry]: """Resolve an async completion to its verified owning gateway session. - A compression rotation ends the physical parent row while continuing the same logical - conversation in a child. Follow that lineage, but never let a late completion override an - unrelated /new or restored route. Unknown ownership stays fail-closed; the result remains - available in the delegation records. + Follow compression-rotation lineage (parent row ended, child continues), but never let a + late completion override an unrelated /new or restored route. Unknown ownership fails + closed; the result stays in the delegation records. """ session_db = cast(Any, self._session_db) if session_db is None: @@ -16883,11 +15683,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return None if _end_reason != "compression": - # Idle/timeout/lifecycle end (scale-to-zero norm): the chat route remains valid and - # ``session_entry`` IS the routing key's current session for this same chat, so - # deliver the finished work there instead of dropping it. This is the delivery leg - # _classify_completion_target promises for non-boundary ends; without it the row is - # acked at adapter acceptance then silently dropped (falsely-acknowledged loss). + # Idle/timeout/lifecycle end (scale-to-zero norm): the chat route is still valid and + # ``session_entry`` is its current session, so deliver here rather than drop — otherwise + # the row is acked at adapter acceptance then silently lost. logger.info( "Async-delegation completion pinned to %s-ended session %s; " "retargeting to the chat's current session %s.", @@ -16997,17 +15795,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # ------------------------------------------------------------------ # Mid-run (busy-session) slash command dispatch — "Guard 2". - # - # Replaces the historical hand-written per-command if-chain: each - # command's mid-run behavior is declared on its CommandDef - # (busy_policy / busy_handler in hermes_cli/commands.py) and resolved - # here through a single handler table. Reply strings are byte-identical - # to the old chain. + # Each command's mid-run behavior is declared on its CommandDef (busy_policy / busy_handler + # in hermes_cli/commands.py) and resolved through a single handler table. # ------------------------------------------------------------------ - # Command-specific mid-run reject texts (busy_policy == "reject" with a - # busy_handler naming an entry here). All other rejected commands get - # the generic catch-all text in _dispatch_busy_slash_command. + # Command-specific mid-run reject texts (busy_policy == "reject" with a busy_handler naming an + # entry here); all other rejected commands get the generic text in _dispatch_busy_slash_command. _BUSY_REJECT_TEXT: Dict[str, str] = { "model": "Agent is running — wait or /stop first, then switch models.", "codex-runtime": ("Agent is running — wait or /stop first, then " @@ -17041,16 +15834,67 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "version": self._handle_version_command, } + async def _send_command_ack(self, source, text: str, label: str) -> None: + """Best-effort acknowledgment for a slash command that falls through to agent processing.""" + try: + adapter = self._adapter_for_source(source) + if adapter: + await adapter.send( + str(source.chat_id), text, metadata=self._thread_metadata_for_source(source) + ) + except Exception: + logger.debug("%s ack send failed", label, exc_info=True) + + def _gateway_idle_command_handlers(self): + """Slash handlers dispatched only when no agent is running for the session (idle path). + + Busy dispatch keeps its own explicit allowlist (``_dispatch_busy_slash_command``).""" + return { + "topic": self._handle_topic_command, + "whoami": self._handle_whoami_command, + "platform": self._handle_platform_command, + "stop": self._handle_stop_command, + "reasoning": self._handle_reasoning_command, + "memory": self._handle_memory_command, + "skills": self._handle_skills_command, + "fast": self._handle_fast_command, + "approvals": self._handle_approvals_command, + "model": self._handle_model_command, + "codex-runtime": self._handle_codex_runtime_command, + "personality": self._handle_personality_command, + "suggestions": self._handle_suggestions_command, + "save": self._handle_save_command, + "retry": self._handle_retry_command, + "sethome": self._handle_set_home_command, + "compress": self._handle_compress_command, + "usage": self._handle_usage_command, + "topup": self._handle_topup_command, + "insights": self._handle_insights_command, + "reload-mcp": self._handle_reload_mcp_command, + "reload-skills": self._handle_reload_skills_command, + "bundles": self._handle_bundles_command, + "debug": self._handle_debug_command, + "title": self._handle_title_command, + "resume": self._handle_resume_command, + "sessions": self._handle_sessions_command, + "branch": self._handle_branch_command, + "rollback": self._handle_rollback_command, + "diff": self._handle_diff_command, + "goal": self._handle_goal_command, + "loop": self._handle_loop_command, + "refine": self._handle_refine_command, + "review": self._handle_review_command, + "voice": self._handle_voice_command, + } + async def _dispatch_busy_slash_command( self, event: MessageEvent, cmd_def, quick_key: str, source, ): """Dispatch a recognized slash command while an agent is running. - Resolution order: 1. ``busy_handler`` — special mid-run variant (e.g. /goal's control- - verb whitelist, /queue's FIFO enqueue, /model's custom reject text). 2. ``busy_policy == - "dispatch"`` — the command's normal handler. 3. Catch-all busy-reject text. Rejecting is - required rather than falling through to interrupt + discard: Discord-registered slash - commands (/model, /reasoning, /reset, ...) would interrupt the agent AND be discarded by + Order: ``busy_handler`` (special mid-run variant) → ``busy_policy == "dispatch"`` (normal + handler) → catch-all busy-reject text. Rejecting beats falling through to interrupt + + discard: Discord-registered slash commands would interrupt the agent AND be discarded by the slash-command safety net, producing a zero-char response. """ name = cmd_def.name @@ -17083,21 +15927,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "falling back to busy-reject", policy, name, ) - # Catch-all: any other recognized slash command reached the - # running-agent guard. Reject gracefully rather than falling - # through to interrupt + discard. + # Catch-all: any other recognized slash command hit the running-agent guard — reject + # gracefully rather than falling through to interrupt + discard. return ( f"⏳ Agent is running — `/{name}` can't run " f"mid-turn. Wait for the current response or `/stop` first." ) async def _handle_pause_command(self, event: MessageEvent): - """`/pause [reason]` engages the global emergency stop; `/pause off` - (aliases: resume/stop) lifts it. + """`/pause [reason]` engages the global emergency stop; `/pause off` (resume/stop) lifts it. - This is the in-band resume path for messaging-only operators — the - estop gate above deliberately lets recognized slash commands through - while paused so a user without host-shell access is never locked out. + In-band resume path for messaging-only operators — the estop gate lets recognized slash + commands through while paused so a user without host-shell access is never locked out. """ from agent import estop @@ -17122,9 +15963,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) async def _busy_start_command(self, event: MessageEvent, quick_key: str, source): - # Telegram sends /start for bot launches/deep-links. Treat it as a - # platform ping, not a user command: no help dump, no agent - # interrupt, no queued text. + # Telegram sends /start for bot launches/deep-links — a platform ping, not a user command: + # no help dump, no agent interrupt, no queued text. logger.info("Ignoring /start platform ping for active session %s", quick_key) return "" @@ -17161,9 +16001,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return await self._handle_reset_command(event) async def _busy_queue_command(self, event: MessageEvent, quick_key: str, source): - # /queue — queue without interrupting. Semantics: each /queue invocation produces - # its own full agent turn, processed in FIFO order after the current run (and any earlier - # /queue items) finishes. Messages are NOT merged. + # /queue — queue without interrupting. Each /queue is its own full agent turn, run + # FIFO after the current run (and earlier /queue items) finish; messages are NOT merged. queued_text = event.get_command_args().strip() # Preserve media/reply payloads: a /queue carrying a photo, document, or reply context is # valid even with no prompt text (e.g. "/queue" as the caption of an image). Dropping these @@ -17252,9 +16091,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # /model so we don't race a second continuation prompt against the current turn. _goal_arg = (event.get_command_args() or "").strip().lower() _goal_verb = _goal_arg.split(None, 1)[0] if _goal_arg else "" - # Exact-match control verbs (unchanged semantics), plus the wait/unwait barrier verbs which - # take a pid argument and the gate management verb (inspection/mutation of the gate list - # only — gates run at turn boundary, so editing them mid-run is safe). + # Exact-match control verbs, plus the wait/unwait barrier verbs (take a pid) and the gate + # management verb (gates run at turn boundary, so editing the gate list mid-run is safe). _is_control = ( not _goal_arg or _goal_arg in {"status", "pause", "resume", "clear", "stop", "done", "unwait"} @@ -17265,9 +16103,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return "Agent is running — use /goal status / pause / clear / wait mid-run, or /stop before setting a new goal." async def _busy_loop_command(self, event: MessageEvent, quick_key: str, source): - # /loop mirrors /goal: control verbs are safe mid-run (state - # only — read at the next idle boundary); setting a new loop - # mid-run is rejected so we don't race the current turn. + # /loop mirrors /goal: control verbs are safe mid-run (state only — read at the next idle + # boundary); setting a new loop mid-run is rejected so we don't race the current turn. _loop_arg = (event.get_command_args() or "").strip().lower() if not _loop_arg or _loop_arg in {"status", "pause", "resume", "stop", "clear", "cancel", "help", "--help", "-h"}: return await self._handle_loop_command(event) @@ -17276,23 +16113,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _handle_message(self, event: MessageEvent) -> Optional[str]: """Handle an incoming message from any platform. - This is the core message processing pipeline: 1. Check user authorization 2. Check for - commands (/new, /reset, etc.) 3. Check for running agent and interrupt if needed 4. Get - or create session 5. Build context for agent 6. Run agent conversation 7. Return response + Pipeline: auth → command check → running-agent interrupt → get/create session → build + context → run agent → return response. """ source = event.source - # 🔴 Cross-session leak guard. This handler runs inside a per-message - # asyncio task created via create_task(), which snapshots the spawning - # context with copy_context(). If a *concurrent* message had already - # bound its session via set_session_vars() when this task was created, - # we inherited ITS HERMES_SESSION_* ContextVars. Until we bind our own - # (a few steps down, in _set_session_env), any subprocess spawned here - # would read the foreign session's identity via the subprocess-env - # bridge — the _UNSET-strip guard there can't help because the vars are - # set-to-foreign, not _UNSET. Reset to _UNSET now so that window strips - # safe (no session) instead of leaking the sibling's. See - # gateway/session_context.reset_session_vars + the inheritance test. + # 🔴 Cross-session leak guard. This per-message task was created via create_task(), which + # copies the spawning context: if a concurrent message had already bound its session via + # set_session_vars(), we inherited ITS HERMES_SESSION_* ContextVars, and until _set_session_env + # binds ours any subprocess would read the foreign identity (the _UNSET-strip guard can't + # help — the vars are set-to-foreign). Reset to _UNSET so that window strips safe instead. try: from gateway.session_context import reset_session_vars reset_session_vars() @@ -17329,9 +16159,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew is_internal = bool(getattr(event, "internal", False)) # Ignored-channel guard runs FIRST — before startup-restore queueing, plugin hooks, auth, - # and session setup — so a configured ignored channel can never reach pairing/auth/session - # state. getattr: bare test runners construct GatewayRunner via object.__new__ without - # config (see AGENTS.md pitfall on object.__new__ test pattern). + # and session setup — so an ignored channel can never reach pairing/auth/session state. + # getattr: bare test runners construct GatewayRunner via object.__new__ without config. if ( not is_internal and getattr(source, "platform", None) == Platform.SLACK @@ -17353,21 +16182,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._queue_startup_restore_event(event) return None - # scale-to-zero (Phase 0, 0.B/F13): stamp the gateway-scoped last-inbound - # clock for real (user-originated) inbound only. Internal/system events - # (background-process completions, startup-restore replays) are NOT - # traffic — counting them would keep a genuinely idle gateway awake. This - # clock is what the idle predicate (gateway/scale_to_zero.is_idle) reads. + # scale-to-zero: stamp the gateway-scoped last-inbound clock (read by is_idle) for real + # user-originated inbound only. Internal/system events are NOT traffic — counting them + # would keep a genuinely idle gateway awake. if not is_internal: self._scale_to_zero_note_real_inbound() - # Fire pre_gateway_dispatch plugin hook for user-originated messages. - # Plugins receive the MessageEvent and may return a dict influencing flow: - # {"action": "skip", "reason": ...} -> drop (no reply, plugin handled) - # {"action": "rewrite", "text": ...} -> replace event.text, continue - # {"action": "allow"} / None -> normal dispatch - # Hook runs BEFORE auth so plugins can handle unauthorized senders - # (e.g. customer handover ingest) without triggering the pairing flow. + # pre_gateway_dispatch plugin hook (user-originated only). Plugins may return + # {"action": "skip", "reason": ...} -> drop; {"action": "rewrite", "text": ...} -> replace + # event.text; {"action": "allow"} / None -> normal dispatch. + # Runs BEFORE auth so plugins can handle unauthorized senders without the pairing flow. if not is_internal: try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -17375,9 +16199,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "pre_gateway_dispatch", event=event, gateway=self, - # getattr: bare-runner tests build GatewayRunner via - # object.__new__ without __init__ (pitfall #17), and the - # hook must not fail dispatch over a missing attribute. + # getattr: bare-runner tests build GatewayRunner via object.__new__ without + # __init__; the hook must not fail dispatch over a missing attribute. session_store=getattr(self, "session_store", None), ) except Exception as _hook_exc: @@ -17409,9 +16232,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew pass elif source.user_id is None: # Messages with no user identity (Telegram service messages, channel forwards, anonymous - # admin posts, sender_chat) can't be paired, but they can still be authorized via a - # chat-scoped allowlist (e.g. TELEGRAM_GROUP_ALLOWED_CHATS authorizes every member of - # the listed chat). Defer to _is_user_authorized so that path runs. + # admin posts, sender_chat) can't be paired but may be authorized via a chat-scoped + # allowlist (e.g. TELEGRAM_GROUP_ALLOWED_CHATS), so defer to _is_user_authorized. if not self._is_user_authorized_for_source(source): logger.debug("Ignoring message with no user_id from %s", source.platform.value) return None @@ -17434,9 +16256,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew platform_name, ) return None - # Rate-limit ALL pairing responses (code or rejection) to - # prevent spamming the user with repeated messages when - # multiple DMs arrive in quick succession. + # Rate-limit ALL pairing responses (code or rejection) so a burst of DMs doesn't + # spam the user with repeated messages. if pairing_store._is_rate_limited(platform_name, source.user_id): return None code = pairing_store.generate_code( @@ -17473,22 +16294,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew pairing_store._record_rate_limit(platform_name, source.user_id) return None - # Global emergency stop (`hermes pause`): give new turns a brief - # paused notice instead of starting an agent run. Internal events - # (background-process completions from IN-FLIGHT work) bypass the - # gate — pause stops NEW work, it never kills or orphans running - # work. Placed after auth so unauthorized senders keep the normal - # silent/pairing behavior and can't probe pause state. - # - # Passthroughs (pause blocks new AGENT turns, not control traffic): - # * recognized slash commands — /status, /help, /new, /approve and - # friends must keep working while paused, and /pause off is the - # in-band resume path for messaging-only users; - # * replies owned by IN-FLIGHT work — a pending detached-update - # prompt, clarify, slash-confirm, or dangerous-command approval, - # plus any message steering a session whose agent is already - # running. Swallowing those would stall work the pause promised - # not to touch. + # Global emergency stop (`hermes pause`): new turns get a brief paused notice instead of an + # agent run. Placed after auth so unauthorized senders can't probe pause state. Pause blocks + # NEW agent turns, never running work or control traffic, so these pass through: internal + # events from IN-FLIGHT work; recognized slash commands (/status, /approve, ... and /pause off + # as the in-band resume path); replies owned by in-flight work — pending update prompt, + # clarify, slash-confirm, dangerous-command approval, or steering an already-running session. if not is_internal: try: from agent.estop import paused_reply as _estop_paused_reply @@ -17520,9 +16331,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): _estop_allow = True if not _estop_allow and self._is_session_running(_estop_key): - # Steering / interrupting in-flight work (which - # also covers pending clarify + tool approvals - # held by the running agent). + # Steering / interrupting in-flight work (also covers pending clarify + + # tool approvals held by the running agent). _estop_allow = True if not _estop_allow: from tools import slash_confirm as _estop_confirm_mod @@ -17544,11 +16354,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return _paused_notice - # Intercept messages that are responses to a pending /update prompt. The update process - # (detached) wrote .update_prompt.json; the watcher forwarded it to the user; now the user's - # reply goes back via .update_response so the update process can continue. IMPORTANT: - # recognized slash commands must bypass this interception or /new, /help etc. get silently - # consumed as update answers. + # Route replies to a pending /update prompt back to the detached update process via + # .update_response. Recognized slash commands must bypass this or /new, /help etc. get + # silently consumed as update answers. _quick_key = self._session_key_for_source(source) allow_gateway_control = event.allow_gateway_control _up_state = self._peek_session_state(_quick_key) @@ -17577,10 +16385,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _recognized_cmd = _cmd_def.name if _cmd_def else None except Exception: _recognized_cmd = None - if _recognized_cmd: - response_text = "" - else: - response_text = raw + response_text = "" if _recognized_cmd else raw if response_text: response_path = _hermes_home / ".update_response" prompt_path = _hermes_home / ".update_prompt.json" @@ -17595,10 +16400,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _up_state.persistent.update_prompt_pending = False label = response_text if len(response_text) <= 20 else response_text[:20] + "…" return f"✓ Sent `{label}` to the update process." - # Recognized slash command during a pending update prompt: unblock the detached update - # subprocess by writing a blank response so ``_gateway_prompt`` returns the prompt's - # default (typically a safe "n" / skip) and exits cleanly instead of blocking on stdin - # until the 30-minute watcher timeout. + # Recognized slash command during a pending update prompt: write a blank response so the + # detached update's ``_gateway_prompt`` returns the prompt's default (typically a safe + # "n" / skip) and exits instead of blocking on stdin until the watcher timeout. if _recognized_cmd: response_path = _hermes_home / ".update_response" prompt_path = _hermes_home / ".update_prompt.json" @@ -17620,9 +16424,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) _up_state.persistent.update_prompt_pending = False - # Intercept messages that are responses to a pending clarify. Open-ended prompts and "Other" - # responses are captured as free text; direct replies to multi-choice prompts are accepted - # too ("2" maps to the second option). + # Intercept replies to a pending clarify: open-ended prompts and "Other" responses are free + # text; direct replies to multi-choice prompts are accepted too ("2" → second option). _clarify_mod = None try: from tools import clarify_gateway as _clarify_mod @@ -17646,9 +16449,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _pending_clarify.clarify_id, ) return "" - # Skip slash commands — the user clearly wanted to issue a command, not answer the - # clarify. Leave the clarify pending so the user can retry; if it times out, the agent - # unblocks with an empty response. + # Skip slash commands — the user wanted a command, not to answer the clarify. Leave it + # pending so they can retry; on timeout the agent unblocks with an empty response. if _raw_clarify_reply and not _raw_clarify_reply.startswith("/"): _text_outcome = _clarify_mod.attempt_text_response_for_session( _quick_key, _raw_clarify_reply, @@ -17670,14 +16472,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Failed to resume typing after clarify response", exc_info=True, ) - # Acknowledge with empty string so adapters that emit - # the agent's response don't double-post. The agent - # itself will produce the next user-facing message. + # Acknowledge with empty string so adapters that emit the agent's response don't + # double-post; the agent itself produces the next user-facing message. return "" if _text_outcome == _clarify_mod.TEXT_REJECTED_SELECTION: - # Selection-shaped but invalid (out-of-range number, unrecognised comma-list). - # Keep the clarify armed so the user can retry — do not cancel and do not treat - # this as an unrelated follow-up turn. + # Selection-shaped but invalid (out-of-range number, bad comma-list): keep the + # clarify armed for retry — don't cancel, don't treat as an unrelated follow-up. logger.info( "Gateway retained pending clarify after invalid " "selection attempt (session=%s, id=%s)", @@ -17694,16 +16494,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "", ) - # Intercept messages that are responses to a pending /reload-mcp - # (or future) slash-confirm prompt. Recognized confirm replies are - # /approve, /always, /cancel (plus short aliases). Anything else - # falls through to normal dispatch — a stale pending confirm does - # NOT block other commands. - # - # Important: if a dangerous-command approval is ALSO pending (agent - # blocked inside tools/approval.py), the tool approval takes - # precedence — /approve there unblocks the waiting tool thread. - # Slash-confirm only catches /approve when no tool approval is live. + # Replies to a pending slash-confirm prompt (/reload-mcp etc.): /approve, /always, /cancel and + # short aliases. Anything else falls through — a stale pending confirm does NOT block other + # commands. A pending dangerous-command approval takes precedence: /approve there unblocks + # the waiting tool thread; slash-confirm only catches it when no tool approval is live. from tools import slash_confirm as _slash_confirm_mod _pending_confirm = _slash_confirm_mod.get_pending(_quick_key) _tool_approval_live = False @@ -17737,9 +16531,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _quick_key, _pending_confirm.get("confirm_id"), _confirm_choice, ) return _resolved or "" - # Stale pending + unrelated command: drop the pending state so - # the confirm doesn't block normal usage indefinitely. The user - # clearly moved on. + # Stale pending + unrelated command: the user moved on, so drop the pending state rather + # than let the confirm block normal usage indefinitely. _slash_confirm_mod.clear_if_stale(_quick_key) # PRIORITY handling when an agent is already running for this session. Default behavior is @@ -17748,19 +16541,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # for photo-only follow-ups; adapter-level batching absorbs them. # Staleness eviction: detect leaked locks from hung/crashed handlers. With inactivity-based - # timeout, active tasks can run for hours, so wall-clock age alone isn't sufficient. Evict - # only when the agent has been *idle* past the threshold (or has no activity tracker and - # wall-clock age is extreme). + # timeout active tasks can run for hours, so evict only when the agent has been *idle* past + # the threshold (or has no activity tracker and its wall-clock age is extreme). _raw_stale_timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) _quick_state = self._peek_session_state(_quick_key) _stale_ts = _quick_state.turn.started_ts if _quick_state else 0 if _quick_state is not None and _quick_state.turn.agent is not None and _stale_ts: _stale_age = time.time() - _stale_ts _stale_agent = _quick_state.turn.agent - # Never evict the pending sentinel — it was just placed moments ago during the async - # setup phase before the real agent is created. Sentinels have no - # get_activity_summary(), so the idle check would read inf >= timeout and evict them - # immediately, racing the setup path. + # Never evict the pending sentinel — it was just placed during async setup before the + # real agent exists. Sentinels have no get_activity_summary(), so the idle check would + # read inf >= timeout and evict them immediately, racing the setup path. _stale_idle = float("inf") # assume idle if we can't check _stale_detail = "" _activity_summary_valid = False @@ -17816,12 +16607,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) self._release_running_agent_state(_quick_key) - # Durable-reaped guard. A session whose routing row was ended in state.db (e.g. - # ``ws_orphan_reap`` / ``agent_close``) while the gateway stayed alive keeps its in-memory - # turn slot (``_is_session_running`` stays True). The priority fast-path would then queue - # every next user message into the dead runtime instead of healing the routing via - # ``get_or_create_session`` → ``reopen``. Evict the stale slot so the next message falls - # through to the cold path and re-attaches or creates a fresh session. + # Durable-reaped guard. A session whose routing row was ended in state.db (``ws_orphan_reap`` + # / ``agent_close``) while the gateway lived keeps its in-memory turn slot, so the fast-path + # would queue every next message into the dead runtime. Evict the stale slot so the cold + # path re-attaches via ``get_or_create_session`` → ``reopen`` or creates a fresh session. if self._is_session_running(_quick_key): try: _reap_store = getattr(self, "session_store", None) @@ -17854,10 +16643,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("reaped-session staleness check failed", exc_info=True) if self._is_session_running(_quick_key): - # Resolve the command once; every command's mid-run behavior is declared on its - # CommandDef (busy_policy / busy_handler in hermes_cli/commands.py) and dispatched - # through the single resolver _dispatch_busy_slash_command below — no per-command if- - # chain here. + # Resolve the command once; each command's mid-run behavior is declared on its + # CommandDef (busy_policy / busy_handler in hermes_cli/commands.py) and dispatched via + # _dispatch_busy_slash_command below — no per-command if-chain here. from hermes_cli.commands import resolve_command as _resolve_cmd_inner _evt_cmd = event.get_command() _cmd_def_inner = _resolve_cmd_inner(_evt_cmd) if _evt_cmd else None @@ -17994,14 +16782,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) self._queue_or_replace_pending_event(_quick_key, event) return None - # #56391 — Compression protection (PRIORITY path). Same - # rationale as ``_handle_active_session_busy_message``: context - # compression is interrupt-protected (#23975), but an interrupt - # here starts a new turn against the pre-rotation parent - # session while the still-running compression later rotates - # the id out from under it, forking orphaned compression - # siblings. Demote to queue semantics so the follow-up waits - # for the in-flight compression + rotation to land. + # Compression protection (PRIORITY path), as in ``_handle_active_session_busy_message``: + # an interrupt would start a new turn on the pre-rotation parent while compression + # rotates the id away, forking orphaned siblings. Demote to queue until rotation lands. if await self._session_has_compression_in_flight(_quick_key): logger.info( "PRIORITY interrupt demoted to queue for session %s " @@ -18010,9 +16793,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) self._queue_or_replace_pending_event(_quick_key, event) return None - # Text-only corrections redirect the live turn (preserving - # displayed context) when the runtime supports it; media/voice and - # older runtimes fall back to the proven interrupt path below. + # Text-only corrections redirect the live turn (preserving displayed context) when the + # runtime supports it; media/voice and older runtimes use the interrupt path below. if ( event.message_type == MessageType.TEXT and not event.media_urls @@ -18045,9 +16827,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew elif not _interrupt_text and _media_urls: _interrupt_text = _build_media_placeholder(event) running_agent.interrupt(_interrupt_text) - # NOTE: self._pending_messages was write-only (never consumed). - # The actual interrupt message is delivered via adapter._pending_messages - # which is read by _run_agent. Removed to prevent unbounded growth. + # The interrupt message is delivered via adapter._pending_messages (read by _run_agent); + # don't also buffer it on self — that copy was never consumed and grew unbounded. return None # Check for commands @@ -18094,11 +16875,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _denied is not None: return _denied - # pre_command observer hook: fires for every recognized slash command BEFORE core handling, - # mirroring the CLI fire-site in cli.py process_command. Observer-only in v1 (returns - # ignored). Placement matters: the running-agent intercept path above (/stop, /approve, - # busy_policy dispatch) deliberately does NOT fire it — a slow or hostile plugin must not - # be able to interfere with the operator's escape hatches for a live agent. + # pre_command observer hook (returns ignored) fires for every recognized slash command + # BEFORE core handling, mirroring cli.py. The running-agent intercept path above (/stop, + # /approve, busy_policy) deliberately does NOT fire it — a slow or hostile plugin must not + # interfere with the operator's escape hatches for a live agent. if command and is_gateway_known_command(canonical): try: from hermes_cli.plugins import fire_pre_command_hook @@ -18116,13 +16896,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _pre_cmd_err, ) - # Fire the ``command:`` hook for any recognized slash - # command — built-in OR plugin-registered. Handlers can return a - # dict with ``{"decision": "deny" | "handled" | "rewrite", ...}`` - # to intercept dispatch before core handling runs. This replaces - # the previous fire-and-forget emit(): return values are now - # honored, but handlers that return nothing behave exactly as - # before (telemetry-style hooks keep working). + # Fire ``command:`` for any recognized slash command (built-in or plugin). + # Handlers may return ``{"decision": "deny" | "handled" | "rewrite", ...}`` to intercept + # dispatch; handlers returning nothing behave as plain observers. if command and is_gateway_known_command(canonical): raw_args = event.get_command_args().strip() hook_ctx = { @@ -18171,7 +16947,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew canonical = _cmd_def.name if _cmd_def else command break - plain_handler = self._gateway_plain_command_handlers().get(canonical) + plain_handler = ( + self._gateway_plain_command_handlers().get(canonical) + or self._gateway_idle_command_handlers().get(canonical) + ) if plain_handler is not None: return await plain_handler(event) @@ -18191,36 +16970,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew execute=_do_reset, ) - if canonical == "topic": - return await self._handle_topic_command(event) - if canonical == "start": logger.info("Ignoring /start platform ping for session %s", _quick_key) return "" - if canonical == "whoami": - return await self._handle_whoami_command(event) - if canonical == "egress": from hermes_cli.proxy_cli import format_status_text return format_status_text() - if canonical == "platform": - return await self._handle_platform_command(event) - - if canonical == "stop": - return await self._handle_stop_command(event) - - if canonical == "reasoning": - return await self._handle_reasoning_command(event) - - if canonical == "memory": - return await self._handle_memory_command(event) - - if canonical == "skills": - return await self._handle_skills_command(event) - if canonical == "learn": # Open-ended: rewrite the turn to a standards-guided prompt and fall through to normal # agent processing. Mirrors the /blueprint fall-through so role alternation is @@ -18233,13 +16991,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _learn_req else "Learning a skill from this conversation…" ) - try: - adapter = self._adapter_for_source(source) - if adapter: - _ack_meta = self._thread_metadata_for_source(source) - await adapter.send(str(source.chat_id), _ack, metadata=_ack_meta) - except Exception: - logger.debug("learn ack send failed", exc_info=True) + await self._send_command_ack(source, _ack, "learn") try: event.text = build_learn_prompt(_learn_req) # fall through to agent processing @@ -18248,8 +17000,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if canonical == "plan": # /plan: rewrite the turn to the plan-mode prompt and fall through to normal agent - # processing (same fall-through as /learn so role alternation is preserved). No engine, - # works on any backend. + # processing (the /learn fall-through keeps role alternation). Works on any backend. from agent.plan_prompt import build_plan_prompt _plan_task = event.get_command_args().strip() @@ -18258,13 +17009,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _plan_task else "Planning from this conversation's context…" ) - try: - adapter = self._adapter_for_source(source) - if adapter: - _ack_meta = self._thread_metadata_for_source(source) - await adapter.send(str(source.chat_id), _ack, metadata=_ack_meta) - except Exception: - logger.debug("plan ack send failed", exc_info=True) + await self._send_command_ack(source, _ack, "plan") try: event.text = build_plan_prompt(_plan_task) # fall through to agent processing @@ -18273,8 +17018,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if canonical == "init": # /init: rewrite the turn to a guidance-laden prompt and fall through to normal agent - # processing (same fall-through as /learn so role alternation is preserved). No engine, - # works on any backend. + # processing (the /learn fall-through keeps role alternation). Works on any backend. from hermes_cli.init_command import build_init_prompt_for_cwd _init_notes = event.get_command_args().strip() @@ -18287,51 +17031,20 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if "UPDATE the existing AGENTS.md" in _init_prompt else "Generating AGENTS.md from a project scan…" ) - try: - adapter = self._adapter_for_source(source) - if adapter: - _ack_meta = self._thread_metadata_for_source(source) - await adapter.send(str(source.chat_id), _ack, metadata=_ack_meta) - except Exception: - logger.debug("init ack send failed", exc_info=True) + await self._send_command_ack(source, _ack, "init") event.text = _init_prompt # fall through to agent processing - if canonical == "fast": - return await self._handle_fast_command(event) - - if canonical == "approvals": - return await self._handle_approvals_command(event) - - if canonical == "model": - return await self._handle_model_command(event) - - if canonical == "codex-runtime": - return await self._handle_codex_runtime_command(event) - - if canonical == "personality": - return await self._handle_personality_command(event) - - if canonical == "suggestions": - return await self._handle_suggestions_command(event) - if canonical == "blueprint": _blueprint_result = await self._handle_blueprint_command(event) _blueprint_seed = getattr(_blueprint_result, "agent_seed", None) if _blueprint_seed: # Blueprint matched — rewrite the turn to the seed and fall through to - # _handle_message_with_agent so the agent asks the user for each slot value - # conversationally and then calls the cronjob tool (the /steer fall-through - # pattern). + # _handle_message_with_agent so the agent collects each slot value conversationally, + # then calls the cronjob tool (the /steer fall-through pattern). _ack = getattr(_blueprint_result, "text", "") or "" if _ack: - try: - adapter = self._adapter_for_source(source) - if adapter: - _ack_meta = self._thread_metadata_for_source(source) - await adapter.send(str(source.chat_id), _ack, metadata=_ack_meta) - except Exception: - logger.debug("blueprint ack send failed", exc_info=True) + await self._send_command_ack(source, _ack, "blueprint") try: event.text = _blueprint_seed except Exception: @@ -18339,12 +17052,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew else: return getattr(_blueprint_result, "text", "") or None - if canonical == "save": - return await self._handle_save_command(event) - - if canonical == "retry": - return await self._handle_retry_command(event) - if canonical == "undo": async def _do_undo(): return await self._handle_undo_command(event) @@ -18368,85 +17075,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew execute=_do_undo, ) - if canonical == "sethome": - return await self._handle_set_home_command(event) - - if canonical == "compress": - return await self._handle_compress_command(event) - - if canonical == "usage": - return await self._handle_usage_command(event) - - if canonical == "topup": - return await self._handle_topup_command(event) - - if canonical == "insights": - return await self._handle_insights_command(event) - - if canonical == "reload-mcp": - return await self._handle_reload_mcp_command(event) - - if canonical == "reload-skills": - return await self._handle_reload_skills_command(event) - - if canonical == "bundles": - return await self._handle_bundles_command(event) - - if canonical == "debug": - return await self._handle_debug_command(event) - - if canonical == "title": - return await self._handle_title_command(event) - - if canonical == "resume": - return await self._handle_resume_command(event) - - if canonical == "sessions": - return await self._handle_sessions_command(event) - - if canonical == "branch": - return await self._handle_branch_command(event) - - if canonical == "rollback": - return await self._handle_rollback_command(event) - - if canonical == "diff": - return await self._handle_diff_command(event) - if canonical == "queue": queue_payload = event.get_command_args().strip() if not queue_payload: return "Usage: /queue " - try: + with suppress(Exception): event.text = queue_payload - except Exception: - pass if canonical == "steer": - # No active agent — /steer has no tool call to inject into. - # Strip the prefix so downstream treats it as a normal user - # message. If the payload is empty, surface the usage hint. + # No active agent — /steer has nothing to inject into. Strip the prefix so downstream + # treats it as a normal user message; an empty payload surfaces the usage hint. steer_payload = event.get_command_args().strip() if not steer_payload: return "Usage: /steer (no agent is running; sending as a normal message)" - try: + with suppress(Exception): event.text = steer_payload - except Exception: - pass - # Do NOT return — fall through to _handle_message_with_agent - # at the end of this function so the rewritten text is sent - # to the agent as a regular user turn. - - if canonical == "goal": - return await self._handle_goal_command(event) - - if canonical == "loop": - return await self._handle_loop_command(event) - - if canonical == "refine": - return await self._handle_refine_command(event) - if canonical == "review": - return await self._handle_review_command(event) + # Do NOT return — fall through to _handle_message_with_agent at the end of this function + # so the rewritten text is sent to the agent as a regular user turn. if canonical == "moa": # /moa is one-shot sugar only: run a single prompt through the default MoA preset, then @@ -18483,9 +17128,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: return "Failed to prepare MoA turn." - if canonical == "voice": - return await self._handle_voice_command(event) - if self._draining: return f"⏳ Gateway is {self._status_action_gerund()} and is not accepting new work right now." @@ -18510,9 +17152,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew exec_cmd = qcmd.get("command", "") if exec_cmd: try: - # Sanitize env to prevent credential leakage — - # quick commands run in the gateway process which - # has all API keys in os.environ. + # Sanitize env to prevent credential leakage — quick commands run in the + # gateway process, which has all API keys in os.environ. from tools.environments.local import build_subprocess_env sanitized_env = build_subprocess_env() proc = await asyncio.create_subprocess_shell( @@ -18552,9 +17193,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if command: try: from hermes_cli.plugins import get_plugin_command_handler - # Normalize underscores to hyphens so Telegram's underscored - # autocomplete form matches plugin commands registered with - # hyphens. See hermes_cli/commands.py:_build_telegram_menu. + # Normalize underscores to hyphens so Telegram's underscored autocomplete form + # matches plugin commands registered with hyphens (see _build_telegram_menu). plugin_handler = get_plugin_command_handler(command.replace("_", "-")) if plugin_handler: user_args = event.get_command_args().strip() @@ -18625,9 +17265,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"Enable it with: `hermes skills config`" ) user_instruction = event.get_command_args().strip() - # Stacked slash-skill invocations: `/skill-a /skill-b do - # XYZ` loads every leading skill (up to 5), not just the - # first. Inspired by Claude Code v2.1.199. Mirrors CLI. + # Stacked slash-skill invocations: `/skill-a /skill-b do XYZ` loads every + # leading skill (up to 5), not just the first. Mirrors CLI. try: from agent.skill_commands import ( build_stacked_skill_invocation_message as _build_stacked, @@ -18682,11 +17321,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _unavail_msg = _check_unavailable_skill(command) if _unavail_msg: return _unavail_msg - # Genuinely unrecognized /command: not a built-in, not a plugin, not a skill, - # not a known-inactive skill. Warn instead of silently forwarding it to the LLM as - # free text (which leads to the model inventing e.g. a delegate_task call). - # Normalize to hyphenated form first: command may be an alias target set by the - # quick-command block above, so _cmd_def can be stale. + # Genuinely unrecognized /command (not built-in/plugin/skill/known-inactive): + # warn instead of forwarding to the LLM as free text (it invents tool calls). + # Normalize to hyphenated form first: the quick-command block may have set an + # alias target, so _cmd_def can be stale. if command.replace("_", "-") not in GATEWAY_KNOWN_COMMANDS: logger.warning( "Unrecognized slash command /%s from %s — " @@ -18703,9 +17341,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.debug("Skill command check failed (non-fatal): %s", e) - # Pending exec approvals are handled by /approve and /deny commands above. - # No bare text matching — "yes" in normal conversation must not trigger - # execution of a dangerous command. + # Pending exec approvals go through /approve and /deny only — no bare-text matching, or a + # conversational "yes" would execute a dangerous command. if not is_internal and await asyncio.to_thread( self._is_telegram_topic_root_lobby, source @@ -18716,12 +17353,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return self._telegram_topic_root_lobby_message() return None - # ── External-drain new-turn gate (Phase 2) ──────────────────── - # When NAS has engaged an external drain (.drain_request.json present, observed by - # _drain_control_watcher), refuse to START a new turn so the in-flight set can only fall to - # zero — eliminating the TOCTOU race (D4a: stop accepting new turns FIRST, then NAS polls - # until active_agents==0). In-flight turns are untouched; internal/system events bypass the - # gate (not user-initiated, must still flow during a drain). Reversible once the marker goes. + # ── External-drain new-turn gate ───────────────────────────── + # When NAS engaged an external drain (.drain_request.json, seen by _drain_control_watcher), + # refuse to START new turns so the in-flight set can only fall to zero (stop accepting + # FIRST, then NAS polls active_agents==0). Internal/system events bypass; reversible. if self._external_drain_active and not is_internal: logger.info( "Refusing new turn for session %s — external drain active.", @@ -18734,10 +17369,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) # ── Claim this session before any await ─────────────────────── - # Between here and _run_agent registering the real AIAgent, there are numerous await points - # (hooks, vision enrichment, STT, session hygiene compression). Without this sentinel a - # second message arriving during any of those yields would pass the "already running" guard - # and spin up a duplicate agent for the same session — corrupting the transcript. + # Many awaits sit between here and _run_agent registering the real AIAgent; without this + # sentinel a second message during any of them passes the "already running" guard and spins + # up a duplicate agent for the same session, corrupting the transcript. _active_session_lease, _limit_message = self._claim_active_session_slot( _quick_key, source, @@ -18749,12 +17383,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return _limit_message - # ── FIFO orphan rescue (#99882) ──────────────────────────────── - # If this session went idle with a populated overflow (queued during a busy window whose - # post-turn drain never promoted — e.g. a compression-demoted follow-up after the - # compression window ended through an exit that skipped the promotion site), those events - # were silently orphaned. Re-stage them in FIFO order and enqueue the incoming event behind - # them so arrival order holds. Skipped for control commands (/stop etc.) and internal events. + # ── FIFO orphan rescue ─────────────────────────────────────── + # A session that went idle with a populated overflow (post-turn drain never promoted, e.g. a + # compression-demoted follow-up) silently orphaned those events. Re-stage them FIFO and + # enqueue this event behind them. Skipped for control commands and internal events. try: _orphan_adapter = self._adapter_for_source(source) if ( @@ -18771,9 +17403,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # picks it up), otherwise into overflow behind the already-staged next orphan. self._enqueue_fifo(_quick_key, event, _orphan_adapter) event = _rescued - # Same session key by construction; carry the orphan's - # own source so reply anchors / thread metadata point - # at the message that is actually being answered. + # Same session key by construction; carry the orphan's own source so reply + # anchors / thread metadata point at the message actually being answered. _rescued_source = getattr(_rescued, "source", None) if _rescued_source is not None: source = _rescued_source @@ -18799,9 +17430,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew event, source, _quick_key, _run_generation ) except TurnLeaseTimeoutError as exc: - # This is a rejected message, not a completed agent turn. Return - # before the /goal judge below so it cannot consume the resend - # notice and enqueue a synthetic continuation loop. + # A rejected message, not a completed turn: return before the /goal judge below so + # it cannot consume the resend notice and enqueue a synthetic continuation loop. logger.error( "Rejecting turn for routing key %s on session %s after " "turn-lease timeout; transcript load was not started and " @@ -18825,34 +17455,29 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("post-turn hook failed: %s", _goal_exc) return _agent_result finally: - # MoA one-shot restore must run on EVERY exit path, not just success. The restore data - # lives on the per-turn event object; if the handler raised, a restore in the try block - # would be skipped and the MoA override would leak permanently. Putting it in - # finally guarantees the revert on success, exception, and interrupt alike. + # MoA one-shot restore must run on EVERY exit path: the restore data lives on the + # per-turn event, so a restore in the try block is skipped when the handler raises and + # the override leaks permanently; finally covers success, exception and interrupt. self._restore_moa_one_shot(event, _quick_key) self._restore_pending_one_turn_model_override(_quick_key) - # Normal completion/exception/interrupt owns and clears this exact - # durable marker. SIGKILL/OOM skips finally, leaving the marker for - # the next unclean startup's recovery pass. + # Normal completion/exception/interrupt clears this durable marker; SIGKILL/OOM skips + # finally, leaving it for the next unclean startup's recovery pass. await self._clear_durable_active_turn(event) - # Unconditional release covers every exit path. _release_running_agent_state is - # idempotent (pop-on-absent is harmless) and, called without a run_generation guard, - # always clears the slot regardless of which generation it holds. This evicts the zombie - # left when session_reset bumps the generation mid-flight: gen-N's guarded release in - # _run_agent returns False, and a sentinel-only check here would lock the session forever. + # Unconditional release covers every exit path: _release_running_agent_state is idempotent + # and, without a run_generation guard, clears the slot whichever generation holds it. This + # evicts the zombie left when session_reset bumps the generation mid-flight (gen-N's + # guarded release in _run_agent returns False; a sentinel-only check would lock forever). self._release_running_agent_state(_quick_key) - # Turn lease (#64934): release THIS turn's lease token — keyed by - # (routing key, run generation) so this unwind can only ever free - # the lease its own turn acquired, never a newer turn's. + # Turn lease: release THIS turn's token — keyed by (routing key, run generation) so this + # unwind can only free the lease its own turn acquired, never a newer turn's. self._release_turn_lease(_quick_key, _run_generation) def _restore_moa_one_shot(self, event: "MessageEvent", quick_key: str) -> None: """Revert a ``/moa `` one-shot model override after its turn. - Called from the ``finally`` of the message-handling path so the revert fires whether the - turn succeeded, raised, or was interrupted. A no-op unless - ``event._moa_disable_after_turn`` is set. ``_moa_restore_override`` carries the prior - per-session override (``None`` = no override, so the MoA override is cleared outright). + Called from the message-handling ``finally`` so it fires on success, error or interrupt. + No-op unless ``event._moa_disable_after_turn``; ``_moa_restore_override`` holds the prior + per-session override (``None`` = clear the MoA override outright). """ if not getattr(event, "_moa_disable_after_turn", False): return @@ -18888,13 +17513,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[str]: """Prepare inbound event text for the agent. - Keep the normal inbound path and the queued follow-up path on the same preprocessing - pipeline so sender attribution, image enrichment, STT, document notes, reply context, and - @ references all behave the same. - - Side effect: buffers per-session native image paths when the model supports native vision - and images are attached; the caller consumes/clears that buffer at the ``run_conversation`` - site. When the list is empty, the ``_enrich_message_with_vision`` text path already ran. + Shared by the normal inbound and queued follow-up paths so attribution, image enrichment, + STT, document notes, reply context and @ references behave the same. Side effect: buffers + per-session native image paths when the model supports native vision; the caller consumes + that buffer at ``run_conversation``. Empty list means the text vision path already ran. """ history = history or [] _pending_stt_prepared = hasattr(event, "_gateway_pending_stt_text") @@ -18905,9 +17527,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) or "" _group_sessions_per_user = getattr(self.config, "group_sessions_per_user", True) _thread_sessions_per_user = getattr(self.config, "thread_sessions_per_user", False) - # Prefer the already resolved session key from the caller so this write - # key matches the consume key at the run_conversation site. Fall back - # to deriving it here for tests and legacy standalone callers. + # Prefer the caller's resolved session key so this write key matches the consume key at the + # run_conversation site; derive it here only for tests and legacy standalone callers. session_key = session_key or self._session_key_for_source(source) # Reset only this session's per-call buffer; other sessions may be # concurrently preparing multimodal turns on the same runner. @@ -18934,9 +17555,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) message_text = f"[{_safe_user_name}] {message_text}" - # Prepend channel context from history backfill (if any). This - # happens after sender-prefix so the prefix only applies to the - # trigger message, not the backfill block. + # Prepend history-backfill channel context after the sender-prefix so the prefix applies + # only to the trigger message, not the backfill block. if getattr(event, "channel_context", None): message_text = f"{event.channel_context}\n\n[New message]\n{message_text}" @@ -19038,11 +17658,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug( "Transcript echo failed (non-fatal): %s", _echo_exc, ) - # NOTE: Previously, when transcription failed (e.g. no STT provider configured), the - # gateway also emitted a hardcoded English notice via `_stt_adapter.send()`. That - # bypassed the LLM and produced two replies (one pre-canned, spoken by TTS in the - # wrong language). Enrichment now leaves a single neutral marker in the prompt so the - # LLM produces one localized reply; the hardcoded send has therefore been removed. + # On transcription failure, do NOT send a hardcoded notice here: that bypassed the + # LLM and produced two replies (one pre-canned, TTS'd in the wrong language). + # Enrichment leaves a single neutral marker so the LLM gives one localized reply. if audio_file_paths: from tools.credential_files import to_agent_visible_cache_path as _to_agent_path @@ -19088,13 +17706,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _TEXT_EXTENSIONS = {".txt", ".md", ".csv", ".log", ".json", ".xml", ".yaml", ".yml", ".toml", ".ini", ".cfg"} for i, path in enumerate(event.media_urls): - # Per-attachment document handling. Skip anything already routed - # as image / audio / video by the buckets above — only genuine - # non-media files get a path-pointing context note. This makes a - # document mixed into a PHOTO/VOICE message (whole-message type - # != DOCUMENT) still reach the agent as a readable cached file, - # instead of being silently dropped because the message-level - # type wasn't DOCUMENT. + # Per-attachment document handling: skip anything already routed as image/audio/video + # above; only genuine non-media files get a context note. A document mixed into a + # PHOTO/VOICE message (message-level type != DOCUMENT) thus still reaches the agent. if ( _event_media_is_image(event, i) or _event_media_is_audio(event, i) @@ -19108,10 +17722,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew mtype = "text/plain" else: guessed, _ = _mimetypes.guess_type(path) - if guessed: - mtype = guessed - else: - mtype = "application/octet-stream" + mtype = guessed or "application/octet-stream" # Any accepted file gets a path-pointing context note — we accept # all file types now, so a non-text/non-application MIME (font/*, # model/*, etc.) must still tell the agent the file exists. @@ -19196,10 +17807,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _msg_custom_providers = _msg_cfg.get("custom_providers") or [] except Exception: pass - # Resolve the session's actual model/provider/base_url the same way the hygiene - # compression block does. GatewayRunner has no self._model/self._base_url (that was - # copy-pasted from HermesCLI); using them raised AttributeError, silently caught - # below, so this feature never ran. + # Resolve the session's actual model/provider/base_url as the hygiene compression + # block does; GatewayRunner has no self._model/self._base_url (AttributeError, + # silently caught below). _msg_model, _msg_runtime = self._resolve_session_agent_runtime( source=source, session_key=session_key, @@ -19399,9 +18009,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew attempt, exc, ) - # Never let marker cleanup block in-memory agent/lease release. A - # stale marker is bounded by the configured agent timeout and the - # clean-start orphan-marker discard path. + # Never let marker cleanup block agent/lease release; a stale marker is bounded by the + # agent timeout and the clean-start orphan-marker discard path. logger.warning( "Could not clear active-turn marker for %s after 3 attempts: %s", session_key, @@ -19413,10 +18022,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "_gateway_active_turn_session_key", "_gateway_active_turn_token", ): - try: + with suppress(AttributeError): delattr(event, attr) - except AttributeError: - pass def _install_plugin_message_injector(self) -> None: """Publish this live gateway's plugin message scheduler.""" @@ -19576,10 +18183,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return None source = cached_sources.get(session_key) if source is not None: - try: + with suppress(Exception): cached_sources.move_to_end(session_key) - except Exception: - pass return source async def _handle_message_with_agent(self, event, source, _quick_key: str, run_generation: int): @@ -19605,10 +18210,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew source.chat_id, source.user_id, source.thread_id, recovered, ) source = dataclasses.replace(source, thread_id=recovered) - try: + with suppress(Exception): event.source = source - except Exception: - pass event_metadata = getattr(event, "metadata", None) or {} expected_session_key = str( @@ -19715,20 +18318,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await asyncio.to_thread(self._record_telegram_topic_binding, source, session_entry) except Exception: logger.debug("Failed to record Telegram topic binding", exc_info=True) - # Capture and immediately consume was_auto_reset so it does not - # re-fire on subsequent messages — preventing the cleanup from - # wiping model/reasoning overrides set between turns (Closes #48031). + # Capture and consume was_auto_reset immediately so it cannot re-fire on later messages and + # wipe model/reasoning overrides set between turns. _was_auto_reset = getattr(session_entry, "was_auto_reset", False) if _was_auto_reset: - # Treat auto-reset as a full conversation boundary — clear every conversation-scoped - # per-session dict in one funnel call so the fresh session does not inherit the previous - # conversation's model/reasoning overrides, a queued "/model switched" note, or a stale - # resolved-model cache. + # Auto-reset is a full conversation boundary: one funnel call clears every conversation- + # scoped per-session dict so the fresh session inherits no model/reasoning overrides, no + # queued "/model switched" note and no stale resolved-model cache. self._clear_conversation_scope(session_key, reason="auto_reset") - # Evict the cached agent so the fresh session does not inherit the previous - # conversation's context_compressor._previous_summary — the cache is keyed on the stable - # session_key, so an auto-reset otherwise reuses the old agent and leaks prior history - # into new compaction summaries. + # Evict the cached agent: the cache is keyed on the stable session_key, so an auto-reset + # would otherwise reuse the old agent and leak context_compressor._previous_summary + # (prior history) into new compaction summaries. self._evict_cached_agent(session_key) session_entry.was_auto_reset = False @@ -19760,11 +18360,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _redact_pii = False persist_user_message = None persist_user_timestamp = None - # Synthetic self-injected turns (async-delegation batch completions, background watch - # notifications, resume wake-ups) arrive as MessageEvent(internal=True). Persist their user - # row with display_kind="internal_notification" so UIs render them as timeline notices, not - # user bubbles. Role/content untouched — display_kind is a DB-only sidecar stripped from - # every provider-bound payload (conversation_loop's api_msg.pop("display_kind")). + # Synthetic self-injected turns (batch completions, watch notifications, resume wake-ups) + # arrive as MessageEvent(internal=True). Persist with display_kind="internal_notification" + # so UIs render timeline notices, not user bubbles. display_kind is a DB-only sidecar + # stripped from every provider-bound payload; role/content untouched. persist_user_display_kind = ( "internal_notification" if getattr(event, "internal", False) else None ) @@ -19774,34 +18373,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass - # Build the context prompt to inject. The render is pinned per - # session, keyed by a hash of the exact renderer inputs - # (_ephemeral_change_key). A key hit reuses the pinned bytes verbatim - # so the composed system prompt cannot drift turn-over-turn; a key - # miss (thread rename, /sethome, redact_pii flip, ...) re-renders - # once — the only legitimate cache busts. + # Build the context prompt. The render is pinned per session, keyed by a hash of the exact + # renderer inputs (_ephemeral_change_key): a hit reuses the pinned bytes so the system prompt + # cannot drift turn-over-turn; a miss (thread rename, /sethome, redact_pii flip) re-renders. context_prompt = self._pinned_session_context_prompt( context, _redact_pii, session_key ) - # Per-turn must-deliver notes. These used to be appended to context_prompt (the ephemeral - # system prompt), which guaranteed a turn1→turn2 system-prompt diff and a full agent - # rebuild. They now ride the current user message via the api_content sidecar (staged - # below, consumed in run_sync → build_turn_context). + # Per-turn must-deliver notes ride the user message via the api_content sidecar (staged + # below, consumed in run_sync → build_turn_context), NOT context_prompt: appending them to + # the ephemeral system prompt guaranteed a turn1→turn2 diff and a full agent rebuild. turn_sidecar_notes: List[str] = [] # If the previous session expired and was auto-reset, deliver a notice # so the agent knows this is a fresh conversation (not an intentional /reset). if _was_auto_reset: reset_reason = getattr(session_entry, 'auto_reset_reason', None) or 'idle' - if reset_reason == "suspended": - context_note = "[System note: The user's previous session was stopped and suspended. This is a fresh conversation with no prior context.]" - elif reset_reason == "daily": - context_note = "[System note: The user's session was automatically reset by the daily schedule. This is a fresh conversation with no prior context.]" - elif reset_reason == "resume_pending_expired": - context_note = "[System note: The previous gateway session could not be recovered after a restart (API recovery timed out). This is a fresh conversation — use /resume to restore history if needed.]" - else: - context_note = "[System note: The user's previous session expired due to inactivity. This is a fresh conversation with no prior context.]" + context_note = _AUTO_RESET_CONTEXT_NOTES.get(reset_reason, _AUTO_RESET_CONTEXT_NOTES["idle"]) # Slack/Discord channels/threads are long-lived: point the agent at the specific prior # same-channel session so it recalls that context via session_search instead of an # unrelated recent session. Deterministic — no extra API/DB calls. @@ -19813,10 +18401,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew context_note = context_note + "\n\n" + continuity_note turn_sidecar_notes.append(context_note) - # Send a user-facing notification explaining the reset, unless: - # - notifications are disabled in config - # - the platform is excluded (e.g. api_server, webhook) - # - the expired session had no activity (nothing was cleared) + # Notify the user about the reset unless notifications are disabled in config, the + # platform is excluded (e.g. api_server, webhook), or the expired session had no + # activity. try: policy = self.session_store.config.get_reset_policy( platform=source.platform, @@ -19835,17 +18422,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if should_notify: adapter = self._adapter_for_source(source) if adapter: - if reset_reason == "suspended": - reason_text = "previous session was stopped or interrupted" - elif reset_reason == "resume_pending_expired": - reason_text = "gateway restart recovery timed out" - elif reset_reason == "daily": - reason_text = f"daily schedule at {policy.at_hour}:00" - else: - hours = policy.idle_minutes // 60 - mins = policy.idle_minutes % 60 - duration = f"{hours}h" if not mins else f"{hours}h {mins}m" if hours else f"{mins}m" - reason_text = f"inactive for {duration}" + reason_text = _auto_reset_reason_text(reset_reason, policy) notice = ( f"◐ Session automatically reset ({reason_text}). " f"Conversation history cleared.\n" @@ -19906,21 +18483,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.warning("[Gateway] Failed to auto-load skill(s) %s: %s", _skill_names, e) - # ── Turn lease (#64934) ──────────────────────────────────────── - # Session resolution is FINAL here (get_or_create → async-delegation - # pinning → topic tip-walk switch_session are all above). Serialize - # the [load history → run → flush] region per resolved SESSION_ID: - # when a second routing key is mapped to this same session_id, its - # turn waits here for the previous turn's flush instead of loading a - # stale history base and interleaving transcript writes. Same-key - # messages never reach this point mid-turn (adapter + runner guards - # hold them), so the lock is uncontended outside the alias-key route. - # Fail-closed on timeout: never enter the transcript region without a - # lease. Outer dispatch returns a bounded rejection/resend notice rather - # than recreating the exact concurrent-turn corruption this lease exists - # to prevent. Released in _handle_message's finally via - # _release_turn_lease — granted per (routing key, run generation) so a - # stale unwind can't release a newer turn's lease. + # ── Turn lease: session resolution is FINAL here. Serialize [load history → run → flush] + # per resolved SESSION_ID: another routing key mapped to the same session_id waits for the + # prior flush instead of loading a stale base and interleaving writes (same-key messages + # never reach here mid-turn thanks to adapter + runner guards, so the lock is otherwise + # uncontended). Fail-closed on timeout: never enter the transcript region without a lease; + # outer dispatch returns a bounded resend notice. Released in _handle_message's finally + # (_release_turn_lease), granted per (routing key, run generation) so a stale unwind can't + # release a newer turn's. _lease_registry = getattr(self, "_turn_leases", None) if _lease_registry is not None: try: @@ -19965,32 +18535,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "if you intentionally want to start a new conversation." ) - # ----------------------------------------------------------------- - # Session hygiene: auto-compress pathologically large transcripts - # - # Long-lived gateway sessions can accumulate enough history that - # every new message rehydrates an oversized transcript, causing - # repeated truncation/context failures. Detect this early and - # compress proactively — before the agent even starts. (#628) - # - # Token source priority: - # 1. Actual API-reported prompt_tokens from the last turn - # (stored in session_entry.last_prompt_tokens) - # 2. Rough char-based estimate (str(msg)//4). Overestimates - # by 30-50% on code/JSON-heavy sessions, but that just - # means hygiene fires a bit early — safe and harmless. - # ----------------------------------------------------------------- + # Session hygiene: auto-compress pathologically large transcripts before the agent starts so + # oversized histories don't cause repeated truncation/context failures. Token source: the + # API's prompt_tokens from the last turn (session_entry.last_prompt_tokens), else a char/4 + # estimate (30-50% high on code-heavy sessions, so hygiene merely fires a bit early). if history and len(history) >= 4: from agent.model_metadata import ( estimate_messages_tokens_rough, get_model_context_length_async, ) - # Read model + compression config from config.yaml. NOTE: hygiene threshold is - # intentionally HIGHER than the agent's own compressor (0.85 vs 0.50). Hygiene is a - # pre-agent safety net for sessions that grew too large between turns; the agent's own - # compressor handles normal context management with real token counts. Having hygiene at - # 0.50 caused premature compression on every turn in long gateway sessions. + # Read model + compression config. Hygiene threshold is intentionally HIGHER than the + # agent's own compressor (0.85 vs 0.50): it is a pre-agent safety net for sessions that + # grew between turns; at 0.50 it compressed prematurely on every turn in long sessions. _hyg_model = "anthropic/claude-sonnet-4.6" _hyg_threshold_pct = 0.85 _hyg_compression_enabled = True @@ -20024,17 +18581,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # (same as run_agent.py lines 995-1005) _raw_ctx = _model_cfg.get("context_length") if _raw_ctx is not None: - try: + with suppress(TypeError, ValueError): _hyg_config_context_length = int(_raw_ctx) - except (TypeError, ValueError): - pass # Read provider for accurate context detection _hyg_provider = _model_cfg.get("provider") or None _hyg_base_url = _model_cfg.get("base_url") or None - # Read compression settings — only use enabled flag. - # The threshold is intentionally separate from the agent's - # compression.threshold (hygiene runs higher). + # Only the enabled flag is read; hygiene's threshold is deliberately separate + # from the agent's compression.threshold (hygiene runs higher). _comp_cfg = _hyg_data.get("compression", {}) if isinstance(_comp_cfg, dict): _hyg_compression_enabled = str( @@ -20118,9 +18672,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: _hyg_config_context_length = None - # Check custom_providers per-model context_length - # (same fallback as run_agent.py lines 1171-1189). - # Must run after runtime resolution so _hyg_base_url is set. + # custom_providers per-model context_length fallback (as in run_agent.py); must run + # after runtime resolution so _hyg_base_url is set. if _hyg_config_context_length is None and _hyg_base_url: try: try: @@ -20169,17 +18722,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew else: _approx_tokens = estimate_messages_tokens_rough(history) _token_source = "estimated" - # Note: rough estimates overestimate by 30-50% for code/JSON-heavy sessions, but - # that just means hygiene fires a bit early — which is safe and harmless. Do NOT - # compensate with a threshold multiplier: 85% * 1.4 = 119% of context, which - # kept hygiene from ever firing for ~200K models. + # Rough estimates run 30-50% high on code/JSON-heavy sessions, which only makes + # hygiene fire early (safe). Do NOT compensate with a threshold multiplier: 85% + # * 1.4 = 119% of context kept hygiene from ever firing for ~200K models. - # Hard safety valve: force compression if message count is extreme, regardless of - # token estimates. This breaks the death spiral where API disconnects prevent token - # data collection, which prevents compression, which causes more disconnects. The - # default (5000) sits well clear of legitimate 1M+ context sessions doing thousands - # of short turns — those compress on the token threshold. Configurable via - # compression.hygiene_hard_message_limit. + # Hard safety valve: force compression at an extreme message count regardless of token + # estimates, breaking the spiral where API disconnects prevent token data → no + # compression → more disconnects. Default 5000 sits clear of legitimate 1M+ context + # sessions (those compress on tokens). Config: compression.hygiene_hard_message_limit. _HARD_MSG_LIMIT = _hyg_hard_msg_limit _needs_compress = ( _approx_tokens >= _compress_token_threshold @@ -20248,15 +18798,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _hyg_runtime.get("api_mode") or "" ).lower() if _hyg_api_mode == "codex_app_server": - # codex app-server runtime: the model's real - # context is the app-server's server-side thread, - # not the transcript mirror. The detached-agent - # block below could only rewrite the mirror (a - # guaranteed no-op for the thread) and its - # finally-clause eviction would destroy the live - # thread — the next turn then starts blank - # (#73503). Route to the live cached agent's - # thread/compact/start instead and KEEP it cached. + # codex app-server runtime: the real context is the server-side thread, + # not the transcript mirror. The detached-agent block below would only + # rewrite the mirror and its finally-eviction would destroy the live + # thread (next turn starts blank). Use the cached agent's + # thread/compact/start and KEEP it cached. _hyg_codex_auto = "native" _hyg_comp_cfg = ( _hyg_data.get("compression") @@ -20289,12 +18835,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"{_approx_tokens:,}", ) elif _hyg_runtime.get("api_key"): - # Pass the FULL transcript (tool results included). Filtering to - # user/assistant starved the compressor: tool results are the bulk of - # context, _prune_old_tool_results never saw them, and short histories - # tripped the protect-first/last early-return so nothing compressed. The - # agent loop passes its full message list to _compress_context — the - # gateway now matches. + # Pass the FULL transcript (tool results included), matching the agent + # loop: filtering to user/assistant starved the compressor — tool results + # are the bulk of context and short histories tripped the + # protect-first/last early-return so nothing compressed. _hyg_msgs = [ m for m in history if m.get("role") in {"user", "assistant", "tool"} @@ -20317,14 +18861,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew exc_info=True, ) _hyg_session_db = getattr(self._session_db, "_db", self._session_db) - # Hygiene performs the same lossy rewrite as - # normal compression. When the operator enabled - # compression.checkpoint_required, the memory - # provider must be loaded so the required - # checkpoint is created before any transcript - # mutation; otherwise keep the historical fast - # path (no provider init, no best-effort hook) - # for hygiene. + # Hygiene is the same lossy rewrite as normal compression: when + # compression.checkpoint_required is on, load the memory provider so + # the checkpoint exists before any mutation; otherwise keep the + # fast path (no provider init, no best-effort hook). from hermes_cli.config import load_config as _load_cfg from utils import is_truthy_value as _is_truthy @@ -20354,13 +18894,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _hyg_agent.platform = _GATEWAY_HYGIENE_PLATFORM _hyg_cleanup_deferred = False try: - # Gateway hygiene runs before the user turn starts and already - # owns the session binding, so prefer in-place compaction: it - # archives old rows under the same session id instead of minting - # a continuation child that must be published back to - # SessionStore/topic bindings. Without a SessionDB, - # compress_context leaves this False and the guard below - # preserves the transcript. + # Hygiene runs before the turn and owns the session binding, so + # prefer in-place compaction: archive old rows under the same + # session id rather than minting a continuation child that must + # be published back to SessionStore/topic bindings. Without a + # SessionDB this stays False and the guard below preserves it. _hyg_agent.compression_in_place = True _bind_hyg_state = getattr( getattr(_hyg_agent, "context_compressor", None), @@ -20382,12 +18920,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew total_ceiling_seconds=_hyg_total_ceiling_seconds ) # Default executor (NOT self._get_executor): a fence-cancelled - # hung summary must never occupy one of the gateway's agent-work - # slots. But it MUST run inside the caller's contextvars: under - # multiplex_profiles the profile secret scope / HERMES_HOME live - # in ContextVars; a bare run_in_executor worker starts with an - # empty Context, get_secret() fails closed, and every hygiene - # compaction silently degrades to lossy truncation. + # hung summary must never occupy an agent-work slot. But it MUST + # run in the caller's contextvars: under multiplex_profiles the + # secret scope / HERMES_HOME live in ContextVars, and an empty + # Context makes get_secret() fail closed → lossy truncation. _hyg_future = loop.run_in_executor( None, copy_context().run, @@ -20398,21 +18934,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ), ) try: - # Progress-aware wait: the timeout is an INACTIVITY budget, - # not a total one — the worker ticks the fence per streamed - # token (touch_progress), so a slow reasoning model that is - # still generating keeps extending the deadline; only a - # genuinely silent worker times out. A hard ceiling bounds - # the total wait so a degenerate trickle stream can't hold - # the turn forever. + # Progress-aware wait: the timeout is an INACTIVITY budget — + # the worker ticks the fence per streamed token, so a slow but + # still-generating model extends the deadline. A hard ceiling + # bounds the total so a trickle stream can't hold the turn. _hyg_wait_started = time.monotonic() while True: if _hyg_commit_fence.is_cancelled: raise asyncio.TimeoutError - # #76354 S3: charge the idle budget from the LAST - # PROGRESS event, not from the start of this wait slice - # — otherwise silence can approach 2x the configured - # timeout. + # Charge the idle budget from the LAST PROGRESS event, + # not from the start of this wait slice — otherwise + # silence can approach 2x the configured timeout. _hyg_waited = ( time.monotonic() - _hyg_wait_started ) @@ -20428,24 +18960,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew 0.005, ), ) - # Bounded turn-hold (#TKT-0029): cap - # this slice at the remaining - # turn-hold budget so the wait is - # re-evaluated against - # _hyg_max_turn_hold_seconds at - # least that often — otherwise a - # continuously-streaming worker - # (which keeps the inactivity slice - # large) would hold the turn until - # the total ceiling before the - # budget check ever runs. + # Bounded turn-hold: cap this slice at the remaining + # turn-hold budget so it is re-checked against + # _hyg_max_turn_hold_seconds at least that often — + # otherwise a continuously-streaming worker keeps the + # slice large and holds the turn until the ceiling. _turn_hold_remaining = ( _hyg_max_turn_hold_seconds - (time.monotonic() - _hyg_wait_started) ) if _turn_hold_remaining <= 0: - # Budget already exhausted — - # force an immediate timeout so + # Budget exhausted: force an immediate timeout so # the abandonment path below runs. _slice = 0.005 else: @@ -20473,20 +18998,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew raise _hyg_waited = time.monotonic() - _hyg_wait_started _idle = _hyg_commit_fence.seconds_since_progress() - # Bounded turn-hold (#TKT-0029): - # never hold the user's TURN - # longer than - # _hyg_max_turn_hold_seconds, - # even if the summary model is - # still streaming. Past the - # budget we stop waiting and - # fall through to the timeout - # path below, which revokes - # commit admission and proceeds - # on the uncompressed - # transcript — the wire never - # stays silent long enough to - # trip a transport idle-timeout. + # Bounded turn-hold: never hold the user's TURN past + # _hyg_max_turn_hold_seconds even if the summary + # model is still streaming; fall through to the + # timeout path, which revokes commit admission and + # proceeds on the uncompressed transcript, so the + # wire never trips a transport idle-timeout. if ( _hyg_waited >= _hyg_max_turn_hold_seconds @@ -20525,43 +19042,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew continue raise except HygieneTurnHoldExceeded: - # Turn-hold expiry is an availability boundary, - # not a failure. The compressor is healthy and - # still streaming; we simply cannot hold the - # current user turn any longer. Share the safe - # mechanics (fence, release, defer, proceed - # uncompressed) but with distinct provenance, - # user message, and NO failure-cooldown - # increment. - # - # #97963: decouple the TURN from the - # COMPRESSION. When the worker's commit is - # watermark-fenced (it captured the session's - # active-row watermark at compression start, - # so rows appended after that point — this - # released turn included — survive its late - # commit verbatim as cloned concurrent tail), - # the already-running attempt KEEPS its commit - # admission: the user's turn proceeds on the - # uncompressed transcript NOW, and the summary - # is adopted when the detached worker reaches - # its own watermark-fenced commit transaction - # (archive_and_compact / the rotation publish - # path — the next safe boundary). Before this, - # the fence was ALWAYS cancelled here, burning - # the full summary attempt — for a thinking - # summary model whose reasoning prefix alone - # exceeds the 10s hold, that made hygiene - # auto-compression fail 100% of the time while - # paying the summary model per turn. The turn - # itself is still released at the same budget: - # only the fate of the detached worker's - # RESULT changes. If the commit is NOT - # watermark-fenced (no session_db, watermark - # capture failed, legacy lock API), a late - # commit could clobber newer turns, so cancel - # exactly as before — never worse than the - # status quo. + # Turn-hold expiry is an availability boundary, not a + # failure: the compressor is healthy and still streaming; we + # just can't hold the turn longer. Share the safe mechanics + # (fence, release, defer, proceed uncompressed) with + # distinct provenance / user message and NO failure-cooldown + # increment. Decouple the TURN from the COMPRESSION: when + # the worker's commit is watermark-fenced (rows appended + # after compression start, this turn included, survive its + # late commit as cloned concurrent tail) the attempt KEEPS + # commit admission — the turn proceeds uncompressed NOW and + # the summary is adopted at the worker's own fenced commit + # (archive_and_compact / rotation publish); always + # cancelling burned every attempt for thinking summary + # models whose reasoning prefix alone exceeds the hold. If + # NOT watermark-fenced (no session_db, capture failed, + # legacy lock API) a late commit could clobber newer turns, + # so cancel. _hyg_keep_admission = bool( getattr( _hyg_commit_fence, @@ -20576,21 +19073,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew context="session hygiene turn-hold", ) _hyg_cleanup_deferred = True - # NO retry-after here (#97963 (b)): the - # attempt is still running toward a real - # commit, and arming the flat 60s - # retry-after would ALSO block the - # agent-side preflight compressor from a - # fresh chance ("Skipping preflight - # compression: same-session cooldown - # active"). Re-attempt spacing is covered - # by the durable compression lock instead: - # the next turn's hygiene pre-check skips - # while this worker's lease is held - # (_session_has_compression_in_flight). - # The flat retry-after is recorded by the - # done-callback below ONLY if the worker - # ends without committing anything. + # NO retry-after here: the attempt is still running + # toward a real commit, and the flat 60s retry-after + # would also block the agent-side preflight compressor + # (same-session cooldown). Re-attempt spacing comes from + # the durable compression lock instead: the next turn's + # hygiene pre-check skips while this worker's lease is + # held (_session_has_compression_in_flight). The + # done-callback below records the flat retry-after ONLY + # if the worker ends without committing anything. _hyg_deferred_sid = session_entry.session_id _hyg_deferred_key = session_key _hyg_deferred_agent = _hyg_agent @@ -20944,10 +19435,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew context="session hygiene unwind", ) _hyg_cleanup_deferred = True - # restart drain / task cancel used to re-raise with no - # cooldown, so the next turn immediately re-armed hygiene - # and waited up to 600s behind a fence that would refuse the - # commit again. + # restart drain / task cancel must record a cooldown, or the + # next turn immediately re-arms hygiene and waits up to 600s + # behind a fence that would refuse the commit again. if _hyg_failure_cooldown_seconds >= 0: try: _hyg_cooldown = _hygiene_cooldown_for_failure( @@ -20993,34 +19483,21 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) _hyg_rotated = False _compressed = history - # Only rewrite the transcript when rotation produced - # a NEW session id. In-place compaction does NOT - # need a rewrite: archive_and_compact() has already - # soft-archived the previous active rows and inserted - # the compacted messages as the new active set inside - # _compress_context(). Calling rewrite_transcript() - # after in-place compaction would invoke - # replace_messages(active_only=False) which DELETEs - # ALL rows — including the archived turns that - # archive_and_compact() deliberately preserved - # (silent data loss, #61145). - # - # The danger this guards against (mirrors the - # /compress fix #44794/#39704): if _compress_context - # returns a summary but neither rotates nor completes - # archive_and_compact(), the session_id is unchanged - # for a FAILURE reason, and an unconditional - # rewrite_transcript() would DELETE the original - # messages and replace them with only the compressed - # summary (permanent data loss, #21301). - # - # Write-before-repoint (mirrors manual /compress): - # if we repointed session_entry onto the child SID - # and rewrite_transcript then failed (lock/ENOSPC), - # the live entry would already reference a brand-new - # empty session while the turn continues — the - # conversation silently vanishes. Persist the child - # transcript first; only then rebind the live entry. + # Rewrite the transcript only when rotation produced a NEW + # session id. In-place compaction needs none: + # archive_and_compact() already soft-archived the previous + # active rows and inserted the compacted set, and + # rewrite_transcript() would run + # replace_messages(active_only=False) and DELETE the archived + # turns. Likewise a summary with neither rotation nor a + # completed archive_and_compact() (unchanged session_id) signals + # FAILURE; an unconditional rewrite would replace the originals + # with only the summary (permanent data loss). + # Write-before-repoint (mirrors manual /compress): if + # session_entry were repointed to the child SID and + # rewrite_transcript then failed (lock/ENOSPC), the live entry + # would reference an empty session — the conversation silently + # vanishes. Persist the child transcript first, then rebind. if _hyg_rotated: if not await self.async_session_store.rewrite_transcript( _hyg_new_sid, _compressed @@ -21071,9 +19548,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _compressed ) else: - # No rewrite happened — transcript preserved - # unchanged, so the post-compression counts equal - # the pre-compression ones. + # No rewrite happened — the transcript is unchanged, so the + # post-compression counts equal the pre-compression ones. _new_count = _msg_count _new_tokens = _approx_tokens logger.warning( @@ -21099,15 +19575,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"{_new_tokens:,}", ) - # If summary generation failed, the - # compressor aborts entirely and returns - # messages unchanged — nothing is dropped. - # Surface a visible warning to the gateway - # user — agent.log alone is invisible on - # TG/Discord/etc. — so they know the chat - # is "frozen" at the current size and can - # /compress to retry or /reset to start - # fresh. + # Summary failure aborts the compressor entirely (messages + # unchanged, nothing dropped). Warn the gateway user visibly + # — agent.log is invisible on TG/Discord/etc. — so they know + # the chat is "frozen" at this size and can /compress to + # retry or /reset to start fresh. _comp = getattr(_hyg_agent, "context_compressor", None) _hyg_aborted = _comp is not None and getattr( _comp, "_last_compress_aborted", False @@ -21127,14 +19599,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _hyg_fence_cancelled: _hyg_aborted = True if not _hyg_aborted: - # Recovery decision lives in the - # extracted, unit-tested predicate — the - # degenerate "did not rotate or compact - # in place" path (#21301) sets both flags - # False and reuses the pre-compression - # counts, so a numbers-only check would - # read a no-op as success and clear the - # streak on every wedged run (#79624). + # Recovery decision lives in the unit-tested predicate: the + # degenerate "neither rotated nor compacted in place" path + # sets both flags False and reuses the pre-compression + # counts, so a numbers-only check would read a no-op as + # success and clear the streak. if hygiene_compaction_recovered( aborted=_hyg_aborted, rotated=_hyg_rotated, @@ -21181,9 +19650,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) if not _hyg_fence_cancelled: _err = getattr(_comp, "_last_summary_error", None) or "unknown error" - # Force-redact: provider exception text - # may contain credentials; this message - # reaches gateway users directly. + # Force-redact: provider exception text may contain + # credentials and this message reaches gateway users. from agent.redact import redact_sensitive_text _err = redact_sensitive_text(_err, force=True) _warn_msg = ( @@ -21226,9 +19694,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _werr, ) finally: - # Evict the cached agent so the next turn - # rebuilds its system prompt from current - # SOUL.md, memory, and skills. + # Evict the cached agent so the next turn rebuilds its system + # prompt from current SOUL.md, memory, and skills. self._evict_cached_agent(session_key) if not _hyg_cleanup_deferred: await self._cleanup_agent_resources_off_loop( @@ -21255,12 +19722,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Briefly introduce yourself and mention that /help shows available commands. " "Keep the introduction concise -- one or two sentences max.]" ) - # Opt-in structured profile-build path. When enabled (default - # "ask") and not yet offered on this install, swap the plain intro - # for a consent-gated directive that offers to build a user - # profile and persists confirmed facts via memory(target="user"). - # The offer fires at most once (onboarding.seen flag); set - # onboarding.profile_build: off in config.yaml to disable. + # Opt-in structured profile-build path: when enabled (default "ask") and not yet offered + # on this install, swap the plain intro for a consent-gated directive that offers to + # build a user profile and persists confirmed facts via memory(target="user"). Fires at + # most once (onboarding.seen flag); onboarding.profile_build: off in config.yaml + # disables it. try: from agent.onboarding import ( PROFILE_BUILD_FLAG, @@ -21339,32 +19805,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) await self._deliver_platform_notice(source, notice) - # ----------------------------------------------------------------- - # Voice channel awareness — deliver current voice channel state so - # the agent knows who is in the channel and who is speaking, without - # needing a separate tool call. Delivered on the current user - # message and ONLY when it changed since the previous turn: the - # member/speaking serialization differs essentially every turn, and - # appending it to the ephemeral system prompt forced a full agent - # rebuild + prompt-cache re-key per message. The system prompt - # carries a static pointer line instead (gateway/session.py). - # ----------------------------------------------------------------- + # Voice channel awareness: deliver voice channel state (who is present / speaking) on the + # user message, ONLY when changed since the previous turn. It differs almost every turn, and + # in the ephemeral system prompt it forced a full agent rebuild + prompt-cache re-key per + # message; the system prompt carries a static pointer line instead (gateway/session.py). _vc_note = self._voice_channel_sidecar_note(event, source, session_key) if _vc_note: turn_sidecar_notes.append(_vc_note) - # ----------------------------------------------------------------- - # Auto-analyze images sent by the user - # - # If the user attached image(s), we run the vision tool eagerly so - # the conversation model always receives a text description. The - # local file path is also included so the model can re-examine the - # image later with a more targeted question via vision_analyze. - # - # We filter to image paths only (by media_type) so that non-image - # attachments (documents, audio, etc.) are not sent to the vision - # tool even when they appear in the same message. - # ----------------------------------------------------------------- + # Auto-analyze user images: run the vision tool eagerly so the model always gets a text + # description plus the local path for re-examination via vision_analyze. Filter to image + # media_type so documents/audio in the same message are not sent to the vision tool. message_text = await self._prepare_profile_scoped_inbound_message_text( event=event, source=source, @@ -21408,15 +19859,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as _ts_err: logger.debug("Message timestamp injection failed (non-fatal): %s", _ts_err) - # Stage the collected must-deliver notes for this turn's agent run (one-shot; consumed in - # run_sync). Staged AFTER the message_text early-out above so an aborted turn cannot leak - # its notes into the next turn's user message. + # Stage this turn's must-deliver notes (one-shot; consumed in run_sync) AFTER the + # message_text early-out so an aborted turn cannot leak its notes into the next turn. if turn_sidecar_notes and session_key: self._set_pending_turn_sidecar_notes(session_key, turn_sidecar_notes) - # Bind this gateway run generation to the adapter's active-session - # event so deferred post-delivery callbacks can be released by the - # same run that registered them. + # Bind this run generation to the adapter's active-session event so deferred post-delivery + # callbacks can be released by the same run that registered them. self._bind_adapter_run_generation( self._adapter_for_source(source), session_key, @@ -21462,9 +19911,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) _turn_seconds = time.monotonic() - _turn_started_monotonic - # Stop persistent typing indicator now that the agent is done. - # Slack AI status is scoped to a thread/workspace, so preserve the - # same routing metadata used by the response delivery path. + # Stop the typing indicator. Slack AI status is scoped to a thread/workspace, so + # preserve the routing metadata used by the response delivery path. try: _typing_adapter = self._adapter_for_source(source) _stop_with_metadata = getattr( @@ -21533,15 +19981,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _response_time, _api_calls, _resp_len, ) - # NOTE: the cross-process cache-coherence re-baseline - # (_refresh_agent_cache_message_count) is intentionally deferred until AFTER this turn's - # transcript persistence block below — it must include the first-turn `session_meta` - # marker row and the compression session_id swap, both of which happen later. + # The cross-process cache-coherence re-baseline (_refresh_agent_cache_message_count) is + # deferred until AFTER the transcript persistence block below: it must include the + # first-turn `session_meta` marker row and the compression session_id swap. - # Successful turn — clear any stuck-loop counter for this session. This ensures the - # counter only accumulates across CONSECUTIVE restarts where the session was active - # (never completed). Also clear resume_pending (set by drain-timeout shutdown): the turn - # completed, so later messages must not get the restart-interruption system note. + # Successful turn: clear the stuck-loop counter (it only accumulates across CONSECUTIVE + # restarts where the session never completed) and resume_pending (set by drain-timeout + # shutdown) so later messages don't get the restart-interruption system note. if session_key and _should_clear_resume_pending_after_turn(agent_result): await self._clear_restart_failure_count(session_key) try: @@ -21591,9 +20037,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew session_entry.session_id, ) - # Prepend reasoning/thinking if display is enabled (per-platform). - # Mattermost requires explicit per-platform opt-in because this is - # scratch text, not ordinary final-answer content. + # Prepend reasoning if display is enabled (per-platform). Mattermost requires explicit + # opt-in because this is scratch text, not ordinary final-answer content. try: _show_reasoning_effective = _resolve_gateway_display_bool( _load_gateway_config(), @@ -21620,9 +20065,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew display_reasoning += f"\n_... ({len(lines) - 15} more lines)_" else: display_reasoning = last_reasoning.strip() - # Render style is per-platform: Discord defaults to "-# " - # subtext (native small grey metadata text); other - # platforms keep the fenced code block. + # Render style is per-platform: Discord defaults to "-# " subtext (native small + # grey metadata text); other platforms keep the fenced code block. try: from gateway.display_config import resolve_display_setting _reasoning_style = resolve_display_setting( @@ -21693,11 +20137,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.error("Process watcher setup error: %s", e) - # Drain watch pattern notifications that arrived during the agent run. Watch events and - # completions share the same queue; process completions are already handled by the per- - # process watcher task above, so we only inject watch-type events here. Async-delegation - # completions also ride this queue but are owned by _async_delegation_watcher (single - # consumer for idle and post-turn cases) — leave them on the queue. + # Drain watch notifications that arrived during the run. The queue also carries process + # completions (handled by the per-process watcher task above) and async-delegation + # completions (owned by _async_delegation_watcher, the single consumer for idle and + # post-turn cases) — inject only watch-type events and leave the rest on the queue. try: from tools.process_registry import process_registry as _pr await self._drain_watch_notifications(_pr.completion_queue) @@ -21707,29 +20150,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # NOTE: Dangerous command approvals are now handled inline by the blocking gateway # approval mechanism in tools/approval.py. - # Save the full conversation to the transcript, including tool calls. - # This preserves the complete agent loop (tool_calls, tool results, - # intermediate reasoning) so sessions can be resumed with full context - # and transcripts are useful for debugging and training data. - # - # IMPORTANT: For context-overflow failures (compression exhausted, - # generic 400 on large sessions) we must NOT persist the user's - # message — doing so would grow the session further and cause the - # same failure on the next attempt, an infinite loop. (#1630, #9893) - # - # Transient failures (429, timeout, connection error, provider 5xx) - # are different: the session is not oversized, and silently dropping - # the user message causes severe context loss on retry — the agent - # forgets what was just asked. Persist the user turn so the - # conversation is preserved. (#7100) + # Persist the full agent loop (tool calls, results, reasoning) so sessions resume with + # full context. IMPORTANT: on context-overflow failures (compression exhausted, generic + # 400 on large sessions) do NOT persist the user message — it would grow the session and + # reproduce the failure forever. Transient failures (429, timeout, connection error, + # 5xx) are different: the session is not oversized and dropping the user turn causes + # severe context loss on retry, so persist it. agent_failed_early = bool(agent_result.get("failed")) hidden_reasoning_incomplete = _is_gateway_hidden_reasoning_incomplete_turn( agent_result ) _err_str_for_classify = str(agent_result.get("error", "")).lower() - # Use specific multi-word phrases (not bare "exceed" or "token") to avoid false - # positives on transient errors like "rate limit exceeded" or "invalid auth token". - # Matches run_agent.py's own context-length classifier. + # Use specific multi-word phrases (not bare "exceed"/"token") to avoid false positives + # on transient errors such as "rate limit exceeded"; matches run_agent.py's classifier. is_context_overflow_failure = agent_failed_early and ( bool(agent_result.get("compression_exhausted")) or any(p in _err_str_for_classify for p in ( @@ -21761,11 +20194,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew agent_result.get("error", "processing incomplete"), ) - # When compression is exhausted, the session is permanently too large to process. Auto- - # reset it so the next message starts fresh instead of replaying the same oversized - # context in an infinite fail loop. A lock-contended defer is the OPPOSITE case: a - # concurrent path holds the compression lock and is shrinking it — never wipe the - # session for that; retry-next-message semantics apply. + # Compression exhausted = permanently too large: auto-reset so the next message starts + # fresh instead of replaying the oversized context forever. A lock-contended defer is + # the OPPOSITE case (a concurrent path holds the lock and is shrinking it): never wipe + # for that. if agent_result.get("compression_deferred"): logger.info( "Compression deferred for session %s — the compression " @@ -21780,25 +20212,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) new_entry = await self.async_session_store.reset_session(session_key) self._evict_cached_agent(session_key) - # Conversation boundary: one funnel call clears every - # conversation-scoped per-session dict (#58403 and siblings). - # See _CONVERSATION_SCOPED_STATE. + # Conversation boundary: one funnel call clears every conversation-scoped + # per-session dict (see _CONVERSATION_SCOPED_STATE). self._clear_conversation_scope( session_key, reason="compression_exhausted_reset" ) if new_entry is not None: - # Drop the stale reference to the bloated compressed child and - # re-point the Telegram topic binding at the fresh session. - # Compression rotated session_entry.session_id to the oversized - # compressed child earlier this turn (the agent-result sync - # above), and that _sync also rewrote the (chat_id, thread_id) - # -> bloated-child binding. reset_session swaps in a clean, - # parentless session, but without re-syncing the binding the - # next inbound message in this topic gets switch_session'd back - # onto the bloated child by the binding-heal walk, reloads the - # oversized transcript, and re-triggers compression exhaustion - # forever (#35809 — regression of the #9893/#10063 auto-reset). - # No-op on non-topic lanes. + # Re-point the Telegram topic binding at the fresh session: compression rotated + # session_entry.session_id to the bloated child earlier this turn and that _sync + # also rewrote the (chat_id, thread_id) binding. Without a re-sync the + # binding-heal walk switches the next inbound message back onto the child and + # re-triggers exhaustion forever. No-op on non-topic lanes. session_entry = new_entry await asyncio.to_thread( self._sync_telegram_topic_binding, @@ -21812,9 +20236,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ts = time.time() # Unix epoch float — consistent with DB storage - # If this is a fresh session (no history), write the full tool - # definitions as the first entry so the transcript is self-describing - # -- the same list of dicts sent as tools=[...] in the API request. + # Fresh session (no history): write the full tool definitions as the first entry so the + # transcript is self-describing — the same dicts sent as tools=[...] in the API request. if is_context_overflow_failure: pass # Skip all transcript writes — don't grow a broken session elif not history: @@ -21830,21 +20253,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew } ) - # The agent already persisted these messages to SQLite via - # _flush_messages_to_session_db(), so skip the DB write here - # to prevent the duplicate-write bug (#860 / #42039). This holds - # for the codex app-server runtime too: although it early-returns - # and bypasses conversation_loop's per-step flushes, it flushes its - # own projected assistant/tool messages before returning and - # reports agent_persisted=True (see agent/codex_runtime.py). Reading - # the flag (default = self._session_db is not None) keeps the - # persistence contract explicit and lets any future non-persisting - # runtime opt into a gateway-side write by returning False. + # The agent already persisted these via _flush_messages_to_session_db(); skip the DB + # write to avoid duplicates. Holds for the codex app-server runtime too (it flushes its + # own projected messages before returning and reports agent_persisted=True). Reading the + # flag (default = self._session_db is not None) keeps the contract explicit; a + # non-persisting runtime opts in via False. agent_persisted = agent_result.get("agent_persisted", self._session_db is not None) - # Find only the NEW messages from this turn (skip history we loaded). Use history_offset - # (what the agent actually saw), not len(history), which includes session_meta entries - # stripped before the agent saw them. + # Only the NEW messages from this turn: use history_offset (what the agent saw), not + # len(history), which counts session_meta entries stripped before the agent saw them. if is_context_overflow_failure: pass # handled above — skip all transcript writes elif agent_failed_early or hidden_reasoning_incomplete: @@ -21869,9 +20286,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _user_entry["display_kind"] = persist_user_display_kind if event.message_id: _user_entry["message_id"] = str(event.message_id) - # Dedupe: skip if this platform message_id is already in the - # transcript (prevents duplicate user turns on Telegram retries - # after transient failures). #47237 + # Dedupe: skip if this platform message_id is already in the transcript (prevents + # duplicate user turns on Telegram retries after transient failures). _skip_persist = ( event.message_id and await self.async_session_store.has_platform_message_id( @@ -21948,9 +20364,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew skip_db=agent_persisted, ) - # Token counts and model are now persisted by the agent directly. - # Keep only last_prompt_tokens here for context-window tracking and - # compression decisions. + # The agent persists token counts and model itself; keep only last_prompt_tokens here + # for context-window tracking and compression decisions. await self.async_session_store.update_session( session_entry.session_key, last_prompt_tokens=agent_result.get("last_prompt_tokens", 0), @@ -21958,12 +20373,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) # Re-baseline the cached agent's message_count snapshot now that ALL of this turn's - # transcript writes are done — the agent's flushed user/assistant/tool rows AND the - # first-turn `session_meta` marker appended above. The cross-process coherence guard - # snapshots at agent-BUILD time and never refreshes on reuse, so without this our own - # writes trigger a rebuild next turn (destroying prompt caching). MUST run after the - # session_meta append (that row also bumps the count; before it, the snapshot was one - # short and turn 2 of every fresh conversation rebuilt). Fail-safe inside the helper. + # transcript writes are done (flushed rows AND the first-turn `session_meta` marker). + # The cross-process coherence guard snapshots at agent-BUILD time and never refreshes + # on reuse, so our own writes would trigger a rebuild next turn (destroying prompt + # caching). MUST run after the session_meta append (that row bumps the count too). await self._refresh_agent_cache_message_count( session_key, session_entry.session_id ) @@ -21992,11 +20405,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): await self._send_voice_reply(event, response) - # If streaming already delivered the response, extract and deliver any MEDIA: files - # before returning None — streamed chunks carry MEDIA: tags verbatim and the normal - # post-processing is skipped when already_sent, so media would never be delivered. - # Never skip when the agent failed: the error text is new content the user hasn't - # seen (streaming only sent partial output before the failure). + # Streamed responses still need MEDIA: files delivered before returning None (chunks + # carry the tags verbatim and post-processing is skipped when already_sent). Never skip + # when the agent failed: the error text is new content streaming didn't show. if agent_result.get("already_sent") and not agent_result.get("failed"): if response: _media_adapter = self._adapter_for_source(source) @@ -22020,10 +20431,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # This branch returns None so the adapter does not send the body twice. /loop and # /goal hooks in _handle_message read the return value, so stash the delivered text # on the event or those hooks never run and a /loop tick stays awaiting. - try: + with suppress(Exception): event._streamed_final_response = str(response or "") - except Exception: - pass return None return response @@ -22049,11 +20458,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass logger.exception("Agent error in session %s", session_key) - # Crash-resilience for failures that happen before AIAgent enters run_conversation() - # (for example: provider/httpx client init failures). In that path the agent cannot - # persist the current inbound turn itself, so append the user message here once. If the - # agent already reached its turn-start persistence, the latest user row matches and - # we skip the duplicate. + # Crash-resilience for failures before AIAgent enters run_conversation() (e.g. provider/ + # httpx client init): the agent can't persist the inbound turn there, so append the user + # message here once; if the agent already reached turn-start persistence the latest user + # row matches and we skip the duplicate. try: if 'message_text' in locals() and message_text is not None and session_entry is not None: _already_persisted = False @@ -22127,9 +20535,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew elif status_code == 529: status_hint = " The API is temporarily overloaded. Please try again shortly." elif status_code in {400, 500}: - # 400 with a large session is context overflow. - # 500 with a large session often means the payload is too large - # for the API to process — treat it the same way. + # 400 on a large session is context overflow; 500 on a large session often means the + # payload is too large for the API — treat it the same way. if _hist_len > 50: return ( "⚠️ Session too large for the model's context window.\n" @@ -22149,14 +20556,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _reset_notice_session_info(self, source: SessionSource) -> str: """Session-info block for the auto-reset notice, profile-scoped. - When multiplexing, resolve model/provider/context inside the profile serving ``source`` — - otherwise the banner advertises the base config's model while the session actually runs - on the profile's. Mirrors ``_run_agent``'s gating so single-profile gateways never enter - the scope. - - Call via ``asyncio.to_thread``: under the scope, resolution can do blocking work - (credential refresh, context-length HTTP probes). The scope is entered inside this method - so contextvars behave correctly in the worker thread. + Under multiplexing, resolve model/provider/context inside the profile serving ``source`` + (mirrors ``_run_agent``'s gating) or the banner advertises the base config's model. Call + via ``asyncio.to_thread``: resolution can block (credential refresh, context-length + probes), and the scope is entered here so contextvars behave in the worker thread. """ if getattr(getattr(self, "config", None), "multiplex_profiles", False): with _profile_runtime_scope(self._resolve_profile_home_for_source(source)): @@ -22208,10 +20611,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[str]: """Return a denial message if ``source`` cannot run ``canonical_cmd``, else None. - Used by both the cold and running-agent dispatch paths in ``_handle_message`` so - admin/user gating can't be bypassed by an in-flight agent. Backward-compat: when the - operator hasn't set ``allow_admin_from`` for the scope, ``policy_for_source`` returns - ``enabled=False`` and this always returns None. + Used by the cold and running-agent dispatch paths in ``_handle_message`` so admin/user gating + can't be bypassed by an in-flight agent. Backward-compat: without ``allow_admin_from`` for the + scope, ``policy_for_source`` returns ``enabled=False`` and this always returns None. """ from gateway.slash_access import policy_for_source as _policy_for_source @@ -22245,14 +20647,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _sibling_thread_run_keys(self, source: SessionSource, own_key: str) -> list: """Find running-agent keys for OTHER participants in the same thread. - Only applies when the message originates in a thread. In per-user thread mode each - participant gets an isolated key (``...:{thread_id}:{user_id}``), so another user's run is - invisible to the caller's own ``/stop``. This returns the keys of any - *actually running* agents (not the pending sentinel, not the caller's own key) whose key - shares the caller's ``{chat_id}:{thread_id}`` prefix. - - Returns an empty list when the source is not in a thread, or when no - sibling runs exist — callers must still gate on authorization. + In per-user thread mode each participant gets an isolated key + (``...:{thread_id}:{user_id}``), so another user's run is invisible to the caller's own + ``/stop``. Returns keys of *actually running* agents (not the pending sentinel, not the + caller's own) sharing the caller's ``{chat_id}:{thread_id}`` prefix; empty when not in a + thread or no sibling runs exist. Callers must still gate on authorization. """ thread_id = getattr(source, "thread_id", None) chat_id = getattr(source, "chat_id", None) @@ -22303,19 +20702,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: marker_path = _hermes_home / ".restart_last_processed.json" if not marker_path.exists(): - # Belt-and-suspenders for when the dedup marker goes missing - # (manually cleaned up, or the previous cycle's write failed). - # Without a marker the update_id comparison below can't run, so - # a redelivered /restart would sail through and re-restart the - # gateway — an infinite loop (issue #18528). - # - # Suppress ONLY when we can independently confirm we just came - # out of a restart cycle: this process booted from a - # chat-originated /restart (_booted_from_restart) AND is still - # within a short post-boot window. This never swallows a - # genuine first /restart on a fresh boot (no restart marker on - # boot → flag stays False). Consume the flag one-shot so a - # legitimate /restart sent later in the same session is honored. + # Belt-and-suspenders for a missing dedup marker (cleaned up, or the previous write + # failed): without it the update_id comparison can't run and a redelivered /restart + # would re-restart the gateway forever. Suppress ONLY when a restart cycle is + # independently confirmed: this process booted from a chat-originated /restart + # (_booted_from_restart) AND is within a short post-boot window; a genuine first + # /restart on a fresh boot is never swallowed (flag stays False). Consume the flag + # one-shot so a later legitimate /restart in the same session is honored. if ( getattr(self, "_booted_from_restart", False) and time.time() - getattr(self, "_startup_time", 0.0) < 60 @@ -22342,9 +20735,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._booted_from_restart = False return True - # Staleness guard: ignore markers older than 5 minutes. A legitimately - # old marker (e.g. crash recovery where notify never fired) should not - # swallow a fresh /restart from the user. + # Staleness guard: ignore markers older than 5 minutes so a legitimately old one (e.g. crash + # recovery where notify never fired) doesn't swallow a fresh /restart. requested_at = data.get("requested_at") if isinstance(requested_at, (int, float)): if time.time() - requested_at > 300: @@ -22358,20 +20750,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew the event source so an accepted suggestion's job delivers back to this chat/thread. """ args = (event.get_command_args() or "").strip() - source = event.source - origin = None - try: - platform = getattr(source.platform, "value", None) or str(getattr(source, "platform", "") or "") - chat_id = getattr(source, "chat_id", None) - if platform and chat_id: - origin = { - "platform": platform, - "chat_id": str(chat_id), - "chat_name": getattr(source, "chat_name", None), - "thread_id": getattr(source, "thread_id", None), - } - except Exception: - origin = None + origin = _command_origin_for_source(event.source) try: from hermes_cli.suggestions_cmd import handle_suggestions_command @@ -22387,20 +20766,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew from the event source so a directly created blueprint job delivers back to this chat. """ args = (event.get_command_args() or "").strip() - source = event.source - origin = None - try: - platform = getattr(source.platform, "value", None) or str(getattr(source, "platform", "") or "") - chat_id = getattr(source, "chat_id", None) - if platform and chat_id: - origin = { - "platform": platform, - "chat_id": str(chat_id), - "chat_name": getattr(source, "chat_name", None), - "thread_id": getattr(source, "thread_id", None), - } - except Exception: - origin = None + origin = _command_origin_for_source(event.source) try: from hermes_cli.blueprint_cmd import handle_blueprint_command @@ -22417,9 +20783,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _goal_max_turns_from_config(self) -> int: """Resolve the configured /goal turn budget for gateway sessions. - GatewayRunner.config is a GatewayConfig dataclass, not the full - user config mapping. Top-level config blocks such as ``goals`` are - therefore only available through hermes_cli.config.load_config(). + GatewayRunner.config is a GatewayConfig dataclass, not the full user config mapping, so + top-level blocks such as ``goals`` are only reachable via hermes_cli.config.load_config(). """ try: goals_cfg = ( @@ -22450,65 +20815,50 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.warning("%s: session DB warm-up failed: %s", label, exc) - async def _get_goal_manager_for_event(self, event: "MessageEvent"): - """Return a GoalManager bound to the session for this gateway event. + async def _session_entry_for_manager(self, event: "MessageEvent", label: str): + """Session entry for a /goal or /heartbeat manager, or None when lookup fails. - Returns ``(manager, session_entry)`` or ``(None, None)`` if the - goals module can't be loaded. + Warms the SessionDB cache off-loop first: a cold cache freezes the loop for the init + duration and drops the first write while the reply claims it was set. Internal events look + the session up WITHOUT touching activity so they never advance the idle/daily reset clock. """ + await self._warm_goals_session_db(label) + try: + session_entry = await self.async_session_store.get_or_create_session( + event.source, + touch_activity=not bool(getattr(event, "internal", False)), + ) + except Exception as exc: + logger.debug("%s: session lookup failed: %s", label, exc) + return None + if not (getattr(session_entry, "session_id", None) or ""): + return None + return session_entry + + async def _get_goal_manager_for_event(self, event: "MessageEvent"): + """Return ``(GoalManager, session_entry)`` for this event, or ``(None, None)``.""" try: from hermes_cli.goals import GoalManager except Exception as exc: logger.debug("goal manager unavailable: %s", exc) return None, None - # Warm the SessionDB cache off-loop. A cold cache freezes the - # loop for the init duration and drops the first write: the - # /goal reply claims the goal was set. - await self._warm_goals_session_db("goal manager") - try: - # Session lookups on behalf of an internal event must not advance - # the user-activity clock that drives idle/daily reset policy - # (same class as the wake fix in _handle_message_with_agent). - session_entry = await self.async_session_store.get_or_create_session( - event.source, - touch_activity=not bool(getattr(event, "internal", False)), - ) - except Exception as exc: - logger.debug("goal manager: session lookup failed: %s", exc) - return None, None - sid = getattr(session_entry, "session_id", None) or "" - if not sid: + session_entry = await self._session_entry_for_manager(event, "goal manager") + if session_entry is None: return None, None max_turns = self._goal_max_turns_from_config() - return GoalManager(session_id=sid, default_max_turns=max_turns), session_entry + return GoalManager(session_id=session_entry.session_id, default_max_turns=max_turns), session_entry async def _get_heartbeat_manager_for_event(self, event: "MessageEvent"): - """Return a HeartbeatManager bound to the session for this event. - - Returns ``(manager, session_entry)`` or ``(None, None)``. - """ + """Return ``(HeartbeatManager, session_entry)`` for this event, or ``(None, None)``.""" try: from hermes_cli.heartbeat import HeartbeatManager except Exception as exc: logger.debug("heartbeat manager unavailable: %s", exc) return None, None - # Warm the SessionDB cache off-loop. A cold cache can drop the - # first /heartbeat write while the reply claims it was set. - await self._warm_goals_session_db("heartbeat manager") - try: - # Same reset-policy contract as _get_goal_manager_for_event: - # internal events look up the session without touching activity. - session_entry = await self.async_session_store.get_or_create_session( - event.source, - touch_activity=not bool(getattr(event, "internal", False)), - ) - except Exception as exc: - logger.debug("heartbeat manager: session lookup failed: %s", exc) + session_entry = await self._session_entry_for_manager(event, "heartbeat manager") + if session_entry is None: return None, None - sid = getattr(session_entry, "session_id", None) or "" - if not sid: - return None, None - return HeartbeatManager(session_id=sid), session_entry + return HeartbeatManager(session_id=session_entry.session_id), session_entry def _register_heartbeat_watch(self, quick_key: str, source: Any, session_id: str) -> None: """Track a session with an active heartbeat and start the poller. @@ -22612,11 +20962,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _defer_goal_status_notice_after_delivery(self, source: Any, message: str) -> None: """Send a /goal status line after the main response is delivered. - The message handler returns the agent response to the adapter, which sends it after this - caller has returned; for natural reading order the goal status belongs after that send. - Platform adapters provide a one-shot post-delivery callback for exactly this boundary; - when unavailable, fall back to direct awaited delivery rather than silently dropping the - notice. + The adapter sends the agent response after this caller returns, so for reading order the + status must follow that send: use the adapter's one-shot post-delivery callback when + available, else fall back to direct awaited delivery rather than dropping the notice. """ adapter = self._adapter_for_source(source) if not adapter: @@ -22661,9 +21009,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Run the goal judge after a gateway turn and, if still active, enqueue a continuation prompt for the same session. - Called from ``_handle_message_with_agent`` at turn boundary, AFTER the response has been - delivered. We use the adapter's pending-message / FIFO machinery so any real user message - that arrives simultaneously is handled by the same queue and takes priority naturally. + Called at turn boundary AFTER delivery. Uses the adapter's pending-message/FIFO machinery + so a simultaneous real user message is handled by the same queue and takes priority. """ try: from hermes_cli.goals import GoalManager @@ -22677,9 +21024,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew max_turns = self._goal_max_turns_from_config() - # Warm the SessionDB cache off-loop. A cold cache runs the state.db init on the loop thread - # at the turn boundary (the 2026-08-14 crash-loop seam). A slow init can drop the goal read - # and silently end the goal loop. + # Warm the SessionDB cache off-loop: a cold cache runs the state.db init on the loop thread + # at the turn boundary; a slow init can drop the goal read and silently end the goal loop. await self._warm_goals_session_db("goal continuation") mgr = GoalManager(session_id=sid, default_max_turns=max_turns) @@ -22692,12 +21038,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: _bg_procs = None - # evaluate_after_turn calls judge_goal() which makes a synchronous HTTP request to the - # auxiliary LLM. Running it on the event-loop thread would block Discord heartbeats for - # 10-40 s and cause connection flaps, so we offload it to a thread-pool executor. - # _run_in_executor_with_context (not bare run_in_executor): the profile secret scope and - # aux runtime context are contextvars; a default-executor hop drops them and aux-client - # credential resolution fails under multiplexing. + # evaluate_after_turn calls judge_goal(), a synchronous HTTP request to the auxiliary LLM; + # on the event-loop thread it blocks Discord heartbeats 10-40 s and flaps connections, so it + # is offloaded to a thread-pool executor. _run_in_executor_with_context (not bare + # run_in_executor): the profile secret scope and aux runtime context are contextvars; a + # default-executor hop drops them and aux credential resolution fails under multiplexing. decision = await self._run_in_executor_with_context( lambda: mgr.evaluate_after_turn( final_response or "", @@ -22819,9 +21164,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not sid: return - # Warm the SessionDB cache off-loop. A cold cache at the turn boundary stalls the loop for - # the init duration and can drop the tick-completion write (the /goal continuation seam, one - # sibling over). + # Warm the SessionDB cache off-loop: a cold cache at the turn boundary stalls the loop for + # the init duration and can drop the tick-completion write (the /goal continuation seam). await self._warm_goals_session_db("loop completion") mgr = LoopManager(session_id=sid) @@ -22857,9 +21201,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew list_active_loops, ) - # Warm the cache off-loop once per scan. The scan reads - # every persisted loop, so a cold cache runs the state.db - # init on the loop thread before the first read. + # Warm the cache off-loop once per scan: the scan reads every persisted loop, so a + # cold cache would run the state.db init on the loop thread before the first read. await self._warm_goals_session_db("loop wakeup") now = time.time() @@ -22934,10 +21277,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew mgr.complete_tick("") except Exception as exc: logger.warning("loop wakeup injection failed for %s: %s", sid, exc) - try: + with suppress(Exception): mgr.abandon_tick() - except Exception: - pass except Exception as exc: logger.debug("loop wakeup watcher error: %s", exc) await asyncio.sleep(interval) @@ -23150,9 +21491,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass - # Build a synthetic MessageEvent and feed through the normal pipeline - # Use SimpleNamespace as raw_message so _get_guild_id() can extract - # guild_id and _send_voice_reply() plays audio in the voice channel. + # Build a synthetic MessageEvent for the normal pipeline; SimpleNamespace raw_message lets + # _get_guild_id() extract guild_id and _send_voice_reply() play audio in the voice channel. from types import SimpleNamespace # Resolve the bound text channel's channel_prompt so voice input gets # the same per-channel context as typed messages (#50149). @@ -23183,14 +21523,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> bool: """Decide whether the runner should send a TTS voice reply. - Returns False when: - - voice_mode is off for this chat - - response is empty or an error - - agent already called text_to_speech tool (dedup) - - voice input and base adapter auto-TTS already handled it (skip_double) - UNLESS streaming already consumed the response (already_sent=True), - in which case the base adapter won't have text for auto-TTS so the - runner must handle it. + False when voice_mode is off for this chat, the response is empty/an error, the agent + already called text_to_speech (dedup), or voice input + base adapter auto-TTS already + handled it (skip_double) — UNLESS streaming consumed the response (already_sent=True), + since then the base adapter has no text for auto-TTS and the runner must handle it. """ if not response or response.startswith("Error:"): return False @@ -23211,9 +21547,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew should = ( (voice_mode == "all") or (voice_mode == "voice_only" and is_voice_input) - # ``voice.auto_tts`` is synced into the adapter on gateway startup. - # It is the fallback only when the chat has no explicit mode; - # otherwise the chat-level all/voice_only/off choice takes precedence. + # ``voice.auto_tts`` (synced into the adapter at startup) is the fallback only when the + # chat has no explicit mode; the chat-level all/voice_only/off choice takes precedence. or (voice_mode is None and adapter_auto_tts) ) if not should: @@ -23241,13 +21576,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return False # Dedup: base adapter auto-TTS already handles voice input (play_tts plays in VC when - # connected, so runner can skip). When streaming already delivered the text - # (already_sent), the base adapter receives None and can't run auto-TTS, so the runner - # must take over. - if is_voice_input and not already_sent: - return False - - return True + # connected), so the runner can skip — unless streaming already delivered the text + # (already_sent): then the base adapter gets None, can't run auto-TTS, and the runner must. + return not (is_voice_input and not already_sent) def _should_echo_stt_transcripts(self) -> bool: """Return whether inbound voice/STT transcripts should be echoed to chat.""" @@ -23264,10 +21595,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not tts_text: return - # Platform-aware output path: platforms whose native voice bubbles require Ogg/Opus - # (OPUS_VOICE_PLATFORMS — Telegram, Matrix, Feishu, WhatsApp, Signal) get an explicit - # .ogg path; the TTS tool's central container repair guarantees real Ogg/Opus bytes for - # every provider. + # Platforms whose native voice bubbles require Ogg/Opus (OPUS_VOICE_PLATFORMS — + # Telegram, Matrix, Feishu, WhatsApp, Signal) get an explicit .ogg path; the TTS tool's + # central container repair guarantees real Ogg/Opus bytes for every provider. audio_path = build_auto_tts_output_path(event.source.platform) result_json = await asyncio.to_thread( @@ -23279,9 +21609,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.warning("Auto voice reply TTS returned invalid JSON: %s", result_json[:200] if result_json else result_json) return - # Final delivery may be one combined file or multiple separately - # valid files when combination is unavailable or would exceed a - # platform limit. Preserve legacy single-file results. + # Delivery may be one combined file or several separately valid files (combination + # unavailable or over a platform limit); preserve legacy single-file results. actual_paths = result.get("file_paths") or [ result.get("file_path", audio_path) ] @@ -23335,10 +21664,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.warning("Auto voice reply failed: %s", e, exc_info=True) finally: for p in ({audio_path, *actual_paths} - {None}): - try: + with suppress(OSError): os.unlink(p) - except OSError: - pass async def _deliver_media_from_response( self, @@ -23347,16 +21674,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew adapter, thread_metadata: Optional[Dict[str, Any]] = None, ) -> None: - """Extract explicit MEDIA: tags from a response and deliver them. + """Extract explicit MEDIA: tags from an already-streamed response and deliver them. - Called after streaming has already sent the text to the user, so the text itself is - already delivered — this only handles file attachments that the normal - _process_message_background path would have caught. - - Unlike the non-streaming path in ``gateway/platforms/base.py``, this rescan is - EXPLICIT-ONLY: a bare local path in an already-streamed reply was either shown to the user - as text or is stale inspected content, and promoting it sent files the model never asked - to deliver. Only ``MEDIA:`` directives trigger post-stream uploads. + The text is already delivered; this only handles file attachments the normal + _process_message_background path would have caught. Unlike the non-streaming path in + ``gateway/platforms/base.py`` this rescan is EXPLICIT-ONLY: a bare local path in a + streamed reply was either shown as text or is stale inspected content, and promoting it + sent files the model never asked to deliver. """ from pathlib import Path from urllib.parse import quote as _quote @@ -23546,11 +21870,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> list: """Resolve enabled toolsets for an agent run, honoring per-source overrides. - Asks the receiving adapter for a ``toolsets_for_source()`` override (e.g. per-route - webhook toolsets). When present, the override list is validated through the SAME - ``_get_platform_tools`` path as normal platform config — by substituting it as the - platform's toolset list — so unknown names and platform-restricted toolsets are dropped - rather than trusted. When absent, falls back to ``platform_toolsets.``. + An adapter ``toolsets_for_source()`` override (e.g. per-route webhook toolsets) is + validated through the SAME ``_get_platform_tools`` path as normal platform config, so + unknown and platform-restricted toolsets are dropped rather than trusted. Absent an + override, falls back to ``platform_toolsets.``. """ from hermes_cli.tools_config import _get_platform_tools @@ -23689,9 +22012,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not response and result and result.get("error"): response = f"Error: {result['error']}" - # Background tasks start a fresh conversation (no prior history), - # so history_offset=0: every message in the run belongs to this - # turn. Mirrors the repair on the main turn path. + # Background tasks start a fresh conversation, so history_offset=0: every message in the + # run belongs to this turn. Mirrors the repair on the main turn path. if response: response = repair_explicit_computer_use_media_paths( response, @@ -23723,19 +22045,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Send extracted images for image_url, alt_text in (images or []): - try: + with suppress(Exception): await adapter.send_image( chat_id=source.chat_id, image_url=image_url, caption=alt_text, metadata=_thread_metadata, ) - except Exception: - pass - # Send media files, routing each by type so a TTS clip - # arrives as a voice bubble / a clip as a video rather than - # a generic document. Mirrors the streaming + kanban paths. + # Route each media file by type so a TTS clip arrives as a voice bubble and a clip + # as a video rather than a generic document. Mirrors the streaming + kanban paths. from gateway.platforms.base import ( should_send_media_as_audio as _should_send_media_as_audio, ) @@ -23781,14 +22100,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as e: logger.exception("Background task %s failed", task_id) - try: + with suppress(Exception): await adapter.send( chat_id=source.chat_id, content=f"❌ Background task {task_id} failed: {e}", metadata=_thread_metadata, ) - except Exception: - pass async def _get_telegram_topic_capabilities(self, source: SessionSource) -> dict: """Read Telegram private-topic capability flags via Bot API getMe.""" @@ -23919,17 +22236,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """(thread_id, initial_name) when the RELAY connector auto-threaded our reply to this source's chat — the title-turn sibling of _is_discord_auto_thread_lane. - The marker-based check above only lights up for events ARRIVING IN an auto-created thread - (turn 2+); the auto-title fires on the FIRST exchange, whose source is the PARENT channel - event, so no markers exist and the native lane never matches on the relay title turn. - - Preferred: the connector stamps ``prospective_thread_id`` on the inbound (the anchor - message id == the thread it will auto-create). Deterministic and per-message, so it names - the EXACT thread even when several auto-threads spawn from one channel — unlike the - per-chat send-result cache, which only ever renamed the FIRST thread. The connector's own - created-name guard (prefer_connector_created) enforces no-clobber, so no initial name is - needed here. Fallback: the send-result thread_id/auto_thread_name cached per chat by the - relay adapter, kept for older connectors. + The marker check only matches events ARRIVING IN an auto-created thread (turn 2+); the + auto-title fires on the FIRST exchange, whose source is the PARENT channel event with no + markers. Preferred: the connector's ``prospective_thread_id`` stamp (anchor message id == + the thread it will create) — per-message, so it names the EXACT thread even when several + auto-threads spawn from one channel; the connector's created-name guard enforces + no-clobber. Fallback: the per-chat send-result thread_id/auto_thread_name cache (older + connectors), which only ever renamed the FIRST thread. """ if source.platform != Platform.DISCORD or not source.chat_id: return None @@ -24031,13 +22344,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew else getattr(source, "auto_thread_initial_name", None) ) thread_name = self._sanitize_discord_thread_title(title) - # Relay lane only: the connector's egress guard resolves the owning tenant from the outbound - # metadata's scope_id (guild) / user_id (author). Those caches are keyed by the PARENT - # channel chat_id (learned at inbound), not the thread id; rename_thread defaults chat_id - # to the thread id, so the lookup misses and the connector declines ("target not routed to - # an onboarded tenant"). Pass the parent channel id (the relay source's chat_id IS the - # parent) so the discriminators resolve. Native lane needs nothing: its source IS - # the thread and it renames via the direct Discord API, not the relay egress guard. + # Relay lane only: the connector's egress guard resolves the owning tenant from the + # outbound scope_id/user_id caches, keyed by the PARENT channel chat_id (learned at + # inbound), not the thread id. rename_thread defaults chat_id to the thread id, so the + # lookup misses and the connector declines; pass the parent channel id (the relay source's + # chat_id). Native lane needs nothing: its source IS the thread, direct Discord API. parent_chat_id = ( str(source.chat_id) if use_connector_guard and source.chat_id else None ) @@ -24075,29 +22386,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: logger.debug("Failed to rename Discord auto-thread for generated session title", exc_info=True) - def _schedule_discord_semantic_thread_rename( - self, - source: SessionSource, - session_id: str, - title: str, - ) -> None: - """Schedule Discord auto-thread rename from the auto-title background thread.""" - relay_info = None - if not title: - return - if not self._is_discord_auto_thread_lane(source): - # Relay title turn: the source is the PARENT channel event (the - # thread didn't exist at ingest, so no auto-thread markers). The - # connector's send-result feedback tells us where the reply - # landed — but the auto-title thread races the delivery that - # produces it, so a cache miss HERE is not a verdict. Schedule - # whenever the SHAPE matches; the async rename lane polls the - # cache (with a bounded wait) and no-ops on a true miss. - relay_info = self._relay_auto_thread_info(source) - if relay_info is None and not self._is_relay_discord_channel_lane( - source - ): - return + def _schedule_rename_from_title_thread(self, source: SessionSource, make_coro, label: str) -> None: + """Schedule a best-effort rename coroutine onto the gateway loop from the auto-title thread. + + The source is copied so the background thread never shares the live dataclass with the + loop; failures are logged at debug and never propagate.""" try: loop = asyncio.get_running_loop() except RuntimeError: @@ -24109,12 +22402,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: copied_source = source future = safe_schedule_threadsafe( - self._rename_discord_auto_thread_for_session_title( - copied_source, session_id, title, relay_info=relay_info - ), + make_coro(copied_source), loop, logger=logger, - log_message="Discord semantic thread rename failed to schedule", + log_message=f"{label} failed to schedule", ) if future is None: return @@ -24123,10 +22414,39 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: fut.result() except Exception: - logger.debug("Discord semantic thread rename failed", exc_info=True) + logger.debug("%s failed", label, exc_info=True) future.add_done_callback(_log_rename_failure) + def _schedule_discord_semantic_thread_rename( + self, + source: SessionSource, + session_id: str, + title: str, + ) -> None: + """Schedule Discord auto-thread rename from the auto-title background thread.""" + relay_info = None + if not title: + return + if not self._is_discord_auto_thread_lane(source): + # Relay title turn: the source is the PARENT channel event (thread didn't exist at + # ingest, no auto-thread markers). The connector's send-result feedback says where the + # reply landed, but the auto-title races that delivery, so a cache miss HERE is not a + # verdict. Schedule whenever the SHAPE matches; the async rename lane polls the cache + # (bounded wait) and no-ops on a true miss. + relay_info = self._relay_auto_thread_info(source) + if relay_info is None and not self._is_relay_discord_channel_lane( + source + ): + return + self._schedule_rename_from_title_thread( + source, + lambda copied: self._rename_discord_auto_thread_for_session_title( + copied, session_id, title, relay_info=relay_info + ), + "Discord semantic thread rename", + ) + async def _rename_telegram_topic_for_session_title( self, source: SessionSource, @@ -24137,9 +22457,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not await asyncio.to_thread(self._is_telegram_topic_lane, source) or not source.chat_id or not source.thread_id: return - # Operator can fully disable per-topic auto-rename via extra.disable_topic_auto_rename. - # Useful when topics are managed by the user (ad-hoc Threaded Mode) and auto-rename would - # overwrite their chosen names every time the auto-title fires. + # extra.disable_topic_auto_rename lets the operator disable per-topic auto-rename entirely, + # e.g. user-managed topics (ad-hoc Threaded Mode) that auto-rename would keep overwriting. if self._telegram_topic_auto_rename_disabled(source): return @@ -24211,8 +22530,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _telegram_topic_auto_rename_disabled(self, source: SessionSource) -> bool: """Return True when operator disabled per-topic auto-rename for this Telegram chat. - Controlled via ``gateway.platforms.telegram.extra.disable_topic_auto_rename``. - Default is False (auto-rename enabled, preserves prior behaviour). + ``gateway.platforms.telegram.extra.disable_topic_auto_rename``; default False (auto-rename on). """ platform_cfg = ( self.config.platforms.get(source.platform) @@ -24242,39 +22560,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return if self._telegram_topic_auto_rename_disabled(source): return - try: - loop = asyncio.get_running_loop() - except RuntimeError: - loop = getattr(self, "_gateway_loop", None) - if loop is None or loop.is_closed(): - return - try: - copied_source = dataclasses.replace(source) - except Exception: - copied_source = source - future = safe_schedule_threadsafe( - self._rename_telegram_topic_for_session_title(copied_source, session_id, title), - loop, - logger=logger, - log_message="Telegram topic title rename failed to schedule", + self._schedule_rename_from_title_thread( + source, + lambda copied: self._rename_telegram_topic_for_session_title(copied, session_id, title), + "Telegram topic title rename", ) - if future is None: - return - def _log_rename_failure(fut) -> None: - try: - fut.result() - except Exception: - logger.debug("Telegram topic title rename failed", exc_info=True) - - future.add_done_callback(_log_rename_failure) _TELEGRAM_CAPABILITY_HINT_COOLDOWN_S = 300.0 def _should_send_telegram_capability_hint(self, source: SessionSource) -> bool: """Rate-limit the BotFather Threads Settings screenshot. - If a user sends /topic repeatedly while Threads Settings are still - off, we shouldn't keep re-uploading the screenshot every time. + Repeated /topic while Threads Settings are still off must not re-upload it every time. """ if not hasattr(self, "_telegram_capability_hint_ts"): self._telegram_capability_hint_ts = {} @@ -24463,11 +22760,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _execute_mcp_reload(self, event: MessageEvent) -> str: """Actually disconnect, reconnect, and notify MCP tool changes. - Split out from ``_handle_reload_mcp_command`` so the confirmation wrapper can invoke the - same path whether the user confirmed via button, text reply, or has the confirm gate - disabled. Under multiplex the reload runs inside the requesting profile's runtime scope - (entered here when the caller — e.g. a button callback — did not), and only that - profile's servers are torn down and rediscovered. + Split out so the confirmation wrapper can invoke the same path for button, text reply, + or disabled confirm gate. Under multiplex the reload runs inside the requesting profile's + runtime scope (entered here when the caller did not) and only that profile's servers are + torn down and rediscovered. """ multiplex = bool(getattr(self.config, "multiplex_profiles", False)) if multiplex and not get_hermes_home_override(): @@ -24537,9 +22833,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _cache = getattr(self, "_agent_cache", None) _cache_lock = getattr(self, "_agent_cache_lock", None) if _cache_lock is not None and _cache: - # Multiplex: only this profile's sessions. Rebuilding - # another profile's agent inside this scope would hand it - # this profile's tool registry. + # Multiplex: only this profile's sessions; rebuilding another profile's agent in + # this scope would hand it this profile's tool registry. _ns_prefix = ( _session_key_namespace(event.source.profile) + ":" if multiplex else None @@ -24566,9 +22861,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _exc, ) - # Inject a message at the END of the session history so the - # model knows tools changed on its next turn. Appended after - # all existing messages to preserve prompt-cache for the prefix. + # Inject a message at the END of the session history so the model knows tools changed + # next turn; appending after all existing messages preserves the prompt-cache prefix. change_parts = [] if added: change_parts.append(f"Added servers: {', '.join(sorted(added))}") @@ -24596,21 +22890,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.warning("MCP reload failed: %s", e) return t("gateway.reload_mcp.failed", error=e) - # ------------------------------------------------------------------ - # Slash-command confirmation primitive (generic) - # ------------------------------------------------------------------ - # Used by slash commands that have a non-destructive but expensive - # side effect worth an explicit user confirmation (currently only - # /reload-mcp, which invalidates the prompt cache). Two delivery - # paths: - # 1. Button UI — adapters that override ``send_slash_confirm`` - # (Telegram, Discord, Slack, Matrix, Feishu) render three - # inline buttons. The adapter routes the button click back via - # ``tools.slash_confirm.resolve(session_key, confirm_id, choice)``. - # 2. Text fallback — adapters that don't override the hook get a - # plain text prompt. Users reply with /approve, /always, or - # /cancel; the early intercept in ``_handle_message`` matches - # those replies against ``tools.slash_confirm.get_pending()``. + # Slash-command confirmation primitive (generic): for slash commands with an expensive side + # effect worth explicit confirmation (currently /reload-mcp, which invalidates the prompt + # cache). Two delivery paths: adapters overriding ``send_slash_confirm`` render inline buttons + # and route the click back via ``tools.slash_confirm.resolve(session_key, confirm_id, choice)``; + # others get a text prompt answered with /approve, /always, or /cancel, matched in + # ``_handle_message`` against ``tools.slash_confirm.get_pending()``. async def _maybe_confirm_destructive_slash( self, @@ -24623,12 +22908,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Union[str, "EphemeralReply", None]: """Gate a destructive session slash command (/new, /reset, /undo). - ``execute`` is an async callable ``execute() -> str | EphemeralReply`` that performs the - destructive action. If the ``approvals.destructive_slash_confirm`` gate is off, ``execute`` - runs immediately. Otherwise this routes through ``_request_slash_confirm`` — native - yes/no buttons on Telegram/Discord/Slack, text fallback elsewhere. Resolution: ``once`` - runs ``execute``; ``always`` persists ``destructive_slash_confirm: false`` then runs it; - ``cancel`` returns a "cancelled" message without running it. + ``execute`` is an async ``execute() -> str | EphemeralReply`` performing the action. It + runs immediately if ``approvals.destructive_slash_confirm`` is off; otherwise this routes + through ``_request_slash_confirm`` (native buttons or text fallback): ``once`` runs it, + ``always`` persists ``destructive_slash_confirm: false`` then runs it, ``cancel`` returns + a "cancelled" message without running it. """ # Gate check. confirm_required = True @@ -24727,22 +23011,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[str]: """Ask the user to confirm an expensive slash command. - ``handler`` is an async callable ``handler(choice: str) -> str`` where ``choice`` is - ``"once"``, ``"always"``, or ``"cancel"``. The handler runs on the event loop when the - user responds; its return value is sent back as a gateway message. - - Returns a short acknowledgment string to send immediately (before - the user's response). If buttons rendered successfully the ack - is ``None`` (buttons are self-explanatory); if we fell back to - text the message itself IS the ack. + ``handler(choice: str) -> str`` runs on the event loop when the user responds with + ``"once"``, ``"always"``, or ``"cancel"``; its return value is sent as a gateway message. + Returns the immediate acknowledgment: ``None`` if buttons rendered (self-explanatory), + otherwise the text-fallback message itself IS the ack. """ from tools import slash_confirm as _slash_confirm_mod source = event.source session_key = self._session_key_for_source(source) - # Bare-runner test harnesses (object.__new__(GatewayRunner)) skip __init__ and don't have - # the counter attribute — fall back to a local counter so tests don't AttributeError. Real - # runs always have the instance attribute. + # Bare-runner test harnesses (object.__new__(GatewayRunner)) skip __init__ and lack the + # counter attribute; fall back to a local counter. Real runs always have the attribute. counter = getattr(self, "_slash_confirm_counter", None) if counter is None: import itertools as _itertools @@ -24809,13 +23088,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew reply_to_message_id=reply_to_message_id or getattr(source, "message_id", None), ) if getattr(source, "platform", None) == Platform.SLACK: - # Per-turn egress identity (R3-5, connector PR gateway-gateway#210). Slack's - # chat.startStream needs recipient_user_id/team_id, which the relay connector fills - # from metadata.user_id/scope_id. The relay adapter's _with_scope fallback reads both - # from per-chat caches keyed only by chat_id — mutable state a CONCURRENT turn - # overwrites (U1's stream opened with U2 as recipient). Stamp this turn's authentic - # values here; _with_scope only fills absent keys, so the cache degrades to a - # restart/synthetic-send fallback. + # Per-turn egress identity. Slack's chat.startStream needs recipient_user_id/team_id, + # which the relay connector fills from metadata.user_id/scope_id; the relay adapter's + # _with_scope fallback reads both from per-chat caches keyed only by chat_id — mutable + # state a CONCURRENT turn overwrites (U1's stream opened with U2 as recipient). Stamp + # this turn's authentic values; _with_scope only fills absent keys, so the cache is + # reduced to a restart/synthetic-send fallback. team_id = getattr(source, "scope_id", None) user_id = getattr(source, "user_id", None) if team_id or user_id: @@ -24856,9 +23134,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew adapter=adapter, ): metadata["telegram_dm_topic_reply_fallback"] = True - # Telegram DM topic lanes need direct_messages_topic_id in metadata - # so synthetic/queued messages (goal continuations, status notices) - # route to the correct topic even when reply anchor is unavailable. + # Telegram DM topic lanes need direct_messages_topic_id in metadata so synthetic/queued + # messages (goal continuations, status notices) reach the topic without a reply anchor. tid = str(thread_id) if tid and tid not in {"", "1"}: metadata["direct_messages_topic_id"] = tid @@ -24942,9 +23219,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Watch ``hermes update --gateway``, streaming output + forwarding prompts. - Polls ``.update_output.txt`` for new content and sends chunks to the user periodically. - Detects ``.update_prompt.json`` (written by the update process when it needs user input) - and forwards the prompt to the messenger. + Polls ``.update_output.txt`` for new content and sends chunks to the user periodically; + detects ``.update_prompt.json`` (written when the update process needs input) and forwards it. """ pending_path = _hermes_home / ".update_pending.json" claimed_path = _hermes_home / ".update_pending.claimed.json" @@ -24990,11 +23266,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not adapter or not chat_id: logger.warning("Update watcher: cannot resolve adapter/chat_id, falling back to completion-only") - # Fall back to completion-only: wait for the exit code and send the final notification. - # _send_update_notification re-resolves the adapter on every call, so when the target - # platform is still reconnecting it returns False and keeps the markers. Keep polling - # until it actually delivers (True) — a platform that reconnects a few seconds after - # completion would otherwise never be notified. + # Completion-only fallback: wait for the exit code, then keep polling until + # _send_update_notification actually delivers (True) — it re-resolves the adapter each + # call and returns False (markers kept) while the platform is still reconnecting. while (pending_path.exists() or claimed_path.exists()) and loop.time() < deadline: if exit_code_path.exists() and await self._send_update_notification(): return @@ -25147,9 +23421,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew f"or type your answer directly.", metadata=_non_conversational_metadata(metadata, platform=platform), ) - # Keep the prompt marker on disk until the user answers. If the gateway - # restarts mid-prompt, the next watcher can recover by re-forwarding it from - # disk. + # Keep the prompt marker on disk until the user answers so a watcher after a + # mid-prompt gateway restart can recover by re-forwarding it. self._session_state( session_key ).persistent.update_prompt_pending = True @@ -25165,14 +23438,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.warning("Update watcher timed out after %.0fs", timeout) exit_code_path.write_text("124", encoding="utf-8") await _flush_buffer() - try: + with suppress(Exception): await adapter.send( chat_id, "❌ Hermes update timed out after 30 minutes.", metadata=_non_conversational_metadata(metadata, platform=platform), ) - except Exception: - pass for p in (pending_path, claimed_path, output_path, exit_code_path, prompt_path): p.unlink(missing_ok=True) @@ -25184,8 +23455,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _send_update_notification(self) -> bool: """If an update finished, notify the user. - Returns False when the update is still running so a caller can retry - later. Returns True after a definitive send/skip decision. + False while the update is still running (caller may retry); True after a definitive send/skip. """ pending_path = _hermes_home / ".update_pending.json" claimed_path = _hermes_home / ".update_pending.claimed.json" @@ -25346,10 +23616,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "♻ Gateway restarted successfully. Your session continues.", metadata=_non_conversational_metadata(metadata, platform=platform), ) - # adapter.send() catches provider errors (e.g. "Chat not found") - # and returns SendResult(success=False) rather than raising, so - # we must inspect the result before claiming success — otherwise - # the log line is misleading and hides real delivery failures. + # adapter.send() catches provider errors (e.g. "Chat not found") and returns + # SendResult(success=False) rather than raising, so inspect the result before claiming + # success — otherwise the log line hides real delivery failures. if result is not None and getattr(result, "success", True) is False: logger.warning( "Restart notification to %s:%s was not delivered: %s", @@ -25538,11 +23807,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _set_session_env(self, context: SessionContext) -> list: """Set session context variables for the current async task. - Uses ``contextvars`` instead of ``os.environ`` so that concurrent - gateway messages cannot overwrite each other's session state. - - Returns a list of reset tokens; pass them to ``_clear_session_env`` - in a ``finally`` block. + Uses ``contextvars`` rather than ``os.environ`` so concurrent gateway messages cannot + overwrite each other's state. Returns reset tokens for ``_clear_session_env`` in a + ``finally`` block. """ from gateway.session_context import set_session_vars # Propagate the adapter's async-delivery capability so async tools (terminal @@ -25659,13 +23926,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Resolve image-input routing for the effective model this turn. - Returns ``"native"`` (attach pixels on the user turn) or ``"text"`` - (pre-analyze with vision_analyze and prepend the description). See - agent/image_routing.py for the full decision table. - - Gateway sessions can have /model overrides outside config.yaml, and image preprocessing - runs before AIAgent sets the auxiliary_client runtime globals — so resolve the same - per-session runtime bundle the upcoming turn will use, not just the persisted default. + Returns ``"native"`` (attach pixels on the user turn) or ``"text"`` (pre-analyze with + vision_analyze and prepend the description); see agent/image_routing.py. Gateway sessions + can carry /model overrides and image preprocessing runs before AIAgent sets the + auxiliary_client runtime globals, so resolve the per-session runtime bundle the upcoming + turn will use, not just the persisted default. """ try: from agent.image_routing import decide_image_input_mode @@ -25727,16 +23992,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Auto-analyze user-attached images with the vision tool and prepend the descriptions to the message text. - Each image is analyzed with a general-purpose prompt. The resulting description *and* the - local cache path are injected so the model can: 1. Immediately understand what the user - sent (no extra tool call). 2. Re-examine the image with vision_analyze if it needs detail. - - Args: - user_text: The user's original caption / message text. - image_paths: List of local file paths to cached images. - - Returns: - The enriched message string with vision descriptions prepended. + Description *and* local cache path are injected so the model understands the image without + a tool call and can re-examine it with vision_analyze. Returns the enriched message string. """ from tools.vision_tools import vision_analyze_tool from agent.memory_manager import sanitize_context @@ -25796,18 +24053,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """Auto-transcribe user voice/audio messages using the configured STT provider and prepend the transcript to the message text. - Args: - user_text: The user's original caption / message text. - audio_paths: List of local file paths to cached audio files. - - Returns: - A tuple of ``(enriched_text, successful_transcripts)``: - - ``enriched_text``: the message string with transcription wrappers - prepended (same as before). - - ``successful_transcripts``: the raw transcript strings for audio - clips that were successfully transcribed, in input order. Empty - list if every clip failed or STT is disabled. Callers can use - this to echo transcripts back to the user before the agent loop. + Returns ``(enriched_text, successful_transcripts)``: the message with transcription + wrappers prepended, and the raw transcripts of successfully transcribed clips in input + order (empty if every clip failed or STT is disabled) so callers can echo them back to + the user before the agent loop. """ seen = set() audio_paths = [p for p in audio_paths if p not in seen and not seen.add(p)] @@ -25868,9 +24117,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew result = fallback if result["success"]: transcript = result["transcript"] - # Speech-to-text can return success=True with an empty or whitespace-only - # transcript on silence, cut-off, or inaudible audio. Emitting empty quotes makes - # the agent reply to nothing and can loop, so that case gets a sentinel note. + # STT may return success=True with an empty/whitespace transcript (silence, cut-off); + # empty quotes make the agent reply to nothing and can loop, so emit a sentinel note. if not (transcript or "").strip(): enriched_parts.append( "[The user sent a voice message but it came through " @@ -25880,17 +24128,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) continue successful_transcripts.append(transcript) - # Pass the transcript through as a plain quoted line. Wrapping it in "The user - # sent a voice message..." read as a meta-instruction and made the LLM comment - # on voice mode instead of replying to the content. + # Pass the transcript as a plain quoted line: a "The user sent a voice message..." + # wrapper read as a meta-instruction and made the LLM comment on voice mode instead. enriched_parts.append(f'"{transcript}"') else: error = result.get("error", "unknown error") - # All failure branches: a single, minimal, neutral marker. Do NOT mention "no STT - # provider configured", setup instructions, or claim a DM was sent — those get - # persisted in history and poison later turns (the model keeps volunteering - # STT-setup advice after transcription works). The cause is logged - # for operator diagnosis but kept out of the LLM-visible prompt. + # All failure branches: one minimal neutral marker. Never mention "no STT provider", + # setup steps, or a DM sent — persisted in history they poison later turns (the model + # keeps volunteering STT-setup advice). Cause is logged for operators, not the prompt. logger.info("Voice transcription failed for %s: %s", path, error) from tools.credential_files import to_agent_visible_cache_path @@ -25937,9 +24182,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> tuple[str | None, List[str]]: """Transcribe a pending audio event once and cache the result on the event. - Voice follow-ups can be inspected first by the interrupt monitor and later consumed by - the pending-drain path. Both need the same transcript, but only one STT call and one - transcript echo should happen for the platform message. + The interrupt monitor and the pending-drain path both need the transcript; caching keeps + it to one STT call and one transcript echo per platform message. """ if hasattr(event, "_gateway_pending_stt_text"): cached_text = getattr(event, "_gateway_pending_stt_text") @@ -25971,12 +24215,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Echo pending-event STT transcripts to the chat at most once. - The already-echoed transcripts are tracked as a COUNT rather than a single boolean: - ``merge_pending_message_event`` can append a second voice note after the first transcript - was echoed and invalidate the cache; the re-run returns the earlier transcripts as a prefix - of the new list, so echoing only the unsent tail suppresses the repeat. A - count rather than a set of seen values because two separate notes that transcribe - identically are two distinct deliveries and both must be echoed. + Tracked as a COUNT (not a set — identical transcripts are distinct deliveries): + ``merge_pending_message_event`` can append a second voice note and invalidate the cache, + and the re-run returns earlier transcripts as a prefix, so only the unsent tail is echoed. """ if ( not transcripts @@ -26009,10 +24250,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> tuple[str, List[str]]: """Transcribe a pending voice event and echo transcripts once. - Returns ``(enriched_text, transcripts)`` so the caller can feed the enriched text into - ``agent.interrupt()`` or the pending-drain flow. If the event has no STT-eligible media, - returns ``(text, [])`` unchanged; the caller owns the ``_build_media_placeholder`` - fallback when ``text`` is empty and the event has non-audio media. + Returns ``(enriched_text, transcripts)`` for ``agent.interrupt()`` or the pending-drain + flow; ``(text, [])`` unchanged when there is no STT-eligible media (caller owns the + ``_build_media_placeholder`` fallback for empty ``text`` with non-audio media). """ if not self._pending_event_audio_paths(event): return text, [] @@ -26041,8 +24281,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _build_process_event_source(self, evt: dict): """Resolve the canonical source for a synthetic background-process event. - Prefer the persisted session-store origin for the event's session key. Falling back to - the currently active foreground event is what causes cross-topic bleed, so don't do that. + Prefer the persisted session-store origin; the active foreground event causes cross-topic bleed. """ from gateway.session import SessionSource @@ -26089,9 +24328,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew try: platform = Platform(platform_name) - # Reject arbitrary strings that create dynamic pseudo-members. - # Built-in platforms are always valid; plugin platforms must be - # registered in the platform registry. + # Reject arbitrary strings (dynamic pseudo-members): built-ins are always valid, plugin + # platforms must be registered in the platform registry. if platform.value not in _BUILTIN_PLATFORM_VALUES: try: from gateway.platform_registry import platform_registry @@ -26108,11 +24346,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew scope_id = str(evt.get("scope_id") or "").strip() or None if scope_id is None and chat_type not in ("dm", "thread"): - # Reconstructed (non-persisted) source for a scoped chat with no scope discriminator: on - # a relay-fronted deployment the connector's fail-closed tenant guard may decline the - # reply unless user_id resolves it (resolveByUser). Don't fail here — DMs and - # author-bound chats still route, and native adapters don't need scope_id — but warn - # so a post-restart egress decline isn't silent. + # Reconstructed (non-persisted) source for a scoped chat with no scope discriminator: a + # relay connector's fail-closed tenant guard may decline the reply unless user_id resolves it + # (resolveByUser). Don't fail — DMs/author-bound chats still route and native adapters need + # no scope_id — but warn so a post-restart egress decline isn't silent. logger.warning( "Synthetic event source for %s chat=%s (%s) reconstructed " "without scope_id; scoped relay egress may be declined by " @@ -26153,18 +24390,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[bool]: """Inject a watch/completion notification as a synthetic message event. - Routing must come from the queued event itself, not from whatever foreground message - happened to be active when the queue was drained. Returns ``True`` after adapter - acceptance, ``False`` after a retryable adapter failure, ``None`` when the event has no - gateway route. This is not a transactional boundary: a - process crash after adapter acceptance can still cause durable at-least-once replay. + Routing comes from the queued event, never the active foreground message. Returns + ``True`` on adapter acceptance, ``False`` on retryable adapter failure, ``None`` with no + gateway route. Not transactional: a crash after acceptance can replay (at-least-once). """ source = await asyncio.to_thread(self._build_process_event_source, evt) if not source: - # API-server-originated sessions bind a RAW session key (the X-Hermes-Session-Id value — - # see _bind_api_server_session), not a structured ``agent:main:...`` key, so - # _build_process_event_source cannot derive routing metadata from it and returns None - # above. + # API-server sessions bind the RAW X-Hermes-Session-Id key (_bind_api_server_session), not a + # structured ``agent:main:...`` key, so _build_process_event_source returned None above. raw_sid = str(evt.get("origin_session_id") or "").strip() if not raw_sid: _sk = str(evt.get("session_key") or "").strip() @@ -26228,11 +24461,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return None platform_name = source.platform.value if hasattr(source.platform, "value") else str(source.platform) - # Alias-aware resolution (relay-plane): a relay-fronted gateway registers ONE adapter under - # Platform.RELAY fronting N logical platforms, so a literal ``p.value == platform_name`` - # scan misses "slack" and silently drops the completion as "no gateway route". Resolve via - # the shared transport resolver — native adapter wins; relay is eligible only when it - # advertises fronting the logical platform. + # Alias-aware resolution (relay-plane): one adapter under Platform.RELAY fronts N logical + # platforms, so a literal ``p.value == platform_name`` scan misses "slack" and drops the + # completion as "no gateway route". Use the shared transport resolver — native adapter wins; + # relay is eligible only when it advertises fronting the logical platform. adapter = None try: _platform_enum = Platform(platform_name) @@ -26248,9 +24480,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _transport is not None: adapter = _transport.adapter if adapter is None: - # Legacy literal scan — still correct for native adapters, and - # keeps minimal runner stubs (tests) and exotic platform strings - # working when the resolver can't run. + # Legacy literal scan — still correct for native adapters; keeps minimal runner stubs (tests) + # and exotic platform strings working when the resolver can't run. for p, a in self.adapters.items(): if p.value == platform_name: adapter = a @@ -26259,16 +24490,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return None from gateway.wake import adapter_supports_push as _wake_push_ok if not _wake_push_ok(adapter): - # Non-push adapter (api_server) resolved WITH routing metadata: its chat_id is the raw - # session id (see _bind_api_server_session, which binds chat_id = session_id). - # handle_message would run the wake under a build_session_key()-derived key that never - # matches the raw X-Hermes-Session-Id session — self-post instead. + # Non-push adapter (api_server) resolved WITH routing metadata: its chat_id is the raw session + # id (_bind_api_server_session binds chat_id = session_id), so handle_message would run the + # wake under a build_session_key() key that never matches the raw session — self-post. from gateway.wake import deliver_wake, persist_delegation_delivery raw_sid = str(evt.get("origin_session_id") or "").strip() or str(source.chat_id or "") if evt.get("type") == "async_delegation": - # #85957: same client-owns-the-turn rule as the raw-key branch - # above — persist the completion as a delivery row, never - # self-post it as a new role=user prompt. + # Same client-owns-the-turn rule as the raw-key branch above: persist the completion as a + # delivery row, never self-post it as a new role=user prompt. try: logger.info( "Async delegation completion — persisting delivery " @@ -26320,10 +24549,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew source.chat_id, source.thread_id, ) - # Relay-plane egress priming (defect #4, staging 2026-08-09): a synthetic turn injected - # right after a restart reaches a relay adapter whose per-chat routing caches are cold - # (they warm only on inbound), so its replies egress without tenant discriminators and - # the connector's fail-closed guard declines them. + # Relay-plane egress priming: a synthetic turn injected right after a restart reaches a relay + # adapter whose per-chat routing caches are cold (they warm only on inbound), so its replies + # egress without tenant discriminators and the connector's fail-closed guard declines them. _prime = getattr(adapter, "prime_routing_cache", None) if callable(_prime): _prime(synth_event) @@ -26337,10 +24565,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _completion_delivery_identity(evt: dict) -> Optional[tuple[str, str, object]]: """Return a producer-stable identity when one is available. - Delegation UUIDs identify one producer completion. Process session IDs are normally - unique too, but include the persisted spawn epoch so an explicitly reused ID represents a - distinct process incarnation. Legacy process events without ``started_at`` are delivered - without deduplication rather than risking suppression of a real completion. + Delegation UUIDs identify one producer completion. Process session IDs include the + persisted spawn epoch so a reused ID is a distinct incarnation; legacy events without + ``started_at`` are delivered undeduplicated rather than risk suppressing a real completion. """ evt_type = str(evt.get("type") or "") if evt_type == "async_delegation": @@ -26356,14 +24583,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew async def _classify_completion_target(self, parent_session_id: str) -> str: """Classify an async-completion delivery target before adapter acceptance. - Returns one of: - - - ``"deliver"`` — spawning session live (or compression-rotated with a live - continuation); only proves deliverability, the resolver still retargets. - - ``"terminal"`` — parent gone for good (unknown / explicit user boundary like /new); - drop the durable row rather than falsely ack or replay forever. - - ``"retry"`` — transient uncertainty (DB unavailable, rotation mid-flight); release - the claim so a later consumer retries; the attempt cap bounds churn. + - ``"deliver"``: spawning session live (or compression-rotated with a live continuation); + proves deliverability only, the resolver still retargets. + - ``"terminal"``: parent gone for good (unknown / explicit user boundary like /new); drop + the durable row rather than falsely ack or replay forever. + - ``"retry"``: transient uncertainty (DB unavailable, rotation mid-flight); release the + claim for a later consumer; the attempt cap bounds churn. """ session_db = getattr(self, "_session_db", None) if session_db is None: @@ -26382,11 +24607,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return "deliver" end_reason = str(parent.get("end_reason") or "") if end_reason != "compression": - # An ended parent is only unreachable when the USER closed the thread of work (explicit - # boundary: /new -> session_reset / new_session, user_exit, session_switch). Idle/timeout - # ends are normal on scale-to-zero relays — the chat stays routable and the resolver - # retargets, so dropping those loses finished work. Boundary set is shared with the - # resolver (_USER_BOUNDARY_END_REASONS) so the two decisions cannot drift. + # An ended parent is unreachable only when the USER closed the thread of work (/new -> + # session_reset / new_session, user_exit, session_switch). Idle/timeout ends are normal on + # scale-to-zero relays — the chat stays routable and the resolver retargets, so dropping loses + # finished work. Boundary set shared with the resolver (_USER_BOUNDARY_END_REASONS): no drift. if end_reason in _USER_BOUNDARY_END_REASONS: return "terminal" return "deliver" @@ -26412,9 +24636,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[bool]: """Deliver once per live gateway, or return False for a retry. - ``True``: adapter accepted; ``False``: injection failed, claim released for retry; - ``None``: another same-lifecycle caller owns/delivered it, or no gateway route. - No cross-process exactly-once guarantee is claimed. + ``True``: adapter accepted; ``False``: injection failed, claim released for retry; ``None``: + another same-lifecycle caller owns/delivered it, or no route. No cross-process exactly-once. """ identity = self._completion_delivery_identity(evt) durable_claim_id = "" @@ -26438,12 +24661,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return False parent_session_id = str(evt.get("parent_session_id") or "").strip() if parent_session_id: - # Pre-flight (#65838-class): adapter acceptance is NOT proof of - # delivery — the inner #55578 resolver can still fail closed - # inside the message pipeline AFTER the adapter accepted, which - # would falsely acknowledge the durable row as delivered. - # Verify the target here, before acceptance, and give drops an - # honest durable disposition. + # Adapter acceptance is not proof of delivery: the inner resolver can still fail closed + # inside the pipeline after acceptance, falsely acking the durable row as delivered. + # Verify the target before acceptance so drops get an honest durable disposition. verdict = await self._classify_completion_target(parent_session_id) if verdict == "terminal": logger.warning( @@ -26480,11 +24700,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return False elif evt.get("type") == "completion": - # Background-process completions carry only session_key (chat/ thread routing), so after - # /new the notification from the OLD session would land in the chat's NEW session. - # Stamped events get the same pre-flight as async delegations — one policy owner - # (_classify_completion_target), never a forked predicate. - # Legacy/unstamped events keep today's behavior and deliver. + # Background completions carry only session_key, so after /new the OLD session's + # notification would land in the chat's NEW session. Stamped events get the same + # pre-flight as async delegations (_classify_completion_target); unstamped ones deliver. parent_session_id = str(evt.get("parent_session_id") or "").strip() if parent_session_id: verdict = await self._classify_completion_target(parent_session_id) @@ -26498,9 +24716,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return None if verdict == "retry": - # Transient uncertainty (session DB unavailable or a compression rotation mid- - # flight): signal the watcher to re-poll and try again rather than dropping or - # misrouting the result. + # Transient uncertainty (session DB down / compression rotation mid-flight): tell the + # watcher to re-poll and retry rather than drop or misroute the result. return False if identity is not None: with self._completion_delivery_lock: @@ -26528,9 +24745,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): self._completion_deliveries_delivered.popitem(last=False) - # If the durable async-delegation producer branch is present, its - # SQLite row remains the authoritative replay state. Acknowledge it - # after adapter acceptance; this gateway keeps no parallel ledger. + # When the durable async-delegation producer branch is present, its SQLite row is the + # authoritative replay state — ack it after adapter acceptance; no parallel ledger here. if durable_claim_id: try: from tools.async_delegation import complete_completion_delivery @@ -26583,10 +24799,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew session_id = str(evt.get("session_id") or "unknown") exit_code = evt.get("exit_code") reason = str(evt.get("completion_reason") or "exited") - # Completion-event output is normally passed through the terminal redactor at the - # producer seam, but that redactor is deliberately configurable; this is user-facing - # input so keep the unconditional gateway floor. Redact BEFORE slicing: truncating - # first can leave a credential fragment that no longer matches the patterns. + # Completion output normally passes the terminal redactor at the producer seam, but that is + # configurable and this is user-facing, so keep the unconditional gateway floor. Redact + # BEFORE slicing: truncating first can leave a credential fragment the patterns miss. output = _redact_gateway_user_facing_secrets( str(evt.get("output") or "") ).strip() @@ -26642,9 +24857,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew else: synth_text = self._format_coalesced_process_completions(entries) - # A duplicate primary can legitimately return None from the - # lifecycle dedupe seam. Try the next batch identity so a - # fresh sibling is never discarded with that duplicate. + # A duplicate primary can legitimately return None from the lifecycle dedupe seam; try the + # next batch identity so a fresh sibling is never discarded with that duplicate. delivered = None for _text, candidate_evt, _future in entries: delivered = await self._deliver_completion_notification( @@ -26657,9 +24871,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew [evt for _text, evt, _future in entries] ) except asyncio.CancelledError: - # Shutdown may cancel us either during the fan-in window or while adapter delivery is - # blocked. Recover entries that have not yet detached and resolve every waiter as - # retryable before adapters are torn down. + # Shutdown may cancel us mid fan-in or while adapter delivery is blocked: recover entries not + # yet detached and resolve every waiter as retryable before adapters are torn down. delivered = False if not entries: entries = self._completion_notification_batches.pop(key, []) @@ -26668,9 +24881,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.exception("Coalesced process completion delivery failed") delivered = False finally: - # Never strand watcher futures if formatting, delivery, or task - # cancellation interrupts a batch. False follows the existing - # watcher retry path; None remains the ordinary dedupe result. + # Never strand watcher futures when formatting, delivery, or cancellation interrupts a batch: + # False follows the existing watcher retry path; None remains the ordinary dedupe result. for _text, _evt, future in entries: if not future.done(): future.set_result(delivered) @@ -26733,9 +24945,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._flush_process_completion_batch(key) ) self._completion_notification_batch_tasks[key] = task - # Keep the flush alive and include it in the gateway's normal - # lifecycle accounting. Focused tests that construct a runner via - # object.__new__ lazily receive the same ownership set. + # Keep the flush alive under the gateway's normal lifecycle accounting; runners built via + # object.__new__ (focused tests) lazily receive the same ownership set. if not hasattr(self, "_background_tasks"): self._background_tasks = set() self._background_tasks.add(task) @@ -26749,10 +24960,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _enrich_async_delegation_routing(self, evt: dict) -> None: """Fill platform/chat_id/thread_id/chat_type on an async-delegation event. - Async-delegation completion events only carry ``session_key`` (the daemon worker has no - access to the per-message routing metadata the terminal background watcher captures at - spawn time). Best-effort: a CLI-origin event (empty session_key) is left as-is and simply - won't route on the gateway. + Such events only carry ``session_key`` (the daemon worker lacks per-message routing + metadata). Best-effort: a CLI-origin event (empty session_key) is left as-is and won't route. """ if evt.get("platform"): return # already enriched @@ -26767,12 +24976,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @staticmethod def _async_delegation_group_key(evt: dict) -> tuple[str, ...]: - """Return the same-session routing key for async completion coalescing. - - Two events coalesce only when every routing dimension matches — the - originating session key, the parent session the result re-enters, and - the full gateway route. Events for different sessions never coalesce. - """ + """Return the async-completion coalescing key: originating session, parent session, route.""" return tuple(str(evt.get(field) or "") for field in ( "session_key", "parent_session_id", @@ -26800,17 +25004,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Optional[bool]: """Deliver a same-session batch of async completions as ONE turn. - Single-event groups ride the per-event path unchanged. Multi-event groups deliver the - primary via ``_deliver_completion_notification`` (owns claim, lifecycle dedupe, target - preflight) with a consolidated text of every sibling THIS runner claimed up front. - Only after adapter acceptance are the sibling claims acknowledged — the durable ledger - never acks work that was not delivered, and a sibling claimed by another consumer is - excluded from the consolidated text entirely so its content cannot be double-delivered. - - Returns ``True`` after adapter acceptance, ``False`` when the caller - should requeue the group for retry, and ``None`` when nothing in the - group is deliverable by this runner (siblings that still need a retry - are requeued here before returning). + Single-event groups ride the per-event path. Multi-event groups deliver the primary via + ``_deliver_completion_notification`` with consolidated text of every sibling THIS runner + claimed; sibling claims are acked only after adapter acceptance, and siblings claimed by + another consumer are excluded (no double delivery). Returns True after acceptance, False + to requeue the group, None when nothing is deliverable here (retry siblings requeued). """ from tools.process_registry import process_registry as _pr @@ -26878,9 +25076,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew [evt for evt, _claim_id in siblings] ) else: - # Not delivered — release every sibling claim so a retry (or - # another consumer) can claim it, honestly leaving the durable - # rows pending. + # Not delivered — release every sibling claim so a retry or another consumer can claim it, + # honestly leaving the durable rows pending. for evt, claim_id in siblings: try: release_event_delivery(evt, claim_id) @@ -26897,20 +25094,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return delivered async def _async_delegation_watcher(self, interval: float = 2.0) -> None: - """Drain async-delegation completions and inject them as new turns. + """Drain async-delegation completions and inject them as new turns (IDLE case). - Background subagents (``delegate_task(background=true)``) run on the async-delegation - daemon executor — they have no per-process watcher task, so their completion events would - only be seen by the post-turn queue drain. This covers the IDLE case (no turn running). - Mirrors the CLI's idle ``process_loop`` drain; ignores non-async event types (handled by - ``_run_process_watcher`` / the post-turn drain). + Background subagents run on the daemon executor with no per-process watcher, so their + completions would otherwise only be seen by the post-turn drain. Ignores non-async events. """ await asyncio.sleep(3) # let platforms finish connecting from tools.process_registry import process_registry as _pr while self._running: try: - # Peek the queue for async-delegation events. We must NOT - # consume watch/completion events here (other drains own them), + # Peek for async-delegation events only; watch/completion events belong to other drains, # so requeue anything that isn't ours. requeue = [] async_events = [] @@ -26925,11 +25118,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew requeue.append(evt) for evt in requeue: _pr.completion_queue.put(evt) - # A same-tick drain often carries several completions for the SAME originating - # session (a fan-out of background subagents finishing together). Delivering each - # individually floods the session with N synthetic turns — group by full gateway - # route + parent session, one consolidated turn per group. Events for different - # sessions never coalesce. + # A same-tick drain often carries several completions for the SAME session (a fan-out + # finishing together); delivering each individually floods it with N synthetic turns. + # Group by full gateway route + parent session: one consolidated turn per group. groups: dict[tuple[str, ...], list[dict]] = {} group_order: list[tuple[str, ...]] = [] for evt in async_events: @@ -27003,10 +25194,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew last_output_len = current_output_len if session.exited: - # --- Agent-triggered completion: inject synthetic message --- - # Skip if the agent already consumed the result via wait/log. - # poll() is read-only and intentionally does NOT mark consumed - # (#10156) — a status check must not suppress this delivery turn. + # Agent-triggered completion: inject a synthetic message unless the agent already consumed + # the result via wait/log. poll() is read-only and deliberately does NOT mark consumed — + # a status check must not suppress this delivery turn. from tools.process_registry import format_process_notification, process_registry as _pr_check if agent_notify and not _pr_check.is_completion_consumed(session_id): from agent.redact import redact_terminal_output @@ -27015,9 +25205,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _raw = strip_ansi(session.output_buffer) if session.output_buffer else "" _raw = redact_terminal_output(_raw, _command) _command = _redact_gateway_user_facing_secrets(_command) - # Truncate at line boundaries so notifications never start mid-line (fixes - # #23284). Keep the last ~2000 chars but snap to the nearest preceding newline, - # then prepend a truncation marker when output was cut. + # Truncate on line boundaries (never start mid-line): keep the last ~2000 chars + # snapped to the preceding newline, prepending a marker when output was cut. _LIMIT = 2000 if len(_raw) > _LIMIT: _tail = _raw[-_LIMIT:] @@ -27044,10 +25233,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "completion_reason": getattr(session, "completion_reason", "exited"), "termination_source": getattr(session, "termination_source", ""), "output": _out, - # Spawning conversation's session-db id (stamped at - # spawn time in terminal_tool). Lets the delivery - # pre-flight drop this completion when the user closed - # that session (/new) before the process finished. + # Spawning conversation's session-db id (stamped in terminal_tool); lets delivery + # pre-flight drop this completion if the user closed that session (/new) first. "parent_session_id": ( watcher.get("parent_session_id") or getattr(session, "parent_session_id", "") @@ -27066,12 +25253,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew continue break - # --- Normal text-only notification --- - # Skip when the agent already consumed this completion via wait/log: process(wait) - # returned the exit code and output inline, so the raw "[Background process ... - # finished with exit code ...]" message would be a duplicate delivery of the same - # completion. The agent_notify branch's skip FALLS THROUGH to here, hence this - # check. poll() is read-only and intentionally does not mark consumed. + # Normal text-only notification. Skip when the agent already consumed this completion via + # wait/log (output returned inline) — the raw "finished" message would be a duplicate. + # The agent_notify skip FALLS THROUGH here, hence this check. poll() is read-only. if _pr_check.is_completion_consumed(session_id): logger.debug( "Process watcher: completion for %s already consumed " @@ -27166,10 +25350,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816) - # Config keys whose values MUST invalidate the gateway's cached agent when they change: the - # agent bakes them in at construction, so a mid-gateway config edit would otherwise be - # silently ignored until some other eviction. (section, key) tuples from the raw config - # dict; add new baked-at-construction settings here. + # Config keys whose values MUST invalidate the cached agent when they change: the agent bakes + # them in at construction, so a mid-gateway edit would otherwise be silently ignored until some + # other eviction. (section, key) tuples from the raw config dict; add new baked-in settings here. _CACHE_BUSTING_CONFIG_KEYS: tuple = ( ("model", "context_length"), ("model", "max_tokens"), @@ -27247,15 +25430,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @classmethod def _extract_cache_busting_config(cls, user_config: dict | None) -> dict: - """Pull values that must bust the cached agent. + """Pull values that must bust the cached agent, as a flat dict keyed by 'section.key'. - Returns a flat dict keyed by 'section.key'. Missing config keys and - non-dict sections yield None values, which still contribute to the - signature (so 'absent' vs 'present-and-null' differ). - - The live tool registry generation is included too: MCP reloads / dynamic tool-list - changes mutate the registry without touching config.yaml, and cached agents freeze - their tool schemas at construction, so a generation change must rebuild the agent. + Missing keys / non-dict sections yield None, which still enters the signature ('absent' vs + 'present-and-null' differ). Includes the live tool registry generation: MCP reloads mutate + the registry without touching config.yaml, and cached agents freeze their tool schemas. """ out: Dict[str, Any] = {} cfg = user_config if isinstance(user_config, dict) else {} @@ -27299,24 +25478,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Compute a stable string key from agent config values. - Signature change → cached AIAgent discarded and rebuilt; unchanged → reused (frozen - system prompt + tool schemas for prompt cache hits). - - Callers pass the output of ``_extract_cache_busting_config(user_config)`` so edits to - model.context_length / compression.* in config.yaml are picked up on the next gateway - message without a manual restart. - - ``user_id`` / ``user_id_alt`` participate because the Honcho memory provider freezes - them into ``HonchoSessionManager`` at first-message init; in shared-thread session keys - (``thread_sessions_per_user=False``) omitting them would attribute the second user's - messages to the first user's Honcho peer. Per-user rebuilds trade cache warmth for - correct memory attribution. + Signature change → cached AIAgent rebuilt; unchanged → reused (frozen prompt + schemas for + cache hits). Callers pass ``_extract_cache_busting_config(user_config)`` so config.yaml + edits apply on the next message. ``user_id`` / ``user_id_alt`` participate because Honcho + freezes them into ``HonchoSessionManager`` at init; omitting them in shared-thread keys + (``thread_sessions_per_user=False``) would attribute one user's messages to another's peer. """ import hashlib, json as _j - # Fingerprint the FULL credential string instead of using a short prefix. OAuth/JWT-style - # tokens frequently share a common prefix (e.g. "eyJhbGci"), which can cause false cache - # hits across auth switches if only the first few characters are considered. + # Fingerprint the FULL credential, not a short prefix: OAuth/JWT-style tokens often share a + # common prefix (e.g. "eyJhbGci"), so a prefix would give false cache hits across auth switches. _api_key = str(runtime.get("api_key", "") or "") _api_key_fingerprint = hashlib.sha256(_api_key.encode()).hexdigest() if _api_key else "" @@ -27338,9 +25509,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _cache_keys_sorted, str(user_id or ""), str(user_id_alt or ""), - # skip_context_files changes the agent's frozen system prompt - # (context files in vs out) — a toggled config edit must - # rebuild the cached agent, not silently reuse it. + # skip_context_files changes the agent's frozen system prompt (context files in vs out): + # a toggled edit must rebuild the cached agent, not silently reuse it. bool(skip_context_files), ], sort_keys=True, @@ -27351,11 +25521,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _rehydrate_session_model_override(self, session_key: str) -> None: """Lazily restore a persisted /model override after a gateway restart. - ``_session_model_overrides`` is in-memory only, so before persistence a restart silently - reverted every session to the global default model. Non-secret parts (model/provider/ - base_url) are written through on /model (cleared on /new); read back here on first use - and credentials re-resolved via normal runtime resolution — api_key is never persisted. - No-op when an in-memory override exists (live state wins) or nothing is persisted. + ``_session_model_overrides`` is in-memory only. Non-secret parts (model/provider/base_url) + are written through on /model (cleared on /new) and read back here on first use; api_key + is never persisted and is re-resolved. No-op when an in-memory override or nothing exists. """ _rehydrate_state = self._peek_session_state(session_key) if ( @@ -27415,9 +25583,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> tuple: """Apply /model session overrides if present, returning (model, runtime_kwargs). - These must take precedence over config.yaml defaults so the switched model is actually - used for subsequent messages. Fields with ``None`` values are skipped so partial - overrides don't clobber valid config defaults. + Overrides take precedence over config.yaml defaults so the switched model is actually used; + ``None`` fields are skipped so partial overrides don't clobber valid defaults. """ _apply_state = self._peek_session_state(session_key) override = _apply_state.conversation.model_override if _apply_state else None @@ -27491,17 +25658,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew *, run_generation: Optional[int] = None, ) -> bool: - """Pop ALL per-running-agent state entries for ``session_key``. + """Pop ALL per-running-agent state entries for ``session_key``; True when cleared. - Replaces ad-hoc ``del self._running_agents[key]`` calls scattered across the gateway. - Each missed entry was a small, persistent leak — a (str_key → float) tuple per session - per gateway lifetime. - - Call at every site that ends a running turn, whatever the cause. Per-session state that - PERSISTS across turns (model overrides, voice mode, pending approvals, update prompt) is - NOT touched — those have their own lifecycles. With ``run_generation``, only clear if - that generation is still current, so a stale async unwind bumped by /stop or /new cannot - clobber a newer run. Returns True when cleared, False when the ownership guard blocked. + Call at every site that ends a running turn, whatever the cause. State that PERSISTS + across turns (model overrides, voice mode, pending approvals, update prompt) is NOT + touched. With ``run_generation``, only clear if that generation is still current, so a + stale async unwind bumped by /stop or /new cannot clobber a newer run (returns False). """ if not session_key: return False @@ -27519,23 +25681,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug( "Failed to release active session slot", exc_info=True ) - # One structured reset instead of the old drifting pop-list (agent / started_ts / lease - # / busy_ack_ts). Turn-lease tokens are deliberately NOT cleared here — - # _release_turn_lease owns them. + # One structured reset instead of a drifting pop-list. Turn-lease tokens are deliberately NOT + # cleared here — _release_turn_lease owns them. state.turn.clear() - # Turn boundary: a running-agent slot was just released. Persist the new (lower) in-flight - # count so the dashboard readout stays current between lifecycle transitions. Preserves - # gateway_state (see _persist_active_agents). + # Turn boundary: a running-agent slot was just released; persist the new (lower) in-flight count + # so the dashboard readout stays current. Preserves gateway_state (see _persist_active_agents). self._persist_active_agents() return True def _release_turn_lease(self, session_key: str, run_generation: int) -> bool: """Release the turn lease acquired by (``session_key``, ``run_generation``). - Companion to the acquisition in ``_handle_message_with_agent``. Token map is keyed by - (routing key, run generation), so this only frees the lease its own turn acquired — a - stale unwind pops ITS token and the registry's identity check refuses it if a newer turn - holds the lease. Idempotent and safe for bare test runners built via ``object.__new__``. + Token map is keyed by (routing key, run generation), so a stale unwind pops only ITS token + and the registry's identity check refuses it if a newer turn holds the lease. Idempotent. """ if not session_key: return False @@ -27560,11 +25718,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> bool: """Follow a mid-turn session_id rotation with the held turn lease. - Compression can rotate ``session_entry.session_id`` mid-turn. The turn's flush targets - the NEW id, so the serialization boundary must follow it — otherwise an alias routing - key resolving the new id (topic tip-walk onto the fresh child) could start a concurrent - turn the lease never sees. Call at every site that reassigns session_id mid-turn. - Fail-open no-op when no token is held. + Compression can rotate ``session_entry.session_id`` mid-turn; the flush targets the NEW id, + so the serialization boundary must follow or an alias key resolving the new id could start + a concurrent turn the lease never sees. Call at every mid-turn reassignment; no-op if no token. """ if not session_key or not new_session_id: return False @@ -27584,19 +25740,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _clear_conversation_scope(self, session_key: str, *, reason: str) -> None: """Clear ALL conversation-scoped per-session state for ``session_key``. - THE single conversation-boundary funnel. Call this — and nothing else — whenever a - session_key crosses a conversation boundary: /new, /resume, auto-reset - (idle/daily/suspended), expiry finalization, and the compression-exhausted auto-reset. - - Why a funnel: each boundary used to carry a hand-copied pop-list that drifted whenever a - dict was added ("boundary X forgot dict Y" bugs). New conversation-scoped dicts go in - _CONVERSATION_SCOPED_STATE and every boundary picks them up. - - Scope: conversation-scoped (cleared): model/reasoning overrides, one-turn restore - snapshots, pending model notes, last-resolved model, queued follow-ups, boundary security - state. Turn-scoped (NOT cleared): _running_agents/_ts, slot leases, turn-lease tokens — - owned by _release_running_agent_state. Idle agent-cache eviction is NOT a boundary (the - session lives on and a resumed turn rebuilds from these overrides). getattr-guarded. + THE single conversation-boundary funnel — call this and nothing else at /new, /resume, + auto-reset (idle/daily/suspended), expiry finalization and compression-exhausted reset. + New conversation-scoped dicts go in _CONVERSATION_SCOPED_STATE so every boundary picks + them up (hand-copied pop-lists drifted). Turn-scoped state (_running_agents/_ts, slot + leases, turn-lease tokens) is owned by _release_running_agent_state and NOT cleared. Idle + agent-cache eviction is NOT a boundary (a resumed turn rebuilds from these). getattr-guarded. """ if not session_key: return @@ -27605,10 +25754,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew state = self._peek_session_state(session_key) if state is not None: state.conversation.clear() - # Legacy plain-dict stores still registered in _CONVERSATION_SCOPED_STATE (not yet folded - # into SessionState), e.g. _pending_model_notes. SessionState-backed names resolve to - # MutableMapping views (not dict), so the isinstance(dict) guard skips them — already - # handled above. + # Legacy plain-dict stores still in _CONVERSATION_SCOPED_STATE (not yet folded into + # SessionState), e.g. _pending_model_notes. SessionState-backed names resolve to MutableMapping + # views (not dict), so the isinstance(dict) guard skips them — already handled above. for attr in _CONVERSATION_SCOPED_STATE: store = getattr(self, attr, None) if isinstance(store, dict): @@ -27663,11 +25811,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) def _begin_session_run_generation(self, session_key: str) -> int: - """Claim a fresh run generation token for ``session_key``. + """Claim a fresh, monotonically increasing run generation token for ``session_key``. - Every top-level gateway turn gets a monotonically increasing token. If a later command - like /stop or /new invalidates that token while the old worker is still unwinding, the - late result can be recognized and dropped instead of bleeding into the fresh session. + If /stop or /new invalidates the token while the old worker is still unwinding, the late + result is recognized and dropped instead of bleeding into the fresh session. """ if not session_key: return 0 @@ -27736,11 +25883,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _process_baseline = getattr( running_agent, "_gateway_turn_process_baseline", None ) - # Bump the generation *before* scheduling the reap thread and capture the post-bump value: - # task_id is session-scoped (task_id == session_id), so if a replacement turn claims this - # session and spawns its own process before the reap thread actually runs, that claim bumps - # the generation again; the closure then sees a stale generation and skips — the - # replacement's own baseline covers its cleanup, so nothing stays unreaped. + # Bump the generation BEFORE scheduling the reap thread and capture the post-bump value: + # task_id is session-scoped, so a replacement turn spawning before the reap runs bumps it + # again and the closure sees a stale generation and skips — the replacement's own baseline + # covers its cleanup, so nothing stays unreaped. _generation_at_interrupt = self._invalidate_session_run_generation( session_key, reason=invalidation_reason ) @@ -27783,12 +25929,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _iac_state.persistent.pending_command_text = None if release_running_state: self._release_running_agent_state(session_key) - # Evict the cached agent: ``_interrupt_requested`` is only cleared by the turn - # finalizer, so on a hung or still-draining run the flag survives the lock release and - # kills the session's NEXT message at the top of the tool loop (interrupted=True, - # api_calls=0, empty response — silently swallowed, #44212). Evicting mirrors /new and - # /model: the next message rebuilds from history while the old agent keeps its - # interrupt flag so a hung drain still dies when it unblocks. + # Evict the cached agent: ``_interrupt_requested`` is only cleared by the turn finalizer, + # so on a hung/still-draining run the flag survives and silently kills the session's NEXT + # message (interrupted=True, api_calls=0, empty response). Like /new and /model, the next + # message rebuilds from history; the old agent keeps its flag so a hung drain still dies. self._evict_cached_agent(session_key) async def _refresh_agent_cache_message_count( @@ -27796,16 +25940,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> None: """Re-baseline a cached agent's stored message_count after THIS turn. - The cross-process coherence guard compares on-disk ``message_count`` against the count - snapshotted with the cached agent and rebuilds on mismatch. But the snapshot is taken at - agent-BUILD time — before this turn writes its own user + assistant (+ tool) rows — and - the cache entry is never rewritten on a reuse, so without re-baselining our OWN writes - would force a rebuild every turn, destroying the prompt caching the cache protects. - Call this once a turn has completed and the agent has flushed its rows to the SessionDB. - Only the count element is refreshed (``_sig`` untouched), and only if the same agent is - still cached. If the entry records a different ``session_id`` (cache built for another - conversation under the same key) leave it alone — overwriting would corrupt that - conversation's baseline. Fail-safe: DB errors leave the snapshot as-is (one spare rebuild). + The coherence guard compares on-disk ``message_count`` against the BUILD-time snapshot and + rebuilds on mismatch; without re-baselining after our own rows flush, every turn would + rebuild and destroy prompt caching. Only the count is refreshed (``_sig`` untouched), only + if the same agent is still cached, never when the entry records a different ``session_id`` + (another conversation's baseline). DB errors leave the snapshot as-is (one spare rebuild). """ if self._session_db is None or not session_id: return @@ -27822,25 +25961,21 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return with _cache_lock: cached = _cache.get(session_key) - # Only re-baseline a live 3-tuple entry; skip pending sentinels, - # legacy 2-tuples (they intentionally opt out of the guard), and - # the case where the entry was evicted/rebuilt mid-turn. + # Only re-baseline a live 3-tuple entry; skip pending sentinels, legacy 2-tuples (they opt + # out of the guard), and entries evicted/rebuilt mid-turn. if ( isinstance(cached, tuple) and len(cached) > 2 and cached[0] is not _AGENT_PENDING_SENTINEL ): - # If the snapshot was taken for a different session_id (same session_key, different - # conversation), leave the snapshot alone — the current session_id's count belongs - # to a different DB row. + # A snapshot taken for a different session_id (same session_key, different conversation) + # belongs to a different DB row — leave it alone. _snapshot_sid = cached[3] if len(cached) > 3 else None if _snapshot_sid is not None and _snapshot_sid != session_id: return if cached[2] != _live: if _snapshot_sid is None: - # Legacy 3-tuple: preserve the original 3-element - # shape so existing entries stay compatible with - # callers that index ``cached[2]`` directly. + # Legacy 3-tuple: preserve the 3-element shape for callers indexing ``cached[2]``. _cache[session_key] = (cached[0], cached[1], _live) else: _cache[session_key] = ( @@ -27866,8 +26001,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _voice_channel_sidecar_note(self, event, source: SessionSource, session_key: str) -> Optional[str]: """Return a ``[Voice channel now: ...]`` note when VC state changed. - Unchanged state returns ``None`` so the per-turn member/speaking serialization cannot - churn the prompt. + Unchanged state returns ``None`` so per-turn member/speaking churn can't touch the prompt. """ if source.platform != Platform.DISCORD: return None @@ -27896,9 +26030,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> str: """Return the session-context prompt, pinned per session. - Key hit → the pinned bytes are reused VERBATIM (immunizes the composed system prompt - against renderer nondeterminism); key miss → re-render ``build_session_context_prompt`` - and re-pin (a legitimate cache bust: rename, topic edit, /sethome, redact_pii flip, ...). + Key hit → pinned bytes reused VERBATIM (immune to renderer nondeterminism); key miss → + re-render ``build_session_context_prompt`` and re-pin (rename, topic edit, /sethome, ...). """ _eph_key = self._ephemeral_change_key(context, redact_pii) _eph_pin = None @@ -27919,10 +26052,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _ephemeral_change_key(context, redact_pii: bool) -> str: """Hash the exact inputs ``build_session_context_prompt`` renders. - This key decides when the pinned per-session context-prompt bytes are reused verbatim vs - re-rendered. Invariant (guarded by tests/gateway/test_prompt_tail_freeze.py): any input - whose change alters the rendered bytes MUST appear here — omission means a stale pinned - prompt; an extra field only costs a spurious re-render. + Invariant (tests/gateway/test_prompt_tail_freeze.py): any input whose change alters the + rendered bytes MUST appear here — omission means a stale pinned prompt; extras only re-render. """ import hashlib @@ -27940,16 +26071,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew str(src.parent_chat_id or ""), str(src.thread_id or ""), str(src.chat_id or ""), - # Only PRESENCE is rendered (the id itself is delivered - # per-turn in the user message) — keying on the value would - # re-render every message for zero byte change. + # Only PRESENCE is rendered (the id itself arrives per-turn in the user message) — + # keying on the value would re-render every message for zero byte change. "1" if src.message_id else "0", ) - # Slack renders a capability-aware platform note gated on _slack_tools_loaded() — the gate - # state must appear in the key (same parity contract as the Discord gate above) so a config - # / MCP-registration flip re-renders once instead of serving a stale pinned note for the - # rest of the session. + # Slack's capability-aware platform note is gated on _slack_tools_loaded() — the gate state must + # be in the key (same parity contract as the Discord gate above) so a config / MCP-registration + # flip re-renders once instead of serving a stale pinned note for the rest of the session. slack_tools = "" if src.platform == Platform.SLACK: from gateway.session import _slack_tools_loaded @@ -27994,20 +26123,15 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _evict_cached_agent(self, session_key: str) -> None: """Remove a cached agent for a session (called on /new, /model, etc). - Pops the entry AND soft-releases the evicted agent's LLM client pool so httpx sockets and - buffers are freed promptly — AIAgent holds reference cycles that delay refcount - collection, so without a manual release gateway RSS grows across /new, /model, reset. - - The release is soft (``release_clients()``): it frees the client pool and per-turn child - subagents but PRESERVES the session's terminal sandbox, browser daemon, and tracked bg - processes (keyed on task_id), because the session may resume with a freshly-built agent. - True boundaries (/new) call ``_cleanup_agent_resources`` first; ``release_clients`` is - idempotent after that. Cleanup runs on a daemon thread so ``_agent_cache_lock`` is never - held across slow socket teardown. + Also soft-releases the evicted agent's LLM client pool (``release_clients()``): AIAgent + holds reference cycles that delay collection, so without it gateway RSS grows across /new. + Soft = frees clients and per-turn child subagents but PRESERVES the session's terminal + sandbox, browser daemon and bg processes (keyed on task_id) since the session may resume. + True boundaries (/new) call ``_cleanup_agent_resources`` first (release is idempotent). + Cleanup runs on a daemon thread so ``_agent_cache_lock`` never spans slow socket teardown. """ - # Prompt-stability state rides the agent-cache lifecycle: a fresh - # agent must re-render its session-context bytes (the pin) and re-see - # the current voice-channel state once. + # Prompt-stability state rides the agent-cache lifecycle: a fresh agent must re-render its + # session-context bytes (the pin) and re-see the current voice-channel state once. _evict_state = self._peek_session_state(session_key) if _evict_state is not None: _evict_state.conversation.ephemeral_pin = None @@ -28029,11 +26153,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Don't tear down an agent that's actively mid-turn — its client, # sandbox and child subagents are in use by the running request. - running_ids = { - id(a) - for _, a in self._running_agent_items() - if a is not None and a is not _AGENT_PENDING_SENTINEL - } + running_ids = self._running_agent_ids() if id(agent) in running_ids: return @@ -28047,21 +26167,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: # If we can't spawn a thread (interpreter shutdown), release # inline as a best-effort fallback. - try: + with suppress(Exception): self._release_evicted_agent_soft(agent) - except Exception: - pass @staticmethod def _init_cached_agent_for_turn(agent: Any, interrupt_depth: int) -> None: """Reset per-turn state on a cached agent before a new turn starts. - ``_last_activity_ts`` / ``_desc`` / ``_provenance`` are a semantic triple (desc and - provenance describe the activity *at* ts) so they are reset together, and only for fresh - external turns (depth 0). For interrupt-recursive turns all three are preserved so the - inactivity watchdog can accumulate stuck-turn idle time and fire the 30-min timeout. The - depth-0 reset is still needed: a session idle 29 min would otherwise trip the watchdog - before the new turn's first API call. + ``_last_activity_ts`` / ``_desc`` / ``_provenance`` are a semantic triple, reset together + and only for fresh external turns (depth 0) — otherwise a session idle 29 min would trip + the watchdog before the first API call. Interrupt-recursive turns preserve all three so + the inactivity watchdog can accumulate stuck-turn idle time and fire the 30-min timeout. """ if interrupt_depth == 0: from agent.session_activity import ActivityProvenance @@ -28069,9 +26185,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew agent._last_activity_ts = time.time() agent._last_activity_desc = "starting new turn (cached)" agent._last_activity_provenance = ActivityProvenance.UNKNOWN - # Reset the SessionDB flush cursor so the new turn's messages are - # fully persisted - a stale value from the previous turn would - # cause `_flush_messages_to_session_db` to skip new rows (#44327). + # Reset the SessionDB flush cursor so the new turn's messages are fully persisted — a stale + # value from the previous turn makes `_flush_messages_to_session_db` skip new rows. if hasattr(agent, "_last_flushed_db_idx"): agent._last_flushed_db_idx = 0 agent._api_call_count = 0 @@ -28079,15 +26194,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _commit_memory_before_soft_evict(self, agent: Any, key: str) -> None: """Fire on_session_end extraction before soft-evicting a live agent. - Soft eviction (``_release_evicted_agent_soft``) deliberately keeps the session resumable - and does NOT fire ``on_session_end`` — that hook is reserved for the true session - boundary, tear-down done by ``_session_expiry_watcher`` when the session finally expires. - But the watcher tears down whatever agent it finds in ``_agent_cache``; if the LRU cap - soft-evicts first, it finds none and ``on_session_end`` is silently skipped — memory - providers never see the transcript. So commit extraction here with the live agent's own - memory manager via ``commit_memory_session`` (extraction WITHOUT teardown, eviction stays - soft). Only for finalizable sessions (finite reset policy); ``mode == "none"`` never - finalizes so nothing to compensate. Best-effort: failures swallowed, eviction proceeds. + Soft eviction keeps the session resumable and does NOT fire ``on_session_end`` — that is + ``_session_expiry_watcher``'s job at true expiry. But the watcher tears down whatever it + finds in ``_agent_cache``; if the LRU cap soft-evicts first, memory providers never see the + transcript. So commit extraction here via ``commit_memory_session`` (no teardown). Only for + finalizable sessions — ``mode == "none"`` never finalizes. Best-effort: failures swallowed. """ if agent is None or not hasattr(agent, "commit_memory_session"): return @@ -28101,9 +26212,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew entry = _store._entries.get(key) if entry is None: return - # Only compensate when the watcher would otherwise expect to find this agent at expiry - # (finite policy, not yet expired). Expired sessions are torn down by the watcher - # directly; mode="none" sessions are never finalized. + # Compensate only when the watcher would expect this agent at expiry (finite policy, not yet + # expired). Expired sessions are torn down by the watcher; mode="none" is never finalized. if not _store.is_session_finalizable(entry): return if _store._is_session_expired(entry): @@ -28120,9 +26230,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _commit_then_release_soft(self, agent: Any, key: str) -> None: """Commit end-of-session memory (if warranted), then soft-release. - Runs on the daemon eviction thread so the memory-provider call and the client teardown - never block the caller's held cache lock. Order matters: commit uses the live agent's - memory manager before ``release_clients`` drops the message buffer. + Runs on the daemon eviction thread so neither blocks the caller's held cache lock. Order + matters: commit needs the live memory manager before ``release_clients`` drops the buffer. """ self._commit_memory_before_soft_evict(agent, key) self._release_evicted_agent_soft(agent) @@ -28130,10 +26239,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _release_evicted_agent_soft(self, agent: Any) -> None: """Soft cleanup for cache-evicted agents — preserves session tool state. - Called from _enforce_agent_cache_cap and _sweep_idle_cached_agents. Distinct from - _cleanup_agent_resources (full teardown): a cache-evicted session may resume, so its - terminal sandbox, browser daemon and bg processes must outlive the AIAgent instance and - be inherited by the next agent built for the same task_id. + Unlike _cleanup_agent_resources (full teardown), an evicted session may resume, so its + terminal sandbox, browser daemon and bg processes must outlive the AIAgent instance. """ if agent is None: return @@ -28146,16 +26253,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._cleanup_agent_resources(agent) except Exception: pass - # Free conversation history memory — can be tens of MB with tool outputs (file reads, - # terminal output, search results) on heavy 100+-tool-call sessions. release_clients() - # deliberately preserves session tool state for resume, but the message list is rebuilt from + # Free conversation history — tens of MB of tool output on heavy 100+-tool-call sessions. + # release_clients() preserves session tool state for resume, but the message list is rebuilt from # persisted session JSON on the next turn, so dropping it here is safe. if hasattr(agent, "_session_messages"): agent._session_messages = [] - # _db_flush_scan_prefix is a shallow copy of the flushed transcript (run_agent.py, stamped - # on every successful flush) — it shares every message dict, so leaving it pins the multi-MB - # content strings the eviction exists to free. Pressure-evictable agents have flushed by - # definition, so it is always populated on exactly the agents the valve targets. + # _db_flush_scan_prefix (run_agent.py, stamped on every successful flush) is a shallow copy + # sharing every message dict of the flushed transcript, so leaving it pins the multi-MB strings + # this eviction frees. Pressure-evictable agents have flushed by definition, so it's populated. if hasattr(agent, "_db_flush_scan_prefix"): agent._db_flush_scan_prefix = None @@ -28173,10 +26278,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew bounds = resolve_agent_cache_bounds(_load_gateway_config()) except Exception as _e: logger.debug("Agent cache bounds config read failed: %s", _e) - # Resolve from an empty config rather than bare AgentCacheBounds(): the dataclass - # default has memory_high_mb=None (pressure pass OFF), but an *absent* config - # section means "auto" — a transient config read failure must not permanently - # disable the OOM valve this feature exists to provide. + # Resolve from an empty config rather than bare AgentCacheBounds(): the dataclass default + # has memory_high_mb=None (pressure pass OFF) but an *absent* section means "auto" — a + # transient config read failure must not permanently disable the OOM valve. bounds = resolve_agent_cache_bounds({}) self._agent_cache_bounds_cache = bounds return bounds @@ -28192,19 +26296,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return configured if configured else _AGENT_CACHE_IDLE_TTL_SECS def _sweep_agent_cache_under_pressure(self) -> int: - """Shed cached transcripts once the gateway's own heap nears its budget. + """Shed cached transcripts once the gateway heap nears its budget; returns count evicted. - The LRU cap counts entries and the idle sweep counts seconds; neither knows one cached - agent pins a full ``_session_messages`` transcript (tens of MB on 100+ tool calls). - A gateway serving many chats therefore holds every warm transcript indefinitely: agents - that took a turn within the TTL are never idle-swept, and the sweep additionally defers - finalizable sessions until they expire. RSS climbs until the cgroup throttles and SIGTERM - can no longer flush inside the stop timeout. This is the missing valve: above the - anonymous-RSS budget it soft-evicts LRU agents (transcript rebuilt from persisted session - next turn). Never touched: agents mid-turn, the most recently used sessions (prompt cache - worth most), and sessions whose transcript has not finished reaching disk. - - Returns the number of entries evicted (0 when memory is fine). + The LRU cap counts entries and the idle sweep counts seconds; neither knows one cached agent + pins a full ``_session_messages`` transcript (tens of MB). Warm and finalizable agents are + never swept, so RSS climbs until the cgroup throttles. Above the anonymous-RSS budget this + soft-evicts LRU agents (transcript rebuilt from the persisted session next turn). Never + touched: agents mid-turn, the most recently used sessions, and transcripts not yet on disk. """ from gateway.agent_cache_pressure import ( plan_pressure_evictions, @@ -28226,11 +26324,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if rss_mb is None or rss_mb < bounds.memory_high_mb: return 0 - running_ids = { - id(a) - for _, a in self._running_agent_items() - if a is not None and a is not _AGENT_PENDING_SENTINEL - } + running_ids = self._running_agent_ids() def _is_evictable(key: str, agent: Any) -> bool: if agent is None or agent is _AGENT_PENDING_SENTINEL: @@ -28294,20 +26388,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ).start() except Exception: self._release_pressure_batch(plan) - # NOTE: _release_pressure_batch drains `plan` in place (so the trim - # runs with no lingering agent references) — len(plan) is 0 by the - # time the daemon thread finishes, hence the pre-captured count. + # NOTE: _release_pressure_batch drains `plan` in place (so the trim runs with no lingering + # agent refs) — len(plan) is 0 once the daemon thread finishes, hence the pre-captured count. return evicted_count def _release_pressure_batch(self, plan: List[tuple]) -> None: """Release a pressure-evicted batch, then return the heap to the OS. - Sequential on one daemon thread rather than a thread per agent: the batch is already - capped, and the point of the pass is to reclaim memory, not to race N teardowns. The - trailing ``malloc_trim`` is what turns "Python dropped it" into "RSS actually fell" — - glibc otherwise keeps the freed arenas. The plan is drained (``pop`` + ``del``), not - iterated, so no local reference pins evicted agents during ``gc.collect`` + trim; - otherwise the trim frees little and the valve over-evicts warm caches every cycle. + Sequential on one daemon thread (the batch is capped; the goal is reclaiming memory, not + racing teardowns). The trailing ``malloc_trim`` makes RSS actually fall — glibc otherwise + keeps freed arenas. The plan is drained (``pop`` + ``del``), not iterated, so no local + reference pins evicted agents during ``gc.collect`` + trim (else the valve over-evicts). """ while plan: key, agent = plan.pop(0) # FIFO — evict LRU-first order preserved @@ -28324,13 +26415,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew pass def _enforce_agent_cache_cap(self) -> None: - """Evict oldest cached agents when cache exceeds the LRU cap. + """Evict oldest cached agents when cache exceeds the LRU cap. Requires _agent_cache_lock. - Must be called with _agent_cache_lock held. Resource cleanup (memory provider shutdown, - tool resource close) is scheduled on a daemon thread so the caller doesn't block on slow - teardown while holding the cache lock. Agents in _running_agents are SKIPPED — their - clients, sandboxes, bg processes and child subagents are in use; evicting would crash - the turn. If every LRU candidate is active the cache stays over cap until the next insert. + Resource cleanup runs on a daemon thread so the lock is not held over slow teardown. + Agents in _running_agents are SKIPPED (their clients/sandboxes/subagents are in use); if + every LRU candidate is active the cache stays over cap until the next insert. """ _cache = getattr(self, "_agent_cache", None) if _cache is None: @@ -28340,18 +26429,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not hasattr(_cache, "move_to_end"): return - # Snapshot of agent instances that are actively mid-turn. Use id() - # so the lookup is O(1) and doesn't depend on AIAgent.__eq__ (which - # MagicMock overrides in tests). - running_ids = { - id(a) - for _, a in self._running_agent_items() - if a is not None and a is not _AGENT_PENDING_SENTINEL - } + # Snapshot of agent instances mid-turn, keyed by id() so lookup is O(1) and independent of + # AIAgent.__eq__ (which MagicMock overrides in tests). + running_ids = self._running_agent_ids() - # Walk LRU → MRU and evict excess-LRU entries that aren't mid-turn. We only consider entries - # in the first (size - cap) LRU positions as eviction candidates. An active slot is SKIPPED - # without evicting a newer entry instead — that would penalise a fresh session (no cache + # Walk LRU → MRU; only the first (size - cap) LRU positions are candidates. An active slot is + # SKIPPED rather than evicting a newer entry — that would penalise a fresh session (no cache # history) to protect a long-running one. Cache may stay over cap until the next insert. cap = self._agent_cache_cap() excess = max(0, len(_cache) - cap) @@ -28382,9 +26465,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew key, len(_cache), ) if agent is not None: - # Commit end-of-session memory extraction, then soft-release, both on the daemon - # thread so the (possibly network-bound) provider call never blocks the held cache - # lock. + # Commit end-of-session memory, then soft-release, both on the daemon thread so the + # (possibly network-bound) provider call never blocks the held cache lock. threading.Thread( target=self._commit_then_release_soft, args=(agent, key), @@ -28393,13 +26475,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ).start() def _sweep_idle_cached_agents(self) -> int: - """Evict cached agents whose AIAgent has been idle past the idle TTL. + """Evict cached agents idle past the idle TTL; returns the number evicted. - Safe to call from the expiry watcher without the cache lock — acquires it internally; - cleanup is scheduled on daemon threads. - Returns the number of entries evicted. Agents currently in _running_agents are SKIPPED - for the same reason as _enforce_agent_cache_cap: tearing down an active turn's clients - mid-flight would crash the request. + Acquires the cache lock internally (safe from the expiry watcher); cleanup on daemon + threads. Agents in _running_agents are SKIPPED — tearing down an active turn crashes it. """ _cache = getattr(self, "_agent_cache", None) _lock = getattr(self, "_agent_cache_lock", None) @@ -28408,11 +26487,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew now = time.time() idle_ttl = self._agent_cache_idle_ttl() to_evict: List[tuple] = [] - running_ids = { - id(a) - for _, a in self._running_agent_items() - if a is not None and a is not _AGENT_PENDING_SENTINEL - } + running_ids = self._running_agent_ids() with _lock: for key, entry in list(_cache.items()): agent = entry[0] if isinstance(entry, tuple) and entry else None @@ -28424,30 +26499,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if last_activity is None: continue if (now - last_activity) > idle_ttl: - # Check whether the session has actually expired in the - # session store. If it hasn't (e.g. daily-reset mode - # where the reset fires hours after the user's last - # message), keep the agent in cache so the session-store - # expiry watcher can still find it and call - # on_session_end() with the live transcript. Skipping - # eviction here means the agent stays alive until the - # session genuinely expires, at which point the watcher - # (gateway/run.py _session_expiry_watcher) tears it down - # properly. (#11205 follow-up) - # - # BUT only defer when the watcher will EVER finalize this - # session. For a mode == "none" session the watcher never - # fires (is_session_finalizable() is False), so deferring - # would pin the agent in cache for the gateway's entire - # lifetime — the exact leak this idle sweep exists to - # relieve. Those sessions fall through to soft eviction - # WITHOUT on_session_end, and that is correct: a mode=="none" - # session never reaches a session-end boundary, so there is - # no missed on_session_end to compensate for. (The finite - # case — a session evicted under LRU-cap pressure before it - # expires — is instead covered by _commit_memory_before_soft_ - # evict on the cap path, which fires on_session_end via the - # live agent's memory manager before releasing it.) + # If the session hasn't actually expired in the store (e.g. daily-reset fires hours + # after the last message), keep the agent cached so the expiry watcher can still find + # it and call on_session_end() with the live transcript. BUT only defer when the + # watcher will EVER finalize it: for mode == "none" (is_session_finalizable() False) + # deferring pins the agent for the gateway's lifetime — the leak this sweep relieves. + # Those fall through to soft eviction WITHOUT on_session_end, correctly (never a + # session-end boundary). Finite sessions evicted under LRU-cap pressure are covered + # by _commit_memory_before_soft_evict on the cap path. session_entry = None _store = getattr(self, "session_store", None) try: @@ -28479,15 +26538,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ).start() return len(to_evict) - # ------------------------------------------------------------------ - # Proxy mode: forward messages to a remote Hermes API server - # ------------------------------------------------------------------ + # ---- Proxy mode: forward messages to a remote Hermes API server ---- def _get_proxy_url(self) -> Optional[str]: """Return the proxy URL if proxy mode is configured, else None. - Checks GATEWAY_PROXY_URL env var first (convenient for Docker), - then ``gateway.proxy_url`` in config.yaml. + GATEWAY_PROXY_URL env var (Docker-friendly) wins over ``gateway.proxy_url`` in config.yaml. """ url = os.getenv("GATEWAY_PROXY_URL", "").strip() if url: @@ -28507,16 +26563,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew *, on_missing_cursor: str, ) -> "tuple[Any, Optional[Callable[[], None]]]": - """Build the shared ``StreamConsumerConfig`` and the optional Telegram pause-typing closure - used by both agent-run paths. + """Build the shared ``StreamConsumerConfig`` and optional Telegram pause-typing closure. - ``on_missing_cursor`` controls how platforms whose adapter sets - ``SUPPORTS_MESSAGE_EDITING = False`` are handled — both semantics are preserved verbatim - from the pre-refactor call sites: ``"fallback"`` (proxy path) streams with an empty - cursor; ``"raise"`` (in-process path) raises ``RuntimeError`` so the caller's ``except`` - skips streaming entirely. - - Returns ``(consumer_cfg, pause_typing_before_finalize)``. + ``on_missing_cursor`` handles adapters with ``SUPPORTS_MESSAGE_EDITING = False``: + ``"fallback"`` (proxy path) streams with an empty cursor; ``"raise"`` (in-process path) + raises ``RuntimeError`` so the caller's ``except`` skips streaming entirely. Returns + ``(consumer_cfg, pause_typing_before_finalize)``. """ from gateway.stream_consumer import StreamConsumerConfig @@ -28527,13 +26579,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _chat_id=source.chat_id, ) -> None: _adapter.pause_typing_for_chat(_chat_id) - # Platforms that don't support editing sent messages (e.g. QQ, WeChat) should skip streaming - # entirely — without edit support, the consumer sends a partial first message that can never - # be updated, resulting in duplicate messages (partial + final). + # Platforms that can't edit sent messages (e.g. QQ, WeChat) skip streaming entirely: the + # partial first message could never be updated, yielding duplicates (partial + final). _adapter_supports_edit = getattr(adapter, "SUPPORTS_MESSAGE_EDITING", True) - # Adapters that can't edit messages but provide a native-streaming transport (e.g. WeCom's - # msgtype: "stream" via send_stream_frame) get past the gate — the consumer's native branch - # delivers the full turn through that transport. + # Adapters that can't edit but have a native-streaming transport (e.g. WeCom msgtype "stream" + # via send_stream_frame) pass the gate — the consumer's native branch delivers the full turn. _adapter_supports_native_stream = bool(getattr( adapter, "SUPPORTS_NATIVE_STREAMING", False, )) @@ -28544,16 +26594,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): raise RuntimeError("skip streaming for non-editable platform") _effective_cursor = scfg.cursor if _adapter_supports_edit else "" - # Some Matrix clients render the streaming cursor - # as a visible tofu/white-box artifact. Keep + # Some Matrix clients render the streaming cursor as a visible tofu/white-box artifact: keep # streaming text on Matrix, but suppress the cursor. _buffer_only = False if source.platform == Platform.MATRIX: _effective_cursor = "" _buffer_only = True - # Fresh-final applies to Telegram only — other platforms either edit in place cheaply - # (Discord, Slack) or don't have the timestamp-on-edit / edit-timestamp-stays-stale problem. - # (Ported from openclaw/openclaw#72038.) + # Fresh-final applies to Telegram only — other platforms edit in place cheaply (Discord, Slack) + # or lack the edit-timestamp-stays-stale problem. _fresh_final_secs = ( float(getattr(scfg, "fresh_final_after_seconds", 0.0) or 0.0) if source.platform == Platform.TELEGRAM @@ -28605,9 +26653,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "tools": [], } - # Scope-aware read: the proxy key is a per-profile credential; under - # multiplex honor the installed scope's verdict (Slack pattern for - # the unscoped default-profile loop). + # Scope-aware read: the proxy key is a per-profile credential; under multiplex honor the + # installed scope's verdict (Slack pattern for the unscoped default-profile loop). try: from agent.secret_scope import UnscopedSecretError, get_secret @@ -28623,10 +26670,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return True return self._is_session_run_current(session_key, run_generation) - # Build messages in OpenAI chat format -------------------------- - # The remote api_server keeps session continuity via X-Hermes-Session-Id and loads its - # own history, so we only send the current message; if the remote has no history yet, - # include a compact text-only local history for context (remote handles tool replay). + # Build messages in OpenAI chat format. The remote api_server keeps continuity via + # X-Hermes-Session-Id and loads its own history, so send only the current message; if the + # remote has no history yet, include a compact text-only local history (remote replays tools). api_messages: List[Dict[str, str]] = [] if context_prompt: @@ -28705,10 +26751,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Send typing indicator _adapter = self._adapter_for_source(source) if _adapter: - try: + with suppress(Exception): await _adapter.send_typing(source.chat_id, metadata=_thread_metadata) - except Exception: - pass # Make the HTTP request with SSE streaming ----------------------- full_response = "" @@ -28862,11 +26906,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) -> Dict[str, Any]: """Profile-scoping wrapper around the agent run. - When multiplexing is active, resolve the source's profile and run the whole turn inside - ``_profile_runtime_scope`` so config/skills/memory resolve to that profile's home AND - credentials come from its secret scope (never process-global ``os.environ``). - When multiplexing is off this is a transparent pass-through — zero behavior change for - single-profile gateways. + Under multiplexing, run the turn inside ``_profile_runtime_scope`` so config/skills/memory + resolve to the source profile's home AND credentials come from its secret scope (never + process-global ``os.environ``). Transparent pass-through when multiplexing is off. """ if not getattr(getattr(self, "config", None), "multiplex_profiles", False): return await self._run_agent_inner( @@ -28898,17 +26940,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _profile_name_for_source(self, source: SessionSource) -> Optional[str]: """Resolve the profile name for an inbound source via configured routes. - Returns ``None`` when multiplexing is off, no routes are configured, or - no route matches. Callers (``build_source``, - ``_resolve_profile_home_for_source``) treat ``None`` as "use the - default/active profile". When ``gateway.profile_routes`` is configured, - the most specific matching route wins (guild < channel < thread). See - :mod:`gateway.profile_routing` for matching rules. - - Gated on ``gateway.multiplex_profiles``: routing stamps ``source.profile`` (session-key - namespace + batch keys) but the profile-scoped run only activates under multiplexing; - without the gate, keys would be namespaced by profile while the agent still ran in - ``agent:main``. + Returns ``None`` (= use the default/active profile) when multiplexing is off, no routes are + configured, or none match. The most specific matching route wins (guild < channel < + thread); see :mod:`gateway.profile_routing`. Gated on ``gateway.multiplex_profiles``: + routing stamps ``source.profile`` but the scoped run only activates under multiplexing, + else keys would be namespaced by profile while the agent still ran in ``agent:main``. """ config = getattr(self, "config", None) if not getattr(config, "multiplex_profiles", False): @@ -28961,10 +26997,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _resolve_profile_home_for_source(self, source: SessionSource) -> "Path": """Resolve which profile's HERMES_HOME should serve this inbound source. - Resolution order: 1. ``source.profile`` — set by /p// URL prefix, per-credential - adapter ownership, OR profile_routes matching at ``build_source`` time. - 2. ``_profile_name_for_source`` — re-run routing as a defensive fallback for sources - that bypass ``build_source``. 3. The active profile (the multiplexer's own home). + Order: ``source.profile`` (URL prefix, adapter ownership, or profile_routes at + ``build_source``), then ``_profile_name_for_source`` (fallback for sources bypassing + ``build_source``), then the active profile. """ from gateway.profile_routing import ProfileRouteRejected from hermes_cli.profiles import ( @@ -29034,13 +27069,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew persist_user_display_kind: Optional[str] = None, message_type: Optional[str] = None, ) -> Dict[str, Any]: - """Run the agent with the given message and context. + """Run the agent; returns the full run_conversation result dict. - Returns the full result dict from run_conversation, including: - - "final_response": str (the text to send back) - - "messages": list (full conversation including tool calls) - - "api_calls": int - - "completed": bool + Keys: "final_response", "messages", "api_calls", "completed". """ # ---- Proxy mode: delegate to remote API server ---- if self._get_proxy_url(): @@ -29078,8 +27109,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not isinstance(display_config, dict): display_config = {} - # Per-platform display settings — resolve via display_config module - # which checks display.platforms.. first, then + # Per-platform display settings via display_config: display.platforms.., then # display. global, then built-in platform defaults. from gateway.display_config import resolve_display_setting @@ -29170,10 +27200,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew from gateway.config import Platform tool_progress_enabled = progress_mode not in {"off", "log"} and source.platform != Platform.WEBHOOK # Live working-state status for text-rendering typing indicators (Slack's assistant status - # line). Independent of tool_progress — Slack defaults tool_progress off (permanent lines - # spam channels) but the status line is ephemeral, so live status stays useful there. - # Rides the existing _keep_typing refresh (callback only stores a phrase) — zero extra - # platform API calls. + # line). Independent of tool_progress (Slack defaults it off; the status line is ephemeral). + # Rides the existing _keep_typing refresh — the callback only stores a phrase, no extra calls. _live_status_mode = resolve_display_setting( user_config, platform_key, "live_status", "full" ) @@ -29186,9 +27214,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # instead of the chat (#3459 / #3458). Gateway-only by design. log_mode_enabled = progress_mode == "log" and source.platform != Platform.WEBHOOK log_queue: "queue.Queue | None" = queue.Queue() if log_mode_enabled else None - # Natural assistant status messages are intentionally independent from - # tool progress and token streaming. Users can keep tool_progress quiet - # in chat platforms while opting into concise mid-turn updates. + # Natural assistant status messages are independent from tool progress and token streaming: + # tool_progress can stay quiet while users opt into concise mid-turn updates. interim_assistant_messages_mode = _display_surface_mode( "interim_assistant_messages", default=True, @@ -29207,10 +27234,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew require_platform_override_for={Platform.MATTERMOST}, ) _thinking_enabled = _thinking_mode != "off" - # Slack-native task cards: when the Slack adapter's opt-in is set, tool progress renders as - # native plan/task cards via chat.startStream — the progress queue is needed even though - # Slack keeps ordinary text tool_progress off by default (requiring both flags would - # silently leave the native feature inactive). + # Slack-native task cards: with the Slack adapter's opt-in, tool progress renders as native + # plan/task cards via chat.startStream, so the progress queue is needed even though Slack keeps + # text tool_progress off by default (requiring both flags would silently disable the feature). _progress_adapter_for_native = self._adapter_for_source(source) _native_slack_task_cards = False if ( @@ -29233,16 +27259,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew last_tool = [None] # Mutable container for tracking in closure last_progress_msg = [None] # Track last message for dedup repeat_count = [0] # How many times the same message repeated - # True when the previously enqueued progress line was a terminal - # fenced code block — consecutive terminal calls then drop the - # repeated "💻 terminal" header and render back-to-back blocks. + # True when the previous progress line was a terminal fenced code block — consecutive terminal + # calls then drop the repeated "💻 terminal" header and render back-to-back blocks. last_was_terminal_block = [False] - # ── Discord voice "verbal ack before tool calls" ──────────────── - # When the bot is in a voice channel with the continuous mixer installed - # (discord.voice_fx.enabled), speak a short phrase ("let me look into that") over the - # ambient idle bed on the FIRST tool call of the turn. Fires from tool_start_callback - # (independent of the tool-progress text gate), at most once per turn; no-op elsewhere. + # Discord voice "verbal ack before tool calls": with the continuous mixer installed + # (discord.voice_fx.enabled), speak a short phrase over the idle bed on the FIRST tool call of + # the turn (from tool_start_callback, independent of the tool-progress text gate); once per turn. _voice_ack_fired = [False] _voice_ack_guild: List[Optional[int]] = [None] if source.platform == Platform.DISCORD: @@ -29266,9 +27289,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew resolve_display_setting(user_config, platform_key, "cleanup_progress") ) _cleanup_adapter = self._adapter_for_source(source) if _cleanup_progress else None - # getattr, not attribute access — same duck-typed-adapter guard as the - # edit_message check in send_progress_messages below: a fake/minimal - # adapter without delete_message means "can't delete", not a crash. + # getattr, not attribute access — same duck-typed-adapter guard as the edit_message check in + # send_progress_messages: a fake adapter without delete_message means "can't delete", not a crash. _cleanup_delete = getattr(type(_cleanup_adapter), "delete_message", None) if _cleanup_adapter is not None else None if _cleanup_adapter is not None and ( _cleanup_delete is None @@ -29339,30 +27361,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew turn_runner.native_tool_complete_callback ) - # Background task to send progress messages - # Accumulates tool lines into a single message that gets edited. - # - # Threading metadata is platform-specific: - # - Slack DM threading needs event_message_id fallback (reply thread) - # - Telegram forum topics use message_thread_id; Hermes-created private - # DM topic lanes require both thread metadata and a reply anchor - # - Feishu only honors reply_in_thread when sending a reply, so topic - # progress uses the triggering event message as the reply target - # - Other platforms should use explicit source.thread_id only - # - # Slack honours platforms.slack.extra.reply_in_thread=false: if the - # user has opted out of threaded replies, don't synthesise a thread - # for progress messages either — the very first progress message - # would otherwise create a thread that all subsequent replies - # (including the final answer) would inherit (#18859). + # Background task accumulating tool lines into one edited progress message. Threading metadata + # is platform-specific: Slack DM threading needs the event_message_id fallback; Telegram forum + # topics use message_thread_id and Hermes-created private DM topic lanes need thread metadata + # plus a reply anchor; Feishu only honors reply_in_thread on a reply, so topic progress replies + # to the triggering event; others use explicit source.thread_id only. Slack honours + # reply_in_thread=false: don't synthesise a thread for progress, or every later reply inherits it. _progress_reply_in_thread = True if source.platform == Platform.SLACK: _slack_adapter_for_progress = self._adapter_for_source(source) if _slack_adapter_for_progress is not None: try: - # Relay lane: the adapter owns mode resolution (nested - # platforms.relay.extra.slack subset with flat-key - # fallback). Native lane: read the flat extra as before. + # Relay lane: adapter owns mode resolution (nested platforms.relay.extra.slack subset, + # flat-key fallback). Native lane: read the flat extra as before. _mode_fn = getattr( _slack_adapter_for_progress, "_effective_reply_in_thread", @@ -29379,9 +27390,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: _progress_reply_in_thread = True elif str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz": - # Buzz honours the same opt-out (reply_to_mode: off / - # extra.reply_in_thread: false). When the user asked for flat - # channel replies, progress must not synthesise a thread either. + # Buzz honours the same opt-out (reply_to_mode: off / extra.reply_in_thread: false): when the + # user asked for flat channel replies, progress must not synthesise a thread either. _buzz_adapter_for_progress = self._adapter_for_source(source) if _buzz_adapter_for_progress is not None: try: @@ -29396,10 +27406,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew reply_in_thread=_progress_reply_in_thread, ) # Relay Discord auto-thread lane: a channel-initiating message has no thread_id at ingest - # (the thread is born on the connector's FIRST send). The connector stamps - # prospective_thread_id (anchor message id == the thread it will create) and auto-threads - # outbound carrying it as reply_to; without it progress bubbles land flat in the PARENT - # channel. Carry the anchor on the progress send so it routes into the SAME auto-thread. + # (thread is born on the connector's FIRST send). The connector stamps prospective_thread_id + # (anchor id == the thread it will create); carry it as reply_to on the progress send so + # bubbles route into the SAME auto-thread instead of landing flat in the parent channel. _relay_prospective_thread_id = ( str(getattr(source, "prospective_thread_id", None)) if source.platform == Platform.DISCORD @@ -29441,9 +27450,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew and event_message_id ) or ( - # Buzz has no native thread_id; threading is always via reply-to - # the triggering event id (channel clutter otherwise). Skipped - # when the user opted out of threaded replies. + # Buzz has no native thread_id; threading is always via reply-to the triggering event id + # (channel clutter otherwise); skipped when the user opted out of threaded replies. str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz" and event_message_id and _progress_reply_in_thread @@ -29453,11 +27461,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) async def write_tool_log(): - """Drain log_queue and append tool-call lines to tool_calls.log. + """Drain log_queue and append tool-call lines to tool_calls.log (tool_progress=log). - Only active when ``display.tool_progress`` is ``log``. Uses a RotatingFileHandler - (5MB × 3 backups) so the audit log can't grow unbounded, and the shared - RedactingFormatter so secrets never land on disk. + RotatingFileHandler (5MB × 3) bounds the log; RedactingFormatter keeps secrets off disk. """ if log_queue is None: return @@ -29506,9 +27512,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: pass - # Extracted to TurnRunner.send_progress_messages. The threading - # metadata computed above is published onto the shared TurnContext - # exactly where the original closure's captured locals were bound. + # Extracted to TurnRunner.send_progress_messages; the threading metadata above is published + # onto the shared TurnContext where the original closure's captured locals were bound. turn_ctx._progress_metadata = _progress_metadata turn_ctx._progress_reply_to = _progress_reply_to send_progress_messages = turn_runner.send_progress_messages @@ -29519,9 +27524,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew result_holder = [None] # Mutable container for the result tools_holder = [None] # Mutable container for the tool definitions stream_consumer_holder = [None] # Mutable container for stream consumer - # streaming PCM audio consumer. Created on the gateway event-loop thread (NOT inside - # run_sync's executor worker) so the outer finalisation / interrupt paths can reference it - # without a cross-scope NameError. + # streaming PCM audio consumer. Created on the gateway event-loop thread (NOT in run_sync's + # executor worker) so outer finalisation / interrupt paths can reference it without a NameError. streaming_tts_consumer_holder: list = [None] turn_ctx.result_holder = result_holder turn_ctx.tools_holder = tools_holder @@ -29538,9 +27542,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew turn_ctx._hooks_ref = _hooks_ref turn_ctx._step_callback_sync = turn_runner._step_callback_sync - # Bridge sync event_callback → async hooks.emit for lifecycle events - # (e.g. session:compress fires after context compression splits a session) - # Bridge extracted to TurnRunner._event_callback_sync. + # Bridge sync event_callback → async hooks.emit for lifecycle events (e.g. session:compress + # after a compression split); extracted to TurnRunner._event_callback_sync. turn_ctx._event_callback_sync = turn_runner._event_callback_sync # Bridge sync status_callback → async adapter.send for context pressure @@ -29567,25 +27570,22 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) ) if _progress_thread_id else None if _status_thread_metadata is None and _relay_prospective_thread_id: - # Relay Discord auto-thread lane (see _progress_metadata above): - # carry the reply anchor so status/interim bubbles route into - # the same connector-created thread as the final reply. + # Relay Discord auto-thread lane (see _progress_metadata): carry the reply anchor so + # status/interim bubbles route into the same connector-created thread as the final reply. _status_thread_metadata = { "reply_to_message_id": event_message_id } - # Bridge extracted to TurnRunner._status_callback_sync; publish the - # status wiring computed above onto the shared TurnContext at the - # exact original binding site. + # Bridge extracted to TurnRunner._status_callback_sync; publish the status wiring computed + # above onto the shared TurnContext at the exact original binding site. turn_ctx._status_adapter = _status_adapter turn_ctx._status_chat_id = _status_chat_id turn_ctx._status_thread_metadata = _status_thread_metadata turn_ctx._status_callback_sync = turn_runner._status_callback_sync - # ---- Streaming TTS consumer setup (#60671) ---- - # Created on the gateway event-loop thread (here, in _run_agent_inner), NOT inside - # run_sync's executor worker. This avoids a cross-scope NameError: the outer interrupt / - # finalisation paths reference the consumer via ``streaming_tts_consumer_holder[0]``. + # Streaming TTS consumer setup. Created on the gateway event-loop thread (here), NOT inside + # run_sync's executor worker: the outer interrupt / finalisation paths reference the consumer + # via ``streaming_tts_consumer_holder[0]`` and would hit a cross-scope NameError. _stts_adapter = self._adapter_for_source(source) _is_voice_input = ( message_type is not None @@ -29616,15 +27616,13 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as _stts_err: logger.debug("Could not set up streaming TTS consumer: %s", _stts_err) - # run_sync extracted to TurnRunner.run_sync (bound method; the - # executor call below is unchanged). Its closed-over locals travel - # on turn_ctx; `nonlocal message` rebinds became ctx.message writes. + # run_sync extracted to TurnRunner.run_sync (bound method; executor call unchanged). Its + # closed-over locals travel on turn_ctx; `nonlocal message` rebinds became ctx.message writes. run_sync = turn_runner.run_sync - # Start progress message sender if enabled. Gate on needs_progress_queue (tool_progress OR - # thinking_progress), not tool_progress alone: the sender drains BOTH tool-progress lines - # and _thinking scratch bubbles — with a tool_progress-only gate, thinking_progress:true / - # tool_progress:off queued _thinking messages nothing ever drained. + # Start the progress sender if enabled. Gate on needs_progress_queue (tool_progress OR + # thinking_progress), not tool_progress alone: the sender drains BOTH tool-progress lines and + # _thinking scratch bubbles — a tool_progress-only gate left thinking-only queues never drained. progress_task = None if needs_progress_queue: progress_task = asyncio.create_task(send_progress_messages()) @@ -29674,10 +27672,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew tracking_task = asyncio.create_task(track_agent()) - # Monitor for interrupts from the adapter (new messages arriving). This is the PRIMARY - # interrupt path for regular text messages — Level 1 (base.py) catches them before - # _handle_message() is reached, so the Level 2 running_agent.interrupt() path never fires. - # The inactivity poll loop has a BACKUP check in case this task dies silently. + # Monitor adapter interrupts (new messages). PRIMARY interrupt path for regular text: Level 1 + # (base.py) catches them before _handle_message(), so the Level 2 running_agent.interrupt() path + # never fires. The inactivity poll loop has a BACKUP check in case this task dies silently. _interrupt_detected = asyncio.Event() # shared with backup check async def monitor_for_interrupt(): @@ -29692,23 +27689,20 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _adapter = self._adapter_for_source(source) if not _adapter: continue - # Check if adapter has a pending interrupt for this session. Must use - # session_key (build_session_key output) — NOT source.chat_id — because the - # adapter stores interrupt events under the full session key. + # Must use session_key (build_session_key output), NOT source.chat_id: the adapter + # stores interrupt events under the full session key. if hasattr(_adapter, 'has_pending_interrupt') and _adapter.has_pending_interrupt(session_key): agent = agent_holder[0] if agent: - # Peek at the pending message text WITHOUT consuming it: the message must - # stay in _pending_messages for the post-run _dequeue_pending_event() - # (full MessageEvent with media). Popping here races: the agent may finish - # before checking _interrupt_requested and the message is lost to both paths. + # Peek WITHOUT consuming: the message must stay in _pending_messages for the + # post-run _dequeue_pending_event() (full MessageEvent + media). Popping here + # races: the agent may finish before checking _interrupt_requested, losing it. _peek_event = _adapter._pending_messages.get(session_key) pending_text = None if _peek_event is not None: pending_text = _peek_event.text or "" - # Transcribe audio media BEFORE signaling the agent, so voice - # messages interrupt with the real transcript instead of an empty - # string (or file-path placeholder). + # Transcribe audio BEFORE signaling the agent, so voice messages interrupt + # with the real transcript, not an empty string / file-path placeholder. _media_urls = getattr(_peek_event, "media_urls", None) or [] if self._pending_event_audio_paths(_peek_event): pending_text, _ = await self._transcribe_and_echo_pending_voice( @@ -29736,9 +27730,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew interrupt_monitor = asyncio.create_task(monitor_for_interrupt()) - # Periodic "still working" notifications for long-running tasks. Fires every N seconds so - # the user knows the agent hasn't died. Config: agent.gateway_notify_interval in - # config.yaml, or HERMES_AGENT_NOTIFY_INTERVAL env var. Default 180s (3 min). + # Periodic "still working" notifications so the user knows the agent hasn't died. Config: + # agent.gateway_notify_interval or HERMES_AGENT_NOTIFY_INTERVAL env; default 180s. _NOTIFY_INTERVAL_RAW = _float_env("HERMES_AGENT_NOTIFY_INTERVAL", 180) _NOTIFY_INTERVAL = _NOTIFY_INTERVAL_RAW if _NOTIFY_INTERVAL_RAW > 0 else None _long_running_mode = _display_surface_mode( @@ -29756,16 +27749,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _notify_adapter = self._adapter_for_source(source) if not _notify_adapter: return - # Track the heartbeat message id so we can edit-in-place on platforms that support it - # (Telegram, Discord, Slack, etc.) instead of spamming a new "Still working" bubble - # every interval. + # Track the heartbeat message id to edit in place where supported (Telegram, Discord, + # Slack, ...) instead of a new "Still working" bubble every interval. _heartbeat_msg_id: Optional[str] = None while True: await asyncio.sleep(_NOTIFY_INTERVAL) - # Stop heartbeating once this run no longer owns the session slot or the executor - # has finished — otherwise a stale "running: delegate_task" bubble can outlive the - # run that spawned it. _executor_task is a closure var bound just after this task - # is scheduled; tolerate the brief window before then. + # Stop heartbeating once this run no longer owns the session slot or the executor has + # finished, else a stale "running: delegate_task" bubble outlives its run. _executor_task + # is bound just after this task is scheduled; tolerate the brief window before then. try: _exec_ref = _executor_task except NameError: @@ -29775,9 +27766,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ): break _elapsed_mins = int((time.time() - _notify_start) // 60) - # Include agent activity context if available. Default heartbeat is terse: elapsed + - # current tool. Verbose iteration counter is gated on busy_ack_detail so users who - # want it can opt in per platform. + # Default heartbeat is terse (elapsed + current tool); the verbose iteration counter is + # gated on busy_ack_detail so users can opt in per platform. _agent_ref = agent_holder[0] _status_detail = "" _want_iteration_detail = bool( @@ -29847,11 +27837,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if consumer is None: return False if getattr(consumer, "final_response_sent", False): - # A successful finalize call is not proof the *content* was final: the edit may have - # carried only the last preview snapshot while the tail generated between that - # snapshot and stream completion never reached any API call. Reconcile the recorded - # turn-final payload: only a demonstrable mismatch (False, incl. payload-less split - # delivery) overrides the flag; None keeps legacy trust so timeout dedup isn't regressed. + # A successful finalize call is not proof the *content* was final: the edit may carry + # only the last preview snapshot. Reconcile against the recorded turn-final payload: + # only a demonstrable mismatch (False, incl. payload-less split delivery) overrides + # the flag; None keeps legacy trust so timeout dedup isn't regressed. matcher = getattr(consumer, "delivered_final_matches", None) if callable(matcher): try: @@ -29870,10 +27859,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return False try: - # Run in thread pool to not block. *Inactivity*-based timeout, not wall-clock: the agent - # may run for hours while actively calling tools / streaming, but a hung API call or - # stuck tool with no activity is killed. Config agent.gateway_timeout or - # HERMES_AGENT_TIMEOUT (env wins); default 1800s; 0 = unlimited. + # Thread pool so we don't block. *Inactivity* timeout, not wall-clock: the agent may run for + # hours while actively calling tools / streaming, but a hung API call or stuck tool is killed. + # agent.gateway_timeout / HERMES_AGENT_TIMEOUT (env wins); default 1800s; 0 = unlimited. _agent_timeout_raw = _float_env("HERMES_AGENT_TIMEOUT", 1800) _agent_timeout = _agent_timeout_raw if _agent_timeout_raw > 0 else None _agent_warning_raw = _float_env("HERMES_AGENT_TIMEOUT_WARNING", 900) @@ -29893,10 +27881,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _turn_worker_done = threading.Event() _turn_timeout_fired = threading.Event() _turn_cleanup_lock = threading.Lock() - # task_id above is session-scoped, not turn-scoped: gate the eventual reap on this exact - # claim still being current, so a replacement turn that starts on the same session - # before the watchdog fires doesn't get its own fresh process killed by this turn's - # stale baseline. + # task_id is session-scoped, not turn-scoped: gate the eventual reap on this exact claim still + # being current, so a replacement turn on the same session that starts before the watchdog + # fires doesn't get its own fresh process killed by this turn's stale baseline. _turn_run_generation = run_generation _turn_is_current = ( (lambda: self._is_session_run_current(session_key, _turn_run_generation)) @@ -29909,12 +27896,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return run_sync() finally: _turn_worker_done.set() - # `.turn.agent` on the session state is only reset to _AGENT_PENDING_SENTINEL - # when the *next* turn is claimed (see _session_state(...).turn.agent = ... at - # claim time), so a stale reference to this exact agent instance stays reachable - # from _interrupt_and_clear_session() until then. Clearing ownership markers the - # instant our worker finishes closes that window: a /stop landing on the finished - # turn no longer reaps background work it deliberately left running. + # `.turn.agent` is only reset to _AGENT_PENDING_SENTINEL when the *next* turn is + # claimed, so this agent stays reachable from _interrupt_and_clear_session() + # until then. Clearing ownership markers the instant our worker finishes means a + # /stop on the finished turn no longer reaps background work it left running. _finished_agent = agent_holder[0] if agent_holder else None if _finished_agent is not None: _finished_agent._gateway_turn_process_task_id = "" @@ -29992,19 +27977,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _stts.abort("barge-in") else: - # Poll loop: check the agent's built-in activity tracker - # (updated by _touch_activity() on every tool call, API - # call, and stream delta) every few seconds. + # Poll the agent's built-in activity tracker (updated by _touch_activity() on every tool + # call, API call, and stream delta) every few seconds. response = None while True: done, _ = await asyncio.wait( {_executor_task}, timeout=_POLL_INTERVAL ) if done: - # Prefer the real result when the worker finished, even if the watchdog - # fired in the same window: the completed run already persisted its reply to - # session history, so surfacing the "agent inactive" diagnostic here would - # contradict the stored transcript. + # Prefer the real result when the worker finished even if the watchdog fired in + # the same window: the completed run already persisted its reply, so the "agent + # inactive" diagnostic would contradict the stored transcript. response = _executor_task.result() break if _turn_timeout_fired.is_set(): @@ -30095,10 +28078,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _timed_out_agent = agent_holder[0] _activity = {} if _timed_out_agent and hasattr(_timed_out_agent, "get_activity_summary"): - try: + with suppress(Exception): _activity = _timed_out_agent.get_activity_summary() - except Exception: - pass _last_desc = _activity.get("last_activity_desc", "unknown") _secs_ago = _activity.get("seconds_since_activity", 0) @@ -30153,21 +28134,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "failed": True, } - # Track fallback model state: if the agent switched to a fallback model during this run, - # persist it so /model shows the actually-active model instead of the config default. - # Skip eviction when the run failed — evicting a failed agent forces MCP reinit on the - # next message for no benefit (same error recurs): bad model → fallback → evict → - # recreate → reinit → same 400 loop burned CPU for hours. + # Persist fallback-model switches so /model shows the actually-active model. Skip + # eviction when the run failed — evicting forces MCP reinit on the next message for no + # benefit (bad model → fallback → evict → recreate → same 400 loop burning CPU). _agent = agent_holder[0] _result_for_fb = result_holder[0] _run_failed = _result_for_fb.get("failed") if _result_for_fb else False if _agent is not None and hasattr(_agent, 'model') and not _run_failed: _cfg_model = _resolve_gateway_model() - # Normalize _cfg_model the same way AIAgent.__init__ does, so a vendor-prefixed - # config value (e.g. "deepseek/deepseek-v4-pro") matches the agent's stripped model on - # native providers — otherwise _agent.model != _cfg_model is always true and the - # cached agent is evicted every turn, destroying prompt caching. Aggregators - # (openrouter, etc.) keep the vendor/model slug, so they're left untouched. + # Normalize _cfg_model as AIAgent.__init__ does so a vendor-prefixed config value + # matches the agent's stripped model on native providers — otherwise the cached agent + # is evicted every turn, destroying prompt caching. Aggregators keep the vendor slug. try: from hermes_cli.model_normalize import ( _AGGREGATOR_PROVIDERS, @@ -30199,9 +28176,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as _stts_done_err: logger.debug("streaming TTS wait_complete error: %s", _stts_done_err) if not _stts.done: - # Timeout before or after audible audio: abort to free - # the consumer task. Audible streams retain suppression; - # silent streams remain eligible for whole-file fallback. + # Timeout before or after audible audio: abort to free the consumer task. Audible + # streams retain suppression; silent streams stay eligible for whole-file fallback. _stts.abort("streaming TTS finalisation timeout") await _stts.wait_complete(timeout=2.0) if _stts.suppress_whole_file and adapter is not None: @@ -30231,9 +28207,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew else: pending = interrupt_message elif pending_event: - # Transcribe audio media on the dequeued event BEFORE it is handed back as the - # next user turn, so queued/interrupting voice messages drain with the real - # transcript instead of a file-path placeholder. + # Transcribe audio on the dequeued event BEFORE it becomes the next user turn, so + # queued/interrupting voice messages drain with the real transcript, not a file path. _pending_text = pending_event.text or "" _media_urls = getattr(pending_event, "media_urls", None) or [] if self._pending_event_audio_paths(pending_event): @@ -30252,9 +28227,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if pending: logger.debug("Processing queued message after agent completion: '%s...'", pending[:40]) - # Leftover /steer: if a steer arrived after the last tool batch (e.g. during the final - # API call), the agent couldn't inject it and returned it in result["pending_steer"]. - # Deliver it as the next user turn so it isn't silently dropped. + # Leftover /steer: a steer arriving after the last tool batch (e.g. during the final API + # call) comes back in result["pending_steer"]; deliver it as the next user turn, not drop it. if result and not pending and not pending_event: _leftover_steer = result.get("pending_steer") if _leftover_steer: @@ -30292,9 +28266,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if pending_event or pending: logger.debug("Processing pending message: '%s...'", pending[:40]) - # Clear the adapter's interrupt event so the next _run_agent call - # doesn't immediately re-trigger the interrupt before the new agent - # even makes its first API call (this was causing an infinite loop). + # Clear the adapter's interrupt event so the next _run_agent call doesn't re-trigger the + # interrupt before the new agent's first API call (infinite loop otherwise). if adapter and hasattr(adapter, '_active_sessions') and session_key and session_key in adapter._active_sessions: adapter._active_sessions[session_key].clear() @@ -30315,19 +28288,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew was_interrupted = result.get("interrupted") if not was_interrupted: - # Queued message after normal completion — deliver the first - # response before processing the queued follow-up. - # Skip if streaming already delivered it. + # Queued message after normal completion: deliver the first response before the + # queued follow-up, unless streaming already delivered it. _sc = stream_consumer_holder[0] if _sc and stream_task: try: await asyncio.wait_for(stream_task, timeout=5.0) except (asyncio.TimeoutError, asyncio.CancelledError): stream_task.cancel() - try: + with suppress(asyncio.CancelledError): await stream_task - except asyncio.CancelledError: - pass except Exception as e: logger.debug("Stream consumer wait before queued message failed: %s", e) # The queued branch needs raw ``result`` for interruption, history, and @@ -30341,9 +28311,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew first_response, previewed=_previewed, ) - # Apply the same predicate as the normal completed-turn path. - # This direct queued-send branch predates intentional-silence - # filtering, so without this check it leaks the literal marker. + # Same predicate as the normal completed-turn path: this direct queued-send branch + # predates intentional-silence filtering and would leak the literal marker. try: from gateway.response_filters import is_intentional_silence_agent_result _intentional_silence = is_intentional_silence_agent_result( @@ -30380,9 +28349,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception as e: logger.warning("Failed to send first response before queued message: %s", e) - # Release deferred bg-review notifications now that the first response has been - # delivered. Pop from the adapter's callback dict (prevents double-fire in - # base.py's finally block) and call it. + # Release deferred bg-review notifications now that the first response is delivered: + # pop from the adapter's callback dict (no double-fire in base.py's finally) and call. if getattr(type(adapter), "pop_post_delivery_callback", None) is not None: _bg_cb = adapter.pop_post_delivery_callback( session_key, @@ -30404,9 +28372,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew await _bg_result except Exception: pass - # else: interrupted — discard the interrupted response ("Operation - # interrupted." is just noise; the user already knows they sent a - # new message). + # else: interrupted — discard the response ("Operation interrupted." is noise; the user + # knows they sent a new message). updated_history = result.get("messages", history) next_source = source @@ -30414,9 +28381,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew next_message_id = None next_channel_prompt = None next_session_key = session_key - # #60671 — carry the pending event's message_type into the - # recursive call so queued voice turns can stream TTS and - # re-mark the generation for the final delivered turn. + # Carry the pending event's message_type into the recursive call so queued voice turns + # can stream TTS and re-mark the generation for the final delivered turn. next_message_type = None if pending_event is not None: next_source = getattr(pending_event, "source", None) or source @@ -30427,9 +28393,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return result # Resolve the follow-up's session key BEFORE preparing the inbound text: - # _prepare_inbound_message_text buffers native image paths under the key it is - # given, and the recursive _run_agent below consumes them under - # next_session_key. Write and consume keys must match or the images drop. + # _prepare_inbound_message_text buffers native image paths under the key given, and + # the recursive _run_agent consumes them under next_session_key — mismatch drops them. try: next_session_key = self._session_key_for_source(next_source) except Exception: @@ -30450,9 +28415,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew next_channel_prompt = getattr(pending_event, "channel_prompt", None) next_message_type = getattr(pending_event, "message_type", None) - # Clear the completed streaming marker from the prior logical - # turn so the recursive turn's streaming TTS is not suppressed - # by the prior turn's completion (#60671). + # Clear the prior logical turn's completed streaming marker so the recursive turn's + # streaming TTS isn't suppressed by that completion. _clear_adapter = self._adapter_for_source(source) if _clear_adapter is not None and session_key and run_generation is not None: _completed_turns = getattr(_clear_adapter, "_streaming_tts_completed_turns", None) @@ -30463,24 +28427,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if _pk: _completed_turns.discard(_pk) - # Restart typing indicator so the user sees activity while - # the follow-up turn runs. The outer _process_message_background - # typing task is still alive but may be stale. + # Restart the typing indicator for the follow-up turn; the outer + # _process_message_background typing task is alive but may be stale. _followup_adapter = self._adapter_for_source(source) if _followup_adapter: - try: + with suppress(Exception): await _followup_adapter.send_typing( source.chat_id, metadata=_status_thread_metadata, ) - except Exception: - pass - # Re-baseline the cached agent's message_count snapshot before recursing into the - # in-band queued (/queue) follow-up turn: the first turn flushed its own rows, so the - # coherence guard the recursive call re-enters would rebuild on OUR OWN writes and - # destroy the prompt-cache prefix. The re-baseline in _handle_message_with_agent runs - # only after the whole chain unwinds — too late. Fail-safe in helper. + # Re-baseline the cached agent's message_count before recursing into the /queue follow-up: + # the coherence guard would otherwise rebuild on OUR OWN flushed rows and destroy the + # prompt-cache prefix; _handle_message_with_agent re-baselines only after the chain ends. await self._refresh_agent_cache_message_count(session_key, session_id) followup_result = await self._run_agent( @@ -30508,47 +28467,38 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # Wait for stream consumer to finish its final edit if stream_task: - # If the agent never created a stream consumer (e.g. non- streaming code path, or a - # test stub returning synchronously) there is nothing to flush — cancel immediately - # instead of waiting out the 5s timeout on a task that's just polling for a consumer - # that will never arrive. + # If the agent never created a stream consumer (non-streaming path, or a test stub + # returning synchronously) there is nothing to flush — cancel now instead of waiting + # out the 5s timeout polling for a consumer that will never arrive. _has_stream_consumer = ( stream_consumer_holder and stream_consumer_holder[0] is not None ) if not _has_stream_consumer: stream_task.cancel() - try: + with suppress(asyncio.CancelledError): await stream_task - except asyncio.CancelledError: - pass else: try: await asyncio.wait_for(stream_task, timeout=5.0) except (asyncio.TimeoutError, asyncio.CancelledError): stream_task.cancel() - try: + with suppress(asyncio.CancelledError): await stream_task - except asyncio.CancelledError: - pass - # Unconditional abort + bounded wait for the streaming-TTS - # consumer (#60671 hardening). Covers cancellation / exception - # paths where the normal finalisation block was skipped. + # Unconditional abort + bounded wait for the streaming-TTS consumer: covers cancellation / + # exception paths where the normal finalisation block was skipped. _stts_finally = streaming_tts_consumer_holder[0] if _stts_finally is not None and not _stts_finally.done: _stts_finally.abort("cleanup") - try: + with suppress(Exception): await _stts_finally.wait_complete(timeout=2.0) - except Exception: - pass # Clean up tracking tracking_task.cancel() if session_key: - # Only release the slot if this run's generation still owns it. A /stop or /new that - # bumped the generation while we were unwinding has already installed its own state; - # this guard prevents an old run from clobbering it on the way out. + # Release the slot only if this run's generation still owns it: a /stop or /new that + # bumped the generation while we unwound already installed its own state; keep it. self._release_running_agent_state( session_key, run_generation=run_generation ) @@ -30571,27 +28521,23 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew exc_info=True, ) - # If streaming already delivered the response, mark it so the caller's send() is skipped - # (avoiding duplicate messages). BUT never suppress when the agent failed — the error is - # new content the user hasn't seen. Also never suppress on "(empty)": the model produced - # nothing after tool calls, and interim text ("Let me search…") set already_sent but is - # NOT the final answer; suppressing would leave the user staring at silence. + # If streaming already delivered the response, skip the caller's send() — but never when the + # agent failed (the error is unseen content) or on "(empty)": interim text ("Let me search…") + # set already_sent but is NOT the final answer; suppressing would leave the user with silence. _sc = stream_consumer_holder[0] if isinstance(response, dict) and not response.get("failed"): _final = response.get("final_response") or "" _is_empty_sentinel = not _final or _final == "(empty)" - # response_previewed means the interim_assistant_callback already saw the final text, - # but only suppress the normal send if that exact final text was delivered. Unrelated - # commentary/progress must not be mistaken for the final response. + # response_previewed means interim_assistant_callback already saw the final text, but only + # suppress the send if that exact text was delivered — unrelated commentary/progress isn't it. _previewed = bool(response.get("response_previewed")) _content_delivered = bool( _sc and getattr(_sc, "final_content_delivered", False) ) - # A *successful* finalize edit can still carry only the last preview snapshot — deltas - # generated between that edit and stream completion never reach any API call, and both - # suppression flags are set from the call's success rather than its content. Reconcile - # the recorded turn-final payload: on mismatch (False, incl. payload-less split delivery) - # neither flag may suppress the final send; None (no record) keeps legacy trust. + # A *successful* finalize edit can still carry only the last preview snapshot, and both + # suppression flags reflect call success, not content. Reconcile against the recorded + # turn-final payload: on mismatch (False, incl. payload-less split delivery) neither flag + # may suppress the final send; None (no record) keeps legacy trust. _stale_finalized = False if _content_delivered and not _is_empty_sentinel: _matcher = getattr(_sc, "delivered_final_matches", None) @@ -30602,16 +28548,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew _stale_finalized = False if _stale_finalized: _content_delivered = False - # Plugin hooks (e.g. transform_llm_output) may have appended content - # after streaming finished — when the response was transformed, always - # send the final version so the appended content reaches the client. + # Plugin hooks (e.g. transform_llm_output) may append content after streaming finished — when + # transformed, always send the final version so the appended content reaches the client. _transformed = bool(response.get("response_transformed")) - # Only suppress the normal send when the actual final reply reached - # the user: the stream consumer streamed it (final_response_sent / - # final_content_delivered), or the interim preview delivered that - # *exact* final text. Unrelated commentary/progress shown during a - # compression/session split must not be mistaken for the final - # response (#14238). + # Suppress the normal send only when the actual final reply reached the user (streamed, or + # interim preview of that *exact* text); commentary shown during a compression/split isn't it. _streamed = _stream_confirmed_final_delivery( _sc, _final, @@ -30627,11 +28568,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) response["already_sent"] = True elif not _is_empty_sentinel and not _transformed and _stale_finalized and _sc is not None: - # Stale finalize: the streamed message holds only the last preview snapshot. Prefer - # editing it up to the complete response (one corrected message); on edit failure - # fall through with already_sent unset so the normal send delivers. Not valid for - # multi-message split delivery: message_id is only the LAST chunk, so editing it - # would repeat every sealed head chunk — fall through to the normal send instead. + # Stale finalize: the streamed message holds only the last preview snapshot. Edit it + # up to the complete response; on edit failure leave already_sent unset so the normal + # send delivers. Not for split delivery: message_id is only the LAST chunk, so editing + # it would repeat every sealed head chunk — fall through to the normal send. _sc_msg_id = _sc.message_id _sc_adapter = getattr(_sc, "adapter", None) if getattr(_sc, "_turn_split_delivery", False): @@ -30708,9 +28648,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew len(_final), ) - # Schedule deletion of tracked temporary progress bubbles after the final response lands. - # Failed runs skip this so bubbles remain as breadcrumbs for the user to see what work - # happened. Only on adapters with ``delete_message``; failures swallowed (best-effort). + # Schedule deletion of tracked temporary progress bubbles after the final response lands; failed + # runs keep them as breadcrumbs. Only on adapters with ``delete_message``; failures swallowed. if ( _cleanup_progress and _cleanup_adapter is not None @@ -30728,20 +28667,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _cleanup_temp_bubbles() -> None: async def _delete_all() -> None: for _mid in _ids_snapshot: - try: + with suppress(Exception): await _adapter_snapshot.delete_message( _chat_id_snapshot, _mid ) - except Exception: - pass - try: + with suppress(Exception): safe_schedule_threadsafe( _delete_all(), _loop_snapshot, logger=logger, log_message="Temp bubble cleanup scheduling error", ) - except Exception: - pass try: _cleanup_adapter.register_post_delivery_callback( @@ -30765,17 +28700,12 @@ def _run_planned_stop_watcher( ) -> None: """Poll for the planned-stop marker and trigger graceful shutdown. - On Windows, ``asyncio.add_signal_handler`` raises NotImplementedError for SIGTERM/SIGINT, so - the standard signal-driven shutdown path never runs when ``hermes gateway stop`` signals the - gateway — the drain loop is skipped, sessions die mid-turn and ``resume_pending`` is never - set, so the next boot cannot auto-resume them. - - This watcher runs on every platform (cheap, defensive) and translates the filesystem marker - written by ``hermes_cli.gateway_windows.stop()`` (``write_planned_stop_marker(pid)``) into - the same shutdown-handler invocation a real SIGTERM would produce. On POSIX it is a no-op - safety net: the synchronous signal handler always consumes the marker first. - ``runner._running`` / ``_draining`` are checked to avoid re-triggering shutdown; the handler - tolerates ``signal=None`` and consumes the marker via ``consume_planned_stop_marker_for_self()``. + On Windows ``asyncio.add_signal_handler`` raises NotImplementedError, so ``hermes gateway + stop`` never reaches the signal-driven drain: sessions die mid-turn and ``resume_pending`` + is never set. This watcher (cheap, runs on every platform) translates the marker written by + ``hermes_cli.gateway_windows.stop()`` into the same shutdown-handler call a SIGTERM would; on + POSIX the synchronous signal handler consumes the marker first. ``_running`` / ``_draining`` + guard against re-triggering; the handler tolerates ``signal=None``. """ from gateway.status import ( _get_planned_stop_marker_path, @@ -30789,19 +28719,15 @@ def _run_planned_stop_watcher( and not getattr(runner, "_draining", False) and getattr(runner, "_running", False) ): - # A marker existing is NOT sufficient — it may have been written for a PREVIOUS - # gateway instance (different PID) and left behind because that process exited - # before the CLI's stop() could clean it up. Firing on a foreign marker drives us - # into shutdown, then the PID-mismatch is logged as an "UNKNOWN" exit and the - # watchdog crash-loops. Only fire when the marker targets us; the probe is - # non-destructive on match and unlinks stale/malformed markers so they can't wedge - # a fresh gateway. + # A marker existing is NOT sufficient — it may target a PREVIOUS gateway instance + # (different PID) left behind when that process exited before stop() cleaned up. + # Firing on it drives us into shutdown, an "UNKNOWN" exit, and a watchdog crash-loop. + # Only fire when the marker targets us; the probe unlinks stale/malformed markers. if not planned_stop_marker_targets_self(): stop_event.wait(poll_interval) continue - # Drive the same path as a real signal handler. Pass signal=None — the handler - # tolerates that and consumes the marker via consume_planned_stop_marker_for_self, - # which also validates target_pid + start_time match us. + # Same path as a real signal handler. signal=None is tolerated; the handler consumes the + # marker via consume_planned_stop_marker_for_self (validates target_pid + start_time). loop.call_soon_threadsafe(shutdown_handler, None) # Done — the handler will set _draining; we exit on next tick. break @@ -30813,12 +28739,10 @@ def _run_planned_stop_watcher( def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop=None, interval: int = 60, cron_provider=None): """Background thread for gateway-only periodic chores (NOT cron). - Split out of the historical ``_start_cron_ticker`` so the cron *trigger* can live behind the - ``CronScheduler`` provider (built-in or external) while these gateway-specific chores keep - running independently of which provider fires cron. An external scale-to-zero provider has - no 60s loop at all, so housekeeping owns its own loop. Refreshes the channel directory every - 5 min; prunes media caches + expired share pastes hourly; polls the curator hourly (its - inner gate enforces the weekly cadence). + Separate from the cron trigger so chores run regardless of which ``CronScheduler`` provider + fires cron (an external scale-to-zero provider has no 60s loop). Refreshes the channel + directory every 5 min; prunes media caches + expired share pastes hourly; polls the curator + hourly (its inner gate enforces the weekly cadence). """ from gateway.platforms.base import ( cleanup_audio_cache, @@ -30865,9 +28789,8 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop try: from gateway.channel_directory import build_channel_directory if loop is not None: - # build_channel_directory is async (Slack web calls), and this runs in a - # background thread. Schedule onto the gateway event loop and wait briefly for - # completion so refresh failures are still logged via the except. + # build_channel_directory is async (Slack web calls) and this is a background thread: + # schedule onto the gateway loop and wait briefly so refresh failures still log. fut = safe_schedule_threadsafe( build_channel_directory(adapters), loop, logger=logger, @@ -30899,10 +28822,9 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop logger.debug("Paste sweep error: %s", e) # Misfire catch-up (external cron providers only): fire jobs whose scheduled time passed - # with no external fire delivered — the backstop for a dead loopback fire hop (gateway - # restart window, api_server not bound, scheduler retries exhausted). The helper no-ops for - # the built-in ticker, enforces cron.misfire_grace_minutes, and the store CAS claim de-dupes - # against a late external retry arriving concurrently. + # with no external fire delivered (dead loopback hop: restart window, api_server not bound, + # retries exhausted). No-op for the built-in ticker; enforces cron.misfire_grace_minutes; + # the store CAS claim de-dupes against a late external retry. if cron_provider is not None and tick_count % MISFIRE_SWEEP_EVERY == 0: try: from cron.scheduler_provider import fire_overdue_jobs @@ -30917,10 +28839,9 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop except Exception as e: logger.debug("Misfire catch-up sweep error: %s", e) - # Curator — piggy-back on the housekeeping loop so long-running gateways get weekly skill - # maintenance without needing restarts. maybe_run_curator() is internally gated by - # config.interval_hours (7 days by default), so CURATOR_EVERY is just the poll rate — the - # real work only fires once per config interval. + # Curator — piggy-back on housekeeping so long-running gateways get weekly skill maintenance + # without restarts. maybe_run_curator() is gated by config.interval_hours (7 days default), so + # CURATOR_EVERY is just the poll rate. if tick_count % CURATOR_EVERY == 0: try: from agent.curator import maybe_run_curator @@ -30931,9 +28852,8 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop except Exception as e: logger.debug("Curator tick error: %s", e) - # Skill Sync — best-effort periodic pull on the same cadence. - # Inert unless the access gate is open and a sync base URL is - # configured; never raises. + # Skill Sync — best-effort periodic pull on the same cadence; inert unless the access gate is + # open and a sync base URL is configured; never raises. try: from tools.skills_sync_client import maybe_pull_skills maybe_pull_skills() @@ -30948,15 +28868,13 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop except Exception as e: logger.debug("Org sync pull tick error: %s", e) - # Stale-session auto-archive — a live timer, so gateways that stay up - # for weeks keep sweeping on schedule (the startup hook fires once). - # maybe_auto_archive() is gated by sessions.min_interval_hours in - # state_meta; this is just the poll rate. Opens its own SessionDB — - # SQLite connections are thread-bound and this runs off-loop. + # Stale-session auto-archive on a live timer so long-running gateways keep sweeping (the + # startup hook fires once). maybe_auto_archive() is gated by sessions.min_interval_hours; + # this is just the poll rate. Opens its own SessionDB — SQLite connections are thread-bound. if tick_count % AUTO_ARCHIVE_EVERY == 0: try: from hermes_cli.config import load_config as _load_full_config - from hermes_state import get_shared_session_db, release_shared_session_db + from hermes_state import get_shared_session_db _sess_cfg = (_load_full_config().get("sessions") or {}) if _sess_cfg.get("auto_archive", False): _adb = get_shared_session_db() @@ -30971,11 +28889,9 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop except Exception as e: logger.debug("Auto-archive tick error: %s", e) - # Deferred stale-FTS rebuild retry. A SessionDB that opened while another process held - # state.db / the rebuild lock fails closed and leaves search on the LIKE fallback; a short- - # lived CLI clears that on its next open, but the gateway opens once and stays up for days. - # Retry against the shared instances this process holds: non-blocking admission, no new - # thread, rate-limited inside SessionDB; no-op when nothing is stale. + # Deferred stale-FTS rebuild retry: a SessionDB opened while another process held state.db + # or the rebuild lock fails closed onto the LIKE fallback, and the gateway stays up for days. + # Non-blocking admission, no new thread, rate-limited inside SessionDB; no-op when not stale. if tick_count % FTS_STALE_RETRY_EVERY == 0: try: from hermes_state_registry import live_shared_session_dbs @@ -30991,18 +28907,16 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop except Exception as exc: logger.debug("Deferred FTS retry tick error: %s", exc) - # This is the long-lived messaging-gateway counterpart to the TUI idle - # reaper. The helper is config-gated and rate-limited, so calling it on - # the 60s housekeeping cadence does not create a trim storm. + # Long-lived messaging-gateway counterpart to the TUI idle reaper; the helper is config-gated + # and rate-limited, so the 60s housekeeping cadence creates no trim storm. if tick_count % MEMORY_TRIM_EVERY == 0: try: from hermes_cli.mem_trim import trim_memory trim_memory(reason="messaging gateway housekeeping") except Exception as exc: - # debug, not warning: sibling housekeeping branches all log failures at debug, and a - # persistent failure (e.g. broken import after a partial update) would otherwise - # warn every 60s forever. + # debug, not warning: sibling branches log failures at debug, and a persistent failure + # (e.g. broken import after a partial update) would otherwise warn every 60s forever. logger.debug( "gateway housekeeping memory trim failed: %s: %s", type(exc).__name__, @@ -31014,13 +28928,10 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop def _start_cron_ticker(stop_event: threading.Event, adapters=None, loop=None, interval: int = 60): - """DEPRECATED shim — preserved for backward compatibility. + """DEPRECATED shim — runs ONLY the built-in in-process cron tick loop. - The cron trigger now lives behind the ``CronScheduler`` provider - (``cron.scheduler_provider``); the gateway resolves a provider and runs its ``start()`` - directly (see ``start_gateway``). This shim runs ONLY the built-in in-process tick loop for - external callers/tests still referencing this symbol; housekeeping moved to - ``_start_gateway_housekeeping``. + The trigger now lives behind the ``CronScheduler`` provider (``cron.scheduler_provider``, + started in ``start_gateway``); housekeeping moved to ``_start_gateway_housekeeping``. """ from cron.scheduler_provider import InProcessCronScheduler InProcessCronScheduler().start(stop_event, adapters=adapters, loop=loop, interval=interval) @@ -31039,30 +28950,24 @@ def _stop_cron_provider(provider) -> None: logger.debug("Cron provider stop() error: %s", exc) -# Upper bound for cooperatively draining the cron ticker on shutdown. The cron thread delivers via -# ``safe_schedule_threadsafe`` and blocks on ``future.result(timeout=60)`` (see -# cron/scheduler.py::_deliver_result), so a single in-flight delivery unblocks within ~60s. +# Upper bound for cooperatively draining the cron ticker on shutdown: the cron thread blocks on +# ``future.result(timeout=60)`` (cron/scheduler.py::_deliver_result), so a delivery unblocks in ~60s. _CRON_SHUTDOWN_DRAIN_TIMEOUT = 65.0 -# Upper bound for cooperatively draining the housekeeping ticker on shutdown. Housekeeping refreshes -# the channel directory via ``safe_schedule_threadsafe(...)`` and blocks on ``fut.result(timeout=30)`` -# (same loop-scheduled-future pattern as cron). So the cooperative bound must cover that 30s future -# (plus margin) rather than the old 5s join, otherwise a channel-directory refresh in flight at -# shutdown gets abandoned mid-resolve (not user-facing — self-heals next tick — but keeps the drain honest). +# Upper bound for draining the housekeeping ticker on shutdown: the channel-directory refresh blocks +# on ``fut.result(timeout=30)``, so cover that 30s plus margin or an in-flight refresh is abandoned. _HOUSEKEEPING_SHUTDOWN_DRAIN_TIMEOUT = 35.0 async def _await_thread_exit( thread: Optional[threading.Thread], timeout: float, poll: float = 0.1 ) -> bool: - """Wait for a daemon thread to exit WITHOUT blocking the event loop. + """Wait for a daemon thread to exit WITHOUT blocking the event loop; True if it exited in time. - A synchronous ``thread.join()`` here would freeze the event loop — fatal for the cron ticker, - whose in-flight delivery is a coroutine scheduled onto *this* loop via - ``safe_schedule_threadsafe``: the loop could never run it, so ``join(timeout=5)`` always timed - out and the message was silently dropped on restart. Polling ``is_alive()`` with - ``await asyncio.sleep`` keeps the loop running so the delivery completes and the ticker sees - ``stop_event``. Returns True if the thread exited within ``timeout``. + A synchronous ``join()`` freezes the loop — fatal for the cron ticker, whose in-flight delivery + is a coroutine scheduled onto *this* loop via ``safe_schedule_threadsafe``: it could never run, + so the join always timed out and the message was dropped. Polling ``is_alive()`` with + ``await asyncio.sleep`` lets the delivery complete and the ticker see ``stop_event``. """ if thread is None: return True @@ -31073,18 +28978,13 @@ async def _await_thread_exit( async def _shutdown_mcp_servers_nonblocking(timeout: float = 5.0) -> bool: - """Close MCP servers off-loop with a bounded wait. + """Close MCP servers off-loop with a bounded wait; True when done within ``timeout``. - ``shutdown_mcp_servers()`` is synchronous and can block for its full 15s internal wait when - the MCP loop and stdio children are torn down concurrently (supervisor SIGTERMs the whole - tree). Calling it on the loop thread freezes the loop, so supervisors with a short kill grace - (s6 default 3s) SIGKILL us before ``lifecycle_ledger.mark_exited()`` runs and every later - boot reports a phantom unclean death. - - Run it on a daemon thread instead and poll with ``_await_thread_exit`` so the loop keeps - servicing teardown. If it does not finish within ``timeout`` we proceed with shutdown; the - daemon thread is left to finish (or die with the process) in the background. Returns True - when it completed within budget. + ``shutdown_mcp_servers()`` is synchronous and can block ~15s when the MCP loop and stdio + children are torn down concurrently. On the loop thread that freezes the loop, so supervisors + with a short kill grace (s6 default 3s) SIGKILL us before ``lifecycle_ledger.mark_exited()`` + runs and every later boot reports a phantom unclean death. Runs on a daemon thread, polled via + ``_await_thread_exit``; on timeout shutdown proceeds and the thread is left to finish or die. """ def _do() -> None: @@ -31129,19 +29029,12 @@ def _gateway_stderr_formatter() -> logging.Formatter: def _replace_target_belongs_to_other_profile(existing_pid: int) -> bool: """Return True when ``--replace`` must refuse to signal ``existing_pid``. - The PID file is HERMES_HOME-scoped, but a poisoned/stale record can point at another - profile's LIVE gateway; signaling it starts a cross-profile SIGTERM restart loop. - This is a destructive-action authority check, so ownership is decided by the persisted - identity record ALONE — exact ``_same_hermes_home`` equality — and only while that record - stays bound to the live target by exact PID + start-time identity: - - * The authorizing record is whichever source produced the PID (PID file, lock record, or - runtime-status fallback). Live argv carries no HERMES_HOME so it can never PROVE ownership; - it is only a consistency check — profile flags that clearly contradict our home refuse. - * Missing, legacy, conflicting, stale-bound, or unprovable identity → refuse (fail closed). - - Same-home targets replace normally; every refusal here only narrows what the legacy - start_time check alone used to allow. + A poisoned/stale PID record can point at another profile's LIVE gateway; signaling it starts + a cross-profile SIGTERM restart loop. Ownership is decided by the persisted identity record + ALONE (exact ``_same_hermes_home``), and only while it stays bound to the live target by exact + PID + start-time identity. Live argv carries no HERMES_HOME so it can never PROVE ownership; + it is only a consistency check (contradicting profile flags refuse). Missing, legacy, + conflicting, stale-bound or unprovable identity → refuse (fail closed). """ try: from gateway.status import ( @@ -31157,9 +29050,8 @@ def _replace_target_belongs_to_other_profile(existing_pid: int) -> bool: our_home = _get_process_hermes_home() - # ── Authorize from the persisted identity record ────────────── - # Bound claim: the record must describe THIS pid with THIS live - # start time, otherwise it is stale/poisoned and proves nothing. + # Authorize from the persisted identity record — bound claim: the record must describe THIS pid + # with THIS live start time, otherwise it is stale/poisoned and proves nothing. record = _read_pid_record(_get_pid_path()) if not isinstance(record, dict) or not _record_looks_like_gateway(record): logger.warning( @@ -31208,10 +29100,8 @@ def _replace_target_belongs_to_other_profile(existing_pid: int) -> bool: ) return True - # ── Readable-argv consistency check (never authority) ───────── - # An explicit profile flag / HERMES_HOME= on the argv that clearly - # contradicts our home refuses even though the record agreed; a bare - # or matching argv adds nothing either way. + # Readable-argv consistency check (never authority): an explicit profile flag / HERMES_HOME= that + # clearly contradicts our home refuses even if the record agreed; bare/matching argv adds nothing. try: live_cmdline = _read_process_cmdline(existing_pid) except Exception: @@ -31278,44 +29168,32 @@ def _looks_like_profile_conflict_from_cmdline(command: str, our_home) -> bool: return None if profile_name is not None and profile_name != "default": - # Our home is a named profile: any explicit DIFFERENT named profile - # on the argv contradicts it. Bare argv stays consistent (legacy - # default-gateway argv never carried profile flags). + # Our home is a named profile: any explicit DIFFERENT named profile on the argv contradicts it; + # bare argv stays consistent (legacy default-gateway argv never carried profile flags). for flag in ("--profile", "-p"): value = _flag_value(flag) if value is not None and value != profile_name: return True home_value = _flag_value("--hermes-home") or _env_home_value() - if home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home))): - return True - return False + return bool(home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home)))) # Our home is the default/root: ANY explicit named-profile flag on the # argv contradicts it. if _flag_value("--profile") is not None or _flag_value("-p") is not None: return True home_value = _flag_value("--hermes-home") or _env_home_value() - if home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home))): - return True - return False + return bool(home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home)))) async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = False, verbosity: Optional[int] = 0) -> bool: """Start the gateway and run until interrupted. - This is the main entry point for running the gateway. Returns True if the gateway ran - successfully, False if it failed to start. A False return causes a non-zero exit code so - systemd can auto-restart. - - Args: - config: Optional gateway configuration override. - replace: If True, kill any existing gateway instance before starting. - Useful for systemd services to avoid restart-loop deadlocks - when the previous process hasn't fully exited yet. + Returns True if the gateway ran, False if it failed to start (non-zero exit so systemd can + auto-restart). ``replace`` kills any existing instance first — avoids systemd restart-loop + deadlocks when the previous process hasn't fully exited. """ - # Enable interactive exec approval for dangerous commands on messaging - # platforms. Set here (not at module import) so incidental imports of - # gateway.run from CLI/tool code do not poison HERMES_EXEC_ASK. + # Enable interactive exec approval on messaging platforms. Set here (not at module import) so + # incidental imports of gateway.run from CLI/tool code don't poison HERMES_EXEC_ASK. os.environ["HERMES_EXEC_ASK"] = "1" from hermes_cli.resource_limits import apply_nofile_soft_limit @@ -31328,10 +29206,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = from gateway.code_skew import record_boot_fingerprint record_boot_fingerprint() - # ── Duplicate-instance guard ────────────────────────────────────── - # Prevent two gateways from running under the same HERMES_HOME. The PID file is scoped to - # HERMES_HOME, so future multi-profile setups (each profile using a distinct HERMES_HOME) will - # naturally allow concurrent instances without tripping this guard. + # Duplicate-instance guard: no two gateways under one HERMES_HOME. The PID file is scoped to + # HERMES_HOME, so multi-profile setups (distinct HERMES_HOME each) run concurrently untripped. from gateway.status import ( acquire_gateway_runtime_lock, get_running_pid, @@ -31370,11 +29246,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = write_takeover_marker(existing_pid) except Exception as e: logger.debug("Could not write takeover marker: %s", e) - # Snapshot the old gateway's child processes BEFORE signalling it: once it exits, - # orphans are reparented and can no longer be found by a parent walk. On POSIX, adapter - # subprocesses outliving the gateway keep holding scoped token locks and block the - # replacement (Windows terminate_pid(force=True) already tree-kills). Best-effort — [] - # on any failure. + # Snapshot the old gateway's children BEFORE signalling it: once it exits, orphans are + # reparented and invisible to a parent walk. On POSIX, surviving adapter subprocesses hold + # scoped token locks and block the replacement (Windows already tree-kills). Best-effort. try: from gateway.status import _snapshot_gateway_children _old_gateway_children = _snapshot_gateway_children(existing_pid) @@ -31397,18 +29271,16 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception: pass return False - # Wait up to 10 seconds for the old process to exit. - # ``os.kill(pid, 0)`` on Windows is NOT a no-op — use the - # handle-based existence check instead. + # Wait up to 10s for the old process to exit. ``os.kill(pid, 0)`` on Windows is NOT a no-op — + # use the handle-based existence check instead. from gateway.status import _pid_exists old_gateway_exited = False for _ in range(20): if not _pid_exists(existing_pid): old_gateway_exited = True break # Process is gone - # start_gateway is async: a blocking sleep here froze the - # event loop (signal handlers, health checks, every other - # coroutine) for up to 10s per replacement (#36163). + # start_gateway is async: a blocking sleep here freezes the event loop (signal handlers, + # health checks, every coroutine) for up to 10s per replacement. await asyncio.sleep(0.5) else: # Still alive after 10s — force kill @@ -31426,12 +29298,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = old_gateway_exited = True except (PermissionError, OSError): pass - # Confirm the force-kill actually reaped the process before we - # clear its PID file / scoped locks. SIGKILL can fail to take - # (e.g. an uninterruptible-sleep or zombie-reaping parent), and - # if we blindly clear the metadata and start a fresh instance - # we end up with two live gateways fighting over the same - # token — the duplicate-gateway failure in #19471. + # Confirm the force-kill actually reaped the process before clearing its PID file / + # scoped locks: SIGKILL can fail to take (uninterruptible sleep, zombie), and blindly + # clearing metadata would leave two live gateways fighting over the same token. if not old_gateway_exited: for _ in range(20): if not _pid_exists(existing_pid): @@ -31468,10 +29337,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = remove_pid_file() # remove_pid_file() is a no-op when the PID doesn't match. # Force-unlink to cover the old-process-crashed case. - try: + with suppress(Exception): (get_hermes_home() / "gateway.pid").unlink(missing_ok=True) - except Exception: - pass # Clean up any takeover marker the old process didn't consume # (e.g. SIGKILL'd before its shutdown handler could read it). try: @@ -31479,9 +29346,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = clear_takeover_marker() except Exception: pass - # Also release all scoped locks left by the old process. - # Stopped (Ctrl+Z) processes don't release locks on exit, - # leaving stale lock files that block the new gateway from starting. + # Release all scoped locks left by the old process: stopped (Ctrl+Z) processes don't release + # locks on exit, leaving stale lock files that block the new gateway. try: from gateway.status import release_all_scoped_locks _released = release_all_scoped_locks( @@ -31514,15 +29380,13 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception: pass - # Centralized logging — agent.log (INFO+), errors.log (WARNING+), - # and gateway.log (INFO+, gateway-component records only). - # Idempotent, so repeated calls from AIAgent.__init__ won't duplicate. + # Centralized logging — agent.log (INFO+), errors.log (WARNING+), gateway.log (INFO+, gateway + # records only). Idempotent, so repeated calls from AIAgent.__init__ don't duplicate. from hermes_logging import setup_logging, _safe_stderr setup_logging(hermes_home=_hermes_home, mode="gateway") - # Startup security posture audit — warn-on-load, never blocks. Surfaces root / weak-SSH / - # ephemeral-container / unauthenticated-listener posture so operators get the "you're exposed" - # signal the June 2026 MCP-config persistence campaign victims never had. + # Startup security posture audit — warn-on-load, never blocks: surfaces root / weak-SSH / + # ephemeral-container / unauthenticated-listener posture so operators see they're exposed. try: from hermes_cli.security_audit_startup import log_startup_security_warnings @@ -31553,27 +29417,23 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = logging.getLogger().setLevel(_stderr_level) runner = GatewayRunner(config) - # Multiplex: swap the launch-home file handlers for per-profile routers so - # each profile's records land in its own logs/ (#82936). Must run after - # the runner resolved the (possibly None) config and after setup_logging. + # Multiplex: swap the launch-home file handlers for per-profile routers so each profile's records + # land in its own logs/. Must run after the runner resolved (possibly None) config and setup_logging. _enable_multiplex_log_routing(runner.config) - # ``--replace`` is explicit startup authority, not a durable reconnect - # policy. GatewayRunner scopes this bit to cold adapter connects and clears - # it before the background reconnect watcher starts. + # ``--replace`` is explicit startup authority, not a durable reconnect policy: GatewayRunner scopes + # it to cold adapter connects and clears it before the background reconnect watcher starts. runner._platform_lock_takeover_on_start = bool(replace) - # Track whether an unexpected signal initiated the shutdown. When an unexpected SIGTERM kills - # the gateway, we exit non-zero so service managers can revive the process. Planned stop paths - # write a marker before signalling us so they can exit cleanly instead. + # Track whether an unexpected signal initiated shutdown: an unexpected SIGTERM exits non-zero so + # service managers revive us; planned stop paths write a marker first so they exit cleanly. _signal_initiated_shutdown = False # Set up signal handlers def shutdown_signal_handler(received_signal=None): nonlocal _signal_initiated_shutdown - # Planned --replace takeover check: when a sibling gateway is taking over via --replace, it - # wrote a marker naming this PID before sending SIGTERM. Treat as planned and exit 0 so - # systemd's Restart=on-failure doesn't revive us and flap-fight the replacer (e.g. when - # both hermes.service and hermes-gateway.service are enabled). + # Planned --replace takeover: the sibling wrote a marker naming this PID before SIGTERM. Treat as + # planned, exit 0 so systemd's Restart=on-failure doesn't revive us to flap-fight the replacer + # (e.g. when both hermes.service and hermes-gateway.service are enabled). planned_takeover = False try: from gateway.status import consume_takeover_marker_for_self @@ -31581,9 +29441,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception as e: logger.debug("Takeover marker check failed: %s", e) - # Planned stop check: service managers and `hermes gateway stop` also send SIGTERM, which is - # indistinguishable from an unexpected external kill unless the CLI marks it first. SIGINT - # comes from an interactive Ctrl+C and is likewise an intentional foreground stop. + # Planned stop: service managers and `hermes gateway stop` also send SIGTERM, indistinguishable + # from an external kill unless the CLI marks it first. SIGINT is an interactive Ctrl+C stop. planned_stop = False if received_signal == signal.SIGINT: planned_stop = True @@ -31594,9 +29453,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception as e: logger.debug("Planned stop marker check failed: %s", e) - # Fast (<10ms) snapshot of who's asking us to shut down — runs synchronously inside the - # asyncio signal handler, so we keep it purely stdlib + /proc reads, no subprocesses (a - # synchronous `ps aux` here once blocked the loop ~3s and delayed adapter teardown). + # Fast (<10ms) snapshot of who's asking us to shut down — runs synchronously inside the asyncio + # signal handler: stdlib + /proc only, no subprocesses (a sync `ps aux` here once blocked ~3s). try: from gateway.shutdown_forensics import ( format_context_for_log, @@ -31620,19 +29478,17 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = ) else: _signal_initiated_shutdown = True - # Mirror onto the runner so _stop_impl can suppress the gateway_state=stopped persist - # for unexpected signals (container/s6 SIGTERM on restart, OOM, bare kill). Operator - # stops set a planned-stop marker, take the `planned_stop` branch above and leave this - # False so they DO persist "stopped". + # Mirror onto the runner so _stop_impl can suppress the gateway_state=stopped persist for + # unexpected signals (container/s6 SIGTERM on restart, OOM, bare kill). Operator stops set a + # planned-stop marker, take the `planned_stop` branch above and leave this False (DO persist). runner._signal_initiated_shutdown = True logger.info( "Received %s — initiating shutdown", _shutdown_ctx["signal"] if _shutdown_ctx else "SIGTERM/SIGINT", ) - # Always log who/what triggered the signal — most useful single - # line when diagnosing "the gateway keeps dying" tickets. Format - # is one line, key=value, parent_cmdline last (often long). + # Always log who/what triggered the signal — the most useful line for "gateway keeps dying" + # tickets. One line, key=value, parent_cmdline last (often long). if _shutdown_ctx is not None: try: logger.warning( @@ -31641,10 +29497,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception as _e: logger.debug("format_context_for_log failed: %s", _e) - # Spawn the heavyweight diagnostic (ps auxf, pstree, dmesg) in - # a detached subprocess so it can finish writing to disk even - # if our cgroup is being torn down. Bounded by an internal - # timeout; never blocks the event loop here. + # Spawn the heavyweight diagnostic (ps auxf, pstree, dmesg) detached so it can finish + # writing even if our cgroup is torn down; bounded by an internal timeout, never blocks. try: _diag_log = _hermes_home / "logs" / "gateway-shutdown-diag.log" spawn_async_diagnostic( @@ -31659,11 +29513,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = loop = asyncio.get_running_loop() - # Install a loop-level exception handler that swallows transient network errors from background - # tasks: an unhandled telegram TimedOut / NetworkError / httpx connection error in any awaited - # coroutine would otherwise kill the whole gateway (every profile) and lose the active turn. - # Deliberately narrow: only well-known transient network errors are swallowed (logged with - # traceback); everything else goes to the default handler so real bugs still surface. + # Loop-level exception handler swallowing transient network errors from background tasks: an + # unhandled telegram TimedOut / NetworkError / httpx connection error in any awaited coroutine + # would kill the whole gateway. Deliberately narrow — everything else hits the default handler. loop.set_exception_handler(_gateway_loop_exception_handler) if threading.current_thread() is threading.main_thread(): @@ -31680,12 +29532,10 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = else: logger.info("Skipping signal handlers (not running in main thread).") - # Windows fallback: asyncio.add_signal_handler raises NotImplementedError on Windows, so `hermes - # gateway stop`'s SIGTERM (which Python maps to TerminateProcess on Windows) never invokes - # shutdown_signal_handler — no drain loop, no mark_resume_pending, sessions lost on restart. - # A marker-polling thread notices the planned-stop marker `hermes gateway stop` writes BEFORE - # killing and drives the same shutdown path. Runs on every platform (cheap, defensive) so - # non-signal-bearing environments (Windows native, CI runners masking SIGTERM) drain cleanly. + # Windows fallback: asyncio.add_signal_handler raises NotImplementedError there, so `hermes + # gateway stop`'s SIGTERM never reaches shutdown_signal_handler (no drain, sessions lost). A + # marker-polling thread notices the planned-stop marker written BEFORE the kill and drives the + # same shutdown path. Runs everywhere (cheap) so environments masking SIGTERM still drain cleanly. _planned_stop_watcher_stop = threading.Event() _planned_stop_watcher_thread = threading.Thread( target=_run_planned_stop_watcher, @@ -31695,12 +29545,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = ) _planned_stop_watcher_thread.start() - # Claim the PID file BEFORE bringing up any platform adapters. - # This closes the --replace race window: two concurrent `gateway run - # --replace` invocations both pass the termination-wait above, but - # only the winner of the O_CREAT|O_EXCL race below will ever open - # Telegram polling, Discord gateway sockets, etc. The loser exits - # cleanly before touching any external service. + # Claim the PID file BEFORE bringing up any platform adapters: two concurrent `gateway run + # --replace` invocations both pass the termination-wait above, but only the O_CREAT|O_EXCL + # winner ever opens Telegram polling, Discord sockets, etc. The loser exits cleanly first. import atexit from gateway.status import write_pid_file, remove_pid_file, get_running_pid _current_pid = get_running_pid() @@ -31726,23 +29573,17 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = atexit.register(remove_pid_file) atexit.register(release_gateway_runtime_lock) - # Control socket (#92091 step 1) — the gateway-owned identify/status surface. Started right - # after the PID-file claim: winning that O_EXCL race is when this process becomes the - # authoritative gateway for its HERMES_HOME, so "does a socket answer?" is a truthful - # liveness/identity query from here on. Strictly non-fatal: a bind failure only means - # consumers fall back to the process-scan/state-file layer, exactly as before this feature. + # Control socket — the gateway-owned identify/status surface. Started right after the PID-file + # claim, since winning that O_EXCL race makes this process the authoritative gateway for its + # HERMES_HOME. Non-fatal: a bind failure just leaves consumers on the process-scan/state-file layer. _control_server = None try: from gateway.control_socket import GatewayControlServer - # pause-for-update (#92091 step 2): the updater asks this gateway to - # drain in-flight turns and exit cleanly — releasing every venv file - # handle — instead of being tree-killed mid-turn. Same drain path as - # SIGUSR1/service restarts (request_restart(via_service=True)); the - # updater (or the service manager) relaunches after the code swap. - # The handler runs on the socket's executor thread, so the restart - # request is marshalled onto the loop thread; the ACK returns the - # drain budget so the caller knows how long to wait for exit. + # pause-for-update: the updater asks this gateway to drain in-flight turns and exit cleanly + # (releasing every venv file handle) instead of being tree-killed mid-turn — same drain path as + # SIGUSR1/service restarts (request_restart(via_service=True)). The handler runs on the socket's + # executor thread, so the request is marshalled onto the loop; the ACK returns the drain budget. _main_loop = asyncio.get_running_loop() def _pause_for_update_handler() -> dict: @@ -31784,11 +29625,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = logger.debug("Control socket startup failed (non-fatal): %s", _cs_exc) _control_server = None - # Lifecycle ledger (NS-608): report if the previous gateway life died - # uncleanly (SIGKILL / OOM / VM death — no exit path ran), then claim - # the sentinel for this life. Placed after the PID-file/lock claim so - # only the authoritative gateway for this HERMES_HOME touches the - # sentinel — a --replace loser exiting above must not clobber it. + # Lifecycle ledger: report if the previous life died uncleanly (SIGKILL / OOM / VM death), then + # claim the sentinel for this life. Placed after the PID-file/lock claim so only the + # authoritative gateway touches it — a --replace loser exiting above must not clobber it. try: from gateway.lifecycle_ledger import record_startup as _lifecycle_record_startup _lifecycle_record_startup() @@ -31804,10 +29643,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = _ensure_windows_gateway_venv_imports() - # MCP tool discovery — run in an executor so the asyncio event loop stays responsive even when a - # configured MCP server is slow or unreachable. discover_mcp_tools() uses a blocking 120s wait - # internally; calling it from the loop thread would freeze platform heartbeats (Discord shard, - # Telegram polling) until it returned. + # MCP tool discovery in an executor so the loop stays responsive when a configured MCP server is + # slow/unreachable: discover_mcp_tools() blocks up to 120s, which on the loop thread would freeze + # platform heartbeats (Discord shard, Telegram polling). try: await _discover_gateway_mcp_tools(runner.config) except Exception as e: @@ -31836,11 +29674,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = _shutdown_gateway_health_export(runner) if runner.exit_reason: logger.error("Gateway exiting cleanly: %s", runner.exit_reason) - # A clean exit that carries an explicit exit code (e.g. a fatal config error stamped with - # GATEWAY_FATAL_CONFIG_EXIT_CODE) must propagate that code to the process so the s6 finish - # script can translate it (78 → 125) and stop the supervisor restart loop. Otherwise the - # early `return True` below exits 0, the finish script's 78 check never matches, and s6 - # crash-loops the gateway anyway. + # A clean exit carrying an explicit exit code (e.g. GATEWAY_FATAL_CONFIG_EXIT_CODE) must + # propagate so the s6 finish script can translate it (78 → 125) and stop the restart loop; + # otherwise the early `return True` exits 0 and s6 crash-loops the gateway anyway. if runner.exit_code is not None: raise SystemExit(runner.exit_code) return True @@ -31853,10 +29689,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = if runner.exit_reason: logger.error("Gateway exiting with failure: %s", runner.exit_reason) return False - try: + with suppress(Exception): await _shutdown_mcp_servers_nonblocking() - except Exception: - pass if runner.exit_code is not None: raise SystemExit(runner.exit_code) return True @@ -31878,12 +29712,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = ) cron_start_kwargs: Dict[str, Any] = {"adapters": runner.adapters, "loop": asyncio.get_running_loop()} - # Multiplex profiles: tell the built-in ticker which profile homes to - # tick so secondary-profile cron jobs actually fire (#69377). - # Without this, only the process-global HERMES_HOME (default profile) - # is iterated and every secondary profile's cron store is silently - # ignored — jobs show as "scheduled" with a valid next_run_at but - # never execute because no ticker owns that store. + # Multiplex profiles: tell the built-in ticker which profile homes to tick. Otherwise only the + # process-global HERMES_HOME is iterated and secondary profiles' cron jobs show as "scheduled" + # with a valid next_run_at but never execute because no ticker owns that store. if ( isinstance(cron_provider, InProcessCronScheduler) and multiplex_cron @@ -31892,18 +29723,15 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = profile_homes = _multiplex_profile_homes(runner.config) if profile_homes: cron_start_kwargs["profile_homes"] = profile_homes - # Per-profile adapters so each profile's cron output is - # delivered via its own bot/adapter instead of the default - # profile's. + # Per-profile adapters so each profile's cron output goes via its own bot/adapter, not the + # default profile's. cron_start_kwargs["profile_adapters"] = getattr( runner, "_profile_adapters", None ) - # runner.adapters belongs to the default profile, which - # profiles_to_serve() names "default" in its multiplex list. - # Thread that identity so the ticker reserves the shared adapters - # for the default profile alone and never routes a secondary's - # cron through the default bot (even before its adapter connects, - # when profile_adapters[name] is still absent/empty). + # runner.adapters belongs to the default profile ("default" in the multiplex list). + # Thread that identity so the ticker reserves the shared adapters for the default + # profile alone and never routes a secondary's cron through the default bot (even + # before its adapter connects, when profile_adapters[name] is still absent/empty). cron_start_kwargs["default_profile"] = "default" logger.info( "Cron scheduler will tick %d profile(s) under multiplex: %s", @@ -31916,9 +29744,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = exc, ) - # External cron providers own their remote scheduling contract. Only the - # in-process ticker polls local due jobs, so only it receives the local - # external-drain dispatch gate. + # External cron providers own their remote scheduling contract; only the in-process ticker polls + # local due jobs, so only it receives the local external-drain dispatch gate. if isinstance(cron_provider, InProcessCronScheduler): cron_start_kwargs["can_dispatch"] = lambda: not ( runner._draining or runner._external_drain_active @@ -31932,11 +29759,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = ) cron_thread.start() - # Preflight tell for the hosted fire path: an external cron provider (Chronos) delivers - # scheduled fires over HTTP to THIS process's api_server adapter on loopback. If that adapter - # never came up (typically API_SERVER_KEY missing, e.g. relaunched outside the supervisor's - # env), every fire fails with ConnectError while manual runs work — misread as a job bug. Say - # it loudly ONCE at startup, when it is fixable, instead of letting the first miss say it at 2am. + # Preflight tell for the hosted fire path: an external cron provider fires over HTTP to THIS + # process's api_server adapter on loopback. If it never came up (typically API_SERVER_KEY missing) + # every fire fails with ConnectError while manual runs work — misread as a job bug. Say it ONCE now. if not isinstance(cron_provider, InProcessCronScheduler): try: _has_api_server = Platform.API_SERVER in (runner.adapters or {}) @@ -31954,9 +29779,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = getattr(cron_provider, "name", "external"), ) - # Gateway-only periodic housekeeping (channel dir, cache cleanup, paste - # sweep, curator) — runs independently of which cron provider is active. - # Shares cron_stop as the shutdown signal. + # Gateway-only periodic housekeeping (channel dir, cache cleanup, paste sweep, curator) — runs + # independently of the active cron provider; shares cron_stop as the shutdown signal. housekeeping_thread = threading.Thread( target=_start_gateway_housekeeping, args=(cron_stop,), @@ -31970,9 +29794,8 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = ) housekeeping_thread.start() - # READY is emitted only after adapters, cron, and housekeeping have all - # reached their running boundary. Missing config/systemd runtime state - # leaves the watchdog disabled without changing gateway behavior. + # READY is emitted only after adapters, cron and housekeeping reach their running boundary; + # missing config/systemd runtime state leaves the watchdog disabled without changing behavior. start_watchdog = getattr(runner, "_start_systemd_watchdog", None) if callable(start_watchdog): start_watchdog() @@ -32002,10 +29825,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = logger.error("Gateway exiting with failure: %s", runner.exit_reason) return False - # Stop cron scheduler + housekeeping cleanly. These MUST be awaited cooperatively, not - # join()ed: an in-flight cron delivery is a coroutine scheduled onto THIS loop while the - # ticker thread blocks on future.result(); a synchronous join would block the loop so the - # delivery could never run and the message was silently dropped. + # Stop cron scheduler + housekeeping cooperatively, never join()ed: an in-flight cron delivery + # is a coroutine scheduled onto THIS loop while the ticker thread blocks on future.result(); a + # synchronous join would block the loop so the delivery never ran and the message was dropped. cron_stop.set() _stop_cron_provider(cron_provider) if not await _await_thread_exit(cron_thread, timeout=_CRON_SHUTDOWN_DRAIN_TIMEOUT): @@ -32022,22 +29844,16 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = _planned_stop_watcher_thread.join(timeout=2) # Close MCP server connections (off-loop, bounded — #82874) - try: + with suppress(Exception): await _shutdown_mcp_servers_nonblocking() - except Exception: - pass if runner.exit_code is not None: raise SystemExit(runner.exit_code) - # When an unexpected SIGTERM caused the shutdown and it wasn't a planned - # restart (/restart, /update, SIGUSR1), exit non-zero so systemd's - # Restart=on-failure revives the process. This covers: - # - hermes update killing the gateway mid-work - # - External kill commands - # - WSL2/container runtime sending unexpected signals - # `hermes gateway stop` and interactive Ctrl+C are handled above as - # planned stops and should not trigger service-manager revival. + # An unexpected SIGTERM that wasn't a planned restart (/restart, /update, SIGUSR1) exits + # non-zero so systemd's Restart=on-failure revives the process (hermes update killing the + # gateway mid-work, external kills, WSL2/container runtime signals). `hermes gateway stop` and + # Ctrl+C are handled above as planned stops and must not trigger revival. if _signal_initiated_shutdown and not runner._restart_requested: logger.info( "Exiting with code 1 (signal-initiated shutdown without restart " @@ -32060,10 +29876,9 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = def _guard_corrupt_user_config() -> None: """Fail closed when the active profile's config.yaml cannot be parsed. - The gateway is a fully non-interactive surface: nobody is present to repair a corrupt - ``config.yaml``, and silently continuing on built-in defaults lets provider auto-detection - adopt credentials from ``.env`` that the config never named. Same policy and escape hatch - (``HERMES_IGNORE_USER_CONFIG=1``) as the non-interactive CLI guard in ``hermes_cli/main.py``. + Nobody is present to repair a corrupt config on this non-interactive surface, and continuing + on defaults lets provider auto-detection adopt ``.env`` credentials the config never named. + Same policy and escape hatch (``HERMES_IGNORE_USER_CONFIG=1``) as ``hermes_cli/main.py``. """ from hermes_cli.config import ( InvalidUserConfigError, @@ -32083,16 +29898,14 @@ def main(): # startup (watchdog, DB opens, provider resolution). See _guard docstring. _guard_corrupt_user_config() - # Advertise the agent harness to child processes (AI_AGENT is the cross-agent standard; - # HERMES_AGENT the Hermes-specific marker — see _advertise_agent_env in hermes_cli/main.py, kept - # inline here to avoid importing that module's startup side effects). Value must equal our - # registry id ``hermes-agent`` (matching is exact); setdefault never clobbers an outer harness. + # Advertise the agent harness to children (AI_AGENT = cross-agent standard, HERMES_AGENT = Hermes + # marker — mirrors _advertise_agent_env in hermes_cli/main.py, inlined to avoid its startup + # side effects). Value must equal registry id ``hermes-agent`` exactly; setdefault never clobbers. os.environ.setdefault("AI_AGENT", "hermes-agent") os.environ.setdefault("HERMES_AGENT", "true") - # Positive process identity: ledger registration + Windows job-object - # self-attach, so update-time reapers can identify this gateway (and its - # child tree dies with it on Windows). Best-effort — never blocks startup. + # Positive process identity: ledger registration + Windows job-object self-attach, so update-time + # reapers can identify this gateway (and its child tree dies with it on Windows). Best-effort. try: from hermes_cli.process_identity import ( attach_self_to_kill_on_close_job, @@ -32104,10 +29917,9 @@ def main(): except Exception: pass - # Startup-liveness watchdog (OOF-298): armed before config load, DB opens, and the rest of pre- - # loop startup so a deadlock in that window still gets the process respawned by the service - # supervisor instead of wedging as a live-PID zombie. (The argv fast-path in hermes_cli.main - # covers import time even earlier.) Disarmed by GatewayRunner once the loop is confirmed live. + # Startup-liveness watchdog: armed before config load, DB opens and the rest of pre-loop startup + # so a deadlock there still gets the process respawned by the supervisor instead of wedging as a + # live-PID zombie (hermes_cli.main's argv fast-path covers import time). GatewayRunner disarms it. try: from gateway.startup_watchdog import arm_startup_watchdog arm_startup_watchdog() @@ -32137,13 +29949,10 @@ def main(): data = yaml.safe_load(f) or {} config = GatewayConfig.from_dict(data) - # start_gateway() performs the full graceful teardown (adapters disconnected, sessions saved + - # flushed, SQLite closed, cron/MCP stopped, PID file + runtime lock released) before it returns - # OR raises SystemExit with an explicit code. Force-exit afterwards so a wedged non-daemon - # worker thread (e.g. a ThreadPoolExecutor call with no timeout) cannot block interpreter - # finalization (Py_FinalizeEx joins all non-daemon threads) and strand the gateway half-shut. - # SystemExit is caught explicitly (clean-fatal-config, planned-restart, service-restart paths - # all complete teardown first) so EVERY exit path goes through the os._exit backstop. + # start_gateway() completes full graceful teardown before returning OR raising SystemExit. Force- + # exit afterwards so a wedged non-daemon worker thread (e.g. an executor call with no timeout) + # can't block Py_FinalizeEx's thread join and strand the gateway half-shut. SystemExit is caught + # explicitly (all its paths finish teardown first) so EVERY exit path hits the os._exit backstop. try: success = asyncio.run(start_gateway(config)) exit_code = 0 if success else 1 @@ -32161,49 +29970,36 @@ def main(): def _exit_after_graceful_shutdown(exit_code: int) -> None: """Flush stdio, release the PID file + runtime lock, then hard-exit. - Graceful teardown is already complete by the time this runs, so there is nothing left that - needs a clean interpreter shutdown. Deliberately ``os._exit`` (not ``sys.exit``): SystemExit - triggers ``Py_FinalizeEx`` → joins every non-daemon thread — exactly the hang a wedged - tool-worker causes. - - ``os._exit`` bypasses ``atexit``, so the atexit-registered ``remove_pid_file`` / - ``release_gateway_runtime_lock`` never run. The full-shutdown path releases both in - ``_stop_impl``, but the EARLY exit paths (clean-fatal-config, startup-aborted) relied on - atexit; now that those paths are routed through this backstop, release both here explicitly - (both idempotent — no-op on the normal path). - - Logging IS drained here: file handlers run behind an async ``QueueListener`` thread, so - records emitted just before shutdown may sit in the queue and the atexit listener drain - never runs — drain explicitly (bounded) or lose the last lines. Stdio is flushed too. + Graceful teardown is already complete, so ``os._exit`` (not ``sys.exit``): SystemExit triggers + ``Py_FinalizeEx`` → joins every non-daemon thread — exactly the hang a wedged worker causes. + ``os._exit`` bypasses ``atexit``, so ``remove_pid_file`` / ``release_gateway_runtime_lock`` are + called here explicitly (idempotent; the EARLY exit paths relied on atexit). Logging is drained + explicitly (bounded): file handlers sit behind a ``QueueListener`` thread whose atexit drain + never runs, so the last records would otherwise be lost. """ for stream in (sys.stdout, sys.stderr): - try: + with suppress(Exception): stream.flush() - except Exception: - pass - # Release PID + runtime lock BEFORE the log drain: the drain is bounded but - # could still take up to its timeout on a wedged disk, and these locks must - # never be stranded. os._exit skips atexit, and the early SystemExit exit - # paths never run _stop_impl, so release here (idempotent). + # Release PID + runtime lock BEFORE the log drain: the drain is bounded but could take its full + # timeout on a wedged disk, and these locks must never be stranded. os._exit skips atexit and the + # early SystemExit paths never run _stop_impl, so release here (idempotent). try: from gateway.status import remove_pid_file, release_gateway_runtime_lock remove_pid_file() release_gateway_runtime_lock() except Exception: pass - # Mark this life cleanly exited in the lifecycle sentinel (NS-608). This is the single funnel - # every graceful exit passes through, so the next boot's unclean-death detector only fires for - # genuine SIGKILL/OOM/VM deaths. Ownership-guarded internally: a --replace old life won't - # clobber the replacement's freshly claimed "running" sentinel. + # Mark this life cleanly exited in the lifecycle sentinel: the single funnel every graceful exit + # passes through, so the next boot's unclean-death detector fires only for genuine SIGKILL/OOM/VM + # deaths. Ownership-guarded: an old --replace life won't clobber the replacement's fresh sentinel. try: from gateway.lifecycle_ledger import mark_exited mark_exited(exit_code, reason="graceful_shutdown") except Exception: pass - # Drain the async log queue: os._exit bypasses atexit, so the listener's atexit drain won't - # fire. Use drain_log_queue() (bounded, no restart), NOT flush_log_queue(): if the listener is - # wedged on the rotation lock an unbounded stop() join would re-freeze the shutdown. No-ops - # when logging never initialized a queue (very early aborts). + # Drain the async log queue (os._exit bypasses the listener's atexit drain). drain_log_queue() is + # bounded with no restart — NOT flush_log_queue(): a listener wedged on the rotation lock would + # re-freeze shutdown in an unbounded stop() join. No-op when logging never initialized a queue. try: from hermes_logging import drain_log_queue drain_log_queue(timeout=1.0) diff --git a/gateway/session.py b/gateway/session.py index 8de09eceb6..b11a83faa6 100644 --- a/gateway/session.py +++ b/gateway/session.py @@ -104,6 +104,7 @@ from .whatsapp_identity import ( ) from utils import atomic_replace from agent.turn_context import extract_api_content_sidecar +import contextlib # Session keys/ids flow into filesystem paths downstream (e.g. # ``sessions_dir / f"{session_id}.json"`` in hermes_state, request-dump @@ -1378,7 +1379,7 @@ class SessionStore: once it expires, one caller reopens while concurrent callers keep using the JSONL fallback. """ - from hermes_state import SessionDB, _default_db_path, get_shared_session_db + from hermes_state import _default_db_path, get_shared_session_db path = Path(db_path) if db_path is not None else Path(_default_db_path()) def _open(): @@ -2547,10 +2548,8 @@ class SessionStore: return try: origin_json = None - try: + with contextlib.suppress(Exception): origin_json = json.dumps(source.to_dict()) - except Exception: - pass recorder( session_id, source=source.platform.value, diff --git a/gateway/session_db_recovery.py b/gateway/session_db_recovery.py index 0a7be4ace3..0315049969 100644 --- a/gateway/session_db_recovery.py +++ b/gateway/session_db_recovery.py @@ -8,6 +8,7 @@ import weakref from dataclasses import dataclass from pathlib import Path from typing import Any, Callable +import contextlib _INITIAL_RETRY_DELAY_SECONDS = 1.0 @@ -137,10 +138,8 @@ class RecoverableHandleCache: close_rejected = self._close_rejected if stale else None if stale: if close_rejected is not None: - try: + with contextlib.suppress(Exception): close_rejected(handle) - except Exception: - pass return None _publish_health(self._health_source, path, "ok") if was_unavailable and on_recovered is not None: @@ -157,10 +156,8 @@ class RecoverableHandleCache: self.handles.clear() self._unavailable.clear() for handle in handles: - try: + with contextlib.suppress(Exception): close(handle) - except Exception: - pass with _health_lock: states = _health_states.get(self._health_source) if states is not None: diff --git a/gateway/shutdown_forensics.py b/gateway/shutdown_forensics.py index 4a4a76925f..b13ef4476a 100644 --- a/gateway/shutdown_forensics.py +++ b/gateway/shutdown_forensics.py @@ -30,6 +30,7 @@ from gateway.restart import ( DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, resolve_systemd_timeout_stop_sec, ) +import contextlib _SIGNAL_NAME_BY_NUM: Dict[int, str] = {} @@ -91,10 +92,8 @@ def _proc_summary(pid: int) -> Dict[str, Any]: summary["state"] = state ppid = _read_proc_field(pid, "PPid") if ppid is not None: - try: + with contextlib.suppress(ValueError): summary["ppid"] = int(ppid) - except ValueError: - pass uid = _read_proc_field(pid, "Uid") if uid is not None: # "real effective saved fs" @@ -149,10 +148,8 @@ def snapshot_shutdown_context(received_signal: Any = None) -> Dict[str, Any]: # Load average — high load points the finger at "something else # crushing the box" rather than "external killer". - try: + with contextlib.suppress(OSError, AttributeError): ctx["loadavg_1m"] = os.getloadavg()[0] - except (OSError, AttributeError): - pass # /proc/self/status TracerPid: nonzero means a debugger / strace is # attached. Useful when "phantom SIGKILL" turns out to be a manual @@ -268,17 +265,13 @@ def spawn_async_diagnostic( close_fds=True, ) except (FileNotFoundError, OSError): - try: + with contextlib.suppress(OSError): os.close(fd) - except OSError: - pass return None finally: # Subprocess inherited the fd; we can drop our handle. - try: + with contextlib.suppress(OSError): os.close(fd) - except OSError: - pass return proc.pid diff --git a/gateway/shutdown_watchdog.py b/gateway/shutdown_watchdog.py index 3dbeff7cf0..1218e4b3d9 100644 --- a/gateway/shutdown_watchdog.py +++ b/gateway/shutdown_watchdog.py @@ -38,6 +38,7 @@ from typing import Any, Callable, Dict, Optional from gateway.restart import GATEWAY_SERVICE_RESTART_EXIT_CODE from hermes_constants import get_hermes_home from utils import atomic_json_write +import contextlib logger = logging.getLogger(__name__) @@ -175,7 +176,7 @@ def start_loop_liveness_watchdog( if stop_event.is_set(): return - try: + with contextlib.suppress(Exception): logger.critical( "Gateway event loop missed %d consecutive liveness probes; " "dumping all thread stacks and exiting with code %d so the " @@ -183,8 +184,6 @@ def start_loop_liveness_watchdog( strikes, exit_code, ) - except Exception: - pass try: faulthandler.dump_traceback(all_threads=True) except Exception: @@ -406,21 +405,17 @@ def arm_shutdown_watchdog( target = dump_path if dump_path is not None else get_shutdown_watchdog_dump_path() _write_watchdog_dump(target, delay_s=delay, snapshot=snapshot) - try: + with contextlib.suppress(Exception): logger.critical( "Shutdown watchdog fired after %.0fs — forcing process exit " "(asyncio drain path appears wedged; see %s)", delay, target, ) - except Exception: - pass for stream in (sys.stdout, sys.stderr): - try: + with contextlib.suppress(Exception): stream.flush() - except Exception: - pass # Mirror _exit_after_graceful_shutdown: release PID file + runtime # lock BEFORE the log drain (locks must never be stranded), then # drain the async log queue so the logger.critical above actually @@ -470,10 +465,8 @@ async def _tick_socket_handler( except Exception: pass finally: - try: + with contextlib.suppress(Exception): writer.close() - except Exception: - pass async def loop_heartbeat_forever( @@ -638,12 +631,8 @@ async def loop_heartbeat_forever( finally: if tick_server is not None: tick_server.close() - try: + with contextlib.suppress(Exception): await tick_server.wait_closed() - except Exception: - pass if tick_socket_path is not None: - try: + with contextlib.suppress(Exception): tick_socket_path.unlink(missing_ok=True) - except Exception: - pass diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 5304ee1eb8..fded4616ee 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -1,16 +1,9 @@ """Gateway slash-command handlers for GatewayRunner. -Extracted from ``gateway/run.py`` (god-file decomposition Phase 3b). These are -the in-session slash commands (/model, /reset, /usage, /compress, ...) the -gateway dispatches from ``_handle_message``. There are 42 of them (~3,200 LOC); -lifting them into a mixin that ``GatewayRunner`` inherits keeps every -``self._handle_*_command`` dispatch + test reference working via the MRO, while -removing the bulk from run.py. - -Module-level run.py helpers a handler needs (``_hermes_home``, -``_load_gateway_config``, ``_resolve_gateway_model``, etc.) are imported lazily -inside the handler body — a deferred ``from gateway.run import ...`` resolves at -call time (run.py fully loaded by then), avoiding an import cycle. +The in-session slash commands (/model, /reset, /usage, ...) dispatched from ``_handle_message``, +lifted out of ``gateway/run.py`` into a mixin so every ``self._handle_*_command`` reference keeps +working via the MRO. run.py helpers (``_hermes_home``, ``_load_gateway_config``, ...) are imported +lazily inside handler bodies — a deferred ``from gateway.run import ...`` avoids the import cycle. """ from __future__ import annotations @@ -46,6 +39,7 @@ from utils import ( base_url_host_matches, is_truthy_value, ) +import contextlib logger = logging.getLogger("gateway.run") @@ -71,12 +65,9 @@ def _int_value(value: Any) -> int: def _model_switch_skew_guard() -> Optional[str]: """Refuse a model switch when the gateway is running stale code. - A long-lived gateway holds its modules in memory from boot. If the checkout changed - underneath it (e.g. a manual ``git pull``), switching models can hit a first-time lazy import - on a new code path and crash on a stale cached dependency — the cryptic - ``cannot import name 'env_float' from 'utils'``. Detect the drift and tell the user to - restart instead. Deliberately scoped to model switching (the highest-risk trigger); other - lazy-import sites are not guarded. + A long-lived gateway keeps boot-time modules in memory; if the checkout changed underneath it, + a first-time lazy import on a new code path can crash on a stale cached dependency. Detect the + drift and ask for a restart. Scoped to model switching only (the highest-risk trigger). """ from gateway.code_skew import detect_code_skew @@ -97,11 +88,9 @@ def _model_switch_skew_guard() -> Optional[str]: def _home_thread_from_source(source) -> Optional[str]: """The thread id /sethome should persist on the home target, or None. - Slack thread-per-message session keying stamps a top-level message's own id as - ``source.thread_id`` (a session KEY, not a durable location). Persisting it would pin HOME to - the ephemeral thread around the /sethome message, so every bare ``deliver="slack"`` would land - there forever. Same recognition as cron origin capture: a Slack thread id equal to the - message's own id is synthetic; a genuine thread (id = parent's) is kept as the home target. + Slack thread-per-message keying stamps a top-level message's own id as ``source.thread_id`` (a + session key, not a location); persisting it would pin HOME to that ephemeral thread. A thread + id equal to the message's own id is synthetic and dropped; a real thread (id = parent's) is kept. """ thread_id = getattr(source, "thread_id", None) if not thread_id: @@ -123,9 +112,8 @@ class GatewaySlashCommandsMixin: def _typed_command_prefix_for(self, platform) -> str: """Return the prefix users can always type to reach Hermes commands. - Reads the adapter's ``typed_command_prefix`` capability flag (default "/"). Slack and - Matrix return "!" because typed "/" commands are blocked in Slack threads / reserved by - Matrix clients; their adapters rewrite "!command" to "/command" on receive. + Adapter ``typed_command_prefix`` capability (default "/"). Slack and Matrix use "!" because + typed "/" is blocked/reserved there; their adapters rewrite "!command" to "/command". """ adapter = self.adapters.get(platform) if getattr(self, "adapters", None) else None return getattr(adapter, "typed_command_prefix", "/") if adapter is not None else "/" @@ -146,12 +134,10 @@ class GatewaySlashCommandsMixin: # expiring session id before reset_session() rotates it. old_entry = self.session_store._entries.get(session_key) - # Close tool resources on the old agent (terminal sandboxes, browser daemons, background - # processes) before evicting from cache. Guard with getattr because test fixtures may skip - # __init__. _cleanup_agent_resources is synchronous and can block (subprocess teardown, - # memory-provider network IO) and this handler runs ON the event loop when a confirm-button - # click resolves the slash-confirm — an inline call wedges the loop and the bot goes silent. - # Offload to a worker thread with a bounded timeout. + # Close the old agent's tool resources (sandboxes, browser daemons, subprocesses) before + # evicting it; getattr-guarded since test fixtures may skip __init__. _cleanup_agent_resources + # is blocking and this handler runs ON the event loop (confirm-button click), so an inline + # call wedges the loop — offload to a worker thread with a bounded timeout. _cache_lock = getattr(self, "_agent_cache_lock", None) if _cache_lock is not None: with _cache_lock: @@ -188,10 +174,9 @@ class GatewaySlashCommandsMixin: # security state in one funnel call. See _CONVERSATION_SCOPED_STATE in gateway/run.py. self._clear_conversation_scope(session_key, reason="session_reset") - # The old conversation's in-flight async delegations end WITH it: after the reset rotates - # the session id, their completions would have no live owner — a dangling subagent can only - # burn tokens and park an orphaned payload on the shared queue. Interrupt by the expiring - # durable session id (parent_session_id) and by routing key as a fallback for older records. + # The old conversation's in-flight async delegations end WITH it: once the session id rotates + # their completions have no live owner (orphaned payload on the shared queue, wasted tokens). + # Interrupt by expiring durable session id (parent_session_id), routing key as legacy fallback. try: from tools.async_delegation import interrupt_for_session @@ -226,7 +211,7 @@ class GatewaySlashCommandsMixin: # Fire plugin on_session_finalize hook (session boundary). Off-loop + bounded: finalize # hooks can block arbitrarily (observability trace exports) and this handler runs on the # gateway event loop (see GatewayRunner._finalize_session_off_loop). - try: + with contextlib.suppress(Exception): await self._finalize_session_off_loop( session_id=_old_sid, platform=source.platform.value if source.platform else "", @@ -234,8 +219,6 @@ class GatewaySlashCommandsMixin: old_session_id=_old_sid, new_session_id=new_entry.session_id if new_entry else None, ) - except Exception: - pass # Emit session:end hook (session is ending) await self.hooks.emit("session:end", { @@ -329,12 +312,10 @@ class GatewaySlashCommandsMixin: async def _handle_profile_command(self, event: MessageEvent) -> str: """Handle /profile — show the profile serving this source and its home. - On a multiplexed gateway the process-level active profile is always the multiplexer's own, - so reporting it would answer "default" in every chat regardless of which profile serves the - room (``source.profile``). When ``multiplex_profiles`` is on, report the stamped profile - and, like the scoped /reset banner, resolve the displayed home under that profile's runtime - scope. When off (default) the stamp is ignored — mirroring ``_run_agent`` — and output is - byte-identical to before. + On a multiplexed gateway the process-level active profile is the multiplexer's own, so it + would read "default" in every chat. With ``multiplex_profiles`` on, report ``source.profile`` + and resolve home under that profile's runtime scope (like the scoped /reset banner); when + off the stamp is ignored, mirroring ``_run_agent``. """ from hermes_constants import display_hermes_home from hermes_cli.slash_exec import CommandContext, execute_command @@ -377,9 +358,8 @@ class GatewaySlashCommandsMixin: async def _handle_whoami_command(self, event: MessageEvent) -> str: """Handle /whoami — show the user's slash command access on this scope. - Always works (it's in the always-allowed floor of slash_access). Reports: platform, scope - (DM vs group), the user's tier (admin / user / unrestricted), and the slash commands they - can actually run on this scope. + Always allowed (slash_access floor). Reports platform, DM-vs-group scope, tier, and the + commands the user can actually run here. """ from gateway.slash_access import policy_for_source as _policy_for_source @@ -427,9 +407,8 @@ class GatewaySlashCommandsMixin: async def _handle_kanban_command(self, event: MessageEvent) -> str: """Handle /kanban — delegate to the shared kanban CLI. - Run the potentially-blocking DB work in a thread pool so the gateway event loop stays - responsive. Reads and mutations are both allowed while an agent runs: the board is - profile-agnostic and never touches the running agent's state. + DB work runs in a thread pool to keep the event loop responsive. Reads and mutations are + allowed while an agent runs: the board is profile-agnostic and never touches agent state. """ import asyncio import re @@ -486,10 +465,9 @@ class GatewaySlashCommandsMixin: chat_type = str(getattr(source, "chat_type", "") or "") or None thread_id = str(getattr(source, "thread_id", "") or "") user_id = str(getattr(source, "user_id", "") or "") or None - # Persist the platform-specific stable alt id (Signal UUID, - # Feishu union_id) too: build_session_key keys the participant - # on ``user_id_alt or user_id``, so a replayed wake only rebuilds - # the same session key when the alt id survives the round-trip. + # Also persist the stable alt id (Signal UUID, Feishu union_id): build_session_key + # keys the participant on ``user_id_alt or user_id``, so a replayed wake rebuilds + # the same session key only when the alt id survives the round-trip. user_id_alt = str(getattr(source, "user_id_alt", "") or "") or None delivery_metadata = self._thread_metadata_for_source( source, self._reply_anchor_for_event(event) @@ -540,7 +518,7 @@ class GatewaySlashCommandsMixin: source = event.source session_entry = await self.async_session_store.get_or_create_session(source) - connected_platforms = [p.value for p in self.adapters.keys()] + connected_platforms = [p.value for p in self.adapters] # Check if there's an active agent. Keep the sentinel distinct: a # starting/pending run should not be treated as a fully usable agent for @@ -614,7 +592,6 @@ class GatewaySlashCommandsMixin: model_name = "" provider_name = "" - base_url = "" route_resolved = False context_used = 0 context_total = 0 @@ -624,7 +601,6 @@ class GatewaySlashCommandsMixin: if live_model and live_provider: model_name = live_model provider_name = live_provider - base_url = _clean_str(getattr(status_agent, "base_url", "")) route_resolved = True ctx = getattr(status_agent, "context_compressor", None) if ctx is not None: @@ -636,12 +612,10 @@ class GatewaySlashCommandsMixin: if not route_resolved and persisted_model and persisted_provider: model_name = persisted_model provider_name = persisted_provider - base_url = _clean_str(persisted_route.get("billing_base_url")) route_resolved = True if not route_resolved: model_name = _clean_str(session_row.get("model")) provider_name = _clean_str(session_row.get("billing_provider")) - base_url = _clean_str(session_row.get("billing_base_url")) context_used = context_used or _int_value(getattr(session_entry, "last_prompt_tokens", 0)) user_config: dict[str, Any] = {} @@ -956,13 +930,10 @@ class GatewaySlashCommandsMixin: and origin.platform == Platform.MATRIX and current.platform == Platform.MATRIX and origin.chat_id == current.chat_id - # thread_id is part of the session key (build_session_key appends it - # for every chat type when present), and Matrix scopes the model's - # turn to the current room/thread. A live session in another thread - # of the SAME room is a DIFFERENT session, so a caller in thread A - # must not resume/enumerate a target whose origin is in thread B. - # Non-threaded rooms have empty thread_id on both sides ("" == ""), - # so room-level sharing is preserved unchanged. + # thread_id is part of the session key (build_session_key appends it for every chat + # type when present) and Matrix scopes a turn to the current room/thread, so a live + # session in another thread of the SAME room is a DIFFERENT session: thread A must not + # resume/enumerate a target from thread B. Non-threaded rooms compare "" == "" unchanged. and str(getattr(current, "thread_id", "") or "") == str(getattr(origin, "thread_id", "") or "") ) @@ -970,11 +941,9 @@ class GatewaySlashCommandsMixin: def _same_origin_chat(self, current: SessionSource, origin: Optional[SessionSource]) -> bool: """Platform-agnostic counterpart to ``_same_matrix_room``. - Group and thread sessions that ``build_session_key`` isolates per participant (the - default ``group_sessions_per_user=True``) must also be scoped by participant here — - otherwise a co-member could resume another member's live per-user group session (IDOR). - Only an explicitly shared group/thread lets co-members share, mirroring the key contract - via ``is_shared_multi_user_session``. + Per-participant sessions (``build_session_key`` with the default ``group_sessions_per_user``) + must be participant-scoped here too, else a co-member could resume another member's live + session (IDOR). Only an explicitly shared group/thread (``is_shared_multi_user_session``) shares. """ if origin is None or current is None: return False @@ -982,13 +951,10 @@ class GatewaySlashCommandsMixin: return False if origin.chat_id != current.chat_id: return False - # thread_id is part of the session key for every chat type when present - # (build_session_key appends it unconditionally), so a session in one - # thread is a DIFFERENT session from another thread of the same parent - # chat. is_shared_multi_user_session only decides participant sharing - # WITHIN a thread, never across threads — require thread equality before - # any sharing logic so a live origin in thread A cannot match a caller in - # thread B of the same parent chat. + # thread_id is part of the session key for every chat type (build_session_key appends it + # unconditionally), so threads of the same parent chat are DIFFERENT sessions. + # is_shared_multi_user_session only decides sharing WITHIN a thread — require thread equality + # before any sharing logic so a live origin in thread A cannot match a caller in thread B. if str(getattr(current, "thread_id", "") or "") != str( getattr(origin, "thread_id", "") or "" ): @@ -996,14 +962,11 @@ class GatewaySlashCommandsMixin: chat_type = (getattr(current, "chat_type", "") or "").lower() # DM-like chats are always per-user. if chat_type in {"dm", "direct", "private", ""}: - # chat_id was already required equal above and, when present, IS the - # DM session key — so an equal non-empty chat_id is sufficient. - # build_session_key only falls back to the participant id - # (``user_id_alt or user_id`` — Signal/Feishu key on user_id_alt) - # when there is NO chat_id; mirror that and fail closed on a - # missing/different participant so two no-chat_id DM origins are - # never conflated (was: compared user_id only and allowed when - # either side was missing). + # chat_id was already required equal above and, when present, IS the DM session key, so + # an equal non-empty chat_id suffices. build_session_key falls back to the participant + # (``user_id_alt or user_id`` — Signal/Feishu key on user_id_alt) only when there is NO + # chat_id; mirror that and fail closed on a missing/different participant so two + # no-chat_id DM origins are never conflated. if str(getattr(current, "chat_id", "") or ""): return True cur_pid = str(current.user_id_alt or current.user_id or "") @@ -1030,13 +993,11 @@ class GatewaySlashCommandsMixin: return False def _resume_caller_is_admin(self, source: SessionSource) -> bool: - """Whether *source* is an EXPLICITLY-configured admin allowed to make a cross-origin /resume - or /sessions listing. + """Whether *source* is an EXPLICITLY-configured admin allowed cross-origin /resume or /sessions. - Deliberately stricter than ``SlashAccessPolicy.is_admin()``: that returns True for every - allowed caller when slash gating is DISABLED (so commands stay runnable by default), but - cross-ORIGIN DATA ACCESS must require a real, configured admin — otherwise the default - (no admin list) config would make every caller cross-origin-capable and re-open the IDOR. + Stricter than ``SlashAccessPolicy.is_admin()``, which returns True for every allowed caller + when slash gating is DISABLED; cross-origin DATA ACCESS needs a real configured admin, else + the default (no admin list) config would make every caller cross-origin-capable (IDOR). """ try: from gateway.slash_access import policy_for_source @@ -1051,11 +1012,9 @@ class GatewaySlashCommandsMixin: ) -> bool: """Whether *source* may resume the persisted session *target_id*. - Generalizes the Matrix-only room guard to every adapter so a caller cannot bind their - gateway session to another user's/room's persisted session id (IDOR). Uses the live origin - when the target is active; otherwise the DB row's source + user_id. A row must PROVE the - same owner; insufficient ownership data fails closed. An explicit admin ``--all`` override - bypasses scoping. + Generalizes the Matrix-only room guard to every adapter so a caller cannot bind to another + user's/room's session (IDOR). Uses the live origin when the target is active, else the DB + row's source + user_id; the row must PROVE ownership or fail closed. Admin ``--all`` bypasses. """ if allow_override and self._resume_caller_is_admin(source): return True @@ -1079,28 +1038,23 @@ class GatewaySlashCommandsMixin: return False # different platform / source caller_uid = str(getattr(source, "user_id", "") or "") row_uid = str(row.get("user_id") or "") - # Chat/thread origin recorded at session creation (see SessionDB._insert_session_row). The - # sessions table historically stored only source + user_id, so a same-user row could belong - # to a DIFFERENT chat; comparing the persisted origin closes that gap. Legacy rows have NULL - # here and fail closed — resume them via a live session or an admin override. + # Chat/thread origin recorded at session creation. Rows once stored only source + user_id, + # so a same-user row could belong to a DIFFERENT chat; comparing the persisted origin closes + # that gap. Legacy rows (NULL) fail closed — resume via a live session or an admin override. caller_chat = str(getattr(source, "chat_id", "") or "") row_chat = str(row.get("chat_id") or "") caller_thread = str(getattr(source, "thread_id", "") or "") row_thread = str(row.get("thread_id") or "") chat_type = (getattr(source, "chat_type", "") or "").lower() caller_is_dm = chat_type in {"dm", "direct", "private", ""} - # build_session_key keys the participant on ``user_id_alt or user_id`` (Signal/Feishu carry - # the canonical participant in user_id_alt), but the sessions table only ever stored user_id - # — it has no user_id_alt column. So a row cannot prove the canonical participant for an - # alt-keyed caller: any per-user comparison relying on row_uid == caller_uid must fail - # closed to stay in lock-step with the key boundary (CWE-639). Shared sessions unaffected. + # build_session_key keys the participant on ``user_id_alt or user_id``, but the sessions table + # has no user_id_alt column, so a row cannot prove the canonical participant for an alt-keyed + # (Signal/Feishu) caller: per-user row_uid == caller_uid checks must fail closed (CWE-639). caller_keys_on_alt = bool(str(getattr(source, "user_id_alt", "") or "")) if caller_uid: - # Identity-bearing caller: allow only when the row PROVES the same owner AND the same - # platform/origin AND the same chat/thread. A blank/legacy source can't prove the - # platform (the row_src check above only rejects a *mismatching* non-blank source); a - # different thread is a different session (build_session_key appends thread_id). Any - # gap fails closed — legacy NULL-owner/blank-source/NULL-chat rows are not resumable here. + # Identity-bearing caller: the row must PROVE the same owner AND platform AND chat/thread. + # A blank/legacy source can't prove the platform (row_src above only rejects a *mismatching* + # non-blank one); a different thread is a different session. Any gap fails closed. origin_ok = ( bool(row_src) and bool(caller_src) and str(row_src) == str(caller_src) @@ -1109,11 +1063,9 @@ class GatewaySlashCommandsMixin: if not origin_ok: return False if caller_is_dm: - # DMs are keyed on user_id; require the same owner. chat_id is legitimately absent - # on both sides for a no-chat_id DM (scoped by user_id), but a mismatching chat_id - # (when present) is still rejected. A no-chat_id DM is keyed PURELY on the - # participant, so an alt-keyed caller fails closed there; with chat_id on both sides - # equal chat_id is the DM key and suffices. + # DMs are keyed on user_id; require the same owner. A no-chat_id DM is keyed PURELY on + # the participant (so an alt-keyed caller fails closed); when both sides carry chat_id, + # equality is the DM key and suffices, and a mismatching chat_id is rejected. if caller_keys_on_alt and not (bool(row_chat) and bool(caller_chat)): return False return ( @@ -1125,13 +1077,10 @@ class GatewaySlashCommandsMixin: # equal — a legacy NULL-chat row fails closed even when both normalize to "". (CWE-639) if not (bool(row_chat) and bool(caller_chat) and row_chat == caller_chat): return False - # Within the same non-DM chat/thread, mirror build_session_key's - # participant scoping: a SHARED group/thread session - # (group_sessions_per_user=False, or a shared thread) is one session - # for every participant, so the same-chat proof above is sufficient — - # do NOT also require user-id equality (otherwise a co-member is - # wrongly blocked from their own shared session). A per-user session - # still requires the same owner. + # Same non-DM chat/thread: mirror build_session_key's participant scoping. A SHARED + # group/thread session (group_sessions_per_user=False, or a shared thread) is one session + # for every participant, so the same-chat proof suffices — do NOT also require user-id + # equality (it would block co-members). A per-user session still requires the same owner. shared = is_shared_multi_user_session( source, group_sessions_per_user=getattr(self.config, "group_sessions_per_user", True), @@ -1145,10 +1094,9 @@ class GatewaySlashCommandsMixin: if caller_keys_on_alt: return False return bool(row_uid) and row_uid == caller_uid - # No caller identity: the persisted row carries only source + user_id (the sessions table - # has no chat_id), so a same-platform row can belong to a DIFFERENT chat or user. Same - # platform alone is NOT ownership proof — fail closed. Same-chat resume of an ACTIVE session - # still works via the live-origin branch above. (CWE-639: IDOR on session routing.) + # No caller identity: the row carries only source + user_id, so a same-platform row can belong + # to a DIFFERENT chat or user — same platform alone is NOT ownership proof; fail closed + # (CWE-639). Same-chat resume of an ACTIVE session still works via the live-origin branch. return False async def _resume_row_visible( @@ -1156,9 +1104,8 @@ class GatewaySlashCommandsMixin: ) -> bool: """Whether a titled-session listing *row* belongs to the caller's origin. - Prevents cross-origin enumeration of session ids/previews via the numbered /resume list. - Preserves the existing Matrix room-scoping semantics; scopes every other platform to the - caller's own sessions unless an admin passes ``--all``. + Prevents cross-origin enumeration of session ids/previews via the numbered /resume list; + keeps Matrix room-scoping, scopes every other platform to the caller unless admin ``--all``. """ sid = str(row.get("id") or "") if source.platform == Platform.MATRIX: @@ -1357,10 +1304,9 @@ class GatewaySlashCommandsMixin: ) return EphemeralReply(t("gateway.stop.stopped")) - # No run under the caller's own session key. In a per-user thread - # (thread_sessions_per_user=True) each participant is isolated even inside one shared - # thread, so a run another user started lives under a different key. Authorized users must - # still be able to /stop it: fall back to sibling runs in this thread, gated on authorization. + # No run under the caller's own key. In a per-user thread (thread_sessions_per_user=True) a + # run another user started lives under a different key, yet authorized users must still be + # able to /stop it: fall back to sibling runs in this thread, gated on authorization. sibling_keys = self._sibling_thread_run_keys(source, session_key) if sibling_keys and self._is_user_authorized(source): for sibling_key in sibling_keys: @@ -1399,13 +1345,8 @@ class GatewaySlashCommandsMixin: return t("gateway.stop.no_active") async def _handle_platform_command(self, event: MessageEvent) -> str: - """Handle ``/platform list|pause|resume [name]`` — surface and manually control - failed/paused gateway adapters. - - Examples: - ``/platform list`` — show connected + failed/paused platforms - ``/platform pause whatsapp`` — stop the reconnect watcher hammering whatsapp - ``/platform resume whatsapp`` — re-queue a paused platform for retry + """Handle ``/platform list|pause|resume [name]`` — inspect and manually control failed/paused + gateway adapters (pause stops the reconnect watcher; resume re-queues for retry). """ text = (getattr(event, "content", "") or "").strip() # Strip the leading "/platform" (or "/PLATFORM") token if present @@ -1426,7 +1367,7 @@ class GatewaySlashCommandsMixin: if action == "list": lines = ["**Gateway platforms**"] - connected = sorted(p.value for p in self.adapters.keys()) + connected = sorted(p.value for p in self.adapters) if connected: lines.append("Connected: " + ", ".join(connected)) else: @@ -1494,11 +1435,9 @@ class GatewaySlashCommandsMixin: async def _handle_restart_command(self, event: MessageEvent) -> Union[str, EphemeralReply]: """Handle /restart command - drain active work, then restart the gateway.""" from gateway.run import _hermes_home - # Defensive idempotency check: if the previous gateway process recorded this same /restart - # (same platform + update_id) and the new process is seeing it *again*, this is a re- - # delivery caused by PTB's graceful-shutdown `get_updates` ACK failing on the way out - # ("updates may be received twice" in gateway.log). Ignoring the stale redelivery prevents - # a self-perpetuating loop where every fresh gateway re-runs /restart and restarts again. + # Idempotency check: if the previous gateway process recorded this same /restart (platform + + # update_id) and we see it *again*, it's a redelivery from PTB's graceful-shutdown get_updates + # ACK failing on the way out. Ignoring it prevents a loop where every fresh gateway re-restarts. if self._is_stale_restart_redelivery(event): logger.info( "Ignoring redelivered /restart (platform=%s, update_id=%s) — " @@ -1571,11 +1510,9 @@ class GatewaySlashCommandsMixin: logger.debug("Failed to write restart dedup marker: %s", e) active_agents = self._running_agent_count() - # When running under a service manager (systemd/launchd) or inside a Docker/Podman - # container, use the service restart path: exit with code 75 so the service manager / - # container restart policy restarts us. The detached setsid+bash approach fails there - # (systemd KillMode=mixed kills the cgroup; tini exits with the gateway). Native supervisor - # markers cover direct starts; the explicit marker covers wrappers like ``sudo env -i``. + # Under a service manager (systemd/launchd) or Docker/Podman, exit 75 so the supervisor / + # restart policy restarts us — detached setsid+bash fails there (systemd KillMode=mixed kills + # the cgroup; tini exits with the gateway). The explicit marker covers ``sudo env -i`` wrappers. from gateway.restart import ( is_container_restart_context, is_gateway_supervisor_process, @@ -1628,6 +1565,291 @@ class GatewaySlashCommandsMixin: getattr(getattr(event, "source", None), "platform", None), ) + async def _perform_model_switch( + self, + switch_model, + *, + raw_input: str, + explicit_provider, + session_key: str, + source, + current_model, + current_provider, + current_base_url, + current_api_key, + persist_global: bool, + user_provs, + custom_provs, + ): + """Resolve a /model switch off-loop. Returns ``(result, None)`` or ``(None, error_text)``.""" + from gateway.run import _load_gateway_config + + skew_error = _model_switch_skew_guard() + if skew_error: + return None, skew_error + # Offload the switch off the event loop — switch_model() can fall through to a synchronous + # models.dev HTTP fetch (requests.get, 15s timeout) on a cold/expired cache, which freezes + # the gateway otherwise. + result = await asyncio.to_thread( + switch_model, + raw_input=raw_input, + current_provider=current_provider, + current_model=current_model, + current_base_url=current_base_url, + current_api_key=current_api_key, + is_global=persist_global, + explicit_provider=explicit_provider, + user_providers=user_provs, + custom_providers=custom_provs, + ) + if not result.success: + return None, t("gateway.model.error_prefix", error=result.error_message) + try: + from hermes_cli.context_switch_guard import enrich_model_switch_warnings_for_gateway + + # Offload: merge_preflight_compression_warning() calls the sync + # resolve_display_context_length() provider probe ladder — must not run on the loop. + await asyncio.to_thread( + enrich_model_switch_warnings_for_gateway, + result, + self, + session_key=session_key, + source=source, + custom_providers=custom_provs, + load_gateway_config=_load_gateway_config, + ) + except Exception as exc: + logger.debug("preflight-compression switch warning failed: %s", exc) + return result, None + + async def _commit_model_switch( + self, + result, + *, + session_key: str, + source, + current_model, + current_base_url, + current_api_key, + custom_provs, + persist_global: bool, + config_path, + one_turn: bool = False, + restore_snapshot=None, + picker: bool = False, + ) -> str: + """Apply a resolved switch (cached agent, session, config) and build the confirmation. + + Shared by the typed ``/model `` path and the picker callback (``picker=True``). + """ + from gateway.run import _load_gateway_config + from hermes_cli.model_switch import format_model_for_display, resolve_display_context_length_async + + # If there's a cached agent, update it in-place + cached_entry = None + _cache_lock = getattr(self, "_agent_cache_lock", None) + _cache = getattr(self, "_agent_cache", None) + if _cache_lock and _cache is not None: + with _cache_lock: + cached_entry = _cache.get(session_key) + if cached_entry and cached_entry[0] is not None: + try: + cached_entry[0].switch_model( + new_model=result.new_model, + new_provider=result.target_provider, + api_key=result.api_key, + base_url=result.base_url, + api_mode=result.api_mode, + capabilities=getattr(result, "runtime_capabilities", None), + ) + except Exception as exc: + # In-place swap rolled back to the OLD working model/client and re-raised. Abort the + # commit (DB persist, session override, cache eviction, config write) so a failed switch + # is a no-op — otherwise the next message rebuilds a broken agent from the override. + logger.warning( + "%s model switch failed for cached agent: %s", "Picker" if picker else "In-place", exc + ) + return t( + "gateway.model.error_prefix", + error=f"Model switch to {result.new_model} failed ({exc}); staying on {current_model}.", + ) + + # Persist the new model to the session DB so the dashboard shows the updated model. + _sess_db = getattr(self, "_session_db", None) + if _sess_db is not None: + try: + _sess_entry = await self.async_session_store.get_or_create_session(source) + # Typed path: if this session was auto-reset, consume the flag so the next regular + # message's cleanup does not wipe the model override just stored below. + if not picker and getattr(_sess_entry, "was_auto_reset", False): + _sess_entry.was_auto_reset = False + await _sess_db.update_session_model( + _sess_entry.session_id, result.new_model, + provider=result.target_provider, + ) + except Exception as exc: + logger.debug("Failed to persist model switch to DB: %s", exc) + + # Store a note to prepend to the next user message so the model knows about the switch + # (avoids system messages mid-history). Display form strips opaque Palantir RID + # prefixes; the override map below keeps the full ID for the wire. + if not hasattr(self, "_pending_model_notes"): + self._pending_model_notes = {} + self._pending_model_notes[session_key] = ( + f"[Note: model was just switched from {format_model_for_display(current_model)} to " + f"{format_model_for_display(result.new_model)} " + f"via {result.provider_label or result.target_provider}. " + f"{'This override applies to the next turn only. ' if one_turn else ''}" + f"Adjust your self-identification accordingly.]" + ) + + # Store session override so next agent creation uses the new model + self._session_model_overrides[session_key] = { + "model": result.new_model, + "provider": result.target_provider, + "api_key": result.api_key, + "base_url": result.base_url, + "api_mode": result.api_mode, + "request_overrides": dict(result.request_overrides or {}), + "capabilities": dict(result.runtime_capabilities or {}), + } + if one_turn: + if not hasattr(self, "_pending_one_turn_model_restores"): + self._pending_one_turn_model_restores = {} + self._pending_one_turn_model_restores[session_key] = ( + restore_snapshot or {"had_override": False, "override": None} + ) + elif not picker and hasattr(self, "_pending_one_turn_model_restores"): + self._pending_one_turn_model_restores.pop(session_key, None) + + # Write-through the non-secret parts (model/provider/base_url) so the override survives a + # restart; api_key/api_mode are never persisted (re-resolved on rehydration). /model --once is + # EXCLUDED: a one-turn override must not outlive a restart; the pre-once value stays persisted. + if not one_turn: + try: + await self.async_session_store.set_model_override( + session_key, self._session_model_overrides[session_key] + ) + except Exception: + logger.debug("Failed to persist session model override", exc_info=True) + + # Evict cached agent so the next turn creates a fresh agent from the + # override rather than relying on cache signature mismatch detection. + self._evict_cached_agent(session_key) + + # Persist to config (default) unless --session opted out + if persist_global: + try: + # Write-back round-trip: raw read is correct (merged + # defaults must not be persisted back to the user's file). + from hermes_cli.config import read_user_config_raw, save_config + cfg = read_user_config_raw(config_path) + # Coerce scalar/None ``model:`` into a dict before mutation — otherwise + # ``cfg.setdefault("model", {})`` returns the existing scalar and the next + # assignment raises ``TypeError: 'str' object does not support item assignment``. + raw_model = cfg.get("model") + if isinstance(raw_model, dict): + model_cfg = raw_model + elif isinstance(raw_model, str) and raw_model.strip(): + model_cfg = {"default": raw_model.strip()} + cfg["model"] = model_cfg + else: + model_cfg = {} + cfg["model"] = model_cfg + try: + from hermes_cli.route_identity import should_clear_context_pin_async + + if await should_clear_context_pin_async( + model_cfg.get("default") or model_cfg.get("model"), + result.new_model, + model_cfg.get("base_url"), + result.base_url, + model_cfg.get("provider"), + result.target_provider, + ): + model_cfg.pop("context_length", None) + except Exception: + model_cfg.pop("context_length", None) + model_cfg["default"] = result.new_model + model_cfg["provider"] = result.target_provider + # Named providers re-resolve base_url/api_mode fresh, so leftovers are cleared + # unconditionally; custom providers have no registry entry to re-derive from, so + # they need an explicit set-or-clear (a lone ``if base_url:`` leaves stale values). + _is_custom_target = str(result.target_provider or "").strip().lower() == "custom" + if result.base_url: + model_cfg["base_url"] = result.base_url + elif _is_custom_target: + model_cfg.pop("base_url", None) + if _is_custom_target: + if result.api_mode: + model_cfg["api_mode"] = result.api_mode + else: + model_cfg.pop("api_mode", None) + else: + clear_model_endpoint_credentials(model_cfg, clear_base_url=True) + save_config(cfg) + except Exception as e: + logger.warning("Failed to persist model switch: %s", e) + + # Build confirmation message with full metadata. Display form shortens opaque Palantir + # IDs (ri.language-model-service..*) to their trailing slug. + provider_label = result.provider_label or result.target_provider + lines = [t("gateway.model.switched", model=format_model_for_display(result.new_model))] + lines.append(t("gateway.model.provider_label", provider=provider_label)) + + # Context: always resolve via the provider-aware chain so Codex OAuth, + # Copilot, and Nous-enforced caps win over the raw models.dev entry. + mi = result.model_info + _sw_config_ctx = None + _sw_model_cfg = {} + try: + _sw_model_cfg = _load_gateway_config().get("model", {}) + if isinstance(_sw_model_cfg, dict): + _sw_raw = _sw_model_cfg.get("context_length") + if _sw_raw is not None: + _sw_config_ctx = int(_sw_raw) + except Exception: + pass + if not isinstance(_sw_model_cfg, dict): + _sw_model_cfg = {} + ctx = await resolve_display_context_length_async( + result.new_model, + result.target_provider, + base_url=result.base_url or current_base_url or "", + api_key=result.api_key or current_api_key or "", + model_info=mi, + custom_providers=custom_provs, + config_context_length=_sw_config_ctx, + configured_model=_sw_model_cfg.get("default") or _sw_model_cfg.get("model"), + configured_provider=_sw_model_cfg.get("provider"), + configured_base_url=_sw_model_cfg.get("base_url"), + ) + if ctx: + lines.append(t("gateway.model.context_label", tokens=f"{ctx:,}")) + if mi: + if mi.max_output: + lines.append(t("gateway.model.max_output_label", tokens=f"{mi.max_output:,}")) + lines.append(t("gateway.model.capabilities_label", capabilities=mi.format_capabilities())) + + if not picker: + cache_enabled = ( + (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) + or result.api_mode == "anthropic_messages" + ) + if cache_enabled: + lines.append(t("gateway.model.prompt_caching_enabled")) + + if result.warning_message: + lines.append(t("gateway.model.warning_prefix", warning=result.warning_message)) + + if persist_global: + lines.append(t("gateway.model.saved_global")) + elif one_turn: + lines.append(" (next turn only — restores after one response)") + else: + lines.append(t("gateway.model.session_only_hint")) + return "\n".join(lines) + async def _handle_model_command(self, event: MessageEvent) -> Optional[str]: """Handle /model command — switch model.""" from gateway.run import _hermes_home, _load_gateway_config @@ -1761,247 +1983,34 @@ class GatewaySlashCommandsMixin: _chat_id: str, model_id: str, provider_slug: str ) -> str: """Perform the model switch and return confirmation text.""" - skew_error = _model_switch_skew_guard() - if skew_error: - return skew_error - # Offload the switch off the event loop — switch_model() can fall through to - # a synchronous models.dev HTTP fetch (requests.get, 15s timeout) on a - # cold/expired cache, which freezes the gateway otherwise. - result = await asyncio.to_thread( + result, error = await _self._perform_model_switch( _switch_model, raw_input=model_id, + explicit_provider=provider_slug, + session_key=_session_key, + source=event.source, + current_model=_cur_model, current_provider=_cur_provider, + current_base_url=_cur_base_url, + current_api_key=_cur_api_key, + persist_global=persist_global, + user_provs=user_provs, + custom_provs=custom_provs, + ) + if error is not None: + return error + return await _self._commit_model_switch( + result, + session_key=_session_key, + source=event.source, current_model=_cur_model, current_base_url=_cur_base_url, current_api_key=_cur_api_key, - is_global=persist_global, - explicit_provider=provider_slug, - user_providers=user_provs, - custom_providers=custom_provs, + custom_provs=custom_provs, + persist_global=persist_global, + config_path=config_path, + picker=True, ) - if not result.success: - return t("gateway.model.error_prefix", error=result.error_message) - - try: - from hermes_cli.context_switch_guard import ( - enrich_model_switch_warnings_for_gateway, - ) - - # Offload: merge_preflight_compression_warning() - # calls the sync resolve_display_context_length() - # provider probe ladder — must not run on the loop. - await asyncio.to_thread( - enrich_model_switch_warnings_for_gateway, - result, - _self, - session_key=_session_key, - source=event.source, - custom_providers=custom_provs, - load_gateway_config=_load_gateway_config, - ) - except Exception as exc: - logger.debug("preflight-compression switch warning failed: %s", exc) - - # Update cached agent in-place - cached_entry = None - _cache_lock = getattr(_self, "_agent_cache_lock", None) - _cache = getattr(_self, "_agent_cache", None) - if _cache_lock and _cache is not None: - with _cache_lock: - cached_entry = _cache.get(_session_key) - if cached_entry and cached_entry[0] is not None: - try: - cached_entry[0].switch_model( - new_model=result.new_model, - new_provider=result.target_provider, - api_key=result.api_key, - base_url=result.base_url, - api_mode=result.api_mode, - capabilities=getattr( - result, "runtime_capabilities", None - ), - ) - except Exception as exc: - # The in-place swap rolled the agent back to the OLD working - # model/client and re-raised. Abort the commit: do NOT persist the - # failed model, set a session override, or evict the working cached - # agent — otherwise the next message rebuilds a dead agent from the - # broken override and the conversation is lost. A failed switch is a no-op. - logger.warning( - "Picker model switch failed for cached agent: %s", exc - ) - return t( - "gateway.model.error_prefix", - error=( - f"Model switch to {result.new_model} failed ({exc}); " - f"staying on {_cur_model}." - ), - ) - - # Persist the new model to the session DB so the - # dashboard shows the updated model (#34850). - _sess_db = getattr(_self, "_session_db", None) - if _sess_db is not None: - try: - _sess_entry = await _self.async_session_store.get_or_create_session( - event.source - ) - await _sess_db.update_session_model( - _sess_entry.session_id, result.new_model, - provider=result.target_provider, - ) - except Exception as exc: - logger.debug( - "Failed to persist model switch to DB: %s", exc - ) - - # Store model note + session override. Use display form (strips opaque - # Palantir prefix) for the user- visible note; session-override map still - # gets the full opaque ID, which is what the wire needs. - from hermes_cli.model_switch import format_model_for_display - _display_cur = format_model_for_display(_cur_model) - _display_new = format_model_for_display(result.new_model) - if not hasattr(_self, "_pending_model_notes"): - _self._pending_model_notes = {} - _self._pending_model_notes[_session_key] = ( - f"[Note: model was just switched from {_display_cur} to {_display_new} " - f"via {result.provider_label or result.target_provider}. " - f"Adjust your self-identification accordingly.]" - ) - _self._session_model_overrides[_session_key] = { - "model": result.new_model, - "provider": result.target_provider, - "api_key": result.api_key, - "base_url": result.base_url, - "api_mode": result.api_mode, - "request_overrides": dict(result.request_overrides or {}), - "capabilities": dict(result.runtime_capabilities or {}), - } - - # Write-through the non-secret parts to the session - # store so the picked model survives a gateway restart - # (api_key is never persisted). - try: - await _self.async_session_store.set_model_override( - _session_key, - _self._session_model_overrides[_session_key], - ) - except Exception: - logger.debug( - "Failed to persist session model override", - exc_info=True, - ) - - # Evict cached agent so the next turn creates a fresh - # agent from the override rather than relying on the - # stale cache signature to trigger a rebuild. - _self._evict_cached_agent(_session_key) - - # Persist to config (default) unless --session opted out, - # mirroring the text /model command path above so a picked - # model survives across sessions like a typed one (#49066). - if persist_global: - try: - # Write-back round-trip: raw read is correct - # (merged defaults must not be persisted). - from hermes_cli.config import read_user_config_raw - _persist_cfg = read_user_config_raw(config_path) - _raw_model = _persist_cfg.get("model") - if isinstance(_raw_model, dict): - _persist_model_cfg = _raw_model - elif isinstance(_raw_model, str) and _raw_model.strip(): - _persist_model_cfg = {"default": _raw_model.strip()} - _persist_cfg["model"] = _persist_model_cfg - else: - _persist_model_cfg = {} - _persist_cfg["model"] = _persist_model_cfg - try: - from hermes_cli.route_identity import should_clear_context_pin_async - - if await should_clear_context_pin_async( - _persist_model_cfg.get("default") - or _persist_model_cfg.get("model"), - result.new_model, - _persist_model_cfg.get("base_url"), - result.base_url, - _persist_model_cfg.get("provider"), - result.target_provider, - ): - _persist_model_cfg.pop("context_length", None) - except Exception: - _persist_model_cfg.pop("context_length", None) - _persist_model_cfg["default"] = result.new_model - _persist_model_cfg["provider"] = result.target_provider - # Named providers always resolve base_url/api_mode fresh, so any - # leftover is cleared unconditionally below. Custom providers have no - # registry entry to re-derive from, so they need an explicit - # set-or-clear here (a lone `if base_url:` left stale values behind). - _is_custom_target = str(result.target_provider or "").strip().lower() == "custom" - if result.base_url: - _persist_model_cfg["base_url"] = result.base_url - elif _is_custom_target: - _persist_model_cfg.pop("base_url", None) - if _is_custom_target: - if result.api_mode: - _persist_model_cfg["api_mode"] = result.api_mode - else: - _persist_model_cfg.pop("api_mode", None) - else: - clear_model_endpoint_credentials(_persist_model_cfg, clear_base_url=True) - from hermes_cli.config import save_config - save_config(_persist_cfg) - except Exception as e: - logger.warning("Failed to persist model switch: %s", e) - - # Build confirmation text. Use display form so opaque - # Palantir IDs (ri.language-model-service..*) get - # shortened to their trailing slug for the UI. - plabel = result.provider_label or result.target_provider - lines = [t("gateway.model.switched", model=format_model_for_display(result.new_model))] - lines.append(t("gateway.model.provider_label", provider=plabel)) - mi = result.model_info - from hermes_cli.model_switch import resolve_display_context_length_async - _sw_config_ctx = None - _sw_model_cfg = {} - try: - _sw_cfg = _load_gateway_config() - _sw_model_cfg = _sw_cfg.get("model", {}) - if isinstance(_sw_model_cfg, dict): - _sw_raw = _sw_model_cfg.get("context_length") - if _sw_raw is not None: - _sw_config_ctx = int(_sw_raw) - except Exception: - pass - if not isinstance(_sw_model_cfg, dict): - _sw_model_cfg = {} - ctx = await resolve_display_context_length_async( - result.new_model, - result.target_provider, - base_url=result.base_url or current_base_url or "", - api_key=result.api_key or current_api_key or "", - model_info=mi, - custom_providers=custom_provs, - config_context_length=_sw_config_ctx, - configured_model=( - _sw_model_cfg.get("default") - or _sw_model_cfg.get("model") - ), - configured_provider=_sw_model_cfg.get("provider"), - configured_base_url=_sw_model_cfg.get("base_url"), - ) - if ctx: - lines.append(t("gateway.model.context_label", tokens=f"{ctx:,}")) - if mi: - if mi.max_output: - lines.append(t("gateway.model.max_output_label", tokens=f"{mi.max_output:,}")) - lines.append(t("gateway.model.capabilities_label", capabilities=mi.format_capabilities())) - if result.warning_message: - lines.append(t("gateway.model.warning_prefix", warning=result.warning_message)) - if persist_global: - lines.append(t("gateway.model.saved_global")) - else: - lines.append(t("gateway.model.session_only_hint")) - return "\n".join(lines) async def _on_model_selected( _chat_id: str, model_id: str, provider_slug: str @@ -2066,277 +2075,42 @@ class GatewaySlashCommandsMixin: return "\n".join(lines) # Perform the switch - skew_error = _model_switch_skew_guard() - if skew_error: - return skew_error - # Offload the switch off the event loop — switch_model() can fall through to a synchronous - # models.dev HTTP fetch (requests.get, 15s timeout) on a cold/expired cache, which freezes - # the gateway otherwise. - result = await asyncio.to_thread( + result, error = await self._perform_model_switch( _switch_model, raw_input=model_input, - current_provider=current_provider, + explicit_provider=explicit_provider, + session_key=session_key, + source=source, current_model=current_model, + current_provider=current_provider, current_base_url=current_base_url, current_api_key=current_api_key, - is_global=persist_global, - explicit_provider=explicit_provider, - user_providers=user_provs, - custom_providers=custom_provs, + persist_global=persist_global, + user_provs=user_provs, + custom_provs=custom_provs, ) - - if not result.success: - return t("gateway.model.error_prefix", error=result.error_message) - - try: - from hermes_cli.context_switch_guard import ( - enrich_model_switch_warnings_for_gateway, - ) - - # Offload: merge_preflight_compression_warning() calls the sync - # resolve_display_context_length() provider probe ladder — must - # not run on the loop. - await asyncio.to_thread( - enrich_model_switch_warnings_for_gateway, - result, - self, - session_key=session_key, - source=source, - custom_providers=custom_provs, - load_gateway_config=_load_gateway_config, - ) - except Exception as exc: - logger.debug("preflight-compression switch warning failed: %s", exc) + if error is not None: + return error async def _finish_switch() -> str: """Apply the resolved switch (agent, session, config) and build the reply.""" - # If there's a cached agent, update it in-place - cached_entry = None - _cache_lock = getattr(self, "_agent_cache_lock", None) - _cache = getattr(self, "_agent_cache", None) - if _cache_lock and _cache is not None: - with _cache_lock: - cached_entry = _cache.get(session_key) - - if cached_entry and cached_entry[0] is not None: - try: - cached_entry[0].switch_model( - new_model=result.new_model, - new_provider=result.target_provider, - api_key=result.api_key, - base_url=result.base_url, - api_mode=result.api_mode, - capabilities=getattr(result, "runtime_capabilities", None), - ) - except Exception as exc: - # In-place swap rolled the agent back to the OLD working model/client and re- - # raised. Abort the commit (skip DB persist, session override, cache eviction, - # config write) so a failed switch is a no-op. Without this early return the next - # message rebuilds a broken agent from the override. - logger.warning("In-place model switch failed for cached agent: %s", exc) - return t( - "gateway.model.error_prefix", - error=( - f"Model switch to {result.new_model} failed ({exc}); " - f"staying on {current_model}." - ), - ) - - # Persist the new model to the session DB so the dashboard - # shows the updated model (#34850). - _sess_db = getattr(self, "_session_db", None) - if _sess_db is not None: - try: - _sess_entry = await self.async_session_store.get_or_create_session(source) - # If this session was auto-reset, consume the flag so the - # next regular message's cleanup does not wipe the model - # override just stored below (Closes #48031). - if getattr(_sess_entry, "was_auto_reset", False): - _sess_entry.was_auto_reset = False - await _sess_db.update_session_model( - _sess_entry.session_id, result.new_model, - provider=result.target_provider, - ) - except Exception as exc: - logger.debug( - "Failed to persist model switch to DB: %s", exc - ) - - # Store a note to prepend to the next user message so the model knows about the switch - # (avoids system messages mid-history). Display form strips opaque Palantir RID - # prefixes; the override map below keeps the full ID for the wire. - from hermes_cli.model_switch import format_model_for_display - if not hasattr(self, "_pending_model_notes"): - self._pending_model_notes = {} - self._pending_model_notes[session_key] = ( - f"[Note: model was just switched from {format_model_for_display(current_model)} to {format_model_for_display(result.new_model)} " - f"via {result.provider_label or result.target_provider}. " - f"{'This override applies to the next turn only. ' if one_turn else ''}" - f"Adjust your self-identification accordingly.]" + return await self._commit_model_switch( + result, + session_key=session_key, + source=source, + current_model=current_model, + current_base_url=current_base_url, + current_api_key=current_api_key, + custom_provs=custom_provs, + persist_global=persist_global, + config_path=config_path, + one_turn=one_turn, + restore_snapshot=restore_snapshot, ) - # Store session override so next agent creation uses the new model - self._session_model_overrides[session_key] = { - "model": result.new_model, - "provider": result.target_provider, - "api_key": result.api_key, - "base_url": result.base_url, - "api_mode": result.api_mode, - "request_overrides": dict(result.request_overrides or {}), - "capabilities": dict(result.runtime_capabilities or {}), - } - if one_turn: - if not hasattr(self, "_pending_one_turn_model_restores"): - self._pending_one_turn_model_restores = {} - self._pending_one_turn_model_restores[session_key] = ( - restore_snapshot or {"had_override": False, "override": None} - ) - elif hasattr(self, "_pending_one_turn_model_restores"): - self._pending_one_turn_model_restores.pop(session_key, None) - - # Write-through the non-secret parts (model/provider/base_url) to the session store so - # the override survives a gateway restart. api_key/api_mode are never persisted — they - # are re-resolved via runtime provider resolution on rehydration. /model --once is - # deliberately EXCLUDED: a one-turn override must never survive a restart, so the - # persisted value stays at the pre-once state the finally-restore reverts to. - if not one_turn: - try: - await self.async_session_store.set_model_override( - session_key, - self._session_model_overrides[session_key], - ) - except Exception: - logger.debug( - "Failed to persist session model override", exc_info=True - ) - - # Evict cached agent so the next turn creates a fresh agent from the - # override rather than relying on cache signature mismatch detection. - self._evict_cached_agent(session_key) - - # Persist to config (default) unless --session opted out - if persist_global: - try: - # Write-back round-trip: raw read is correct (merged - # defaults must not be persisted back to the user's file). - from hermes_cli.config import read_user_config_raw - cfg = read_user_config_raw(config_path) - # Coerce scalar/None ``model:`` into a dict before mutation — otherwise - # ``cfg.setdefault("model", {})`` returns the existing scalar and the next - # assignment raises ``TypeError: 'str' object does not support item - # assignment``. - raw_model = cfg.get("model") - if isinstance(raw_model, dict): - model_cfg = raw_model - elif isinstance(raw_model, str) and raw_model.strip(): - model_cfg = {"default": raw_model.strip()} - cfg["model"] = model_cfg - else: - model_cfg = {} - cfg["model"] = model_cfg - try: - from hermes_cli.route_identity import should_clear_context_pin_async - - if await should_clear_context_pin_async( - model_cfg.get("default") or model_cfg.get("model"), - result.new_model, - model_cfg.get("base_url"), - result.base_url, - model_cfg.get("provider"), - result.target_provider, - ): - model_cfg.pop("context_length", None) - except Exception: - model_cfg.pop("context_length", None) - model_cfg["default"] = result.new_model - model_cfg["provider"] = result.target_provider - # See the picker handler above for why custom providers need an - # explicit set-or-clear instead of the old lone truthy check (#25107). - _is_custom_target = str(result.target_provider or "").strip().lower() == "custom" - if result.base_url: - model_cfg["base_url"] = result.base_url - elif _is_custom_target: - model_cfg.pop("base_url", None) - if _is_custom_target: - if result.api_mode: - model_cfg["api_mode"] = result.api_mode - else: - model_cfg.pop("api_mode", None) - else: - clear_model_endpoint_credentials(model_cfg, clear_base_url=True) - from hermes_cli.config import save_config - save_config(cfg) - except Exception as e: - logger.warning("Failed to persist model switch: %s", e) - - # Build confirmation message with full metadata - provider_label = result.provider_label or result.target_provider - lines = [t("gateway.model.switched", model=format_model_for_display(result.new_model))] - lines.append(t("gateway.model.provider_label", provider=provider_label)) - - # Context: always resolve via the provider-aware chain so Codex OAuth, - # Copilot, and Nous-enforced caps win over the raw models.dev entry. - mi = result.model_info - from hermes_cli.model_switch import resolve_display_context_length_async - _sw2_config_ctx = None - _sw2_model_cfg = {} - try: - _sw2_cfg = _load_gateway_config() - _sw2_model_cfg = _sw2_cfg.get("model", {}) - if isinstance(_sw2_model_cfg, dict): - _sw2_raw = _sw2_model_cfg.get("context_length") - if _sw2_raw is not None: - _sw2_config_ctx = int(_sw2_raw) - except Exception: - pass - if not isinstance(_sw2_model_cfg, dict): - _sw2_model_cfg = {} - ctx = await resolve_display_context_length_async( - result.new_model, - result.target_provider, - base_url=result.base_url or current_base_url or "", - api_key=result.api_key or current_api_key or "", - model_info=mi, - custom_providers=custom_provs, - config_context_length=_sw2_config_ctx, - configured_model=( - _sw2_model_cfg.get("default") - or _sw2_model_cfg.get("model") - ), - configured_provider=_sw2_model_cfg.get("provider"), - configured_base_url=_sw2_model_cfg.get("base_url"), - ) - if ctx: - lines.append(t("gateway.model.context_label", tokens=f"{ctx:,}")) - if mi: - if mi.max_output: - lines.append(t("gateway.model.max_output_label", tokens=f"{mi.max_output:,}")) - lines.append(t("gateway.model.capabilities_label", capabilities=mi.format_capabilities())) - - # Cache notice - cache_enabled = ( - (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) - or result.api_mode == "anthropic_messages" - ) - if cache_enabled: - lines.append(t("gateway.model.prompt_caching_enabled")) - - if result.warning_message: - lines.append(t("gateway.model.warning_prefix", warning=result.warning_message)) - - if persist_global: - lines.append(t("gateway.model.saved_global")) - elif one_turn: - lines.append(" (next turn only — restores after one response)") - else: - lines.append(t("gateway.model.session_only_hint")) - - return "\n".join(lines) - - # Selection-guard confirmation gate (typed /model path). The pickers already confirm - # via their own UI; this covers the direct text command, which previously bypassed the - # guard. Runs the unified registry (cost + data-policy + future guards). Pricing lookups may - # hit models.dev or a /models endpoint on a cache miss, so run it off the event loop. + # Selection-guard confirmation for the typed /model path (pickers confirm via their own + # UI). Runs the unified registry (cost + data-policy guards); pricing lookups may hit + # models.dev or a /models endpoint on a cache miss, so run it off the event loop. _cost_warning = None try: from hermes_cli.model_selection_guards import combined_selection_warning @@ -2420,8 +2194,7 @@ class GatewaySlashCommandsMixin: async def _handle_personality_command(self, event: MessageEvent) -> str: """Handle /personality command - list or set a personality. - All resolution/persistence goes through hermes_cli.personality — - the single owner of personality state on every surface. + All resolution/persistence goes through hermes_cli.personality, the single owner of state. """ from gateway.run import _load_gateway_config from hermes_cli.personality import ( @@ -2462,10 +2235,9 @@ class GatewaySlashCommandsMixin: available = "`none`, " + ", ".join(f"`{n}`" for n in personalities) return t("gateway.personality.unknown", name=args.lower(), available=available) - # Persist the selection only — hermes_cli.personality never writes agent.system_prompt - # (user-owned manual overlay). persist_personality writes get_hermes_home()/config.yaml, - # i.e. the routed profile under multiplex; the next turn re-resolves the prompt from that - # file (_get_system_prompt_for_channel), so no process-global state to update. + # Persist the selection only — hermes_cli.personality never writes agent.system_prompt (user- + # owned overlay). persist_personality writes get_hermes_home()/config.yaml (the routed profile + # under multiplex) and the next turn re-resolves the prompt from it: no process-global state. if not persist_personality(name): return t("gateway.personality.save_failed", error="config write failed") @@ -2560,9 +2332,8 @@ class GatewaySlashCommandsMixin: async def _handle_goal_command(self, event: "MessageEvent") -> str: """Handle /goal for gateway platforms. - Subcommands: ``/goal`` / ``/goal status`` / ``/goal pause`` / ``/goal resume`` / ``/goal - clear``. Setting a new goal queues the goal text as the next turn so the agent starts - working on it immediately — the post-turn continuation hook then takes over from there. + Subcommands: status / pause / resume / clear. Setting a new goal queues the goal text as the + next turn so the agent starts immediately; the post-turn continuation hook takes over after. """ args = (event.get_command_args() or "").strip() lower = args.lower() @@ -2707,7 +2478,6 @@ class GatewaySlashCommandsMixin: if not objective: return "Usage: /goal draft " try: - import asyncio from hermes_cli.goals import draft_contract # _run_in_executor_with_context, not a bare hop: drafting a contract calls the @@ -2765,9 +2535,8 @@ class GatewaySlashCommandsMixin: async def _handle_heartbeat_command(self, event: "MessageEvent") -> str: """Handle /heartbeat for gateway platforms (mirror of CLI handler). - Sets/manages the session's one recurring re-entry prompt. The - gateway-wide poller injects due heartbeats through the adapter FIFO - as ordinary user turns, so alternation and caching are untouched. + Manages the session's one recurring re-entry prompt. The gateway-wide poller injects due + heartbeats through the adapter FIFO as ordinary user turns, so alternation and caching hold. """ from hermes_cli.heartbeat import parse_interval, format_interval, MIN_INTERVAL_SECONDS @@ -2837,9 +2606,8 @@ class GatewaySlashCommandsMixin: async def _handle_refine_command(self, event: "MessageEvent") -> str: """Handle /refine — run the memory/skill review fork on demand. - Uses the session's cached AIAgent (idle agents live in ``_agent_cache``). The review runs - in a daemon thread against a snapshot of the conversation; the live session and prompt - cache are untouched. Requires the session to have at least one completed turn. + Runs in a daemon thread against a snapshot of the cached AIAgent's conversation; the live + session and prompt cache are untouched. Requires at least one completed turn. """ args = (event.get_command_args() or "").strip() quick_key = self._session_key_for_source(event.source) if event.source else None @@ -2880,9 +2648,8 @@ class GatewaySlashCommandsMixin: async def _handle_review_command(self, event: "MessageEvent") -> str: """Handle /review — spawn an independent reviewer subagent. - The approval session-key contextvar is only bound during agent turns, so it is bound - explicitly here — without it the completion event would carry no gateway route and never - re-enter this chat. + The approval session-key contextvar is only bound during agent turns, so bind it explicitly + here or the completion event carries no gateway route and never re-enters this chat. """ args = (event.get_command_args() or "").strip() quick_key = self._session_key_for_source(event.source) if event.source else None @@ -2983,9 +2750,8 @@ class GatewaySlashCommandsMixin: async def _get_loop_manager_for_event(self, event: "MessageEvent"): """Return a LoopManager bound to the session for this gateway event. - Returns ``(manager, session_entry)`` or ``(None, None)`` when the - loops module or session can't be loaded. Mirrors - ``_get_goal_manager_for_event``. + Returns ``(manager, session_entry)``, or ``(None, None)`` when the loops module or session + can't be loaded. Mirrors ``_get_goal_manager_for_event``. """ try: from hermes_cli.loops import LoopManager @@ -3008,9 +2774,8 @@ class GatewaySlashCommandsMixin: async def _handle_loop_command(self, event: "MessageEvent") -> str: """Handle /loop for gateway platforms — recurring in-session wakeups. - Mirrors the CLI handler via the shared ``dispatch_loop_command``. New loops capture the - event's routing (platform/chat/thread) so the gateway's idle loop-wakeup watcher can - inject ticks back into this chat even after a restart. + Mirrors the CLI handler via ``dispatch_loop_command``. New loops capture the event's routing + (platform/chat/thread) so the idle loop-wakeup watcher can inject ticks here after a restart. """ try: from hermes_cli.loops import dispatch_loop_command, goal_blocks_loop_tick @@ -3054,12 +2819,9 @@ class GatewaySlashCommandsMixin: return output async def _handle_undo_command(self, event: MessageEvent) -> str: - """Handle /undo [N] — back up N user turns (default 1), soft-deleting the truncated rows on - disk and echoing the backed-up message text so the user can copy/edit and resend. - - The cached agent is evicted so the next message rebuilds context from the truncated - (active-only) transcript — the gateway's equivalent of the CLI's in-place history surgery - + memory-cache invalidation. + """Handle /undo [N] — back up N user turns (default 1), soft-deleting the truncated rows and + echoing the backed-up text. Evicts the cached agent so the next message rebuilds context + from the active-only transcript (gateway analogue of the CLI's history surgery). """ source = event.source @@ -3339,9 +3101,7 @@ class GatewaySlashCommandsMixin: async def _handle_diff_command(self, event: MessageEvent) -> str: """Handle /diff — show git changes in the working directory. - The diff body is truncated hard here (messaging surfaces are not a pager); platform - senders additionally split/clamp long messages to per-platform limits, the same way tool- - progress output is truncated in three layers before delivery. + Diff body is truncated hard here (chat is not a pager); platform senders clamp further. """ args = event.get_command_args().strip() @@ -3443,11 +3203,8 @@ class GatewaySlashCommandsMixin: return f"```diff\n{diff}{note}\n```" async def _handle_background_command(self, event: MessageEvent) -> str: - """Handle /bg — run a prompt in a separate background session. - - Spawns a new AIAgent in a background thread with its own session. - When it completes, sends the result back to the same chat without - modifying the active session's conversation history. + """Handle /bg — run a prompt in a background thread with its own session; the + result is sent to the same chat without touching the active session's history. """ prompt = event.get_command_args().strip() if not prompt: @@ -3480,12 +3237,9 @@ class GatewaySlashCommandsMixin: return t("gateway.background.started", preview=preview, task_id=task_id) async def _handle_btw_command(self, event: MessageEvent) -> str: - """Handle /btw — answer a side question about this conversation. - - Snapshots the session transcript and answers the question with a one-shot auxiliary LLM - call (main model by default) — the live session's history is never touched, so role - alternation and the prompt cache stay intact and the current turn keeps running. - Deliberately different from /bg, which spawns a fresh contextless agent session. + """Handle /btw — answer a side question via a one-shot auxiliary LLM call on a + transcript snapshot; live history is never touched (alternation + prompt cache intact, + current turn keeps running). Unlike /bg, which spawns a fresh contextless session. """ question = event.get_command_args().strip() if not question: @@ -3514,11 +3268,9 @@ class GatewaySlashCommandsMixin: "api_mode": runtime_kwargs.get("api_mode"), } history_snapshot = list(history) - # Prefer the cache-parity fork when this chat has a live cached AIAgent: the fork replays - # the snapshot against the warm provider prefix cache (same mechanism as the background - # self-improvement review), giving the side answer FULL conversation context at cache-read - # prices. With no cached agent the provider cache is cold anyway — answer_side_question's - # one-shot digest fallback handles it. + # Prefer the cache-parity fork when a live cached AIAgent exists: it replays the snapshot + # against the warm provider prefix cache, giving FULL context at cache-read prices. With no + # cached agent the cache is cold anyway — answer_side_question's digest fallback handles it. parent_agent = None try: session_key = self._session_key_for_source(source) @@ -3600,8 +3352,7 @@ class GatewaySlashCommandsMixin: ) -> str: """Apply a /reasoning argument (typed or picked) and return the reply. - Single application path shared by the typed `/reasoning ` branch and the interactive - choice picker, so both surfaces stay in lockstep with the canonical parser. + Single path shared by `/reasoning ` and the choice picker so both match the parser. """ from hermes_constants import parse_reasoning_effort @@ -3685,9 +3436,8 @@ class GatewaySlashCommandsMixin: ) -> bool: """Send an interactive choice picker when the platform supports it. - Mirrors the `/model` picker gate: the capability is detected on the - adapter *type* (``send_choice_picker``), and a failed send falls back - to the text path (returns False) instead of erroring the command. + Mirrors the `/model` gate: capability is detected on the adapter *type* + (``send_choice_picker``); a failed send returns False (text fallback) instead of erroring. """ adapter = getattr(self, "_adapter_for_source")(event.source) has_picker = ( @@ -3800,8 +3550,7 @@ class GatewaySlashCommandsMixin: async def _handle_memory_command(self, event: MessageEvent) -> str: """Handle /memory — review pending memory writes + toggle the approval gate. - Memory entries are small enough to review inline in a chat bubble, so the full - pending/approve/reject/approval flow works on every platform. + Entries are small enough to review inline, so the full flow works on every platform. """ from gateway.run import _gateway_config_home from hermes_cli.write_approval_commands import handle_pending_subcommand @@ -3837,13 +3586,10 @@ class GatewaySlashCommandsMixin: return out async def _handle_skills_command(self, event: MessageEvent) -> str: - """Handle /skills on the gateway — pending skill-write review only. + """Handle /skills on the gateway — pending skill-write review only (hub stays CLI-only). - The full skills hub (search/browse/install) stays CLI-only; this handler covers the - write-approval review surface (pending / approve / reject / diff / approval) so a skill - staged from a gateway session can be reviewed from that same session. Gated by - ``skills.write_approval``, but still answers when staged writes exist after the gate was - turned off (so they are never stranded). ``diff`` output is truncated for chat bubbles. + Gated by ``skills.write_approval`` but still answers when staged writes exist after the + gate is off (never stranded). ``diff`` is truncated for chat. """ from gateway.run import _gateway_config_home from hermes_cli.write_approval_commands import handle_pending_subcommand @@ -3879,10 +3625,8 @@ class GatewaySlashCommandsMixin: "approve , reject , diff , approval . " "(Search/install are CLI-only.)") - # Chat bubbles can't hold a full skill diff — truncate and point at - # the real review surface. (Note: `hermes skills diff ` is a - # *different* command — it diffs a bundled skill against its stock - # version — so we point at the pending JSON file, not that command.) + # Chat bubbles can't hold a full skill diff — truncate and point at the pending JSON file + # (NOT `hermes skills diff `, which diffs a bundled skill against its stock version). if args and args[0].lower() == "diff" and len(out) > 3000: pending_id = args[1] if len(args) > 1 else "" out = (out[:3000] @@ -3893,8 +3637,7 @@ class GatewaySlashCommandsMixin: async def _handle_fast_command(self, event: MessageEvent) -> Optional[str]: """Handle /fast — mirror the CLI Priority Processing toggle in gateway chats. - Session-scoped by default; ``--global`` persists agent.service_tier - to config.yaml (parity with /model and /reasoning). + Session-scoped by default; ``--global`` persists agent.service_tier (parity with /model). """ from gateway.run import _load_gateway_config, _resolve_gateway_model from hermes_cli.models import model_supports_fast_mode @@ -4026,10 +3769,8 @@ class GatewaySlashCommandsMixin: async def _handle_verbose_command(self, event: MessageEvent) -> str: """Handle /verbose command — cycle tool progress display mode. - Gated by ``display.tool_progress_command`` in config.yaml (default off). Cycles the mode - off → new → all → verbose → off for the *current platform*. The setting is saved to - ``display.platforms..tool_progress`` so each channel can have its own - verbosity level independently. + Gated by ``display.tool_progress_command`` (default off). Cycles off → new → all → verbose + per *current platform*, saved to ``display.platforms..tool_progress``. """ from gateway.run import _gateway_config_home, _load_gateway_config, _platform_config_key @@ -4121,10 +3862,8 @@ class GatewaySlashCommandsMixin: ) else: self._busy_input_mode = arg - # busy_input_mode is the source of truth for the text mode too - # (run.py:_load_busy_text_mode) — re-derive it so the adapter refresh below doesn't - # read a stale value and keep interrupting after e.g. /busy queue (config IS saved; - # only the live session lagged until restart). + # busy_input_mode is also the source of truth for the text mode — re-derive it so the + # adapter refresh below doesn't keep a stale value and keep interrupting. self._busy_text_mode = self._load_busy_text_mode() adapter = self._adapter_for_source(event.source) @@ -4223,11 +3962,9 @@ class GatewaySlashCommandsMixin: async def _handle_compress_command(self, event: MessageEvent) -> str: """Profile-scoping wrapper around manual /compress. - Multiplexed gateways resolve credentials through the fail-closed per-profile secret scope - (``agent.secret_scope``, Workstream A). Agent turns install it via ``_run_agent``'s - wrapper but slash dispatch does not, so an unscoped /compress died with - ``UnscopedSecretError``. Install the source profile's scope around the whole handler, - mirroring ``_run_agent``. Single-profile gateways skip this — zero behavior change. + Multiplexed gateways resolve credentials through the fail-closed per-profile secret scope; + slash dispatch (unlike ``_run_agent``) does not install it, so an unscoped /compress would + raise ``UnscopedSecretError``. Single-profile gateways skip this. """ if not getattr(getattr(self, "config", None), "multiplex_profiles", False): return await self._handle_compress_command_inner(event) @@ -4243,11 +3980,9 @@ class GatewaySlashCommandsMixin: ) -> str: """Manual /compress for codex_app_server sessions. - Compacts the LIVE cached agent's app-server thread via ``thread/compact/start`` with - ``force=True`` (bypasses the ``codex_app_server_auto`` mode gate — a manual /compress is an - explicit user decision) and keeps that agent cached so the next turn continues from it. - Never builds a temporary compression agent and never rewrites the transcript mirror: - neither can shrink the server-side thread that is the model's real context. + Compacts the LIVE cached agent's app-server thread (``thread/compact/start``, ``force=True`` + bypasses the ``codex_app_server_auto`` gate) and keeps the agent cached. Never builds a + temporary agent or rewrites the mirror: neither can shrink the server-side thread. """ agent = None lock = getattr(self, "_agent_cache_lock", None) @@ -4299,9 +4034,7 @@ class GatewaySlashCommandsMixin: async def _handle_compress_command_inner(self, event: MessageEvent) -> str: """Handle /compress command -- manually compress conversation context. - Accepts an optional focus topic: ``/compress `` guides the summariser to preserve - information related to *focus* while being more aggressive about discarding everything - else. + Optional ``/compress `` tells the summariser what to preserve, discarding the rest. """ source = event.source session_entry = await self.async_session_store.get_or_create_session(source) @@ -4360,10 +4093,8 @@ class GatewaySlashCommandsMixin: from agent.model_metadata import estimate_request_tokens_rough session_key = self._session_key_for_source(source) - # Preserve the same platform + stable gateway session identity that a normal gateway - # turn passes (gateway/run.py main turn), so external context engines bind this - # temporary compression agent to the original platform conversation instead of falling - # back to an unbound/default "cli" host source — see #50422. + # Preserve the platform + stable gateway session identity of a normal turn so external + # context engines bind this agent to the original conversation, not a default "cli" host. from gateway.run import ( _GATEWAY_HYGIENE_PLATFORM, _platform_config_key, @@ -4377,22 +4108,17 @@ class GatewaySlashCommandsMixin: session_key=session_key, ) if str(runtime_kwargs.get("api_mode") or "").lower() == "codex_app_server": - # codex app-server runtime: the model's working context is the app-server's server- - # side thread, owned by the LIVE cached agent (agent/codex_runtime.py — one - # CodexAppServerSession per AIAgent, spawned lazily on first turn). A temporary - # compression agent has no thread (codex route bails, and the finally-eviction below - # would destroy the only real context), so compact the live thread and KEEP the agent - # cached. No transcript fallback: rewriting the mirror cannot shrink the thread. + # codex app-server: the model's context is the server-side thread owned by the LIVE + # cached agent; a temporary agent has none (and finally-eviction would destroy the + # real context). Compact the live thread and KEEP the agent cached; no mirror fallback. return await self._compress_codex_app_server_session( session_key, session_entry.session_id ) if not runtime_kwargs.get("api_key"): return t("gateway.compress.no_provider") - # Pass the FULL transcript (tool results included) — same rationale as the session- - # hygiene auto-compress in gateway/run.py: filtering to user/assistant-only starves the - # compressor's tool-result pruning and can trip the protect-first/last early-return on - # short filtered histories. + # Pass the FULL transcript (tool results included), like auto-compress: user/assistant- + # only starves tool-result pruning and can trip the protect-first/last early-return. msgs = [ m for m in history if m.get("role") in {"user", "assistant", "tool"} @@ -4410,20 +4136,16 @@ class GatewaySlashCommandsMixin: partial = False head = msgs - # Bind the temporary compression agent to the originating source's platform + stable - # gateway session key. These are authoritative identity invariants, so assign directly - # (not setdefault — a resolver-supplied value would be a stale placeholder and must not - # win; assigning also avoids a duplicate-kwarg TypeError). platform is only set when - # known so AIAgent's default (None -> "cli") still applies. _resolve_session_agent_runtime - # does not set either key today, so in practice this just adds them. + # Bind the temporary compression agent to the source's platform + stable gateway session + # key. Assign directly (not setdefault: a resolver value would be a stale placeholder, + # and it avoids duplicate-kwarg TypeError); platform only when known so None -> "cli" holds. if platform_key is not None: runtime_kwargs["platform"] = platform_key runtime_kwargs["gateway_session_key"] = session_key - # The manual compression helper runs outside the live session's fully initialized prompt - # environment (it loads the memory provider only when compression.checkpoint_required - # demands it), and _compress_context may persist its cached system prompt. Restore the - # exact live-session prompt so provider blocks are retained. + # The manual compression helper runs outside the live session's fully initialized + # prompt environment and _compress_context may persist its cached system prompt — + # restore the exact live-session prompt so provider blocks are retained. session_row = None get_session = getattr(self._session_db, "get_session", None) if callable(get_session): @@ -4486,10 +4208,9 @@ class GatewaySlashCommandsMixin: if not compressor.has_content_to_compress(head): return t("gateway.compress.nothing_to_do") - # _run_in_executor_with_context (not a bare run_in_executor): the profile secret - # scope installed by the wrapper is a contextvar, and the default-executor hop would - # drop it — the compressor's aux-client provider resolution would then read - # credentials unscoped and fail closed under multiplexing. + # Not a bare run_in_executor: the profile secret scope is a contextvar and the + # default-executor hop would drop it, making the compressor's aux-client credential + # resolution fail closed under multiplexing. compressed, _ = await self._run_in_executor_with_context( lambda: tmp_agent._compress_context( head, @@ -4514,25 +4235,20 @@ class GatewaySlashCommandsMixin: if partial and tail: compressed = rejoin_compressed_head_and_tail(compressed, tail) - # _compress_context either rotated (legacy: ended the old session, created a - # continuation id — write compressed messages into the NEW session so the original - # stays searchable) or compacted in place (compression.in_place / #38763: same id, - # transcript replaced with the compacted set). + # _compress_context either rotated (new continuation id — write compressed messages + # into the NEW session so the original stays searchable) or compacted in place + # (compression.in_place: same id, transcript replaced). new_session_id = tmp_agent.session_id rotated = new_session_id != session_entry.session_id _in_place = bool(getattr(tmp_agent, "_last_compaction_in_place", False)) - # Persist the compressed transcript BEFORE repointing the live session onto the new - # session_id. Order matters: repoint first + failed DB write (lock contention, ENOSPC, - # IO error) would leave the entry on an empty session while reporting success — the - # conversation silently vanishes. Write first and treat failure as fatal so old - # history stays reachable and the outer handler can surface a "compress failed". - # - # Only rewrite when rotation produced a NEW session id. In-place compaction already - # archived + inserted rows inside _compress_context(); rewrite_transcript() would call - # replace_messages(active_only=False) and DELETE the archived turns (silent data loss). - # Unchanged id without in-place means _compress_context FAILED to rotate — a rewrite - # there would replace the original messages with only the summary (permanent loss). + # Persist the compressed transcript BEFORE repointing the live session: repoint first + # + failed DB write would leave the entry on an empty session while reporting success. + # Write first, treat failure as fatal so old history stays reachable. + # Only rewrite when rotation produced a NEW id: in-place compaction already archived + + # inserted rows and rewrite_transcript() (active_only=False) would DELETE the archived + # turns; an unchanged id without in-place means rotation FAILED and a rewrite would + # leave only the summary. if rotated: if not await self.async_session_store.rewrite_transcript( new_session_id, compressed @@ -4576,10 +4292,8 @@ class GatewaySlashCommandsMixin: new_tokens, compression_state=compressor, ) - # Detect summary-generation failure so we can surface a visible warning to the user - # even on the manual /compress path (otherwise the failure is silently logged). - # _last_compress_aborted = aux LLM returned no usable summary and messages were kept - # unchanged. force=True above bypasses any active cooldown. + # Surface summary-generation failure on the manual path (_last_compress_aborted = + # no usable summary, messages unchanged). force=True above bypasses any cooldown. _summary_aborted = bool(getattr(compressor, "_last_compress_aborted", False)) _summary_err = getattr(compressor, "_last_summary_error", None) # Force-redact provider exception text at this UI boundary @@ -5127,10 +4841,8 @@ class GatewaySlashCommandsMixin: ) async def _handle_branch_command(self, event: MessageEvent) -> str: - """Handle /branch [name] — fork the current session into a new independent copy. - - Copies conversation history to a new session so the user can explore a different approach - without losing the original. Inspired by Claude Code's /branch command. + """Handle /branch [name] — fork the current session into a new independent copy so the + user can explore a different approach without losing the original. """ import uuid as _uuid @@ -5189,24 +4901,12 @@ class GatewaySlashCommandsMixin: model=(self.config.get("model", {}) or {}).get("default") if isinstance(self.config, dict) else None, model_config={"_branched_from": parent_session_id}, parent_session_id=parent_session_id, - # Gateway routing columns — forward ALL of them at CREATE time, - # same fix as the compression-rotation bug in - # agent/conversation_compression.py. Without these, the branched - # child row has NULL routing columns until switch_session() below - # calls _record_gateway_session_peer() — a crash/kill anywhere - # between here and there (most plausibly mid-history-copy, since - # each append_message call a few lines down is independently - # best-effort) leaves the branch permanently unroutable: - # unreachable by chat/thread lookup, and unreachable via /resume's - # IDOR guard too (which requires the row's chat_id/thread_id to - # match the caller's). user_id is critical for the fallback lookup - # path (hermes_state.py:1994-2009) that searches by the complete - # peer tuple when session_key doesn't match. origin_json and - # display_name complete the identity (same shape as the reset - # path's db_create_kwargs in gateway/session.py, #82633) so - # consumers that read routing/presentation data from state.db - # (mcp_serve, mirror, channel directory) see the branch row - # fully formed with zero backfill gap. + # Forward ALL gateway routing columns at CREATE time: otherwise they're NULL until + # switch_session() calls _record_gateway_session_peer(), and a crash in between (each + # append_message is best-effort) leaves the branch unroutable — by chat/thread lookup + # and by /resume's IDOR guard. user_id feeds the full-peer-tuple fallback lookup; + # origin_json/display_name complete the identity (same shape as session.py's reset + # path) so state.db consumers see a fully formed row with no backfill gap. user_id=source.user_id, session_key=session_key, chat_id=source.chat_id, @@ -5252,10 +4952,8 @@ class GatewaySlashCommandsMixin: pass # Best-effort copy # Set title - try: + with contextlib.suppress(Exception): await self._session_db.set_session_title(new_session_id, branch_title) - except Exception: - pass # Switch the session store entry to the new session new_entry = await self.async_session_store.switch_session(session_key, new_session_id) @@ -5273,8 +4971,7 @@ class GatewaySlashCommandsMixin: async def _handle_topup_command(self, event: MessageEvent) -> str: """Handle /topup -- show the Nous balance and hand off to the portal. - Remote spending is managed on the portal: this messaging command does NOT charge, - confirm, or track payment here — everything happens in the browser and the next /topup + Does NOT charge, confirm, or track payment — that happens in the browser; the next /topup shows the new balance. Fetched off the event loop; fail-open. """ from agent.account_usage import build_credits_view @@ -5304,8 +5001,7 @@ class GatewaySlashCommandsMixin: def _context_breakdown_block(self, agent, source, expanded: bool) -> list[str]: """Render the /context per-category block (plain text, no grid). - Estimated (chars/4) — same engine as the desktop popover and /usage. Runs in a thread - (sync store reads); returns [] and never raises so /context stays robust. + Estimated (chars/4), same engine as /usage. Runs in a thread; returns [] and never raises. """ try: from agent.context_breakdown import ( @@ -5339,8 +5035,7 @@ class GatewaySlashCommandsMixin: def _context_breakdown_lines(self, agent, source) -> list[str]: """Render the per-category context breakdown for /usage. - Estimated (chars/4) — same engine the desktop popover uses. Returns an - empty list and never raises on failure so /usage stays robust. + Estimated (chars/4). Returns [] and never raises so /usage stays robust. """ try: from agent.context_breakdown import compute_session_context_breakdown @@ -5380,9 +5075,8 @@ class GatewaySlashCommandsMixin: async def _handle_usage_command(self, event: MessageEvent) -> str: """Handle /usage command -- show token usage for the current session. - Checks both _running_agents (mid-turn) and _agent_cache (between turns) - so that rate limits, cost estimates, and detailed token breakdowns are - available whenever the user asks, not only while the agent is running. + Checks both _running_agents (mid-turn) and _agent_cache (between turns) so details are + available whenever the user asks. """ from gateway.run import _AGENT_PENDING_SENTINEL source = event.source @@ -5465,11 +5159,9 @@ class GatewaySlashCommandsMixin: account_lines = render_account_usage_lines(account_snapshot, markdown=True) # ── Nous credits magnitudes + monthly-grant % gauge ───────────── - # Shared with the CLI / TUI /usage block via nous_credits_lines(): a single auth-gate + - # portal-fetch + render path (which also honors the dev fixture). Run off the event loop. - # Gates on "a Nous account is logged in" — NOT the inference provider, NOT nested under - # `if provider:` — so a Nous-credentialled user inferring elsewhere still sees a balance. - # No recovery trigger (messaging binds no notice consumer). Fail-open: never break /usage. + # Shared with CLI/TUI via nous_credits_lines(); run off the event loop. Gates on "a Nous + # account is logged in" — NOT the inference provider, NOT under `if provider:` — so a Nous + # user inferring elsewhere still sees a balance. No recovery trigger; fail-open. try: from agent.account_usage import nous_credits_lines @@ -5587,7 +5279,7 @@ class GatewaySlashCommandsMixin: i += 1 try: - from hermes_state import get_shared_session_db, release_shared_session_db + from hermes_state import get_shared_session_db from agent.insights import InsightsEngine def _run_insights(): @@ -5601,10 +5293,9 @@ class GatewaySlashCommandsMixin: from hermes_state import release_or_close release_or_close(db) - # _run_in_executor_with_context, not a bare hop: ``SessionDB()`` with no explicit path - # resolves ``get_hermes_home()`` at call time, and that override is a contextvar - # installed by ``_profile_runtime_scope``. A default-executor hop starts with an EMPTY - # context, so /insights would read the DEFAULT profile's state.db. + # Not a bare hop: ``SessionDB()`` resolves ``get_hermes_home()`` at call time, which is + # a contextvar set by ``_profile_runtime_scope``; a default-executor hop starts with an + # EMPTY context and would read the DEFAULT profile's state.db. return await self._run_in_executor_with_context(_run_insights) except Exception as e: logger.error("Insights command error: %s", e, exc_info=True) @@ -5613,11 +5304,9 @@ class GatewaySlashCommandsMixin: async def _handle_reload_mcp_command(self, event: MessageEvent) -> Optional[str]: """Handle /reload-mcp — reconnect MCP servers and rebuild the cached agent. - Reloading MCP tools invalidates the provider prompt cache (tool schemas are baked into the - system prompt): the next message re-sends full input tokens, which is expensive on - long-context or high-reasoning models. To surface that cost the command routes through the - slash-confirm primitive. "Always Approve" persists ``approvals.mcp_reload_confirm: false`` - so the prompt is silenced for subsequent reloads in any session. + Reloading invalidates the provider prompt cache (tool schemas live in the system prompt), + so it routes through slash-confirm; "Always Approve" persists + ``approvals.mcp_reload_confirm: false``. """ source = event.source session_key = self._session_key_for_source(source) @@ -5668,12 +5357,9 @@ class GatewaySlashCommandsMixin: async def _handle_reload_skills_command(self, event: MessageEvent) -> str: """Handle /reload-skills — rescan skills dir, queue a note for next turn. - Skills don't need to be in the system prompt for the model to use them (they're invoked - via ``/skill-name``, ``skills_list``, or ``skill_view`` at runtime), so this does NOT - clear the prompt cache — prefix caching stays intact. Added/removed skills are reported via - a one-shot note in ``_pending_skills_reload_notes[session_key]`` that the gateway prepends - to the NEXT user message (then clears) — nothing is written to the transcript out-of-band, - so message alternation is preserved. + Skills are invoked at runtime, not baked into the system prompt, so this does NOT clear the + prompt cache. Added/removed skills go into ``_pending_skills_reload_notes[session_key]``, + prepended to the NEXT user message — nothing out-of-band, so alternation is preserved. """ try: from agent.skill_commands import reload_skills @@ -5685,14 +5371,9 @@ class GatewaySlashCommandsMixin: removed = result.get("removed", []) # [{"name", "description"}, ...] total = result.get("total", 0) - # Let each connected adapter refresh any platform-side state - # that cached the skill list at startup. Today that's the - # Discord /skill autocomplete (registered once per connect); - # without this call, new skills stay invisible in the - # dropdown and deleted skills error out when clicked. Other - # adapters that don't override refresh_skill_group (Telegram's - # BotCommand menu, Slack subcommand map, etc.) are silently - # skipped — the in-process reload above is enough for them. + # Let adapters refresh platform-side state that cached the skill list at startup (today: + # Discord /skill autocomplete — otherwise new skills stay invisible and deleted ones + # error). Adapters without refresh_skill_group are skipped; the in-process reload suffices. for adapter in list(self.adapters.values()): refresh = getattr(adapter, "refresh_skill_group", None) if not callable(refresh): @@ -5761,11 +5442,9 @@ class GatewaySlashCommandsMixin: return t("gateway.reload_skills.failed", error=e) async def _handle_bundles_command(self, event: MessageEvent) -> str: - """Handle /bundles — list installed skill bundles. + """Handle /bundles — list installed skill bundles (mirrors the CLI handler). - Mirrors the CLI ``/bundles`` handler. Returns a single text - message suitable for any gateway adapter; bundles are loaded by - invoking the bundle's own ``/`` command, not by this one. + Bundles are loaded by invoking their own ``/`` command, not by this one. """ from hermes_cli.slash_exec import CommandContext, execute_command @@ -5799,9 +5478,8 @@ class GatewaySlashCommandsMixin: async def _handle_approve_command(self, event: MessageEvent) -> Optional[str]: """Handle /approve command — unblock waiting agent thread(s). - The agent thread(s) are blocked inside tools/approval.py waiting for the user to respond. - This handler signals the event so the agent resumes and the terminal_tool executes the - command inline — the same flow as the CLI's synchronous input() approval. + Agent threads block inside tools/approval.py; signalling the event resumes them so the + command executes inline — same flow as the CLI's synchronous approval. """ source = event.source session_key = self._session_key_for_source(source) @@ -5841,9 +5519,8 @@ class GatewaySlashCommandsMixin: plural = "plural" if count > 1 else "singular" confirmation_text = t(f"gateway.approve.{choice}_{plural}", count=count) # Native-streaming adapters (WeCom msgtype:"stream") need the confirmation sent directly - # with control-lane metadata so it lands via a reliable proactive send instead of the - # (already-finalized) reply stream. Everyone else returns the text for normal delivery. - # (`is not True` — mock adapters auto-create truthy attributes.) + # with control-lane metadata (reliable proactive send, not the finalized reply stream). + # Everyone else returns text for normal delivery. (`is not True`: mocks auto-create attrs.) if getattr(_adapter, "SUPPORTS_NATIVE_STREAMING", False) is not True: return confirmation_text if _adapter: @@ -5870,9 +5547,8 @@ class GatewaySlashCommandsMixin: async def _handle_deny_command(self, event: MessageEvent) -> str: """Handle /deny command — reject pending dangerous command(s). - Signals blocked agent thread(s) with a 'deny' result so they receive a definitive BLOCKED - message, same as the CLI deny flow. ``/deny`` denies the oldest; ``/deny all`` denies - everything. Ported from qwibitai/nanoclaw#2832. + Signals blocked thread(s) with a 'deny' result so they get a definitive BLOCKED message, + as in the CLI. ``/deny`` denies the oldest; ``/deny all`` denies everything. """ source = event.source session_key = self._session_key_for_source(source) @@ -5893,10 +5569,7 @@ class GatewaySlashCommandsMixin: raw_args = event.get_command_args().strip() tokens = raw_args.split() resolve_all = bool(tokens) and tokens[0].lower() == "all" - if resolve_all: - reason = raw_args[len(tokens[0]):].strip() - else: - reason = raw_args + reason = raw_args[len(tokens[0]):].strip() if resolve_all else raw_args # Cap to a sane one-liner; the agent only needs a short hint. if reason: reason = reason[:280].strip() @@ -5956,11 +5629,9 @@ class GatewaySlashCommandsMixin: async def _handle_debug_command(self, event: MessageEvent) -> str: """Handle /debug — upload debug report (summary only) and return paste URLs. - Gateway uploads ONLY the summary report (system info + log tails), - NOT full log files, to protect conversation privacy. Users who need - full log uploads should use ``hermes debug share`` from the CLI. + Uploads ONLY the summary (system info + log tails), never full logs, to protect privacy; + use ``hermes debug share`` from the CLI for full uploads. """ - import asyncio from hermes_cli.debug import ( _capture_dump, collect_debug_report, upload_to_pastebin, _schedule_auto_delete, @@ -6001,9 +5672,8 @@ class GatewaySlashCommandsMixin: async def _handle_update_command(self, event: MessageEvent) -> str: """Handle /update command — update Hermes Agent to the latest version. - Spawns ``hermes update`` in a detached session (via ``setsid``) so it survives the - gateway restart that ``hermes update`` may trigger. Marker files are written so either - the current gateway process or the next one can notify the user when the update finishes. + Spawns ``hermes update`` detached (``setsid``) so it survives the gateway restart it may + trigger; marker files let this or the next gateway process notify the user on completion. """ from gateway.run import _hermes_home, _resolve_hermes_bin import json @@ -6059,30 +5729,12 @@ class GatewaySlashCommandsMixin: _tmp_pending.replace(pending_path) exit_code_path.unlink(missing_ok=True) - # Spawn `hermes update --gateway` detached so it survives gateway restart. - # --gateway enables file-based IPC for interactive prompts (stash - # restore, config migration) so the gateway can forward them to the - # user instead of silently skipping them. - # Use setsid for portable session detach (works under system services - # where systemd-run --user fails due to missing D-Bus session). - # PYTHONUNBUFFERED ensures output is flushed line-by-line so the - # gateway can stream it to the messenger in near-real-time. - # Spawn `hermes update --gateway` detached so it survives gateway restart. - # --gateway enables file-based IPC for interactive prompts (stash - # restore, config migration) so the gateway can forward them to the - # user instead of silently skipping them. - # Use setsid for portable session detach (works under system services - # where systemd-run --user fails due to missing D-Bus session). - # PYTHONUNBUFFERED ensures output is flushed line-by-line so the - # gateway can stream it to the messenger in near-real-time. - # - # Windows: no bash/setsid chain. Run `hermes update --gateway` - # directly via sys.executable; redirect stdout/stderr to the same - # output files via Popen file handles; write the exit code in a - # follow-up write. A tiny Python watcher would be cleaner but - # we're already inside gateway/run.py's update path which is async, - # so the simplest correct thing is: launch an inline Python helper - # that runs the command and writes both outputs. + # Spawn `hermes update --gateway` detached (setsid: portable, works where systemd-run --user + # lacks a D-Bus session) so it survives gateway restart. --gateway enables file-based IPC + # for interactive prompts so the gateway forwards them instead of skipping. PYTHONUNBUFFERED + # lets the gateway stream output in near-real-time. + # Windows: no setsid chain — an inline Python helper via sys.executable runs the command, + # redirects both outputs to the same files, and writes the exit code. try: if sys.platform == "win32": import textwrap @@ -6122,10 +5774,8 @@ class GatewaySlashCommandsMixin: update_cmd = ( f"PYTHONUNBUFFERED=1 {hermes_cmd_str} update --gateway" f" > {shlex.quote(str(output_path))} 2>&1; " - # Avoid `status=$?`: `status` is a read-only special parameter - # in zsh, and this command string is copied/reused in macOS/zsh - # operator wrappers. Keep the template zsh-safe even though this - # specific subprocess currently runs under bash. + # Avoid `status=$?`: `status` is read-only in zsh and this template is reused in + # macOS/zsh operator wrappers, so keep it zsh-safe even though bash runs it here. f"rc=$?; printf '%s' \"$rc\" > {shlex.quote(str(exit_code_path))}" ) setsid_bin = shutil.which("setsid") diff --git a/gateway/status.py b/gateway/status.py index 0df8708158..5ce7187cc0 100644 --- a/gateway/status.py +++ b/gateway/status.py @@ -29,6 +29,7 @@ from pathlib import Path from hermes_constants import get_hermes_home, _get_platform_default_hermes_home from typing import Any, Callable, NamedTuple, Optional from utils import atomic_json_write +import contextlib if sys.platform == "win32": import msvcrt @@ -659,9 +660,7 @@ def _command_line_belongs_to_profile(command: str, profile_home: Path) -> bool: # absence is not disqualifying — only a conflicting explicit value is. if "--profile " in command_lc or " -p " in command_lc: return False - if "hermes_home=" in command_lc and f"hermes_home={home_lc}" not in command_lc: - return False - return True + return not ("hermes_home=" in command_lc and f"hermes_home={home_lc}" not in command_lc) def _record_matches_live_gateway_pid( @@ -688,11 +687,7 @@ def _record_matches_live_gateway_pid( if live_cmdline: if not looks_like_gateway_runtime_command_line(live_cmdline): return False - if expected_home is not None and not _command_line_belongs_to_profile( - live_cmdline, expected_home - ): - return False - return True + return not (expected_home is not None and not _command_line_belongs_to_profile(live_cmdline, expected_home)) return _record_looks_like_gateway(record) @@ -877,14 +872,10 @@ def _cleanup_invalid_pid_path(pid_path: Path, *, cleanup_stale: bool) -> None: if not cleanup_stale: return _clear_running_pid_cache() - try: + with contextlib.suppress(Exception): pid_path.unlink(missing_ok=True) - except Exception: - pass - try: + with contextlib.suppress(Exception): _get_gateway_lock_path(pid_path).unlink(missing_ok=True) - except Exception: - pass def _write_gateway_lock_record(handle) -> None: @@ -892,10 +883,8 @@ def _write_gateway_lock_record(handle) -> None: handle.truncate() json.dump(_build_pid_record(), handle) handle.flush() - try: + with contextlib.suppress(OSError): os.fsync(handle.fileno()) - except OSError: - pass def _try_acquire_file_lock(handle) -> bool: @@ -1089,10 +1078,8 @@ def release_gateway_runtime_lock() -> None: return _gateway_lock_handle = None _release_file_lock(handle) - try: + with contextlib.suppress(OSError): handle.close() - except OSError: - pass _clear_running_pid_cache() @@ -1127,10 +1114,8 @@ def is_gateway_runtime_lock_active(lock_path: Optional[Path] = None) -> bool: # session that ran as root. The parent directory owner can unlink # files even when they don't own them, so remove the stale lock # and report inactive — the new process will create a fresh one. - try: + with contextlib.suppress(OSError): resolved_lock_path.unlink() - except OSError: - pass return False try: if _try_acquire_file_lock(handle): @@ -1138,10 +1123,8 @@ def is_gateway_runtime_lock_active(lock_path: Optional[Path] = None) -> bool: return False return True finally: - try: + with contextlib.suppress(OSError): handle.close() - except OSError: - pass def _strict_path_exists(path: Path, label: str) -> bool: @@ -1170,10 +1153,8 @@ def _is_gateway_runtime_lock_active_strict(lock_path: Path) -> bool: except OSError as exc: raise RuntimeError(f"gateway runtime lock probe failed: {exc}") from exc finally: - try: + with contextlib.suppress(OSError): handle.close() - except OSError: - pass def write_pid_file() -> None: @@ -1195,10 +1176,8 @@ def write_pid_file() -> None: f.write(record) _clear_running_pid_cache() except Exception: - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass raise @@ -1358,13 +1337,7 @@ def runtime_status_pid_is_live(record: Optional[dict[str, Any]]) -> bool: return False recorded_start = (record or {}).get("start_time") current_start = _get_process_start_time(pid) - if ( - recorded_start is not None - and current_start is not None - and current_start != recorded_start - ): - return False - return True + return not (recorded_start is not None and current_start is not None and current_start != recorded_start) def parse_active_agents(raw: Any) -> int: @@ -1673,10 +1646,8 @@ def acquire_scoped_lock(scope: str, identity: str, metadata: Optional[dict[str, # stale. This happens when a previous process was killed between # O_CREAT|O_EXCL and the subsequent json.dump() (e.g. DNS failure # during rapid Slack reconnect retries). - try: + with contextlib.suppress(OSError): lock_path.unlink(missing_ok=True) - except OSError: - pass if existing: try: existing_pid = int(existing["pid"]) @@ -1772,10 +1743,8 @@ def acquire_scoped_lock(scope: str, identity: str, metadata: Optional[dict[str, except OSError: pass else: - try: + with contextlib.suppress(OSError): tombstone.unlink(missing_ok=True) - except OSError: - pass else: return False, existing @@ -1787,10 +1756,8 @@ def acquire_scoped_lock(scope: str, identity: str, metadata: Optional[dict[str, with os.fdopen(fd, "w", encoding="utf-8") as handle: json.dump(record, handle) except Exception: - try: + with contextlib.suppress(OSError): lock_path.unlink(missing_ok=True) - except OSError: - pass raise return True, None @@ -1807,10 +1774,8 @@ def release_scoped_lock(scope: str, identity: str) -> None: # start_time equality: on-disk null vs a live fingerprint (macOS/psutil # timing) would otherwise leave the lock stuck across Discord/Telegram # reconnects (#81468). start_time only guards PID reuse for *other* PIDs. - try: + with contextlib.suppress(OSError): lock_path.unlink(missing_ok=True) - except OSError: - pass def release_all_scoped_locks( @@ -1924,17 +1889,13 @@ def _consume_pid_marker_for_self( target_start_time = record.get(start_time_field) written_at = record.get("written_at") or "" except (KeyError, TypeError, ValueError): - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass return False if _marker_is_stale(written_at, ttl_s): - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass return False # Cross-profile guard (#29092): new markers explicitly name the verified @@ -1977,10 +1938,8 @@ def _consume_pid_marker_for_self( else: matches = True - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass return matches @@ -2050,10 +2009,8 @@ def consume_takeover_marker_for_self() -> bool: def clear_takeover_marker(target_home: Optional[Path] = None) -> None: """Remove the takeover marker unconditionally. Safe to call repeatedly.""" - try: + with contextlib.suppress(OSError): _get_takeover_marker_path(target_home).unlink(missing_ok=True) - except OSError: - pass def _validated_scoped_lock_gateway_owner( @@ -2420,19 +2377,15 @@ def planned_stop_marker_targets_self() -> bool: written_at = record.get("written_at") or "" except (KeyError, TypeError, ValueError): # Malformed marker can never match anyone — drop it. - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass return False if _marker_is_stale(written_at, _PLANNED_STOP_MARKER_TTL_S): # A marker this old is past its useful life regardless of target — # clean it up so it cannot crash-loop a freshly booted gateway. - try: + with contextlib.suppress(OSError): path.unlink(missing_ok=True) - except OSError: - pass return False our_pid = os.getpid() @@ -2455,10 +2408,8 @@ def planned_stop_marker_targets_self() -> bool: def clear_planned_stop_marker() -> None: """Remove the planned-stop marker unconditionally.""" - try: + with contextlib.suppress(OSError): _get_planned_stop_marker_path().unlink(missing_ok=True) - except OSError: - pass def get_running_pid( diff --git a/gateway/sticker_cache.py b/gateway/sticker_cache.py index 58f1ee1c87..f336e8fab9 100644 --- a/gateway/sticker_cache.py +++ b/gateway/sticker_cache.py @@ -11,6 +11,7 @@ import time from typing import Optional from hermes_cli.config import get_hermes_home +import contextlib CACHE_PATH = get_hermes_home() / "sticker_cache.json" @@ -42,10 +43,8 @@ def _save_cache(cache: dict) -> None: os.fsync(f.fileno()) os.replace(tmp_path, str(CACHE_PATH)) except BaseException: - try: + with contextlib.suppress(OSError): os.unlink(tmp_path) - except OSError: - pass raise diff --git a/gateway/stream_consumer.py b/gateway/stream_consumer.py index 7b4d905ea1..07ed8f65b7 100644 --- a/gateway/stream_consumer.py +++ b/gateway/stream_consumer.py @@ -38,6 +38,7 @@ from gateway.response_filters import ( is_intentional_silence_response as _is_intentional_silence_response, is_partial_silence_marker as _is_partial_silence_marker, ) +import contextlib logger = logging.getLogger("gateway.stream_consumer") @@ -685,9 +686,7 @@ class GatewayStreamConsumer: return True # A segment break / commentary may have delivered the final text # earlier in the turn under a different record. - if self.has_delivered_text(final_text): - return True - return False + return bool(self.has_delivered_text(final_text)) def has_delivered_text(self, text: str) -> bool: """Return True if *text* was already delivered as visible chat content.""" @@ -752,10 +751,8 @@ class GatewayStreamConsumer: race conditions with pending deltas or other queue items. """ loop = None - try: + with contextlib.suppress(RuntimeError): loop = asyncio.get_running_loop() - except RuntimeError: - pass if not self._use_native_streaming: # No native stream to close — return resolved future @@ -774,10 +771,7 @@ class GatewayStreamConsumer: # Create a future that run() will resolve after processing. # cancelled_flag is retained for backward compatibility with callers # (run.py sets it on timeout) but the handler always finalizes regardless. - if loop: - boundary_future = loop.create_future() - else: - boundary_future = concurrent.futures.Future() + boundary_future = loop.create_future() if loop else concurrent.futures.Future() cancelled_flag = {"cancelled": False} self._queue.put((_APPROVAL_BOUNDARY, boundary_future, cancelled_flag)) @@ -854,10 +848,8 @@ class GatewayStreamConsumer: """ if flush_event is None: return - try: + with contextlib.suppress(Exception): flush_event.set() - except Exception: - pass def _reset_segment_state(self, *, preserve_no_edit: bool = False) -> None: if preserve_no_edit and self._message_id == "__no_edit__": @@ -1040,10 +1032,7 @@ class GatewayStreamConsumer: # Resolve future so approval callback knows the result if boundary_future is not None: try: - if isinstance(boundary_future, asyncio.Future): - if not boundary_future.done(): - boundary_future.set_result(boundary_ok) - elif isinstance(boundary_future, concurrent.futures.Future): + if isinstance(boundary_future, (asyncio.Future, concurrent.futures.Future)): if not boundary_future.done(): boundary_future.set_result(boundary_ok) except Exception: @@ -1995,14 +1984,12 @@ class GatewayStreamConsumer: # _final_response_sent itself; this handler owns the flags. _best_effort_ok = False if self._accumulated and self._message_id: - try: + with contextlib.suppress(Exception): _best_effort_ok = bool( await self._send_or_edit( self._accumulated, finalize=True, is_turn_final=False, ) ) - except Exception: - pass elif self._message_id is None: # Native draft path deliberately keeps _message_id=None, so # the best-effort edit above never runs for it — the stream diff --git a/gateway/streaming_tts_consumer.py b/gateway/streaming_tts_consumer.py index b796385c59..e25982e713 100644 --- a/gateway/streaming_tts_consumer.py +++ b/gateway/streaming_tts_consumer.py @@ -33,6 +33,7 @@ import threading from typing import Any, Dict, Optional from gateway.platforms.base import AudioFormat, StreamingTTSHandle +import contextlib logger = logging.getLogger("gateway.streaming_tts_consumer") @@ -300,16 +301,12 @@ class StreamingTTSConsumer: else: logger.debug("streaming TTS _ABORT sentinel could not be enqueued") if self._handle is not None and not self._handle.aborted: - try: + with contextlib.suppress(Exception): self._loop.call_soon_threadsafe(asyncio.create_task, self._safe_abort(reason)) - except Exception: - pass async def wait_complete(self, timeout: float = 10.0) -> bool: """Wait for the drain task to finish. Returns True only on full success.""" if self._task is not None: - try: + with contextlib.suppress(asyncio.CancelledError, Exception): await asyncio.wait_for(asyncio.shield(self._task), timeout=timeout) - except (asyncio.CancelledError, Exception): - pass return self._completed