diff --git a/gateway/run.py b/gateway/run.py index d05b1d49ec..a478de1177 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -16,13 +16,9 @@ except ModuleNotFoundError: import asyncio import concurrent.futures import dataclasses -import faulthandler -import functools -import inspect import json import logging import os -import queue import re import shlex import site @@ -32,12 +28,12 @@ import threading import time import traceback from collections import OrderedDict -from contextvars import Context, copy_context +from contextvars import copy_context from pathlib import Path -from datetime import datetime, timedelta, timezone -from typing import Awaitable, Callable, Dict, Optional, Any, List, Tuple, Union, cast +from datetime import datetime +from typing import Callable, Dict, Optional, Any, List, Tuple, cast -from agent.async_utils import consume_detached_task_result, safe_schedule_threadsafe +from agent.async_utils import safe_schedule_threadsafe from agent.conversation_compression import ( COMPACTION_DONE_STATUS, COMPACTION_STATUS, @@ -50,8 +46,6 @@ from agent.conversation_compression import ( PREFLIGHT_COMPRESSION_STATUS_TEMPLATE, ) from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX -from agent.compaction_display import project_compaction_message_for_display -from agent.i18n import t from agent.interrupt_compat import request_hard_interrupt from agent.turn_context import ( compression_made_progress, @@ -1642,7 +1636,6 @@ _TOOL_MEDIA_RE = re.compile( # Shared with cron delivery and gateway background tasks — the repair must run on every surface # that feeds a final response into media extraction; canonical names live in gateway.media_repair. from gateway.media_repair import ( # noqa: E402 - repair_explicit_computer_use_media_paths, tool_name_by_call_id as _tool_name_by_call_id, ) @@ -1853,7 +1846,7 @@ sys.path.insert(0, str(Path(__file__).parent.parent)) # Resolve Hermes home directory (respects HERMES_HOME override) from hermes_constants import get_hermes_home, get_hermes_home_override -from utils import atomic_json_write, base_url_hostname, is_truthy_value +from utils import atomic_json_write, base_url_hostname, is_truthy_value # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) _hermes_home = get_hermes_home() # Load environment variables from ~/.hermes/.env first. @@ -1952,9 +1945,6 @@ suppress, # shared listener (serving every profile via the /p// prefix), so a SECONDARY profile # enabling one is always a misconfiguration and is skipped (SecondaryPortBindingConfigError) rather # than taking down the multiplexer. Lives in gateway.config so dashboard validation enforces it too. -from gateway.config import ( - platform_binds_port as _platform_binds_port, -) class MultiplexConfigError(RuntimeError): @@ -2515,7 +2505,6 @@ if not _configured_cwd or _configured_cwd in CWD_PLACEHOLDERS: from gateway.config import ( ChannelOverride, Platform, - _BUILTIN_PLATFORM_VALUES, GatewayConfig, PlatformConfig, _getenv, @@ -2523,31 +2512,19 @@ from gateway.config import ( ) from gateway.session import ( AsyncSessionStore, - SessionEntry, SessionStore, SessionSource, SessionContext, - TranscriptReadError, - _session_key_namespace, - build_session_context, - build_session_context_prompt, - build_channel_continuity_note, build_session_key, - is_shared_multi_user_session, - neutralize_untrusted_inline_text, ) from gateway.delivery import ( DeliveryRouter, - looks_like_telegram_private_chat_id, - resolve_delivery_transport, + resolve_delivery_transport, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) ) from gateway.turn_lease import ( - DEFAULT_LEASE_WAIT, SessionTurnLeaseRegistry, - TurnLeaseTimeoutError, ) from gateway.session_state import ( - SERVICE_TIER_UNSET as _SERVICE_TIER_UNSET, SessionState, legacy_dict_property, legacy_lease_token_property, @@ -2555,43 +2532,37 @@ from gateway.session_state import ( from gateway.authz_mixin import GatewayAuthorizationMixin from gateway.kanban_watchers import GatewayKanbanWatchersMixin from gateway.slash_commands import GatewaySlashCommandsMixin -from gateway.turn_context import TurnContext +from gateway.run_voice import GatewayVoiceMixin +from gateway.run_adapters import GatewayAdapterLifecycleMixin +from gateway.run_topics import GatewayTopicThreadsMixin +from gateway.run_turn import GatewayTurnMixin +from gateway.run_shutdown import GatewayShutdownMixin +from gateway.run_busy import GatewayBusySessionMixin +from gateway.run_config_loaders import GatewayConfigLoadersMixin +from gateway.run_startup import GatewayStartupMixin +from gateway.run_watchers import GatewaySessionWatchersMixin +from gateway.run_notifications import GatewayNotificationsMixin +from gateway.run_inbound import GatewayInboundMixin +from gateway.run_goals import GatewayGoalsMixin +from gateway.run_agent_cache import GatewayAgentCacheMixin +from gateway.run_turn_runner import TurnRunner # noqa: F401 (re-exported; run.py callers + tests) from gateway.platforms.base import ( BasePlatformAdapter, - EphemeralReply, MessageEvent, MessageType, - _prefix_within_utf16_limit, _reply_anchor_for_event, - build_auto_tts_output_path, - merge_pending_message_event, - utf16_len, -) + merge_pending_message_event, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) + ) from gateway.shutdown_watchdog import ( - DEFAULT_HEARTBEAT_INTERVAL_S, - DEFAULT_LOOP_WATCHDOG_INTERVAL_S, - DEFAULT_LOOP_WATCHDOG_MAX_STRIKES, - DEFAULT_LOOP_WATCHDOG_TIMEOUT_S, - _arm_loop_floor_timer, - arm_shutdown_watchdog, - loop_heartbeat_forever, - resolve_shutdown_watchdog_delay, - start_loop_liveness_watchdog, + _arm_loop_floor_timer, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) + start_loop_liveness_watchdog, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) ) from gateway.restart import ( DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, - DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT, + DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT, DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, - GATEWAY_FATAL_CONFIG_EXIT_CODE, - GATEWAY_SERVICE_RESTART_EXIT_CODE, - is_global_startup_conflict, - parse_cron_drain_timeout, - parse_restart_after_turn_timeout, - parse_restart_drain_timeout, - parse_signal_interrupt_grace_timeout, - resolve_cron_drain_budget, ) @@ -2681,8 +2652,7 @@ _CONVERSATION_SCOPED_STATE: tuple = ( "_pending_turn_sidecar_notes", ) -# Sentinel for "caller did not pass metadata" vs "caller passed None". -_UNSET = object() +from gateway.run_common import _UNSET # noqa: F401 (def-time sentinel shared with run_* mixins) def _resolve_runtime_agent_kwargs() -> dict: @@ -4017,2235 +3987,6 @@ def _reconnect_needs_attention(info: dict, now: float) -> bool: return (now - queued_at) >= _RECONNECT_ATTENTION_AFTER_SECONDS -class TurnRunner: - """Per-turn collaborator carrying ``GatewayRunner._run_agent_inner``'s tool-progress callbacks. - - Module-global references (logger, cfg_get, BasePlatformAdapter, ...) resolve in this module. - """ - - def __init__(self, runner: "GatewayRunner", ctx: TurnContext) -> None: - self._runner = runner - self._ctx = ctx - - def progress_callback(self, event_type: str, tool_name: str = None, preview: str = None, args: dict = None, **kwargs): - """Callback invoked by agent on tool lifecycle events.""" - ctx = self._ctx - # Failed subagent → one clean user-facing notice, handled FIRST, before every progress-queue - # gate: platforms with tool_progress off must still hear about a dead delegation. Only - # terminal failure statuses render (same notice rail as credit warnings); success/interrupt - # stay quiet. - if event_type == "subagent.complete": - _sub_status = kwargs.get("status") - try: - from tools.delegate_tool import ( - SUBAGENT_FAILURE_STATUSES, - format_subagent_failure_line, - ) - if _sub_status in SUBAGENT_FAILURE_STATUSES and ctx._run_still_current(): - _line = format_subagent_failure_line( - kwargs.get("goal"), - _sub_status, - error=kwargs.get("summary") or preview, - duration_seconds=kwargs.get("duration_seconds"), - ) - safe_schedule_threadsafe( - self._runner._deliver_platform_notice(ctx.source, _line), - ctx._loop_for_step, - logger=logger, - log_message="subagent failure notice scheduling error", - ) - except Exception: - logger.debug("subagent failure notice failed", exc_info=True) - return - # Live status line (Slack assistant status): stash the tool phrase on the adapter; the - # _keep_typing refresh renders it. Plain dict write, safe from the sync worker thread. - if ( - ctx._live_status_adapter is not None - and ctx._live_status_mode != "off" - and tool_name != "_thinking" - ): - try: - if event_type == "tool.started" and tool_name and ctx._run_still_current(): - from agent.display import build_status_phrase - _phrase = build_status_phrase( - tool_name, - args if ctx._live_status_mode == "full" else None, - ) - ctx._live_status_adapter.set_status_text(ctx.source.chat_id, _phrase) - elif event_type == "tool.completed": - # Between tools the model is genuinely "thinking" - # again — revert to the static default. - ctx._live_status_adapter.set_status_text(ctx.source.chat_id, None) - except Exception as _ls_err: - logger.debug("live status update failed: %s", _ls_err) - # "log" mode: append tool.started lines to the log queue, silent in chat. Handled before - # the progress_queue guard because log mode runs without a chat progress queue. - if ctx.log_queue is not None: - if event_type == "tool.started" and tool_name and tool_name != "_thinking": - ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S") - preview_str = f' "{preview}"' if preview else "" - ctx.log_queue.put(f"{ts} {tool_name}:{preview_str}".rstrip()) - if not ctx.progress_queue: - return - if not ctx.progress_queue or not ctx._run_still_current(): - return - - # First-touch onboarding: the first time a tool exceeds _LONG_TOOL_THRESHOLD_S while - # streaming every tool (progress_mode == "all"), append a one-time /verbose hint. - if event_type == "tool.completed" and not ctx.long_tool_hint_fired[0]: - try: - duration = kwargs.get("duration") or 0 - if duration >= ctx._LONG_TOOL_THRESHOLD_S and ctx.progress_mode == "all": - from agent.onboarding import ( - TOOL_PROGRESS_FLAG, - is_seen, - mark_seen, - tool_progress_hint_gateway, - ) - _cfg = _load_gateway_config() - gate_on = is_truthy_value( - cfg_get(_cfg, "display", "tool_progress_command"), - default=False, - ) - if gate_on and not is_seen(_cfg, TOOL_PROGRESS_FLAG): - ctx.long_tool_hint_fired[0] = True - ctx.progress_queue.put(tool_progress_hint_gateway()) - mark_seen(_hermes_home / "config.yaml", TOOL_PROGRESS_FLAG) - except Exception as _hint_err: - logger.debug("tool-progress onboarding hint failed: %s", _hint_err) - return - - # "_thinking" is assistant scratch text between tool calls. It is never ordinary tool - # progress: only relay it when the platform explicitly opted into thinking_progress. - if event_type == "_thinking" or tool_name == "_thinking": - if not ctx._thinking_enabled: - return - thinking_text = preview if tool_name == "_thinking" else tool_name - msg = f"💬 {thinking_text}" if thinking_text else None - if msg: - ctx.progress_queue.put(msg) - return - - # Native task cards consume the ID-bearing tool_start/tool_complete callbacks instead; - # name-correlated text events would duplicate cards and mispair concurrent same-tool calls. - if ctx._native_slack_task_cards and event_type in { - "tool.started", - "tool.completed", - }: - return - - # If tool_progress is off, only _thinking passes through (above). - # Regular tool calls are suppressed. - if not ctx.tool_progress_enabled: - return - - # Only act on tool.started events (ignore tool.completed, reasoning.available, etc.) - if event_type not in {"tool.started",}: - return - - # Never render a progress bubble for clarify: send_clarify IS the user-facing rendering, so - # a bubble is duplication, and verbose mode would dump the raw tool-call args JSON, which - # (progress queue drains on a background task) lands right under the rendered prompt. - if tool_name == "clarify": - return - - # Suppress tool-progress bubbles once the user sent `stop`: N parallel tool calls fire N - # "tool.started" events before the interrupt check, so a late `stop` would still render - # all N bubbles. (agent_holder[0] is the shared agent handle across nested scopes.) - try: - _agent_for_interrupt = ctx.agent_holder[0] if ctx.agent_holder else None - if _agent_for_interrupt is not None and getattr( - _agent_for_interrupt, "is_interrupted", False - ): - return - except Exception: - pass - - # "new" mode: only report when tool changes - if ctx.progress_mode == "new" and tool_name == ctx.last_tool[0]: - return - ctx.last_tool[0] = tool_name - - # Build progress message with primary argument preview - from agent.display import get_tool_emoji - emoji = get_tool_emoji(tool_name, default="⚙️") - - # Markdown platforms (``supports_code_blocks``) fence terminal commands; plain-text ones - # keep the compact `terminal: "cmd…"` line. No language tag: Slack mrkdwn renders it as a - # literal first code line. Verbose shows the FULL command; "all"/"new" fence but truncate to - # one line capped at ``tool_preview_length`` (default 40), the non-terminal preview budget. - _code_block_full = None - _code_block_short = None - try: - _progress_adapter = self._runner._adapter_for_source(ctx.source) - except Exception: - _progress_adapter = None - if ( - getattr(_progress_adapter, "supports_code_blocks", False) - and tool_name == "terminal" - and isinstance(args, dict) - and isinstance(args.get("command"), str) - and args["command"].strip() - ): - from agent.display import get_tool_preview_max_len - _cmd_full = args["command"].rstrip() - # Consecutive terminal calls drop the repeated "💻 terminal" header so back-to-back - # commands render as adjacent code blocks under one header. - _block_header = ( - "" if ctx.last_was_terminal_block[0] else f"{emoji} {tool_name}\n" - ) - _code_block_full = f"{_block_header}```\n{_cmd_full}\n```" - # Single-line, capped preview for non-verbose modes. - _pl = get_tool_preview_max_len() - _cap = _pl if _pl > 0 else 40 - _lines = _cmd_full.splitlines() - _cmd_short = _lines[0] if _lines else _cmd_full - _multiline = len(_lines) > 1 - if len(_cmd_short) > _cap: - _cmd_short = _cmd_short[:_cap - 3] + "..." - elif _multiline: - _cmd_short = _cmd_short + " ..." - _code_block_short = f"{_block_header}```\n{_cmd_short}\n```" - - # Verbose mode: show detailed arguments, respects tool_preview_length - if ctx.progress_mode == "verbose": - if _code_block_full is not None: - ctx.last_was_terminal_block[0] = True - ctx.progress_queue.put(_code_block_full) - return - ctx.last_was_terminal_block[0] = False - if args: - from agent.display import get_tool_preview_max_len - _pl = get_tool_preview_max_len() - args_str = json.dumps(args, ensure_ascii=False, default=str) - # tool_preview_length 0 (default) = no truncation in verbose mode; the user asked - # for full detail and platform message-length limits handle the rest. - if _pl > 0 and len(args_str) > _pl: - args_str = args_str[:_pl - 3] + "..." - msg = f"{emoji} {tool_name}({list(args.keys())})\n{args_str}" - elif preview: - msg = f"{emoji} {tool_name}: \"{preview}\"" - else: - msg = f"{emoji} {tool_name}..." - ctx.progress_queue.put(msg) - return - - # "all" / "new" modes: short preview capped by tool_preview_length (default 40; gateway - # messages persist, unlike CLI spinners). Markdown terminal commands use the fence above. - if _code_block_short is not None: - msg = _code_block_short - ctx.last_was_terminal_block[0] = True - elif preview: - from agent.display import ( - get_tool_preview_max_len, - get_tool_verb, - prepare_tool_preview, - tool_verb_connector, - verb_drops_preview, - ) - _pl = get_tool_preview_max_len() - _cap = _pl if _pl > 0 else 40 - _prepared_preview = prepare_tool_preview( - tool_name, - args, - fallback=preview, - max_len=_cap, - ) - if _progress_adapter is not None: - preview = _progress_adapter.format_tool_preview(_prepared_preview) - else: - preview = _prepared_preview.text - # Friendly labels: human-phrased line for built-in tools ("🔍 Searching the web for ...") - # by prefixing the verb onto the computed preview, so the command/url/query is kept. - _verb = get_tool_verb(tool_name) - if _verb: - if verb_drops_preview(tool_name): - msg = f"{emoji} {_verb}" - else: - msg = f"{emoji} {_verb}{tool_verb_connector(tool_name)}{preview}" - else: - msg = f"{emoji} {tool_name}: \"{preview}\"" - ctx.last_was_terminal_block[0] = False - else: - msg = f"{emoji} {tool_name}..." - ctx.last_was_terminal_block[0] = False - - # Dedup consecutive identical progress messages (common with execute_code: same - # boilerplate imports → identical previews). - if msg == ctx.last_progress_msg[0]: - ctx.repeat_count[0] += 1 - # Native-stream-progress routing: dedup updates the last line - # in the overlay rather than sending a queue signal. - _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None - if _sc is not None and getattr(_sc, "accepts_tool_progress", False): - # Replace the last progress line with the dedup version - _sc.on_tool_progress(f"{msg} (×{ctx.repeat_count[0] + 1})") - return - # Update the last line in progress_lines with a counter - # via a special "dedup" queue message. - ctx.progress_queue.put(("__dedup__", msg, ctx.repeat_count[0])) - return - ctx.last_progress_msg[0] = msg - ctx.repeat_count[0] = 0 - - # If the stream consumer is active with native streaming, inject progress into the stream - # bubble instead of the separate progress queue. - _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None - if _sc is not None and getattr(_sc, "accepts_tool_progress", False): - _sc.on_tool_progress(msg) - return - - ctx.progress_queue.put(msg) - - async def _send_native_task_card_progress(self, adapter) -> None: - """Drain the progress queue into Slack-native plan/task cards. - - On any native failure, fall back to an editable in-thread message so progress stays live. - """ - ctx = self._ctx - tasks: Dict[str, Dict[str, str]] = {} - task_order: List[str] = [] - fallback_msg_id: Optional[str] = None - native_failed = False - anonymous_seq = 0 - - def _compact(value: Any, limit: int = 120) -> str: - text = re.sub(r"\s+", " ", str(value or "")).strip() - if len(text) <= limit: - return text - return text[: limit - 3].rstrip() + "..." - - def _visible_tasks() -> List[Dict[str, str]]: - return [tasks[task_id] for task_id in task_order[-8:]] - - def _fallback_text() -> str: - labels = { - "in_progress": "running", - "complete": "complete", - "error": "error", - } - lines = [ - f"- {task['title']} - {labels.get(task['status'], task['status'])}" - for task in _visible_tasks() - ] - return "Hermes is working\n" + "\n".join(lines) - - def _apply_native_event(raw: Any) -> bool: - nonlocal anonymous_seq - if not isinstance(raw, dict): - return False - event_type = raw.get("type") - if event_type not in {"tool.started", "tool.completed"}: - return False - call_id = str(raw.get("tool_call_id") or "") - if not call_id: - anonymous_seq += 1 - call_id = f"anonymous_{anonymous_seq}" - tool_name = str(raw.get("tool_name") or "tool") - - if event_type == "tool.started": - title = tool_name - preview = _compact(raw.get("preview"), 64) - if preview: - title = f"{tool_name} - {preview}" - if call_id not in tasks: - task_order.append(call_id) - tasks[call_id] = { - "id": call_id, - "title": _compact(title), - "status": "in_progress", - } - return True - - task = tasks.get(call_id) - if task is None: - # Completion-only events are rare but valid on some runtimes; keep their real ID - # instead of guessing a same-name pending call. - task = { - "id": call_id, - "title": _compact(tool_name), - "status": "in_progress", - } - tasks[call_id] = task - task_order.append(call_id) - task["status"] = "error" if raw.get("is_error") else "complete" - return True - - async def _send_or_edit_fallback() -> None: - nonlocal fallback_msg_id - text = _fallback_text() - if fallback_msg_id: - result = await adapter.edit_message( - chat_id=ctx.source.chat_id, - message_id=fallback_msg_id, - content=text, - metadata=ctx._progress_metadata, - ) - if getattr(result, "success", False): - return - result = await adapter.send( - chat_id=ctx.source.chat_id, - content=text, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - if getattr(result, "success", False) and getattr( - result, "message_id", None - ): - fallback_msg_id = str(result.message_id) - if ctx._cleanup_progress: - ctx._cleanup_msg_ids.append(fallback_msg_id) - - async def _publish_native_progress() -> None: - nonlocal native_failed - if not tasks: - return - if not native_failed: - result = await adapter.send_native_task_card_progress( - chat_id=ctx.source.chat_id, - tasks=_visible_tasks(), - title="Hermes is working", - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - fallback_text=_fallback_text(), - ) - if getattr(result, "success", False): - return - native_failed = True - logger.warning( - "Slack native task-card progress failed; falling back " - "to an editable text update: %s", - getattr(result, "error", "unknown error"), - ) - # Once the native rail fails, every later lifecycle event - # edits the same fallback message so progress remains live. - await _send_or_edit_fallback() - - def _drain_native_queue() -> bool: - changed = False - while True: - try: - changed = _apply_native_event( - ctx.progress_queue.get_nowait() - ) or changed - except queue.Empty: - return changed - except Exception: - logger.debug( - "Slack native progress queue drain failed", - exc_info=True, - ) - return changed - - def _agent_interrupted() -> bool: - try: - _agent = ctx.agent_holder[0] if ctx.agent_holder else None - return bool( - _agent is not None and getattr(_agent, "is_interrupted", False) - ) - except Exception: - return False - - try: - while True: - if not ctx._run_still_current(): - return - try: - raw = ctx.progress_queue.get_nowait() - except queue.Empty: - await asyncio.sleep(0.1) - continue - - if _agent_interrupted(): - continue - - if _apply_native_event(raw): - await _publish_native_progress() - except asyncio.CancelledError: - if _drain_native_queue() and ctx._run_still_current(): - if not _agent_interrupted(): - await _publish_native_progress() - return - finally: - if hasattr(adapter, "stop_native_task_card_progress"): - # Best-effort on the turn-cleanup path: an escaping transport exception would skip - # final-delivery logic (cleanup awaits catch only CancelledError). - try: - await adapter.stop_native_task_card_progress( - ctx.source.chat_id, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - except asyncio.CancelledError: - raise - except Exception: - logger.debug( - "task-card stop failed during turn cleanup", - exc_info=True, - ) - - async def send_progress_messages(self): - ctx = self._ctx - if not ctx.progress_queue: - return - - adapter = self._runner._adapter_for_source(ctx.source) - if not adapter: - return - - if ctx._native_slack_task_cards and hasattr( - adapter, "send_native_task_card_progress" - ): - await self._send_native_task_card_progress(adapter) - return - - # Skip tool progress for platforms that can't edit messages (e.g. iMessage/BlueBubbles): - # each update would be a separate bubble. getattr, not attribute access: duck-typed - # adapters (test fakes, minimal plugins) may lack edit_message — treated as "can't edit". - _adapter_edit = getattr(type(adapter), "edit_message", None) - if _adapter_edit is None or _adapter_edit is BasePlatformAdapter.edit_message: - while not ctx.progress_queue.empty(): - try: - ctx.progress_queue.get_nowait() - except Exception: - break - return - - progress_lines = [] # Accumulated tool lines for the CURRENT editable bubble - progress_msg_id = None # ID of the current progress message to edit - can_edit = ctx.progress_grouping != "separate" # "separate" = one message per tool (pre-v0.9 behavior) - _last_edit_ts = 0.0 # Throttle edits to avoid Telegram flood control - _PROGRESS_EDIT_INTERVAL = 1.5 # Minimum seconds between edits - - _progress_len_fn = ( - adapter.message_len_fn - if isinstance(adapter, BasePlatformAdapter) - else len - ) - try: - _raw_progress_limit = int(getattr(adapter, "MAX_MESSAGE_LENGTH", 4000) or 4000) - except Exception: - _raw_progress_limit = 4000 - # Per-chat resolution (relay adapter fronting N platforms): cap and length unit follow the - # chat's underlying platform; native adapters return their scalar/property unchanged. - if isinstance(adapter, BasePlatformAdapter): - try: - _raw_progress_limit = int( - adapter.max_message_length_for_chat(ctx.source.chat_id) or 4000 - ) - _progress_len_fn = adapter.message_len_fn_for_chat(ctx.source.chat_id) - except Exception: - pass - # Leave a little room for platform quirks / formatting. For tiny - # test adapters keep the limit usable instead of clamping to 500+. - _PROGRESS_TEXT_LIMIT = max( - 1, - _raw_progress_limit - (64 if _raw_progress_limit > 128 else 0), - ) - - # Detect whether the adapter's edit_message accepts metadata so - # overflow edits preserve Telegram topic/thread routing (#27487). - _edit_accepts_metadata = False - if ctx._progress_metadata: - try: - _edit_params = inspect.signature(adapter.edit_message).parameters - _edit_accepts_metadata = ( - "metadata" in _edit_params - or any( - param.kind is inspect.Parameter.VAR_KEYWORD - for param in _edit_params.values() - ) - ) - except (TypeError, ValueError): - _edit_accepts_metadata = False - - async def _edit_progress_message(message_id: str, content: str): - kwargs = { - "chat_id": ctx.source.chat_id, - "message_id": message_id, - "content": content, - } - if getattr(adapter, "REQUIRES_EDIT_FINALIZE", False): - kwargs["finalize"] = True - if _edit_accepts_metadata: - kwargs["metadata"] = ctx._progress_metadata - return await adapter.edit_message(**kwargs) - - def _progress_text(lines: list) -> str: - return "\n".join(str(line) for line in lines) - - def _split_progress_groups(lines: list) -> list[list]: - """Partition progress lines into platform-sized editable bubbles.""" - groups: list[list] = [] - current: list = [] - for line in lines: - candidate = current + [line] - if current and _progress_len_fn(_progress_text(candidate)) > _PROGRESS_TEXT_LIMIT: - groups.append(current) - current = [line] - else: - current = candidate - if current: - groups.append(current) - return groups - - def _track_progress_result(result) -> None: - if ( - ctx._cleanup_progress - and getattr(result, "success", False) - and getattr(result, "message_id", None) - ): - ctx._cleanup_msg_ids.append(str(result.message_id)) - - async def _send_progress_text(text: str): - result = await adapter.send( - chat_id=ctx.source.chat_id, - content=text, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - _track_progress_result(result) - return result - - async def _roll_progress_overflow_if_needed() -> bool: - """Start fresh editable progress bubbles before a bubble exceeds limit. - - Returns True when it delivered/split the buffer or a transient edit failure left it - intact for retry — either way the caller skips the normal send/edit path this tick. - """ - nonlocal progress_msg_id, progress_lines, can_edit - if not progress_lines or not can_edit: - return False - groups = _split_progress_groups(progress_lines) - if len(groups) <= 1: - return False - - first_text = _progress_text(groups[0]) - if progress_msg_id is not None: - result = await _edit_progress_message(progress_msg_id, first_text) - if not result.success: - if getattr(result, "retryable", False): - logger.debug( - "[%s] Transient overflow edit failure — keeping can_edit=True", - adapter.name, - ) - return True - can_edit = False - # Fall back to the existing non-edit behavior below. - return False - else: - result = await _send_progress_text(first_text) - if result.success and result.message_id: - progress_msg_id = result.message_id - - for group in groups[1:]: - result = await _send_progress_text(_progress_text(group)) - if result.success and result.message_id: - progress_msg_id = result.message_id - - # The newest continuation is the only mutable bubble: keep just its lines so later - # edits update it instead of replaying the full transcript into new messages. - progress_lines = groups[-1] - return True - - while True: - try: - if not ctx._run_still_current(): - while not ctx.progress_queue.empty(): - try: - ctx.progress_queue.get_nowait() - except Exception: - break - return - - raw = ctx.progress_queue.get_nowait() - - # Drain silently when interrupted: events queued in the window between tool parse - # and interrupt processing should not render as bubbles. - try: - _agent_for_interrupt = ctx.agent_holder[0] if ctx.agent_holder else None - if _agent_for_interrupt is not None and getattr( - _agent_for_interrupt, "is_interrupted", False - ): - # Drop this event and continue draining. - await asyncio.sleep(0) - continue - except Exception: - pass - - # Handle dedup messages: update last line with repeat counter - if isinstance(raw, tuple) and len(raw) == 3 and raw[0] == "__dedup__": - _, base_msg, count = raw - if progress_lines: - progress_lines[-1] = f"{base_msg} (×{count + 1})" - msg = progress_lines[-1] if progress_lines else base_msg - elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": - # Content bubble landed — close the tool-progress bubble so the next tool starts - # fresh below it; else tool edits hit the ORIGINAL message above (out of order). - progress_msg_id = None - progress_lines = [] - ctx.last_progress_msg[0] = None - ctx.repeat_count[0] = 0 - continue - else: - msg = raw - progress_lines.append(msg) - - if await _roll_progress_overflow_if_needed(): - _last_edit_ts = time.monotonic() - await asyncio.sleep(0.3) - if ctx._run_still_current(): - await adapter.send_typing(ctx.source.chat_id, metadata=ctx._progress_metadata) - continue - - # Throttle edits: batch rapid tool updates into fewer API calls to avoid Telegram - # flood control (grammY pattern: proactively rate-limit rather than react to 429s). - _now = time.monotonic() - _remaining = _PROGRESS_EDIT_INTERVAL - (_now - _last_edit_ts) - if _remaining > 0: - # Wait out the throttle interval, then loop back to drain any further queued - # messages before sending a single batched edit. - await asyncio.sleep(_remaining) - continue - - if not ctx._run_still_current(): - return - - if can_edit and progress_msg_id is not None: - # Try to edit the existing progress message - full_text = "\n".join(progress_lines) - result = await _edit_progress_message(progress_msg_id, full_text) - if not result.success: - _err = (getattr(result, "error", "") or "").lower() - # Transient network errors (ConnectError, timeouts) must not disable editing; - # only permanent failures (flood, not found, permissions) set can_edit = False. - if getattr(result, "retryable", False): - logger.debug( - "[%s] Transient edit failure — keeping can_edit=True", - adapter.name, - ) - continue - if "flood" in _err or "retry after" in _err: - # Flood control hit — backoff but keep editing. - # Only disable edits for non-recoverable errors. - logger.info( - "[%s] Progress edit flood control, backing off", - adapter.name, - ) - _last_edit_ts = time.monotonic() - else: - can_edit = False - _flood_result = await adapter.send( - chat_id=ctx.source.chat_id, - content=msg, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - if ( - ctx._cleanup_progress - and getattr(_flood_result, "success", False) - and getattr(_flood_result, "message_id", None) - ): - ctx._cleanup_msg_ids.append(str(_flood_result.message_id)) - else: - if can_edit: - # First tool: send all accumulated text as new message - full_text = "\n".join(progress_lines) - result = await adapter.send( - chat_id=ctx.source.chat_id, - content=full_text, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - else: - # Editing unsupported: send just this line - result = await adapter.send( - chat_id=ctx.source.chat_id, - content=msg, - reply_to=ctx._progress_reply_to, - metadata=ctx._progress_metadata, - ) - if result.success and result.message_id: - progress_msg_id = result.message_id - if ctx._cleanup_progress: - ctx._cleanup_msg_ids.append(str(result.message_id)) - - _last_edit_ts = time.monotonic() - - # Restore typing indicator - await asyncio.sleep(0.3) - if ctx._run_still_current(): - await adapter.send_typing(ctx.source.chat_id, metadata=ctx._progress_metadata) - - except queue.Empty: - await asyncio.sleep(0.3) - except asyncio.CancelledError: - # Drain remaining queued messages - while not ctx.progress_queue.empty(): - try: - raw = ctx.progress_queue.get_nowait() - if isinstance(raw, tuple) and len(raw) == 3 and raw[0] == "__dedup__": - _, base_msg, count = raw - if progress_lines: - progress_lines[-1] = f"{base_msg} (×{count + 1})" - await _roll_progress_overflow_if_needed() - elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": - # Content-bubble marker during drain: close the current progress bubble - # and start a fresh one for tool lines that arrived after. - await _roll_progress_overflow_if_needed() - if can_edit and progress_lines and progress_msg_id: - _pending_text = _progress_text(progress_lines) - with suppress(Exception): - await _edit_progress_message(progress_msg_id, _pending_text) - progress_msg_id = None - progress_lines = [] - ctx.last_progress_msg[0] = None - ctx.repeat_count[0] = 0 - else: - progress_lines.append(raw) - await _roll_progress_overflow_if_needed() - except Exception: - break - # Final edit with all remaining tools (only if editing works) - if can_edit and progress_lines and progress_msg_id: - await _roll_progress_overflow_if_needed() - if can_edit and progress_lines and progress_msg_id: - full_text = _progress_text(progress_lines) - with suppress(Exception): - await _edit_progress_message(progress_msg_id, full_text) - return - except Exception as e: - logger.error("Progress message error: %s", e) - await asyncio.sleep(1) - - def voice_ack_callback(self, call_id, tool_name, args): - """tool_start_callback: speak a one-time ack in the voice channel.""" - ctx = self._ctx - if ctx._voice_ack_fired[0] or ctx._voice_ack_guild[0] is None: - return - if not ctx._run_still_current(): - return - ctx._voice_ack_fired[0] = True - _adapter = self._runner.adapters.get(Platform.DISCORD) - if _adapter is None or not hasattr(_adapter, "play_ack_in_voice"): - return - try: - safe_schedule_threadsafe( - _adapter.play_ack_in_voice(ctx._voice_ack_guild[0]), - ctx._voice_ack_loop, - logger=logger, - log_message="voice ack scheduling error", - ) - except Exception as _ack_err: - logger.debug("voice ack schedule failed: %s", _ack_err) - - # ── Slack-native task cards: ID-bearing lifecycle callbacks ── ride agent.tool_start_callback / - # agent.tool_complete_callback so start/completion correlate by the REAL tool-call id; the - # name-correlated progress_callback text events would duplicate cards and mispair concurrent calls. - - def native_tool_start_callback(self, call_id, tool_name, args): - """Queue an ID-correlated native progress start from the agent thread.""" - ctx = self._ctx - if not ctx.progress_queue or not ctx._run_still_current(): - return - try: - _agent = ctx.agent_holder[0] if ctx.agent_holder else None - if _agent is not None and getattr(_agent, "is_interrupted", False): - return - except Exception: - pass - from agent.display import build_tool_preview - - ctx.progress_queue.put( - { - "type": "tool.started", - "tool_call_id": str(call_id or ""), - "tool_name": str(tool_name or "tool"), - "preview": build_tool_preview( - str(tool_name or "tool"), args or {}, max_len=64 - ) - or "", - } - ) - - def native_tool_complete_callback(self, call_id, tool_name, args, result): - """Queue the matching native completion using the real tool-call ID.""" - ctx = self._ctx - if not ctx.progress_queue or not ctx._run_still_current(): - return - try: - _agent = ctx.agent_holder[0] if ctx.agent_holder else None - if _agent is not None and getattr(_agent, "is_interrupted", False): - return - except Exception: - pass - from agent.display import _detect_tool_failure - - is_error, _ = _detect_tool_failure(str(tool_name or "tool"), result) - ctx.progress_queue.put( - { - "type": "tool.completed", - "tool_call_id": str(call_id or ""), - "tool_name": str(tool_name or "tool"), - "is_error": bool(is_error), - } - ) - - def combined_tool_start_callback(self, call_id, tool_name, args): - """Compose the voice ack + native task-card start consumers.""" - ctx = self._ctx - if ctx._voice_ack_guild[0] is not None: - self.voice_ack_callback(call_id, tool_name, args) - if ctx._native_slack_task_cards: - self.native_tool_start_callback(call_id, tool_name, args) - - def _step_callback_sync(self, iteration: int, prev_tools: list) -> None: - ctx = self._ctx - if not ctx._run_still_current(): - return - # prev_tools may be list[str] or list[dict] with "name"/"result" keys. Normalise so - # "tool_names" stays backward-compatible for user hooks that do ', '.join(tool_names). - _names: list[str] = [] - for _t in (prev_tools or []): - if isinstance(_t, dict): - _names.append(_t.get("name") or "") - else: - _names.append(str(_t)) - safe_schedule_threadsafe( - ctx._hooks_ref.emit("agent:step", { - "platform": ctx.source.platform.value if ctx.source.platform else "", - "user_id": ctx.source.user_id, - "session_id": ctx.session_id, - "iteration": iteration, - "tool_names": _names, - "tools": prev_tools, - }), - ctx._loop_for_step, - logger=logger, - log_message="agent:step hook scheduling error", - ) - - def _event_callback_sync(self, event_type: str, context: dict) -> None: - ctx = self._ctx - try: - asyncio.run_coroutine_threadsafe( - ctx._hooks_ref.emit(event_type, context), - ctx._loop_for_step, - ) - except Exception as _e: - logger.debug("event_callback hook error: %s", _e) - - def _attach_session_title_callback(self, agent, ctx) -> None: - """Wire the platform thread-rename lane onto the agent as `_on_session_title`. - - The titler runs in the turn prologue, so attach before the run, not after it. - """ - try: - # Gateway auto-title failures are not user-actionable, so never surface them as messages; - # overriding the failure sink keeps CLI on _emit_auxiliary_failure while gateway logs debug. - def _title_failure_cb(task: str, exc: BaseException) -> None: - logger.debug( - "Gateway auto-title failure suppressed (not user-visible): %s: %s", - task, exc, - ) - - agent._title_failure_callback = _title_failure_cb - - session_id = getattr(agent, "session_id", None) - source = ctx.source - - # Both lanes spend a rate-limited platform call per title, so they use the model's title - # only (TitleCallback); renaming twice burns Discord's 2-per-10-min budget on a throwaway. - if self._runner._is_telegram_topic_lane(source): - agent._on_session_title = lambda title, title_source: ( - title_source == "llm" - and self._runner._schedule_telegram_topic_title_rename( - source, session_id, title, - ) - ) - elif self._runner._is_discord_auto_thread_lane(source) or ( - self._runner._is_relay_discord_channel_lane(source) - ): - # Relay note: the second predicate is shape-only (relay Discord channel event). - # Whether the connector auto-threaded our reply is only knowable AFTER delivery, so - # the callback must be registered eagerly and the rename lane does the cache lookup - # at fire time — gating registration on the cache read meant it never registered. - agent._on_session_title = lambda title, title_source: ( - title_source == "llm" - and self._runner._schedule_discord_semantic_thread_rename( - source, session_id, title, - ) - ) - except Exception: - logger.debug("Failed to attach session title callback", exc_info=True) - - def _status_callback_sync(self, event_type: str, message: str) -> None: - ctx = self._ctx - if not ctx._status_adapter or not ctx._run_still_current(): - return - prepared_message = _prepare_gateway_status_message( - ctx.source.platform, - event_type, - message, - ) - if prepared_message is None: - logger.debug( - "status_callback suppressed for %s/%s: %s", - ctx.source.platform.value if ctx.source.platform else "unknown", - event_type, - _redact_gateway_user_facing_secrets(str(message or ""))[:160], - ) - return - _fut = safe_schedule_threadsafe( - _send_or_update_status_coro(ctx._status_adapter, ctx._status_chat_id, event_type, prepared_message, ctx._status_thread_metadata), - ctx._loop_for_step, - logger=logger, - log_message=f"status_callback ({event_type}) scheduling error", - ) - if _fut is None: - return - if ctx._cleanup_progress: - def _track_status_id(fut) -> None: - try: - res = fut.result() - except Exception: - return - mid = getattr(res, "message_id", None) - if getattr(res, "success", False) and mid: - ctx._cleanup_msg_ids.append(str(mid)) - _fut.add_done_callback(_track_status_id) - - def run_sync(self): - ctx = self._ctx - # As a method the turn message lives on the shared TurnContext: every rebind writes - # `ctx.message`, so the outer `_run_agent_inner` body sees the update as via the closure cell. - - # session_key propagates via contextvars (_set_session_env / set_current_session_key): - # concurrency-safe and inherited by tool worker threads. Deliberately do NOT write - # os.environ["HERMES_SESSION_KEY"]: it is process-global, so concurrent sessions would clobber - # each other and a tool thread with an unset contextvar would read the wrong key, misrouting - # approvals. Only the TUI slash-worker subprocess exports the env var (from its own argv). - - # Map platform enum to the platform hint key the agent understands. - # Platform.LOCAL ("local") maps to "cli"; others pass through as-is. - platform_key = "cli" if ctx.source.platform == Platform.LOCAL else ctx.source.platform.value - - # Combine platform context, YAML channel_prompts hint for this chat, channel_overrides - # system_prompt (or global ephemeral), and the gateway ephemeral prompt. - combined_ephemeral = ctx.context_prompt or "" - event_channel_prompt = (ctx.channel_prompt or "").strip() - if event_channel_prompt: - combined_ephemeral = (combined_ephemeral + "\n\n" + event_channel_prompt).strip() - cfg_channel_prompt = self._runner._get_system_prompt_for_channel( - ctx.source.platform, - ctx.source.chat_id or "", - thread_id=getattr(ctx.source, "thread_id", None), - parent_id=getattr(ctx.source, "parent_chat_id", None), - ) - if cfg_channel_prompt: - combined_ephemeral = (combined_ephemeral + "\n\n" + cfg_channel_prompt).strip() - - max_iterations = _current_max_iterations() - - try: - model, runtime_kwargs = self._runner._resolve_session_agent_runtime( - source=ctx.source, - session_key=ctx.session_key, - user_config=ctx.user_config, - ) - logger.debug( - "run_agent resolved: model=%s provider=%s session=%s", - model, runtime_kwargs.get("provider"), ctx.session_key or "", - ) - except Exception as exc: - return { - "final_response": f"⚠️ Provider authentication failed: {exc}", - "messages": [], - "api_calls": 0, - "tools": [], - } - - pr = self._runner._provider_routing - reasoning_config = self._runner._resolve_session_reasoning_config( - source=ctx.source, - session_key=ctx.session_key, - model=model, - ) - self._runner._reasoning_config = reasoning_config - self._runner._service_tier = self._runner._resolve_session_service_tier( - source=ctx.source, session_key=ctx.session_key - ) - # Set up stream consumer for token streaming or interim commentary. - _stream_consumer = None - _stream_delta_cb = None - # streaming TTS consumer is created on the outer event-loop thread before run_sync launches. - # run_sync only reads it via ``streaming_tts_consumer_holder[0]`` for delta callback wiring. - _stts_consumer_ref = ctx.streaming_tts_consumer_holder[0] - _scfg = getattr(getattr(self._runner, 'config', None), 'streaming', None) - if _scfg is None: - from gateway.config import StreamingConfig - _scfg = StreamingConfig() - - # Per-platform streaming gate: display.platforms..streaming can disable streaming - # for specific platforms even when the global streaming config is enabled. - _plat_streaming = ctx.resolve_display_setting( - ctx.user_config, platform_key, "streaming" - ) - # None = no per-platform override → follow global config - _streaming_enabled = ( - _scfg.enabled and _scfg.transport != "off" - if _plat_streaming is None - else bool(_plat_streaming) - ) - _want_stream_deltas = _streaming_enabled - _want_interim_messages = ctx.interim_assistant_messages_enabled - _want_interim_consumer = _want_interim_messages - if _want_stream_deltas or _want_interim_consumer: - try: - from gateway.stream_consumer import GatewayStreamConsumer - _adapter = self._runner._adapter_for_source(ctx.source) - if _adapter: - _consumer_cfg, _pause_typing_before_finalize = ( - self._runner._build_stream_consumer_config( - ctx.source, _scfg, _adapter, - on_missing_cursor="raise", - ) - ) - _stream_consumer = GatewayStreamConsumer( - adapter=_adapter, - chat_id=ctx.source.chat_id, - config=_consumer_cfg, - metadata=ctx._status_thread_metadata, - on_new_message=( - (lambda: ctx.progress_queue.put(("__reset__",))) - if ctx.progress_queue is not None - else None - ), - on_before_finalize=_pause_typing_before_finalize, - initial_reply_to_id=ctx.event_message_id, - run_still_current=ctx._run_still_current, - ) - if _want_stream_deltas: - def _stream_delta_cb(text: str) -> None: - if ctx._run_still_current(): - _stream_consumer.on_delta(text) - # Tee to the streaming-TTS consumer (#60671). - if _stts_consumer_ref is not None: - _stts_consumer_ref.on_delta(text) - ctx.stream_consumer_holder[0] = _stream_consumer - except Exception as _sc_err: - logger.debug("Could not set up stream consumer: %s", _sc_err) - - # Text streaming off but streaming TTS active: install a TTS-only delta callback so the - # consumer still receives LLM deltas for audio synthesis. - if _stream_delta_cb is None and _stts_consumer_ref is not None: - def _stream_delta_cb(text: str) -> None: - if ctx._run_still_current(): - _stts_consumer_ref.on_delta(text) - - def _interim_assistant_cb(text: str, *, already_streamed: bool = False) -> None: - if not ctx._run_still_current(): - return - display_text = text - if _stream_consumer is not None: - if already_streamed: - _stream_consumer.on_segment_break() - else: - _stream_consumer.on_commentary(display_text) - return - if already_streamed or not ctx._status_adapter or not str(display_text or "").strip(): - return - safe_schedule_threadsafe( - ctx._status_adapter.send( - ctx._status_chat_id, - display_text, - metadata=ctx._status_thread_metadata, - ), - ctx._loop_for_step, - logger=logger, - log_message="interim_assistant_callback scheduling error", - ) - - turn_route = self._runner._resolve_turn_agent_config(ctx.message, model, runtime_kwargs) - - # Per-platform skip_context_files — messaging platforms can opt out of filesystem-heavy - # context-file discovery (SOUL.md, AGENTS.md, .cursorrules) to cut AIAgent build latency. - _platforms_gw_cfg = (ctx.user_config.get("gateway") or {}).get("platforms") or {} - # ``hermes gateway setup`` writes ``gateway.platforms`` as a LIST of enabled platform names, - # not a dict; treat any non-dict shape as "no per-platform overrides" rather than crashing. - if not isinstance(_platforms_gw_cfg, dict): - _platforms_gw_cfg = {} - _plat_gw_cfg = _platforms_gw_cfg.get(platform_key) or {} - _skip_context = _plat_gw_cfg.get("skip_context_files") - skip_context_files = bool(_skip_context) if _skip_context is not None else False - - # Agent cache: reuse this session's previous AIAgent to preserve the frozen system prompt - # and tool schemas for prompt cache hits. - _sig = self._runner._agent_config_signature( - turn_route["model"], - turn_route["runtime"], - ctx.enabled_toolsets, - combined_ephemeral, - cache_keys=self._runner._extract_cache_busting_config(ctx.user_config), - user_id=getattr(ctx.source, "user_id", None), - user_id_alt=getattr(ctx.source, "user_id_alt", None), - skip_context_files=skip_context_files, - ) - agent = None - reused_cached_agent = False - _cache_lock = getattr(self._runner, "_agent_cache_lock", None) - _cache = getattr(self._runner, "_agent_cache", None) - - # Peek at the cached entry's snapshot session_id so we can check, OUTSIDE the cache lock, - # whether it is a DEAD session in state.db. "cached sid != current sid" normally means an - # intentional switch (reuse the agent), but the routing-key self-heal yields the same shape - # with an agent bound to a DEAD session; reusing it re-binds the dead sid and loops. - _peek_cached_sid = None - if _cache_lock and _cache is not None: - with _cache_lock: - _peek_entry = _cache.get(ctx.session_key) - if _peek_entry and len(_peek_entry) > 3: - _peek_cached_sid = _peek_entry[3] - _cached_sid_is_dead = False - if ( - _peek_cached_sid is not None - and ctx.session_id is not None - and _peek_cached_sid != ctx.session_id - ): - try: - _cached_sid_is_dead = self._runner.session_store._is_session_ended_in_db( - _peek_cached_sid - ) - except Exception: - _cached_sid_is_dead = False - - # Cross-process write guard: another process (e.g. hermes dashboard) appending to the same - # SessionDB session makes the cached agent's transcript stale. On message_count mismatch vs - # the count recorded at cache time, invalidate so a fresh agent re-reads from disk. - _current_msg_count = None - if self._runner._session_db is not None and ctx.session_id: - try: - # run_sync is off-loop (executor); sync DB is fine. - _sess_row = self._runner._session_db._db.get_session(ctx.session_id) - if _sess_row: - _current_msg_count = _sess_row.get("message_count", 0) - except Exception: - pass - - _xproc_evicted_agent = None - if _cache_lock and _cache is not None: - with _cache_lock: - cached = _cache.get(ctx.session_key) - if cached and cached[1] == _sig: - # cached[2] is the message_count at cache time; stale when a second process - # appended rows. cached[3] (when present) is the session_id the snapshot was - # taken for — used to skip the guard when the active session_id differs. - _cached_mc = cached[2] if len(cached) > 2 else None - _cached_sid = cached[3] if len(cached) > 3 else None - # Snapshot from a different session_id (same session_key, other conversation): the - # counts track DIFFERENT DB rows, so the comparison is meaningless. REUSE the cached - # agent rather than rebuild and bust the prompt cache on every session switch. - _session_id_mismatch = ( - _cached_sid is not None - and ctx.session_id is not None - and _cached_sid != ctx.session_id - ) - # Re-validate the OUTSIDE-lock dead-session peek against the tuple read under THIS - # lock: the entry may have been replaced between peek and acquisition, and a stale - # "dead" verdict must never be applied to a different (possibly live) cached agent. - _stale_dead_sid_reuse = ( - _session_id_mismatch - and _cached_sid_is_dead - and _cached_sid == _peek_cached_sid - ) - if _stale_dead_sid_reuse: - # The routing key was just self-healed away from a session state.db marked - # ended, but this cached AIAgent still belongs to that DEAD session_id. - # Reusing it would re-bind the dead sid and undo the self-heal; rebuild fresh. - logger.info( - "Agent cache invalidated for session %s: " - "cached agent's session_id %s is ended in " - "state.db (stale self-heal artifact, " - "#54878 x #54947) — discarding instead of " - "reusing across the routing recovery", - ctx.session_key, _cached_sid, - ) - evicted = self._runner._agent_cache.pop(ctx.session_key, None) - _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None - if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: - # Same deferred-cleanup rationale as the cross-process branch below: don't - # block the event loop / cache lock on memory-provider or socket teardown. - _xproc_evicted_agent = _ev_agent - elif ( - not _session_id_mismatch - and _cached_mc is not None - and _current_msg_count is not None - and _current_msg_count != _cached_mc - ): - # Cross-process write detected — discard stale - # agent so it rebuilds from fresh DB transcript. - logger.info( - "Agent cache invalidated for session %s: " - "message_count changed (%s -> %s), " - "possible cross-process write", - ctx.session_key, _cached_mc, _current_msg_count, - ) - evicted = self._runner._agent_cache.pop(ctx.session_key, None) - _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None - if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: - # Defer cleanup until AFTER the lock is released: release_clients can - # block on memory-provider/socket teardown, stalling the event loop while - # the idle sweeper waits on this lock (blocking Discord heartbeats). The - # session rebuilds a fresh agent below, so use the SOFT release that keeps - # its terminal sandbox / browser / bg processes for the new agent to - # inherit — mirrors _evict_cached_agent / idle-sweep. - _xproc_evicted_agent = _ev_agent - else: - agent = cached[0] - # Refresh LRU order so the cap enforcement evicts - # truly-oldest entries, not the one we just used. - if hasattr(_cache, "move_to_end"): - with suppress(KeyError): - _cache.move_to_end(ctx.session_key) - self._runner._init_cached_agent_for_turn(agent, ctx._interrupt_depth) - # Refresh agent max_iterations from current config - # (cached agent may have been created with old config) - agent.max_iterations = max_iterations - logger.debug("Reusing cached agent for session %s", ctx.session_key) - reused_cached_agent = True - - # Lock released — refresh the reused agent's fallback chain from disk OUTSIDE the cache lock - # (disk I/O under the lock stalls the idle-sweep watcher and Discord heartbeats). A chain - # configured after caching must reach the next turn; per-session serialization keeps it safe. - if reused_cached_agent and agent is not None: - self._runner._apply_fallback_chain_to_agent( - agent, self._runner._refresh_fallback_model(), - ) - - # Lock released — schedule cleanup of any cross-process-evicted agent on a daemon thread so - # memory-provider/socket teardown never blocks the gateway loop or the expiry watcher's lock. - if _xproc_evicted_agent is not None: - try: - threading.Thread( - target=self._runner._release_evicted_agent_soft, - args=(_xproc_evicted_agent,), - daemon=True, - name=f"agent-xproc-evict-{str(ctx.session_key)[:24]}", - ).start() - except Exception: - # Interpreter shutdown or thread-spawn failure — release - # inline as a best-effort fallback. - with suppress(Exception): - self._runner._release_evicted_agent_soft(_xproc_evicted_agent) - - if agent is None: - # Config changed or first message — create fresh agent - agent = ctx.AIAgent( - model=turn_route["model"], - **turn_route["runtime"], - **_checkpoint_agent_kwargs(ctx.user_config), - max_iterations=max_iterations, - quiet_mode=True, - verbose_logging=False, - enabled_toolsets=ctx.enabled_toolsets, - disabled_toolsets=ctx.disabled_toolsets, - ephemeral_system_prompt=combined_ephemeral or None, - prefill_messages=self._runner._prefill_messages or None, - reasoning_config=reasoning_config, - service_tier=self._runner._service_tier, - request_overrides=turn_route.get("request_overrides"), - providers_allowed=pr.get("only"), - providers_ignored=pr.get("ignore"), - providers_order=pr.get("order"), - provider_sort=pr.get("sort"), - provider_require_parameters=pr.get("require_parameters", False), - provider_data_collection=pr.get("data_collection"), - session_id=ctx.session_id, - platform=platform_key, - user_id=ctx.source.user_id, - user_id_alt=ctx.source.user_id_alt, - user_name=ctx.source.user_name, - chat_id=ctx.source.chat_id, - chat_name=ctx.source.chat_name, - chat_type=ctx.source.chat_type, - thread_id=ctx.source.thread_id, - gateway_session_key=ctx.session_key, - session_db=getattr(self._runner._session_db, "_db", self._runner._session_db), - # Reload from disk — do not reuse the startup snapshot (#60955). - fallback_model=self._runner._refresh_fallback_model(), - skip_context_files=skip_context_files, - # Keep the persona even with minimal context: soul identity is - # a single small file, not part of the expensive walk. - load_soul_identity=True, - ) - if _cache_lock and _cache is not None: - with _cache_lock: - # Record the snapshot's session_id with message_count so the cross-process guard can - # skip the meaningless count comparison if the active session_id later switches. - _cache[ctx.session_key] = ( - agent, _sig, _current_msg_count, ctx.session_id, - ) - self._runner._enforce_agent_cache_cap() - logger.debug("Created new agent for session %s (sig=%s)", ctx.session_key, _sig) - - # Per-message state — callbacks and reasoning config change every turn, so they aren't baked - # into the cached agent. The progress callback is ALWAYS attached (never gated to None): its - # body gates each event class, and subagent-failure notices must fire even with - # tool_progress/thinking off — a None gate made dead subagents vanish silently. - agent.tool_progress_callback = ctx.progress_callback - # Compose ID-bearing lifecycle consumers: Discord's one-time voice ack and Slack's task cards - # both ride the authoritative start callback, so neither infers identity from tool names. - _combined_start_cb = ctx.native_tool_start_callback or ctx.voice_ack_callback - agent.tool_start_callback = ( - _combined_start_cb - if ( - ctx._voice_ack_guild[0] is not None - or ctx._native_slack_task_cards - ) - else None - ) - agent.tool_complete_callback = ( - ctx.native_tool_complete_callback - if ctx._native_slack_task_cards - and ctx.native_tool_complete_callback is not None - else None - ) - agent.step_callback = ctx._step_callback_sync if ctx._hooks_ref.loaded_hooks else None - agent.stream_delta_callback = _stream_delta_cb - agent.interim_assistant_callback = _interim_assistant_cb if _want_interim_messages else None - agent.status_callback = ctx._status_callback_sync - # Credits / out-of-band notices (usage bands, depletion, restored) fire from the agent's sync - # worker thread, so hop onto the gateway loop via safe_schedule_threadsafe. Fired-once latch - # lives on the cached agent (no per-turn re-nag); clear is a no-op — sends can't be retracted. - def _notice_callback_sync(notice) -> None: - if not ctx._status_adapter or not ctx._run_still_current(): - return - try: - line = render_notice_line(notice) - except Exception: - logger.debug("render_notice_line failed", exc_info=True) - return - if not line: - return - safe_schedule_threadsafe( - self._runner._deliver_platform_notice(ctx.source, line), - ctx._loop_for_step, - logger=logger, - log_message="notice_callback delivery scheduling error", - ) - - agent.notice_callback = _notice_callback_sync - agent.notice_clear_callback = None - agent.event_callback = ctx._event_callback_sync - agent.reasoning_config = reasoning_config - agent.service_tier = self._runner._service_tier - # Merge, never overwrite: init-time request overrides (e.g. a custom provider's extra_body - # merged at agent construction) must survive every reused-agent turn. Drop only the PREVIOUS - # turn's routing overrides before layering this turn's, so stale per-turn values never linger. - request_overrides = dict(getattr(agent, "request_overrides", {}) or {}) - previous_turn_overrides = dict( - getattr(agent, "_gateway_turn_request_overrides", {}) or {} - ) - for key, value in previous_turn_overrides.items(): - if request_overrides.get(key) == value: - request_overrides.pop(key, None) - turn_request_overrides = dict(turn_route.get("request_overrides") or {}) - request_overrides.update(turn_request_overrides) - agent.request_overrides = request_overrides - agent._gateway_turn_request_overrides = turn_request_overrides - # Must-deliver notes for THIS turn ride the current user message (api_content sidecar), never - # the system prompt: staged by _handle_message_with_agent (auto-reset, first-contact intro, - # voice-channel change). Assigned unconditionally so a reused agent never replays a stale note. - agent._gateway_turn_context_notes = "\n\n".join( - self._runner._consume_pending_turn_sidecar_notes(ctx.session_key) - ) - - _bg_review_release = threading.Event() - _bg_review_pending: list[str] = [] - _bg_review_pending_lock = threading.Lock() - - def _deliver_bg_review_message(message: str) -> None: - if not ctx._status_adapter or not ctx._run_still_current(): - return - safe_schedule_threadsafe( - ctx._status_adapter.send( - ctx._status_chat_id, - message, - metadata=_interim_metadata(_non_conversational_metadata(ctx._status_thread_metadata, platform=ctx.source.platform)), - ), - ctx._loop_for_step, - logger=logger, - log_message="background_review_callback scheduling error", - ) - - def _release_bg_review_messages() -> None: - _bg_review_release.set() - with _bg_review_pending_lock: - pending = list(_bg_review_pending) - _bg_review_pending.clear() - for queued in pending: - _deliver_bg_review_message(queued) - - # Background review delivery — send "💾 Memory updated" etc. to user - def _bg_review_send(message: str) -> None: - if not ctx._status_adapter or not ctx._run_still_current(): - return - if not _bg_review_release.is_set(): - with _bg_review_pending_lock: - if not _bg_review_release.is_set(): - _bg_review_pending.append(message) - return - _deliver_bg_review_message(message) - - agent.background_review_callback = _bg_review_send - # Register the release hook on the adapter so base.py's finally - # block can fire it after delivering the main response. - if ctx._status_adapter and ctx.session_key: - if getattr(type(ctx._status_adapter), "register_post_delivery_callback", None) is not None: - ctx._status_adapter.register_post_delivery_callback( - ctx.session_key, - _release_bg_review_messages, - generation=ctx.run_generation, - ) - else: - _pdc = getattr(ctx._status_adapter, "_post_delivery_callbacks", None) - if _pdc is not None: - _pdc[ctx.session_key] = _release_bg_review_messages - # Memory update notifications in chat. Config: display.memory_notifications - # off — no chat notification (still logged to stdout) - # on — generic "💾 Memory updated" (default) - # verbose — content preview: "💾 Memory ➕ Hermes Repo..." - _mem_notif = ctx.user_config.get("display", {}).get("memory_notifications") - if isinstance(_mem_notif, bool): - _mem_notif = "on" if _mem_notif else "off" - agent.memory_notifications = str(_mem_notif).lower() if _mem_notif else "on" - - # ------------------------------------------------------------------ - # Shared native-stream boundary close: for native-streaming platforms (e.g. WeCom), an - # interrupting interaction (approval or clarify prompt) must finalize the current stream - # and disable native streaming first, or post-interaction output keeps updating the OLD - # bubble above the prompt. Runs on the agent thread; the consumer serializes via its queue. - def _close_native_stream_boundary( - _reason: str, _placeholder: str | None = None, _reopen: bool = False, - ) -> bool: - _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None - if not (_sc and getattr(_sc, "_use_native_streaming", False)): - return True - _cancelled_flag = None - try: - _boundary_result = _sc.close_for_approval_prompt( - _placeholder, reason=_reason, reopen=_reopen, - ) - # Returns (future, cancelled_flag) or just a future. - if isinstance(_boundary_result, tuple): - _boundary_future, _cancelled_flag = _boundary_result - else: - _boundary_future = _boundary_result - if hasattr(_boundary_future, "result"): - _ok = _boundary_future.result(timeout=10) - if not _ok: - logger.warning( - "%s boundary failed to close stream properly — " - "prompt may still appear in typing bubble", _reason, - ) - return bool(_ok) - return True - except (TimeoutError, Exception) as _boundary_err: - if _cancelled_flag is not None: - _cancelled_flag["cancelled"] = True - logger.warning( - "%s boundary timed out or failed: %s", _reason, _boundary_err, - ) - return False - - # ------------------------------------------------------------------ - # Clarify callback: present a clarify prompt and block on a response. Runs on the agent's - # worker thread (clarify_tool's synchronous contract): schedules the adapter's send_clarify - # on the gateway loop, then blocks on the primitive's threading.Event with a timeout. - # Returns the response string, or a sentinel explaining no response arrived. - # ------------------------------------------------------------------ - def _clarify_callback_sync(question: str, choices, multi_select: bool = False) -> str: - from tools import clarify_gateway as _clarify_mod - import uuid as _uuid - - if not ctx._status_adapter: - return "" - - clarify_id = _uuid.uuid4().hex[:10] - _clarify_mod.register( - clarify_id=clarify_id, - session_key=ctx.session_key or "", - question=question, - choices=list(choices) if choices else None, - multi_select=bool(multi_select), - ) - - # WeCom native streaming: finalize the current stream before the clarify prompt so the - # post-answer output opens a fresh bubble below the question ("气泡割裂" otherwise). Unlike - # approval, clarify passes reopen=True so the continuation re-opens a native stream; if - # the re-seed fails the consumer degrades to send() automatically. - _close_native_stream_boundary( - "Clarify", "💬 等待你的选择...", _reopen=True, - ) - - # Pause typing — as with approval, a "thinking..." status must not obscure the prompt or - # block an "Other" reply on platforms that disable input while typing (Slack Assistant). - with suppress(Exception): - ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) - - # Ordering barrier: flush buffered assistant prose to the platform BEFORE sending the - # poll, which goes out on a separate agent-thread-blocking path and would otherwise - # render ABOVE its own explanation. Best-effort + short timeout so the agent thread - # never hangs if the consumer task isn't running. - try: - _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None - _flush = getattr(_sc, "flush_pending_sync", None) - if callable(_flush): - _flush(timeout=3.0) - except Exception: - logger.debug( - "Stream-consumer flush before clarify prompt failed", - exc_info=True, - ) - - fut = safe_schedule_threadsafe( - ctx._status_adapter.send_clarify( - chat_id=ctx._status_chat_id, - question=question, - choices=list(choices) if choices else None, - clarify_id=clarify_id, - session_key=ctx.session_key or "", - metadata=ctx._status_thread_metadata, - ), - ctx._loop_for_step, - logger=logger, - log_message="Clarify send failed to schedule", - ) - # Boundary rule (see _approval_send_outcome): a send timeout is AMBIGUOUS — the card may - # have posted with a late ack. Only a definitive failure tears down the registration; - # ambiguous falls through to the bounded wait so a late reply resolves. - _clarify_response = _clarify_send_then_wait( - fut, - clarify_id=clarify_id, - session_key=ctx.session_key or "", - clarify_mod=_clarify_mod, - ) - # Only re-arm typing when the user actually answered — the undeliverable sentinel and the - # timeout/cancellation strings start with '[' and must pass through untouched. - if not ( - isinstance(_clarify_response, str) - and _clarify_response.startswith("[") - ): - # User answered: reopen typing IMMEDIATELY, not on the LLM's first post-answer token - # (native streaming otherwise re-seeds lazily on the first delta: ~48s of dead air). - # request_reopen_seed is a no-op outside the reopen-pending native state; always safe. - _sc_reopen = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None - if _sc_reopen is not None: - try: - _sc_reopen.request_reopen_seed() - except Exception: - logger.debug( - "request_reopen_seed after clarify answer failed", - exc_info=True, - ) - try: - ctx._status_adapter.resume_typing_for_chat(ctx._status_chat_id) - except Exception: - logger.debug( - "resume_typing_for_chat after clarify answer failed", - exc_info=True, - ) - return _clarify_response - - agent.clarify_callback = _clarify_callback_sync - - # Show assistant thinking between tool calls — independent of tool_progress mode. Mattermost - # needs an explicit per-platform opt-in so global scratch-text doesn't leak into threads. - agent.thinking_progress = ctx._thinking_enabled - # Store agent reference for interrupt support - ctx.agent_holder[0] = agent - # Wire the platform thread-rename lane onto the agent: the titler fires from the turn prologue, - # not after the response, so titles are pushed the moment they land. - self._attach_session_title_callback(agent, ctx) - # Publish turn ownership for explicit /stop, /new, disconnect, and shutdown interrupts. - # Older session processes are outside this baseline and remain alive. - agent._gateway_turn_process_task_id = ctx.process_task_id - agent._gateway_turn_process_baseline = ctx.process_baseline - # Capture the full tool definitions for transcript logging - ctx.tools_holder[0] = agent.tools if hasattr(agent, 'tools') else None - - # Convert history to agent format. Transcript path: {role, content, timestamp} dicts — strip - # timestamps. Interrupt path (agent result["messages"]): full agent messages with - # tool_calls/tool_call_id/reasoning — pass through intact so the API sees valid assistant→tool - # sequences (dropping tool_calls causes 500s). Telegram observed group context: observed=True - # rows are withheld from replayable history and attached to the current addressed message as - # API-only context, so persisted history stores only the real addressed user turn. - agent_history, observed_group_context = _build_gateway_agent_history( - ctx.history, - channel_prompt=ctx.channel_prompt, - inject_timestamps=_message_timestamps_enabled(ctx.user_config), - ) - - # FTS write-corruption guard: if persistence failed silently via corrupt FTS triggers, the - # reloaded transcript is stale/empty while the SAME cached agent still holds the full live - # conversation in `_session_messages`; replacing it causes same-session amnesia. Only for - # a reused agent bound to this exact session_id. - if reused_cached_agent and getattr(agent, "session_id", None) == ctx.session_id: - _selected = _select_cached_agent_history( - agent_history, getattr(agent, "_session_messages", None) - ) - if _selected is not agent_history: - logger.warning( - "Persisted transcript lagged live cached history for " - "session %s (disk=%d, memory=%d); preserving live " - "conversation context (possible FTS write corruption)", - ctx.session_key, len(agent_history), len(_selected), - ) - # The live in-memory history bypassed the _build_gateway_agent_history cleanup above — - # re-apply the stale-confirmation expiry so a dangerous confirmation can't slip through. - agent_history = strip_stale_dangerous_confirmations( - _selected, now=time.time() - ) - - # Collect MEDIA paths already in history to exclude them from this turn's extraction. - # Compression-safe: even if the message list shrinks, we know which paths are old. - _history_media_paths: set = _collect_history_media_paths(agent_history) - - # Per-session gateway approval callback: dangerous-command approval blocks the agent thread - # (mirrors CLI input()); the callback bridges sync→async to send the request immediately. - from tools.approval import ( - register_gateway_notify, - reset_current_session_key, - set_current_session_key, - unregister_gateway_notify, - ) - - def _approval_notify_sync(approval_data: dict) -> None: - """Send the approval request to the user from the agent thread. - - Uses the adapter's interactive button approvals (e.g. ``send_exec_approval``) when - available, else a plain text message with ``/approve`` instructions. - """ - # Pause typing while awaiting approval: Slack's assistant_threads_setStatus disables the - # compose box, so the user can't type /approve while "is thinking..." shows. The approval - # send auto-clears it; pausing stops _keep_typing re-setting it. Resumed in approve/deny. - ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) - - # WeCom native streaming: ask the stream consumer to close the current stream before the - # approval prompt — via the consumer's queue, so it serializes with pending deltas. - _close_native_stream_boundary("Approval") - - cmd = approval_data.get("command", "") - desc = approval_data.get("description", "dangerous command") - - # Redact credentials from the command before display — Tirith's findings are already - # redacted, but the raw command string still leaks secrets to the chat platform. Done - # here so BOTH the button-based and plain-text fallback paths use the redacted value. - cmd = _redact_approval_command(cmd) - - # Prefer button-based approval when the adapter supports it. Check the *class*, not the - # instance — avoids false positives from MagicMock auto-attribute creation in tests. - if getattr(type(ctx._status_adapter), "send_exec_approval", None) is not None: - try: - _approval_fut = safe_schedule_threadsafe( - ctx._status_adapter.send_exec_approval( - chat_id=ctx._status_chat_id, - command=cmd, - session_key=_approval_session_key, - description=desc, - metadata=ctx._status_thread_metadata, - allow_permanent=approval_data.get("allow_permanent", True), - allow_session=approval_data.get("allow_session", True), - smart_denied=approval_data.get("smart_denied", False), - ), - ctx._loop_for_step, - logger=logger, - log_message="send_exec_approval scheduling error", - ) - if _approval_fut is None: - raise RuntimeError("send_exec_approval: loop unavailable") - _outcome = _approval_send_outcome(_approval_fut, timeout=15) - if _outcome == "sent": - return - if _outcome == "ambiguous": - # Timeout ≠ failure: the card may have posted with a late ack (slow API or - # backpressure). The prompt registration stays alive so a tap still resolves; - # re-sending made duplicate cards + orphaned "/approve: nothing pending". Skip. - logger.warning( - "Button-based approval send timed out — treating " - "as possibly-delivered (no re-send; the prompt " - "stays armed for a late tap)" - ) - return - logger.warning( - "Button-based approval failed (send returned error), falling back to text" - ) - except Exception as _e: - logger.warning( - "Button-based approval failed, falling back to text: %s", _e - ) - - # Fallback: plain-text approval prompt with the adapter's typed prefix (e.g. `!approve`) — - # typed "/" is blocked in Slack threads and reserved by Matrix clients. - _p = getattr(ctx._status_adapter, "typed_command_prefix", "/") - msg = _format_exec_approval_fallback( - cmd, - desc, - _p, - allow_permanent=approval_data.get("allow_permanent", True), - allow_session=approval_data.get("allow_session", True), - smart_denied=approval_data.get("smart_denied", False), - ) - try: - # Mark as approval prompt so WeCom routes through control lane - _approval_metadata = dict(ctx._status_thread_metadata or {}) - _approval_metadata["is_approval_prompt"] = True - - _approval_send_fut = safe_schedule_threadsafe( - ctx._status_adapter.send( - ctx._status_chat_id, - msg, - metadata=_interim_metadata(_approval_metadata), - ), - ctx._loop_for_step, - logger=logger, - log_message="Approval text-send scheduling error", - ) - if _approval_send_fut is not None: - _approval_send_fut.result(timeout=15) - except Exception as _e: - logger.error("Failed to send approval request: %s", _e) - - # Keep real user text separate from API-only recovery guidance: if an auto-continue note is - # prepended below, persist the original so stale guidance never replays as user text. - _persist_user_message_override: Optional[Any] = ctx.persist_user_message - _persist_user_timestamp_override: Optional[float] = ctx.persist_user_timestamp - - # Prepend pending model switch note so the model knows about the switch - _pending_notes = getattr(self._runner, '_pending_model_notes', {}) - _msn = _pending_notes.pop(ctx.session_key, None) if ctx.session_key else None - if _msn: - ctx.message = _msn + "\n\n" + ctx.message - - # Auto-continue: history ending with a tool result means the previous turn was cut off - # (restart, crash, SIGTERM) — prepend a system note so the model finishes the pending tool - # results first. Session-level resume_pending (drain-timeout shutdown) uses stronger - # reason-aware wording that subsumes this case. Both gate on the age of ``history[-1]`` (not - # agent_history, which stripped ``timestamp`` off tool rows); rows without one are fresh. - _freshness_window = _auto_continue_freshness_window() - _interruption_is_fresh = _is_fresh_gateway_interruption( - _last_transcript_timestamp(ctx.history), - window_secs=_freshness_window, - ) - - _resume_entry = None - if ctx.session_key: - try: - _resume_entry = self._runner.session_store._entries.get(ctx.session_key) - except Exception: - _resume_entry = None - - # resume_pending freshness also uses the restart watchdog's ``last_resume_marked_at`` (the - # true interruption stamp): the transcript clock (_interruption_is_fresh) can be hours older - # for an active thread, so gating on it alone drops the recovery note — and the startup - # auto-resume turn has empty text, so the model gets a blank user message. Fresh if EITHER is. - _resume_mark_is_fresh = False - if _resume_entry is not None and getattr(_resume_entry, "resume_pending", False): - _resume_mark_is_fresh = _is_fresh_gateway_interruption( - getattr(_resume_entry, "last_resume_marked_at", None), - window_secs=_freshness_window, - ) - _is_resume_pending = bool( - _resume_entry is not None - and getattr(_resume_entry, "resume_pending", False) - and (_interruption_is_fresh or _resume_mark_is_fresh) - ) - _has_fresh_tool_tail = bool( - agent_history - and agent_history[-1].get("role") == "tool" - and _interruption_is_fresh - ) - - if _is_resume_pending: - _reason = getattr(_resume_entry, "resume_reason", None) or "restart_timeout" - # Empty message = the startup auto-resume turn from _schedule_resume_pending_sessions; - # there is no NEW user message. Interactive platforms report the restore and ask what - # next; event platforms (webhook, API server) continue the work — nobody is present to - # answer, and an acknowledgement would silently abandon the task. - _resume_adapter = self._runner._adapter_for_source(ctx.source) - _interactive_resume = bool( - getattr(_resume_adapter, "interactive_resume", True) - ) - ctx.message, _persist_user_message_override = _prepare_resume_pending_message( - _reason, ctx.message, interactive=_interactive_resume, - ) - elif _has_fresh_tool_tail: - _persist_user_message_override = ctx.message - ctx.message = ( - "[System note: A new message has arrived. The conversation " - "history contains pending tool outputs from an interrupted turn. " - "IGNORE those pending results. Address the user's NEW message " - "below FIRST. Do NOT re-execute old tool calls from the history.]\n\n" - + ctx.message - ) - - # Consume one-shot /reload-skills note (same queue pattern as CLI): prepend to the NEXT user - # message, then clear. Nothing hit the transcript out-of-band, so alternation stays intact. - _pending_notes = getattr(self._runner, "_pending_skills_reload_notes", None) - if _pending_notes and ctx.session_key and ctx.session_key in _pending_notes: - _srn = _pending_notes.pop(ctx.session_key, None) - if _srn: - ctx.message = _srn + "\n\n" + ctx.message - - # Safety net: a startup auto-resume event carries empty text and relies on the resume_pending - # branch above for the recovery note. If it did not fire (freshness signals disagreed, marker - # cleared before dispatch) we must NOT hand the model a blank user turn. Restricted to - # resume_pending sessions so legitimately empty turns (caption-less image) are untouched. - if ( - isinstance(ctx.message, str) - and not ctx.message.strip() - and _resume_entry is not None - and getattr(_resume_entry, "resume_pending", False) - ): - _sn_reason = ( - getattr(_resume_entry, "resume_reason", None) or "restart_timeout" - ) - _sn_adapter = self._runner._adapter_for_source(ctx.source) - ctx.message = build_resume_recovery_note( - _sn_reason, - "", - interactive=bool( - getattr(_sn_adapter, "interactive_resume", True) - ), - ) - - _approval_session_key = ctx.session_key or "" - _approval_session_token = set_current_session_key(_approval_session_key) - register_gateway_notify(_approval_session_key, _approval_notify_sync) - try: - # If _prepare_inbound_message_text buffered image paths for native attachment, wrap the - # user turn as an OpenAI-style multimodal content list. Consume-and-clear so subsequent - # turns on the same runner instance don't re-attach stale images. - _native_imgs = self._runner._consume_pending_native_image_paths(ctx.session_key) - if _native_imgs: - try: - from agent.image_routing import build_native_content_parts - _parts, _skipped = build_native_content_parts( - ctx.message, - _native_imgs, - ) - if _skipped: - logger.warning( - "Native image attachment: skipped %d unreadable path(s): %s", - len(_skipped), _skipped, - ) - if any(p.get("type") == "image_url" for p in _parts): - _run_message: Any = _parts - else: - # All images failed to read — fall back to plain text. - _run_message = ctx.message - except Exception as _img_exc: - logger.warning( - "Native image attachment failed, falling back to text: %s", - _img_exc, - ) - _run_message = ctx.message - else: - _run_message = ctx.message - - _api_run_message = _wrap_current_message_with_observed_context( - _run_message, - observed_group_context, - ) - _conversation_kwargs = { - "conversation_history": agent_history, - "task_id": ctx.session_id, - } - if _persist_user_message_override is not None: - _conversation_kwargs["persist_user_message"] = _persist_user_message_override - elif observed_group_context: - _conversation_kwargs["persist_user_message"] = ctx.message - if ctx.persist_user_display_kind: - # Internal self-injected turn: type the persisted user row at turn start so UIs - # render it as a timeline notice, not a user bubble. Role/content are untouched and - # the key is stripped from provider-bound payloads in conversation_loop. - _conversation_kwargs["persist_user_display_kind"] = ( - ctx.persist_user_display_kind - ) - if ctx.moa_config is not None: - _conversation_kwargs["moa_config"] = ctx.moa_config - if _persist_user_timestamp_override is not None: - _conversation_kwargs["persist_user_timestamp"] = _persist_user_timestamp_override - # Thread the platform-side inbound message id onto the persisted user turn so a turn - # interrupted by a restart is recorded WITH its id — drain-window recovery dedups on - # has_platform_message_id. Uses the raw inbound id, NOT event_message_id (reply anchor). - if ctx.inbound_message_id is not None: - _conversation_kwargs["persist_user_platform_id"] = str(ctx.inbound_message_id) - result = agent.run_conversation(_api_run_message, **_conversation_kwargs) - finally: - unregister_gateway_notify(_approval_session_key) - # Cancel any pending clarify entries so blocked agent threads don't hang past the end of - # the run (interrupt, completion, gateway shutdown). Idempotent. - try: - from tools.clarify_gateway import clear_session as _clear_clarify_session - _clear_clarify_session(_approval_session_key) - except Exception: - pass - reset_current_session_key(_approval_session_token) - # Canonicalize a model-emitted computer-use screenshot path at the common result boundary: the - # streaming finalizer below and the non-streaming delivery path must see the same response; - # repairing only in later media scanning leaves streaming a mangled path + rejected attachment. - if isinstance(result, dict): - _result_final = result.get("final_response") - if isinstance(_result_final, str): - result["final_response"] = repair_explicit_computer_use_media_paths( - _result_final, - result.get("messages", []), - history_offset=len(agent_history), - ) - - ctx.result_holder[0] = result - - # Signal the stream consumer that the agent is done, passing final_response as the - # authoritative finalize payload: it includes post-stream augmentation (verifier footer, - # explainer) the accumulator never saw, so the seal delivers the TRUE final with no - # corrective send. Failed turns pass nothing — error text goes via the normal path. - if _stream_consumer is not None: - _final_for_stream = None - # Adopt ONLY a genuinely completed final: interrupt paths return {interrupted: True, - # completed: False} with a DIAGNOSTIC final_response and no failed key — adopting it - # would seal the streamed partial answer over with the diagnostic AND make - # delivered_final_matches reconcile, suppressing the gateway's own error delivery. - if ( - isinstance(result, dict) - and not result.get("failed") - and not result.get("interrupted") - and result.get("completed") is not False - ): - _fr = result.get("final_response") - if isinstance(_fr, str) and _fr.strip() and _fr != "(empty)": - _final_for_stream = _fr - if _final_for_stream is not None: - # Duck-type safe: test doubles / older consumers may expose a zero-arg finish(). The - # payload is an optimization, not a requirement — fall back to the bare signal. - try: - _stream_consumer.finish(_final_for_stream) - except TypeError: - _stream_consumer.finish() - else: - _stream_consumer.finish() - - # Signal the streaming-TTS consumer that the agent is done. finish() runs on the outer - # event-loop thread after the executor returns, so early run_sync returns are also finalised. - - # Return final response, or a message if something went wrong - final_response = result.get("final_response") - - # Extract actual token counts from the agent instance used for this run - _last_prompt_toks = 0 - _input_toks = 0 - _output_toks = 0 - _context_length = 0 - _agent = ctx.agent_holder[0] - if _agent and hasattr(_agent, "context_compressor"): - _last_prompt_toks = getattr(_agent.context_compressor, "last_prompt_tokens", 0) - _input_toks = getattr(_agent, "session_prompt_tokens", 0) - _output_toks = getattr(_agent, "session_completion_tokens", 0) - _context_length = getattr(_agent.context_compressor, "context_length", 0) or 0 - _resolved_model = getattr(_agent, "model", None) if _agent else None - - # Sync session_id right after run_conversation(): compression can rotate before a follow-up - # model call fails, and the failure return below must still point at the compressed child. - agent = ctx.agent_holder[0] - _session_was_split = False - # In-place compaction (compression.in_place) compacts the transcript WITHOUT rotating the id, - # so the id-change diff below can't see it. compress_context() sets this flag on the agent; the - # gateway re-baselines (history_offset=0 + JSONL rewrite) as for a split despite unchanged id. - _compacted_in_place = bool(getattr(agent, "_last_compaction_in_place", False)) if agent else False - agent_session_id = getattr(agent, 'session_id', ctx.session_id) if agent else ctx.session_id - if agent and ctx.session_key and agent_session_id != ctx.session_id: - _session_was_split = True - logger.info( - "Session split detected: %s → %s (compression)", - ctx.session_id, agent_session_id, - ) - entry = self._runner.session_store._entries.get(ctx.session_key) - _session_split_entry_persisted = False - if entry: - entry_session_id = getattr(entry, "session_id", None) - if not ctx._run_still_current(): - logger.info( - "Skipping session split sync for stale run %s — " - "generation %s is no longer current", - ctx.session_key or "?", - ctx.run_generation, - ) - elif entry_session_id == agent_session_id: - _session_split_entry_persisted = True - elif entry_session_id != ctx.session_id: - logger.info( - "Skipping session split sync for %s because the " - "session binding moved from %s to %s before " - "compression finished", - ctx.session_key or "?", - ctx.session_id, - entry_session_id, - ) - else: - entry.session_id = agent_session_id - self._runner.session_store._save() - self._runner.session_store._record_gateway_session_peer( - agent_session_id, - ctx.session_key, - ctx.source, - ) - _session_split_entry_persisted = True - - # Telegram DM whose source.thread_id was lost in the session split (synthetic/recovered - # event): restore it from the binding so _thread_metadata_for_source yields the right - # message_thread_id instead of the General thread (non-fatal). Only after this run - # published its split — a stale /stop→/new predecessor must not mutate routing state. - if _session_split_entry_persisted and ( - getattr(ctx.source, "platform", None) == Platform.TELEGRAM - and getattr(ctx.source, "chat_type", None) == "dm" - and getattr(ctx.source, "thread_id", None) is None - and self._runner._session_db is not None - ): - try: - # run_sync is off-loop (executor); sync DB is fine. - _binding = self._runner._session_db._db.get_telegram_topic_binding_by_session( - session_id=agent_session_id, - ) - if _binding and _binding.get("thread_id"): - ctx.source.thread_id = str(_binding["thread_id"]) - logger.debug( - "Restored source.thread_id=%s from binding after session split %s → %s", - ctx.source.thread_id, - ctx.session_id, - agent_session_id, - ) - except Exception: - logger.debug( - "Failed to restore thread_id from binding after session split", - exc_info=True, - ) - if _session_split_entry_persisted: - self._runner._sync_telegram_topic_binding( - ctx.source, entry, reason="agent-run-compression", - ) - - effective_session_id = agent_session_id - self._runner._sync_session_model_from_agent(effective_session_id, agent) - # history_offset=0 whenever the agent's message list lost the original history prefix: rotation - # (split) OR in-place compaction. Either way the returned `messages` is the compacted set, so - # persist all of it; slicing past the pre-compaction length would drop everything. - _effective_history_offset = ( - 0 if (_session_was_split or _compacted_in_place) else len(agent_history) - ) - - if not final_response: - final_response = _normalize_empty_agent_response( - result, final_response or "", history_len=len(agent_history), - ) - final_response = _sanitize_gateway_final_response(ctx.source.platform, final_response) - if not final_response: - final_response = f"⚠️ {result['error']}" if result.get("error") else "" - return { - "final_response": final_response, - "messages": result.get("messages", []), - "api_calls": result.get("api_calls", 0), - "failed": result.get("failed", False), - # Sibling of the non-empty-response return below: the classifier's failure_reason - # must survive the empty-response path too, or downstream consumers (TUI billing, - # transient-failure persistence) lose the structured reason when no text was produced. - "failure_reason": result.get("failure_reason"), - "partial": result.get("partial", False), - "completed": result.get("completed"), - "interrupted": result.get("interrupted", False), - "interrupt_message": result.get("interrupt_message"), - "error": result.get("error"), - "compression_exhausted": result.get("compression_exhausted", False), - "compression_deferred": result.get("compression_deferred", False), - "tools": ctx.tools_holder[0] or [], - "history_offset": _effective_history_offset, - "compacted_in_place": _compacted_in_place, - "session_id": effective_session_id, - "last_prompt_tokens": _last_prompt_toks, - "input_tokens": _input_toks, - "output_tokens": _output_toks, - "model": _resolved_model, - "context_length": _context_length, - } - - # Append MEDIA: tags from tool results (e.g. TTS) that the model's final text omits, so - # extract_media() delivers each file once. Scope to THIS turn (slice at ``len(agent_history)``) - # so a stale MEDIA: path from an earlier turn doesn't ride a later text-only reply; dedup - # against _history_media_paths is the secondary guard — and the sole one on the fallback - # branch when mid-run compression shrank the list below the history length. - if "MEDIA:" not in final_response: - media_tags, has_voice_directive = _collect_auto_append_media_tags( - result.get("messages", []), - history_offset=len(agent_history), - history_media_paths=_history_media_paths, - ) - - if media_tags: - seen = set() - unique_tags = [] - for tag in media_tags: - if tag not in seen: - seen.add(tag) - unique_tags.append(tag) - if has_voice_directive: - unique_tags.insert(0, "[[audio_as_voice]]") - final_response = final_response + "\n" + "\n".join(unique_tags) - - # Auto-titling runs at TURN START (agent/turn_context.py) from the user's message alone, so a - # failed/interrupted turn is still titled. Thread-rename callbacks are attached as - # `_on_session_title` before the run because the titler fires from the turn prologue. - - return { - "final_response": final_response, - "last_reasoning": result.get("last_reasoning"), - "messages": ctx.result_holder[0].get("messages", []) if ctx.result_holder[0] else [], - "api_calls": ctx.result_holder[0].get("api_calls", 0) if ctx.result_holder[0] else 0, - "failed": ctx.result_holder[0].get("failed", False) if ctx.result_holder[0] else False, - "failure_reason": ( - ctx.result_holder[0].get("failure_reason") if ctx.result_holder[0] else None - ), - "completed": ctx.result_holder[0].get("completed") if ctx.result_holder[0] else None, - "interrupted": ctx.result_holder[0].get("interrupted", False) if ctx.result_holder[0] else False, - "partial": ctx.result_holder[0].get("partial", False) if ctx.result_holder[0] else False, - "error": ctx.result_holder[0].get("error") if ctx.result_holder[0] else None, - "interrupt_message": ctx.result_holder[0].get("interrupt_message") if ctx.result_holder[0] else None, - "compression_exhausted": ( - ctx.result_holder[0].get("compression_exhausted", False) - if ctx.result_holder[0] else False - ), - # Soft lock-contention defer: distinct from compression_exhausted so the gateway never - # auto-resets a session that a concurrent compressor is about to shrink. - "compression_deferred": ( - ctx.result_holder[0].get("compression_deferred", False) - if ctx.result_holder[0] else False - ), - "tools": ctx.tools_holder[0] or [], - "history_offset": _effective_history_offset, - "compacted_in_place": _compacted_in_place, - "last_prompt_tokens": _last_prompt_toks, - "input_tokens": _input_toks, - "output_tokens": _output_toks, - "model": _resolved_model, - "context_length": _context_length, - "session_id": effective_session_id, - "response_previewed": result.get("response_previewed", False), - "response_transformed": result.get("response_transformed", False), - # Pass through agent_persisted so the persistence block above can tell whether the codex - # app-server path self-persisted (it didn't — see codex_runtime.py); default True keeps the - # skip-db behaviour for the standard runtime. - "agent_persisted": (ctx.result_holder[0].get("agent_persisted", True) if ctx.result_holder[0] else True), - } - - # Sentinel for "no explicit session DB pinned on this runner", so ``_session_db`` can distinguish # "resolve from the active profile scope" from a deliberate ``runner._session_db = None`` (disables # DB-backed commands, as many test suites do). Mirrors ``gateway.session._DB_UNPINNED``. @@ -6353,7 +4094,24 @@ def _instantiate_builtin_adapter(platform: Platform, config: Any) -> Optional[Ba return adapter_cls(config) -class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, GatewaySlashCommandsMixin): +class GatewayRunner( + GatewayAuthorizationMixin, + GatewayKanbanWatchersMixin, + GatewaySlashCommandsMixin, + GatewayVoiceMixin, + GatewayAdapterLifecycleMixin, + GatewayTopicThreadsMixin, + GatewayTurnMixin, + GatewayShutdownMixin, + GatewayBusySessionMixin, + GatewayConfigLoadersMixin, + GatewayStartupMixin, + GatewaySessionWatchersMixin, + GatewayNotificationsMixin, + GatewayInboundMixin, + GatewayGoalsMixin, + GatewayAgentCacheMixin, +): """Main gateway controller: manages adapter lifecycles, routes messages to/from the agent.""" # Class-level defaults so partial construction in tests doesn't @@ -6487,6 +4245,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._warn_if_docker_media_delivery_is_risky() _gateway_runner_ref = _weakref.ref(self) + self._init_runtime_settings() + self._init_session_store() + self._init_lifecycle_state() + self._init_runtime_caches() + self._init_startup_checks() + self._init_session_db() + self._init_registries_and_clocks() + + def _init_runtime_settings(self) -> None: + """Load ephemeral per-call config (prefill, reasoning, busy modes, timeouts, routing).""" # Load ephemeral config from config.yaml / env vars. # Both are injected at API-call time only and never persisted. self._prefill_messages = self._load_prefill_messages() @@ -6508,6 +4276,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._provider_routing = self._load_provider_routing() self._fallback_model = self._load_fallback_model() + def _init_session_store(self) -> None: + """Build the SessionStore (with process-registry reset guard), its async facade and the router.""" # Wire process registry into session store for reset protection. A background process older # than session_reset.bg_process_max_age_hours (default 24h) is stale and no longer blocks # idle/daily reset. The process is NOT killed, only ignored by the reset guard. @@ -6528,6 +4298,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # ``session_store`` directly; async gateway handlers call this facade and await every op. self._async_session_store = AsyncSessionStore(self.session_store) self.delivery_router = DeliveryRouter(self.config) + + def _init_lifecycle_state(self) -> None: + """Initialise run/exit/restart flags, per-session state, and completion-delivery bookkeeping.""" self._running = False self._gateway_loop: Optional[asyncio.AbstractEventLoop] = None self._shutdown_event = asyncio.Event() @@ -6613,6 +4386,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._completion_notification_batch_window = 0.1 self._completion_notification_batches_stopping = False + def _init_runtime_caches(self) -> None: + """Agent cache, profile identity, Teams runtime, failed-platform tracking, slash-confirm counter.""" # Cache AIAgent instances per session to preserve prompt caching (a fresh agent per message # rebuilds the system prompt and breaks the prefix cache, ~10x cost on Anthropic). Value: # (AIAgent, config_signature_str). OrderedDict for LRU eviction in _enforce_agent_cache_cap(); @@ -6650,6 +4425,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew import itertools as _itertools self._slash_confirm_counter = _itertools.count(1) + def _init_startup_checks(self) -> None: + """Ensure tirith is installed and warn when manual approvals have no automated assessor.""" # Persistent Honcho managers keyed by gateway session key: preserves write_frequency="session" # semantics across short-lived per-message AIAgent instances. @@ -6683,6 +4460,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: logger.debug("approvals.mode startup check skipped", exc_info=True) + def _init_session_db(self) -> None: + """Open the session DB for the active scope and run opportunistic state.db / checkpoint maintenance.""" # Session DB for session_search: a property caches one AsyncSessionDB per path (not a handle # bound here, which would pin the root home — /resume, /title, /history and search run inside # _profile_runtime_scope under multiplex); priming here keeps startup diagnostics at init. @@ -6751,6 +4530,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception as exc: logger.debug("checkpoint auto-maintenance skipped: %s", exc) + def _init_registries_and_clocks(self) -> None: + """Pairing stores, hook registry, voice modes, background-task set, liveness and idle clocks.""" # DM pairing store for code-based user authorization. ``pairing_store`` is the global/default # store (``hermes pairing`` CLI, callers without profile context); ``pairing_stores`` is the # per-profile map ``authz_mixin._is_user_authorized`` routes through (one whitelist/profile). @@ -6954,347 +4735,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "for container-local paths like '/workspace/...' or '/output/...'." ) - # -- Setup skill availability ---------------------------------------- - - def _has_setup_skill(self) -> bool: - """Check if the hermes-agent-setup skill is installed.""" - try: - from tools.skill_manager_tool import _find_skill - return _find_skill("hermes-agent-setup") is not None - except Exception: - return False - # -- Voice mode persistence ------------------------------------------ _VOICE_MODE_PATH = _hermes_home / "gateway_voice_mode.json" - def _voice_key( - self, platform: Platform, chat_id: str, profile: Optional[str] = None - ) -> str: - """Return a platform-namespaced key for voice mode state. - - Under multiplexing the key is ``::`` (profile whose bot speaks); - the default profile keeps ``:`` so persisted state stays valid. Otherwise - two bots in one Discord channel share a key and one profile's ``/voice`` flips the other's. - """ - base = f"{platform.value}:{chat_id}" - profile = profile.strip() if isinstance(profile, str) else "" - if not profile or profile == "default": - return base - return f"{profile}:{base}" - - def _voice_key_for_source(self, source: SessionSource) -> str: - """Voice-state key for an inbound source, namespaced by its transport owner. - - Voice mode belongs to the (bot, chat) pair, so the namespace is the profile that OWNS the - receiving adapter (matching ``_sync_voice_mode_state_to_adapter``), not the routed profile. - """ - return self._voice_key( - source.platform, - source.chat_id, - profile=self._adapter_profile_for_source(source), - ) - - def _bind_voice_input_callback(self, adapter) -> None: - """Route voice transcripts back through the adapter that captured them.""" - if hasattr(adapter, "_voice_input_callback"): - adapter._voice_input_callback = functools.partial( - self._handle_voice_channel_input, adapter=adapter - ) - - def _load_voice_modes(self) -> Dict[str, str]: - try: - data = json.loads(self._VOICE_MODE_PATH.read_text(encoding="utf-8")) - except (FileNotFoundError, json.JSONDecodeError, OSError): - return {} - - if not isinstance(data, dict): - return {} - - valid_modes = {"off", "voice_only", "all"} - result = {} - for chat_id, mode in data.items(): - if mode not in valid_modes: - continue - key = str(chat_id) - # Skip legacy unprefixed keys (warn and skip) - if ":" not in key: - logger.warning( - "Skipping legacy unprefixed voice mode key %r during migration. " - "Re-enable voice mode on that chat to rebuild the prefixed key.", - key, - ) - continue - result[key] = mode - return result - - def _save_voice_modes(self) -> None: - try: - self._VOICE_MODE_PATH.parent.mkdir(parents=True, exist_ok=True) - self._VOICE_MODE_PATH.write_text( - json.dumps(self._voice_mode, indent=2), encoding="utf-8" - ) - except OSError as e: - logger.warning("Failed to save voice modes: %s", e) - - @staticmethod - def _toggle_adapter_auto_tts_set(adapter, chat_id: str, on: bool, *, add_to: str, clear_from: str) -> None: - """Add/discard ``chat_id`` in the adapter's ``add_to`` set; adding also clears it from ``clear_from``. - - ``/voice off`` and an explicit ``/voice on``/``/voice tts`` are hard overrides of each other.""" - target = getattr(adapter, add_to, None) - if not isinstance(target, set): - return - if on: - target.add(chat_id) - other = getattr(adapter, clear_from, None) - if isinstance(other, set): - other.discard(chat_id) - else: - target.discard(chat_id) - - def _set_adapter_auto_tts_disabled(self, adapter, chat_id: str, disabled: bool) -> None: - """Update an adapter's in-memory auto-TTS suppression set if present.""" - self._toggle_adapter_auto_tts_set( - adapter, chat_id, disabled, add_to="_auto_tts_disabled_chats", clear_from="_auto_tts_enabled_chats" - ) - - def _set_adapter_auto_tts_enabled(self, adapter, chat_id: str, enabled: bool) -> None: - """Update an adapter's per-chat auto-TTS opt-in set (auto-TTS even when ``voice.auto_tts`` is False).""" - self._toggle_adapter_auto_tts_set( - adapter, chat_id, enabled, add_to="_auto_tts_enabled_chats", clear_from="_auto_tts_disabled_chats" - ) - - def _sync_voice_mode_state_to_adapter(self, adapter) -> None: - """Restore persisted /voice state into a live platform adapter. - - Sets ``_auto_tts_default`` (from ``voice.auto_tts``) and, from ``self._voice_mode``, - ``_auto_tts_enabled_chats`` (modes ``voice_only``/``all``) and ``_auto_tts_disabled_chats`` - (mode ``off``). - """ - platform = getattr(adapter, "platform", None) - if not isinstance(platform, Platform): - return - - disabled_chats = getattr(adapter, "_auto_tts_disabled_chats", None) - enabled_chats = getattr(adapter, "_auto_tts_enabled_chats", None) - if not isinstance(disabled_chats, set) and not isinstance(enabled_chats, set): - return - - # Push the global voice.auto_tts default (config.yaml) onto the adapter. - # Lazy import to avoid adding a module-level dep from gateway → hermes_cli. - try: - from hermes_cli.config import load_config as _load_full_config - _full_cfg = _load_full_config() - _auto_tts_default = bool( - (_full_cfg.get("voice") or {}).get("auto_tts", False) - ) - except Exception: - _auto_tts_default = False - if hasattr(adapter, "_auto_tts_default"): - adapter._auto_tts_default = _auto_tts_default - - prefix = self._voice_key(platform, "", profile=getattr(adapter, "_owner_profile", None)) - if isinstance(disabled_chats, set): - disabled_chats.clear() - disabled_chats.update( - key[len(prefix):] for key, mode in self._voice_mode.items() - if mode == "off" and key.startswith(prefix) - ) - if isinstance(enabled_chats, set): - enabled_chats.clear() - enabled_chats.update( - key[len(prefix):] for key, mode in self._voice_mode.items() - if mode in {"voice_only", "all"} and key.startswith(prefix) - ) - - async def _await_adapter_cleanup_with_timeout( - self, awaitable: Awaitable[Any], timeout: float - ) -> bool: - """Wait for adapter cleanup without letting cancellation swallowing hang us. - - ``asyncio.wait_for`` cancels an overdue child but then waits for it to exit. An adapter - close path that catches ``CancelledError`` can therefore block recovery forever. Keep - ownership of the old task through its done callback, but release the runner at the deadline. - """ - if timeout <= 0: - await awaitable - return True - - task = asyncio.ensure_future(awaitable) - try: - done, _pending = await asyncio.wait({task}, timeout=timeout) - except asyncio.CancelledError: - task.cancel() - task.add_done_callback(consume_detached_task_result) - raise - if task in done: - await task - return True - - task.cancel() - task.add_done_callback(consume_detached_task_result) - return False - - async def _safe_adapter_disconnect(self, adapter, platform) -> None: - """Call adapter.disconnect() defensively, swallowing any error. - - For a failed/raised connect(): partial resources (aiohttp.ClientSession, poll tasks, child - subprocesses) would otherwise leak. Must tolerate partial-init state and never raise. - """ - timeout = self._adapter_disconnect_timeout_secs() - try: - completed = await self._await_adapter_cleanup_with_timeout( - adapter.disconnect(), timeout - ) - if not completed: - logger.warning( - "Timed out after %.1fs while disconnecting %s adapter; continuing shutdown", - timeout, - platform.value if platform is not None else "adapter", - ) - except Exception as e: - logger.debug( - "Defensive %s disconnect after failed connect raised: %s", - platform.value if platform is not None else "adapter", - e, - ) - - async def _bounded_adapter_teardown( - self, adapter, platform, *, profile: Optional[str] = None - ) -> None: - """Tear down one adapter on the shutdown path with bounded awaits. - - ``cancel_background_tasks()`` and ``disconnect()`` can block forever on half-dead network - state (e.g. a wedged WebSocket thread), stalling shutdown past systemd's ``TimeoutStopSec``; - the SIGKILL skips ``atexit`` PID-file cleanup and the next start dies with "PID file race - lost". Each await uses ``HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT``; on timeout the task is - cancelled and detached so a cancellation-swallowing adapter can't hang the loop. Never raises. - """ - timeout = self._adapter_disconnect_timeout_secs() - suffix = f" (profile: {profile})" if profile else "" - started_at = time.monotonic() - try: - cancelled = await self._await_adapter_cleanup_with_timeout( - adapter.cancel_background_tasks(), timeout - ) - if not cancelled: - logger.warning( - "✗ %s background-task cancel timed out after %.1fs - forcing continue%s", - platform.value, timeout, suffix, - ) - except Exception as e: - logger.debug("✗ %s background-task cancel error%s: %s", platform.value, suffix, e) - try: - disconnected = await self._await_adapter_cleanup_with_timeout( - adapter.disconnect(), timeout - ) - if disconnected: - logger.info( - "✓ %s disconnected (%.2fs)%s", - platform.value, time.monotonic() - started_at, suffix, - ) - else: - logger.warning( - "✗ %s disconnect timed out after %.1fs - forcing continue%s", - platform.value, timeout, suffix, - ) - except Exception as e: - logger.error( - "✗ %s disconnect error after %.2fs%s: %s", - platform.value, time.monotonic() - started_at, suffix, e, - ) - - def _adapter_disconnect_timeout_secs(self) -> float: - """Return the per-adapter disconnect timeout used during shutdown.""" - raw = os.getenv("HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT", "").strip() - if raw: - try: - timeout = float(raw) - except ValueError: - logger.warning( - "Ignoring invalid HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT=%r", - raw, - ) - else: - return max(0.0, timeout) - return _ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT - - def _platform_connect_timeout_secs(self, platform=None, *, initial: bool = False) -> float: - """Return the per-platform connect timeout used during startup/retry. - - Telegram's full 180s connect budget is deliberately NOT spent at cold start: an unreachable - Telegram would hold the gateway out of ``running`` for the whole budget. The cold-start wait - is capped and the platform handed to the reconnect watcher, which retries with the full - budget and ``is_reconnect=True`` (preserving the offline update queue). - """ - raw = os.getenv("HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT", "").strip() - if raw: - try: - timeout = float(raw) - except ValueError: - logger.warning( - "Ignoring invalid HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT=%r", - raw, - ) - else: - return max(0.0, timeout) - if platform == Platform.TELEGRAM: - if initial: - return _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT - return _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT - return _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT - - async def _connect_adapter_with_timeout( - self, adapter, platform, *, is_reconnect: bool = False, initial: bool = False - ) -> bool: - """Connect an adapter without allowing one platform to block others. - - ``is_reconnect`` lets adapters distinguish a cold first boot (drop any stale server-side - queue) from a watcher reconnect (preserve the queue so interim messages aren't dropped). - ``initial`` selects the capped cold-start budget for platforms whose full connect budget is - too long to spend before the gateway reaches ``running`` (Telegram's 180s). - """ - timeout = self._platform_connect_timeout_secs(platform, initial=initial) - if timeout <= 0: - return await adapter.connect(is_reconnect=is_reconnect) - # Detach-on-timeout rather than plain asyncio.wait_for: wait_for cancels the overdue task but - # then waits for it to exit, so a connect() that catches CancelledError blocks recovery - # forever (watcher never retries). Keep ownership via its done callback; release at deadline. - task = asyncio.ensure_future( - adapter.connect(is_reconnect=is_reconnect) - ) - try: - done, _pending = await asyncio.wait({task}, timeout=timeout) - except asyncio.CancelledError: - task.cancel() - task.add_done_callback(consume_detached_task_result) - raise - if task in done: - result = await task - return bool(result) - task.cancel() - task.add_done_callback(consume_detached_task_result) - raise TimeoutError( - f"{platform.value} connect timed out after {timeout:g}s" - ) - - async def _connect_initial_adapter_with_timeout(self, adapter, platform) -> bool: - """Connect one cold-start adapter with tightly scoped replace intent. - - The capability is visible only while this initial connect is awaited. Reconnects call - ``_connect_adapter_with_timeout`` directly and adapters also default to deny, so a later - network recovery can never evict a healthy token holder. - """ - adapter._platform_lock_takeover_allowed = bool( - self._platform_lock_takeover_on_start - ) - try: - return await self._connect_adapter_with_timeout( - adapter, platform, initial=True - ) - finally: - adapter._platform_lock_takeover_allowed = False @property def should_exit_cleanly(self) -> bool: @@ -7341,207 +4785,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew profile=_profile, ) - @staticmethod - def _telegram_topic_profile_name(source: SessionSource) -> str: - """Profile namespace for Telegram topic-mode rows. - - Use the profile stamped on the routed event (``source.profile``), never the process-global - active profile — under multiplex that mis-attributes topic state across bots sharing state.db. - """ - name = str(getattr(source, "profile", None) or "").strip() - return name if name else "default" - - def _telegram_topic_mode_enabled(self, source: SessionSource) -> bool: - """Return whether Telegram DM topic mode is active for this chat.""" - if source.platform != Platform.TELEGRAM or source.chat_type != "dm": - return False - session_db = getattr(self, "_session_db", None) - if session_db is None: - return False - # Runs off-loop (always via asyncio.to_thread); use the sync handle. - session_db = getattr(session_db, "_db", session_db) - try: - raw = session_db.is_telegram_topic_mode_enabled( - chat_id=str(source.chat_id), - user_id=str(source.user_id), - profile_name=self._telegram_topic_profile_name(source), - ) - except Exception: - logger.debug("Failed to read Telegram topic mode state", exc_info=True) - return False - # Only a real True from the SessionDB enables topic mode; anything else (including MagicMock - # from test fixtures that didn't opt in) means off for this chat. - return raw is True # Telegram's General (pinned top) topic in forum-enabled private chats: clients variously omit # message_thread_id or send "1" for it. Treat both as "root" for lobby/lane purposes. _TELEGRAM_GENERAL_TOPIC_IDS = frozenset({"", "1"}) - def _is_telegram_topic_root_lobby(self, source: SessionSource) -> bool: - """True for the main Telegram DM (or General topic) when topic mode has made it a lobby.""" - if source.platform != Platform.TELEGRAM or source.chat_type != "dm": - return False - if not self._telegram_topic_mode_enabled(source): - return False - tid = str(source.thread_id or "") - return tid in self._TELEGRAM_GENERAL_TOPIC_IDS - - def _is_telegram_topic_lane(self, source: SessionSource) -> bool: - """True for a user-created Telegram private-chat topic lane.""" - if source.platform != Platform.TELEGRAM or source.chat_type != "dm": - return False - if not self._telegram_topic_mode_enabled(source): - return False - tid = str(source.thread_id or "") - return bool(tid) and tid not in self._TELEGRAM_GENERAL_TOPIC_IDS _TELEGRAM_LOBBY_REMINDER_COOLDOWN_S = 30.0 - def _telegram_topic_cooldown_key(self, source: SessionSource) -> Optional[str]: - """Cooldown key for topic-mode cooldowns: (profile, chat_id). - - Profiles sharing a Telegram private chat_id under multiplex must not - suppress each other's lobby reminders / capability hints (#76423). - """ - chat_id = str(source.chat_id or "") - if not chat_id: - return None - return f"{self._telegram_topic_profile_name(source)}:{chat_id}" - - def _should_send_telegram_lobby_reminder(self, source: SessionSource) -> bool: - """Rate-limit root-DM lobby reminders to one per cooldown window, not one per prompt typed.""" - if not hasattr(self, "_telegram_lobby_reminder_ts"): - self._telegram_lobby_reminder_ts = {} - key = self._telegram_topic_cooldown_key(source) - if not key: - return True - import time as _time - now = _time.monotonic() - last = self._telegram_lobby_reminder_ts.get(key, 0.0) - if now - last < self._TELEGRAM_LOBBY_REMINDER_COOLDOWN_S: - return False - self._telegram_lobby_reminder_ts[key] = now - return True - - def _telegram_topic_root_lobby_message(self) -> str: - return ( - "This main chat is reserved for system commands.\n\n" - "To start a new Hermes chat, open the All Messages topic at the top " - "of this bot interface and send any message there. Telegram will " - "create a new topic for that message; each topic works as an " - "independent Hermes session." - ) - - def _telegram_topic_root_new_message(self) -> str: - return ( - "To start a new parallel Hermes chat, open the All Messages topic " - "at the top of this bot interface and send any message there. " - "Telegram will create a new topic for it.\n\n" - "Each topic is an independent Hermes session. Use /new inside an " - "existing topic only if you want to replace that topic's current session." - ) - - def _telegram_topic_new_header(self, source: SessionSource) -> Optional[str]: - if not self._is_telegram_topic_lane(source): - return None - return ( - "Started a new Hermes session in this topic.\n\n" - "Tip: for parallel work, open All Messages and send a message there " - "to create a separate topic instead of using /new here. /new replaces " - "the session attached to the current topic." - ) - - def _record_telegram_topic_binding( - self, - source: SessionSource, - session_entry, - ) -> None: - """Persist the Telegram topic -> Hermes session binding for topic lanes.""" - session_db = getattr(self, "_session_db", None) - if session_db is None or not source.chat_id or not source.thread_id: - return - # Runs off-loop (always via asyncio.to_thread); use the sync handle. - session_db = getattr(session_db, "_db", session_db) - session_db.bind_telegram_topic( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - user_id=str(source.user_id or ""), - session_key=session_entry.session_key, - session_id=session_entry.session_id, - profile_name=self._telegram_topic_profile_name(source), - ) - - def _sync_telegram_topic_binding( - self, - source: SessionSource, - session_entry, - *, - reason: str, - ) -> None: - """Update the topic binding to point at ``session_entry.session_id``. - - Topic lanes persist (chat_id, thread_id) -> session_id so reopening a topic resumes the - right session. When compression rotates the id mid-turn a stale binding reloads the - oversized parent next message, retriggering preflight compression — sometimes in a loop. - """ - if not self._is_telegram_topic_lane(source): - return - try: - self._record_telegram_topic_binding(source, session_entry) - except Exception: - logger.debug( - "telegram topic binding refresh failed (%s)", reason, exc_info=True, - ) - - def _recover_telegram_topic_thread_id( - self, - source: SessionSource, - ) -> Optional[str]: - """Pin DM-topic routing to the user's last-active topic. - - Telegram can omit ``message_thread_id`` or surface General (``1``) for topic-mode DM - replies; in those lobby-shaped cases keep the conversation on the user's most-recent bound - topic. Do not rewrite a non-lobby, previously-unbound thread id: a brand-new DM topic is - also "unknown" until its first inbound message is recorded, and rewriting would send its - answer into an older lane. Returns None to leave the source alone. - """ - if ( - source.platform != Platform.TELEGRAM - or source.chat_type != "dm" - or not source.chat_id - or not source.user_id - or not self._telegram_topic_mode_enabled(source) - ): - return None - inbound = str(source.thread_id or "") - is_lobby = not inbound or inbound in self._TELEGRAM_GENERAL_TOPIC_IDS - if not is_lobby: - # A non-lobby, unknown thread_id is likely the first message of a new Telegram DM topic: - # preserve it to be recorded as a new lane below rather than hijack the latest binding. - return None - session_db = getattr(self, "_session_db", None) - if session_db is None: - return None - # Runs off-loop (always via asyncio.to_thread); use the sync handle. - session_db = getattr(session_db, "_db", session_db) - try: - bindings = session_db.list_telegram_topic_bindings_for_chat( - chat_id=str(source.chat_id), - profile_name=self._telegram_topic_profile_name(source), - ) - except Exception: - logger.debug("topic-recover: read failed", exc_info=True) - return None - if not bindings: - return None - user_id = str(source.user_id) - for b in bindings: # newest-first - if str(b.get("user_id") or "") == user_id: - recovered = str(b.get("thread_id") or "") - if recovered and recovered != inbound: - return recovered - return None - return None def _normalize_source_for_session_key( self, @@ -7573,933 +4824,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew except Exception: return None - def _resolve_session_agent_runtime( - self, - *, - source: Optional[SessionSource] = None, - session_key: Optional[str] = None, - user_config: Optional[dict] = None, - ) -> tuple[str, dict]: - """Resolve model/runtime for a session. - - Priority (highest first): session ``/model`` → ``channel_overrides`` → global config/env - (``_resolve_gateway_model(user_config)`` and default provider resolution). - """ - resolved_session_key = self._resolve_session_key_or_none(source, session_key) - - model = _resolve_gateway_model(user_config) - if resolved_session_key: - self._rehydrate_session_model_override(resolved_session_key) - _override_state = ( - self._peek_session_state(resolved_session_key) - if resolved_session_key - else None - ) - override = ( - _override_state.conversation.model_override if _override_state else None - ) - if override: - override_model = override.get("model", model) - override_runtime = { - "provider": override.get("provider"), - "requested_provider": override.get("requested_provider"), - "api_key": override.get("api_key"), - "base_url": override.get("base_url"), - "api_mode": override.get("api_mode"), - "max_tokens": override.get("max_tokens"), - "credential_pool": override.get("credential_pool"), - "request_overrides": override.get("request_overrides"), - "capabilities": dict(override.get("capabilities") or {}), - } - if override_runtime.get("api_key"): - if override_runtime.get("credential_pool") is None: - override_runtime["credential_pool"] = _credential_pool_for_provider( - override.get("provider") - ) - logger.debug( - "Session model override (fast): session=%s config_model=%s -> override_model=%s provider=%s", - resolved_session_key or "", model, override_model, - override_runtime.get("provider"), - ) - return override_model, override_runtime - # Override exists but has no api_key — fall through to env-based - # resolution and apply model/provider from the override on top. - logger.debug( - "Session model override (no api_key, fallback): session=%s config_model=%s override_model=%s", - resolved_session_key or "", model, override_model, - ) - else: - logger.debug( - "No session model override: session=%s config_model=%s override_keys=%s", - resolved_session_key or "", model, - [ - _key - for _key, _st in list(self._sessions_map().items()) - if _st.conversation.model_override is not None - ][:5] or "[]", - ) - - runtime_kwargs = _resolve_runtime_agent_kwargs() - runtime_model = runtime_kwargs.pop("model", None) - if runtime_model: - logger.info( - "Runtime provider supplied explicit model override: %s -> %s", - model, - runtime_model, - ) - model = runtime_model - - cfg = getattr(self, "config", None) - if cfg and source is not None: - chat_id = str(source.chat_id) if source.chat_id else "" - thread_id = ( - str(source.thread_id) if getattr(source, "thread_id", None) else None - ) - parent_id = ( - str(source.parent_chat_id) - if getattr(source, "parent_chat_id", None) - else None - ) - ch = _get_channel_override( - cfg, - source.platform, - chat_id, - thread_id=thread_id, - parent_id=parent_id, - ) - if ch: - if ch.model: - model = ch.model - if ch.provider: - runtime_kwargs = _resolve_runtime_agent_kwargs_for_provider( - ch.provider - ) - ch_runtime_model = runtime_kwargs.pop("model", None) - # Only adopt the provider's bundled model when the override - # did not specify an explicit model. - if ch_runtime_model and not ch.model: - model = ch_runtime_model - - if override and resolved_session_key: - model, runtime_kwargs = self._apply_session_model_override( - resolved_session_key, model, runtime_kwargs - ) - - # No model.default but a provider resolved (e.g. `hermes auth add openai-codex` without - # `hermes model`): fall back to the provider's first catalog model so the API call has one. - if not model and runtime_kwargs.get("provider"): - try: - from hermes_cli.models import get_default_model_for_provider - model = get_default_model_for_provider(runtime_kwargs["provider"]) - if model: - logger.info( - "No model configured — defaulting to %s for provider %s", - model, runtime_kwargs["provider"], - ) - except Exception: - pass - - # Final safety net: if resolution still produced an empty model (e.g. a transient config-cache - # miss on a post-interrupt recovery turn), reuse the last model resolved for this session, - # else the most recent process-wide — model="" makes every API call fail HTTP 400 and the - # session goes silent. ``getattr`` guards bare test runners built via ``object.__new__``. - if not model: - _lr_state = ( - self._peek_session_state(resolved_session_key) - if resolved_session_key - else None - ) - _lr_star = self._peek_session_state("*") - _recovered = ( - (_lr_state.conversation.last_resolved_model if _lr_state else "") - or (_lr_star.conversation.last_resolved_model if _lr_star else "") - ) - if _recovered: - logger.warning( - "Empty model resolved for session=%s — recovering " - "last-known-good model %s (config read likely returned " - "empty; see #35314)", - resolved_session_key or "", _recovered, - ) - model = _recovered - elif model: - # Cache the good resolution for future recovery turns. - if resolved_session_key: - self._session_state( - resolved_session_key - ).conversation.last_resolved_model = model - self._session_state("*").conversation.last_resolved_model = model - - return model, runtime_kwargs - - def _resolve_turn_agent_config(self, user_message: str, model: str, runtime_kwargs: dict) -> dict: - """Build the effective model/runtime config for a single turn. - - Always uses the session's primary model/provider. If `/fast` is enabled and the model - supports it, attach `request_overrides` for priority processing. Per-provider - ``request_overrides`` from ``resolve_runtime_provider`` (e.g. ``custom_providers`` - ``extra_body``) are merged *under* the fast-mode overrides so they still reach the model. - """ - from hermes_cli.models import resolve_fast_mode_overrides - - runtime = { - "api_key": runtime_kwargs.get("api_key"), - "base_url": runtime_kwargs.get("base_url"), - "provider": runtime_kwargs.get("provider"), - "requested_provider": runtime_kwargs.get("requested_provider"), - "api_mode": runtime_kwargs.get("api_mode"), - "command": runtime_kwargs.get("command"), - "args": list(runtime_kwargs.get("args") or []), - "credential_pool": runtime_kwargs.get("credential_pool"), - "max_tokens": runtime_kwargs.get("max_tokens"), - "capabilities": dict(runtime_kwargs.get("capabilities") or {}), - } - base_request_overrides = dict(runtime_kwargs.get("request_overrides") or {}) - route = { - "model": model, - "runtime": runtime, - "signature": ( - model, - runtime["provider"], - runtime["requested_provider"], - runtime["base_url"], - runtime["api_mode"], - runtime["command"], - tuple(runtime["args"]), - ), - } - - # Provider-level request_overrides (e.g. a custom_providers extra_body) resolved upstream by - # resolve_runtime_provider(). - service_tier = getattr(self, "_service_tier", None) - if service_tier != "priority": - # None (normal) or auto/cold — the bounded window is applied per - # request by agent.fast_mode, not pinned into request_overrides. - route["request_overrides"] = base_request_overrides - return route - - try: - overrides = resolve_fast_mode_overrides( - route["model"], - provider=runtime["provider"], - base_url=runtime["base_url"], - ) - except Exception: - overrides = None - # Fast-mode overrides (service_tier / speed) are top-level keys and do - # not collide with extra_body; deep-merge them over the provider overrides. - route["request_overrides"] = _deep_merge_request_overrides( - base_request_overrides, - overrides or {}, - ) - return route - - def _sync_session_model_from_agent(self, session_id: str, agent: Any) -> None: - """Persist the runtime model/provider actually used by a gateway turn. - - Provider fallback can switch ``agent.model``/``agent.provider`` after the session row was - created; keep the DB metadata in sync so session lists and tooling report the backend that - actually answered. Runs in the ``run_sync`` closure (executor thread), so it uses the sync - ``SessionDB`` (``_db``) directly rather than the AsyncSessionDB forwarder. - """ - if not session_id or agent is None or self._session_db is None: - return - model = getattr(agent, "model", None) - if not model: - return - runtime = { - "provider": getattr(agent, "provider", None), - "base_url": getattr(agent, "base_url", None), - "api_mode": getattr(agent, "api_mode", None), - "fallback_active": bool(getattr(agent, "_fallback_activated", False)), - } - runtime = {k: v for k, v in runtime.items() if v not in (None, "")} - - try: - db = self._session_db._db - row = db.get_session(session_id) - if not row: - return - current_model = row.get("model") - raw_config = row.get("model_config") - try: - config = json.loads(raw_config) if raw_config else {} - except Exception: - config = {} - if not isinstance(config, dict): - config = {} - gateway_runtime = dict(config.get("gateway_runtime") or {}) - if current_model == model and all( - gateway_runtime.get(k) == v for k, v in runtime.items() - ): - return - config["gateway_runtime"] = runtime - db.update_session_meta(session_id, json.dumps(config), model=model) - except Exception: - logger.debug("Failed to sync gateway session model metadata", exc_info=True) - - async def _handle_reaction_event(self, ctx: Dict[str, Any]) -> None: - """Fan a normalised platform reaction event out to the HookRegistry. - - The adapter-supplied ``event_name`` ("reaction:added"/"reaction:removed") is the hook event, - matching the ``agent:*`` naming scheme. Errors never block the adapter's event loop. - """ - event_name = str(ctx.get("event_name") or "reaction:added") - try: - await self.hooks.emit(event_name, ctx) - except Exception: - logger.debug("[Gateway] reaction hook emit failed", exc_info=True) - - async def _handle_adapter_fatal_error(self, adapter: BasePlatformAdapter) -> None: - """React to an adapter failure after startup. - - Retryable errors (network blip, DNS) queue the platform for background reconnection. - The notification arrives on the failing adapter's own polling task, and the disconnect in - the handler can cancel that task mid-flight (disconnect()'s current-task guard misses it - because _safe_adapter_disconnect closes in a wrapper task), stranding the platform between - the fatal log and the reconnect queue — so the real work runs in a detached task. - """ - tasks = getattr(self, "_fatal_handler_tasks", None) - if tasks is None: - tasks = self._fatal_handler_tasks = set() - task = asyncio.create_task(self._handle_adapter_fatal_error_detached(adapter)) - tasks.add(task) - task.add_done_callback(tasks.discard) - # Await so callers that expect completion still get it — but through shield(): Task.cancel() - # on the caller also cancels the future it is awaiting (_fut_waiter), so a plain `await - # task` would tunnel the cancellation straight into the "detached" task. shield() absorbs - # it: the caller sees CancelledError, the handler runs to completion. - await asyncio.shield(task) - - def _queue_retryable_fatal_platform(self, adapter: BasePlatformAdapter) -> bool: - """Queue a retryable fatal adapter for background reconnection. - - Returns True when newly queued; idempotent if already queued. Must not await: callers - invoke this *before* any disconnect await so a wedged close cannot strand the platform. - """ - if not adapter.fatal_error_retryable: - return False - platform_config = self.config.platforms.get(adapter.platform) - if not platform_config: - return False - if adapter.platform in self._failed_platforms: - # Nothing to enqueue — but "already queued" is exactly when the watcher may have died, - # and the enqueue branch below holds the ONLY _ensure_reconnect_watcher_running() call. - # _spawn_supervised gives up after _MAX_SUPERVISED_RESTARTS; without this backstop a - # queued platform is a silent permanent outage (nothing retries, and the stranded check - # treats a queued platform as safe so the process never restarts either). - self._ensure_reconnect_watcher_running() - return False - self._failed_platforms[adapter.platform] = { - "config": platform_config, - "attempts": 0, - "next_retry": time.monotonic(), - "queued_at": time.monotonic(), - "credential_claim": self._adapter_credential_claim( - adapter.platform, adapter - ), - "listener_claim": self._adapter_listener_claim( - adapter.platform, adapter - ), - } - logger.info( - "%s queued for background reconnection", - adapter.platform.value, - ) - # Ensure the reconnect watcher is alive — respawn if it died (e.g. restart budget exhausted) - # so queued platforms are not permanently stranded. - self._ensure_reconnect_watcher_running() - return True - - async def _handle_adapter_fatal_error_detached( - self, adapter: BasePlatformAdapter - ) -> None: - """Run the fatal handler; if the platform still ends up stranded (not reconnected, not - queued, not intentionally disabled), exit the gateway with failure so the service manager - restarts it instead of leaving a silent partial outage.""" - try: - # Outer hard deadline: even with queue-before-disconnect, a hang anywhere in the impl - # (status write side effects, detach races, etc.) must not leave this task wedged - # forever — the stranded check in ``finally`` only runs when we return. - timeout = self._adapter_disconnect_timeout_secs() - if timeout <= 0: - await self._handle_adapter_fatal_error_impl(adapter) - else: - # Disconnect budget plus a little queue/status bookkeeping overhead; keep the extra - # proportional so tests that shrink the disconnect timeout still finish promptly. - outer = timeout + min(2.0, max(0.05, timeout)) - completed = await self._await_adapter_cleanup_with_timeout( - self._handle_adapter_fatal_error_impl(adapter), - outer, - ) - if not completed: - logger.error( - "Fatal-error handling for %s timed out after %.1fs; " - "ensuring reconnect queue is populated", - adapter.platform.value, - outer, - ) - self._queue_retryable_fatal_platform(adapter) - except asyncio.CancelledError: - # Best-effort queue before re-raising: a cancelled fatal handler - # must not strand a retryable platform (#80598). - try: - self._queue_retryable_fatal_platform(adapter) - except Exception: - logger.debug( - "Failed to queue %s after fatal-handler cancellation", - adapter.platform.value, - exc_info=True, - ) - raise - except Exception: - logger.exception( - "Fatal-error handling for %s raised unexpectedly", - adapter.platform.value, - ) - # Best-effort queue so an unexpected raise mid-handler cannot - # leave a retryable platform permanently deaf (#80598). - try: - self._queue_retryable_fatal_platform(adapter) - except Exception: - logger.debug( - "Failed to queue %s after fatal-handler exception", - adapter.platform.value, - exc_info=True, - ) - finally: - platform = adapter.platform - shutdown_event = getattr(self, "_shutdown_event", None) - stranded = ( - adapter.fatal_error_retryable - and platform not in self.adapters - and platform not in getattr(self, "_failed_platforms", {}) - and not (shutdown_event is not None and shutdown_event.is_set()) - ) - if stranded: - logger.error( - "%s adapter was lost without entering the reconnection " - "queue; exiting gateway so the service manager restarts it.", - platform.value, - ) - self._exit_reason = ( - f"{platform.value} adapter lost without reconnection queue" - ) - self._exit_with_failure = True - await self.stop() - - async def _handle_adapter_fatal_error_impl(self, adapter: BasePlatformAdapter) -> None: - # Snapshot this platform slot's current owner first: acting on a stale notification would - # overwrite a healthy platform's runtime status and wrongly re-queue it for reconnection. - existing = self.adapters.get(adapter.platform) - if existing is not None and existing is not adapter: - logger.debug( - "Ignoring stale fatal error from a superseded %s adapter instance: %s", - adapter.platform.value, - adapter.fatal_error_code or "unknown", - ) - return - - logger.error( - "Fatal %s adapter error (%s): %s", - adapter.platform.value, - adapter.fatal_error_code or "unknown", - adapter.fatal_error_message or "unknown error", - ) - # A relay credential revoked by opt-out is not an error to retry: render a clean "disabled" - # state, not red "fatal"/"retrying" (non-retryable code, so it also leaves the queue below). - if adapter.fatal_error_code == "relay_disabled": - platform_state = "disabled" - elif adapter.fatal_error_retryable: - platform_state = "retrying" - else: - platform_state = "fatal" - self._update_platform_runtime_status( - adapter.platform.value, - platform_state=platform_state, - error_code=adapter.fatal_error_code, - error_message=adapter.fatal_error_message, - ) - - if existing is adapter: - # Claim this adapter for teardown before awaiting disconnect(): a second fatal-error - # notification for the same adapter (e.g. a concurrent recovery path) would otherwise - # still see itself as "existing" during the await and disconnect() the same object twice. - self.adapters.pop(adapter.platform, None) - self.delivery_router.adapters = self.adapters - - # Queue retryable failures BEFORE any disconnect await: a half-dead transport can wedge - # native close() (or swallow CancelledError), so "disconnect then queue" left platforms - # permanently deaf in a live process after the network recovered. Populate the queue first so the - # reconnect watcher always has work; teardown is best-effort after. - self._queue_retryable_fatal_platform(adapter) - - if existing is adapter: - # A half-closed transport can wedge native close() indefinitely; reuse the shutdown-path - # timeout so this runtime fatal handler always returns to the stay-alive / stranded path. - await self._safe_adapter_disconnect(adapter, adapter.platform) - - if not self.adapters and not self._failed_platforms: - self._exit_reason = adapter.fatal_error_message or "All messaging adapters disconnected" - if adapter.fatal_error_retryable: - self._exit_with_failure = True - logger.error("No connected messaging platforms remain. Shutting down gateway for service restart.") - else: - logger.error("No connected messaging platforms remain. Shutting down gateway cleanly.") - await self.stop() - elif not self.adapters and self._failed_platforms: - # All platforms are down and queued for reconnection. Keep the gateway alive so cron jobs - # still run and the watcher can recover platforms when the problem clears; exiting for a - # systemd restart would turn a transient outage into a state-killing restart loop. - logger.warning( - "No connected messaging platforms remain, but %d platform(s) " - "queued for reconnection — gateway staying alive, watcher will " - "retry in background.", - len(self._failed_platforms), - ) - - def _request_clean_exit(self, reason: str) -> None: - self._exit_cleanly = True - self._exit_reason = reason - self._shutdown_event.set() def _running_agent_count(self) -> int: return len(self._running_agents) - def _active_work_count(self) -> int: - """All agent work the gateway must expose and drain as one total.""" - return ( - self._running_agent_count() - + self._active_cron_job_count() - + self._active_api_run_count() - + self._active_deferred_agent_worker_count() - ) - - def _active_cron_job_count(self) -> int: - """Count of cron jobs currently executing (``cron.scheduler._running_job_ids``). - - Cron jobs run on the scheduler's own thread pool, outside ``self._running_agents`` which - every OTHER active-work check reads; without this the shutdown drain can kill a cron job's - tool subprocess mid-run. Best-effort: returns 0 if the cron module can't be imported. - """ - try: - from cron.scheduler import get_running_job_ids - return len(get_running_job_ids()) - except Exception: - return 0 - - def _active_api_run_count(self) -> int: - """Count API-server work that is outside ``_running_agents``. - - Only the primary API server owns the HTTP listener (secondary multiplex profiles cannot - bind a port), so only the primary registry is a source of this work. - """ - try: - adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) - helper = getattr(adapter, "active_agent_work_count", None) - return max(0, int(helper())) if callable(helper) else 0 - except Exception: - return 0 - - def _interrupt_api_server_runs(self, reason: str) -> int: - """Interrupt API-server agents that are not in ``_running_agents``. - - Counterpart of ``_active_api_run_count()``: must reach the same agents when the drain times - out. Duck-typed so an adapter (or test double) without the hook is skipped, not raised on. - """ - try: - adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) - helper = getattr(adapter, "interrupt_active_runs", None) - return max(0, int(helper(reason))) if callable(helper) else 0 - except Exception as exc: - logger.debug("Failed interrupting api_server runs during shutdown: %s", exc) - return 0 - - def _active_deferred_agent_worker_count(self) -> int: - """Count executor workers that outlived their owning gateway turn. - - A timed-out hygiene compression keeps running in its executor thread. - Some paths defer agent cleanup; the live Codex path keeps its cached - agent. In both cases the turn can finish before the worker does, so - ``_running_agents`` no longer represents it. Count the worker itself. - """ - workers = getattr(self, "_deferred_agent_workers", None) - if not isinstance(workers, dict): - return 0 - return sum(1 for future in list(workers) if not future.done()) - - def _track_deferred_agent_worker( - self, - future: asyncio.Future, - agent: Any, - ) -> None: - """Expose an executor worker to drain/interrupt until it really exits.""" - workers = getattr(self, "_deferred_agent_workers", None) - if workers is None: - workers = {} - self._deferred_agent_workers = workers - workers[future] = agent - - def _discard_worker(done_future: asyncio.Future) -> None: - workers.pop(done_future, None) - # Some tracked workers intentionally outlive the coroutine that - # started them and therefore have no later waiter. Consume their - # terminal exception so asyncio does not emit an unhandled-future - # warning after the worker eventually unwinds (#98973). - if not done_future.cancelled(): - try: - done_future.exception() - except Exception: - pass - - future.add_done_callback(_discard_worker) - - def _interrupt_deferred_agent_workers(self, reason: str) -> int: - """Request cancellation of detached executor-backed agent work.""" - workers = getattr(self, "_deferred_agent_workers", None) - if not isinstance(workers, dict): - return 0 - interrupted = 0 - seen: set[int] = set() - for future, agent in list(workers.items()): - if future.done() or agent is None or id(agent) in seen: - continue - seen.add(id(agent)) - try: - request_hard_interrupt(agent, reason) - interrupted += 1 - except Exception as exc: - logger.debug( - "Failed interrupting deferred agent worker during shutdown: %s", - exc, - ) - return interrupted # ── scale-to-zero idle detection / dormant-quiesce (Phase 0) ────────────── # The gateway-side BEHAVIOUR that consumes the relay scale-to-zero primitives (gateway-gateway # Phase 5). Pure logic lives in gateway/scale_to_zero.py; the methods here bind it to the live # runner/transport. - def _scale_to_zero_has_live_background_work(self) -> bool: - """Live background work that must block a suspend. - - Backgrounded delegate_task / kanban / terminal(background=true) are NOT counted by - _running_agent_count() but suspending loses them; checks tracked tasks + process registry + - pending completion watchers. PERMANENT supervised watchers (_hermes_supervised_watcher) are - excluded — they live for the whole process (including the scale-to-zero watcher itself), so - counting them would make this True forever and the gateway could never go dormant. - """ - if any( - not t.done() and not getattr(t, "_hermes_supervised_watcher", False) - for t in self._background_tasks - ): - return True - try: - from tools.async_delegation import active_count - - if active_count() > 0: - return True - except Exception: # noqa: BLE001 - never let the idle check raise - logger.debug("scale-to-zero async-delegation check failed", exc_info=True) - try: - from tools.process_registry import process_registry - - if process_registry.has_any_active(): - return True - if process_registry.pending_watchers: - return True - except Exception: # noqa: BLE001 - never let the idle check raise - logger.debug("scale-to-zero bg-work check failed", exc_info=True) - return False - - def _scale_to_zero_idle_timeout_seconds(self) -> float: - from gateway.scale_to_zero import parse_idle_timeout_seconds - - raw = None - try: - user_cfg = _load_gateway_config() - gw = user_cfg.get("gateway") if isinstance(user_cfg, dict) else None - stz = gw.get("scale_to_zero") if isinstance(gw, dict) else None - if isinstance(stz, dict): - raw = stz.get("idle_timeout_minutes") - except Exception: # noqa: BLE001 - raw = None - return parse_idle_timeout_seconds(raw) - - def _restart_loop_guard_config(self) -> tuple: - """Return ``(max_restarts, window_seconds, max_gap_seconds)`` for the auto-resume - restart-loop breaker, from ``gateway.restart_loop_guard`` with module defaults as fallback. - - ``max_restarts <= 0`` disables the breaker. ``max_gap_seconds`` is the longest spacing - between consecutive restart-interrupted boots that still counts as the same loop, so a - crash cycle slower than ``window_seconds`` stays visible. - """ - from gateway import restart_loop_guard as _rlg - - max_restarts = _rlg.DEFAULT_MAX_RESTARTS - window_seconds = _rlg.DEFAULT_WINDOW_SECONDS - max_gap_seconds = _rlg.DEFAULT_MAX_GAP_SECONDS - try: - user_cfg = _load_gateway_config() - gw = user_cfg.get("gateway") if isinstance(user_cfg, dict) else None - rlg = gw.get("restart_loop_guard") if isinstance(gw, dict) else None - if isinstance(rlg, dict): - if isinstance(rlg.get("max_restarts"), int): - max_restarts = rlg["max_restarts"] - if isinstance(rlg.get("window_seconds"), int) and rlg["window_seconds"] > 0: - window_seconds = rlg["window_seconds"] - if ( - isinstance(rlg.get("max_gap_seconds"), int) - and rlg["max_gap_seconds"] > 0 - ): - max_gap_seconds = rlg["max_gap_seconds"] - except Exception: # noqa: BLE001 - pass - return max_restarts, window_seconds, max_gap_seconds - - def _scale_to_zero_active_messaging_platforms(self) -> list: - """ENABLED platforms that count for the relay-only arm gate. - - Two load-bearing filters: enabled only (config.platforms is pre-seeded with disabled - placeholders for the whole catalog) and MESSAGING only (the api_server is a loopback listener - force-enabled on every hosted container with no outbound socket; counting it silently - disarmed the feature everywhere). Mirrors the non-messaging exclusion in _connect_platforms. - """ - if not self.config: - return [] - non_messaging = {Platform.LOCAL, Platform.API_SERVER, Platform.WEBHOOK} - try: - return [ - p - for p, pc in self.config.platforms.items() - if getattr(pc, "enabled", False) and p not in non_messaging - ] - except Exception: # noqa: BLE001 - return [] - - def _scale_to_zero_should_arm(self) -> bool: - """Whether to start the idle watcher (D1/D11/§3.4(1)).""" - from gateway.relay import relay_wake_url - from gateway.scale_to_zero import ( - messaging_is_relay_only_or_absent, - scale_to_zero_enabled, - should_arm, - ) - - platforms = self._scale_to_zero_active_messaging_platforms() - try: - wake_url = relay_wake_url() - except Exception: # noqa: BLE001 - wake_url = None - return should_arm( - enabled=scale_to_zero_enabled(), - relay_only_or_absent=messaging_is_relay_only_or_absent(platforms), - wake_url=wake_url, - ) - - def _log_scale_to_zero_not_armed_reason(self) -> None: - """Log why the idle watcher did NOT arm — but only for an OPTED-IN instance. - - A non-opted instance (no HERMES_SCALE_TO_ZERO stamp) not arming is normal and stays silent; - with the stamp set, the surprise earns one INFO line so the answer is a log grep. - """ - from gateway.relay import relay_wake_url - from gateway.scale_to_zero import ( - messaging_is_relay_only_or_absent, - scale_to_zero_enabled, - ) - - try: - enabled = scale_to_zero_enabled() - if not enabled: - return # not opted in — normal, stay quiet - active = [ - getattr(p, "value", p) - for p in self._scale_to_zero_active_messaging_platforms() - ] - relay_only = messaging_is_relay_only_or_absent(active) - try: - wake_url = relay_wake_url() - except Exception: # noqa: BLE001 - wake_url = None - logger.info( - "scale-to-zero: NOT armed despite opt-in — " - "relay_only_or_absent=%s (enabled platforms=%s), wake_url=%s. " - "Need relay-only messaging + a registered wake URL.", - relay_only, - active or "none", - "set" if wake_url else "MISSING", - ) - except Exception: # noqa: BLE001 - diagnostics must never block startup - logger.debug("scale-to-zero: not-armed reason logging failed", exc_info=True) - - def _scale_to_zero_is_idle(self) -> bool: - from gateway.scale_to_zero import is_idle - - # The FULL work aggregate, not _running_agent_count(): cron jobs and API-server runs live - # outside _running_agents, so counting agents alone let a suspend land mid-cron-job. - # Fail-AWAKE accounting: the shutdown-drain counters swallow exceptions to 0, which is fine - # for a drain but unsafe for a suspend predicate (a transient read failure would look idle). - # Here an unreadable source counts as work (sentinel 1) so the machine stays awake. - try: - from cron.scheduler import get_running_job_ids - - cron_count = len(get_running_job_ids()) - except Exception: # noqa: BLE001 - unreadable source => assume busy - logger.debug("scale-to-zero: cron work count unreadable — staying awake", exc_info=True) - cron_count = 1 - try: - adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) - helper = getattr(adapter, "active_agent_work_count", None) - api_count = max(0, int(helper())) if callable(helper) else 0 - except Exception: # noqa: BLE001 - unreadable source => assume busy - logger.debug("scale-to-zero: api work count unreadable — staying awake", exc_info=True) - api_count = 1 - # An attached dashboard/desktop/TUI client is inbound activity too; it lives in the DASHBOARD - # process and reaches us as a file mtime refreshed on every WS frame (gateway/scale_to_zero.py). - # Folded into the inbound clock rather than a conjunct: same idle_timeout grace after - # disconnect as a chat message, and a lingering marker cannot pin the box. - last_inbound = self._last_inbound_at - try: - from gateway.scale_to_zero import dashboard_client_last_seen - - seen = dashboard_client_last_seen() - except Exception: # noqa: BLE001 - unreadable source => assume busy - logger.debug("scale-to-zero: dashboard heartbeat unreadable — staying awake", exc_info=True) - seen = time.time() - if seen is not None and seen > last_inbound: - last_inbound = seen - return is_idle( - active_work_count=self._running_agent_count() + cron_count + api_count, - seconds_since_last_inbound=time.time() - last_inbound, - idle_timeout_seconds=self._scale_to_zero_idle_timeout_seconds(), - has_live_background_work=self._scale_to_zero_has_live_background_work(), - ) - - def _scale_to_zero_note_real_inbound(self) -> None: - """Stamp real inbound and restore lifecycle after a dormant wake. - - Dormancy marks status `draining` but is not the stop/restart drain: the process stays alive - and should present as running once real traffic wakes it. Internal completion/replay events - deliberately do not call this, so they don't keep an idle gateway awake. - """ - self._last_inbound_at = time.time() - if getattr(self, "_scale_to_zero_cooldown_until", 0.0) > 0: - try: - self._update_runtime_status("running") - except Exception: # noqa: BLE001 - status restoration is best-effort - logger.debug("scale-to-zero: status restore failed", exc_info=True) - self._scale_to_zero_cooldown_until = 0.0 - - def _relay_adapter_for_dormancy(self): - """Return the connected RELAY adapter, if any (the one go_dormant targets).""" - try: - from gateway.platforms.base import Platform - except Exception: # noqa: BLE001 - return None - return self.adapters.get(Platform.RELAY) - - async def _scale_to_zero_watcher(self, interval: float = 30.0) -> None: - """Watch for idle, drive the relay dormant, then self-suspend the machine. - - Armed ONLY via _scale_to_zero_should_arm() (HERMES_SCALE_TO_ZERO stamp + relay-only/absent - messaging + wakeUrl). On sustained idle: mark status `draining` (NOT _running=False), relay - adapter.go_dormant() (supervisor-preserving socket close, NOT disconnect()), NO - mark_resume_pending (suspend preserves RAM), THEN suspend via the local flaps socket. The - gateway owns the suspend because Fly autostop sees only INBOUND connections and would freeze - mid-job or before the relay flip (machines run autostop:"off"); autostart stays platform-side. - A re-arm cooldown keeps a wake's drained backlog from being re-quiesced. Off-Fly (no flaps - socket) the watcher does not quiesce at all. - """ - await asyncio.sleep(min(interval, 30.0)) # let startup settle - while self._running: - try: - await asyncio.sleep(interval) - if not self._running: - return - if time.time() < self._scale_to_zero_cooldown_until: - continue - if not self._scale_to_zero_is_idle(): - continue - adapter = self._relay_adapter_for_dormancy() - if adapter is None: - continue - go_dormant = getattr(adapter, "go_dormant", None) - if not callable(go_dormant): - continue - # Quiesce only when a suspend can follow. Off-Fly the platform owns the freeze and - # go_dormant()'s socket close arms the reconnect supervisor (re-dial ~1.4s, unflipped - # at freeze, inbound dropped not buffered); stay connected, orphan detection adopts it. - from gateway.scale_to_zero import self_suspend_available - - if not self_suspend_available(): - if not self._scale_to_zero_no_suspend_logged: - self._scale_to_zero_no_suspend_logged = True - logger.info( - "scale-to-zero: idle, but this platform suspends on " - "its own timer (no in-machine suspend API); staying " - "connected rather than quiescing" - ) - continue - logger.info( - "scale-to-zero: gateway idle for >= %.0fs — going dormant " - "(relay buffered, socket closed) then self-suspending", - self._scale_to_zero_idle_timeout_seconds(), - ) - try: - self._update_runtime_status("draining") - except Exception: # noqa: BLE001 - status is best-effort - logger.debug("scale-to-zero: status mark failed", exc_info=True) - dormant_ok = True - try: - result = go_dormant() - if asyncio.iscoroutine(result): - await result - except Exception: # noqa: BLE001 - dormancy is best-effort - dormant_ok = False - logger.debug("scale-to-zero: go_dormant failed", exc_info=True) - # After a wake the drained inbound updates _last_inbound_at; give it a window so we - # don't immediately re-go-dormant on the same idle reading before traffic lands. - self._scale_to_zero_cooldown_until = time.time() + max(interval, 60.0) - # Self-suspend ONLY after a clean quiesce: the relay flip (buffered delivery + wake - # poke armed) must be set before the freeze, or inbound black-holes while we sleep. - # Re-check idle one last time — inbound may have landed during the quiesce await. - if not dormant_ok: - continue - if not self._scale_to_zero_is_idle(): - logger.info( - "scale-to-zero: inbound arrived during quiesce — skipping suspend" - ) - continue - await self._scale_to_zero_self_suspend() - except asyncio.CancelledError: - raise - except Exception: # noqa: BLE001 - the watcher must never crash the gateway - logger.debug("scale-to-zero watcher iteration error", exc_info=True) - - async def _scale_to_zero_self_suspend(self) -> None: - """Suspend this Fly machine via the local flaps socket (fail-awake). - - Blocking unix-socket call runs in a worker thread so the loop stays live until the kernel - freeze; nothing meaningful runs until wake. Off-Fly this is a silent no-op. - """ - from gateway.scale_to_zero import self_suspend_available, suspend_self - - try: - if not self_suspend_available(): - logger.debug( - "scale-to-zero: flaps socket / machine identity absent — " - "dormant without platform suspend" - ) - return - accepted = await asyncio.to_thread(suspend_self) - if not accepted: - logger.warning( - "scale-to-zero: self-suspend not accepted — machine stays " - "awake (fail-awake); will retry on the next idle window" - ) - except Exception: # noqa: BLE001 - suspend is best-effort, never crash - logger.debug("scale-to-zero: self-suspend failed", exc_info=True) def _status_action_label(self) -> str: return "restart" if self._restart_requested else "shutdown" @@ -8507,160 +4841,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew def _status_action_gerund(self) -> str: return "restarting" if self._restart_requested else "shutting down" - def _queue_during_drain_enabled( - self, busy_input_mode: Optional[str] = None - ) -> bool: - # "queue" and "steer" both mean messages must not be lost across restart: queue them for - # the newly-spawned gateway process to pick up. "interrupt" mode drops them. - mode = busy_input_mode or self._busy_input_mode - return self._restart_requested and mode in {"queue", "steer"} # -------- /queue FIFO helpers -------------------------------------- # /queue yields one full agent turn per invocation, FIFO, no merging. _pending_messages is a # single "next-up" slot (shared with photo-burst follow-ups) holding the head; an overflow list # holds the tail. Promotion after each run's drain refills the slot. Cleared on /new and /reset. - def _enqueue_fifo(self, session_key: str, queued_event: "MessageEvent", adapter: Any) -> None: - """Append a /queue event to the FIFO chain for a session.""" - if adapter is None: - return - pending_slot = getattr(adapter, "_pending_messages", None) - if pending_slot is None: - return - if session_key in pending_slot: - self._session_state(session_key).conversation.queued_events.append( - queued_event - ) - else: - pending_slot[session_key] = queued_event - - def _promote_queued_event( - self, - session_key: str, - adapter: Any, - pending_event: Optional["MessageEvent"], - ) -> Optional["MessageEvent"]: - """Promote the next overflow item after the slot was drained. - - If pending_event is None, return the overflow head as the new pending_event; if the slot is - already populated (interrupt follow-up etc.), stage the head there for the NEXT recursion. - Returns the (possibly updated) pending_event. - """ - _q_state = self._peek_session_state(session_key) - overflow = _q_state.conversation.queued_events if _q_state else None - if not overflow: - return pending_event - next_queued = overflow.pop(0) - if pending_event is None: - return next_queued - if adapter is not None and hasattr(adapter, "_pending_messages"): - adapter._pending_messages[session_key] = next_queued - else: - # No adapter — push back so we don't silently drop the item. - overflow.insert(0, next_queued) - return pending_event - - def _queue_depth(self, session_key: str, *, adapter: Any = None) -> int: - """Total pending /queue items for a session — slot + overflow.""" - _q_state = self._peek_session_state(session_key) - depth = len(_q_state.conversation.queued_events) if _q_state else 0 - if adapter is not None and session_key in getattr(adapter, "_pending_messages", {}): - depth += 1 - return depth - - def _rescue_orphaned_overflow( - self, session_key: str, adapter: Any - ) -> Optional["MessageEvent"]: - """Pop the oldest orphaned FIFO overflow event for an idle session. - - ``queued_events`` drains only at the post-turn promotion site in ``_run_agent``; if a busy - window ends without that drain (early recursion exit, exception/interrupt/generation-bump), - the overflow is silently orphaned. Called when a NEW event arrives for a NON-busy session: - the oldest orphan is returned to run as THIS turn, the next is staged into the slot so the - chain continues in arrival order, and the caller enqueues the incoming event behind it. The - returned event is REMOVED from both stores, else the post-turn dequeue would run it twice. - Returns ``None`` when there is nothing to rescue (no overflow, slot occupied, or no slot). - """ - try: - _q_state = self._peek_session_state(session_key) - overflow = _q_state.conversation.queued_events if _q_state else None - if not overflow: - return None - pending_slot = getattr(adapter, "_pending_messages", None) - if not isinstance(pending_slot, dict) or pending_slot.get(session_key): - # Slot occupied (busy) or no slot storage — promotion owns - # this; do not fight it from the idle path. - return None - head = overflow.pop(0) - # Keep the slot occupied for the rest of the chain so the drain promotes in order and - # any mid-chain arrival routes to overflow instead of jumping the queue (same invariant - # as the drain's own _promote_queued_event). Only ONE event fits the slot. - if overflow: - pending_slot[session_key] = overflow.pop(0) - logger.warning( - "Rescued orphaned FIFO overflow event for idle session " - "%s — it was queued during a busy window but the post-turn " - "drain never promoted it (#99882)", - session_key, - ) - if overflow: - logger.warning( - "%d overflow event(s) still queued for session %s after " - "rescue staging (will drain via normal promotion)", - len(overflow), - session_key, - ) - return head - except Exception: - logger.debug("FIFO overflow rescue failed for %s", session_key, exc_info=True) - return None - - @staticmethod - def _is_goal_continuation_event(event_or_text: Any) -> bool: - """Return True for synthetic /goal continuation turns. - - Goal continuations are normal queued user-role events, so pause/clear must distinguish - them from real user /queue messages before removing or suppressing them. - """ - text = getattr(event_or_text, "text", event_or_text) or "" - return str(text).startswith("[Continuing toward your standing goal]\nGoal:") - - def _clear_goal_pending_continuations(self, session_key: str, adapter: Any) -> int: - """Remove queued synthetic /goal continuations for one session. - - User /goal pause/clear can race a judge-queued continuation; only synthetic goal - continuations are removed, normal /queue and user follow-up events are preserved. - """ - removed = 0 - pending_slot = getattr(adapter, "_pending_messages", None) if adapter is not None else None - if isinstance(pending_slot, dict): - pending_event = pending_slot.get(session_key) - if self._is_goal_continuation_event(pending_event): - pending_slot.pop(session_key, None) - removed += 1 - - _q_state = self._peek_session_state(session_key) - overflow = _q_state.conversation.queued_events if _q_state else [] - if overflow: - kept = [] - for queued_event in overflow: - if self._is_goal_continuation_event(queued_event): - removed += 1 - else: - kept.append(queued_event) - _q_state.conversation.queued_events = kept - return removed - - def _goal_still_active_for_session(self, session_id: str) -> bool: - """Best-effort fresh DB check before running a queued continuation.""" - if not session_id: - return False - try: - from hermes_cli.goals import GoalManager - return GoalManager(session_id=session_id).is_active() - except Exception as exc: - logger.debug("goal continuation: active-state recheck failed: %s", exc) - return False def _update_runtime_status(self, gateway_state: Optional[str] = None, exit_reason: Optional[str] = None) -> None: _write_runtime_status_quiet( @@ -8680,726 +4866,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew """ _write_runtime_status_quiet(active_agents=self._active_work_count()) - # ------------------------------------------------------------------ - # External drain control (NAS-driven quiesce-without-restart). The dashboard's - # begin/cancel-drain endpoint writes/removes the ``.drain_request.json`` marker - # (gateway/drain_control.py); this watcher flips the gateway between accepting and refusing - # NEW turns WITHOUT exiting. Reversible: NAS begins drain, polls /api/status until - # active_agents hits 0, acts; on cancel/abort the marker is removed and turns resume. - # ------------------------------------------------------------------ - def _enter_external_drain(self) -> None: - """Begin external drain: refuse NEW turns (in-flight ones are NOT interrupted), flip state. - - Idempotent: re-entry only re-writes status. - """ - if self._external_drain_active: - return - self._external_drain_active = True - logger.info( - "External drain ENGAGED (.drain_request.json present) — refusing " - "new turns; %d in-flight turn(s) will finish. Process stays up.", - self._active_work_count(), - ) - # Flip persisted lifecycle state so /api/status.gateway_busy / gateway_drainable track the - # drain; active_agents is preserved (read-merge keeps the live count), only state changes. - self._update_runtime_status("draining") - - def _exit_external_drain(self) -> None: - """Cancel external drain: revert state, re-accept new turns. - - Idempotent. Reverts to ``running`` only when actually mid-drain AND not shutting down — - a real shutdown ``_draining`` must win; never resurrect a stopping gateway. - """ - if not self._external_drain_active: - return - self._external_drain_active = False - if self._draining or not self._running: - # A shutdown drain is in progress / the loop has stopped — do not - # clobber the terminal state back to running. - logger.info( - "External drain marker cleared during shutdown — not reverting " - "to running (shutdown takes precedence)." - ) - return - logger.info( - "External drain RELEASED (.drain_request.json removed) — " - "re-accepting new turns; gateway_state -> running." - ) - self._update_runtime_status("running") - - async def _drain_control_watcher(self, interval: float = 1.0) -> None: - """Background task: reconcile gateway accept-state with the drain marker. - - Polls ``.drain_request.json`` (presence-based) at 1s: present -> enter drain, absent -> exit; - reconciles once at startup. A marker from a PRIOR instantiation epoch (survived a machine - restart) is treated as absent. Best-effort: tick errors are logged and the loop continues. - """ - from gateway.drain_control import drain_requested - - while self._running: - try: - # drain_requested() does a synchronous read_text() on the marker file: at 1s cadence - # that is a blocking disk read on the event loop ~86k times/day, and under host I/O - # pressure one read can stall 30s+ and take every platform heartbeat down. Off-thread it. - if await asyncio.to_thread(drain_requested): - self._enter_external_drain() - # API and cron work live outside messaging's _running_agents map; refresh the - # aggregate while an external caller polls this reversible drain state. - self._persist_active_agents() - else: - self._exit_external_drain() - except asyncio.CancelledError: - raise - except Exception as exc: - logger.debug("Drain-control watcher tick error: %s", exc, exc_info=True) - await asyncio.sleep(interval) - - def _update_platform_runtime_status( - self, - platform: str, - *, - platform_state: Optional[str] = None, - error_code: Optional[str] = None, - error_message: Optional[str] = None, - needs_attention: Optional[bool] = None, - retrying_since: Any = _UNSET, - ) -> None: - try: - from gateway.status import write_runtime_status - extra: Dict[str, Any] = {} - if needs_attention is not None: - extra["needs_attention"] = needs_attention - if retrying_since is not _UNSET: - extra["retrying_since"] = retrying_since - write_runtime_status( - platform=platform, - platform_state=platform_state, - error_code=error_code, - error_message=error_message, - **extra, - ) - except Exception: - pass - - # ------------------------------------------------------------------ - # Per-platform circuit breaker (pause/resume): reconnect watcher + /platform pause|resume. - # ------------------------------------------------------------------ - def _pause_failed_platform(self, platform, *, reason: str = "") -> None: - """Mark a queued platform as paused — stays in ``_failed_platforms`` but the reconnect - watcher stops hammering it. - - Manual (``/platform pause ``) only: the watcher never auto-pauses — retryable failures - keep retrying at the backoff cap so a transient outage self-heals. - """ - info = getattr(self, "_failed_platforms", {}).get(platform) - if info is None: - return - if info.get("paused"): - return - info["paused"] = True - info["pause_reason"] = reason or "auto-paused after repeated failures" - # Push next_retry far enough out that even if "paused" is missed - # by a stale code path, the watcher won't fire on it. - info["next_retry"] = float("inf") - with suppress(Exception): - self._update_platform_runtime_status( - platform.value, - platform_state="paused", - error_code=None, - error_message=info["pause_reason"], - ) - logger.warning( - "%s paused after %d consecutive failures (%s) — " - "fix the underlying issue then run `/platform resume %s` " - "to retry, or `hermes gateway restart` to restart the gateway.", - platform.value, info.get("attempts", 0), - info["pause_reason"], platform.value, - ) - - def _resume_paused_platform(self, platform) -> bool: - """Unpause a platform — reset its attempt counter and schedule an - immediate retry. Returns True if the platform was paused and is - now queued; False if it wasn't paused (or wasn't in the queue). - """ - info = getattr(self, "_failed_platforms", {}).get(platform) - if info is None: - return False - if not info.get("paused"): - return False - info["paused"] = False - info.pop("pause_reason", None) - info["attempts"] = 0 - info["next_retry"] = time.monotonic() # retry on next watcher tick - with suppress(Exception): - self._update_platform_runtime_status( - platform.value, - platform_state="retrying", - error_code=None, - error_message=None, - ) - logger.info("%s resumed — retrying on next watcher tick", platform.value) - return True - - @staticmethod - def _load_prefill_messages() -> List[Dict[str, Any]]: - """Load ephemeral prefill messages from config or env var. - - HERMES_PREFILL_MESSAGES_FILE env wins, then top-level prefill_messages_file in config.yaml, - then legacy agent.prefill_messages_file. Relative paths resolve from ~/.hermes/. - """ - file_path = os.getenv("HERMES_PREFILL_MESSAGES_FILE", "") - if not file_path: - cfg = _load_gateway_runtime_config() - file_path = str(cfg.get("prefill_messages_file", "") or "") - if not file_path: - file_path = str(cfg_get(cfg, "agent", "prefill_messages_file", default="") or "") - if not file_path: - return [] - path = Path(file_path).expanduser() - if not path.is_absolute(): - path = _hermes_home / path - if not path.exists(): - logger.warning("Prefill messages file not found: %s", path) - return [] - try: - with open(path, "r", encoding="utf-8") as f: - data = json.load(f) - if not isinstance(data, list): - logger.warning("Prefill messages file must contain a JSON array: %s", path) - return [] - return data - except Exception as e: - logger.warning("Failed to load prefill messages from %s: %s", path, e) - return [] - - @staticmethod - def _load_ephemeral_system_prompt() -> str: - """Load ephemeral system prompt: HERMES_EPHEMERAL_SYSTEM_PROMPT env var first, then - ``display.personality`` / ``agent.system_prompt`` in config.yaml. - """ - from hermes_cli.config import resolve_ephemeral_system_prompt_from_config - - prompt = os.getenv("HERMES_EPHEMERAL_SYSTEM_PROMPT", "") - if prompt: - return prompt - cfg = _load_gateway_runtime_config() - return resolve_ephemeral_system_prompt_from_config(cfg) - - def _resolve_model_for_channel( - self, - platform: Platform, - chat_id: str, - *, - user_config: Optional[dict] = None, - thread_id: Optional[str] = None, - parent_id: Optional[str] = None, - ) -> str: - """Resolve model for this channel: channel_overrides else global default. - - Precedence lives in :func:`hermes_cli.model_switch.resolve_effective_model` (shared with the - API server so the surfaces cannot diverge). No session tier here: session /model overrides - are applied later by ``_apply_session_model_override``. - """ - from hermes_cli.model_switch import resolve_effective_model - - override = None - config = getattr(self, "config", None) - if config: - override = _get_channel_override( - config, - platform, - chat_id, - thread_id=thread_id, - parent_id=parent_id, - ) - return resolve_effective_model( - None, # session tier applied downstream (_apply_session_model_override) - override, - _resolve_gateway_model(user_config), - ) - - def _get_system_prompt_for_channel( - self, - platform: Platform, - chat_id: str, - *, - thread_id: Optional[str] = None, - parent_id: Optional[str] = None, - ) -> str: - """Ephemeral system prompt for this channel/thread. - - ``channel_overrides`` when set, else the gateway prompt resolved from the CURRENT profile's - config on every call (callers run inside ``_profile_runtime_scope``, so routed multiplex - profiles get their own personality/system_prompt and ``/personality`` edits apply next turn). - Legacy ``channel_prompts`` are applied separately via ``event.channel_prompt`` in ``run_sync``. - """ - config = getattr(self, "config", None) - if config: - override = _get_channel_override( - config, - platform, - chat_id, - thread_id=thread_id, - parent_id=parent_id, - ) - if override and override.system_prompt: - return (override.system_prompt or "").strip() - return self._load_ephemeral_system_prompt() - - @staticmethod - def _load_reasoning_config(model: str = "") -> dict | None: - """Load reasoning effort from config.yaml, respecting per-model overrides. - - Thin wrapper over :func:`hermes_constants.resolve_reasoning_config` (per-model override > - global ``agent.reasoning_effort``; YAML False = disabled). Empty ``model`` uses ``model.default``. - """ - from hermes_constants import resolve_reasoning_config - cfg = _load_gateway_runtime_config() - return resolve_reasoning_config(cfg, model) - - @staticmethod - def _parse_reasoning_command_args(raw_args: str) -> tuple[str, bool]: - """Parse `/reasoning` args into `(value, persist_global)`. - - Session-scoped by default; `--global` in any position persists the change to config.yaml. - """ - import shlex - - text = str(raw_args or "").strip().replace("—", "--") - if not text: - return "", False - try: - tokens = shlex.split(text) - except ValueError: - tokens = text.split() - - persist_global = False - value_tokens = [] - for token in tokens: - if token == "--global": - persist_global = True - else: - value_tokens.append(token) - return " ".join(value_tokens).strip().lower(), persist_global - - def _resolve_session_reasoning_config( - self, - *, - source: Optional[SessionSource] = None, - session_key: Optional[str] = None, - model: str = "", - ) -> dict | None: - """Resolve reasoning effort for a session, honoring session overrides. - - Priority: session ``/reasoning --session`` > per-model ``agent.reasoning_overrides`` > global - ``agent.reasoning_effort``. ``model`` must be the session's *effective* model (session - ``/model`` override included); empty uses ``model.default``. - """ - resolved_session_key = self._resolve_session_key_or_none(source, session_key) - - if resolved_session_key: - _r_state = self._peek_session_state(resolved_session_key) - if _r_state is not None and _r_state.conversation.reasoning_override is not None: - return _r_state.conversation.reasoning_override - return self._load_reasoning_config(model) - - def _set_session_reasoning_override( - self, - session_key: str, - reasoning_config: Optional[dict], - ) -> None: - """Set or clear the session-scoped reasoning override.""" - if not session_key: - return - # Per-session field write: a lazy ``_session_reasoning_overrides = {}`` init replaced the - # WHOLE dict, racing concurrent sessions; a SessionState field reset cannot cross sessions. - self._session_state(session_key).conversation.reasoning_override = ( - None if reasoning_config is None else dict(reasoning_config) - ) - - def _resolve_session_service_tier( - self, - source=None, - session_key: Optional[str] = None, - ) -> Optional[str]: - """Resolve the effective service tier for a session. - - A session-scoped /fast override beats the config default; the override dict stores - "priority" or None (explicit normal), so key presence — not truthiness — decides. - """ - resolved_session_key = self._resolve_session_key_or_none(source, session_key) - - if resolved_session_key: - _t_state = self._peek_session_state(resolved_session_key) - if ( - _t_state is not None - and _t_state.conversation.service_tier_override - is not _SERVICE_TIER_UNSET - ): - return _t_state.conversation.service_tier_override - return self._load_service_tier() - - def _set_session_service_tier_override( - self, - session_key: str, - service_tier, - clear: bool = False, - ) -> None: - """Set or clear the session-scoped /fast override. - - ``service_tier`` is "priority" or None (explicit normal). Pass - ``clear=True`` to remove the override entirely (fall back to config). - """ - if not session_key: - return - # Presence-sensitive: "priority" or None (explicit normal) both count as an override; the - # sentinel means "no override". Per-session field write: a lazy dict replace races sessions. - self._session_state(session_key).conversation.service_tier_override = ( - _SERVICE_TIER_UNSET if clear else service_tier - ) - - @staticmethod - def _load_service_tier() -> str | None: - """Load Priority Processing (agent.service_tier) from config.yaml: "fast"/"priority"/"on" => - "priority"; "normal"/"off" disable; None when unset/unsupported. - """ - cfg = _load_gateway_runtime_config() - raw = str(cfg_get(cfg, "agent", "service_tier", default="") or "").strip() - - value = raw.lower() - if not value or value in {"normal", "default", "standard", "off", "none"}: - return None - if value in {"fast", "priority", "on"}: - return "priority" - if value in {"auto", "cold"}: - return value - logger.warning("Unknown service_tier '%s', ignoring", raw) - return None - - @staticmethod - def _load_show_reasoning() -> bool: - """Load show_reasoning toggle from config.yaml display section.""" - cfg = _load_gateway_runtime_config() - return is_truthy_value( - cfg_get(cfg, "display", "show_reasoning"), - default=False, - ) - - @staticmethod - def _load_busy_input_mode() -> str: - """Load gateway drain-time busy-input behavior from config/env.""" - mode = os.getenv("HERMES_GATEWAY_BUSY_INPUT_MODE", "").strip().lower() - if not mode: - cfg = _load_gateway_runtime_config() - mode = str(cfg_get(cfg, "display", "busy_input_mode", default="") or "").strip().lower() - if mode == "queue": - return "queue" - if mode == "steer": - return "steer" - return "interrupt" - - @staticmethod - def _load_busy_text_mode() -> str: - """Resolve normal busy TEXT follow-up behavior. - - ``busy_input_mode`` is the source of truth (default ``interrupt``); legacy ``busy_text_mode`` - is honored only when explicitly set so existing queue setups keep working. - """ - # Legacy explicit override wins for backward compat. - legacy = os.getenv("HERMES_GATEWAY_BUSY_TEXT_MODE", "").strip().lower() - if not legacy: - cfg = _load_gateway_runtime_config() - legacy = str(cfg_get(cfg, "display", "busy_text_mode", default="") or "").strip().lower() - if legacy == "interrupt": - return "interrupt" - if legacy == "queue": - return "queue" - # No explicit legacy knob → follow busy_input_mode. - input_mode = GatewayRunner._load_busy_input_mode() - return "queue" if input_mode == "queue" else "interrupt" - - @staticmethod - def _busy_modes_from_config( - config: dict, - *, - fallback_input: str, - fallback_text: str, - ) -> tuple[str, str]: - """Resolve one profile's busy modes without consulting process env.""" - raw_input = str( - cfg_get(config, "display", "busy_input_mode", default="") or "" - ).strip().lower() - input_mode = ( - raw_input - if raw_input in {"interrupt", "queue", "steer"} - else fallback_input - ) - - raw_text = str( - cfg_get(config, "display", "busy_text_mode", default="") or "" - ).strip().lower() - if raw_text in {"interrupt", "queue"}: - text_mode = raw_text - elif raw_input in {"interrupt", "queue", "steer"}: - text_mode = "queue" if input_mode == "queue" else "interrupt" - else: - text_mode = fallback_text - return input_mode, text_mode - - def _snapshot_profile_busy_modes(self, profile_name: str, config: dict) -> None: - """Cache a routed profile's busy policy for this gateway lifetime.""" - input_mode, text_mode = self._busy_modes_from_config( - config, - fallback_input=getattr(self, "_busy_input_mode", "interrupt"), - fallback_text=getattr(self, "_busy_text_mode", "interrupt"), - ) - input_modes = self.__dict__.setdefault("_busy_input_modes_by_profile", {}) - text_modes = self.__dict__.setdefault("_busy_text_modes_by_profile", {}) - input_modes[profile_name] = input_mode - text_modes[profile_name] = text_mode - - def _busy_profile_name_for_source(self, source: SessionSource) -> Optional[str]: - """Return the routed profile whose busy policy applies, if any.""" - if not getattr(getattr(self, "config", None), "multiplex_profiles", False): - return None - name = str(getattr(source, "profile", "") or "").strip() - if not name: - try: - name = str(self._profile_name_for_source(source) or "").strip() - except Exception: - name = "" - return name or None - - def _effective_busy_input_mode(self, source: SessionSource) -> str: - """Resolve busy input mode from the routed profile startup snapshot.""" - fallback = getattr(self, "_busy_input_mode", "interrupt") - profile_name = self._busy_profile_name_for_source(source) - if not profile_name: - return fallback - modes = getattr(self, "_busy_input_modes_by_profile", None) - return modes.get(profile_name, fallback) if isinstance(modes, dict) else fallback - - def _effective_busy_text_mode(self, source: SessionSource) -> str: - """Resolve legacy busy text mode from the routed profile snapshot.""" - fallback = getattr(self, "_busy_text_mode", "interrupt") - profile_name = self._busy_profile_name_for_source(source) - if not profile_name: - return fallback - modes = getattr(self, "_busy_text_modes_by_profile", None) - return modes.get(profile_name, fallback) if isinstance(modes, dict) else fallback - - @staticmethod - def _load_restart_drain_timeout() -> float: - """Load graceful gateway restart/stop drain timeout in seconds.""" - raw = os.getenv("HERMES_RESTART_DRAIN_TIMEOUT", "").strip() - if not raw: - cfg = _load_gateway_runtime_config() - raw = str(cfg_get(cfg, "agent", "restart_drain_timeout", default="") or "").strip() - value = parse_restart_drain_timeout(raw) - if raw and value == DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT: - try: - float(raw) - except (TypeError, ValueError): - logger.warning( - "Invalid restart_drain_timeout '%s', using default %.0fs", - raw, - DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT, - ) - return value - - @staticmethod - def _load_env_or_agent_cfg_timeout(env_var: str, cfg_key: str, parse, default: float) -> float: - """Env var (non-empty) else ``agent.``; warn once when a supplied value fails to parse. - - ``0`` is a valid value; the parser falls back to ``default`` on garbage.""" - env_raw = os.getenv(env_var) - if env_raw is not None and str(env_raw).strip() != "": - raw: object = env_raw - else: - cfg = _load_gateway_runtime_config() - raw = cfg_get(cfg, "agent", cfg_key, default=None) - value = parse(raw) - if raw is not None and str(raw).strip() != "": - try: - float(raw) - except (TypeError, ValueError): - logger.warning("Invalid %s '%s', using default %.0fs", cfg_key, raw, default) - return value - - @classmethod - def _load_restart_after_turn_timeout(cls) -> float: - """Load in-band restart wait-for-idle timeout in seconds.""" - return cls._load_env_or_agent_cfg_timeout( - "HERMES_RESTART_AFTER_TURN_TIMEOUT", "restart_after_turn_timeout", - parse_restart_after_turn_timeout, DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, - ) - - @classmethod - def _load_cron_drain_timeout(cls) -> float: - """Load the cron-only floor under the stop()/drain wait.""" - return cls._load_env_or_agent_cfg_timeout( - "HERMES_CRON_DRAIN_TIMEOUT", "cron_drain_timeout", - parse_cron_drain_timeout, DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, - ) - - @staticmethod - def _load_signal_interrupt_grace_timeout() -> float: - """Load the unexpected-signal post-interrupt grace in seconds.""" - cfg = _load_gateway_runtime_config() - raw = cfg_get( - cfg, - "gateway", - "signal_interrupt_grace_timeout", - default=None, - ) - value = parse_signal_interrupt_grace_timeout(raw) - if raw is not None and raw != "": - try: - float(raw) - except (TypeError, ValueError): - logger.warning( - "Invalid signal_interrupt_grace_timeout '%s', using default %.0fs", - raw, - DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, - ) - return value - - def _post_interrupt_grace_timeout(self) -> float: - """Return the grace before teardown after forcibly interrupting agents.""" - if ( - getattr(self, "_signal_initiated_shutdown", False) - and not getattr(self, "_restart_requested", False) - ): - return max( - 0.0, - float( - getattr( - self, - "_signal_interrupt_grace_timeout", - DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, - ) - ), - ) - return DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT - - @staticmethod - def _load_background_notifications_mode() -> str: - """Load background process notification mode from config or env var.""" - mode = os.getenv("HERMES_BACKGROUND_NOTIFICATIONS", "") - if not mode: - cfg = _load_gateway_runtime_config() - raw = cfg_get(cfg, "display", "background_process_notifications") - if raw is False: - mode = "off" - elif raw not in {None, ""}: - mode = str(raw) - mode = (mode or "concise").strip().lower() - valid = {"concise", "all", "result", "error", "off"} - if mode not in valid: - logger.warning( - "Unknown background_process_notifications '%s', defaulting to 'concise'", - mode, - ) - return "concise" - return mode - - @staticmethod - def _load_provider_routing() -> dict: - """Load OpenRouter provider routing preferences from config.yaml.""" - try: - # Canonical gateway loader (fail-open): managed overlay + ${VAR} - # expansion now apply to provider_routing too. - cfg = _load_gateway_runtime_config() - return cfg.get("provider_routing", {}) or {} - except Exception: - pass - return {} - - @staticmethod - def _load_fallback_model() -> list | None: - """Load fallback provider chain from config.yaml. - - Merges ``fallback_providers`` (kept first) with legacy ``fallback_model`` entries. - """ - try: - # Canonical gateway loader (fail-open): managed overlay + ${VAR} - # expansion now apply to the fallback chain too. - cfg = _load_gateway_runtime_config() - fb = get_fallback_chain(cfg) - if fb: - return fb - except Exception: - pass - return None - - def _refresh_fallback_model(self) -> list | None: - """Re-read fallback_providers from disk for the next agent create/reuse. - - Lets a chain edited after startup reach messaging sessions (cron already re-reads per job). - A TRANSIENT read/parse failure (user mid-edit, non-atomic write) keeps the last known-good - chain; only a successful read that genuinely lacks the key clears it. - """ - try: - from hermes_cli.config import read_user_config_raw - cfg_path = _hermes_home / "config.yaml" - if not cfg_path.exists(): - self._fallback_model = None - return self._fallback_model - # Raw primitive (raises on parse failure) is required here: the canonical fail-open - # loader would return {} on a torn mid-edit write and WIPE the last known-good chain. - # The overlay/expansion below fixes the managed-scope/${VAR} drift without losing that. - cfg = read_user_config_raw(cfg_path) - try: - from hermes_cli import managed_scope - cfg = managed_scope.apply_managed_overlay(cfg) - except Exception: - pass - try: - from hermes_cli.config import _expand_env_vars - expanded = _expand_env_vars(cfg) - if isinstance(expanded, dict): - cfg = expanded - except Exception: - pass - except Exception: - # Transient failure — keep last known-good chain. - logger.debug( - "fallback_providers refresh: config.yaml read failed; " - "keeping last known-good chain", exc_info=True, - ) - return self._fallback_model - self._fallback_model = get_fallback_chain(cfg) or None - return self._fallback_model - - @staticmethod - def _apply_fallback_chain_to_agent(agent: Any, chain: list | None) -> None: - """Keep a cached agent's fallback chain aligned with current config. - - Skips the rewrite while a cooldown holds the agent on an activated fallback provider - (``restore_primary_runtime`` owns that lifecycle); otherwise replaces the chain so - mid-uptime ``fallback_providers`` edits apply without a restart. - """ - if agent is None: - return - new_chain = list(chain or []) - rate_limited_until = getattr(agent, "_rate_limited_until", 0) or 0 - if ( - getattr(agent, "_fallback_activated", False) - and rate_limited_until > time.monotonic() - ): - return - old_chain = list(getattr(agent, "_fallback_chain", []) or []) - agent._fallback_chain = new_chain - agent._fallback_model = new_chain[0] if new_chain else None - if not getattr(agent, "_fallback_activated", False): - agent._fallback_index = 0 - # A config edit means the user changed something — drop the session-scoped unavailability - # memo so re-configured entries (e.g. credentials added mid-uptime) get retried. Only on real - # content change, so the per-message no-op refresh keeps the memo's rate-limiting benefit. - if new_chain != old_chain: - unavailable = getattr(agent, "_unavailable_fallback_keys", None) - if unavailable: - unavailable.clear() def _running_agent_ids(self) -> set: """``id()`` of every agent mid-turn — identity-keyed so the lookup is O(1) and independent of @@ -9417,1620 +4883,36 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if agent is not _AGENT_PENDING_SENTINEL } - def _get_max_concurrent_sessions(self) -> Optional[int]: - """Return the configured active chat session cap, if enabled.""" - try: - from hermes_cli.active_sessions import resolve_max_concurrent_sessions - - return resolve_max_concurrent_sessions(getattr(self, "config", None)) - except Exception: - return None - - def _active_session_limit_message(self, session_key: str) -> Optional[str]: - """Return a user-facing rejection when starting a new session exceeds the cap.""" - max_sessions = self._get_max_concurrent_sessions() - if max_sessions is None: - return None - if self._is_session_running(session_key): - return None - active_count = self._running_agent_count() - if active_count < max_sessions: - return None - from hermes_cli.active_sessions import active_session_limit_message - - return active_session_limit_message(active_count, max_sessions) - - def _claim_active_session_slot( - self, - session_key: str, - source: SessionSource, - ) -> tuple[Any, Optional[str]]: - """Claim a cross-process active-session slot for a new gateway turn.""" - if self._is_session_running(session_key): - return None, None - local_limit_message = self._active_session_limit_message(session_key) - if local_limit_message is not None: - return None, local_limit_message - try: - from hermes_cli.active_sessions import try_acquire_active_session - - platform = source.platform.value if source and source.platform else "gateway" - return try_acquire_active_session( - session_id=session_key, - surface=f"gateway:{platform}", - config=getattr(self, "config", None), - metadata={ - "platform": platform, - "chat_id": getattr(source, "chat_id", "") or "", - "user_id": getattr(source, "user_id", "") or "", - # Writer identity for re-entrancy: if this process leaks a lease for this session - # (exception path skipped release), the next turn re-acquires its own entry rather - # than being fenced out forever — pruning only reclaims entries whose PROCESS died. - "live_session_id": str(session_key), - }, - ) - except Exception as exc: - logger.warning("Failed to claim active session slot: %s", exc) - return None, None - - @staticmethod - def _agent_has_active_subagents(running_agent: Any) -> bool: - """Return True when *running_agent* is driving subagents via ``delegate_task``. - - ``AIAgent.interrupt()`` cascades through ``_active_children`` and aborts in-flight subagent - work, so callers demote ``busy_input_mode='interrupt'`` to ``queue`` while this is True; - explicit ``/stop`` is untouched. Fail-safe: returns False on any attribute/lock error. - """ - if running_agent is None or running_agent is _AGENT_PENDING_SENTINEL: - return False - children = getattr(running_agent, "_active_children", None) - # AIAgent always initialises this as a concrete list. Reject anything that isn't a real - # collection — guards against ``MagicMock()._active_children`` auto-creating a truthy stub - # in tests and triggering the demotion for an agent with no subagents. - if not isinstance(children, (list, tuple, set)): - return False - if not children: - return False - lock = getattr(running_agent, "_active_children_lock", None) - try: - if lock is not None: - with lock: - return bool(children) - return bool(children) - except Exception: - return False - - async def _session_has_compression_in_flight(self, session_key: str) -> bool: - """Return True when a compression lock is held for this session's id. - - Gateway ``interrupt`` busy mode could start a follow-up against the pre-rotation parent while - compression is mid-flight, producing orphaned compression siblings; callers demote interrupt - to queue when True. Both blocking sources (``session_store`` lock + JSON load, SQLite lock - holder SELECT) run in a worker thread so a large state.db never freezes the event loop. - """ - session_store = getattr(self, "session_store", None) - if not session_key or session_store is None: - return False - try: - session_id = await asyncio.to_thread( - self._lookup_session_id_under_store_lock, session_store, session_key - ) - except (AttributeError, TypeError): - return False - except Exception: - logger.warning( - "Compression in-flight check failed while reading session %s; " - "treating compression as active to avoid interrupting a possible " - "parent-session rotation", - session_key, - exc_info=True, - ) - return True - if not session_id: - return False - session_db = getattr(self, "_session_db", None) - if session_db is None: - return False - raw_db = getattr(session_db, "_db", session_db) - try: - holder = await asyncio.to_thread( - raw_db.get_compression_lock_holder, str(session_id) - ) - # Production returns Optional[str]. Reject non-strings so a MagicMock auto-attr (or any - # unexpected truthy) cannot look like a held lock and skip hygiene. - return isinstance(holder, str) and bool(holder) - except (AttributeError, TypeError): - return False - except Exception: - logger.warning( - "Compression in-flight check failed while reading lock holder " - "for session %s; treating compression as active to avoid " - "interrupting a possible parent-session rotation", - session_id, - exc_info=True, - ) - return True - - @staticmethod - def _lookup_session_id_under_store_lock(session_store, session_key: str): - """Sync helper run in the thread pool: read session_id under the store lock.""" - # noqa: SLF001 — intentional private access; runs off the event loop. - with session_store._lock: # noqa: SLF001 - session_store._ensure_loaded_locked() # noqa: SLF001 - entry = session_store._entries.get(session_key) # noqa: SLF001 - return getattr(entry, "session_id", None) if entry is not None else None # Hard cap on per-session pending follow-ups for busy_input_mode=queue (and the draining/steer- # fallback/subagent-demotion paths that share this entry point). Without a cap, a stuck agent + # a rapid-fire user could grow the overflow list unboundedly. _BUSY_QUEUE_MAX_PENDING = 32 - def _queue_or_replace_pending_event(self, session_key: str, event: MessageEvent) -> None: - adapter = self._adapter_for_source(event.source) - if not adapter: - return - # Route through the ``/queue`` FIFO infrastructure so each follow-up gets its own turn in - # arrival order (merge_text=False silently OVERWROTE the single pending slot). Photo bursts - # still merge into the head slot (album semantics); everything else appends to the tail. - pending_slot = getattr(adapter, "_pending_messages", None) - existing = pending_slot.get(session_key) if isinstance(pending_slot, dict) else None - security_metadata_keys = ( - "hermes_plugin_id", - "hermes_plugin_injection", - "gateway_session_key", - "gateway_session_id", - "gateway_session_strict", - ) - same_security_context = existing is not None and ( - getattr(existing, "internal", False) == getattr(event, "internal", False) - and getattr(existing, "allow_gateway_control", True) - == getattr(event, "allow_gateway_control", True) - and all( - (getattr(existing, "metadata", None) or {}).get(key) - == (getattr(event, "metadata", None) or {}).get(key) - for key in security_metadata_keys - ) - ) - if same_security_context and ( - getattr(existing, "message_type", None) == MessageType.PHOTO - or event.message_type == MessageType.PHOTO - or bool(getattr(existing, "media_urls", None)) - or bool(getattr(event, "media_urls", None)) - ): - # Preserve photo-burst / media-merge semantics for the head slot. - merge_pending_message_event( - adapter._pending_messages, - session_key, - event, - merge_text=event.message_type == MessageType.TEXT, - ) - return - if self._queue_depth(session_key, adapter=adapter) >= self._BUSY_QUEUE_MAX_PENDING: - logger.warning( - "Dropping busy-mode follow-up for session %s — pending queue at cap (%d).", - session_key, - self._BUSY_QUEUE_MAX_PENDING, - ) - return + @dataclasses.dataclass + class _BusySteerOutcome: + effective_mode: str + demoted_for_subagents: bool + demoted_for_compression: bool + steered: bool + redirected: bool - self._enqueue_fifo(session_key, event, adapter) - - async def _prepare_busy_steer_text(self, event: MessageEvent) -> str: - """Return steerable text for a busy follow-up, transcribing voice first. - - Successful steer messages bypass the inbound STT queue, so without this a media-only voice - follow-up has empty text and steer silently degrades to queue mode. Only voice-message media - (not audio file attachments) is transcribed; on failure keep any caption and let the steer - fallback handle it. Goes through ``_transcribe_and_echo_pending_voice`` — the single - out-of-band STT choke point — so STT runs at most once per message (cached on the event). - """ - text = (event.text or "").strip() - if not self._pending_event_audio_paths(event): - return text - - adapter = self._adapter_for_source(event.source) - enriched_text, successful_transcripts = await self._transcribe_and_echo_pending_voice( - event, - adapter, - event.source, - text, - log_context="Busy-steer", - ) - if not successful_transcripts: - return text - return (enriched_text or text).strip() - - async def _handle_active_session_busy_message(self, event: MessageEvent, session_key: str) -> bool: - # Authorization gate: the cold path (_handle_message) checks _is_user_authorized before - # creating a session; the busy path must enforce the same check, else unauthorized users in - # shared threads (Slack/Telegram/Discord) inject messages into a session they don't own. - if not self._is_user_authorized(event.source): - logger.warning( - "Dropping message from unauthorized user in active session: " - "user=%s (%s), platform=%s, session=%s", - event.source.user_id, - event.source.user_name, - event.source.platform.value if event.source.platform else "unknown", - session_key, - ) - return True # handled (silently dropped); do not fall through - - effective_mode = self._effective_busy_input_mode(event.source) - - # --- Draining case (gateway restarting/stopping) --- - if self._draining: - adapter = self._adapter_for_source(event.source) - if not adapter: - return True - - reply_anchor = self._reply_anchor_for_event(event) - thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) - if self._queue_during_drain_enabled(effective_mode): - self._queue_or_replace_pending_event(session_key, event) - message = f"⏳ Gateway {self._status_action_gerund()} — queued for the next turn after it comes back." - else: - message = f"⏳ Gateway is {self._status_action_gerund()} and is not accepting another turn right now." - - await adapter._send_with_retry( - chat_id=event.source.chat_id, - content=message, - reply_to=( - reply_anchor - if event.source.platform == Platform.TELEGRAM - and event.source.chat_type == "dm" - and event.source.thread_id - else (None if event.source.platform == Platform.TELEGRAM and event.source.thread_id else event.message_id) - ), - metadata=thread_meta, - ) - return True - - # Approval routing: while blocked on a dangerous-command approval, a bare "yes" must reach the - # approval handler, not be steered/queued/interrupted (else it queues behind a turn that can't - # start until the approval resolves -> auto-deny deadlock). Slash forms already bypass at the - # base-adapter guard. Gated on has_blocking_approval so a conversational "yes" never fires a - # command. Reuse the /approve and /deny handlers; the busy path does not auto-send their return. - try: - from tools.approval import has_blocking_approval - if event.allow_gateway_control and has_blocking_approval(session_key): - _raw_text = (event.text or "").strip().lower() - _approve_words = {"approve", "yes", "ok", "okay", "confirm", "y", "👍"} - _deny_words = {"deny", "no", "reject", "cancel", "n", "👎"} - _approval_handler = None - _normalized_args = "" - if _raw_text in _approve_words: - _approval_handler = self._handle_approve_command - elif _raw_text in _deny_words: - _approval_handler = self._handle_deny_command - elif _raw_text in {"always", "approve always", "always approve"}: - _approval_handler = self._handle_approve_command - _normalized_args = "always" - elif _raw_text in {"session", "approve session", "session approve"}: - _approval_handler = self._handle_approve_command - _normalized_args = "session" - if _approval_handler is not None: - # Synthesize "/approve [args]" / "/deny" so the slash handlers parse modifiers via - # event.get_command_args(). Always a literal "/": is_command()/get_command_args() - # don't recognize per-platform display prefixes ("!" on Slack/Matrix). - _verb = "approve" if _approval_handler is self._handle_approve_command else "deny" - _synth = f"/{_verb}" - if _normalized_args: - _synth = f"{_synth} {_normalized_args}" - event.text = _synth - _reply = await _approval_handler(event) - logger.info( - "Approval response via plain text: session=%s verb=%s args=%r", - session_key, _verb, _normalized_args, - ) - _adapter = self._adapter_for_source(event.source) - if _adapter and _reply: - _text, _eph_ttl = _adapter._unwrap_ephemeral(_reply) - if _text: - _anchor = self._reply_anchor_for_event(event) - await _adapter._send_with_retry( - chat_id=event.source.chat_id, - content=_text, - reply_to=_anchor, - metadata=self._thread_metadata_for_source(event.source, _anchor), - ) - return True - except Exception: - logger.warning( - "Plain-text approval routing failed for session %s; " - "falling through to busy handling", - session_key, exc_info=True, - ) - - # Normal busy case (agent actively running a task) - adapter = self._adapter_for_source(event.source) - if not adapter: - return False # let default path handle it - - # Internal synthetic events (async-delegation / background-process completions) must never - # interrupt/steer: treated as user TEXT while busy, interrupt mode would abort the active turn; - # a completion surfaces as a NEW turn only when idle. Plugin events carry untrusted payload - # text, so queue them through the gateway FIFO (security metadata kept apart). - if getattr(event, "internal", False) and not event.allow_gateway_control: - self._queue_or_replace_pending_event(session_key, event) - return True - if getattr(event, "internal", False): - return False - - _busy_state = self._peek_session_state(session_key) - running_agent = _busy_state.turn.agent if _busy_state else None - - busy_text_mode = self._effective_busy_text_mode(event.source) - if ( - event.message_type == MessageType.TEXT - and busy_text_mode == "queue" - and effective_mode != "steer" - ): - return False - - # Steer mode injects mid-run via running_agent.steer(); fall back to queue (nothing lost) if the - # agent isn't running yet (sentinel), lacks steer(), or the payload is empty. interrupt() - # cascades to ``_active_children`` and aborts delegate_task work, so demote ``interrupt`` to - # ``queue`` while the parent drives subagents; explicit /stop and /new still force-cancel all. - demoted_for_subagents = ( - effective_mode == "interrupt" - and self._agent_has_active_subagents(running_agent) - ) - if demoted_for_subagents: - logger.info( - "Demoting busy_input_mode 'interrupt' to 'queue' for session %s " - "because the running agent has active subagents (#30170)", - session_key, - ) - effective_mode = "queue" - demoted_for_compression = ( - effective_mode == "interrupt" - and await self._session_has_compression_in_flight(session_key) - ) - if demoted_for_compression: - logger.info( - "Demoting busy_input_mode 'interrupt' to 'queue' for session %s " - "because context compression is in flight (#56391)", - session_key, - ) - effective_mode = "queue" - steered = False - redirected = False - if effective_mode == "steer": - steer_text = await self._prepare_busy_steer_text(event) - # Steerable: plain text, OR every attachment is STT-eligible voice media whose transcript - # was folded into steer_text — else a voice note in steer mode silently degrades to queue. - _steer_media_urls = getattr(event, "media_urls", None) or [] - _steer_all_voice = bool(_steer_media_urls) and ( - len(self._pending_event_audio_paths(event)) == len(_steer_media_urls) - ) - can_steer = ( - steer_text - and ( - ( - event.message_type == MessageType.TEXT - and not event.media_urls - and not event.media_types - ) - or _steer_all_voice - ) - and running_agent is not None - and running_agent is not _AGENT_PENDING_SENTINEL - and hasattr(running_agent, "steer") - ) - if can_steer: - try: - steered = bool(running_agent.steer(steer_text)) - except Exception as exc: - logger.warning("Gateway steer failed for session %s: %s", session_key, exc) - steered = False - if not steered: - # Fall back to queue (merge into pending messages, no interrupt) - effective_mode = "queue" - elif ( - effective_mode == "interrupt" - and event.message_type == MessageType.TEXT - and not event.media_urls - and not event.media_types - and running_agent is not None - and running_agent is not _AGENT_PENDING_SENTINEL - and getattr(running_agent, "_supports_active_turn_redirect", False) is True - and hasattr(running_agent, "redirect") - ): - try: - redirected = bool(running_agent.redirect((event.text or "").strip())) - except Exception as exc: - logger.warning("Gateway redirect failed for session %s: %s", session_key, exc) - redirected = False - - # Queue as the next turn after the current run ends. Skip after a successful steer — the text - # is already in the run and must NOT replay. Use the FIFO helper, not raw - # merge_pending_message_event (merge_text=True newline-joins consecutive TEXT follow-ups into - # ONE turn); FIFO gives each text its own turn while keeping photo-burst / album merge for media. - if not steered and not redirected: - self._queue_or_replace_pending_event(session_key, event) - - is_queue_mode = effective_mode == "queue" - is_steer_mode = effective_mode == "steer" - is_redirect_mode = effective_mode == "interrupt" and redirected - - # Interrupt mode: abort in-flight tool calls; the agent loop exits at its next check point. - if ( - effective_mode == "interrupt" - and not redirected - and running_agent - and running_agent is not _AGENT_PENDING_SENTINEL - ): - try: - _interrupt_text = event.text - _media_urls = getattr(event, "media_urls", None) or [] - if self._pending_event_audio_paths(event): - _interrupt_text, _ = await self._transcribe_and_echo_pending_voice( - event, - adapter, - event.source, - event.text or "", - log_context="Voice-busy-interrupt", - ) - elif not _interrupt_text and _media_urls: - _interrupt_text = _build_media_placeholder(event) - running_agent.interrupt(_interrupt_text) - except Exception: - pass # don't let interrupt failure block the ack - - # Disabled ack: skip sending, still process input. Checked before debounce so we never stamp a - # "last ack" timestamp for an ack that was not delivered. - busy_ack_enabled = os.environ.get("HERMES_GATEWAY_BUSY_ACK_ENABLED", "true").lower() == "true" - if not busy_ack_enabled: - logger.debug("Busy ack suppressed for session %s", session_key) - return True # input still processed, just no ack sent - - # Debounce before the config-heavy display lookup: rapid follow-ups are still processed but - # shouldn't cost a config read just to learn no ack will be sent. - _BUSY_ACK_COOLDOWN = 30 - now = time.time() - last_ack = _busy_state.turn.busy_ack_ts if _busy_state else 0 - if now - last_ack < _BUSY_ACK_COOLDOWN: - return True # interrupt sent (if not queue), ack already delivered recently - - from gateway.display_config import resolve_display_setting - platform_key = _platform_config_key(event.source.platform) - - # Steer mode already injected the text; some mobile chat setups want silent steering (like STT - # echo suppression) — keep the behavior, drop only the confirmation bubble. - if is_steer_mode: - steer_ack_env = os.environ.get("HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED") - if steer_ack_env is not None: - steer_ack_enabled = steer_ack_env.strip().lower() in {"1", "true", "yes", "on"} - else: - steer_ack_enabled = bool( - resolve_display_setting( - _load_gateway_config(), - platform_key, - "busy_steer_ack_enabled", - True, - ) - ) - if not steer_ack_enabled: - logger.debug("Busy steer ack suppressed for session %s", session_key) - return True - - self._session_state(session_key).turn.busy_ack_ts = now - - # Mobile chat defaults keep the ack terse; iteration/tool detail stays in logs and can be opted - # in per platform via display.platforms..busy_ack_detail. - status_parts = [] - busy_ack_detail_enabled = bool( - resolve_display_setting( - _load_gateway_config(), - _platform_config_key(event.source.platform), - "busy_ack_detail", - True, - ) - ) - - if busy_ack_detail_enabled and running_agent and running_agent is not _AGENT_PENDING_SENTINEL: - try: - summary = running_agent.get_activity_summary() - iteration = summary.get("api_call_count", 0) - max_iter = summary.get("max_iterations", 0) - current_tool = summary.get("current_tool") - start_ts = _busy_state.turn.started_ts if _busy_state else 0 - if start_ts: - elapsed_min = int((now - start_ts) / 60) - if elapsed_min > 0: - status_parts.append(f"{elapsed_min} min elapsed") - if max_iter: - status_parts.append(f"iteration {iteration}/{max_iter}") - if current_tool: - status_parts.append(f"running: {current_tool}") - except Exception: - pass - - status_detail = f" ({', '.join(status_parts)})" if status_parts else "" - if is_steer_mode: - message = ( - f"⏩ Steered into current run{status_detail}. " - f"Your message arrives after the next tool call." - ) - elif is_redirect_mode: - message = ( - f"↪ Redirected current run{status_detail}. " - f"I'll adjust using your correction." - ) - elif is_queue_mode and demoted_for_subagents: - # Explain the demotion: the follow-up didn't kill the subagent; /stop is the escape hatch. - message = ( - f"⏳ Subagent working{status_detail} — your message is queued for " - f"when it finishes (use /stop to cancel everything)." - ) - elif is_queue_mode and demoted_for_compression: - message = ( - f"⏳ Compressing context{status_detail} — your message is queued for " - f"when it finishes (use /stop to cancel everything)." - ) - elif is_queue_mode: - message = ( - f"⏳ Queued for the next turn{status_detail}. " - f"I'll respond once the current task finishes." - ) - else: - message = ( - f"⚡ Interrupting current task{status_detail}. " - f"I'll respond to your message shortly." - ) - - # First-touch onboarding: one-time hint about the queue/interrupt knob; the flag is persisted to - # config.yaml so it never fires again on this install. - try: - from agent.onboarding import ( - BUSY_INPUT_FLAG, - busy_input_hint_gateway, - is_seen, - mark_seen, - ) - _user_cfg = _load_gateway_config() - if not is_seen(_user_cfg, BUSY_INPUT_FLAG): - if is_steer_mode: - _hint_mode = "steer" - elif is_queue_mode: - _hint_mode = "queue" - elif is_redirect_mode: - _hint_mode = "redirect" - else: - _hint_mode = "interrupt" - message = ( - f"{message}\n\n" - f"{busy_input_hint_gateway(_hint_mode)}" - ) - mark_seen(_hermes_home / "config.yaml", BUSY_INPUT_FLAG) - except Exception as _onb_err: - logger.debug("Failed to apply busy-input onboarding hint: %s", _onb_err) - - reply_anchor = self._reply_anchor_for_event(event) - thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) - try: - await adapter._send_with_retry( - chat_id=event.source.chat_id, - content=message, - reply_to=( - reply_anchor - if event.source.platform == Platform.TELEGRAM - and event.source.chat_type == "dm" - and event.source.thread_id - else (None if event.source.platform == Platform.TELEGRAM and event.source.thread_id else event.message_id) - ), - metadata=thread_meta, - ) - except Exception as e: - logger.debug("Failed to send busy-ack: %s", e) - - return True - - async def _drain_active_agents( - self, timeout: float, cron_timeout: Optional[float] = None - ) -> tuple[Dict[str, Any], bool]: - snapshot = self._snapshot_running_agents() - last_active_count = self._running_agent_count() - last_cron_count = self._active_cron_job_count() - last_api_count = self._active_api_run_count() - last_deferred_count = self._active_deferred_agent_worker_count() - last_status_at = 0.0 - - def _maybe_update_status(force: bool = False) -> None: - nonlocal last_active_count, last_cron_count, last_api_count - nonlocal last_deferred_count, last_status_at - now = asyncio.get_running_loop().time() - active_count = self._running_agent_count() - cron_count = self._active_cron_job_count() - api_count = self._active_api_run_count() - deferred_count = self._active_deferred_agent_worker_count() - if ( - force - or active_count != last_active_count - or cron_count != last_cron_count - or api_count != last_api_count - or deferred_count != last_deferred_count - or (now - last_status_at) >= 1.0 - ): - self._update_runtime_status("draining") - last_active_count = active_count - last_cron_count = cron_count - last_api_count = api_count - last_deferred_count = deferred_count - last_status_at = now - - # Cron jobs run on the scheduler's pool, outside ``self._running_agents`` — fold their in-flight - # count into this wait, or a cron job's tool work is killed without warning once it's the only - # active thing running. API-server/desk sessions and detached deferred workers share the gap. - if ( - not self._running_agents - and last_cron_count == 0 - and last_api_count == 0 - and last_deferred_count == 0 - ): - _maybe_update_status(force=True) - return snapshot, False - - _maybe_update_status(force=True) - - # Cron drains on its own deadline: ``timeout`` (``restart_drain_timeout``) defaults to 0 since - # an interrupted chat turn is announced and resumable, while a cron run killed mid-flight is a - # permanent failure nobody is waiting on. One shared budget would kill cron after 0.00s. - loop = asyncio.get_running_loop() - started = loop.time() - deadline = started + timeout - cron_deadline = started + (timeout if cron_timeout is None else cron_timeout) - - def _still_draining() -> bool: - now = loop.time() - if ( - len(self._running_agents) - or self._active_api_run_count() - or self._active_deferred_agent_worker_count() - ) and now < deadline: - return True - return bool(self._active_cron_job_count()) and now < cron_deadline - - # Both budgets at 0 leave this loop unentered ("interrupt immediately") as an expired deadline, - # not a special case, so timed_out below is always computed from real state. - while _still_draining(): - _maybe_update_status() - await asyncio.sleep(0.1) - timed_out = ( - bool(len(self._running_agents)) - or bool(self._active_cron_job_count()) - or bool(self._active_api_run_count()) - or bool(self._active_deferred_agent_worker_count()) - ) - _maybe_update_status(force=True) - return snapshot, timed_out - - def _interrupt_running_agents(self, reason: str) -> None: - for session_key, agent in list(self._running_agents.items()): - if agent is _AGENT_PENDING_SENTINEL: - continue - try: - request_hard_interrupt(agent, reason) - logger.debug("Interrupted running agent for session %s during shutdown", session_key) - except Exception as e: - logger.debug("Failed interrupting agent during shutdown: %s", e) - # API-server / desk turns are adapter-owned and never enter _running_agents, so the loop above - # cannot see them even though _drain_active_agents() waited for them. - interrupted_api = self._interrupt_api_server_runs(reason) - if interrupted_api: - logger.debug("Interrupted %d api_server run(s) during shutdown", interrupted_api) - interrupted_deferred = self._interrupt_deferred_agent_workers(reason) - if interrupted_deferred: - logger.debug( - "Interrupted %d deferred agent worker(s) during shutdown", - interrupted_deferred, - ) - - async def _notify_interrupted_cron_jobs(self, job_ids) -> int: - """Tell the owner of each just-interrupted cron job that its run died. - - The cron worker can't: its thread reaches ``_deliver_result`` after teardown closed the - transport. Must run post-interrupt while adapters are still connected (the window - ``_notify_active_sessions_of_shutdown`` uses, which is blind to cron work). Best-effort: every - failure is swallowed so a wedged adapter can't extend shutdown. Returns notices sent. - """ - if not job_ids: - return 0 - try: - from cron.jobs import get_job - from cron.scheduler import _resolve_delivery_targets - except Exception as e: - logger.debug("Cron interrupt notification unavailable: %s", e) - return 0 - - action = "restarting" if self._restart_requested else "shutting down" - notified: set = set() - for job_id in job_ids: - try: - job = get_job(job_id) - if not job: - continue - # deliver=local jobs, and deliver=origin jobs with no resolvable origin, resolve to zero - # targets and must stay silent rather than fall back to a home channel. Interrupted - # notices are failure-category engine status, so they honor failure_deliver. - targets = _resolve_delivery_targets(job, for_failure=True) - except Exception as e: - logger.debug("Cron interrupt targets unresolved for %s: %s", job_id, e) - continue - if not targets: - continue - - msg = ( - f"⚠️ Cron job '{job.get('name') or job_id}' was interrupted — " - f"the gateway is {action} and killed the run before it " - "finished. No result was produced for this run." - ) - for target in targets: - try: - platform = Platform(str(target.get("platform", "")).lower()) - except Exception: - continue - adapter = self.adapters.get(platform) - if adapter is None: - continue - platform_cfg = self.config.platforms.get(platform) - if platform_cfg is not None and not platform_cfg.gateway_restart_notification: - continue - - chat_id = str(target.get("chat_id")) - thread_id = target.get("thread_id") - dedup_key = ( - job_id, - platform.value, - chat_id, - str(thread_id) if thread_id else None, - ) - if dedup_key in notified: - continue - try: - metadata = self._thread_metadata_for_target( - platform, chat_id, thread_id, adapter=adapter - ) - result = await adapter.send(chat_id, msg, metadata=metadata) - if result is not None and getattr(result, "success", True) is False: - logger.debug( - "Cron interrupt notice to %s:%s failed: %s", - platform.value, chat_id, - getattr(result, "error", "send returned success=False"), - ) - continue - notified.add(dedup_key) - except Exception as e: - logger.debug( - "Cron interrupt notice to %s:%s raised: %s", - platform.value, chat_id, e, - ) - if notified: - logger.info( - "Shutdown: delivered %d interrupted-cron-job notice(s)", - len(notified), - ) - return len(notified) - - async def _notify_active_sessions_of_shutdown(self) -> None: - """Send shutdown/restart notifications to active chats and home channels. - - Called at the start of stop() while adapters are connected; send failures never block shutdown. - """ - active = self._snapshot_running_agents() - restart_source = self._restart_command_source if self._restart_requested else None - - action = "restarting" if self._restart_requested else "shutting down" - hint = ( - "Your current task will be interrupted. " - "Send any message after restart and I'll try to resume where you left off." - if self._restart_requested - else "Your current task will be interrupted." - ) - msg = f"⚠️ Gateway {action} — {hint}" - - notified: set[tuple[str, str, Optional[str]]] = set() - for session_key in active: - source = None - try: - if getattr(self, "session_store", None) is not None: - await self.async_session_store._ensure_loaded() - entry = self.session_store._entries.get(session_key) - source = getattr(entry, "origin", None) if entry else None - except Exception as e: - logger.debug( - "Failed to load session origin for shutdown notification %s: %s", - session_key, - e, - ) - - if source is None: - source = self._get_cached_session_source(session_key) - - if source is not None: - platform_str = source.platform.value - chat_id = str(source.chat_id) - thread_id = source.thread_id - else: - # Fall back to parsing the session key when no persisted - # origin is available (legacy sessions/tests). - _parsed = _parse_session_key(session_key) - if not _parsed: - continue - platform_str = _parsed["platform"] - chat_id = _parsed["chat_id"] - thread_id = _parsed.get("thread_id") - - # Dedupe only identical targets: thread/topic platforms share a parent chat yet route to - # distinct destinations via metadata. - dedup_key = (platform_str, chat_id, str(thread_id) if thread_id else None) - if dedup_key in notified: - continue - - try: - platform = Platform(platform_str) - adapter = self.adapters.get(platform) - if not adapter: - continue - - platform_cfg = self.config.platforms.get(platform) - if platform_cfg is not None and not platform_cfg.gateway_restart_notification: - logger.info( - "Shutdown notification suppressed for active session: %s has gateway_restart_notification=false", - platform_str, - ) - continue - - reply_to_message_id = getattr(source, "message_id", None) if source is not None else None - if reply_to_message_id is None and restart_source is not None: - try: - restart_platform = restart_source.platform.value - restart_chat_id = str(restart_source.chat_id) - restart_thread_id = str(restart_source.thread_id) if restart_source.thread_id else None - if (restart_platform, restart_chat_id, restart_thread_id) == dedup_key: - reply_to_message_id = getattr(restart_source, "message_id", None) - except Exception: - pass - - metadata = self._thread_metadata_for_target( - platform, - chat_id, - thread_id, - chat_type=getattr(source, "chat_type", None) if source is not None else None, - reply_to_message_id=reply_to_message_id, - adapter=adapter, - ) - - result = await adapter.send(chat_id, msg, metadata=metadata) - if result is not None and getattr(result, "success", True) is False: - logger.debug( - "Failed to send shutdown notification to %s:%s: %s", - platform_str, - chat_id, - getattr(result, "error", "send returned success=False"), - ) - continue - - notified.add(dedup_key) - logger.info( - "Sent shutdown notification to active chat %s:%s", - platform_str, chat_id, - ) - except Exception as e: - logger.debug( - "Failed to send shutdown notification to %s:%s: %s", - platform_str, chat_id, e, - ) - - if self._restart_requested and restart_source is not None: - logger.debug("Skipping home-channel shutdown notifications for in-chat restart") - return - - # Suppress ONLY the home-channel broadcast when the drain asked to be quiet (e.g. routine - # auto-update on an always-on fleet). Per-session interrupt pings above are NOT gated: empty by - # construction on a drained shutdown, and useful ("task cut off, message me to resume") on a - # force-interrupt. Honoured only for a CURRENT-epoch marker (staleness check inside - # drain_notification_suppressed), so an orphaned marker can't silence a fresh gateway. - try: - from gateway.drain_control import drain_notification_suppressed - if drain_notification_suppressed(): - logger.info( - "Home-channel shutdown broadcast suppressed by drain marker " - "(suppress_notification=true)" - ) - return - except Exception as e: - # Never let the suppression check block the shutdown broadcast — - # fail toward the louder, more-visible behaviour. - logger.debug("drain_notification_suppressed check failed: %s", e) - - # Snapshot adapters: adapter.send() can hit a fatal path (_handle_fatal) that pops the adapter - # from self.adapters -> ``RuntimeError: dictionary changed size during iteration``. - for platform, adapter in list(self.adapters.items()): - home = self.config.get_home_channel(platform) - if not home or not home.chat_id: - continue - - platform_cfg = self.config.platforms.get(platform) - if platform_cfg is not None and not platform_cfg.gateway_restart_notification: - logger.info( - "Shutdown notification suppressed for home channel: %s has gateway_restart_notification=false", - platform.value, - ) - continue - - dedup_key = (platform.value, str(home.chat_id), str(home.thread_id) if home.thread_id else None) - if dedup_key in notified: - continue - - try: - metadata = self._thread_metadata_for_target( - platform, - home.chat_id, - home.thread_id, - adapter=adapter, - ) - if metadata: - result = await adapter.send(str(home.chat_id), msg, metadata=metadata) - else: - result = await adapter.send(str(home.chat_id), msg) - if result is not None and getattr(result, "success", True) is False: - logger.debug( - "Failed to send shutdown notification to home channel %s:%s: %s", - platform.value, - home.chat_id, - getattr(result, "error", "send returned success=False"), - ) - continue - - notified.add(dedup_key) - logger.info( - "Sent shutdown notification to home channel %s:%s", - platform.value, - home.chat_id, - ) - except Exception as e: - logger.debug( - "Failed to send shutdown notification to home channel %s:%s: %s", - platform.value, - home.chat_id, - e, - ) - - async def _finalize_shutdown_agents(self, active_agents: Dict[str, Any]) -> None: - for agent in active_agents.values(): - # Persist in-flight transcripts before teardown: a force-interrupted agent may never reach - # finalize_turn (the only mid-turn flush), so its tool rounds would vanish from - # load_transcript() on resume (resume already tolerates a pending-tool-result tail). The - # flush is idempotent (identity-tracked); gracefully finished agents re-flush nothing. - try: - _flush = getattr(agent, "_flush_messages_to_session_db", None) - _session_messages = getattr(agent, "_session_messages", None) - if callable(_flush) and isinstance(_session_messages, list) and _session_messages: - # Strip empty-response retry scaffolding from the tail first (as ``_persist_session`` - # does) so a resumed turn doesn't replay synthetic recovery nudges. - _strip = getattr( - agent, "_drop_trailing_empty_response_scaffolding", None - ) - if callable(_strip): - with suppress(Exception): - _strip(_session_messages) - try: - _flush(_session_messages) - except Exception as _flush_err: - # Transcript could not be persisted (e.g. FTS/SQLite index corruption). A log - # line alone loses the conversation at exit, so dump the live history to an - # external JSON recovery snapshot. Non-fatal: shutdown never blocks on a backup. - logger.warning( - "Shutdown transcript flush failed (%s); preserving " - "%d in-memory message(s) to recovery snapshot", - _flush_err, - len(_session_messages), - ) - from gateway.shutdown_flush import flush_agent_history_to_file - flush_agent_history_to_file( - getattr(agent, "session_id", None), - _session_messages, - ) - except Exception as _e: - logger.debug("Shutdown transcript flush failed: %s", _e) - # Off-loop + bounded: plugin on_session_finalize hooks can do arbitrary synchronous work - # (e.g. a full-session trace export) — same hang class as the memory provider below. - await self._finalize_session_off_loop( - session_id=getattr(agent, "session_id", None), - platform="gateway", - reason="shutdown", - ) - # Off-loop + bounded: a wedged memory provider here used to hang - # the whole shutdown so SIGTERM never completed (#53175). - await self._cleanup_agent_resources_off_loop( - agent, context="shutdown finalize" - ) - - def _should_emit_long_running_notification( - self, - session_key: Optional[str], - agent: Any, - executor_task: Optional[Any], - ) -> bool: - """Only emit the heartbeat while this task still owns the live run. - - Stop once the executor finishes, the agent is gone, or the session key was rebound (e.g. - ``/new`` mid-run) — else a stale ``running: delegate_task`` heartbeat outlives its run. - """ - if agent is None: - return False - if executor_task is not None and executor_task.done(): - return False - if session_key: - _hb_state = self._peek_session_state(session_key) - if (_hb_state.turn.agent if _hb_state else None) is not agent: - return False - return True # Bound for off-loop agent-resource cleanup from event-loop coroutines (expiry sweep, cache-hygiene # re-eviction). _cleanup_agent_resources is synchronous and can block long (subprocess teardown, # memory-provider network/SQLite IO); inline it wedges the loop, so it runs in a worker thread. _CLEANUP_TIMEOUT_S = 30.0 - def _defer_agent_cleanup_until_future_done( - self, - future: asyncio.Future, - agent: Any, - *, - context: str, - ) -> None: - """Clean up ``agent`` only after its executor future has finished. - - A timed-out executor call keeps running in its worker thread; closing the agent first can - tear down clients it still uses, so hold a strong task ref and await the real future. - """ - - async def _cleanup_when_done() -> None: - try: - await asyncio.shield(future) - except asyncio.CancelledError: - # Loop shutdown can cancel this waiter while the executor still - # runs. Never turn that cancellation into premature cleanup. - return - except Exception as exc: - logger.debug( - "Deferred agent worker%s finished with an error: %s", - f" ({context})" if context else "", - exc, - ) - await self._cleanup_agent_resources_off_loop(agent, context=context) - - self._track_deferred_agent_worker(future, agent) - - task = asyncio.create_task(_cleanup_when_done()) - tasks = getattr(self, "_deferred_agent_cleanup_tasks", None) - if tasks is None: - tasks = set() - self._deferred_agent_cleanup_tasks = tasks - tasks.add(task) - task.add_done_callback(tasks.discard) # Budget for one finalize_session() dispatch (plugin on_session_finalize hooks + Relay close): # enough for a normal trace-export flush, small enough a wedged plugin can't eat the stop window. _FINALIZE_TIMEOUT_S = 10.0 - async def _finalize_session_off_loop( - self, - *, - session_id: Any, - platform: str, - reason: str, - **extra: Any, - ) -> None: - """Run hermes_cli.lifecycle.finalize_session off the event loop, bounded. - - On timeout the worker thread is left to finish (or leak) and the caller proceeds. - """ - - def _call() -> None: - from hermes_cli.lifecycle import finalize_session - - finalize_session( - session_id=session_id, - platform=platform, - reason=reason, - **extra, - ) - - try: - await asyncio.wait_for( - self._run_in_executor_with_context(_call), - timeout=self._FINALIZE_TIMEOUT_S, - ) - except asyncio.TimeoutError: - logger.warning( - "Session finalize hooks (%s, reason=%s) exceeded %ss; " - "proceeding without blocking the event loop (the worker " - "thread is left to finish on its own).", - session_id, - reason, - self._FINALIZE_TIMEOUT_S, - ) - except Exception as finalize_exc: - logger.debug( - "Session finalize hooks (%s, reason=%s) failed: %s", - session_id, - reason, - finalize_exc, - ) - - async def _cleanup_agent_resources_off_loop( - self, agent: Any, *, context: str = "" - ) -> None: - """Run _cleanup_agent_resources in a worker thread with a bounded wait. - - On timeout the worker thread is left to finish (or leak) and the caller proceeds, as /new does. - """ - if agent is None: - return - if context.startswith("shutdown") or context == "session expiry": - with suppress(Exception): - agent._end_session_on_close = False - try: - await asyncio.wait_for( - self._run_in_executor_with_context( - self._cleanup_agent_resources, agent - ), - timeout=self._CLEANUP_TIMEOUT_S, - ) - except asyncio.TimeoutError: - logger.warning( - "Agent resource cleanup%s exceeded %ss; proceeding without " - "blocking the event loop (the worker thread is left to finish " - "on its own). (#53175)", - f" ({context})" if context else "", - self._CLEANUP_TIMEOUT_S, - ) - except Exception as cleanup_exc: - logger.warning( - "Agent resource cleanup%s failed: %s (#53175)", - f" ({context})" if context else "", - cleanup_exc, - ) - - def _cleanup_agent_resources(self, agent: Any) -> None: - """Best-effort cleanup for temporary or cached agent instances.""" - if agent is None: - return - try: - if hasattr(agent, "shutdown_memory_provider"): - # Drain queued memory writes BEFORE tearing the provider down: shutdown_all() gives - # the serialized memory worker only ~5s and cancels the rest, so a /reset or rotation - # could drop handed-off writes and the next session loads stale memory. Bounded head - # start via the manager's own barrier (mirrors CLI exit); a failure never blocks teardown. - _mm = getattr(agent, "_memory_manager", None) - if _mm is not None and hasattr(_mm, "flush_pending"): - with suppress(Exception): - _mm.flush_pending(timeout=10) - # Pass the real transcript so ``on_session_end`` hooks don't see the empty default. - # ``_session_messages`` may be absent on ``object.__new__`` test stubs, hence getattr. - session_messages = getattr(agent, "_session_messages", None) - if isinstance(session_messages, list): - agent.shutdown_memory_provider(session_messages) - else: - agent.shutdown_memory_provider() - except Exception: - pass - # Close tool resources (sandboxes, browser daemons, background processes, httpx clients). - try: - if hasattr(agent, "close"): - agent.close() - except Exception: - pass - # Auxiliary async clients live in a process-global cache created from worker threads; drop - # entries whose event loop is dead so httpx transports don't accumulate across turns. - try: - from agent.auxiliary_client import cleanup_stale_async_clients - cleanup_stale_async_clients() - except Exception: - pass _STUCK_LOOP_THRESHOLD = 3 # restarts while active before auto-suspend _STUCK_LOOP_FILE = ".restart_failure_counts" - def _increment_restart_failure_counts(self, active_session_keys: set) -> None: - """Increment restart-failure counters for sessions active at shutdown. - - Persists to a JSON file so counters survive across restarts. Sessions NOT in - active_session_keys are removed (they completed successfully, so the loop is broken). - """ - import json - - path = _hermes_home / self._STUCK_LOOP_FILE - try: - counts = json.loads(path.read_text(encoding="utf-8")) if path.exists() else {} - except Exception: - counts = {} - - # Increment active sessions, remove inactive ones (loop broken) - new_counts = {} - for key in active_session_keys: - new_counts[key] = counts.get(key, 0) + 1 - # Keep any entries that are still above 0 even if not active now - # (they might become active again next restart) - - with suppress(Exception): - atomic_json_write(path, new_counts, indent=None) - - def _suspend_stuck_loop_sessions(self) -> int: - """Suspend sessions active across too many restarts (load → stuck → restart loop). - - Runs at startup AFTER suspend_recently_active(). Returns the number suspended. - """ - import json - - path = _hermes_home / self._STUCK_LOOP_FILE - if not path.exists(): - return 0 - - try: - counts = json.loads(path.read_text(encoding="utf-8")) - except Exception: - return 0 - - suspended = 0 - stuck_keys = [k for k, v in counts.items() if v >= self._STUCK_LOOP_THRESHOLD] - - for session_key in stuck_keys: - try: - entry = self.session_store._entries.get(session_key) - if entry and not entry.suspended: - entry.suspended = True - suspended += 1 - logger.warning( - "Auto-suspended stuck session %s (active across %d " - "consecutive restarts — likely a stuck loop)", - session_key, counts[session_key], - ) - except Exception: - pass - - if suspended: - with suppress(Exception): - self.session_store._save() - - # Clear the file — counters start fresh after suspension - with suppress(Exception): - path.unlink(missing_ok=True) - - return suspended - - async def _clear_restart_failure_count(self, session_key: str) -> None: - """Clear a completed session's restart-failure counter off-loop (atomic_json_write fsyncs).""" - import json - - path = _hermes_home / self._STUCK_LOOP_FILE - if not path.exists(): - return - try: - counts = json.loads(path.read_text(encoding="utf-8")) - if session_key in counts: - del counts[session_key] - if counts: - await asyncio.to_thread(atomic_json_write, path, counts, indent=None) - else: - path.unlink(missing_ok=True) - except Exception: - pass - - async def _launch_detached_restart_command(self) -> None: - import shutil - import subprocess - - hermes_cmd = _resolve_hermes_bin() - if not hermes_cmd: - logger.error("Could not locate hermes binary for detached /restart") - return - if self._detached_restart_helper_started: - return - self._detached_restart_helper_started = True - - current_pid = os.getpid() - restart_after_s = max(float(getattr(self, "_restart_drain_timeout", 0.0) or 0.0) + 5.0, 5.0) - - # On Windows there's no bash/setsid chain — spawn a tiny Python watcher directly via - # sys.executable instead. - if sys.platform == "win32": - import textwrap - from hermes_cli._subprocess_compat import ( - windows_detach_flags_without_breakaway, - windows_detach_popen_kwargs, - ) - - cmd_argv = [*hermes_cmd, "gateway", "restart"] - watcher = textwrap.dedent( - """ - import os, subprocess, sys, time - from hermes_cli._subprocess_compat import windows_detach_flags_without_breakaway - pid = int(sys.argv[1]) - restart_after_s = float(sys.argv[2]) - cmd = sys.argv[3:] - deadline = time.monotonic() + restart_after_s - - def _alive(p): - # On Windows, os.kill(pid, 0) is NOT a no-op — it maps to - # GenerateConsoleCtrlEvent(0, pid) (bpo-14484). Use the - # Win32 handle-based existence check instead. - if os.name == 'nt': - import ctypes - k32 = ctypes.windll.kernel32 - k32.OpenProcess.restype = ctypes.c_void_p - k32.WaitForSingleObject.restype = ctypes.c_uint - k32.GetLastError.restype = ctypes.c_uint - h = k32.OpenProcess(0x1000 | 0x100000, False, int(p)) - if not h: - return k32.GetLastError() != 87 - try: - return k32.WaitForSingleObject(h, 0) == 0x102 - finally: - k32.CloseHandle(h) - try: - os.kill(int(p), 0) - return True - except ProcessLookupError: - return False - except PermissionError: - return True - except OSError: - return False - - while time.monotonic() < deadline: - if not _alive(pid): - break - time.sleep(0.2) - subprocess.Popen( - cmd, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - creationflags=windows_detach_flags_without_breakaway(), - ) - """ - ).strip() - from tools.environments.local import build_subprocess_env - watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) - # The watcher must not inherit the gateway marker, else `hermes gateway restart` refuses to - # run (self-restart loop guard) and the gateway stays stopped. - watcher_env.pop("_HERMES_GATEWAY", None) - project_root = Path(__file__).resolve().parent.parent - # Console python under CREATE_NO_WINDOW owns one hidden console inherited by the restart - # child, so nothing flashes. Do NOT swap in pythonw.exe — a console-less watcher forces - # every console-subsystem descendant to allocate a visible conhost. - watcher_python = sys.executable - venv_dir = Path(watcher_env.get("VIRTUAL_ENV") or project_root / "venv") - site_packages = venv_dir / "Lib" / "site-packages" - if site_packages.exists(): - watcher_env["VIRTUAL_ENV"] = str(venv_dir) - pythonpath = [str(project_root), str(site_packages)] - if watcher_env.get("PYTHONPATH"): - pythonpath.append(watcher_env["PYTHONPATH"]) - watcher_env["PYTHONPATH"] = os.pathsep.join(dict.fromkeys(pythonpath)) - watcher_argv = [ - watcher_python, - "-c", - watcher, - str(current_pid), - str(restart_after_s), - *cmd_argv, - ] - # The watcher must break away from any job object the parent CLI lives in (Desktop - # wrappers, Windows Terminal, schtasks), else it is reaped when the CLI exits and the - # gateway never respawns. windows_detach_popen_kwargs() sets CREATE_BREAKAWAY_FROM_JOB, - # but a job without JOB_OBJECT_LIMIT_BREAKAWAY_OK rejects it (ERROR_ACCESS_DENIED as - # OSError); retry once without the bit, preserving argv and the scrubbed watcher_env. - try: - subprocess.Popen( - watcher_argv, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=watcher_env, - **windows_detach_popen_kwargs(), - ) - except OSError: - try: - subprocess.Popen( - watcher_argv, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=watcher_env, - creationflags=windows_detach_flags_without_breakaway(), - ) - except OSError as exc: - # Both spawns failed. Log only the interpreter basename and numeric errno — never - # argv, env, watcher source, or str(exc) (may carry a full path) — and return. - winerror = getattr(exc, "winerror", None) - error_code = winerror if winerror is not None else exc.errno - error_field = "winerror" if winerror is not None else "errno" - logger.warning( - "Detached restart watcher was not started after the " - "no-breakaway retry (%s; %s=%r). The gateway will not " - "be respawned by this restart attempt.", - os.path.basename(watcher_python), - error_field, - error_code, - ) - return - - cmd = " ".join(shlex.quote(part) for part in hermes_cmd) - shell_cmd = ( - f"deadline=$(( $(date +%s) + {int(restart_after_s)} )); " - f"while kill -0 {current_pid} 2>/dev/null && [ $(date +%s) -lt $deadline ]; do sleep 0.2; done; " - f"{cmd} gateway restart" - ) - # Same marker scrub as the Windows watcher: an inherited _HERMES_GATEWAY=1 makes the CLI's - # self-restart loop guard refuse silently (DEVNULL), so the gateway stops and never comes back. - from tools.environments.local import build_subprocess_env - watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) - watcher_env.pop("_HERMES_GATEWAY", None) - setsid_bin = shutil.which("setsid") - if setsid_bin: - subprocess.Popen( - [setsid_bin, "bash", "-lc", shell_cmd], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=watcher_env, - start_new_session=True, - ) - else: - subprocess.Popen( - ["bash", "-lc", shell_cmd], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=watcher_env, - start_new_session=True, - ) - - def _wedged_agent_count(self) -> int: - """Count running chat agents already past the inactivity timeout. - - No activity (API bytes, tool progress) for ``agent.gateway_timeout`` = wedged (the turn reaper's - threshold). Returns 0 when the timeout is disabled (the after-turn cap still bounds the wait). - Cron/API-server work has no activity clock and pending sentinels are brand-new, so neither - counts. Fail-open per agent: an unreadable activity summary means "not wedged". - """ - timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) - if timeout <= 0: - return 0 - wedged = 0 - for agent in list((getattr(self, "_running_agents", None) or {}).values()): - if agent is None or agent is _AGENT_PENDING_SENTINEL: - continue - summary_fn = getattr(agent, "get_activity_summary", None) - if not callable(summary_fn): - continue - try: - summary = summary_fn() - if not isinstance(summary, dict): - continue - idle = float(summary.get("seconds_since_activity", 0.0)) - except Exception: - continue - if idle >= timeout: - wedged += 1 - return wedged - - def _awaitable_work_count(self) -> int: - """Active work minus wedged turns — what the restart wait waits on.""" - return max(0, self._active_work_count() - self._wedged_agent_count()) - - async def _await_active_work_before_restart(self) -> bool: - """Wait for in-flight work to finish before entering ``stop()``. - - Calling ``stop()`` immediately would fold the requesting turn into the drain set and - force-interrupt it at ``restart_drain_timeout``; instead refuse new turns, wait for active - agents/cron/api work to reach zero, then ``stop()`` an idle gateway. Wedged turns - (``_wedged_agent_count``) are excluded — restart is the remedy, so ``stop()``'s drain - interrupts them. Returns True when drained to zero, False when the safety cap elapsed or - only wedged work remains (caller proceeds to ``stop()``). - """ - active = self._active_work_count() - if active <= 0: - return True - - awaitable = self._awaitable_work_count() - if awaitable <= 0: - logger.warning( - "Restart requested with %d active work unit(s), all wedged " - "past the inactivity timeout; skipping the after-turn wait " - "and proceeding to stop()/drain which will interrupt them", - active, - ) - return False - - timeout = float(getattr(self, "_restart_after_turn_timeout", 0.0) or 0.0) - if timeout <= 0: - logger.info( - "Restart requested with %d active work unit(s); " - "restart_after_turn_timeout=0 — entering stop()/drain immediately", - active, - ) - return False - - logger.info( - "Restart requested with %d active work unit(s); " - "deferring stop() until they finish (cap=%.0fs) so in-flight " - "turns are not amputated (#77184)", - active, - timeout, - ) - with suppress(Exception): - self._update_runtime_status("draining") - - loop = asyncio.get_running_loop() - deadline = loop.time() + timeout - last_status_at = 0.0 - while self._awaitable_work_count() > 0: - now = loop.time() - if now >= deadline: - logger.warning( - "Restart after-turn wait timed out after %.0fs with %d " - "still active; proceeding to stop()/drain which may " - "interrupt remaining work (#77184)", - timeout, - self._active_work_count(), - ) - return False - if (now - last_status_at) >= 30.0: - logger.info( - "Restart deferred: waiting on %d active work unit(s) " - "(%d wedged and excluded; %.0fs remaining before force drain)", - self._awaitable_work_count(), - self._wedged_agent_count(), - deadline - now, - ) - with suppress(Exception): - self._update_runtime_status("draining") - last_status_at = now - await asyncio.sleep(0.1) - - if self._active_work_count() > 0: - logger.warning( - "Restart deferred wait: %d wedged work unit(s) remain; " - "proceeding to stop()/drain which will interrupt them", - self._active_work_count(), - ) - return False - - logger.info( - "Restart deferred wait complete — active work drained; " - "proceeding to stop()" - ) - return True - - def request_restart(self, *, detached: bool = False, via_service: bool = False) -> bool: - if self._restart_task_started: - return False - self._restart_requested = True - self._restart_detached = detached - self._restart_via_service = via_service - self._restart_task_started = True - # Refuse new turns while in-flight work finishes. Keep ``_running`` True so adapters stay - # connected and the active turn can still deliver its final response. - self._draining = True - - async def _run_restart() -> None: - await self._await_active_work_before_restart() - # Launch the detached helper only AFTER the after-turn wait: its drain_timeout+5 deadline - # covers stop() teardown; earlier it would fire the restart mid-turn. - if detached: - try: - await self._launch_detached_restart_command() - except Exception as e: - logger.error("Failed to launch detached gateway restart helper: %s", e) - await asyncio.sleep(0.05) - await self.stop(restart=True, detached_restart=detached, service_restart=via_service) - - # Do NOT add _run_restart to _background_tasks: _stop_impl cancels every entry there, which - # would cancel it while awaiting _stop_task and propagate CancelledError into _stop_impl, - # skipping _shutdown_event.set() / _exit_code = 75. Keep a strong ref in self._restart_task. - self._restart_task = asyncio.create_task(_run_restart()) - return True # Reasons set by _stop_impl() on force-interrupt; "restart_interrupted" by suspend_recently_active() # on crash recovery (no .clean_shutdown marker). All mean "killed mid-turn" -> startup auto-resume. @@ -11038,2578 +4920,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew {"restart_timeout", "shutdown_timeout", "restart_interrupted"} ) - async def _run_startup_resume_event( - self, - adapter: BasePlatformAdapter, - event: MessageEvent, - session_key: str, - ) -> None: - """Dispatch one synthetic startup resume and wait for its agent turn. - - Inbound messages stay queued until the resumed turn finishes, else a user message can race it. - """ - try: - await adapter.handle_message(event) - session_tasks = getattr(adapter, "_session_tasks", {}) - task = session_tasks.get(session_key) if isinstance(session_tasks, dict) else None - if task is not None: - await asyncio.shield(task) - finally: - # The runner slot was pre-claimed before this task spawned; release it if handle_message - # raises before _handle_message takes ownership, else the real run's cleanup owns it. - _pre_state = self._peek_session_state(session_key) - if (_pre_state.turn.agent if _pre_state else None) is _AGENT_PENDING_SENTINEL: - self._release_running_agent_state(session_key) - - def _queue_startup_restore_event(self, event: MessageEvent) -> None: - queue = getattr(self, "_startup_restore_queue", None) - if queue is None: - queue = [] - self._startup_restore_queue = queue - queue.append(event) - try: - source = event.source - logger.info( - "Queued inbound message during gateway startup restore: platform=%s chat=%s", - source.platform.value if source and source.platform else "unknown", - source.chat_id if source else "unknown", - ) - except Exception: - pass - - async def _drain_startup_restore_queue(self) -> int: - """Replay inbound messages queued while startup auto-resume ran.""" - drained = 0 - queue = getattr(self, "_startup_restore_queue", None) - if queue is None: - return 0 - while queue: - event = queue.pop(0) - source = getattr(event, "source", None) - adapter = self._adapter_for_source(source) - if adapter is None: - logger.debug( - "Dropping startup-restore queued message: adapter unavailable for %s", - getattr(getattr(source, "platform", None), "value", None), - ) - continue - # Mark this replay so _handle_message does not queue it again while - # the restore gate remains closed for any fresh inbound arrivals. - with suppress(Exception): - setattr(event, "_hermes_startup_restore_replay", True) - await adapter.handle_message(event) - drained += 1 - return drained - - def _start_startup_warmup(self) -> None: - """Kick off the boot turn-machinery warm-up in the background. - - Called from ``start()`` right after the startup-restore gate closes so the warm-up overlaps - the network-bound platform connects; ``_finish_startup_restore`` awaits it (bounded). - """ - timeout = _startup_warmup_timeout_secs() - if timeout <= 0: - self._startup_warmup_task = None - return - self._startup_warmup_task = asyncio.ensure_future( - self._warm_turn_prerequisites() - ) - - async def _warm_turn_prerequisites(self) -> None: - """Initialize turn machinery on an executor thread before the gate opens. - - Never raises: a failed warm-up degrades to lazy init and must not block startup. - """ - try: - loop = asyncio.get_running_loop() - t0 = time.monotonic() - tool_count = await loop.run_in_executor(None, _warm_turn_machinery_sync) - logger.info( - "Turn machinery warmed in %.1fs (%d tool schema(s) materialized)", - time.monotonic() - t0, - tool_count, - ) - except Exception: - logger.warning( - "Turn-machinery warm-up failed; first inbound turn will " - "initialize lazily", - exc_info=True, - ) - - async def _await_startup_warmup(self) -> None: - """Bounded wait for the boot warm-up before the inbound gate opens. - - On timeout the gate opens anyway (availability outranks prompt completeness for a WEDGED - init) and the warm-up continues in the background; a late failure is still logged. - """ - task = getattr(self, "_startup_warmup_task", None) - if task is None or task.done(): - return - timeout = _startup_warmup_timeout_secs() - if timeout <= 0: - return - done, pending = await asyncio.wait({task}, timeout=timeout) - if pending: - logger.warning( - "Turn-machinery warm-up still running after %.0fs; opening " - "inbound gate anyway — the first turn may see lazily " - "initialized machinery (#99373). Warm-up continues in the " - "background.", - timeout, - ) - task.add_done_callback( - lambda t: GatewayRunner._log_late_background_failure( - t, - "boot turn-machinery warm-up failed after gate release", - level=logging.DEBUG, - ) - ) - - async def _finish_startup_restore(self) -> None: - """Wait (BOUNDED) for startup auto-resume, then release + drain inbound. - - Bounded by ``_startup_restore_drain_timeout_secs`` so one pathological boot-resume turn - cannot hold the gate shut for every channel; on timeout the gate opens and resume turns - finish in the background (NOT cancelled). Safe because ``_schedule_resume_pending_sessions`` - claims each ``_running_agents`` slot SYNCHRONOUSLY first, so drained inbound queues behind. - """ - tasks = list(getattr(self, "_startup_restore_tasks", []) or []) - if tasks: - timeout = _startup_restore_drain_timeout_secs() - if timeout > 0: - # asyncio.wait (unlike wait_for / gather+timeout) does NOT cancel pending tasks on - # timeout — the slow resume turn keeps running in the background. - done, pending = await asyncio.wait(tasks, timeout=timeout) - if pending: - logger.warning( - "Startup-restore gate released after %.0fs with %d boot " - "auto-resume turn(s) still running; draining inbound " - "queue now (resume slots already claimed, so no " - "duplicate agents). Slow turn(s) continue in the " - "background.", - timeout, - len(pending), - ) - # These tasks outlive the gate. Their normal done-callback only discards them - # from _background_tasks, so a LATER failure would be silently swallowed. - for task in pending: - task.add_done_callback(self._log_background_resume_result) - else: - # Non-positive timeout => opt out of the bound (historical - # "wait forever" behaviour). - await asyncio.gather(*tasks, return_exceptions=True) - done = set(tasks) - for task in done: - if task.cancelled(): - continue - exc = task.exception() - if exc is not None: - logger.debug( - "startup auto-resume task failed", - exc_info=(type(exc), exc, exc.__traceback__), - ) - self._startup_restore_tasks = [] - # Warm the turn machinery BEFORE the queue drains: replayed (and - # fresh) inbound turns must not build skeleton prompts (#99373). - await self._await_startup_warmup() - drained = await self._drain_startup_restore_queue() - self._startup_restore_in_progress = False - if drained: - logger.info("Drained %d inbound message(s) queued during startup restore", drained) - - @staticmethod - def _log_background_resume_result(task: "asyncio.Task") -> None: - """Done-callback for a boot-resume turn that outlived the startup-restore gate.""" - GatewayRunner._log_late_background_failure( - task, - "background startup auto-resume task failed after gate release", - level=logging.DEBUG, - ) - - @staticmethod - def _log_late_background_failure( - task: "asyncio.Task", message: str, *, level: int = logging.WARNING - ) -> None: - """Shared done-callback body for boot-path tasks that outlive the startup-restore gate: - surface a late failure otherwise swallowed once the task leaves ``_background_tasks``. - Cancellation (shutdown) is expected, not an error.""" - if task.cancelled(): - return - exc = task.exception() - if exc is not None: - logger.log( - level, - message, - exc_info=(type(exc), exc, exc.__traceback__), - ) - - async def _await_startup_boot_sends( - self, - *, - planned_restart_notification_pending: bool, - ) -> None: - """Run boot-path sends without letting them pin the inbound restore gate. - - Awaiting ``_send_restart_notification`` / ``_redeliver_pending_obligations`` inline before - the gate releases lets one Telegram flood-control sleep freeze inbound on every platform. - Same bounded ``asyncio.wait`` as the resume gate: on timeout return and let the sends finish - in the background (not cancelled). The ledger claim + ``resume_pending`` clear run INLINE - before the send task exists: bounded DB work, and deferring it let a hung notification - expire the gate with zero rows claimed, so answered turns were replayed AND redelivered. - """ - claimed = await self._claim_pending_obligations() - - async def _boot_sends() -> None: - await self._send_restart_notification() - if planned_restart_notification_pending: - try: - await self._send_home_channel_startup_notifications( - skip_targets=None, - ) - finally: - _clear_planned_restart_notification() - await self._redeliver_claimed_obligations(claimed) - - boot_task = asyncio.create_task(_boot_sends()) - timeout = _startup_restore_drain_timeout_secs() - if timeout > 0: - _done, pending = await asyncio.wait({boot_task}, timeout=timeout) - if pending: - logger.warning( - "Boot-path sends still running after %.0fs; releasing " - "inbound gate so other platforms are not frozen. " - "Restart notification / obligation redelivery continue " - "in the background.", - timeout, - ) - boot_task.add_done_callback(self._log_background_boot_send_result) - tasks = getattr(self, "_background_tasks", None) - if tasks is None: - self._background_tasks = set() - tasks = self._background_tasks - tasks.add(boot_task) - boot_task.add_done_callback(tasks.discard) - else: - await boot_task - - @staticmethod - def _log_background_boot_send_result(task: "asyncio.Task") -> None: - """Done-callback for boot-path sends that outlived the restore gate.""" - GatewayRunner._log_late_background_failure( - task, "background boot-path send failed after gate release: see traceback" - ) - - async def _clear_resume_pending_for_claimed_obligations( - self, claimed: list, *, require_success: bool = False - ) -> list: - """Clear resume flags and return rows safe to redeliver. - - Startup recovery stays best-effort. Runtime reconnect recovery is stricter: if the - session-store write fails the response must not be sent, or the turn could be resumed too. - """ - sendable = [] - for row in claimed: - session_key = row.get("session_key") or "" - if not session_key: - sendable.append(row) - continue - try: - await self.async_session_store.clear_resume_pending(session_key) - except Exception: - logger.debug( - "clear_resume_pending failed for %s", session_key, - exc_info=True, - ) - if not require_success: - sendable.append(row) - else: - sendable.append(row) - return sendable - - async def _claim_pending_obligations(self) -> list: - """Claim recoverable delivery-ledger rows and clear their ``resume_pending`` flags. - - Pure DB work, no sends. Must run INLINE at startup BEFORE ``_schedule_resume_pending_sessions`` - and before the abandonable boot-send task exists: these sessions already produced their - answer, so clearing ``resume_pending`` here stops the resume path from re-running (and - re-paying for) the turn however long the sends take. Rows that were mid-send or previously - rejected carry a visible recovered-reply marker so a possible duplicate is labeled, never - silent (gateway/delivery_ledger.py). Returns the claimed rows for redelivery. - """ - try: - from gateway.delivery_ledger import ( - ledger_enabled, - sweep_recoverable, - ) - - if not await asyncio.to_thread(ledger_enabled): - return [] - # Only claim rows whose exact transport owner is connected this boot. A multiplexed - # gateway can host several bot identities for one platform; platform-only filtering - # would spend a disconnected bot's retry budget merely because another bot is online. - _profile_adapters = getattr(self, "_profile_adapters", None) or {} - _deliverable_targets = { - (getattr(p, "value", str(p)), "default") for p in self.adapters - } - # Legacy rows predate adapter_profile. They are unambiguous only in a non-multiplexed - # gateway; fail closed when multiple bot identities share the process. - if not _profile_adapters: - _deliverable_targets.update( - (getattr(p, "value", str(p)), None) for p in self.adapters - ) - for _profile, _adapters in _profile_adapters.items(): - _deliverable_targets.update( - (getattr(p, "value", str(p)), _profile) for p in _adapters - ) - _deliverable = {platform for platform, _ in _deliverable_targets} - claimed = await asyncio.to_thread( - sweep_recoverable, - None, - deliverable_platforms=_deliverable, - deliverable_targets=_deliverable_targets, - ) - except Exception: - logger.debug("delivery ledger sweep failed", exc_info=True) - return [] - if not claimed: - return [] - - # Clear resume_pending for EVERY claimed row before any send: claiming already spent one - # redelivery attempt and the answer is in the ledger, so the resume path must never re-run. - await self._clear_resume_pending_for_claimed_obligations(claimed) - return claimed - - async def _redeliver_claimed_obligations(self, claimed: list) -> int: - """Redeliver final responses for rows claimed by :meth:`_claim_pending_obligations`. - - Network half of the split: runs inside the bounded boot-send task, so a flood-limited send - can be abandoned by the restore gate without reopening the turn-replay window. Returns count. - """ - if not claimed: - return 0 - try: - from gateway.delivery_ledger import ( - RECOVERED_MARKER, - mark_delivered, - mark_failed, - release_runtime_claim, - ) - except Exception: - logger.debug("delivery ledger import failed", exc_info=True) - return 0 - - redelivered = 0 - for row in claimed: - try: - platform = Platform(row["platform"]) - except Exception: - logger.debug( - "obligation %s: unknown platform %r", - row["obligation_id"], row.get("platform"), - ) - continue - if "profile" in row: - adapter = self._authorization_adapter( - platform, row.get("profile") - ) - else: - # Startup rows preserve the historical default-adapter route. - adapter = self.adapters.get(platform) - if adapter is None: - # Runtime claims have not reached a transport yet. If the - # reconnect vanished before dispatch, release the claim without - # spending an attempt so the next reconnect can retry it. - if row.get("runtime_recovery"): - try: - await asyncio.to_thread( - release_runtime_claim, - row["obligation_id"], - "send_path_degraded", - ) - except Exception: - logger.debug( - "failed to release undispatched runtime obligation %s", - row["obligation_id"], - exc_info=True, - ) - # Startup claims preserve their historical state; attempts cap - # + stale cutoff bound later retries. - continue - content = row["content"] - if row.get("needs_marker"): - content = row.get("marker", RECOVERED_MARKER) + content - metadata = ( - {"thread_id": row["thread_id"]} if row.get("thread_id") else None - ) - - try: - result = await adapter.send( - chat_id=row["chat_id"], - content=content, - metadata=metadata, - ) - except Exception as send_err: - logger.warning( - "obligation %s: redelivery send raised: %s", - row["obligation_id"], send_err, - ) - result = None - try: - if result is not None and getattr(result, "success", False): - await asyncio.to_thread(mark_delivered, row["obligation_id"]) - redelivered += 1 - logger.info( - "Redelivered recovered final response to %s:%s " - "(obligation %s, attempt %d)", - row["platform"], row["chat_id"], - row["obligation_id"], row["attempts"], - ) - else: - await asyncio.to_thread( - mark_failed, - row["obligation_id"], - str(getattr(result, "error", "") or "send failed"), - ) - except Exception: - logger.debug("delivery ledger update failed", exc_info=True) - return redelivered - - async def _redeliver_pending_obligations(self) -> int: - """Claim + redeliver in one call (:meth:`_claim_pending_obligations` then - :meth:`_redeliver_claimed_obligations`). Stable public shape for tests/external callers; - the startup path calls the halves separately so the DB half runs inline before the - abandonable send task. - """ - return await self._redeliver_claimed_obligations( - await self._claim_pending_obligations() - ) - - async def _redeliver_failed_obligations_for_platform( - self, - platform: Platform, - *, - profile: Optional[str] = None, - ) -> int: - """Replay one adapter identity's transient failures after reconnect. - - The startup sweep cannot claim live-owner rows, and an adapter can reconnect without the - process exiting, so ``send_path_degraded`` responses would otherwise stay failed until the - next restart. Claim/clear/send are best-effort and reuse the startup redelivery contract. - """ - try: - from gateway.delivery_ledger import ( - ledger_enabled, - release_runtime_claim, - sweep_failed_for_runtime, - ) - - if not await asyncio.to_thread(ledger_enabled): - return 0 - claimed = await asyncio.to_thread( - sweep_failed_for_runtime, - platform.value, - profile=profile, - ) - except Exception: - logger.debug( - "runtime delivery ledger sweep failed after %s reconnect", - platform.value, - exc_info=True, - ) - return 0 - if not claimed: - return 0 - - # Clear before any send so the reconnect path cannot both redeliver an - # already-produced answer and schedule the same agent turn for resume. - sendable = await self._clear_resume_pending_for_claimed_obligations( - claimed, require_success=True - ) - sendable_ids = {row["obligation_id"] for row in sendable} - for row in claimed: - if row["obligation_id"] in sendable_ids: - continue - try: - await asyncio.to_thread( - release_runtime_claim, - row["obligation_id"], - "send_path_degraded", - ) - except Exception: - logger.debug( - "failed to release runtime delivery claim %s", - row["obligation_id"], - exc_info=True, - ) - return await self._redeliver_claimed_obligations(sendable) - - def _schedule_resume_pending_sessions(self, platform=None) -> int: - """Auto-continue fresh restart-interrupted sessions after startup. - - Synthesizes the next turn once adapters are back online; the event text is empty so the - existing ``_is_resume_pending`` injection path owns the recovery wording. Sessions whose - adapter is not in ``self.adapters`` stay ``resume_pending`` for the reconnect watcher, which - re-calls this scoped to that ``platform`` (a reconnecting platform never touches another's - recoveries); sessions with a running agent are skipped so none is resumed twice. - """ - window = _auto_continue_freshness_window() - try: - with self.session_store._lock: # noqa: SLF001 — snapshot under lock - self.session_store._ensure_loaded_locked() # noqa: SLF001 - candidates = [ - entry for entry in self.session_store._entries.values() # noqa: SLF001 - if entry.resume_pending - and not entry.suspended - and entry.origin is not None - and entry.resume_reason in self._AUTO_RESUME_REASONS - and (platform is None or entry.origin.platform == platform) - ] - except Exception as exc: - logger.warning("Failed to enumerate resume-pending sessions: %s", exc) - return 0 - - # Defense-3: break the SIGTERM-respawn loop. Only count this boot when there are restart- - # interrupted sessions to resume — a clean boot must not accrue toward the breaker. If too - # many such boots hit the window, skip auto-resume for THIS boot only: the gateway still - # serves inbound; the session stays resume_pending so a real user message can continue it. - if candidates: - try: - from gateway import restart_loop_guard as _rlg - - _max_restarts, _window, _max_gap = self._restart_loop_guard_config() - if _rlg.check_and_record( - _max_restarts, _window, max_gap_seconds=_max_gap - ): - return 0 - except Exception as exc: # noqa: BLE001 — breaker must fail OPEN - logger.debug("Restart-loop guard check skipped: %s", exc) - - now = datetime.now() - scheduled = 0 - for entry in candidates: - marker = entry.last_resume_marked_at or entry.updated_at - if marker is not None and (now - marker).total_seconds() > window: - continue - - # Already being resumed (e.g. scheduled at startup and still - # in-flight) — don't synthesize a second continuation turn. - if self._is_session_running(entry.session_key): - continue - - source = entry.origin - adapter = self._adapter_for_source(source) - if adapter is None: - logger.debug( - "Skipping auto-resume for %s: adapter not ready for %s", - entry.session_key, - getattr(source.platform, "value", source.platform), - ) - continue - - # Validate the session owner against the current allowlist before auto-resuming: a - # session created before the allowlist existed (or whose owner was since removed) must - # not silently receive a full agent response just because it carries a resume marker. - try: - if not self._is_user_authorized(source): - logger.warning( - "Skipping auto-resume for %s: session owner is no " - "longer authorized under the current allowlist", - entry.session_key, - ) - continue - except Exception as exc: - logger.warning( - "Skipping auto-resume for %s: authorization check failed: %s", - entry.session_key, exc, - ) - continue - - # Claim the session slot *before* spawning the task so an inbound message arriving - # between task creation and the task's first await (where _process_message_background - # sets the real sentinel) sees the slot occupied and queues, not a duplicate AIAgent. - _resume_state = self._session_state(entry.session_key) - _resume_state.turn.agent = _AGENT_PENDING_SENTINEL - _resume_state.turn.started_ts = time.time() - self._persist_active_agents() - - # Empty-text internal event: the _is_resume_pending branch in _handle_message_with_agent - # prepends the reason-aware system note before the turn runs. - event = MessageEvent( - text="", - message_type=MessageType.TEXT, - source=source, - internal=True, - ) - task = asyncio.create_task( - self._run_startup_resume_event(adapter, event, entry.session_key) - ) - self._background_tasks.add(task) - task.add_done_callback(self._background_tasks.discard) - if getattr(self, "_startup_restore_in_progress", False): - tasks = getattr(self, "_startup_restore_tasks", None) - if tasks is None: - tasks = [] - self._startup_restore_tasks = tasks - tasks.append(task) - scheduled += 1 - if scheduled: - logger.info( - "Scheduled auto-resume for %d restart-interrupted session(s)", - scheduled, - ) - return scheduled - - def _startup_should_abort(self) -> bool: - return ( - self._restart_requested - or self._draining - or self._shutdown_event.is_set() - ) - - async def _abort_startup_if_shutdown_requested( - self, - adapter: Optional[BasePlatformAdapter] = None, - platform: Optional[Platform] = None, - ) -> bool: - """Clean up and exit startup when restart/shutdown begins mid-startup.""" - if not self._startup_should_abort(): - return False - if adapter is not None and platform is not None: - try: - await adapter.cancel_background_tasks() - except Exception as e: - logger.debug("✗ %s background-task cancel error: %s", platform.value, e) - await self._safe_adapter_disconnect(adapter, platform) - stop_task = self._stop_task - current_task = asyncio.current_task() - if stop_task is not None and stop_task is not current_task: - await stop_task - elif not self._shutdown_event.is_set(): - await self.stop( - restart=self._restart_requested, - detached_restart=self._restart_detached, - service_restart=self._restart_via_service, - ) - return True - - def _start_loop_liveness_guards(self, loop: asyncio.AbstractEventLoop) -> None: - """Arm the selector floor and out-of-loop watchdog before adapters. - - Disabled entirely with ``gateway.loop_watchdog: false`` in config.yaml (config-only knob). - """ - config = getattr(self, "config", None) - if config is not None and not getattr(config, "loop_watchdog", True): - return - if getattr(self, "_loop_floor_timer_handle", None) is None: - try: - self._loop_floor_timer_handle = _arm_loop_floor_timer(loop) - except Exception: - logger.debug("Failed to arm gateway loop floor timer", exc_info=True) - - watchdog = getattr(self, "_loop_liveness_watchdog", None) - if watchdog is None or not watchdog.is_alive(): - try: - # getattr defaults cover the config=None / bare-object test path; config-loaded - # values are already validated+clamped by GatewayConfig.from_dict; no re-clamping. - interval = getattr( - config, - "loop_watchdog_probe_interval_s", - DEFAULT_LOOP_WATCHDOG_INTERVAL_S, - ) - timeout = getattr( - config, - "loop_watchdog_probe_timeout_s", - DEFAULT_LOOP_WATCHDOG_TIMEOUT_S, - ) - strikes = getattr( - config, - "loop_watchdog_max_strikes", - DEFAULT_LOOP_WATCHDOG_MAX_STRIKES, - ) - self._loop_liveness_watchdog = start_loop_liveness_watchdog( - loop, - probe_interval=float(interval), - probe_timeout=float(timeout), - max_strikes=int(strikes), - ) - except Exception: - logger.debug("Failed to start gateway loop liveness watchdog", exc_info=True) - - def _stop_loop_liveness_guards(self) -> None: - """Disarm lifetime liveness guards before shutdown can load the loop.""" - watchdog = getattr(self, "_loop_liveness_watchdog", None) - self._loop_liveness_watchdog = None - if watchdog is not None: - try: - watchdog.stop() - except Exception: - logger.debug("Failed to stop gateway loop liveness watchdog", exc_info=True) - - floor_timer = getattr(self, "_loop_floor_timer_handle", None) - self._loop_floor_timer_handle = None - if floor_timer is not None: - try: - floor_timer.cancel() - except Exception: - logger.debug("Failed to cancel gateway loop floor timer", exc_info=True) - - # Also disarm the heartbeat writer task: once shutdown starts loading the loop, a heartbeat - # that keeps refreshing the file makes a draining gateway look healthy to external probes. - heartbeat = getattr(self, "_loop_heartbeat_task", None) - self._loop_heartbeat_task = None - if heartbeat is not None: - try: - heartbeat.cancel() - except Exception: - logger.debug("Failed to cancel gateway loop heartbeat task", exc_info=True) - - async def _consume_clean_shutdown_marker(self, marker_path) -> int: - """Discard orphan turn markers before consuming a clean-exit receipt. - - If persistence or marker removal fails, startup must fail closed: continuing with the old - receipt would let a later unclean exit masquerade as clean and discard interrupted turns. - """ - discarded = await self.async_session_store.discard_active_turn_markers() - marker_path.unlink() - return discarded - - async def _recover_unclean_sessions(self) -> tuple[int, int]: - """Recover exact active turns, then run the legacy recency fallback.""" - exact = 0 - fallback = 0 - try: - agent_timeout = max(1.0, _float_env("HERMES_AGENT_TIMEOUT", 1800)) - marker_max_age = max(60 * 60, int(agent_timeout * 2)) - exact = await self.async_session_store.recover_interrupted_turns( - max_age_seconds=marker_max_age - ) - except Exception as exc: - logger.warning("Exact active-turn recovery on startup failed: %s", exc) - try: - fallback = await self.async_session_store.suspend_recently_active( - max_age_seconds=120 - ) - except Exception as exc: - logger.warning("Legacy session recovery on startup failed: %s", exc) - return exact, fallback - - @staticmethod - def _start_hosted_room_worker_sync(): - """Start the local Group Chat worker without importing the dashboard.""" - - import tui_gateway.server # noqa: F401 - from tui_gateway import methods_groups - - service = methods_groups.get_hosted_room_service() - if service is None: - service = methods_groups.start_hosted_room_service() - if service is None: - raise RuntimeError("Group Chat worker has no bound session backend") - status = service.runtime.status() - if not status.get("running") or status.get("stopping"): - raise RuntimeError("Group Chat worker did not start") - return service - - async def _ensure_hosted_room_worker(self): - return await asyncio.to_thread(self._start_hosted_room_worker_sync) - - async def _hosted_room_worker_watcher(self, interval: float = 1.0) -> None: - """Keep the room worker alive for the messaging gateway lifetime.""" - - while self._running: - await self._ensure_hosted_room_worker() - await asyncio.sleep(interval) - - async def _stop_hosted_room_worker(self, timeout: float = 5.0) -> bool: - """Pause room execution durably without interrupting accepted turns.""" - - from tui_gateway import methods_groups - - return await asyncio.to_thread( - methods_groups.stop_hosted_room_service, - timeout=timeout, - ) - - def _start_loop_heartbeat_task(self) -> None: - """Start the loop-liveness heartbeat task, idempotent. - - An asyncio task so a frozen loop stops refreshing ``state/gateway.heartbeat``; cancelled - with the other background tasks in stop(). Best-effort — must never abort startup. - """ - try: - _existing_hb = getattr(self, "_loop_heartbeat_task", None) - if _existing_hb is not None and not _existing_hb.done(): - return - self._loop_heartbeat_task = asyncio.create_task( - loop_heartbeat_forever( - interval_s=DEFAULT_HEARTBEAT_INTERVAL_S, - start_time=getattr(self, "_gateway_started_at", 0.0), - ) - ) - # PERMANENT for the process lifetime, same as a _spawn_supervised watcher — tag it so - # _scale_to_zero_has_live_background_work() doesn't treat an armed, otherwise-idle - # gateway as busy forever. - self._loop_heartbeat_task._hermes_supervised_watcher = True # type: ignore[attr-defined] - _bg = getattr(self, "_background_tasks", None) - if _bg is not None: - _bg.add(self._loop_heartbeat_task) - self._loop_heartbeat_task.add_done_callback(_bg.discard) - except Exception: - logger.debug("Failed to start gateway loop heartbeat", exc_info=True) - - async def start(self) -> bool: - """Start the gateway and all configured platform adapters. - - Returns True if at least one adapter connected successfully. - """ - logger.info("Starting Hermes Gateway...") - # Enable faulthandler for stack dumps on freezes/crashes. Falls back to a log file when - # sys.stderr is None (Windows VBS / pythonw / detached service) — otherwise the gateway - # would die here and take every adapter offline. - try: - faulthandler.enable() - except (RuntimeError, ValueError, OSError): - try: - _fh_log_dir = getattr(self.config, "log_dir", None) or os.path.join( - str(get_hermes_home()), - "logs", - ) - os.makedirs(_fh_log_dir, exist_ok=True) - _fh_enable_path = os.path.join(_fh_log_dir, "gateway_faulthandler.log") - _fh_enable_file = open(_fh_enable_path, "a", encoding="utf-8") - faulthandler.enable(file=_fh_enable_file, all_threads=True) - except Exception: - logger.debug("faulthandler.enable() unavailable", exc_info=True) - # Also dump stacks to a rotating file for off-line analysis under a service manager that - # doesn't capture stderr. faulthandler.register()/SIGUSR2 are POSIX-only: skip the signal- - # triggered file dump on Windows (faulthandler.enable() above still covers fatal errors). - _sigusr2 = getattr(signal, "SIGUSR2", None) - if _sigusr2 is not None and hasattr(faulthandler, "register"): - try: - _log_dir = getattr(self.config, "log_dir", None) or os.path.join( - str(get_hermes_home()), - "logs", - ) - _faulthandler_path = os.path.join(_log_dir, "gateway_faulthandler.log") - os.makedirs(_log_dir, exist_ok=True) - _fh = open(_faulthandler_path, "a", encoding="utf-8") - faulthandler.register( - _sigusr2, - file=_fh, - all_threads=True, - chain=True, - ) - except Exception: - logger.debug("Could not set up faulthandler file logging", exc_info=True) - - try: - self._gateway_loop = asyncio.get_running_loop() - except RuntimeError: - self._gateway_loop = None - if self._gateway_loop is not None: - self._start_loop_liveness_guards(self._gateway_loop) - # Loop confirmed live: the startup-liveness watchdog is done and the loop-liveness - # watchdog (armed above) takes over. Disarm even when loop guards are config-disabled — - # the startup watchdog covers only the pre-loop window. Deliberately inside this branch: - # if the loop isn't live, startup has NOT reached the milestone and it must stay armed. - try: - from gateway.startup_watchdog import disarm_startup_watchdog - - disarm_startup_watchdog() - except Exception: - logger.debug("Startup watchdog disarm failed", exc_info=True) - logger.info("Session storage: %s", self.config.sessions_dir) - - # Sanity-check that systemd's TimeoutStopSec covers our drain window: a unit file from - # before a hermes-agent upgrade (no ``hermes setup`` re-run) may encode the old default, - # so SIGKILL hits mid-drain and looks like a phantom kill in the journal. Never raises. - try: - from gateway.shutdown_forensics import check_systemd_timing_alignment - _alignment = check_systemd_timing_alignment( - self._restart_drain_timeout, - getattr(self, "_cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT), - ) - if _alignment is not None and _alignment.get("mismatch"): - logger.warning( - "Stale systemd unit detected: %s has TimeoutStopSec=%.0fs but " - "drain_timeout=%.0fs cron_drain_timeout=%.0fs (expected >=%.0fs). " - "systemd may SIGKILL the gateway mid-drain. Run " - "`hermes gateway install --force` to regenerate the unit, or " - "shorten agent.restart_drain_timeout / agent.cron_drain_timeout.", - _alignment.get("unit", "(unknown)"), - _alignment["timeout_stop_sec"], - _alignment["drain_timeout"], - _alignment.get( - "cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT - ), - _alignment["expected_min"], - ) - except Exception as _e: - logger.debug("check_systemd_timing_alignment failed: %s", _e) - # Log the resolved max_iterations budget so operators can verify the config.yaml → env - # bridge at a glance (instead of silently running at a stale .env value for weeks). - try: - _effective_max_iter = int(os.getenv("HERMES_MAX_ITERATIONS", "500")) - logger.info( - "Agent budget: max_iterations=%d (agent.max_turns from config.yaml, " - "or HERMES_MAX_ITERATIONS from .env, or default 500)", - _effective_max_iter, - ) - except Exception: - pass - # Redaction is ON by default; warn prominently when an operator has explicitly opted out so - # the downgrade isn't forgotten. The redactor snapshots its state at import time, so this - # log line is the source of truth for the process lifetime. - try: - _redact_raw = os.getenv("HERMES_REDACT_SECRETS", "true") - _redact_on = _redact_raw.lower() in {"1", "true", "yes", "on"} - if _redact_on: - logger.info( - "Secret redaction: ENABLED (tool output, logs, and chat " - "responses are scrubbed before delivery)" - ) - else: - logger.warning( - "Secret redaction: DISABLED (HERMES_REDACT_SECRETS=%s). " - "API keys and tokens may appear verbatim in chat output, " - "session JSONs, and logs. Set security.redact_secrets: true " - "in config.yaml to re-enable.", - _redact_raw, - ) - except Exception: - pass - try: - from hermes_cli.profiles import get_active_profile_name - _profile = get_active_profile_name() - if _profile and _profile != "default": - logger.info("Active profile: %s", _profile) - except Exception: - pass - try: - from gateway.status import write_runtime_status - write_runtime_status( - gateway_state="starting", - exit_reason=None, - clear_profile_platforms=True, - ) - except Exception: - pass - try: - from hermes_cli.config import load_config - from agent.monitoring.gateway_health_export import start_gateway_health_export - self._gateway_health_export_runtime = start_gateway_health_export(load_config()) - if getattr(self._gateway_health_export_runtime, "enabled", False): - logger.info("Gateway health OTLP export: enabled") - except Exception: - logger.debug("gateway health OTLP export startup failed", exc_info=True) - - # Log any active supply-chain security advisories. Deliberately does NOT block startup or - # surface inline to users — only the operator can act (uninstall, rotate credentials). - try: - from hermes_cli.security_advisories import ( - detect_compromised, - gateway_log_message, - ) - _adv_hits = detect_compromised() - _adv_msg = gateway_log_message(_adv_hits) - if _adv_msg: - logger.warning("%s", _adv_msg) - logger.warning( - "Run `hermes doctor` on the gateway host for full " - "remediation steps." - ) - except Exception: - logger.debug( - "security advisory check failed at gateway startup", - exc_info=True, - ) - if await self._abort_startup_if_shutdown_requested(): - return True - - # Warn if no user allowlists are configured and open access is not opted in - _builtin_allowed_vars = ( - "TELEGRAM_ALLOWED_USERS", "DISCORD_ALLOWED_USERS", - "WHATSAPP_ALLOWED_USERS", "WHATSAPP_CLOUD_ALLOWED_USERS", - "SLACK_ALLOWED_USERS", - "SIGNAL_ALLOWED_USERS", "SIGNAL_GROUP_ALLOWED_USERS", - "TELEGRAM_GROUP_ALLOWED_USERS", - "TELEGRAM_GROUP_ALLOWED_CHATS", - "EMAIL_ALLOWED_USERS", - "SMS_ALLOWED_USERS", "MATTERMOST_ALLOWED_USERS", - "MATRIX_ALLOWED_USERS", "DINGTALK_ALLOWED_USERS", - "FEISHU_ALLOWED_USERS", - "WECOM_ALLOWED_USERS", - "WECOM_CALLBACK_ALLOWED_USERS", - "WEIXIN_ALLOWED_USERS", - "BLUEBUBBLES_ALLOWED_USERS", - "QQ_ALLOWED_USERS", - "YUANBAO_ALLOWED_USERS", - "GATEWAY_ALLOWED_USERS", - ) - _builtin_allow_all_vars = ( - "TELEGRAM_ALLOW_ALL_USERS", "DISCORD_ALLOW_ALL_USERS", - "WHATSAPP_ALLOW_ALL_USERS", "WHATSAPP_CLOUD_ALLOW_ALL_USERS", - "SLACK_ALLOW_ALL_USERS", - "SIGNAL_ALLOW_ALL_USERS", "EMAIL_ALLOW_ALL_USERS", - "SMS_ALLOW_ALL_USERS", "MATTERMOST_ALLOW_ALL_USERS", - "MATRIX_ALLOW_ALL_USERS", "DINGTALK_ALLOW_ALL_USERS", - "FEISHU_ALLOW_ALL_USERS", - "WECOM_ALLOW_ALL_USERS", - "WECOM_CALLBACK_ALLOW_ALL_USERS", - "WEIXIN_ALLOW_ALL_USERS", - "BLUEBUBBLES_ALLOW_ALL_USERS", - "QQ_ALLOW_ALL_USERS", - "YUANBAO_ALLOW_ALL_USERS", - ) - # Also pick up plugin-registered platforms — each entry can declare its own - # allowed_users_env / allow_all_env, so the warning stays accurate as plugins (IRC) arrive. - _plugin_allowed_vars: tuple = () - _plugin_allow_all_vars: tuple = () - try: - from gateway.platform_registry import platform_registry - _plugin_allowed_vars = tuple( - e.allowed_users_env for e in platform_registry.plugin_entries() - if e.allowed_users_env - ) - _plugin_allow_all_vars = tuple( - e.allow_all_env for e in platform_registry.plugin_entries() - if e.allow_all_env - ) - except Exception: - pass - _any_allowlist = any( - os.getenv(v) for v in _builtin_allowed_vars + _plugin_allowed_vars - ) - _allow_all = os.getenv("GATEWAY_ALLOW_ALL_USERS", "").lower() in {"true", "1", "yes"} or any( - os.getenv(v, "").lower() in {"true", "1", "yes"} - for v in _builtin_allow_all_vars + _plugin_allow_all_vars - ) - if not _any_allowlist and not _allow_all: - logger.warning( - "No env user allowlists configured. Messaging platforms default to " - "pairing/allowlist policies and will deny unknown senders unless you " - "configure platform allowlists (e.g., TELEGRAM_ALLOWED_USERS=your_id) " - "or explicitly opt in with GATEWAY_ALLOW_ALL_USERS=true plus " - "dm_policy/group_policy: open on the platform." - ) - - reason = _own_policy_open_startup_violation(self.config) - if reason: - platform_value = reason.split(":", 1)[0] - allow_all_env = None - for platform, open_env in _OWN_POLICY_OPEN_ENV.items(): - if platform.value == platform_value: - allow_all_env = open_env[2] - break - logger.error( - "Refusing to start: %s has dm_policy/group_policy set to 'open' " - "but neither GATEWAY_ALLOW_ALL_USERS nor %s is enabled.", - platform_value, - allow_all_env or "a platform allow-all flag", - ) - _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) - self._request_clean_exit(reason) - return True - - # Discover Python plugins before shell hooks so plugin block decisions take precedence in - # tie cases. Explicit here because the gateway lazily imports run_agent per request, so - # the discover_plugins() side-effect in model_tools.py is NOT guaranteed to have run yet. - try: - from hermes_cli.plugins import discover_plugins - discover_plugins() - except Exception: - logger.warning( - "plugin discovery failed at gateway startup", exc_info=True, - ) - - # Register the generic relay adapter only if GATEWAY_RELAY_URL / gateway.relay_url is set. - # No URL -> no-op, so direct/single-tenant deployments are unaffected. - try: - from gateway.relay import ( - register_relay_adapter, - relay_url, - self_provision_relay, - send_relay_policy, - ) - - # Boot-time relay self-provision: resolve the agent's NAS token -> POST /relay/provision - # -> set GATEWAY_RELAY_* in os.environ BEFORE registration reads them. Never raises. - self_provision_relay() - - if register_relay_adapter(): - logger.info("relay adapter registered (connector at %s)", relay_url()) - # Declare this gateway's relevance policy (mention-gating / free-response / allow- - # bots) to the connector so the SAME behavior governs relay delivery (Phase 6 Unit - # ζ). Runs after the secret is resolved; never raises, never blocks boot. - send_relay_policy() - except Exception: - logger.warning( - "relay adapter registration failed at gateway startup", exc_info=True, - ) - - # Register declarative shell hooks from cli-config.yaml. Gateway has no TTY, so consent must - # come from --accept-hooks, HERMES_ACCEPT_HOOKS, or hooks_auto_accept: true; pass - # accept_hooks=False and let register_from_config resolve env + config. Never blocks startup. - try: - from hermes_cli.config import load_config - from agent.shell_hooks import register_from_config - _hooks_cfg = load_config() - register_from_config(_hooks_cfg, accept_hooks=False) - - from agent.outbound_webhooks import ( - register_from_config as register_outbound_webhooks, - ) - register_outbound_webhooks(_hooks_cfg) - except Exception: - logger.debug( - "shell-hook registration failed at gateway startup", - exc_info=True, - ) - - # Discover and load event hooks - self.hooks.discover_and_load() - - # Recover background processes from checkpoint (crash recovery) - try: - from tools.process_registry import process_registry - recovered = process_registry.recover_from_checkpoint() - if recovered: - logger.info("Recovered %s background process(es) from previous run", recovered) - except Exception as e: - logger.warning("Process checkpoint recovery: %s", e) - - # Recover sessions active when the gateway last exited. Exact durable turn markers cover - # long-running work; the 120s recency heuristic remains as a fallback for turns from older - # versions without markers. SKIP after a clean shutdown — the previous process already drained. - _clean_marker = _hermes_home / ".clean_shutdown" - if _clean_marker.exists(): - logger.info("Previous gateway exited cleanly — skipping session suspension") - try: - discarded = await self._consume_clean_shutdown_marker(_clean_marker) - except Exception as exc: - logger.error( - "Clean-start marker cleanup failed; refusing startup so the " - "clean-exit receipt cannot mask a later unclean exit: %s", - exc, - ) - raise RuntimeError("clean-start recovery cleanup failed") from exc - if discarded: - logger.info( - "Discarded %d orphan active-turn marker(s) after clean shutdown", - discarded, - ) - else: - exact, fallback = await self._recover_unclean_sessions() - recovered = exact + fallback - if recovered: - logger.info( - "Marked %d in-flight session(s) as resumable from previous run " - "(%d exact, %d legacy)", - recovered, - exact, - fallback, - ) - - # Stuck-loop detection: a session active across 3+ consecutive restarts is probably looping - # (its history keeps hanging the agent); auto-suspend so the next message starts clean. - try: - stuck = self._suspend_stuck_loop_sessions() - if stuck: - logger.warning("Auto-suspended %d stuck-loop session(s)", stuck) - except Exception as e: - logger.debug("Stuck-loop detection failed: %s", e) - - # Serialize startup restore against inbound dispatch: adapters can receive messages as soon - # as they connect, but restart-interrupted sessions are not auto-resumed until all startup - # wiring below completes, so inbound queues until every synthetic resume turn has finished. - self._startup_restore_in_progress = True - self._startup_restore_queue = [] - self._startup_restore_tasks = [] - # Fresh-boot readiness: with no resume_pending sessions the gate opens almost immediately - # while the turn machinery is still cold, so a message in that window got a skeleton system - # prompt. Warm NOW to overlap the connects below; _finish_startup_restore awaits it (bounded). - self._start_startup_warmup() - - connected_count = 0 - enabled_platform_count = 0 - startup_nonretryable_errors: list[str] = [] - startup_retryable_errors: list[str] = [] - _multiplex_on = bool(getattr(self.config, "multiplex_profiles", False)) - _multiplex_skipped_platforms: list[Platform] = [] - # Initialize and connect each configured platform. connect() calls run concurrently so one - # slow/failing platform (e.g. Telegram behind a dead proxy) cannot delay the others by a - # full timeout window; the cheap serial pre-filter and per-platform timeouts are unchanged. - _pending_connects = [] # (platform, platform_config, adapter) - for platform, platform_config in self.config.platforms.items(): - if await self._abort_startup_if_shutdown_requested(): - return True - if not platform_config.enabled: - continue - # Under multiplexing, a platform may be enabled on the default profile's config.yaml - # while its bot token lives only in a secondary profile's .env. Starting the primary with - # an empty token fails at once and queues a reconnect loop that can never heal; the - # secondary starts its own adapter with the real token, so skip the empty primary. - if _multiplex_on and not _platform_has_bot_credential(platform, platform_config): - logger.info( - "Skipping %s on default profile: no bot credential in this " - "profile's secrets. Secondary multiplexed profiles that " - "provide the token will still connect.", - platform.value, - ) - _multiplex_skipped_platforms.append(platform) - continue - enabled_platform_count += 1 - - adapter = self._create_adapter(platform, platform_config) - if not adapter: - # Distinguish between missing builtin deps and missing plugin - _pval = platform.value - _builtin_names = {m.value for m in Platform.__members__.values()} - if _pval not in _builtin_names: - logger.warning( - "No adapter for '%s' -- is the plugin installed? " - "(platform is enabled in config.yaml but no plugin registered it)", - _pval, - ) - else: - logger.warning("No adapter available for %s", _pval) - continue - - # Set up message + fatal error handlers. Under multiplexing the default profile needs - # the same whole-handler runtime scope as a secondary profile: authorization and prompt - # rendering both run before the narrower agent-turn scope is installed. - adapter.set_message_handler(self._primary_message_handler()) - adapter.set_fatal_error_handler(self._handle_adapter_fatal_error) - adapter.set_session_store(self.session_store) - adapter.set_busy_session_handler(self._handle_active_session_busy_message) - _set_reaction = getattr(adapter, "set_reaction_handler", None) - if callable(_set_reaction): - _set_reaction(self._handle_reaction_event) - adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) - adapter.set_authorization_check(self._make_adapter_auth_check(adapter.platform)) - adapter.set_platform_event_handler(self._primary_platform_event_handler()) - adapter._busy_text_mode = self._busy_text_mode - _pending_connects.append((platform, platform_config, adapter)) - - if await self._abort_startup_if_shutdown_requested(): - return True - - async def _connect_one_startup(p, p_cfg, adp): - """Connect a single platform; never let one block the others (#83791).""" - if await self._abort_startup_if_shutdown_requested(adp, p): - return (p, adp, p_cfg, "aborted", None) - logger.info("Connecting to %s...", p.value) - self._update_platform_runtime_status( - p.value, platform_state="connecting", error_code=None, error_message=None, - ) - try: - ok = await self._connect_initial_adapter_with_timeout(adp, p) - except Exception as _exc: # noqa: BLE001 - surfaced below as a retryable error - return (p, adp, p_cfg, "exception", _exc) - return (p, adp, p_cfg, "ok" if ok else "failed", None) - - if _pending_connects: - # Abort-aware concurrent wait (parity with the serial loop's between-platforms check): a - # restart/shutdown requested mid-connect must cancel still-pending connects, clean up the - # ones already completed, and abort startup. - _task_map: dict = {} - for (p, c, a) in _pending_connects: - _t = asyncio.ensure_future(_connect_one_startup(p, c, a)) - _task_map[_t] = (p, c, a) - _pending_tasks = set(_task_map) - _abort_mid_connect = False - while _pending_tasks: - _done, _pending_tasks = await asyncio.wait( - _pending_tasks, timeout=0.05 - ) - if _pending_tasks and self._startup_should_abort(): - _abort_mid_connect = True - break - if _abort_mid_connect: - # Cancel and fully settle the in-flight connects FIRST, so a completed adapter's - # disconnect cannot unblock a sibling's connect() before the sibling is cancelled. - for _t in _pending_tasks: - _t.cancel() - await asyncio.gather(*_pending_tasks, return_exceptions=True) - for _t in _pending_tasks: - _p, _c, _a = _task_map[_t] - try: - await _a.cancel_background_tasks() - except Exception as e: - logger.debug( - "✗ %s background-task cancel error: %s", _p.value, e - ) - await self._safe_adapter_disconnect(_a, _p) - # Tear down adapters whose connect already succeeded — they - # were never registered, so stop() won't reach them. - for _t, (_p, _c, _a) in _task_map.items(): - if _t in _pending_tasks or _t.cancelled(): - continue - _res = _t.exception() is None and _t.result() or None - if _res and _res[3] == "ok": - try: - await _a.cancel_background_tasks() - except Exception as e: - logger.debug( - "✗ %s background-task cancel error: %s", - _p.value, e, - ) - await self._safe_adapter_disconnect(_a, _p) - await self._abort_startup_if_shutdown_requested() - return True - _raw = [ - _t.exception() or _t.result() for _t in _task_map - ] - else: - _raw = [] - - # Aggregate results single-threaded so shared state (self.adapters, self._failed_platforms, - # the error lists, connected_count) is mutated exactly as the original serial loop did -- - # only the connect() wall-clock overlap changed. - for _item in _raw: - if isinstance(_item, Exception): - # Unexpected escape from _connect_one_startup (shouldn't happen); - # log and skip rather than aborting the whole startup. - logger.error("Unexpected startup connect error: %s", _item) - continue - platform, adapter, platform_config, outcome, exc = _item - if outcome == "aborted": - continue - if outcome == "exception": - logger.error("\u2717 %s error: %s", platform.value, exc) - # Same defensive cleanup path for exceptions -- an adapter that raised mid-connect - # may still have a live aiohttp.ClientSession or child subprocess. - await self._safe_adapter_disconnect(adapter, platform) - self._update_platform_runtime_status( - platform.value, platform_state="retrying", error_code=None, error_message=str(exc), - ) - startup_retryable_errors.append(f"{platform.value}: {exc}") - # Unexpected exceptions are typically transient -- queue for retry - self._failed_platforms[platform] = { - "config": platform_config, - "attempts": 1, - "next_retry": time.monotonic() + 30, - "queued_at": time.monotonic(), - "credential_claim": self._adapter_credential_claim(platform, adapter), - "listener_claim": self._adapter_listener_claim(platform, adapter), - } - continue - if outcome == "ok": - self.adapters[platform] = adapter - self._sync_voice_mode_state_to_adapter(adapter) - # Wire voice input callback at connect time so voice - # transcription is forwarded without requiring /voice join. - self._bind_voice_input_callback(adapter) - connected_count += 1 - self._update_platform_runtime_status( - platform.value, platform_state="connected", error_code=None, error_message=None, - ) - logger.info("\u2713 %s connected", platform.value) - else: # outcome == "failed" - logger.warning("\u2717 %s failed to connect", platform.value) - # Defensive cleanup: a failed connect() may have allocated resources - # (aiohttp.ClientSession, poll tasks, bridge subprocesses) before giving up. - await self._safe_adapter_disconnect(adapter, platform) - if adapter.has_fatal_error: - # A live foreign holder of this bot token is a single-writer ownership conflict, - # not a blip — ``_acquire_platform_lock`` emits it retryable only so a MID-RUN - # reconnect can recover. At startup route it non-retryable: with nothing connected - # the gateway exits 78 instead of sitting alive and deaf in the retry queue. - _retryable = adapter.fatal_error_retryable and not ( - is_global_startup_conflict(adapter.fatal_error_code) - ) - self._update_platform_runtime_status( - platform.value, - platform_state="retrying" if _retryable else "fatal", - error_code=adapter.fatal_error_code, - error_message=adapter.fatal_error_message, - ) - target = ( - startup_retryable_errors - if _retryable - else startup_nonretryable_errors - ) - target.append(f"{platform.value}: {adapter.fatal_error_message}") - # Queue for reconnection if the error is retryable - if _retryable: - self._failed_platforms[platform] = { - "config": platform_config, - "attempts": 1, - "next_retry": time.monotonic() + 30, - "credential_claim": self._adapter_credential_claim(platform, adapter), - "listener_claim": self._adapter_listener_claim(platform, adapter), - } - else: - self._update_platform_runtime_status( - platform.value, platform_state="retrying", error_code=None, error_message="failed to connect", - ) - startup_retryable_errors.append(f"{platform.value}: failed to connect") - # No fatal error info means likely a transient issue -- queue for retry - self._failed_platforms[platform] = { - "config": platform_config, - "attempts": 1, - "next_retry": time.monotonic() + 30, - "queued_at": time.monotonic(), - "credential_claim": self._adapter_credential_claim(platform, adapter), - "listener_claim": self._adapter_listener_claim(platform, adapter), - } - - if await self._abort_startup_if_shutdown_requested(): - return True - # Multi-profile multiplexing: bring up adapters for every OTHER profile this gateway serves. - # Each profile's adapters connect under that profile's home + credential scope and stamp - # their inbound events with the profile so the agent turn resolves correctly. - try: - _secondary_connected = await self._start_secondary_profile_adapters() - connected_count += _secondary_connected - except MultiplexConfigError as e: - # Invalid multiplexer config — abort startup cleanly so the operator - # fixes config.yaml rather than running a half-wired gateway. - reason = str(e) - logger.error("Gateway multiplexer config error: %s", reason) - _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) - self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE - self._request_clean_exit(reason) - self._startup_restore_in_progress = False - return True - except Exception as e: - logger.error("Secondary-profile adapter startup failed: %s", e, exc_info=True) - finally: - # Startup authority is one phase, not a persistent runner mode. - # From this point onward every adapter retry is non-evicting. - self._platform_lock_takeover_on_start = False - - # A platform skipped on the primary for a missing credential should have been picked up by - # a secondary profile owning the token. If none did, it is enabled in config.yaml yet - # silently unserved — surface it loudly instead of leaving a quiet dead channel. - for _skipped in _multiplex_skipped_platforms: - _served_by_secondary = any( - _skipped in _profile_map - for _profile_map in self._profile_adapters.values() - ) - if not _served_by_secondary: - logger.warning( - "%s is enabled but no profile (default or secondary) " - "provided a bot credential for it — the platform is not " - "being served. Add its token to the profile that should " - "own it, or disable the platform.", - _skipped.value, - ) - - if connected_count == 0: - if startup_nonretryable_errors and not startup_retryable_errors: - reason = "; ".join(startup_nonretryable_errors) - logger.error("Gateway hit a non-retryable startup conflict: %s", reason) - _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) - self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE - self._request_clean_exit(reason) - self._startup_restore_in_progress = False - return True - if startup_nonretryable_errors: - # Mixed failure mode: some platforms fatally misconfigured (e.g. WhatsApp never - # paired), others merely transient (e.g. Telegram TimedOut). Exiting 78 here would - # let exit-78 supervisors take the gateway PERMANENTLY down over a network blip and - # deny the retryable ones their retry. Log the fatal side loudly, then fall through to - # the degraded/retry path: the watcher recovers the retryable; the rest stay parked. - logger.error( - "%d platform(s) fatally misconfigured and parked: %s. " - "Staying alive so retryable platforms can recover.", - len(startup_nonretryable_errors), - "; ".join(startup_nonretryable_errors), - ) - if enabled_platform_count > 0: - if startup_retryable_errors: - # All enabled platforms hit retryable failures (network blip, bridge not paired, - # npm install timeout...). Keep the gateway alive so cron jobs still run and the - # reconnect watcher can recover the platforms once the cause is fixed; exiting - # here would turn one misconfigured platform into an infinite systemd restart loop. - reason = "; ".join(startup_retryable_errors) - logger.warning( - "Gateway started with no connected platforms — " - "%d platform(s) queued for retry: %s", - len(self._failed_platforms), reason, - ) - try: - from gateway.status import write_runtime_status - write_runtime_status( - gateway_state="degraded", - exit_reason=None, - ) - except Exception: - pass - # Fall through to the normal "running" state — reconnect watcher takes it from here. - # All enabled platforms had no adapter (missing library or credentials). Fleet nodes - # share one config.yaml but hold credentials for only a subset of platforms, so - # degrade gracefully and let cron jobs run. - logger.warning( - "No adapter could be created for any of the %d configured platform(s). " - "Check that required dependencies are installed and credentials are set. " - "Gateway will continue for cron job execution.", - enabled_platform_count, - ) - else: - logger.warning("No messaging platforms enabled.") - logger.info("Gateway will continue running for cron job execution.") - - # Update delivery router with adapters - if await self._abort_startup_if_shutdown_requested(): - return True - self.delivery_router.adapters = self.adapters - self._wire_teams_pipeline_runtime() - - self._running = True - self._install_plugin_message_injector() - self._update_runtime_status("running") - - try: - await self._ensure_hosted_room_worker() - except Exception: - logger.error( - "Group Chat worker failed to start; mutating Group Chat commands " - "will fail closed until supervision recovers it", - exc_info=True, - ) - self._spawn_supervised( - self._hosted_room_worker_watcher, - "hosted_room_worker", - ) - - self._start_loop_heartbeat_task() - - # Emit gateway:startup hook - hook_count = len(self.hooks.loaded_hooks) - if hook_count: - logger.info("%s hook(s) loaded", hook_count) - await self.hooks.emit("gateway:startup", { - "platforms": [p.value for p in self.adapters], - }) - - if connected_count > 0: - logger.info("Gateway running with %s platform(s)", connected_count) - - # Build initial channel directory for send_message name resolution - try: - from gateway.channel_directory import build_channel_directory - directory = await build_channel_directory(self.adapters) - ch_count = sum(len(chs) for chs in directory.get("platforms", {}).values()) - logger.info("Channel directory built: %d target(s)", ch_count) - except Exception as e: - logger.warning("Channel directory build failed: %s", e) - - # Check if we're restarting after a /update command. If the update is - # still running, keep watching so we notify once it actually finishes. - notified = await self._send_update_notification() - if not notified and any( - path.exists() - for path in ( - _hermes_home / ".update_pending.json", - _hermes_home / ".update_pending.claimed.json", - ) - ): - self._schedule_update_notification_watch() - - # Give freshly connected adapters a brief moment to settle before sending restart/startup - # lifecycle messages; in practice this helps Discord thread deliveries after reconnect. - if connected_count > 0: - await asyncio.sleep(1.0) - - # Notify the chat that initiated /restart that the gateway is back. - chat_restart_notification_pending = _restart_notification_pending() - planned_restart_notification_pending = _planned_restart_notification_pending() - # Capture, before _send_restart_notification() unlinks the marker, whether this process - # booted from a chat-originated /restart. One-shot signal for the /restart redelivery - # guard (_is_stale_restart_redelivery): a missing dedup marker only suppresses a /restart - # when we KNOW we just came out of a restart cycle. - if chat_restart_notification_pending: - self._booted_from_restart = True - # Restart notification, home-channel startup notice, and obligation redelivery all call - # adapter.send(). Those sends must not pin the inbound restore gate — a Telegram flood- - # control sleep on this path froze every platform for the full penalty. - await self._await_startup_boot_sends( - planned_restart_notification_pending=planned_restart_notification_pending, - ) - - # Auto-continue fresh sessions interrupted by the previous restart/shutdown. resume_pending - # is cleared by the normal successful-turn path, so a failed auto-resume stays visible on the - # next user message. _await_startup_boot_sends already cleared sessions answered in the ledger. - self._schedule_resume_pending_sessions() - await self._finish_startup_restore() - - # Surface state.db init failures to the user's messaging platforms - # so they know persistence is broken before losing data (#88235). - await self._send_session_db_warning_notifications() - - # Drain any recovered process watchers (from crash recovery checkpoint) - try: - from tools.process_registry import process_registry - # Detach the current batch atomically: reassigning to a fresh list takes ownership of - # exactly the watchers present now, so any watcher appended concurrently during the - # yield below isn't silently dropped by a clear() on the shared list. - watchers = process_registry.pending_watchers - process_registry.pending_watchers = [] - # Process in batches of 100 with event-loop yield points to avoid - # O(n^2) event-loop blocking when recovering thousands of watchers. - for i, watcher in enumerate(watchers): - self._spawn_supervised( - lambda w=watcher: self._run_process_watcher(w), - f"process_watcher:{watcher.get('session_id')}", - restart=False, - ) - logger.info("Resumed watcher for recovered process %s", watcher.get("session_id")) - if i % 100 == 99: - await asyncio.sleep(0) - except Exception as e: - logger.error("Recovered watcher setup error: %s", e) - - # Start background session expiry watcher to finalize expired sessions - self._spawn_supervised(self._session_expiry_watcher, "session_expiry_watcher") - - # Keep the /model picker's remote catalogs (curated manifest, OpenRouter live list, Nous - # Portal recommendations) warm on disk so a delisted or newly-published model reaches the - # picker within one TTL window (model_catalog.ttl_minutes, default 20) without a cold open. - self._spawn_supervised(self._model_catalog_refresh_watcher, "model_catalog_refresh_watcher") - - # Stall watchdog: pending inbound + stale agent activity → warn user - # to /new (does not kill the turn; see agent.session_stall_timeout). - self._spawn_supervised(self._session_stall_watcher, "session_stall_watcher") - - # Start the kanban notifier — each gateway delivers events for subscriptions owned by the - # profiles whose adapters it hosts, even when another gateway owns the single dispatcher. - self._spawn_supervised(self._kanban_notifier_watcher, "kanban_notifier_watcher") - - # Start background kanban dispatcher — spawns workers for ready tasks. Gated by - # `kanban.dispatch_in_gateway` (default True). When false, users run `hermes kanban daemon` - # externally or simply don't use kanban; this loop becomes a no-op. - self._spawn_supervised(self._kanban_dispatcher_watcher, "kanban_dispatcher_watcher") - - # Start background reconnection watcher for platforms that failed at startup - if self._failed_platforms: - logger.info( - "Starting reconnection watcher for %d failed platform(s): %s", - len(self._failed_platforms), - ", ".join(p.value for p in self._failed_platforms), - ) - # Track the reconnect watcher task so _ensure_reconnect_watcher_running can detect death - # and respawn it. Spawned via _spawn_supervised so an exception escaping the watcher's OUTER - # loop is caught, logged, and restarted with backoff instead of silently killing it (else a - # platform already queued in _failed_platforms stays stranded: the ensure hook only runs on - # a NEW fatal-error arrival). ``on_spawn`` keeps ``_reconnect_watcher_task`` on the CURRENT - # live task across backoff respawns so a superseded handle never looks like a dead watcher. - self._spawn_reconnect_watcher() - - # Start background handoff watcher — picks up CLI sessions marked handoff_state='pending' in - # state.db and re-binds them to the destination platform's home channel, then forges a - # synthetic user turn so the agent kicks off the new chat. - self._spawn_supervised(self._handoff_watcher, "handoff_watcher") - - # Async-delegation watcher: drains delegate_task(background=true) completions and injects - # each result into its originating session as a new turn (covers the idle, no-turn case). - self._spawn_supervised(self._async_delegation_watcher, "async_delegation_watcher") - - # /loop wakeup watcher: scans persisted loops (SessionDB loop:* rows) and injects due - # wakeup prompts into their originating chats while the session is idle. - self._spawn_supervised(self._loop_wakeup_watcher, "loop_wakeup_watcher") - - # Start the scale-to-zero idle watcher ONLY when opted in (HERMES_SCALE_TO_ZERO stamp), - # messaging is relay-only/absent, and a wakeUrl is registered. When armed it drives the relay - # dormant on sustained idle, then suspends via flaps — Fly autostop is inbound-only, job-blind. - try: - if self._scale_to_zero_should_arm(): - logger.info( - "scale-to-zero: armed (idle timeout %.0fs) — watching for idle", - self._scale_to_zero_idle_timeout_seconds(), - ) - self._spawn_supervised(self._scale_to_zero_watcher, "scale_to_zero_watcher") - else: - # Surface WHY an OPTED-IN instance didn't arm (non-opted not arming is normal — - # stay silent); otherwise a failed arm is invisible and needs a box-dive. - self._log_scale_to_zero_not_armed_reason() - except Exception: # noqa: BLE001 - arming must never block startup - logger.debug("scale-to-zero: arm check failed at startup", exc_info=True) - - # Drain-control watcher: reconciles the gateway's new-turn accept-state with the external - # ``.drain_request.json`` marker the dashboard begin/cancel-drain endpoint writes. A marker - # from a prior instantiation (durable-volume restart) is ignored via its epoch. - self._spawn_supervised(self._drain_control_watcher, "drain_control_watcher") - - logger.info("Press Ctrl+C to stop") - - return True _MAX_SUPERVISED_RESTARTS = 5 # A task that ran at least this long before crashing is HEALTHY: an isolated crash, not a # crash-loop; the consecutive-restart counter resets so a long-lived daemon isn't abandoned. _SUPERVISED_HEALTHY_SECS = 300 - @staticmethod - def _supervised_backoff(attempt: int) -> float: - """Delay before the supervisor's next respawn, in seconds (capped exponential). - - A method so tests can collapse the schedule instead of sleeping through the real curve. - """ - return min(60, 2 ** min(attempt, 6)) - - def _spawn_supervised( - self, coro_factory, name, *, restart=True, _attempt=0, on_spawn=None, - on_give_up=None, - ): - """Launch a long-lived background task with task-level supervision. - - Catches what a per-iteration try/except cannot — exceptions in the OUTER loop or pre-try - setup — which a bare ``asyncio.create_task`` drops silently. Restarts with capped backoff up - to ``_MAX_SUPERVISED_RESTARTS`` rapid failures; the counter resets after a run healthy for - ``_SUPERVISED_HEALTHY_SECS``. Each spawn uses a fresh ``Context``: an inherited - delegated-child marker would make the Kanban dispatcher reject its own writes. - ``on_spawn`` fires on EVERY spawn incl. respawns; callers tracking the handle elsewhere - (e.g. ``_reconnect_watcher_task``) MUST pass it or a respawn leaves a stale handle and a - SECOND watcher. ``on_give_up(name)`` fires when the restart budget is spent. - """ - if getattr(self, "_background_tasks", None) is None: - self._background_tasks = set() - - # Monotonic spawn timestamp captured per spawn: the ``_done`` callback - # uses it to distinguish a rapid crash-loop from a healthy-run-then-crash. - _started = time.monotonic() - - # Deliberately no kwargs to create_task (some test doubles mock a narrow signature); calling - # it from a fresh Context gives the same isolation as create_task(..., context=Context()). - task = Context().run(lambda: asyncio.create_task(coro_factory())) - # PERMANENT supervised watcher, not transient background WORK: the scale-to-zero idle check - # must ignore process-lifetime watchers or the gateway counts itself busy forever. Transient - # tasks added to _background_tasks elsewhere (startup-resume events etc.) stay counted. - task._hermes_supervised_watcher = True # type: ignore[attr-defined] - self._background_tasks.add(task) - if on_spawn is not None: - # Record the live handle NOW so an external tracker (e.g. _reconnect_watcher_task) - # points at the current task, not a dead one left by a prior supervised respawn. - try: - on_spawn(task) - except Exception: # pragma: no cover - defensive; a tracker must never kill the spawn - logger.debug("on_spawn callback for %s raised", name, exc_info=True) - - def _done(t): - self._background_tasks.discard(t) - if t.cancelled(): - return - exc = t.exception() - if exc is None: - # Clean return == deliberate shutdown or a self-disabling watcher (e.g. a gated - # no-op returning at once); respawning would busy-spin it — NEVER restart on it. - return - logger.error("Supervised task %s died: %r", name, exc, exc_info=exc) - if restart and self._running: - ran_for = time.monotonic() - _started - if ran_for >= self._SUPERVISED_HEALTHY_SECS: - # Ran healthily before crashing — a FRESH failure, not a rapid crash-loop. Reset - # the counter so a daemon crashing a few times over days is never abandoned. - effective_attempt = 0 - else: - effective_attempt = _attempt - if effective_attempt >= self._MAX_SUPERVISED_RESTARTS: - logger.error( - "Supervised task %s died %d times in rapid succession " - "(each within %ds of restart) — giving up restarts", - name, - effective_attempt, - self._SUPERVISED_HEALTHY_SECS, - ) - if on_give_up is not None: - try: - on_give_up(name) - except Exception: # pragma: no cover - defensive - logger.debug( - "on_give_up callback for %s raised", - name, exc_info=True, - ) - return - backoff = self._supervised_backoff(effective_attempt) - - async def _respawn(): - await asyncio.sleep(backoff) - if self._running: - self._spawn_supervised( - coro_factory, - name, - restart=restart, - _attempt=effective_attempt + 1, - on_spawn=on_spawn, - # Threaded through the recursion like on_spawn: only the LAST respawn's give-up - # matters, and dropping the callback leaves the exhaustion branch with no owner. - on_give_up=on_give_up, - ) - - # The done callback retains its registration context, so isolate the backoff task - # too; otherwise a restart could reintroduce the original caller's turn scope. - respawn_task = Context().run(lambda: asyncio.create_task(_respawn())) - self._background_tasks.add(respawn_task) - respawn_task.add_done_callback(self._background_tasks.discard) - - task.add_done_callback(_done) - return task - - async def _handoff_watcher( - self, interval: float = 2.0, drain_timeout: float = 30.0, - ) -> None: - """Background task that processes pending CLI→gateway session handoffs. - - Polls ``state.db`` for ``handoff_state='pending'`` rows: claim atomically (pending → - running), re-bind the home channel's session_key to the CLI session_id via - ``switch_session``, dispatch a synthetic ``MessageEvent``, mark ``completed``/``failed``. - """ - # Initial delay so the gateway is fully connected to its platforms - # before we try to dispatch handoffs through them. - await asyncio.sleep(5) - - # Does _process_handoff accept the profile argument? The real one does; test stand-ins bind - # a one-parameter callable. Probed once, outside the loop. - try: - import inspect as _inspect - _process_takes_profile = len( - _inspect.signature(self._process_handoff).parameters - ) >= 2 - except Exception: - _process_takes_profile = False - - # In-flight dispatches keyed by session id. A handoff runs a FULL agent turn plus delivery - # (far longer than the CLI's 60s wait); inline processing would let one slow handoff block - # every other profile's poll and time them out. Fire-and-forget; the poll loop only claims. - inflight: Dict[str, "asyncio.Task"] = {} - - async def _dispatch(row, session_id, session_db, profile_name) -> None: - """Run one claimed handoff to a terminal state, off the poll path.""" - try: - if _process_takes_profile: - await self._process_handoff(row, profile_name) - else: - await self._process_handoff(row) - await session_db.complete_handoff(session_id) - except asyncio.CancelledError: - # Gateway shutting down: leave the row 'running' so the next - # start's reclaim marks it failed with a clear reason. - raise - except Exception as exc: - logger.warning( - "Handoff for session %s failed: %s", - session_id, exc, exc_info=True, - ) - try: - await session_db.fail_handoff(session_id, str(exc)) - except Exception: - logger.debug("Could not record handoff failure", exc_info=True) - finally: - inflight.pop(session_id, None) - - async def _tick(profile_name: Optional[str] = None) -> None: - """One poll of the CURRENTLY-SCOPED session store. - - A closure over ``self``, not a method: unit tests bind ``_handoff_watcher`` onto a - ``SimpleNamespace`` exposing only ``_session_db``, ``_running`` and ``_process_handoff``; - any other ``self.`` would raise, be swallowed by the loop, and silently no-op the - watcher. ``profile_name`` (``None`` = root) makes delivery use that profile's OWN adapter. - """ - session_db = getattr(self, "_session_db", None) - if session_db is None: - return - pending = await session_db.list_pending_handoffs() - for row in pending: - session_id = row.get("id") - if not session_id or session_id in inflight: - continue - if not await session_db.claim_handoff(session_id): - # Another tick or another gateway already claimed it. - continue - # Positional, not keyword: tests bind a one-arg ``_process_handoff(row)`` stand-in and a - # keyword call would TypeError into the failure branch (arity probed above). - # INVARIANT (do not weaken): this task is created inside _profile_runtime_scope but - # typically RUNS after it exits; it sees the profile's home/secret scope only because - # those seams are ContextVar-based and ensure_future copies the Context. - inflight[session_id] = asyncio.ensure_future( - _dispatch(row, session_id, session_db, profile_name) - ) - - # A row still 'running' at startup belongs to a gateway that died mid-dispatch: it can never - # reach a terminal state, and request_handoff refuses new requests while it sits there. - for _pname, _phome in _handoff_watch_scopes(self): - try: - if _phome is None: - await _reclaim_stale(self) - else: - async with _async_profile_runtime_scope(_phome): - await _reclaim_stale(self) - except Exception: - logger.debug("Stale-handoff reclaim failed", exc_info=True) - - try: - while self._running: - try: - for profile_name, profile_home in _handoff_watch_scopes(self): - if profile_home is None: - await _tick(profile_name) - else: - async with _async_profile_runtime_scope(profile_home): - await _tick(profile_name) - except asyncio.CancelledError: - raise - except Exception as exc: - logger.debug("Handoff watcher tick error: %s", exc, exc_info=True) - await asyncio.sleep(interval) - finally: - # Drain in-flight dispatches before returning: cancelling would strand their rows in - # 'running'; a bounded grace period lets an almost-done handoff record its own state. - pending_tasks = [t for t in inflight.values() if not t.done()] - if pending_tasks: - try: - await asyncio.wait(pending_tasks, timeout=drain_timeout) - except Exception: - logger.debug("Handoff drain raised", exc_info=True) - for task in pending_tasks: - if not task.done(): - task.cancel() - - async def _process_handoff( - self, row: Dict[str, Any], profile_name: Optional[str] = None, - ) -> None: - """Execute one handoff row. Raises on failure (caller marks failed). - - ``profile_name`` (``None`` = root) is the profile whose store queued this handoff. Under - multiplex it is load-bearing: ``self.adapters``/``self.config`` are the primary's (secondaries - live in ``_profile_adapters``), and the session key must be namespaced ``agent::...`` - or it binds a key nobody reads. Passing the name beats re-deriving it from the contextvar. - """ - from gateway.config import Platform - from gateway.session import SessionSource, build_session_key - from gateway.platforms.base import MessageEvent - - cli_session_id = row["id"] - platform_name = (row.get("handoff_platform") or "").strip().lower() - if not platform_name: - raise RuntimeError("handoff_platform is empty") - - # Resolve platform enum - try: - platform = Platform(platform_name) - except (ValueError, KeyError): - raise RuntimeError(f"unknown platform '{platform_name}'") - - # Resolve the config + adapter map for the profile that queued this handoff; single-profile - # gateways (or a default-profile handoff) fall back to self.config/self.adapters. - handoff_config = self.config - handoff_adapters = self.adapters - if profile_name and profile_name != "default": - secondary = (self._profile_adapters or {}).get(profile_name) - if not secondary: - raise RuntimeError( - f"profile '{profile_name}' has no live adapters in this gateway" - ) - handoff_adapters = secondary - # The watcher already entered _profile_runtime_scope, so a fresh load resolves THIS - # profile's config. Fail closed — self.config would deliver to the WRONG chat. - try: - handoff_config = load_gateway_config() - except Exception as exc: - logger.error( - "Handoff: could not load config for profile %s; " - "failing the handoff instead of delivering via the " - "primary's config", - profile_name, exc_info=True, - ) - raise RuntimeError( - f"could not load config for profile '{profile_name}': {exc}" - ) from exc - - # Adapter must be live. A relay-fronted gateway registers ONE adapter under Platform.RELAY - # fronting N logical platforms, so a literal adapters.get(discord) misses a deliverable - # platform; resolve_delivery_transport is the alias-aware resolver (native adapter wins). - transport = resolve_delivery_transport(platform, handoff_config, handoff_adapters) - if not transport: - raise RuntimeError( - f"platform '{platform_name}' is not active in this gateway" - ) - adapter = transport.adapter - - # Home channel must be configured - home = handoff_config.get_home_channel(platform) - if not home or not home.chat_id: - raise RuntimeError( - f"no home channel configured for {platform_name}; " - f"run /sethome on the desired chat first" - ) - - cli_title = row.get("title") or cli_session_id[:8] - - # Create a fresh thread on the destination so the handoff has its own scrollback. Adapter - # returns None if threading is unsupported (Matrix/WhatsApp/Signal/SMS) or creation failed. - thread_name = f"Hermes — {cli_title}" - try: - new_thread_id = await adapter.create_handoff_thread( - str(home.chat_id), thread_name, - ) - except Exception as exc: - logger.debug( - "Handoff: create_handoff_thread raised on %s: %s", - platform_name, exc, exc_info=True, - ) - new_thread_id = None - - effective_thread_id = new_thread_id or ( - str(home.thread_id) if home.thread_id else None - ) - - # Telegram private-chat DM topics are shaped differently from group/forum threads by the - # inbound adapter: a handoff-created topic in a positive chat_id must use the DM-topic source - # shape, or the synthetic turn binds a `thread` key while real replies arrive on a `dm` key. - home_chat_id = str(home.chat_id) - is_telegram_private_chat = ( - platform == Platform.TELEGRAM - and looks_like_telegram_private_chat_id(home_chat_id) - ) - - if new_thread_id and not is_telegram_private_chat: - dest_chat_type = "thread" - dest_user_id = "system:handoff" - else: - # No thread — assume DM-style. For Telegram private-chat topics use the real user id - # (== chat_id) so topic-mode checks and binding persistence match later inbound turns. - dest_chat_type = "dm" - dest_user_id = home_chat_id if is_telegram_private_chat else "system:handoff" - - # Discord (unlike Slack/Telegram) builds in-thread messages with ``chat_id == thread id``, - # so key on the thread's OWN id; keying on the parent would make the next reply spawn anew. - if platform == Platform.DISCORD and dest_chat_type == "thread" and effective_thread_id: - dest_chat_id = str(effective_thread_id) - else: - dest_chat_id = home_chat_id - dest_source = SessionSource( - platform=platform, - chat_id=dest_chat_id, - chat_name=home.name, - chat_type=dest_chat_type, - user_id=dest_user_id, - user_name="Handoff", - thread_id=effective_thread_id, - profile=profile_name, - ) - - # Build the session_key with the adapters' own rules so switch_session hits the right entry. - # Thread keys omit user_id (thread_sessions_per_user default) so the next message shares it. - platform_cfg = handoff_config.platforms.get(platform) - extra = platform_cfg.extra if platform_cfg else {} - # Namespace the key to the queuing profile: a multiplexed gateway would otherwise build - # ``agent:main:...`` while the profile's adapter routes inbound on ``agent::...``. - # The resolver is only the root fallback (None when multiplexing is off; old key unchanged). - # The isinstance check is load-bearing: a Mock store returns a truthy MagicMock. - handoff_profile = profile_name if (profile_name and profile_name != "default") else None - if handoff_profile is None: - try: - store = getattr(self.async_session_store, "_store", self.async_session_store) - resolver = getattr(store, "_resolve_profile_for_key", None) - if callable(resolver): - resolved = resolver(dest_source) - if isinstance(resolved, str) and resolved.strip(): - handoff_profile = resolved - except Exception: - logger.debug("Handoff: could not resolve profile namespace", exc_info=True) - session_key = build_session_key( - dest_source, - group_sessions_per_user=extra.get("group_sessions_per_user", True), - thread_sessions_per_user=extra.get("thread_sessions_per_user", False), - profile=handoff_profile, - ) - - # Ensure a session_store entry exists for this key (get_or_create_session creates one for a - # never-used home channel); switch_session then re-points it. - await self.async_session_store.get_or_create_session(dest_source) - - # Re-bind the destination key to the CLI session_id: switch_session ends the prior session - # in SQLite and reopens the CLI session under the new key; its transcript is now active. - switched = await self.async_session_store.switch_session(session_key, cli_session_id) - if switched is None: - raise RuntimeError( - f"could not switch session key {session_key} → {cli_session_id}" - ) - - # Evict any cached AIAgent for this session_key so the next dispatch - # rebuilds it against the CLI session_id (mirrors /resume / /branch). - self._evict_cached_agent(session_key) - - # Cancel any in-flight running-agent state for the destination key - # so the synthetic turn isn't queued behind a stale running flag. - self._release_running_agent_state(session_key) - - synthetic_text = ( - f"[Session was just handed off from CLI (\"{cli_title}\") to this " - f"channel. The full prior conversation history is loaded above. " - f"Briefly confirm you're working here and summarize what we were " - f"working on, so the user can continue from this device.]" - ) - - synthetic_event = MessageEvent( - text=synthetic_text, - source=dest_source, - internal=True, - ) - - logger.info( - "Handoff: dispatching synthetic turn for CLI session %s → %s " - "(home=%s, thread=%s, session_key=%s)", - cli_session_id, platform_name, home.chat_id, effective_thread_id, - session_key, - ) - - # Dispatch through the runner directly: adapter.handle_message would spawn a background task - # and lose error visibility; inline _handle_message keeps success/failure observable. - response_text = await self._handle_message(synthetic_event) - if not response_text: - # Streaming may have already delivered the response inline. - # Either way, agent ran without raising — count as success. - return - - # Send the reply to the new thread if we created one, else the configured home channel - # (which may carry a thread_id). Use the resolved transport (not adapter.send) so a - # relay-fronted logical platform is stamped on the outbound frame (send_for_platform). - send_metadata: Dict[str, Any] = {} - if effective_thread_id: - send_metadata["thread_id"] = effective_thread_id - try: - result = await transport.send( - platform, - str(home.chat_id), - response_text, - send_metadata or None, - ) - except Exception as exc: - raise RuntimeError(f"adapter.send failed: {exc}") from exc - - if not getattr(result, "success", True): - err = getattr(result, "error", "send returned success=False") - raise RuntimeError(f"adapter.send failed: {err}") - - async def _session_expiry_watcher(self, interval: int = 300): - """Background task that finalizes expired sessions: runs ``on_session_finalize`` hooks, - cleans up the cached agent's tool resources, evicts the cache entry, and marks the session - finalized so it is not finalized again. - """ - await asyncio.sleep(60) # initial delay — let the gateway fully start - _finalize_failures: dict[str, int] = {} # session_id -> consecutive failure count - _MAX_FINALIZE_RETRIES = 3 - while self._running: - try: - await self.async_session_store._ensure_loaded() - # Collect expired sessions first, then log a single summary. - _expired_entries = [] - for key, entry in list(self.session_store._entries.items()): - if entry.expiry_finalized: - continue - if not await self.async_session_store._is_session_expired(entry): - continue - _expired_entries.append((key, entry)) - - if _expired_entries: - # Extract platform names from session keys for a compact summary. - # Keys look like "agent:main:telegram:dm:12345" — platform is field [2]. - _platforms: dict[str, int] = {} - for _k, _e in _expired_entries: - _parts = _k.split(":") - _plat = _parts[2] if len(_parts) > 2 else "unknown" - _platforms[_plat] = _platforms.get(_plat, 0) + 1 - _plat_summary = ", ".join( - f"{p}:{c}" for p, c in sorted(_platforms.items()) - ) - logger.info( - "Session expiry: %d sessions to finalize (%s)", - len(_expired_entries), _plat_summary, - ) - - for key, entry in _expired_entries: - try: - try: - _parts = key.split(":") - _platform = _parts[2] if len(_parts) > 2 else "" - # Off-loop + bounded: plugin finalize hooks can block arbitrarily, and - # this watcher runs on the gateway event loop. - await self._finalize_session_off_loop( - session_id=entry.session_id, - platform=_platform, - reason="session_expired", - ) - except Exception: - pass - # Close the cached agent's memory provider and tool resources. Idle agents - # live in _agent_cache (not _running_agents), so look there. - _cached_agent = None - _cache_lock = getattr(self, "_agent_cache_lock", None) - if _cache_lock is not None: - with _cache_lock: - _cached = self._agent_cache.get(key) - _cached_agent = _cached[0] if isinstance(_cached, tuple) else _cached if _cached else None - # Fall back to _running_agents in case the agent is - # still mid-turn when the expiry fires. - if _cached_agent is None: - _exp_state = self._peek_session_state(key) - _cached_agent = _exp_state.turn.agent if _exp_state else None - if _cached_agent and _cached_agent is not _AGENT_PENDING_SENTINEL: - await self._cleanup_agent_resources_off_loop( - _cached_agent, context="session expiry" - ) - # Drop the cache entry so the AIAgent (LLM clients, tool schemas, memory - # provider refs) can be GC'd; otherwise the cache grows unbounded. - self._evict_cached_agent(key) - # Permanent finalization: one funnel call drops every conversation-scoped - # dict AND boundary security state so they don't grow unbounded. Idle - # agent-cache eviction must NOT do this — that session is still alive and a - # resumed turn rebuilds from these overrides. Only finalize, /new, /reset clear. - self._clear_conversation_scope( - key, reason="expiry_finalized" - ) - # Persist finalized flag (sessions.json AND state.db, single write-path); - # also drops the /model override — finalization is a conversation boundary. - await self.async_session_store.set_expiry_finalized(entry) - logger.debug( - "Session expiry finalized for %s", - entry.session_id, - ) - _finalize_failures.pop(entry.session_id, None) - except Exception as e: - failures = _finalize_failures.get(entry.session_id, 0) + 1 - _finalize_failures[entry.session_id] = failures - if failures >= _MAX_FINALIZE_RETRIES: - logger.warning( - "Session finalize gave up after %d attempts for %s: %s. " - "Marking as finalized to prevent infinite retry loop.", - failures, entry.session_id, e, - ) - await self.async_session_store.set_expiry_finalized( - entry, clear_model_override=False - ) - _finalize_failures.pop(entry.session_id, None) - else: - logger.debug( - "Session finalize failed (%d/%d) for %s: %s", - failures, _MAX_FINALIZE_RETRIES, entry.session_id, e, - ) - - if _expired_entries: - _done = sum( - 1 for _, e in _expired_entries if e.expiry_finalized - ) - _failed = len(_expired_entries) - _done - if _failed: - logger.info( - "Session expiry done: %d finalized, %d pending retry", - _done, _failed, - ) - else: - logger.info( - "Session expiry done: %d finalized", _done, - ) - - # Sweep agents idle beyond the TTL regardless of session reset policy: sessions with - # long / "never" reset windows would otherwise pin memory for the gateway's life. - try: - _idle_evicted = self._sweep_idle_cached_agents() - if _idle_evicted: - logger.info( - "Agent cache idle sweep: evicted %d agent(s)", - _idle_evicted, - ) - except Exception as _e: - logger.debug("Idle agent sweep failed: %s", _e) - - # Neither LRU cap nor idle TTL knows what a cached transcript costs in memory, so a - # busy gateway keeps every warm session's tool output resident until the RSS limit. - try: - self._sweep_agent_cache_under_pressure() - except Exception as _e: - logger.debug("Agent cache pressure sweep failed: %s", _e) - - # Prune stale SessionStore entries; the in-memory dict (and sessions.json) would - # otherwise grow unbounded with many rotating chats / threads / users. - _last_prune_ts = getattr(self, "_last_session_store_prune_ts", 0.0) - _prune_interval = 3600.0 # once per hour - if time.time() - _last_prune_ts > _prune_interval: - try: - _max_age = int( - getattr(self.config, "session_store_max_age_days", 0) or 0 - ) - if _max_age > 0: - _pruned = await self.async_session_store.prune_old_entries(_max_age) - if _pruned: - logger.info( - "SessionStore prune: dropped %d stale entries", - _pruned, - ) - except Exception as _e: - logger.debug("SessionStore prune failed: %s", _e) - self._last_session_store_prune_ts = time.time() - except Exception as e: - logger.debug("Session expiry watcher error: %s", e) - # Sleep in small increments so we can stop quickly - for _ in range(interval): - if not self._running: - break - await asyncio.sleep(1) - - def _session_stall_timeout_seconds(self) -> float: - """Return configured stall timeout (seconds); 0 disables the watchdog.""" - return _float_env("HERMES_SESSION_STALL_TIMEOUT", 300) - - def _iter_gateway_adapters(self): - """Yield every live platform adapter (default + multiplex profiles).""" - seen: set[int] = set() - for adapter in list(getattr(self, "adapters", {}).values()): - if adapter is None: - continue - aid = id(adapter) - if aid in seen: - continue - seen.add(aid) - yield adapter - for amap in list(getattr(self, "_profile_adapters", {}).values()): - for adapter in list(amap.values()): - if adapter is None: - continue - aid = id(adapter) - if aid in seen: - continue - seen.add(aid) - yield adapter - - def _session_activity_for_stall(self, session_key: str) -> Optional[dict]: - """Return the shared activity snapshot for stall progress: the single source is - ``AIAgent.get_activity_summary()`` / ``agent.session_activity``; no turn-start or - pending-inbound clocks. - """ - agent = (getattr(self, "_running_agents", None) or {}).get(session_key) - if agent is None or agent is _AGENT_PENDING_SENTINEL: - return None - if not hasattr(agent, "get_activity_summary"): - return None - try: - summary = agent.get_activity_summary() - except Exception: - return None - return summary if isinstance(summary, dict) else None - - async def _check_session_stalls(self, timeout_seconds: float) -> int: - """Scan pending inbound sessions and notify once per stall episode; returns the number of - notifications sent this pass (for tests). - """ - from gateway.session_stall import ( - format_session_stall_notification, - resolve_session_idle_seconds_from_activity, - should_clear_session_stall_notification, - should_emit_session_stall_notification, - ) - - notified_map = getattr(self, "_session_stall_notified", None) - if notified_map is None: - notified_map = {} - self._session_stall_notified = notified_map - - sent = 0 - now = time.time() - candidates: Dict[str, tuple[Any, Any]] = {} - - for adapter in self._iter_gateway_adapters(): - pending_slot = getattr(adapter, "_pending_messages", None) or {} - for session_key, event in list(pending_slot.items()): - if session_key and session_key not in candidates and event is not None: - candidates[session_key] = (adapter, event) - - for session_key, overflow in list( - (getattr(self, "_queued_events", None) or {}).items() - ): - if not session_key or session_key in candidates or not overflow: - continue - event = overflow[0] - source = getattr(event, "source", None) - adapter = ( - self._adapter_for_source(source) if source is not None else None - ) - if adapter is None: - continue - candidates[session_key] = (adapter, event) - - for session_key, (adapter, pending_event) in list(candidates.items()): - has_pending = pending_event is not None - activity = ( - self._session_activity_for_stall(session_key) if has_pending else None - ) - idle_seconds = ( - resolve_session_idle_seconds_from_activity(activity, now=now) - if has_pending - else None - ) - already = bool(notified_map.get(session_key)) - if should_clear_session_stall_notification( - timeout_seconds=timeout_seconds, - idle_seconds=idle_seconds, - has_pending_inbound=has_pending, - ): - notified_map.pop(session_key, None) - already = False - if not should_emit_session_stall_notification( - timeout_seconds=timeout_seconds, - idle_seconds=idle_seconds, - has_pending_inbound=has_pending, - already_notified=already, - ): - continue - - if idle_seconds is None: - continue - mins = max(1, int(idle_seconds // 60)) - activity = activity or {} - logger.warning( - "Session stall detected: session=%s idle=%.0fs " - "(timeout=%.0fs, ~%d min); pending inbound present " - "| last_activity=%s | provenance=%s " - "(agent.session_stall_timeout)", - session_key, - idle_seconds, - timeout_seconds, - mins, - activity.get("last_activity_desc") - or activity.get("last_activity_description") - or "unknown", - activity.get("provenance") - or activity.get("last_activity_provenance") - or "unknown", - ) - source = getattr(pending_event, "source", None) - chat_id = getattr(source, "chat_id", None) if source is not None else None - if not chat_id: - logger.warning( - "Session stall notify skipped (no chat_id): session=%s", - session_key, - ) - # Cannot deliver; latch to avoid log spam every tick. - notified_map[session_key] = True - continue - # Re-read pending state + activity IMMEDIATELY before delivery: the snapshot above ages - # while earlier candidates await sends; an agent that progressed (or drained its queue) - # must not get a false stall notice. Abort, latch un-set, so the next tick re-evaluates. - still_pending = ( - (getattr(adapter, "_pending_messages", None) or {}).get( - session_key - ) - is not None - or bool( - (getattr(self, "_queued_events", None) or {}).get( - session_key - ) - ) - ) - fresh_idle = resolve_session_idle_seconds_from_activity( - self._session_activity_for_stall(session_key), - now=time.time(), - ) - if not still_pending or ( - fresh_idle is not None and fresh_idle < timeout_seconds - ): - logger.info( - "Session stall notify aborted (no longer stale): " - "session=%s pending=%s fresh_idle=%s", - session_key, - still_pending, - fresh_idle, - ) - # Re-arm: drop any stale latch so a FUTURE genuine stall - # episode notifies again. - notified_map.pop(session_key, None) - continue - try: - metadata = ( - self._thread_metadata_for_source(source) - if source is not None and hasattr(self, "_thread_metadata_for_source") - else None - ) - # Bound the send: a wedged adapter transport (network hang, dead websocket) must not - # block the watcher pass — siblings would go unevaluated and the watcher stop. - try: - result = await asyncio.wait_for( - adapter.send( - str(chat_id), - format_session_stall_notification(idle_seconds), - metadata=metadata, - ), - timeout=_STALL_NOTIFY_SEND_TIMEOUT_SECONDS, - ) - except asyncio.TimeoutError: - logger.warning( - "Session stall notify send timed out after %.0fs " - "for %s; will retry next tick", - _STALL_NOTIFY_SEND_TIMEOUT_SECONDS, - session_key, - ) - continue # do not latch; retry next tick - # Adapters often return SendResult(success=False) instead of raising. - if result is not None and getattr(result, "success", True) is False: - logger.warning( - "Session stall notify failed for %s: %s", - session_key, - getattr(result, "error", "send returned success=False"), - ) - continue # do not latch; retry next tick - sent += 1 - notified_map[session_key] = True - except Exception as exc: - logger.warning( - "Session stall notify failed for %s: %s", - session_key, - exc, - ) - # Do not latch — retry next watcher tick until delivery or episode clear. - - # Drop latches for sessions that no longer appear in any pending map. - for key in list(notified_map.keys()): - if key not in candidates: - notified_map.pop(key, None) - - return sent - - async def _model_catalog_refresh_watcher(self) -> None: - """Refresh the /model picker's remote catalogs every TTL window. The picker itself only - refreshes on a cold/stale open, so if nobody opens ``/model`` the cache never updates. - """ - from hermes_cli.model_catalog import refresh_catalogs, refresh_interval_seconds - - await asyncio.sleep(30) # let startup settle - while self._running: - try: - await asyncio.to_thread(refresh_catalogs) - except Exception as exc: - logger.debug("Model catalog refresh failed: %s", exc) - try: - interval = refresh_interval_seconds() - except Exception: - interval = 1200.0 - deadline = time.monotonic() + interval - while self._running and time.monotonic() < deadline: - await asyncio.sleep(min(30.0, max(0.0, deadline - time.monotonic()))) - - async def _session_stall_watcher(self, interval: float = 30.0): - """Periodic pending-inbound + stale-activity stall watchdog. - - Progress comes only from ``get_activity_summary()``. Pending inbound is a notify policy - gate, not a progress clock. Notify-only: does not kill the turn (contrast - ``gateway_timeout`` / ``shutdown_watchdog``). - """ - # Short initial delay so startup reconnect noise does not false-fire. - await asyncio.sleep(min(30.0, max(1.0, float(interval)))) - while self._running: - try: - timeout = self._session_stall_timeout_seconds() - if timeout > 0: - await self._check_session_stalls(timeout) - except Exception as exc: - logger.debug("Session stall watcher error: %s", exc) - # Interruptible sleep - steps = max(1, int(float(interval))) - for _ in range(steps): - if not self._running: - break - await asyncio.sleep(1) def _active_profile_name(self) -> str: """Return the profile name this gateway represents.""" @@ -13631,1733 +4947,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew #: five-minute retries cannot keep a watcher alive, the fault is not transient — fail loudly. _MAX_SLOW_WATCHER_RESPAWNS = 6 - def _on_reconnect_watcher_gave_up(self, name: str = "") -> None: - """Own the reconnect invariant once supervision has abandoned it. - Invariant: while running and ``_failed_platforms`` is non-empty, a reconnect watcher is live - or a bounded respawn is scheduled. Event-coupled recovery is not enough: the failed adapter - is dropped from the live map, so no later event may ever arrive to notice a dead watcher. - Deliberately NOT done here: requesting a process restart when the slow tier is exhausted — - a blast-radius policy call; a single loud error names the still-queued platforms instead. - """ - if not getattr(self, "_running", False): - return - if not getattr(self, "_failed_platforms", None): - # No queued work depends on the watcher; leaving it dead is correct — the enqueue path - # spawns a fresh one the moment a platform is queued again. - logger.warning( - "Reconnect watcher supervision exhausted with an empty retry " - "queue — leaving it down until a platform is queued." - ) - return - self._schedule_slow_reconnect_watcher_respawn(attempt=0) - - def _schedule_slow_reconnect_watcher_respawn(self, *, attempt: int) -> None: - """Bounded slow-tier respawn of the reconnect watcher.""" - if attempt >= self._MAX_SLOW_WATCHER_RESPAWNS: - logger.error( - "Reconnect watcher could not be kept alive after %d slow " - "respawns; %d platform(s) remain queued and unattended: %s. " - "Manual intervention or a gateway restart is required.", - attempt, - len(self._failed_platforms), - ", ".join(str(p) for p in self._failed_platforms), - ) - return - - async def _slow_respawn() -> None: - await asyncio.sleep(self._RECONNECT_WATCHER_SLOW_RETRY_SECS) - if not getattr(self, "_running", False): - return - if not getattr(self, "_failed_platforms", None): - # The queue drained while we waited -- something else healed - # it. Nothing to own any more. - return - task = getattr(self, "_reconnect_watcher_task", None) - if task is not None and not task.done(): - return # a watcher came back on its own; stand down - logger.warning( - "Reconnect watcher still down with %d platform(s) queued — " - "slow respawn %d/%d", - len(self._failed_platforms), - attempt + 1, - self._MAX_SLOW_WATCHER_RESPAWNS, - ) - self._spawn_reconnect_watcher( - on_give_up=lambda _name: self._schedule_slow_reconnect_watcher_respawn( - attempt=attempt + 1 - ) - ) - - respawn_task = asyncio.create_task(_slow_respawn()) - if getattr(self, "_background_tasks", None) is None: - self._background_tasks = set() - self._background_tasks.add(respawn_task) - respawn_task.add_done_callback(self._background_tasks.discard) - - def _spawn_reconnect_watcher(self, *, on_give_up=None): - """Single place that knows how to launch the reconnect watcher. - - ``on_spawn`` is load-bearing: without it the supervisor's own respawn leaves - ``_reconnect_watcher_task`` at a dead handle and ``_ensure_...`` spawns a second watcher. - """ - self._reconnect_watcher_task = self._spawn_supervised( - self._platform_reconnect_watcher, - "platform_reconnect_watcher", - on_spawn=lambda t: setattr(self, "_reconnect_watcher_task", t), - on_give_up=on_give_up or self._on_reconnect_watcher_gave_up, - ) - return self._reconnect_watcher_task - - def _ensure_reconnect_watcher_running(self) -> None: - """Ensure the platform reconnect watcher background task is alive. - - Respawns a dead watcher (exhausted restart budget, unrecoverable exception) so queued - platforms are not stranded. Called on BOTH _queue_retryable_fatal_platform paths: the - re-fatal of an already-queued platform is the only case where the budget can be exhausted. - """ - if not getattr(self, "_running", False): - return - task = getattr(self, "_reconnect_watcher_task", None) - if task is not None and not task.done(): - return # already alive - logger.warning( - "Reconnect watcher task is dead (done=%s) — respawning", - task.done() if task is not None else "N/A", - ) - self._spawn_reconnect_watcher() - - async def _platform_reconnect_watcher(self) -> None: - """Background task that periodically retries connecting failed platforms. - - Exponential backoff 30s → 300s cap; retryable failures (network/DNS) retry at the cap - indefinitely so transient outages self-heal, non-retryable (bad auth) drop out immediately. - The circuit breaker (``/platform pause``) is manual only — auto-pausing left bots dead. - """ - await asyncio.sleep(10) # initial delay — let startup finish - while self._running: - if not self._failed_platforms: - # Nothing to reconnect — sleep and check again - for _ in range(30): - if not self._running: - return - if self._failed_platforms: - break - await asyncio.sleep(1) - continue - - now = time.monotonic() - for platform in list(self._failed_platforms.keys()): - if not self._running: - return - info = self._failed_platforms.get(platform) - if info is None: - # Removed concurrently (/platform resume, reconnect via another path) between - # the snapshot above and this lookup — not an error, nothing to do this pass. - continue - # Skip paused platforms entirely — they need explicit - # /platform resume to come back. - if info.get("paused"): - continue - # Long-lived retry escalation: past the attention threshold flag the platform - # NEEDS_ATTENTION in runtime status so a dead token/revoked intent doesn't look - # like ordinary "retrying" forever. A signal, NOT a circuit breaker — retries continue. - if not info.get("attention_flagged") and _reconnect_needs_attention(info, now): - info["attention_flagged"] = True - queued_for = now - info.get("queued_at", now) - retrying_since_iso = ( - datetime.now(timezone.utc) - timedelta(seconds=queued_for) - ).isoformat() - logger.warning( - "%s has been failing/reconnecting continuously for " - "%.1f hours (%d attempts) — flagging NEEDS_ATTENTION. " - "Retries continue, but this usually means a permanent " - "problem (revoked credentials, missing intents, broken " - "sidecar). Check `hermes status` / `/platform list`.", - platform.value, - queued_for / 3600.0, - info.get("attempts", 0), - ) - self._update_platform_runtime_status( - platform.value, - platform_state="retrying", - needs_attention=True, - retrying_since=retrying_since_iso, - ) - if now < info["next_retry"]: - continue # not time yet - - platform_config = info["config"] - attempt = info["attempts"] + 1 - # Empty-token primary configs can never reconnect; drop them so multiplex setups - # where a secondary profile owns the bot do not spin forever. - if not _platform_has_bot_credential(platform, platform_config): - logger.warning( - "Reconnect %s: no bot credential on queued config, " - "removing from retry queue", - platform.value, - ) - del self._failed_platforms[platform] - continue - logger.info( - "Reconnecting %s (attempt %d)...", - platform.value, attempt, - ) - - adapter = None - try: - adapter = self._create_adapter(platform, platform_config) - if not adapter: - logger.warning( - "Reconnect %s: adapter creation returned None, removing from retry queue", - platform.value, - ) - del self._failed_platforms[platform] - continue - - adapter.set_message_handler(self._primary_message_handler()) - adapter.set_fatal_error_handler(self._handle_adapter_fatal_error) - adapter.set_session_store(self.session_store) - adapter.set_busy_session_handler(self._handle_active_session_busy_message) - _set_reaction = getattr(adapter, "set_reaction_handler", None) - if callable(_set_reaction): - _set_reaction(self._handle_reaction_event) - adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) - adapter.set_authorization_check(self._make_adapter_auth_check(adapter.platform)) - adapter.set_platform_event_handler(self._primary_platform_event_handler()) - adapter._busy_text_mode = self._busy_text_mode - - # Reconnect after outage: keep the platform's server-side update queue so - # messages sent while the bot was offline are delivered rather than dropped. - success = await self._connect_adapter_with_timeout( - adapter, platform, is_reconnect=True - ) - if success: - self.adapters[platform] = adapter - self._sync_voice_mode_state_to_adapter(adapter) - # Wire voice input callback on reconnect as well (#60623). - self._bind_voice_input_callback(adapter) - self.delivery_router.adapters = self.adapters - del self._failed_platforms[platform] - self._update_platform_runtime_status( - platform.value, - platform_state="connected", - error_code=None, - error_message=None, - needs_attention=False, - retrying_since=None, - ) - logger.info("✓ %s reconnected successfully", platform.value) - - # Final responses rejected while this adapter was down are still owned by - # this live process, so startup recovery cannot claim them. Replay the - # explicitly transient subset now that the platform is usable. - try: - await self._redeliver_failed_obligations_for_platform( - platform - ) - except Exception: - logger.debug( - "failed-obligation redelivery after %s reconnect failed", - platform.value, - exc_info=True, - ) - - # Rebuild channel directory with the new adapter - try: - from gateway.channel_directory import build_channel_directory - await build_channel_directory(self.adapters) - except Exception: - pass - - # A platform that was offline at gateway startup never got its restart- - # interrupted sessions auto-resumed — the startup pass skips sessions whose - # adapter isn't connected yet. - try: - self._schedule_resume_pending_sessions(platform=platform) - except Exception: - logger.debug( - "resume-pending reschedule after %s reconnect failed", - platform.value, - exc_info=True, - ) - # Check if the failure is non-retryable - elif adapter.has_fatal_error and not adapter.fatal_error_retryable: - self._update_platform_runtime_status( - platform.value, - platform_state="fatal", - error_code=adapter.fatal_error_code, - error_message=adapter.fatal_error_message, - ) - logger.warning( - "Reconnect %s: non-retryable error (%s), removing from retry queue", - platform.value, adapter.fatal_error_message, - ) - # The adapter is about to be dropped from the queue without ever being - # installed on self.adapters, so nothing else will call disconnect() on it. - # Dispose here or the resource owners built in __init__ (ResponseStore etc.) - # leak ~2 fds each; at the 300s cap the gateway hits the fd limit in ~12h. - await _dispose_unused_adapter(adapter) - del self._failed_platforms[platform] - else: - self._update_platform_runtime_status( - platform.value, - platform_state="retrying", - error_code=adapter.fatal_error_code, - error_message=adapter.fatal_error_message or "failed to reconnect", - ) - backoff = _reconnect_backoff(attempt) - info["attempts"] = attempt - info["next_retry"] = time.monotonic() + backoff - logger.info( - "Reconnect %s failed, next retry in %ds", - platform.value, backoff, - ) - # Same fd-leak concern as the non-retryable branch above: the adapter failed - # to connect and is being thrown away. - await _dispose_unused_adapter(adapter) - # Retryable failures (network/DNS blips) retry at the backoff cap forever, - # self-healing when connectivity returns. Never auto-pause them: a transient - # outage must not need `/platform resume`. Everything here is retryable. - except Exception as e: - if adapter is not None: - # An exception escaping connect (DNS timeout, aiohttp server.start() crash, - # etc.) leaves the adapter in the same unowned state as the branches above. - await _dispose_unused_adapter(adapter) - self._update_platform_runtime_status( - platform.value, - platform_state="retrying", - error_code=None, - error_message=str(e), - ) - backoff = _reconnect_backoff(attempt) - info["attempts"] = attempt - info["next_retry"] = time.monotonic() + backoff - logger.warning( - "Reconnect %s error: %s, next retry in %ds", - platform.value, e, backoff, - ) - # A reconnect exception (connect timeout, DNS failure, ...) is transient; keep - # retrying at the backoff cap rather than auto-pausing. - - # Check every 10 seconds for platforms that need reconnection - for _ in range(10): - if not self._running: - return - await asyncio.sleep(1) - - async def _cancel_secondary_profile_reconnect_tasks(self) -> None: - """Cancel profile-scoped reconnects before tearing down their registry. - - A reconnect can be waiting in adapter setup while shutdown begins. It must not republish - an adapter after the secondary registry is drained. Waiting is bounded by the adapter- - cleanup budget; a task that overruns is still blocked by the stopped runner state. - """ - pending = self._profile_failed_platforms - if not isinstance(pending, dict): - return - current = asyncio.current_task() - tasks: list[asyncio.Task] = [] - for profile_pending in pending.values(): - if not isinstance(profile_pending, dict): - continue - for task in profile_pending.values(): - if isinstance(task, asyncio.Task) and task is not current and not task.done(): - tasks.append(task) - for task in tasks: - task.cancel() - timeout = self._adapter_disconnect_timeout_secs() - if tasks and timeout > 0: - _done, unfinished = await asyncio.wait(tasks, timeout=timeout) - if unfinished: - logger.warning( - "Timed out waiting for %d secondary profile reconnect task(s) during shutdown", - len(unfinished), - ) - pending.clear() - - def _start_systemd_watchdog(self) -> bool: - """Start sd_notify only after a configured gateway is truly running.""" - if not self._running or self.config.systemd_watchdog_seconds <= 0: - return False - if self._systemd_watchdog is not None: - return True - - from gateway.systemd_notify import SystemdWatchdog - - watchdog = SystemdWatchdog(config_enabled=True) - if not watchdog.start(): - return False - self._systemd_watchdog = watchdog - watchdog.ready("Hermes Gateway running") - return True - - async def _stop_systemd_watchdog(self) -> None: - """Stop heartbeats before any potentially long shutdown drain.""" - watchdog = self._systemd_watchdog - if watchdog is None: - return - self._systemd_watchdog = None - await watchdog.stop() - - async def stop( - self, - *, - restart: bool = False, - detached_restart: bool = False, - service_restart: bool = False, - ) -> None: - """Stop the gateway and disconnect all adapters.""" - # getattr-guard: shutdown-path tests build bare runners via - # object.__new__ that lack the liveness-guard machinery. - _stop_guards = getattr(self, "_stop_loop_liveness_guards", None) - if callable(_stop_guards): - _stop_guards() - if restart: - self._restart_requested = True - self._restart_detached = detached_restart - self._restart_via_service = service_restart - if self._stop_task is not None: - await self._stop_task - return - - async def _stop_impl() -> None: - def _kill_tool_subprocesses(phase: str) -> list: - """Kill tool subprocesses + tear down terminal envs + browsers. - - Returns the cron job IDs marked interrupted so the caller can notify owners while - adapters are still up. Called twice: eagerly after a drain timeout forces interrupt - (reclaim children before systemd SIGKILLs) and as a final catch-all in _stop_impl(). - Best-effort; exceptions swallowed so one subsystem cannot block the rest. - """ - try: - from tools.process_registry import process_registry - _killed = process_registry.kill_all() - if _killed: - logger.info( - "Shutdown (%s): killed %d tool subprocess(es)", - phase, _killed, - ) - except Exception as _e: - logger.debug("process_registry.kill_all (%s) error: %s", phase, _e) - _marked_cron_jobs: list = [] - try: - # kill_all() is a global sweep, so any cron job dispatched right now lost its tool - # subprocess; its agent thread may still emit a plausible response from truncated - # output. Mark the run interrupted so it can never be reported as success. - from cron.scheduler import mark_running_jobs_interrupted - _interrupted = _marked_cron_jobs = mark_running_jobs_interrupted( - f"Gateway shutdown ({phase}) killed the job's tool " - "subprocess before the run finished." - ) - if _interrupted: - logger.warning( - "Shutdown (%s): marked %d in-flight cron job(s) interrupted: %s", - phase, len(_interrupted), ", ".join(_interrupted), - ) - except Exception as _e: - logger.debug("mark_running_jobs_interrupted (%s) error: %s", phase, _e) - try: - from tools.async_delegation import interrupt_all as _interrupt_async - _async_n = _interrupt_async(reason=f"gateway shutdown ({phase})") - if _async_n: - logger.info( - "Shutdown (%s): interrupted %d background delegation(s)", - phase, _async_n, - ) - except Exception as _e: - logger.debug("async interrupt_all (%s) error: %s", phase, _e) - try: - from tools.terminal_tool import cleanup_all_environments - cleanup_all_environments() - except Exception as _e: - logger.debug("cleanup_all_environments (%s) error: %s", phase, _e) - try: - from tools.browser_tool import cleanup_all_browsers - cleanup_all_browsers() - except Exception as _e: - logger.debug("cleanup_all_browsers (%s) error: %s", phase, _e) - return _marked_cron_jobs - - # Thread-based shutdown watchdog: asyncio timeouts cannot recover a frozen loop. Arm a - # plain OS thread at the start of stop(); if teardown never finishes within drain+grace - # it dumps faulthandler stacks and os._exit so KeepAlive/systemd can revive. Skipped - # under pytest so stop()-driving tests don't get a delayed hard-exit in the worker. - _watchdog_done = threading.Event() - self._shutdown_watchdog_done = _watchdog_done - _stop_started_at_box: dict[str, float] = {} - - def _shutdown_watchdog_snapshot() -> dict: - started = _stop_started_at_box.get("t") - return { - "restart_requested": bool(self._restart_requested), - "draining": bool(self._draining), - "running": bool(self._running), - "active_agents": self._running_agent_count(), - "active_cron_jobs": self._active_cron_job_count(), - "active_api_runs": self._active_api_run_count(), - "active_deferred_agent_workers": getattr( - self, - "_active_deferred_agent_worker_count", - lambda: 0, - )(), - "restart_drain_timeout": self._restart_drain_timeout, - "watchdog_delay_s": resolve_shutdown_watchdog_delay( - self._restart_drain_timeout - ), - "phase_elapsed_s": ( - time.monotonic() - started if started is not None else None - ), - } - - if not os.environ.get("PYTEST_CURRENT_TEST"): - arm_shutdown_watchdog( - resolve_shutdown_watchdog_delay(self._restart_drain_timeout), - done_event=_watchdog_done, - snapshot_fn=_shutdown_watchdog_snapshot, - exit_code=1, - ) - - try: - await _stop_impl_body( - _kill_tool_subprocesses, - _stop_started_at_box, - ) - finally: - _watchdog_done.set() - - async def _stop_impl_body(_kill_tool_subprocesses, _stop_started_at_box) -> None: - # Shutdown-path tests and third-party runner doubles may only - # implement the older drain-count surface. - _deferred_worker_count = getattr( - self, - "_active_deferred_agent_worker_count", - lambda: 0, - ) - logger.info( - "Stopping gateway%s...", - " for restart" if self._restart_requested else "", - ) - _stop_started_at = time.monotonic() - _stop_started_at_box["t"] = _stop_started_at - - def _phase_elapsed() -> float: - return time.monotonic() - _stop_started_at - - self._running = False - self._clear_plugin_message_injector() - self._draining = True - - stop_room_worker = getattr(self, "_stop_hosted_room_worker", None) - if callable(stop_room_worker): - try: - stopped = await stop_room_worker(timeout=5.0) - if not stopped: - logger.warning( - "Group Chat worker is still settling durable work; " - "the next gateway start will recover it" - ) - except Exception: - logger.warning( - "Group Chat worker could not stop cleanly; the next gateway " - "start will recover durable work", - exc_info=True, - ) - - stop_watchdog = getattr(self, "_stop_systemd_watchdog", None) - if callable(stop_watchdog): - await stop_watchdog() - - await self._cancel_secondary_profile_reconnect_tasks() - - # Notify all chats with active agents BEFORE draining. - # Adapters are still connected here, so messages can be sent. - await self._notify_active_sessions_of_shutdown() - logger.info( - "Shutdown phase: notify_active_sessions done at +%.2fs", - _phase_elapsed(), - ) - - timeout = self._restart_drain_timeout - - # Pre-mark sessions resume_pending BEFORE the drain wait: if the service manager kills - # the process mid-drain, the durable marker already lets the next boot recover them. - _pre_drain_keys: list[str] = [] - for _sk, _agent in list(self._running_agents.items()): - if _agent is _AGENT_PENDING_SENTINEL: - continue - try: - await self.async_session_store.mark_resume_pending( - _sk, - "restart_timeout" if self._restart_requested else "shutdown_timeout", - ) - _pre_drain_keys.append(_sk) - except Exception as _e: - logger.debug("pre-drain mark_resume_pending failed for %s: %s", _sk, _e) - - _cron_at_start = self._active_cron_job_count() - _api_at_start = self._active_api_run_count() - _deferred_at_start = _deferred_worker_count() - # In-flight cron work gets its own floor, clamped to the watchdog leash so the extra - # wait never costs the post-drain cleanup window. getattr-guard: shutdown-path tests - # drive _stop_impl_body from bare doubles (not GatewayRunner) lacking the class default. - _cron_drain_cfg = getattr( - self, "_cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT - ) - _cron_timeout = resolve_cron_drain_budget( - timeout, - _cron_drain_cfg, - watchdog_delay=resolve_shutdown_watchdog_delay(timeout), - elapsed=_phase_elapsed(), - ) - if _cron_at_start and _cron_timeout > timeout: - logger.info( - "Shutdown drain: %d in-flight cron job(s) — waiting up to " - "%.0fs for them (cron_drain_timeout=%.0fs, " - "restart_drain_timeout=%.0fs)", - _cron_at_start, - _cron_timeout, - _cron_drain_cfg, - timeout, - ) - _drain_started_at = time.monotonic() - active_agents, timed_out = await self._drain_active_agents( - timeout, _cron_timeout - ) - _drain_elapsed = time.monotonic() - _drain_started_at - logger.info( - "Shutdown phase: drain done at +%.2fs (drain took %.2fs, " - "timed_out=%s, active_at_start=%d, active_now=%d, " - "cron_at_start=%d, cron_now=%d, " - "api_at_start=%d, api_now=%d, " - "deferred_at_start=%d, deferred_now=%d)", - _phase_elapsed(), - _drain_elapsed, - timed_out, - len(active_agents), - self._running_agent_count(), - _cron_at_start, - self._active_cron_job_count(), - _api_at_start, - self._active_api_run_count(), - _deferred_at_start, - _deferred_worker_count(), - ) - - if not timed_out: - # Graceful drain: clear the pre-drain resume_pending markers so sessions that - # finished during the drain window don't carry a stale flag. - for _sk in _pre_drain_keys: - if _sk not in self._running_agents: - try: - await self.async_session_store.clear_resume_pending(_sk) - except Exception as _e: - logger.debug( - "clear_resume_pending after drain failed for %s: %s", - _sk, _e, - ) - - if timed_out: - logger.warning( - "Gateway drain timed out after %.1fs with %d active agent(s), " - "%d in-flight cron job(s), %d api_server run(s), and " - "%d deferred agent worker(s); " - "interrupting remaining work.", - _drain_elapsed, - self._running_agent_count(), - self._active_cron_job_count(), - self._active_api_run_count(), - _deferred_worker_count(), - ) - # Mark forcibly-interrupted sessions resume_pending BEFORE interrupting, so the next - # message on the same session_key auto-resumes instead of being converted to a fresh - # session by suspend_recently_active(). Genuinely stuck sessions still escalate via - # ``.restart_failure_counts`` (threshold 3), which sets ``suspended=True`` and wins. - # - # Iterate self._running_agents (current), not the drain-start snapshot: sessions that - # finished cleanly during the drain would otherwise get a stray interruption note. - # Skip pending sentinels as _interrupt_running_agents() does — nothing has started. - _resume_reason = ( - "restart_timeout" if self._restart_requested else "shutdown_timeout" - ) - for _sk, _agent in list(self._running_agents.items()): - if _agent is _AGENT_PENDING_SENTINEL: - continue - try: - await self.async_session_store.mark_resume_pending(_sk, _resume_reason) - except Exception as _e: - logger.debug( - "mark_resume_pending failed for %s: %s", - _sk, _e, - ) - self._interrupt_running_agents( - _INTERRUPT_REASON_GATEWAY_RESTART if self._restart_requested else _INTERRUPT_REASON_GATEWAY_SHUTDOWN - ) - interrupt_grace_timeout = ( - GatewayRunner._post_interrupt_grace_timeout(self) - ) - interrupt_deadline = ( - asyncio.get_running_loop().time() + interrupt_grace_timeout - ) - logger.info( - "Shutdown phase: allowing %.1fs for interrupted agents to unwind", - interrupt_grace_timeout, - ) - # Wait on API-server work too: the interrupt is cooperative, and without this the - # settle window closes as soon as _running_agents is empty, so an API turn just asked - # to stop has its tool subprocesses killed below before it can unwind. - while ( - self._running_agents - or self._active_api_run_count() - or _deferred_worker_count() - ) and asyncio.get_running_loop().time() < interrupt_deadline: - self._update_runtime_status("draining") - await asyncio.sleep(0.1) - - # The interrupt fires once, but work can materialize AFTER it: a /v1/runs task enters - # _active_run_agents only when _create_agent returns, and a _AGENT_PENDING_SENTINEL - # entry is promoted by track_agent() on its own schedule. Re-signal anything still - # live so it gets a cooperative interrupt instead of a bare tool-subprocess kill. - if ( - self._running_agents - or self._active_api_run_count() - or _deferred_worker_count() - ): - self._interrupt_running_agents( - _INTERRUPT_REASON_GATEWAY_RESTART - if self._restart_requested - else _INTERRUPT_REASON_GATEWAY_SHUTDOWN - ) - logger.debug( - "Re-signaled interrupt for work still live at settle-window exit" - ) - - # Kill lingering tool subprocesses NOW, before adapter disconnect / DB close: under - # systemd (TimeoutStopSec ≈ drain_timeout + headroom) deferring risks the cgroup - # SIGKILL reaping orphaned children instead of us. The final catch-all still runs. - _interrupted_cron_jobs = _kill_tool_subprocesses("post-interrupt") - logger.info( - "Shutdown phase: post-interrupt tool kill done at +%.2fs", - _phase_elapsed(), - ) - # Last window where the transport is still up. The cron worker whose run we just - # killed will try to deliver its own "interrupted" notice, but it gets there after - # the adapter teardown below and the message is lost. - try: - await self._notify_interrupted_cron_jobs(_interrupted_cron_jobs) - except Exception as _e: - logger.debug("Cron interrupt notification failed: %s", _e) - logger.info( - "Shutdown phase: cron interrupt notices done at +%.2fs", - _phase_elapsed(), - ) - - if self._restart_requested and self._restart_detached: - try: - await self._launch_detached_restart_command() - except Exception as e: - logger.error("Failed to launch detached gateway restart: %s", e) - - await self._finalize_shutdown_agents(active_agents) - - # Also shut down memory providers on idle cached agents. _finalize_shutdown_agents only - # handles agents that were mid-turn at drain time; the _agent_cache may still hold idle - # agents whose MemoryProviders never received on_session_end(). - _cache_lock = getattr(self, "_agent_cache_lock", None) - _cache = getattr(self, "_agent_cache", None) - if _cache_lock is not None and _cache is not None: - with _cache_lock: - _idle_agents = list(_cache.values()) - _cache.clear() - for _entry in _idle_agents: - _agent = ( - _entry[0] if isinstance(_entry, tuple) else _entry - ) - # Bounded + off-loop so a wedged memory provider can't hang shutdown forever - # (this path is why SIGTERM once failed to kill the process). - await self._cleanup_agent_resources_off_loop( - _agent, context="shutdown idle-cache" - ) - - # Completion flush tasks can be sleeping in their fan-in window or blocked in adapter - # delivery. Cancel and await them while adapters are still alive so every watcher - # receives a retryable result before platform teardown begins. - cancel_completion_batches = getattr( - self, "_cancel_process_completion_batch_tasks", None - ) - if cancel_completion_batches is not None: - await cancel_completion_batches() - - for platform, adapter in list(self.adapters.items()): - await self._bounded_adapter_teardown(adapter, platform) - - # Disconnect secondary-profile adapters (multiplex mode). - for _prof, _amap in list(getattr(self, "_profile_adapters", {}).items()): - for platform, adapter in list(_amap.items()): - await self._bounded_adapter_teardown( - adapter, platform, profile=_prof - ) - _amap.clear() - if hasattr(self, "_profile_adapters"): - self._profile_adapters.clear() - logger.info( - "Shutdown phase: all adapters disconnected at +%.2fs", - _phase_elapsed(), - ) - - for _task in list(self._background_tasks): - if _task is self._stop_task: - continue - if _task is self._restart_task: - # The restart orchestration task is awaiting _stop_task right now; cancelling it - # would propagate CancelledError into this _stop_impl and skip - # _shutdown_event.set() / _exit_code = 75. It self-terminates anyway. - continue - _task.cancel() - self._background_tasks.clear() - - self.adapters.clear() - for _session_key in list(self._running_agents): - self._release_running_agent_state(_session_key) - # Flush pending messages to disk before clearing: under FTS5 corruption the in-memory - # pending text is the only surviving copy; clearing unflushed loses it permanently. - try: - from gateway.shutdown_flush import flush_pending_to_file - flush_pending_to_file(dict(self._pending_messages), reason="shutdown") - except Exception: - pass - # The FIFO tail lives in SessionState.conversation.queued_events, not the slot dict - # above — flush it too or every follow-up parked in overflow at restart time is lost. - try: - from gateway.shutdown_flush import flush_overflow_to_file - flush_overflow_to_file( - { - _k: list(_v) - for _k, _v in dict(getattr(self, "_queued_events", None) or {}).items() - if _v - }, - reason="shutdown", - ) - except Exception: - pass - # On the real runner these are live SessionState views whose clear() resets one field - # per session — never a wholesale dict swap, so a concurrent writer on another session - # can't lose its entry. Test fakes borrowing _stop_impl keep plain dicts. - self._running_agents.clear() - self._running_agents_ts.clear() - if hasattr(self, "_active_session_leases"): - self._active_session_leases.clear() - self._pending_messages.clear() - self._pending_approvals.clear() - if hasattr(self, '_busy_ack_ts'): - self._busy_ack_ts.clear() - self._shutdown_event.set() - - # Global catch-all subprocess kill (safe to repeat): covers the graceful path and - # anything respawned since the drain-timeout path's post-interrupt kill. - _kill_tool_subprocesses("final-cleanup") - logger.info( - "Shutdown phase: final-cleanup tool kill done at +%.2fs", - _phase_elapsed(), - ) - - # Reap the process-global auxiliary-client cache once at the end of teardown. Per-turn - # cleanup misses clients bound to worker-thread loops that died with their executor - # (cron ticks); without this sweep async httpx transports accumulate until EMFILE. - try: - from agent.auxiliary_client import shutdown_cached_clients - shutdown_cached_clients() - except Exception as _e: - logger.debug("shutdown_cached_clients error: %s", _e) - - # Quiesce the gateway thread pool BEFORE the session databases are closed. Running it - # after the close left two holes: (a) ``_executor_closing`` was still False, so any - # coroutine reaching ``_run_in_executor_with_context`` minted a fresh pool and ran more - # blocking DB work against just-closed handles; (b) cancelling ``self._background_tasks`` - # does not stop a ``run_in_executor`` future that already started — the task dies, the - # worker keeps writing. Either way a write lands after ``SessionDB.close()`` has - # checkpointed the WAL and let SQLite unlink the sidecar; the late write silently - # reopens the handle and mints a fresh WAL generation behind that checkpoint, so - # teardown checkpoints the same file twice from an unaccounted connection - # (close-time page-write corruption / split WAL generation). - # The wait is bounded and clamped to what is left of the shutdown watchdog leash - # (minus a second for the close itself), so a stuck worker can never cost us the - # post-close cleanup window. - _exec_quiesce_budget = max( - 0.0, - min( - _EXECUTOR_QUIESCE_TIMEOUT, - resolve_shutdown_watchdog_delay(timeout) - - _phase_elapsed() - - 1.0, - ), - ) - _exec_live = GatewayRunner._shutdown_executor( - self, drain_timeout=_exec_quiesce_budget - ) - if _exec_live: - # A live worker can still be mid-write against a SessionDB - # handle. Checkpointing/closing it now is exactly the - # sequence that produced the wrong-page-number corruption in - # #101093, so the close path below is skipped entirely - # rather than raced — the handle is left open for SQLite to - # recover from its own WAL on the next open, which is a - # transient "database is locked" on an immediate --replace - # at worst, not a corrupt file. - logger.warning( - "Shutdown phase: %d executor worker(s) still running after " - "a %.2fs quiesce — skipping the SessionDB close/checkpoint " - "to avoid racing a live write (#101093); handles are left " - "open for SQLite to recover on next open", - _exec_live, - _exec_quiesce_budget, - ) - else: - logger.info( - "Shutdown phase: executor quiesced at +%.2fs", - _phase_elapsed(), - ) - - # Close SQLite session DBs so the WAL lock is released; otherwise --replace leaves the old - # connection holding it until exit and the new gateway gets 'database is locked'. - # ``_session_db`` is an AsyncSessionDB facade — unwrap; ``session_store`` holds ``_db``. - _self_db = getattr(self, "_session_db", None) - _self_db = getattr(_self_db, "_db", _self_db) - for _db in (_self_db, getattr(getattr(self, "session_store", None), "_db", None)): - if _db is None or not hasattr(_db, "close"): - continue - try: - _db.close() - except Exception as _e: - logger.debug("SessionDB close error: %s", _e) - # A multiplexed session_store caches one SessionDB per profile; ``_db`` above only covered - # the root scope. Sweep the rest so secondary WAL locks are released before --replace. - _sweep = getattr( - getattr(self, "session_store", None), "close_all_db_handles", None - ) - if _sweep is not None: - try: - _sweep() - except Exception as _e: - logger.debug("SessionDB handle sweep error: %s", _e) - # Same sweep for the runner's own per-profile session_search - # handles (slash commands resolve them under profile scopes). - try: - GatewayRunner.close_all_session_db_handles(self) - except Exception as _e: - logger.debug("Runner SessionDB handle sweep error: %s", _e) - # Final sweep: close shared SessionDB instances still held by the process-wide registry - # (tools, cron, mirror, etc. opened via get_shared_session_db but not released above). - try: - from hermes_state import close_shared_session_dbs - closed = close_shared_session_dbs() - if closed: - logger.debug("Closed %d shared SessionDB instance(s) at shutdown", closed) - except Exception as _e: - logger.debug("Shared SessionDB close error: %s", _e) - logger.info( - "Shutdown phase: SessionDB close done at +%.2fs", - _phase_elapsed(), - ) - - from gateway.status import remove_pid_file, release_gateway_runtime_lock - remove_pid_file() - release_gateway_runtime_lock() - - # Clean-shutdown marker: suspend_recently_active() need only run after unexpected exits. - # If the drain timed out and agents were force-interrupted, sessions may be half-finished - # — skip the marker so the next startup suspends them. - if not timed_out: - with suppress(Exception): - (_hermes_home / ".clean_shutdown").touch() - else: - logger.info( - "Skipping .clean_shutdown marker — drain timed out with " - "interrupted agents; next startup will suspend recently " - "active sessions." - ) - - # Stuck-loop detection: the counter increments for sessions active at each restart; at - # the threshold (3 consecutive) the next startup auto-suspends the session. - if active_agents: - self._increment_restart_failure_counts(set(active_agents.keys())) - - if self._restart_requested and self._restart_command_source is None: - try: - atomic_json_write( - _planned_restart_notification_path(), - { - "requested_at": time.time(), - "via_service": bool(self._restart_via_service), - "detached": bool(self._restart_detached), - }, - indent=None, - ) - except Exception as e: - logger.debug("Failed to write planned restart notification marker: %s", e) - - if self._restart_requested and self._restart_via_service: - # Service manager owns restarts: exit 75 + ``RestartForceExitStatus=75`` has systemd - # replace this process without a second helper racing the unit's stop/start job. - self._exit_code = GATEWAY_SERVICE_RESTART_EXIT_CODE - self._exit_reason = self._exit_reason or "Gateway restart requested" - - self._draining = False - # Persist terminal gateway_state: "stopped" by default, but "running" on an UNEXPECTED - # external signal (s6 SIGTERM on docker restart, OOM-kill, kill) — container_boot.py - # only auto-starts gateways last seen "running", so "stopped"/"draining" after a routine - # recreate would leave channels dark. Operator stops write a planned-stop marker BEFORE - # signalling and persist "stopped"; a restart also persists "stopped". - if getattr(self, "_signal_initiated_shutdown", False) and not self._restart_requested: - logger.info( - "Gateway stopped by an unexpected signal — persisting " - "gateway_state=running so container_boot auto-starts on " - "the next boot (issue #42675)" - ) - self._update_runtime_status("running", self._exit_reason) - else: - self._update_runtime_status("stopped", self._exit_reason) - _shutdown_gateway_health_export(self) - logger.info("Gateway stopped (total teardown %.2fs)", _phase_elapsed()) - - self._stop_task = asyncio.create_task(_stop_impl()) - await self._stop_task - - async def wait_for_shutdown(self) -> None: - """Wait for shutdown signal.""" - await self._shutdown_event.wait() - - async def _start_secondary_profile_adapters(self) -> int: - """Bring up adapters for every non-active profile this gateway serves. - - Returns the count of connected secondary adapters; 0 unless ``gateway.multiplex_profiles``. - Each profile's adapters connect under its HERMES_HOME + secret scope, live in - ``self._profile_adapters[profile]``, and get a handler stamping ``source.profile``. Same- - platform credential collisions are refused here — the only point seeing every profile's - resolved credentials together. - """ - if not getattr(self.config, "multiplex_profiles", False): - return 0 - - try: - from hermes_cli.profiles import get_active_profile_name - except Exception: - return 0 - - active = get_active_profile_name() or "default" - connected = 0 - # Resource claim -> owning profile. Credential claims stop two profiles polling the same - # account; listener claims stop sidecars with distinct credentials binding one endpoint. - claimed: Dict[tuple, str] = {} - for _plat, _ad in self.adapters.items(): - fp = self._adapter_credential_fingerprint(_ad) - if fp is not None: - claimed[(_plat, fp)] = active - listener_claim = self._adapter_listener_claim(_plat, _ad) - if listener_claim is not None: - claimed[listener_claim] = active - # A retryable primary still owns its credential and listener; reserve both while queued - # so a secondary cannot take the endpoint before the reconnect watcher retries it. - for retry_info in getattr(self, "_failed_platforms", {}).values(): - for claim_name in ("credential_claim", "listener_claim"): - retry_claim = retry_info.get(claim_name) - if isinstance(retry_claim, tuple): - claimed[retry_claim] = active - - profile_homes = _multiplex_profile_homes(self.config) - for profile_name, profile_home in profile_homes: - if profile_name == active: - continue # handled by the primary startup loop - try: - connected += await self._start_one_profile_adapters( - profile_name, profile_home, claimed - ) - except SecondaryPortBindingConfigError as e: - logger.warning( - "Skipping secondary profile '%s' due to port-binding config error: %s", - profile_name, - e, - ) - except MultiplexConfigError: - raise - except Exception as e: - logger.error( - "Failed to start adapters for profile '%s': %s", - profile_name, e, exc_info=True, - ) - - # Record the authoritative served set in runtime status for `hermes status`. "Served" - # means eligible for shared routing, HTTP prefixes, cron, and profile runtime scope — - # intentionally broader than profiles with a connected (or any) secondary adapter. - try: - from gateway.status import write_runtime_status - from gateway.pairing import PairingStore - served = [active] + sorted( - name for name, _home in profile_homes if name != active - ) - # Per-profile PairingStores so authz_mixin routes pairing checks to the right whitelist; - # the active profile's store is at its HERMES_HOME, other served profiles at their own. - for name in served: - if name and name not in self.pairing_stores: - self.pairing_stores[name] = ( - self.pairing_store - if name == active - else PairingStore(profile=name) - ) - write_runtime_status(served_profiles=served) - except Exception: - logger.debug("could not record served_profiles", exc_info=True) - - return connected - - async def _start_one_profile_adapters( - self, profile_name: str, profile_home: "Path", claimed: Dict[tuple, str] - ) -> int: - """Create+connect one profile's adapters under its runtime scope.""" - from gateway.config import load_gateway_config - from hermes_cli.env_loader import hydrate_profile_secret_sources - - # Hydrate external secret sources (1Password/vault/...) off-loop ONCE, then enter the scope - # without re-hydrating: the sync hydration is network-bound and would otherwise stall every - # other profile's heartbeat while this one boots (same class as the reconnect path). - await asyncio.to_thread(hydrate_profile_secret_sources, profile_home) - - with _profile_runtime_scope(profile_home, hydrate_secrets=False): - profile_runtime_cfg = _load_gateway_runtime_config() - from hermes_cli.plugins import discover_plugins - - discover_plugins() - - # Register this profile's own declarative shell hooks and outbound webhooks. The - # registration in start() runs before any profile scope exists and only sees the root - # profile's config, so without this a secondary profile's `hooks:` block is silently - # inert (its turns use a plugin manager keyed by resolved home). - try: - from hermes_cli.config import load_config as _load_profile_config - from agent.shell_hooks import ( - register_from_config as _register_shell_hooks, - ) - from agent.outbound_webhooks import ( - register_from_config as _register_outbound_webhooks, - ) - - _profile_hooks_cfg = _load_profile_config() - _register_shell_hooks(_profile_hooks_cfg, accept_hooks=False) - _register_outbound_webhooks(_profile_hooks_cfg) - except Exception: - logger.warning( - "shell-hook/webhook registration failed for profile '%s'", - profile_name, - exc_info=True, - ) - - profile_cfg = load_gateway_config() - violation = _own_policy_open_startup_violation(profile_cfg) - self._snapshot_profile_busy_modes(profile_name, profile_runtime_cfg) - if violation: - raise MultiplexConfigError( - f"Profile '{profile_name}' enables {violation}. " - "Enable GATEWAY_ALLOW_ALL_USERS or the platform allow-all flag " - "for that profile, or change dm_policy/group_policy away from " - "'open'." - ) - - port_binding_platforms = sorted( - platform.value - for platform, platform_config in profile_cfg.platforms.items() - if platform_config.enabled - and _platform_binds_port(platform.value, platform_config.extra) - ) - if port_binding_platforms: - joined = ", ".join(port_binding_platforms) - raise SecondaryPortBindingConfigError( - f"Profile '{profile_name}' enables port-binding platform(s) " - f"{joined}, but gateway.multiplex_profiles is on. The default " - f"profile owns the single shared HTTP listener and serves every " - f"profile through the /p/{profile_name}/ URL prefix. Remove " - f"these platform entries from profile '{profile_name}'s config.yaml " - f"or configure them only on the default profile." - ) - - profile_map = self._profile_adapters.setdefault(profile_name, {}) - connected = 0 - for platform, platform_config in profile_cfg.platforms.items(): - if not platform_config.enabled: - continue - # A platform enabled in a secondary profile's config.yaml may have no credential in that - # profile's secret scope — the shared YAML enables it for the default profile only. - # Building an adapter anyway would fan one inbound message out across every - # credential-less profile; mirror the primary loop's credential gate and skip. - if ( - getattr(self.config, "multiplex_profiles", False) - and not _platform_has_bot_credential(platform, platform_config) - ): - logger.info( - "[MULTIPLEX] Profile '%s': skipping %s - no bot credential " - "in this profile's secrets", - profile_name, - platform.value, - ) - continue - # Relay and WhatsApp are shared process-level ingress in multiplex mode (one connection - # owned by the active profile, route-stamped source.profile fans out). WhatsApp is one - # session per phone number; a secondary adapter would only retry-loop and stall startup. - if ( - getattr(self.config, "multiplex_profiles", False) - and platform in (Platform.RELAY, Platform.WHATSAPP) - ): - continue - try: - with _profile_runtime_scope(profile_home, hydrate_secrets=False): - adapter = self._create_adapter(platform, platform_config) - except Exception as e: - logger.error( - "[MULTIPLEX] Profile '%s': _create_adapter('%s') raised %s", - profile_name, - platform.value, - e, - exc_info=True, - ) - continue - if not adapter: - logger.warning( - "[MULTIPLEX] Profile '%s': skipping platform '%s' - adapter creation returned None", - profile_name, - platform.value, - ) - continue - - # Same-token conflict detection — refuse a duplicate poll. - credential_claim = self._adapter_credential_claim(platform, adapter) - if credential_claim is not None: - owner = claimed.get(credential_claim) - if owner is not None: - message = ( - f"Profile '{owner}' and '{profile_name}' both configure " - f"{platform.value} with the same credential. Give each " - f"profile its own {platform.value} credential." - ) - logger.error( - "Profile '%s' and '%s' both configure %s with the same " - "credential — refusing to start the duplicate (one " - "credential cannot be consumed twice). Give each profile " - "its own %s credential.", - owner, profile_name, platform.value, platform.value, - ) - self._update_platform_runtime_status( - f"{profile_name}:{platform.value}", - platform_state="fatal", - error_code="duplicate_credential", - error_message=message, - ) - # This adapter has not connected and therefore owns no resources to clean up. - # Calling disconnect here can mutate the shared platform state and, for a same- - # credential Photon adapter, shut down the primary profile's live sidecar. - continue - - listener_claim = self._adapter_listener_claim(platform, adapter) - if listener_claim is not None: - owner = claimed.get(listener_claim) - if owner is not None: - bind, port = listener_claim[-2:] - message = ( - f"Profile '{owner}' and '{profile_name}' both configure " - f"{platform.value} sidecars on the same listener. Configure " - f"a distinct listener for profile '{profile_name}'." - ) - logger.error( - "Profile '%s' and '%s' both configure %s sidecars on " - "%s:%s — refusing to start the duplicate listener. " - "Set platforms.%s.extra.sidecar_port to a distinct port " - "for profile '%s'.", - owner, - profile_name, - platform.value, - bind, - port, - platform.value, - profile_name, - ) - self._update_platform_runtime_status( - f"{profile_name}:{platform.value}", - platform_state="fatal", - error_code="duplicate_listener", - error_message=message, - ) - # Like credential conflicts, this adapter never connected - # and owns no resources that should be disconnected. - continue - - self._configure_profile_adapter(adapter, profile_name, platform) - - try: - with _profile_runtime_scope(profile_home, hydrate_secrets=False): - success = await self._connect_initial_adapter_with_timeout( - adapter, platform - ) - if success: - profile_map[platform] = adapter - # Restore persisted /voice state for this bot (#84872) — - # primary startup and every reconnect path already do. - self._sync_voice_mode_state_to_adapter(adapter) - if credential_claim is not None: - claimed[credential_claim] = profile_name - if listener_claim is not None: - claimed[listener_claim] = profile_name - connected += 1 - logger.info("✓ %s connected (profile: %s)", platform.value, profile_name) - else: - logger.warning("✗ %s failed to connect (profile: %s)", platform.value, profile_name) - await self._safe_adapter_disconnect(adapter, platform) - self._schedule_secondary_profile_startup_reconnect( - profile_name, platform, adapter - ) - except Exception as e: - logger.error("✗ %s error (profile: %s): %s", platform.value, profile_name, e) - await self._safe_adapter_disconnect(adapter, platform) - self._schedule_secondary_profile_startup_reconnect( - profile_name, platform, adapter - ) - return connected - - def _configure_profile_adapter( - self, - adapter: BasePlatformAdapter, - profile_name: str, - platform: Platform, - ) -> None: - """Install the profile-scoped handlers shared by startup and reconnect.""" - # Runtime status is process-scoped while message/config work is profile-scoped. Keep both - # dimensions in the key so dashboard/NAS health aggregation sees which secondary failed. - adapter._runtime_status_platform_key = f"{profile_name}:{platform.value}" - adapter.set_message_handler(self._make_profile_message_handler(profile_name)) - adapter.set_fatal_error_handler( - self._make_profile_fatal_error_handler(profile_name, platform) - ) - adapter.set_session_store(self.session_store) - # Declare credential ownership BEFORE any inbound event can be handled: adapter-level - # session keys (batching, _active_sessions, busy guard) are derived at ingress, before the - # handler stamps source.profile — without this every secondary bot would key into the - # default profile's `agent:main:` lane (see BasePlatformAdapter._session_key_profile). - _set_owner = getattr(adapter, "set_owner_profile", None) - if callable(_set_owner): - _set_owner(profile_name) - adapter.set_busy_session_handler( - self._make_profile_busy_session_handler(profile_name) - ) - _set_reaction = getattr(adapter, "set_reaction_handler", None) - if callable(_set_reaction): - _set_reaction(self._handle_reaction_event) - adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) - adapter.set_authorization_check( - self._make_adapter_auth_check(platform, profile_name=profile_name) - ) - adapter.set_platform_event_handler( - self._make_profile_platform_event_handler(profile_name) - ) - # Voice transcripts from this bot's channels dispatch through THIS - # adapter (primary wiring lives at connect time; see #75198). - self._bind_voice_input_callback(adapter) - text_modes = getattr(self, "_busy_text_modes_by_profile", None) - adapter._busy_text_mode = ( - text_modes.get(profile_name, self._busy_text_mode) - if isinstance(text_modes, dict) - else self._busy_text_mode - ) - # Secondary adapters always carry the profile they serve so prune - # paths namespace topic bindings correctly under multiplex (#76423). - adapter._hermes_profile_name = profile_name - - async def _run_secondary_profile_reconnect( - self, profile_name: str, platform: Platform - ) -> None: - """Reconnect a retryable secondary adapter under its own profile scope.""" - attempts = 0 - current_task = asyncio.current_task() - try: - while self._running: - adapter = None - try: - from hermes_cli.profiles import get_profile_dir - from hermes_cli.env_loader import hydrate_profile_secret_sources - from gateway.config import load_gateway_config - - profile_home = get_profile_dir(profile_name) - # Like the #16856 MCP discovery path, hydrate external secret - # sources off-loop so they cannot starve platform heartbeats. - await asyncio.to_thread( - hydrate_profile_secret_sources, profile_home - ) - with _profile_runtime_scope(profile_home, hydrate_secrets=False): - profile_config = load_gateway_config().platforms.get(platform) - if profile_config is None or not profile_config.enabled: - return - # Mirrors the startup credential gate: a credential removed from this - # profile's scope must not rebuild an adapter that would fan out turns. - if not _platform_has_bot_credential(platform, profile_config): - logger.info( - "Secondary %s reconnect skipped: no bot credential " - "(profile: %s)", - platform.value, - profile_name, - ) - return - adapter = self._create_adapter(platform, profile_config) - if adapter is None: - logger.warning( - "Secondary %s reconnect skipped: adapter unavailable (profile: %s)", - platform.value, - profile_name, - ) - return - self._configure_profile_adapter( - adapter, profile_name, platform - ) - success = await self._connect_adapter_with_timeout( - adapter, platform, is_reconnect=True - ) - - if success and self._running: - profile_map = self._profile_adapters.setdefault(profile_name, {}) - if platform not in profile_map: - profile_map[platform] = adapter - self._sync_voice_mode_state_to_adapter(adapter) - logger.info( - "✓ %s reconnected (profile: %s)", - platform.value, - profile_name, - ) - await self._redeliver_failed_obligations_for_platform( - platform, profile=profile_name - ) - return - # A newer reconnect already won the slot while this - # attempt was awaiting connect; do not replace it. - await self._safe_adapter_disconnect(adapter, platform) - return - - # Shutdown can begin mid-connect(): never republish a newly connected adapter - # after the registry has been drained; release its partial resources instead. - if success: - await self._safe_adapter_disconnect(adapter, platform) - return - - await self._safe_adapter_disconnect(adapter, platform) - if ( - getattr(adapter, "has_fatal_error", False) - and not getattr(adapter, "fatal_error_retryable", True) - ): - return - except asyncio.CancelledError: - if adapter is not None: - await self._safe_adapter_disconnect(adapter, platform) - raise - except Exception: - if adapter is not None: - await self._safe_adapter_disconnect(adapter, platform) - logger.debug( - "Secondary %s reconnect attempt failed (profile: %s)", - platform.value, - profile_name, - exc_info=True, - ) - - if not self._running: - return - attempts += 1 - backoff = _reconnect_backoff(attempts) - logger.info( - "Secondary %s reconnect retry in %ds (profile: %s)", - platform.value, - backoff, - profile_name, - ) - await asyncio.sleep(backoff) - finally: - pending = self._profile_failed_platforms - if isinstance(pending, dict): - profile_pending = pending.get(profile_name) - task = profile_pending.get(platform) if isinstance(profile_pending, dict) else None - if not isinstance(task, asyncio.Task) or task is current_task: - if isinstance(profile_pending, dict): - profile_pending.pop(platform, None) - if not profile_pending: - pending.pop(profile_name, None) - - def _schedule_secondary_profile_startup_reconnect( - self, profile_name: str, platform: Platform, adapter: BasePlatformAdapter - ) -> None: - """Queue a cold-start reconnect for a secondary adapter. - - Startup failures happen BEFORE ``self._running`` flips True, so the regular scheduler's - guard would drop the request. Park a task across startup and hand off to the scheduler once - live (``_profile_failed_platforms`` dedupes); release it if shutdown begins first. - Non-retryable failures are dropped as the regular scheduler would. - """ - if not getattr(adapter, "fatal_error_retryable", True): - return - if is_global_startup_conflict(getattr(adapter, "fatal_error_code", None)): - # Same startup contract as the primary path: a live foreign holder of this profile's - # token/identity is an ownership conflict, not a transient blip. Park it fatal (like - # ``duplicate_credential``) instead of retry-storming the token every backoff. - logger.error( - "[MULTIPLEX] Profile '%s': %s credential is held by another " - "gateway (%s) — parked, not retried. %s", - profile_name, - platform.value, - adapter.fatal_error_code, - adapter.fatal_error_message or "", - ) - self._update_platform_runtime_status( - f"{profile_name}:{platform.value}", - platform_state="fatal", - error_code=adapter.fatal_error_code, - error_message=adapter.fatal_error_message, - ) - return - - async def _await_running_then_schedule() -> None: - if self._running: - try: - self._schedule_secondary_profile_reconnect( - profile_name, platform, adapter - ) - except Exception: - # Same GC-time-exception hazard as the post-poll handoff - # below; surface it in gateway.log instead. - logger.exception( - "secondary-startup-reconnect handoff failed " - "(profile=%s platform=%s)", - profile_name, - platform.value, - ) - return - # Modest poll: startup completion has no dedicated event, and the reconnect runner's own - # backoff makes sub-100ms precision irrelevant. Bounded so a wedged startup cannot spin. - while not self._running and not self._shutdown_event.is_set(): - await asyncio.sleep(0.1) - if self._running and not self._shutdown_event.is_set(): - try: - self._schedule_secondary_profile_reconnect( - profile_name, platform, adapter - ) - except Exception: - # The handoff touches live registries; if it raises, the parked task dies as an - # unretrieved-task exception logged only at GC. Surface it where operators look. - logger.exception( - "secondary-startup-reconnect handoff failed " - "(profile=%s platform=%s)", - profile_name, - platform.value, - ) - - task = asyncio.create_task( - _await_running_then_schedule(), - name=f"secondary-startup-reconnect:{profile_name}:{platform.value}", - ) - background_tasks = getattr(self, "_background_tasks", None) - if not isinstance(background_tasks, set): - background_tasks = set() - self._background_tasks = background_tasks - background_tasks.add(task) - task.add_done_callback(background_tasks.discard) - - def _schedule_secondary_profile_reconnect( - self, profile_name: str, platform: Platform, adapter: BasePlatformAdapter - ) -> None: - """Schedule one runner-owned reconnect without sharing primary secrets.""" - if not self._running or not adapter.fatal_error_retryable: - return - pending = self._profile_failed_platforms - if not isinstance(pending, dict): - pending = {} - self._profile_failed_platforms = pending - profile_pending = pending.setdefault(profile_name, {}) - if platform in profile_pending: - return - task = asyncio.create_task( - self._run_secondary_profile_reconnect(profile_name, platform), - name=f"secondary-reconnect:{profile_name}:{platform.value}", - ) - profile_pending[platform] = task - background_tasks = getattr(self, "_background_tasks", None) - if not isinstance(background_tasks, set): - background_tasks = set() - self._background_tasks = background_tasks - background_tasks.add(task) - task.add_done_callback(background_tasks.discard) - - def _make_profile_fatal_error_handler( - self, profile_name: str, platform: Platform - ) -> Callable[[BasePlatformAdapter], Awaitable[None]]: - """Route a secondary-profile fatal error to that profile's reconnect slot.""" - async def _handler(adapter: BasePlatformAdapter) -> None: - await self._handle_profile_adapter_fatal_error(profile_name, platform, adapter) - - return _handler - - async def _handle_profile_adapter_fatal_error( - self, - profile_name: str, - platform: Platform, - adapter: BasePlatformAdapter, - ) -> None: - """Remove a failed multiplexed adapter without touching the primary slot. - - Secondaries live in ``_profile_adapters``, which the primary-only fatal handler ignores; - without this route a fatal secondary Discord client stayed live forever. - """ - profile_map = getattr(self, "_profile_adapters", {}).get(profile_name) - if not isinstance(profile_map, dict) or profile_map.get(platform) is not adapter: - logger.debug( - "Ignoring stale fatal error from secondary %s adapter (profile: %s)", - platform.value, - profile_name, - ) - return - profile_map.pop(platform, None) - await self._safe_adapter_disconnect(adapter, platform) - if not self._running: - return - self._schedule_secondary_profile_reconnect(profile_name, platform, adapter) - logger.error( - "Fatal %s adapter error for multiplexed profile %s (%s)", - platform.value, - profile_name, - adapter.fatal_error_code or "unknown", - ) # Reconnect is scoped to the profile's own config and secret mapping; # never rebuild a secondary adapter with the default profile's credentials. - def _make_profile_message_handler(self, profile_name: str): - """Return a message handler that stamps source.profile then delegates. - - Auth runs inside ``_handle_message`` *before* the agent-turn scope is installed. For - secondary profiles under multiplex, wrap the whole handler in ``_profile_runtime_scope`` - so allowlists/tokens from that profile's ``.env`` are visible to ``get_secret`` / authz. - """ - from hermes_cli.profiles import get_profile_dir - - try: - profile_home = get_profile_dir(profile_name) - except Exception: - profile_home = None - - async def _handler(event): - try: - if getattr(event, "source", None) is not None and not event.source.profile: - event.source.profile = profile_name - except Exception: - pass - if profile_home is not None: - async with _async_profile_runtime_scope(profile_home): - return await self._handle_message(event) - return await self._handle_message(event) - - return _handler - - def _make_profile_busy_session_handler(self, profile_name: str): - """Stamp an owning adapter's profile before resolving busy policy.""" - async def _handler(event, _session_key): - try: - if getattr(event, "source", None) is not None and not event.source.profile: - event.source.profile = profile_name - except Exception: - pass - routed_session_key = self._session_key_for_source(event.source) - return await self._handle_active_session_busy_message( - event, routed_session_key - ) - - return _handler - - def _make_default_profile_message_handler(self): - """Scope primary-adapter messages to their routed multiplex profile. - - Resolve the home per event so session lookup and transcript loading use the same profile - store as the agent run. Authorization stays with the transport profile (a routed profile - may intentionally have no bot credential/allowlist): the transport home is preserved on the - live source and never re-checked against the routed scope. Unrouted events keep the default. - """ - default_home = Path(get_hermes_home()) - - async def _handler(event): - source = event.source - # In-process only (SessionSource serialization ignores dynamic attrs). The route selects - # agent/session state, not which bot admitted the message — separate trust domains. - source._authorization_profile_home = default_home - if ( - not getattr(source, "profile", None) - and getattr(source, "profile_route_rejected", False) is not True - ): - from gateway.profile_routing import ProfileRouteRejected - - try: - source.profile = self._profile_name_for_source(source) - except ProfileRouteRejected: - # NOT write-only: the ``_handle_message`` ingress gate reads this exact marker - # and drops the message fail-closed (explicit route to an unserved profile). - source.profile_route_rejected = True - - profile_home = ( - self._resolve_profile_home_for_source(source) - if getattr(source, "profile", None) - else default_home - ) - async with _async_profile_runtime_scope(profile_home): - return await self._handle_message(event) - - return _handler - - def _primary_message_handler(self): - """Return the correctly scoped handler for a primary adapter.""" - if getattr(self.config, "multiplex_profiles", False): - return self._make_default_profile_message_handler() - return self._handle_message - - async def _handle_gateway_platform_event(self, event: dict, source) -> None: - """Authorize and publish one normalized adapter event to plugin hooks.""" - try: - from hermes_cli.lifecycle import has_hook, invoke_hook - - if not has_hook("gateway_platform_event"): - return - if not self._is_user_authorized_for_source(source): - return - invoke_hook("gateway_platform_event", **event) - except Exception: - # Observer failures must never break the adapter's update loop. - logger.debug("gateway_platform_event hook dispatch failed", exc_info=True) - - def _make_profile_platform_event_handler(self, profile_name: str): - """Bind platform-event auth and hook dispatch to one multiplex profile.""" - from hermes_cli.profiles import get_profile_dir - - try: - profile_home = get_profile_dir(profile_name) - except Exception: - profile_home = None - - async def _handler(event, source): - if getattr(source, "profile", None) is None: - source.profile = profile_name - if profile_home is not None: - with _profile_runtime_scope(profile_home): - return await self._handle_gateway_platform_event(event, source) - return await self._handle_gateway_platform_event(event, source) - - return _handler - - def _make_default_profile_platform_event_handler(self): - """Scope primary-transport events to their routed multiplex profile.""" - default_home = Path(get_hermes_home()) - - async def _handler(event, source): - source._authorization_profile_home = default_home - with _profile_runtime_scope(self._resolve_profile_home_for_source(source)): - return await self._handle_gateway_platform_event(event, source) - - return _handler def _is_user_authorized_for_source( self, @@ -15387,411 +4980,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return _check() return _check() - def _primary_platform_event_handler(self): - if getattr(self.config, "multiplex_profiles", False): - return self._make_default_profile_platform_event_handler() - return self._handle_gateway_platform_event - - @staticmethod - def _adapter_credential_claim( - platform: Platform, adapter: Any - ) -> Optional[tuple]: - """Return the exclusive credential resource claimed by an adapter.""" - fingerprint = GatewayRunner._adapter_credential_fingerprint(adapter) - if fingerprint is None: - return None - return (platform, fingerprint) - - @staticmethod - def _adapter_listener_claim(platform: Platform, adapter: Any) -> Optional[tuple]: - """Return the exclusive listener resource claimed by an adapter. - - Sidecars with different credentials still cannot share a bind+port; expose it as a claim so - multiplex startup rejects the later adapter before connect()/disconnect() disturb the first. - """ - if getattr(platform, "value", None) != "photon": - return None - bind = getattr(adapter, "_sidecar_bind", None) - port = getattr(adapter, "_sidecar_port", None) - if not isinstance(bind, str) or not bind.strip(): - return None - try: - port = int(port) - except (TypeError, ValueError): - return None - return ("listener", "photon", bind.strip().lower(), port) - - @staticmethod - def _adapter_credential_fingerprint(adapter: Any) -> Optional[str]: - """Return a stable, log-safe fingerprint of an adapter's credential. - - Salted hash (never the credential) used to detect two profiles sharing one platform - credential; None when no credential is discoverable (conflict detection is then skipped). - """ - token = None - for attr in ( - "token", - "bot_token", - "_token", - "api_token", - "_bot_token", - # Photon/Spectrum authenticates with project credentials, not a bot token; including - # its secret stops multiplexed profiles spawning rival sidecars for one account/port. - "_project_secret", - # Feishu/Lark authenticates with an app_id/app_secret pair (one WebSocket per app). - # app_id is stable, log-safe and already the adapter's _app_lock_identity, so including - # it lets the multiplex guard refuse cloned profiles competing for the same app. - "_app_id", - # Same class: Teams (client_id/client_secret) and WeCom - # (bot_id/secret) authenticate with an app-style id pair too. - "_client_id", - "_bot_id", - ): - val = getattr(adapter, attr, None) - if isinstance(val, str) and val.strip(): - token = val.strip() - break - # Many adapters (e.g. Discord) store the token on their `config` sub-object. Without this - # lookup they return None, the same-token check is silently skipped, and every profile's - # adapter polls the same bot token — a per-message race over which one answers. - if not token: - cfg = getattr(adapter, "config", None) - if cfg is not None: - for attr in ("token", "bot_token"): - val = getattr(cfg, attr, None) - if isinstance(val, str) and val.strip(): - token = val.strip() - break - if not token: - config = getattr(adapter, "config", None) - val = getattr(config, "token", None) - if isinstance(val, str) and val.strip(): - token = val.strip() - if not token: - return None - import hashlib - return hashlib.sha256(("hermes-mux:" + token).encode("utf-8")).hexdigest()[:16] - - def _create_adapter( - self, - platform: Platform, - config: Any, - ) -> Optional[BasePlatformAdapter]: - """Create an adapter and bind it to this gateway runner. - - Every lifecycle path (primary/secondary startup, reconnect) uses this method; keep runner - binding here so adapters can resolve inbound profile routes before handlers or connect(). - """ - adapter = self._instantiate_adapter(platform, config) - if adapter is not None: - adapter.gateway_runner = self - return adapter - - def _instantiate_adapter( - self, - platform: Platform, - config: Any, - ) -> Optional[BasePlatformAdapter]: - """Instantiate the appropriate adapter for a platform. - - Checks platform_registry (plugin adapters) first, then the built-in table of core platforms. - """ - if hasattr(config, "extra") and isinstance(config.extra, dict): - config.extra.setdefault( - "group_sessions_per_user", - self.config.group_sessions_per_user, - ) - config.extra.setdefault( - "thread_sessions_per_user", - getattr(self.config, "thread_sessions_per_user", False), - ) - - # ── Plugin-registered platforms (checked first) ─────────────────── - try: - from gateway.platform_registry import platform_registry - if platform_registry.is_registered(platform.value): - adapter = platform_registry.create_adapter(platform.value, config) - if adapter is not None: - return adapter - # Registered but failed to instantiate — don't silently fall - # through to built-ins (there are none for plugin platforms). - logger.error( - "Platform '%s' is registered but adapter creation failed " - "(check dependencies and config)", - platform.value, - ) - return None - except Exception as e: - logger.debug("Platform registry lookup for '%s' failed: %s", platform.value, e) - # Fall through to built-in adapters below - - return _instantiate_builtin_adapter(platform, config) - - def _make_adapter_auth_check( - self, - platform: Platform, - profile_name: Optional[str] = None, - ) -> Callable[[str, Optional[str], Optional[str]], bool]: - """Build a platform-bound auth callback for adapter use. - - Adapters fetching external context (e.g. Slack ``conversations.replies``) use it via - ``_is_sender_authorized`` to mark non-allowlisted senders unverified (prompt-injection - mitigation). Delegates to :meth:`_is_user_authorized` so the full auth chain stays the single - source of truth. ``profile_name`` binds a secondary adapter to its own secret scope; for the - shared primary (None) the ``profile_routes`` match is stamped on the source so the routed - profile's pairing store is consulted while allowlist reads stay under the transport home. - """ - multiplex = bool(getattr(self.config, "multiplex_profiles", False)) - transport_home = ( - Path(get_hermes_home()) if multiplex and profile_name is None else None - ) - - def check( - user_id: str, - chat_type: Optional[str] = None, - chat_id: Optional[str] = None, - *, - is_bot: bool = False, - thread_id: Optional[str] = None, - ) -> bool: - if not user_id: - return False - source = SessionSource( - platform=platform, - chat_id=chat_id or "", - chat_type=chat_type or "group", - user_id=user_id, - thread_id=thread_id, - is_bot=bool(is_bot), - profile=profile_name, - ) - # Same in-process transport provenance ``build_source`` retains, so adapter-level policy - # reads (config.yaml group_allowed_chats, allow_from) resolve the receiving adapter even - # once the routed profile is stamped below. - registry = ( - (getattr(self, "_profile_adapters", None) or {}).get(profile_name) - if profile_name - else getattr(self, "adapters", None) - ) or {} - adapter = registry.get(platform) - if adapter is not None: - source._transport_adapter_ref = _weakref.ref(adapter) - if transport_home is None: - return self._is_user_authorized(source) - source._authorization_profile_home = transport_home - from gateway.profile_routing import ProfileRouteRejected - - try: - source.profile = self._profile_name_for_source(source) - except ProfileRouteRejected: - # Same fail-closed outcome as the ingress gate in - # ``_handle_message`` for a route to an unserved profile. - return False - return self._is_user_authorized_for_source(source) - return check - - async def _deliver_platform_notice(self, source, content: str) -> None: - """Deliver a setup/operational notice using platform-specific privacy rules.""" - adapter = self._adapter_for_source(source) - if not adapter: - return - - config = getattr(self, "config", None) - if ( - config - and getattr(source, "platform", None) == Platform.SLACK - and _is_slack_ignored_channel(config, getattr(source, "chat_id", None)) - ): - logger.info( - "Skipping Slack platform notice for configured ignored channel %s", - getattr(source, "chat_id", None), - ) - return - - notice_delivery = "public" - if config and hasattr(config, "get_notice_delivery"): - notice_delivery = config.get_notice_delivery(source.platform) - - metadata = self._thread_metadata_for_source(source) - if notice_delivery == "private" and getattr(source, "user_id", None): - try: - result = await adapter.send_private_notice( - source.chat_id, - source.user_id, - content, - metadata=metadata, - ) - if getattr(result, "success", False): - return - except Exception: - logger.debug( - "[%s] send_private_notice failed, falling back to public", - getattr(source, "platform", "?"), - exc_info=True, - ) - - await adapter.send(source.chat_id, content, metadata=metadata) - - async def _resolve_async_delegation_session( - self, - session_entry: SessionEntry, - pinned_session_id: str, - ) -> Optional[SessionEntry]: - """Resolve an async completion to its verified owning gateway session. - - Follow compression-rotation lineage (parent row ended, child continues), but never let a - late completion override an unrelated /new or restored route. Unknown ownership fails - closed; the result stays in the delegation records. - """ - session_db = cast(Any, self._session_db) - if session_db is None: - logger.warning( - "Async-delegation completion has no session database; " - "dropping injection (#55578 fail-closed)." - ) - return None - - pinned_row = None - try: - pinned_row = await session_db.get_session(pinned_session_id) - except Exception: - logger.debug( - "Async-delegation parent lookup failed for %s", - pinned_session_id, - exc_info=True, - ) - - if pinned_row is None: - logger.warning( - "Async-delegation completion has unknown spawning session %s; " - "dropping injection (#55578 fail-closed).", - pinned_session_id, - ) - return None - - target_session_id = pinned_session_id - follows_compression = False - if pinned_row.get("ended_at"): - _end_reason = str(pinned_row.get("end_reason") or "") - if _end_reason in _USER_BOUNDARY_END_REASONS: - logger.warning( - "Async-delegation completion pinned to user-closed session %s " - "(end_reason=%r); dropping injection instead of resurrecting it " - "(#55578 fail-closed).", - pinned_session_id, - _end_reason, - ) - return None - if _end_reason != "compression": - # Idle/timeout/lifecycle end (scale-to-zero norm): the chat route is still valid and - # ``session_entry`` is its current session, so deliver here rather than drop — otherwise - # the row is acked at adapter acceptance then silently lost. - logger.info( - "Async-delegation completion pinned to %s-ended session %s; " - "retargeting to the chat's current session %s.", - _end_reason or "idle", - pinned_session_id, - session_entry.session_id, - ) - return session_entry - - follows_compression = True - try: - target_session_id = await session_db.get_compression_tip( - pinned_session_id - ) - except Exception: - logger.debug( - "Async-delegation compression-tip lookup failed for %s", - pinned_session_id, - exc_info=True, - ) - target_session_id = None - - if not target_session_id or target_session_id == pinned_session_id: - logger.warning( - "Async-delegation completion pinned to compressed session %s " - "without a continuation; dropping injection.", - pinned_session_id, - ) - return None - - try: - tip_row = await session_db.get_session(target_session_id) - except Exception: - tip_row = None - if tip_row is None or tip_row.get("ended_at"): - logger.warning( - "Async-delegation compression continuation %s is %s; " - "dropping injection.", - target_session_id, - "unknown" if tip_row is None else "ended", - ) - return None - - route_owns_lineage = session_entry.session_id in { - pinned_session_id, - target_session_id, - } - if not route_owns_lineage: - # A long-running delegation may survive multiple compression - # rotations. Accept an intermediate stale route only when its - # own verified compression tip is the same live target. - try: - route_row = await session_db.get_session(session_entry.session_id) - route_tip = ( - await session_db.get_compression_tip(session_entry.session_id) - if route_row is not None - and route_row.get("ended_at") - and route_row.get("end_reason") == "compression" - else None - ) - except Exception: - route_tip = None - route_owns_lineage = route_tip == target_session_id - - if not route_owns_lineage: - logger.warning( - "Async-delegation completion for compression lineage %s -> %s " - "does not own current route %s; dropping injection.", - pinned_session_id, - target_session_id, - session_entry.session_id, - ) - return None - - if target_session_id == session_entry.session_id: - return session_entry - - prior_session_id = session_entry.session_id - if follows_compression: - switched = await self.async_session_store.advance_compression_session( - session_entry.session_key, - prior_session_id, - target_session_id, - ) - else: - switched = await self.async_session_store.switch_session( - session_entry.session_key, - target_session_id, - ) - if switched is None: - logger.warning( - "Async-delegation completion could not bind routing key %s to " - "owning session %s; dropping injection.", - session_entry.session_key, - target_session_id, - ) - return None - - logger.info( - "Pinned async-delegation completion to owning session %s " - "(was %s) for routing key %s (#57498)", - target_session_id, - prior_session_id, - session_entry.session_key, - ) - return switched # ------------------------------------------------------------------ # Mid-run (busy-session) slash command dispatch — "Guard 2". @@ -15808,2129 +4996,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "moa": "Agent is running — wait or /stop first, then run /moa.", } - def _gateway_plain_command_handlers(self): - """Return ordinary slash handlers shared by idle and busy dispatch.""" - return { - "status": self._handle_status_command, - "context": self._handle_context_command, - "restart": self._handle_restart_command, - "approve": self._handle_approve_command, - "deny": self._handle_deny_command, - "pause": self._handle_pause_command, - "agents": self._handle_agents_command, - "bg": self._handle_background_command, - "btw": self._handle_btw_command, - "kanban": self._handle_kanban_command, - "subgoal": self._handle_subgoal_command, - "heartbeat": self._handle_heartbeat_command, - "busy": self._handle_busy_command, - "yolo": self._handle_yolo_command, - "verbose": self._handle_verbose_command, - "footer": self._handle_footer_command, - "help": self._handle_help_command, - "commands": self._handle_commands_command, - "profile": self._handle_profile_command, - "update": self._handle_update_command, - "version": self._handle_version_command, - } - - async def _send_command_ack(self, source, text: str, label: str) -> None: - """Best-effort acknowledgment for a slash command that falls through to agent processing.""" - try: - adapter = self._adapter_for_source(source) - if adapter: - await adapter.send( - str(source.chat_id), text, metadata=self._thread_metadata_for_source(source) - ) - except Exception: - logger.debug("%s ack send failed", label, exc_info=True) - - def _gateway_idle_command_handlers(self): - """Slash handlers dispatched only when no agent is running for the session (idle path). - - Busy dispatch keeps its own explicit allowlist (``_dispatch_busy_slash_command``).""" - return { - "topic": self._handle_topic_command, - "whoami": self._handle_whoami_command, - "platform": self._handle_platform_command, - "stop": self._handle_stop_command, - "reasoning": self._handle_reasoning_command, - "memory": self._handle_memory_command, - "skills": self._handle_skills_command, - "fast": self._handle_fast_command, - "approvals": self._handle_approvals_command, - "model": self._handle_model_command, - "codex-runtime": self._handle_codex_runtime_command, - "personality": self._handle_personality_command, - "suggestions": self._handle_suggestions_command, - "save": self._handle_save_command, - "retry": self._handle_retry_command, - "sethome": self._handle_set_home_command, - "compress": self._handle_compress_command, - "usage": self._handle_usage_command, - "topup": self._handle_topup_command, - "insights": self._handle_insights_command, - "reload-mcp": self._handle_reload_mcp_command, - "reload-skills": self._handle_reload_skills_command, - "bundles": self._handle_bundles_command, - "debug": self._handle_debug_command, - "title": self._handle_title_command, - "resume": self._handle_resume_command, - "sessions": self._handle_sessions_command, - "branch": self._handle_branch_command, - "rollback": self._handle_rollback_command, - "diff": self._handle_diff_command, - "goal": self._handle_goal_command, - "loop": self._handle_loop_command, - "refine": self._handle_refine_command, - "review": self._handle_review_command, - "voice": self._handle_voice_command, - } - - async def _dispatch_busy_slash_command( - self, event: MessageEvent, cmd_def, quick_key: str, source, - ): - """Dispatch a recognized slash command while an agent is running. - - Order: ``busy_handler`` (special mid-run variant) → ``busy_policy == "dispatch"`` (normal - handler) → catch-all busy-reject text. Rejecting beats falling through to interrupt + - discard: Discord-registered slash commands would interrupt the agent AND be discarded by - the slash-command safety net, producing a zero-char response. - """ - name = cmd_def.name - policy = getattr(cmd_def, "busy_policy", "reject") - handler_key = getattr(cmd_def, "busy_handler", None) - - if handler_key: - special = { - "start": self._busy_start_command, - "stop": self._busy_stop_command, - "new": self._busy_new_command, - "queue": self._busy_queue_command, - "steer": self._busy_steer_command, - "egress": self._busy_egress_command, - "goal": self._busy_goal_command, - "loop": self._busy_loop_command, - }.get(handler_key) - if special is not None: - return await special(event, quick_key, source) - reject_text = self._BUSY_REJECT_TEXT.get(handler_key) - if reject_text is not None: - return reject_text - - if policy in ("dispatch", "interrupt_then_dispatch"): - plain = self._gateway_plain_command_handlers().get(name) - if plain is not None: - return await plain(event) - logger.warning( - "busy_policy=%s for /%s has no mid-run handler — " - "falling back to busy-reject", policy, name, - ) - - # Catch-all: any other recognized slash command hit the running-agent guard — reject - # gracefully rather than falling through to interrupt + discard. - return ( - f"⏳ Agent is running — `/{name}` can't run " - f"mid-turn. Wait for the current response or `/stop` first." - ) - - async def _handle_pause_command(self, event: MessageEvent): - """`/pause [reason]` engages the global emergency stop; `/pause off` (resume/stop) lifts it. - - In-band resume path for messaging-only operators — the estop gate lets recognized slash - commands through while paused so a user without host-shell access is never locked out. - """ - from agent import estop - - args = (event.get_command_args() or "").strip() - if args.lower() in {"off", "resume", "stop", "disengage"}: - if estop.disengage(): - return "▶️ Resumed — new work is accepted again." - return "Hermes wasn't paused." - state = estop.get_state() - if state is not None and not args: - reason = state.get("reason") - suffix = f" (reason: {reason})" if reason else "" - return ( - f"⏸️ Hermes is already paused{suffix}. " - "Use `/pause off` to resume." - ) - estop.engage(reason=args or None) - suffix = f" (reason: {args})" if args else "" - return ( - f"⏸️ Paused{suffix}. New cron/kanban/gateway work is on hold; " - "in-flight work finishes normally. Use `/pause off` to resume." - ) - - async def _busy_start_command(self, event: MessageEvent, quick_key: str, source): - # Telegram sends /start for bot launches/deep-links — a platform ping, not a user command: - # no help dump, no agent interrupt, no queued text. - logger.info("Ignoring /start platform ping for active session %s", quick_key) - return "" - - async def _busy_egress_command(self, event: MessageEvent, quick_key: str, source): - from hermes_cli.proxy_cli import format_status_text - - return format_status_text() - - async def _busy_stop_command(self, event: MessageEvent, quick_key: str, source): - # /stop must hard-kill the session when an agent is running. A soft interrupt - # (agent.interrupt()) doesn't help when the agent is truly hung — the executor thread is - # blocked and never checks _interrupt_requested. - await self._interrupt_and_clear_session( - quick_key, - source, - interrupt_reason=_INTERRUPT_REASON_STOP, - invalidation_reason="stop_command", - ) - logger.info("STOP for session %s — agent interrupted, session lock released", quick_key) - return EphemeralReply(t("gateway.stop.stopped")) - - async def _busy_new_command(self, event: MessageEvent, quick_key: str, source): - # /reset and /new must bypass the running-agent guard so they actually dispatch as commands - # instead of being queued as user text (which would be fed back to the agent with the same - # broken history — #2170). Clear any pending messages so the old text doesn't replay - await self._interrupt_and_clear_session( - quick_key, - source, - interrupt_reason=_INTERRUPT_REASON_RESET, - invalidation_reason="new_command", - ) - # Clean up the running agent entry so the reset handler - # doesn't think an agent is still active. - return await self._handle_reset_command(event) - - async def _busy_queue_command(self, event: MessageEvent, quick_key: str, source): - # /queue — queue without interrupting. Each /queue is its own full agent turn, run - # FIFO after the current run (and earlier /queue items) finish; messages are NOT merged. - queued_text = event.get_command_args().strip() - # Preserve media/reply payloads: a /queue carrying a photo, document, or reply context is - # valid even with no prompt text (e.g. "/queue" as the caption of an image). Dropping these - # fields silently lost the attachment when the queued turn ran. - has_media = bool(getattr(event, "media_urls", None)) - if not queued_text and not has_media: - return "Usage: /queue " - adapter = self._adapter_for_source(source) - if adapter: - queued_event = MessageEvent( - text=queued_text, - message_type=event.message_type if has_media else MessageType.TEXT, - source=event.source, - raw_message=event.raw_message, - message_id=event.message_id, - media_urls=list(getattr(event, "media_urls", []) or []), - media_types=list(getattr(event, "media_types", []) or []), - media_text_inlined=list(getattr(event, "media_text_inlined", []) or []), - reply_to_message_id=event.reply_to_message_id, - reply_to_text=event.reply_to_text, - reply_to_author_id=event.reply_to_author_id, - reply_to_author_name=event.reply_to_author_name, - reply_to_is_own_message=event.reply_to_is_own_message, - auto_skill=event.auto_skill, - channel_prompt=event.channel_prompt, - channel_context=event.channel_context, - internal=event.internal, - timestamp=event.timestamp, - ) - self._enqueue_fifo(quick_key, queued_event, adapter) - depth = self._queue_depth(quick_key, adapter=self._adapter_for_source(source)) - if depth <= 1: - return "Queued for the next turn." - return f"Queued for the next turn. ({depth} queued)" - - async def _busy_steer_command(self, event: MessageEvent, quick_key: str, source): - # /steer — inject mid-run after the next tool call. Unlike /queue (turn boundary), - # /steer lands BETWEEN tool-call iterations inside the same agent run, by appending to the - # last tool result's content. No interrupt, no new user turn, no role-alternation violation. - steer_text = event.get_command_args().strip() - if not steer_text: - return "Usage: /steer " - _steer_state = self._peek_session_state(quick_key) - running_agent = _steer_state.turn.agent if _steer_state else None - if running_agent is _AGENT_PENDING_SENTINEL: - # Agent hasn't started yet — queue as turn-boundary fallback. - adapter = self._adapter_for_source(source) - if adapter: - queued_event = MessageEvent( - text=steer_text, - message_type=MessageType.TEXT, - source=event.source, - message_id=event.message_id, - channel_prompt=event.channel_prompt, - channel_context=event.channel_context, - ) - self._enqueue_fifo(quick_key, queued_event, adapter) - return "Agent still starting — /steer queued for the next turn." - if running_agent and hasattr(running_agent, "steer"): - try: - accepted = running_agent.steer(steer_text) - except Exception as exc: - logger.warning("Steer failed for session %s: %s", quick_key, exc) - return f"⚠️ Steer failed: {exc}" - if accepted: - preview = steer_text[:60] + ("..." if len(steer_text) > 60 else "") - return f"⏩ Steer queued — arrives after the next tool call: '{preview}'" - return "Steer rejected (empty payload)." - # Running agent is missing or lacks steer() — fall back to queue. - adapter = self._adapter_for_source(source) - if adapter: - queued_event = MessageEvent( - text=steer_text, - message_type=MessageType.TEXT, - source=event.source, - message_id=event.message_id, - channel_prompt=event.channel_prompt, - channel_context=event.channel_context, - ) - self._enqueue_fifo(quick_key, queued_event, adapter) - return "No active agent — /steer queued for the next turn." - - async def _busy_goal_command(self, event: MessageEvent, quick_key: str, source): - # /goal is safe mid-run for status/pause/clear/wait (inspection and control-plane only — - # doesn't interrupt the running turn). Setting new goal text mid-run is rejected like - # /model so we don't race a second continuation prompt against the current turn. - _goal_arg = (event.get_command_args() or "").strip().lower() - _goal_verb = _goal_arg.split(None, 1)[0] if _goal_arg else "" - # Exact-match control verbs, plus the wait/unwait barrier verbs (take a pid) and the gate - # management verb (gates run at turn boundary, so editing the gate list mid-run is safe). - _is_control = ( - not _goal_arg - or _goal_arg in {"status", "pause", "resume", "clear", "stop", "done", "unwait"} - or _goal_verb in {"wait", "gate"} - ) - if _is_control: - return await self._handle_goal_command(event) - return "Agent is running — use /goal status / pause / clear / wait mid-run, or /stop before setting a new goal." - - async def _busy_loop_command(self, event: MessageEvent, quick_key: str, source): - # /loop mirrors /goal: control verbs are safe mid-run (state only — read at the next idle - # boundary); setting a new loop mid-run is rejected so we don't race the current turn. - _loop_arg = (event.get_command_args() or "").strip().lower() - if not _loop_arg or _loop_arg in {"status", "pause", "resume", "stop", "clear", "cancel", "help", "--help", "-h"}: - return await self._handle_loop_command(event) - return "Agent is running — use /loop status / pause / stop mid-run, or /stop before setting a new loop." - - async def _handle_message(self, event: MessageEvent) -> Optional[str]: - """Handle an incoming message from any platform. - - Pipeline: auth → command check → running-agent interrupt → get/create session → build - context → run agent → return response. - """ - source = event.source - - # 🔴 Cross-session leak guard. This per-message task was created via create_task(), which - # copies the spawning context: if a concurrent message had already bound its session via - # set_session_vars(), we inherited ITS HERMES_SESSION_* ContextVars, and until _set_session_env - # binds ours any subprocess would read the foreign identity (the _UNSET-strip guard can't - # help — the vars are set-to-foreign). Reset to _UNSET so that window strips safe instead. - try: - from gateway.session_context import reset_session_vars - reset_session_vars() - except Exception: - logger.debug("reset_session_vars failed at handler entry", exc_info=True) - - # Most adapters resolve profile routes in build_source(), before they hand us the event. A - # few internal/voice paths construct SessionSource directly, so resolve those here as the - # shared fail-closed ingress gate before authorization, hooks, or session side effects. - if ( - getattr(getattr(self, "config", None), "multiplex_profiles", False) - and not getattr(source, "profile", None) - and getattr(source, "profile_route_rejected", False) is not True - ): - from gateway.profile_routing import ProfileRouteRejected - - try: - source.profile = self._profile_name_for_source(source) - except ProfileRouteRejected: - source.profile_route_rejected = True - - # SessionSource owns a strict boolean marker. Require the literal value - # so duck-typed test/internal sources with dynamic attributes are not - # mistaken for an explicit matched-route rejection. - if getattr(source, "profile_route_rejected", False) is True: - logger.warning( - "Dropping inbound message because its explicit profile route " - "targets an unserved profile" - ) - return None - - # Internal events (e.g. background-process completion notifications) - # are system-generated and must skip user authorization. - is_internal = bool(getattr(event, "internal", False)) - - # Ignored-channel guard runs FIRST — before startup-restore queueing, plugin hooks, auth, - # and session setup — so an ignored channel can never reach pairing/auth/session state. - # getattr: bare test runners construct GatewayRunner via object.__new__ without config. - if ( - not is_internal - and getattr(source, "platform", None) == Platform.SLACK - and _is_slack_ignored_channel( - getattr(self, "config", None), getattr(source, "chat_id", None) - ) - ): - logger.info( - "Dropping Slack message from configured ignored channel %s", - getattr(source, "chat_id", None), - ) - return None - - if ( - getattr(self, "_startup_restore_in_progress", False) - and not is_internal - and not getattr(event, "_hermes_startup_restore_replay", False) - ): - self._queue_startup_restore_event(event) - return None - - # scale-to-zero: stamp the gateway-scoped last-inbound clock (read by is_idle) for real - # user-originated inbound only. Internal/system events are NOT traffic — counting them - # would keep a genuinely idle gateway awake. - if not is_internal: - self._scale_to_zero_note_real_inbound() - - # pre_gateway_dispatch plugin hook (user-originated only). Plugins may return - # {"action": "skip", "reason": ...} -> drop; {"action": "rewrite", "text": ...} -> replace - # event.text; {"action": "allow"} / None -> normal dispatch. - # Runs BEFORE auth so plugins can handle unauthorized senders without the pairing flow. - if not is_internal: - try: - from hermes_cli.lifecycle import invoke_hook as _invoke_hook - _hook_results = _invoke_hook( - "pre_gateway_dispatch", - event=event, - gateway=self, - # getattr: bare-runner tests build GatewayRunner via object.__new__ without - # __init__; the hook must not fail dispatch over a missing attribute. - session_store=getattr(self, "session_store", None), - ) - except Exception as _hook_exc: - logger.warning("pre_gateway_dispatch invocation failed: %s", _hook_exc) - _hook_results = [] - - for _result in _hook_results: - if not isinstance(_result, dict): - continue - _action = _result.get("action") - if _action == "skip": - logger.info( - "pre_gateway_dispatch skip: reason=%s platform=%s chat=%s", - _result.get("reason"), - source.platform.value if source.platform else "unknown", - source.chat_id or "unknown", - ) - return None - if _action == "rewrite": - _new_text = _result.get("text") - if isinstance(_new_text, str): - event = dataclasses.replace(event, text=_new_text) - source = event.source - break - if _action == "allow": - break - - if is_internal: - pass - elif source.user_id is None: - # Messages with no user identity (Telegram service messages, channel forwards, anonymous - # admin posts, sender_chat) can't be paired but may be authorized via a chat-scoped - # allowlist (e.g. TELEGRAM_GROUP_ALLOWED_CHATS), so defer to _is_user_authorized. - if not self._is_user_authorized_for_source(source): - logger.debug("Ignoring message with no user_id from %s", source.platform.value) - return None - elif not self._is_user_authorized_for_source(source): - logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value) - # In DMs: offer pairing code. In groups: silently ignore. - if ( - source.chat_type == "dm" - and self._get_unauthorized_dm_behavior( - source.platform, - profile=source.profile, - ) - == "pair" - ): - platform_name = source.platform.value if source.platform else "unknown" - pairing_store = self._pairing_store_for(source) - if pairing_store is None: - logger.error( - "Cannot offer pairing code on %s: no pairing store", - platform_name, - ) - return None - # Rate-limit ALL pairing responses (code or rejection) so a burst of DMs doesn't - # spam the user with repeated messages. - if pairing_store._is_rate_limited(platform_name, source.user_id): - return None - code = pairing_store.generate_code( - platform_name, source.user_id, source.user_name or "" - ) - if code: - adapter = self._adapter_for_source(source) - if adapter: - store_profile = getattr(pairing_store, "profile", None) - profile_arg = ( - f"-p {store_profile} " - if isinstance(store_profile, str) - and store_profile - and store_profile != "default" - else "" - ) - await adapter.send( - source.chat_id, - f"Hi~ I don't recognize you yet!\n\n" - f"Here's your pairing code: `{code}`\n\n" - f"Ask the bot owner to run:\n" - f"`hermes {profile_arg}pairing approve " - f"{platform_name} {code}`" - ) - else: - adapter = self._adapter_for_source(source) - if adapter: - await adapter.send( - source.chat_id, - "Too many pairing requests right now~ " - "Please try again later!" - ) - # Record rate limit so subsequent messages are silently ignored - pairing_store._record_rate_limit(platform_name, source.user_id) - return None - - # Global emergency stop (`hermes pause`): new turns get a brief paused notice instead of an - # agent run. Placed after auth so unauthorized senders can't probe pause state. Pause blocks - # NEW agent turns, never running work or control traffic, so these pass through: internal - # events from IN-FLIGHT work; recognized slash commands (/status, /approve, ... and /pause off - # as the in-band resume path); replies owned by in-flight work — pending update prompt, - # clarify, slash-confirm, dangerous-command approval, or steering an already-running session. - if not is_internal: - try: - from agent.estop import paused_reply as _estop_paused_reply - _paused_notice = _estop_paused_reply() - except ImportError: - _paused_notice = None - if _paused_notice is not None: - _estop_allow = False - _estop_cmd = None - try: - _estop_cmd = event.get_command() - except Exception: - _estop_cmd = None - if _estop_cmd: - try: - from hermes_cli.commands import ( - resolve_command as _resolve_estop_cmd, - ) - _estop_allow = _resolve_estop_cmd(_estop_cmd) is not None - except Exception: - _estop_allow = False - if not _estop_allow: - try: - _estop_key = self._session_key_for_source(source) - _estop_state = self._peek_session_state(_estop_key) - if ( - _estop_state is not None - and _estop_state.persistent.update_prompt_pending - ): - _estop_allow = True - if not _estop_allow and self._is_session_running(_estop_key): - # Steering / interrupting in-flight work (also covers pending clarify + - # tool approvals held by the running agent). - _estop_allow = True - if not _estop_allow: - from tools import slash_confirm as _estop_confirm_mod - if _estop_confirm_mod.get_pending(_estop_key): - _estop_allow = True - if not _estop_allow: - from tools.approval import ( - has_blocking_approval as _estop_has_approval, - ) - if _estop_has_approval(_estop_key): - _estop_allow = True - except Exception: - pass - if not _estop_allow: - logger.info( - "Gateway turn paused by global emergency stop (platform=%s chat=%s)", - getattr(getattr(source, "platform", None), "value", "unknown"), - getattr(source, "chat_id", None) or "unknown", - ) - return _paused_notice - - # Route replies to a pending /update prompt back to the detached update process via - # .update_response. Recognized slash commands must bypass this or /new, /help etc. get - # silently consumed as update answers. - _quick_key = self._session_key_for_source(source) - allow_gateway_control = event.allow_gateway_control - _up_state = self._peek_session_state(_quick_key) - if ( - allow_gateway_control - and _up_state is not None - and _up_state.persistent.update_prompt_pending - ): - raw = (event.text or "").strip() - # Accept /approve and /deny as shorthand for yes/no - cmd = event.get_command() - if cmd in {"approve", "yes"}: - response_text = "y" - elif cmd in {"deny", "no"}: - response_text = "n" - else: - _recognized_cmd = None - if cmd: - try: - from hermes_cli.commands import resolve_command as _resolve_update_cmd - except Exception: - _resolve_update_cmd = None - if _resolve_update_cmd is not None: - try: - _cmd_def = _resolve_update_cmd(cmd) - _recognized_cmd = _cmd_def.name if _cmd_def else None - except Exception: - _recognized_cmd = None - response_text = "" if _recognized_cmd else raw - if response_text: - response_path = _hermes_home / ".update_response" - prompt_path = _hermes_home / ".update_prompt.json" - try: - tmp = response_path.with_suffix(".tmp") - tmp.write_text(response_text, encoding="utf-8") - tmp.replace(response_path) - prompt_path.unlink(missing_ok=True) - except OSError as e: - logger.warning("Failed to write update response: %s", e) - return f"✗ Failed to send response to update process: {e}" - _up_state.persistent.update_prompt_pending = False - label = response_text if len(response_text) <= 20 else response_text[:20] + "…" - return f"✓ Sent `{label}` to the update process." - # Recognized slash command during a pending update prompt: write a blank response so the - # detached update's ``_gateway_prompt`` returns the prompt's default (typically a safe - # "n" / skip) and exits instead of blocking on stdin until the watcher timeout. - if _recognized_cmd: - response_path = _hermes_home / ".update_response" - prompt_path = _hermes_home / ".update_prompt.json" - try: - tmp = response_path.with_suffix(".tmp") - tmp.write_text("", encoding="utf-8") - tmp.replace(response_path) - prompt_path.unlink(missing_ok=True) - logger.info( - "Recognized /%s during pending update prompt for %s; " - "cancelled prompt with default and dispatching command", - _recognized_cmd, - _quick_key, - ) - except OSError as e: - logger.warning( - "Failed to write cancel response for pending update prompt: %s", - e, - ) - _up_state.persistent.update_prompt_pending = False - - # Intercept replies to a pending clarify: open-ended prompts and "Other" responses are free - # text; direct replies to multi-choice prompts are accepted too ("2" → second option). - _clarify_mod = None - try: - from tools import clarify_gateway as _clarify_mod - _pending_clarify = _clarify_mod.get_pending_for_session( - _quick_key, include_choice_prompts=True, - ) - except Exception: - _pending_clarify = None - if ( - allow_gateway_control - and _pending_clarify is not None - and _clarify_mod is not None - ): - _clarify_has_audio = bool(self._pending_event_audio_paths(event)) - _raw_clarify_reply = await self._prepare_clarify_reply_text(event) - if _clarify_has_audio and not _raw_clarify_reply: - logger.info( - "Gateway retained pending clarify after voice transcription " - "produced no usable text (session=%s, id=%s)", - _quick_key, - _pending_clarify.clarify_id, - ) - return "" - # Skip slash commands — the user wanted a command, not to answer the clarify. Leave it - # pending so they can retry; on timeout the agent unblocks with an empty response. - if _raw_clarify_reply and not _raw_clarify_reply.startswith("/"): - _text_outcome = _clarify_mod.attempt_text_response_for_session( - _quick_key, _raw_clarify_reply, - ) - if _text_outcome == _clarify_mod.TEXT_RESOLVED: - logger.info( - "Gateway intercepted clarify text response (session=%s, id=%s)", - _quick_key, _pending_clarify.clarify_id, - ) - # The clarify callback pauses the platform typing/status indicator while waiting - # so Slack users can type their answer. The active agent resumes as soon as this - # reply resolves the wait, so re-enable its indicator here too. - _clarify_adapter = self._adapter_for_source(source) - if _clarify_adapter: - try: - _clarify_adapter.resume_typing_for_chat(source.chat_id) - except Exception: - logger.debug( - "Failed to resume typing after clarify response", - exc_info=True, - ) - # Acknowledge with empty string so adapters that emit the agent's response don't - # double-post; the agent itself produces the next user-facing message. - return "" - if _text_outcome == _clarify_mod.TEXT_REJECTED_SELECTION: - # Selection-shaped but invalid (out-of-range number, bad comma-list): keep the - # clarify armed for retry — don't cancel, don't treat as an unrelated follow-up. - logger.info( - "Gateway retained pending clarify after invalid " - "selection attempt (session=%s, id=%s)", - _quick_key, _pending_clarify.clarify_id, - ) - return "" - if _text_outcome == _clarify_mod.TEXT_REJECTED_PROSE: - # Native-choice prompts deliberately reject unmatched prose so it can continue - # through normal busy-message routing. Release this clarify first: redirect() - # degrades to steer() while tools execute, and that steer cannot drain until - # the clarify tool returns. - _clarify_mod.resolve_gateway_clarify( - _pending_clarify.clarify_id, - "", - ) - - # Replies to a pending slash-confirm prompt (/reload-mcp etc.): /approve, /always, /cancel and - # short aliases. Anything else falls through — a stale pending confirm does NOT block other - # commands. A pending dangerous-command approval takes precedence: /approve there unblocks - # the waiting tool thread; slash-confirm only catches it when no tool approval is live. - from tools import slash_confirm as _slash_confirm_mod - _pending_confirm = _slash_confirm_mod.get_pending(_quick_key) - _tool_approval_live = False - try: - from tools.approval import has_blocking_approval - _tool_approval_live = has_blocking_approval(_quick_key) - except Exception: - _tool_approval_live = False - if allow_gateway_control and _pending_confirm and not _tool_approval_live: - _raw_reply = (event.text or "").strip() - # Accept bang-prefixed replies (`!always`, `!cancel`) verbatim: Slack/Matrix show the - # `!` prefix (typed `/` is blocked in Slack threads) and adapters only rewrite - # `!` — confirm keywords aren't commands, so the `!` survives to here. - _norm_reply = _raw_reply.lstrip("!/").lower() - _cmd_reply = event.get_command() - _confirm_choice = None - if _cmd_reply in {"approve", "yes", "ok", "confirm"}: - _confirm_choice = "once" - elif _cmd_reply in {"always", "remember"}: - _confirm_choice = "always" - elif _cmd_reply in {"cancel", "no", "deny", "nevermind"}: - _confirm_choice = "cancel" - elif _norm_reply in {"approve", "approve once", "once"}: - _confirm_choice = "once" - elif _norm_reply in {"always", "always approve"}: - _confirm_choice = "always" - elif _norm_reply in {"cancel", "nevermind", "no"}: - _confirm_choice = "cancel" - if _confirm_choice is not None: - _resolved = await _slash_confirm_mod.resolve( - _quick_key, _pending_confirm.get("confirm_id"), _confirm_choice, - ) - return _resolved or "" - # Stale pending + unrelated command: the user moved on, so drop the pending state rather - # than let the confirm block normal usage indefinitely. - _slash_confirm_mod.clear_if_stale(_quick_key) - - # PRIORITY handling when an agent is already running for this session. Default behavior is - # to interrupt immediately so user text/stop messages are handled with minimal latency. - # Exception: Telegram photo bursts arrive as near-simultaneous updates — do NOT interrupt - # for photo-only follow-ups; adapter-level batching absorbs them. - - # Staleness eviction: detect leaked locks from hung/crashed handlers. With inactivity-based - # timeout active tasks can run for hours, so evict only when the agent has been *idle* past - # the threshold (or has no activity tracker and its wall-clock age is extreme). - _raw_stale_timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) - _quick_state = self._peek_session_state(_quick_key) - _stale_ts = _quick_state.turn.started_ts if _quick_state else 0 - if _quick_state is not None and _quick_state.turn.agent is not None and _stale_ts: - _stale_age = time.time() - _stale_ts - _stale_agent = _quick_state.turn.agent - # Never evict the pending sentinel — it was just placed during async setup before the - # real agent exists. Sentinels have no get_activity_summary(), so the idle check would - # read inf >= timeout and evict them immediately, racing the setup path. - _stale_idle = float("inf") # assume idle if we can't check - _stale_detail = "" - _activity_summary_valid = False - if _stale_agent and hasattr(_stale_agent, "get_activity_summary"): - try: - _sa = _stale_agent.get_activity_summary() - from gateway.session_stall import ( - resolve_session_idle_seconds_from_activity, - ) - - _resolved_idle = resolve_session_idle_seconds_from_activity( - _sa if isinstance(_sa, dict) else None, - now=time.time(), - ) - if _resolved_idle is not None: - _stale_idle = _resolved_idle - _activity_summary_valid = True - _stale_detail = ( - f" | last_activity={_sa.get('last_activity_desc', 'unknown') if isinstance(_sa, dict) else 'unknown'} " - f"({_stale_idle:.0f}s ago) " - f"| iteration={_sa.get('api_call_count', 0) if isinstance(_sa, dict) else 0}/{_sa.get('max_iterations', 0) if isinstance(_sa, dict) else 0}" - ) - except Exception: - pass - # A valid activity clock is authoritative: total age alone never - # makes an actively progressing turn stale. The emergency wall TTL - # is only a fallback when the agent cannot report usable activity. - _wall_ttl = max(_raw_stale_timeout * 10, 7200) if _raw_stale_timeout > 0 else float("inf") - _should_evict = ( - _stale_agent is not _AGENT_PENDING_SENTINEL - and ( - ( - _activity_summary_valid - and _raw_stale_timeout > 0 - and _stale_idle >= _raw_stale_timeout - ) - or ( - not _activity_summary_valid - and _stale_age > _wall_ttl - ) - ) - ) - if _should_evict: - logger.warning( - "Evicting stale _running_agents entry for %s " - "(age: %.0fs, idle: %.0fs, timeout: %.0fs)%s", - _quick_key, _stale_age, _stale_idle, - _raw_stale_timeout, _stale_detail, - ) - self._invalidate_session_run_generation( - _quick_key, - reason="stale_running_agent_eviction", - ) - self._release_running_agent_state(_quick_key) - - # Durable-reaped guard. A session whose routing row was ended in state.db (``ws_orphan_reap`` - # / ``agent_close``) while the gateway lived keeps its in-memory turn slot, so the fast-path - # would queue every next message into the dead runtime. Evict the stale slot so the cold - # path re-attaches via ``get_or_create_session`` → ``reopen`` or creates a fresh session. - if self._is_session_running(_quick_key): - try: - _reap_store = getattr(self, "session_store", None) - # Use the public, lock-held accessors: peek_session_id resolves key -> session_id - # under the store lock, and returns a non-str on stubbed stores in bare test runners - # — both the isinstance() gate and the ``is True`` gate below keep this guard inert - # unless a real SessionStore answers. - _reap_peek = getattr(_reap_store, "peek_session_id", None) - _is_ended = getattr(_reap_store, "_is_session_ended_in_db", None) - _reap_sid = _reap_peek(_quick_key) if callable(_reap_peek) else None - if ( - isinstance(_reap_sid, str) - and _reap_sid - and callable(_is_ended) - and _is_ended(_reap_sid) is True - ): - logger.warning( - "Evicting stale _running_agents entry for %s — " - "durable session %s is ended (reaped) in state.db; " - "healing routing on next message (#99106)", - _quick_key, - _reap_sid, - ) - self._invalidate_session_run_generation( - _quick_key, - reason="reaped_session_eviction", - ) - self._release_running_agent_state(_quick_key) - except Exception: - logger.debug("reaped-session staleness check failed", exc_info=True) - - if self._is_session_running(_quick_key): - # Resolve the command once; each command's mid-run behavior is declared on its - # CommandDef (busy_policy / busy_handler in hermes_cli/commands.py) and dispatched via - # _dispatch_busy_slash_command below — no per-command if-chain here. - from hermes_cli.commands import resolve_command as _resolve_cmd_inner - _evt_cmd = event.get_command() - _cmd_def_inner = _resolve_cmd_inner(_evt_cmd) if _evt_cmd else None - - # /status and /context are intentionally pre-gate so users - # always see session state. - if _cmd_def_inner and _cmd_def_inner.name == "status": - return await self._handle_status_command(event) - if _cmd_def_inner and _cmd_def_inner.name == "context": - return await self._handle_context_command(event) - - # Slash command access control on the running-agent fast-path. Mirrors the cold-path - # gate below so non-admins can't bypass gating just because an agent is busy. /status - # above is intentionally pre-gate; /help and /whoami are the always-allowed floor. - if _evt_cmd and _cmd_def_inner is not None: - _denied = self._check_slash_access(source, _cmd_def_inner.name) - if _denied is not None: - return _denied - - # Any recognized slash command: dispatch according to its declared busy_policy (dispatch - # / interrupt_then_dispatch / reject). Unrecognized commands and plain text fall through - # to the interrupt/queue logic below. - if _cmd_def_inner: - return await self._dispatch_busy_slash_command( - event, _cmd_def_inner, _quick_key, source, - ) - - if event.message_type == MessageType.PHOTO: - logger.debug("PRIORITY photo follow-up for session %s — queueing without interrupt", _quick_key) - adapter = self._adapter_for_source(source) - if adapter: - merge_pending_message_event(adapter._pending_messages, _quick_key, event) - return None - - effective_busy_input_mode = self._effective_busy_input_mode(source) - _telegram_followup_grace = float( - os.getenv("HERMES_TELEGRAM_FOLLOWUP_GRACE_SECONDS", "3.0") - ) - _grace_state = self._peek_session_state(_quick_key) - _started_at = _grace_state.turn.started_ts if _grace_state else 0 - if ( - source.platform == Platform.TELEGRAM - and event.message_type == MessageType.TEXT - and _telegram_followup_grace > 0 - and _started_at - and (time.time() - _started_at) <= _telegram_followup_grace - ): - logger.debug( - "Telegram follow-up arrived %.2fs after run start for %s — queueing without interrupt", - time.time() - _started_at, - _quick_key, - ) - adapter = self._adapter_for_source(source) - if adapter: - if effective_busy_input_mode == "queue": - self._enqueue_fifo(_quick_key, event, adapter) - else: - merge_pending_message_event( - adapter._pending_messages, - _quick_key, - event, - merge_text=True, - ) - return None - - _ra_state = self._peek_session_state(_quick_key) - running_agent = _ra_state.turn.agent if _ra_state else None - if running_agent is _AGENT_PENDING_SENTINEL: - # Agent is being set up but not ready yet. - if event.get_command() == "stop": - # Force-clean the sentinel so the session is unlocked. - self._release_running_agent_state(_quick_key) - logger.info("HARD STOP (pending) for session %s — sentinel cleared", _quick_key) - return EphemeralReply("⚡ Force-stopped. The agent was still starting — session unlocked.") - # Queue the message so it will be picked up after the - # agent starts. - adapter = self._adapter_for_source(source) - if adapter: - merge_pending_message_event( - adapter._pending_messages, - _quick_key, - event, - merge_text=True, - ) - return None - if self._draining: - queue_during_drain = self._queue_during_drain_enabled( - effective_busy_input_mode - ) - if queue_during_drain: - self._queue_or_replace_pending_event(_quick_key, event) - return ( - f"⏳ Gateway {self._status_action_gerund()} — queued for the next turn after it comes back." - if queue_during_drain - else f"⏳ Gateway is {self._status_action_gerund()} and is not accepting another turn right now." - ) - if effective_busy_input_mode == "queue": - logger.debug("PRIORITY queue follow-up for session %s", _quick_key) - self._queue_or_replace_pending_event(_quick_key, event) - return None - if effective_busy_input_mode == "steer": - # Steer mode: inject text into the running agent mid-run via - # agent.steer(). Falls back to queue semantics if the payload - # is empty, the agent lacks steer(), or steer() rejects. - steer_text = (event.text or "").strip() - steered = False - if ( - event.message_type == MessageType.TEXT - and not event.media_urls - and not event.media_types - and steer_text - and hasattr(running_agent, "steer") - ): - try: - steered = bool(running_agent.steer(steer_text)) - except Exception as exc: - logger.warning("PRIORITY steer failed for session %s: %s", _quick_key, exc) - steered = False - if steered: - logger.debug("PRIORITY steer for session %s", _quick_key) - return None - logger.debug("PRIORITY steer-fallback-to-queue for session %s", _quick_key) - self._queue_or_replace_pending_event(_quick_key, event) - return None - # Subagent protection (PRIORITY path). Same rationale as - # ``_handle_active_session_busy_message``: an interrupt cascades through - # ``_active_children`` and aborts in-flight delegate_task work, so demote to queue - # semantics while subagents run. /stop reached its handler above — still an escape hatch. - if self._agent_has_active_subagents(running_agent): - logger.info( - "PRIORITY interrupt demoted to queue for session %s " - "because the running agent has active subagents (#30170)", - _quick_key, - ) - self._queue_or_replace_pending_event(_quick_key, event) - return None - # Compression protection (PRIORITY path), as in ``_handle_active_session_busy_message``: - # an interrupt would start a new turn on the pre-rotation parent while compression - # rotates the id away, forking orphaned siblings. Demote to queue until rotation lands. - if await self._session_has_compression_in_flight(_quick_key): - logger.info( - "PRIORITY interrupt demoted to queue for session %s " - "because context compression is in flight (#56391)", - _quick_key, - ) - self._queue_or_replace_pending_event(_quick_key, event) - return None - # Text-only corrections redirect the live turn (preserving displayed context) when the - # runtime supports it; media/voice and older runtimes use the interrupt path below. - if ( - event.message_type == MessageType.TEXT - and not event.media_urls - and not event.media_types - and getattr(running_agent, "_supports_active_turn_redirect", False) - is True - and hasattr(running_agent, "redirect") - ): - try: - if running_agent.redirect((event.text or "").strip()): - logger.debug("PRIORITY redirect for session %s", _quick_key) - return None - except Exception as exc: - logger.warning( - "PRIORITY redirect failed for session %s: %s", - _quick_key, - exc, - ) - logger.debug("PRIORITY interrupt for session %s", _quick_key) - _interrupt_text = event.text - _media_urls = getattr(event, "media_urls", None) or [] - if self._pending_event_audio_paths(event): - _interrupt_text, _ = await self._transcribe_and_echo_pending_voice( - event, - self._adapter_for_source(source), - source, - event.text or "", - log_context="Voice-priority-interrupt", - ) - elif not _interrupt_text and _media_urls: - _interrupt_text = _build_media_placeholder(event) - running_agent.interrupt(_interrupt_text) - # The interrupt message is delivered via adapter._pending_messages (read by _run_agent); - # don't also buffer it on self — that copy was never consumed and grew unbounded. - return None - - # Check for commands - command = event.get_command() - - from hermes_cli.commands import ( - GATEWAY_KNOWN_COMMANDS, - is_gateway_known_command, - resolve_command as _resolve_cmd, - ) - - # Resolve aliases to canonical name so dispatch and hook names - # don't depend on the exact alias the user typed. - _cmd_def = _resolve_cmd(command) if command else None - canonical = _cmd_def.name if _cmd_def else command - - # Expand alias quick commands before built-in dispatch so targets like /model openai/gpt-5.5 - # --provider openrouter reach the /model handler. Preserve built-in precedence; aliases only - # need early handling when the typed command is not already known. - if command and _cmd_def is None: - if isinstance(self.config, dict): - quick_commands = self.config.get("quick_commands", {}) or {} - else: - quick_commands = getattr(self.config, "quick_commands", {}) or {} - if isinstance(quick_commands, dict) and command in quick_commands: - qcmd = quick_commands[command] - if qcmd.get("type") == "alias": - target = (qcmd.get("target") or "").strip() - if target: - target = target if target.startswith("/") else f"/{target}" - target_command = target.lstrip("/") - user_args = event.get_command_args().strip() - event.text = f"{target} {user_args}".strip() - command = target_command.split()[0] if target_command else target_command - _cmd_def = _resolve_cmd(command) if command else None - canonical = _cmd_def.name if _cmd_def else command - - # Per-platform slash command access control. Only kicks in when the operator has set - # ``allow_admin_from`` for the source's scope (DM vs group). When unset → backward-compat: - # every allowed user can run every command. When set → non-admins get only - # ``user_allowed_commands`` plus the /help, /whoami floor. Plain chat is never gated. - if command and canonical and is_gateway_known_command(canonical): - _denied = self._check_slash_access(source, canonical) - if _denied is not None: - return _denied - - # pre_command observer hook (returns ignored) fires for every recognized slash command - # BEFORE core handling, mirroring cli.py. The running-agent intercept path above (/stop, - # /approve, busy_policy) deliberately does NOT fire it — a slow or hostile plugin must not - # interfere with the operator's escape hatches for a live agent. - if command and is_gateway_known_command(canonical): - try: - from hermes_cli.plugins import fire_pre_command_hook - fire_pre_command_hook( - surface="gateway", - command=str(canonical), - alias_used=str(command), - args_raw=event.get_command_args().strip(), - session_key=_quick_key, - platform=source.platform.value if source.platform else "", - ) - except Exception as _pre_cmd_err: - logger.debug( - "pre_command hook dispatch failed (non-fatal): %s", - _pre_cmd_err, - ) - - # Fire ``command:`` for any recognized slash command (built-in or plugin). - # Handlers may return ``{"decision": "deny" | "handled" | "rewrite", ...}`` to intercept - # dispatch; handlers returning nothing behave as plain observers. - if command and is_gateway_known_command(canonical): - raw_args = event.get_command_args().strip() - hook_ctx = { - "platform": source.platform.value if source.platform else "", - "user_id": source.user_id, - "command": canonical, - "raw_command": command, - "args": raw_args, - "raw_args": raw_args, - } - try: - hook_results = await self.hooks.emit_collect( - f"command:{canonical}", hook_ctx - ) - except Exception as _hook_err: - logger.debug( - "command:%s hook dispatch failed (non-fatal): %s", - canonical, _hook_err, - ) - hook_results = [] - - for hook_result in hook_results: - if not isinstance(hook_result, dict): - continue - decision = str(hook_result.get("decision", "")).strip().lower() - if not decision or decision == "allow": - continue - if decision == "deny": - message = hook_result.get("message") - if isinstance(message, str) and message: - return message - return f"Command `/{command}` was blocked by a hook." - if decision == "handled": - message = hook_result.get("message") - return message if isinstance(message, str) and message else None - if decision == "rewrite": - new_command = str( - hook_result.get("command_name", "") - ).strip().lstrip("/") - if not new_command: - continue - new_args = str(hook_result.get("raw_args", "")).strip() - event.text = f"/{new_command} {new_args}".strip() - command = event.get_command() - _cmd_def = _resolve_cmd(command) if command else None - canonical = _cmd_def.name if _cmd_def else command - break - - plain_handler = ( - self._gateway_plain_command_handlers().get(canonical) - or self._gateway_idle_command_handlers().get(canonical) - ) - if plain_handler is not None: - return await plain_handler(event) - - if canonical == "new": - if await asyncio.to_thread(self._is_telegram_topic_root_lobby, source): - return self._telegram_topic_root_new_message() - async def _do_reset(): - return await self._handle_reset_command(event) - return await self._maybe_confirm_destructive_slash( - event=event, - command="new", - title="/new", - detail=( - "This starts a fresh session and discards the current " - "conversation history." - ), - execute=_do_reset, - ) - - if canonical == "start": - logger.info("Ignoring /start platform ping for session %s", _quick_key) - return "" - - if canonical == "egress": - from hermes_cli.proxy_cli import format_status_text - - return format_status_text() - - if canonical == "learn": - # Open-ended: rewrite the turn to a standards-guided prompt and fall through to normal - # agent processing. Mirrors the /blueprint fall-through so role alternation is - # preserved. No engine, works on any backend. - from agent.learn_prompt import build_learn_prompt - - _learn_req = event.get_command_args().strip() - _ack = ( - "Learning a skill from what you described…" - if _learn_req - else "Learning a skill from this conversation…" - ) - await self._send_command_ack(source, _ack, "learn") - try: - event.text = build_learn_prompt(_learn_req) - # fall through to agent processing - except Exception: - return "Could not start /learn — please try again." - - if canonical == "plan": - # /plan: rewrite the turn to the plan-mode prompt and fall through to normal agent - # processing (the /learn fall-through keeps role alternation). Works on any backend. - from agent.plan_prompt import build_plan_prompt - - _plan_task = event.get_command_args().strip() - _ack = ( - f"Planning: {_plan_task[:80]}{'…' if len(_plan_task) > 80 else ''}" - if _plan_task - else "Planning from this conversation's context…" - ) - await self._send_command_ack(source, _ack, "plan") - try: - event.text = build_plan_prompt(_plan_task) - # fall through to agent processing - except Exception: - return "Could not start /plan — please try again." - - if canonical == "init": - # /init: rewrite the turn to a guidance-laden prompt and fall through to normal agent - # processing (the /learn fall-through keeps role alternation). Works on any backend. - from hermes_cli.init_command import build_init_prompt_for_cwd - - _init_notes = event.get_command_args().strip() - try: - _init_prompt = build_init_prompt_for_cwd(extra=_init_notes) - except Exception: - return "Could not start /init — please try again." - _ack = ( - "Updating AGENTS.md from a project scan…" - if "UPDATE the existing AGENTS.md" in _init_prompt - else "Generating AGENTS.md from a project scan…" - ) - await self._send_command_ack(source, _ack, "init") - event.text = _init_prompt - # fall through to agent processing - - if canonical == "blueprint": - _blueprint_result = await self._handle_blueprint_command(event) - _blueprint_seed = getattr(_blueprint_result, "agent_seed", None) - if _blueprint_seed: - # Blueprint matched — rewrite the turn to the seed and fall through to - # _handle_message_with_agent so the agent collects each slot value conversationally, - # then calls the cronjob tool (the /steer fall-through pattern). - _ack = getattr(_blueprint_result, "text", "") or "" - if _ack: - await self._send_command_ack(source, _ack, "blueprint") - try: - event.text = _blueprint_seed - except Exception: - return getattr(_blueprint_result, "text", "") or None - else: - return getattr(_blueprint_result, "text", "") or None - - if canonical == "undo": - async def _do_undo(): - return await self._handle_undo_command(event) - _undo_n = 1 - _undo_raw = event.get_command_args().strip() - if _undo_raw: - try: - _undo_n = max(1, int(_undo_raw.split()[0])) - except (ValueError, IndexError): - _undo_n = 1 - _undo_detail = ( - "This removes the last user/assistant exchange from history." - if _undo_n == 1 - else f"This removes the last {_undo_n} user turns from history." - ) - return await self._maybe_confirm_destructive_slash( - event=event, - command="undo", - title="/undo", - detail=_undo_detail, - execute=_do_undo, - ) - - if canonical == "queue": - queue_payload = event.get_command_args().strip() - if not queue_payload: - return "Usage: /queue " - with suppress(Exception): - event.text = queue_payload - - if canonical == "steer": - # No active agent — /steer has nothing to inject into. Strip the prefix so downstream - # treats it as a normal user message; an empty payload surfaces the usage hint. - steer_payload = event.get_command_args().strip() - if not steer_payload: - return "Usage: /steer (no agent is running; sending as a normal message)" - with suppress(Exception): - event.text = steer_payload - # Do NOT return — fall through to _handle_message_with_agent at the end of this function - # so the rewritten text is sent to the agent as a regular user turn. - - if canonical == "moa": - # /moa is one-shot sugar only: run a single prompt through the default MoA preset, then - # restore the prior model. To *switch* to a MoA preset for the session, pick it from the - # model picker (MoA presets surface as a virtual "Mixture of Agents" provider). - from hermes_cli.moa_config import ( - moa_usage, - normalize_moa_config, - ) - from hermes_cli.config import load_config - - moa_payload = event.get_command_args().strip() - if not moa_payload: - return moa_usage() - try: - cfg = load_config() - moa_cfg = normalize_moa_config(cfg.get("moa") if isinstance(cfg, dict) else {}) - except Exception: - moa_cfg = normalize_moa_config({}) - preset = moa_cfg["default_preset"] - try: - event.text = moa_payload - _moa_state = self._session_state(_quick_key) - event._moa_restore_override = _moa_state.conversation.model_override - _moa_state.conversation.model_override = { - "provider": "moa", - "model": preset, - "base_url": "moa://local", - "api_key": "moa-virtual-provider", - "api_mode": "chat_completions", - } - self._evict_cached_agent(_quick_key) - event._moa_disable_after_turn = True - except Exception: - return "Failed to prepare MoA turn." - - if self._draining: - return f"⏳ Gateway is {self._status_action_gerund()} and is not accepting new work right now." - - # User-defined quick commands (bypass agent loop, no LLM call) - if command: - if isinstance(self.config, dict): - quick_commands = self.config.get("quick_commands", {}) or {} - else: - quick_commands = getattr(self.config, "quick_commands", {}) or {} - if not isinstance(quick_commands, dict): - quick_commands = {} - if command in quick_commands: - # Quick commands are slash capabilities too — and type:exec ones run a shell command - # in the gateway process. The early gate only fires for registry-known commands and - # quick commands are never in the registry, so apply the same admin/user policy to - # the raw typed name here so non-admins can't invoke admin-only quick commands. - _denied = self._check_slash_access(source, command) - if _denied is not None: - return _denied - qcmd = quick_commands[command] - if qcmd.get("type") == "exec": - exec_cmd = qcmd.get("command", "") - if exec_cmd: - try: - # Sanitize env to prevent credential leakage — quick commands run in the - # gateway process, which has all API keys in os.environ. - from tools.environments.local import build_subprocess_env - sanitized_env = build_subprocess_env() - proc = await asyncio.create_subprocess_shell( - exec_cmd, - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.PIPE, - env=sanitized_env, - ) - stdout, stderr = await asyncio.wait_for(proc.communicate(), timeout=30) - output = (stdout or stderr).decode().strip() - # Redact any remaining sensitive patterns in output - if output: - from agent.redact import redact_sensitive_text - output = redact_sensitive_text(output) - return output if output else "Command returned no output." - except asyncio.TimeoutError: - return "Quick command timed out (30s)." - except Exception as e: - return f"Quick command error: {e}" - else: - return f"Quick command '/{command}' has no command defined." - elif qcmd.get("type") == "alias": - target = (qcmd.get("target") or "").strip() - if target: - target = target if target.startswith("/") else f"/{target}" - target_command = target.lstrip("/") - user_args = event.get_command_args().strip() - event.text = f"{target} {user_args}".strip() - command = target_command.split()[0] if target_command else target_command - # Fall through to normal command dispatch below - else: - return f"Quick command '/{command}' has no target defined." - else: - return f"Quick command '/{command}' has unsupported type (supported: 'exec', 'alias')." - - # Plugin-registered slash commands - if command: - try: - from hermes_cli.plugins import get_plugin_command_handler - # Normalize underscores to hyphens so Telegram's underscored autocomplete form - # matches plugin commands registered with hyphens (see _build_telegram_menu). - plugin_handler = get_plugin_command_handler(command.replace("_", "-")) - if plugin_handler: - user_args = event.get_command_args().strip() - result = plugin_handler(user_args) - if asyncio.iscoroutine(result): - result = await result - return str(result) if result else None - except Exception as e: - logger.warning("Plugin command dispatch failed: %s", e) - - # Skill slash commands: /skill-name loads the skill and sends to agent. - # resolve_skill_command_key() handles the Telegram underscore/hyphen round-trip so - # /claude_code from Telegram autocomplete still resolves to the claude-code skill. - if command: - # Skill bundles take precedence over individual skill commands — - # / loads multiple skills at once. Mirrors CLI dispatch. - _bundle_handled = False - try: - from agent.skill_bundles import ( - build_bundle_invocation_message, - resolve_bundle_command_key, - ) - bundle_key = resolve_bundle_command_key(command) - if bundle_key is not None: - user_instruction = event.get_command_args().strip() - # Pass the platform explicitly: bundle skill loading bypasses - # get_skill_commands()' scan-time disabled filter, and the gateway serves - # multiple platforms in one process, so env-var platform resolution can't be - # trusted here. - _bundle_plat = source.platform.value if source.platform else None - bundle_result = build_bundle_invocation_message( - bundle_key, user_instruction, task_id=_quick_key, - platform=_bundle_plat, - ) - if bundle_result: - msg, _loaded, missing = bundle_result - event.text = msg - _bundle_handled = True - if missing: - logger.info( - "Bundle %s skipped missing skills: %s", - bundle_key, ", ".join(missing), - ) - # Fall through to normal message processing with bundle content - except Exception as exc: - logger.warning("Bundle dispatch failed: %s", exc) - - if command and not locals().get("_bundle_handled", False): - try: - from agent.skill_commands import ( - get_skill_commands, - build_skill_invocation_message, - resolve_skill_command_key, - ) - skill_cmds = get_skill_commands() - cmd_key = resolve_skill_command_key(command) - if cmd_key is not None: - # Check per-platform disabled status before executing. get_skill_commands() only - # applies the *global* disabled list at scan time; per-platform overrides need - # checking here because the cache is process-global across platforms. - _skill_name = skill_cmds[cmd_key].get("name", "") - _plat = source.platform.value if source.platform else None - if _plat and _skill_name: - from agent.skill_utils import get_disabled_skill_names as _get_plat_disabled - if _skill_name in _get_plat_disabled(platform=_plat): - return ( - f"The **{_skill_name}** skill is disabled for {_plat}.\n" - f"Enable it with: `hermes skills config`" - ) - user_instruction = event.get_command_args().strip() - # Stacked slash-skill invocations: `/skill-a /skill-b do XYZ` loads every - # leading skill (up to 5), not just the first. Mirrors CLI. - try: - from agent.skill_commands import ( - build_stacked_skill_invocation_message as _build_stacked, - split_stacked_skill_commands, - ) - extra_keys, stacked_instruction = ( - split_stacked_skill_commands(user_instruction) - ) - except Exception: - _build_stacked = None - extra_keys, stacked_instruction = [], user_instruction - if extra_keys and _plat: - # split_stacked_skill_commands() only resolves that each extra token is a - # KNOWN skill command — like get_skill_commands() itself, it has no per- - # platform view. Re-check every stacked skill against the same disabled list, - # or a skill disabled for this platform still gets loaded via the stack. - from agent.skill_utils import get_disabled_skill_names as _get_plat_disabled - _plat_disabled = _get_plat_disabled(platform=_plat) - _disabled_extra = [ - skill_cmds.get(k, {}).get("name", "") - for k in extra_keys - if skill_cmds.get(k, {}).get("name", "") in _plat_disabled - ] - if _disabled_extra: - return ( - f"The **{', '.join(_disabled_extra)}** skill(s) in this " - f"stacked invocation are disabled for {_plat}.\n" - f"Enable them with: `hermes skills config`" - ) - if extra_keys and _build_stacked is not None: - stacked_result = _build_stacked( - [cmd_key, *extra_keys], - stacked_instruction, - task_id=_quick_key, - ) - if stacked_result: - msg, _loaded, _missing = stacked_result - event.text = msg - # Fall through to normal message processing - else: - return f"Failed to load stacked skills for /{command}." - else: - msg = build_skill_invocation_message( - cmd_key, user_instruction, task_id=_quick_key - ) - if msg: - event.text = msg - # Fall through to normal message processing with skill content - else: - # Not an active skill — check if it's a known-but-disabled or - # uninstalled skill and give actionable guidance. - _unavail_msg = _check_unavailable_skill(command) - if _unavail_msg: - return _unavail_msg - # Genuinely unrecognized /command (not built-in/plugin/skill/known-inactive): - # warn instead of forwarding to the LLM as free text (it invents tool calls). - # Normalize to hyphenated form first: the quick-command block may have set an - # alias target, so _cmd_def can be stale. - if command.replace("_", "-") not in GATEWAY_KNOWN_COMMANDS: - logger.warning( - "Unrecognized slash command /%s from %s — " - "replying with unknown-command notice", - command, - source.platform.value if source.platform else "?", - ) - return ( - f"Unknown command `/{command}`. " - f"Type /commands to see what's available, " - f"or resend without the leading slash to send " - f"as a regular message." - ) - except Exception as e: - logger.debug("Skill command check failed (non-fatal): %s", e) - - # Pending exec approvals go through /approve and /deny only — no bare-text matching, or a - # conversational "yes" would execute a dangerous command. - - if not is_internal and await asyncio.to_thread( - self._is_telegram_topic_root_lobby, source - ): - # Debounce the lobby reminder so a user who forgets about - # topic mode and fires ten prompts doesn't get ten copies. - if self._should_send_telegram_lobby_reminder(source): - return self._telegram_topic_root_lobby_message() - return None - - # ── External-drain new-turn gate ───────────────────────────── - # When NAS engaged an external drain (.drain_request.json, seen by _drain_control_watcher), - # refuse to START new turns so the in-flight set can only fall to zero (stop accepting - # FIRST, then NAS polls active_agents==0). Internal/system events bypass; reversible. - if self._external_drain_active and not is_internal: - logger.info( - "Refusing new turn for session %s — external drain active.", - _quick_key, - ) - return ( - "⏳ This agent is draining for a maintenance action and isn't " - "accepting new turns right now. It'll be back in a moment — " - "please resend shortly." - ) - - # ── Claim this session before any await ─────────────────────── - # Many awaits sit between here and _run_agent registering the real AIAgent; without this - # sentinel a second message during any of them passes the "already running" guard and spins - # up a duplicate agent for the same session, corrupting the transcript. - _active_session_lease, _limit_message = self._claim_active_session_slot( - _quick_key, - source, - ) - if _limit_message is not None: - logger.info( - "Rejecting new active session %s: max_concurrent_sessions reached", - _quick_key, - ) - return _limit_message - - # ── FIFO orphan rescue ─────────────────────────────────────── - # A session that went idle with a populated overflow (post-turn drain never promoted, e.g. a - # compression-demoted follow-up) silently orphaned those events. Re-stage them FIFO and - # enqueue this event behind them. Skipped for control commands and internal events. - try: - _orphan_adapter = self._adapter_for_source(source) - if ( - _orphan_adapter is not None - and not bool(getattr(event, "internal", False)) - and not event.get_command() - ): - _rescued = self._rescue_orphaned_overflow( - _quick_key, _orphan_adapter - ) - if _rescued is not None: - # The oldest orphan runs as THIS turn. Park the incoming event behind the rest of - # the chain: into the slot when the chain was a single orphan (post-turn drain - # picks it up), otherwise into overflow behind the already-staged next orphan. - self._enqueue_fifo(_quick_key, event, _orphan_adapter) - event = _rescued - # Same session key by construction; carry the orphan's own source so reply - # anchors / thread metadata point at the message actually being answered. - _rescued_source = getattr(_rescued, "source", None) - if _rescued_source is not None: - source = _rescued_source - is_internal = bool(getattr(_rescued, "internal", False)) - except Exception: - logger.debug( - "FIFO orphan rescue pre-claim failed for %s", - _quick_key, - exc_info=True, - ) - - _claim_state = self._session_state(_quick_key) - if _active_session_lease is not None: - _claim_state.turn.lease = _active_session_lease - _claim_state.turn.agent = _AGENT_PENDING_SENTINEL - _claim_state.turn.started_ts = time.time() - self._persist_active_agents() - _run_generation = self._begin_session_run_generation(_quick_key) - - try: - try: - _agent_result = await self._handle_message_with_agent( - event, source, _quick_key, _run_generation - ) - except TurnLeaseTimeoutError as exc: - # A rejected message, not a completed turn: return before the /goal judge below so - # it cannot consume the resend notice and enqueue a synthetic continuation loop. - logger.error( - "Rejecting turn for routing key %s on session %s after " - "turn-lease timeout; transcript load was not started and " - "the user must resend", - _quick_key, - exc.session_id, - ) - return ( - "⏳ Another turn is still running on this session. To " - "protect the transcript, this message was not processed. " - "Wait for the active turn to finish, then resend it." - ) - try: - await self._run_post_turn_hooks( - agent_result=_agent_result, - source=source, - is_internal=is_internal, - event=event, - ) - except Exception as _goal_exc: - logger.debug("post-turn hook failed: %s", _goal_exc) - return _agent_result - finally: - # MoA one-shot restore must run on EVERY exit path: the restore data lives on the - # per-turn event, so a restore in the try block is skipped when the handler raises and - # the override leaks permanently; finally covers success, exception and interrupt. - self._restore_moa_one_shot(event, _quick_key) - self._restore_pending_one_turn_model_override(_quick_key) - # Normal completion/exception/interrupt clears this durable marker; SIGKILL/OOM skips - # finally, leaving it for the next unclean startup's recovery pass. - await self._clear_durable_active_turn(event) - # Unconditional release covers every exit path: _release_running_agent_state is idempotent - # and, without a run_generation guard, clears the slot whichever generation holds it. This - # evicts the zombie left when session_reset bumps the generation mid-flight (gen-N's - # guarded release in _run_agent returns False; a sentinel-only check would lock forever). - self._release_running_agent_state(_quick_key) - # Turn lease: release THIS turn's token — keyed by (routing key, run generation) so this - # unwind can only free the lease its own turn acquired, never a newer turn's. - self._release_turn_lease(_quick_key, _run_generation) - - def _restore_moa_one_shot(self, event: "MessageEvent", quick_key: str) -> None: - """Revert a ``/moa `` one-shot model override after its turn. - - Called from the message-handling ``finally`` so it fires on success, error or interrupt. - No-op unless ``event._moa_disable_after_turn``; ``_moa_restore_override`` holds the prior - per-session override (``None`` = clear the MoA override outright). - """ - if not getattr(event, "_moa_disable_after_turn", False): - return - try: - _restore = getattr(event, "_moa_restore_override", None) - self._session_state(quick_key).conversation.model_override = _restore - self._evict_cached_agent(quick_key) - except Exception: - pass - - def _restore_pending_one_turn_model_override(self, session_key: str) -> None: - """Restore a per-session model override after ``/model --once`` runs.""" - if not session_key: - return - try: - _otr_state = self._peek_session_state(session_key) - snapshot = _otr_state.conversation.one_turn_restore if _otr_state else None - if _otr_state is not None: - _otr_state.conversation.one_turn_restore = None - if not snapshot: - return - self._restore_session_model_override(session_key, snapshot) - except Exception: - logger.debug("Failed to restore one-turn model override", exc_info=True) - - async def _prepare_inbound_message_text( - self, - *, - event: MessageEvent, - source: SessionSource, - history: List[Dict[str, Any]], - session_key: Optional[str] = None, - ) -> Optional[str]: - """Prepare inbound event text for the agent. - - Shared by the normal inbound and queued follow-up paths so attribution, image enrichment, - STT, document notes, reply context and @ references behave the same. Side effect: buffers - per-session native image paths when the model supports native vision; the caller consumes - that buffer at ``run_conversation``. Empty list means the text vision path already ran. - """ - history = history or [] - _pending_stt_prepared = hasattr(event, "_gateway_pending_stt_text") - message_text = ( - getattr(event, "_gateway_pending_stt_text", None) - if _pending_stt_prepared - else event.text - ) or "" - _group_sessions_per_user = getattr(self.config, "group_sessions_per_user", True) - _thread_sessions_per_user = getattr(self.config, "thread_sessions_per_user", False) - # Prefer the caller's resolved session key so this write key matches the consume key at the - # run_conversation site; derive it here only for tests and legacy standalone callers. - session_key = session_key or self._session_key_for_source(source) - # Reset only this session's per-call buffer; other sessions may be - # concurrently preparing multimodal turns on the same runner. - self._consume_pending_native_image_paths(session_key) - - _is_shared_multi_user = is_shared_multi_user_session( - source, - group_sessions_per_user=_group_sessions_per_user, - thread_sessions_per_user=_thread_sessions_per_user, - ) - if _is_shared_multi_user and source.user_name: - # source.user_name is the platform display name — attacker-influenceable on any - # platform that lets participants set their own name. Neutralize newlines/control chars - # before interpolating it into every message, or a hostile name can masquerade as a - # fake markdown section (mirrors build_session_context_prompt's treatment). - _safe_user_name = neutralize_untrusted_inline_text(source.user_name) - # On Slack, expose the current author's verifiable user ID next to the display name: - # "mention me again" requests need a trusted `<@U...>` target for the CURRENT speaker — - # display names are ambiguous and historical mentions may point at someone else. The - # user_id comes from the Slack event envelope (not user-editable), so no neutralization. - if source.platform == Platform.SLACK and source.user_id: - _safe_user_name = ( - f"{_safe_user_name} | Slack user <@{source.user_id}>" - ) - message_text = f"[{_safe_user_name}] {message_text}" - - # Prepend history-backfill channel context after the sender-prefix so the prefix applies - # only to the trigger message, not the backfill block. - if getattr(event, "channel_context", None): - message_text = f"{event.channel_context}\n\n[New message]\n{message_text}" - - # Declare at outer scope so the audio-file-paths handling block below - # remains safe when ``event.media_urls`` is empty (no inner block runs). - audio_file_paths: list[str] = [] - video_paths: list[str] = [] - - if event.media_urls: - image_paths = [] - audio_paths = [] - for i, path in enumerate(event.media_urls): - mtype = event.media_types[i] if i < len(event.media_types) else "" - # Classify images per-attachment: trust this attachment's own MIME, and only honour - # the message-level PHOTO type when the per-attachment MIME is unknown. Otherwise a - # document sent alongside an image gets mis-routed as an image and the provider 400s. - if _event_media_is_image(event, i): - image_paths.append(path) - # MessageType.AUDIO = audio file attachment (e.g. .mp3, .m4a) — never STT. - # Mixed DOCUMENT events also preserve audio as a file path instead of - # dropping it or treating it as a voice note. - if _event_media_is_audio(event, i): - if event.message_type in {MessageType.AUDIO, MessageType.DOCUMENT}: - audio_file_paths.append(path) - elif not _pending_stt_prepared and _event_media_is_stt_input(event, i): - audio_paths.append(path) - if mtype.startswith("video/") or (not mtype and event.message_type == MessageType.VIDEO): - video_paths.append(path) - - if image_paths: - # Decide routing: native (attach pixels) vs text (vision_analyze pre-run + prepend - # description). See agent/image_routing.py. Offloaded to a thread: the decision does - # blocking network I/O (models.dev fetch on cache miss, Ollama /api/show probe) whose - # timeout would otherwise stall the whole gateway event loop. - _img_mode = await asyncio.to_thread( - self._decide_image_input_mode, - source=source, - session_key=session_key, - ) - if _img_mode == "native": - # Defer attachment to the run_conversation call site. - self._session_state( - session_key - ).persistent.native_image_paths = list(image_paths) - logger.info( - "Image routing: native (model supports vision). %d image(s) will be attached inline.", - len(image_paths), - ) - else: - logger.info( - "Image routing: text (mode=%s). Pre-analyzing %d image(s) via vision_analyze.", - _img_mode, len(image_paths), - ) - # Vision enrichment runs before AIAgent.run_conversation(), - # so bind this session's resolved runtime explicitly rather - # than consulting process-global compatibility mirrors. - vision_runtime = None - try: - turn_model, runtime_kwargs = self._resolve_session_agent_runtime( - source=source, - session_key=session_key, - ) - vision_runtime = dict(runtime_kwargs or {}) - vision_runtime["model"] = turn_model - except Exception: - logger.debug( - "vision enrichment: session runtime resolution failed", - exc_info=True, - ) - - from agent.auxiliary_client import scoped_runtime_main - - with scoped_runtime_main(vision_runtime): - message_text = await self._enrich_message_with_vision( - message_text, - image_paths, - ) - - if audio_paths: - message_text, _successful_transcripts = await self._enrich_message_with_transcription( - message_text, - audio_paths, - ) - # Echo each successful transcript back to the user immediately when configured. Lets - # users verify STT quality in real-time, while allowing quiet STT for users who only - # want the agent to receive the transcription. - if _successful_transcripts and self._should_echo_stt_transcripts(): - _echo_adapter = self._adapter_for_source(source) - _echo_meta = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) - if _echo_adapter: - for _tx in _successful_transcripts: - try: - await _echo_adapter.send( - source.chat_id, - f'🎙️ "{_tx}"', - metadata=_echo_meta, - ) - except Exception as _echo_exc: - logger.debug( - "Transcript echo failed (non-fatal): %s", _echo_exc, - ) - # On transcription failure, do NOT send a hardcoded notice here: that bypassed the - # LLM and produced two replies (one pre-canned, TTS'd in the wrong language). - # Enrichment leaves a single neutral marker so the LLM gives one localized reply. - - if audio_file_paths: - from tools.credential_files import to_agent_visible_cache_path as _to_agent_path - for _apath in audio_file_paths: - _basename = os.path.basename(_apath) - _parts = _basename.split("_", 2) - _display = _parts[2] if len(_parts) >= 3 else _basename - _display = re.sub(r'[^\w.\- ]', '_', _display) - _agent_path = _to_agent_path(_apath) - _note = ( - f"[The user sent an audio file attachment: '{_display}'. " - f"It is saved at: {_agent_path}. " - f"Its content is not inlined here. If the user's request involves " - f"what the audio contains, transcribe or process it yourself — for " - f"example by passing the path to a transcription or media tool — " - f"instead of asking the user to describe it. Only ask what to do " - f"with it if their intent is genuinely unclear.]" - ) - message_text = f"{_note}\n\n{message_text}" - - if video_paths: - from tools.credential_files import to_agent_visible_cache_path as _to_agent_path - for _vpath in video_paths: - _basename = os.path.basename(_vpath) - _parts = _basename.split("_", 2) - _display = _parts[2] if len(_parts) >= 3 else _basename - _display = re.sub(r'[^\w.\- ]', '_', _display) - _agent_path = _to_agent_path(_vpath) - _note = ( - f"[The user sent a video attachment: '{_display}'. " - f"It is saved at: {_agent_path}. " - f"Its content is not inlined here. If the user's request involves " - f"what the video contains, inspect or process it yourself — for " - f"example by passing the path to a video analysis or media tool — " - f"instead of asking the user to describe it. Only ask what to do " - f"with it if their intent is genuinely unclear.]" - ) - message_text = f"{_note}\n\n{message_text}" - - if event.media_urls: - import mimetypes as _mimetypes - from tools.credential_files import to_agent_visible_cache_path - - _TEXT_EXTENSIONS = {".txt", ".md", ".csv", ".log", ".json", ".xml", ".yaml", ".yml", ".toml", ".ini", ".cfg"} - for i, path in enumerate(event.media_urls): - # Per-attachment document handling: skip anything already routed as image/audio/video - # above; only genuine non-media files get a context note. A document mixed into a - # PHOTO/VOICE message (message-level type != DOCUMENT) thus still reaches the agent. - if ( - _event_media_is_image(event, i) - or _event_media_is_audio(event, i) - or _event_media_is_video(event, i) - ): - continue - mtype = event.media_types[i] if i < len(event.media_types) else "" - if mtype in {"", "application/octet-stream"}: - _ext = os.path.splitext(path)[1].lower() - if _ext in _TEXT_EXTENSIONS: - mtype = "text/plain" - else: - guessed, _ = _mimetypes.guess_type(path) - mtype = guessed or "application/octet-stream" - # Any accepted file gets a path-pointing context note — we accept - # all file types now, so a non-text/non-application MIME (font/*, - # model/*, etc.) must still tell the agent the file exists. - - basename = os.path.basename(path) - parts = basename.split("_", 2) - display_name = parts[2] if len(parts) >= 3 else basename - display_name = re.sub(r'[^\w.\- ]', '_', display_name) - - # Translate host cache path to in-container path if running under Docker backend. - # This ensures the agent receives a path it can open inside its sandbox, as the - # cache directories are auto-mounted at /root/.hermes/cache/* by get_cache_directory_mounts(). - agent_path = to_agent_visible_cache_path(path) - - inline_flags = getattr(event, "media_text_inlined", None) or [] - inline_flag = inline_flags[i] if i < len(inline_flags) else None - context_note = _build_document_context_note( - display_name, - agent_path, - mtype, - content_inlined=inline_flag is not False, - ) - message_text = f"{context_note}\n\n{message_text}" - - # Discord: surface the triggering message id per-turn on the user message rather than in the - # cached system prompt. message_id changes every turn, so baking it into - # build_session_context_prompt() would bust the agent-cache signature and rebuild the - # AIAgent every message (destroying prompt caching). - if ( - source is not None - and getattr(source, "platform", None) == Platform.DISCORD - and getattr(event, "message_id", None) - ): - from gateway.session import _discord_tools_loaded as _disc_tools_loaded - if _disc_tools_loaded(): - message_text = ( - f"[Triggering message id: `{event.message_id}` — use as " - f"`message_id` for reply/react/pin via the discord tools.]\n\n" - f"{message_text}" - ) - - if getattr(event, "reply_to_text", None) and event.reply_to_message_id: - # Always inject the reply-to pointer — even when the quoted text already appears in - # history. The prefix isn't deduplication, it's disambiguation: it tells the agent - # *which* prior message the user is referencing. Token overhead is minimal. - reply_snippet = event.reply_to_text[:500] - if getattr(event, "reply_to_is_own_message", False): - message_text = ( - f'[Replying to your previous message: "{reply_snippet}"]\n\n' - f"{message_text}" - ) - else: - message_text = f'[Replying to: "{reply_snippet}"]\n\n{message_text}' - - if "@" in message_text: - try: - from agent.context_references import preprocess_context_references_async - from agent.model_metadata import get_model_context_length_async - - try: - from tools.terminal_scope import terminal_env as _ts_env - except ImportError: - _msg_cwd = os.environ.get("TERMINAL_CWD", os.path.expanduser("~")) - else: - _msg_cwd = _ts_env("TERMINAL_CWD", os.path.expanduser("~")) - _msg_config_ctx = None - _msg_cfg = None - _msg_model_cfg = {} - _msg_custom_providers = [] - try: - _msg_cfg = _load_gateway_config() - _msg_model_cfg = _msg_cfg.get("model", {}) - if isinstance(_msg_model_cfg, dict): - _msg_raw_ctx = _msg_model_cfg.get("context_length") - if _msg_raw_ctx is not None: - _msg_config_ctx = int(_msg_raw_ctx) - try: - from hermes_cli.config import get_compatible_custom_providers - - _msg_custom_providers = get_compatible_custom_providers(_msg_cfg) - except Exception: - _msg_custom_providers = _msg_cfg.get("custom_providers") or [] - except Exception: - pass - # Resolve the session's actual model/provider/base_url as the hygiene compression - # block does; GatewayRunner has no self._model/self._base_url (AttributeError, - # silently caught below). - _msg_model, _msg_runtime = self._resolve_session_agent_runtime( - source=source, - session_key=session_key, - user_config=_msg_cfg, - ) - _msg_base_url = _msg_runtime.get("base_url") or "" - # A global model.context_length belongs to the configured - # model, not a session /model or channel override. Prefer a - # matching per-custom-provider model limit when available. - _msg_configured_model = ( - _msg_model_cfg.get("default") or _msg_model_cfg.get("model") - if isinstance(_msg_model_cfg, dict) - else _msg_model_cfg - ) - if _msg_model != _msg_configured_model: - _msg_config_ctx = None - if _msg_config_ctx is not None and isinstance(_msg_model_cfg, dict): - try: - from hermes_cli.route_identity import should_clear_context_pin_async - - if await should_clear_context_pin_async( - None, # model match already checked above - None, - _msg_model_cfg.get("base_url"), - _msg_base_url, - _msg_model_cfg.get("provider"), - _msg_runtime.get("provider"), - ): - _msg_config_ctx = None - except Exception: - _msg_config_ctx = None - if _msg_custom_providers and _msg_base_url: - try: - from hermes_cli.config import get_custom_provider_context_length - - _msg_custom_ctx = get_custom_provider_context_length( - model=_msg_model, - base_url=_msg_base_url, - custom_providers=_msg_custom_providers, - ) - if _msg_custom_ctx: - _msg_config_ctx = _msg_custom_ctx - except Exception: - pass - _msg_ctx_len = await get_model_context_length_async( - _msg_model, - base_url=_msg_base_url, - api_key=_msg_runtime.get("api_key") or "", - config_context_length=_msg_config_ctx, - provider=_msg_runtime.get("provider") or "", - custom_providers=_msg_custom_providers, - ) - _ctx_result = await preprocess_context_references_async( - message_text, - cwd=_msg_cwd, - context_length=_msg_ctx_len, - allowed_root=_msg_cwd, - ) - if _ctx_result.blocked: - _adapter = self._adapter_for_source(source) - if _adapter: - await _adapter.send( - source.chat_id, - "\n".join(_ctx_result.warnings) or "Context injection refused.", - ) - return None - if _ctx_result.expanded: - message_text = _ctx_result.message - except Exception as exc: - logger.warning("@ context reference expansion failed: %s", exc) - logger.debug("@ context reference expansion failure detail", exc_info=True) - - return message_text - - async def _prepare_profile_scoped_inbound_message_text( - self, - *, - event: MessageEvent, - source: SessionSource, - history: List[Dict[str, Any]], - session_key: Optional[str] = None, - ) -> Optional[str]: - """Run inbound preprocessing under the routed profile when multiplexed.""" - if getattr(getattr(self, "config", None), "multiplex_profiles", False): - async with _async_profile_runtime_scope( - self._resolve_profile_home_for_source(source) - ): - return await self._prepare_inbound_message_text( - event=event, - source=source, - history=history, - session_key=session_key, - ) - return await self._prepare_inbound_message_text( - event=event, - source=source, - history=history, - session_key=session_key, - ) - - async def _prepare_clarify_reply_text(self, event) -> str: - """Return raw text or successful voice transcripts for a clarify reply.""" - if not self._pending_event_audio_paths(event): - return (event.text or "").strip() - - _, successful_transcripts = await self._transcribe_pending_audio_event_once( - event, "", - ) - return "\n\n".join( - transcript.strip() - for transcript in successful_transcripts - if transcript.strip() - ) - - def _consume_pending_native_image_paths(self, session_key: str) -> List[str]: - state = self._peek_session_state(session_key) - if state is None or not state.persistent.native_image_paths: - return [] - paths = list(state.persistent.native_image_paths) - state.persistent.native_image_paths = [] - return paths def _cache_session_source(self, session_key: str, source) -> None: if not session_key or source is None: @@ -17962,218 +5027,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew self._async_session_store = facade return facade - async def _mark_durable_active_turn( - self, - event: "MessageEvent", - session_key: str, - ) -> bool: - """Persist the exact resolved routing key for this running turn.""" - try: - token = await self.async_session_store.mark_turn_active(session_key) - except Exception as exc: - logger.warning( - "Could not persist active-turn marker for %s: %s", - session_key, - exc, - ) - return False - if not token: - return False - # Private event attributes are process-local ownership state. Keep the - # token out of public metadata, transcripts, and platform payloads. - setattr(event, "_gateway_active_turn_session_key", session_key) - setattr(event, "_gateway_active_turn_token", token) - return True - - async def _clear_durable_active_turn(self, event: "MessageEvent") -> bool: - """Best-effort CAS clear of the marker owned by *event*.""" - session_key = getattr(event, "_gateway_active_turn_session_key", None) - token = getattr(event, "_gateway_active_turn_token", None) - try: - if not session_key or not token: - return False - last_error: Optional[Exception] = None - for attempt in range(1, 4): - try: - return bool( - await self.async_session_store.clear_turn_active( - session_key, token - ) - ) - except Exception as exc: - last_error = exc - if attempt < 3: - logger.debug( - "Retrying active-turn marker cleanup for %s (%d/3): %s", - session_key, - attempt, - exc, - ) - # Never let marker cleanup block agent/lease release; a stale marker is bounded by the - # agent timeout and the clean-start orphan-marker discard path. - logger.warning( - "Could not clear active-turn marker for %s after 3 attempts: %s", - session_key, - last_error, - ) - return False - finally: - for attr in ( - "_gateway_active_turn_session_key", - "_gateway_active_turn_token", - ): - with suppress(AttributeError): - delattr(event, attr) - - def _install_plugin_message_injector(self) -> None: - """Publish this live gateway's plugin message scheduler.""" - from hermes_cli.plugins import get_plugin_manager - - get_plugin_manager().set_gateway_message_injector( - self, - self._schedule_plugin_message_injection, - ) - - def _clear_plugin_message_injector(self) -> None: - """Remove this runner's scheduler without clobbering a newer owner.""" - from hermes_cli.plugins import get_plugin_manager - - get_plugin_manager().clear_gateway_message_injector(self) - - def _schedule_plugin_message_injection( - self, - *, - session_key: str, - content: str, - plugin_id: str, - ) -> bool: - """Schedule a plugin-triggered turn on the live gateway loop.""" - loop = getattr(self, "_gateway_loop", None) - if not getattr(self, "_running", False) or loop is None or loop.is_closed(): - return False - - coro = self._dispatch_plugin_message_injection( - session_key=session_key, - content=content, - plugin_id=plugin_id, - ) - try: - current_loop = asyncio.get_running_loop() - except RuntimeError: - current_loop = None - - if current_loop is loop: - try: - future = loop.create_task(coro) - except Exception: - coro.close() - logger.warning( - "Plugin message injection scheduling failed", - exc_info=True, - ) - return False - self._background_tasks.add(future) - future.add_done_callback(self._background_tasks.discard) - else: - future = safe_schedule_threadsafe( - coro, - loop, - logger=logger, - log_message="Plugin message injection scheduling failed", - log_level=logging.WARNING, - ) - if future is None: - return False - - def _log_result(completed) -> None: - try: - accepted = completed.result() - except (asyncio.CancelledError, concurrent.futures.CancelledError): - return - except Exception: - logger.warning( - "Plugin message injection failed: plugin=%s session=%s", - plugin_id, - session_key, - exc_info=True, - ) - return - if not accepted: - logger.warning( - "Plugin message injection was not routed: plugin=%s session=%s", - plugin_id, - session_key, - ) - - future.add_done_callback(_log_result) - return True - - async def _dispatch_plugin_message_injection( - self, - *, - session_key: str, - content: str, - plugin_id: str, - ) -> bool: - """Route a plugin-triggered turn through the session's live adapter.""" - if not getattr(self, "_running", False) or getattr(self, "_draining", False): - return False - - entry = await self.async_session_store.lookup_by_session_key(session_key) - if entry is None or entry.origin is None: - return False - if not getattr(self, "_running", False) or getattr(self, "_draining", False): - return False - - source = dataclasses.replace(entry.origin) - try: - if not self._is_user_authorized( - source, - allow_adapter_delegation=False, - ): - logger.warning( - "Plugin message injection denied by current gateway authorization: " - "plugin=%s session=%s", - plugin_id, - session_key, - ) - return False - except Exception: - logger.warning( - "Plugin message injection authorization check failed: " - "plugin=%s session=%s", - plugin_id, - session_key, - exc_info=True, - ) - return False - - adapter = self._adapter_for_source(source) - if adapter is None: - return False - - event = MessageEvent( - text=content, - message_type=MessageType.TEXT, - source=source, - internal=True, - allow_gateway_control=False, - metadata={ - "hermes_plugin_id": plugin_id, - "hermes_plugin_injection": True, - "gateway_session_key": session_key, - "gateway_session_id": entry.session_id, - "gateway_session_strict": True, - }, - ) - await adapter.handle_message(event) - logger.info( - "Plugin message injection dispatched: plugin=%s session=%s session_id=%s", - plugin_id, - session_key, - entry.session_id, - ) - return True def _get_cached_session_source(self, session_key: str): if not session_key: @@ -18187,4708 +5040,43 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew cached_sources.move_to_end(session_key) return source - async def _handle_message_with_agent(self, event, source, _quick_key: str, run_generation: int): - """Inner handler that runs under the _running_agents sentinel guard.""" - _msg_start_time = time.time() - _platform_name = source.platform.value if hasattr(source.platform, "value") else str(source.platform) - _msg_preview = (event.text or "")[:80].replace("\n", " ") - _reply_id = getattr(event, "reply_to_message_id", None) - _reply_txt = (getattr(event, "reply_to_text", None) or "")[:80].replace("\n", " ") - logger.info( - "inbound message: platform=%s user=%s chat=%s msg=%r reply_to_id=%s reply_to_text=%r", - _platform_name, source.user_name or source.user_id or "unknown", - source.chat_id or "unknown", _msg_preview, _reply_id, _reply_txt, - ) - - # Get or create session Topic-mode DMs: rewrite a stale/foreign thread_id to the user's - # last-active topic so a cross-topic Reply or stripped plain reply doesn't fragment the - # conversation across sessions. - recovered = await asyncio.to_thread(self._recover_telegram_topic_thread_id, source) - if recovered is not None: - logger.info( - "telegram topic recovery: chat=%s user=%s %r -> %s", - source.chat_id, source.user_id, source.thread_id, recovered, - ) - source = dataclasses.replace(source, thread_id=recovered) - with suppress(Exception): - event.source = source - - event_metadata = getattr(event, "metadata", None) or {} - expected_session_key = str( - event_metadata.get("gateway_session_key") or "" - ).strip() - if expected_session_key: - derived_session_key = self._session_key_for_source(source) - if derived_session_key != expected_session_key: - logger.warning( - "Dropping internally routed event after route recovery: " - "expected session=%s derived=%s", - expected_session_key, - derived_session_key, - ) - return - - strict_session = bool(event_metadata.get("gateway_session_strict")) - pinned_session_id = str(event_metadata.get("gateway_session_id") or "").strip() - if strict_session: - session_entry = await self.async_session_store.lookup_by_session_key( - expected_session_key - ) - if ( - session_entry is None - or not pinned_session_id - or session_entry.session_id != pinned_session_id - ): - logger.warning( - "Dropping internally routed event: expected session id=%s is no " - "longer current for key=%s", - pinned_session_id or "missing", - expected_session_key or "missing", - ) - return - else: - # Internal wakes must observe reset policy without becoming user activity themselves. - # Otherwise periodic Kanban/process notifications keep the stable routing key alive - # across every daily/idle boundary. - session_entry = await self.async_session_store.get_or_create_session( - source, - touch_activity=not bool(getattr(event, "internal", False)), - ) - session_key = session_entry.session_key - if not strict_session and pinned_session_id: - resolved_entry = await self._resolve_async_delegation_session( - session_entry, - pinned_session_id, - ) - if resolved_entry is None: - return - session_entry = resolved_entry - self._cache_session_source(session_key, source) - if await asyncio.to_thread(self._is_telegram_topic_lane, source): - try: - binding = (await self._session_db.get_telegram_topic_binding( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - profile_name=self._telegram_topic_profile_name(source), - )) if self._session_db else None - except Exception: - logger.debug("Failed to read Telegram topic binding", exc_info=True) - binding = None - if binding: - bound_session_id = str(binding.get("session_id") or "") - # Heal bindings that point at a pre-compression parent: walk the compression- - # continuation chain forward to its tip so the next message resumes the compressed - # child instead of reloading the oversized parent transcript. - if bound_session_id and self._session_db is not None: - try: - canonical_session_id = await self._session_db.get_compression_tip( - bound_session_id, - ) - except Exception: - logger.debug( - "compression-tip lookup failed for %s", - bound_session_id, exc_info=True, - ) - canonical_session_id = bound_session_id - if ( - canonical_session_id - and canonical_session_id != bound_session_id - ): - bound_session_id = canonical_session_id - if bound_session_id and bound_session_id != session_entry.session_id: - # Route the override through SessionStore so the session_key → session_id - # mapping is persisted to disk and the previous lane session is ended cleanly. - # Mutating session_entry in place created a split-brain: the JSON index pointed - # at one id while downstream code used another. - switched = await self.async_session_store.switch_session(session_key, bound_session_id) - if switched is not None: - session_entry = switched - # If the stored binding pointed at a parent, rewrite it to the - # canonical descendant now that we've followed the chain. - if ( - bound_session_id - and bound_session_id != str(binding.get("session_id") or "") - ): - await asyncio.to_thread( - self._sync_telegram_topic_binding, - source, session_entry, reason="compression-tip-walk", - ) - else: - try: - await asyncio.to_thread(self._record_telegram_topic_binding, source, session_entry) - except Exception: - logger.debug("Failed to record Telegram topic binding", exc_info=True) - # Capture and consume was_auto_reset immediately so it cannot re-fire on later messages and - # wipe model/reasoning overrides set between turns. - _was_auto_reset = getattr(session_entry, "was_auto_reset", False) - if _was_auto_reset: - # Auto-reset is a full conversation boundary: one funnel call clears every conversation- - # scoped per-session dict so the fresh session inherits no model/reasoning overrides, no - # queued "/model switched" note and no stale resolved-model cache. - self._clear_conversation_scope(session_key, reason="auto_reset") - # Evict the cached agent: the cache is keyed on the stable session_key, so an auto-reset - # would otherwise reuse the old agent and leak context_compressor._previous_summary - # (prior history) into new compaction summaries. - self._evict_cached_agent(session_key) - session_entry.was_auto_reset = False - - # Emit session:start for new or auto-reset sessions - _is_new_session = ( - session_entry.created_at == session_entry.updated_at - or _was_auto_reset - or getattr(session_entry, "is_fresh_reset", False) - ) - # Consume the is_fresh_reset flag immediately so it doesn't leak - # onto subsequent messages in the same session (issue #6508). - if getattr(session_entry, "is_fresh_reset", False): - session_entry.is_fresh_reset = False - if _is_new_session: - await self.hooks.emit("session:start", { - "platform": source.platform.value if source.platform else "", - "user_id": source.user_id, - "session_id": session_entry.session_id, - "session_key": session_key, - }) - - # Build session context - context = build_session_context(source, self.config, session_entry) - - # Set session context variables for tools (task-local, concurrency-safe) - _session_env_tokens = self._set_session_env(context) - - # Read privacy.redact_pii from config (re-read per message) - _redact_pii = False - persist_user_message = None - persist_user_timestamp = None - # Synthetic self-injected turns (batch completions, watch notifications, resume wake-ups) - # arrive as MessageEvent(internal=True). Persist with display_kind="internal_notification" - # so UIs render timeline notices, not user bubbles. display_kind is a DB-only sidecar - # stripped from every provider-bound payload; role/content untouched. - persist_user_display_kind = ( - "internal_notification" if getattr(event, "internal", False) else None - ) - try: - _pcfg = _load_gateway_config() - _redact_pii = bool((_pcfg.get("privacy") or {}).get("redact_pii", False)) - except Exception: - pass - - # Build the context prompt. The render is pinned per session, keyed by a hash of the exact - # renderer inputs (_ephemeral_change_key): a hit reuses the pinned bytes so the system prompt - # cannot drift turn-over-turn; a miss (thread rename, /sethome, redact_pii flip) re-renders. - context_prompt = self._pinned_session_context_prompt( - context, _redact_pii, session_key - ) - - # Per-turn must-deliver notes ride the user message via the api_content sidecar (staged - # below, consumed in run_sync → build_turn_context), NOT context_prompt: appending them to - # the ephemeral system prompt guaranteed a turn1→turn2 diff and a full agent rebuild. - turn_sidecar_notes: List[str] = [] - - # If the previous session expired and was auto-reset, deliver a notice - # so the agent knows this is a fresh conversation (not an intentional /reset). - if _was_auto_reset: - reset_reason = getattr(session_entry, 'auto_reset_reason', None) or 'idle' - context_note = _AUTO_RESET_CONTEXT_NOTES.get(reset_reason, _AUTO_RESET_CONTEXT_NOTES["idle"]) - # Slack/Discord channels/threads are long-lived: point the agent at the specific prior - # same-channel session so it recalls that context via session_search instead of an - # unrelated recent session. Deterministic — no extra API/DB calls. - try: - continuity_note = build_channel_continuity_note(session_entry, source) - except Exception: - continuity_note = None - if continuity_note: - context_note = context_note + "\n\n" + continuity_note - turn_sidecar_notes.append(context_note) - - # Notify the user about the reset unless notifications are disabled in config, the - # platform is excluded (e.g. api_server, webhook), or the expired session had no - # activity. - try: - policy = self.session_store.config.get_reset_policy( - platform=source.platform, - session_type=getattr(source, 'chat_type', 'dm'), - ) - platform_name = source.platform.value if source.platform else "" - had_activity = getattr(session_entry, 'reset_had_activity', False) - # Suspended and restart-recovery-expired sessions always notify regardless of - # policy.notify — the user had an active session that was silently replaced, so they - # need to know they can /resume it. Idle/daily resets respect the policy flag. - should_notify = reset_reason in {"suspended", "resume_pending_expired"} or ( - policy.notify - and had_activity - and platform_name not in policy.notify_exclude_platforms - ) - if should_notify: - adapter = self._adapter_for_source(source) - if adapter: - reason_text = _auto_reset_reason_text(reset_reason, policy) - notice = ( - f"◐ Session automatically reset ({reason_text}). " - f"Conversation history cleared.\n" - f"Use /resume to browse and restore a previous session.\n" - f"Adjust reset timing in config.yaml under session_reset." - ) - try: - session_info = await asyncio.to_thread( - self._reset_notice_session_info, source - ) - if session_info: - notice = f"{notice}\n\n{session_info}" - except Exception: - pass - await adapter.send( - source.chat_id, notice, - metadata=self._thread_metadata_for_source(source), - ) - except Exception as e: - logger.debug("Auto-reset notification failed (non-fatal): %s", e) - - # was_auto_reset is already consumed in the cleanup block above - # (single source of truth); only the reset reason needs clearing here. - session_entry.auto_reset_reason = None - - # Auto-load skill(s) for topic/channel bindings (Telegram DM Topics, Discord - # channel_skill_bindings). Supports a single name or ordered list. Only inject on NEW - # sessions — ongoing conversations already carry the skill content in their history. - _auto = getattr(event, "auto_skill", None) - if _is_new_session and _auto: - _skill_names = [_auto] if isinstance(_auto, str) else list(_auto) - try: - from agent.skill_commands import _load_skill_payload, _build_skill_message - _combined_parts: list[str] = [] - _loaded_names: list[str] = [] - for _sname in _skill_names: - _loaded = _load_skill_payload(_sname, task_id=_quick_key) - if _loaded: - _loaded_skill, _skill_dir, _display_name = _loaded - _note = ( - f'[IMPORTANT: The "{_display_name}" skill is auto-loaded. ' - f"Follow its instructions for this session.]" - ) - _part = _build_skill_message(_loaded_skill, _skill_dir, _note) - if _part: - _combined_parts.append(_part) - _loaded_names.append(_sname) - else: - logger.warning("[Gateway] Auto-skill '%s' not found", _sname) - if _combined_parts: - # Append the user's original text after all skill payloads - _combined_parts.append(event.text) - event.text = "\n\n".join(_combined_parts) - logger.info( - "[Gateway] Auto-loaded skill(s) %s for session %s", - _loaded_names, session_key, - ) - except Exception as e: - logger.warning("[Gateway] Failed to auto-load skill(s) %s: %s", _skill_names, e) - - # ── Turn lease: session resolution is FINAL here. Serialize [load history → run → flush] - # per resolved SESSION_ID: another routing key mapped to the same session_id waits for the - # prior flush instead of loading a stale base and interleaving writes (same-key messages - # never reach here mid-turn thanks to adapter + runner guards, so the lock is otherwise - # uncontended). Fail-closed on timeout: never enter the transcript region without a lease; - # outer dispatch returns a bounded resend notice. Released in _handle_message's finally - # (_release_turn_lease), granted per (routing key, run generation) so a stale unwind can't - # release a newer turn's. - _lease_registry = getattr(self, "_turn_leases", None) - if _lease_registry is not None: - try: - _lease_token = await _lease_registry.acquire( - session_entry.session_id, - owner_key=_quick_key, - generation=run_generation, - timeout=_float_env( - "HERMES_TURN_LEASE_TIMEOUT", DEFAULT_LEASE_WAIT - ), - ) - except TurnLeaseTimeoutError: - # The broad session-context cleanup finally starts later in this - # method. Restore the tokens here before propagating the rejection - # to outer dispatch, or this early exit leaks task-local identity. - self._clear_session_env(_session_env_tokens) - raise - if _lease_token is not None: - _lease_state = self._session_state(_quick_key).turn - _lease_state.lease_token = _lease_token - _lease_state.lease_generation = run_generation - - # A turn only becomes durable recovery work after it owns (or has explicitly degraded past) - # the per-session lease. Marking before the await above would falsely recover an alias- - # routed message that never began processing if the gateway died while it was still waiting. - await self._mark_durable_active_turn(event, session_entry.session_key) - - # Load conversation history from transcript. An unreadable canonical store is not an empty - # conversation: stop before the agent can invent continuity from a plausible-looking []. - # This return happens before the broad cleanup finally below, so restore task-local context - # here; the outer dispatch still clears the durable marker and turn lease. - try: - history = await self.async_session_store.load_transcript( - session_entry.session_id - ) - except TranscriptReadError: - self._clear_session_env(_session_env_tokens) - return ( - "⚠️ This session's history is temporarily unavailable, so " - "this message was not processed. Ask the operator to inspect " - "state.db, then resend after it is healthy. Use /reset only " - "if you intentionally want to start a new conversation." - ) - - # Session hygiene: auto-compress pathologically large transcripts before the agent starts so - # oversized histories don't cause repeated truncation/context failures. Token source: the - # API's prompt_tokens from the last turn (session_entry.last_prompt_tokens), else a char/4 - # estimate (30-50% high on code-heavy sessions, so hygiene merely fires a bit early). - if history and len(history) >= 4: - from agent.model_metadata import ( - estimate_messages_tokens_rough, - get_model_context_length_async, - ) - - # Read model + compression config. Hygiene threshold is intentionally HIGHER than the - # agent's own compressor (0.85 vs 0.50): it is a pre-agent safety net for sessions that - # grew between turns; at 0.50 it compressed prematurely on every turn in long sessions. - _hyg_model = "anthropic/claude-sonnet-4.6" - _hyg_threshold_pct = 0.85 - _hyg_compression_enabled = True - _hyg_hard_msg_limit = 5000 - _hyg_timeout_seconds = 30.0 - _hyg_total_ceiling_seconds = 600.0 - # Max wall-clock the user's TURN is held waiting on hygiene compression before the - # gateway stops waiting and proceeds on the uncompressed transcript. The compressor keeps - # running detached; its commit is fenced (revoke_commit_admission) so a stale result can - # never clobber later turns. Kept well below transport idle-timeouts (Telegram ~30s). - _hyg_max_turn_hold_seconds = 10.0 - _hyg_failure_cooldown_seconds = 300.0 - _hyg_config_context_length = None - _hyg_provider = None - _hyg_base_url = None - _hyg_api_key = None - _hyg_configured_model = None - _hyg_configured_provider = None - _hyg_configured_base_url = None - _hyg_data = {} - try: - _hyg_data = _load_gateway_config() - if _hyg_data: - # Resolve model name (same logic as run_sync) - _model_cfg = _hyg_data.get("model", {}) - if isinstance(_model_cfg, str): - _hyg_model = _model_cfg - elif isinstance(_model_cfg, dict): - _hyg_model = _model_cfg.get("default") or _model_cfg.get("model") or _hyg_model - # Read explicit context_length override from model config - # (same as run_agent.py lines 995-1005) - _raw_ctx = _model_cfg.get("context_length") - if _raw_ctx is not None: - with suppress(TypeError, ValueError): - _hyg_config_context_length = int(_raw_ctx) - # Read provider for accurate context detection - _hyg_provider = _model_cfg.get("provider") or None - _hyg_base_url = _model_cfg.get("base_url") or None - - # Only the enabled flag is read; hygiene's threshold is deliberately separate - # from the agent's compression.threshold (hygiene runs higher). - _comp_cfg = _hyg_data.get("compression", {}) - if isinstance(_comp_cfg, dict): - _hyg_compression_enabled = str( - _comp_cfg.get("enabled", True) - ).lower() in {"true", "1", "yes"} - _raw_hard_limit = _comp_cfg.get("hygiene_hard_message_limit") - if _raw_hard_limit is not None: - try: - _parsed = int(_raw_hard_limit) - if _parsed > 0: - _hyg_hard_msg_limit = _parsed - except (TypeError, ValueError): - pass - _raw_timeout = _comp_cfg.get("hygiene_timeout_seconds") - if _raw_timeout is not None: - try: - _parsed = float(_raw_timeout) - if _parsed > 0: - _hyg_timeout_seconds = _parsed - except (TypeError, ValueError): - pass - _raw_ceiling = _comp_cfg.get("hygiene_total_ceiling_seconds") - if _raw_ceiling is not None: - try: - _parsed = float(_raw_ceiling) - if _parsed > 0: - _hyg_total_ceiling_seconds = _parsed - except (TypeError, ValueError): - pass - # The ceiling can never be tighter than one idle - # window, or the extension loop would be dead code. - _hyg_total_ceiling_seconds = max( - _hyg_total_ceiling_seconds, _hyg_timeout_seconds, - ) - _raw_turn_hold = _comp_cfg.get("hygiene_max_turn_hold_seconds") - if _raw_turn_hold is not None: - try: - _parsed = float(_raw_turn_hold) - if _parsed > 0: - _hyg_max_turn_hold_seconds = _parsed - except (TypeError, ValueError): - pass - _raw_cooldown = _comp_cfg.get("hygiene_failure_cooldown_seconds") - if _raw_cooldown is not None: - try: - _parsed = float(_raw_cooldown) - if _parsed >= 0: - _hyg_failure_cooldown_seconds = _parsed - except (TypeError, ValueError): - pass - - _hyg_configured_model = _hyg_model - _hyg_configured_provider = _hyg_provider - _hyg_configured_base_url = _hyg_base_url - - try: - _hyg_model, _hyg_runtime = self._resolve_session_agent_runtime( - source=source, - session_key=session_key, - user_config=_hyg_data if isinstance(_hyg_data, dict) else None, - ) - _hyg_provider = _hyg_runtime.get("provider") or _hyg_provider - _hyg_base_url = _hyg_runtime.get("base_url") or _hyg_base_url - _hyg_api_key = _hyg_runtime.get("api_key") or _hyg_api_key - except Exception: - pass - - if _hyg_config_context_length is not None: - try: - from hermes_cli.route_identity import should_clear_context_pin_async - - if await should_clear_context_pin_async( - _hyg_configured_model, - _hyg_model, - _hyg_configured_base_url, - _hyg_base_url, - _hyg_configured_provider, - _hyg_provider, - ): - _hyg_config_context_length = None - except Exception: - _hyg_config_context_length = None - - # custom_providers per-model context_length fallback (as in run_agent.py); must run - # after runtime resolution so _hyg_base_url is set. - if _hyg_config_context_length is None and _hyg_base_url: - try: - try: - from hermes_cli.config import ( - get_compatible_custom_providers as _gw_gcp, - get_custom_provider_context_length as _gw_gccl, - ) - _hyg_custom_providers = _gw_gcp(_hyg_data) - except Exception: - _hyg_custom_providers = _hyg_data.get("custom_providers") - if not isinstance(_hyg_custom_providers, list): - _hyg_custom_providers = [] - _hyg_custom_ctx = _gw_gccl( - model=_hyg_model, - base_url=_hyg_base_url, - custom_providers=_hyg_custom_providers, - ) - if _hyg_custom_ctx: - _hyg_config_context_length = int(_hyg_custom_ctx) - except (TypeError, ValueError): - pass - except Exception: - pass - - if _hyg_compression_enabled: - _hyg_context_length = await get_model_context_length_async( - _hyg_model, - base_url=_hyg_base_url or "", - api_key=_hyg_api_key or "", - config_context_length=_hyg_config_context_length, - provider=_hyg_provider or "", - ) - _compress_token_threshold = int( - _hyg_context_length * _hyg_threshold_pct - ) - _warn_token_threshold = int(_hyg_context_length * 0.95) - - _msg_count = len(history) - - # Prefer actual API-reported tokens from the last turn - # (stored in session entry) over the rough char-based estimate. - _stored_tokens = session_entry.last_prompt_tokens - if _stored_tokens > 0: - _approx_tokens = _stored_tokens - _token_source = "actual" - else: - _approx_tokens = estimate_messages_tokens_rough(history) - _token_source = "estimated" - # Rough estimates run 30-50% high on code/JSON-heavy sessions, which only makes - # hygiene fire early (safe). Do NOT compensate with a threshold multiplier: 85% - # * 1.4 = 119% of context kept hygiene from ever firing for ~200K models. - - # Hard safety valve: force compression at an extreme message count regardless of token - # estimates, breaking the spiral where API disconnects prevent token data → no - # compression → more disconnects. Default 5000 sits clear of legitimate 1M+ context - # sessions (those compress on tokens). Config: compression.hygiene_hard_message_limit. - _HARD_MSG_LIMIT = _hyg_hard_msg_limit - _needs_compress = ( - _approx_tokens >= _compress_token_threshold - or _msg_count >= _HARD_MSG_LIMIT - ) - - if _needs_compress: - # Use the persistent DB-backed cooldown (same as the in-conversation compression - # path in context_compressor.py) so the cooldown survives gateway restarts. The - # in-memory dict reset on every restart, re-triggering the same failing - # compression and wedging session storage. - _session_db = getattr(self, "_session_db", None) - if _session_db is not None: - _session_db = getattr(_session_db, "_db", _session_db) - _getter = getattr(_session_db, "get_compression_failure_cooldown", None) - if _getter is not None: - try: - _cooldown_state = _getter(session_entry.session_id) - except Exception: - _cooldown_state = None - if _cooldown_state and _cooldown_state.get("remaining_seconds", 0) > 0: - logger.info( - "Session hygiene: skipping compression for %s; " - "previous failure cooldown active for %.1fs", - session_entry.session_id, - _cooldown_state["remaining_seconds"], - ) - _needs_compress = False - - if _needs_compress and await self._session_has_compression_in_flight( - session_key - ): - # A prior hygiene/agent compression still holds the durable lock (typically a - # shielded worker left behind by /stop or /restart). Starting another attempt - # would wait up to the 600s ceiling behind a commit the fence will refuse, while - # inbound messages demote to queue. - logger.info( - "Session hygiene: skipping compression for %s; " - "another compression is already in flight", - session_entry.session_id, - ) - _needs_compress = False - - if _needs_compress: - logger.info( - "Session hygiene: %s messages, ~%s tokens (%s) — auto-compressing " - "(threshold: %s%% of %s = %s tokens)", - _msg_count, f"{_approx_tokens:,}", _token_source, - int(_hyg_threshold_pct * 100), - f"{_hyg_context_length:,}", - f"{_compress_token_threshold:,}", - ) - - _hyg_meta = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) - - try: - from agent.conversation_compression import CompressionCommitFence - from run_agent import AIAgent - - _hyg_model, _hyg_runtime = self._resolve_session_agent_runtime( - source=source, - session_key=session_key, - user_config=_hyg_data if isinstance(_hyg_data, dict) else None, - ) - _hyg_api_mode = str( - _hyg_runtime.get("api_mode") or "" - ).lower() - if _hyg_api_mode == "codex_app_server": - # codex app-server runtime: the real context is the server-side thread, - # not the transcript mirror. The detached-agent block below would only - # rewrite the mirror and its finally-eviction would destroy the live - # thread (next turn starts blank). Use the cached agent's - # thread/compact/start and KEEP it cached. - _hyg_codex_auto = "native" - _hyg_comp_cfg = ( - _hyg_data.get("compression") - if isinstance(_hyg_data, dict) - else None - ) - if isinstance(_hyg_comp_cfg, dict): - _hyg_codex_auto = str( - _hyg_comp_cfg.get( - "codex_app_server_auto", "native" - ) - or "native" - ) - _hyg_codex_outcome = await run_codex_hygiene_compaction( - self, - session_key, - session_entry.session_id, - auto_mode=_hyg_codex_auto, - history=history, - approx_tokens=_approx_tokens, - timeout_seconds=_hyg_total_ceiling_seconds, - failure_cooldown_seconds=_hyg_failure_cooldown_seconds, - ) - logger.info( - "Session hygiene (codex app-server): %s " - "(session=%s, mode=%s, ~%s tokens)", - _hyg_codex_outcome, - session_entry.session_id, - _hyg_codex_auto, - f"{_approx_tokens:,}", - ) - elif _hyg_runtime.get("api_key"): - # Pass the FULL transcript (tool results included), matching the agent - # loop: filtering to user/assistant starved the compressor — tool results - # are the bulk of context and short histories tripped the - # protect-first/last early-return so nothing compressed. - _hyg_msgs = [ - m for m in history - if m.get("role") in {"user", "assistant", "tool"} - ] - - if len(_hyg_msgs) >= 4: - try: - _hyg_session_row = await self._session_db.get_session( - session_entry.session_id - ) - except Exception as exc: - _hyg_session_row = None - logger.warning( - "Session hygiene could not restore the system " - "prompt for session %s: %s. Preserving an empty " - "prompt so the live turn rebuilds it with its " - "configured providers.", - session_entry.session_id, - exc, - exc_info=True, - ) - _hyg_session_db = getattr(self._session_db, "_db", self._session_db) - # Hygiene is the same lossy rewrite as normal compression: when - # compression.checkpoint_required is on, load the memory provider so - # the checkpoint exists before any mutation; otherwise keep the - # fast path (no provider init, no best-effort hook). - from hermes_cli.config import load_config as _load_cfg - from utils import is_truthy_value as _is_truthy - - _hyg_checkpoint_required = _is_truthy( - ((_load_cfg() or {}).get("compression") or {}).get( - "checkpoint_required" - ), - default=False, - ) - _hyg_agent = AIAgent( - **_hyg_runtime, - model=_hyg_model, - max_iterations=4, - quiet_mode=True, - skip_memory=not _hyg_checkpoint_required, - enabled_toolsets=["memory"], - session_id=session_entry.session_id, - session_db=_hyg_session_db, - ) - _seed_hygiene_system_prompt( - _hyg_agent, - _hyg_session_row, - ) - # If compression must rebuild instead of retaining - # the cached prompt, make the persisted result - # deliberately stale for every real gateway surface. - _hyg_agent.platform = _GATEWAY_HYGIENE_PLATFORM - _hyg_cleanup_deferred = False - try: - # Hygiene runs before the turn and owns the session binding, so - # prefer in-place compaction: archive old rows under the same - # session id rather than minting a continuation child that must - # be published back to SessionStore/topic bindings. Without a - # SessionDB this stays False and the guard below preserves it. - _hyg_agent.compression_in_place = True - _bind_hyg_state = getattr( - getattr(_hyg_agent, "context_compressor", None), - "bind_session_state", - None, - ) - if callable(_bind_hyg_state): - _bind_hyg_state( - _hyg_session_db, - session_entry.session_id, - ) - # It must never finalize on close() — close() - # would end the live gateway session row. - _hyg_agent._end_session_on_close = False - _hyg_agent._print_fn = lambda *a, **kw: None - - loop = asyncio.get_running_loop() - _hyg_commit_fence = CompressionCommitFence( - total_ceiling_seconds=_hyg_total_ceiling_seconds - ) - # Default executor (NOT self._get_executor): a fence-cancelled - # hung summary must never occupy an agent-work slot. But it MUST - # run in the caller's contextvars: under multiplex_profiles the - # secret scope / HERMES_HOME live in ContextVars, and an empty - # Context makes get_secret() fail closed → lossy truncation. - _hyg_future = loop.run_in_executor( - None, - copy_context().run, - lambda: _hyg_agent._compress_context( - _hyg_msgs, "", - approx_tokens=_approx_tokens, - commit_fence=_hyg_commit_fence, - ), - ) - try: - # Progress-aware wait: the timeout is an INACTIVITY budget — - # the worker ticks the fence per streamed token, so a slow but - # still-generating model extends the deadline. A hard ceiling - # bounds the total so a trickle stream can't hold the turn. - _hyg_wait_started = time.monotonic() - while True: - if _hyg_commit_fence.is_cancelled: - raise asyncio.TimeoutError - # Charge the idle budget from the LAST PROGRESS event, - # not from the start of this wait slice — otherwise - # silence can approach 2x the configured timeout. - _hyg_waited = ( - time.monotonic() - _hyg_wait_started - ) - _slice = min( - max( - _hyg_timeout_seconds - - _hyg_commit_fence.seconds_since_progress(), - 0.005, - ), - max( - _hyg_total_ceiling_seconds - - _hyg_waited, - 0.005, - ), - ) - # Bounded turn-hold: cap this slice at the remaining - # turn-hold budget so it is re-checked against - # _hyg_max_turn_hold_seconds at least that often — - # otherwise a continuously-streaming worker keeps the - # slice large and holds the turn until the ceiling. - _turn_hold_remaining = ( - _hyg_max_turn_hold_seconds - - (time.monotonic() - _hyg_wait_started) - ) - if _turn_hold_remaining <= 0: - # Budget exhausted: force an immediate timeout so - # the abandonment path below runs. - _slice = 0.005 - else: - _slice = min( - _slice, - max(_turn_hold_remaining, 0.005), - ) - # Re-check the fence on a short poll so a - # /stop or /restart cancel is not stuck - # behind a full idle window (#96953). - _idle_left = max( - _hyg_timeout_seconds - - _hyg_commit_fence.seconds_since_progress(), - 0.005, - ) - _slice = min(_slice, 0.25) - try: - _compressed, _ = await asyncio.wait_for( - asyncio.shield(_hyg_future), - timeout=_slice, - ) - break - except asyncio.TimeoutError: - if _hyg_commit_fence.is_cancelled: - raise - _hyg_waited = time.monotonic() - _hyg_wait_started - _idle = _hyg_commit_fence.seconds_since_progress() - # Bounded turn-hold: never hold the user's TURN past - # _hyg_max_turn_hold_seconds even if the summary - # model is still streaming; fall through to the - # timeout path, which revokes commit admission and - # proceeds on the uncompressed transcript, so the - # wire never trips a transport idle-timeout. - if ( - _hyg_waited - >= _hyg_max_turn_hold_seconds - ): - logger.info( - "Session hygiene compression for " - "session %s exceeded the turn-hold " - "budget (%.1fs >= %.1fs) — " - "abandoning inline wait, proceeding " - "without compression this turn", - session_entry.session_id, - _hyg_waited, - _hyg_max_turn_hold_seconds, - ) - raise HygieneTurnHoldExceeded( - f"turn-hold budget {_hyg_max_turn_hold_seconds:.1f}s " - f"elapsed after {_hyg_waited:.1f}s" - ) - if hygiene_wait_should_extend( - idle=_idle, - timeout=_hyg_timeout_seconds, - waited=_hyg_waited, - ceiling=_hyg_total_ceiling_seconds, - fence_cancelled=_hyg_commit_fence.is_cancelled, - ): - if _slice >= _idle_left - 1e-9: - logger.info( - "Session hygiene compression for " - "session %s still streaming after " - "%.0fs (last progress %.1fs ago) — " - "extending wait (ceiling %.0fs)", - session_entry.session_id, - _hyg_waited, _idle, - _hyg_total_ceiling_seconds, - ) - continue - raise - except HygieneTurnHoldExceeded: - # Turn-hold expiry is an availability boundary, not a - # failure: the compressor is healthy and still streaming; we - # just can't hold the turn longer. Share the safe mechanics - # (fence, release, defer, proceed uncompressed) with - # distinct provenance / user message and NO failure-cooldown - # increment. Decouple the TURN from the COMPRESSION: when - # the worker's commit is watermark-fenced (rows appended - # after compression start, this turn included, survive its - # late commit as cloned concurrent tail) the attempt KEEPS - # commit admission — the turn proceeds uncompressed NOW and - # the summary is adopted at the worker's own fenced commit - # (archive_and_compact / rotation publish); always - # cancelling burned every attempt for thinking summary - # models whose reasoning prefix alone exceeds the hold. If - # NOT watermark-fenced (no session_db, capture failed, - # legacy lock API) a late commit could clobber newer turns, - # so cancel. - _hyg_keep_admission = bool( - getattr( - _hyg_commit_fence, - "commit_watermark_fenced", - False, - ) - ) and not _hyg_commit_fence.is_cancelled - if _hyg_keep_admission: - self._defer_agent_cleanup_until_future_done( - _hyg_future, - _hyg_agent, - context="session hygiene turn-hold", - ) - _hyg_cleanup_deferred = True - # NO retry-after here: the attempt is still running - # toward a real commit, and the flat 60s retry-after - # would also block the agent-side preflight compressor - # (same-session cooldown). Re-attempt spacing comes from - # the durable compression lock instead: the next turn's - # hygiene pre-check skips while this worker's lease is - # held (_session_has_compression_in_flight). The - # done-callback below records the flat retry-after ONLY - # if the worker ends without committing anything. - _hyg_deferred_sid = session_entry.session_id - _hyg_deferred_key = session_key - _hyg_deferred_agent = _hyg_agent - - def _hyg_adopt_or_space_retry( - _fut, - _gw=self, - _sid=_hyg_deferred_sid, - _skey=_hyg_deferred_key, - _agent=_hyg_deferred_agent, - ): - try: - _exc = _fut.exception() - except ( - asyncio.CancelledError, - Exception, - ): - _exc = None - _committed = False - else: - _committed = _exc is None and ( - bool( - getattr( - _agent, - "_last_compaction_in_place", - False, - ) - ) - or getattr( - _agent, "session_id", _sid - ) - != _sid - ) - if _committed: - logger.info( - "Session hygiene compression for " - "session %s finished after the " - "turn-hold was released — summary " - "adopted at the watermark-fenced " - "commit boundary (#97963)", - _sid, - ) - try: - _reset_hygiene_failure_streak( - _gw, _skey - ) - except Exception as _rs_err: - logger.debug( - "hygiene streak reset after " - "deferred adoption failed: %s", - _rs_err, - ) - else: - # Nothing to adopt (summary failed, fence - # refused the commit, or the attempt was - # superseded). Record flat spacing so sustained - # traffic does not spawn and abandon a fresh - # compressor every turn; non-escalating — the - # failure streak must not advance for a deferral. - _record_hygiene_cooldown( - _gw, _sid, - _HYGIENE_TURNHOLD_RETRY_SECONDS, - "hygiene compression deferred: " - "turn-hold budget expired and the " - "detached attempt did not commit", - ) - - _hyg_future.add_done_callback( - _hyg_adopt_or_space_retry - ) - from agent.session_activity import ( - ActivityProvenance, - ) - _stamp_hygiene_compression_provenance( - _hyg_agent, - "session hygiene compression turn-hold", - ActivityProvenance.AGENT_COMPRESSION_TURNHOLD, - "hygiene compression turn-hold " - "activity stamp failed", - ) - logger.info( - "Session hygiene compression for session %s " - "exceeded turn-hold budget (%.1fs); " - "proceeding without compression this turn — " - "the watermark-fenced worker keeps its " - "commit admission and the summary will be " - "adopted when it finishes", - session_entry.session_id, - time.monotonic() - _hyg_wait_started, - ) - _turnhold_msg = t( - "gateway.compress.turnhold_deferred" - ) - try: - _adapter = self._adapter_for_source(source) - if _adapter and source.chat_id: - await _adapter.send( - source.chat_id, - _turnhold_msg, - metadata=_hyg_meta, - ) - except Exception as _werr: - logger.warning( - "Failed to deliver compression-turnhold " - "notice to user: %s", - _werr, - ) - raise - _cancelled = None - while _cancelled is None: - if _hyg_commit_fence.commit_in_flight: - _cancelled = False - break - _cancelled = ( - _hyg_commit_fence.try_cancel_before_commit() - ) - if _cancelled is None: - await asyncio.sleep(0.025) - if not _cancelled: - # NOTE: bounded overshoot by design: the turn can be held - # past _hyg_max_turn_hold_seconds by up to the commit - # duration. Aborting mid-commit would corrupt the - # message-store transaction — the overshoot is the - # cheaper failure mode. Do NOT "fix" this into a - # mid-commit cancellation. - _compressed, _ = await _hyg_future - else: - _hyg_commit_fence.release_cancelled_compression_lock() - self._defer_agent_cleanup_until_future_done( - _hyg_future, - _hyg_agent, - context="session hygiene turn-hold", - ) - _hyg_cleanup_deferred = True - # Short, NON-escalating retry-after. Without it every - # turn re-spawns a compressor, holds it for the turn-hold - # budget and cancels it — token burn that never commits. - # Deliberately NOT _hygiene_cooldown_for_failure: the - # compressor is healthy, so the failure streak must not - # advance; only flat retry spacing is recorded. - _record_hygiene_cooldown( - self, session_entry.session_id, - _HYGIENE_TURNHOLD_RETRY_SECONDS, - "hygiene compression deferred: " - "turn-hold budget expired while the " - "summary was still streaming", - ) - from agent.session_activity import ( - ActivityProvenance, - ) - _stamp_hygiene_compression_provenance( - _hyg_agent, - "session hygiene compression turn-hold", - ActivityProvenance.AGENT_COMPRESSION_TURNHOLD, - "hygiene compression turn-hold " - "activity stamp failed", - ) - logger.info( - "Session hygiene compression for session %s " - "exceeded turn-hold budget (%.1fs); " - "proceeding without compression this turn", - session_entry.session_id, - time.monotonic() - _hyg_wait_started, - ) - _turnhold_msg = t( - "gateway.compress.turnhold_deferred" - ) - try: - _adapter = self._adapter_for_source(source) - if _adapter and source.chat_id: - await _adapter.send( - source.chat_id, - _turnhold_msg, - metadata=_hyg_meta, - ) - except Exception as _werr: - logger.warning( - "Failed to deliver compression-turnhold " - "notice to user: %s", - _werr, - ) - raise - except asyncio.TimeoutError: - _hyg_waited = time.monotonic() - _hyg_wait_started - _hyg_total_exhausted = ( - _hyg_waited >= _hyg_total_ceiling_seconds - or _hyg_commit_fence.deadline_exceeded - ) - if _hyg_total_exhausted: - # The worker cooperatively checks this deadline between - # digest calls. Keep its lease until it exits so an - # unchanged session cannot overlap a retry. - _hyg_commit_fence.retain_compression_lock_until_worker_done() - # Capture fence state BEFORE try_cancel — that call itself - # sets is_cancelled, which would mis-label a genuine idle - # timeout as a fence cancel. - _hyg_fence_cancelled = ( - _hyg_commit_fence.is_cancelled - ) - _cancelled = None - while _cancelled is None: - # #76354 F1: a hung commit retains the fence lock; the - # lock-free phase marker keeps this loop from spinning - # forever while the commit blocks. - if _hyg_commit_fence.commit_in_flight: - _cancelled = False - break - _cancelled = ( - _hyg_commit_fence.try_cancel_before_commit() - ) - if _cancelled is None: - # Round-2 #5: transient lock-setup windows ride - # write patience for seconds; 25ms keeps sub-tick - # latency without 1kHz spin. - await asyncio.sleep(0.025) - if not _cancelled: - # The worker crossed the commit boundary just before the - # timeout; the fence poll waited for it to finish, so - # consume the result instead of treating a successful - # compaction as a timeout. - _compressed, _ = await _hyg_future - else: - # Release an inactivity-timed-out worker's holder- - # qualified lease promptly. Total-ceiling attempts - # retained it above, so this is a no-op for them. - _hyg_commit_fence.release_cancelled_compression_lock() - self._defer_agent_cleanup_until_future_done( - _hyg_future, - _hyg_agent, - context="session hygiene timeout", - ) - _hyg_cleanup_deferred = True - _hyg_timeout_error = ( - "session hygiene compression " - "cancelled at commit fence" - if _hyg_fence_cancelled - else ( - "session hygiene compression " - "timed out with no output from " - "the summary model" - ) - ) - if _hyg_failure_cooldown_seconds >= 0: - _hyg_cooldown = await asyncio.to_thread( - _hygiene_cooldown_for_failure, - self, - session_key, - _hyg_failure_cooldown_seconds, - ) - _timeout_reason = ( - _hyg_timeout_error - if _hyg_fence_cancelled - else ( - "session hygiene compression total " - "ceiling exhausted" - if _hyg_total_exhausted - else "session hygiene compression " - "timed out with no output from the " - "summary model" - ) - ) - _record_hygiene_cooldown( - self, session_entry.session_id, - _hyg_cooldown, - _timeout_reason, - ) - from agent.session_activity import ( - ActivityProvenance, - ) - _stamp_hygiene_compression_provenance( - _hyg_agent, - ( - "session hygiene compression " - "cancelled at commit fence" - if _hyg_fence_cancelled - else "session hygiene compression timed out" - ), - ActivityProvenance.AGENT_COMPRESSION_TIMEOUT, - "hygiene compression timeout " - "activity stamp failed", - ) - if _hyg_fence_cancelled: - logger.warning( - "Session hygiene compression for " - "session %s was cancelled at the " - "commit fence; continuing without " - "compression", - session_entry.session_id, - ) - else: - _hyg_elapsed = ( - time.monotonic() - _hyg_wait_started - ) - if _hyg_total_exhausted: - logger.warning( - "Session hygiene compression for session %s " - "reached its total ceiling after %.1fs " - "(progress observed=%s); continuing without " - "compression", - session_entry.session_id, - _hyg_elapsed, - _hyg_commit_fence.progress_observed, - ) - else: - logger.warning( - "Session hygiene compression for session %s " - "made no progress for %.1fs (total wait " - "%.1fs, ceiling %.1fs); continuing without " - "compression", - session_entry.session_id, - _hyg_commit_fence.seconds_since_progress(), - _hyg_elapsed, - _hyg_total_ceiling_seconds, - ) - _timeout_msg = ( - _hygiene_compression_timeout_message( - total_exhausted=_hyg_total_exhausted, - elapsed=_hyg_elapsed, - idle_timeout=_hyg_timeout_seconds, - progress_observed=( - _hyg_commit_fence.progress_observed - ), - ) - ) - try: - _adapter = self._adapter_for_source(source) - if _adapter and source.chat_id: - await _adapter.send( - source.chat_id, - _timeout_msg, - metadata=_hyg_meta, - ) - except Exception as _werr: - logger.warning( - "Failed to deliver compression-timeout " - "warning to user: %s", - _werr, - ) - raise - except BaseException: - # #76354 F2: non-timeout unwind while the detached hygiene - # worker may still run — KeyboardInterrupt, task - # cancellation, or any unexpected error. Revoke commit - # admission (and release the worker's durable lease) BEFORE - # the host unwinds so the worker can never commit later. - _hyg_commit_fence.revoke_commit_admission() - if not _hyg_cleanup_deferred: - self._defer_agent_cleanup_until_future_done( - _hyg_future, - _hyg_agent, - context="session hygiene unwind", - ) - _hyg_cleanup_deferred = True - # restart drain / task cancel must record a cooldown, or the - # next turn immediately re-arms hygiene and waits up to 600s - # behind a fence that would refuse the commit again. - if _hyg_failure_cooldown_seconds >= 0: - try: - _hyg_cooldown = _hygiene_cooldown_for_failure( - self, - session_key, - _hyg_failure_cooldown_seconds, - ) - _record_hygiene_cooldown( - self, session_entry.session_id, - _hyg_cooldown, - "session hygiene compression " - "cancelled at commit fence", - ) - except Exception as _cd_err: - logger.debug( - "hygiene unwind cooldown " - "record failed: %s", - _cd_err, - ) - raise - - # _compress_context ends the old session and creates a new - # session_id. Write compressed messages into the NEW session so - # the old transcript stays intact and searchable. - _hyg_new_sid = _hyg_agent.session_id - _hyg_rotated = _hyg_new_sid != session_entry.session_id - _hyg_in_place = bool( - getattr(_hyg_agent, "_last_compaction_in_place", False) - ) - # Anti-growth guard: refuse a compression that did not shrink - # the transcript (observed: 427K -> 598K). Compare like-for-like - # rough estimates. - _hyg_in_toks = estimate_messages_tokens_rough(history) - _hyg_out_toks = estimate_messages_tokens_rough(_compressed) - if _hyg_rotated and _hyg_out_toks > _hyg_in_toks: - logger.warning( - "Gateway hygiene compression for session %s " - "would grow transcript (~%s -> ~%s tokens); " - "keeping the original transcript unchanged", - session_entry.session_id, - f"{_hyg_in_toks:,}", - f"{_hyg_out_toks:,}", - ) - _hyg_rotated = False - _compressed = history - # Rewrite the transcript only when rotation produced a NEW - # session id. In-place compaction needs none: - # archive_and_compact() already soft-archived the previous - # active rows and inserted the compacted set, and - # rewrite_transcript() would run - # replace_messages(active_only=False) and DELETE the archived - # turns. Likewise a summary with neither rotation nor a - # completed archive_and_compact() (unchanged session_id) signals - # FAILURE; an unconditional rewrite would replace the originals - # with only the summary (permanent data loss). - # Write-before-repoint (mirrors manual /compress): if - # session_entry were repointed to the child SID and - # rewrite_transcript then failed (lock/ENOSPC), the live entry - # would reference an empty session — the conversation silently - # vanishes. Persist the child transcript first, then rebind. - if _hyg_rotated: - if not await self.async_session_store.rewrite_transcript( - _hyg_new_sid, _compressed - ): - logger.error( - "Session hygiene: failed to persist " - "compressed transcript for rotated " - "session %s → %s; keeping the live " - "entry on the original session so the " - "conversation is not dropped", - session_entry.session_id, - _hyg_new_sid, - ) - # Fail closed: treat like no rotation. - _hyg_rotated = False - _hyg_in_place = False - else: - session_entry.session_id = _hyg_new_sid - # The held turn lease follows the rotation so an alias - # key resolving the fresh child still serializes against - # this turn. - self._rebind_turn_lease( - _quick_key, run_generation, _hyg_new_sid - ) - await self.async_session_store._save() - await asyncio.to_thread( - self._sync_telegram_topic_binding, - source, session_entry, - reason="hygiene-compression", - ) - - if _hyg_rotated: - # Reset stored token count — transcript rewritten - session_entry.last_prompt_tokens = 0 - history = _compressed - _new_count = len(_compressed) - _new_tokens = estimate_messages_tokens_rough( - _compressed - ) - elif _hyg_in_place: - # archive_and_compact() already persisted the - # compacted transcript inside _compress_context. - # Reset counts to match the new active set. - session_entry.last_prompt_tokens = 0 - history = _compressed - _new_count = len(_compressed) - _new_tokens = estimate_messages_tokens_rough( - _compressed - ) - else: - # No rewrite happened — the transcript is unchanged, so the - # post-compression counts equal the pre-compression ones. - _new_count = _msg_count - _new_tokens = _approx_tokens - logger.warning( - "Gateway hygiene compression for session %s " - "did not rotate or compact in place " - "(no session_db on the hygiene agent) — " - "preserving the original transcript instead " - "of overwriting it with the summary (#21301).", - session_entry.session_id, - ) - - logger.info( - "Session hygiene: compressed %s → %s msgs, " - "~%s → ~%s tokens", - _msg_count, _new_count, - f"{_approx_tokens:,}", f"{_new_tokens:,}", - ) - - if _new_tokens >= _warn_token_threshold: - logger.warning( - "Session hygiene: still ~%s tokens after " - "compression", - f"{_new_tokens:,}", - ) - - # Summary failure aborts the compressor entirely (messages - # unchanged, nothing dropped). Warn the gateway user visibly - # — agent.log is invisible on TG/Discord/etc. — so they know - # the chat is "frozen" at this size and can /compress to - # retry or /reset to start fresh. - _comp = getattr(_hyg_agent, "context_compressor", None) - _hyg_aborted = _comp is not None and getattr( - _comp, "_last_compress_aborted", False - ) - # Fence-cancelled _compress_context returns the original - # transcript with _last_compress_aborted still False - # (failure_class=commit_fence_cancelled, chunk_count=0). Treat - # that no-op as an abort so hygiene records a cooldown instead - # of retrying into the 600s wait. A successful rotate/in-place - # commit is not an abort even if a later invalidation flipped - # the fence. - _hyg_fence_cancelled = bool( - _hyg_commit_fence.is_cancelled - and not _hyg_rotated - and not _hyg_in_place - ) - if _hyg_fence_cancelled: - _hyg_aborted = True - if not _hyg_aborted: - # Recovery decision lives in the unit-tested predicate: the - # degenerate "neither rotated nor compacted in place" path - # sets both flags False and reuses the pre-compression - # counts, so a numbers-only check would read a no-op as - # success and clear the streak. - if hygiene_compaction_recovered( - aborted=_hyg_aborted, - rotated=_hyg_rotated, - in_place=_hyg_in_place, - msg_count=_msg_count, - new_count=_new_count, - approx_tokens=_approx_tokens, - new_tokens=_new_tokens, - ): - await asyncio.to_thread( - _reset_hygiene_failure_streak, - self, - session_key, - ) - if _hyg_aborted: - if _hyg_failure_cooldown_seconds >= 0: - _hyg_cooldown = await asyncio.to_thread( - _hygiene_cooldown_for_failure, - self, - session_key, - _hyg_failure_cooldown_seconds, - ) - _record_hygiene_cooldown( - self, session_entry.session_id, - _hyg_cooldown, - ( - "session hygiene compression " - "cancelled at commit fence" - if _hyg_fence_cancelled - else getattr( - _comp, "_last_summary_error", None - ) - ), - ) - from agent.session_activity import ( - ActivityProvenance, - ) - _stamp_hygiene_compression_provenance( - _hyg_agent, - "session hygiene compression aborted", - ActivityProvenance.AGENT_COMPRESSION_COOLDOWN, - "hygiene compression abort " - "activity stamp failed", - ) - if not _hyg_fence_cancelled: - _err = getattr(_comp, "_last_summary_error", None) or "unknown error" - # Force-redact: provider exception text may contain - # credentials and this message reaches gateway users. - from agent.redact import redact_sensitive_text - _err = redact_sensitive_text(_err, force=True) - _warn_msg = ( - "⚠️ Context compression aborted " - f"({_err}). No messages were dropped — " - "conversation is unchanged. Run /compress " - "to retry, /reset for a clean session, or " - "check your auxiliary.compression model " - "configuration." - ) - try: - _adapter = self._adapter_for_source(source) - if _adapter and source.chat_id: - await _adapter.send(source.chat_id, _warn_msg, metadata=_hyg_meta) - except Exception as _werr: - logger.warning( - "Failed to deliver compression-failure warning to user: %s", - _werr, - ) - # Separately: if the user's CONFIGURED aux model failed and we - # recovered by falling back to the main model, tell them — a - # misconfigured auxiliary.compression.model is something only - # they can fix, and silent recovery would hide it. - elif _comp is not None and getattr(_comp, "_last_aux_model_failure_model", None): - _aux_model = getattr(_comp, "_last_aux_model_failure_model", "") - _aux_err = getattr(_comp, "_last_aux_model_failure_error", None) or "unknown error" - _aux_msg = ( - f"ℹ️ Configured compression model `{_aux_model}` " - f"failed ({_aux_err}). Recovered using your main " - "model — context is intact — but you may want to " - "check `auxiliary.compression.model` in config.yaml." - ) - try: - _adapter = self._adapter_for_source(source) - if _adapter and source.chat_id: - await _adapter.send(source.chat_id, _aux_msg, metadata=_hyg_meta) - except Exception as _werr: - logger.warning( - "Failed to deliver aux-model-fallback notice to user: %s", - _werr, - ) - finally: - # Evict the cached agent so the next turn rebuilds its system - # prompt from current SOUL.md, memory, and skills. - self._evict_cached_agent(session_key) - if not _hyg_cleanup_deferred: - await self._cleanup_agent_resources_off_loop( - _hyg_agent, context="session hygiene" - ) - - except HygieneTurnHoldExceeded: - # Availability boundary, not a failure — already logged at INFO by the turn- - # hold handler. Must not hit the generic "auto-compress failed" warning - # below: that log made thinking-model deployments read as permanently broken. - pass - except Exception as e: - logger.warning( - "Session hygiene auto-compress failed: %s", e - ) - - # First-message onboarding -- only on the very first interaction ever. Delivered on the - # current user message (sidecar), NOT the ephemeral system prompt: present-on-turn-1/absent- - # on-turn-2 was a guaranteed system-prompt diff and agent rebuild. - if not history and not await self.async_session_store.has_any_sessions(): - # Default first-contact note: a brief self-introduction. - _intro_note = ( - "[System note: This is the user's very first message ever. " - "Briefly introduce yourself and mention that /help shows available commands. " - "Keep the introduction concise -- one or two sentences max.]" - ) - # Opt-in structured profile-build path: when enabled (default "ask") and not yet offered - # on this install, swap the plain intro for a consent-gated directive that offers to - # build a user profile and persists confirmed facts via memory(target="user"). Fires at - # most once (onboarding.seen flag); onboarding.profile_build: off in config.yaml - # disables it. - try: - from agent.onboarding import ( - PROFILE_BUILD_FLAG, - is_seen, - mark_seen, - profile_build_directive, - profile_build_mode, - ) - _onb_cfg = _load_gateway_config() - if ( - profile_build_mode(_onb_cfg) == "ask" - and not is_seen(_onb_cfg, PROFILE_BUILD_FLAG) - ): - turn_sidecar_notes.append(profile_build_directive().strip()) - mark_seen(_hermes_home / "config.yaml", PROFILE_BUILD_FLAG) - else: - turn_sidecar_notes.append(_intro_note) - except Exception as _pb_err: - logger.debug( - "Profile-build onboarding directive failed, using plain intro: %s", - _pb_err, - ) - turn_sidecar_notes.append(_intro_note) - - # One-time prompt if no home channel is set for this platform - # Skip for webhooks - they deliver directly to configured targets (github_comment, etc.) - if not history and source.platform and source.platform != Platform.LOCAL and source.platform != Platform.WEBHOOK: - platform_name = source.platform.value - env_key = _home_target_env_var(platform_name) - # Multiplex: home channel may live only in the profile secret - # scope / PlatformConfig, not process os.environ. - home_env = "" - try: - from agent.secret_scope import get_secret - - home_env = (get_secret(env_key) or "").strip() if env_key else "" - except Exception: - home_env = "" - if not home_env: - home_env = (os.getenv(env_key) or "").strip() if env_key else "" - # Also honor in-memory / yaml home_channel on this platform. - try: - if not home_env and self.config.get_home_channel(source.platform): - home_env = "set" - except Exception: - pass - # Secondary-profile platforms (e.g. Slack on yolo) may only exist - # under that profile's loaded config — check after scope install. - if not home_env: - try: - from gateway.config import load_gateway_config as _lgc - prof = (getattr(source, "profile", None) or "").strip() - if prof and prof != "default": - # Already inside profile scope for secondary handlers; - # re-read live config for home_channel. - _pcfg = _lgc() - if _pcfg.get_home_channel(source.platform): - home_env = "set" - except Exception: - pass - if not home_env: - # Slack dispatches all Hermes commands through a single - # parent slash command `/hermes`; bare `/sethome` is not - # registered and would fail with "app did not respond". - sethome_cmd = ( - "/hermes sethome" - if source.platform == Platform.SLACK - else "/sethome" - ) - notice = ( - f"📬 No home channel is set for {platform_name.title()}. " - f"A home channel is where Hermes delivers cron job results " - f"and cross-platform messages.\n\n" - f"Type {sethome_cmd} to make this chat your home channel, " - f"or ignore to skip." - ) - await self._deliver_platform_notice(source, notice) - - # Voice channel awareness: deliver voice channel state (who is present / speaking) on the - # user message, ONLY when changed since the previous turn. It differs almost every turn, and - # in the ephemeral system prompt it forced a full agent rebuild + prompt-cache re-key per - # message; the system prompt carries a static pointer line instead (gateway/session.py). - _vc_note = self._voice_channel_sidecar_note(event, source, session_key) - if _vc_note: - turn_sidecar_notes.append(_vc_note) - - # Auto-analyze user images: run the vision tool eagerly so the model always gets a text - # description plus the local path for re-examination via vision_analyze. Filter to image - # media_type so documents/audio in the same message are not sent to the vision tool. - message_text = await self._prepare_profile_scoped_inbound_message_text( - event=event, - source=source, - history=history, - session_key=session_key, - ) - if message_text is None: - return - - # Capture the platform event time as message metadata and keep the persisted transcript - # clean (strip any leading timestamp prefix). This runs regardless of the toggle so storage - # stays clean and the send-time is preserved. Only the in-context RENDER (the prefix the - # model sees) is gated behind gateway.message_timestamps.enabled — default OFF. - try: - from hermes_time import get_timezone as _get_evt_tz - from gateway.message_timestamps import ( - coerce_message_timestamp as _coerce_msg_ts, - render_user_content_with_timestamp as _render_msg_ts, - strip_leading_message_timestamps as _strip_msg_ts, - ) - _evt_tz = _get_evt_tz() - _evt_ts = getattr(event, "timestamp", None) - if message_text and isinstance(message_text, str): - _clean_message_text, _embedded_ts = _strip_msg_ts( - message_text, tz=_evt_tz) - persist_user_message = _clean_message_text - _event_epoch = _coerce_msg_ts(_evt_ts, tz=_evt_tz) - persist_user_timestamp = ( - _event_epoch if _event_epoch is not None else _embedded_ts - ) - if _message_timestamps_enabled(_load_gateway_config()): - message_text = _render_msg_ts( - _clean_message_text, - persist_user_timestamp, - tz=_evt_tz, - ) - else: - # Toggle off: model sees the clean message; the timestamp - # is still stored as metadata for later opt-in. - message_text = _clean_message_text - except Exception as _ts_err: - logger.debug("Message timestamp injection failed (non-fatal): %s", _ts_err) - - # Stage this turn's must-deliver notes (one-shot; consumed in run_sync) AFTER the - # message_text early-out so an aborted turn cannot leak its notes into the next turn. - if turn_sidecar_notes and session_key: - self._set_pending_turn_sidecar_notes(session_key, turn_sidecar_notes) - - # Bind this run generation to the adapter's active-session event so deferred post-delivery - # callbacks can be released by the same run that registered them. - self._bind_adapter_run_generation( - self._adapter_for_source(source), - session_key, - run_generation, - ) - - try: - # Emit agent:start hook - hook_ctx = { - "platform": source.platform.value if source.platform else "", - "user_id": source.user_id, - "chat_id": source.chat_id or "", - "thread_id": str(getattr(source, "thread_id", None)) if getattr(source, "thread_id", None) else "", - "chat_type": getattr(source, "chat_type", "") or "", - "session_id": session_entry.session_id, - "message": message_text[:500], - } - await self.hooks.emit("agent:start", hook_ctx) - - # Run the agent. Capture the session id that this run was launched against so post-run - # compression publication can be identity-guarded below; a /new or another lifecycle - # transition may move session_entry.session_id while the old run is still unwinding. - _run_start_session_id = session_entry.session_id - _turn_started_monotonic = time.monotonic() - agent_result = await self._run_agent( - message=message_text, - context_prompt=context_prompt, - history=history, - source=source, - session_id=_run_start_session_id, - session_key=session_key, - run_generation=run_generation, - event_message_id=self._reply_anchor_for_event(event), - inbound_message_id=( - str(event.message_id) if event.message_id else None - ), - channel_prompt=event.channel_prompt, - moa_config=getattr(event, "_moa_config", None), - persist_user_message=persist_user_message, - persist_user_timestamp=persist_user_timestamp, - persist_user_display_kind=persist_user_display_kind, - message_type=event.message_type, - ) - _turn_seconds = time.monotonic() - _turn_started_monotonic - - # Stop the typing indicator. Slack AI status is scoped to a thread/workspace, so - # preserve the routing metadata used by the response delivery path. - try: - _typing_adapter = self._adapter_for_source(source) - _stop_with_metadata = getattr( - type(_typing_adapter), "_stop_typing_with_metadata", None - ) - _stop_typing = getattr(type(_typing_adapter), "stop_typing", None) - if _typing_adapter and callable(_stop_with_metadata): - await _typing_adapter._stop_typing_with_metadata( - source.chat_id, - self._thread_metadata_for_source( - source, self._reply_anchor_for_event(event) - ), - ) - elif _typing_adapter and callable(_stop_typing): - await _typing_adapter.stop_typing(source.chat_id) - except Exception: - pass - - if not self._is_session_run_current(_quick_key, run_generation): - logger.info( - "Discarding stale agent result for %s — generation %d is no longer current", - _quick_key or "?", - run_generation, - ) - _stale_adapter = self._adapter_for_source(source) - if getattr(type(_stale_adapter), "pop_post_delivery_callback", None) is not None: - _stale_adapter.pop_post_delivery_callback( - _quick_key, - generation=run_generation, - ) - elif _stale_adapter and hasattr(_stale_adapter, "_post_delivery_callbacks"): - _stale_adapter._post_delivery_callbacks.pop(_quick_key, None) - return None - - response = agent_result.get("final_response") or "" - # Hidden-reasoning-only retry exhaustion: the loop's sentinel text ("Codex response - # remained incomplete after 3 continuation attempts") doubles as final_response, so it - # would be delivered verbatim into the channel — where peer agents can ingest it as a - # completed assistant turn. - if _is_gateway_hidden_reasoning_incomplete_turn(agent_result): - response = "" - try: - from gateway.response_filters import is_intentional_silence_agent_result - _intentional_silence = is_intentional_silence_agent_result( - agent_result, response, - ) - except Exception: - _intentional_silence = False - - # Convert the agent's internal "(empty)" sentinel into a user-friendly message. - # "(empty)" means the model failed to produce visible content after exhausting all - # retries (nudge, prefill, empty-retry, fallback). - if response == "(empty)" and not _intentional_silence: - response = ( - "⚠️ The model returned no response after processing tool " - "results. This can happen with some models — try again or " - "rephrase your question." - ) - agent_messages = agent_result.get("messages", []) - _response_time = time.time() - _msg_start_time - _api_calls = agent_result.get("api_calls", 0) - _resp_len = len(response) - logger.info( - "response ready: platform=%s chat=%s time=%.1fs api_calls=%d response=%d chars", - _platform_name, source.chat_id or "unknown", - _response_time, _api_calls, _resp_len, - ) - - # The cross-process cache-coherence re-baseline (_refresh_agent_cache_message_count) is - # deferred until AFTER the transcript persistence block below: it must include the - # first-turn `session_meta` marker row and the compression session_id swap. - - # Successful turn: clear the stuck-loop counter (it only accumulates across CONSECUTIVE - # restarts where the session never completed) and resume_pending (set by drain-timeout - # shutdown) so later messages don't get the restart-interruption system note. - if session_key and _should_clear_resume_pending_after_turn(agent_result): - await self._clear_restart_failure_count(session_key) - try: - await self.async_session_store.clear_resume_pending(session_key) - except Exception as _e: - logger.debug( - "clear_resume_pending failed for %s: %s", - session_key, _e, - ) - - # Normalize empty responses: surface errors, partial failures, and - # the case where agent did work but returned no text. Fix for #18765. - if not _intentional_silence: - response = _normalize_empty_agent_response( - agent_result, response, history_len=len(history), - ) - response = _sanitize_gateway_final_response(source.platform, response) - - # Ordering contract: the agent thread already updated the contextvar in - # conversation_compression.py; propagate to SessionEntry + _save(). - if agent_result.get("session_id") and agent_result["session_id"] != session_entry.session_id: - if session_entry.session_id == _run_start_session_id: - session_entry.session_id = agent_result["session_id"] - # The held turn lease follows the rotation: the transcript persistence below - # writes to the NEW id, so the serialization boundary must move with it or an - # alias key resolving the fresh child could interleave. - self._rebind_turn_lease( - _quick_key, run_generation, session_entry.session_id - ) - await self.async_session_store._save() - await self.async_session_store._record_gateway_session_peer( - session_entry.session_id, - session_key, - source, - ) - await asyncio.to_thread( - self._sync_telegram_topic_binding, - source, session_entry, reason="agent-result-compression", - ) - else: - logger.info( - "Skipping agent-result session split sync for %s because " - "the session binding moved from %s to %s before " - "compression finished", - session_key or "?", - _run_start_session_id, - session_entry.session_id, - ) - - # Prepend reasoning if display is enabled (per-platform). Mattermost requires explicit - # opt-in because this is scratch text, not ordinary final-answer content. - try: - _show_reasoning_effective = _resolve_gateway_display_bool( - _load_gateway_config(), - _platform_config_key(source.platform), - "show_reasoning", - default=bool(getattr(self, "_show_reasoning", False)), - platform=source.platform, - require_platform_override_for={Platform.MATTERMOST}, - ) - except Exception: - _show_reasoning_effective = ( - False - if source.platform == Platform.MATTERMOST - else getattr(self, "_show_reasoning", False) - ) - if _show_reasoning_effective and response and not _intentional_silence: - last_reasoning = agent_result.get("last_reasoning") - if last_reasoning: - from gateway.stream_consumer import escape_code_fences_for_display - # Collapse long reasoning to keep messages readable - lines = last_reasoning.strip().splitlines() - if len(lines) > 15: - display_reasoning = "\n".join(lines[:15]) - display_reasoning += f"\n_... ({len(lines) - 15} more lines)_" - else: - display_reasoning = last_reasoning.strip() - # Render style is per-platform: Discord defaults to "-# " subtext (native small - # grey metadata text); other platforms keep the fenced code block. - try: - from gateway.display_config import resolve_display_setting - _reasoning_style = resolve_display_setting( - _load_gateway_config(), - _platform_config_key(source.platform), - "reasoning_style", - "code", - ) - except Exception: - _reasoning_style = "code" - if _reasoning_style == "subtext": - _quoted = "\n".join( - f"-# {ln}" if ln else "-#" for ln in display_reasoning.splitlines() - ) - response = f"-# 💭 Reasoning\n{_quoted}\n\n{response}" - elif _reasoning_style == "blockquote": - _quoted = "\n".join( - f"> {ln}" if ln else ">" for ln in display_reasoning.splitlines() - ) - response = f"> 💭 **Reasoning:**\n{_quoted}\n\n{response}" - else: - # Escape ``` inside reasoning so inner fences don't - # break the outer code block used to render it. - display_reasoning = escape_code_fences_for_display(display_reasoning) - response = f"💭 **Reasoning:**\n```\n{display_reasoning}\n```\n\n{response}" - - # Runtime-metadata footer — only on the FINAL message of the turn. Off by default - # (display.runtime_footer.enabled=false). When streaming already delivered the body, we - # can't mutate the sent text, so we fire a separate trailing send below. - _footer_line = "" - try: - from gateway.runtime_footer import build_footer_line as _bfl - _footer_line = _bfl( - user_config=_load_gateway_config(), - platform_key=_platform_config_key(source.platform), - model=agent_result.get("model"), - context_tokens=agent_result.get("last_prompt_tokens", 0) or 0, - context_length=agent_result.get("context_length") or None, - cwd=_terminal_scope_cwd(""), - turn_seconds=_turn_seconds, - ) - except Exception as _footer_err: - logger.debug("runtime_footer build failed: %s", _footer_err) - _footer_line = "" - if _footer_line and response and not agent_result.get("already_sent") and not _intentional_silence: - response = f"{response}\n\n{_footer_line}" - - # Emit agent:end hook - await self.hooks.emit("agent:end", { - **hook_ctx, - "response": (response or "")[:500], - "model": agent_result.get("model", ""), - "provider": agent_result.get("provider", ""), - }) - - # Check for pending process watchers (check_interval on background processes) - try: - from tools.process_registry import process_registry - # Detach the current batch atomically (see crash-recovery drain - # above): reassign to a fresh list so a watcher appended by a - # concurrent session during the yield isn't dropped by clear(). - watchers = process_registry.pending_watchers - process_registry.pending_watchers = [] - for i, watcher in enumerate(watchers): - asyncio.create_task(self._run_process_watcher(watcher)) - if i % 100 == 99: - await asyncio.sleep(0) - except Exception as e: - logger.error("Process watcher setup error: %s", e) - - # Drain watch notifications that arrived during the run. The queue also carries process - # completions (handled by the per-process watcher task above) and async-delegation - # completions (owned by _async_delegation_watcher, the single consumer for idle and - # post-turn cases) — inject only watch-type events and leave the rest on the queue. - try: - from tools.process_registry import process_registry as _pr - await self._drain_watch_notifications(_pr.completion_queue) - except Exception as e: - logger.debug("Watch queue drain error: %s", e) - - # NOTE: Dangerous command approvals are now handled inline by the blocking gateway - # approval mechanism in tools/approval.py. - - # Persist the full agent loop (tool calls, results, reasoning) so sessions resume with - # full context. IMPORTANT: on context-overflow failures (compression exhausted, generic - # 400 on large sessions) do NOT persist the user message — it would grow the session and - # reproduce the failure forever. Transient failures (429, timeout, connection error, - # 5xx) are different: the session is not oversized and dropping the user turn causes - # severe context loss on retry, so persist it. - agent_failed_early = bool(agent_result.get("failed")) - hidden_reasoning_incomplete = _is_gateway_hidden_reasoning_incomplete_turn( - agent_result - ) - _err_str_for_classify = str(agent_result.get("error", "")).lower() - # Use specific multi-word phrases (not bare "exceed"/"token") to avoid false positives - # on transient errors such as "rate limit exceeded"; matches run_agent.py's classifier. - is_context_overflow_failure = agent_failed_early and ( - bool(agent_result.get("compression_exhausted")) - or any(p in _err_str_for_classify for p in ( - "context length", "context size", "context window", - "maximum context", "token limit", "too many tokens", - "reduce the length", "exceeds the limit", - "request entity too large", "prompt is too long", - "payload too large", "input is too long", - )) - or ("400" in _err_str_for_classify and len(history) > 50) - ) - if is_context_overflow_failure: - logger.info( - "Skipping transcript persistence for context-overflow " - "failure in session %s to prevent session growth loop.", - session_entry.session_id, - ) - elif agent_failed_early: - logger.info( - "Transient agent failure in session %s — persisting user " - "message so conversation context is preserved on retry.", - session_entry.session_id, - ) - elif hidden_reasoning_incomplete: - logger.warning( - "Suppressing hidden-reasoning-only incomplete gateway turn " - "for session %s: %s", - session_entry.session_id, - agent_result.get("error", "processing incomplete"), - ) - - # Compression exhausted = permanently too large: auto-reset so the next message starts - # fresh instead of replaying the oversized context forever. A lock-contended defer is - # the OPPOSITE case (a concurrent path holds the lock and is shrinking it): never wipe - # for that. - if agent_result.get("compression_deferred"): - logger.info( - "Compression deferred for session %s — the compression " - "lock is held by a concurrent compressor. Keeping the " - "session intact; the next message retries normally.", - session_entry.session_id if session_entry else "?", - ) - elif agent_result.get("compression_exhausted") and session_entry and session_key: - logger.info( - "Auto-resetting session %s after compression exhaustion.", - session_entry.session_id, - ) - new_entry = await self.async_session_store.reset_session(session_key) - self._evict_cached_agent(session_key) - # Conversation boundary: one funnel call clears every conversation-scoped - # per-session dict (see _CONVERSATION_SCOPED_STATE). - self._clear_conversation_scope( - session_key, reason="compression_exhausted_reset" - ) - if new_entry is not None: - # Re-point the Telegram topic binding at the fresh session: compression rotated - # session_entry.session_id to the bloated child earlier this turn and that _sync - # also rewrote the (chat_id, thread_id) binding. Without a re-sync the - # binding-heal walk switches the next inbound message back onto the child and - # re-triggers exhaustion forever. No-op on non-topic lanes. - session_entry = new_entry - await asyncio.to_thread( - self._sync_telegram_topic_binding, - source, session_entry, reason="compression-exhausted-reset", - ) - response = (response or "") + ( - "\n\n🔄 Session auto-reset — the conversation exceeded the " - "maximum context size and could not be compressed further. " - "Your next message will start a fresh session." - ) - - ts = time.time() # Unix epoch float — consistent with DB storage - - # Fresh session (no history): write the full tool definitions as the first entry so the - # transcript is self-describing — the same dicts sent as tools=[...] in the API request. - if is_context_overflow_failure: - pass # Skip all transcript writes — don't grow a broken session - elif not history: - tool_defs = agent_result.get("tools", []) - await self.async_session_store.append_to_transcript( - session_entry.session_id, - { - "role": "session_meta", - "tools": tool_defs or [], - "model": _resolve_gateway_model(), - "platform": source.platform.value if source.platform else "", - "timestamp": ts, - } - ) - - # The agent already persisted these via _flush_messages_to_session_db(); skip the DB - # write to avoid duplicates. Holds for the codex app-server runtime too (it flushes its - # own projected messages before returning and reports agent_persisted=True). Reading the - # flag (default = self._session_db is not None) keeps the contract explicit; a - # non-persisting runtime opts in via False. - agent_persisted = agent_result.get("agent_persisted", self._session_db is not None) - - # Only the NEW messages from this turn: use history_offset (what the agent saw), not - # len(history), which counts session_meta entries stripped before the agent saw them. - if is_context_overflow_failure: - pass # handled above — skip all transcript writes - elif agent_failed_early or hidden_reasoning_incomplete: - # Transient failure (429/timeout/5xx): persist only the user message so the next - # message can load a transcript that reflects what was said. Skip the assistant - # error text since it's a gateway-generated hint, not model output. Hidden-reasoning - # incomplete turns follow the same rule so peer-agent channels don't ingest them. - _user_entry = { - "role": "user", - "content": ( - persist_user_message - if persist_user_message is not None - else message_text - ), - "timestamp": ( - persist_user_timestamp - if persist_user_timestamp is not None - else ts - ), - } - if persist_user_display_kind: - _user_entry["display_kind"] = persist_user_display_kind - if event.message_id: - _user_entry["message_id"] = str(event.message_id) - # Dedupe: skip if this platform message_id is already in the transcript (prevents - # duplicate user turns on Telegram retries after transient failures). - _skip_persist = ( - event.message_id - and await self.async_session_store.has_platform_message_id( - session_entry.session_id, str(event.message_id) - ) - ) - if _skip_persist: - logger.info( - "Skipping duplicate user turn " - "(message_id=%s) in session %s", - event.message_id, session_entry.session_id, - ) - else: - await self.async_session_store.append_to_transcript( - session_entry.session_id, - _user_entry, - skip_db=agent_persisted, - ) - else: - history_len = agent_result.get("history_offset", len(history)) - new_messages = agent_messages[history_len:] if len(agent_messages) > history_len else [] - - # If no new messages found (edge case), fall back to simple user/assistant - if not new_messages: - _user_entry = { - "role": "user", - "content": ( - persist_user_message - if persist_user_message is not None - else message_text - ), - "timestamp": ( - persist_user_timestamp - if persist_user_timestamp is not None - else ts - ), - } - if persist_user_display_kind: - _user_entry["display_kind"] = persist_user_display_kind - if event.message_id: - _user_entry["message_id"] = str(event.message_id) - await self.async_session_store.append_to_transcript( - session_entry.session_id, - _user_entry, - skip_db=agent_persisted, - ) - if response: - await self.async_session_store.append_to_transcript( - session_entry.session_id, - {"role": "assistant", "content": response, "timestamp": ts}, - skip_db=agent_persisted, - ) - else: - # Attach the inbound platform message_id to the first user entry written this - # turn so platform-level quote-resolution (e.g. Yuanbao QuoteContextMiddleware's - # transcript fallback) can find earlier @bot messages by their original id. - _user_msg_id_attached = False - for msg in new_messages: - # Skip system messages (they're rebuilt each run) - if msg.get("role") == "system": - continue - # Add timestamp to each message for debugging - entry = {**msg, "timestamp": ts} - if ( - not _user_msg_id_attached - and msg.get("role") == "user" - and event.message_id - and "message_id" not in entry - ): - entry["message_id"] = str(event.message_id) - _user_msg_id_attached = True - await self.async_session_store.append_to_transcript( - session_entry.session_id, entry, - skip_db=agent_persisted, - ) - - # The agent persists token counts and model itself; keep only last_prompt_tokens here - # for context-window tracking and compression decisions. - await self.async_session_store.update_session( - session_entry.session_key, - last_prompt_tokens=agent_result.get("last_prompt_tokens", 0), - touch_activity=not bool(getattr(event, "internal", False)), - ) - - # Re-baseline the cached agent's message_count snapshot now that ALL of this turn's - # transcript writes are done (flushed rows AND the first-turn `session_meta` marker). - # The cross-process coherence guard snapshots at agent-BUILD time and never refreshes - # on reuse, so our own writes would trigger a rebuild next turn (destroying prompt - # caching). MUST run after the session_meta append (that row bumps the count too). - await self._refresh_agent_cache_message_count( - session_key, session_entry.session_id - ) - - # Intentional silence is a delivery decision, not a transcript mutation: the [SILENT] - # assistant turn stays persisted so later turns keep user/assistant alternation; only - # the outbound delivery is suppressed. - if _intentional_silence: - logger.info( - "Suppressing intentional silence marker for session %s", - session_entry.session_id, - ) - response = "" - - # Auto voice reply: send TTS audio before the text response - _already_sent = bool(agent_result.get("already_sent")) - # Skip when streaming TTS already delivered audio for this turn (#60671). - _stts_adapter = self._adapter_for_source(source) - _streaming_tts_done = ( - _stts_adapter is not None - and bool(getattr(_stts_adapter, "_streaming_tts_turn_completed", lambda *_a, **_k: False)(session_key, run_generation)) - ) - if ( - not _streaming_tts_done - and self._should_send_voice_reply(event, response, agent_messages, already_sent=_already_sent) - ): - await self._send_voice_reply(event, response) - - # Streamed responses still need MEDIA: files delivered before returning None (chunks - # carry the tags verbatim and post-processing is skipped when already_sent). Never skip - # when the agent failed: the error text is new content streaming didn't show. - if agent_result.get("already_sent") and not agent_result.get("failed"): - if response: - _media_adapter = self._adapter_for_source(source) - if _media_adapter: - await self._deliver_media_from_response( - response, event, _media_adapter, - ) - # Streaming already delivered the body text, but the footer was intentionally held - # back (see the `not already_sent` gate above). - if _footer_line: - try: - _foot_adapter = self._adapter_for_source(source) - if _foot_adapter: - await _foot_adapter.send( - source.chat_id, - _footer_line, - metadata=self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)), - ) - except Exception as _e: - logger.debug("trailing footer send failed: %s", _e) - # This branch returns None so the adapter does not send the body twice. /loop and - # /goal hooks in _handle_message read the return value, so stash the delivered text - # on the event or those hooks never run and a /loop tick stays awaiting. - with suppress(Exception): - event._streamed_final_response = str(response or "") - return None - - return response - - except Exception as e: - # Stop typing indicator on error too, retaining Slack thread/workspace - # routing so a failed turn cannot leave its status visible. - try: - _err_adapter = self._adapter_for_source(source) - _stop_with_metadata = getattr( - type(_err_adapter), "_stop_typing_with_metadata", None - ) - _stop_typing = getattr(type(_err_adapter), "stop_typing", None) - if _err_adapter and callable(_stop_with_metadata): - await _err_adapter._stop_typing_with_metadata( - source.chat_id, - self._thread_metadata_for_source( - source, self._reply_anchor_for_event(event) - ), - ) - elif _err_adapter and callable(_stop_typing): - await _err_adapter.stop_typing(source.chat_id) - except Exception: - pass - logger.exception("Agent error in session %s", session_key) - # Crash-resilience for failures before AIAgent enters run_conversation() (e.g. provider/ - # httpx client init): the agent can't persist the inbound turn there, so append the user - # message here once; if the agent already reached turn-start persistence the latest user - # row matches and we skip the duplicate. - try: - if 'message_text' in locals() and message_text is not None and session_entry is not None: - _already_persisted = False - try: - _recent_transcript = await self.async_session_store.load_transcript(session_entry.session_id) - except Exception: - _recent_transcript = [] - for _msg in reversed(_recent_transcript[-10:]): - if _msg.get("role") == "user": - _expected_user_content = ( - persist_user_message - if persist_user_message is not None - else message_text - ) - _already_persisted = (_msg.get("content") == _expected_user_content) - break - if not _already_persisted: - _user_entry = { - "role": "user", - "content": ( - persist_user_message - if persist_user_message is not None - else message_text - ), - "timestamp": ( - persist_user_timestamp - if persist_user_timestamp is not None - else time.time() - ), - } - if 'persist_user_display_kind' in locals() and persist_user_display_kind: - _user_entry["display_kind"] = persist_user_display_kind - if getattr(event, "message_id", None): - _user_entry["message_id"] = str(event.message_id) - await self.async_session_store.append_to_transcript( - session_entry.session_id, - _user_entry, - ) - except Exception: - logger.debug("Failed to persist inbound user message after agent exception", exc_info=True) - # Log full details server-side only; never expose raw exception - # types or messages to end users (info-leakage risk). - status_hint = "" - status_code = getattr(e, "status_code", None) - _hist_len = len(history) if 'history' in locals() else 0 - if status_code == 401: - status_hint = " Check your API key or run `claude /login` to refresh OAuth credentials." - elif status_code == 402: - status_hint = " Your API balance or quota is exhausted. Check your provider dashboard." - elif status_code == 429: - # Check if this is a plan usage limit (resets on a schedule) vs a transient rate limit - _err_body = getattr(e, "response", None) - _err_json = {} - try: - if _err_body is not None: - _err_json = _err_body.json().get("error", {}) - if not isinstance(_err_json, dict): - _err_json = {} - except Exception: - pass - if _err_json.get("type") == "usage_limit_reached": - _resets_in = _err_json.get("resets_in_seconds") - if _resets_in and _resets_in > 0: - import math - _hours = math.ceil(_resets_in / 3600) - status_hint = f" Your plan's usage limit has been reached. It resets in ~{_hours}h." - else: - status_hint = " Your plan's usage limit has been reached. Please wait until it resets." - else: - status_hint = " You are being rate-limited. Please wait a moment and try again." - elif status_code == 529: - status_hint = " The API is temporarily overloaded. Please try again shortly." - elif status_code in {400, 500}: - # 400 on a large session is context overflow; 500 on a large session often means the - # payload is too large for the API — treat it the same way. - if _hist_len > 50: - return ( - "⚠️ Session too large for the model's context window.\n" - "Use /compact to compress the conversation, or " - "/reset to start fresh." - ) - elif status_code == 400: - status_hint = " The request was rejected by the API." - return ( - f"Sorry, I encountered an unexpected error.{status_hint}\n" - "Try again or use /reset to start a fresh session." - ) - finally: - # Restore session context variables to their pre-handler state - self._clear_session_env(_session_env_tokens) - - def _reset_notice_session_info(self, source: SessionSource) -> str: - """Session-info block for the auto-reset notice, profile-scoped. - - Under multiplexing, resolve model/provider/context inside the profile serving ``source`` - (mirrors ``_run_agent``'s gating) or the banner advertises the base config's model. Call - via ``asyncio.to_thread``: resolution can block (credential refresh, context-length - probes), and the scope is entered here so contextvars behave in the worker thread. + @dataclasses.dataclass + class _HygieneSettings: + """Resolved session-hygiene configuration for one inbound turn.""" + + model: str + threshold_pct: float + compression_enabled: bool + hard_msg_limit: int + timeout_seconds: float + total_ceiling_seconds: float + max_turn_hold_seconds: float + failure_cooldown_seconds: float + config_context_length: Optional[int] + provider: Optional[str] + base_url: Optional[str] + api_key: Optional[str] + data: Any + + @dataclasses.dataclass + class _HygieneAttempt: + """One detached hygiene compression attempt (agent, worker future, commit fence). + + ``cleanup_deferred`` is shared mutable state: the wait handlers set it on their raise + paths and the owning ``finally`` reads it to decide whether to clean the agent up now. """ - if getattr(getattr(self, "config", None), "multiplex_profiles", False): - with _profile_runtime_scope(self._resolve_profile_home_for_source(source)): - return self._format_session_info() - return self._format_session_info() - def _format_session_info(self) -> str: - """Resolve current model config and return a formatted info block. + agent: Any + meta: Any + commit_fence: Any = None + future: Any = None + wait_started: float = 0.0 + cleanup_deferred: bool = False + history: Any = None - Surfaces model, provider, context length, and endpoint so gateway users can immediately - see if context detection went wrong (e.g. local models falling to the 128K default). - """ - resolved = _resolve_gateway_model_context() - model = resolved.model - provider = resolved.provider - base_url = resolved.base_url - context_length = resolved.context_length - - # Format context source hint - if resolved.context_source == "config": - ctx_source = "config" - elif resolved.context_source == "default": - ctx_source = "default — set model.context_length in config to override" - else: - ctx_source = "detected" - - # Format context length for display - if context_length >= 1_000_000: - ctx_display = f"{context_length / 1_000_000:.1f}M" - elif context_length >= 1_000: - ctx_display = f"{context_length // 1_000}K" - else: - ctx_display = str(context_length) - - lines = [ - f"◆ Model: `{model}`", - f"◆ Provider: {provider or 'openrouter'}", - f"◆ Context: {ctx_display} tokens ({ctx_source})", - ] - - # Show endpoint for local/custom setups - if base_url and base_url_hostname(base_url) in ("localhost", "127.0.0.1", "0.0.0.0"): - lines.append(f"◆ Endpoint: {base_url}") - - return "\n".join(lines) - - def _check_slash_access( - self, source: SessionSource, canonical_cmd: str - ) -> Optional[str]: - """Return a denial message if ``source`` cannot run ``canonical_cmd``, else None. - - Used by the cold and running-agent dispatch paths in ``_handle_message`` so admin/user gating - can't be bypassed by an in-flight agent. Backward-compat: without ``allow_admin_from`` for the - scope, ``policy_for_source`` returns ``enabled=False`` and this always returns None. - """ - from gateway.slash_access import policy_for_source as _policy_for_source - - if not canonical_cmd: - return None - policy = _policy_for_source(self.config, source) - if not policy.enabled or policy.can_run(source.user_id, canonical_cmd): - return None - logger.info( - "Slash command /%s denied for %s:%s (not admin, not in user_allowed_commands)", - canonical_cmd, - source.platform.value if source.platform else "?", - source.user_id, - ) - allowed_preview = sorted(policy.user_allowed_commands) - if allowed_preview: - suffix = ( - "You can run: " - + ", ".join(f"/{c}" for c in allowed_preview[:12]) - + ("…" if len(allowed_preview) > 12 else "") - + ". Use /whoami for the full list." - ) - else: - suffix = ( - "No slash commands are enabled for non-admins on this " - "platform. Ask an admin to add you to allow_admin_from " - "or to set user_allowed_commands." - ) - return f"⛔ /{canonical_cmd} is admin-only here. {suffix}" - - def _sibling_thread_run_keys(self, source: SessionSource, own_key: str) -> list: - """Find running-agent keys for OTHER participants in the same thread. - - In per-user thread mode each participant gets an isolated key - (``...:{thread_id}:{user_id}``), so another user's run is invisible to the caller's own - ``/stop``. Returns keys of *actually running* agents (not the pending sentinel, not the - caller's own) sharing the caller's ``{chat_id}:{thread_id}`` prefix; empty when not in a - thread or no sibling runs exist. Callers must still gate on authorization. - """ - thread_id = getattr(source, "thread_id", None) - chat_id = getattr(source, "chat_id", None) - if not thread_id or not chat_id: - return [] - platform = source.platform.value - chat_type = getattr(source, "chat_type", None) or "" - # Prefix that every per-user key in this thread shares, up to and including the thread_id - # segment. Match the exact key or prefix + ":" (a further user_id segment) so an unrelated - # thread whose id merely starts with this one is not matched. - prefix = ":".join( - ["agent:main", platform, chat_type, str(chat_id), str(thread_id)] - ) - matches = [] - for key, agent in self._running_agent_items(): - if key == own_key: - continue - if agent is _AGENT_PENDING_SENTINEL or not agent: - continue - if key == prefix or key.startswith(prefix + ":"): - matches.append(key) - return matches - - def _is_stale_restart_redelivery(self, event: MessageEvent) -> bool: - """Return True if this /restart is a Telegram re-delivery we already handled. - - The previous gateway wrote ``.restart_last_processed.json`` with the triggering platform - + update_id when it processed the /restart. A /restart on the same platform with - update_id <= that value is a redelivery when this process booted from that restart; - otherwise the marker must still be recent (< 5 minutes). Telegram only (the only platform - with a numeric cross-session update ordering); other platforms return False. - """ - if event is None or event.source is None: - return False - if event.platform_update_id is None: - return False - if event.source.platform is None: - return False - # Only Telegram populates platform_update_id currently; be explicit - # so future platforms aren't accidentally gated by this check. - try: - platform_value = event.source.platform.value - except Exception: - return False - if platform_value != "telegram": - return False - - try: - marker_path = _hermes_home / ".restart_last_processed.json" - if not marker_path.exists(): - # Belt-and-suspenders for a missing dedup marker (cleaned up, or the previous write - # failed): without it the update_id comparison can't run and a redelivered /restart - # would re-restart the gateway forever. Suppress ONLY when a restart cycle is - # independently confirmed: this process booted from a chat-originated /restart - # (_booted_from_restart) AND is within a short post-boot window; a genuine first - # /restart on a fresh boot is never swallowed (flag stays False). Consume the flag - # one-shot so a later legitimate /restart in the same session is honored. - if ( - getattr(self, "_booted_from_restart", False) - and time.time() - getattr(self, "_startup_time", 0.0) < 60 - ): - self._booted_from_restart = False - return True - return False - data = json.loads(marker_path.read_text(encoding="utf-8")) - except Exception: - return False - - if data.get("platform") != platform_value: - return False - recorded_uid = data.get("update_id") - if not isinstance(recorded_uid, int): - return False - if event.platform_update_id > recorded_uid: - return False - - # A service-managed restart can legitimately take longer than the marker's normal five- - # minute trust window while adapters, cron, and in-flight deliveries drain. Consume the boot - # signal one-shot so a later genuine command is evaluated normally. - if getattr(self, "_booted_from_restart", False): - self._booted_from_restart = False - return True - - # Staleness guard: ignore markers older than 5 minutes so a legitimately old one (e.g. crash - # recovery where notify never fired) doesn't swallow a fresh /restart. - requested_at = data.get("requested_at") - if isinstance(requested_at, (int, float)): - if time.time() - requested_at > 300: - return False - return True - - async def _handle_suggestions_command(self, event: MessageEvent) -> str: - """Handle /suggestions in the gateway. - - Delegates to the shared handler so CLI and gateway never drift. The origin is built from - the event source so an accepted suggestion's job delivers back to this chat/thread. - """ - args = (event.get_command_args() or "").strip() - origin = _command_origin_for_source(event.source) - try: - from hermes_cli.suggestions_cmd import handle_suggestions_command - - return handle_suggestions_command(args, origin=origin, surface="gateway") - except Exception as e: - logger.debug("suggestions command failed: %s", e) - return f"Suggestions command failed: {e}" - - async def _handle_blueprint_command(self, event: MessageEvent): - """Handle /blueprint in the gateway. - - Delegates to the shared handler so CLI, TUI, and gateway never drift. Origin is built - from the event source so a directly created blueprint job delivers back to this chat. - """ - args = (event.get_command_args() or "").strip() - origin = _command_origin_for_source(event.source) - try: - from hermes_cli.blueprint_cmd import handle_blueprint_command - - return handle_blueprint_command(args, origin=origin, surface="gateway") - except Exception as e: - logger.debug("blueprint command failed: %s", e) - from hermes_cli.blueprint_cmd import BlueprintCommandResult - - return BlueprintCommandResult(f"Cron blueprint command failed: {e}") - - # ──────────────────────────────────────────────────────────────── - # /goal — persistent cross-turn goals (Ralph-style loop) - # ──────────────────────────────────────────────────────────────── - def _goal_max_turns_from_config(self) -> int: - """Resolve the configured /goal turn budget for gateway sessions. - - GatewayRunner.config is a GatewayConfig dataclass, not the full user config mapping, so - top-level blocks such as ``goals`` are only reachable via hermes_cli.config.load_config(). - """ - try: - goals_cfg = ( - (self.config or {}).get("goals", {}) - if isinstance(self.config, dict) - else getattr(self.config, "goals", {}) or {} - ) - if not goals_cfg: - from hermes_cli.config import load_config - - goals_cfg = (load_config() or {}).get("goals") or {} - return int(goals_cfg.get("max_turns", 20) or 20) - except Exception: - return 20 - - async def _warm_goals_session_db(self, label: str) -> None: - """Warm the goals SessionDB cache off-loop (best-effort). - - A cold cache runs the state.db init on the loop thread and freezes the loop for the init - duration. The executor hop keeps the profile home override alive under multiplex, so the - warm cache belongs to the caller's profile. On failure the caller falls back to the - bootstrap windows, so a dropped warm-up is a bounded stall, never a crash. - """ - try: - from hermes_cli.goals import _get_session_db as _warm_goals_db - - await self._run_in_executor_with_context(_warm_goals_db) - except Exception as exc: - logger.warning("%s: session DB warm-up failed: %s", label, exc) - - async def _session_entry_for_manager(self, event: "MessageEvent", label: str): - """Session entry for a /goal or /heartbeat manager, or None when lookup fails. - - Warms the SessionDB cache off-loop first: a cold cache freezes the loop for the init - duration and drops the first write while the reply claims it was set. Internal events look - the session up WITHOUT touching activity so they never advance the idle/daily reset clock. - """ - await self._warm_goals_session_db(label) - try: - session_entry = await self.async_session_store.get_or_create_session( - event.source, - touch_activity=not bool(getattr(event, "internal", False)), - ) - except Exception as exc: - logger.debug("%s: session lookup failed: %s", label, exc) - return None - if not (getattr(session_entry, "session_id", None) or ""): - return None - return session_entry - - async def _get_goal_manager_for_event(self, event: "MessageEvent"): - """Return ``(GoalManager, session_entry)`` for this event, or ``(None, None)``.""" - try: - from hermes_cli.goals import GoalManager - except Exception as exc: - logger.debug("goal manager unavailable: %s", exc) - return None, None - session_entry = await self._session_entry_for_manager(event, "goal manager") - if session_entry is None: - return None, None - max_turns = self._goal_max_turns_from_config() - return GoalManager(session_id=session_entry.session_id, default_max_turns=max_turns), session_entry - - async def _get_heartbeat_manager_for_event(self, event: "MessageEvent"): - """Return ``(HeartbeatManager, session_entry)`` for this event, or ``(None, None)``.""" - try: - from hermes_cli.heartbeat import HeartbeatManager - except Exception as exc: - logger.debug("heartbeat manager unavailable: %s", exc) - return None, None - session_entry = await self._session_entry_for_manager(event, "heartbeat manager") - if session_entry is None: - return None, None - return HeartbeatManager(session_id=session_entry.session_id), session_entry - - def _register_heartbeat_watch(self, quick_key: str, source: Any, session_id: str) -> None: - """Track a session with an active heartbeat and start the poller. - - The registry maps ``quick_key`` → ``(source, session_id)`` so the poller can rebuild a - MessageEvent and enqueue via the adapter FIFO. In-memory by design: heartbeat STATE - survives restarts in SessionDB, but firing resumes only when the user touches /heartbeat - again (durable schedules belong to cron). - """ - watch = getattr(self, "_heartbeat_watch", None) - if watch is None: - watch = {} - self._heartbeat_watch = watch - watch[quick_key] = (source, session_id) - self._start_heartbeat_poller() - - def _unregister_heartbeat_watch(self, quick_key: str) -> None: - watch = getattr(self, "_heartbeat_watch", None) - if watch: - watch.pop(quick_key, None) - - def _start_heartbeat_poller(self) -> None: - """Start the single gateway-wide heartbeat poll task (idempotent).""" - existing = getattr(self, "_heartbeat_poll_task", None) - if existing is not None and not existing.done(): - return - - from hermes_cli.heartbeat import POLL_SECONDS - - async def _poll_loop(): - while True: - await asyncio.sleep(POLL_SECONDS) - watch = getattr(self, "_heartbeat_watch", None) - if not watch: - continue - # Warm the cache off-loop once per poll. A watch can only be registered through the - # warmed /heartbeat command, so this covers only the degraded path where that warm- - # up failed. - await self._warm_goals_session_db("heartbeat poll") - for quick_key, (source, session_id) in list(watch.items()): - try: - # Busy sessions coalesce their tick to the next idle poll. - if quick_key in self._running_agents: - continue - from hermes_cli.heartbeat import HeartbeatManager - - mgr = HeartbeatManager(session_id=session_id) - if not mgr.has_heartbeat(): - watch.pop(quick_key, None) - continue - prompt = mgr.due_prompt() - if not prompt: - continue - adapter = self._adapter_for_source(source) - if adapter is None: - continue - hb_event = MessageEvent( - text=prompt, - message_type=MessageType.TEXT, - source=source, - message_id=None, - channel_prompt=None, - ) - self._enqueue_fifo(quick_key, hb_event, adapter) - except Exception as exc: - logger.debug("heartbeat poll for %s failed: %s", quick_key, exc) - - try: - task = asyncio.create_task(_poll_loop()) - self._heartbeat_poll_task = task - # PERMANENT once started (an infinite while-True loop, no exit condition) — same as a - # _spawn_supervised watcher. Tag it so _scale_to_zero_has_live_background_work() doesn't - # treat a gateway with an active heartbeat watch as busy forever. - task._hermes_supervised_watcher = True # type: ignore[attr-defined] - _bg = getattr(self, "_background_tasks", None) - if _bg is not None: - _bg.add(task) - task.add_done_callback(_bg.discard) - except Exception: - logger.debug("Failed to start heartbeat poller", exc_info=True) - - async def _send_goal_status_notice(self, source: Any, message: str) -> None: - """Send a /goal judge status line back to the originating chat/thread.""" - adapter = self._adapter_for_source(source) - if not adapter: - logger.debug("goal continuation: no adapter for %s", getattr(source, "platform", None)) - return - - try: - metadata = self._thread_metadata_for_source(source) - except Exception: - metadata = None - - result = await adapter.send(source.chat_id, message, metadata=metadata) - if result is not None and not getattr(result, "success", True): - logger.warning( - "goal continuation: status send failed: %s", - getattr(result, "error", "unknown error"), - ) - - async def _defer_goal_status_notice_after_delivery(self, source: Any, message: str) -> None: - """Send a /goal status line after the main response is delivered. - - The adapter sends the agent response after this caller returns, so for reading order the - status must follow that send: use the adapter's one-shot post-delivery callback when - available, else fall back to direct awaited delivery rather than dropping the notice. - """ - adapter = self._adapter_for_source(source) - if not adapter: - logger.debug("goal continuation: no adapter for %s", getattr(source, "platform", None)) - return - - async def _deliver() -> None: - try: - await self._send_goal_status_notice(source, message) - except Exception as exc: - logger.warning("goal continuation: status send failed: %s", exc, exc_info=True) - - try: - session_key = self._session_key_for_source(source) - except Exception: - session_key = None - - if session_key and hasattr(adapter, "register_post_delivery_callback"): - try: - generation = None - active = getattr(adapter, "_active_sessions", {}).get(session_key) - if active is not None: - generation = getattr(active, "_hermes_run_generation", None) - adapter.register_post_delivery_callback( - session_key, - _deliver, - generation=generation, - ) - return - except Exception as exc: - logger.debug("goal continuation: post-delivery callback registration failed: %s", exc) - - await _deliver() - - async def _post_turn_goal_continuation( - self, - *, - session_entry: Any, - source: Any, - final_response: str, - ) -> None: - """Run the goal judge after a gateway turn and, if still active, enqueue a continuation - prompt for the same session. - - Called at turn boundary AFTER delivery. Uses the adapter's pending-message/FIFO machinery - so a simultaneous real user message is handled by the same queue and takes priority. - """ - try: - from hermes_cli.goals import GoalManager - except Exception as exc: - logger.debug("goal continuation: goals module unavailable: %s", exc) - return - - sid = getattr(session_entry, "session_id", None) or "" - if not sid: - return - - max_turns = self._goal_max_turns_from_config() - - # Warm the SessionDB cache off-loop: a cold cache runs the state.db init on the loop thread - # at the turn boundary; a slow init can drop the goal read and silently end the goal loop. - await self._warm_goals_session_db("goal continuation") - - mgr = GoalManager(session_id=sid, default_max_turns=max_turns) - if not mgr.is_active(): - return - - try: - from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() - except Exception: - _bg_procs = None - - # evaluate_after_turn calls judge_goal(), a synchronous HTTP request to the auxiliary LLM; - # on the event-loop thread it blocks Discord heartbeats 10-40 s and flaps connections, so it - # is offloaded to a thread-pool executor. _run_in_executor_with_context (not bare - # run_in_executor): the profile secret scope and aux runtime context are contextvars; a - # default-executor hop drops them and aux credential resolution fails under multiplexing. - decision = await self._run_in_executor_with_context( - lambda: mgr.evaluate_after_turn( - final_response or "", - user_initiated=True, - background_processes=_bg_procs, - ), - ) - msg = decision.get("message") or "" - - # Defer the status line until after the adapter has delivered the agent's visible final - # response. The judge runs after the response is produced but before BasePlatformAdapter - # sends it, so sending here would show "✓ Goal achieved" before the answer itself. - if msg and source is not None: - await self._defer_goal_status_notice_after_delivery(source, msg) - - if not decision.get("should_continue"): - return - - prompt = decision.get("continuation_prompt") or "" - if not prompt or source is None: - return - - # Enqueue via the adapter's FIFO so a user message already in - # flight preempts the continuation naturally. - try: - adapter = self._adapter_for_source(source) - _quick_key = self._session_key_for_source(source) - if adapter and _quick_key: - cont_event = MessageEvent( - text=prompt, - message_type=MessageType.TEXT, - source=source, - message_id=None, - channel_prompt=None, - ) - self._enqueue_fifo(_quick_key, cont_event, adapter) - except Exception as exc: - logger.debug("goal continuation: enqueue failed: %s", exc) - - async def _run_post_turn_hooks( - self, - *, - agent_result: Any, - source: Any, - is_internal: bool, - event: Any = None, - ) -> None: - """Run goal and loop bookkeeping after an agent turn returns.""" - final_text = self._final_text_for_post_turn_hooks(agent_result, event) - - try: - session_entry = await self.async_session_store.get_or_create_session( - source, - touch_activity=not is_internal, - ) - except Exception as exc: - logger.debug("post-turn session resolution failed: %s", exc) - return - - # Empty interrupted/errored responses must not drive /goal, but an - # in-flight /loop tick still needs to be released and rescheduled. - if final_text.strip(): - try: - await self._post_turn_goal_continuation( - session_entry=session_entry, - source=source, - final_response=final_text, - ) - except Exception as exc: - logger.debug("goal continuation hook failed: %s", exc) - try: - await self._post_turn_loop_completion( - session_entry=session_entry, - source=source, - final_response=final_text, - ) - except Exception as exc: - logger.debug("loop completion hook failed: %s", exc) - - @staticmethod - def _final_text_for_post_turn_hooks(agent_result, event=None) -> str: - """Text for /goal and /loop after a gateway turn. - - Streamed turns return None from _handle_message_with_agent (already_sent). The delivered - reply is stashed on the event so those hooks still see it. - """ - text = "" - if isinstance(agent_result, dict): - text = str(agent_result.get("final_response") or "") - elif isinstance(agent_result, str): - text = agent_result - if text.strip(): - return text - streamed = getattr(event, "_streamed_final_response", None) - if isinstance(streamed, str) and streamed.strip(): - return streamed - return text - - async def _post_turn_loop_completion( - self, - *, - session_entry: Any, - source: Any, - final_response: str, - ) -> None: - """Complete a /loop wakeup tick after a gateway turn. - - No-op unless the session has a loop whose tick is in flight (``awaiting_response`` — set - when the wakeup was injected). Applies the LOOP_COMPLETE marker / --until judge / caps - and schedules the next tick; the idle wakeup watcher fires it when due. - """ - try: - from hermes_cli.loops import LoopManager - except Exception as exc: - logger.debug("loop completion: loops module unavailable: %s", exc) - return - - sid = getattr(session_entry, "session_id", None) or "" - if not sid: - return - - # Warm the SessionDB cache off-loop: a cold cache at the turn boundary stalls the loop for - # the init duration and can drop the tick-completion write (the /goal continuation seam). - await self._warm_goals_session_db("loop completion") - - mgr = LoopManager(session_id=sid) - state = mgr.state - if state is None or not state.awaiting_response: - return - - # The --until judge is a sync aux-LLM call — keep it off the event loop. - decision = await asyncio.get_running_loop().run_in_executor( - None, mgr.complete_tick, final_response or "" - ) - msg = decision.get("message") or "" - if msg and source is not None: - await self._defer_goal_status_notice_after_delivery(source, msg) - - async def _loop_wakeup_watcher(self, interval: float = 15.0) -> None: - """Fire due /loop wakeups for idle gateway sessions. - - The gateway has no per-session scheduler thread, so a coarse ticker scans persisted loops - (SessionDB ``loop:*`` rows) and injects the wakeup prompt into each due session's chat - via the same synthetic-message path used by watch notifications. Deferrals: session - currently running a turn → skip (the FIFO would race the live turn); active non-parked - /goal → skip (goal owns the idle boundary); no routing metadata → skip with a one-time - warning (CLI/TUI loops carry no route). - """ - await asyncio.sleep(5) # let platforms finish connecting - warned_no_route: set = set() - while self._running: - try: - from hermes_cli.loops import ( - LoopManager, - goal_blocks_loop_tick, - list_active_loops, - ) - - # Warm the cache off-loop once per scan: the scan reads every persisted loop, so a - # cold cache would run the state.db init on the loop thread before the first read. - await self._warm_goals_session_db("loop wakeup") - - now = time.time() - for sid, state in list_active_loops(): - if state.awaiting_response or now < state.next_due_at: - continue - route = state.route or {} - platform_name = route.get("platform", "") - chat_id = route.get("chat_id", "") - if not platform_name or not chat_id: - # CLI / TUI-owned loop — their own schedulers drive it. - continue - adapter = None - for p, a in self.adapters.items(): - if p.value == platform_name: - adapter = a - break - if adapter is None: - if sid not in warned_no_route: - warned_no_route.add(sid) - logger.debug( - "loop wakeup: no adapter for platform %r (session %s)", - platform_name, sid, - ) - continue - - # Build the source + session key to check business. - evt_stub = { - "session_key": "", - "platform": platform_name, - "chat_id": chat_id, - "chat_type": route.get("chat_type", ""), - "thread_id": route.get("thread_id", ""), - "user_id": route.get("user_id", ""), - "user_name": route.get("user_name", ""), - } - source = self._build_process_event_source(evt_stub) - if source is None: - continue - try: - session_key = self._session_key_for_source(source) - except Exception: - session_key = None - if session_key and session_key in self._running_agents: - continue # busy — stays due, next scan retries - if goal_blocks_loop_tick(sid): - continue - - mgr = LoopManager(session_id=sid) - if not mgr.is_due(now): - continue - wakeup = mgr.fire_tick() - if not wakeup: - continue - try: - synth_event = MessageEvent( - text=wakeup, - message_type=MessageType.TEXT, - source=source, - internal=True, - ) - logger.info( - "loop wakeup #%s — injecting for %s chat=%s thread=%s", - mgr.state.ticks_fired if mgr.state else "?", - platform_name, source.chat_id, source.thread_id, - ) - await adapter.handle_message(synth_event) - # Slash-command loops dispatch through the command - # path and never hit the post-turn completion hook — - # complete the tick immediately (caps + scheduling). - if wakeup.lstrip().startswith("/"): - mgr.complete_tick("") - except Exception as exc: - logger.warning("loop wakeup injection failed for %s: %s", sid, exc) - with suppress(Exception): - mgr.abandon_tick() - except Exception as exc: - logger.debug("loop wakeup watcher error: %s", exc) - await asyncio.sleep(interval) - - @staticmethod - def _get_guild_id(event: MessageEvent) -> Optional[int]: - """Extract Discord guild_id from the raw message object.""" - raw = getattr(event, "raw_message", None) - if raw is None: - return None - # Slash command interaction - if hasattr(raw, "guild_id") and raw.guild_id: - return int(raw.guild_id) - # Regular message - if hasattr(raw, "guild") and raw.guild: - return raw.guild.id - return None - - async def _handle_voice_channel_join(self, event: MessageEvent) -> str: - """Join the user's current Discord voice channel.""" - adapter = self._adapter_for_source(event.source) - if not hasattr(adapter, "join_voice_channel"): - return "Voice channels are not supported on this platform." - - guild_id = self._get_guild_id(event) - if not guild_id: - return "This command only works in a Discord server." - - voice_channel = await adapter.get_user_voice_channel( - guild_id, event.source.user_id - ) - if not voice_channel: - return "You need to be in a voice channel first." - - # Wire callbacks BEFORE join so voice input arriving immediately - # after connection is not lost. - self._bind_voice_input_callback(adapter) - voice_profile = self._adapter_profile_for_source(event.source) - if hasattr(adapter, "_on_voice_disconnect"): - adapter._on_voice_disconnect = functools.partial( - self._handle_voice_timeout_cleanup, adapter=adapter - ) - # Let the adapter's inactivity timer see the live voice-reply mode so it - # doesn't disconnect a deliberately text-only (/voice off) session. - if hasattr(adapter, "_voice_mode_getter"): - adapter._voice_mode_getter = lambda chat_id: self._voice_mode.get( - self._voice_key(Platform.DISCORD, str(chat_id), profile=voice_profile), - "off", - ) - - try: - success = await adapter.join_voice_channel(voice_channel) - except Exception as e: - logger.warning("Failed to join voice channel: %s", e) - adapter._voice_input_callback = None - err_lower = str(e).lower() - if "pynacl" in err_lower or "nacl" in err_lower or "davey" in err_lower: - return ( - "Voice dependencies are missing (PyNaCl / davey). " - f"Install with: `{sys.executable} -m pip install PyNaCl`" - ) - return f"Failed to join voice channel: {e}" - - if success: - adapter._voice_text_channels[guild_id] = int(event.source.chat_id) - if hasattr(adapter, "_voice_sources"): - adapter._voice_sources[guild_id] = event.source.to_dict() - self._voice_mode[self._voice_key_for_source(event.source)] = "all" - self._save_voice_modes() - self._set_adapter_auto_tts_enabled(adapter, event.source.chat_id, enabled=True) - return ( - f"Joined voice channel **{voice_channel.name}**.\n" - f"I'll speak my replies and listen to you. Use /voice leave to disconnect." - ) - # Join failed — clear callback - adapter._voice_input_callback = None - return "Failed to join voice channel. Check bot permissions (Connect + Speak)." - - async def _handle_voice_channel_leave(self, event: MessageEvent) -> str: - """Leave the Discord voice channel.""" - adapter = self._adapter_for_source(event.source) - guild_id = self._get_guild_id(event) - - if not guild_id or not hasattr(adapter, "leave_voice_channel"): - return "Not in a voice channel." - - if not hasattr(adapter, "is_in_voice_channel") or not adapter.is_in_voice_channel(guild_id): - return "Not in a voice channel." - - try: - await adapter.leave_voice_channel(guild_id) - except Exception as e: - logger.warning("Error leaving voice channel: %s", e) - # Always clean up state even if leave raised an exception - self._voice_mode[self._voice_key_for_source(event.source)] = "off" - self._save_voice_modes() - self._set_adapter_auto_tts_disabled(adapter, event.source.chat_id, disabled=True) - if hasattr(adapter, "_voice_input_callback"): - adapter._voice_input_callback = None - return "Left voice channel." - - def _handle_voice_timeout_cleanup(self, chat_id: str, *, adapter=None) -> None: - """Called by the adapter when a voice channel times out. - - Cleans up runner-side voice_mode state that the adapter cannot reach. ``adapter`` is the - Discord adapter that timed out (bound at join time); under multiplexing that is a - specific profile's bot, not necessarily ``self.adapters[DISCORD]``. - """ - if adapter is None: - adapter = self.adapters.get(Platform.DISCORD) - profile = getattr(adapter, "_owner_profile", None) - self._voice_mode[self._voice_key(Platform.DISCORD, chat_id, profile=profile)] = "off" - self._save_voice_modes() - self._set_adapter_auto_tts_disabled(adapter, chat_id, disabled=True) - - def _is_duplicate_voice_transcript(self, guild_id: int, user_id: int, transcript: str) -> bool: - """Suppress repeated STT outputs for the same recent utterance. - - Voice capture can occasionally emit the same utterance twice a few seconds apart, which - creates a second queued agent run and overlapping spoken replies. - """ - from difflib import SequenceMatcher - - normalized = re.sub(r"\s+", " ", transcript).strip().lower() - normalized = re.sub(r"[^\w\s]", "", normalized) - if not normalized: - return False - - now = time.monotonic() - window_seconds = 12.0 - key = (guild_id, user_id) - recent_store = getattr(self, "_recent_voice_transcripts", None) - if not isinstance(recent_store, dict): - recent_store = {} - self._recent_voice_transcripts = recent_store - recent = [ - (ts, txt) - for ts, txt in recent_store.get(key, []) - if now - ts <= window_seconds - ] - - for _, prior in recent: - if prior == normalized: - recent_store[key] = recent - return True - if len(prior) >= 16 and len(normalized) >= 16: - if SequenceMatcher(None, prior, normalized).ratio() >= 0.95: - recent_store[key] = recent - return True - - recent.append((now, normalized)) - recent_store[key] = recent[-5:] - return False - - async def _handle_voice_channel_input( - self, guild_id: int, user_id: int, transcript: str, *, adapter=None - ): - """Handle transcribed voice from a user in a voice channel. - - ``adapter`` is the Discord adapter that captured the audio (bound via - ``_bind_voice_input_callback``); under multiplexing each profile's bot must dispatch - through its own adapter, never the default profile's. - """ - if adapter is None: - adapter = self.adapters.get(Platform.DISCORD) - if not adapter: - return - - text_ch_id = adapter._voice_text_channels.get(guild_id) - if not text_ch_id: - return - - # Build source — reuse the linked text channel's metadata when available - # so voice input shares the same session as the bound text conversation. - source_data = getattr(adapter, "_voice_sources", {}).get(guild_id) - if source_data: - source = SessionSource.from_dict(source_data) - source.user_id = str(user_id) - source.user_name = str(user_id) - else: - source = SessionSource( - platform=Platform.DISCORD, - chat_id=str(text_ch_id), - user_id=str(user_id), - user_name=str(user_id), - chat_type="channel", - profile=getattr(adapter, "_owner_profile", None), - ) - - # Check authorization before processing voice input - if not self._is_user_authorized(source): - logger.debug("Unauthorized voice input from user %d, ignoring", user_id) - return - - if self._is_duplicate_voice_transcript(guild_id, user_id, transcript): - logger.info( - "Suppressing duplicate voice transcript for guild=%s user=%s: %s", - guild_id, - user_id, - transcript[:100], - ) - return - - # Show transcript in text channel (after auth, with mention sanitization) - try: - channel = adapter._client.get_channel(text_ch_id) - if channel: - safe_text = transcript[:2000].replace("@everyone", "@\u200beveryone").replace("@here", "@\u200bhere") - await channel.send(f"**[Voice]** <@{user_id}>: {safe_text}") - except Exception: - pass - - # Build a synthetic MessageEvent for the normal pipeline; SimpleNamespace raw_message lets - # _get_guild_id() extract guild_id and _send_voice_reply() play audio in the voice channel. - from types import SimpleNamespace - # Resolve the bound text channel's channel_prompt so voice input gets - # the same per-channel context as typed messages (#50149). - channel_prompt: Optional[str] = None - resolver = getattr(adapter, "_resolve_channel_prompt", None) - if callable(resolver): - try: - resolved = resolver(str(text_ch_id)) - channel_prompt = resolved if isinstance(resolved, str) else None - except Exception: - channel_prompt = None - event = MessageEvent( - source=source, - text=transcript, - message_type=MessageType.VOICE, - raw_message=SimpleNamespace(guild_id=guild_id, guild=None), - channel_prompt=channel_prompt, - ) - - await adapter.handle_message(event) - - def _should_send_voice_reply( - self, - event: MessageEvent, - response: str, - agent_messages: list, - already_sent: bool = False, - ) -> bool: - """Decide whether the runner should send a TTS voice reply. - - False when voice_mode is off for this chat, the response is empty/an error, the agent - already called text_to_speech (dedup), or voice input + base adapter auto-TTS already - handled it (skip_double) — UNLESS streaming consumed the response (already_sent=True), - since then the base adapter has no text for auto-TTS and the runner must handle it. - """ - if not response or response.startswith("Error:"): - return False - - chat_id = event.source.chat_id - voice_key = self._voice_key_for_source(event.source) - voice_mode = self._voice_mode.get(voice_key) - is_voice_input = (event.message_type == MessageType.VOICE) - - adapter = self._adapter_for_source(event.source) - adapter_auto_tts = False - if adapter and hasattr(adapter, "_should_auto_tts_for_chat"): - try: - adapter_auto_tts = bool(adapter._should_auto_tts_for_chat(chat_id)) - except Exception: - adapter_auto_tts = False - - should = ( - (voice_mode == "all") - or (voice_mode == "voice_only" and is_voice_input) - # ``voice.auto_tts`` (synced into the adapter at startup) is the fallback only when the - # chat has no explicit mode; the chat-level all/voice_only/off choice takes precedence. - or (voice_mode is None and adapter_auto_tts) - ) - if not should: - logger.debug( - "Auto voice reply skipped: mode=%s adapter_auto_tts=%s chat=%s platform=%s", - voice_mode, adapter_auto_tts, chat_id, event.source.platform.value, - ) - return False - - # Dedup: agent already called TTS tool in THIS turn only - last_user_idx = None - for i, msg in enumerate(reversed(agent_messages)): - if msg.get("role") == "user": - last_user_idx = len(agent_messages) - 1 - i; break - turn_messages = agent_messages[last_user_idx:] if last_user_idx is not None else agent_messages - has_agent_tts = any( - msg.get("role") == "assistant" - and any( - (tc.get("function") or {}).get("name") == "text_to_speech" - for tc in (msg.get("tool_calls") or []) - ) - for msg in turn_messages - ) - if has_agent_tts: - return False - - # Dedup: base adapter auto-TTS already handles voice input (play_tts plays in VC when - # connected), so the runner can skip — unless streaming already delivered the text - # (already_sent): then the base adapter gets None, can't run auto-TTS, and the runner must. - return not (is_voice_input and not already_sent) - - def _should_echo_stt_transcripts(self) -> bool: - """Return whether inbound voice/STT transcripts should be echoed to chat.""" - return bool(getattr(self.config, "stt_echo_transcripts", True)) - - async def _send_voice_reply(self, event: MessageEvent, text: str) -> None: - """Generate TTS audio and send as a voice message before the text reply.""" - audio_path = None - actual_paths: List[str] = [] - try: - from tools.tts_tool import text_to_speech_tool, _strip_markdown_for_tts - - tts_text = _strip_markdown_for_tts(text) - if not tts_text: - return - - # Platforms whose native voice bubbles require Ogg/Opus (OPUS_VOICE_PLATFORMS — - # Telegram, Matrix, Feishu, WhatsApp, Signal) get an explicit .ogg path; the TTS tool's - # central container repair guarantees real Ogg/Opus bytes for every provider. - audio_path = build_auto_tts_output_path(event.source.platform) - - result_json = await asyncio.to_thread( - text_to_speech_tool, text=tts_text, output_path=audio_path - ) - try: - result = json.loads(result_json) - except (json.JSONDecodeError, TypeError): - logger.warning("Auto voice reply TTS returned invalid JSON: %s", result_json[:200] if result_json else result_json) - return - - # Delivery may be one combined file or several separately valid files (combination - # unavailable or over a platform limit); preserve legacy single-file results. - actual_paths = result.get("file_paths") or [ - result.get("file_path", audio_path) - ] - actual_paths = [ - str(path) for path in actual_paths - if path and os.path.isfile(path) - ] - if not result.get("success") or not actual_paths: - logger.warning("Auto voice reply TTS failed: %s", result.get("error")) - return - - adapter = self._adapter_for_source(event.source) - - # If connected to a voice channel, play there instead of sending a file - guild_id = self._get_guild_id(event) - play_in_voice_channel = getattr(adapter, "play_in_voice_channel", None) - is_in_voice_channel = getattr(adapter, "is_in_voice_channel", None) - send_voice = getattr(adapter, "send_voice", None) - in_voice_channel = bool( - guild_id - and callable(play_in_voice_channel) - and callable(is_in_voice_channel) - and is_in_voice_channel(guild_id) - ) - reply_anchor = self._reply_anchor_for_event(event) - thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) - if not in_voice_channel and callable(send_voice): - # Mark the auto voice reply as notify-worthy (mirrors the final-text path in - # platforms/base.py) so adapters that gate push notifications (Telegram "important" - # mode) deliver it as a normal notification, not a silent message. Clone first so - # we don't mutate metadata shared with concurrent typing-indicator state. - if thread_meta is not None: - thread_meta = dict(thread_meta) - thread_meta["notify"] = True - else: - thread_meta = {"notify": True} - for actual_path in actual_paths: - if in_voice_channel: - play_voice = cast(Callable[..., Awaitable[Any]], play_in_voice_channel) - await play_voice(guild_id, actual_path) - elif callable(send_voice): - send_voice_call = cast(Callable[..., Awaitable[Any]], send_voice) - send_kwargs: Dict[str, Any] = { - "chat_id": event.source.chat_id, - "audio_path": actual_path, - "reply_to": reply_anchor, - "metadata": thread_meta, - } - await send_voice_call(**send_kwargs) - except Exception as e: - logger.warning("Auto voice reply failed: %s", e, exc_info=True) - finally: - for p in ({audio_path, *actual_paths} - {None}): - with suppress(OSError): - os.unlink(p) - - async def _deliver_media_from_response( - self, - response: str, - event: MessageEvent, - adapter, - thread_metadata: Optional[Dict[str, Any]] = None, - ) -> None: - """Extract explicit MEDIA: tags from an already-streamed response and deliver them. - - The text is already delivered; this only handles file attachments the normal - _process_message_background path would have caught. Unlike the non-streaming path in - ``gateway/platforms/base.py`` this rescan is EXPLICIT-ONLY: a bare local path in a - streamed reply was either shown as text or is stale inspected content, and promoting it - sent files the model never asked to deliver. - """ - from pathlib import Path - from urllib.parse import quote as _quote - - try: - # Capture [[as_document]] before extract_media strips it, so the dispatch partition - # below can route image-extension files through send_document (preserving bytes) instead - # of send_multiple_images (Telegram sendPhoto recompresses to ~1280px). - force_document_attachments = "[[as_document]]" in response - - from gateway.platforms.base import BasePlatformAdapter, should_send_media_as_audio - - media_files, cleaned = adapter.extract_media(response) - media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files) - # Do NOT deduplicate explicit MEDIA tags against prior turns here. This rescan is - # already EXPLICIT-ONLY (see docstring): a MEDIA: directive in the final streamed reply - # is the model deliberately attaching a file — including a user-requested resend. Stale - # auto-appended tags are deduped upstream (_collect_auto_append_media_tags). Strip image - # URLs for parity with the non-streaming chain, but do NOT run extract_local_files here. - adapter.extract_images(cleaned) - - _thread_meta = ( - dict(thread_metadata) - if thread_metadata is not None - else self._thread_metadata_for_source( - event.source, - self._reply_anchor_for_event(event), - ) - ) - - _VIDEO_EXTS = {'.mp4', '.mov', '.avi', '.mkv', '.webm', '.3gp'} - _IMAGE_EXTS = {'.jpg', '.jpeg', '.png', '.webp', '.gif'} - - # Partition out images so they can be sent as a single batch (e.g. Signal's multi- - # attachment RPC). When [[as_document]] was set, image-extension files skip the photo - # path and route to send_document below — preserving original bytes. - image_paths: list = [] - non_image_media: list = [] - for media_path, is_voice in media_files: - ext = Path(media_path).suffix.lower() - if (ext in _IMAGE_EXTS - and not is_voice - and not force_document_attachments): - image_paths.append(media_path) - else: - non_image_media.append((media_path, is_voice)) - - if image_paths: - try: - images = [(f"file://{_quote(p)}", "") for p in image_paths] - await adapter.send_multiple_images( - chat_id=event.source.chat_id, - images=images, - metadata=_thread_meta, - ) - except Exception as e: - logger.warning("[%s] Post-stream image batch delivery failed: %s", adapter.name, e) - - for media_path, is_voice in non_image_media: - try: - ext = Path(media_path).suffix.lower() - if should_send_media_as_audio(event.source.platform, ext, is_voice=is_voice): - await adapter.send_voice( - chat_id=event.source.chat_id, - audio_path=media_path, - metadata=_thread_meta, - is_voice=is_voice, - ) - elif ext in _VIDEO_EXTS: - await adapter.send_video( - chat_id=event.source.chat_id, - video_path=media_path, - metadata=_thread_meta, - ) - else: - await adapter.send_document( - chat_id=event.source.chat_id, - file_path=media_path, - metadata=_thread_meta, - ) - except Exception as e: - logger.warning("[%s] Post-stream media delivery failed: %s", adapter.name, e) - - except Exception as e: - logger.warning("Post-stream media extraction failed: %s", e) - - async def _deliver_queued_first_response( - self, - response: str, - source: SessionSource, - adapter, - metadata: Optional[Dict[str, Any]] = None, - event_message_id: Optional[str] = None, - text_already_delivered: bool = False, - deliver_media: bool = True, - stream_consumer=None, - ) -> None: - """Deliver a queued response using the normal text+attachment split.""" - if not text_already_delivered: - text_content = _strip_response_attachments_for_direct_send(response, adapter) - if text_content: - # Reconcile-by-edit first (live finding, 2026-08-16 canary): when the stream - # consumer delivered/sealed a message but its recorded payload didn't confirm the - # final (post-stream mutation), plain-sending here creates the duplicate — the - # sealed message already carries most of the answer. - _reconciled = False - _sc_msg_id = getattr(stream_consumer, "message_id", None) - if ( - _sc_msg_id - and _sc_msg_id != "__no_edit__" - and not getattr(stream_consumer, "_turn_split_delivery", False) - ): - try: - _edit_res = await adapter.edit_message( - chat_id=source.chat_id, - message_id=_sc_msg_id, - content=text_content, - finalize=True, - ) - if getattr(_edit_res, "success", False): - _reconciled = True - logger.info( - "Queued-lane final reconciled by editing message %s in place (no duplicate send).", - _sc_msg_id, - ) - except Exception as _qe: - logger.debug( - "Queued-lane reconcile edit failed (%s); falling back to send.", - _qe, - ) - if not _reconciled: - await adapter.send( - source.chat_id, - text_content, - metadata=metadata, - ) - - # Failed turns still deliver their (normalized failure) text above, but must not upload - # attachments as if the turn succeeded — mirrors the ``not agent_result.get("failed")`` - # guard on the completed-turn delivery path. - if not deliver_media: - return - - synthetic_event = MessageEvent( - text="", - source=source, - message_id=event_message_id, - ) - await self._deliver_media_from_response( - response, - synthetic_event, - adapter, - thread_metadata=metadata, - ) - - async def _run_background_task( - self, - prompt: str, - source: "SessionSource", - task_id: str, - event_message_id: Optional[str] = None, - media_urls: Optional[List[str]] = None, - media_types: Optional[List[str]] = None, - ) -> None: - """Profile-scoping wrapper around the background agent task. - - When multiplexing is active, resolve the inbound source's profile and run the whole task - inside ``_profile_runtime_scope`` so credentials resolve from that profile's secret - scope. Mirrors the pattern in ``_run_agent``. - """ - if not getattr(getattr(self, "config", None), "multiplex_profiles", False): - return await self._run_background_task_inner( - prompt, source, task_id, event_message_id, media_urls, media_types, - ) - - profile_home = self._resolve_profile_home_for_source(source) - with _profile_runtime_scope(profile_home): - return await self._run_background_task_inner( - prompt, source, task_id, event_message_id, media_urls, media_types, - ) - - def _resolve_enabled_toolsets_for_source( - self, - user_config: dict, - source: "SessionSource", - platform_key: str, - ) -> list: - """Resolve enabled toolsets for an agent run, honoring per-source overrides. - - An adapter ``toolsets_for_source()`` override (e.g. per-route webhook toolsets) is - validated through the SAME ``_get_platform_tools`` path as normal platform config, so - unknown and platform-restricted toolsets are dropped rather than trusted. Absent an - override, falls back to ``platform_toolsets.``. - """ - from hermes_cli.tools_config import _get_platform_tools - - override = None - try: - adapter = self._adapter_for_source(source) - if adapter is not None: - override = adapter.toolsets_for_source(source) - except Exception: - override = None - - if override and isinstance(override, list): - cfg = dict(user_config) - pts = dict(cfg.get("platform_toolsets") or {}) - pts[platform_key] = [str(t) for t in override] - cfg["platform_toolsets"] = pts - return sorted(_get_platform_tools(cfg, platform_key)) - - return sorted(_get_platform_tools(user_config, platform_key)) - - async def _run_background_task_inner( - self, - prompt: str, - source: "SessionSource", - task_id: str, - event_message_id: Optional[str] = None, - media_urls: Optional[List[str]] = None, - media_types: Optional[List[str]] = None, - ) -> None: - """Execute a background agent task and deliver the result to the chat.""" - from run_agent import AIAgent - - media_urls = media_urls or [] - media_types = media_types or [] - - adapter = self._adapter_for_source(source) - if not adapter: - logger.warning("No adapter for platform %s in background task %s", source.platform, task_id) - return - - _thread_metadata = self._thread_metadata_for_source(source, event_message_id) - - try: - user_config = _load_gateway_config() - model, runtime_kwargs = self._resolve_session_agent_runtime( - source=source, - user_config=user_config, - ) - if not runtime_kwargs.get("api_key"): - await adapter.send( - source.chat_id, - f"❌ Background task {task_id} failed: no provider credentials configured.", - metadata=_thread_metadata, - ) - return - - platform_key = _platform_config_key(source.platform) - - enabled_toolsets = self._resolve_enabled_toolsets_for_source( - user_config, source, platform_key - ) - agent_cfg = user_config.get("agent") or {} - from agent.skill_utils import parse_config_string_list - - disabled_toolsets = parse_config_string_list(agent_cfg.get("disabled_toolsets")) or None - - pr = self._provider_routing - max_iterations = _current_max_iterations() - reasoning_config = self._resolve_session_reasoning_config( - source=source, model=model - ) - self._reasoning_config = reasoning_config - self._service_tier = self._resolve_session_service_tier(source=source) - turn_route = self._resolve_turn_agent_config(prompt, model, runtime_kwargs) - - # Enrich the prompt with image descriptions so the background - # agent can see user-attached images (same as the main flow). - enriched_prompt = prompt - if media_urls: - image_paths = [] - for i, path in enumerate(media_urls): - mtype = media_types[i] if i < len(media_types) else "" - if mtype.startswith("image/"): - image_paths.append(path) - if image_paths: - try: - enriched_prompt = await self._enrich_message_with_vision( - prompt, image_paths, - ) - except Exception as e: - logger.warning("Background task vision enrichment failed: %s", e) - - def run_sync(): - agent = AIAgent( - model=turn_route["model"], - **turn_route["runtime"], - **_checkpoint_agent_kwargs(user_config), - max_iterations=max_iterations, - quiet_mode=True, - verbose_logging=False, - enabled_toolsets=enabled_toolsets, - disabled_toolsets=disabled_toolsets, - reasoning_config=reasoning_config, - service_tier=self._service_tier, - request_overrides=turn_route.get("request_overrides"), - providers_allowed=pr.get("only"), - providers_ignored=pr.get("ignore"), - providers_order=pr.get("order"), - provider_sort=pr.get("sort"), - provider_require_parameters=pr.get("require_parameters", False), - provider_data_collection=pr.get("data_collection"), - session_id=task_id, - platform=platform_key, - user_id=source.user_id, - user_id_alt=source.user_id_alt, - user_name=source.user_name, - chat_id=source.chat_id, - chat_name=source.chat_name, - chat_type=source.chat_type, - thread_id=source.thread_id, - session_db=getattr(self._session_db, "_db", self._session_db), - # Reload from disk — do not reuse the startup snapshot (#60955). - fallback_model=self._refresh_fallback_model(), - ) - try: - return agent.run_conversation( - user_message=enriched_prompt, - task_id=task_id, - ) - finally: - self._cleanup_agent_resources(agent) - - result = await self._run_in_executor_with_context(run_sync) - - response = result.get("final_response", "") if result else "" - if not response and result and result.get("error"): - response = f"Error: {result['error']}" - - # Background tasks start a fresh conversation, so history_offset=0: every message in the - # run belongs to this turn. Mirrors the repair on the main turn path. - if response: - response = repair_explicit_computer_use_media_paths( - response, - result.get("messages", []), - ) - - # Extract media files from the response - if response: - media_files, response = adapter.extract_media(response) - from gateway.platforms.base import BasePlatformAdapter - media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files) - images, text_content = adapter.extract_images(response) - - preview = prompt[:60] + ("..." if len(prompt) > 60 else "") - header = f'✅ Background task complete\nPrompt: "{preview}"\n\n' - - if text_content: - await adapter.send( - chat_id=source.chat_id, - content=header + text_content, - metadata=_thread_metadata, - ) - elif not images and not media_files: - await adapter.send( - chat_id=source.chat_id, - content=header + "(No response generated)", - metadata=_thread_metadata, - ) - - # Send extracted images - for image_url, alt_text in (images or []): - with suppress(Exception): - await adapter.send_image( - chat_id=source.chat_id, - image_url=image_url, - caption=alt_text, - metadata=_thread_metadata, - ) - - # Route each media file by type so a TTS clip arrives as a voice bubble and a clip - # as a video rather than a generic document. Mirrors the streaming + kanban paths. - from gateway.platforms.base import ( - should_send_media_as_audio as _should_send_media_as_audio, - ) - _IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp"} - _VIDEO_EXTS = {".mp4", ".mov", ".avi", ".mkv", ".webm", ".3gp"} - for media_path, _is_voice in (media_files or []): - _ext = os.path.splitext(media_path)[1].lower() - try: - if _should_send_media_as_audio(source.platform, _ext, _is_voice): - await adapter.send_voice( - chat_id=source.chat_id, - audio_path=media_path, - metadata=_thread_metadata, - is_voice=_is_voice, - ) - elif _ext in _VIDEO_EXTS: - await adapter.send_video( - chat_id=source.chat_id, - video_path=media_path, - metadata=_thread_metadata, - ) - elif _ext in _IMAGE_EXTS: - await adapter.send_image_file( - chat_id=source.chat_id, - image_path=media_path, - metadata=_thread_metadata, - ) - else: - await adapter.send_document( - chat_id=source.chat_id, - file_path=media_path, - metadata=_thread_metadata, - ) - except Exception: - pass - else: - preview = prompt[:60] + ("..." if len(prompt) > 60 else "") - await adapter.send( - chat_id=source.chat_id, - content=f'✅ Background task complete\nPrompt: "{preview}"\n\n(No response generated)', - metadata=_thread_metadata, - ) - - except Exception as e: - logger.exception("Background task %s failed", task_id) - with suppress(Exception): - await adapter.send( - chat_id=source.chat_id, - content=f"❌ Background task {task_id} failed: {e}", - metadata=_thread_metadata, - ) - - async def _get_telegram_topic_capabilities(self, source: SessionSource) -> dict: - """Read Telegram private-topic capability flags via Bot API getMe.""" - adapter = self._adapter_for_source(source) - bot = getattr(adapter, "_bot", None) - if bot is None or not hasattr(bot, "get_me"): - return {"checked": False} - try: - me = await bot.get_me() - except Exception: - logger.debug("Failed to fetch Telegram getMe topic capabilities", exc_info=True) - return {"checked": False} - - def _field(name: str): - if hasattr(me, name): - return getattr(me, name) - api_kwargs = getattr(me, "api_kwargs", None) - if isinstance(api_kwargs, dict) and name in api_kwargs: - return api_kwargs.get(name) - if isinstance(me, dict): - return me.get(name) - return None - - return { - "checked": True, - "has_topics_enabled": _field("has_topics_enabled"), - "allows_users_to_create_topics": _field("allows_users_to_create_topics"), - } - - async def _ensure_telegram_system_topic(self, source: SessionSource) -> None: - """Create/pin the managed System topic after /topic activation when possible.""" - adapter = self._adapter_for_source(source) - if adapter is None or not source.chat_id: - return - - thread_id = None - create_topic = getattr(adapter, "_create_dm_topic", None) - if callable(create_topic): - try: - thread_id = await create_topic(int(source.chat_id), "System") - except Exception: - logger.debug("Failed to create Telegram System topic", exc_info=True) - if not thread_id: - return - - message_id = None - try: - send_result = await adapter.send( - source.chat_id, - "System topic for Hermes commands and status.", - metadata={"thread_id": str(thread_id)}, - ) - message_id = getattr(send_result, "message_id", None) - except Exception: - logger.debug("Failed to send Telegram System topic intro", exc_info=True) - if not message_id: - return - - bot = getattr(adapter, "_bot", None) - if bot is None or not hasattr(bot, "pin_chat_message"): - return - try: - await bot.pin_chat_message( - chat_id=int(source.chat_id), - message_id=int(message_id), - disable_notification=True, - ) - except Exception: - logger.debug("Failed to pin Telegram System topic intro", exc_info=True) - - async def _send_telegram_topic_setup_image(self, source: SessionSource) -> None: - """Send the bundled BotFather Threads Settings screenshot when available.""" - adapter = self._adapter_for_source(source) - if adapter is None or not source.chat_id or not hasattr(adapter, "send_image_file"): - return - image_path = Path(__file__).resolve().parent / "assets" / "telegram-botfather-threads-settings.jpg" - if not image_path.exists(): - return - try: - await adapter.send_image_file( - chat_id=source.chat_id, - image_path=str(image_path), - caption="BotFather → Bot Settings → Threads Settings", - metadata={"thread_id": str(source.thread_id)} if source.thread_id else None, - ) - except Exception: - logger.debug("Failed to send Telegram topic setup image", exc_info=True) - - def _sanitize_telegram_topic_title(self, title: str) -> str: - """Return a Bot API-safe forum topic name from a generated session title.""" - cleaned = re.sub(r"\s+", " ", str(title or "")).strip() - if not cleaned: - return "Hermes Chat" - # Telegram forum topic names are short (currently 1-128 chars). Keep - # extra room for multi-byte titles and avoid trailing ellipsis churn. - if len(cleaned) > 120: - cleaned = cleaned[:117].rstrip() + "..." - return cleaned - - def _is_discord_auto_thread_lane(self, source: SessionSource) -> bool: - """Return True only for Discord threads Hermes just auto-created.""" - return ( - source.platform == Platform.DISCORD - and source.chat_type == "thread" - and bool(getattr(source, "auto_thread_created", False)) - and bool(source.thread_id) - and bool(getattr(source, "auto_thread_initial_name", None)) - ) - - def _is_relay_discord_channel_lane(self, source: SessionSource) -> bool: - """Shape-only check: a relay-delivered Discord CHANNEL event whose - reply the connector MAY auto-thread (title-turn registration gate). - - Deliberately does NOT consult the send-result cache: at registration - time (before delivery) the feedback can't exist yet. The rename lane - polls the cache at fire time instead.""" - return ( - source.platform == Platform.DISCORD - and bool(source.chat_id) - and not source.thread_id - and source.chat_type in ("group", "channel") - and getattr(source, "delivered_via_upstream_relay", False) is True - ) - - def _relay_auto_thread_info( - self, source: SessionSource - ) -> Optional[Tuple[str, str]]: - """(thread_id, initial_name) when the RELAY connector auto-threaded our reply to this - source's chat — the title-turn sibling of _is_discord_auto_thread_lane. - - The marker check only matches events ARRIVING IN an auto-created thread (turn 2+); the - auto-title fires on the FIRST exchange, whose source is the PARENT channel event with no - markers. Preferred: the connector's ``prospective_thread_id`` stamp (anchor message id == - the thread it will create) — per-message, so it names the EXACT thread even when several - auto-threads spawn from one channel; the connector's created-name guard enforces - no-clobber. Fallback: the per-chat send-result thread_id/auto_thread_name cache (older - connectors), which only ever renamed the FIRST thread. - """ - if source.platform != Platform.DISCORD or not source.chat_id: - return None - if not getattr(source, "delivered_via_upstream_relay", False): - return None - prospective = getattr(source, "prospective_thread_id", None) - if prospective: - # Deterministic per-thread identity; the empty initial-name marker - # signals the caller to rely on the connector-side no-clobber guard. - return (str(prospective), "") - adapter = self._adapter_for_source(source) - info_fn = getattr(adapter, "auto_thread_info_for_chat", None) - if not callable(info_fn): - return None - try: - return _as_thread_info(info_fn(str(source.chat_id))) - except Exception: - return None - - async def _await_relay_auto_thread_info( - self, source: SessionSource - ) -> Optional[Tuple[str, str]]: - """``_relay_auto_thread_info``, waited out until this turn delivers. - - The legacy send-result path can only answer once the reply is sent, and the caller asks - at title time — one turn early. The adapter answers on the send either way, so the - timeout is only a backstop for a turn that never sends at all; the turn's own inactivity - limit is exactly how long that turn could still be alive. - """ - # The connector-stamped prospective id is known at ingest, so most - # sessions answer here and never wait at all. - known = self._relay_auto_thread_info(source) - if known is not None: - return known - adapter = self._adapter_for_source(source) - wait_fn = getattr(adapter, "wait_for_auto_thread_info", None) - if not callable(wait_fn) or not source.chat_id: - return None - # 0 means the operator disabled the turn limit; the backstop still needs one. - timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) or 1800 - try: - return _as_thread_info(await wait_fn(str(source.chat_id), timeout)) - except Exception: - return None - - def _sanitize_discord_thread_title(self, title: str) -> str: - """Return a Discord-safe semantic thread title from a session title. - - Discord thread names are capped at 100 characters measured in UTF-16 code units (emoji - count double), so truncate with the UTF-16 helpers rather than Python code-point slices. - """ - cleaned = re.sub(r"\s+", " ", str(title or "")).strip() - if not cleaned: - return "Hermes Chat" - if utf16_len(cleaned) > 80: - cleaned = _prefix_within_utf16_limit(cleaned, 77).rstrip() + "..." - return cleaned - - async def _rename_discord_auto_thread_for_session_title( - self, - source: SessionSource, - session_id: str, - title: str, - relay_info: Optional[Tuple[str, str]] = None, - ) -> None: - """Best-effort semantic rename of a newly auto-created Discord thread. - - ``relay_info`` is the (thread_id, initial_name) pair from the relay connector's send- - result feedback — supplied on the title turn, where the source is the parent-channel - event and carries no auto-thread markers (see _relay_auto_thread_info). - """ - if relay_info is None and not await asyncio.to_thread( - self._is_discord_auto_thread_lane, source - ): - # Relay title turn with no feedback captured at schedule time: the title comes off the - # user's opening message, so it beats the delivery that produces the connector's send- - # result feedback (thread_id + initial name) by the whole length of the turn. - if not self._is_relay_discord_channel_lane(source): - return - relay_info = await self._await_relay_auto_thread_info(source) - if relay_info is None: - # True miss: the connector did not auto-thread this reply - # (policy off, DM, already-threaded, or send failed). - return - adapter = self._adapter_for_source(source) if getattr(self, "adapters", None) else None - if adapter is None: - return - rename_thread = getattr(adapter, "rename_thread", None) - if rename_thread is None: - return - target_thread_id = relay_info[0] if relay_info else str(source.thread_id) - # Relay lane (relay_info present): ask the CONNECTOR to enforce the no-clobber guard from - # its own created-name memory — the gateway can't reliably reproduce the thread's initial - # name byte-for-byte (normalization drift silently declined every rename before this). - use_connector_guard = relay_info is not None - guard_name = ( - None - if use_connector_guard - else getattr(source, "auto_thread_initial_name", None) - ) - thread_name = self._sanitize_discord_thread_title(title) - # Relay lane only: the connector's egress guard resolves the owning tenant from the - # outbound scope_id/user_id caches, keyed by the PARENT channel chat_id (learned at - # inbound), not the thread id. rename_thread defaults chat_id to the thread id, so the - # lookup misses and the connector declines; pass the parent channel id (the relay source's - # chat_id). Native lane needs nothing: its source IS the thread, direct Discord API. - parent_chat_id = ( - str(source.chat_id) if use_connector_guard and source.chat_id else None - ) - logger.info( - "discord auto-thread rename: thread=%s lane=%s new_title=%r", - target_thread_id, - "relay" if use_connector_guard else "native", - thread_name, - ) - rename_kwargs = ( - { - "prefer_connector_created": True, - "parent_chat_id": parent_chat_id, - } - if use_connector_guard - else {"only_if_current_name": guard_name} - ) - try: - renamed = await rename_thread( - target_thread_id, - thread_name, - **rename_kwargs, - ) - logger.info( - "discord auto-thread rename result: thread=%s applied=%s", - target_thread_id, - bool(renamed), - ) - except TypeError: - logger.warning( - "Discord semantic thread rename raised TypeError (adapter=%s)", - type(adapter).__name__, - exc_info=True, - ) - except Exception: - logger.debug("Failed to rename Discord auto-thread for generated session title", exc_info=True) - - def _schedule_rename_from_title_thread(self, source: SessionSource, make_coro, label: str) -> None: - """Schedule a best-effort rename coroutine onto the gateway loop from the auto-title thread. - - The source is copied so the background thread never shares the live dataclass with the - loop; failures are logged at debug and never propagate.""" - try: - loop = asyncio.get_running_loop() - except RuntimeError: - loop = getattr(self, "_gateway_loop", None) - if loop is None or loop.is_closed(): - return - try: - copied_source = dataclasses.replace(source) - except Exception: - copied_source = source - future = safe_schedule_threadsafe( - make_coro(copied_source), - loop, - logger=logger, - log_message=f"{label} failed to schedule", - ) - if future is None: - return - - def _log_rename_failure(fut) -> None: - try: - fut.result() - except Exception: - logger.debug("%s failed", label, exc_info=True) - - future.add_done_callback(_log_rename_failure) - - def _schedule_discord_semantic_thread_rename( - self, - source: SessionSource, - session_id: str, - title: str, - ) -> None: - """Schedule Discord auto-thread rename from the auto-title background thread.""" - relay_info = None - if not title: - return - if not self._is_discord_auto_thread_lane(source): - # Relay title turn: the source is the PARENT channel event (thread didn't exist at - # ingest, no auto-thread markers). The connector's send-result feedback says where the - # reply landed, but the auto-title races that delivery, so a cache miss HERE is not a - # verdict. Schedule whenever the SHAPE matches; the async rename lane polls the cache - # (bounded wait) and no-ops on a true miss. - relay_info = self._relay_auto_thread_info(source) - if relay_info is None and not self._is_relay_discord_channel_lane( - source - ): - return - self._schedule_rename_from_title_thread( - source, - lambda copied: self._rename_discord_auto_thread_for_session_title( - copied, session_id, title, relay_info=relay_info - ), - "Discord semantic thread rename", - ) - - async def _rename_telegram_topic_for_session_title( - self, - source: SessionSource, - session_id: str, - title: str, - ) -> None: - """Best-effort rename of a Telegram DM topic when Hermes auto-titles a session.""" - if not await asyncio.to_thread(self._is_telegram_topic_lane, source) or not source.chat_id or not source.thread_id: - return - - # extra.disable_topic_auto_rename lets the operator disable per-topic auto-rename entirely, - # e.g. user-managed topics (ad-hoc Threaded Mode) that auto-rename would keep overwriting. - if self._telegram_topic_auto_rename_disabled(source): - return - - # Skip rename when the topic is operator-declared via extra.dm_topics. Those topics have - # fixed names chosen by the operator (plus optional skill binding); auto-renaming would - # silently mutate operator config. Check the class, not the instance — getattr() on a - # MagicMock auto-creates attributes, so an instance hasattr() is True for every test double. - adapter = self._adapter_for_source(source) - if adapter is not None: - get_info = getattr(type(adapter), "_get_dm_topic_info", None) - if callable(get_info): - try: - operator_topic = get_info(adapter, str(source.chat_id), str(source.thread_id)) - except Exception: - operator_topic = None - # Only treat dict-shaped returns as operator-declared; a - # bare MagicMock or other sentinel shouldn't count. - if isinstance(operator_topic, dict): - return - - session_db = getattr(self, "_session_db", None) - if session_db is not None: - try: - binding = await session_db.get_telegram_topic_binding( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - profile_name=self._telegram_topic_profile_name(source), - ) - if binding and str(binding.get("session_id") or "") != str(session_id): - return - except Exception: - logger.debug("Failed to verify Telegram topic binding before rename", exc_info=True) - return - - if adapter is None: - return - topic_name = self._sanitize_telegram_topic_title(title) - try: - rename_topic = getattr(adapter, "rename_dm_topic", None) - if rename_topic is not None: - await rename_topic( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - name=topic_name, - ) - return - - bot = getattr(adapter, "_bot", None) - edit_forum_topic = getattr(bot, "edit_forum_topic", None) if bot is not None else None - if edit_forum_topic is None: - edit_forum_topic = getattr(bot, "editForumTopic", None) if bot is not None else None - if edit_forum_topic is None: - return - try: - await edit_forum_topic( - chat_id=int(source.chat_id), - message_thread_id=int(source.thread_id), - name=topic_name, - ) - except (TypeError, ValueError): - await edit_forum_topic( - chat_id=source.chat_id, - message_thread_id=source.thread_id, - name=topic_name, - ) - except Exception: - logger.debug("Failed to rename Telegram topic for auto-generated title", exc_info=True) - - def _telegram_topic_auto_rename_disabled(self, source: SessionSource) -> bool: - """Return True when operator disabled per-topic auto-rename for this Telegram chat. - - ``gateway.platforms.telegram.extra.disable_topic_auto_rename``; default False (auto-rename on). - """ - platform_cfg = ( - self.config.platforms.get(source.platform) - if getattr(self, "config", None) and getattr(self.config, "platforms", None) - else None - ) - if platform_cfg is None: - return False - extra = getattr(platform_cfg, "extra", None) or {} - value = extra.get("disable_topic_auto_rename") - if value is None: - return False - if isinstance(value, bool): - return value - if isinstance(value, str): - return value.strip().lower() in {"1", "true", "yes", "on"} - return bool(value) - - def _schedule_telegram_topic_title_rename( - self, - source: SessionSource, - session_id: str, - title: str, - ) -> None: - """Schedule a topic rename from the auto-title background thread.""" - if not title or not self._is_telegram_topic_lane(source): - return - if self._telegram_topic_auto_rename_disabled(source): - return - self._schedule_rename_from_title_thread( - source, - lambda copied: self._rename_telegram_topic_for_session_title(copied, session_id, title), - "Telegram topic title rename", - ) _TELEGRAM_CAPABILITY_HINT_COOLDOWN_S = 300.0 - def _should_send_telegram_capability_hint(self, source: SessionSource) -> bool: - """Rate-limit the BotFather Threads Settings screenshot. - - Repeated /topic while Threads Settings are still off must not re-upload it every time. - """ - if not hasattr(self, "_telegram_capability_hint_ts"): - self._telegram_capability_hint_ts = {} - key = self._telegram_topic_cooldown_key(source) - if not key: - return True - import time as _time - now = _time.monotonic() - last = self._telegram_capability_hint_ts.get(key, 0.0) - if now - last < self._TELEGRAM_CAPABILITY_HINT_COOLDOWN_S: - return False - self._telegram_capability_hint_ts[key] = now - return True - - def _telegram_topic_help_text(self) -> str: - return ( - "/topic — enable multi-session DM mode (one bot, many parallel chats)\n" - "\n" - "Usage:\n" - " /topic Enable topic mode, or show status if already on\n" - " /topic help Show this message\n" - " /topic off Disable topic mode and clear topic bindings\n" - " /topic Inside a topic: restore a previous session by ID\n" - "\n" - "How it works:\n" - "1. Run /topic once in this DM — Hermes checks BotFather Threads\n" - " Settings are enabled and flips on multi-session mode.\n" - "2. Tap All Messages at the top of the bot and send any message.\n" - " Telegram creates a new topic for that message; each topic is\n" - " an independent Hermes session (fresh history, fresh context).\n" - "3. The root DM becomes a system lobby — send /topic, /status,\n" - " /help, /usage there. Normal prompts go in a topic.\n" - "4. /new inside a topic resets just that topic's session.\n" - "5. /topic inside a topic restores an old session into it." - ) - - async def _disable_telegram_topic_mode_for_chat(self, source: SessionSource) -> str: - """Cleanly disable topic mode for a chat via /topic off.""" - if not self._session_db: - from hermes_state import format_session_db_unavailable - return format_session_db_unavailable(prefix=t("gateway.shared.session_db_unavailable_prefix")) - chat_id = str(source.chat_id or "") - if not chat_id: - return "Could not determine chat ID." - # No-op if never enabled. - try: - currently_enabled = await self._session_db.is_telegram_topic_mode_enabled( - chat_id=chat_id, - user_id=str(source.user_id or ""), - profile_name=self._telegram_topic_profile_name(source), - ) - except Exception: - currently_enabled = False - if not currently_enabled: - return "Multi-session topic mode is not currently enabled for this chat." - try: - await self._session_db.disable_telegram_topic_mode( - chat_id=chat_id, - profile_name=self._telegram_topic_profile_name(source), - ) - except Exception as exc: - logger.exception("Failed to disable Telegram topic mode") - return f"Failed to disable topic mode: {exc}" - # Reset per-profile+chat debounce state so the user doesn't see a - # stale cooldown on the next activation (issue #76423). - cooldown_key = self._telegram_topic_cooldown_key(source) - if cooldown_key: - for attr in ("_telegram_lobby_reminder_ts", "_telegram_capability_hint_ts"): - store = getattr(self, attr, None) - if isinstance(store, dict): - store.pop(cooldown_key, None) - return ( - "Multi-session topic mode is now OFF for this chat.\n\n" - "Existing topics in Telegram aren't removed — they'll just stop " - "being gated as independent sessions. The root DM works as a " - "normal Hermes chat again. Run /topic to re-enable later." - ) - - async def _telegram_topic_root_status_message(self, source: SessionSource) -> str: - lines = [ - "Telegram multi-session topics are enabled.", - "", - "To create a new Hermes chat, open All Messages at the top of this " - "bot interface and send any message there. Telegram will create a " - "new topic for it.", - "", - ] - try: - sessions = await self._session_db.list_unlinked_telegram_sessions_for_user( - chat_id=str(source.chat_id), - user_id=str(source.user_id), - profile_name=self._telegram_topic_profile_name(source), - limit=10, - ) - except Exception: - logger.debug("Failed to list unlinked Telegram sessions", exc_info=True) - sessions = [] - - if sessions: - lines.append("Previous unlinked sessions:") - for session in sessions: - session_id = str(session.get("id") or "") - title = str(session.get("title") or "Untitled session") - preview = str(session.get("preview") or "").strip() - line = f"- {title} — `{session_id}`" - if preview: - line += f" — {preview}" - lines.append(line) - lines.extend([ - "", - "To restore one:", - "1. Create or open a topic. To create a new one, open All Messages and send any message there.", - "2. Send /topic inside that topic.", - f"Example: Send /topic {sessions[0].get('id')} inside a topic.", - ]) - else: - lines.extend([ - "No previous unlinked Telegram sessions found.", - "", - "To restore a previous session later:", - "1. Create or open a topic. To create a new one, open All Messages and send any message there.", - "2. Send /topic inside that topic.", - ]) - return "\n".join(lines) - - async def _restore_telegram_topic_session(self, event: MessageEvent, raw_session_id: str) -> str: - """Restore an existing Telegram-owned Hermes session into this topic.""" - source = event.source - session_id = await self._session_db.resolve_session_id(raw_session_id.strip()) - if not session_id: - return f"Session not found: {raw_session_id.strip()}" - - session = await self._session_db.get_session(session_id) - if not session: - return f"Session not found: {raw_session_id.strip()}" - if str(session.get("source") or "") != "telegram": - return "That session is not a Telegram session and cannot be restored into this topic." - if str(session.get("user_id") or "") != str(source.user_id): - return "That session does not belong to this Telegram user." - - linked = await self._session_db.is_telegram_session_linked_to_topic(session_id=session_id) - topic_profile = self._telegram_topic_profile_name(source) - current_binding = await self._session_db.get_telegram_topic_binding( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - profile_name=topic_profile, - ) - if linked: - if not current_binding or current_binding.get("session_id") != session_id: - return "That session is already linked to another Telegram topic." - - session_key = self._session_key_for_source(source) - try: - await self._session_db.bind_telegram_topic( - chat_id=str(source.chat_id), - thread_id=str(source.thread_id), - user_id=str(source.user_id), - session_key=session_key, - session_id=session_id, - managed_mode="restored", - profile_name=topic_profile, - ) - except ValueError as exc: - if "already linked" in str(exc): - return "That session is already linked to another Telegram topic." - raise - - title = await self._session_db.get_session_title(session_id) or session_id - last_assistant = None - try: - for message in reversed(await self._session_db.get_messages(session_id)): - if message.get("role") != "assistant": - continue - projected = project_compaction_message_for_display(message) - if projected is not None and projected.get("content"): - last_assistant = str(projected.get("content")) - break - except Exception: - last_assistant = None - - response = f"Session restored: {title}" - if last_assistant: - response += f"\n\nLast Hermes message:\n{last_assistant}" - return response - - async def _execute_mcp_reload(self, event: MessageEvent) -> str: - """Actually disconnect, reconnect, and notify MCP tool changes. - - Split out so the confirmation wrapper can invoke the same path for button, text reply, - or disabled confirm gate. Under multiplex the reload runs inside the requesting profile's - runtime scope (entered here when the caller did not) and only that profile's servers are - torn down and rediscovered. - """ - multiplex = bool(getattr(self.config, "multiplex_profiles", False)) - if multiplex and not get_hermes_home_override(): - profile_home = self._resolve_profile_home_for_source(event.source) - with _profile_runtime_scope(Path(profile_home)): - return await self._execute_mcp_reload(event) - try: - from tools.mcp_tool import shutdown_mcp_servers, discover_mcp_tools, _servers, _lock - from tools.mcp_tool import _server_scope_keys, reprobe_tool_availability - from tools.registry import registry - - reload_scope = registry.current_scope_key() if multiplex else None - - def _scoped_server_names() -> set: - with _lock: - return { - name for name in _servers - if reload_scope is None or _server_scope_keys.get(name) == reload_scope - } - - # Capture old server names before shutdown - old_servers = _scoped_server_names() - - # Read new config before shutting down, so we know what will be added/removed - # Shutdown existing connections - await self._run_in_executor_with_context( - lambda: shutdown_mcp_servers(scope=reload_scope) - ) - # Explicit reload also re-probes tool availability (check_fn). - reprobe_tool_availability() - - # Reconnect by discovering tools (reads config.yaml fresh) - new_tools = await self._run_in_executor_with_context(discover_mcp_tools) - - # Compute what changed - connected_servers = _scoped_server_names() - if reload_scope is not None: - from tools.mcp_tool import _mcp_tool_server_names - - with _lock: - new_tools = [ - n for n in new_tools - if _mcp_tool_server_names.get(n) in connected_servers - ] - - added = connected_servers - old_servers - removed = old_servers - connected_servers - reconnected = connected_servers & old_servers - - lines = [t("gateway.reload_mcp.header")] - if reconnected: - lines.append(t("gateway.reload_mcp.reconnected", names=", ".join(sorted(reconnected)))) - if added: - lines.append(t("gateway.reload_mcp.added", names=", ".join(sorted(added)))) - if removed: - lines.append(t("gateway.reload_mcp.removed", names=", ".join(sorted(removed)))) - if not connected_servers: - lines.append(t("gateway.reload_mcp.none_connected")) - else: - lines.append(t("gateway.reload_mcp.tools_available", tools=len(new_tools), servers=len(connected_servers))) - - # Refresh cached agents so existing sessions see new MCP tools on their next turn — - # without this, the user has to `/new` (which discards conversation history) to pick up - # tools from a server that was just added or reconnected. - try: - from tools.mcp_tool import refresh_agent_mcp_tools - _cache = getattr(self, "_agent_cache", None) - _cache_lock = getattr(self, "_agent_cache_lock", None) - if _cache_lock is not None and _cache: - # Multiplex: only this profile's sessions; rebuilding another profile's agent in - # this scope would hand it this profile's tool registry. - _ns_prefix = ( - _session_key_namespace(event.source.profile) + ":" - if multiplex else None - ) - with _cache_lock: - for _sess_key, _entry in list(_cache.items()): - if _ns_prefix and not str(_sess_key).startswith(_ns_prefix): - continue - try: - _agent = _entry[0] if isinstance(_entry, tuple) else _entry - except Exception: - continue - if _agent is None: - continue - # Preserve each cached agent's build-time toolset selection EXACTLY: a - # gateway session built with a restricted enabled_toolsets (e.g. - # ["safe"]) must NOT silently gain tools after a reload. Unlike the - # CLI/TUI /reload-mcp (one user re-applying their own config), gateway - # agents are per-session and may be deliberately locked down. - refresh_agent_mcp_tools(_agent, quiet_mode=True) - except Exception as _exc: - logger.debug( - "Failed to update cached agent tools after MCP reload: %s", - _exc, - ) - - # Inject a message at the END of the session history so the model knows tools changed - # next turn; appending after all existing messages preserves the prompt-cache prefix. - change_parts = [] - if added: - change_parts.append(f"Added servers: {', '.join(sorted(added))}") - if removed: - change_parts.append(f"Removed servers: {', '.join(sorted(removed))}") - if reconnected: - change_parts.append(f"Reconnected servers: {', '.join(sorted(reconnected))}") - tool_summary = f"{len(new_tools)} MCP tool(s) now available" if new_tools else "No MCP tools available" - change_detail = ". ".join(change_parts) + ". " if change_parts else "" - reload_msg = { - "role": "user", - "content": f"[IMPORTANT: MCP servers have been reloaded. {change_detail}{tool_summary}. The tool list for this conversation has been updated accordingly.]", - } - try: - session_entry = await self.async_session_store.get_or_create_session(event.source) - await self.async_session_store.append_to_transcript( - session_entry.session_id, reload_msg - ) - except Exception: - pass # Best-effort; don't fail the reload over a transcript write - - return "\n".join(lines) - - except Exception as e: - logger.warning("MCP reload failed: %s", e) - return t("gateway.reload_mcp.failed", error=e) # Slash-command confirmation primitive (generic): for slash commands with an expensive side # effect worth explicit confirmation (currently /reload-mcp, which invalidates the prompt @@ -22897,182 +5085,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # others get a text prompt answered with /approve, /always, or /cancel, matched in # ``_handle_message`` against ``tools.slash_confirm.get_pending()``. - async def _maybe_confirm_destructive_slash( - self, - *, - event: MessageEvent, - command: str, - title: str, - detail: str, - execute, - ) -> Union[str, "EphemeralReply", None]: - """Gate a destructive session slash command (/new, /reset, /undo). - - ``execute`` is an async ``execute() -> str | EphemeralReply`` performing the action. It - runs immediately if ``approvals.destructive_slash_confirm`` is off; otherwise this routes - through ``_request_slash_confirm`` (native buttons or text fallback): ``once`` runs it, - ``always`` persists ``destructive_slash_confirm: false`` then runs it, ``cancel`` returns - a "cancelled" message without running it. - """ - # Gate check. - confirm_required = True - try: - cfg = self._read_user_config() - approvals = cfg.get("approvals") if isinstance(cfg, dict) else None - if isinstance(approvals, dict): - confirm_required = bool(approvals.get("destructive_slash_confirm", True)) - except Exception: - pass - - if not confirm_required: - return await execute() - - session_key = self._session_key_for_source(event.source) - - async def _on_confirm(choice: str): - if choice == "cancel": - return f"🟡 /{command} cancelled. Conversation unchanged." - persisted = False - if choice == "always": - try: - from cli import save_config_value - # save_config_value swallows its own errors and reports the - # outcome in the return value, so the try block alone says - # nothing about whether the write landed. - persisted = bool( - save_config_value("approvals.destructive_slash_confirm", False) - ) - if persisted: - logger.info( - "User opted out of destructive slash confirm (session=%s)", - session_key, - ) - else: - logger.warning( - "Could not persist destructive_slash_confirm=false " - "(session=%s); config.yaml is not writable", - session_key, - ) - except Exception as exc: - logger.warning( - "Failed to persist destructive_slash_confirm=false: %s", exc, - ) - result = await execute() - if choice == "always": - if persisted: - note = ( - "\n\nℹ️ Future /clear, /new, /reset, and /undo will run " - "without confirmation. Re-enable via " - "`approvals.destructive_slash_confirm: true` in config.yaml." - ) - else: - # The user did approve this run, so the action still goes ahead, but the - # preference did not stick and the prompt will be back next time. Say so rather - # than promising an opt-out that was never written. - note = ( - "\n\n⚠️ Could not save that preference (config.yaml is not " - "writable), so /clear, /new, /reset, and /undo will ask " - "again next time. To silence it permanently, set " - "`approvals.destructive_slash_confirm: false` in config.yaml." - ) - if isinstance(result, str): - return result + note - # EphemeralReply or other: leave untouched, since the note would - # mangle structured replies. - return result - return result - - _p = self._typed_command_prefix_for(event.source.platform) - prompt_message = ( - f"⚠️ **Confirm /{command}**\n\n" - f"{detail}\n\n" - "Choose:\n" - "• **Approve Once** — proceed this time only\n" - "• **Always Approve** — proceed and silence this prompt permanently\n" - "• **Cancel** — keep current conversation\n\n" - f"_Text fallback: reply `{_p}approve`, `{_p}always`, or `{_p}cancel`._" - ) - return await self._request_slash_confirm( - event=event, - command=command, - title=title, - message=prompt_message, - handler=_on_confirm, - ) - - async def _request_slash_confirm( - self, - *, - event: MessageEvent, - command: str, - title: str, - message: str, - handler, - ) -> Optional[str]: - """Ask the user to confirm an expensive slash command. - - ``handler(choice: str) -> str`` runs on the event loop when the user responds with - ``"once"``, ``"always"``, or ``"cancel"``; its return value is sent as a gateway message. - Returns the immediate acknowledgment: ``None`` if buttons rendered (self-explanatory), - otherwise the text-fallback message itself IS the ack. - """ - from tools import slash_confirm as _slash_confirm_mod - - source = event.source - session_key = self._session_key_for_source(source) - # Bare-runner test harnesses (object.__new__(GatewayRunner)) skip __init__ and lack the - # counter attribute; fall back to a local counter. Real runs always have the attribute. - counter = getattr(self, "_slash_confirm_counter", None) - if counter is None: - import itertools as _itertools - counter = _itertools.count(1) - self._slash_confirm_counter = counter - confirm_id = f"{next(counter)}" - - # Register the pending confirm FIRST so a super-fast button click - # cannot race the send_slash_confirm return. - _slash_confirm_mod.register(session_key, confirm_id, command, handler) - - adapter = self._adapter_for_source(source) - metadata = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) - - used_buttons = False - if adapter is not None: - try: - button_result = await adapter.send_slash_confirm( - chat_id=source.chat_id, - title=title, - message=message, - session_key=session_key, - confirm_id=confirm_id, - metadata=metadata, - ) - if button_result and getattr(button_result, "success", False): - used_buttons = True - except Exception as exc: - logger.debug( - "send_slash_confirm failed for %s on %s: %s", - command, source.platform, exc, - ) - - if used_buttons: - # Buttons rendered — no redundant text ack. - return None - # Text fallback — return the prompt message as the direct reply. - return message - - def _read_user_config(self) -> Dict[str, Any]: - """Read the user's raw config.yaml (cached) for gate lookups. - - Used by slash-confirm gates that must reflect on-disk state changes - (e.g. a prior "Always Approve" click) without a gateway restart. - """ - try: - from hermes_cli.config import load_config - cfg = load_config() - return cfg if isinstance(cfg, dict) else {} - except Exception: - return {} def _thread_metadata_for_source( self, @@ -23198,611 +5210,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew Platform.FEISHU, Platform.WECOM, Platform.WECOM_CALLBACK, Platform.WEIXIN, Platform.BLUEBUBBLES, Platform.QQBOT, Platform.LOCAL, }) - def _schedule_update_notification_watch(self) -> None: - """Ensure a background task is watching for update completion.""" - existing_task = getattr(self, "_update_notification_task", None) - if existing_task and not existing_task.done(): - return - - try: - self._update_notification_task = asyncio.create_task( - self._watch_update_progress() - ) - except RuntimeError: - logger.debug("Skipping update notification watcher: no running event loop") - - async def _watch_update_progress( - self, - poll_interval: float = 2.0, - stream_interval: float = 4.0, - timeout: float = 1800.0, - ) -> None: - """Watch ``hermes update --gateway``, streaming output + forwarding prompts. - - Polls ``.update_output.txt`` for new content and sends chunks to the user periodically; - detects ``.update_prompt.json`` (written when the update process needs input) and forwards it. - """ - pending_path = _hermes_home / ".update_pending.json" - claimed_path = _hermes_home / ".update_pending.claimed.json" - output_path = _hermes_home / ".update_output.txt" - exit_code_path = _hermes_home / ".update_exit_code" - prompt_path = _hermes_home / ".update_prompt.json" - - loop = asyncio.get_running_loop() - deadline = loop.time() + timeout - - # Resolve the adapter and chat_id for sending messages - adapter = None - chat_id = None - session_key = None - metadata = None - for path in (claimed_path, pending_path): - if path.exists(): - try: - pending = json.loads(path.read_text(encoding="utf-8")) - platform_str = pending.get("platform") - chat_id = pending.get("chat_id") - chat_type = pending.get("chat_type") - session_key = pending.get("session_key") - thread_id = pending.get("thread_id") - message_id = pending.get("message_id") - if platform_str and chat_id: - platform = Platform(platform_str) - adapter = self.adapters.get(platform) - metadata = self._thread_metadata_for_target( - platform, - chat_id, - thread_id, - chat_type=chat_type, - reply_to_message_id=message_id, - adapter=adapter, - ) - # Fallback session key if not stored (old pending files) - if not session_key: - session_key = f"{platform_str}:{chat_id}" - break - except Exception: - pass - - if not adapter or not chat_id: - logger.warning("Update watcher: cannot resolve adapter/chat_id, falling back to completion-only") - # Completion-only fallback: wait for the exit code, then keep polling until - # _send_update_notification actually delivers (True) — it re-resolves the adapter each - # call and returns False (markers kept) while the platform is still reconnecting. - while (pending_path.exists() or claimed_path.exists()) and loop.time() < deadline: - if exit_code_path.exists() and await self._send_update_notification(): - return - await asyncio.sleep(poll_interval) - if (pending_path.exists() or claimed_path.exists()) and not exit_code_path.exists(): - exit_code_path.write_text("124", encoding="utf-8") - await self._send_update_notification() - return - - def _strip_ansi(text: str) -> str: - from tools.ansi_strip import strip_ansi - return strip_ansi(text) - - def _read_output_since(path: Path, offset: int) -> tuple[str, int]: - """Read update output defensively; logs may contain invalid UTF-8.""" - try: - data = path.read_bytes() - except OSError: - return "", offset - if len(data) <= offset: - return "", len(data) - return data[offset:].decode("utf-8", errors="replace"), len(data) - - bytes_sent = 0 - last_stream_time = loop.time() - buffer = "" - - async def _flush_buffer() -> None: - """Send buffered output to the user.""" - nonlocal buffer, last_stream_time - if not buffer.strip(): - buffer = "" - return - # Chunk to fit message limits (Telegram: 4096, others: generous) - clean = _strip_ansi(buffer).strip() - buffer = "" - last_stream_time = loop.time() - if not clean: - return - # Split into chunks if too long - max_chunk = 3500 - chunks = [clean[i:i + max_chunk] for i in range(0, len(clean), max_chunk)] - for chunk in chunks: - try: - await adapter.send( - chat_id, - f"```\n{chunk}\n```", - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - except Exception as e: - logger.debug("Update stream send failed: %s", e) - - while loop.time() < deadline: - # Check for completion - if exit_code_path.exists(): - # Read any remaining output - if output_path.exists(): - try: - chunk, bytes_sent = _read_output_since(output_path, bytes_sent) - if chunk: - buffer += chunk - except OSError: - pass - await _flush_buffer() - - # Send final status - try: - exit_code_raw = exit_code_path.read_text(encoding="utf-8").strip() or "1" - exit_code = int(exit_code_raw) - if exit_code == 0: - await adapter.send( - chat_id, - "✅ Hermes update finished.", - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - else: - await adapter.send( - chat_id, - "❌ Hermes update failed (exit code {}).".format(exit_code), - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - logger.info("Update finished (exit=%s), notified %s", exit_code, session_key) - except Exception as e: - logger.warning("Update final notification failed: %s", e) - - # Cleanup - for p in (pending_path, claimed_path, output_path, - exit_code_path, prompt_path): - p.unlink(missing_ok=True) - (_hermes_home / ".update_response").unlink(missing_ok=True) - _up_done = self._peek_session_state(session_key) - if _up_done is not None: - _up_done.persistent.update_prompt_pending = False - return - - # Check for new output - if output_path.exists(): - try: - chunk, bytes_sent = _read_output_since(output_path, bytes_sent) - if chunk: - buffer += chunk - except OSError: - pass - - # Flush buffer periodically - if buffer.strip() and (loop.time() - last_stream_time) >= stream_interval: - await _flush_buffer() - - # Check for prompts — only forward if we haven't already sent one that's still awaiting - # a response. Without this guard the watcher would re-read the same .update_prompt.json - # every poll cycle and spam the user with duplicate prompt messages. - _up_pending_state = ( - self._peek_session_state(session_key) if session_key else None - ) - if (prompt_path.exists() and session_key - and not ( - _up_pending_state is not None - and _up_pending_state.persistent.update_prompt_pending - )): - try: - prompt_data = json.loads(prompt_path.read_text(encoding="utf-8")) - prompt_text = prompt_data.get("prompt", "") - default = prompt_data.get("default", "") - if prompt_text: - # Flush any buffered output first so the user sees - # context before the prompt - await _flush_buffer() - # Try platform-native buttons first (Discord, Telegram) - sent_buttons = False - if getattr(type(adapter), "send_update_prompt", None) is not None: - try: - await adapter.send_update_prompt( - chat_id=chat_id, - prompt=prompt_text, - default=default, - session_key=session_key, - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - sent_buttons = True - except Exception as btn_err: - logger.debug("Button-based update prompt failed: %s", btn_err) - if not sent_buttons: - default_hint = f" (default: {default})" if default else "" - _p = getattr(adapter, "typed_command_prefix", "/") - await adapter.send( - chat_id, - f"⚕ **Update needs your input:**\n\n" - f"{prompt_text}{default_hint}\n\n" - f"Reply `{_p}approve` (yes) or `{_p}deny` (no), " - f"or type your answer directly.", - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - # Keep the prompt marker on disk until the user answers so a watcher after a - # mid-prompt gateway restart can recover by re-forwarding it. - self._session_state( - session_key - ).persistent.update_prompt_pending = True - # .update_response to continue — it doesn't re-check - logger.info("Forwarded update prompt to %s: %s", session_key, prompt_text[:80]) - except (json.JSONDecodeError, OSError) as e: - logger.debug("Failed to read update prompt: %s", e) - - await asyncio.sleep(poll_interval) - - # Timeout - if not exit_code_path.exists(): - logger.warning("Update watcher timed out after %.0fs", timeout) - exit_code_path.write_text("124", encoding="utf-8") - await _flush_buffer() - with suppress(Exception): - await adapter.send( - chat_id, - "❌ Hermes update timed out after 30 minutes.", - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - for p in (pending_path, claimed_path, output_path, - exit_code_path, prompt_path): - p.unlink(missing_ok=True) - (_hermes_home / ".update_response").unlink(missing_ok=True) - _up_timeout_state = self._peek_session_state(session_key) - if _up_timeout_state is not None: - _up_timeout_state.persistent.update_prompt_pending = False - - async def _send_update_notification(self) -> bool: - """If an update finished, notify the user. - - False while the update is still running (caller may retry); True after a definitive send/skip. - """ - pending_path = _hermes_home / ".update_pending.json" - claimed_path = _hermes_home / ".update_pending.claimed.json" - output_path = _hermes_home / ".update_output.txt" - exit_code_path = _hermes_home / ".update_exit_code" - - if not pending_path.exists() and not claimed_path.exists(): - return False - - cleanup = True - active_pending_path = claimed_path - try: - if pending_path.exists(): - try: - pending_path.replace(claimed_path) - except FileNotFoundError: - if not claimed_path.exists(): - return True - elif not claimed_path.exists(): - return True - - pending = json.loads(claimed_path.read_text(encoding="utf-8")) - platform_str = pending.get("platform") - chat_id = pending.get("chat_id") - chat_type = pending.get("chat_type") - thread_id = pending.get("thread_id") - message_id = pending.get("message_id") - - if not exit_code_path.exists(): - logger.info("Update notification deferred: update still running") - cleanup = False - active_pending_path = pending_path - claimed_path.replace(pending_path) - return False - - exit_code_raw = exit_code_path.read_text(encoding="utf-8").strip() or "1" - exit_code = int(exit_code_raw) - - # Read the captured update output - output = "" - if output_path.exists(): - output = output_path.read_bytes().decode("utf-8", errors="replace") - - # Resolve adapter - platform = Platform(platform_str) - adapter = self.adapters.get(platform) - - if not adapter and chat_id: - # The update finished, but the target platform has not reconnected yet (common right - # after the restart that `hermes update` triggers). Treating "adapter missing" as a - # definitive skip would delete the markers and silently lose the notification; - # preserve them so a later retry (watcher poll or next startup) can deliver it. - logger.info( - "Update notification deferred: %s adapter not connected yet", - platform_str, - ) - cleanup = False - active_pending_path = pending_path - claimed_path.replace(pending_path) - return False - - if adapter and chat_id: - metadata = self._thread_metadata_for_target( - platform, - chat_id, - thread_id, - chat_type=chat_type, - reply_to_message_id=message_id, - adapter=adapter, - ) - # Strip ANSI escape codes for clean display - from tools.ansi_strip import strip_ansi - output = strip_ansi(output).strip() - if output: - if len(output) > 3500: - output = "…" + output[-3500:] - if exit_code == 0: - msg = f"✅ Hermes update finished.\n\n```\n{output}\n```" - else: - msg = f"❌ Hermes update failed.\n\n```\n{output}\n```" - elif exit_code == 0: - msg = "✅ Hermes update finished successfully." - else: - msg = "❌ Hermes update failed. Check the gateway logs or run `hermes update` manually for details." - await adapter.send( - chat_id, - msg, - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - logger.info( - "Sent post-update notification to %s:%s (exit=%s)", - platform_str, - chat_id, - exit_code, - ) - except Exception as e: - logger.warning("Post-update notification failed: %s", e) - finally: - if cleanup: - active_pending_path.unlink(missing_ok=True) - claimed_path.unlink(missing_ok=True) - output_path.unlink(missing_ok=True) - exit_code_path.unlink(missing_ok=True) - - return True - - async def _send_restart_notification(self) -> Optional[tuple[str, str, Optional[str]]]: - """Notify the chat that initiated /restart that the gateway is back.""" - notify_path = _hermes_home / ".restart_notify.json" - if not notify_path.exists(): - return None - - try: - data = json.loads(notify_path.read_text(encoding="utf-8")) - platform_str = data.get("platform") - chat_id = data.get("chat_id") - chat_type = data.get("chat_type") - thread_id = data.get("thread_id") - message_id = data.get("message_id") - - if not platform_str or not chat_id: - return None - - platform = Platform(platform_str) - transport = resolve_delivery_transport(platform, self.config, self.adapters) - if transport is None: - logger.debug( - "Restart notification skipped: no live transport for %s", - platform_str, - ) - return None - - platform_cfg = self.config.platforms.get(platform) - if platform_cfg is not None and not platform_cfg.gateway_restart_notification: - logger.info( - "Restart notification suppressed: %s has gateway_restart_notification=false", - platform_str, - ) - return None - - metadata = self._thread_metadata_for_target( - platform, - chat_id, - thread_id, - chat_type=chat_type, - reply_to_message_id=message_id, - adapter=transport.adapter, - ) - if data.get("delivered_via_upstream_relay") is True: - metadata = dict(metadata or {}) - if data.get("user_id"): - metadata["user_id"] = str(data["user_id"]) - if data.get("scope_id"): - metadata["scope_id"] = str(data["scope_id"]) - result = await transport.send( - platform, - str(chat_id), - "♻ Gateway restarted successfully. Your session continues.", - metadata=_non_conversational_metadata(metadata, platform=platform), - ) - # adapter.send() catches provider errors (e.g. "Chat not found") and returns - # SendResult(success=False) rather than raising, so inspect the result before claiming - # success — otherwise the log line hides real delivery failures. - if result is not None and getattr(result, "success", True) is False: - logger.warning( - "Restart notification to %s:%s was not delivered: %s", - platform_str, - chat_id, - getattr(result, "error", "send returned success=False"), - ) - return None - - logger.info( - "Sent restart notification to %s:%s", - platform_str, - chat_id, - ) - return str(platform_str), str(chat_id), str(thread_id) if thread_id else None - except Exception as e: - logger.warning("Restart notification failed: %s", e) - return None - finally: - notify_path.unlink(missing_ok=True) - - async def _send_home_channel_startup_notifications( - self, - *, - skip_targets: Optional[set[tuple[str, str, Optional[str]]]] = None, - ) -> set[tuple[str, str, Optional[str]]]: - """Notify configured home channels that the gateway is back online. - - The notification is best-effort and sent once per connected platform - home channel. ``skip_targets`` lets startup avoid duplicate messages - when a more specific restart notification is queued for the same chat. - """ - delivered: set[tuple[str, str, Optional[str]]] = set() - skipped = skip_targets or set() - message = "♻️ Gateway online — Hermes is back and ready." - - for platform, platform_cfg in self.config.platforms.items(): - home = platform_cfg.home_channel - if not home or not home.chat_id: - continue - - transport = resolve_delivery_transport(platform, self.config, self.adapters) - if transport is None: - continue - - if not platform_cfg.gateway_restart_notification: - logger.info( - "Home-channel startup notification suppressed: %s has gateway_restart_notification=false", - platform.value, - ) - continue - - target = (platform.value, str(home.chat_id), str(home.thread_id) if home.thread_id else None) - if target in skipped or target in delivered: - continue - - try: - metadata = self._thread_metadata_for_target( - platform, - home.chat_id, - home.thread_id, - adapter=transport.adapter, - ) - if transport.is_relay: - metadata = dict(metadata or {}) - if home.user_id: - metadata["user_id"] = home.user_id - if home.scope_id: - metadata["scope_id"] = home.scope_id - send_metadata = _non_conversational_metadata(metadata, platform=platform) - if send_metadata is not None or transport.is_relay: - result = await transport.send( - platform, - str(home.chat_id), - message, - metadata=send_metadata, - ) - else: - result = await transport.adapter.send(str(home.chat_id), message) - if result is not None and getattr(result, "success", True) is False: - logger.warning( - "Home-channel startup notification failed for %s:%s: %s", - platform.value, - home.chat_id, - getattr(result, "error", "send returned success=False"), - ) - continue - - delivered.add(target) - logger.info( - "Sent home-channel startup notification to %s:%s", - platform.value, - home.chat_id, - ) - except Exception as exc: - logger.warning( - "Home-channel startup notification failed for %s:%s: %s", - platform.value, - home.chat_id, - exc, - ) - - return delivered - - async def _send_session_db_warning_notifications(self) -> None: - """Broadcast a state.db failure warning to all home channels. - - When SessionDB init fails at gateway startup, messages may flow but nothing is persisted - — /resume, /history, and session_search all silently break. Best-effort: failures are - logged, not raised. - """ - error = getattr(self, "_session_db_init_error", None) - if not error: - return - - from hermes_state import classify_persistence_error, format_session_db_unavailable - - cause = classify_persistence_error(error) - hint = format_session_db_unavailable() - if cause == "corrupt": - message = ( - "⚠️ Session database corruption detected. Messages may not be " - "persisted. Recovery options:\n" - "1. Run `hermes doctor --fix`\n" - "2. Salvage with: sqlite3 ~/.hermes/state.db \".recover\" " - "(then replace state.db)\n" - "3. Restore from a backup in ~/.hermes/backups/\n" - "Run `hermes doctor` for sanitized diagnostics." - ) - else: - message = ( - f"⚠️ Session database unavailable — messages may not be persisted. " - f"{hint}\n" - f"Run `hermes doctor` for diagnostics." - ) - - logger.warning( - "Broadcasting state.db failure warning to home channels: %s", error - ) - - for platform, platform_cfg in self.config.platforms.items(): - home = platform_cfg.home_channel - if not home or not home.chat_id: - continue - transport = resolve_delivery_transport(platform, self.config, self.adapters) - if transport is None: - continue - try: - metadata = self._thread_metadata_for_target( - platform, - home.chat_id, - home.thread_id, - adapter=transport.adapter, - ) - if transport.is_relay: - metadata = dict(metadata or {}) - if home.user_id: - metadata["user_id"] = home.user_id - if home.scope_id: - metadata["scope_id"] = home.scope_id - send_metadata = _non_conversational_metadata(metadata, platform=platform) - if send_metadata is not None or transport.is_relay: - result = await transport.send( - platform, - str(home.chat_id), - message, - metadata=send_metadata, - ) - else: - result = await transport.adapter.send(str(home.chat_id), message) - if result is not None and getattr(result, "success", True) is False: - logger.warning( - "state.db warning notification failed for %s:%s: %s", - platform.value, - home.chat_id, - getattr(result, "error", "send returned success=False"), - ) - except Exception as exc: - logger.warning( - "state.db warning notification failed for %s:%s: %s", - platform.value, - home.chat_id, - exc, - ) def _set_session_env(self, context: SessionContext) -> list: """Set session context variables for the current async task. @@ -23915,1438 +5322,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew worker.join(remaining) return sum(1 for worker in workers if worker.is_alive()) - def _decide_image_input_mode( - self, - *, - source: Optional[SessionSource] = None, - session_key: Optional[str] = None, - user_config: Optional[dict] = None, - provider: Optional[str] = None, - model: Optional[str] = None, - ) -> str: - """Resolve image-input routing for the effective model this turn. - - Returns ``"native"`` (attach pixels on the user turn) or ``"text"`` (pre-analyze with - vision_analyze and prepend the description); see agent/image_routing.py. Gateway sessions - can carry /model overrides and image preprocessing runs before AIAgent sets the - auxiliary_client runtime globals, so resolve the per-session runtime bundle the upcoming - turn will use, not just the persisted default. - """ - try: - from agent.image_routing import decide_image_input_mode - from agent.auxiliary_client import _read_main_model, _read_main_provider - from hermes_cli.config import load_config - - cfg = user_config if isinstance(user_config, dict) else load_config() - resolved_provider = (provider or "").strip() - resolved_model = (model or "").strip() - resolved_requested_provider = "" - - needs_session_runtime = not resolved_provider or not resolved_model - has_session_identity = source is not None or session_key - if needs_session_runtime and has_session_identity: - try: - turn_model, runtime_kwargs = self._resolve_session_agent_runtime( - source=source, - session_key=session_key, - user_config=cfg, - ) - if not resolved_model and isinstance(turn_model, str): - resolved_model = turn_model.strip() - runtime_provider = runtime_kwargs.get("provider") if isinstance(runtime_kwargs, dict) else None - runtime_requested_provider = ( - runtime_kwargs.get("requested_provider") - if isinstance(runtime_kwargs, dict) - else None - ) - if not resolved_provider and isinstance(runtime_provider, str): - resolved_provider = runtime_provider.strip() - if isinstance(runtime_requested_provider, str): - resolved_requested_provider = runtime_requested_provider.strip() - except Exception as exc: - logger.debug( - "image_routing: session runtime resolution failed, falling back to config — %s", - exc, - ) - - if not resolved_provider: - resolved_provider = _read_main_provider() - if not resolved_model: - resolved_model = _read_main_model() - - return decide_image_input_mode( - resolved_provider, - resolved_model, - cfg, - requested_provider=resolved_requested_provider, - ) - except Exception as exc: - logger.debug("image_routing: decision failed, falling back to text — %s", exc) - return "text" - - async def _enrich_message_with_vision( - self, - user_text: str, - image_paths: List[str], - ) -> str: - """Auto-analyze user-attached images with the vision tool and prepend the descriptions to - the message text. - - Description *and* local cache path are injected so the model understands the image without - a tool call and can re-examine it with vision_analyze. Returns the enriched message string. - """ - from tools.vision_tools import vision_analyze_tool - from agent.memory_manager import sanitize_context - - analysis_prompt = ( - "Concisely describe this image in 2-4 sentences " - "(~200 Chinese characters or ~150 English words). " - "Cover the main subject, key visible text/data/code, and overall context. " - "If it is a chart, diagram, or scientific figure, include the important " - "labels, legend, and key values. Skip decorative details." - ) - - enriched_parts = [] - for path in image_paths: - try: - logger.debug("Auto-analyzing user image: %s", path) - result_json = await vision_analyze_tool( - image_url=path, - user_prompt=analysis_prompt, - ) - result = json.loads(result_json) - if result.get("success"): - description = result.get("analysis", "") - description = sanitize_context(description) - enriched_parts.append( - f"[The user sent an image~ Here's what I can see:\n{description}]\n" - f"[If you need a closer look, use vision_analyze with " - f"image_url: {path} ~]" - ) - else: - enriched_parts.append( - "[The user sent an image but I couldn't quite see it " - "this time (>_<) You can try looking at it yourself " - f"with vision_analyze using image_url: {path}]" - ) - except Exception as e: - logger.error("Vision auto-analysis error: %s", e) - enriched_parts.append( - f"[The user sent an image but something went wrong when I " - f"tried to look at it~ You can try examining it yourself " - f"with vision_analyze using image_url: {path}]" - ) - - # Combine: vision descriptions first, then the user's original text - if enriched_parts: - prefix = "\n\n".join(enriched_parts) - if user_text: - return f"{prefix}\n\n{user_text}" - return prefix - return user_text - - async def _enrich_message_with_transcription( - self, - user_text: str, - audio_paths: List[str], - ) -> tuple[str, List[str]]: - """Auto-transcribe user voice/audio messages using the configured STT provider and prepend - the transcript to the message text. - - Returns ``(enriched_text, successful_transcripts)``: the message with transcription - wrappers prepended, and the raw transcripts of successfully transcribed clips in input - order (empty if every clip failed or STT is disabled) so callers can echo them back to - the user before the agent loop. - """ - seen = set() - audio_paths = [p for p in audio_paths if p not in seen and not seen.add(p)] - if not getattr(self.config, "stt_enabled", True): - notes = [] - for path in audio_paths: - abs_path = os.path.abspath(path) - duration_str = await _probe_audio_duration(abs_path) - if duration_str: - notes.append( - f"[The user sent a voice message: {abs_path} (duration: {duration_str})]" - ) - else: - notes.append(f"[The user sent a voice message: {abs_path}]") - if not notes: - return user_text, [] - prefix = "\n\n".join(notes) - _placeholder = "(The user sent a message with no text content)" - if user_text and user_text.strip() == _placeholder: - return prefix, [] - if user_text: - return f"{prefix}\n\n{user_text}", [] - return prefix, [] - - try: - from tools.transcription_tools import ( - transcribe_audio, - transcribe_audio_local_fallback, - ) - except ModuleNotFoundError as e: - logger.error("Transcription module unavailable: %s", e) - unavailable_note = "[voice message could not be transcribed]" - _placeholder = "(The user sent a message with no text content)" - if user_text and user_text.strip() == _placeholder: - return unavailable_note, [] - if user_text: - return f"{unavailable_note}\n\n{user_text}", [] - return unavailable_note, [] - - enriched_parts = [] - successful_transcripts: List[str] = [] - for path in audio_paths: - try: - logger.debug("Transcribing user voice: %s", path) - result = await asyncio.to_thread( - transcribe_audio, path, None, "gateway", - ) - if not result.get("success"): - fallback = await asyncio.to_thread( - transcribe_audio_local_fallback, - path, - ) - if fallback.get("success"): - logger.info( - "Configured STT failed for %s; recovered with local STT", - path, - ) - result = fallback - if result["success"]: - transcript = result["transcript"] - # STT may return success=True with an empty/whitespace transcript (silence, cut-off); - # empty quotes make the agent reply to nothing and can loop, so emit a sentinel note. - if not (transcript or "").strip(): - enriched_parts.append( - "[The user sent a voice message but it came through " - "empty or inaudible — speech-to-text returned no " - "words. Do not guess at the content; ask the user " - "to resend or type it out.]" - ) - continue - successful_transcripts.append(transcript) - # Pass the transcript as a plain quoted line: a "The user sent a voice message..." - # wrapper read as a meta-instruction and made the LLM comment on voice mode instead. - enriched_parts.append(f'"{transcript}"') - else: - error = result.get("error", "unknown error") - # All failure branches: one minimal neutral marker. Never mention "no STT provider", - # setup steps, or a DM sent — persisted in history they poison later turns (the model - # keeps volunteering STT-setup advice). Cause is logged for operators, not the prompt. - logger.info("Voice transcription failed for %s: %s", path, error) - from tools.credential_files import to_agent_visible_cache_path - - agent_path = to_agent_visible_cache_path(os.path.abspath(path)) - enriched_parts.append( - "[voice message could not be transcribed automatically; " - f"the audio is available at: {agent_path}]" - ) - except Exception as e: - logger.error("Transcription error: %s", e) - from tools.credential_files import to_agent_visible_cache_path - - agent_path = to_agent_visible_cache_path(os.path.abspath(path)) - enriched_parts.append( - "[voice message could not be transcribed automatically; " - f"the audio is available at: {agent_path}]" - ) - - if enriched_parts: - prefix = "\n\n".join(enriched_parts) - # Strip the empty-content placeholder from the Discord adapter - # when we successfully transcribed the audio — it's redundant. - _placeholder = "(The user sent a message with no text content)" - if user_text and user_text.strip() == _placeholder: - return prefix, successful_transcripts - if user_text: - return f"{prefix}\n\n{user_text}", successful_transcripts - return prefix, successful_transcripts - return user_text, successful_transcripts - - def _pending_event_audio_paths(self, event) -> List[str]: - """Return STT-eligible paths from a pending voice message.""" - audio_paths: List[str] = [] - media_urls = getattr(event, "media_urls", None) or [] - for i, path in enumerate(media_urls): - if _event_media_is_stt_input(event, i): - audio_paths.append(path) - return audio_paths - - async def _transcribe_pending_audio_event_once( - self, - event, - user_text: Optional[str] = None, - ) -> tuple[str | None, List[str]]: - """Transcribe a pending audio event once and cache the result on the event. - - The interrupt monitor and the pending-drain path both need the transcript; caching keeps - it to one STT call and one transcript echo per platform message. - """ - if hasattr(event, "_gateway_pending_stt_text"): - cached_text = getattr(event, "_gateway_pending_stt_text") - cached_transcripts = getattr(event, "_gateway_pending_stt_transcripts", []) or [] - return cached_text, list(cached_transcripts) - - audio_paths = self._pending_event_audio_paths(event) - if not audio_paths: - return user_text if user_text is not None else (getattr(event, "text", None) or None), [] - - text = user_text if user_text is not None else (getattr(event, "text", "") or "") - enriched_text, successful_transcripts = await self._enrich_message_with_transcription( - text, - audio_paths, - ) - setattr(event, "_gateway_pending_stt_text", enriched_text) - setattr(event, "_gateway_pending_stt_transcripts", list(successful_transcripts)) - return enriched_text, successful_transcripts - - async def _echo_pending_stt_transcripts_once( - self, - event, - adapter, - source, - transcripts: List[str], - *, - metadata=None, - log_context: str = "Transcript", - ) -> None: - """Echo pending-event STT transcripts to the chat at most once. - - Tracked as a COUNT (not a set — identical transcripts are distinct deliveries): - ``merge_pending_message_event`` can append a second voice note and invalidate the cache, - and the re-run returns earlier transcripts as a prefix, so only the unsent tail is echoed. - """ - if ( - not transcripts - or not self._should_echo_stt_transcripts() - or adapter is None - ): - return - already_echoed = int(getattr(event, "_gateway_pending_stt_echoed", 0) or 0) - unsent = transcripts[already_echoed:] - setattr(event, "_gateway_pending_stt_echoed", already_echoed + len(unsent)) - for tx in unsent: - try: - await adapter.send( - source.chat_id, - f'🎙️ "{tx}"', - metadata=metadata, - ) - except Exception as echo_exc: - logger.debug("%s echo failed (non-fatal): %s", log_context, echo_exc) - - async def _transcribe_and_echo_pending_voice( - self, - event, - adapter, - source, - text: str, - *, - log_context: str, - metadata=_UNSET, - ) -> tuple[str, List[str]]: - """Transcribe a pending voice event and echo transcripts once. - - Returns ``(enriched_text, transcripts)`` for ``agent.interrupt()`` or the pending-drain - flow; ``(text, [])`` unchanged when there is no STT-eligible media (caller owns the - ``_build_media_placeholder`` fallback for empty ``text`` with non-audio media). - """ - if not self._pending_event_audio_paths(event): - return text, [] - try: - enriched_text, transcripts = await self._transcribe_pending_audio_event_once( - event, - text, - ) - echo_meta = self._thread_metadata_for_source( - source, - self._reply_anchor_for_event(event), - ) if metadata is _UNSET else metadata - await self._echo_pending_stt_transcripts_once( - event, - adapter, - source, - transcripts, - metadata=echo_meta, - log_context=log_context, - ) - return enriched_text or text, transcripts - except Exception as trans_exc: - logger.warning("%s transcription failed: %s", log_context, trans_exc) - return text, [] - - def _build_process_event_source(self, evt: dict): - """Resolve the canonical source for a synthetic background-process event. - - Prefer the persisted session-store origin; the active foreground event causes cross-topic bleed. - """ - from gateway.session import SessionSource - - session_key = str(evt.get("session_key") or "").strip() - derived_platform = "" - derived_chat_type = "" - derived_chat_id = "" - - if session_key: - try: - self.session_store._ensure_loaded() - entry = self.session_store._entries.get(session_key) - if entry and getattr(entry, "origin", None): - return entry.origin - except Exception as exc: - logger.debug( - "Synthetic process-event session-store lookup failed for %s: %s", - session_key, - exc, - ) - - cached_source = self._get_cached_session_source(session_key) - if cached_source is not None: - return cached_source - - _parsed = _parse_session_key(session_key) - if _parsed: - derived_platform = _parsed["platform"] - derived_chat_type = _parsed["chat_type"] - derived_chat_id = _parsed["chat_id"] - - platform_name = str(evt.get("platform") or derived_platform or "").strip().lower() - chat_type = str(evt.get("chat_type") or derived_chat_type or "").strip().lower() - chat_id = str(evt.get("chat_id") or derived_chat_id or "").strip() - if not platform_name or not chat_type or not chat_id: - logger.warning( - "Synthetic event source unresolvable: " - "session_key=%r platform=%r chat_type=%r chat_id=%r " - "evt_type=%s", - session_key, platform_name, chat_type, chat_id, - evt.get("type", "?"), - ) - return None - - try: - platform = Platform(platform_name) - # Reject arbitrary strings (dynamic pseudo-members): built-ins are always valid, plugin - # platforms must be registered in the platform registry. - if platform.value not in _BUILTIN_PLATFORM_VALUES: - try: - from gateway.platform_registry import platform_registry - if not platform_registry.is_registered(platform.value): - raise ValueError(platform_name) - except Exception: - raise ValueError(platform_name) - except Exception: - logger.warning( - "Synthetic process event has invalid platform metadata: %r", - platform_name, - ) - return None - - scope_id = str(evt.get("scope_id") or "").strip() or None - if scope_id is None and chat_type not in ("dm", "thread"): - # Reconstructed (non-persisted) source for a scoped chat with no scope discriminator: a - # relay connector's fail-closed tenant guard may decline the reply unless user_id resolves it - # (resolveByUser). Don't fail — DMs/author-bound chats still route and native adapters need - # no scope_id — but warn so a post-restart egress decline isn't silent. - logger.warning( - "Synthetic event source for %s chat=%s (%s) reconstructed " - "without scope_id; scoped relay egress may be declined by " - "the connector's tenant guard (user_id fallback only).", - platform_name, chat_id, chat_type, - ) - return SessionSource( - platform=platform, - chat_id=chat_id, - chat_type=chat_type, - thread_id=str(evt.get("thread_id") or "").strip() or None, - user_id=str(evt.get("user_id") or "").strip() or None, - user_name=str(evt.get("user_name") or "").strip() or None, - scope_id=scope_id, - ) - - async def _drain_watch_notifications(self, completion_queue) -> None: - """Consume queued watch events and inject them when notifications are enabled. - - The queue is ALWAYS drained (so watch events don't rot or requeue-spin) but injection is - skipped entirely when ``display.background_process_notifications`` is ``off``. - """ - watch_events = _drain_gateway_watch_events(completion_queue) - if self._load_background_notifications_mode() == "off": - return - - for evt in watch_events: - synth_text = _format_gateway_process_notification(evt) - if not synth_text: - continue - try: - await self._inject_watch_notification(synth_text, evt) - except Exception as exc: - logger.error("Watch notification injection error: %s", exc) - - async def _inject_watch_notification( - self, synth_text: str, evt: dict, - ) -> Optional[bool]: - """Inject a watch/completion notification as a synthetic message event. - - Routing comes from the queued event, never the active foreground message. Returns - ``True`` on adapter acceptance, ``False`` on retryable adapter failure, ``None`` with no - gateway route. Not transactional: a crash after acceptance can replay (at-least-once). - """ - source = await asyncio.to_thread(self._build_process_event_source, evt) - if not source: - # API-server sessions bind the RAW X-Hermes-Session-Id key (_bind_api_server_session), not a - # structured ``agent:main:...`` key, so _build_process_event_source returned None above. - raw_sid = str(evt.get("origin_session_id") or "").strip() - if not raw_sid: - _sk = str(evt.get("session_key") or "").strip() - if _sk and _parse_session_key(_sk) is None: - raw_sid = _sk - if raw_sid: - adapter = self.adapters.get(Platform.API_SERVER) - from gateway.wake import ( - adapter_supports_push, - deliver_wake, - persist_delegation_delivery, - ) - if adapter is not None and not adapter_supports_push(adapter): - if evt.get("type") == "async_delegation": - # after the parent turn's event.complete the CLIENT owns the next turn on - # this stateless surface. Persist the completion as a durable delivery row — - # never self-post it as a new role=user prompt. - try: - logger.info( - "Async delegation completion — persisting " - "delivery row for api_server session %s " - "(no wake turn)", - raw_sid, - ) - await persist_delegation_delivery( - adapter, text=synth_text, - session_id=raw_sid, evt=evt, - ) - return True - except Exception as e: - logger.warning( - "Async delegation delivery persist failed " - "for session %s: %s", - raw_sid, e, - ) - return False - try: - logger.info( - "Watch pattern notification — waking api_server " - "session %s via self-post", - raw_sid, - ) - await deliver_wake(adapter, text=synth_text, session_id=raw_sid) - return True - except Exception as e: - logger.warning( - "Watch notification self-post wake failed for " - "session %s: %s", - raw_sid, e, - ) - return False - logger.warning( - "Dropping watch notification for raw session %s: no " - "api_server adapter to self-post through", - raw_sid, - ) - return None - logger.warning( - "Dropping watch notification with no routing metadata for process %s", - evt.get("session_id", "unknown"), - ) - return None - platform_name = source.platform.value if hasattr(source.platform, "value") else str(source.platform) - # Alias-aware resolution (relay-plane): one adapter under Platform.RELAY fronts N logical - # platforms, so a literal ``p.value == platform_name`` scan misses "slack" and drops the - # completion as "no gateway route". Use the shared transport resolver — native adapter wins; - # relay is eligible only when it advertises fronting the logical platform. - adapter = None - try: - _platform_enum = Platform(platform_name) - except (ValueError, KeyError): - _platform_enum = None - if _platform_enum is not None: - try: - _transport = resolve_delivery_transport( - _platform_enum, self.config, self.adapters, - ) - except Exception: - _transport = None - if _transport is not None: - adapter = _transport.adapter - if adapter is None: - # Legacy literal scan — still correct for native adapters; keeps minimal runner stubs (tests) - # and exotic platform strings working when the resolver can't run. - for p, a in self.adapters.items(): - if p.value == platform_name: - adapter = a - break - if not adapter: - return None - from gateway.wake import adapter_supports_push as _wake_push_ok - if not _wake_push_ok(adapter): - # Non-push adapter (api_server) resolved WITH routing metadata: its chat_id is the raw session - # id (_bind_api_server_session binds chat_id = session_id), so handle_message would run the - # wake under a build_session_key() key that never matches the raw session — self-post. - from gateway.wake import deliver_wake, persist_delegation_delivery - raw_sid = str(evt.get("origin_session_id") or "").strip() or str(source.chat_id or "") - if evt.get("type") == "async_delegation": - # Same client-owns-the-turn rule as the raw-key branch above: persist the completion as a - # delivery row, never self-post it as a new role=user prompt. - try: - logger.info( - "Async delegation completion — persisting delivery " - "row for api_server session %s (no wake turn)", - raw_sid, - ) - await persist_delegation_delivery( - adapter, text=synth_text, session_id=raw_sid, evt=evt, - ) - return True - except Exception as e: - logger.warning( - "Async delegation delivery persist failed for " - "session %s: %s", - raw_sid, e, - ) - return False - try: - logger.info( - "Watch pattern notification — waking api_server session " - "%s via self-post", - raw_sid, - ) - await deliver_wake(adapter, text=synth_text, session_id=raw_sid) - return True - except Exception as e: - logger.warning( - "Watch notification self-post wake failed for session " - "%s: %s", - raw_sid, e, - ) - return False - try: - metadata = {} - parent_session_id = str(evt.get("parent_session_id") or "").strip() - if parent_session_id: - metadata["gateway_session_id"] = parent_session_id - synth_event = MessageEvent( - text=synth_text, - message_type=MessageType.TEXT, - source=source, - internal=True, - message_id=str(evt.get("message_id") or "").strip() or None, - metadata=metadata, - ) - logger.info( - "Watch pattern notification — injecting for %s chat=%s thread=%s", - platform_name, - source.chat_id, - source.thread_id, - ) - # Relay-plane egress priming: a synthetic turn injected right after a restart reaches a relay - # adapter whose per-chat routing caches are cold (they warm only on inbound), so its replies - # egress without tenant discriminators and the connector's fail-closed guard declines them. - _prime = getattr(adapter, "prime_routing_cache", None) - if callable(_prime): - _prime(synth_event) - await adapter.handle_message(synth_event) - return True - except Exception as e: - logger.error("Watch notification injection error: %s", e) - return False - - @staticmethod - def _completion_delivery_identity(evt: dict) -> Optional[tuple[str, str, object]]: - """Return a producer-stable identity when one is available. - - Delegation UUIDs identify one producer completion. Process session IDs include the - persisted spawn epoch so a reused ID is a distinct incarnation; legacy events without - ``started_at`` are delivered undeduplicated rather than risk suppressing a real completion. - """ - evt_type = str(evt.get("type") or "") - if evt_type == "async_delegation": - producer_id = str(evt.get("delegation_id") or "") - return (evt_type, producer_id, "") if producer_id else None - if evt_type == "completion": - producer_id = str(evt.get("session_id") or "") - started_at = evt.get("started_at") - if producer_id and started_at is not None: - return (evt_type, producer_id, started_at) - return None - - async def _classify_completion_target(self, parent_session_id: str) -> str: - """Classify an async-completion delivery target before adapter acceptance. - - - ``"deliver"``: spawning session live (or compression-rotated with a live continuation); - proves deliverability only, the resolver still retargets. - - ``"terminal"``: parent gone for good (unknown / explicit user boundary like /new); drop - the durable row rather than falsely ack or replay forever. - - ``"retry"``: transient uncertainty (DB unavailable, rotation mid-flight); release the - claim for a later consumer; the attempt cap bounds churn. - """ - session_db = getattr(self, "_session_db", None) - if session_db is None: - return "retry" - try: - parent = await session_db.get_session(parent_session_id) - except Exception: - logger.debug( - "Async-completion pre-flight parent lookup failed for %s", - parent_session_id, exc_info=True, - ) - return "retry" - if parent is None: - return "terminal" - if not parent.get("ended_at"): - return "deliver" - end_reason = str(parent.get("end_reason") or "") - if end_reason != "compression": - # An ended parent is unreachable only when the USER closed the thread of work (/new -> - # session_reset / new_session, user_exit, session_switch). Idle/timeout ends are normal on - # scale-to-zero relays — the chat stays routable and the resolver retargets, so dropping loses - # finished work. Boundary set shared with the resolver (_USER_BOUNDARY_END_REASONS): no drift. - if end_reason in _USER_BOUNDARY_END_REASONS: - return "terminal" - return "deliver" - try: - tip_session_id = await session_db.get_compression_tip(parent_session_id) - if not tip_session_id or tip_session_id == parent_session_id: - # Rotation caught mid-flight: parent is compression-ended but - # its continuation isn't visible yet. Retry, don't drop. - return "retry" - tip = await session_db.get_session(tip_session_id) - except Exception: - logger.debug( - "Async-completion pre-flight tip lookup failed for %s", - parent_session_id, exc_info=True, - ) - return "retry" - if tip is None or tip.get("ended_at"): - return "retry" - return "deliver" - - async def _deliver_completion_notification( - self, synth_text: str, evt: dict, - ) -> Optional[bool]: - """Deliver once per live gateway, or return False for a retry. - - ``True``: adapter accepted; ``False``: injection failed, claim released for retry; ``None``: - another same-lifecycle caller owns/delivered it, or no route. No cross-process exactly-once. - """ - identity = self._completion_delivery_identity(evt) - durable_claim_id = "" - durable_delegation_id = "" - if evt.get("type") == "async_delegation": - durable_delegation_id = str(evt.get("delegation_id") or "") - if durable_delegation_id: - try: - from tools.async_delegation import claim_completion_delivery - - durable_claim_id = f"gateway:{id(self)}:{__import__('uuid').uuid4().hex}" - if not claim_completion_delivery( - durable_delegation_id, durable_claim_id, - ): - return None - except Exception as exc: - logger.warning( - "Could not claim durable async completion %s: %s", - durable_delegation_id, exc, - ) - return False - parent_session_id = str(evt.get("parent_session_id") or "").strip() - if parent_session_id: - # Adapter acceptance is not proof of delivery: the inner resolver can still fail closed - # inside the pipeline after acceptance, falsely acking the durable row as delivered. - # Verify the target before acceptance so drops get an honest durable disposition. - verdict = await self._classify_completion_target(parent_session_id) - if verdict == "terminal": - logger.warning( - "Async delegation %s targets permanently-gone session %s; " - "terminally dropping delivery (result remains in the " - "delegation records).", - durable_delegation_id or "", parent_session_id, - ) - if durable_claim_id: - try: - from tools.async_delegation import drop_completion_delivery - - drop_completion_delivery( - durable_delegation_id, durable_claim_id, - ) - except Exception: - logger.debug( - "Could not drop durable completion claim", - exc_info=True, - ) - return None - if verdict == "retry": - if durable_claim_id: - try: - from tools.async_delegation import release_completion_delivery - - release_completion_delivery( - durable_delegation_id, durable_claim_id, - ) - except Exception: - logger.debug( - "Could not release durable completion claim", - exc_info=True, - ) - return False - elif evt.get("type") == "completion": - # Background completions carry only session_key, so after /new the OLD session's - # notification would land in the chat's NEW session. Stamped events get the same - # pre-flight as async delegations (_classify_completion_target); unstamped ones deliver. - parent_session_id = str(evt.get("parent_session_id") or "").strip() - if parent_session_id: - verdict = await self._classify_completion_target(parent_session_id) - if verdict == "terminal": - logger.warning( - "Background process %s completion targets " - "permanently-gone session %s (user boundary such as " - "/new); dropping notification (output remains " - "available via process(action='log')).", - evt.get("session_id") or "", parent_session_id, - ) - return None - if verdict == "retry": - # Transient uncertainty (session DB down / compression rotation mid-flight): tell the - # watcher to re-poll and retry rather than drop or misroute the result. - return False - if identity is not None: - with self._completion_delivery_lock: - if ( - identity in self._completion_deliveries_inflight - or identity in self._completion_deliveries_delivered - ): - return None - self._completion_deliveries_inflight.add(identity) - - accepted = False - try: - injection_result = await self._inject_watch_notification(synth_text, evt) - if injection_result is not True: - return injection_result - accepted = True - - if identity is not None: - with self._completion_delivery_lock: - self._completion_deliveries_inflight.discard(identity) - self._completion_deliveries_delivered[identity] = None - while ( - len(self._completion_deliveries_delivered) - > self._completion_delivery_retention - ): - self._completion_deliveries_delivered.popitem(last=False) - - # When the durable async-delegation producer branch is present, its SQLite row is the - # authoritative replay state — ack it after adapter acceptance; no parallel ledger here. - if durable_claim_id: - try: - from tools.async_delegation import complete_completion_delivery - - complete_completion_delivery( - durable_delegation_id, durable_claim_id, - ) - except Exception as exc: - logger.warning( - "Could not acknowledge durable async completion %s: %s", - durable_delegation_id, exc, - ) - return True - finally: - if identity is not None and not accepted: - with self._completion_delivery_lock: - self._completion_deliveries_inflight.discard(identity) - if durable_claim_id and not accepted: - try: - from tools.async_delegation import release_completion_delivery - - release_completion_delivery( - durable_delegation_id, durable_claim_id, - ) - except Exception: - logger.debug("Could not release durable completion claim", exc_info=True) - - @staticmethod - def _completion_notification_batch_key(evt: dict) -> tuple[str, ...]: - """Return a routing-complete key for short-window process fan-in.""" - return tuple(str(evt.get(field) or "") for field in ( - "session_key", - "platform", - "chat_type", - "chat_id", - "thread_id", - "user_id", - )) - - @staticmethod - def _format_coalesced_process_completions(entries: list[tuple[str, dict, asyncio.Future]]) -> str: - """Build one bounded synthetic event from several redacted completions.""" - lines = [ - f"[IMPORTANT: {len(entries)} background processes completed for this session.", - "Treat these results as one completion batch and send at most one " - "consolidated user-facing response.", - ] - shown = entries[:10] - for _text, evt, _future in shown: - session_id = str(evt.get("session_id") or "unknown") - exit_code = evt.get("exit_code") - reason = str(evt.get("completion_reason") or "exited") - # Completion output normally passes the terminal redactor at the producer seam, but that is - # configurable and this is user-facing, so keep the unconditional gateway floor. Redact - # BEFORE slicing: truncating first can leave a credential fragment the patterns miss. - output = _redact_gateway_user_facing_secrets( - str(evt.get("output") or "") - ).strip() - if len(output) > 800: - output = f"[… truncated …]\n{output[-800:]}" - lines.append( - f"\n- {session_id}: exit_code={exit_code}, reason={reason}" - ) - if output: - lines.append(output) - omitted = len(entries) - len(shown) - if omitted: - lines.append( - f"\n- … and {omitted} more completion(s); inspect them with " - "the process tool if they affect the conclusion." - ) - lines.append( - "If a result does not change the current conclusion, absorb it silently.]" - ) - return "\n".join(lines) - - def _record_coalesced_completion_siblings(self, events: list[dict]) -> None: - """Extend a successful primary delivery claim to its batched siblings.""" - with self._completion_delivery_lock: - for evt in events: - identity = self._completion_delivery_identity(evt) - if identity is None: - continue - self._completion_deliveries_inflight.discard(identity) - self._completion_deliveries_delivered[identity] = None - while ( - len(self._completion_deliveries_delivered) - > self._completion_delivery_retention - ): - self._completion_deliveries_delivered.popitem(last=False) - - async def _flush_process_completion_batch(self, key: tuple[str, ...]) -> None: - """Deliver one short-window completion batch and resolve its waiters.""" - current_task = asyncio.current_task() - entries: list[tuple[str, dict, asyncio.Future]] = [] - delivered: Optional[bool] = False - try: - await asyncio.sleep(self._completion_notification_batch_window) - entries = self._completion_notification_batches.pop(key, []) - # Detach before adapter delivery. A completion that arrives while - # this batch is in flight must be able to schedule the next flush. - if self._completion_notification_batch_tasks.get(key) is current_task: - self._completion_notification_batch_tasks.pop(key, None) - if not entries: - return - if len(entries) == 1: - synth_text = entries[0][0] - else: - synth_text = self._format_coalesced_process_completions(entries) - - # A duplicate primary can legitimately return None from the lifecycle dedupe seam; try the - # next batch identity so a fresh sibling is never discarded with that duplicate. - delivered = None - for _text, candidate_evt, _future in entries: - delivered = await self._deliver_completion_notification( - synth_text, candidate_evt, - ) - if delivered is not None: - break - if delivered is True and len(entries) > 1: - self._record_coalesced_completion_siblings( - [evt for _text, evt, _future in entries] - ) - except asyncio.CancelledError: - # Shutdown may cancel us mid fan-in or while adapter delivery is blocked: recover entries not - # yet detached and resolve every waiter as retryable before adapters are torn down. - delivered = False - if not entries: - entries = self._completion_notification_batches.pop(key, []) - raise - except Exception: - logger.exception("Coalesced process completion delivery failed") - delivered = False - finally: - # Never strand watcher futures when formatting, delivery, or cancellation interrupts a batch: - # False follows the existing watcher retry path; None remains the ordinary dedupe result. - for _text, _evt, future in entries: - if not future.done(): - future.set_result(delivered) - # Do not remove a newer flush task that reused the same route key. - if self._completion_notification_batch_tasks.get(key) is current_task: - self._completion_notification_batch_tasks.pop(key, None) - - async def _cancel_process_completion_batch_tasks(self) -> None: - """Settle pending completion batches before adapter teardown.""" - self._completion_notification_batches_stopping = True - tasks = { - task - for task in getattr( - self, "_completion_notification_batch_flush_tasks", set() - ) - if not task.done() - } - for task in tasks: - task.cancel() - if tasks: - await asyncio.gather(*tasks, return_exceptions=True) - - # Defensive cleanup for an orphaned queue with no live flush task. - batches = getattr(self, "_completion_notification_batches", {}) - for entries in batches.values(): - for _text, _evt, future in entries: - if not future.done(): - future.set_result(False) - batches.clear() - getattr(self, "_completion_notification_batch_tasks", {}).clear() - getattr(self, "_completion_notification_batch_flush_tasks", set()).clear() - - async def _enqueue_process_completion_notification( - self, synth_text: str, evt: dict, - ) -> Optional[bool]: - """Fan in concurrent process completions that share one conversation.""" - # Some unit tests construct GatewayRunner with object.__new__. Keep the - # batching seam lazy so those focused lifecycle tests remain valid. - if not hasattr(self, "_completion_notification_batches"): - self._completion_notification_batches = {} - if not hasattr(self, "_completion_notification_batch_tasks"): - self._completion_notification_batch_tasks = {} - if not hasattr(self, "_completion_notification_batch_flush_tasks"): - self._completion_notification_batch_flush_tasks = set() - if not hasattr(self, "_completion_notification_batch_window"): - self._completion_notification_batch_window = 0.1 - if not hasattr(self, "_completion_notification_batches_stopping"): - self._completion_notification_batches_stopping = False - - if self._completion_notification_batches_stopping: - return False - - key = self._completion_notification_batch_key(evt) - future = asyncio.get_running_loop().create_future() - self._completion_notification_batches.setdefault(key, []).append( - (synth_text, evt, future) - ) - if key not in self._completion_notification_batch_tasks: - task = asyncio.create_task( - self._flush_process_completion_batch(key) - ) - self._completion_notification_batch_tasks[key] = task - # Keep the flush alive under the gateway's normal lifecycle accounting; runners built via - # object.__new__ (focused tests) lazily receive the same ownership set. - if not hasattr(self, "_background_tasks"): - self._background_tasks = set() - self._background_tasks.add(task) - self._completion_notification_batch_flush_tasks.add(task) - task.add_done_callback(self._background_tasks.discard) - task.add_done_callback( - self._completion_notification_batch_flush_tasks.discard - ) - return await future - - def _enrich_async_delegation_routing(self, evt: dict) -> None: - """Fill platform/chat_id/thread_id/chat_type on an async-delegation event. - - Such events only carry ``session_key`` (the daemon worker lacks per-message routing - metadata). Best-effort: a CLI-origin event (empty session_key) is left as-is and won't route. - """ - if evt.get("platform"): - return # already enriched - parsed = _parse_session_key(evt.get("session_key", "") or "") - if not parsed: - return - evt["platform"] = parsed.get("platform", "") - evt["chat_type"] = parsed.get("chat_type", "") - evt["chat_id"] = parsed.get("chat_id", "") - if parsed.get("thread_id"): - evt["thread_id"] = parsed["thread_id"] - - @staticmethod - def _async_delegation_group_key(evt: dict) -> tuple[str, ...]: - """Return the async-completion coalescing key: originating session, parent session, route.""" - return tuple(str(evt.get(field) or "") for field in ( - "session_key", - "parent_session_id", - "platform", - "chat_type", - "chat_id", - "thread_id", - "user_id", - )) - - @staticmethod - def _format_coalesced_async_delegations(blocks: list[str]) -> str: - """Join per-delegation formatted blocks into one consolidated turn.""" - header = ( - f"[IMPORTANT: {len(blocks)} background subagent delegations " - "completed for this session. Treat these results as one " - "completion batch and send at most one consolidated user-facing " - "response. If a result does not change the current conclusion, " - "absorb it silently.]" - ) - return "\n\n".join([header, *blocks]) - - async def _deliver_async_delegation_group( - self, group: list[dict], - ) -> Optional[bool]: - """Deliver a same-session batch of async completions as ONE turn. - - Single-event groups ride the per-event path. Multi-event groups deliver the primary via - ``_deliver_completion_notification`` with consolidated text of every sibling THIS runner - claimed; sibling claims are acked only after adapter acceptance, and siblings claimed by - another consumer are excluded (no double delivery). Returns True after acceptance, False - to requeue the group, None when nothing is deliverable here (retry siblings requeued). - """ - from tools.process_registry import process_registry as _pr - - deliverable: list[tuple[dict, str]] = [] - for evt in group: - synth_text = _format_gateway_process_notification(evt) - if not synth_text: - continue - identity = self._completion_delivery_identity(evt) - if identity is not None: - with self._completion_delivery_lock: - if ( - identity in self._completion_deliveries_inflight - or identity in self._completion_deliveries_delivered - ): - continue - deliverable.append((evt, synth_text)) - - if not deliverable: - return None - if len(deliverable) == 1: - evt, synth_text = deliverable[0] - return await self._deliver_completion_notification(synth_text, evt) - - from tools.async_delegation import ( - claim_event_delivery, - complete_event_delivery, - release_event_delivery, - ) - - primary_evt, primary_text = deliverable[0] - blocks = [primary_text] - siblings: list[tuple[dict, str]] = [] - for evt, synth_text in deliverable[1:]: - claim_id = claim_event_delivery(evt, f"gateway-batch:{id(self)}") - if claim_id is None: - # Another consumer owns this row's delivery; keep its result - # out of our consolidated text so it is never double-injected. - continue - siblings.append((evt, claim_id)) - blocks.append(synth_text) - - if not siblings: - return await self._deliver_completion_notification( - primary_text, primary_evt, - ) - - consolidated = self._format_coalesced_async_delegations(blocks) - delivered: Optional[bool] = False - try: - delivered = await self._deliver_completion_notification( - consolidated, primary_evt, - ) - finally: - if delivered is True: - for evt, claim_id in siblings: - try: - complete_event_delivery(evt, claim_id) - except Exception: - logger.debug( - "Could not acknowledge coalesced durable completion", - exc_info=True, - ) - self._record_coalesced_completion_siblings( - [evt for evt, _claim_id in siblings] - ) - else: - # Not delivered — release every sibling claim so a retry or another consumer can claim it, - # honestly leaving the durable rows pending. - for evt, claim_id in siblings: - try: - release_event_delivery(evt, claim_id) - except Exception: - logger.debug( - "Could not release coalesced durable claim", - exc_info=True, - ) - if delivered is None: - # The primary was dropped/owned elsewhere but the siblings - # still need delivery — requeue just them for the next tick. - for evt, _claim_id in siblings: - _pr.completion_queue.put(evt) - return delivered - - async def _async_delegation_watcher(self, interval: float = 2.0) -> None: - """Drain async-delegation completions and inject them as new turns (IDLE case). - - Background subagents run on the daemon executor with no per-process watcher, so their - completions would otherwise only be seen by the post-turn drain. Ignores non-async events. - """ - await asyncio.sleep(3) # let platforms finish connecting - from tools.process_registry import process_registry as _pr - while self._running: - try: - # Peek for async-delegation events only; watch/completion events belong to other drains, - # so requeue anything that isn't ours. - requeue = [] - async_events = [] - while not _pr.completion_queue.empty(): - try: - evt = _pr.completion_queue.get_nowait() - except Exception: - break - if evt.get("type") == "async_delegation": - async_events.append(evt) - else: - requeue.append(evt) - for evt in requeue: - _pr.completion_queue.put(evt) - # A same-tick drain often carries several completions for the SAME session (a fan-out - # finishing together); delivering each individually floods it with N synthetic turns. - # Group by full gateway route + parent session: one consolidated turn per group. - groups: dict[tuple[str, ...], list[dict]] = {} - group_order: list[tuple[str, ...]] = [] - for evt in async_events: - self._enrich_async_delegation_routing(evt) - key = self._async_delegation_group_key(evt) - if key not in groups: - groups[key] = [] - group_order.append(key) - groups[key].append(evt) - for key in group_order: - group = groups[key] - try: - delivered = await self._deliver_async_delegation_group(group) - if delivered is False: - for evt in group: - _pr.completion_queue.put(evt) - except Exception as e: - for evt in group: - _pr.completion_queue.put(evt) - logger.error("Async delegation injection error: %s", e) - except Exception as e: - logger.debug("Async delegation watcher error: %s", e) - await asyncio.sleep(interval) - - async def _run_process_watcher(self, watcher: dict) -> None: - """Periodically check a background process and push updates to the user. - - Runs as an asyncio task. Stays silent when nothing changed. Auto-removes when the process - exits or is killed. Notification mode (``display.background_process_notifications``): - concise (default, one-line; failures append output tail) / all (running updates + final - raw output) / result (final raw only) / error (final raw only if exit != 0) / off. - """ - from tools.process_registry import process_registry - - session_id = watcher["session_id"] - interval = watcher["check_interval"] - session_key = watcher.get("session_key", "") - platform_name = watcher.get("platform", "") - chat_id = watcher.get("chat_id", "") - thread_id = watcher.get("thread_id", "") - user_id = watcher.get("user_id", "") - user_name = watcher.get("user_name", "") - message_id = str(watcher.get("message_id") or "").strip() or None - agent_notify = watcher.get("notify_on_complete", False) - notify_mode = self._load_background_notifications_mode() - - logger.debug("Process watcher started: %s (every %ss, notify=%s, agent_notify=%s)", - session_id, interval, notify_mode, agent_notify) - - if notify_mode == "off" and not agent_notify: - # Still wait for the process to exit so we can log it, but don't - # push any messages to the user. - while True: - await asyncio.sleep(interval) - session = process_registry.get(session_id) - if session is None or session.exited: - break - logger.debug("Process watcher ended (silent): %s", session_id) - return - - last_output_len = 0 - while True: - await asyncio.sleep(interval) - - session = process_registry.get(session_id) - if session is None: - break - - current_output_len = len(session.output_buffer) - has_new_output = current_output_len > last_output_len - last_output_len = current_output_len - - if session.exited: - # Agent-triggered completion: inject a synthetic message unless the agent already consumed - # the result via wait/log. poll() is read-only and deliberately does NOT mark consumed — - # a status check must not suppress this delivery turn. - from tools.process_registry import format_process_notification, process_registry as _pr_check - if agent_notify and not _pr_check.is_completion_consumed(session_id): - from agent.redact import redact_terminal_output - from tools.ansi_strip import strip_ansi - _command = getattr(session, "command", "") or "" - _raw = strip_ansi(session.output_buffer) if session.output_buffer else "" - _raw = redact_terminal_output(_raw, _command) - _command = _redact_gateway_user_facing_secrets(_command) - # Truncate on line boundaries (never start mid-line): keep the last ~2000 chars - # snapped to the preceding newline, prepending a marker when output was cut. - _LIMIT = 2000 - if len(_raw) > _LIMIT: - _tail = _raw[-_LIMIT:] - _nl = _tail.find("\n") - _tail = _tail[_nl + 1:] if _nl != -1 else _tail - _out = f"[… output truncated — showing last {len(_tail)} chars]\n{_tail}" - else: - _out = _raw - _out = _redact_gateway_user_facing_secrets(_out) - completion_evt = { - "type": "completion", - "session_id": session_id, - "session_key": session_key, - "platform": platform_name, - "chat_type": watcher.get("chat_type", ""), - "chat_id": chat_id, - "thread_id": thread_id, - "user_id": user_id, - "user_name": user_name, - "message_id": message_id, - "started_at": getattr(session, "started_at", None), - "command": _command, - "exit_code": session.exit_code, - "completion_reason": getattr(session, "completion_reason", "exited"), - "termination_source": getattr(session, "termination_source", ""), - "output": _out, - # Spawning conversation's session-db id (stamped in terminal_tool); lets delivery - # pre-flight drop this completion if the user closed that session (/new) first. - "parent_session_id": ( - watcher.get("parent_session_id") - or getattr(session, "parent_session_id", "") - or "" - ), - } - synth_text = format_process_notification(completion_evt) - if not synth_text: - break - delivered = await self._enqueue_process_completion_notification( - synth_text, completion_evt, - ) - if delivered is False: - # The process remains terminal; retry after failed - # adapter injection instead of suppressing the result. - continue - break - - # Normal text-only notification. Skip when the agent already consumed this completion via - # wait/log (output returned inline) — the raw "finished" message would be a duplicate. - # The agent_notify skip FALLS THROUGH here, hence this check. poll() is read-only. - if _pr_check.is_completion_consumed(session_id): - logger.debug( - "Process watcher: completion for %s already consumed " - "via wait/log — skipping raw notification (#65379)", - session_id, - ) - break - # Decide whether to notify based on mode - should_notify = ( - notify_mode in {"concise", "all", "result"} - or (notify_mode == "error" and session.exit_code not in {0, None}) - ) - if should_notify: - new_output = session.output_buffer[-1000:] if session.output_buffer else "" - if new_output: - from agent.redact import redact_terminal_output - new_output = redact_terminal_output( - new_output, getattr(session, "command", "") or "" - ) - # redact_terminal_output() is unforced, so it returns raw text when - # security.redact_secrets is off. This goes straight to the platform - # adapter, so it needs the same unconditional floor as agent-notify. - new_output = _redact_gateway_user_facing_secrets(new_output) - if notify_mode == "concise": - _cmd_disp = _redact_gateway_user_facing_secrets( - getattr(session, "command", "") or "" - ) - _started = getattr(session, "started_at", None) - _dur = None - if isinstance(_started, (int, float)): - _dur = max(0.0, time.time() - _started) - message_text = _format_concise_process_notification( - session_id, - _cmd_disp, - session.exit_code, - new_output, - duration_seconds=_dur, - ) - else: - message_text = ( - f"[Background process {session_id} finished with exit code {session.exit_code}~ " - f"Here's the final output:\n{new_output}]" - ) - adapter = None - for p, a in self.adapters.items(): - if p.value == platform_name: - adapter = a - break - if adapter and chat_id: - try: - send_meta = {"thread_id": thread_id} if thread_id else None - await adapter.send( - chat_id, - message_text, - metadata=_non_conversational_metadata(send_meta, platform=platform_name), - ) - except Exception as e: - logger.error("Watcher delivery error: %s", e) - break - - elif has_new_output and notify_mode == "all" and not agent_notify: - # New output available -- deliver status update (only in "all" mode) - # Skip periodic updates for agent_notify watchers (they only care about completion) - new_output = session.output_buffer[-500:] if session.output_buffer else "" - if new_output: - from agent.redact import redact_terminal_output - new_output = redact_terminal_output( - new_output, getattr(session, "command", "") or "" - ) - new_output = _redact_gateway_user_facing_secrets(new_output) - message_text = ( - f"[Background process {session_id} is still running~ " - f"New output:\n{new_output}]" - ) - adapter = None - for p, a in self.adapters.items(): - if p.value == platform_name: - adapter = a - break - if adapter and chat_id: - try: - send_meta = {"thread_id": thread_id} if thread_id else None - await adapter.send( - chat_id, - message_text, - metadata=_non_conversational_metadata(send_meta, platform=platform_name), - ) - except Exception as e: - logger.error("Watcher delivery error: %s", e) - - logger.debug("Process watcher ended: %s", session_id) _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816) @@ -25394,781 +5369,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) _HONCHO_CACHE_BUSTING_MEMO: dict[tuple[str, int | None], dict[str, Any]] = {} - @classmethod - def _empty_honcho_cache_busting_config(cls) -> dict[str, Any]: - return {key: None for key in cls._HONCHO_CACHE_BUSTING_KEYS} - - @classmethod - def _extract_honcho_cache_busting_config(cls) -> dict[str, Any]: - """Extract Honcho identity keys, memoized by honcho.json mtime.""" - try: - from plugins.memory.honcho.client import HonchoClientConfig, resolve_config_path - - path = resolve_config_path() - try: - mtime_ns = path.stat().st_mtime_ns - except OSError: - mtime_ns = None - memo_key = (str(path), mtime_ns) - cached = cls._HONCHO_CACHE_BUSTING_MEMO.get(memo_key) - if cached is not None: - return dict(cached) - - hcfg = HonchoClientConfig.from_global_config(config_path=path) - aliases = hcfg.user_peer_aliases or {} - values = { - "honcho.peer_name": hcfg.peer_name, - "honcho.ai_peer": hcfg.ai_peer, - "honcho.pin_peer_name": bool(hcfg.pin_peer_name), - "honcho.runtime_peer_prefix": hcfg.runtime_peer_prefix or "", - "honcho.user_peer_aliases": sorted(aliases.items()) if isinstance(aliases, dict) else [], - } - cls._HONCHO_CACHE_BUSTING_MEMO = {memo_key: values} - return dict(values) - except Exception: - return cls._empty_honcho_cache_busting_config() - - @classmethod - def _extract_cache_busting_config(cls, user_config: dict | None) -> dict: - """Pull values that must bust the cached agent, as a flat dict keyed by 'section.key'. - - Missing keys / non-dict sections yield None, which still enters the signature ('absent' vs - 'present-and-null' differ). Includes the live tool registry generation: MCP reloads mutate - the registry without touching config.yaml, and cached agents freeze their tool schemas. - """ - out: Dict[str, Any] = {} - cfg = user_config if isinstance(user_config, dict) else {} - for section, key in cls._CACHE_BUSTING_CONFIG_KEYS: - section_val = cfg.get(section) - if section == "checkpoints" and isinstance(section_val, bool): - # Preserve legacy ``checkpoints: true`` behavior. A live - # toggle must still rebuild the cached agent. - out[f"{section}.{key}"] = section_val if key == "enabled" else None - elif isinstance(section_val, dict): - out[f"{section}.{key}"] = section_val.get(key) - else: - out[f"{section}.{key}"] = None - try: - from tools.registry import registry - - out["tools.registry_generation"] = getattr(registry, "_generation", None) - except Exception: - out["tools.registry_generation"] = None - - # Honcho identity-mapping keys live in honcho.json, not user_config. - # Only read that file when Honcho is the active memory provider. - provider = cfg_get(cfg, "memory", "provider") - if isinstance(provider, str) and provider.lower() == "honcho": - out.update(cls._extract_honcho_cache_busting_config()) - else: - out.update(cls._empty_honcho_cache_busting_config()) - - return out - - @staticmethod - def _agent_config_signature( - model: str, - runtime: dict, - enabled_toolsets: list, - ephemeral_prompt: str, - cache_keys: dict | None = None, - user_id: str | None = None, - user_id_alt: str | None = None, - skip_context_files: bool = False, - ) -> str: - """Compute a stable string key from agent config values. - - Signature change → cached AIAgent rebuilt; unchanged → reused (frozen prompt + schemas for - cache hits). Callers pass ``_extract_cache_busting_config(user_config)`` so config.yaml - edits apply on the next message. ``user_id`` / ``user_id_alt`` participate because Honcho - freezes them into ``HonchoSessionManager`` at init; omitting them in shared-thread keys - (``thread_sessions_per_user=False``) would attribute one user's messages to another's peer. - """ - import hashlib, json as _j - - # Fingerprint the FULL credential, not a short prefix: OAuth/JWT-style tokens often share a - # common prefix (e.g. "eyJhbGci"), so a prefix would give false cache hits across auth switches. - _api_key = str(runtime.get("api_key", "") or "") - _api_key_fingerprint = hashlib.sha256(_api_key.encode()).hexdigest() if _api_key else "" - - _cache_keys_sorted = sorted((cache_keys or {}).items()) - - blob = _j.dumps( - [ - model, - _api_key_fingerprint, - runtime.get("base_url", ""), - runtime.get("provider", ""), - runtime.get("requested_provider", ""), - runtime.get("api_mode", ""), - sorted((runtime.get("capabilities") or {}).items()), - sorted(enabled_toolsets) if enabled_toolsets else [], - # reasoning_config excluded — it's set per-message on the - # cached agent and doesn't affect system prompt or tools. - ephemeral_prompt or "", - _cache_keys_sorted, - str(user_id or ""), - str(user_id_alt or ""), - # skip_context_files changes the agent's frozen system prompt (context files in vs out): - # a toggled edit must rebuild the cached agent, not silently reuse it. - bool(skip_context_files), - ], - sort_keys=True, - default=str, - ) - return hashlib.sha256(blob.encode()).hexdigest()[:16] - - def _rehydrate_session_model_override(self, session_key: str) -> None: - """Lazily restore a persisted /model override after a gateway restart. - - ``_session_model_overrides`` is in-memory only. Non-secret parts (model/provider/base_url) - are written through on /model (cleared on /new) and read back here on first use; api_key - is never persisted and is re-resolved. No-op when an in-memory override or nothing exists. - """ - _rehydrate_state = self._peek_session_state(session_key) - if ( - _rehydrate_state is not None - and _rehydrate_state.conversation.model_override is not None - ): - return - store = getattr(self, "session_store", None) - if store is None: - return - try: - persisted = store.get_model_override(session_key) - except Exception: - logger.debug( - "Failed to read persisted session model override", exc_info=True - ) - return - if not persisted: - return - override: Dict[str, Any] = { - "model": persisted.get("model"), - "provider": persisted.get("provider"), - "base_url": persisted.get("base_url"), - } - provider = persisted.get("provider") - if provider: - # Re-resolve credentials for the persisted provider. On failure (e.g. credentials - # removed since the switch) keep the credential-less override — - # _resolve_session_agent_runtime falls back to env resolution and layers model/provider. - try: - runtime = _resolve_runtime_agent_kwargs_for_provider(provider) - override["api_key"] = runtime.get("api_key") - override["api_mode"] = runtime.get("api_mode") - override["credential_pool"] = runtime.get("credential_pool") - override["request_overrides"] = dict( - runtime.get("request_overrides") or {} - ) - override["requested_provider"] = runtime.get("requested_provider") - override["capabilities"] = dict(runtime.get("capabilities") or {}) - override["max_tokens"] = runtime.get("max_tokens") - if not override.get("base_url"): - override["base_url"] = runtime.get("base_url") - except Exception: - logger.debug( - "Credential re-resolution failed for persisted override " - "(provider=%s); using credential-less override", - provider, exc_info=True, - ) - self._session_state(session_key).conversation.model_override = override - logger.info( - "Rehydrated persisted /model override for session=%s: model=%s provider=%s", - session_key, override.get("model"), provider or "", - ) - - def _apply_session_model_override( - self, session_key: str, model: str, runtime_kwargs: dict - ) -> tuple: - """Apply /model session overrides if present, returning (model, runtime_kwargs). - - Overrides take precedence over config.yaml defaults so the switched model is actually used; - ``None`` fields are skipped so partial overrides don't clobber valid defaults. - """ - _apply_state = self._peek_session_state(session_key) - override = _apply_state.conversation.model_override if _apply_state else None - if not override: - return model, runtime_kwargs - model = override.get("model", model) - for key in ( - "provider", - "requested_provider", - "api_key", - "base_url", - "api_mode", - "credential_pool", - "capabilities", - "max_tokens", - ): - val = override.get(key) - if val is not None: - runtime_kwargs[key] = val - # request_overrides reflects the switched-to provider; apply whenever the override recorded - # it (even as None) so switching to a provider without configured overrides clears a stale - # value left by the default provider's runtime resolution. - if "request_overrides" in override: - override_request_overrides = override.get("request_overrides") - if isinstance(override_request_overrides, dict) and override_request_overrides: - runtime_kwargs["request_overrides"] = dict(override_request_overrides) - else: - runtime_kwargs["request_overrides"] = override_request_overrides - if ( - runtime_kwargs.get("api_key") - and runtime_kwargs.get("credential_pool") is None - and override.get("provider") - ): - runtime_kwargs["credential_pool"] = _credential_pool_for_provider( - override.get("provider") - ) - return model, runtime_kwargs - - def _snapshot_session_model_override(self, session_key: str) -> dict: - """Capture a gateway session override before a one-turn switch.""" - _snap_state = self._peek_session_state(session_key) - override = _snap_state.conversation.model_override if _snap_state else None - return { - "had_override": override is not None, - "override": dict(override) if override is not None else None, - } - - def _restore_session_model_override(self, session_key: str, snapshot: dict) -> None: - """Restore the session override captured before a one-turn switch.""" - if not session_key: - return - if snapshot.get("had_override"): - self._session_state(session_key).conversation.model_override = dict( - snapshot.get("override") or {} - ) - else: - _rst_state = self._peek_session_state(session_key) - if _rst_state is not None: - _rst_state.conversation.model_override = None - self._evict_cached_agent(session_key) - - def _is_intentional_model_switch(self, session_key: str, agent_model: str) -> bool: - """Return True if *agent_model* matches an active /model session override.""" - _ims_state = self._peek_session_state(session_key) - override = _ims_state.conversation.model_override if _ims_state else None - return override is not None and override.get("model") == agent_model - - def _release_running_agent_state( - self, - session_key: str, - *, - run_generation: Optional[int] = None, - ) -> bool: - """Pop ALL per-running-agent state entries for ``session_key``; True when cleared. - - Call at every site that ends a running turn, whatever the cause. State that PERSISTS - across turns (model overrides, voice mode, pending approvals, update prompt) is NOT - touched. With ``run_generation``, only clear if that generation is still current, so a - stale async unwind bumped by /stop or /new cannot clobber a newer run (returns False). - """ - if not session_key: - return False - if run_generation is not None and not self._is_session_run_current( - session_key, run_generation - ): - return False - state = self._peek_session_state(session_key) - if state is not None: - lease = state.turn.lease - if lease is not None: - try: - lease.release() - except Exception: - logger.debug( - "Failed to release active session slot", exc_info=True - ) - # One structured reset instead of a drifting pop-list. Turn-lease tokens are deliberately NOT - # cleared here — _release_turn_lease owns them. - state.turn.clear() - # Turn boundary: a running-agent slot was just released; persist the new (lower) in-flight count - # so the dashboard readout stays current. Preserves gateway_state (see _persist_active_agents). - self._persist_active_agents() - return True - - def _release_turn_lease(self, session_key: str, run_generation: int) -> bool: - """Release the turn lease acquired by (``session_key``, ``run_generation``). - - Token map is keyed by (routing key, run generation), so a stale unwind pops only ITS token - and the registry's identity check refuses it if a newer turn holds the lease. Idempotent. - """ - if not session_key: - return False - registry = getattr(self, "_turn_leases", None) - state = self._peek_session_state(session_key) - if state is None or registry is None: - return False - turn = state.turn - if turn.lease_token is None or turn.lease_generation != run_generation: - return False - token = turn.lease_token - turn.lease_token = None - turn.lease_generation = None - try: - return registry.release(token) - except Exception: - logger.debug("Failed to release turn lease", exc_info=True) - return False - - def _rebind_turn_lease( - self, session_key: str, run_generation: int, new_session_id: str - ) -> bool: - """Follow a mid-turn session_id rotation with the held turn lease. - - Compression can rotate ``session_entry.session_id`` mid-turn; the flush targets the NEW id, - so the serialization boundary must follow or an alias key resolving the new id could start - a concurrent turn the lease never sees. Call at every mid-turn reassignment; no-op if no token. - """ - if not session_key or not new_session_id: - return False - registry = getattr(self, "_turn_leases", None) - state = self._peek_session_state(session_key) - if state is None or registry is None: - return False - turn = state.turn - if turn.lease_token is None or turn.lease_generation != run_generation: - return False - try: - return registry.rebind(turn.lease_token, new_session_id) - except Exception: - logger.debug("Failed to rebind turn lease", exc_info=True) - return False - - def _clear_conversation_scope(self, session_key: str, *, reason: str) -> None: - """Clear ALL conversation-scoped per-session state for ``session_key``. - - THE single conversation-boundary funnel — call this and nothing else at /new, /resume, - auto-reset (idle/daily/suspended), expiry finalization and compression-exhausted reset. - New conversation-scoped dicts go in _CONVERSATION_SCOPED_STATE so every boundary picks - them up (hand-copied pop-lists drifted). Turn-scoped state (_running_agents/_ts, slot - leases, turn-lease tokens) is owned by _release_running_agent_state and NOT cleared. Idle - agent-cache eviction is NOT a boundary (a resumed turn rebuilds from these). getattr-guarded. - """ - if not session_key: - return - # Structural clear: every conversation-scoped field resets in one - # call — no per-attribute pop-list to drift. - state = self._peek_session_state(session_key) - if state is not None: - state.conversation.clear() - # Legacy plain-dict stores still in _CONVERSATION_SCOPED_STATE (not yet folded into - # SessionState), e.g. _pending_model_notes. SessionState-backed names resolve to MutableMapping - # views (not dict), so the isinstance(dict) guard skips them — already handled above. - for attr in _CONVERSATION_SCOPED_STATE: - store = getattr(self, attr, None) - if isinstance(store, dict): - store.pop(session_key, None) - self._clear_session_boundary_security_state(session_key) - logger.debug( - "Cleared conversation scope for %s (%s)", session_key, reason - ) - - def _clear_session_boundary_security_state(self, session_key: str) -> None: - """Clear per-session control state that must not survive a boundary switch.""" - if not session_key: - return - - pending_skills_reload_notes = getattr( - self, "_pending_skills_reload_notes", None - ) - if isinstance(pending_skills_reload_notes, dict): - pending_skills_reload_notes.pop(session_key, None) - - _sec_state = self._peek_session_state(session_key) - if _sec_state is not None: - _sec_state.persistent.approvals = None - _sec_state.persistent.update_prompt_pending = False - - try: - from tools import slash_confirm as _slash_confirm_mod - except Exception: - _slash_confirm_mod = None - if _slash_confirm_mod is not None: - try: - _slash_confirm_mod.clear(session_key) - except Exception as e: - logger.debug( - "Failed to clear slash-confirm state for session boundary %s: %s", - session_key, - e, - ) - - try: - from tools.approval import clear_session as _clear_approval_session - except Exception: - return - - try: - _clear_approval_session(session_key) - except Exception as e: - logger.debug( - "Failed to clear approval state for session boundary %s: %s", - session_key, - e, - ) - - def _begin_session_run_generation(self, session_key: str) -> int: - """Claim a fresh, monotonically increasing run generation token for ``session_key``. - - If /stop or /new invalidates the token while the old worker is still unwinding, the late - result is recognized and dropped instead of bleeding into the fresh session. - """ - if not session_key: - return 0 - persistent = self._session_state(session_key).persistent - # Monotonic by design (#28686): incremented here, NEVER reset. - persistent.run_generation = int(persistent.run_generation) + 1 - return persistent.run_generation - - def _invalidate_session_run_generation(self, session_key: str, *, reason: str = "") -> int: - """Invalidate any in-flight run token for ``session_key``.""" - generation = self._begin_session_run_generation(session_key) - if reason: - logger.info( - "Invalidated run generation for %s → %d (%s)", - session_key, - generation, - reason, - ) - return generation - - def _is_session_run_current(self, session_key: str, generation: int) -> bool: - """Return True when ``generation`` is still current for ``session_key``.""" - if not session_key: - return True - state = self._peek_session_state(session_key) - current = state.persistent.run_generation if state is not None else 0 - return int(current) == int(generation) - - def _bind_adapter_run_generation( - self, - adapter: Any, - session_key: str, - generation: int | None, - ) -> None: - """Bind a gateway run generation to the adapter's active-session event.""" - if not adapter or not session_key or generation is None: - return - try: - interrupt_event = getattr(adapter, "_active_sessions", {}).get(session_key) - if interrupt_event is not None: - setattr(interrupt_event, "_hermes_run_generation", int(generation)) - except Exception: - pass - - async def _interrupt_and_clear_session( - self, - session_key: str, - source: SessionSource, - *, - interrupt_reason: str, - invalidation_reason: str, - release_running_state: bool = True, - ) -> None: - """Interrupt the current run and clear queued session state consistently.""" - if not session_key: - return - _iac_state = self._peek_session_state(session_key) - running_agent = _iac_state.turn.agent if _iac_state else None - _process_task_id = "" - _process_baseline = None - if running_agent and running_agent is not _AGENT_PENDING_SENTINEL: - request_hard_interrupt(running_agent, interrupt_reason) - _process_task_id = getattr( - running_agent, "_gateway_turn_process_task_id", "" - ) - _process_baseline = getattr( - running_agent, "_gateway_turn_process_baseline", None - ) - # Bump the generation BEFORE scheduling the reap thread and capture the post-bump value: - # task_id is session-scoped, so a replacement turn spawning before the reap runs bumps it - # again and the closure sees a stale generation and skips — the replacement's own baseline - # covers its cleanup, so nothing stays unreaped. - _generation_at_interrupt = self._invalidate_session_run_generation( - session_key, reason=invalidation_reason - ) - if _process_task_id and _process_baseline is not None: - threading.Thread( - target=_reap_gateway_turn_processes, - args=(_process_task_id, _process_baseline), - kwargs={ - "source": "gateway_turn_interrupt", - "is_still_current": lambda: self._is_session_run_current( - session_key, _generation_at_interrupt - ), - }, - name=f"gateway-turn-reaper-{_process_task_id[:12]}", - daemon=True, - ).start() - adapter = self._adapter_for_source(source) - interrupt_session_activity = getattr( - type(adapter), "interrupt_session_activity", None - ) - if adapter and callable(interrupt_session_activity): - metadata = self._thread_metadata_for_source(source) - try: - params = inspect.signature(interrupt_session_activity).parameters - accepts_metadata = "metadata" in params or any( - param.kind is inspect.Parameter.VAR_KEYWORD - for param in params.values() - ) - except (TypeError, ValueError): - accepts_metadata = False - if accepts_metadata: - await adapter.interrupt_session_activity( - session_key, source.chat_id, metadata=metadata - ) - else: - await adapter.interrupt_session_activity(session_key, source.chat_id) - if adapter and hasattr(adapter, "get_pending_message"): - adapter.get_pending_message(session_key) # consume and discard - if _iac_state is not None: - _iac_state.persistent.pending_command_text = None - if release_running_state: - self._release_running_agent_state(session_key) - # Evict the cached agent: ``_interrupt_requested`` is only cleared by the turn finalizer, - # so on a hung/still-draining run the flag survives and silently kills the session's NEXT - # message (interrupted=True, api_calls=0, empty response). Like /new and /model, the next - # message rebuilds from history; the old agent keeps its flag so a hung drain still dies. - self._evict_cached_agent(session_key) - - async def _refresh_agent_cache_message_count( - self, session_key: str, session_id: Optional[str] - ) -> None: - """Re-baseline a cached agent's stored message_count after THIS turn. - - The coherence guard compares on-disk ``message_count`` against the BUILD-time snapshot and - rebuilds on mismatch; without re-baselining after our own rows flush, every turn would - rebuild and destroy prompt caching. Only the count is refreshed (``_sig`` untouched), only - if the same agent is still cached, never when the entry records a different ``session_id`` - (another conversation's baseline). DB errors leave the snapshot as-is (one spare rebuild). - """ - if self._session_db is None or not session_id: - return - _cache_lock = getattr(self, "_agent_cache_lock", None) - _cache = getattr(self, "_agent_cache", None) - if not _cache_lock or _cache is None: - return - try: - _sess_row = await self._session_db.get_session(session_id) - _live = _sess_row.get("message_count", 0) if _sess_row else None - except Exception: - return - if _live is None: - return - with _cache_lock: - cached = _cache.get(session_key) - # Only re-baseline a live 3-tuple entry; skip pending sentinels, legacy 2-tuples (they opt - # out of the guard), and entries evicted/rebuilt mid-turn. - if ( - isinstance(cached, tuple) - and len(cached) > 2 - and cached[0] is not _AGENT_PENDING_SENTINEL - ): - # A snapshot taken for a different session_id (same session_key, different conversation) - # belongs to a different DB row — leave it alone. - _snapshot_sid = cached[3] if len(cached) > 3 else None - if _snapshot_sid is not None and _snapshot_sid != session_id: - return - if cached[2] != _live: - if _snapshot_sid is None: - # Legacy 3-tuple: preserve the 3-element shape for callers indexing ``cached[2]``. - _cache[session_key] = (cached[0], cached[1], _live) - else: - _cache[session_key] = ( - cached[0], cached[1], _live, _snapshot_sid, - ) - - def _set_pending_turn_sidecar_notes(self, session_key: str, notes: List[str]) -> None: - """Stage per-turn must-deliver notes for the next agent run (one-shot).""" - if not session_key or not notes: - return - self._session_state(session_key).conversation.sidecar_notes = list(notes) - - def _consume_pending_turn_sidecar_notes(self, session_key: str) -> List[str]: - if not session_key: - return [] - state = self._peek_session_state(session_key) - if state is None: - return [] - staged = state.conversation.sidecar_notes - state.conversation.sidecar_notes = [] - return list(staged) if isinstance(staged, list) else [] - - def _voice_channel_sidecar_note(self, event, source: SessionSource, session_key: str) -> Optional[str]: - """Return a ``[Voice channel now: ...]`` note when VC state changed. - - Unchanged state returns ``None`` so per-turn member/speaking churn can't touch the prompt. - """ - if source.platform != Platform.DISCORD: - return None - adapter = self.adapters.get(Platform.DISCORD) - guild_id = self._get_guild_id(event) - if not (guild_id and adapter and hasattr(adapter, "get_voice_channel_context")): - return None - try: - vc_now = adapter.get_voice_channel_context(guild_id) or "" - except Exception: - logger.debug("voice-channel context read failed", exc_info=True) - return None - vc_prev = None - if session_key: - _vc_state = self._session_state(session_key) - vc_prev = _vc_state.conversation.vc_last - _vc_state.conversation.vc_last = vc_now - if vc_now == (vc_prev if vc_prev is not None else ""): - return None - if not vc_now: - return "[Voice channel now: not connected to a voice channel]" - return f"[Voice channel now: {vc_now}]" - - def _pinned_session_context_prompt( - self, context, redact_pii: bool, session_key: Optional[str] - ) -> str: - """Return the session-context prompt, pinned per session. - - Key hit → pinned bytes reused VERBATIM (immune to renderer nondeterminism); key miss → - re-render ``build_session_context_prompt`` and re-pin (rename, topic edit, /sethome, ...). - """ - _eph_key = self._ephemeral_change_key(context, redact_pii) - _eph_pin = None - if session_key: - _pin_state = self._peek_session_state(session_key) - _eph_pin = _pin_state.conversation.ephemeral_pin if _pin_state else None - if _eph_pin is not None and _eph_pin[0] == _eph_key: - return _eph_pin[1] - text = build_session_context_prompt(context, redact_pii=redact_pii) - if session_key: - self._session_state(session_key).conversation.ephemeral_pin = ( - _eph_key, - text, - ) - return text - - @staticmethod - def _ephemeral_change_key(context, redact_pii: bool) -> str: - """Hash the exact inputs ``build_session_context_prompt`` renders. - - Invariant (tests/gateway/test_prompt_tail_freeze.py): any input whose change alters the - rendered bytes MUST appear here — omission means a stale pinned prompt; extras only re-render. - """ - import hashlib - - src = context.source - platform = src.platform.value if src.platform else "" - - discord_ids: tuple = () - discord_tools = "" - if src.platform == Platform.DISCORD: - from gateway.session import _discord_tools_loaded - - discord_tools = "1" if _discord_tools_loaded() else "0" - discord_ids = ( - str(src.guild_id or ""), - str(src.parent_chat_id or ""), - str(src.thread_id or ""), - str(src.chat_id or ""), - # Only PRESENCE is rendered (the id itself arrives per-turn in the user message) — - # keying on the value would re-render every message for zero byte change. - "1" if src.message_id else "0", - ) - - # Slack's capability-aware platform note is gated on _slack_tools_loaded() — the gate state must - # be in the key (same parity contract as the Discord gate above) so a config / MCP-registration - # flip re-renders once instead of serving a stale pinned note for the rest of the session. - slack_tools = "" - if src.platform == Platform.SLACK: - from gateway.session import _slack_tools_loaded - - slack_tools = "1" if _slack_tools_loaded() else "0" - - try: - from hermes_constants import display_hermes_home - - home_display = str(display_hermes_home()) - except Exception: - home_display = "" - - key_tuple = ( - platform, - str(src.chat_id or ""), - str(src.thread_id or ""), - str(src.chat_type or ""), - str(src.chat_name or ""), - str(src.chat_topic or ""), - str(src.user_name or ""), - str(src.user_id or ""), - str(getattr(src, "profile", None) or ""), - bool(context.shared_multi_user_session), - discord_ids, - discord_tools, - slack_tools, - tuple(p.value for p in context.connected_platforms), - tuple( - ( - p.value, - str(getattr(hc, "name", "") or ""), - str(getattr(hc, "chat_id", "") or ""), - ) - for p, hc in context.home_channels.items() - ), - bool(redact_pii), - home_display, - ) - return hashlib.sha256(repr(key_tuple).encode("utf-8")).hexdigest() - - def _evict_cached_agent(self, session_key: str) -> None: - """Remove a cached agent for a session (called on /new, /model, etc). - - Also soft-releases the evicted agent's LLM client pool (``release_clients()``): AIAgent - holds reference cycles that delay collection, so without it gateway RSS grows across /new. - Soft = frees clients and per-turn child subagents but PRESERVES the session's terminal - sandbox, browser daemon and bg processes (keyed on task_id) since the session may resume. - True boundaries (/new) call ``_cleanup_agent_resources`` first (release is idempotent). - Cleanup runs on a daemon thread so ``_agent_cache_lock`` never spans slow socket teardown. - """ - # Prompt-stability state rides the agent-cache lifecycle: a fresh agent must re-render its - # session-context bytes (the pin) and re-see the current voice-channel state once. - _evict_state = self._peek_session_state(session_key) - if _evict_state is not None: - _evict_state.conversation.ephemeral_pin = None - _evict_state.conversation.vc_last = None - - _lock = getattr(self, "_agent_cache_lock", None) - evicted = None - if _lock: - with _lock: - evicted = self._agent_cache.pop(session_key, None) - else: - _cache = getattr(self, "_agent_cache", None) - if _cache is not None: - evicted = _cache.pop(session_key, None) - - agent = evicted[0] if isinstance(evicted, tuple) and evicted else evicted - if agent is None or agent is _AGENT_PENDING_SENTINEL: - return - - # Don't tear down an agent that's actively mid-turn — its client, - # sandbox and child subagents are in use by the running request. - running_ids = self._running_agent_ids() - if id(agent) in running_ids: - return - - try: - threading.Thread( - target=self._release_evicted_agent_soft, - args=(agent,), - daemon=True, - name=f"agent-evict-{str(session_key)[:24]}", - ).start() - except Exception: - # If we can't spawn a thread (interpreter shutdown), release - # inline as a best-effort fallback. - with suppress(Exception): - self._release_evicted_agent_soft(agent) @staticmethod def _init_cached_agent_for_turn(agent: Any, interrupt_depth: int) -> None: @@ -26191,751 +5391,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew agent._last_flushed_db_idx = 0 agent._api_call_count = 0 - def _commit_memory_before_soft_evict(self, agent: Any, key: str) -> None: - """Fire on_session_end extraction before soft-evicting a live agent. - - Soft eviction keeps the session resumable and does NOT fire ``on_session_end`` — that is - ``_session_expiry_watcher``'s job at true expiry. But the watcher tears down whatever it - finds in ``_agent_cache``; if the LRU cap soft-evicts first, memory providers never see the - transcript. So commit extraction here via ``commit_memory_session`` (no teardown). Only for - finalizable sessions — ``mode == "none"`` never finalizes. Best-effort: failures swallowed. - """ - if agent is None or not hasattr(agent, "commit_memory_session"): - return - if getattr(agent, "_memory_manager", None) is None: - return # no external memory provider — nothing to commit - try: - _store = getattr(self, "session_store", None) - if _store is None: - return - _store._ensure_loaded() - entry = _store._entries.get(key) - if entry is None: - return - # Compensate only when the watcher would expect this agent at expiry (finite policy, not yet - # expired). Expired sessions are torn down by the watcher; mode="none" is never finalized. - if not _store.is_session_finalizable(entry): - return - if _store._is_session_expired(entry): - return - messages = getattr(agent, "_session_messages", None) - agent.commit_memory_session(messages if isinstance(messages, list) else None) - logger.debug( - "Committed on_session_end extraction before soft-evicting " - "finalizable session=%s (cache pressure, pre-expiry)", key, - ) - except Exception as _e: - logger.debug("Pre-evict memory commit failed for %s: %s", key, _e) - - def _commit_then_release_soft(self, agent: Any, key: str) -> None: - """Commit end-of-session memory (if warranted), then soft-release. - - Runs on the daemon eviction thread so neither blocks the caller's held cache lock. Order - matters: commit needs the live memory manager before ``release_clients`` drops the buffer. - """ - self._commit_memory_before_soft_evict(agent, key) - self._release_evicted_agent_soft(agent) - - def _release_evicted_agent_soft(self, agent: Any) -> None: - """Soft cleanup for cache-evicted agents — preserves session tool state. - - Unlike _cleanup_agent_resources (full teardown), an evicted session may resume, so its - terminal sandbox, browser daemon and bg processes must outlive the AIAgent instance. - """ - if agent is None: - return - try: - if hasattr(agent, "release_clients"): - agent.release_clients() - else: - # Older agent instance (shouldn't happen in practice) — - # fall back to the legacy full-close path. - self._cleanup_agent_resources(agent) - except Exception: - pass - # Free conversation history — tens of MB of tool output on heavy 100+-tool-call sessions. - # release_clients() preserves session tool state for resume, but the message list is rebuilt from - # persisted session JSON on the next turn, so dropping it here is safe. - if hasattr(agent, "_session_messages"): - agent._session_messages = [] - # _db_flush_scan_prefix (run_agent.py, stamped on every successful flush) is a shallow copy - # sharing every message dict of the flushed transcript, so leaving it pins the multi-MB strings - # this eviction frees. Pressure-evictable agents have flushed by definition, so it's populated. - if hasattr(agent, "_db_flush_scan_prefix"): - agent._db_flush_scan_prefix = None - - def _agent_cache_bounds(self): - """Operator-configured agent-cache bounds, resolved once per process. - - Resolved lazily rather than in ``__init__`` so it also works for the - ``__new__``-constructed runners used by tests and by the slash-command mixin. - """ - bounds = getattr(self, "_agent_cache_bounds_cache", None) - if bounds is None: - from gateway.agent_cache_pressure import resolve_agent_cache_bounds - - try: - bounds = resolve_agent_cache_bounds(_load_gateway_config()) - except Exception as _e: - logger.debug("Agent cache bounds config read failed: %s", _e) - # Resolve from an empty config rather than bare AgentCacheBounds(): the dataclass default - # has memory_high_mb=None (pressure pass OFF) but an *absent* section means "auto" — a - # transient config read failure must not permanently disable the OOM valve. - bounds = resolve_agent_cache_bounds({}) - self._agent_cache_bounds_cache = bounds - return bounds - - def _agent_cache_cap(self) -> int: - """Effective LRU cap — the configured override, else the default.""" - configured = self._agent_cache_bounds().max_size - return configured if configured else _AGENT_CACHE_MAX_SIZE - - def _agent_cache_idle_ttl(self) -> float: - """Effective idle TTL in seconds — configured override, else default.""" - configured = self._agent_cache_bounds().idle_ttl_secs - return configured if configured else _AGENT_CACHE_IDLE_TTL_SECS - - def _sweep_agent_cache_under_pressure(self) -> int: - """Shed cached transcripts once the gateway heap nears its budget; returns count evicted. - - The LRU cap counts entries and the idle sweep counts seconds; neither knows one cached agent - pins a full ``_session_messages`` transcript (tens of MB). Warm and finalizable agents are - never swept, so RSS climbs until the cgroup throttles. Above the anonymous-RSS budget this - soft-evicts LRU agents (transcript rebuilt from the persisted session next turn). Never - touched: agents mid-turn, the most recently used sessions, and transcripts not yet on disk. - """ - from gateway.agent_cache_pressure import ( - plan_pressure_evictions, - read_anon_rss_mb, - transcript_persistence_caught_up, - ) - - bounds = self._agent_cache_bounds() - if not bounds.memory_high_mb: - return 0 - _cache = getattr(self, "_agent_cache", None) - _lock = getattr(self, "_agent_cache_lock", None) - if not _cache or _lock is None: - # Nothing cached — whatever is using the heap, it isn't us, and - # warning about it every tick would point at the wrong subsystem. - return 0 - - rss_mb = read_anon_rss_mb() - if rss_mb is None or rss_mb < bounds.memory_high_mb: - return 0 - - running_ids = self._running_agent_ids() - - def _is_evictable(key: str, agent: Any) -> bool: - if agent is None or agent is _AGENT_PENDING_SENTINEL: - return False - if id(agent) in running_ids: - return False - return transcript_persistence_caught_up(agent) - - with _lock: - ordered = [ - (key, entry[0] if isinstance(entry, tuple) and entry else entry) - for key, entry in _cache.items() - ] - plan = plan_pressure_evictions( - ordered, - is_evictable=_is_evictable, - max_evictions=bounds.max_evictions_per_pass, - protect_recent=bounds.protect_recent, - ) - for key, _ in plan: - _cache.pop(key, None) - - if not plan: - _mid_turn = sum(1 for _, a in ordered if a is not None and id(a) in running_ids) - _unflushed = sum( - 1 - for _, a in ordered - if a is not None - and a is not _AGENT_PENDING_SENTINEL - and id(a) not in running_ids - and not transcript_persistence_caught_up(a) - ) - logger.warning( - "Agent cache pressure: anon RSS %dMB over budget %dMB but no " - "evictable session (%d cached, %d mid-turn, %d blocked on " - "un-flushed persistence)%s", - rss_mb, bounds.memory_high_mb, len(ordered), _mid_turn, _unflushed, - ( - " — transcripts are not reaching the session DB " - "(session persistence disabled or failing?); the memory " - "valve cannot shed sessions until they persist." - if _unflushed and not _mid_turn - else " — memory will keep climbing until those turns finish." - ), - ) - return 0 - - evicted_count = len(plan) - logger.warning( - "Agent cache pressure: anon RSS %dMB over budget %dMB — evicting " - "%d LRU session(s): %s", - rss_mb, bounds.memory_high_mb, evicted_count, - ", ".join(key for key, _ in plan), - ) - try: - threading.Thread( - target=self._release_pressure_batch, - args=(plan,), - daemon=True, - name="agent-cache-pressure", - ).start() - except Exception: - self._release_pressure_batch(plan) - # NOTE: _release_pressure_batch drains `plan` in place (so the trim runs with no lingering - # agent refs) — len(plan) is 0 once the daemon thread finishes, hence the pre-captured count. - return evicted_count - - def _release_pressure_batch(self, plan: List[tuple]) -> None: - """Release a pressure-evicted batch, then return the heap to the OS. - - Sequential on one daemon thread (the batch is capped; the goal is reclaiming memory, not - racing teardowns). The trailing ``malloc_trim`` makes RSS actually fall — glibc otherwise - keeps freed arenas. The plan is drained (``pop`` + ``del``), not iterated, so no local - reference pins evicted agents during ``gc.collect`` + trim (else the valve over-evicts). - """ - while plan: - key, agent = plan.pop(0) # FIFO — evict LRU-first order preserved - try: - self._commit_then_release_soft(agent, key) - except Exception as _e: - logger.debug("Pressure release failed for %s: %s", key, _e) - del agent - try: - from hermes_cli.mem_trim import trim_memory - - trim_memory(force=True, reason="agent_cache_pressure") - except Exception: - pass - - def _enforce_agent_cache_cap(self) -> None: - """Evict oldest cached agents when cache exceeds the LRU cap. Requires _agent_cache_lock. - - Resource cleanup runs on a daemon thread so the lock is not held over slow teardown. - Agents in _running_agents are SKIPPED (their clients/sandboxes/subagents are in use); if - every LRU candidate is active the cache stays over cap until the next insert. - """ - _cache = getattr(self, "_agent_cache", None) - if _cache is None: - return - # OrderedDict.popitem(last=False) pops oldest; plain dict lacks the - # arg so skip enforcement if a test fixture swapped the cache type. - if not hasattr(_cache, "move_to_end"): - return - - # Snapshot of agent instances mid-turn, keyed by id() so lookup is O(1) and independent of - # AIAgent.__eq__ (which MagicMock overrides in tests). - running_ids = self._running_agent_ids() - - # Walk LRU → MRU; only the first (size - cap) LRU positions are candidates. An active slot is - # SKIPPED rather than evicting a newer entry — that would penalise a fresh session (no cache - # history) to protect a long-running one. Cache may stay over cap until the next insert. - cap = self._agent_cache_cap() - excess = max(0, len(_cache) - cap) - evict_plan: List[tuple] = [] # [(key, agent), ...] - if excess > 0: - ordered_keys = list(_cache.keys()) - for key in ordered_keys[:excess]: - entry = _cache.get(key) - agent = entry[0] if isinstance(entry, tuple) and entry else None - if agent is not None and id(agent) in running_ids: - continue # active mid-turn; don't evict, don't substitute - evict_plan.append((key, agent)) - - for key, _ in evict_plan: - _cache.pop(key, None) - - remaining_over_cap = len(_cache) - cap - if remaining_over_cap > 0: - logger.warning( - "Agent cache over cap (%d > %d); %d excess slot(s) held by " - "mid-turn agents — will re-check on next insert.", - len(_cache), cap, remaining_over_cap, - ) - - for key, agent in evict_plan: - logger.info( - "Agent cache at cap; evicting LRU session=%s (cache_size=%d)", - key, len(_cache), - ) - if agent is not None: - # Commit end-of-session memory, then soft-release, both on the daemon thread so the - # (possibly network-bound) provider call never blocks the held cache lock. - threading.Thread( - target=self._commit_then_release_soft, - args=(agent, key), - daemon=True, - name=f"agent-cache-evict-{key[:24]}", - ).start() - - def _sweep_idle_cached_agents(self) -> int: - """Evict cached agents idle past the idle TTL; returns the number evicted. - - Acquires the cache lock internally (safe from the expiry watcher); cleanup on daemon - threads. Agents in _running_agents are SKIPPED — tearing down an active turn crashes it. - """ - _cache = getattr(self, "_agent_cache", None) - _lock = getattr(self, "_agent_cache_lock", None) - if _cache is None or _lock is None: - return 0 - now = time.time() - idle_ttl = self._agent_cache_idle_ttl() - to_evict: List[tuple] = [] - running_ids = self._running_agent_ids() - with _lock: - for key, entry in list(_cache.items()): - agent = entry[0] if isinstance(entry, tuple) and entry else None - if agent is None: - continue - if id(agent) in running_ids: - continue # mid-turn — don't tear it down - last_activity = getattr(agent, "_last_activity_ts", None) - if last_activity is None: - continue - if (now - last_activity) > idle_ttl: - # If the session hasn't actually expired in the store (e.g. daily-reset fires hours - # after the last message), keep the agent cached so the expiry watcher can still find - # it and call on_session_end() with the live transcript. BUT only defer when the - # watcher will EVER finalize it: for mode == "none" (is_session_finalizable() False) - # deferring pins the agent for the gateway's lifetime — the leak this sweep relieves. - # Those fall through to soft eviction WITHOUT on_session_end, correctly (never a - # session-end boundary). Finite sessions evicted under LRU-cap pressure are covered - # by _commit_memory_before_soft_evict on the cap path. - session_entry = None - _store = getattr(self, "session_store", None) - try: - if _store is not None: - _store._ensure_loaded() - session_entry = _store._entries.get(key) - except Exception: - session_entry = None - if ( - session_entry is not None - and _store is not None - and _store.is_session_finalizable(session_entry) - and not _store._is_session_expired(session_entry) - ): - continue # keep agent — finite session hasn't expired - to_evict.append((key, agent)) - for key, _ in to_evict: - _cache.pop(key, None) - for key, agent in to_evict: - logger.info( - "Agent cache idle-TTL evict: session=%s (idle=%.0fs)", - key, now - getattr(agent, "_last_activity_ts", now), - ) - threading.Thread( - target=self._release_evicted_agent_soft, - args=(agent,), - daemon=True, - name=f"agent-cache-idle-{key[:24]}", - ).start() - return len(to_evict) # ---- Proxy mode: forward messages to a remote Hermes API server ---- - def _get_proxy_url(self) -> Optional[str]: - """Return the proxy URL if proxy mode is configured, else None. - - GATEWAY_PROXY_URL env var (Docker-friendly) wins over ``gateway.proxy_url`` in config.yaml. - """ - url = os.getenv("GATEWAY_PROXY_URL", "").strip() - if url: - return url.rstrip("/") - cfg = _load_gateway_config() - url = (cfg.get("gateway") or {}).get("proxy_url") - url = (url or "").strip() - if url: - return url.rstrip("/") - return None - - def _build_stream_consumer_config( - self, - source: "SessionSource", - scfg: Any, - adapter: Any, - *, - on_missing_cursor: str, - ) -> "tuple[Any, Optional[Callable[[], None]]]": - """Build the shared ``StreamConsumerConfig`` and optional Telegram pause-typing closure. - - ``on_missing_cursor`` handles adapters with ``SUPPORTS_MESSAGE_EDITING = False``: - ``"fallback"`` (proxy path) streams with an empty cursor; ``"raise"`` (in-process path) - raises ``RuntimeError`` so the caller's ``except`` skips streaming entirely. Returns - ``(consumer_cfg, pause_typing_before_finalize)``. - """ - from gateway.stream_consumer import StreamConsumerConfig - - _pause_typing_before_finalize = None - if source.platform == Platform.TELEGRAM and hasattr(adapter, "pause_typing_for_chat"): - def _pause_typing_before_finalize( - _adapter=adapter, - _chat_id=source.chat_id, - ) -> None: - _adapter.pause_typing_for_chat(_chat_id) - # Platforms that can't edit sent messages (e.g. QQ, WeChat) skip streaming entirely: the - # partial first message could never be updated, yielding duplicates (partial + final). - _adapter_supports_edit = getattr(adapter, "SUPPORTS_MESSAGE_EDITING", True) - # Adapters that can't edit but have a native-streaming transport (e.g. WeCom msgtype "stream" - # via send_stream_frame) pass the gate — the consumer's native branch delivers the full turn. - _adapter_supports_native_stream = bool(getattr( - adapter, "SUPPORTS_NATIVE_STREAMING", False, - )) - if ( - not _adapter_supports_edit - and not _adapter_supports_native_stream - and on_missing_cursor == "raise" - ): - raise RuntimeError("skip streaming for non-editable platform") - _effective_cursor = scfg.cursor if _adapter_supports_edit else "" - # Some Matrix clients render the streaming cursor as a visible tofu/white-box artifact: keep - # streaming text on Matrix, but suppress the cursor. - _buffer_only = False - if source.platform == Platform.MATRIX: - _effective_cursor = "" - _buffer_only = True - # Fresh-final applies to Telegram only — other platforms edit in place cheaply (Discord, Slack) - # or lack the edit-timestamp-stays-stale problem. - _fresh_final_secs = ( - float(getattr(scfg, "fresh_final_after_seconds", 0.0) or 0.0) - if source.platform == Platform.TELEGRAM - else 0.0 - ) - _consumer_cfg = StreamConsumerConfig( - edit_interval=scfg.edit_interval, - buffer_threshold=scfg.buffer_threshold, - cursor=_effective_cursor, - buffer_only=_buffer_only, - fresh_final_after_seconds=_fresh_final_secs, - transport=scfg.transport or "edit", - chat_type=getattr(source, "chat_type", "") or "", - ) - return _consumer_cfg, _pause_typing_before_finalize - - async def _run_agent_via_proxy( - self, - message: str, - context_prompt: str, - history: List[Dict[str, Any]], - source: "SessionSource", - session_id: str, - session_key: str = None, - run_generation: Optional[int] = None, - event_message_id: Optional[str] = None, - ) -> Dict[str, Any]: - """Forward the message to a remote Hermes API server instead of running a local AIAgent. - - This lets a Docker container handle Matrix E2EE while the actual agent runs on the host - with full access to local files, memory, skills, and a unified session store. - """ - try: - from aiohttp import ClientSession as _AioClientSession, ClientTimeout - except ImportError: - return { - "final_response": "⚠️ Proxy mode requires aiohttp. Install with: pip install aiohttp", - "messages": [], - "api_calls": 0, - "tools": [], - } - - proxy_url = self._get_proxy_url() - if not proxy_url: - return { - "final_response": "⚠️ Proxy URL not configured (GATEWAY_PROXY_URL or gateway.proxy_url)", - "messages": [], - "api_calls": 0, - "tools": [], - } - - # Scope-aware read: the proxy key is a per-profile credential; under multiplex honor the - # installed scope's verdict (Slack pattern for the unscoped default-profile loop). - try: - from agent.secret_scope import UnscopedSecretError, get_secret - - try: - proxy_key = (get_secret("GATEWAY_PROXY_KEY") or "").strip() - except UnscopedSecretError: - proxy_key = os.getenv("GATEWAY_PROXY_KEY", "").strip() - except Exception: - proxy_key = os.getenv("GATEWAY_PROXY_KEY", "").strip() - - def _run_still_current() -> bool: - if run_generation is None or not session_key: - return True - return self._is_session_run_current(session_key, run_generation) - - # Build messages in OpenAI chat format. The remote api_server keeps continuity via - # X-Hermes-Session-Id and loads its own history, so send only the current message; if the - # remote has no history yet, include a compact text-only local history (remote replays tools). - api_messages: List[Dict[str, str]] = [] - - if context_prompt: - api_messages.append({"role": "system", "content": context_prompt}) - - for msg in history: - role = msg.get("role") - content = msg.get("content") - if role in {"user", "assistant"} and content: - api_messages.append({"role": role, "content": content}) - - api_messages.append({"role": "user", "content": message}) - - # HTTP headers --------------------------------------------------- - headers: Dict[str, str] = {"Content-Type": "application/json"} - if proxy_key: - headers["Authorization"] = f"Bearer {proxy_key}" - if session_id: - headers["X-Hermes-Session-Id"] = session_id - - body = { - "model": "hermes-agent", - "messages": api_messages, - "stream": True, - } - - # Set up platform streaming if available ------------------------- - _stream_consumer = None - _scfg = getattr(getattr(self, "config", None), "streaming", None) - if _scfg is None: - from gateway.config import StreamingConfig - _scfg = StreamingConfig() - - platform_key = _platform_config_key(source.platform) - user_config = _load_gateway_config() - from gateway.display_config import resolve_display_setting - _plat_streaming = resolve_display_setting( - user_config, platform_key, "streaming" - ) - _streaming_enabled = ( - _scfg.enabled and _scfg.transport != "off" - if _plat_streaming is None - else bool(_plat_streaming) - ) - - _thread_metadata: Optional[Dict[str, Any]] = self._thread_metadata_for_source(source, event_message_id) - - if _streaming_enabled: - try: - from gateway.stream_consumer import GatewayStreamConsumer - _adapter = self._adapter_for_source(source) - if _adapter: - _consumer_cfg, _pause_typing_before_finalize = ( - self._build_stream_consumer_config( - source, _scfg, _adapter, - on_missing_cursor="fallback", - ) - ) - _stream_consumer = GatewayStreamConsumer( - adapter=_adapter, - chat_id=source.chat_id, - config=_consumer_cfg, - metadata=_thread_metadata, - on_before_finalize=_pause_typing_before_finalize, - initial_reply_to_id=event_message_id, - run_still_current=_run_still_current, - ) - except Exception as _sc_err: - logger.debug("Proxy: could not set up stream consumer: %s", _sc_err) - - # Run the stream consumer task in the background - stream_task = None - if _stream_consumer: - stream_task = asyncio.create_task(_stream_consumer.run()) - - # Send typing indicator - _adapter = self._adapter_for_source(source) - if _adapter: - with suppress(Exception): - await _adapter.send_typing(source.chat_id, metadata=_thread_metadata) - - # Make the HTTP request with SSE streaming ----------------------- - full_response = "" - _start = time.time() - - try: - _timeout = ClientTimeout(total=0, sock_read=1800) - async with _AioClientSession(timeout=_timeout) as session: - async with session.post( - f"{proxy_url}/v1/chat/completions", - json=body, - headers=headers, - ) as resp: - if resp.status != 200: - error_text = await resp.text() - logger.warning( - "Proxy error (%d) from %s: %s", - resp.status, proxy_url, error_text[:500], - ) - return { - "final_response": f"⚠️ Proxy error ({resp.status}): {error_text[:300]}", - "messages": [], - "api_calls": 0, - "tools": [], - } - - # Parse SSE stream - buffer = "" - async for chunk in resp.content.iter_any(): - if not _run_still_current(): - logger.info( - "Discarding stale proxy stream for %s — generation %d is no longer current", - session_key or "?", - run_generation or 0, - ) - return { - "final_response": "", - "messages": [], - "api_calls": 0, - "tools": [], - "history_offset": len(history), - "session_id": session_id, - "response_previewed": False, - } - text = chunk.decode("utf-8", errors="replace") - buffer += text - - # Process complete SSE lines - while "\n" in buffer: - line, buffer = buffer.split("\n", 1) - line = line.strip() - if not line: - continue - if line.startswith("data: "): - data = line[6:] - if data.strip() == "[DONE]": - break - try: - obj = json.loads(data) - choices = obj.get("choices", []) - if choices: - delta = choices[0].get("delta", {}) - content = delta.get("content", "") - if content: - full_response += content - if _stream_consumer: - _stream_consumer.on_delta(content) - except json.JSONDecodeError: - pass - if len(buffer) > _GATEWAY_PROXY_SSE_BUFFER_MAX_CHARS: - raise ValueError( - "Proxy SSE stream exceeded max buffer size without a line boundary" - ) - - except asyncio.CancelledError: - raise - except Exception as e: - logger.error("Proxy connection error to %s: %s", proxy_url, e) - if not full_response: - return { - "final_response": f"⚠️ Proxy connection error: {e}", - "messages": [], - "api_calls": 0, - "tools": [], - } - # Partial response — return what we got - finally: - # Finalize stream consumer - if _stream_consumer: - _stream_consumer.finish() - if stream_task: - try: - await asyncio.wait_for(stream_task, timeout=5.0) - except (asyncio.TimeoutError, asyncio.CancelledError): - stream_task.cancel() - - _elapsed = time.time() - _start - if not _run_still_current(): - logger.info( - "Discarding stale proxy result for %s — generation %d is no longer current", - session_key or "?", - run_generation or 0, - ) - return { - "final_response": "", - "messages": [], - "api_calls": 0, - "tools": [], - "history_offset": len(history), - "session_id": session_id, - "response_previewed": False, - } - logger.info( - "proxy response: url=%s session=%s time=%.1fs response=%d chars", - proxy_url, (session_id or "")[:20], _elapsed, len(full_response), - ) - - return { - "final_response": full_response or "(No response from remote agent)", - "messages": [ - {"role": "user", "content": message}, - {"role": "assistant", "content": full_response}, - ], - "api_calls": 1, - "tools": [], - "history_offset": len(history), - "session_id": session_id, - "response_previewed": _stream_consumer is not None and bool(full_response), - } # ------------------------------------------------------------------ - async def _run_agent( - self, - message: str, - context_prompt: str, - history: List[Dict[str, Any]], - source: SessionSource, - session_id: str, - session_key: str = None, - run_generation: Optional[int] = None, - _interrupt_depth: int = 0, - event_message_id: Optional[str] = None, - inbound_message_id: Optional[str] = None, - channel_prompt: Optional[str] = None, - moa_config: Optional[dict] = None, - persist_user_message: Optional[Any] = None, - persist_user_timestamp: Optional[float] = None, - persist_user_display_kind: Optional[str] = None, - message_type: Optional[str] = None, - ) -> Dict[str, Any]: - """Profile-scoping wrapper around the agent run. - - Under multiplexing, run the turn inside ``_profile_runtime_scope`` so config/skills/memory - resolve to the source profile's home AND credentials come from its secret scope (never - process-global ``os.environ``). Transparent pass-through when multiplexing is off. - """ - if not getattr(getattr(self, "config", None), "multiplex_profiles", False): - return await self._run_agent_inner( - message, context_prompt, history, source, session_id, - session_key=session_key, run_generation=run_generation, - _interrupt_depth=_interrupt_depth, event_message_id=event_message_id, - inbound_message_id=inbound_message_id, - channel_prompt=channel_prompt, moa_config=moa_config, - persist_user_message=persist_user_message, - persist_user_timestamp=persist_user_timestamp, - persist_user_display_kind=persist_user_display_kind, - message_type=message_type, - ) - - profile_home = self._resolve_profile_home_for_source(source) - with _profile_runtime_scope(profile_home): - return await self._run_agent_inner( - message, context_prompt, history, source, session_id, - session_key=session_key, run_generation=run_generation, - _interrupt_depth=_interrupt_depth, event_message_id=event_message_id, - inbound_message_id=inbound_message_id, - channel_prompt=channel_prompt, moa_config=moa_config, - persist_user_message=persist_user_message, - persist_user_timestamp=persist_user_timestamp, - persist_user_display_kind=persist_user_display_kind, - message_type=message_type, - ) def _profile_name_for_source(self, source: SessionSource) -> Optional[str]: """Resolve the profile name for an inbound source via configured routes. @@ -27050,1644 +5511,42 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) return get_hermes_home() - async def _run_agent_inner( - self, - message: str, - context_prompt: str, - history: List[Dict[str, Any]], - source: SessionSource, - session_id: str, - session_key: str = None, - run_generation: Optional[int] = None, - _interrupt_depth: int = 0, - event_message_id: Optional[str] = None, - inbound_message_id: Optional[str] = None, - channel_prompt: Optional[str] = None, - moa_config: Optional[dict] = None, - persist_user_message: Optional[Any] = None, - persist_user_timestamp: Optional[float] = None, - persist_user_display_kind: Optional[str] = None, - message_type: Optional[str] = None, - ) -> Dict[str, Any]: - """Run the agent; returns the full run_conversation result dict. - - Keys: "final_response", "messages", "api_calls", "completed". - """ - # ---- Proxy mode: delegate to remote API server ---- - if self._get_proxy_url(): - return await self._run_agent_via_proxy( - message=message, - context_prompt=context_prompt, - history=history, - source=source, - session_id=session_id, - session_key=session_key, - run_generation=run_generation, - event_message_id=event_message_id, - ) - - from run_agent import AIAgent - import queue - - def _run_still_current() -> bool: - if run_generation is None or not session_key: - return True - return self._is_session_run_current(session_key, run_generation) - - user_config = _load_gateway_config() - platform_key = _platform_config_key(source.platform) - - enabled_toolsets = self._resolve_enabled_toolsets_for_source( - user_config, source, platform_key - ) - agent_cfg_local = user_config.get("agent") or {} - from agent.skill_utils import parse_config_string_list - - disabled_toolsets = parse_config_string_list(agent_cfg_local.get("disabled_toolsets")) or None - - display_config = user_config.get("display", {}) - if not isinstance(display_config, dict): - display_config = {} - - # Per-platform display settings via display_config: display.platforms.., then - # display. global, then built-in platform defaults. - from gateway.display_config import resolve_display_setting - - # Apply tool preview length config (0 = no limit) - try: - from agent.display import set_tool_preview_max_len - _tpl = resolve_display_setting(user_config, platform_key, "tool_preview_length", 0) - set_tool_preview_max_len(int(_tpl) if _tpl else 0) - except Exception: - pass - - # Apply friendly tool labels config (default on) — per-platform aware - try: - from agent.display import set_friendly_tool_labels - _ftl = resolve_display_setting(user_config, platform_key, "friendly_tool_labels", True) - set_friendly_tool_labels(bool(_ftl)) - except Exception: - pass - - # Tool progress mode — resolved per-platform with env var fallback - _resolved_tp = resolve_display_setting(user_config, platform_key, "tool_progress") - _env_tp = os.getenv("HERMES_TOOL_PROGRESS_MODE") - _display_cfg = display_config if isinstance(display_config, dict) else {} - _platforms_cfg = _display_cfg.get("platforms") or {} - _platform_cfg = _platforms_cfg.get(platform_key) or {} - _legacy_tp_overrides = _display_cfg.get("tool_progress_overrides") or {} - _tool_progress_configured = ( - "tool_progress" in _display_cfg - or ( - isinstance(_platform_cfg, dict) - and "tool_progress" in _platform_cfg - ) - or ( - isinstance(_legacy_tp_overrides, dict) - and platform_key in _legacy_tp_overrides - ) - ) - progress_mode = ( - _env_tp - if _env_tp and not _tool_progress_configured - else (_resolved_tp or _env_tp or "all") - ) - # Tool progress grouping: "accumulate" (edit one bubble) or "separate" (one msg per tool) - progress_grouping = resolve_display_setting(user_config, platform_key, "tool_progress_grouping") or "accumulate" - from gateway.status_phrases import choose_status_phrase, resolve_status_phrase_catalog - _generic_status_recent: List[str] = [] - _generic_status_catalog = resolve_status_phrase_catalog(user_config, platform_key) - - def _display_surface_mode( - setting: str, - *, - default: bool = False, - require_platform_override_for: set[Any] | None = None, - allow_generic: bool = False, - ) -> str: - """Return off|raw|generic for a gateway visibility surface.""" - if require_platform_override_for: - current_platform = _gateway_platform_value(source.platform) - platform_only = { - _gateway_platform_value(item) - for item in require_platform_override_for - } - if ( - current_platform in platform_only - and not _has_platform_display_override(user_config, platform_key, setting) - ): - return "off" - value = resolve_display_setting(user_config, platform_key, setting, default) - if isinstance(value, str) and value.strip().lower() == "generic": - return "generic" if allow_generic else "off" - return "raw" if bool(value) else "off" - - def _generic_status_phrase(kind: str, *, tool_name: str | None = None, preview: str | None = None, args: Any = None) -> str: - try: - return choose_status_phrase( - kind, - tool_name=tool_name, - preview=preview, - args=args, - recent=_generic_status_recent, - catalog=_generic_status_catalog, - ) - except Exception as _phrase_err: - logger.debug("generic status phrase selection failed: %s", _phrase_err) - return "still on it" if kind in {"heartbeat", "waiting", "long_running", "status"} else "one sec" - # Disable tool progress for webhooks - they don't support message editing, - # so each progress line would be sent as a separate message. - from gateway.config import Platform - tool_progress_enabled = progress_mode not in {"off", "log"} and source.platform != Platform.WEBHOOK - # Live working-state status for text-rendering typing indicators (Slack's assistant status - # line). Independent of tool_progress (Slack defaults it off; the status line is ephemeral). - # Rides the existing _keep_typing refresh — the callback only stores a phrase, no extra calls. - _live_status_mode = resolve_display_setting( - user_config, platform_key, "live_status", "full" - ) - _live_status_adapter = self._adapter_for_source(source) - if not getattr(_live_status_adapter, "supports_status_text", False): - _live_status_adapter = None - if _live_status_mode == "off": - _live_status_adapter = None - # "log" mode: tool calls are written to ~/.hermes/logs/tool_calls.log - # instead of the chat (#3459 / #3458). Gateway-only by design. - log_mode_enabled = progress_mode == "log" and source.platform != Platform.WEBHOOK - log_queue: "queue.Queue | None" = queue.Queue() if log_mode_enabled else None - # Natural assistant status messages are independent from tool progress and token streaming: - # tool_progress can stay quiet while users opt into concise mid-turn updates. - interim_assistant_messages_mode = _display_surface_mode( - "interim_assistant_messages", - default=True, - require_platform_override_for={Platform.MATTERMOST}, - ) - interim_assistant_messages_enabled = ( - source.platform != Platform.WEBHOOK - and interim_assistant_messages_mode != "off" - ) - # thinking_progress is independent — if enabled, we need the progress queue even when - # tool_progress is off (thinking relay uses same infra). Mattermost requires a per-platform - # opt-in: global scratch-text display is too easy to leak into busy public threads. - _thinking_mode = _display_surface_mode( - "thinking_progress", - default=False, - require_platform_override_for={Platform.MATTERMOST}, - ) - _thinking_enabled = _thinking_mode != "off" - # Slack-native task cards: with the Slack adapter's opt-in, tool progress renders as native - # plan/task cards via chat.startStream, so the progress queue is needed even though Slack keeps - # text tool_progress off by default (requiring both flags would silently disable the feature). - _progress_adapter_for_native = self._adapter_for_source(source) - _native_slack_task_cards = False - if ( - source.platform == Platform.SLACK - and _progress_adapter_for_native is not None - and hasattr(_progress_adapter_for_native, "native_task_cards_enabled") - ): - try: - _native_slack_task_cards = bool( - _progress_adapter_for_native.native_task_cards_enabled() - ) - except Exception: - logger.debug("Slack native task-card config check failed", exc_info=True) - needs_progress_queue = ( - tool_progress_enabled or _thinking_enabled or _native_slack_task_cards - ) - - # Queue for progress messages (thread-safe) - progress_queue = queue.Queue() if needs_progress_queue else None - last_tool = [None] # Mutable container for tracking in closure - last_progress_msg = [None] # Track last message for dedup - repeat_count = [0] # How many times the same message repeated - # True when the previous progress line was a terminal fenced code block — consecutive terminal - # calls then drop the repeated "💻 terminal" header and render back-to-back blocks. - last_was_terminal_block = [False] - - # Discord voice "verbal ack before tool calls": with the continuous mixer installed - # (discord.voice_fx.enabled), speak a short phrase over the idle bed on the FIRST tool call of - # the turn (from tool_start_callback, independent of the tool-progress text gate); once per turn. - _voice_ack_fired = [False] - _voice_ack_guild: List[Optional[int]] = [None] - if source.platform == Platform.DISCORD: - _va = self.adapters.get(Platform.DISCORD) - # source.chat_id is the linked text channel; resolve the guild whose - # voice connection is bound to it (mirrors DiscordAdapter.play_tts). - _vtc = getattr(_va, "_voice_text_channels", None) - if isinstance(_vtc, dict) and hasattr(_va, "voice_mixer_active"): - for _gid, _tc in _vtc.items(): - if str(_tc) == str(source.chat_id) and _va.voice_mixer_active(_gid): - _voice_ack_guild[0] = _gid - break - _voice_ack_loop = asyncio.get_running_loop() - - # voice_ack_callback extracted to TurnRunner.voice_ack_callback - # (published onto turn_ctx after the runner is constructed below). - - # Auto-cleanup of temporary progress bubbles (Telegram + any adapter that implements - # ``delete_message``). Failed runs skip cleanup so the bubbles remain as breadcrumbs. - _cleanup_progress = bool( - resolve_display_setting(user_config, platform_key, "cleanup_progress") - ) - _cleanup_adapter = self._adapter_for_source(source) if _cleanup_progress else None - # getattr, not attribute access — same duck-typed-adapter guard as the edit_message check in - # send_progress_messages: a fake adapter without delete_message means "can't delete", not a crash. - _cleanup_delete = getattr(type(_cleanup_adapter), "delete_message", None) if _cleanup_adapter is not None else None - if _cleanup_adapter is not None and ( - _cleanup_delete is None - or _cleanup_delete is BasePlatformAdapter.delete_message - ): - # Adapter doesn't support deletion — silently disable. - _cleanup_progress = False - _cleanup_adapter = None - _cleanup_msg_ids: List[str] = [] - # First-touch onboarding latch: fires at most once per run, even if - # several tools exceed the threshold. - long_tool_hint_fired = [False] - _LONG_TOOL_THRESHOLD_S = 30.0 - - turn_ctx = TurnContext( - source=source, - _run_still_current=_run_still_current, - _live_status_adapter=_live_status_adapter, - _live_status_mode=_live_status_mode, - _thinking_enabled=_thinking_enabled, - progress_mode=progress_mode, - progress_grouping=progress_grouping, - tool_progress_enabled=tool_progress_enabled, - progress_queue=progress_queue, - log_queue=log_queue, - last_progress_msg=last_progress_msg, - last_tool=last_tool, - last_was_terminal_block=last_was_terminal_block, - repeat_count=repeat_count, - long_tool_hint_fired=long_tool_hint_fired, - _LONG_TOOL_THRESHOLD_S=_LONG_TOOL_THRESHOLD_S, - _cleanup_progress=_cleanup_progress, - _cleanup_msg_ids=_cleanup_msg_ids, - message=message, - AIAgent=AIAgent, - resolve_display_setting=resolve_display_setting, - user_config=user_config, - enabled_toolsets=enabled_toolsets, - disabled_toolsets=disabled_toolsets, - log_mode_enabled=log_mode_enabled, - interim_assistant_messages_enabled=interim_assistant_messages_enabled, - needs_progress_queue=needs_progress_queue, - _native_slack_task_cards=_native_slack_task_cards, - _voice_ack_fired=_voice_ack_fired, - _voice_ack_guild=_voice_ack_guild, - _voice_ack_loop=_voice_ack_loop, - history=history, - context_prompt=context_prompt, - channel_prompt=channel_prompt, - session_id=session_id, - session_key=session_key, - run_generation=run_generation, - _interrupt_depth=_interrupt_depth, - event_message_id=event_message_id, - inbound_message_id=inbound_message_id, - moa_config=moa_config, - persist_user_message=persist_user_message, - persist_user_timestamp=persist_user_timestamp, - persist_user_display_kind=persist_user_display_kind, - ) - turn_runner = TurnRunner(self, turn_ctx) - # Callback invoked by agent on tool lifecycle events — extracted to - # TurnRunner.progress_callback (bound method, same signature). - turn_ctx.progress_callback = turn_runner.progress_callback - turn_ctx.voice_ack_callback = turn_runner.voice_ack_callback - turn_ctx.native_tool_start_callback = turn_runner.combined_tool_start_callback - turn_ctx.native_tool_complete_callback = ( - turn_runner.native_tool_complete_callback - ) - - # Background task accumulating tool lines into one edited progress message. Threading metadata - # is platform-specific: Slack DM threading needs the event_message_id fallback; Telegram forum - # topics use message_thread_id and Hermes-created private DM topic lanes need thread metadata - # plus a reply anchor; Feishu only honors reply_in_thread on a reply, so topic progress replies - # to the triggering event; others use explicit source.thread_id only. Slack honours - # reply_in_thread=false: don't synthesise a thread for progress, or every later reply inherits it. - _progress_reply_in_thread = True - if source.platform == Platform.SLACK: - _slack_adapter_for_progress = self._adapter_for_source(source) - if _slack_adapter_for_progress is not None: - try: - # Relay lane: adapter owns mode resolution (nested platforms.relay.extra.slack subset, - # flat-key fallback). Native lane: read the flat extra as before. - _mode_fn = getattr( - _slack_adapter_for_progress, - "_effective_reply_in_thread", - None, - ) - if callable(_mode_fn): - _progress_reply_in_thread = bool(_mode_fn()) - else: - _progress_reply_in_thread = bool( - _slack_adapter_for_progress.config.extra.get( - "reply_in_thread", True - ) - ) - except Exception: - _progress_reply_in_thread = True - elif str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz": - # Buzz honours the same opt-out (reply_to_mode: off / extra.reply_in_thread: false): when the - # user asked for flat channel replies, progress must not synthesise a thread either. - _buzz_adapter_for_progress = self._adapter_for_source(source) - if _buzz_adapter_for_progress is not None: - try: - _progress_reply_in_thread = ( - getattr(_buzz_adapter_for_progress, "_reply_to_mode", "first") - != "off" - ) - except Exception: - _progress_reply_in_thread = True - _progress_thread_id = _resolve_progress_thread_id( - source.platform, source.thread_id, event_message_id, - reply_in_thread=_progress_reply_in_thread, - ) - # Relay Discord auto-thread lane: a channel-initiating message has no thread_id at ingest - # (thread is born on the connector's FIRST send). The connector stamps prospective_thread_id - # (anchor id == the thread it will create); carry it as reply_to on the progress send so - # bubbles route into the SAME auto-thread instead of landing flat in the parent channel. - _relay_prospective_thread_id = ( - str(getattr(source, "prospective_thread_id", None)) - if source.platform == Platform.DISCORD - and getattr(source, "delivered_via_upstream_relay", False) - and getattr(source, "prospective_thread_id", None) - and not source.thread_id - else None - ) - _progress_metadata = ( - self._thread_metadata_for_source(source, event_message_id) - if _progress_thread_id == source.thread_id - else self._thread_metadata_for_target( - source.platform, - source.chat_id, - _progress_thread_id, - chat_type=getattr(source, "chat_type", None), - reply_to_message_id=event_message_id, - ) - ) if _progress_thread_id else None - if _progress_metadata is None and _relay_prospective_thread_id: - # No real thread yet, but the connector will auto-thread on the - # reply anchor; carry it so progress joins that thread. - _progress_metadata = {"reply_to_message_id": event_message_id} - _progress_metadata = _non_conversational_metadata(_progress_metadata, platform=source.platform) - if _native_slack_task_cards: - # chat.startStream in channels requires the recipient team/user - # pair; harmless extras elsewhere, so stamp them whenever known. - _progress_metadata = dict(_progress_metadata or {}) - if source.scope_id: - _progress_metadata.setdefault("recipient_team_id", source.scope_id) - _progress_metadata.setdefault("slack_team_id", source.scope_id) - if source.user_id: - _progress_metadata.setdefault("recipient_user_id", source.user_id) - _progress_reply_to = ( - event_message_id - if ( - source.platform in (Platform.FEISHU, Platform.MATTERMOST) - and source.thread_id - and event_message_id - ) - or ( - # Buzz has no native thread_id; threading is always via reply-to the triggering event id - # (channel clutter otherwise); skipped when the user opted out of threaded replies. - str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz" - and event_message_id - and _progress_reply_in_thread - ) - or _relay_prospective_thread_id - else None - ) - - async def write_tool_log(): - """Drain log_queue and append tool-call lines to tool_calls.log (tool_progress=log). - - RotatingFileHandler (5MB × 3) bounds the log; RedactingFormatter keeps secrets off disk. - """ - if log_queue is None: - return - from logging.handlers import RotatingFileHandler - - from agent.redact import RedactingFormatter - - log_dir = _hermes_home / "logs" - log_dir.mkdir(parents=True, exist_ok=True) - file_handler = RotatingFileHandler( - log_dir / "tool_calls.log", - maxBytes=5 * 1024 * 1024, - backupCount=3, - encoding="utf-8", - ) - file_handler.setFormatter(RedactingFormatter("%(message)s")) - tool_logger = logging.getLogger(f"hermes.tool_calls.{id(log_queue)}") - tool_logger.setLevel(logging.INFO) - tool_logger.propagate = False - tool_logger.addHandler(file_handler) - try: - while True: - try: - tool_logger.info("%s", log_queue.get_nowait()) - except queue.Empty: - await asyncio.sleep(0.3) - except Exception as e: - logger.error("write_tool_log error: %s", e) - await asyncio.sleep(1) - except asyncio.CancelledError: - pass - finally: - # Drain remaining entries before closing so late tool calls - # from the final iteration aren't lost. - while True: - try: - tool_logger.info("%s", log_queue.get_nowait()) - except queue.Empty: - break - except Exception: - break - tool_logger.removeHandler(file_handler) - try: - file_handler.flush() - file_handler.close() - except Exception: - pass - - # Extracted to TurnRunner.send_progress_messages; the threading metadata above is published - # onto the shared TurnContext where the original closure's captured locals were bound. - turn_ctx._progress_metadata = _progress_metadata - turn_ctx._progress_reply_to = _progress_reply_to - send_progress_messages = turn_runner.send_progress_messages - - # We need to share the agent instance for interrupt support - agent_holder = [None] # Mutable container for the agent instance - turn_ctx.agent_holder = agent_holder - result_holder = [None] # Mutable container for the result - tools_holder = [None] # Mutable container for the tool definitions - stream_consumer_holder = [None] # Mutable container for stream consumer - # streaming PCM audio consumer. Created on the gateway event-loop thread (NOT in run_sync's - # executor worker) so outer finalisation / interrupt paths can reference it without a NameError. - streaming_tts_consumer_holder: list = [None] - turn_ctx.result_holder = result_holder - turn_ctx.tools_holder = tools_holder - turn_ctx.stream_consumer_holder = stream_consumer_holder - turn_ctx.streaming_tts_consumer_holder = streaming_tts_consumer_holder - - # Bridge sync step_callback → async hooks.emit for agent:step events - _loop_for_step = asyncio.get_running_loop() - _hooks_ref = self.hooks - - # Bridge extracted to TurnRunner._step_callback_sync; the loop and - # hooks refs bound just above are published at their original site. - turn_ctx._loop_for_step = _loop_for_step - turn_ctx._hooks_ref = _hooks_ref - turn_ctx._step_callback_sync = turn_runner._step_callback_sync - - # Bridge sync event_callback → async hooks.emit for lifecycle events (e.g. session:compress - # after a compression split); extracted to TurnRunner._event_callback_sync. - turn_ctx._event_callback_sync = turn_runner._event_callback_sync - - # Bridge sync status_callback → async adapter.send for context pressure - _status_adapter = self._adapter_for_source(source) - _status_chat_id = source.chat_id - if source.platform == Platform.FEISHU and source.thread_id and event_message_id: - # Feishu topics only keep messages inside the topic when they are sent via the reply API - # with reply_in_thread=true. Status/approval/stream paths usually only get metadata, so - # carry the triggering message id as a Feishu-specific fallback. - _status_thread_metadata: Optional[Dict[str, Any]] = { - "thread_id": _progress_thread_id, - "reply_to_message_id": event_message_id, - } - else: - _status_thread_metadata = ( - self._thread_metadata_for_source(source, event_message_id) - if _progress_thread_id == source.thread_id - else self._thread_metadata_for_target( - source.platform, - source.chat_id, - _progress_thread_id, - chat_type=getattr(source, "chat_type", None), - reply_to_message_id=event_message_id, - ) - ) if _progress_thread_id else None - if _status_thread_metadata is None and _relay_prospective_thread_id: - # Relay Discord auto-thread lane (see _progress_metadata): carry the reply anchor so - # status/interim bubbles route into the same connector-created thread as the final reply. - _status_thread_metadata = { - "reply_to_message_id": event_message_id - } - - # Bridge extracted to TurnRunner._status_callback_sync; publish the status wiring computed - # above onto the shared TurnContext at the exact original binding site. - turn_ctx._status_adapter = _status_adapter - turn_ctx._status_chat_id = _status_chat_id - turn_ctx._status_thread_metadata = _status_thread_metadata - turn_ctx._status_callback_sync = turn_runner._status_callback_sync - - # Streaming TTS consumer setup. Created on the gateway event-loop thread (here), NOT inside - # run_sync's executor worker: the outer interrupt / finalisation paths reference the consumer - # via ``streaming_tts_consumer_holder[0]`` and would hit a cross-scope NameError. - _stts_adapter = self._adapter_for_source(source) - _is_voice_input = ( - message_type is not None - and str(getattr(message_type, "value", message_type)).lower() == "voice" - ) - if ( - _stts_adapter is not None - and _is_voice_input - and _stts_adapter._should_auto_tts_for_chat(source.chat_id) - ): - try: - from gateway.streaming_tts_consumer import StreamingTTSConsumer - from tools.tts_tool import _load_tts_config - _tts_cfg = _load_tts_config() - _gateway_loop = self._gateway_loop or asyncio.get_event_loop() - _stts_consumer = StreamingTTSConsumer( - adapter=_stts_adapter, - chat_id=source.chat_id, - tts_config=_tts_cfg, - loop=_gateway_loop, - metadata=_status_thread_metadata, - ) - if _stts_consumer.active: - streaming_tts_consumer_holder[0] = _stts_consumer - _stts_consumer.start() - # else: consumer inactive (no streaming provider) — leave - # the holder as None so the whole-file fallback path runs. - except Exception as _stts_err: - logger.debug("Could not set up streaming TTS consumer: %s", _stts_err) - - # run_sync extracted to TurnRunner.run_sync (bound method; executor call unchanged). Its - # closed-over locals travel on turn_ctx; `nonlocal message` rebinds became ctx.message writes. - run_sync = turn_runner.run_sync - - # Start the progress sender if enabled. Gate on needs_progress_queue (tool_progress OR - # thinking_progress), not tool_progress alone: the sender drains BOTH tool-progress lines and - # _thinking scratch bubbles — a tool_progress-only gate left thinking-only queues never drained. - progress_task = None - if needs_progress_queue: - progress_task = asyncio.create_task(send_progress_messages()) - - # Start the tool-call log writer when tool_progress == "log". - log_task = None - if log_mode_enabled: - log_task = asyncio.create_task(write_tool_log()) - - # Start stream consumer task — polls for consumer creation since it - # happens inside run_sync (thread pool) after the agent is constructed. - stream_task = None - - async def _start_stream_consumer(): - """Wait for the stream consumer to be created, then run it.""" - for _ in range(200): # Up to 10s wait - if stream_consumer_holder[0] is not None: - await stream_consumer_holder[0].run() - return - await asyncio.sleep(0.05) - - stream_task = asyncio.create_task(_start_stream_consumer()) - - # Track this agent as running for this session (for interrupt support) - # We do this in a callback after the agent is created - async def track_agent(): - # Wait for agent to be created - while agent_holder[0] is None: - await asyncio.sleep(0.05) - if not session_key: - return - # Only promote the sentinel to the real agent if this run is still current. If /stop or - # /new bumped the generation while we were spinning up, leave the newer run's slot alone - # — we'll be discarded by the stale-result check in _handle_message_with_agent. - if run_generation is not None and not self._is_session_run_current( - session_key, run_generation - ): - logger.info( - "Skipping stale agent promotion for %s — generation %s is no longer current", - session_key or "", - run_generation, - ) - return - self._session_state(session_key).turn.agent = agent_holder[0] - if self._draining: - self._update_runtime_status("draining") - - tracking_task = asyncio.create_task(track_agent()) - - # Monitor adapter interrupts (new messages). PRIMARY interrupt path for regular text: Level 1 - # (base.py) catches them before _handle_message(), so the Level 2 running_agent.interrupt() path - # never fires. The inactivity poll loop has a BACKUP check in case this task dies silently. - _interrupt_detected = asyncio.Event() # shared with backup check - - async def monitor_for_interrupt(): - if not session_key: - return - - while True: - await asyncio.sleep(0.2) # Check every 200ms - try: - # Re-resolve adapter each iteration so reconnects don't - # leave us holding a stale reference. - _adapter = self._adapter_for_source(source) - if not _adapter: - continue - # Must use session_key (build_session_key output), NOT source.chat_id: the adapter - # stores interrupt events under the full session key. - if hasattr(_adapter, 'has_pending_interrupt') and _adapter.has_pending_interrupt(session_key): - agent = agent_holder[0] - if agent: - # Peek WITHOUT consuming: the message must stay in _pending_messages for the - # post-run _dequeue_pending_event() (full MessageEvent + media). Popping here - # races: the agent may finish before checking _interrupt_requested, losing it. - _peek_event = _adapter._pending_messages.get(session_key) - pending_text = None - if _peek_event is not None: - pending_text = _peek_event.text or "" - # Transcribe audio BEFORE signaling the agent, so voice messages interrupt - # with the real transcript, not an empty string / file-path placeholder. - _media_urls = getattr(_peek_event, "media_urls", None) or [] - if self._pending_event_audio_paths(_peek_event): - pending_text, _ = await self._transcribe_and_echo_pending_voice( - _peek_event, - _adapter, - source, - pending_text, - log_context="Voice-interrupt", - metadata={"thread_id": source.thread_id} if source.thread_id else None, - ) - elif not pending_text and _media_urls: - pending_text = _build_media_placeholder(_peek_event) - logger.debug("Interrupt detected from adapter, signaling agent...") - agent.interrupt(pending_text) - _interrupt_detected.set() - # Abort streaming TTS on barge-in (#60671). - _stts = streaming_tts_consumer_holder[0] - if _stts is not None: - _stts.abort("barge-in") - break - except asyncio.CancelledError: - raise - except Exception as _mon_err: - logger.debug("monitor_for_interrupt error (will retry): %s", _mon_err) - - interrupt_monitor = asyncio.create_task(monitor_for_interrupt()) - - # Periodic "still working" notifications so the user knows the agent hasn't died. Config: - # agent.gateway_notify_interval or HERMES_AGENT_NOTIFY_INTERVAL env; default 180s. - _NOTIFY_INTERVAL_RAW = _float_env("HERMES_AGENT_NOTIFY_INTERVAL", 180) - _NOTIFY_INTERVAL = _NOTIFY_INTERVAL_RAW if _NOTIFY_INTERVAL_RAW > 0 else None - _long_running_mode = _display_surface_mode( - "long_running_notifications", - default=True, - allow_generic=True, - ) - if _long_running_mode == "off": - _NOTIFY_INTERVAL = None - _notify_start = time.time() - - async def _notify_long_running(): - if _NOTIFY_INTERVAL is None: - return # Notifications disabled (gateway_notify_interval: 0) - _notify_adapter = self._adapter_for_source(source) - if not _notify_adapter: - return - # Track the heartbeat message id to edit in place where supported (Telegram, Discord, - # Slack, ...) instead of a new "Still working" bubble every interval. - _heartbeat_msg_id: Optional[str] = None - while True: - await asyncio.sleep(_NOTIFY_INTERVAL) - # Stop heartbeating once this run no longer owns the session slot or the executor has - # finished, else a stale "running: delegate_task" bubble outlives its run. _executor_task - # is bound just after this task is scheduled; tolerate the brief window before then. - try: - _exec_ref = _executor_task - except NameError: - _exec_ref = None - if not self._should_emit_long_running_notification( - session_key, agent_holder[0], _exec_ref - ): - break - _elapsed_mins = int((time.time() - _notify_start) // 60) - # Default heartbeat is terse (elapsed + current tool); the verbose iteration counter is - # gated on busy_ack_detail so users can opt in per platform. - _agent_ref = agent_holder[0] - _status_detail = "" - _want_iteration_detail = bool( - resolve_display_setting( - user_config, - platform_key, - "busy_ack_detail", - True, - ) - ) - if _agent_ref and hasattr(_agent_ref, "get_activity_summary"): - try: - _a = _agent_ref.get_activity_summary() - _parts = [] - if _want_iteration_detail: - _parts.append( - f"iteration {_a['api_call_count']}/{_a['max_iterations']}" - ) - _action = _a.get("current_tool") or _a.get("last_activity_desc") - if _action: - _parts.append(str(_action)) - if _parts: - _status_detail = " — " + ", ".join(_parts) - except Exception: - pass - _heartbeat_text = ( - _generic_status_phrase("status") - if _long_running_mode == "generic" - else f"⏳ Working — {_elapsed_mins} min{_status_detail}" - ) - try: - _notify_res = None - if _heartbeat_msg_id: - try: - _notify_res = await _notify_adapter.edit_message( - source.chat_id, - _heartbeat_msg_id, - _heartbeat_text, - ) - except Exception as _ee: - logger.debug("Heartbeat edit failed: %s", _ee) - _notify_res = None - if not (_notify_res and getattr(_notify_res, "success", False)): - _notify_res = await _notify_adapter.send( - source.chat_id, - _heartbeat_text, - metadata=_interim_metadata(_non_conversational_metadata(_status_thread_metadata, platform=source.platform)), - ) - if getattr(_notify_res, "success", False) and getattr( - _notify_res, "message_id", None - ): - _heartbeat_msg_id = str(_notify_res.message_id) - if _cleanup_progress: - _cleanup_msg_ids.append(_heartbeat_msg_id) - except Exception as _ne: - logger.debug("Long-running notification error: %s", _ne) - - _notify_task = asyncio.create_task(_notify_long_running()) - - def _stream_confirmed_final_delivery( - consumer, - final_text: str, - *, - previewed: bool = False, - ) -> bool: - """Return True only when the actual final reply reached the user.""" - if consumer is None: - return False - if getattr(consumer, "final_response_sent", False): - # A successful finalize call is not proof the *content* was final: the edit may carry - # only the last preview snapshot. Reconcile against the recorded turn-final payload: - # only a demonstrable mismatch (False, incl. payload-less split delivery) overrides - # the flag; None keeps legacy trust so timeout dedup isn't regressed. - matcher = getattr(consumer, "delivered_final_matches", None) - if callable(matcher): - try: - if matcher(final_text) is False: - return False - except Exception: - pass - return True - if previewed: - has_delivered_text = getattr(consumer, "has_delivered_text", None) - if callable(has_delivered_text): - try: - return bool(has_delivered_text(final_text)) - except Exception: - return False - return False - - try: - # Thread pool so we don't block. *Inactivity* timeout, not wall-clock: the agent may run for - # hours while actively calling tools / streaming, but a hung API call or stuck tool is killed. - # agent.gateway_timeout / HERMES_AGENT_TIMEOUT (env wins); default 1800s; 0 = unlimited. - _agent_timeout_raw = _float_env("HERMES_AGENT_TIMEOUT", 1800) - _agent_timeout = _agent_timeout_raw if _agent_timeout_raw > 0 else None - _agent_warning_raw = _float_env("HERMES_AGENT_TIMEOUT_WARNING", 900) - _agent_warning = _agent_warning_raw if _agent_warning_raw > 0 else None - _warning_fired = False - - # A background=true process intentionally survives a successful turn, so capture - # existing IDs and reap only children created by THIS turn if it times out. The daemon - # watchdog is independent of asyncio: cgroup memory reclaim can starve the loop that - # runs the normal timeout poll, and cleanup must not wait for the loop to recover. - from tools.process_registry import process_registry - - _turn_task_id = session_id or "" - _turn_process_baseline = process_registry.snapshot_running_ids(_turn_task_id) - turn_ctx.process_task_id = _turn_task_id - turn_ctx.process_baseline = _turn_process_baseline - _turn_worker_done = threading.Event() - _turn_timeout_fired = threading.Event() - _turn_cleanup_lock = threading.Lock() - # task_id is session-scoped, not turn-scoped: gate the eventual reap on this exact claim still - # being current, so a replacement turn on the same session that starts before the watchdog - # fires doesn't get its own fresh process killed by this turn's stale baseline. - _turn_run_generation = run_generation - _turn_is_current = ( - (lambda: self._is_session_run_current(session_key, _turn_run_generation)) - if _turn_run_generation is not None - else (lambda: True) - ) - - def _run_sync_with_timeout_lifecycle(): - try: - return run_sync() - finally: - _turn_worker_done.set() - # `.turn.agent` is only reset to _AGENT_PENDING_SENTINEL when the *next* turn is - # claimed, so this agent stays reachable from _interrupt_and_clear_session() - # until then. Clearing ownership markers the instant our worker finishes means a - # /stop on the finished turn no longer reaps background work it left running. - _finished_agent = agent_holder[0] if agent_holder else None - if _finished_agent is not None: - _finished_agent._gateway_turn_process_task_id = "" - _finished_agent._gateway_turn_process_baseline = frozenset() - - if _agent_timeout is not None: - threading.Thread( - target=_watch_gateway_turn_inactivity, - kwargs={ - "agent_holder": agent_holder, - "task_id": _turn_task_id, - "process_baseline": _turn_process_baseline, - "timeout": _agent_timeout, - "worker_done": _turn_worker_done, - "timeout_fired": _turn_timeout_fired, - "cleanup_lock": _turn_cleanup_lock, - "poll_interval": 5.0, - "is_still_current": _turn_is_current, - }, - name=f"gateway-turn-watchdog-{_turn_task_id[:12]}", - daemon=True, - ).start() - _executor_task = asyncio.ensure_future( - self._run_in_executor_with_context(_run_sync_with_timeout_lifecycle) - ) - - _inactivity_timeout = False - _POLL_INTERVAL = 5.0 - - if _agent_timeout is None: - # Unlimited — still poll periodically for backup interrupt - # detection in case monitor_for_interrupt() silently died. - response = None - while True: - done, _ = await asyncio.wait( - {_executor_task}, timeout=_POLL_INTERVAL - ) - if done: - response = _executor_task.result() - break - # Backup interrupt check: if the monitor task died or - # missed the interrupt, catch it here. - if not _interrupt_detected.is_set() and session_key: - _backup_adapter = self._adapter_for_source(source) - _backup_agent = agent_holder[0] - if (_backup_adapter and _backup_agent - and hasattr(_backup_adapter, 'has_pending_interrupt') - and _backup_adapter.has_pending_interrupt(session_key)): - _bp_event = _backup_adapter._pending_messages.get(session_key) - _bp_text = _bp_event.text if _bp_event else None - if _bp_event is not None: - _bp_media_urls = getattr(_bp_event, "media_urls", None) or [] - if self._pending_event_audio_paths(_bp_event): - _bp_text, _ = await self._transcribe_and_echo_pending_voice( - _bp_event, - _backup_adapter, - source, - _bp_text or "", - log_context="Voice-backup-interrupt", - metadata={"thread_id": source.thread_id} if source.thread_id else None, - ) - elif not _bp_text and _bp_media_urls: - _bp_text = _build_media_placeholder(_bp_event) - logger.info( - "Backup interrupt detected for session %s " - "(monitor task state: %s)", - session_key, - "done" if interrupt_monitor.done() else "running", - ) - _backup_agent.interrupt(_bp_text) - _interrupt_detected.set() - # Abort streaming TTS on barge-in (#60671). - _stts = streaming_tts_consumer_holder[0] - if _stts is not None: - _stts.abort("barge-in") - - else: - # Poll the agent's built-in activity tracker (updated by _touch_activity() on every tool - # call, API call, and stream delta) every few seconds. - response = None - while True: - done, _ = await asyncio.wait( - {_executor_task}, timeout=_POLL_INTERVAL - ) - if done: - # Prefer the real result when the worker finished even if the watchdog fired in - # the same window: the completed run already persisted its reply, so the "agent - # inactive" diagnostic would contradict the stored transcript. - response = _executor_task.result() - break - if _turn_timeout_fired.is_set(): - _inactivity_timeout = True - break - # Agent still running — check inactivity. - _agent_ref = agent_holder[0] - _idle_secs = 0.0 - if _agent_ref and hasattr(_agent_ref, "get_activity_summary"): - try: - _act = _agent_ref.get_activity_summary() - _idle_secs = _act.get("seconds_since_activity", 0.0) - except Exception: - pass - # Staged warning: fire once before escalating to full timeout. - if (not _warning_fired and _agent_warning is not None - and _idle_secs >= _agent_warning): - _warning_fired = True - _warn_adapter = self._adapter_for_source(source) - if _warn_adapter: - _elapsed_warn = int(_agent_warning // 60) or 1 - _remaining_mins = int((_agent_timeout - _agent_warning) // 60) or 1 - try: - await _warn_adapter.send( - source.chat_id, - f"⚠️ No activity for {_elapsed_warn} min. " - f"If the agent does not respond soon, it will " - f"be timed out in {_remaining_mins} min. " - f"You can continue waiting or use /reset.", - metadata=_interim_metadata(_status_thread_metadata), - ) - except Exception as _warn_err: - logger.debug("Inactivity warning send error: %s", _warn_err) - if _idle_secs >= _agent_timeout: - _inactivity_timeout = True - threading.Thread( - target=_abandon_timed_out_gateway_turn, - kwargs={ - "agent_holder": agent_holder, - "task_id": _turn_task_id, - "process_baseline": _turn_process_baseline, - "worker_done": _turn_worker_done, - "timeout_fired": _turn_timeout_fired, - "cleanup_lock": _turn_cleanup_lock, - "is_still_current": _turn_is_current, - }, - name=f"gateway-turn-reaper-{_turn_task_id[:12]}", - daemon=True, - ).start() - break - # Backup interrupt check (same as unlimited path). - if not _interrupt_detected.is_set() and session_key: - _backup_adapter = self._adapter_for_source(source) - _backup_agent = agent_holder[0] - if (_backup_adapter and _backup_agent - and hasattr(_backup_adapter, 'has_pending_interrupt') - and _backup_adapter.has_pending_interrupt(session_key)): - _bp_event = _backup_adapter._pending_messages.get(session_key) - _bp_text = _bp_event.text if _bp_event else None - if _bp_event is not None: - _bp_media_urls = getattr(_bp_event, "media_urls", None) or [] - if self._pending_event_audio_paths(_bp_event): - _bp_text, _ = await self._transcribe_and_echo_pending_voice( - _bp_event, - _backup_adapter, - source, - _bp_text or "", - log_context="Voice-backup-interrupt", - metadata={"thread_id": source.thread_id} if source.thread_id else None, - ) - elif not _bp_text and _bp_media_urls: - _bp_text = _build_media_placeholder(_bp_event) - logger.info( - "Backup interrupt detected for session %s " - "(monitor task state: %s)", - session_key, - "done" if interrupt_monitor.done() else "running", - ) - _backup_agent.interrupt(_bp_text) - _interrupt_detected.set() - # Abort streaming TTS on barge-in (#60671). - _stts = streaming_tts_consumer_holder[0] - if _stts is not None: - _stts.abort("barge-in") - - if _inactivity_timeout: - # Build a diagnostic summary from the agent's activity tracker. - _timed_out_agent = agent_holder[0] - _activity = {} - if _timed_out_agent and hasattr(_timed_out_agent, "get_activity_summary"): - with suppress(Exception): - _activity = _timed_out_agent.get_activity_summary() - - _last_desc = _activity.get("last_activity_desc", "unknown") - _secs_ago = _activity.get("seconds_since_activity", 0) - _cur_tool = _activity.get("current_tool") - _iter_n = _activity.get("api_call_count", 0) - _iter_max = _activity.get("max_iterations", 0) - - logger.error( - "Agent idle for %.0fs (timeout %.0fs) in session %s " - "| last_activity=%s | iteration=%s/%s | tool=%s", - _secs_ago, _agent_timeout, session_key, - _last_desc, _iter_n, _iter_max, - _cur_tool or "none", - ) - - # Interrupt the agent if it's still running so the thread - # pool worker is freed. - if _timed_out_agent: - request_hard_interrupt(_timed_out_agent, _INTERRUPT_REASON_TIMEOUT) - - _timeout_mins = int(_agent_timeout // 60) or 1 - - # Construct a user-facing message with diagnostic context. - _diag_lines = [ - f"⏱️ Agent inactive for {_timeout_mins} min — no tool calls " - f"or API responses." - ] - if _cur_tool: - _diag_lines.append( - f"The agent appears stuck on tool `{_cur_tool}` " - f"({_secs_ago:.0f}s since last activity, " - f"iteration {_iter_n}/{_iter_max})." - ) - else: - _diag_lines.append( - f"Last activity: {_last_desc} ({_secs_ago:.0f}s ago, " - f"iteration {_iter_n}/{_iter_max}). " - "The agent may have been waiting on an API response." - ) - _diag_lines.append( - "To increase the limit, set agent.gateway_timeout in config.yaml " - "(value in seconds, 0 = no limit) and restart the gateway.\n" - "Try again, or use /reset to start fresh." - ) - - response = { - "final_response": "\n".join(_diag_lines), - "messages": result_holder[0].get("messages", []) if result_holder[0] else [], - "api_calls": _iter_n, - "tools": tools_holder[0] or [], - "history_offset": 0, - "failed": True, - } - - # Persist fallback-model switches so /model shows the actually-active model. Skip - # eviction when the run failed — evicting forces MCP reinit on the next message for no - # benefit (bad model → fallback → evict → recreate → same 400 loop burning CPU). - _agent = agent_holder[0] - _result_for_fb = result_holder[0] - _run_failed = _result_for_fb.get("failed") if _result_for_fb else False - if _agent is not None and hasattr(_agent, 'model') and not _run_failed: - _cfg_model = _resolve_gateway_model() - # Normalize _cfg_model as AIAgent.__init__ does so a vendor-prefixed config value - # matches the agent's stripped model on native providers — otherwise the cached agent - # is evicted every turn, destroying prompt caching. Aggregators keep the vendor slug. - try: - from hermes_cli.model_normalize import ( - _AGGREGATOR_PROVIDERS, - normalize_model_for_provider, - ) - _agent_provider = getattr(_agent, 'provider', '') or '' - if _agent_provider and _agent_provider not in _AGGREGATOR_PROVIDERS: - _cfg_model = normalize_model_for_provider(_cfg_model, _agent_provider) - except Exception: - pass - if _agent.model != _cfg_model and not self._is_intentional_model_switch(session_key, _agent.model): - # Fallback activated on a successful run — evict cached - # agent so the next message retries the primary model. - self._evict_cached_agent(session_key) - - # Check if we were interrupted OR have a queued message (/queue). - result = result_holder[0] - adapter = self._adapter_for_source(source) - - # Finalize the streaming-TTS consumer. finish() runs on the outer event-loop thread so - # early returns from run_sync are also finalised. wait_complete() drains queued audio; - # on timeout abort unconditionally — if audio was audible keep suppression (no replay - # from the start); if not, the whole-file fallback is permitted. - _stts = streaming_tts_consumer_holder[0] - if _stts is not None: - _stts.finish() - try: - await _stts.wait_complete(timeout=10.0) - except Exception as _stts_done_err: - logger.debug("streaming TTS wait_complete error: %s", _stts_done_err) - if not _stts.done: - # Timeout before or after audible audio: abort to free the consumer task. Audible - # streams retain suppression; silent streams stay eligible for whole-file fallback. - _stts.abort("streaming TTS finalisation timeout") - await _stts.wait_complete(timeout=2.0) - if _stts.suppress_whole_file and adapter is not None: - _mark_turn = getattr(adapter, "_mark_streaming_tts_completed_turn", None) - if callable(_mark_turn): - _mark_turn(session_key, run_generation) - - # Get pending message from adapter. - # Use session_key (not source.chat_id) to match adapter's storage keys. - pending_event = None - pending = None - if result and adapter and session_key: - pending_event = _dequeue_pending_event(adapter, session_key) - # /queue overflow: after consuming the adapter's "next-up" slot, promote the next - # queued event into it so the recursive run's drain will see it. Keeping the slot - # occupied for the whole FIFO chain preserves order and makes a mid-chain /queue - # route to overflow instead of jumping the queue. - pending_event = self._promote_queued_event(session_key, adapter, pending_event) - if result.get("interrupted") and not pending_event and result.get("interrupt_message"): - interrupt_message = result.get("interrupt_message") - if _is_control_interrupt_message(interrupt_message): - logger.info( - "Ignoring control interrupt message for session %s: %s", - session_key or "?", - interrupt_message, - ) - else: - pending = interrupt_message - elif pending_event: - # Transcribe audio on the dequeued event BEFORE it becomes the next user turn, so - # queued/interrupting voice messages drain with the real transcript, not a file path. - _pending_text = pending_event.text or "" - _media_urls = getattr(pending_event, "media_urls", None) or [] - if self._pending_event_audio_paths(pending_event): - pending, _ = await self._transcribe_and_echo_pending_voice( - pending_event, - adapter, - source, - _pending_text, - log_context="Voice-drain", - metadata={"thread_id": source.thread_id} if source.thread_id else None, - ) - if not pending: - pending = _build_media_placeholder(pending_event) - else: - pending = _pending_text or _build_media_placeholder(pending_event) - if pending: - logger.debug("Processing queued message after agent completion: '%s...'", pending[:40]) - - # Leftover /steer: a steer arriving after the last tool batch (e.g. during the final API - # call) comes back in result["pending_steer"]; deliver it as the next user turn, not drop it. - if result and not pending and not pending_event: - _leftover_steer = result.get("pending_steer") - if _leftover_steer: - pending = _leftover_steer - logger.debug("Delivering leftover /steer as next turn: '%s...'", pending[:40]) - - # Safety net: if the pending text is a slash command (e.g. "/stop", "/new"), discard it - # — commands should never be passed to the agent as user input. - if pending and pending.strip().startswith("/"): - _pending_parts = pending.strip().split(None, 1) - _pending_cmd_word = _pending_parts[0][1:].lower() if _pending_parts else "" - if _pending_cmd_word: - try: - from hermes_cli.commands import resolve_command as _rc_pending - if _rc_pending(_pending_cmd_word): - logger.info( - "Discarding command '/%s' from pending queue — " - "commands must not be passed as agent input", - _pending_cmd_word, - ) - pending_event = None - pending = None - except Exception: - pass - - if self._draining and (pending_event or pending): - logger.info( - "Discarding pending follow-up for session %s during gateway %s", - session_key or "?", - self._status_action_label(), - ) - pending_event = None - pending = None - - if pending_event or pending: - logger.debug("Processing pending message: '%s...'", pending[:40]) - - # Clear the adapter's interrupt event so the next _run_agent call doesn't re-trigger the - # interrupt before the new agent's first API call (infinite loop otherwise). - if adapter and hasattr(adapter, '_active_sessions') and session_key and session_key in adapter._active_sessions: - adapter._active_sessions[session_key].clear() - - # Cap recursion depth to prevent resource exhaustion when the - # user sends multiple messages while the agent keeps failing. (#816) - if _interrupt_depth >= self._MAX_INTERRUPT_DEPTH: - logger.warning( - "Interrupt recursion depth %d reached for session %s — " - "queueing message instead of recursing.", - _interrupt_depth, session_key, - ) - adapter = self._adapter_for_source(source) - if adapter and pending_event: - merge_pending_message_event(adapter._pending_messages, session_key, pending_event) - elif adapter and hasattr(adapter, 'queue_message'): - adapter.queue_message(session_key, pending) - return result_holder[0] or {"final_response": response, "messages": history} - - was_interrupted = result.get("interrupted") - if not was_interrupted: - # Queued message after normal completion: deliver the first response before the - # queued follow-up, unless streaming already delivered it. - _sc = stream_consumer_holder[0] - if _sc and stream_task: - try: - await asyncio.wait_for(stream_task, timeout=5.0) - except (asyncio.TimeoutError, asyncio.CancelledError): - stream_task.cancel() - with suppress(asyncio.CancelledError): - await stream_task - except Exception as e: - logger.debug("Stream consumer wait before queued message failed: %s", e) - # The queued branch needs raw ``result`` for interruption, history, and - # recursion state, but delivery must use the finalized task result — it carries - # empty/failure normalization and final-response processing from _run_agent_task. - _delivery_result = response if isinstance(response, dict) else (result or {}) - _previewed = bool(_delivery_result.get("response_previewed")) - first_response = _delivery_result.get("final_response", "") - _already_streamed = _stream_confirmed_final_delivery( - _sc, - first_response, - previewed=_previewed, - ) - # Same predicate as the normal completed-turn path: this direct queued-send branch - # predates intentional-silence filtering and would leak the literal marker. - try: - from gateway.response_filters import is_intentional_silence_agent_result - _intentional_silence = is_intentional_silence_agent_result( - _delivery_result, first_response, - ) - except Exception: - _intentional_silence = False - if _intentional_silence: - logger.info( - "Queued follow-up for session %s: suppressing intentional silence marker before continuing.", - session_key or "?", - ) - elif first_response: - try: - if _already_streamed: - logger.info( - "Queued follow-up for session %s: final text delivery confirmed; delivering explicit media before continuing.", - session_key or "?", - ) - else: - logger.info( - "Queued follow-up for session %s: final stream delivery not confirmed; sending first response before continuing.", - session_key or "?", - ) - await self._deliver_queued_first_response( - first_response, - source=source, - adapter=adapter, - metadata=_status_thread_metadata, - event_message_id=event_message_id, - text_already_delivered=_already_streamed, - deliver_media=not _delivery_result.get("failed"), - stream_consumer=_sc, - ) - except Exception as e: - logger.warning("Failed to send first response before queued message: %s", e) - # Release deferred bg-review notifications now that the first response is delivered: - # pop from the adapter's callback dict (no double-fire in base.py's finally) and call. - if getattr(type(adapter), "pop_post_delivery_callback", None) is not None: - _bg_cb = adapter.pop_post_delivery_callback( - session_key, - generation=run_generation, - ) - if callable(_bg_cb): - try: - _bg_result = _bg_cb() - if inspect.isawaitable(_bg_result): - await _bg_result - except Exception: - pass - elif adapter and hasattr(adapter, "_post_delivery_callbacks"): - _bg_cb = adapter._post_delivery_callbacks.pop(session_key, None) - if callable(_bg_cb): - try: - _bg_result = _bg_cb() - if inspect.isawaitable(_bg_result): - await _bg_result - except Exception: - pass - # else: interrupted — discard the response ("Operation interrupted." is noise; the user - # knows they sent a new message). - - updated_history = result.get("messages", history) - next_source = source - next_message = pending - next_message_id = None - next_channel_prompt = None - next_session_key = session_key - # Carry the pending event's message_type into the recursive call so queued voice turns - # can stream TTS and re-mark the generation for the final delivered turn. - next_message_type = None - if pending_event is not None: - next_source = getattr(pending_event, "source", None) or source - if self._is_goal_continuation_event(pending_event) and not self._goal_still_active_for_session(session_id): - logger.info( - "Discarding stale goal continuation for session %s — goal is no longer active", - session_key or "?", - ) - return result - # Resolve the follow-up's session key BEFORE preparing the inbound text: - # _prepare_inbound_message_text buffers native image paths under the key given, and - # the recursive _run_agent consumes them under next_session_key — mismatch drops them. - try: - next_session_key = self._session_key_for_source(next_source) - except Exception: - logger.debug( - "Queued follow-up session-key resolution failed; reusing %s", - session_key or "?", - exc_info=True, - ) - next_message = await self._prepare_profile_scoped_inbound_message_text( - event=pending_event, - source=next_source, - history=updated_history, - session_key=next_session_key, - ) - if next_message is None: - return result - next_message_id = self._reply_anchor_for_event(pending_event) - next_channel_prompt = getattr(pending_event, "channel_prompt", None) - next_message_type = getattr(pending_event, "message_type", None) - - # Clear the prior logical turn's completed streaming marker so the recursive turn's - # streaming TTS isn't suppressed by that completion. - _clear_adapter = self._adapter_for_source(source) - if _clear_adapter is not None and session_key and run_generation is not None: - _completed_turns = getattr(_clear_adapter, "_streaming_tts_completed_turns", None) - if _completed_turns is not None: - _prior_key = getattr(_clear_adapter, "_streaming_tts_turn_key", None) - if callable(_prior_key): - _pk = _prior_key(session_key, run_generation) - if _pk: - _completed_turns.discard(_pk) - - # Restart the typing indicator for the follow-up turn; the outer - # _process_message_background typing task is alive but may be stale. - _followup_adapter = self._adapter_for_source(source) - if _followup_adapter: - with suppress(Exception): - await _followup_adapter.send_typing( - source.chat_id, - metadata=_status_thread_metadata, - ) - - # Re-baseline the cached agent's message_count before recursing into the /queue follow-up: - # the coherence guard would otherwise rebuild on OUR OWN flushed rows and destroy the - # prompt-cache prefix; _handle_message_with_agent re-baselines only after the chain ends. - await self._refresh_agent_cache_message_count(session_key, session_id) - - followup_result = await self._run_agent( - message=next_message, - context_prompt=context_prompt, - history=updated_history, - source=next_source, - session_id=session_id, - session_key=next_session_key, - run_generation=run_generation, - _interrupt_depth=_interrupt_depth + 1, - event_message_id=next_message_id, - channel_prompt=next_channel_prompt, - message_type=next_message_type, - ) - return _preserve_queued_followup_history_offset(result, followup_result) - finally: - # Stop progress sender, interrupt monitor, and notification task - if progress_task: - progress_task.cancel() - if log_task: - log_task.cancel() - interrupt_monitor.cancel() - _notify_task.cancel() - - # Wait for stream consumer to finish its final edit - if stream_task: - # If the agent never created a stream consumer (non-streaming path, or a test stub - # returning synchronously) there is nothing to flush — cancel now instead of waiting - # out the 5s timeout polling for a consumer that will never arrive. - _has_stream_consumer = ( - stream_consumer_holder - and stream_consumer_holder[0] is not None - ) - if not _has_stream_consumer: - stream_task.cancel() - with suppress(asyncio.CancelledError): - await stream_task - else: - try: - await asyncio.wait_for(stream_task, timeout=5.0) - except (asyncio.TimeoutError, asyncio.CancelledError): - stream_task.cancel() - with suppress(asyncio.CancelledError): - await stream_task - - # Unconditional abort + bounded wait for the streaming-TTS consumer: covers cancellation / - # exception paths where the normal finalisation block was skipped. - _stts_finally = streaming_tts_consumer_holder[0] - if _stts_finally is not None and not _stts_finally.done: - _stts_finally.abort("cleanup") - with suppress(Exception): - await _stts_finally.wait_complete(timeout=2.0) - - # Clean up tracking - tracking_task.cancel() - if session_key: - # Release the slot only if this run's generation still owns it: a /stop or /new that - # bumped the generation while we unwound already installed its own state; keep it. - self._release_running_agent_state( - session_key, run_generation=run_generation - ) - if self._draining: - self._update_runtime_status("draining") - - # Wait for cancelled tasks - for task in [progress_task, log_task, interrupt_monitor, tracking_task, _notify_task]: - if task: - try: - await task - except asyncio.CancelledError: - pass - except Exception: - # A background task that died of a non-cancellation error (transport drop in - # a progress/card publish) must not abort the cleanup path — everything - # after this loop (final-delivery bookkeeping) still runs (review B7). - logger.debug( - "background turn task failed during cleanup", - exc_info=True, - ) - - # If streaming already delivered the response, skip the caller's send() — but never when the - # agent failed (the error is unseen content) or on "(empty)": interim text ("Let me search…") - # set already_sent but is NOT the final answer; suppressing would leave the user with silence. - _sc = stream_consumer_holder[0] - if isinstance(response, dict) and not response.get("failed"): - _final = response.get("final_response") or "" - _is_empty_sentinel = not _final or _final == "(empty)" - # response_previewed means interim_assistant_callback already saw the final text, but only - # suppress the send if that exact text was delivered — unrelated commentary/progress isn't it. - _previewed = bool(response.get("response_previewed")) - _content_delivered = bool( - _sc and getattr(_sc, "final_content_delivered", False) - ) - # A *successful* finalize edit can still carry only the last preview snapshot, and both - # suppression flags reflect call success, not content. Reconcile against the recorded - # turn-final payload: on mismatch (False, incl. payload-less split delivery) neither flag - # may suppress the final send; None (no record) keeps legacy trust. - _stale_finalized = False - if _content_delivered and not _is_empty_sentinel: - _matcher = getattr(_sc, "delivered_final_matches", None) - if callable(_matcher): - try: - _stale_finalized = _matcher(_final) is False - except Exception: - _stale_finalized = False - if _stale_finalized: - _content_delivered = False - # Plugin hooks (e.g. transform_llm_output) may append content after streaming finished — when - # transformed, always send the final version so the appended content reaches the client. - _transformed = bool(response.get("response_transformed")) - # Suppress the normal send only when the actual final reply reached the user (streamed, or - # interim preview of that *exact* text); commentary shown during a compression/split isn't it. - _streamed = _stream_confirmed_final_delivery( - _sc, - _final, - previewed=_previewed, - ) - if not _is_empty_sentinel and not _transformed and (_streamed or _content_delivered): - logger.info( - "Suppressing normal final send for session %s: final delivery already confirmed (streamed=%s previewed=%s content_delivered=%s).", - session_key or "?", - _streamed, - _previewed, - _content_delivered, - ) - response["already_sent"] = True - elif not _is_empty_sentinel and not _transformed and _stale_finalized and _sc is not None: - # Stale finalize: the streamed message holds only the last preview snapshot. Edit it - # up to the complete response; on edit failure leave already_sent unset so the normal - # send delivers. Not for split delivery: message_id is only the LAST chunk, so editing - # it would repeat every sealed head chunk — fall through to the normal send. - _sc_msg_id = _sc.message_id - _sc_adapter = getattr(_sc, "adapter", None) - if getattr(_sc, "_turn_split_delivery", False): - logger.info( - "Stale streamed finalize detected for session %s on a multi-message split; skipping the in-place reconciliation edit and delivering the complete response via normal final send (#78541).", - session_key or "?", - ) - elif _sc_msg_id and _sc_msg_id != "__no_edit__" and _sc_adapter is not None: - try: - _reconcile_res = await _sc_adapter.edit_message( - chat_id=source.chat_id, - message_id=_sc_msg_id, - content=_final, - finalize=True, - ) - if getattr(_reconcile_res, "success", True): - response["already_sent"] = True - logger.info( - "Reconciled stale streamed finalize for session %s: edited message %s with the complete response (#71643).", - session_key or "?", _sc_msg_id, - ) - else: - logger.warning( - "Stale-finalize reconciliation edit failed for session %s (%s); sending complete response via normal final send.", - session_key or "?", - getattr(_reconcile_res, "error", None), - ) - except Exception as _edit_err: - logger.warning( - "Stale-finalize reconciliation edit failed for session %s: %s; sending complete response via normal final send.", - session_key or "?", _edit_err, - ) - else: - logger.info( - "Stale streamed finalize detected for session %s with no editable message; delivering complete response via normal final send (#71643).", - session_key or "?", - ) - elif not _is_empty_sentinel and _transformed and _sc is not None: - # Plugin hooks transformed the response after streaming — edit the - # existing streamed message instead of sending a duplicate. - _sc_msg_id = _sc.message_id - if _sc_msg_id: - try: - await _sc.adapter.edit_message( - chat_id=source.chat_id, - message_id=_sc_msg_id, - content=response["final_response"], - finalize=True, - ) - response["already_sent"] = True - logger.info( - "Edited streamed message %s for session %s to include plugin-transformed content.", - _sc_msg_id, session_key or "?", - ) - except Exception as _edit_err: - logger.warning( - "Failed to edit streamed message for session %s: %s", - session_key or "?", _edit_err, - ) - elif _sc is not None and not _is_empty_sentinel: - # DUPLICATE-RISK DIAGNOSTIC: a stream consumer existed for this turn but suppression - # did NOT fire, so the gateway's normal final-send is about to run. Log the decision - # inputs so a recurrence can be pinned to "signal never set" vs "ack-pending race". - logger.warning( - "Normal final-send NOT suppressed despite active stream " - "consumer for session %s: streamed=%s previewed=%s " - "content_delivered=%s transformed=%s final_len=%d — " - "possible duplicate send (see wecom ack-timeout RCA).", - session_key or "?", - _streamed, - _previewed, - _content_delivered, - _transformed, - len(_final), - ) - - # Schedule deletion of tracked temporary progress bubbles after the final response lands; failed - # runs keep them as breadcrumbs. Only on adapters with ``delete_message``; failures swallowed. - if ( - _cleanup_progress - and _cleanup_adapter is not None - and _cleanup_msg_ids - and session_key - and isinstance(response, dict) - and not response.get("failed") - and hasattr(_cleanup_adapter, "register_post_delivery_callback") - ): - _ids_snapshot = list(_cleanup_msg_ids) - _chat_id_snapshot = source.chat_id - _adapter_snapshot = _cleanup_adapter - _loop_snapshot = asyncio.get_running_loop() - - def _cleanup_temp_bubbles() -> None: - async def _delete_all() -> None: - for _mid in _ids_snapshot: - with suppress(Exception): - await _adapter_snapshot.delete_message( - _chat_id_snapshot, _mid - ) - with suppress(Exception): - safe_schedule_threadsafe( - _delete_all(), _loop_snapshot, - logger=logger, - log_message="Temp bubble cleanup scheduling error", - ) - - try: - _cleanup_adapter.register_post_delivery_callback( - session_key, - _cleanup_temp_bubbles, - generation=run_generation, - ) - except Exception as _rpe: - logger.debug("Post-delivery cleanup registration failed: %s", _rpe) - - return response + @dataclasses.dataclass + class _RunAgentDisplay: + """Per-turn display / progress settings resolved by ``_run_agent_display_settings``.""" + + user_config: Any = None + platform_key: Any = None + enabled_toolsets: Any = None + disabled_toolsets: Any = None + resolve_display_setting: Any = None + progress_mode: Any = None + progress_grouping: Any = None + _display_surface_mode: Any = None + tool_progress_enabled: Any = None + _live_status_mode: Any = None + _live_status_adapter: Any = None + log_mode_enabled: Any = None + log_queue: Any = None + interim_assistant_messages_enabled: Any = None + _thinking_enabled: Any = None + _native_slack_task_cards: Any = None + needs_progress_queue: Any = None + _generic_status_phrase: Any = None + + @dataclasses.dataclass + class _RunAgentWorker: + """Executor future + inactivity-watchdog handles for one ``_run_agent_inner`` turn.""" + + executor_task: Any = None + agent_timeout: Optional[float] = None + agent_warning: Optional[float] = None + task_id: str = "" + process_baseline: Any = None + worker_done: Any = None + timeout_fired: Any = None + cleanup_lock: Any = None + is_current: Any = None def _run_planned_stop_watcher( @@ -29185,194 +6044,175 @@ def _looks_like_profile_conflict_from_cmdline(command: str, our_home) -> bool: return bool(home_value is not None and os.path.normcase(os.path.normpath(home_value)) != os.path.normcase(os.path.normpath(str(our_home)))) -async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = False, verbosity: Optional[int] = 0) -> bool: - """Start the gateway and run until interrupted. +async def _start_gateway_replace_existing_instance(existing_pid: int, replace: bool) -> bool: + """Handle a live gateway PID under this HERMES_HOME: replace it (``--replace``) or refuse. - Returns True if the gateway ran, False if it failed to start (non-zero exit so systemd can - auto-restart). ``replace`` kills any existing instance first — avoids systemd restart-loop - deadlocks when the previous process hasn't fully exited. + Returns False when startup must abort (refused, permission denied, target still alive). """ - # Enable interactive exec approval on messaging platforms. Set here (not at module import) so - # incidental imports of gateway.run from CLI/tool code don't poison HERMES_EXEC_ASK. - os.environ["HERMES_EXEC_ASK"] = "1" - - from hermes_cli.resource_limits import apply_nofile_soft_limit - - apply_nofile_soft_limit() - - # Snapshot the checkout revision now, while sys.modules still matches disk, so a later `git - # pull` under this long-lived process can be detected (and risky work like model switching - # refused) instead of crashing on a stale in-memory module. - from gateway.code_skew import record_boot_fingerprint - record_boot_fingerprint() - - # Duplicate-instance guard: no two gateways under one HERMES_HOME. The PID file is scoped to - # HERMES_HOME, so multi-profile setups (distinct HERMES_HOME each) run concurrently untripped. from gateway.status import ( - acquire_gateway_runtime_lock, - get_running_pid, get_process_start_time, - release_gateway_runtime_lock, remove_pid_file, terminate_pid, ) - existing_pid = get_running_pid() - if existing_pid is not None and existing_pid != os.getpid(): - if replace: - # Cross-profile ownership gate: never signal a live process we cannot prove belongs to - # this HERMES_HOME. A poisoned PID record steering --replace at another profile's - # gateway is exactly the restart-loop shape this flow must not allow. - if _replace_target_belongs_to_other_profile(existing_pid): - from gateway.status import _get_process_hermes_home + if replace: + # Cross-profile ownership gate: never signal a live process we cannot prove belongs to + # this HERMES_HOME. A poisoned PID record steering --replace at another profile's + # gateway is exactly the restart-loop shape this flow must not allow. + if _replace_target_belongs_to_other_profile(existing_pid): + from gateway.status import _get_process_hermes_home - logger.error( - "Refusing --replace: PID %d cannot be proven to belong " - "to this profile's gateway (HERMES_HOME %s). Remove the " - "stale PID record or stop the owning profile explicitly.", - existing_pid, - _get_process_hermes_home(), - ) - return False - existing_start_time = get_process_start_time(existing_pid) - logger.info( - "Replacing existing gateway instance (PID %d) with --replace.", + logger.error( + "Refusing --replace: PID %d cannot be proven to belong " + "to this profile's gateway (HERMES_HOME %s). Remove the " + "stale PID record or stop the owning profile explicitly.", + existing_pid, + _get_process_hermes_home(), + ) + return False + existing_start_time = get_process_start_time(existing_pid) + logger.info( + "Replacing existing gateway instance (PID %d) with --replace.", + existing_pid, + ) + # Record a takeover marker so the target's shutdown handler recognises its SIGTERM as a + # planned takeover and exits 0 (rather than exit 1, which would trigger systemd's + # Restart=on-failure and start a flap loop against us). Best-effort — proceed on failure. + try: + from gateway.status import write_takeover_marker + write_takeover_marker(existing_pid) + except Exception as e: + logger.debug("Could not write takeover marker: %s", e) + # Snapshot the old gateway's children BEFORE signalling it: once it exits, orphans are + # reparented and invisible to a parent walk. On POSIX, surviving adapter subprocesses hold + # scoped token locks and block the replacement (Windows already tree-kills). Best-effort. + try: + from gateway.status import _snapshot_gateway_children + _old_gateway_children = _snapshot_gateway_children(existing_pid) + except Exception: + _old_gateway_children = [] + try: + terminate_pid(existing_pid, force=False) + except ProcessLookupError: + pass # Already gone + except (PermissionError, OSError): + logger.error( + "Permission denied killing PID %d. Cannot replace.", existing_pid, ) - # Record a takeover marker so the target's shutdown handler recognises its SIGTERM as a - # planned takeover and exits 0 (rather than exit 1, which would trigger systemd's - # Restart=on-failure and start a flap loop against us). Best-effort — proceed on failure. + # Marker is scoped to a specific target; clean it up on + # give-up so it doesn't grief an unrelated future shutdown. try: - from gateway.status import write_takeover_marker - write_takeover_marker(existing_pid) - except Exception as e: - logger.debug("Could not write takeover marker: %s", e) - # Snapshot the old gateway's children BEFORE signalling it: once it exits, orphans are - # reparented and invisible to a parent walk. On POSIX, surviving adapter subprocesses hold - # scoped token locks and block the replacement (Windows already tree-kills). Best-effort. - try: - from gateway.status import _snapshot_gateway_children - _old_gateway_children = _snapshot_gateway_children(existing_pid) + from gateway.status import clear_takeover_marker + clear_takeover_marker() except Exception: - _old_gateway_children = [] + pass + return False + # Wait up to 10s for the old process to exit. ``os.kill(pid, 0)`` on Windows is NOT a no-op — + # use the handle-based existence check instead. + from gateway.status import _pid_exists + old_gateway_exited = False + for _ in range(20): + if not _pid_exists(existing_pid): + old_gateway_exited = True + break # Process is gone + # start_gateway is async: a blocking sleep here freezes the event loop (signal handlers, + # health checks, every coroutine) for up to 10s per replacement. + await asyncio.sleep(0.5) + else: + # Still alive after 10s — force kill + logger.warning( + "Old gateway (PID %d) did not exit after SIGTERM, sending SIGKILL.", + existing_pid, + ) try: - terminate_pid(existing_pid, force=False) + terminate_pid( + existing_pid, + force=True, + expected_start_time=existing_start_time, + ) except ProcessLookupError: - pass # Already gone + old_gateway_exited = True except (PermissionError, OSError): + pass + # Confirm the force-kill actually reaped the process before clearing its PID file / + # scoped locks: SIGKILL can fail to take (uninterruptible sleep, zombie), and blindly + # clearing metadata would leave two live gateways fighting over the same token. + if not old_gateway_exited: + for _ in range(20): + if not _pid_exists(existing_pid): + old_gateway_exited = True + break + # Async context — never block the loop (#36163). + await asyncio.sleep(0.25) + if not old_gateway_exited: logger.error( - "Permission denied killing PID %d. Cannot replace.", + "Old gateway (PID %d) still appears alive after SIGKILL; " + "aborting replacement to avoid a duplicate gateway.", existing_pid, ) - # Marker is scoped to a specific target; clean it up on - # give-up so it doesn't grief an unrelated future shutdown. try: from gateway.status import clear_takeover_marker clear_takeover_marker() except Exception: pass return False - # Wait up to 10s for the old process to exit. ``os.kill(pid, 0)`` on Windows is NOT a no-op — - # use the handle-based existence check instead. - from gateway.status import _pid_exists - old_gateway_exited = False - for _ in range(20): - if not _pid_exists(existing_pid): - old_gateway_exited = True - break # Process is gone - # start_gateway is async: a blocking sleep here freezes the event loop (signal handlers, - # health checks, every coroutine) for up to 10s per replacement. - await asyncio.sleep(0.5) - else: - # Still alive after 10s — force kill - logger.warning( - "Old gateway (PID %d) did not exit after SIGTERM, sending SIGKILL.", - existing_pid, - ) - try: - terminate_pid( - existing_pid, - force=True, - expected_start_time=existing_start_time, - ) - except ProcessLookupError: - old_gateway_exited = True - except (PermissionError, OSError): - pass - # Confirm the force-kill actually reaped the process before clearing its PID file / - # scoped locks: SIGKILL can fail to take (uninterruptible sleep, zombie), and blindly - # clearing metadata would leave two live gateways fighting over the same token. - if not old_gateway_exited: - for _ in range(20): - if not _pid_exists(existing_pid): - old_gateway_exited = True - break - # Async context — never block the loop (#36163). - await asyncio.sleep(0.25) - if not old_gateway_exited: - logger.error( - "Old gateway (PID %d) still appears alive after SIGKILL; " - "aborting replacement to avoid a duplicate gateway.", - existing_pid, - ) - try: - from gateway.status import clear_takeover_marker - clear_takeover_marker() - except Exception: - pass - return False - # Old gateway confirmed dead — reap any orphaned child processes it left behind (POSIX; - # mirrors Windows taskkill /T tree-kill). Orphaned adapter subprocesses would otherwise - # keep holding scoped token locks against us. Best-effort, never raises. - try: - from gateway.status import reap_gateway_children - reap_gateway_children( - _old_gateway_children, parent_pid=existing_pid - ) - except Exception: - logger.debug( - "Child reap for replaced gateway PID %d failed", - existing_pid, - exc_info=True, - ) - remove_pid_file() - # remove_pid_file() is a no-op when the PID doesn't match. - # Force-unlink to cover the old-process-crashed case. - with suppress(Exception): - (get_hermes_home() / "gateway.pid").unlink(missing_ok=True) - # Clean up any takeover marker the old process didn't consume - # (e.g. SIGKILL'd before its shutdown handler could read it). - try: - from gateway.status import clear_takeover_marker - clear_takeover_marker() - except Exception: - pass - # Release all scoped locks left by the old process: stopped (Ctrl+Z) processes don't release - # locks on exit, leaving stale lock files that block the new gateway. - try: - from gateway.status import release_all_scoped_locks - _released = release_all_scoped_locks( - owner_pid=existing_pid, - owner_start_time=existing_start_time, - ) - if _released: - logger.info("Released %d stale scoped lock(s) from old gateway.", _released) - except Exception: - pass - else: - hermes_home = str(get_hermes_home()) - logger.error( - "Another gateway instance is already running (PID %d, HERMES_HOME=%s). " - "Use 'hermes gateway restart' to replace it, or 'hermes gateway stop' first.", - existing_pid, hermes_home, + # Old gateway confirmed dead — reap any orphaned child processes it left behind (POSIX; + # mirrors Windows taskkill /T tree-kill). Orphaned adapter subprocesses would otherwise + # keep holding scoped token locks against us. Best-effort, never raises. + try: + from gateway.status import reap_gateway_children + reap_gateway_children( + _old_gateway_children, parent_pid=existing_pid ) - print( - f"\n❌ Gateway already running (PID {existing_pid}).\n" - f" Use 'hermes gateway restart' to replace it,\n" - f" or 'hermes gateway stop' to kill it first.\n" - f" Or use 'hermes gateway run --replace' to auto-replace.\n" + except Exception: + logger.debug( + "Child reap for replaced gateway PID %d failed", + existing_pid, + exc_info=True, ) - return False + remove_pid_file() + # remove_pid_file() is a no-op when the PID doesn't match. + # Force-unlink to cover the old-process-crashed case. + with suppress(Exception): + (get_hermes_home() / "gateway.pid").unlink(missing_ok=True) + # Clean up any takeover marker the old process didn't consume + # (e.g. SIGKILL'd before its shutdown handler could read it). + try: + from gateway.status import clear_takeover_marker + clear_takeover_marker() + except Exception: + pass + # Release all scoped locks left by the old process: stopped (Ctrl+Z) processes don't release + # locks on exit, leaving stale lock files that block the new gateway. + try: + from gateway.status import release_all_scoped_locks + _released = release_all_scoped_locks( + owner_pid=existing_pid, + owner_start_time=existing_start_time, + ) + if _released: + logger.info("Released %d stale scoped lock(s) from old gateway.", _released) + except Exception: + pass + else: + hermes_home = str(get_hermes_home()) + logger.error( + "Another gateway instance is already running (PID %d, HERMES_HOME=%s). " + "Use 'hermes gateway restart' to replace it, or 'hermes gateway stop' first.", + existing_pid, hermes_home, + ) + print( + f"\n❌ Gateway already running (PID {existing_pid}).\n" + f" Use 'hermes gateway restart' to replace it,\n" + f" or 'hermes gateway stop' to kill it first.\n" + f" Or use 'hermes gateway run --replace' to auto-replace.\n" + ) + return False + return True + +def _start_gateway_configure_logging(verbosity: Optional[int]) -> None: + """Sync bundled skills, set up file logging + startup security audit, and the -v/-q stderr handler.""" # Sync bundled skills on gateway start (fast -- skips unchanged) try: from tools.skills_sync import sync_skills @@ -29416,21 +6256,10 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = if _stderr_level < logging.getLogger().level: logging.getLogger().setLevel(_stderr_level) - runner = GatewayRunner(config) - # Multiplex: swap the launch-home file handlers for per-profile routers so each profile's records - # land in its own logs/. Must run after the runner resolved (possibly None) config and setup_logging. - _enable_multiplex_log_routing(runner.config) - # ``--replace`` is explicit startup authority, not a durable reconnect policy: GatewayRunner scopes - # it to cold adapter connects and clears it before the background reconnect watcher starts. - runner._platform_lock_takeover_on_start = bool(replace) - # Track whether an unexpected signal initiated shutdown: an unexpected SIGTERM exits non-zero so - # service managers revive us; planned stop paths write a marker first so they exit cleanly. - _signal_initiated_shutdown = False - - # Set up signal handlers +def _start_gateway_make_shutdown_signal_handler(runner, _signal_initiated_shutdown: list): + """Build the SIGINT/SIGTERM handler; ``_signal_initiated_shutdown[0]`` records an unplanned signal.""" def shutdown_signal_handler(received_signal=None): - nonlocal _signal_initiated_shutdown # Planned --replace takeover: the sibling wrote a marker naming this PID before SIGTERM. Treat as # planned, exit 0 so systemd's Restart=on-failure doesn't revive us to flap-fight the replacer # (e.g. when both hermes.service and hermes-gateway.service are enabled). @@ -29477,7 +6306,7 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = _shutdown_ctx["signal"] if _shutdown_ctx else "SIGTERM/SIGINT", ) else: - _signal_initiated_shutdown = True + _signal_initiated_shutdown[0] = True # Mirror onto the runner so _stop_impl can suppress the gateway_state=stopped persist for # unexpected signals (container/s6 SIGTERM on restart, OOM, bare kill). Operator stops set a # planned-stop marker, take the `planned_stop` branch above and leave this False (DO persist). @@ -29507,49 +6336,19 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception as _e: logger.debug("spawn_async_diagnostic failed: %s", _e) asyncio.create_task(runner.stop()) + return shutdown_signal_handler - def restart_signal_handler(): - runner.request_restart(detached=False, via_service=True) - loop = asyncio.get_running_loop() - - # Loop-level exception handler swallowing transient network errors from background tasks: an - # unhandled telegram TimedOut / NetworkError / httpx connection error in any awaited coroutine - # would kill the whole gateway. Deliberately narrow — everything else hits the default handler. - loop.set_exception_handler(_gateway_loop_exception_handler) - - if threading.current_thread() is threading.main_thread(): - for sig in (signal.SIGINT, signal.SIGTERM): - try: - loop.add_signal_handler(sig, shutdown_signal_handler, sig) # windows-footgun: ok — wrapped in try/except NotImplementedError for Windows - except NotImplementedError: - pass - if hasattr(signal, "SIGUSR1"): - try: - loop.add_signal_handler(signal.SIGUSR1, restart_signal_handler) # windows-footgun: ok — POSIX signal, guarded by hasattr above + try/except NotImplementedError - except NotImplementedError: - pass - else: - logger.info("Skipping signal handlers (not running in main thread).") - - # Windows fallback: asyncio.add_signal_handler raises NotImplementedError there, so `hermes - # gateway stop`'s SIGTERM never reaches shutdown_signal_handler (no drain, sessions lost). A - # marker-polling thread notices the planned-stop marker written BEFORE the kill and drives the - # same shutdown path. Runs everywhere (cheap) so environments masking SIGTERM still drain cleanly. - _planned_stop_watcher_stop = threading.Event() - _planned_stop_watcher_thread = threading.Thread( - target=_run_planned_stop_watcher, - args=(_planned_stop_watcher_stop, runner, loop, shutdown_signal_handler), - daemon=True, - name="planned-stop-watcher", - ) - _planned_stop_watcher_thread.start() - - # Claim the PID file BEFORE bringing up any platform adapters: two concurrent `gateway run - # --replace` invocations both pass the termination-wait above, but only the O_CREAT|O_EXCL - # winner ever opens Telegram polling, Discord sockets, etc. The loser exits cleanly first. +def _start_gateway_claim_pid_file() -> bool: + """Claim the runtime lock + PID file (O_EXCL winner is the authoritative gateway). False = lost.""" import atexit - from gateway.status import write_pid_file, remove_pid_file, get_running_pid + from gateway.status import ( + acquire_gateway_runtime_lock, + get_running_pid, + release_gateway_runtime_lock, + remove_pid_file, + write_pid_file, + ) _current_pid = get_running_pid() if _current_pid is not None and _current_pid != os.getpid(): logger.error( @@ -29572,10 +6371,12 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = return False atexit.register(remove_pid_file) atexit.register(release_gateway_runtime_lock) + return True - # Control socket — the gateway-owned identify/status surface. Started right after the PID-file - # claim, since winning that O_EXCL race makes this process the authoritative gateway for its - # HERMES_HOME. Non-fatal: a bind failure just leaves consumers on the process-scan/state-file layer. + +async def _start_gateway_start_control_socket(runner): + """Start the gateway control socket (identify/status/pause-for-update); None when unavailable.""" + import atexit _control_server = None try: from gateway.control_socket import GatewayControlServer @@ -29624,79 +6425,14 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = except Exception as _cs_exc: logger.debug("Control socket startup failed (non-fatal): %s", _cs_exc) _control_server = None + return _control_server - # Lifecycle ledger: report if the previous life died uncleanly (SIGKILL / OOM / VM death), then - # claim the sentinel for this life. Placed after the PID-file/lock claim so only the - # authoritative gateway touches it — a --replace loser exiting above must not clobber it. - try: - from gateway.lifecycle_ledger import record_startup as _lifecycle_record_startup - _lifecycle_record_startup() - except Exception as _lc_exc: - logger.debug("Lifecycle ledger startup record failed: %s", _lc_exc) - try: - from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive - - start_nous_auth_keepalive() - except Exception as exc: - logger.debug("Nous auth keepalive did not start: %s", exc) - - _ensure_windows_gateway_venv_imports() - - # MCP tool discovery in an executor so the loop stays responsive when a configured MCP server is - # slow/unreachable: discover_mcp_tools() blocks up to 120s, which on the loop thread would freeze - # platform heartbeats (Discord shard, Telegram polling). - try: - await _discover_gateway_mcp_tools(runner.config) - except Exception as e: - logger.debug("MCP tool discovery failed: %s", e) - - # Start the gateway - try: - success = await runner.start() - except BaseException: - _shutdown_gateway_health_export(runner) - raise - if not success: - _shutdown_gateway_health_export(runner) - return False - # Recover any pending messages flushed during a previous shutdown (#72680). - try: - from gateway.shutdown_flush import recover_pending_to_db - recovered = recover_pending_to_db() - if recovered: - logger.info( - "Recovered %d pending message(s) from shutdown flush", recovered, - ) - except Exception: - pass - if runner.should_exit_cleanly: - _shutdown_gateway_health_export(runner) - if runner.exit_reason: - logger.error("Gateway exiting cleanly: %s", runner.exit_reason) - # A clean exit carrying an explicit exit code (e.g. GATEWAY_FATAL_CONFIG_EXIT_CODE) must - # propagate so the s6 finish script can translate it (78 → 125) and stop the restart loop; - # otherwise the early `return True` exits 0 and s6 crash-loops the gateway anyway. - if runner.exit_code is not None: - raise SystemExit(runner.exit_code) - return True - if not runner._running: - # Startup was intentionally aborted by restart/shutdown before entering - # running mode; preserve that lifecycle path without starting cron. - try: - await runner.wait_for_shutdown() - if runner.should_exit_with_failure: - if runner.exit_reason: - logger.error("Gateway exiting with failure: %s", runner.exit_reason) - return False - with suppress(Exception): - await _shutdown_mcp_servers_nonblocking() - if runner.exit_code is not None: - raise SystemExit(runner.exit_code) - return True - finally: - _shutdown_gateway_health_export(runner) +def _start_gateway_start_cron_and_housekeeping(runner): + """Start the cron scheduler thread + gateway housekeeping thread. + Returns ``(cron_stop, cron_provider, cron_thread, housekeeping_thread)``. + """ # Start the background cron scheduler via the resolved provider so scheduled jobs fire # automatically. Pass the event loop so cron delivery can use live adapters (E2EE support). from cron.scheduler_provider import ( @@ -29793,16 +6529,21 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = name="gateway-housekeeping", ) housekeeping_thread.start() + return cron_stop, cron_provider, cron_thread, housekeeping_thread - # READY is emitted only after adapters, cron and housekeeping reach their running boundary; - # missing config/systemd runtime state leaves the watchdog disabled without changing behavior. - start_watchdog = getattr(runner, "_start_systemd_watchdog", None) - if callable(start_watchdog): - start_watchdog() - - # Wait for shutdown - await runner.wait_for_shutdown() +async def _start_gateway_shutdown_tail( + runner, + _control_server, + cron_stop: threading.Event, + cron_provider, + cron_thread: threading.Thread, + housekeeping_thread: threading.Thread, + _planned_stop_watcher_stop: threading.Event, + _planned_stop_watcher_thread: threading.Thread, + _signal_initiated_shutdown: list, +) -> bool: + """Post-``wait_for_shutdown`` teardown; returns the process exit verdict (True = exit 0).""" # Stop the control socket first: once shutdown begins this process is no longer a truthful "the # gateway is serving here" answer, and a successor (--replace / supervisor respawn) must be able # to bind. Early-exit paths above don't reach this; their atexit cleanup_files hook runs, and a @@ -29854,7 +6595,7 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = # non-zero so systemd's Restart=on-failure revives the process (hermes update killing the # gateway mid-work, external kills, WSL2/container runtime signals). `hermes gateway stop` and # Ctrl+C are handled above as planned stops and must not trigger revival. - if _signal_initiated_shutdown and not runner._restart_requested: + if _signal_initiated_shutdown[0] and not runner._restart_requested: logger.info( "Exiting with code 1 (signal-initiated shutdown without restart " "request) so systemd Restart=on-failure can revive the gateway." @@ -29873,6 +6614,200 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = return True +async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = False, verbosity: Optional[int] = 0) -> bool: + """Start the gateway and run until interrupted. + + Returns True if the gateway ran, False if it failed to start (non-zero exit so systemd can + auto-restart). ``replace`` kills any existing instance first — avoids systemd restart-loop + deadlocks when the previous process hasn't fully exited. + """ + # Enable interactive exec approval on messaging platforms. Set here (not at module import) so + # incidental imports of gateway.run from CLI/tool code don't poison HERMES_EXEC_ASK. + os.environ["HERMES_EXEC_ASK"] = "1" + + from hermes_cli.resource_limits import apply_nofile_soft_limit + + apply_nofile_soft_limit() + + # Snapshot the checkout revision now, while sys.modules still matches disk, so a later `git + # pull` under this long-lived process can be detected (and risky work like model switching + # refused) instead of crashing on a stale in-memory module. + from gateway.code_skew import record_boot_fingerprint + record_boot_fingerprint() + + # Duplicate-instance guard: no two gateways under one HERMES_HOME. The PID file is scoped to + # HERMES_HOME, so multi-profile setups (distinct HERMES_HOME each) run concurrently untripped. + from gateway.status import get_running_pid + existing_pid = get_running_pid() + if existing_pid is not None and existing_pid != os.getpid(): + if not await _start_gateway_replace_existing_instance(existing_pid, replace): + return False + + _start_gateway_configure_logging(verbosity) + + runner = GatewayRunner(config) + # Multiplex: swap the launch-home file handlers for per-profile routers so each profile's records + # land in its own logs/. Must run after the runner resolved (possibly None) config and setup_logging. + _enable_multiplex_log_routing(runner.config) + # ``--replace`` is explicit startup authority, not a durable reconnect policy: GatewayRunner scopes + # it to cold adapter connects and clears it before the background reconnect watcher starts. + runner._platform_lock_takeover_on_start = bool(replace) + + # Track whether an unexpected signal initiated shutdown: an unexpected SIGTERM exits non-zero so + # service managers revive us; planned stop paths write a marker first so they exit cleanly. + _signal_initiated_shutdown = [False] + + # Set up signal handlers + shutdown_signal_handler = _start_gateway_make_shutdown_signal_handler( + runner, _signal_initiated_shutdown + ) + + def restart_signal_handler(): + runner.request_restart(detached=False, via_service=True) + + loop = asyncio.get_running_loop() + + # Loop-level exception handler swallowing transient network errors from background tasks: an + # unhandled telegram TimedOut / NetworkError / httpx connection error in any awaited coroutine + # would kill the whole gateway. Deliberately narrow — everything else hits the default handler. + loop.set_exception_handler(_gateway_loop_exception_handler) + + if threading.current_thread() is threading.main_thread(): + for sig in (signal.SIGINT, signal.SIGTERM): + try: + loop.add_signal_handler(sig, shutdown_signal_handler, sig) # windows-footgun: ok — wrapped in try/except NotImplementedError for Windows + except NotImplementedError: + pass + if hasattr(signal, "SIGUSR1"): + try: + loop.add_signal_handler(signal.SIGUSR1, restart_signal_handler) # windows-footgun: ok — POSIX signal, guarded by hasattr above + try/except NotImplementedError + except NotImplementedError: + pass + else: + logger.info("Skipping signal handlers (not running in main thread).") + + # Windows fallback: asyncio.add_signal_handler raises NotImplementedError there, so `hermes + # gateway stop`'s SIGTERM never reaches shutdown_signal_handler (no drain, sessions lost). A + # marker-polling thread notices the planned-stop marker written BEFORE the kill and drives the + # same shutdown path. Runs everywhere (cheap) so environments masking SIGTERM still drain cleanly. + _planned_stop_watcher_stop = threading.Event() + _planned_stop_watcher_thread = threading.Thread( + target=_run_planned_stop_watcher, + args=(_planned_stop_watcher_stop, runner, loop, shutdown_signal_handler), + daemon=True, + name="planned-stop-watcher", + ) + _planned_stop_watcher_thread.start() + + # Claim the PID file BEFORE bringing up any platform adapters: two concurrent `gateway run + # --replace` invocations both pass the termination-wait above, but only the O_CREAT|O_EXCL + # winner ever opens Telegram polling, Discord sockets, etc. The loser exits cleanly first. + if not _start_gateway_claim_pid_file(): + return False + + # Control socket — the gateway-owned identify/status surface. Started right after the PID-file + # claim, since winning that O_EXCL race makes this process the authoritative gateway for its + # HERMES_HOME. Non-fatal: a bind failure just leaves consumers on the process-scan/state-file layer. + _control_server = await _start_gateway_start_control_socket(runner) + + # Lifecycle ledger: report if the previous life died uncleanly (SIGKILL / OOM / VM death), then + # claim the sentinel for this life. Placed after the PID-file/lock claim so only the + # authoritative gateway touches it — a --replace loser exiting above must not clobber it. + try: + from gateway.lifecycle_ledger import record_startup as _lifecycle_record_startup + _lifecycle_record_startup() + except Exception as _lc_exc: + logger.debug("Lifecycle ledger startup record failed: %s", _lc_exc) + + try: + from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive + + start_nous_auth_keepalive() + except Exception as exc: + logger.debug("Nous auth keepalive did not start: %s", exc) + + _ensure_windows_gateway_venv_imports() + + # MCP tool discovery in an executor so the loop stays responsive when a configured MCP server is + # slow/unreachable: discover_mcp_tools() blocks up to 120s, which on the loop thread would freeze + # platform heartbeats (Discord shard, Telegram polling). + try: + await _discover_gateway_mcp_tools(runner.config) + except Exception as e: + logger.debug("MCP tool discovery failed: %s", e) + + # Start the gateway + try: + success = await runner.start() + except BaseException: + _shutdown_gateway_health_export(runner) + raise + if not success: + _shutdown_gateway_health_export(runner) + return False + # Recover any pending messages flushed during a previous shutdown (#72680). + try: + from gateway.shutdown_flush import recover_pending_to_db + recovered = recover_pending_to_db() + if recovered: + logger.info( + "Recovered %d pending message(s) from shutdown flush", recovered, + ) + except Exception: + pass + if runner.should_exit_cleanly: + _shutdown_gateway_health_export(runner) + if runner.exit_reason: + logger.error("Gateway exiting cleanly: %s", runner.exit_reason) + # A clean exit carrying an explicit exit code (e.g. GATEWAY_FATAL_CONFIG_EXIT_CODE) must + # propagate so the s6 finish script can translate it (78 → 125) and stop the restart loop; + # otherwise the early `return True` exits 0 and s6 crash-loops the gateway anyway. + if runner.exit_code is not None: + raise SystemExit(runner.exit_code) + return True + if not runner._running: + # Startup was intentionally aborted by restart/shutdown before entering + # running mode; preserve that lifecycle path without starting cron. + try: + await runner.wait_for_shutdown() + if runner.should_exit_with_failure: + if runner.exit_reason: + logger.error("Gateway exiting with failure: %s", runner.exit_reason) + return False + with suppress(Exception): + await _shutdown_mcp_servers_nonblocking() + if runner.exit_code is not None: + raise SystemExit(runner.exit_code) + return True + finally: + _shutdown_gateway_health_export(runner) + + cron_stop, cron_provider, cron_thread, housekeeping_thread = ( + _start_gateway_start_cron_and_housekeeping(runner) + ) + + # READY is emitted only after adapters, cron and housekeeping reach their running boundary; + # missing config/systemd runtime state leaves the watchdog disabled without changing behavior. + start_watchdog = getattr(runner, "_start_systemd_watchdog", None) + if callable(start_watchdog): + start_watchdog() + + # Wait for shutdown + await runner.wait_for_shutdown() + + return await _start_gateway_shutdown_tail( + runner, + _control_server, + cron_stop, + cron_provider, + cron_thread, + housekeeping_thread, + _planned_stop_watcher_stop, + _planned_stop_watcher_thread, + _signal_initiated_shutdown, + ) + + def _guard_corrupt_user_config() -> None: """Fail closed when the active profile's config.yaml cannot be parsed. diff --git a/gateway/run_adapters.py b/gateway/run_adapters.py new file mode 100644 index 0000000000..1c15a1d9aa --- /dev/null +++ b/gateway/run_adapters.py @@ -0,0 +1,1984 @@ +"""Adapter connect/disconnect, fatal-error recovery, reconnect watcher and multiplex profile adapter methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import os +import time +import weakref as _weakref +from agent.async_utils import consume_detached_task_result +from contextvars import Context +from datetime import datetime, timedelta, timezone +from gateway.config import Platform, platform_binds_port as _platform_binds_port +from gateway.platforms.base import BasePlatformAdapter +from gateway.restart import is_global_startup_conflict +from gateway.session import SessionSource +from pathlib import Path +from typing import Any, Awaitable, Callable, Dict, Optional + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayAdapterLifecycleMixin: + """Adapter connect/disconnect, fatal-error recovery, reconnect watcher and multiplex profile adapter methods for GatewayRunner.""" + + async def _await_adapter_cleanup_with_timeout( + self, awaitable: Awaitable[Any], timeout: float + ) -> bool: + """Wait for adapter cleanup without letting cancellation swallowing hang us. + + ``asyncio.wait_for`` cancels an overdue child but then waits for it to exit. An adapter + close path that catches ``CancelledError`` can therefore block recovery forever. Keep + ownership of the old task through its done callback, but release the runner at the deadline. + """ + if timeout <= 0: + await awaitable + return True + + task = asyncio.ensure_future(awaitable) + try: + done, _pending = await asyncio.wait({task}, timeout=timeout) + except asyncio.CancelledError: + task.cancel() + task.add_done_callback(consume_detached_task_result) + raise + if task in done: + await task + return True + + task.cancel() + task.add_done_callback(consume_detached_task_result) + return False + + async def _safe_adapter_disconnect(self, adapter, platform) -> None: + """Call adapter.disconnect() defensively, swallowing any error. + + For a failed/raised connect(): partial resources (aiohttp.ClientSession, poll tasks, child + subprocesses) would otherwise leak. Must tolerate partial-init state and never raise. + """ + timeout = self._adapter_disconnect_timeout_secs() + try: + completed = await self._await_adapter_cleanup_with_timeout( + adapter.disconnect(), timeout + ) + if not completed: + logger.warning( + "Timed out after %.1fs while disconnecting %s adapter; continuing shutdown", + timeout, + platform.value if platform is not None else "adapter", + ) + except Exception as e: + logger.debug( + "Defensive %s disconnect after failed connect raised: %s", + platform.value if platform is not None else "adapter", + e, + ) + + async def _bounded_adapter_teardown( + self, adapter, platform, *, profile: Optional[str] = None + ) -> None: + """Tear down one adapter on the shutdown path with bounded awaits. + + ``cancel_background_tasks()`` and ``disconnect()`` can block forever on half-dead network + state (e.g. a wedged WebSocket thread), stalling shutdown past systemd's ``TimeoutStopSec``; + the SIGKILL skips ``atexit`` PID-file cleanup and the next start dies with "PID file race + lost". Each await uses ``HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT``; on timeout the task is + cancelled and detached so a cancellation-swallowing adapter can't hang the loop. Never raises. + """ + timeout = self._adapter_disconnect_timeout_secs() + suffix = f" (profile: {profile})" if profile else "" + started_at = time.monotonic() + try: + cancelled = await self._await_adapter_cleanup_with_timeout( + adapter.cancel_background_tasks(), timeout + ) + if not cancelled: + logger.warning( + "✗ %s background-task cancel timed out after %.1fs - forcing continue%s", + platform.value, timeout, suffix, + ) + except Exception as e: + logger.debug("✗ %s background-task cancel error%s: %s", platform.value, suffix, e) + try: + disconnected = await self._await_adapter_cleanup_with_timeout( + adapter.disconnect(), timeout + ) + if disconnected: + logger.info( + "✓ %s disconnected (%.2fs)%s", + platform.value, time.monotonic() - started_at, suffix, + ) + else: + logger.warning( + "✗ %s disconnect timed out after %.1fs - forcing continue%s", + platform.value, timeout, suffix, + ) + except Exception as e: + logger.error( + "✗ %s disconnect error after %.2fs%s: %s", + platform.value, time.monotonic() - started_at, suffix, e, + ) + + def _adapter_disconnect_timeout_secs(self) -> float: + """Return the per-adapter disconnect timeout used during shutdown.""" + from gateway.run import _ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT + raw = os.getenv("HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT", "").strip() + if raw: + try: + timeout = float(raw) + except ValueError: + logger.warning( + "Ignoring invalid HERMES_GATEWAY_ADAPTER_DISCONNECT_TIMEOUT=%r", + raw, + ) + else: + return max(0.0, timeout) + return _ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT + + def _platform_connect_timeout_secs(self, platform=None, *, initial: bool = False) -> float: + """Return the per-platform connect timeout used during startup/retry. + + Telegram's full 180s connect budget is deliberately NOT spent at cold start: an unreachable + Telegram would hold the gateway out of ``running`` for the whole budget. The cold-start wait + is capped and the platform handed to the reconnect watcher, which retries with the full + budget and ``is_reconnect=True`` (preserving the offline update queue). + """ + from gateway.run import ( + _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT, + _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT, + _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT, + ) + raw = os.getenv("HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT", "").strip() + if raw: + try: + timeout = float(raw) + except ValueError: + logger.warning( + "Ignoring invalid HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT=%r", + raw, + ) + else: + return max(0.0, timeout) + if platform == Platform.TELEGRAM: + if initial: + return _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT + return _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT + return _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT + + async def _connect_adapter_with_timeout( + self, adapter, platform, *, is_reconnect: bool = False, initial: bool = False + ) -> bool: + """Connect an adapter without allowing one platform to block others. + + ``is_reconnect`` lets adapters distinguish a cold first boot (drop any stale server-side + queue) from a watcher reconnect (preserve the queue so interim messages aren't dropped). + ``initial`` selects the capped cold-start budget for platforms whose full connect budget is + too long to spend before the gateway reaches ``running`` (Telegram's 180s). + """ + timeout = self._platform_connect_timeout_secs(platform, initial=initial) + if timeout <= 0: + return await adapter.connect(is_reconnect=is_reconnect) + # Detach-on-timeout rather than plain asyncio.wait_for: wait_for cancels the overdue task but + # then waits for it to exit, so a connect() that catches CancelledError blocks recovery + # forever (watcher never retries). Keep ownership via its done callback; release at deadline. + task = asyncio.ensure_future( + adapter.connect(is_reconnect=is_reconnect) + ) + try: + done, _pending = await asyncio.wait({task}, timeout=timeout) + except asyncio.CancelledError: + task.cancel() + task.add_done_callback(consume_detached_task_result) + raise + if task in done: + result = await task + return bool(result) + task.cancel() + task.add_done_callback(consume_detached_task_result) + raise TimeoutError( + f"{platform.value} connect timed out after {timeout:g}s" + ) + + async def _connect_initial_adapter_with_timeout(self, adapter, platform) -> bool: + """Connect one cold-start adapter with tightly scoped replace intent. + + The capability is visible only while this initial connect is awaited. Reconnects call + ``_connect_adapter_with_timeout`` directly and adapters also default to deny, so a later + network recovery can never evict a healthy token holder. + """ + adapter._platform_lock_takeover_allowed = bool( + self._platform_lock_takeover_on_start + ) + try: + return await self._connect_adapter_with_timeout( + adapter, platform, initial=True + ) + finally: + adapter._platform_lock_takeover_allowed = False + + async def _handle_reaction_event(self, ctx: Dict[str, Any]) -> None: + """Fan a normalised platform reaction event out to the HookRegistry. + + The adapter-supplied ``event_name`` ("reaction:added"/"reaction:removed") is the hook event, + matching the ``agent:*`` naming scheme. Errors never block the adapter's event loop. + """ + event_name = str(ctx.get("event_name") or "reaction:added") + try: + await self.hooks.emit(event_name, ctx) + except Exception: + logger.debug("[Gateway] reaction hook emit failed", exc_info=True) + + async def _handle_adapter_fatal_error(self, adapter: BasePlatformAdapter) -> None: + """React to an adapter failure after startup. + + Retryable errors (network blip, DNS) queue the platform for background reconnection. + The notification arrives on the failing adapter's own polling task, and the disconnect in + the handler can cancel that task mid-flight (disconnect()'s current-task guard misses it + because _safe_adapter_disconnect closes in a wrapper task), stranding the platform between + the fatal log and the reconnect queue — so the real work runs in a detached task. + """ + tasks = getattr(self, "_fatal_handler_tasks", None) + if tasks is None: + tasks = self._fatal_handler_tasks = set() + task = asyncio.create_task(self._handle_adapter_fatal_error_detached(adapter)) + tasks.add(task) + task.add_done_callback(tasks.discard) + # Await so callers that expect completion still get it — but through shield(): Task.cancel() + # on the caller also cancels the future it is awaiting (_fut_waiter), so a plain `await + # task` would tunnel the cancellation straight into the "detached" task. shield() absorbs + # it: the caller sees CancelledError, the handler runs to completion. + await asyncio.shield(task) + + def _queue_retryable_fatal_platform(self, adapter: BasePlatformAdapter) -> bool: + """Queue a retryable fatal adapter for background reconnection. + + Returns True when newly queued; idempotent if already queued. Must not await: callers + invoke this *before* any disconnect await so a wedged close cannot strand the platform. + """ + if not adapter.fatal_error_retryable: + return False + platform_config = self.config.platforms.get(adapter.platform) + if not platform_config: + return False + if adapter.platform in self._failed_platforms: + # Nothing to enqueue — but "already queued" is exactly when the watcher may have died, + # and the enqueue branch below holds the ONLY _ensure_reconnect_watcher_running() call. + # _spawn_supervised gives up after _MAX_SUPERVISED_RESTARTS; without this backstop a + # queued platform is a silent permanent outage (nothing retries, and the stranded check + # treats a queued platform as safe so the process never restarts either). + self._ensure_reconnect_watcher_running() + return False + self._failed_platforms[adapter.platform] = { + "config": platform_config, + "attempts": 0, + "next_retry": time.monotonic(), + "queued_at": time.monotonic(), + "credential_claim": self._adapter_credential_claim( + adapter.platform, adapter + ), + "listener_claim": self._adapter_listener_claim( + adapter.platform, adapter + ), + } + logger.info( + "%s queued for background reconnection", + adapter.platform.value, + ) + # Ensure the reconnect watcher is alive — respawn if it died (e.g. restart budget exhausted) + # so queued platforms are not permanently stranded. + self._ensure_reconnect_watcher_running() + return True + + async def _handle_adapter_fatal_error_detached( + self, adapter: BasePlatformAdapter + ) -> None: + """Run the fatal handler; if the platform still ends up stranded (not reconnected, not + queued, not intentionally disabled), exit the gateway with failure so the service manager + restarts it instead of leaving a silent partial outage.""" + try: + # Outer hard deadline: even with queue-before-disconnect, a hang anywhere in the impl + # (status write side effects, detach races, etc.) must not leave this task wedged + # forever — the stranded check in ``finally`` only runs when we return. + timeout = self._adapter_disconnect_timeout_secs() + if timeout <= 0: + await self._handle_adapter_fatal_error_impl(adapter) + else: + # Disconnect budget plus a little queue/status bookkeeping overhead; keep the extra + # proportional so tests that shrink the disconnect timeout still finish promptly. + outer = timeout + min(2.0, max(0.05, timeout)) + completed = await self._await_adapter_cleanup_with_timeout( + self._handle_adapter_fatal_error_impl(adapter), + outer, + ) + if not completed: + logger.error( + "Fatal-error handling for %s timed out after %.1fs; " + "ensuring reconnect queue is populated", + adapter.platform.value, + outer, + ) + self._queue_retryable_fatal_platform(adapter) + except asyncio.CancelledError: + # Best-effort queue before re-raising: a cancelled fatal handler + # must not strand a retryable platform (#80598). + try: + self._queue_retryable_fatal_platform(adapter) + except Exception: + logger.debug( + "Failed to queue %s after fatal-handler cancellation", + adapter.platform.value, + exc_info=True, + ) + raise + except Exception: + logger.exception( + "Fatal-error handling for %s raised unexpectedly", + adapter.platform.value, + ) + # Best-effort queue so an unexpected raise mid-handler cannot + # leave a retryable platform permanently deaf (#80598). + try: + self._queue_retryable_fatal_platform(adapter) + except Exception: + logger.debug( + "Failed to queue %s after fatal-handler exception", + adapter.platform.value, + exc_info=True, + ) + finally: + platform = adapter.platform + shutdown_event = getattr(self, "_shutdown_event", None) + stranded = ( + adapter.fatal_error_retryable + and platform not in self.adapters + and platform not in getattr(self, "_failed_platforms", {}) + and not (shutdown_event is not None and shutdown_event.is_set()) + ) + if stranded: + logger.error( + "%s adapter was lost without entering the reconnection " + "queue; exiting gateway so the service manager restarts it.", + platform.value, + ) + self._exit_reason = ( + f"{platform.value} adapter lost without reconnection queue" + ) + self._exit_with_failure = True + await self.stop() + + async def _handle_adapter_fatal_error_impl(self, adapter: BasePlatformAdapter) -> None: + # Snapshot this platform slot's current owner first: acting on a stale notification would + # overwrite a healthy platform's runtime status and wrongly re-queue it for reconnection. + existing = self.adapters.get(adapter.platform) + if existing is not None and existing is not adapter: + logger.debug( + "Ignoring stale fatal error from a superseded %s adapter instance: %s", + adapter.platform.value, + adapter.fatal_error_code or "unknown", + ) + return + + logger.error( + "Fatal %s adapter error (%s): %s", + adapter.platform.value, + adapter.fatal_error_code or "unknown", + adapter.fatal_error_message or "unknown error", + ) + # A relay credential revoked by opt-out is not an error to retry: render a clean "disabled" + # state, not red "fatal"/"retrying" (non-retryable code, so it also leaves the queue below). + if adapter.fatal_error_code == "relay_disabled": + platform_state = "disabled" + elif adapter.fatal_error_retryable: + platform_state = "retrying" + else: + platform_state = "fatal" + self._update_platform_runtime_status( + adapter.platform.value, + platform_state=platform_state, + error_code=adapter.fatal_error_code, + error_message=adapter.fatal_error_message, + ) + + if existing is adapter: + # Claim this adapter for teardown before awaiting disconnect(): a second fatal-error + # notification for the same adapter (e.g. a concurrent recovery path) would otherwise + # still see itself as "existing" during the await and disconnect() the same object twice. + self.adapters.pop(adapter.platform, None) + self.delivery_router.adapters = self.adapters + + # Queue retryable failures BEFORE any disconnect await: a half-dead transport can wedge + # native close() (or swallow CancelledError), so "disconnect then queue" left platforms + # permanently deaf in a live process after the network recovered. Populate the queue first so the + # reconnect watcher always has work; teardown is best-effort after. + self._queue_retryable_fatal_platform(adapter) + + if existing is adapter: + # A half-closed transport can wedge native close() indefinitely; reuse the shutdown-path + # timeout so this runtime fatal handler always returns to the stay-alive / stranded path. + await self._safe_adapter_disconnect(adapter, adapter.platform) + + if not self.adapters and not self._failed_platforms: + self._exit_reason = adapter.fatal_error_message or "All messaging adapters disconnected" + if adapter.fatal_error_retryable: + self._exit_with_failure = True + logger.error("No connected messaging platforms remain. Shutting down gateway for service restart.") + else: + logger.error("No connected messaging platforms remain. Shutting down gateway cleanly.") + await self.stop() + elif not self.adapters and self._failed_platforms: + # All platforms are down and queued for reconnection. Keep the gateway alive so cron jobs + # still run and the watcher can recover platforms when the problem clears; exiting for a + # systemd restart would turn a transient outage into a state-killing restart loop. + logger.warning( + "No connected messaging platforms remain, but %d platform(s) " + "queued for reconnection — gateway staying alive, watcher will " + "retry in background.", + len(self._failed_platforms), + ) + + def _request_clean_exit(self, reason: str) -> None: + self._exit_cleanly = True + self._exit_reason = reason + self._shutdown_event.set() + + @staticmethod + def _supervised_backoff(attempt: int) -> float: + """Delay before the supervisor's next respawn, in seconds (capped exponential). + + A method so tests can collapse the schedule instead of sleeping through the real curve. + """ + return min(60, 2 ** min(attempt, 6)) + + def _spawn_supervised( + self, coro_factory, name, *, restart=True, _attempt=0, on_spawn=None, + on_give_up=None, + ): + """Launch a long-lived background task with task-level supervision. + + Catches what a per-iteration try/except cannot — exceptions in the OUTER loop or pre-try + setup — which a bare ``asyncio.create_task`` drops silently. Restarts with capped backoff up + to ``_MAX_SUPERVISED_RESTARTS`` rapid failures; the counter resets after a run healthy for + ``_SUPERVISED_HEALTHY_SECS``. Each spawn uses a fresh ``Context``: an inherited + delegated-child marker would make the Kanban dispatcher reject its own writes. + ``on_spawn`` fires on EVERY spawn incl. respawns; callers tracking the handle elsewhere + (e.g. ``_reconnect_watcher_task``) MUST pass it or a respawn leaves a stale handle and a + SECOND watcher. ``on_give_up(name)`` fires when the restart budget is spent. + """ + if getattr(self, "_background_tasks", None) is None: + self._background_tasks = set() + + # Monotonic spawn timestamp captured per spawn: the ``_done`` callback + # uses it to distinguish a rapid crash-loop from a healthy-run-then-crash. + _started = time.monotonic() + + # Deliberately no kwargs to create_task (some test doubles mock a narrow signature); calling + # it from a fresh Context gives the same isolation as create_task(..., context=Context()). + task = Context().run(lambda: asyncio.create_task(coro_factory())) + # PERMANENT supervised watcher, not transient background WORK: the scale-to-zero idle check + # must ignore process-lifetime watchers or the gateway counts itself busy forever. Transient + # tasks added to _background_tasks elsewhere (startup-resume events etc.) stay counted. + task._hermes_supervised_watcher = True # type: ignore[attr-defined] + self._background_tasks.add(task) + if on_spawn is not None: + # Record the live handle NOW so an external tracker (e.g. _reconnect_watcher_task) + # points at the current task, not a dead one left by a prior supervised respawn. + try: + on_spawn(task) + except Exception: # pragma: no cover - defensive; a tracker must never kill the spawn + logger.debug("on_spawn callback for %s raised", name, exc_info=True) + + def _done(t): + self._background_tasks.discard(t) + if t.cancelled(): + return + exc = t.exception() + if exc is None: + # Clean return == deliberate shutdown or a self-disabling watcher (e.g. a gated + # no-op returning at once); respawning would busy-spin it — NEVER restart on it. + return + logger.error("Supervised task %s died: %r", name, exc, exc_info=exc) + if restart and self._running: + ran_for = time.monotonic() - _started + if ran_for >= self._SUPERVISED_HEALTHY_SECS: + # Ran healthily before crashing — a FRESH failure, not a rapid crash-loop. Reset + # the counter so a daemon crashing a few times over days is never abandoned. + effective_attempt = 0 + else: + effective_attempt = _attempt + if effective_attempt >= self._MAX_SUPERVISED_RESTARTS: + logger.error( + "Supervised task %s died %d times in rapid succession " + "(each within %ds of restart) — giving up restarts", + name, + effective_attempt, + self._SUPERVISED_HEALTHY_SECS, + ) + if on_give_up is not None: + try: + on_give_up(name) + except Exception: # pragma: no cover - defensive + logger.debug( + "on_give_up callback for %s raised", + name, exc_info=True, + ) + return + backoff = self._supervised_backoff(effective_attempt) + + async def _respawn(): + await asyncio.sleep(backoff) + if self._running: + self._spawn_supervised( + coro_factory, + name, + restart=restart, + _attempt=effective_attempt + 1, + on_spawn=on_spawn, + # Threaded through the recursion like on_spawn: only the LAST respawn's give-up + # matters, and dropping the callback leaves the exhaustion branch with no owner. + on_give_up=on_give_up, + ) + + # The done callback retains its registration context, so isolate the backoff task + # too; otherwise a restart could reintroduce the original caller's turn scope. + respawn_task = Context().run(lambda: asyncio.create_task(_respawn())) + self._background_tasks.add(respawn_task) + respawn_task.add_done_callback(self._background_tasks.discard) + + task.add_done_callback(_done) + return task + + async def _handoff_watcher( + self, interval: float = 2.0, drain_timeout: float = 30.0, + ) -> None: + """Background task that processes pending CLI→gateway session handoffs. + + Polls ``state.db`` for ``handoff_state='pending'`` rows: claim atomically (pending → + running), re-bind the home channel's session_key to the CLI session_id via + ``switch_session``, dispatch a synthetic ``MessageEvent``, mark ``completed``/``failed``. + """ + from gateway.run import _async_profile_runtime_scope, _handoff_watch_scopes, _reclaim_stale + # Initial delay so the gateway is fully connected to its platforms + # before we try to dispatch handoffs through them. + await asyncio.sleep(5) + + # Does _process_handoff accept the profile argument? The real one does; test stand-ins bind + # a one-parameter callable. Probed once, outside the loop. + try: + import inspect as _inspect + _process_takes_profile = len( + _inspect.signature(self._process_handoff).parameters + ) >= 2 + except Exception: + _process_takes_profile = False + + # In-flight dispatches keyed by session id. A handoff runs a FULL agent turn plus delivery + # (far longer than the CLI's 60s wait); inline processing would let one slow handoff block + # every other profile's poll and time them out. Fire-and-forget; the poll loop only claims. + inflight: Dict[str, "asyncio.Task"] = {} + + async def _dispatch(row, session_id, session_db, profile_name) -> None: + """Run one claimed handoff to a terminal state, off the poll path.""" + try: + if _process_takes_profile: + await self._process_handoff(row, profile_name) + else: + await self._process_handoff(row) + await session_db.complete_handoff(session_id) + except asyncio.CancelledError: + # Gateway shutting down: leave the row 'running' so the next + # start's reclaim marks it failed with a clear reason. + raise + except Exception as exc: + logger.warning( + "Handoff for session %s failed: %s", + session_id, exc, exc_info=True, + ) + try: + await session_db.fail_handoff(session_id, str(exc)) + except Exception: + logger.debug("Could not record handoff failure", exc_info=True) + finally: + inflight.pop(session_id, None) + + async def _tick(profile_name: Optional[str] = None) -> None: + """One poll of the CURRENTLY-SCOPED session store. + + A closure over ``self``, not a method: unit tests bind ``_handoff_watcher`` onto a + ``SimpleNamespace`` exposing only ``_session_db``, ``_running`` and ``_process_handoff``; + any other ``self.`` would raise, be swallowed by the loop, and silently no-op the + watcher. ``profile_name`` (``None`` = root) makes delivery use that profile's OWN adapter. + """ + session_db = getattr(self, "_session_db", None) + if session_db is None: + return + pending = await session_db.list_pending_handoffs() + for row in pending: + session_id = row.get("id") + if not session_id or session_id in inflight: + continue + if not await session_db.claim_handoff(session_id): + # Another tick or another gateway already claimed it. + continue + # Positional, not keyword: tests bind a one-arg ``_process_handoff(row)`` stand-in and a + # keyword call would TypeError into the failure branch (arity probed above). + # INVARIANT (do not weaken): this task is created inside _profile_runtime_scope but + # typically RUNS after it exits; it sees the profile's home/secret scope only because + # those seams are ContextVar-based and ensure_future copies the Context. + inflight[session_id] = asyncio.ensure_future( + _dispatch(row, session_id, session_db, profile_name) + ) + + # A row still 'running' at startup belongs to a gateway that died mid-dispatch: it can never + # reach a terminal state, and request_handoff refuses new requests while it sits there. + for _pname, _phome in _handoff_watch_scopes(self): + try: + if _phome is None: + await _reclaim_stale(self) + else: + async with _async_profile_runtime_scope(_phome): + await _reclaim_stale(self) + except Exception: + logger.debug("Stale-handoff reclaim failed", exc_info=True) + + try: + while self._running: + try: + for profile_name, profile_home in _handoff_watch_scopes(self): + if profile_home is None: + await _tick(profile_name) + else: + async with _async_profile_runtime_scope(profile_home): + await _tick(profile_name) + except asyncio.CancelledError: + raise + except Exception as exc: + logger.debug("Handoff watcher tick error: %s", exc, exc_info=True) + await asyncio.sleep(interval) + finally: + # Drain in-flight dispatches before returning: cancelling would strand their rows in + # 'running'; a bounded grace period lets an almost-done handoff record its own state. + pending_tasks = [t for t in inflight.values() if not t.done()] + if pending_tasks: + try: + await asyncio.wait(pending_tasks, timeout=drain_timeout) + except Exception: + logger.debug("Handoff drain raised", exc_info=True) + for task in pending_tasks: + if not task.done(): + task.cancel() + + def _on_reconnect_watcher_gave_up(self, name: str = "") -> None: + """Own the reconnect invariant once supervision has abandoned it. + + Invariant: while running and ``_failed_platforms`` is non-empty, a reconnect watcher is live + or a bounded respawn is scheduled. Event-coupled recovery is not enough: the failed adapter + is dropped from the live map, so no later event may ever arrive to notice a dead watcher. + Deliberately NOT done here: requesting a process restart when the slow tier is exhausted — + a blast-radius policy call; a single loud error names the still-queued platforms instead. + """ + if not getattr(self, "_running", False): + return + if not getattr(self, "_failed_platforms", None): + # No queued work depends on the watcher; leaving it dead is correct — the enqueue path + # spawns a fresh one the moment a platform is queued again. + logger.warning( + "Reconnect watcher supervision exhausted with an empty retry " + "queue — leaving it down until a platform is queued." + ) + return + self._schedule_slow_reconnect_watcher_respawn(attempt=0) + + def _schedule_slow_reconnect_watcher_respawn(self, *, attempt: int) -> None: + """Bounded slow-tier respawn of the reconnect watcher.""" + if attempt >= self._MAX_SLOW_WATCHER_RESPAWNS: + logger.error( + "Reconnect watcher could not be kept alive after %d slow " + "respawns; %d platform(s) remain queued and unattended: %s. " + "Manual intervention or a gateway restart is required.", + attempt, + len(self._failed_platforms), + ", ".join(str(p) for p in self._failed_platforms), + ) + return + + async def _slow_respawn() -> None: + await asyncio.sleep(self._RECONNECT_WATCHER_SLOW_RETRY_SECS) + if not getattr(self, "_running", False): + return + if not getattr(self, "_failed_platforms", None): + # The queue drained while we waited -- something else healed + # it. Nothing to own any more. + return + task = getattr(self, "_reconnect_watcher_task", None) + if task is not None and not task.done(): + return # a watcher came back on its own; stand down + logger.warning( + "Reconnect watcher still down with %d platform(s) queued — " + "slow respawn %d/%d", + len(self._failed_platforms), + attempt + 1, + self._MAX_SLOW_WATCHER_RESPAWNS, + ) + self._spawn_reconnect_watcher( + on_give_up=lambda _name: self._schedule_slow_reconnect_watcher_respawn( + attempt=attempt + 1 + ) + ) + + respawn_task = asyncio.create_task(_slow_respawn()) + if getattr(self, "_background_tasks", None) is None: + self._background_tasks = set() + self._background_tasks.add(respawn_task) + respawn_task.add_done_callback(self._background_tasks.discard) + + def _spawn_reconnect_watcher(self, *, on_give_up=None): + """Single place that knows how to launch the reconnect watcher. + + ``on_spawn`` is load-bearing: without it the supervisor's own respawn leaves + ``_reconnect_watcher_task`` at a dead handle and ``_ensure_...`` spawns a second watcher. + """ + self._reconnect_watcher_task = self._spawn_supervised( + self._platform_reconnect_watcher, + "platform_reconnect_watcher", + on_spawn=lambda t: setattr(self, "_reconnect_watcher_task", t), + on_give_up=on_give_up or self._on_reconnect_watcher_gave_up, + ) + return self._reconnect_watcher_task + + def _ensure_reconnect_watcher_running(self) -> None: + """Ensure the platform reconnect watcher background task is alive. + + Respawns a dead watcher (exhausted restart budget, unrecoverable exception) so queued + platforms are not stranded. Called on BOTH _queue_retryable_fatal_platform paths: the + re-fatal of an already-queued platform is the only case where the budget can be exhausted. + """ + if not getattr(self, "_running", False): + return + task = getattr(self, "_reconnect_watcher_task", None) + if task is not None and not task.done(): + return # already alive + logger.warning( + "Reconnect watcher task is dead (done=%s) — respawning", + task.done() if task is not None else "N/A", + ) + self._spawn_reconnect_watcher() + + async def _platform_reconnect_watcher(self) -> None: + """Background task that periodically retries connecting failed platforms. + + Exponential backoff 30s → 300s cap; retryable failures (network/DNS) retry at the cap + indefinitely so transient outages self-heal, non-retryable (bad auth) drop out immediately. + The circuit breaker (``/platform pause``) is manual only — auto-pausing left bots dead. + """ + from gateway.run import ( + _dispose_unused_adapter, + _platform_has_bot_credential, + _reconnect_backoff, + _reconnect_needs_attention, + ) + await asyncio.sleep(10) # initial delay — let startup finish + while self._running: + if not self._failed_platforms: + # Nothing to reconnect — sleep and check again + for _ in range(30): + if not self._running: + return + if self._failed_platforms: + break + await asyncio.sleep(1) + continue + + now = time.monotonic() + for platform in list(self._failed_platforms.keys()): + if not self._running: + return + info = self._failed_platforms.get(platform) + if info is None: + # Removed concurrently (/platform resume, reconnect via another path) between + # the snapshot above and this lookup — not an error, nothing to do this pass. + continue + # Skip paused platforms entirely — they need explicit + # /platform resume to come back. + if info.get("paused"): + continue + # Long-lived retry escalation: past the attention threshold flag the platform + # NEEDS_ATTENTION in runtime status so a dead token/revoked intent doesn't look + # like ordinary "retrying" forever. A signal, NOT a circuit breaker — retries continue. + if not info.get("attention_flagged") and _reconnect_needs_attention(info, now): + info["attention_flagged"] = True + queued_for = now - info.get("queued_at", now) + retrying_since_iso = ( + datetime.now(timezone.utc) - timedelta(seconds=queued_for) + ).isoformat() + logger.warning( + "%s has been failing/reconnecting continuously for " + "%.1f hours (%d attempts) — flagging NEEDS_ATTENTION. " + "Retries continue, but this usually means a permanent " + "problem (revoked credentials, missing intents, broken " + "sidecar). Check `hermes status` / `/platform list`.", + platform.value, + queued_for / 3600.0, + info.get("attempts", 0), + ) + self._update_platform_runtime_status( + platform.value, + platform_state="retrying", + needs_attention=True, + retrying_since=retrying_since_iso, + ) + if now < info["next_retry"]: + continue # not time yet + + platform_config = info["config"] + attempt = info["attempts"] + 1 + # Empty-token primary configs can never reconnect; drop them so multiplex setups + # where a secondary profile owns the bot do not spin forever. + if not _platform_has_bot_credential(platform, platform_config): + logger.warning( + "Reconnect %s: no bot credential on queued config, " + "removing from retry queue", + platform.value, + ) + del self._failed_platforms[platform] + continue + logger.info( + "Reconnecting %s (attempt %d)...", + platform.value, attempt, + ) + + adapter = None + try: + adapter = self._create_adapter(platform, platform_config) + if not adapter: + logger.warning( + "Reconnect %s: adapter creation returned None, removing from retry queue", + platform.value, + ) + del self._failed_platforms[platform] + continue + + adapter.set_message_handler(self._primary_message_handler()) + adapter.set_fatal_error_handler(self._handle_adapter_fatal_error) + adapter.set_session_store(self.session_store) + adapter.set_busy_session_handler(self._handle_active_session_busy_message) + _set_reaction = getattr(adapter, "set_reaction_handler", None) + if callable(_set_reaction): + _set_reaction(self._handle_reaction_event) + adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) + adapter.set_authorization_check(self._make_adapter_auth_check(adapter.platform)) + adapter.set_platform_event_handler(self._primary_platform_event_handler()) + adapter._busy_text_mode = self._busy_text_mode + + # Reconnect after outage: keep the platform's server-side update queue so + # messages sent while the bot was offline are delivered rather than dropped. + success = await self._connect_adapter_with_timeout( + adapter, platform, is_reconnect=True + ) + if success: + self.adapters[platform] = adapter + self._sync_voice_mode_state_to_adapter(adapter) + # Wire voice input callback on reconnect as well (#60623). + self._bind_voice_input_callback(adapter) + self.delivery_router.adapters = self.adapters + del self._failed_platforms[platform] + self._update_platform_runtime_status( + platform.value, + platform_state="connected", + error_code=None, + error_message=None, + needs_attention=False, + retrying_since=None, + ) + logger.info("✓ %s reconnected successfully", platform.value) + + # Final responses rejected while this adapter was down are still owned by + # this live process, so startup recovery cannot claim them. Replay the + # explicitly transient subset now that the platform is usable. + try: + await self._redeliver_failed_obligations_for_platform( + platform + ) + except Exception: + logger.debug( + "failed-obligation redelivery after %s reconnect failed", + platform.value, + exc_info=True, + ) + + # Rebuild channel directory with the new adapter + try: + from gateway.channel_directory import build_channel_directory + await build_channel_directory(self.adapters) + except Exception: + pass + + # A platform that was offline at gateway startup never got its restart- + # interrupted sessions auto-resumed — the startup pass skips sessions whose + # adapter isn't connected yet. + try: + self._schedule_resume_pending_sessions(platform=platform) + except Exception: + logger.debug( + "resume-pending reschedule after %s reconnect failed", + platform.value, + exc_info=True, + ) + # Check if the failure is non-retryable + elif adapter.has_fatal_error and not adapter.fatal_error_retryable: + self._update_platform_runtime_status( + platform.value, + platform_state="fatal", + error_code=adapter.fatal_error_code, + error_message=adapter.fatal_error_message, + ) + logger.warning( + "Reconnect %s: non-retryable error (%s), removing from retry queue", + platform.value, adapter.fatal_error_message, + ) + # The adapter is about to be dropped from the queue without ever being + # installed on self.adapters, so nothing else will call disconnect() on it. + # Dispose here or the resource owners built in __init__ (ResponseStore etc.) + # leak ~2 fds each; at the 300s cap the gateway hits the fd limit in ~12h. + await _dispose_unused_adapter(adapter) + del self._failed_platforms[platform] + else: + self._update_platform_runtime_status( + platform.value, + platform_state="retrying", + error_code=adapter.fatal_error_code, + error_message=adapter.fatal_error_message or "failed to reconnect", + ) + backoff = _reconnect_backoff(attempt) + info["attempts"] = attempt + info["next_retry"] = time.monotonic() + backoff + logger.info( + "Reconnect %s failed, next retry in %ds", + platform.value, backoff, + ) + # Same fd-leak concern as the non-retryable branch above: the adapter failed + # to connect and is being thrown away. + await _dispose_unused_adapter(adapter) + # Retryable failures (network/DNS blips) retry at the backoff cap forever, + # self-healing when connectivity returns. Never auto-pause them: a transient + # outage must not need `/platform resume`. Everything here is retryable. + except Exception as e: + if adapter is not None: + # An exception escaping connect (DNS timeout, aiohttp server.start() crash, + # etc.) leaves the adapter in the same unowned state as the branches above. + await _dispose_unused_adapter(adapter) + self._update_platform_runtime_status( + platform.value, + platform_state="retrying", + error_code=None, + error_message=str(e), + ) + backoff = _reconnect_backoff(attempt) + info["attempts"] = attempt + info["next_retry"] = time.monotonic() + backoff + logger.warning( + "Reconnect %s error: %s, next retry in %ds", + platform.value, e, backoff, + ) + # A reconnect exception (connect timeout, DNS failure, ...) is transient; keep + # retrying at the backoff cap rather than auto-pausing. + + # Check every 10 seconds for platforms that need reconnection + for _ in range(10): + if not self._running: + return + await asyncio.sleep(1) + + async def _cancel_secondary_profile_reconnect_tasks(self) -> None: + """Cancel profile-scoped reconnects before tearing down their registry. + + A reconnect can be waiting in adapter setup while shutdown begins. It must not republish + an adapter after the secondary registry is drained. Waiting is bounded by the adapter- + cleanup budget; a task that overruns is still blocked by the stopped runner state. + """ + pending = self._profile_failed_platforms + if not isinstance(pending, dict): + return + current = asyncio.current_task() + tasks: list[asyncio.Task] = [] + for profile_pending in pending.values(): + if not isinstance(profile_pending, dict): + continue + for task in profile_pending.values(): + if isinstance(task, asyncio.Task) and task is not current and not task.done(): + tasks.append(task) + for task in tasks: + task.cancel() + timeout = self._adapter_disconnect_timeout_secs() + if tasks and timeout > 0: + _done, unfinished = await asyncio.wait(tasks, timeout=timeout) + if unfinished: + logger.warning( + "Timed out waiting for %d secondary profile reconnect task(s) during shutdown", + len(unfinished), + ) + pending.clear() + + async def _start_secondary_profile_adapters(self) -> int: + """Bring up adapters for every non-active profile this gateway serves. + + Returns the count of connected secondary adapters; 0 unless ``gateway.multiplex_profiles``. + Each profile's adapters connect under its HERMES_HOME + secret scope, live in + ``self._profile_adapters[profile]``, and get a handler stamping ``source.profile``. Same- + platform credential collisions are refused here — the only point seeing every profile's + resolved credentials together. + """ + from gateway.run import ( + MultiplexConfigError, + SecondaryPortBindingConfigError, + _multiplex_profile_homes, + ) + if not getattr(self.config, "multiplex_profiles", False): + return 0 + + try: + from hermes_cli.profiles import get_active_profile_name + except Exception: + return 0 + + active = get_active_profile_name() or "default" + connected = 0 + # Resource claim -> owning profile. Credential claims stop two profiles polling the same + # account; listener claims stop sidecars with distinct credentials binding one endpoint. + claimed: Dict[tuple, str] = {} + for _plat, _ad in self.adapters.items(): + fp = self._adapter_credential_fingerprint(_ad) + if fp is not None: + claimed[(_plat, fp)] = active + listener_claim = self._adapter_listener_claim(_plat, _ad) + if listener_claim is not None: + claimed[listener_claim] = active + # A retryable primary still owns its credential and listener; reserve both while queued + # so a secondary cannot take the endpoint before the reconnect watcher retries it. + for retry_info in getattr(self, "_failed_platforms", {}).values(): + for claim_name in ("credential_claim", "listener_claim"): + retry_claim = retry_info.get(claim_name) + if isinstance(retry_claim, tuple): + claimed[retry_claim] = active + + profile_homes = _multiplex_profile_homes(self.config) + for profile_name, profile_home in profile_homes: + if profile_name == active: + continue # handled by the primary startup loop + try: + connected += await self._start_one_profile_adapters( + profile_name, profile_home, claimed + ) + except SecondaryPortBindingConfigError as e: + logger.warning( + "Skipping secondary profile '%s' due to port-binding config error: %s", + profile_name, + e, + ) + except MultiplexConfigError: + raise + except Exception as e: + logger.error( + "Failed to start adapters for profile '%s': %s", + profile_name, e, exc_info=True, + ) + + # Record the authoritative served set in runtime status for `hermes status`. "Served" + # means eligible for shared routing, HTTP prefixes, cron, and profile runtime scope — + # intentionally broader than profiles with a connected (or any) secondary adapter. + try: + from gateway.status import write_runtime_status + from gateway.pairing import PairingStore + served = [active] + sorted( + name for name, _home in profile_homes if name != active + ) + # Per-profile PairingStores so authz_mixin routes pairing checks to the right whitelist; + # the active profile's store is at its HERMES_HOME, other served profiles at their own. + for name in served: + if name and name not in self.pairing_stores: + self.pairing_stores[name] = ( + self.pairing_store + if name == active + else PairingStore(profile=name) + ) + write_runtime_status(served_profiles=served) + except Exception: + logger.debug("could not record served_profiles", exc_info=True) + + return connected + + async def _start_one_profile_adapters( + self, profile_name: str, profile_home: "Path", claimed: Dict[tuple, str] + ) -> int: + """Create+connect one profile's adapters under its runtime scope.""" + from gateway.run import ( + MultiplexConfigError, + SecondaryPortBindingConfigError, + _load_gateway_runtime_config, + _own_policy_open_startup_violation, + _platform_has_bot_credential, + _profile_runtime_scope, + ) + from gateway.config import load_gateway_config + from hermes_cli.env_loader import hydrate_profile_secret_sources + + # Hydrate external secret sources (1Password/vault/...) off-loop ONCE, then enter the scope + # without re-hydrating: the sync hydration is network-bound and would otherwise stall every + # other profile's heartbeat while this one boots (same class as the reconnect path). + await asyncio.to_thread(hydrate_profile_secret_sources, profile_home) + + with _profile_runtime_scope(profile_home, hydrate_secrets=False): + profile_runtime_cfg = _load_gateway_runtime_config() + from hermes_cli.plugins import discover_plugins + + discover_plugins() + + # Register this profile's own declarative shell hooks and outbound webhooks. The + # registration in start() runs before any profile scope exists and only sees the root + # profile's config, so without this a secondary profile's `hooks:` block is silently + # inert (its turns use a plugin manager keyed by resolved home). + try: + from hermes_cli.config import load_config as _load_profile_config + from agent.shell_hooks import ( + register_from_config as _register_shell_hooks, + ) + from agent.outbound_webhooks import ( + register_from_config as _register_outbound_webhooks, + ) + + _profile_hooks_cfg = _load_profile_config() + _register_shell_hooks(_profile_hooks_cfg, accept_hooks=False) + _register_outbound_webhooks(_profile_hooks_cfg) + except Exception: + logger.warning( + "shell-hook/webhook registration failed for profile '%s'", + profile_name, + exc_info=True, + ) + + profile_cfg = load_gateway_config() + violation = _own_policy_open_startup_violation(profile_cfg) + self._snapshot_profile_busy_modes(profile_name, profile_runtime_cfg) + if violation: + raise MultiplexConfigError( + f"Profile '{profile_name}' enables {violation}. " + "Enable GATEWAY_ALLOW_ALL_USERS or the platform allow-all flag " + "for that profile, or change dm_policy/group_policy away from " + "'open'." + ) + + port_binding_platforms = sorted( + platform.value + for platform, platform_config in profile_cfg.platforms.items() + if platform_config.enabled + and _platform_binds_port(platform.value, platform_config.extra) + ) + if port_binding_platforms: + joined = ", ".join(port_binding_platforms) + raise SecondaryPortBindingConfigError( + f"Profile '{profile_name}' enables port-binding platform(s) " + f"{joined}, but gateway.multiplex_profiles is on. The default " + f"profile owns the single shared HTTP listener and serves every " + f"profile through the /p/{profile_name}/ URL prefix. Remove " + f"these platform entries from profile '{profile_name}'s config.yaml " + f"or configure them only on the default profile." + ) + + profile_map = self._profile_adapters.setdefault(profile_name, {}) + connected = 0 + for platform, platform_config in profile_cfg.platforms.items(): + if not platform_config.enabled: + continue + # A platform enabled in a secondary profile's config.yaml may have no credential in that + # profile's secret scope — the shared YAML enables it for the default profile only. + # Building an adapter anyway would fan one inbound message out across every + # credential-less profile; mirror the primary loop's credential gate and skip. + if ( + getattr(self.config, "multiplex_profiles", False) + and not _platform_has_bot_credential(platform, platform_config) + ): + logger.info( + "[MULTIPLEX] Profile '%s': skipping %s - no bot credential " + "in this profile's secrets", + profile_name, + platform.value, + ) + continue + # Relay and WhatsApp are shared process-level ingress in multiplex mode (one connection + # owned by the active profile, route-stamped source.profile fans out). WhatsApp is one + # session per phone number; a secondary adapter would only retry-loop and stall startup. + if ( + getattr(self.config, "multiplex_profiles", False) + and platform in (Platform.RELAY, Platform.WHATSAPP) + ): + continue + try: + with _profile_runtime_scope(profile_home, hydrate_secrets=False): + adapter = self._create_adapter(platform, platform_config) + except Exception as e: + logger.error( + "[MULTIPLEX] Profile '%s': _create_adapter('%s') raised %s", + profile_name, + platform.value, + e, + exc_info=True, + ) + continue + if not adapter: + logger.warning( + "[MULTIPLEX] Profile '%s': skipping platform '%s' - adapter creation returned None", + profile_name, + platform.value, + ) + continue + + # Same-token conflict detection — refuse a duplicate poll. + credential_claim = self._adapter_credential_claim(platform, adapter) + if credential_claim is not None: + owner = claimed.get(credential_claim) + if owner is not None: + message = ( + f"Profile '{owner}' and '{profile_name}' both configure " + f"{platform.value} with the same credential. Give each " + f"profile its own {platform.value} credential." + ) + logger.error( + "Profile '%s' and '%s' both configure %s with the same " + "credential — refusing to start the duplicate (one " + "credential cannot be consumed twice). Give each profile " + "its own %s credential.", + owner, profile_name, platform.value, platform.value, + ) + self._update_platform_runtime_status( + f"{profile_name}:{platform.value}", + platform_state="fatal", + error_code="duplicate_credential", + error_message=message, + ) + # This adapter has not connected and therefore owns no resources to clean up. + # Calling disconnect here can mutate the shared platform state and, for a same- + # credential Photon adapter, shut down the primary profile's live sidecar. + continue + + listener_claim = self._adapter_listener_claim(platform, adapter) + if listener_claim is not None: + owner = claimed.get(listener_claim) + if owner is not None: + bind, port = listener_claim[-2:] + message = ( + f"Profile '{owner}' and '{profile_name}' both configure " + f"{platform.value} sidecars on the same listener. Configure " + f"a distinct listener for profile '{profile_name}'." + ) + logger.error( + "Profile '%s' and '%s' both configure %s sidecars on " + "%s:%s — refusing to start the duplicate listener. " + "Set platforms.%s.extra.sidecar_port to a distinct port " + "for profile '%s'.", + owner, + profile_name, + platform.value, + bind, + port, + platform.value, + profile_name, + ) + self._update_platform_runtime_status( + f"{profile_name}:{platform.value}", + platform_state="fatal", + error_code="duplicate_listener", + error_message=message, + ) + # Like credential conflicts, this adapter never connected + # and owns no resources that should be disconnected. + continue + + self._configure_profile_adapter(adapter, profile_name, platform) + + try: + with _profile_runtime_scope(profile_home, hydrate_secrets=False): + success = await self._connect_initial_adapter_with_timeout( + adapter, platform + ) + if success: + profile_map[platform] = adapter + # Restore persisted /voice state for this bot (#84872) — + # primary startup and every reconnect path already do. + self._sync_voice_mode_state_to_adapter(adapter) + if credential_claim is not None: + claimed[credential_claim] = profile_name + if listener_claim is not None: + claimed[listener_claim] = profile_name + connected += 1 + logger.info("✓ %s connected (profile: %s)", platform.value, profile_name) + else: + logger.warning("✗ %s failed to connect (profile: %s)", platform.value, profile_name) + await self._safe_adapter_disconnect(adapter, platform) + self._schedule_secondary_profile_startup_reconnect( + profile_name, platform, adapter + ) + except Exception as e: + logger.error("✗ %s error (profile: %s): %s", platform.value, profile_name, e) + await self._safe_adapter_disconnect(adapter, platform) + self._schedule_secondary_profile_startup_reconnect( + profile_name, platform, adapter + ) + return connected + + def _configure_profile_adapter( + self, + adapter: BasePlatformAdapter, + profile_name: str, + platform: Platform, + ) -> None: + """Install the profile-scoped handlers shared by startup and reconnect.""" + # Runtime status is process-scoped while message/config work is profile-scoped. Keep both + # dimensions in the key so dashboard/NAS health aggregation sees which secondary failed. + adapter._runtime_status_platform_key = f"{profile_name}:{platform.value}" + adapter.set_message_handler(self._make_profile_message_handler(profile_name)) + adapter.set_fatal_error_handler( + self._make_profile_fatal_error_handler(profile_name, platform) + ) + adapter.set_session_store(self.session_store) + # Declare credential ownership BEFORE any inbound event can be handled: adapter-level + # session keys (batching, _active_sessions, busy guard) are derived at ingress, before the + # handler stamps source.profile — without this every secondary bot would key into the + # default profile's `agent:main:` lane (see BasePlatformAdapter._session_key_profile). + _set_owner = getattr(adapter, "set_owner_profile", None) + if callable(_set_owner): + _set_owner(profile_name) + adapter.set_busy_session_handler( + self._make_profile_busy_session_handler(profile_name) + ) + _set_reaction = getattr(adapter, "set_reaction_handler", None) + if callable(_set_reaction): + _set_reaction(self._handle_reaction_event) + adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) + adapter.set_authorization_check( + self._make_adapter_auth_check(platform, profile_name=profile_name) + ) + adapter.set_platform_event_handler( + self._make_profile_platform_event_handler(profile_name) + ) + # Voice transcripts from this bot's channels dispatch through THIS + # adapter (primary wiring lives at connect time; see #75198). + self._bind_voice_input_callback(adapter) + text_modes = getattr(self, "_busy_text_modes_by_profile", None) + adapter._busy_text_mode = ( + text_modes.get(profile_name, self._busy_text_mode) + if isinstance(text_modes, dict) + else self._busy_text_mode + ) + # Secondary adapters always carry the profile they serve so prune + # paths namespace topic bindings correctly under multiplex (#76423). + adapter._hermes_profile_name = profile_name + + async def _run_secondary_profile_reconnect( + self, profile_name: str, platform: Platform + ) -> None: + """Reconnect a retryable secondary adapter under its own profile scope.""" + from gateway.run import _platform_has_bot_credential, _profile_runtime_scope, _reconnect_backoff + attempts = 0 + current_task = asyncio.current_task() + try: + while self._running: + adapter = None + try: + from hermes_cli.profiles import get_profile_dir + from hermes_cli.env_loader import hydrate_profile_secret_sources + from gateway.config import load_gateway_config + + profile_home = get_profile_dir(profile_name) + # Like the #16856 MCP discovery path, hydrate external secret + # sources off-loop so they cannot starve platform heartbeats. + await asyncio.to_thread( + hydrate_profile_secret_sources, profile_home + ) + with _profile_runtime_scope(profile_home, hydrate_secrets=False): + profile_config = load_gateway_config().platforms.get(platform) + if profile_config is None or not profile_config.enabled: + return + # Mirrors the startup credential gate: a credential removed from this + # profile's scope must not rebuild an adapter that would fan out turns. + if not _platform_has_bot_credential(platform, profile_config): + logger.info( + "Secondary %s reconnect skipped: no bot credential " + "(profile: %s)", + platform.value, + profile_name, + ) + return + adapter = self._create_adapter(platform, profile_config) + if adapter is None: + logger.warning( + "Secondary %s reconnect skipped: adapter unavailable (profile: %s)", + platform.value, + profile_name, + ) + return + self._configure_profile_adapter( + adapter, profile_name, platform + ) + success = await self._connect_adapter_with_timeout( + adapter, platform, is_reconnect=True + ) + + if success and self._running: + profile_map = self._profile_adapters.setdefault(profile_name, {}) + if platform not in profile_map: + profile_map[platform] = adapter + self._sync_voice_mode_state_to_adapter(adapter) + logger.info( + "✓ %s reconnected (profile: %s)", + platform.value, + profile_name, + ) + await self._redeliver_failed_obligations_for_platform( + platform, profile=profile_name + ) + return + # A newer reconnect already won the slot while this + # attempt was awaiting connect; do not replace it. + await self._safe_adapter_disconnect(adapter, platform) + return + + # Shutdown can begin mid-connect(): never republish a newly connected adapter + # after the registry has been drained; release its partial resources instead. + if success: + await self._safe_adapter_disconnect(adapter, platform) + return + + await self._safe_adapter_disconnect(adapter, platform) + if ( + getattr(adapter, "has_fatal_error", False) + and not getattr(adapter, "fatal_error_retryable", True) + ): + return + except asyncio.CancelledError: + if adapter is not None: + await self._safe_adapter_disconnect(adapter, platform) + raise + except Exception: + if adapter is not None: + await self._safe_adapter_disconnect(adapter, platform) + logger.debug( + "Secondary %s reconnect attempt failed (profile: %s)", + platform.value, + profile_name, + exc_info=True, + ) + + if not self._running: + return + attempts += 1 + backoff = _reconnect_backoff(attempts) + logger.info( + "Secondary %s reconnect retry in %ds (profile: %s)", + platform.value, + backoff, + profile_name, + ) + await asyncio.sleep(backoff) + finally: + pending = self._profile_failed_platforms + if isinstance(pending, dict): + profile_pending = pending.get(profile_name) + task = profile_pending.get(platform) if isinstance(profile_pending, dict) else None + if not isinstance(task, asyncio.Task) or task is current_task: + if isinstance(profile_pending, dict): + profile_pending.pop(platform, None) + if not profile_pending: + pending.pop(profile_name, None) + + def _schedule_secondary_profile_startup_reconnect( + self, profile_name: str, platform: Platform, adapter: BasePlatformAdapter + ) -> None: + """Queue a cold-start reconnect for a secondary adapter. + + Startup failures happen BEFORE ``self._running`` flips True, so the regular scheduler's + guard would drop the request. Park a task across startup and hand off to the scheduler once + live (``_profile_failed_platforms`` dedupes); release it if shutdown begins first. + Non-retryable failures are dropped as the regular scheduler would. + """ + if not getattr(adapter, "fatal_error_retryable", True): + return + if is_global_startup_conflict(getattr(adapter, "fatal_error_code", None)): + # Same startup contract as the primary path: a live foreign holder of this profile's + # token/identity is an ownership conflict, not a transient blip. Park it fatal (like + # ``duplicate_credential``) instead of retry-storming the token every backoff. + logger.error( + "[MULTIPLEX] Profile '%s': %s credential is held by another " + "gateway (%s) — parked, not retried. %s", + profile_name, + platform.value, + adapter.fatal_error_code, + adapter.fatal_error_message or "", + ) + self._update_platform_runtime_status( + f"{profile_name}:{platform.value}", + platform_state="fatal", + error_code=adapter.fatal_error_code, + error_message=adapter.fatal_error_message, + ) + return + + async def _await_running_then_schedule() -> None: + if self._running: + try: + self._schedule_secondary_profile_reconnect( + profile_name, platform, adapter + ) + except Exception: + # Same GC-time-exception hazard as the post-poll handoff + # below; surface it in gateway.log instead. + logger.exception( + "secondary-startup-reconnect handoff failed " + "(profile=%s platform=%s)", + profile_name, + platform.value, + ) + return + # Modest poll: startup completion has no dedicated event, and the reconnect runner's own + # backoff makes sub-100ms precision irrelevant. Bounded so a wedged startup cannot spin. + while not self._running and not self._shutdown_event.is_set(): + await asyncio.sleep(0.1) + if self._running and not self._shutdown_event.is_set(): + try: + self._schedule_secondary_profile_reconnect( + profile_name, platform, adapter + ) + except Exception: + # The handoff touches live registries; if it raises, the parked task dies as an + # unretrieved-task exception logged only at GC. Surface it where operators look. + logger.exception( + "secondary-startup-reconnect handoff failed " + "(profile=%s platform=%s)", + profile_name, + platform.value, + ) + + task = asyncio.create_task( + _await_running_then_schedule(), + name=f"secondary-startup-reconnect:{profile_name}:{platform.value}", + ) + background_tasks = getattr(self, "_background_tasks", None) + if not isinstance(background_tasks, set): + background_tasks = set() + self._background_tasks = background_tasks + background_tasks.add(task) + task.add_done_callback(background_tasks.discard) + + def _schedule_secondary_profile_reconnect( + self, profile_name: str, platform: Platform, adapter: BasePlatformAdapter + ) -> None: + """Schedule one runner-owned reconnect without sharing primary secrets.""" + if not self._running or not adapter.fatal_error_retryable: + return + pending = self._profile_failed_platforms + if not isinstance(pending, dict): + pending = {} + self._profile_failed_platforms = pending + profile_pending = pending.setdefault(profile_name, {}) + if platform in profile_pending: + return + task = asyncio.create_task( + self._run_secondary_profile_reconnect(profile_name, platform), + name=f"secondary-reconnect:{profile_name}:{platform.value}", + ) + profile_pending[platform] = task + background_tasks = getattr(self, "_background_tasks", None) + if not isinstance(background_tasks, set): + background_tasks = set() + self._background_tasks = background_tasks + background_tasks.add(task) + task.add_done_callback(background_tasks.discard) + + def _make_profile_fatal_error_handler( + self, profile_name: str, platform: Platform + ) -> Callable[[BasePlatformAdapter], Awaitable[None]]: + """Route a secondary-profile fatal error to that profile's reconnect slot.""" + async def _handler(adapter: BasePlatformAdapter) -> None: + await self._handle_profile_adapter_fatal_error(profile_name, platform, adapter) + + return _handler + + async def _handle_profile_adapter_fatal_error( + self, + profile_name: str, + platform: Platform, + adapter: BasePlatformAdapter, + ) -> None: + """Remove a failed multiplexed adapter without touching the primary slot. + + Secondaries live in ``_profile_adapters``, which the primary-only fatal handler ignores; + without this route a fatal secondary Discord client stayed live forever. + """ + profile_map = getattr(self, "_profile_adapters", {}).get(profile_name) + if not isinstance(profile_map, dict) or profile_map.get(platform) is not adapter: + logger.debug( + "Ignoring stale fatal error from secondary %s adapter (profile: %s)", + platform.value, + profile_name, + ) + return + profile_map.pop(platform, None) + await self._safe_adapter_disconnect(adapter, platform) + if not self._running: + return + self._schedule_secondary_profile_reconnect(profile_name, platform, adapter) + logger.error( + "Fatal %s adapter error for multiplexed profile %s (%s)", + platform.value, + profile_name, + adapter.fatal_error_code or "unknown", + ) + + def _make_profile_message_handler(self, profile_name: str): + """Return a message handler that stamps source.profile then delegates. + + Auth runs inside ``_handle_message`` *before* the agent-turn scope is installed. For + secondary profiles under multiplex, wrap the whole handler in ``_profile_runtime_scope`` + so allowlists/tokens from that profile's ``.env`` are visible to ``get_secret`` / authz. + """ + from gateway.run import _async_profile_runtime_scope + from hermes_cli.profiles import get_profile_dir + + try: + profile_home = get_profile_dir(profile_name) + except Exception: + profile_home = None + + async def _handler(event): + try: + if getattr(event, "source", None) is not None and not event.source.profile: + event.source.profile = profile_name + except Exception: + pass + if profile_home is not None: + async with _async_profile_runtime_scope(profile_home): + return await self._handle_message(event) + return await self._handle_message(event) + + return _handler + + def _make_profile_busy_session_handler(self, profile_name: str): + """Stamp an owning adapter's profile before resolving busy policy.""" + async def _handler(event, _session_key): + try: + if getattr(event, "source", None) is not None and not event.source.profile: + event.source.profile = profile_name + except Exception: + pass + routed_session_key = self._session_key_for_source(event.source) + return await self._handle_active_session_busy_message( + event, routed_session_key + ) + + return _handler + + def _make_default_profile_message_handler(self): + """Scope primary-adapter messages to their routed multiplex profile. + + Resolve the home per event so session lookup and transcript loading use the same profile + store as the agent run. Authorization stays with the transport profile (a routed profile + may intentionally have no bot credential/allowlist): the transport home is preserved on the + live source and never re-checked against the routed scope. Unrouted events keep the default. + """ + from gateway.run import _async_profile_runtime_scope, get_hermes_home + default_home = Path(get_hermes_home()) + + async def _handler(event): + source = event.source + # In-process only (SessionSource serialization ignores dynamic attrs). The route selects + # agent/session state, not which bot admitted the message — separate trust domains. + source._authorization_profile_home = default_home + if ( + not getattr(source, "profile", None) + and getattr(source, "profile_route_rejected", False) is not True + ): + from gateway.profile_routing import ProfileRouteRejected + + try: + source.profile = self._profile_name_for_source(source) + except ProfileRouteRejected: + # NOT write-only: the ``_handle_message`` ingress gate reads this exact marker + # and drops the message fail-closed (explicit route to an unserved profile). + source.profile_route_rejected = True + + profile_home = ( + self._resolve_profile_home_for_source(source) + if getattr(source, "profile", None) + else default_home + ) + async with _async_profile_runtime_scope(profile_home): + return await self._handle_message(event) + + return _handler + + def _primary_message_handler(self): + """Return the correctly scoped handler for a primary adapter.""" + if getattr(self.config, "multiplex_profiles", False): + return self._make_default_profile_message_handler() + return self._handle_message + + async def _handle_gateway_platform_event(self, event: dict, source) -> None: + """Authorize and publish one normalized adapter event to plugin hooks.""" + try: + from hermes_cli.lifecycle import has_hook, invoke_hook + + if not has_hook("gateway_platform_event"): + return + if not self._is_user_authorized_for_source(source): + return + invoke_hook("gateway_platform_event", **event) + except Exception: + # Observer failures must never break the adapter's update loop. + logger.debug("gateway_platform_event hook dispatch failed", exc_info=True) + + def _make_profile_platform_event_handler(self, profile_name: str): + """Bind platform-event auth and hook dispatch to one multiplex profile.""" + from gateway.run import _profile_runtime_scope + from hermes_cli.profiles import get_profile_dir + + try: + profile_home = get_profile_dir(profile_name) + except Exception: + profile_home = None + + async def _handler(event, source): + if getattr(source, "profile", None) is None: + source.profile = profile_name + if profile_home is not None: + with _profile_runtime_scope(profile_home): + return await self._handle_gateway_platform_event(event, source) + return await self._handle_gateway_platform_event(event, source) + + return _handler + + def _make_default_profile_platform_event_handler(self): + """Scope primary-transport events to their routed multiplex profile.""" + from gateway.run import _profile_runtime_scope, get_hermes_home + default_home = Path(get_hermes_home()) + + async def _handler(event, source): + source._authorization_profile_home = default_home + with _profile_runtime_scope(self._resolve_profile_home_for_source(source)): + return await self._handle_gateway_platform_event(event, source) + + return _handler + + def _primary_platform_event_handler(self): + if getattr(self.config, "multiplex_profiles", False): + return self._make_default_profile_platform_event_handler() + return self._handle_gateway_platform_event + + @staticmethod + def _adapter_credential_claim( + platform: Platform, adapter: Any + ) -> Optional[tuple]: + """Return the exclusive credential resource claimed by an adapter.""" + from gateway.run import GatewayRunner + fingerprint = GatewayRunner._adapter_credential_fingerprint(adapter) + if fingerprint is None: + return None + return (platform, fingerprint) + + @staticmethod + def _adapter_listener_claim(platform: Platform, adapter: Any) -> Optional[tuple]: + """Return the exclusive listener resource claimed by an adapter. + + Sidecars with different credentials still cannot share a bind+port; expose it as a claim so + multiplex startup rejects the later adapter before connect()/disconnect() disturb the first. + """ + if getattr(platform, "value", None) != "photon": + return None + bind = getattr(adapter, "_sidecar_bind", None) + port = getattr(adapter, "_sidecar_port", None) + if not isinstance(bind, str) or not bind.strip(): + return None + try: + port = int(port) + except (TypeError, ValueError): + return None + return ("listener", "photon", bind.strip().lower(), port) + + @staticmethod + def _adapter_credential_fingerprint(adapter: Any) -> Optional[str]: + """Return a stable, log-safe fingerprint of an adapter's credential. + + Salted hash (never the credential) used to detect two profiles sharing one platform + credential; None when no credential is discoverable (conflict detection is then skipped). + """ + token = None + for attr in ( + "token", + "bot_token", + "_token", + "api_token", + "_bot_token", + # Photon/Spectrum authenticates with project credentials, not a bot token; including + # its secret stops multiplexed profiles spawning rival sidecars for one account/port. + "_project_secret", + # Feishu/Lark authenticates with an app_id/app_secret pair (one WebSocket per app). + # app_id is stable, log-safe and already the adapter's _app_lock_identity, so including + # it lets the multiplex guard refuse cloned profiles competing for the same app. + "_app_id", + # Same class: Teams (client_id/client_secret) and WeCom + # (bot_id/secret) authenticate with an app-style id pair too. + "_client_id", + "_bot_id", + ): + val = getattr(adapter, attr, None) + if isinstance(val, str) and val.strip(): + token = val.strip() + break + # Many adapters (e.g. Discord) store the token on their `config` sub-object. Without this + # lookup they return None, the same-token check is silently skipped, and every profile's + # adapter polls the same bot token — a per-message race over which one answers. + if not token: + cfg = getattr(adapter, "config", None) + if cfg is not None: + for attr in ("token", "bot_token"): + val = getattr(cfg, attr, None) + if isinstance(val, str) and val.strip(): + token = val.strip() + break + if not token: + config = getattr(adapter, "config", None) + val = getattr(config, "token", None) + if isinstance(val, str) and val.strip(): + token = val.strip() + if not token: + return None + import hashlib + return hashlib.sha256(("hermes-mux:" + token).encode("utf-8")).hexdigest()[:16] + + def _create_adapter( + self, + platform: Platform, + config: Any, + ) -> Optional[BasePlatformAdapter]: + """Create an adapter and bind it to this gateway runner. + + Every lifecycle path (primary/secondary startup, reconnect) uses this method; keep runner + binding here so adapters can resolve inbound profile routes before handlers or connect(). + """ + adapter = self._instantiate_adapter(platform, config) + if adapter is not None: + adapter.gateway_runner = self + return adapter + + def _instantiate_adapter( + self, + platform: Platform, + config: Any, + ) -> Optional[BasePlatformAdapter]: + """Instantiate the appropriate adapter for a platform. + + Checks platform_registry (plugin adapters) first, then the built-in table of core platforms. + """ + from gateway.run import _instantiate_builtin_adapter + if hasattr(config, "extra") and isinstance(config.extra, dict): + config.extra.setdefault( + "group_sessions_per_user", + self.config.group_sessions_per_user, + ) + config.extra.setdefault( + "thread_sessions_per_user", + getattr(self.config, "thread_sessions_per_user", False), + ) + + # ── Plugin-registered platforms (checked first) ─────────────────── + try: + from gateway.platform_registry import platform_registry + if platform_registry.is_registered(platform.value): + adapter = platform_registry.create_adapter(platform.value, config) + if adapter is not None: + return adapter + # Registered but failed to instantiate — don't silently fall + # through to built-ins (there are none for plugin platforms). + logger.error( + "Platform '%s' is registered but adapter creation failed " + "(check dependencies and config)", + platform.value, + ) + return None + except Exception as e: + logger.debug("Platform registry lookup for '%s' failed: %s", platform.value, e) + # Fall through to built-in adapters below + + return _instantiate_builtin_adapter(platform, config) + + def _make_adapter_auth_check( + self, + platform: Platform, + profile_name: Optional[str] = None, + ) -> Callable[[str, Optional[str], Optional[str]], bool]: + """Build a platform-bound auth callback for adapter use. + + Adapters fetching external context (e.g. Slack ``conversations.replies``) use it via + ``_is_sender_authorized`` to mark non-allowlisted senders unverified (prompt-injection + mitigation). Delegates to :meth:`_is_user_authorized` so the full auth chain stays the single + source of truth. ``profile_name`` binds a secondary adapter to its own secret scope; for the + shared primary (None) the ``profile_routes`` match is stamped on the source so the routed + profile's pairing store is consulted while allowlist reads stay under the transport home. + """ + from gateway.run import get_hermes_home + multiplex = bool(getattr(self.config, "multiplex_profiles", False)) + transport_home = ( + Path(get_hermes_home()) if multiplex and profile_name is None else None + ) + + def check( + user_id: str, + chat_type: Optional[str] = None, + chat_id: Optional[str] = None, + *, + is_bot: bool = False, + thread_id: Optional[str] = None, + ) -> bool: + if not user_id: + return False + source = SessionSource( + platform=platform, + chat_id=chat_id or "", + chat_type=chat_type or "group", + user_id=user_id, + thread_id=thread_id, + is_bot=bool(is_bot), + profile=profile_name, + ) + # Same in-process transport provenance ``build_source`` retains, so adapter-level policy + # reads (config.yaml group_allowed_chats, allow_from) resolve the receiving adapter even + # once the routed profile is stamped below. + registry = ( + (getattr(self, "_profile_adapters", None) or {}).get(profile_name) + if profile_name + else getattr(self, "adapters", None) + ) or {} + adapter = registry.get(platform) + if adapter is not None: + source._transport_adapter_ref = _weakref.ref(adapter) + if transport_home is None: + return self._is_user_authorized(source) + source._authorization_profile_home = transport_home + from gateway.profile_routing import ProfileRouteRejected + + try: + source.profile = self._profile_name_for_source(source) + except ProfileRouteRejected: + # Same fail-closed outcome as the ingress gate in + # ``_handle_message`` for a route to an unserved profile. + return False + return self._is_user_authorized_for_source(source) + return check diff --git a/gateway/run_agent_cache.py b/gateway/run_agent_cache.py new file mode 100644 index 0000000000..d8be1190dc --- /dev/null +++ b/gateway/run_agent_cache.py @@ -0,0 +1,1162 @@ +"""Agent cache, session model overrides, turn leases, run generations and conversation-scope reset methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import inspect +import threading +import time +from contextlib import suppress +from gateway.config import Platform +from gateway.session import SessionSource, build_session_context_prompt +from hermes_cli.config import cfg_get +from typing import Any, Dict, List, Optional + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayAgentCacheMixin: + """Agent cache, session model overrides, turn leases, run generations and conversation-scope reset methods for GatewayRunner.""" + + @classmethod + def _empty_honcho_cache_busting_config(cls) -> dict[str, Any]: + return {key: None for key in cls._HONCHO_CACHE_BUSTING_KEYS} + + @classmethod + def _extract_honcho_cache_busting_config(cls) -> dict[str, Any]: + """Extract Honcho identity keys, memoized by honcho.json mtime.""" + try: + from plugins.memory.honcho.client import HonchoClientConfig, resolve_config_path + + path = resolve_config_path() + try: + mtime_ns = path.stat().st_mtime_ns + except OSError: + mtime_ns = None + memo_key = (str(path), mtime_ns) + cached = cls._HONCHO_CACHE_BUSTING_MEMO.get(memo_key) + if cached is not None: + return dict(cached) + + hcfg = HonchoClientConfig.from_global_config(config_path=path) + aliases = hcfg.user_peer_aliases or {} + values = { + "honcho.peer_name": hcfg.peer_name, + "honcho.ai_peer": hcfg.ai_peer, + "honcho.pin_peer_name": bool(hcfg.pin_peer_name), + "honcho.runtime_peer_prefix": hcfg.runtime_peer_prefix or "", + "honcho.user_peer_aliases": sorted(aliases.items()) if isinstance(aliases, dict) else [], + } + cls._HONCHO_CACHE_BUSTING_MEMO = {memo_key: values} + return dict(values) + except Exception: + return cls._empty_honcho_cache_busting_config() + + @classmethod + def _extract_cache_busting_config(cls, user_config: dict | None) -> dict: + """Pull values that must bust the cached agent, as a flat dict keyed by 'section.key'. + + Missing keys / non-dict sections yield None, which still enters the signature ('absent' vs + 'present-and-null' differ). Includes the live tool registry generation: MCP reloads mutate + the registry without touching config.yaml, and cached agents freeze their tool schemas. + """ + out: Dict[str, Any] = {} + cfg = user_config if isinstance(user_config, dict) else {} + for section, key in cls._CACHE_BUSTING_CONFIG_KEYS: + section_val = cfg.get(section) + if section == "checkpoints" and isinstance(section_val, bool): + # Preserve legacy ``checkpoints: true`` behavior. A live + # toggle must still rebuild the cached agent. + out[f"{section}.{key}"] = section_val if key == "enabled" else None + elif isinstance(section_val, dict): + out[f"{section}.{key}"] = section_val.get(key) + else: + out[f"{section}.{key}"] = None + try: + from tools.registry import registry + + out["tools.registry_generation"] = getattr(registry, "_generation", None) + except Exception: + out["tools.registry_generation"] = None + + # Honcho identity-mapping keys live in honcho.json, not user_config. + # Only read that file when Honcho is the active memory provider. + provider = cfg_get(cfg, "memory", "provider") + if isinstance(provider, str) and provider.lower() == "honcho": + out.update(cls._extract_honcho_cache_busting_config()) + else: + out.update(cls._empty_honcho_cache_busting_config()) + + return out + + @staticmethod + def _agent_config_signature( + model: str, + runtime: dict, + enabled_toolsets: list, + ephemeral_prompt: str, + cache_keys: dict | None = None, + user_id: str | None = None, + user_id_alt: str | None = None, + skip_context_files: bool = False, + ) -> str: + """Compute a stable string key from agent config values. + + Signature change → cached AIAgent rebuilt; unchanged → reused (frozen prompt + schemas for + cache hits). Callers pass ``_extract_cache_busting_config(user_config)`` so config.yaml + edits apply on the next message. ``user_id`` / ``user_id_alt`` participate because Honcho + freezes them into ``HonchoSessionManager`` at init; omitting them in shared-thread keys + (``thread_sessions_per_user=False``) would attribute one user's messages to another's peer. + """ + import hashlib, json as _j + + # Fingerprint the FULL credential, not a short prefix: OAuth/JWT-style tokens often share a + # common prefix (e.g. "eyJhbGci"), so a prefix would give false cache hits across auth switches. + _api_key = str(runtime.get("api_key", "") or "") + _api_key_fingerprint = hashlib.sha256(_api_key.encode()).hexdigest() if _api_key else "" + + _cache_keys_sorted = sorted((cache_keys or {}).items()) + + blob = _j.dumps( + [ + model, + _api_key_fingerprint, + runtime.get("base_url", ""), + runtime.get("provider", ""), + runtime.get("requested_provider", ""), + runtime.get("api_mode", ""), + sorted((runtime.get("capabilities") or {}).items()), + sorted(enabled_toolsets) if enabled_toolsets else [], + # reasoning_config excluded — it's set per-message on the + # cached agent and doesn't affect system prompt or tools. + ephemeral_prompt or "", + _cache_keys_sorted, + str(user_id or ""), + str(user_id_alt or ""), + # skip_context_files changes the agent's frozen system prompt (context files in vs out): + # a toggled edit must rebuild the cached agent, not silently reuse it. + bool(skip_context_files), + ], + sort_keys=True, + default=str, + ) + return hashlib.sha256(blob.encode()).hexdigest()[:16] + + def _rehydrate_session_model_override(self, session_key: str) -> None: + """Lazily restore a persisted /model override after a gateway restart. + + ``_session_model_overrides`` is in-memory only. Non-secret parts (model/provider/base_url) + are written through on /model (cleared on /new) and read back here on first use; api_key + is never persisted and is re-resolved. No-op when an in-memory override or nothing exists. + """ + from gateway.run import _resolve_runtime_agent_kwargs_for_provider + _rehydrate_state = self._peek_session_state(session_key) + if ( + _rehydrate_state is not None + and _rehydrate_state.conversation.model_override is not None + ): + return + store = getattr(self, "session_store", None) + if store is None: + return + try: + persisted = store.get_model_override(session_key) + except Exception: + logger.debug( + "Failed to read persisted session model override", exc_info=True + ) + return + if not persisted: + return + override: Dict[str, Any] = { + "model": persisted.get("model"), + "provider": persisted.get("provider"), + "base_url": persisted.get("base_url"), + } + provider = persisted.get("provider") + if provider: + # Re-resolve credentials for the persisted provider. On failure (e.g. credentials + # removed since the switch) keep the credential-less override — + # _resolve_session_agent_runtime falls back to env resolution and layers model/provider. + try: + runtime = _resolve_runtime_agent_kwargs_for_provider(provider) + override["api_key"] = runtime.get("api_key") + override["api_mode"] = runtime.get("api_mode") + override["credential_pool"] = runtime.get("credential_pool") + override["request_overrides"] = dict( + runtime.get("request_overrides") or {} + ) + override["requested_provider"] = runtime.get("requested_provider") + override["capabilities"] = dict(runtime.get("capabilities") or {}) + override["max_tokens"] = runtime.get("max_tokens") + if not override.get("base_url"): + override["base_url"] = runtime.get("base_url") + except Exception: + logger.debug( + "Credential re-resolution failed for persisted override " + "(provider=%s); using credential-less override", + provider, exc_info=True, + ) + self._session_state(session_key).conversation.model_override = override + logger.info( + "Rehydrated persisted /model override for session=%s: model=%s provider=%s", + session_key, override.get("model"), provider or "", + ) + + def _apply_session_model_override( + self, session_key: str, model: str, runtime_kwargs: dict + ) -> tuple: + """Apply /model session overrides if present, returning (model, runtime_kwargs). + + Overrides take precedence over config.yaml defaults so the switched model is actually used; + ``None`` fields are skipped so partial overrides don't clobber valid defaults. + """ + from gateway.run import _credential_pool_for_provider + _apply_state = self._peek_session_state(session_key) + override = _apply_state.conversation.model_override if _apply_state else None + if not override: + return model, runtime_kwargs + model = override.get("model", model) + for key in ( + "provider", + "requested_provider", + "api_key", + "base_url", + "api_mode", + "credential_pool", + "capabilities", + "max_tokens", + ): + val = override.get(key) + if val is not None: + runtime_kwargs[key] = val + # request_overrides reflects the switched-to provider; apply whenever the override recorded + # it (even as None) so switching to a provider without configured overrides clears a stale + # value left by the default provider's runtime resolution. + if "request_overrides" in override: + override_request_overrides = override.get("request_overrides") + if isinstance(override_request_overrides, dict) and override_request_overrides: + runtime_kwargs["request_overrides"] = dict(override_request_overrides) + else: + runtime_kwargs["request_overrides"] = override_request_overrides + if ( + runtime_kwargs.get("api_key") + and runtime_kwargs.get("credential_pool") is None + and override.get("provider") + ): + runtime_kwargs["credential_pool"] = _credential_pool_for_provider( + override.get("provider") + ) + return model, runtime_kwargs + + def _snapshot_session_model_override(self, session_key: str) -> dict: + """Capture a gateway session override before a one-turn switch.""" + _snap_state = self._peek_session_state(session_key) + override = _snap_state.conversation.model_override if _snap_state else None + return { + "had_override": override is not None, + "override": dict(override) if override is not None else None, + } + + def _restore_session_model_override(self, session_key: str, snapshot: dict) -> None: + """Restore the session override captured before a one-turn switch.""" + if not session_key: + return + if snapshot.get("had_override"): + self._session_state(session_key).conversation.model_override = dict( + snapshot.get("override") or {} + ) + else: + _rst_state = self._peek_session_state(session_key) + if _rst_state is not None: + _rst_state.conversation.model_override = None + self._evict_cached_agent(session_key) + + def _is_intentional_model_switch(self, session_key: str, agent_model: str) -> bool: + """Return True if *agent_model* matches an active /model session override.""" + _ims_state = self._peek_session_state(session_key) + override = _ims_state.conversation.model_override if _ims_state else None + return override is not None and override.get("model") == agent_model + + def _release_running_agent_state( + self, + session_key: str, + *, + run_generation: Optional[int] = None, + ) -> bool: + """Pop ALL per-running-agent state entries for ``session_key``; True when cleared. + + Call at every site that ends a running turn, whatever the cause. State that PERSISTS + across turns (model overrides, voice mode, pending approvals, update prompt) is NOT + touched. With ``run_generation``, only clear if that generation is still current, so a + stale async unwind bumped by /stop or /new cannot clobber a newer run (returns False). + """ + if not session_key: + return False + if run_generation is not None and not self._is_session_run_current( + session_key, run_generation + ): + return False + state = self._peek_session_state(session_key) + if state is not None: + lease = state.turn.lease + if lease is not None: + try: + lease.release() + except Exception: + logger.debug( + "Failed to release active session slot", exc_info=True + ) + # One structured reset instead of a drifting pop-list. Turn-lease tokens are deliberately NOT + # cleared here — _release_turn_lease owns them. + state.turn.clear() + # Turn boundary: a running-agent slot was just released; persist the new (lower) in-flight count + # so the dashboard readout stays current. Preserves gateway_state (see _persist_active_agents). + self._persist_active_agents() + return True + + def _release_turn_lease(self, session_key: str, run_generation: int) -> bool: + """Release the turn lease acquired by (``session_key``, ``run_generation``). + + Token map is keyed by (routing key, run generation), so a stale unwind pops only ITS token + and the registry's identity check refuses it if a newer turn holds the lease. Idempotent. + """ + if not session_key: + return False + registry = getattr(self, "_turn_leases", None) + state = self._peek_session_state(session_key) + if state is None or registry is None: + return False + turn = state.turn + if turn.lease_token is None or turn.lease_generation != run_generation: + return False + token = turn.lease_token + turn.lease_token = None + turn.lease_generation = None + try: + return registry.release(token) + except Exception: + logger.debug("Failed to release turn lease", exc_info=True) + return False + + def _rebind_turn_lease( + self, session_key: str, run_generation: int, new_session_id: str + ) -> bool: + """Follow a mid-turn session_id rotation with the held turn lease. + + Compression can rotate ``session_entry.session_id`` mid-turn; the flush targets the NEW id, + so the serialization boundary must follow or an alias key resolving the new id could start + a concurrent turn the lease never sees. Call at every mid-turn reassignment; no-op if no token. + """ + if not session_key or not new_session_id: + return False + registry = getattr(self, "_turn_leases", None) + state = self._peek_session_state(session_key) + if state is None or registry is None: + return False + turn = state.turn + if turn.lease_token is None or turn.lease_generation != run_generation: + return False + try: + return registry.rebind(turn.lease_token, new_session_id) + except Exception: + logger.debug("Failed to rebind turn lease", exc_info=True) + return False + + def _clear_conversation_scope(self, session_key: str, *, reason: str) -> None: + """Clear ALL conversation-scoped per-session state for ``session_key``. + + THE single conversation-boundary funnel — call this and nothing else at /new, /resume, + auto-reset (idle/daily/suspended), expiry finalization and compression-exhausted reset. + New conversation-scoped dicts go in _CONVERSATION_SCOPED_STATE so every boundary picks + them up (hand-copied pop-lists drifted). Turn-scoped state (_running_agents/_ts, slot + leases, turn-lease tokens) is owned by _release_running_agent_state and NOT cleared. Idle + agent-cache eviction is NOT a boundary (a resumed turn rebuilds from these). getattr-guarded. + """ + from gateway.run import _CONVERSATION_SCOPED_STATE + if not session_key: + return + # Structural clear: every conversation-scoped field resets in one + # call — no per-attribute pop-list to drift. + state = self._peek_session_state(session_key) + if state is not None: + state.conversation.clear() + # Legacy plain-dict stores still in _CONVERSATION_SCOPED_STATE (not yet folded into + # SessionState), e.g. _pending_model_notes. SessionState-backed names resolve to MutableMapping + # views (not dict), so the isinstance(dict) guard skips them — already handled above. + for attr in _CONVERSATION_SCOPED_STATE: + store = getattr(self, attr, None) + if isinstance(store, dict): + store.pop(session_key, None) + self._clear_session_boundary_security_state(session_key) + logger.debug( + "Cleared conversation scope for %s (%s)", session_key, reason + ) + + def _clear_session_boundary_security_state(self, session_key: str) -> None: + """Clear per-session control state that must not survive a boundary switch.""" + if not session_key: + return + + pending_skills_reload_notes = getattr( + self, "_pending_skills_reload_notes", None + ) + if isinstance(pending_skills_reload_notes, dict): + pending_skills_reload_notes.pop(session_key, None) + + _sec_state = self._peek_session_state(session_key) + if _sec_state is not None: + _sec_state.persistent.approvals = None + _sec_state.persistent.update_prompt_pending = False + + try: + from tools import slash_confirm as _slash_confirm_mod + except Exception: + _slash_confirm_mod = None + if _slash_confirm_mod is not None: + try: + _slash_confirm_mod.clear(session_key) + except Exception as e: + logger.debug( + "Failed to clear slash-confirm state for session boundary %s: %s", + session_key, + e, + ) + + try: + from tools.approval import clear_session as _clear_approval_session + except Exception: + return + + try: + _clear_approval_session(session_key) + except Exception as e: + logger.debug( + "Failed to clear approval state for session boundary %s: %s", + session_key, + e, + ) + + def _begin_session_run_generation(self, session_key: str) -> int: + """Claim a fresh, monotonically increasing run generation token for ``session_key``. + + If /stop or /new invalidates the token while the old worker is still unwinding, the late + result is recognized and dropped instead of bleeding into the fresh session. + """ + if not session_key: + return 0 + persistent = self._session_state(session_key).persistent + # Monotonic by design (#28686): incremented here, NEVER reset. + persistent.run_generation = int(persistent.run_generation) + 1 + return persistent.run_generation + + def _invalidate_session_run_generation(self, session_key: str, *, reason: str = "") -> int: + """Invalidate any in-flight run token for ``session_key``.""" + generation = self._begin_session_run_generation(session_key) + if reason: + logger.info( + "Invalidated run generation for %s → %d (%s)", + session_key, + generation, + reason, + ) + return generation + + def _is_session_run_current(self, session_key: str, generation: int) -> bool: + """Return True when ``generation`` is still current for ``session_key``.""" + if not session_key: + return True + state = self._peek_session_state(session_key) + current = state.persistent.run_generation if state is not None else 0 + return int(current) == int(generation) + + def _bind_adapter_run_generation( + self, + adapter: Any, + session_key: str, + generation: int | None, + ) -> None: + """Bind a gateway run generation to the adapter's active-session event.""" + if not adapter or not session_key or generation is None: + return + try: + interrupt_event = getattr(adapter, "_active_sessions", {}).get(session_key) + if interrupt_event is not None: + setattr(interrupt_event, "_hermes_run_generation", int(generation)) + except Exception: + pass + + async def _interrupt_and_clear_session( + self, + session_key: str, + source: SessionSource, + *, + interrupt_reason: str, + invalidation_reason: str, + release_running_state: bool = True, + ) -> None: + """Interrupt the current run and clear queued session state consistently.""" + from gateway.run import _AGENT_PENDING_SENTINEL, _reap_gateway_turn_processes, request_hard_interrupt + if not session_key: + return + _iac_state = self._peek_session_state(session_key) + running_agent = _iac_state.turn.agent if _iac_state else None + _process_task_id = "" + _process_baseline = None + if running_agent and running_agent is not _AGENT_PENDING_SENTINEL: + request_hard_interrupt(running_agent, interrupt_reason) + _process_task_id = getattr( + running_agent, "_gateway_turn_process_task_id", "" + ) + _process_baseline = getattr( + running_agent, "_gateway_turn_process_baseline", None + ) + # Bump the generation BEFORE scheduling the reap thread and capture the post-bump value: + # task_id is session-scoped, so a replacement turn spawning before the reap runs bumps it + # again and the closure sees a stale generation and skips — the replacement's own baseline + # covers its cleanup, so nothing stays unreaped. + _generation_at_interrupt = self._invalidate_session_run_generation( + session_key, reason=invalidation_reason + ) + if _process_task_id and _process_baseline is not None: + threading.Thread( + target=_reap_gateway_turn_processes, + args=(_process_task_id, _process_baseline), + kwargs={ + "source": "gateway_turn_interrupt", + "is_still_current": lambda: self._is_session_run_current( + session_key, _generation_at_interrupt + ), + }, + name=f"gateway-turn-reaper-{_process_task_id[:12]}", + daemon=True, + ).start() + adapter = self._adapter_for_source(source) + interrupt_session_activity = getattr( + type(adapter), "interrupt_session_activity", None + ) + if adapter and callable(interrupt_session_activity): + metadata = self._thread_metadata_for_source(source) + try: + params = inspect.signature(interrupt_session_activity).parameters + accepts_metadata = "metadata" in params or any( + param.kind is inspect.Parameter.VAR_KEYWORD + for param in params.values() + ) + except (TypeError, ValueError): + accepts_metadata = False + if accepts_metadata: + await adapter.interrupt_session_activity( + session_key, source.chat_id, metadata=metadata + ) + else: + await adapter.interrupt_session_activity(session_key, source.chat_id) + if adapter and hasattr(adapter, "get_pending_message"): + adapter.get_pending_message(session_key) # consume and discard + if _iac_state is not None: + _iac_state.persistent.pending_command_text = None + if release_running_state: + self._release_running_agent_state(session_key) + # Evict the cached agent: ``_interrupt_requested`` is only cleared by the turn finalizer, + # so on a hung/still-draining run the flag survives and silently kills the session's NEXT + # message (interrupted=True, api_calls=0, empty response). Like /new and /model, the next + # message rebuilds from history; the old agent keeps its flag so a hung drain still dies. + self._evict_cached_agent(session_key) + + async def _refresh_agent_cache_message_count( + self, session_key: str, session_id: Optional[str] + ) -> None: + """Re-baseline a cached agent's stored message_count after THIS turn. + + The coherence guard compares on-disk ``message_count`` against the BUILD-time snapshot and + rebuilds on mismatch; without re-baselining after our own rows flush, every turn would + rebuild and destroy prompt caching. Only the count is refreshed (``_sig`` untouched), only + if the same agent is still cached, never when the entry records a different ``session_id`` + (another conversation's baseline). DB errors leave the snapshot as-is (one spare rebuild). + """ + from gateway.run import _AGENT_PENDING_SENTINEL + if self._session_db is None or not session_id: + return + _cache_lock = getattr(self, "_agent_cache_lock", None) + _cache = getattr(self, "_agent_cache", None) + if not _cache_lock or _cache is None: + return + try: + _sess_row = await self._session_db.get_session(session_id) + _live = _sess_row.get("message_count", 0) if _sess_row else None + except Exception: + return + if _live is None: + return + with _cache_lock: + cached = _cache.get(session_key) + # Only re-baseline a live 3-tuple entry; skip pending sentinels, legacy 2-tuples (they opt + # out of the guard), and entries evicted/rebuilt mid-turn. + if ( + isinstance(cached, tuple) + and len(cached) > 2 + and cached[0] is not _AGENT_PENDING_SENTINEL + ): + # A snapshot taken for a different session_id (same session_key, different conversation) + # belongs to a different DB row — leave it alone. + _snapshot_sid = cached[3] if len(cached) > 3 else None + if _snapshot_sid is not None and _snapshot_sid != session_id: + return + if cached[2] != _live: + if _snapshot_sid is None: + # Legacy 3-tuple: preserve the 3-element shape for callers indexing ``cached[2]``. + _cache[session_key] = (cached[0], cached[1], _live) + else: + _cache[session_key] = ( + cached[0], cached[1], _live, _snapshot_sid, + ) + + def _set_pending_turn_sidecar_notes(self, session_key: str, notes: List[str]) -> None: + """Stage per-turn must-deliver notes for the next agent run (one-shot).""" + if not session_key or not notes: + return + self._session_state(session_key).conversation.sidecar_notes = list(notes) + + def _consume_pending_turn_sidecar_notes(self, session_key: str) -> List[str]: + if not session_key: + return [] + state = self._peek_session_state(session_key) + if state is None: + return [] + staged = state.conversation.sidecar_notes + state.conversation.sidecar_notes = [] + return list(staged) if isinstance(staged, list) else [] + + def _voice_channel_sidecar_note(self, event, source: SessionSource, session_key: str) -> Optional[str]: + """Return a ``[Voice channel now: ...]`` note when VC state changed. + + Unchanged state returns ``None`` so per-turn member/speaking churn can't touch the prompt. + """ + if source.platform != Platform.DISCORD: + return None + adapter = self.adapters.get(Platform.DISCORD) + guild_id = self._get_guild_id(event) + if not (guild_id and adapter and hasattr(adapter, "get_voice_channel_context")): + return None + try: + vc_now = adapter.get_voice_channel_context(guild_id) or "" + except Exception: + logger.debug("voice-channel context read failed", exc_info=True) + return None + vc_prev = None + if session_key: + _vc_state = self._session_state(session_key) + vc_prev = _vc_state.conversation.vc_last + _vc_state.conversation.vc_last = vc_now + if vc_now == (vc_prev if vc_prev is not None else ""): + return None + if not vc_now: + return "[Voice channel now: not connected to a voice channel]" + return f"[Voice channel now: {vc_now}]" + + def _pinned_session_context_prompt( + self, context, redact_pii: bool, session_key: Optional[str] + ) -> str: + """Return the session-context prompt, pinned per session. + + Key hit → pinned bytes reused VERBATIM (immune to renderer nondeterminism); key miss → + re-render ``build_session_context_prompt`` and re-pin (rename, topic edit, /sethome, ...). + """ + _eph_key = self._ephemeral_change_key(context, redact_pii) + _eph_pin = None + if session_key: + _pin_state = self._peek_session_state(session_key) + _eph_pin = _pin_state.conversation.ephemeral_pin if _pin_state else None + if _eph_pin is not None and _eph_pin[0] == _eph_key: + return _eph_pin[1] + text = build_session_context_prompt(context, redact_pii=redact_pii) + if session_key: + self._session_state(session_key).conversation.ephemeral_pin = ( + _eph_key, + text, + ) + return text + + @staticmethod + def _ephemeral_change_key(context, redact_pii: bool) -> str: + """Hash the exact inputs ``build_session_context_prompt`` renders. + + Invariant (tests/gateway/test_prompt_tail_freeze.py): any input whose change alters the + rendered bytes MUST appear here — omission means a stale pinned prompt; extras only re-render. + """ + import hashlib + + src = context.source + platform = src.platform.value if src.platform else "" + + discord_ids: tuple = () + discord_tools = "" + if src.platform == Platform.DISCORD: + from gateway.session import _discord_tools_loaded + + discord_tools = "1" if _discord_tools_loaded() else "0" + discord_ids = ( + str(src.guild_id or ""), + str(src.parent_chat_id or ""), + str(src.thread_id or ""), + str(src.chat_id or ""), + # Only PRESENCE is rendered (the id itself arrives per-turn in the user message) — + # keying on the value would re-render every message for zero byte change. + "1" if src.message_id else "0", + ) + + # Slack's capability-aware platform note is gated on _slack_tools_loaded() — the gate state must + # be in the key (same parity contract as the Discord gate above) so a config / MCP-registration + # flip re-renders once instead of serving a stale pinned note for the rest of the session. + slack_tools = "" + if src.platform == Platform.SLACK: + from gateway.session import _slack_tools_loaded + + slack_tools = "1" if _slack_tools_loaded() else "0" + + try: + from hermes_constants import display_hermes_home + + home_display = str(display_hermes_home()) + except Exception: + home_display = "" + + key_tuple = ( + platform, + str(src.chat_id or ""), + str(src.thread_id or ""), + str(src.chat_type or ""), + str(src.chat_name or ""), + str(src.chat_topic or ""), + str(src.user_name or ""), + str(src.user_id or ""), + str(getattr(src, "profile", None) or ""), + bool(context.shared_multi_user_session), + discord_ids, + discord_tools, + slack_tools, + tuple(p.value for p in context.connected_platforms), + tuple( + ( + p.value, + str(getattr(hc, "name", "") or ""), + str(getattr(hc, "chat_id", "") or ""), + ) + for p, hc in context.home_channels.items() + ), + bool(redact_pii), + home_display, + ) + return hashlib.sha256(repr(key_tuple).encode("utf-8")).hexdigest() + + def _evict_cached_agent(self, session_key: str) -> None: + """Remove a cached agent for a session (called on /new, /model, etc). + + Also soft-releases the evicted agent's LLM client pool (``release_clients()``): AIAgent + holds reference cycles that delay collection, so without it gateway RSS grows across /new. + Soft = frees clients and per-turn child subagents but PRESERVES the session's terminal + sandbox, browser daemon and bg processes (keyed on task_id) since the session may resume. + True boundaries (/new) call ``_cleanup_agent_resources`` first (release is idempotent). + Cleanup runs on a daemon thread so ``_agent_cache_lock`` never spans slow socket teardown. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + # Prompt-stability state rides the agent-cache lifecycle: a fresh agent must re-render its + # session-context bytes (the pin) and re-see the current voice-channel state once. + _evict_state = self._peek_session_state(session_key) + if _evict_state is not None: + _evict_state.conversation.ephemeral_pin = None + _evict_state.conversation.vc_last = None + + _lock = getattr(self, "_agent_cache_lock", None) + evicted = None + if _lock: + with _lock: + evicted = self._agent_cache.pop(session_key, None) + else: + _cache = getattr(self, "_agent_cache", None) + if _cache is not None: + evicted = _cache.pop(session_key, None) + + agent = evicted[0] if isinstance(evicted, tuple) and evicted else evicted + if agent is None or agent is _AGENT_PENDING_SENTINEL: + return + + # Don't tear down an agent that's actively mid-turn — its client, + # sandbox and child subagents are in use by the running request. + running_ids = self._running_agent_ids() + if id(agent) in running_ids: + return + + try: + threading.Thread( + target=self._release_evicted_agent_soft, + args=(agent,), + daemon=True, + name=f"agent-evict-{str(session_key)[:24]}", + ).start() + except Exception: + # If we can't spawn a thread (interpreter shutdown), release + # inline as a best-effort fallback. + with suppress(Exception): + self._release_evicted_agent_soft(agent) + + def _commit_memory_before_soft_evict(self, agent: Any, key: str) -> None: + """Fire on_session_end extraction before soft-evicting a live agent. + + Soft eviction keeps the session resumable and does NOT fire ``on_session_end`` — that is + ``_session_expiry_watcher``'s job at true expiry. But the watcher tears down whatever it + finds in ``_agent_cache``; if the LRU cap soft-evicts first, memory providers never see the + transcript. So commit extraction here via ``commit_memory_session`` (no teardown). Only for + finalizable sessions — ``mode == "none"`` never finalizes. Best-effort: failures swallowed. + """ + if agent is None or not hasattr(agent, "commit_memory_session"): + return + if getattr(agent, "_memory_manager", None) is None: + return # no external memory provider — nothing to commit + try: + _store = getattr(self, "session_store", None) + if _store is None: + return + _store._ensure_loaded() + entry = _store._entries.get(key) + if entry is None: + return + # Compensate only when the watcher would expect this agent at expiry (finite policy, not yet + # expired). Expired sessions are torn down by the watcher; mode="none" is never finalized. + if not _store.is_session_finalizable(entry): + return + if _store._is_session_expired(entry): + return + messages = getattr(agent, "_session_messages", None) + agent.commit_memory_session(messages if isinstance(messages, list) else None) + logger.debug( + "Committed on_session_end extraction before soft-evicting " + "finalizable session=%s (cache pressure, pre-expiry)", key, + ) + except Exception as _e: + logger.debug("Pre-evict memory commit failed for %s: %s", key, _e) + + def _commit_then_release_soft(self, agent: Any, key: str) -> None: + """Commit end-of-session memory (if warranted), then soft-release. + + Runs on the daemon eviction thread so neither blocks the caller's held cache lock. Order + matters: commit needs the live memory manager before ``release_clients`` drops the buffer. + """ + self._commit_memory_before_soft_evict(agent, key) + self._release_evicted_agent_soft(agent) + + def _release_evicted_agent_soft(self, agent: Any) -> None: + """Soft cleanup for cache-evicted agents — preserves session tool state. + + Unlike _cleanup_agent_resources (full teardown), an evicted session may resume, so its + terminal sandbox, browser daemon and bg processes must outlive the AIAgent instance. + """ + if agent is None: + return + try: + if hasattr(agent, "release_clients"): + agent.release_clients() + else: + # Older agent instance (shouldn't happen in practice) — + # fall back to the legacy full-close path. + self._cleanup_agent_resources(agent) + except Exception: + pass + # Free conversation history — tens of MB of tool output on heavy 100+-tool-call sessions. + # release_clients() preserves session tool state for resume, but the message list is rebuilt from + # persisted session JSON on the next turn, so dropping it here is safe. + if hasattr(agent, "_session_messages"): + agent._session_messages = [] + # _db_flush_scan_prefix (run_agent.py, stamped on every successful flush) is a shallow copy + # sharing every message dict of the flushed transcript, so leaving it pins the multi-MB strings + # this eviction frees. Pressure-evictable agents have flushed by definition, so it's populated. + if hasattr(agent, "_db_flush_scan_prefix"): + agent._db_flush_scan_prefix = None + + def _agent_cache_bounds(self): + """Operator-configured agent-cache bounds, resolved once per process. + + Resolved lazily rather than in ``__init__`` so it also works for the + ``__new__``-constructed runners used by tests and by the slash-command mixin. + """ + from gateway.run import _load_gateway_config + bounds = getattr(self, "_agent_cache_bounds_cache", None) + if bounds is None: + from gateway.agent_cache_pressure import resolve_agent_cache_bounds + + try: + bounds = resolve_agent_cache_bounds(_load_gateway_config()) + except Exception as _e: + logger.debug("Agent cache bounds config read failed: %s", _e) + # Resolve from an empty config rather than bare AgentCacheBounds(): the dataclass default + # has memory_high_mb=None (pressure pass OFF) but an *absent* section means "auto" — a + # transient config read failure must not permanently disable the OOM valve. + bounds = resolve_agent_cache_bounds({}) + self._agent_cache_bounds_cache = bounds + return bounds + + def _agent_cache_cap(self) -> int: + """Effective LRU cap — the configured override, else the default.""" + from gateway.run import _AGENT_CACHE_MAX_SIZE + configured = self._agent_cache_bounds().max_size + return configured if configured else _AGENT_CACHE_MAX_SIZE + + def _agent_cache_idle_ttl(self) -> float: + """Effective idle TTL in seconds — configured override, else default.""" + from gateway.run import _AGENT_CACHE_IDLE_TTL_SECS + configured = self._agent_cache_bounds().idle_ttl_secs + return configured if configured else _AGENT_CACHE_IDLE_TTL_SECS + + def _sweep_agent_cache_under_pressure(self) -> int: + """Shed cached transcripts once the gateway heap nears its budget; returns count evicted. + + The LRU cap counts entries and the idle sweep counts seconds; neither knows one cached agent + pins a full ``_session_messages`` transcript (tens of MB). Warm and finalizable agents are + never swept, so RSS climbs until the cgroup throttles. Above the anonymous-RSS budget this + soft-evicts LRU agents (transcript rebuilt from the persisted session next turn). Never + touched: agents mid-turn, the most recently used sessions, and transcripts not yet on disk. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + from gateway.agent_cache_pressure import ( + plan_pressure_evictions, + read_anon_rss_mb, + transcript_persistence_caught_up, + ) + + bounds = self._agent_cache_bounds() + if not bounds.memory_high_mb: + return 0 + _cache = getattr(self, "_agent_cache", None) + _lock = getattr(self, "_agent_cache_lock", None) + if not _cache or _lock is None: + # Nothing cached — whatever is using the heap, it isn't us, and + # warning about it every tick would point at the wrong subsystem. + return 0 + + rss_mb = read_anon_rss_mb() + if rss_mb is None or rss_mb < bounds.memory_high_mb: + return 0 + + running_ids = self._running_agent_ids() + + def _is_evictable(key: str, agent: Any) -> bool: + if agent is None or agent is _AGENT_PENDING_SENTINEL: + return False + if id(agent) in running_ids: + return False + return transcript_persistence_caught_up(agent) + + with _lock: + ordered = [ + (key, entry[0] if isinstance(entry, tuple) and entry else entry) + for key, entry in _cache.items() + ] + plan = plan_pressure_evictions( + ordered, + is_evictable=_is_evictable, + max_evictions=bounds.max_evictions_per_pass, + protect_recent=bounds.protect_recent, + ) + for key, _ in plan: + _cache.pop(key, None) + + if not plan: + _mid_turn = sum(1 for _, a in ordered if a is not None and id(a) in running_ids) + _unflushed = sum( + 1 + for _, a in ordered + if a is not None + and a is not _AGENT_PENDING_SENTINEL + and id(a) not in running_ids + and not transcript_persistence_caught_up(a) + ) + logger.warning( + "Agent cache pressure: anon RSS %dMB over budget %dMB but no " + "evictable session (%d cached, %d mid-turn, %d blocked on " + "un-flushed persistence)%s", + rss_mb, bounds.memory_high_mb, len(ordered), _mid_turn, _unflushed, + ( + " — transcripts are not reaching the session DB " + "(session persistence disabled or failing?); the memory " + "valve cannot shed sessions until they persist." + if _unflushed and not _mid_turn + else " — memory will keep climbing until those turns finish." + ), + ) + return 0 + + evicted_count = len(plan) + logger.warning( + "Agent cache pressure: anon RSS %dMB over budget %dMB — evicting " + "%d LRU session(s): %s", + rss_mb, bounds.memory_high_mb, evicted_count, + ", ".join(key for key, _ in plan), + ) + try: + threading.Thread( + target=self._release_pressure_batch, + args=(plan,), + daemon=True, + name="agent-cache-pressure", + ).start() + except Exception: + self._release_pressure_batch(plan) + # NOTE: _release_pressure_batch drains `plan` in place (so the trim runs with no lingering + # agent refs) — len(plan) is 0 once the daemon thread finishes, hence the pre-captured count. + return evicted_count + + def _release_pressure_batch(self, plan: List[tuple]) -> None: + """Release a pressure-evicted batch, then return the heap to the OS. + + Sequential on one daemon thread (the batch is capped; the goal is reclaiming memory, not + racing teardowns). The trailing ``malloc_trim`` makes RSS actually fall — glibc otherwise + keeps freed arenas. The plan is drained (``pop`` + ``del``), not iterated, so no local + reference pins evicted agents during ``gc.collect`` + trim (else the valve over-evicts). + """ + while plan: + key, agent = plan.pop(0) # FIFO — evict LRU-first order preserved + try: + self._commit_then_release_soft(agent, key) + except Exception as _e: + logger.debug("Pressure release failed for %s: %s", key, _e) + del agent + try: + from hermes_cli.mem_trim import trim_memory + + trim_memory(force=True, reason="agent_cache_pressure") + except Exception: + pass + + def _enforce_agent_cache_cap(self) -> None: + """Evict oldest cached agents when cache exceeds the LRU cap. Requires _agent_cache_lock. + + Resource cleanup runs on a daemon thread so the lock is not held over slow teardown. + Agents in _running_agents are SKIPPED (their clients/sandboxes/subagents are in use); if + every LRU candidate is active the cache stays over cap until the next insert. + """ + _cache = getattr(self, "_agent_cache", None) + if _cache is None: + return + # OrderedDict.popitem(last=False) pops oldest; plain dict lacks the + # arg so skip enforcement if a test fixture swapped the cache type. + if not hasattr(_cache, "move_to_end"): + return + + # Snapshot of agent instances mid-turn, keyed by id() so lookup is O(1) and independent of + # AIAgent.__eq__ (which MagicMock overrides in tests). + running_ids = self._running_agent_ids() + + # Walk LRU → MRU; only the first (size - cap) LRU positions are candidates. An active slot is + # SKIPPED rather than evicting a newer entry — that would penalise a fresh session (no cache + # history) to protect a long-running one. Cache may stay over cap until the next insert. + cap = self._agent_cache_cap() + excess = max(0, len(_cache) - cap) + evict_plan: List[tuple] = [] # [(key, agent), ...] + if excess > 0: + ordered_keys = list(_cache.keys()) + for key in ordered_keys[:excess]: + entry = _cache.get(key) + agent = entry[0] if isinstance(entry, tuple) and entry else None + if agent is not None and id(agent) in running_ids: + continue # active mid-turn; don't evict, don't substitute + evict_plan.append((key, agent)) + + for key, _ in evict_plan: + _cache.pop(key, None) + + remaining_over_cap = len(_cache) - cap + if remaining_over_cap > 0: + logger.warning( + "Agent cache over cap (%d > %d); %d excess slot(s) held by " + "mid-turn agents — will re-check on next insert.", + len(_cache), cap, remaining_over_cap, + ) + + for key, agent in evict_plan: + logger.info( + "Agent cache at cap; evicting LRU session=%s (cache_size=%d)", + key, len(_cache), + ) + if agent is not None: + # Commit end-of-session memory, then soft-release, both on the daemon thread so the + # (possibly network-bound) provider call never blocks the held cache lock. + threading.Thread( + target=self._commit_then_release_soft, + args=(agent, key), + daemon=True, + name=f"agent-cache-evict-{key[:24]}", + ).start() + + def _sweep_idle_cached_agents(self) -> int: + """Evict cached agents idle past the idle TTL; returns the number evicted. + + Acquires the cache lock internally (safe from the expiry watcher); cleanup on daemon + threads. Agents in _running_agents are SKIPPED — tearing down an active turn crashes it. + """ + _cache = getattr(self, "_agent_cache", None) + _lock = getattr(self, "_agent_cache_lock", None) + if _cache is None or _lock is None: + return 0 + now = time.time() + idle_ttl = self._agent_cache_idle_ttl() + to_evict: List[tuple] = [] + running_ids = self._running_agent_ids() + with _lock: + for key, entry in list(_cache.items()): + agent = entry[0] if isinstance(entry, tuple) and entry else None + if agent is None: + continue + if id(agent) in running_ids: + continue # mid-turn — don't tear it down + last_activity = getattr(agent, "_last_activity_ts", None) + if last_activity is None: + continue + if (now - last_activity) > idle_ttl: + # If the session hasn't actually expired in the store (e.g. daily-reset fires hours + # after the last message), keep the agent cached so the expiry watcher can still find + # it and call on_session_end() with the live transcript. BUT only defer when the + # watcher will EVER finalize it: for mode == "none" (is_session_finalizable() False) + # deferring pins the agent for the gateway's lifetime — the leak this sweep relieves. + # Those fall through to soft eviction WITHOUT on_session_end, correctly (never a + # session-end boundary). Finite sessions evicted under LRU-cap pressure are covered + # by _commit_memory_before_soft_evict on the cap path. + session_entry = None + _store = getattr(self, "session_store", None) + try: + if _store is not None: + _store._ensure_loaded() + session_entry = _store._entries.get(key) + except Exception: + session_entry = None + if ( + session_entry is not None + and _store is not None + and _store.is_session_finalizable(session_entry) + and not _store._is_session_expired(session_entry) + ): + continue # keep agent — finite session hasn't expired + to_evict.append((key, agent)) + for key, _ in to_evict: + _cache.pop(key, None) + for key, agent in to_evict: + logger.info( + "Agent cache idle-TTL evict: session=%s (idle=%.0fs)", + key, now - getattr(agent, "_last_activity_ts", now), + ) + threading.Thread( + target=self._release_evicted_agent_soft, + args=(agent,), + daemon=True, + name=f"agent-cache-idle-{key[:24]}", + ).start() + return len(to_evict) diff --git a/gateway/run_busy.py b/gateway/run_busy.py new file mode 100644 index 0000000000..53946c7587 --- /dev/null +++ b/gateway/run_busy.py @@ -0,0 +1,1523 @@ +"""Busy-session queueing, slot claims, slash dispatch tables and destructive-slash confirmation for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import json +import os +import time +from agent.i18n import t +from gateway.config import Platform +from gateway.platforms.base import EphemeralReply, MessageEvent, MessageType +from gateway.session import SessionSource +from typing import Any, Dict, Optional, Union + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayBusySessionMixin: + """Busy-session queueing, slot claims, slash dispatch tables and destructive-slash confirmation for GatewayRunner.""" + + def _queue_during_drain_enabled( + self, busy_input_mode: Optional[str] = None + ) -> bool: + # "queue" and "steer" both mean messages must not be lost across restart: queue them for + # the newly-spawned gateway process to pick up. "interrupt" mode drops them. + mode = busy_input_mode or self._busy_input_mode + return self._restart_requested and mode in {"queue", "steer"} + + def _enqueue_fifo(self, session_key: str, queued_event: "MessageEvent", adapter: Any) -> None: + """Append a /queue event to the FIFO chain for a session.""" + if adapter is None: + return + pending_slot = getattr(adapter, "_pending_messages", None) + if pending_slot is None: + return + if session_key in pending_slot: + self._session_state(session_key).conversation.queued_events.append( + queued_event + ) + else: + pending_slot[session_key] = queued_event + + def _promote_queued_event( + self, + session_key: str, + adapter: Any, + pending_event: Optional["MessageEvent"], + ) -> Optional["MessageEvent"]: + """Promote the next overflow item after the slot was drained. + + If pending_event is None, return the overflow head as the new pending_event; if the slot is + already populated (interrupt follow-up etc.), stage the head there for the NEXT recursion. + Returns the (possibly updated) pending_event. + """ + _q_state = self._peek_session_state(session_key) + overflow = _q_state.conversation.queued_events if _q_state else None + if not overflow: + return pending_event + next_queued = overflow.pop(0) + if pending_event is None: + return next_queued + if adapter is not None and hasattr(adapter, "_pending_messages"): + adapter._pending_messages[session_key] = next_queued + else: + # No adapter — push back so we don't silently drop the item. + overflow.insert(0, next_queued) + return pending_event + + def _queue_depth(self, session_key: str, *, adapter: Any = None) -> int: + """Total pending /queue items for a session — slot + overflow.""" + _q_state = self._peek_session_state(session_key) + depth = len(_q_state.conversation.queued_events) if _q_state else 0 + if adapter is not None and session_key in getattr(adapter, "_pending_messages", {}): + depth += 1 + return depth + + def _rescue_orphaned_overflow( + self, session_key: str, adapter: Any + ) -> Optional["MessageEvent"]: + """Pop the oldest orphaned FIFO overflow event for an idle session. + + ``queued_events`` drains only at the post-turn promotion site in ``_run_agent``; if a busy + window ends without that drain (early recursion exit, exception/interrupt/generation-bump), + the overflow is silently orphaned. Called when a NEW event arrives for a NON-busy session: + the oldest orphan is returned to run as THIS turn, the next is staged into the slot so the + chain continues in arrival order, and the caller enqueues the incoming event behind it. The + returned event is REMOVED from both stores, else the post-turn dequeue would run it twice. + Returns ``None`` when there is nothing to rescue (no overflow, slot occupied, or no slot). + """ + try: + _q_state = self._peek_session_state(session_key) + overflow = _q_state.conversation.queued_events if _q_state else None + if not overflow: + return None + pending_slot = getattr(adapter, "_pending_messages", None) + if not isinstance(pending_slot, dict) or pending_slot.get(session_key): + # Slot occupied (busy) or no slot storage — promotion owns + # this; do not fight it from the idle path. + return None + head = overflow.pop(0) + # Keep the slot occupied for the rest of the chain so the drain promotes in order and + # any mid-chain arrival routes to overflow instead of jumping the queue (same invariant + # as the drain's own _promote_queued_event). Only ONE event fits the slot. + if overflow: + pending_slot[session_key] = overflow.pop(0) + logger.warning( + "Rescued orphaned FIFO overflow event for idle session " + "%s — it was queued during a busy window but the post-turn " + "drain never promoted it (#99882)", + session_key, + ) + if overflow: + logger.warning( + "%d overflow event(s) still queued for session %s after " + "rescue staging (will drain via normal promotion)", + len(overflow), + session_key, + ) + return head + except Exception: + logger.debug("FIFO overflow rescue failed for %s", session_key, exc_info=True) + return None + + @staticmethod + def _is_goal_continuation_event(event_or_text: Any) -> bool: + """Return True for synthetic /goal continuation turns. + + Goal continuations are normal queued user-role events, so pause/clear must distinguish + them from real user /queue messages before removing or suppressing them. + """ + text = getattr(event_or_text, "text", event_or_text) or "" + return str(text).startswith("[Continuing toward your standing goal]\nGoal:") + + def _clear_goal_pending_continuations(self, session_key: str, adapter: Any) -> int: + """Remove queued synthetic /goal continuations for one session. + + User /goal pause/clear can race a judge-queued continuation; only synthetic goal + continuations are removed, normal /queue and user follow-up events are preserved. + """ + removed = 0 + pending_slot = getattr(adapter, "_pending_messages", None) if adapter is not None else None + if isinstance(pending_slot, dict): + pending_event = pending_slot.get(session_key) + if self._is_goal_continuation_event(pending_event): + pending_slot.pop(session_key, None) + removed += 1 + + _q_state = self._peek_session_state(session_key) + overflow = _q_state.conversation.queued_events if _q_state else [] + if overflow: + kept = [] + for queued_event in overflow: + if self._is_goal_continuation_event(queued_event): + removed += 1 + else: + kept.append(queued_event) + _q_state.conversation.queued_events = kept + return removed + + def _goal_still_active_for_session(self, session_id: str) -> bool: + """Best-effort fresh DB check before running a queued continuation.""" + if not session_id: + return False + try: + from hermes_cli.goals import GoalManager + return GoalManager(session_id=session_id).is_active() + except Exception as exc: + logger.debug("goal continuation: active-state recheck failed: %s", exc) + return False + + def _get_max_concurrent_sessions(self) -> Optional[int]: + """Return the configured active chat session cap, if enabled.""" + try: + from hermes_cli.active_sessions import resolve_max_concurrent_sessions + + return resolve_max_concurrent_sessions(getattr(self, "config", None)) + except Exception: + return None + + def _active_session_limit_message(self, session_key: str) -> Optional[str]: + """Return a user-facing rejection when starting a new session exceeds the cap.""" + max_sessions = self._get_max_concurrent_sessions() + if max_sessions is None: + return None + if self._is_session_running(session_key): + return None + active_count = self._running_agent_count() + if active_count < max_sessions: + return None + from hermes_cli.active_sessions import active_session_limit_message + + return active_session_limit_message(active_count, max_sessions) + + def _claim_active_session_slot( + self, + session_key: str, + source: SessionSource, + ) -> tuple[Any, Optional[str]]: + """Claim a cross-process active-session slot for a new gateway turn.""" + if self._is_session_running(session_key): + return None, None + local_limit_message = self._active_session_limit_message(session_key) + if local_limit_message is not None: + return None, local_limit_message + try: + from hermes_cli.active_sessions import try_acquire_active_session + + platform = source.platform.value if source and source.platform else "gateway" + return try_acquire_active_session( + session_id=session_key, + surface=f"gateway:{platform}", + config=getattr(self, "config", None), + metadata={ + "platform": platform, + "chat_id": getattr(source, "chat_id", "") or "", + "user_id": getattr(source, "user_id", "") or "", + # Writer identity for re-entrancy: if this process leaks a lease for this session + # (exception path skipped release), the next turn re-acquires its own entry rather + # than being fenced out forever — pruning only reclaims entries whose PROCESS died. + "live_session_id": str(session_key), + }, + ) + except Exception as exc: + logger.warning("Failed to claim active session slot: %s", exc) + return None, None + + @staticmethod + def _agent_has_active_subagents(running_agent: Any) -> bool: + """Return True when *running_agent* is driving subagents via ``delegate_task``. + + ``AIAgent.interrupt()`` cascades through ``_active_children`` and aborts in-flight subagent + work, so callers demote ``busy_input_mode='interrupt'`` to ``queue`` while this is True; + explicit ``/stop`` is untouched. Fail-safe: returns False on any attribute/lock error. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + if running_agent is None or running_agent is _AGENT_PENDING_SENTINEL: + return False + children = getattr(running_agent, "_active_children", None) + # AIAgent always initialises this as a concrete list. Reject anything that isn't a real + # collection — guards against ``MagicMock()._active_children`` auto-creating a truthy stub + # in tests and triggering the demotion for an agent with no subagents. + if not isinstance(children, (list, tuple, set)): + return False + if not children: + return False + lock = getattr(running_agent, "_active_children_lock", None) + try: + if lock is not None: + with lock: + return bool(children) + return bool(children) + except Exception: + return False + + async def _session_has_compression_in_flight(self, session_key: str) -> bool: + """Return True when a compression lock is held for this session's id. + + Gateway ``interrupt`` busy mode could start a follow-up against the pre-rotation parent while + compression is mid-flight, producing orphaned compression siblings; callers demote interrupt + to queue when True. Both blocking sources (``session_store`` lock + JSON load, SQLite lock + holder SELECT) run in a worker thread so a large state.db never freezes the event loop. + """ + session_store = getattr(self, "session_store", None) + if not session_key or session_store is None: + return False + try: + session_id = await asyncio.to_thread( + self._lookup_session_id_under_store_lock, session_store, session_key + ) + except (AttributeError, TypeError): + return False + except Exception: + logger.warning( + "Compression in-flight check failed while reading session %s; " + "treating compression as active to avoid interrupting a possible " + "parent-session rotation", + session_key, + exc_info=True, + ) + return True + if not session_id: + return False + session_db = getattr(self, "_session_db", None) + if session_db is None: + return False + raw_db = getattr(session_db, "_db", session_db) + try: + holder = await asyncio.to_thread( + raw_db.get_compression_lock_holder, str(session_id) + ) + # Production returns Optional[str]. Reject non-strings so a MagicMock auto-attr (or any + # unexpected truthy) cannot look like a held lock and skip hygiene. + return isinstance(holder, str) and bool(holder) + except (AttributeError, TypeError): + return False + except Exception: + logger.warning( + "Compression in-flight check failed while reading lock holder " + "for session %s; treating compression as active to avoid " + "interrupting a possible parent-session rotation", + session_id, + exc_info=True, + ) + return True + + @staticmethod + def _lookup_session_id_under_store_lock(session_store, session_key: str): + """Sync helper run in the thread pool: read session_id under the store lock.""" + # noqa: SLF001 — intentional private access; runs off the event loop. + with session_store._lock: # noqa: SLF001 + session_store._ensure_loaded_locked() # noqa: SLF001 + entry = session_store._entries.get(session_key) # noqa: SLF001 + return getattr(entry, "session_id", None) if entry is not None else None + + def _queue_or_replace_pending_event(self, session_key: str, event: MessageEvent) -> None: + from gateway.run import merge_pending_message_event + adapter = self._adapter_for_source(event.source) + if not adapter: + return + # Route through the ``/queue`` FIFO infrastructure so each follow-up gets its own turn in + # arrival order (merge_text=False silently OVERWROTE the single pending slot). Photo bursts + # still merge into the head slot (album semantics); everything else appends to the tail. + pending_slot = getattr(adapter, "_pending_messages", None) + existing = pending_slot.get(session_key) if isinstance(pending_slot, dict) else None + security_metadata_keys = ( + "hermes_plugin_id", + "hermes_plugin_injection", + "gateway_session_key", + "gateway_session_id", + "gateway_session_strict", + ) + same_security_context = existing is not None and ( + getattr(existing, "internal", False) == getattr(event, "internal", False) + and getattr(existing, "allow_gateway_control", True) + == getattr(event, "allow_gateway_control", True) + and all( + (getattr(existing, "metadata", None) or {}).get(key) + == (getattr(event, "metadata", None) or {}).get(key) + for key in security_metadata_keys + ) + ) + if same_security_context and ( + getattr(existing, "message_type", None) == MessageType.PHOTO + or event.message_type == MessageType.PHOTO + or bool(getattr(existing, "media_urls", None)) + or bool(getattr(event, "media_urls", None)) + ): + # Preserve photo-burst / media-merge semantics for the head slot. + merge_pending_message_event( + adapter._pending_messages, + session_key, + event, + merge_text=event.message_type == MessageType.TEXT, + ) + return + + if self._queue_depth(session_key, adapter=adapter) >= self._BUSY_QUEUE_MAX_PENDING: + logger.warning( + "Dropping busy-mode follow-up for session %s — pending queue at cap (%d).", + session_key, + self._BUSY_QUEUE_MAX_PENDING, + ) + return + + self._enqueue_fifo(session_key, event, adapter) + + async def _prepare_busy_steer_text(self, event: MessageEvent) -> str: + """Return steerable text for a busy follow-up, transcribing voice first. + + Successful steer messages bypass the inbound STT queue, so without this a media-only voice + follow-up has empty text and steer silently degrades to queue mode. Only voice-message media + (not audio file attachments) is transcribed; on failure keep any caption and let the steer + fallback handle it. Goes through ``_transcribe_and_echo_pending_voice`` — the single + out-of-band STT choke point — so STT runs at most once per message (cached on the event). + """ + text = (event.text or "").strip() + if not self._pending_event_audio_paths(event): + return text + + adapter = self._adapter_for_source(event.source) + enriched_text, successful_transcripts = await self._transcribe_and_echo_pending_voice( + event, + adapter, + event.source, + text, + log_context="Busy-steer", + ) + if not successful_transcripts: + return text + return (enriched_text or text).strip() + + @staticmethod + def _busy_reply_to(event: MessageEvent, reply_anchor): + # Telegram DM topics anchor on the thread; other Telegram threads send unanchored. + return ( + reply_anchor + if event.source.platform == Platform.TELEGRAM + and event.source.chat_type == "dm" + and event.source.thread_id + else (None if event.source.platform == Platform.TELEGRAM and event.source.thread_id else event.message_id) + ) + + async def _send_busy_drain_notice(self, event: MessageEvent, session_key: str, effective_mode: str) -> None: + """Busy path while the gateway is restarting/stopping: queue (if allowed) and tell the user.""" + adapter = self._adapter_for_source(event.source) + if not adapter: + return + + reply_anchor = self._reply_anchor_for_event(event) + thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) + if self._queue_during_drain_enabled(effective_mode): + self._queue_or_replace_pending_event(session_key, event) + message = f"⏳ Gateway {self._status_action_gerund()} — queued for the next turn after it comes back." + else: + message = f"⏳ Gateway is {self._status_action_gerund()} and is not accepting another turn right now." + + await adapter._send_with_retry( + chat_id=event.source.chat_id, + content=message, + reply_to=self._busy_reply_to(event, reply_anchor), + metadata=thread_meta, + ) + + async def _route_plaintext_approval_while_busy(self, event: MessageEvent, session_key: str) -> bool: + """Route a bare "yes"/"no" to the approval handlers while a dangerous-command approval blocks. + + Returns True when the message was consumed as an approval response. + """ + # Approval routing: while blocked on a dangerous-command approval, a bare "yes" must reach the + # approval handler, not be steered/queued/interrupted (else it queues behind a turn that can't + # start until the approval resolves -> auto-deny deadlock). Slash forms already bypass at the + # base-adapter guard. Gated on has_blocking_approval so a conversational "yes" never fires a + # command. Reuse the /approve and /deny handlers; the busy path does not auto-send their return. + try: + from tools.approval import has_blocking_approval + if event.allow_gateway_control and has_blocking_approval(session_key): + _raw_text = (event.text or "").strip().lower() + _approve_words = {"approve", "yes", "ok", "okay", "confirm", "y", "👍"} + _deny_words = {"deny", "no", "reject", "cancel", "n", "👎"} + _approval_handler = None + _normalized_args = "" + if _raw_text in _approve_words: + _approval_handler = self._handle_approve_command + elif _raw_text in _deny_words: + _approval_handler = self._handle_deny_command + elif _raw_text in {"always", "approve always", "always approve"}: + _approval_handler = self._handle_approve_command + _normalized_args = "always" + elif _raw_text in {"session", "approve session", "session approve"}: + _approval_handler = self._handle_approve_command + _normalized_args = "session" + if _approval_handler is not None: + # Synthesize "/approve [args]" / "/deny" so the slash handlers parse modifiers via + # event.get_command_args(). Always a literal "/": is_command()/get_command_args() + # don't recognize per-platform display prefixes ("!" on Slack/Matrix). + _verb = "approve" if _approval_handler is self._handle_approve_command else "deny" + _synth = f"/{_verb}" + if _normalized_args: + _synth = f"{_synth} {_normalized_args}" + event.text = _synth + _reply = await _approval_handler(event) + logger.info( + "Approval response via plain text: session=%s verb=%s args=%r", + session_key, _verb, _normalized_args, + ) + _adapter = self._adapter_for_source(event.source) + if _adapter and _reply: + _text, _eph_ttl = _adapter._unwrap_ephemeral(_reply) + if _text: + _anchor = self._reply_anchor_for_event(event) + await _adapter._send_with_retry( + chat_id=event.source.chat_id, + content=_text, + reply_to=_anchor, + metadata=self._thread_metadata_for_source(event.source, _anchor), + ) + return True + except Exception: + logger.warning( + "Plain-text approval routing failed for session %s; " + "falling through to busy handling", + session_key, exc_info=True, + ) + return False + + async def _resolve_busy_steer_or_redirect( + self, + event: MessageEvent, + session_key: str, + effective_mode: str, + running_agent: Any, + ) -> "GatewayRunner._BusySteerOutcome": + """Apply interrupt->queue demotions, then attempt steer (steer mode) or redirect (interrupt mode).""" + from gateway.run import _AGENT_PENDING_SENTINEL + # Steer mode injects mid-run via running_agent.steer(); fall back to queue (nothing lost) if the + # agent isn't running yet (sentinel), lacks steer(), or the payload is empty. interrupt() + # cascades to ``_active_children`` and aborts delegate_task work, so demote ``interrupt`` to + # ``queue`` while the parent drives subagents; explicit /stop and /new still force-cancel all. + demoted_for_subagents = ( + effective_mode == "interrupt" + and self._agent_has_active_subagents(running_agent) + ) + if demoted_for_subagents: + logger.info( + "Demoting busy_input_mode 'interrupt' to 'queue' for session %s " + "because the running agent has active subagents (#30170)", + session_key, + ) + effective_mode = "queue" + demoted_for_compression = ( + effective_mode == "interrupt" + and await self._session_has_compression_in_flight(session_key) + ) + if demoted_for_compression: + logger.info( + "Demoting busy_input_mode 'interrupt' to 'queue' for session %s " + "because context compression is in flight (#56391)", + session_key, + ) + effective_mode = "queue" + steered = False + redirected = False + if effective_mode == "steer": + steer_text = await self._prepare_busy_steer_text(event) + # Steerable: plain text, OR every attachment is STT-eligible voice media whose transcript + # was folded into steer_text — else a voice note in steer mode silently degrades to queue. + _steer_media_urls = getattr(event, "media_urls", None) or [] + _steer_all_voice = bool(_steer_media_urls) and ( + len(self._pending_event_audio_paths(event)) == len(_steer_media_urls) + ) + can_steer = ( + steer_text + and ( + ( + event.message_type == MessageType.TEXT + and not event.media_urls + and not event.media_types + ) + or _steer_all_voice + ) + and running_agent is not None + and running_agent is not _AGENT_PENDING_SENTINEL + and hasattr(running_agent, "steer") + ) + if can_steer: + try: + steered = bool(running_agent.steer(steer_text)) + except Exception as exc: + logger.warning("Gateway steer failed for session %s: %s", session_key, exc) + steered = False + if not steered: + # Fall back to queue (merge into pending messages, no interrupt) + effective_mode = "queue" + elif ( + effective_mode == "interrupt" + and event.message_type == MessageType.TEXT + and not event.media_urls + and not event.media_types + and running_agent is not None + and running_agent is not _AGENT_PENDING_SENTINEL + and getattr(running_agent, "_supports_active_turn_redirect", False) is True + and hasattr(running_agent, "redirect") + ): + try: + redirected = bool(running_agent.redirect((event.text or "").strip())) + except Exception as exc: + logger.warning("Gateway redirect failed for session %s: %s", session_key, exc) + redirected = False + return self._BusySteerOutcome( + effective_mode=effective_mode, + demoted_for_subagents=demoted_for_subagents, + demoted_for_compression=demoted_for_compression, + steered=steered, + redirected=redirected, + ) + + async def _interrupt_running_agent_for_busy_event(self, event: MessageEvent, adapter, running_agent) -> None: + """Interrupt mode: abort in-flight tool calls; the agent loop exits at its next check point.""" + from gateway.run import _build_media_placeholder + try: + _interrupt_text = event.text + _media_urls = getattr(event, "media_urls", None) or [] + if self._pending_event_audio_paths(event): + _interrupt_text, _ = await self._transcribe_and_echo_pending_voice( + event, + adapter, + event.source, + event.text or "", + log_context="Voice-busy-interrupt", + ) + elif not _interrupt_text and _media_urls: + _interrupt_text = _build_media_placeholder(event) + running_agent.interrupt(_interrupt_text) + except Exception: + pass # don't let interrupt failure block the ack + + def _busy_steer_ack_enabled(self, event: MessageEvent, session_key: str) -> bool: + # Steer mode already injected the text; some mobile chat setups want silent steering (like STT + # echo suppression) — keep the behavior, drop only the confirmation bubble. + from gateway.run import _load_gateway_config, _platform_config_key + from gateway.display_config import resolve_display_setting + platform_key = _platform_config_key(event.source.platform) + steer_ack_env = os.environ.get("HERMES_GATEWAY_BUSY_STEER_ACK_ENABLED") + if steer_ack_env is not None: + steer_ack_enabled = steer_ack_env.strip().lower() in {"1", "true", "yes", "on"} + else: + steer_ack_enabled = bool( + resolve_display_setting( + _load_gateway_config(), + platform_key, + "busy_steer_ack_enabled", + True, + ) + ) + if not steer_ack_enabled: + logger.debug("Busy steer ack suppressed for session %s", session_key) + return steer_ack_enabled + + def _compose_busy_ack_message( + self, + event: MessageEvent, + now: float, + _busy_state, + running_agent: Any, + *, + is_steer_mode: bool, + is_queue_mode: bool, + is_redirect_mode: bool, + demoted_for_subagents: bool, + demoted_for_compression: bool, + ) -> str: + from gateway.run import ( + _AGENT_PENDING_SENTINEL, + _hermes_home, + _load_gateway_config, + _platform_config_key, + ) + from gateway.display_config import resolve_display_setting + + # Mobile chat defaults keep the ack terse; iteration/tool detail stays in logs and can be opted + # in per platform via display.platforms..busy_ack_detail. + status_parts = [] + busy_ack_detail_enabled = bool( + resolve_display_setting( + _load_gateway_config(), + _platform_config_key(event.source.platform), + "busy_ack_detail", + True, + ) + ) + + if busy_ack_detail_enabled and running_agent and running_agent is not _AGENT_PENDING_SENTINEL: + try: + summary = running_agent.get_activity_summary() + iteration = summary.get("api_call_count", 0) + max_iter = summary.get("max_iterations", 0) + current_tool = summary.get("current_tool") + start_ts = _busy_state.turn.started_ts if _busy_state else 0 + if start_ts: + elapsed_min = int((now - start_ts) / 60) + if elapsed_min > 0: + status_parts.append(f"{elapsed_min} min elapsed") + if max_iter: + status_parts.append(f"iteration {iteration}/{max_iter}") + if current_tool: + status_parts.append(f"running: {current_tool}") + except Exception: + pass + + status_detail = f" ({', '.join(status_parts)})" if status_parts else "" + if is_steer_mode: + message = ( + f"⏩ Steered into current run{status_detail}. " + f"Your message arrives after the next tool call." + ) + elif is_redirect_mode: + message = ( + f"↪ Redirected current run{status_detail}. " + f"I'll adjust using your correction." + ) + elif is_queue_mode and demoted_for_subagents: + # Explain the demotion: the follow-up didn't kill the subagent; /stop is the escape hatch. + message = ( + f"⏳ Subagent working{status_detail} — your message is queued for " + f"when it finishes (use /stop to cancel everything)." + ) + elif is_queue_mode and demoted_for_compression: + message = ( + f"⏳ Compressing context{status_detail} — your message is queued for " + f"when it finishes (use /stop to cancel everything)." + ) + elif is_queue_mode: + message = ( + f"⏳ Queued for the next turn{status_detail}. " + f"I'll respond once the current task finishes." + ) + else: + message = ( + f"⚡ Interrupting current task{status_detail}. " + f"I'll respond to your message shortly." + ) + + # First-touch onboarding: one-time hint about the queue/interrupt knob; the flag is persisted to + # config.yaml so it never fires again on this install. + try: + from agent.onboarding import ( + BUSY_INPUT_FLAG, + busy_input_hint_gateway, + is_seen, + mark_seen, + ) + _user_cfg = _load_gateway_config() + if not is_seen(_user_cfg, BUSY_INPUT_FLAG): + if is_steer_mode: + _hint_mode = "steer" + elif is_queue_mode: + _hint_mode = "queue" + elif is_redirect_mode: + _hint_mode = "redirect" + else: + _hint_mode = "interrupt" + message = ( + f"{message}\n\n" + f"{busy_input_hint_gateway(_hint_mode)}" + ) + mark_seen(_hermes_home / "config.yaml", BUSY_INPUT_FLAG) + except Exception as _onb_err: + logger.debug("Failed to apply busy-input onboarding hint: %s", _onb_err) + return message + + async def _send_busy_ack_reply(self, event: MessageEvent, adapter, message: str) -> None: + reply_anchor = self._reply_anchor_for_event(event) + thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) + try: + await adapter._send_with_retry( + chat_id=event.source.chat_id, + content=message, + reply_to=self._busy_reply_to(event, reply_anchor), + metadata=thread_meta, + ) + except Exception as e: + logger.debug("Failed to send busy-ack: %s", e) + + async def _handle_active_session_busy_message(self, event: MessageEvent, session_key: str) -> bool: + # Authorization gate: the cold path (_handle_message) checks _is_user_authorized before + # creating a session; the busy path must enforce the same check, else unauthorized users in + # shared threads (Slack/Telegram/Discord) inject messages into a session they don't own. + from gateway.run import _AGENT_PENDING_SENTINEL + if not self._is_user_authorized(event.source): + logger.warning( + "Dropping message from unauthorized user in active session: " + "user=%s (%s), platform=%s, session=%s", + event.source.user_id, + event.source.user_name, + event.source.platform.value if event.source.platform else "unknown", + session_key, + ) + return True # handled (silently dropped); do not fall through + + effective_mode = self._effective_busy_input_mode(event.source) + + # --- Draining case (gateway restarting/stopping) --- + if self._draining: + await self._send_busy_drain_notice(event, session_key, effective_mode) + return True + + if await self._route_plaintext_approval_while_busy(event, session_key): + return True + + # Normal busy case (agent actively running a task) + adapter = self._adapter_for_source(event.source) + if not adapter: + return False # let default path handle it + + # Internal synthetic events (async-delegation / background-process completions) must never + # interrupt/steer: treated as user TEXT while busy, interrupt mode would abort the active turn; + # a completion surfaces as a NEW turn only when idle. Plugin events carry untrusted payload + # text, so queue them through the gateway FIFO (security metadata kept apart). + if getattr(event, "internal", False) and not event.allow_gateway_control: + self._queue_or_replace_pending_event(session_key, event) + return True + if getattr(event, "internal", False): + return False + + _busy_state = self._peek_session_state(session_key) + running_agent = _busy_state.turn.agent if _busy_state else None + + busy_text_mode = self._effective_busy_text_mode(event.source) + if ( + event.message_type == MessageType.TEXT + and busy_text_mode == "queue" + and effective_mode != "steer" + ): + return False + + _steer = await self._resolve_busy_steer_or_redirect(event, session_key, effective_mode, running_agent) + effective_mode = _steer.effective_mode + demoted_for_subagents = _steer.demoted_for_subagents + demoted_for_compression = _steer.demoted_for_compression + steered = _steer.steered + redirected = _steer.redirected + + # Queue as the next turn after the current run ends. Skip after a successful steer — the text + # is already in the run and must NOT replay. Use the FIFO helper, not raw + # merge_pending_message_event (merge_text=True newline-joins consecutive TEXT follow-ups into + # ONE turn); FIFO gives each text its own turn while keeping photo-burst / album merge for media. + if not steered and not redirected: + self._queue_or_replace_pending_event(session_key, event) + + is_queue_mode = effective_mode == "queue" + is_steer_mode = effective_mode == "steer" + is_redirect_mode = effective_mode == "interrupt" and redirected + + # Interrupt mode: abort in-flight tool calls; the agent loop exits at its next check point. + if ( + effective_mode == "interrupt" + and not redirected + and running_agent + and running_agent is not _AGENT_PENDING_SENTINEL + ): + await self._interrupt_running_agent_for_busy_event(event, adapter, running_agent) + + # Disabled ack: skip sending, still process input. Checked before debounce so we never stamp a + # "last ack" timestamp for an ack that was not delivered. + busy_ack_enabled = os.environ.get("HERMES_GATEWAY_BUSY_ACK_ENABLED", "true").lower() == "true" + if not busy_ack_enabled: + logger.debug("Busy ack suppressed for session %s", session_key) + return True # input still processed, just no ack sent + + # Debounce before the config-heavy display lookup: rapid follow-ups are still processed but + # shouldn't cost a config read just to learn no ack will be sent. + _BUSY_ACK_COOLDOWN = 30 + now = time.time() + last_ack = _busy_state.turn.busy_ack_ts if _busy_state else 0 + if now - last_ack < _BUSY_ACK_COOLDOWN: + return True # interrupt sent (if not queue), ack already delivered recently + + if is_steer_mode and not self._busy_steer_ack_enabled(event, session_key): + return True + + self._session_state(session_key).turn.busy_ack_ts = now + + message = self._compose_busy_ack_message( + event, + now, + _busy_state, + running_agent, + is_steer_mode=is_steer_mode, + is_queue_mode=is_queue_mode, + is_redirect_mode=is_redirect_mode, + demoted_for_subagents=demoted_for_subagents, + demoted_for_compression=demoted_for_compression, + ) + await self._send_busy_ack_reply(event, adapter, message) + return True + + def _gateway_plain_command_handlers(self): + """Return ordinary slash handlers shared by idle and busy dispatch.""" + return { + "status": self._handle_status_command, + "context": self._handle_context_command, + "restart": self._handle_restart_command, + "approve": self._handle_approve_command, + "deny": self._handle_deny_command, + "pause": self._handle_pause_command, + "agents": self._handle_agents_command, + "bg": self._handle_background_command, + "btw": self._handle_btw_command, + "kanban": self._handle_kanban_command, + "subgoal": self._handle_subgoal_command, + "heartbeat": self._handle_heartbeat_command, + "busy": self._handle_busy_command, + "yolo": self._handle_yolo_command, + "verbose": self._handle_verbose_command, + "footer": self._handle_footer_command, + "help": self._handle_help_command, + "commands": self._handle_commands_command, + "profile": self._handle_profile_command, + "update": self._handle_update_command, + "version": self._handle_version_command, + } + + async def _send_command_ack(self, source, text: str, label: str) -> None: + """Best-effort acknowledgment for a slash command that falls through to agent processing.""" + try: + adapter = self._adapter_for_source(source) + if adapter: + await adapter.send( + str(source.chat_id), text, metadata=self._thread_metadata_for_source(source) + ) + except Exception: + logger.debug("%s ack send failed", label, exc_info=True) + + def _gateway_idle_command_handlers(self): + """Slash handlers dispatched only when no agent is running for the session (idle path). + + Busy dispatch keeps its own explicit allowlist (``_dispatch_busy_slash_command``).""" + return { + "topic": self._handle_topic_command, + "whoami": self._handle_whoami_command, + "platform": self._handle_platform_command, + "stop": self._handle_stop_command, + "reasoning": self._handle_reasoning_command, + "memory": self._handle_memory_command, + "skills": self._handle_skills_command, + "fast": self._handle_fast_command, + "approvals": self._handle_approvals_command, + "model": self._handle_model_command, + "codex-runtime": self._handle_codex_runtime_command, + "personality": self._handle_personality_command, + "suggestions": self._handle_suggestions_command, + "save": self._handle_save_command, + "retry": self._handle_retry_command, + "sethome": self._handle_set_home_command, + "compress": self._handle_compress_command, + "usage": self._handle_usage_command, + "topup": self._handle_topup_command, + "insights": self._handle_insights_command, + "reload-mcp": self._handle_reload_mcp_command, + "reload-skills": self._handle_reload_skills_command, + "bundles": self._handle_bundles_command, + "debug": self._handle_debug_command, + "title": self._handle_title_command, + "resume": self._handle_resume_command, + "sessions": self._handle_sessions_command, + "branch": self._handle_branch_command, + "rollback": self._handle_rollback_command, + "diff": self._handle_diff_command, + "goal": self._handle_goal_command, + "loop": self._handle_loop_command, + "refine": self._handle_refine_command, + "review": self._handle_review_command, + "voice": self._handle_voice_command, + } + + async def _dispatch_busy_slash_command( + self, event: MessageEvent, cmd_def, quick_key: str, source, + ): + """Dispatch a recognized slash command while an agent is running. + + Order: ``busy_handler`` (special mid-run variant) → ``busy_policy == "dispatch"`` (normal + handler) → catch-all busy-reject text. Rejecting beats falling through to interrupt + + discard: Discord-registered slash commands would interrupt the agent AND be discarded by + the slash-command safety net, producing a zero-char response. + """ + name = cmd_def.name + policy = getattr(cmd_def, "busy_policy", "reject") + handler_key = getattr(cmd_def, "busy_handler", None) + + if handler_key: + special = { + "start": self._busy_start_command, + "stop": self._busy_stop_command, + "new": self._busy_new_command, + "queue": self._busy_queue_command, + "steer": self._busy_steer_command, + "egress": self._busy_egress_command, + "goal": self._busy_goal_command, + "loop": self._busy_loop_command, + }.get(handler_key) + if special is not None: + return await special(event, quick_key, source) + reject_text = self._BUSY_REJECT_TEXT.get(handler_key) + if reject_text is not None: + return reject_text + + if policy in ("dispatch", "interrupt_then_dispatch"): + plain = self._gateway_plain_command_handlers().get(name) + if plain is not None: + return await plain(event) + logger.warning( + "busy_policy=%s for /%s has no mid-run handler — " + "falling back to busy-reject", policy, name, + ) + + # Catch-all: any other recognized slash command hit the running-agent guard — reject + # gracefully rather than falling through to interrupt + discard. + return ( + f"⏳ Agent is running — `/{name}` can't run " + f"mid-turn. Wait for the current response or `/stop` first." + ) + + async def _handle_pause_command(self, event: MessageEvent): + """`/pause [reason]` engages the global emergency stop; `/pause off` (resume/stop) lifts it. + + In-band resume path for messaging-only operators — the estop gate lets recognized slash + commands through while paused so a user without host-shell access is never locked out. + """ + from agent import estop + + args = (event.get_command_args() or "").strip() + if args.lower() in {"off", "resume", "stop", "disengage"}: + if estop.disengage(): + return "▶️ Resumed — new work is accepted again." + return "Hermes wasn't paused." + state = estop.get_state() + if state is not None and not args: + reason = state.get("reason") + suffix = f" (reason: {reason})" if reason else "" + return ( + f"⏸️ Hermes is already paused{suffix}. " + "Use `/pause off` to resume." + ) + estop.engage(reason=args or None) + suffix = f" (reason: {args})" if args else "" + return ( + f"⏸️ Paused{suffix}. New cron/kanban/gateway work is on hold; " + "in-flight work finishes normally. Use `/pause off` to resume." + ) + + async def _busy_start_command(self, event: MessageEvent, quick_key: str, source): + # Telegram sends /start for bot launches/deep-links — a platform ping, not a user command: + # no help dump, no agent interrupt, no queued text. + logger.info("Ignoring /start platform ping for active session %s", quick_key) + return "" + + async def _busy_egress_command(self, event: MessageEvent, quick_key: str, source): + from hermes_cli.proxy_cli import format_status_text + + return format_status_text() + + async def _busy_stop_command(self, event: MessageEvent, quick_key: str, source): + # /stop must hard-kill the session when an agent is running. A soft interrupt + # (agent.interrupt()) doesn't help when the agent is truly hung — the executor thread is + # blocked and never checks _interrupt_requested. + from gateway.run import _INTERRUPT_REASON_STOP + await self._interrupt_and_clear_session( + quick_key, + source, + interrupt_reason=_INTERRUPT_REASON_STOP, + invalidation_reason="stop_command", + ) + logger.info("STOP for session %s — agent interrupted, session lock released", quick_key) + return EphemeralReply(t("gateway.stop.stopped")) + + async def _busy_new_command(self, event: MessageEvent, quick_key: str, source): + # /reset and /new must bypass the running-agent guard so they actually dispatch as commands + # instead of being queued as user text (which would be fed back to the agent with the same + # broken history — #2170). Clear any pending messages so the old text doesn't replay + from gateway.run import _INTERRUPT_REASON_RESET + await self._interrupt_and_clear_session( + quick_key, + source, + interrupt_reason=_INTERRUPT_REASON_RESET, + invalidation_reason="new_command", + ) + # Clean up the running agent entry so the reset handler + # doesn't think an agent is still active. + return await self._handle_reset_command(event) + + async def _busy_queue_command(self, event: MessageEvent, quick_key: str, source): + # /queue — queue without interrupting. Each /queue is its own full agent turn, run + # FIFO after the current run (and earlier /queue items) finish; messages are NOT merged. + queued_text = event.get_command_args().strip() + # Preserve media/reply payloads: a /queue carrying a photo, document, or reply context is + # valid even with no prompt text (e.g. "/queue" as the caption of an image). Dropping these + # fields silently lost the attachment when the queued turn ran. + has_media = bool(getattr(event, "media_urls", None)) + if not queued_text and not has_media: + return "Usage: /queue " + adapter = self._adapter_for_source(source) + if adapter: + queued_event = MessageEvent( + text=queued_text, + message_type=event.message_type if has_media else MessageType.TEXT, + source=event.source, + raw_message=event.raw_message, + message_id=event.message_id, + media_urls=list(getattr(event, "media_urls", []) or []), + media_types=list(getattr(event, "media_types", []) or []), + media_text_inlined=list(getattr(event, "media_text_inlined", []) or []), + reply_to_message_id=event.reply_to_message_id, + reply_to_text=event.reply_to_text, + reply_to_author_id=event.reply_to_author_id, + reply_to_author_name=event.reply_to_author_name, + reply_to_is_own_message=event.reply_to_is_own_message, + auto_skill=event.auto_skill, + channel_prompt=event.channel_prompt, + channel_context=event.channel_context, + internal=event.internal, + timestamp=event.timestamp, + ) + self._enqueue_fifo(quick_key, queued_event, adapter) + depth = self._queue_depth(quick_key, adapter=self._adapter_for_source(source)) + if depth <= 1: + return "Queued for the next turn." + return f"Queued for the next turn. ({depth} queued)" + + async def _busy_steer_command(self, event: MessageEvent, quick_key: str, source): + # /steer — inject mid-run after the next tool call. Unlike /queue (turn boundary), + # /steer lands BETWEEN tool-call iterations inside the same agent run, by appending to the + # last tool result's content. No interrupt, no new user turn, no role-alternation violation. + from gateway.run import _AGENT_PENDING_SENTINEL + steer_text = event.get_command_args().strip() + if not steer_text: + return "Usage: /steer " + _steer_state = self._peek_session_state(quick_key) + running_agent = _steer_state.turn.agent if _steer_state else None + if running_agent is _AGENT_PENDING_SENTINEL: + # Agent hasn't started yet — queue as turn-boundary fallback. + adapter = self._adapter_for_source(source) + if adapter: + queued_event = MessageEvent( + text=steer_text, + message_type=MessageType.TEXT, + source=event.source, + message_id=event.message_id, + channel_prompt=event.channel_prompt, + channel_context=event.channel_context, + ) + self._enqueue_fifo(quick_key, queued_event, adapter) + return "Agent still starting — /steer queued for the next turn." + if running_agent and hasattr(running_agent, "steer"): + try: + accepted = running_agent.steer(steer_text) + except Exception as exc: + logger.warning("Steer failed for session %s: %s", quick_key, exc) + return f"⚠️ Steer failed: {exc}" + if accepted: + preview = steer_text[:60] + ("..." if len(steer_text) > 60 else "") + return f"⏩ Steer queued — arrives after the next tool call: '{preview}'" + return "Steer rejected (empty payload)." + # Running agent is missing or lacks steer() — fall back to queue. + adapter = self._adapter_for_source(source) + if adapter: + queued_event = MessageEvent( + text=steer_text, + message_type=MessageType.TEXT, + source=event.source, + message_id=event.message_id, + channel_prompt=event.channel_prompt, + channel_context=event.channel_context, + ) + self._enqueue_fifo(quick_key, queued_event, adapter) + return "No active agent — /steer queued for the next turn." + + async def _busy_goal_command(self, event: MessageEvent, quick_key: str, source): + # /goal is safe mid-run for status/pause/clear/wait (inspection and control-plane only — + # doesn't interrupt the running turn). Setting new goal text mid-run is rejected like + # /model so we don't race a second continuation prompt against the current turn. + _goal_arg = (event.get_command_args() or "").strip().lower() + _goal_verb = _goal_arg.split(None, 1)[0] if _goal_arg else "" + # Exact-match control verbs, plus the wait/unwait barrier verbs (take a pid) and the gate + # management verb (gates run at turn boundary, so editing the gate list mid-run is safe). + _is_control = ( + not _goal_arg + or _goal_arg in {"status", "pause", "resume", "clear", "stop", "done", "unwait"} + or _goal_verb in {"wait", "gate"} + ) + if _is_control: + return await self._handle_goal_command(event) + return "Agent is running — use /goal status / pause / clear / wait mid-run, or /stop before setting a new goal." + + async def _busy_loop_command(self, event: MessageEvent, quick_key: str, source): + # /loop mirrors /goal: control verbs are safe mid-run (state only — read at the next idle + # boundary); setting a new loop mid-run is rejected so we don't race the current turn. + _loop_arg = (event.get_command_args() or "").strip().lower() + if not _loop_arg or _loop_arg in {"status", "pause", "resume", "stop", "clear", "cancel", "help", "--help", "-h"}: + return await self._handle_loop_command(event) + return "Agent is running — use /loop status / pause / stop mid-run, or /stop before setting a new loop." + + def _check_slash_access( + self, source: SessionSource, canonical_cmd: str + ) -> Optional[str]: + """Return a denial message if ``source`` cannot run ``canonical_cmd``, else None. + + Used by the cold and running-agent dispatch paths in ``_handle_message`` so admin/user gating + can't be bypassed by an in-flight agent. Backward-compat: without ``allow_admin_from`` for the + scope, ``policy_for_source`` returns ``enabled=False`` and this always returns None. + """ + from gateway.slash_access import policy_for_source as _policy_for_source + + if not canonical_cmd: + return None + policy = _policy_for_source(self.config, source) + if not policy.enabled or policy.can_run(source.user_id, canonical_cmd): + return None + logger.info( + "Slash command /%s denied for %s:%s (not admin, not in user_allowed_commands)", + canonical_cmd, + source.platform.value if source.platform else "?", + source.user_id, + ) + allowed_preview = sorted(policy.user_allowed_commands) + if allowed_preview: + suffix = ( + "You can run: " + + ", ".join(f"/{c}" for c in allowed_preview[:12]) + + ("…" if len(allowed_preview) > 12 else "") + + ". Use /whoami for the full list." + ) + else: + suffix = ( + "No slash commands are enabled for non-admins on this " + "platform. Ask an admin to add you to allow_admin_from " + "or to set user_allowed_commands." + ) + return f"⛔ /{canonical_cmd} is admin-only here. {suffix}" + + def _sibling_thread_run_keys(self, source: SessionSource, own_key: str) -> list: + """Find running-agent keys for OTHER participants in the same thread. + + In per-user thread mode each participant gets an isolated key + (``...:{thread_id}:{user_id}``), so another user's run is invisible to the caller's own + ``/stop``. Returns keys of *actually running* agents (not the pending sentinel, not the + caller's own) sharing the caller's ``{chat_id}:{thread_id}`` prefix; empty when not in a + thread or no sibling runs exist. Callers must still gate on authorization. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + thread_id = getattr(source, "thread_id", None) + chat_id = getattr(source, "chat_id", None) + if not thread_id or not chat_id: + return [] + platform = source.platform.value + chat_type = getattr(source, "chat_type", None) or "" + # Prefix that every per-user key in this thread shares, up to and including the thread_id + # segment. Match the exact key or prefix + ":" (a further user_id segment) so an unrelated + # thread whose id merely starts with this one is not matched. + prefix = ":".join( + ["agent:main", platform, chat_type, str(chat_id), str(thread_id)] + ) + matches = [] + for key, agent in self._running_agent_items(): + if key == own_key: + continue + if agent is _AGENT_PENDING_SENTINEL or not agent: + continue + if key == prefix or key.startswith(prefix + ":"): + matches.append(key) + return matches + + def _is_stale_restart_redelivery(self, event: MessageEvent) -> bool: + """Return True if this /restart is a Telegram re-delivery we already handled. + + The previous gateway wrote ``.restart_last_processed.json`` with the triggering platform + + update_id when it processed the /restart. A /restart on the same platform with + update_id <= that value is a redelivery when this process booted from that restart; + otherwise the marker must still be recent (< 5 minutes). Telegram only (the only platform + with a numeric cross-session update ordering); other platforms return False. + """ + from gateway.run import _hermes_home + if event is None or event.source is None: + return False + if event.platform_update_id is None: + return False + if event.source.platform is None: + return False + # Only Telegram populates platform_update_id currently; be explicit + # so future platforms aren't accidentally gated by this check. + try: + platform_value = event.source.platform.value + except Exception: + return False + if platform_value != "telegram": + return False + + try: + marker_path = _hermes_home / ".restart_last_processed.json" + if not marker_path.exists(): + # Belt-and-suspenders for a missing dedup marker (cleaned up, or the previous write + # failed): without it the update_id comparison can't run and a redelivered /restart + # would re-restart the gateway forever. Suppress ONLY when a restart cycle is + # independently confirmed: this process booted from a chat-originated /restart + # (_booted_from_restart) AND is within a short post-boot window; a genuine first + # /restart on a fresh boot is never swallowed (flag stays False). Consume the flag + # one-shot so a later legitimate /restart in the same session is honored. + if ( + getattr(self, "_booted_from_restart", False) + and time.time() - getattr(self, "_startup_time", 0.0) < 60 + ): + self._booted_from_restart = False + return True + return False + data = json.loads(marker_path.read_text(encoding="utf-8")) + except Exception: + return False + + if data.get("platform") != platform_value: + return False + recorded_uid = data.get("update_id") + if not isinstance(recorded_uid, int): + return False + if event.platform_update_id > recorded_uid: + return False + + # A service-managed restart can legitimately take longer than the marker's normal five- + # minute trust window while adapters, cron, and in-flight deliveries drain. Consume the boot + # signal one-shot so a later genuine command is evaluated normally. + if getattr(self, "_booted_from_restart", False): + self._booted_from_restart = False + return True + + # Staleness guard: ignore markers older than 5 minutes so a legitimately old one (e.g. crash + # recovery where notify never fired) doesn't swallow a fresh /restart. + requested_at = data.get("requested_at") + if isinstance(requested_at, (int, float)): + if time.time() - requested_at > 300: + return False + return True + + async def _handle_suggestions_command(self, event: MessageEvent) -> str: + """Handle /suggestions in the gateway. + + Delegates to the shared handler so CLI and gateway never drift. The origin is built from + the event source so an accepted suggestion's job delivers back to this chat/thread. + """ + from gateway.run import _command_origin_for_source + args = (event.get_command_args() or "").strip() + origin = _command_origin_for_source(event.source) + try: + from hermes_cli.suggestions_cmd import handle_suggestions_command + + return handle_suggestions_command(args, origin=origin, surface="gateway") + except Exception as e: + logger.debug("suggestions command failed: %s", e) + return f"Suggestions command failed: {e}" + + async def _handle_blueprint_command(self, event: MessageEvent): + """Handle /blueprint in the gateway. + + Delegates to the shared handler so CLI, TUI, and gateway never drift. Origin is built + from the event source so a directly created blueprint job delivers back to this chat. + """ + from gateway.run import _command_origin_for_source + args = (event.get_command_args() or "").strip() + origin = _command_origin_for_source(event.source) + try: + from hermes_cli.blueprint_cmd import handle_blueprint_command + + return handle_blueprint_command(args, origin=origin, surface="gateway") + except Exception as e: + logger.debug("blueprint command failed: %s", e) + from hermes_cli.blueprint_cmd import BlueprintCommandResult + + return BlueprintCommandResult(f"Cron blueprint command failed: {e}") + + async def _maybe_confirm_destructive_slash( + self, + *, + event: MessageEvent, + command: str, + title: str, + detail: str, + execute, + ) -> Union[str, "EphemeralReply", None]: + """Gate a destructive session slash command (/new, /reset, /undo). + + ``execute`` is an async ``execute() -> str | EphemeralReply`` performing the action. It + runs immediately if ``approvals.destructive_slash_confirm`` is off; otherwise this routes + through ``_request_slash_confirm`` (native buttons or text fallback): ``once`` runs it, + ``always`` persists ``destructive_slash_confirm: false`` then runs it, ``cancel`` returns + a "cancelled" message without running it. + """ + # Gate check. + confirm_required = True + try: + cfg = self._read_user_config() + approvals = cfg.get("approvals") if isinstance(cfg, dict) else None + if isinstance(approvals, dict): + confirm_required = bool(approvals.get("destructive_slash_confirm", True)) + except Exception: + pass + + if not confirm_required: + return await execute() + + session_key = self._session_key_for_source(event.source) + + async def _on_confirm(choice: str): + if choice == "cancel": + return f"🟡 /{command} cancelled. Conversation unchanged." + persisted = False + if choice == "always": + try: + from cli import save_config_value + # save_config_value swallows its own errors and reports the + # outcome in the return value, so the try block alone says + # nothing about whether the write landed. + persisted = bool( + save_config_value("approvals.destructive_slash_confirm", False) + ) + if persisted: + logger.info( + "User opted out of destructive slash confirm (session=%s)", + session_key, + ) + else: + logger.warning( + "Could not persist destructive_slash_confirm=false " + "(session=%s); config.yaml is not writable", + session_key, + ) + except Exception as exc: + logger.warning( + "Failed to persist destructive_slash_confirm=false: %s", exc, + ) + result = await execute() + if choice == "always": + if persisted: + note = ( + "\n\nℹ️ Future /clear, /new, /reset, and /undo will run " + "without confirmation. Re-enable via " + "`approvals.destructive_slash_confirm: true` in config.yaml." + ) + else: + # The user did approve this run, so the action still goes ahead, but the + # preference did not stick and the prompt will be back next time. Say so rather + # than promising an opt-out that was never written. + note = ( + "\n\n⚠️ Could not save that preference (config.yaml is not " + "writable), so /clear, /new, /reset, and /undo will ask " + "again next time. To silence it permanently, set " + "`approvals.destructive_slash_confirm: false` in config.yaml." + ) + if isinstance(result, str): + return result + note + # EphemeralReply or other: leave untouched, since the note would + # mangle structured replies. + return result + return result + + _p = self._typed_command_prefix_for(event.source.platform) + prompt_message = ( + f"⚠️ **Confirm /{command}**\n\n" + f"{detail}\n\n" + "Choose:\n" + "• **Approve Once** — proceed this time only\n" + "• **Always Approve** — proceed and silence this prompt permanently\n" + "• **Cancel** — keep current conversation\n\n" + f"_Text fallback: reply `{_p}approve`, `{_p}always`, or `{_p}cancel`._" + ) + return await self._request_slash_confirm( + event=event, + command=command, + title=title, + message=prompt_message, + handler=_on_confirm, + ) + + async def _request_slash_confirm( + self, + *, + event: MessageEvent, + command: str, + title: str, + message: str, + handler, + ) -> Optional[str]: + """Ask the user to confirm an expensive slash command. + + ``handler(choice: str) -> str`` runs on the event loop when the user responds with + ``"once"``, ``"always"``, or ``"cancel"``; its return value is sent as a gateway message. + Returns the immediate acknowledgment: ``None`` if buttons rendered (self-explanatory), + otherwise the text-fallback message itself IS the ack. + """ + from tools import slash_confirm as _slash_confirm_mod + + source = event.source + session_key = self._session_key_for_source(source) + # Bare-runner test harnesses (object.__new__(GatewayRunner)) skip __init__ and lack the + # counter attribute; fall back to a local counter. Real runs always have the attribute. + counter = getattr(self, "_slash_confirm_counter", None) + if counter is None: + import itertools as _itertools + counter = _itertools.count(1) + self._slash_confirm_counter = counter + confirm_id = f"{next(counter)}" + + # Register the pending confirm FIRST so a super-fast button click + # cannot race the send_slash_confirm return. + _slash_confirm_mod.register(session_key, confirm_id, command, handler) + + adapter = self._adapter_for_source(source) + metadata = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) + + used_buttons = False + if adapter is not None: + try: + button_result = await adapter.send_slash_confirm( + chat_id=source.chat_id, + title=title, + message=message, + session_key=session_key, + confirm_id=confirm_id, + metadata=metadata, + ) + if button_result and getattr(button_result, "success", False): + used_buttons = True + except Exception as exc: + logger.debug( + "send_slash_confirm failed for %s on %s: %s", + command, source.platform, exc, + ) + + if used_buttons: + # Buttons rendered — no redundant text ack. + return None + # Text fallback — return the prompt message as the direct reply. + return message + + def _read_user_config(self) -> Dict[str, Any]: + """Read the user's raw config.yaml (cached) for gate lookups. + + Used by slash-confirm gates that must reflect on-disk state changes + (e.g. a prior "Always Approve" click) without a gateway restart. + """ + try: + from hermes_cli.config import load_config + cfg = load_config() + return cfg if isinstance(cfg, dict) else {} + except Exception: + return {} diff --git a/gateway/run_common.py b/gateway/run_common.py new file mode 100644 index 0000000000..6d92cec9aa --- /dev/null +++ b/gateway/run_common.py @@ -0,0 +1,8 @@ +"""Leaf constants shared by ``gateway/run.py`` and its ``run_*`` mixin modules. + +Kept import-cycle free (imports nothing from ``gateway.run``) because these values +are used as default-argument sentinels, which must resolve at ``def`` time. +""" + +# Sentinel for "caller did not pass metadata" vs "caller passed None". +_UNSET = object() diff --git a/gateway/run_config_loaders.py b/gateway/run_config_loaders.py new file mode 100644 index 0000000000..bf64fe6b67 --- /dev/null +++ b/gateway/run_config_loaders.py @@ -0,0 +1,620 @@ +"""Config/env loaders for runtime knobs (busy modes, reasoning, service tier, timeouts, fallback) for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import json +import os +import time +from gateway.config import Platform +from gateway.restart import ( + DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, + DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT, + DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, + DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT, + DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, + parse_cron_drain_timeout, + parse_restart_after_turn_timeout, + parse_restart_drain_timeout, + parse_signal_interrupt_grace_timeout, +) +from gateway.session import SessionSource +from gateway.session_state import SERVICE_TIER_UNSET as _SERVICE_TIER_UNSET +from hermes_cli.config import cfg_get +from hermes_cli.fallback_config import get_fallback_chain +from pathlib import Path +from typing import Any, Dict, List, Optional +from utils import is_truthy_value + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayConfigLoadersMixin: + """Config/env loaders for runtime knobs (busy modes, reasoning, service tier, timeouts, fallback) for GatewayRunner.""" + + @staticmethod + def _load_prefill_messages() -> List[Dict[str, Any]]: + """Load ephemeral prefill messages from config or env var. + + HERMES_PREFILL_MESSAGES_FILE env wins, then top-level prefill_messages_file in config.yaml, + then legacy agent.prefill_messages_file. Relative paths resolve from ~/.hermes/. + """ + from gateway.run import _hermes_home, _load_gateway_runtime_config + file_path = os.getenv("HERMES_PREFILL_MESSAGES_FILE", "") + if not file_path: + cfg = _load_gateway_runtime_config() + file_path = str(cfg.get("prefill_messages_file", "") or "") + if not file_path: + file_path = str(cfg_get(cfg, "agent", "prefill_messages_file", default="") or "") + if not file_path: + return [] + path = Path(file_path).expanduser() + if not path.is_absolute(): + path = _hermes_home / path + if not path.exists(): + logger.warning("Prefill messages file not found: %s", path) + return [] + try: + with open(path, "r", encoding="utf-8") as f: + data = json.load(f) + if not isinstance(data, list): + logger.warning("Prefill messages file must contain a JSON array: %s", path) + return [] + return data + except Exception as e: + logger.warning("Failed to load prefill messages from %s: %s", path, e) + return [] + + @staticmethod + def _load_ephemeral_system_prompt() -> str: + """Load ephemeral system prompt: HERMES_EPHEMERAL_SYSTEM_PROMPT env var first, then + ``display.personality`` / ``agent.system_prompt`` in config.yaml. + """ + from gateway.run import _load_gateway_runtime_config + from hermes_cli.config import resolve_ephemeral_system_prompt_from_config + + prompt = os.getenv("HERMES_EPHEMERAL_SYSTEM_PROMPT", "") + if prompt: + return prompt + cfg = _load_gateway_runtime_config() + return resolve_ephemeral_system_prompt_from_config(cfg) + + def _resolve_model_for_channel( + self, + platform: Platform, + chat_id: str, + *, + user_config: Optional[dict] = None, + thread_id: Optional[str] = None, + parent_id: Optional[str] = None, + ) -> str: + """Resolve model for this channel: channel_overrides else global default. + + Precedence lives in :func:`hermes_cli.model_switch.resolve_effective_model` (shared with the + API server so the surfaces cannot diverge). No session tier here: session /model overrides + are applied later by ``_apply_session_model_override``. + """ + from gateway.run import _get_channel_override, _resolve_gateway_model + from hermes_cli.model_switch import resolve_effective_model + + override = None + config = getattr(self, "config", None) + if config: + override = _get_channel_override( + config, + platform, + chat_id, + thread_id=thread_id, + parent_id=parent_id, + ) + return resolve_effective_model( + None, # session tier applied downstream (_apply_session_model_override) + override, + _resolve_gateway_model(user_config), + ) + + def _get_system_prompt_for_channel( + self, + platform: Platform, + chat_id: str, + *, + thread_id: Optional[str] = None, + parent_id: Optional[str] = None, + ) -> str: + """Ephemeral system prompt for this channel/thread. + + ``channel_overrides`` when set, else the gateway prompt resolved from the CURRENT profile's + config on every call (callers run inside ``_profile_runtime_scope``, so routed multiplex + profiles get their own personality/system_prompt and ``/personality`` edits apply next turn). + Legacy ``channel_prompts`` are applied separately via ``event.channel_prompt`` in ``run_sync``. + """ + from gateway.run import _get_channel_override + config = getattr(self, "config", None) + if config: + override = _get_channel_override( + config, + platform, + chat_id, + thread_id=thread_id, + parent_id=parent_id, + ) + if override and override.system_prompt: + return (override.system_prompt or "").strip() + return self._load_ephemeral_system_prompt() + + @staticmethod + def _load_reasoning_config(model: str = "") -> dict | None: + """Load reasoning effort from config.yaml, respecting per-model overrides. + + Thin wrapper over :func:`hermes_constants.resolve_reasoning_config` (per-model override > + global ``agent.reasoning_effort``; YAML False = disabled). Empty ``model`` uses ``model.default``. + """ + from gateway.run import _load_gateway_runtime_config + from hermes_constants import resolve_reasoning_config + cfg = _load_gateway_runtime_config() + return resolve_reasoning_config(cfg, model) + + @staticmethod + def _parse_reasoning_command_args(raw_args: str) -> tuple[str, bool]: + """Parse `/reasoning` args into `(value, persist_global)`. + + Session-scoped by default; `--global` in any position persists the change to config.yaml. + """ + import shlex + + text = str(raw_args or "").strip().replace("—", "--") + if not text: + return "", False + try: + tokens = shlex.split(text) + except ValueError: + tokens = text.split() + + persist_global = False + value_tokens = [] + for token in tokens: + if token == "--global": + persist_global = True + else: + value_tokens.append(token) + return " ".join(value_tokens).strip().lower(), persist_global + + def _resolve_session_reasoning_config( + self, + *, + source: Optional[SessionSource] = None, + session_key: Optional[str] = None, + model: str = "", + ) -> dict | None: + """Resolve reasoning effort for a session, honoring session overrides. + + Priority: session ``/reasoning --session`` > per-model ``agent.reasoning_overrides`` > global + ``agent.reasoning_effort``. ``model`` must be the session's *effective* model (session + ``/model`` override included); empty uses ``model.default``. + """ + resolved_session_key = self._resolve_session_key_or_none(source, session_key) + + if resolved_session_key: + _r_state = self._peek_session_state(resolved_session_key) + if _r_state is not None and _r_state.conversation.reasoning_override is not None: + return _r_state.conversation.reasoning_override + return self._load_reasoning_config(model) + + def _set_session_reasoning_override( + self, + session_key: str, + reasoning_config: Optional[dict], + ) -> None: + """Set or clear the session-scoped reasoning override.""" + if not session_key: + return + # Per-session field write: a lazy ``_session_reasoning_overrides = {}`` init replaced the + # WHOLE dict, racing concurrent sessions; a SessionState field reset cannot cross sessions. + self._session_state(session_key).conversation.reasoning_override = ( + None if reasoning_config is None else dict(reasoning_config) + ) + + def _resolve_session_service_tier( + self, + source=None, + session_key: Optional[str] = None, + ) -> Optional[str]: + """Resolve the effective service tier for a session. + + A session-scoped /fast override beats the config default; the override dict stores + "priority" or None (explicit normal), so key presence — not truthiness — decides. + """ + resolved_session_key = self._resolve_session_key_or_none(source, session_key) + + if resolved_session_key: + _t_state = self._peek_session_state(resolved_session_key) + if ( + _t_state is not None + and _t_state.conversation.service_tier_override + is not _SERVICE_TIER_UNSET + ): + return _t_state.conversation.service_tier_override + return self._load_service_tier() + + def _set_session_service_tier_override( + self, + session_key: str, + service_tier, + clear: bool = False, + ) -> None: + """Set or clear the session-scoped /fast override. + + ``service_tier`` is "priority" or None (explicit normal). Pass + ``clear=True`` to remove the override entirely (fall back to config). + """ + if not session_key: + return + # Presence-sensitive: "priority" or None (explicit normal) both count as an override; the + # sentinel means "no override". Per-session field write: a lazy dict replace races sessions. + self._session_state(session_key).conversation.service_tier_override = ( + _SERVICE_TIER_UNSET if clear else service_tier + ) + + @staticmethod + def _load_service_tier() -> str | None: + """Load Priority Processing (agent.service_tier) from config.yaml: "fast"/"priority"/"on" => + "priority"; "normal"/"off" disable; None when unset/unsupported. + """ + from gateway.run import _load_gateway_runtime_config + cfg = _load_gateway_runtime_config() + raw = str(cfg_get(cfg, "agent", "service_tier", default="") or "").strip() + + value = raw.lower() + if not value or value in {"normal", "default", "standard", "off", "none"}: + return None + if value in {"fast", "priority", "on"}: + return "priority" + if value in {"auto", "cold"}: + return value + logger.warning("Unknown service_tier '%s', ignoring", raw) + return None + + @staticmethod + def _load_show_reasoning() -> bool: + """Load show_reasoning toggle from config.yaml display section.""" + from gateway.run import _load_gateway_runtime_config + cfg = _load_gateway_runtime_config() + return is_truthy_value( + cfg_get(cfg, "display", "show_reasoning"), + default=False, + ) + + @staticmethod + def _load_busy_input_mode() -> str: + """Load gateway drain-time busy-input behavior from config/env.""" + from gateway.run import _load_gateway_runtime_config + mode = os.getenv("HERMES_GATEWAY_BUSY_INPUT_MODE", "").strip().lower() + if not mode: + cfg = _load_gateway_runtime_config() + mode = str(cfg_get(cfg, "display", "busy_input_mode", default="") or "").strip().lower() + if mode == "queue": + return "queue" + if mode == "steer": + return "steer" + return "interrupt" + + @staticmethod + def _load_busy_text_mode() -> str: + """Resolve normal busy TEXT follow-up behavior. + + ``busy_input_mode`` is the source of truth (default ``interrupt``); legacy ``busy_text_mode`` + is honored only when explicitly set so existing queue setups keep working. + """ + from gateway.run import GatewayRunner, _load_gateway_runtime_config + # Legacy explicit override wins for backward compat. + legacy = os.getenv("HERMES_GATEWAY_BUSY_TEXT_MODE", "").strip().lower() + if not legacy: + cfg = _load_gateway_runtime_config() + legacy = str(cfg_get(cfg, "display", "busy_text_mode", default="") or "").strip().lower() + if legacy == "interrupt": + return "interrupt" + if legacy == "queue": + return "queue" + # No explicit legacy knob → follow busy_input_mode. + input_mode = GatewayRunner._load_busy_input_mode() + return "queue" if input_mode == "queue" else "interrupt" + + @staticmethod + def _busy_modes_from_config( + config: dict, + *, + fallback_input: str, + fallback_text: str, + ) -> tuple[str, str]: + """Resolve one profile's busy modes without consulting process env.""" + raw_input = str( + cfg_get(config, "display", "busy_input_mode", default="") or "" + ).strip().lower() + input_mode = ( + raw_input + if raw_input in {"interrupt", "queue", "steer"} + else fallback_input + ) + + raw_text = str( + cfg_get(config, "display", "busy_text_mode", default="") or "" + ).strip().lower() + if raw_text in {"interrupt", "queue"}: + text_mode = raw_text + elif raw_input in {"interrupt", "queue", "steer"}: + text_mode = "queue" if input_mode == "queue" else "interrupt" + else: + text_mode = fallback_text + return input_mode, text_mode + + def _snapshot_profile_busy_modes(self, profile_name: str, config: dict) -> None: + """Cache a routed profile's busy policy for this gateway lifetime.""" + input_mode, text_mode = self._busy_modes_from_config( + config, + fallback_input=getattr(self, "_busy_input_mode", "interrupt"), + fallback_text=getattr(self, "_busy_text_mode", "interrupt"), + ) + input_modes = self.__dict__.setdefault("_busy_input_modes_by_profile", {}) + text_modes = self.__dict__.setdefault("_busy_text_modes_by_profile", {}) + input_modes[profile_name] = input_mode + text_modes[profile_name] = text_mode + + def _busy_profile_name_for_source(self, source: SessionSource) -> Optional[str]: + """Return the routed profile whose busy policy applies, if any.""" + if not getattr(getattr(self, "config", None), "multiplex_profiles", False): + return None + name = str(getattr(source, "profile", "") or "").strip() + if not name: + try: + name = str(self._profile_name_for_source(source) or "").strip() + except Exception: + name = "" + return name or None + + def _effective_busy_input_mode(self, source: SessionSource) -> str: + """Resolve busy input mode from the routed profile startup snapshot.""" + fallback = getattr(self, "_busy_input_mode", "interrupt") + profile_name = self._busy_profile_name_for_source(source) + if not profile_name: + return fallback + modes = getattr(self, "_busy_input_modes_by_profile", None) + return modes.get(profile_name, fallback) if isinstance(modes, dict) else fallback + + def _effective_busy_text_mode(self, source: SessionSource) -> str: + """Resolve legacy busy text mode from the routed profile snapshot.""" + fallback = getattr(self, "_busy_text_mode", "interrupt") + profile_name = self._busy_profile_name_for_source(source) + if not profile_name: + return fallback + modes = getattr(self, "_busy_text_modes_by_profile", None) + return modes.get(profile_name, fallback) if isinstance(modes, dict) else fallback + + @staticmethod + def _load_restart_drain_timeout() -> float: + """Load graceful gateway restart/stop drain timeout in seconds.""" + from gateway.run import _load_gateway_runtime_config + raw = os.getenv("HERMES_RESTART_DRAIN_TIMEOUT", "").strip() + if not raw: + cfg = _load_gateway_runtime_config() + raw = str(cfg_get(cfg, "agent", "restart_drain_timeout", default="") or "").strip() + value = parse_restart_drain_timeout(raw) + if raw and value == DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT: + try: + float(raw) + except (TypeError, ValueError): + logger.warning( + "Invalid restart_drain_timeout '%s', using default %.0fs", + raw, + DEFAULT_GATEWAY_RESTART_DRAIN_TIMEOUT, + ) + return value + + @staticmethod + def _load_env_or_agent_cfg_timeout(env_var: str, cfg_key: str, parse, default: float) -> float: + """Env var (non-empty) else ``agent.``; warn once when a supplied value fails to parse. + + ``0`` is a valid value; the parser falls back to ``default`` on garbage.""" + from gateway.run import _load_gateway_runtime_config + env_raw = os.getenv(env_var) + if env_raw is not None and str(env_raw).strip() != "": + raw: object = env_raw + else: + cfg = _load_gateway_runtime_config() + raw = cfg_get(cfg, "agent", cfg_key, default=None) + value = parse(raw) + if raw is not None and str(raw).strip() != "": + try: + float(raw) + except (TypeError, ValueError): + logger.warning("Invalid %s '%s', using default %.0fs", cfg_key, raw, default) + return value + + @classmethod + def _load_restart_after_turn_timeout(cls) -> float: + """Load in-band restart wait-for-idle timeout in seconds.""" + return cls._load_env_or_agent_cfg_timeout( + "HERMES_RESTART_AFTER_TURN_TIMEOUT", "restart_after_turn_timeout", + parse_restart_after_turn_timeout, DEFAULT_GATEWAY_RESTART_AFTER_TURN_TIMEOUT, + ) + + @classmethod + def _load_cron_drain_timeout(cls) -> float: + """Load the cron-only floor under the stop()/drain wait.""" + return cls._load_env_or_agent_cfg_timeout( + "HERMES_CRON_DRAIN_TIMEOUT", "cron_drain_timeout", + parse_cron_drain_timeout, DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, + ) + + @staticmethod + def _load_signal_interrupt_grace_timeout() -> float: + """Load the unexpected-signal post-interrupt grace in seconds.""" + from gateway.run import _load_gateway_runtime_config + cfg = _load_gateway_runtime_config() + raw = cfg_get( + cfg, + "gateway", + "signal_interrupt_grace_timeout", + default=None, + ) + value = parse_signal_interrupt_grace_timeout(raw) + if raw is not None and raw != "": + try: + float(raw) + except (TypeError, ValueError): + logger.warning( + "Invalid signal_interrupt_grace_timeout '%s', using default %.0fs", + raw, + DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, + ) + return value + + def _post_interrupt_grace_timeout(self) -> float: + """Return the grace before teardown after forcibly interrupting agents.""" + if ( + getattr(self, "_signal_initiated_shutdown", False) + and not getattr(self, "_restart_requested", False) + ): + return max( + 0.0, + float( + getattr( + self, + "_signal_interrupt_grace_timeout", + DEFAULT_GATEWAY_SIGNAL_INTERRUPT_GRACE_TIMEOUT, + ) + ), + ) + return DEFAULT_GATEWAY_POST_INTERRUPT_GRACE_TIMEOUT + + @staticmethod + def _load_background_notifications_mode() -> str: + """Load background process notification mode from config or env var.""" + from gateway.run import _load_gateway_runtime_config + mode = os.getenv("HERMES_BACKGROUND_NOTIFICATIONS", "") + if not mode: + cfg = _load_gateway_runtime_config() + raw = cfg_get(cfg, "display", "background_process_notifications") + if raw is False: + mode = "off" + elif raw not in {None, ""}: + mode = str(raw) + mode = (mode or "concise").strip().lower() + valid = {"concise", "all", "result", "error", "off"} + if mode not in valid: + logger.warning( + "Unknown background_process_notifications '%s', defaulting to 'concise'", + mode, + ) + return "concise" + return mode + + @staticmethod + def _load_provider_routing() -> dict: + """Load OpenRouter provider routing preferences from config.yaml.""" + from gateway.run import _load_gateway_runtime_config + try: + # Canonical gateway loader (fail-open): managed overlay + ${VAR} + # expansion now apply to provider_routing too. + cfg = _load_gateway_runtime_config() + return cfg.get("provider_routing", {}) or {} + except Exception: + pass + return {} + + @staticmethod + def _load_fallback_model() -> list | None: + """Load fallback provider chain from config.yaml. + + Merges ``fallback_providers`` (kept first) with legacy ``fallback_model`` entries. + """ + from gateway.run import _load_gateway_runtime_config + try: + # Canonical gateway loader (fail-open): managed overlay + ${VAR} + # expansion now apply to the fallback chain too. + cfg = _load_gateway_runtime_config() + fb = get_fallback_chain(cfg) + if fb: + return fb + except Exception: + pass + return None + + def _refresh_fallback_model(self) -> list | None: + """Re-read fallback_providers from disk for the next agent create/reuse. + + Lets a chain edited after startup reach messaging sessions (cron already re-reads per job). + A TRANSIENT read/parse failure (user mid-edit, non-atomic write) keeps the last known-good + chain; only a successful read that genuinely lacks the key clears it. + """ + from gateway.run import _hermes_home + try: + from hermes_cli.config import read_user_config_raw + cfg_path = _hermes_home / "config.yaml" + if not cfg_path.exists(): + self._fallback_model = None + return self._fallback_model + # Raw primitive (raises on parse failure) is required here: the canonical fail-open + # loader would return {} on a torn mid-edit write and WIPE the last known-good chain. + # The overlay/expansion below fixes the managed-scope/${VAR} drift without losing that. + cfg = read_user_config_raw(cfg_path) + try: + from hermes_cli import managed_scope + cfg = managed_scope.apply_managed_overlay(cfg) + except Exception: + pass + try: + from hermes_cli.config import _expand_env_vars + expanded = _expand_env_vars(cfg) + if isinstance(expanded, dict): + cfg = expanded + except Exception: + pass + except Exception: + # Transient failure — keep last known-good chain. + logger.debug( + "fallback_providers refresh: config.yaml read failed; " + "keeping last known-good chain", exc_info=True, + ) + return self._fallback_model + self._fallback_model = get_fallback_chain(cfg) or None + return self._fallback_model + + @staticmethod + def _apply_fallback_chain_to_agent(agent: Any, chain: list | None) -> None: + """Keep a cached agent's fallback chain aligned with current config. + + Skips the rewrite while a cooldown holds the agent on an activated fallback provider + (``restore_primary_runtime`` owns that lifecycle); otherwise replaces the chain so + mid-uptime ``fallback_providers`` edits apply without a restart. + """ + if agent is None: + return + new_chain = list(chain or []) + rate_limited_until = getattr(agent, "_rate_limited_until", 0) or 0 + if ( + getattr(agent, "_fallback_activated", False) + and rate_limited_until > time.monotonic() + ): + return + old_chain = list(getattr(agent, "_fallback_chain", []) or []) + agent._fallback_chain = new_chain + agent._fallback_model = new_chain[0] if new_chain else None + if not getattr(agent, "_fallback_activated", False): + agent._fallback_index = 0 + # A config edit means the user changed something — drop the session-scoped unavailability + # memo so re-configured entries (e.g. credentials added mid-uptime) get retried. Only on real + # content change, so the per-message no-op refresh keeps the memo's rate-limiting benefit. + if new_chain != old_chain: + unavailable = getattr(agent, "_unavailable_fallback_keys", None) + if unavailable: + unavailable.clear() diff --git a/gateway/run_goals.py b/gateway/run_goals.py new file mode 100644 index 0000000000..f48ccbbe9d --- /dev/null +++ b/gateway/run_goals.py @@ -0,0 +1,532 @@ +"""Goal/heartbeat continuation, post-turn hooks and loop-wakeup watcher methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import time +from contextlib import suppress +from gateway.platforms.base import MessageEvent, MessageType +from typing import Any + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayGoalsMixin: + """Goal/heartbeat continuation, post-turn hooks and loop-wakeup watcher methods for GatewayRunner.""" + + # ──────────────────────────────────────────────────────────────── + # /goal — persistent cross-turn goals (Ralph-style loop) + # ──────────────────────────────────────────────────────────────── + def _goal_max_turns_from_config(self) -> int: + """Resolve the configured /goal turn budget for gateway sessions. + + GatewayRunner.config is a GatewayConfig dataclass, not the full user config mapping, so + top-level blocks such as ``goals`` are only reachable via hermes_cli.config.load_config(). + """ + try: + goals_cfg = ( + (self.config or {}).get("goals", {}) + if isinstance(self.config, dict) + else getattr(self.config, "goals", {}) or {} + ) + if not goals_cfg: + from hermes_cli.config import load_config + + goals_cfg = (load_config() or {}).get("goals") or {} + return int(goals_cfg.get("max_turns", 20) or 20) + except Exception: + return 20 + + async def _warm_goals_session_db(self, label: str) -> None: + """Warm the goals SessionDB cache off-loop (best-effort). + + A cold cache runs the state.db init on the loop thread and freezes the loop for the init + duration. The executor hop keeps the profile home override alive under multiplex, so the + warm cache belongs to the caller's profile. On failure the caller falls back to the + bootstrap windows, so a dropped warm-up is a bounded stall, never a crash. + """ + try: + from hermes_cli.goals import _get_session_db as _warm_goals_db + + await self._run_in_executor_with_context(_warm_goals_db) + except Exception as exc: + logger.warning("%s: session DB warm-up failed: %s", label, exc) + + async def _session_entry_for_manager(self, event: "MessageEvent", label: str): + """Session entry for a /goal or /heartbeat manager, or None when lookup fails. + + Warms the SessionDB cache off-loop first: a cold cache freezes the loop for the init + duration and drops the first write while the reply claims it was set. Internal events look + the session up WITHOUT touching activity so they never advance the idle/daily reset clock. + """ + await self._warm_goals_session_db(label) + try: + session_entry = await self.async_session_store.get_or_create_session( + event.source, + touch_activity=not bool(getattr(event, "internal", False)), + ) + except Exception as exc: + logger.debug("%s: session lookup failed: %s", label, exc) + return None + if not (getattr(session_entry, "session_id", None) or ""): + return None + return session_entry + + async def _get_goal_manager_for_event(self, event: "MessageEvent"): + """Return ``(GoalManager, session_entry)`` for this event, or ``(None, None)``.""" + try: + from hermes_cli.goals import GoalManager + except Exception as exc: + logger.debug("goal manager unavailable: %s", exc) + return None, None + session_entry = await self._session_entry_for_manager(event, "goal manager") + if session_entry is None: + return None, None + max_turns = self._goal_max_turns_from_config() + return GoalManager(session_id=session_entry.session_id, default_max_turns=max_turns), session_entry + + async def _get_heartbeat_manager_for_event(self, event: "MessageEvent"): + """Return ``(HeartbeatManager, session_entry)`` for this event, or ``(None, None)``.""" + try: + from hermes_cli.heartbeat import HeartbeatManager + except Exception as exc: + logger.debug("heartbeat manager unavailable: %s", exc) + return None, None + session_entry = await self._session_entry_for_manager(event, "heartbeat manager") + if session_entry is None: + return None, None + return HeartbeatManager(session_id=session_entry.session_id), session_entry + + def _register_heartbeat_watch(self, quick_key: str, source: Any, session_id: str) -> None: + """Track a session with an active heartbeat and start the poller. + + The registry maps ``quick_key`` → ``(source, session_id)`` so the poller can rebuild a + MessageEvent and enqueue via the adapter FIFO. In-memory by design: heartbeat STATE + survives restarts in SessionDB, but firing resumes only when the user touches /heartbeat + again (durable schedules belong to cron). + """ + watch = getattr(self, "_heartbeat_watch", None) + if watch is None: + watch = {} + self._heartbeat_watch = watch + watch[quick_key] = (source, session_id) + self._start_heartbeat_poller() + + def _unregister_heartbeat_watch(self, quick_key: str) -> None: + watch = getattr(self, "_heartbeat_watch", None) + if watch: + watch.pop(quick_key, None) + + def _start_heartbeat_poller(self) -> None: + """Start the single gateway-wide heartbeat poll task (idempotent).""" + existing = getattr(self, "_heartbeat_poll_task", None) + if existing is not None and not existing.done(): + return + + from hermes_cli.heartbeat import POLL_SECONDS + + async def _poll_loop(): + while True: + await asyncio.sleep(POLL_SECONDS) + watch = getattr(self, "_heartbeat_watch", None) + if not watch: + continue + # Warm the cache off-loop once per poll. A watch can only be registered through the + # warmed /heartbeat command, so this covers only the degraded path where that warm- + # up failed. + await self._warm_goals_session_db("heartbeat poll") + for quick_key, (source, session_id) in list(watch.items()): + try: + # Busy sessions coalesce their tick to the next idle poll. + if quick_key in self._running_agents: + continue + from hermes_cli.heartbeat import HeartbeatManager + + mgr = HeartbeatManager(session_id=session_id) + if not mgr.has_heartbeat(): + watch.pop(quick_key, None) + continue + prompt = mgr.due_prompt() + if not prompt: + continue + adapter = self._adapter_for_source(source) + if adapter is None: + continue + hb_event = MessageEvent( + text=prompt, + message_type=MessageType.TEXT, + source=source, + message_id=None, + channel_prompt=None, + ) + self._enqueue_fifo(quick_key, hb_event, adapter) + except Exception as exc: + logger.debug("heartbeat poll for %s failed: %s", quick_key, exc) + + try: + task = asyncio.create_task(_poll_loop()) + self._heartbeat_poll_task = task + # PERMANENT once started (an infinite while-True loop, no exit condition) — same as a + # _spawn_supervised watcher. Tag it so _scale_to_zero_has_live_background_work() doesn't + # treat a gateway with an active heartbeat watch as busy forever. + task._hermes_supervised_watcher = True # type: ignore[attr-defined] + _bg = getattr(self, "_background_tasks", None) + if _bg is not None: + _bg.add(task) + task.add_done_callback(_bg.discard) + except Exception: + logger.debug("Failed to start heartbeat poller", exc_info=True) + + async def _send_goal_status_notice(self, source: Any, message: str) -> None: + """Send a /goal judge status line back to the originating chat/thread.""" + adapter = self._adapter_for_source(source) + if not adapter: + logger.debug("goal continuation: no adapter for %s", getattr(source, "platform", None)) + return + + try: + metadata = self._thread_metadata_for_source(source) + except Exception: + metadata = None + + result = await adapter.send(source.chat_id, message, metadata=metadata) + if result is not None and not getattr(result, "success", True): + logger.warning( + "goal continuation: status send failed: %s", + getattr(result, "error", "unknown error"), + ) + + async def _defer_goal_status_notice_after_delivery(self, source: Any, message: str) -> None: + """Send a /goal status line after the main response is delivered. + + The adapter sends the agent response after this caller returns, so for reading order the + status must follow that send: use the adapter's one-shot post-delivery callback when + available, else fall back to direct awaited delivery rather than dropping the notice. + """ + adapter = self._adapter_for_source(source) + if not adapter: + logger.debug("goal continuation: no adapter for %s", getattr(source, "platform", None)) + return + + async def _deliver() -> None: + try: + await self._send_goal_status_notice(source, message) + except Exception as exc: + logger.warning("goal continuation: status send failed: %s", exc, exc_info=True) + + try: + session_key = self._session_key_for_source(source) + except Exception: + session_key = None + + if session_key and hasattr(adapter, "register_post_delivery_callback"): + try: + generation = None + active = getattr(adapter, "_active_sessions", {}).get(session_key) + if active is not None: + generation = getattr(active, "_hermes_run_generation", None) + adapter.register_post_delivery_callback( + session_key, + _deliver, + generation=generation, + ) + return + except Exception as exc: + logger.debug("goal continuation: post-delivery callback registration failed: %s", exc) + + await _deliver() + + async def _post_turn_goal_continuation( + self, + *, + session_entry: Any, + source: Any, + final_response: str, + ) -> None: + """Run the goal judge after a gateway turn and, if still active, enqueue a continuation + prompt for the same session. + + Called at turn boundary AFTER delivery. Uses the adapter's pending-message/FIFO machinery + so a simultaneous real user message is handled by the same queue and takes priority. + """ + try: + from hermes_cli.goals import GoalManager + except Exception as exc: + logger.debug("goal continuation: goals module unavailable: %s", exc) + return + + sid = getattr(session_entry, "session_id", None) or "" + if not sid: + return + + max_turns = self._goal_max_turns_from_config() + + # Warm the SessionDB cache off-loop: a cold cache runs the state.db init on the loop thread + # at the turn boundary; a slow init can drop the goal read and silently end the goal loop. + await self._warm_goals_session_db("goal continuation") + + mgr = GoalManager(session_id=sid, default_max_turns=max_turns) + if not mgr.is_active(): + return + + try: + from hermes_cli.goals import gather_background_processes as _gather_bg + _bg_procs = _gather_bg() + except Exception: + _bg_procs = None + + # evaluate_after_turn calls judge_goal(), a synchronous HTTP request to the auxiliary LLM; + # on the event-loop thread it blocks Discord heartbeats 10-40 s and flaps connections, so it + # is offloaded to a thread-pool executor. _run_in_executor_with_context (not bare + # run_in_executor): the profile secret scope and aux runtime context are contextvars; a + # default-executor hop drops them and aux credential resolution fails under multiplexing. + decision = await self._run_in_executor_with_context( + lambda: mgr.evaluate_after_turn( + final_response or "", + user_initiated=True, + background_processes=_bg_procs, + ), + ) + msg = decision.get("message") or "" + + # Defer the status line until after the adapter has delivered the agent's visible final + # response. The judge runs after the response is produced but before BasePlatformAdapter + # sends it, so sending here would show "✓ Goal achieved" before the answer itself. + if msg and source is not None: + await self._defer_goal_status_notice_after_delivery(source, msg) + + if not decision.get("should_continue"): + return + + prompt = decision.get("continuation_prompt") or "" + if not prompt or source is None: + return + + # Enqueue via the adapter's FIFO so a user message already in + # flight preempts the continuation naturally. + try: + adapter = self._adapter_for_source(source) + _quick_key = self._session_key_for_source(source) + if adapter and _quick_key: + cont_event = MessageEvent( + text=prompt, + message_type=MessageType.TEXT, + source=source, + message_id=None, + channel_prompt=None, + ) + self._enqueue_fifo(_quick_key, cont_event, adapter) + except Exception as exc: + logger.debug("goal continuation: enqueue failed: %s", exc) + + async def _run_post_turn_hooks( + self, + *, + agent_result: Any, + source: Any, + is_internal: bool, + event: Any = None, + ) -> None: + """Run goal and loop bookkeeping after an agent turn returns.""" + final_text = self._final_text_for_post_turn_hooks(agent_result, event) + + try: + session_entry = await self.async_session_store.get_or_create_session( + source, + touch_activity=not is_internal, + ) + except Exception as exc: + logger.debug("post-turn session resolution failed: %s", exc) + return + + # Empty interrupted/errored responses must not drive /goal, but an + # in-flight /loop tick still needs to be released and rescheduled. + if final_text.strip(): + try: + await self._post_turn_goal_continuation( + session_entry=session_entry, + source=source, + final_response=final_text, + ) + except Exception as exc: + logger.debug("goal continuation hook failed: %s", exc) + try: + await self._post_turn_loop_completion( + session_entry=session_entry, + source=source, + final_response=final_text, + ) + except Exception as exc: + logger.debug("loop completion hook failed: %s", exc) + + @staticmethod + def _final_text_for_post_turn_hooks(agent_result, event=None) -> str: + """Text for /goal and /loop after a gateway turn. + + Streamed turns return None from _handle_message_with_agent (already_sent). The delivered + reply is stashed on the event so those hooks still see it. + """ + text = "" + if isinstance(agent_result, dict): + text = str(agent_result.get("final_response") or "") + elif isinstance(agent_result, str): + text = agent_result + if text.strip(): + return text + streamed = getattr(event, "_streamed_final_response", None) + if isinstance(streamed, str) and streamed.strip(): + return streamed + return text + + async def _post_turn_loop_completion( + self, + *, + session_entry: Any, + source: Any, + final_response: str, + ) -> None: + """Complete a /loop wakeup tick after a gateway turn. + + No-op unless the session has a loop whose tick is in flight (``awaiting_response`` — set + when the wakeup was injected). Applies the LOOP_COMPLETE marker / --until judge / caps + and schedules the next tick; the idle wakeup watcher fires it when due. + """ + try: + from hermes_cli.loops import LoopManager + except Exception as exc: + logger.debug("loop completion: loops module unavailable: %s", exc) + return + + sid = getattr(session_entry, "session_id", None) or "" + if not sid: + return + + # Warm the SessionDB cache off-loop: a cold cache at the turn boundary stalls the loop for + # the init duration and can drop the tick-completion write (the /goal continuation seam). + await self._warm_goals_session_db("loop completion") + + mgr = LoopManager(session_id=sid) + state = mgr.state + if state is None or not state.awaiting_response: + return + + # The --until judge is a sync aux-LLM call — keep it off the event loop. + decision = await asyncio.get_running_loop().run_in_executor( + None, mgr.complete_tick, final_response or "" + ) + msg = decision.get("message") or "" + if msg and source is not None: + await self._defer_goal_status_notice_after_delivery(source, msg) + + async def _loop_wakeup_watcher(self, interval: float = 15.0) -> None: + """Fire due /loop wakeups for idle gateway sessions. + + The gateway has no per-session scheduler thread, so a coarse ticker scans persisted loops + (SessionDB ``loop:*`` rows) and injects the wakeup prompt into each due session's chat + via the same synthetic-message path used by watch notifications. Deferrals: session + currently running a turn → skip (the FIFO would race the live turn); active non-parked + /goal → skip (goal owns the idle boundary); no routing metadata → skip with a one-time + warning (CLI/TUI loops carry no route). + """ + await asyncio.sleep(5) # let platforms finish connecting + warned_no_route: set = set() + while self._running: + try: + from hermes_cli.loops import ( + LoopManager, + goal_blocks_loop_tick, + list_active_loops, + ) + + # Warm the cache off-loop once per scan: the scan reads every persisted loop, so a + # cold cache would run the state.db init on the loop thread before the first read. + await self._warm_goals_session_db("loop wakeup") + + now = time.time() + for sid, state in list_active_loops(): + if state.awaiting_response or now < state.next_due_at: + continue + route = state.route or {} + platform_name = route.get("platform", "") + chat_id = route.get("chat_id", "") + if not platform_name or not chat_id: + # CLI / TUI-owned loop — their own schedulers drive it. + continue + adapter = None + for p, a in self.adapters.items(): + if p.value == platform_name: + adapter = a + break + if adapter is None: + if sid not in warned_no_route: + warned_no_route.add(sid) + logger.debug( + "loop wakeup: no adapter for platform %r (session %s)", + platform_name, sid, + ) + continue + + # Build the source + session key to check business. + evt_stub = { + "session_key": "", + "platform": platform_name, + "chat_id": chat_id, + "chat_type": route.get("chat_type", ""), + "thread_id": route.get("thread_id", ""), + "user_id": route.get("user_id", ""), + "user_name": route.get("user_name", ""), + } + source = self._build_process_event_source(evt_stub) + if source is None: + continue + try: + session_key = self._session_key_for_source(source) + except Exception: + session_key = None + if session_key and session_key in self._running_agents: + continue # busy — stays due, next scan retries + if goal_blocks_loop_tick(sid): + continue + + mgr = LoopManager(session_id=sid) + if not mgr.is_due(now): + continue + wakeup = mgr.fire_tick() + if not wakeup: + continue + try: + synth_event = MessageEvent( + text=wakeup, + message_type=MessageType.TEXT, + source=source, + internal=True, + ) + logger.info( + "loop wakeup #%s — injecting for %s chat=%s thread=%s", + mgr.state.ticks_fired if mgr.state else "?", + platform_name, source.chat_id, source.thread_id, + ) + await adapter.handle_message(synth_event) + # Slash-command loops dispatch through the command + # path and never hit the post-turn completion hook — + # complete the tick immediately (caps + scheduling). + if wakeup.lstrip().startswith("/"): + mgr.complete_tick("") + except Exception as exc: + logger.warning("loop wakeup injection failed for %s: %s", sid, exc) + with suppress(Exception): + mgr.abandon_tick() + except Exception as exc: + logger.debug("loop wakeup watcher error: %s", exc) + await asyncio.sleep(interval) diff --git a/gateway/run_inbound.py b/gateway/run_inbound.py new file mode 100644 index 0000000000..8b014b1d57 --- /dev/null +++ b/gateway/run_inbound.py @@ -0,0 +1,2634 @@ +"""Inbound message pipeline (_handle_message, text/media preparation, durable-turn markers, plugin injection) for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import concurrent.futures +import dataclasses +import json +import os +import re +import time +from contextlib import suppress +from gateway.config import Platform +from gateway.platforms.base import EphemeralReply, MessageEvent, MessageType +from gateway.run_common import _UNSET +from gateway.session import ( + SessionSource, + is_shared_multi_user_session, + neutralize_untrusted_inline_text, +) +from gateway.turn_lease import TurnLeaseTimeoutError +from typing import Any, Dict, List, Optional, Tuple + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayInboundMixin: + """Inbound message pipeline (_handle_message, text/media preparation, durable-turn markers, plugin injection) for GatewayRunner.""" + + async def _hm_admit_event( + self, event: "MessageEvent" + ) -> Optional[Tuple["MessageEvent", SessionSource, bool]]: + """Ingress gates for ``_handle_message``: leak guard, profile route, ignored channels, + startup-restore queueing, ``pre_gateway_dispatch`` hook, authorization/pairing. + + Returns ``None`` when the message is dropped, else ``(event, source, is_internal)`` — + the hook may have rewritten ``event``. + """ + from gateway.run import _is_slack_ignored_channel + source = event.source + + # 🔴 Cross-session leak guard. This per-message task was created via create_task(), which + # copies the spawning context: if a concurrent message had already bound its session via + # set_session_vars(), we inherited ITS HERMES_SESSION_* ContextVars, and until _set_session_env + # binds ours any subprocess would read the foreign identity (the _UNSET-strip guard can't + # help — the vars are set-to-foreign). Reset to _UNSET so that window strips safe instead. + try: + from gateway.session_context import reset_session_vars + reset_session_vars() + except Exception: + logger.debug("reset_session_vars failed at handler entry", exc_info=True) + + # Most adapters resolve profile routes in build_source(), before they hand us the event. A + # few internal/voice paths construct SessionSource directly, so resolve those here as the + # shared fail-closed ingress gate before authorization, hooks, or session side effects. + if ( + getattr(getattr(self, "config", None), "multiplex_profiles", False) + and not getattr(source, "profile", None) + and getattr(source, "profile_route_rejected", False) is not True + ): + from gateway.profile_routing import ProfileRouteRejected + + try: + source.profile = self._profile_name_for_source(source) + except ProfileRouteRejected: + source.profile_route_rejected = True + + # SessionSource owns a strict boolean marker. Require the literal value + # so duck-typed test/internal sources with dynamic attributes are not + # mistaken for an explicit matched-route rejection. + if getattr(source, "profile_route_rejected", False) is True: + logger.warning( + "Dropping inbound message because its explicit profile route " + "targets an unserved profile" + ) + return None + + # Internal events (e.g. background-process completion notifications) + # are system-generated and must skip user authorization. + is_internal = bool(getattr(event, "internal", False)) + + # Ignored-channel guard runs FIRST — before startup-restore queueing, plugin hooks, auth, + # and session setup — so an ignored channel can never reach pairing/auth/session state. + # getattr: bare test runners construct GatewayRunner via object.__new__ without config. + if ( + not is_internal + and getattr(source, "platform", None) == Platform.SLACK + and _is_slack_ignored_channel( + getattr(self, "config", None), getattr(source, "chat_id", None) + ) + ): + logger.info( + "Dropping Slack message from configured ignored channel %s", + getattr(source, "chat_id", None), + ) + return None + + if ( + getattr(self, "_startup_restore_in_progress", False) + and not is_internal + and not getattr(event, "_hermes_startup_restore_replay", False) + ): + self._queue_startup_restore_event(event) + return None + + # scale-to-zero: stamp the gateway-scoped last-inbound clock (read by is_idle) for real + # user-originated inbound only. Internal/system events are NOT traffic — counting them + # would keep a genuinely idle gateway awake. + if not is_internal: + self._scale_to_zero_note_real_inbound() + + # pre_gateway_dispatch plugin hook (user-originated only). Plugins may return + # {"action": "skip", "reason": ...} -> drop; {"action": "rewrite", "text": ...} -> replace + # event.text; {"action": "allow"} / None -> normal dispatch. + # Runs BEFORE auth so plugins can handle unauthorized senders without the pairing flow. + if not is_internal: + try: + from hermes_cli.lifecycle import invoke_hook as _invoke_hook + _hook_results = _invoke_hook( + "pre_gateway_dispatch", + event=event, + gateway=self, + # getattr: bare-runner tests build GatewayRunner via object.__new__ without + # __init__; the hook must not fail dispatch over a missing attribute. + session_store=getattr(self, "session_store", None), + ) + except Exception as _hook_exc: + logger.warning("pre_gateway_dispatch invocation failed: %s", _hook_exc) + _hook_results = [] + + for _result in _hook_results: + if not isinstance(_result, dict): + continue + _action = _result.get("action") + if _action == "skip": + logger.info( + "pre_gateway_dispatch skip: reason=%s platform=%s chat=%s", + _result.get("reason"), + source.platform.value if source.platform else "unknown", + source.chat_id or "unknown", + ) + return None + if _action == "rewrite": + _new_text = _result.get("text") + if isinstance(_new_text, str): + event = dataclasses.replace(event, text=_new_text) + source = event.source + break + if _action == "allow": + break + + if is_internal: + pass + elif source.user_id is None: + # Messages with no user identity (Telegram service messages, channel forwards, anonymous + # admin posts, sender_chat) can't be paired but may be authorized via a chat-scoped + # allowlist (e.g. TELEGRAM_GROUP_ALLOWED_CHATS), so defer to _is_user_authorized. + if not self._is_user_authorized_for_source(source): + logger.debug("Ignoring message with no user_id from %s", source.platform.value) + return None + elif not self._is_user_authorized_for_source(source): + logger.warning("Unauthorized user: %s (%s) on %s", source.user_id, source.user_name, source.platform.value) + # In DMs: offer pairing code. In groups: silently ignore. + if ( + source.chat_type == "dm" + and self._get_unauthorized_dm_behavior( + source.platform, + profile=source.profile, + ) + == "pair" + ): + platform_name = source.platform.value if source.platform else "unknown" + pairing_store = self._pairing_store_for(source) + if pairing_store is None: + logger.error( + "Cannot offer pairing code on %s: no pairing store", + platform_name, + ) + return None + # Rate-limit ALL pairing responses (code or rejection) so a burst of DMs doesn't + # spam the user with repeated messages. + if pairing_store._is_rate_limited(platform_name, source.user_id): + return None + code = pairing_store.generate_code( + platform_name, source.user_id, source.user_name or "" + ) + if code: + adapter = self._adapter_for_source(source) + if adapter: + store_profile = getattr(pairing_store, "profile", None) + profile_arg = ( + f"-p {store_profile} " + if isinstance(store_profile, str) + and store_profile + and store_profile != "default" + else "" + ) + await adapter.send( + source.chat_id, + f"Hi~ I don't recognize you yet!\n\n" + f"Here's your pairing code: `{code}`\n\n" + f"Ask the bot owner to run:\n" + f"`hermes {profile_arg}pairing approve " + f"{platform_name} {code}`" + ) + else: + adapter = self._adapter_for_source(source) + if adapter: + await adapter.send( + source.chat_id, + "Too many pairing requests right now~ " + "Please try again later!" + ) + # Record rate limit so subsequent messages are silently ignored + pairing_store._record_rate_limit(platform_name, source.user_id) + return None + + return event, source, is_internal + + def _hm_estop_gate( + self, event: "MessageEvent", source: SessionSource, is_internal: bool + ) -> Optional[str]: + """Return the global emergency-stop notice when this turn must be blocked, else None.""" + # Global emergency stop (`hermes pause`): new turns get a brief paused notice instead of an + # agent run. Placed after auth so unauthorized senders can't probe pause state. Pause blocks + # NEW agent turns, never running work or control traffic, so these pass through: internal + # events from IN-FLIGHT work; recognized slash commands (/status, /approve, ... and /pause off + # as the in-band resume path); replies owned by in-flight work — pending update prompt, + # clarify, slash-confirm, dangerous-command approval, or steering an already-running session. + if not is_internal: + try: + from agent.estop import paused_reply as _estop_paused_reply + _paused_notice = _estop_paused_reply() + except ImportError: + _paused_notice = None + if _paused_notice is not None: + _estop_allow = False + _estop_cmd = None + try: + _estop_cmd = event.get_command() + except Exception: + _estop_cmd = None + if _estop_cmd: + try: + from hermes_cli.commands import ( + resolve_command as _resolve_estop_cmd, + ) + _estop_allow = _resolve_estop_cmd(_estop_cmd) is not None + except Exception: + _estop_allow = False + if not _estop_allow: + try: + _estop_key = self._session_key_for_source(source) + _estop_state = self._peek_session_state(_estop_key) + if ( + _estop_state is not None + and _estop_state.persistent.update_prompt_pending + ): + _estop_allow = True + if not _estop_allow and self._is_session_running(_estop_key): + # Steering / interrupting in-flight work (also covers pending clarify + + # tool approvals held by the running agent). + _estop_allow = True + if not _estop_allow: + from tools import slash_confirm as _estop_confirm_mod + if _estop_confirm_mod.get_pending(_estop_key): + _estop_allow = True + if not _estop_allow: + from tools.approval import ( + has_blocking_approval as _estop_has_approval, + ) + if _estop_has_approval(_estop_key): + _estop_allow = True + except Exception: + pass + if not _estop_allow: + logger.info( + "Gateway turn paused by global emergency stop (platform=%s chat=%s)", + getattr(getattr(source, "platform", None), "value", "unknown"), + getattr(source, "chat_id", None) or "unknown", + ) + return _paused_notice + return None + + def _hm_update_prompt_reply( + self, event: "MessageEvent", _quick_key: str, allow_gateway_control: bool + ) -> Optional[str]: + """Consume a reply to a pending ``/update`` prompt; None when nothing was consumed.""" + from gateway.run import _hermes_home + # Route replies to a pending /update prompt back to the detached update process via + # .update_response. Recognized slash commands must bypass this or /new, /help etc. get + # silently consumed as update answers. + _up_state = self._peek_session_state(_quick_key) + if ( + allow_gateway_control + and _up_state is not None + and _up_state.persistent.update_prompt_pending + ): + raw = (event.text or "").strip() + # Accept /approve and /deny as shorthand for yes/no + cmd = event.get_command() + if cmd in {"approve", "yes"}: + response_text = "y" + elif cmd in {"deny", "no"}: + response_text = "n" + else: + _recognized_cmd = None + if cmd: + try: + from hermes_cli.commands import resolve_command as _resolve_update_cmd + except Exception: + _resolve_update_cmd = None + if _resolve_update_cmd is not None: + try: + _cmd_def = _resolve_update_cmd(cmd) + _recognized_cmd = _cmd_def.name if _cmd_def else None + except Exception: + _recognized_cmd = None + response_text = "" if _recognized_cmd else raw + if response_text: + response_path = _hermes_home / ".update_response" + prompt_path = _hermes_home / ".update_prompt.json" + try: + tmp = response_path.with_suffix(".tmp") + tmp.write_text(response_text, encoding="utf-8") + tmp.replace(response_path) + prompt_path.unlink(missing_ok=True) + except OSError as e: + logger.warning("Failed to write update response: %s", e) + return f"✗ Failed to send response to update process: {e}" + _up_state.persistent.update_prompt_pending = False + label = response_text if len(response_text) <= 20 else response_text[:20] + "…" + return f"✓ Sent `{label}` to the update process." + # Recognized slash command during a pending update prompt: write a blank response so the + # detached update's ``_gateway_prompt`` returns the prompt's default (typically a safe + # "n" / skip) and exits instead of blocking on stdin until the watcher timeout. + if _recognized_cmd: + response_path = _hermes_home / ".update_response" + prompt_path = _hermes_home / ".update_prompt.json" + try: + tmp = response_path.with_suffix(".tmp") + tmp.write_text("", encoding="utf-8") + tmp.replace(response_path) + prompt_path.unlink(missing_ok=True) + logger.info( + "Recognized /%s during pending update prompt for %s; " + "cancelled prompt with default and dispatching command", + _recognized_cmd, + _quick_key, + ) + except OSError as e: + logger.warning( + "Failed to write cancel response for pending update prompt: %s", + e, + ) + _up_state.persistent.update_prompt_pending = False + return None + + async def _hm_clarify_reply( + self, + event: "MessageEvent", + source: SessionSource, + _quick_key: str, + allow_gateway_control: bool, + ) -> Optional[str]: + """Intercept a reply to a pending clarify prompt; None when the message falls through.""" + # Intercept replies to a pending clarify: open-ended prompts and "Other" responses are free + # text; direct replies to multi-choice prompts are accepted too ("2" → second option). + _clarify_mod = None + try: + from tools import clarify_gateway as _clarify_mod + _pending_clarify = _clarify_mod.get_pending_for_session( + _quick_key, include_choice_prompts=True, + ) + except Exception: + _pending_clarify = None + if ( + allow_gateway_control + and _pending_clarify is not None + and _clarify_mod is not None + ): + _clarify_has_audio = bool(self._pending_event_audio_paths(event)) + _raw_clarify_reply = await self._prepare_clarify_reply_text(event) + if _clarify_has_audio and not _raw_clarify_reply: + logger.info( + "Gateway retained pending clarify after voice transcription " + "produced no usable text (session=%s, id=%s)", + _quick_key, + _pending_clarify.clarify_id, + ) + return "" + # Skip slash commands — the user wanted a command, not to answer the clarify. Leave it + # pending so they can retry; on timeout the agent unblocks with an empty response. + if _raw_clarify_reply and not _raw_clarify_reply.startswith("/"): + _text_outcome = _clarify_mod.attempt_text_response_for_session( + _quick_key, _raw_clarify_reply, + ) + if _text_outcome == _clarify_mod.TEXT_RESOLVED: + logger.info( + "Gateway intercepted clarify text response (session=%s, id=%s)", + _quick_key, _pending_clarify.clarify_id, + ) + # The clarify callback pauses the platform typing/status indicator while waiting + # so Slack users can type their answer. The active agent resumes as soon as this + # reply resolves the wait, so re-enable its indicator here too. + _clarify_adapter = self._adapter_for_source(source) + if _clarify_adapter: + try: + _clarify_adapter.resume_typing_for_chat(source.chat_id) + except Exception: + logger.debug( + "Failed to resume typing after clarify response", + exc_info=True, + ) + # Acknowledge with empty string so adapters that emit the agent's response don't + # double-post; the agent itself produces the next user-facing message. + return "" + if _text_outcome == _clarify_mod.TEXT_REJECTED_SELECTION: + # Selection-shaped but invalid (out-of-range number, bad comma-list): keep the + # clarify armed for retry — don't cancel, don't treat as an unrelated follow-up. + logger.info( + "Gateway retained pending clarify after invalid " + "selection attempt (session=%s, id=%s)", + _quick_key, _pending_clarify.clarify_id, + ) + return "" + if _text_outcome == _clarify_mod.TEXT_REJECTED_PROSE: + # Native-choice prompts deliberately reject unmatched prose so it can continue + # through normal busy-message routing. Release this clarify first: redirect() + # degrades to steer() while tools execute, and that steer cannot drain until + # the clarify tool returns. + _clarify_mod.resolve_gateway_clarify( + _pending_clarify.clarify_id, + "", + ) + return None + + async def _hm_slash_confirm_reply( + self, event: "MessageEvent", _quick_key: str, allow_gateway_control: bool + ) -> Optional[str]: + """Resolve a reply to a pending slash-confirm prompt; None when the message falls through.""" + # Replies to a pending slash-confirm prompt (/reload-mcp etc.): /approve, /always, /cancel and + # short aliases. Anything else falls through — a stale pending confirm does NOT block other + # commands. A pending dangerous-command approval takes precedence: /approve there unblocks + # the waiting tool thread; slash-confirm only catches it when no tool approval is live. + from tools import slash_confirm as _slash_confirm_mod + _pending_confirm = _slash_confirm_mod.get_pending(_quick_key) + _tool_approval_live = False + try: + from tools.approval import has_blocking_approval + _tool_approval_live = has_blocking_approval(_quick_key) + except Exception: + _tool_approval_live = False + if allow_gateway_control and _pending_confirm and not _tool_approval_live: + _raw_reply = (event.text or "").strip() + # Accept bang-prefixed replies (`!always`, `!cancel`) verbatim: Slack/Matrix show the + # `!` prefix (typed `/` is blocked in Slack threads) and adapters only rewrite + # `!` — confirm keywords aren't commands, so the `!` survives to here. + _norm_reply = _raw_reply.lstrip("!/").lower() + _cmd_reply = event.get_command() + _confirm_choice = None + if _cmd_reply in {"approve", "yes", "ok", "confirm"}: + _confirm_choice = "once" + elif _cmd_reply in {"always", "remember"}: + _confirm_choice = "always" + elif _cmd_reply in {"cancel", "no", "deny", "nevermind"}: + _confirm_choice = "cancel" + elif _norm_reply in {"approve", "approve once", "once"}: + _confirm_choice = "once" + elif _norm_reply in {"always", "always approve"}: + _confirm_choice = "always" + elif _norm_reply in {"cancel", "nevermind", "no"}: + _confirm_choice = "cancel" + if _confirm_choice is not None: + _resolved = await _slash_confirm_mod.resolve( + _quick_key, _pending_confirm.get("confirm_id"), _confirm_choice, + ) + return _resolved or "" + # Stale pending + unrelated command: the user moved on, so drop the pending state rather + # than let the confirm block normal usage indefinitely. + _slash_confirm_mod.clear_if_stale(_quick_key) + return None + + def _hm_evict_stale_running_agent(self, _quick_key: str) -> None: + """Evict a leaked/reaped ``_running_agents`` slot before the busy-session fast-path.""" + from gateway.run import _AGENT_PENDING_SENTINEL, _float_env + # Staleness eviction: detect leaked locks from hung/crashed handlers. With inactivity-based + # timeout active tasks can run for hours, so evict only when the agent has been *idle* past + # the threshold (or has no activity tracker and its wall-clock age is extreme). + _raw_stale_timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) + _quick_state = self._peek_session_state(_quick_key) + _stale_ts = _quick_state.turn.started_ts if _quick_state else 0 + if _quick_state is not None and _quick_state.turn.agent is not None and _stale_ts: + _stale_age = time.time() - _stale_ts + _stale_agent = _quick_state.turn.agent + # Never evict the pending sentinel — it was just placed during async setup before the + # real agent exists. Sentinels have no get_activity_summary(), so the idle check would + # read inf >= timeout and evict them immediately, racing the setup path. + _stale_idle = float("inf") # assume idle if we can't check + _stale_detail = "" + _activity_summary_valid = False + if _stale_agent and hasattr(_stale_agent, "get_activity_summary"): + try: + _sa = _stale_agent.get_activity_summary() + from gateway.session_stall import ( + resolve_session_idle_seconds_from_activity, + ) + + _resolved_idle = resolve_session_idle_seconds_from_activity( + _sa if isinstance(_sa, dict) else None, + now=time.time(), + ) + if _resolved_idle is not None: + _stale_idle = _resolved_idle + _activity_summary_valid = True + _stale_detail = ( + f" | last_activity={_sa.get('last_activity_desc', 'unknown') if isinstance(_sa, dict) else 'unknown'} " + f"({_stale_idle:.0f}s ago) " + f"| iteration={_sa.get('api_call_count', 0) if isinstance(_sa, dict) else 0}/{_sa.get('max_iterations', 0) if isinstance(_sa, dict) else 0}" + ) + except Exception: + pass + # A valid activity clock is authoritative: total age alone never + # makes an actively progressing turn stale. The emergency wall TTL + # is only a fallback when the agent cannot report usable activity. + _wall_ttl = max(_raw_stale_timeout * 10, 7200) if _raw_stale_timeout > 0 else float("inf") + _should_evict = ( + _stale_agent is not _AGENT_PENDING_SENTINEL + and ( + ( + _activity_summary_valid + and _raw_stale_timeout > 0 + and _stale_idle >= _raw_stale_timeout + ) + or ( + not _activity_summary_valid + and _stale_age > _wall_ttl + ) + ) + ) + if _should_evict: + logger.warning( + "Evicting stale _running_agents entry for %s " + "(age: %.0fs, idle: %.0fs, timeout: %.0fs)%s", + _quick_key, _stale_age, _stale_idle, + _raw_stale_timeout, _stale_detail, + ) + self._invalidate_session_run_generation( + _quick_key, + reason="stale_running_agent_eviction", + ) + self._release_running_agent_state(_quick_key) + + # Durable-reaped guard. A session whose routing row was ended in state.db (``ws_orphan_reap`` + # / ``agent_close``) while the gateway lived keeps its in-memory turn slot, so the fast-path + # would queue every next message into the dead runtime. Evict the stale slot so the cold + # path re-attaches via ``get_or_create_session`` → ``reopen`` or creates a fresh session. + if self._is_session_running(_quick_key): + try: + _reap_store = getattr(self, "session_store", None) + # Use the public, lock-held accessors: peek_session_id resolves key -> session_id + # under the store lock, and returns a non-str on stubbed stores in bare test runners + # — both the isinstance() gate and the ``is True`` gate below keep this guard inert + # unless a real SessionStore answers. + _reap_peek = getattr(_reap_store, "peek_session_id", None) + _is_ended = getattr(_reap_store, "_is_session_ended_in_db", None) + _reap_sid = _reap_peek(_quick_key) if callable(_reap_peek) else None + if ( + isinstance(_reap_sid, str) + and _reap_sid + and callable(_is_ended) + and _is_ended(_reap_sid) is True + ): + logger.warning( + "Evicting stale _running_agents entry for %s — " + "durable session %s is ended (reaped) in state.db; " + "healing routing on next message (#99106)", + _quick_key, + _reap_sid, + ) + self._invalidate_session_run_generation( + _quick_key, + reason="reaped_session_eviction", + ) + self._release_running_agent_state(_quick_key) + except Exception: + logger.debug("reaped-session staleness check failed", exc_info=True) + + async def _hm_handle_running_session_message( + self, event: "MessageEvent", source: SessionSource, _quick_key: str + ) -> Optional[str]: + """Fast-path for a message that arrives while this session's agent is running.""" + from gateway.run import ( + _AGENT_PENDING_SENTINEL, + _build_media_placeholder, + merge_pending_message_event, + ) + # PRIORITY handling when an agent is already running for this session. Default behavior is + # to interrupt immediately so user text/stop messages are handled with minimal latency. + # Exception: Telegram photo bursts arrive as near-simultaneous updates — do NOT interrupt + # for photo-only follow-ups; adapter-level batching absorbs them. + # Resolve the command once; each command's mid-run behavior is declared on its + # CommandDef (busy_policy / busy_handler in hermes_cli/commands.py) and dispatched via + # _dispatch_busy_slash_command below — no per-command if-chain here. + from hermes_cli.commands import resolve_command as _resolve_cmd_inner + _evt_cmd = event.get_command() + _cmd_def_inner = _resolve_cmd_inner(_evt_cmd) if _evt_cmd else None + + # /status and /context are intentionally pre-gate so users + # always see session state. + if _cmd_def_inner and _cmd_def_inner.name == "status": + return await self._handle_status_command(event) + if _cmd_def_inner and _cmd_def_inner.name == "context": + return await self._handle_context_command(event) + + # Slash command access control on the running-agent fast-path. Mirrors the cold-path + # gate below so non-admins can't bypass gating just because an agent is busy. /status + # above is intentionally pre-gate; /help and /whoami are the always-allowed floor. + if _evt_cmd and _cmd_def_inner is not None: + _denied = self._check_slash_access(source, _cmd_def_inner.name) + if _denied is not None: + return _denied + + # Any recognized slash command: dispatch according to its declared busy_policy (dispatch + # / interrupt_then_dispatch / reject). Unrecognized commands and plain text fall through + # to the interrupt/queue logic below. + if _cmd_def_inner: + return await self._dispatch_busy_slash_command( + event, _cmd_def_inner, _quick_key, source, + ) + + if event.message_type == MessageType.PHOTO: + logger.debug("PRIORITY photo follow-up for session %s — queueing without interrupt", _quick_key) + adapter = self._adapter_for_source(source) + if adapter: + merge_pending_message_event(adapter._pending_messages, _quick_key, event) + return None + + effective_busy_input_mode = self._effective_busy_input_mode(source) + _telegram_followup_grace = float( + os.getenv("HERMES_TELEGRAM_FOLLOWUP_GRACE_SECONDS", "3.0") + ) + _grace_state = self._peek_session_state(_quick_key) + _started_at = _grace_state.turn.started_ts if _grace_state else 0 + if ( + source.platform == Platform.TELEGRAM + and event.message_type == MessageType.TEXT + and _telegram_followup_grace > 0 + and _started_at + and (time.time() - _started_at) <= _telegram_followup_grace + ): + logger.debug( + "Telegram follow-up arrived %.2fs after run start for %s — queueing without interrupt", + time.time() - _started_at, + _quick_key, + ) + adapter = self._adapter_for_source(source) + if adapter: + if effective_busy_input_mode == "queue": + self._enqueue_fifo(_quick_key, event, adapter) + else: + merge_pending_message_event( + adapter._pending_messages, + _quick_key, + event, + merge_text=True, + ) + return None + + _ra_state = self._peek_session_state(_quick_key) + running_agent = _ra_state.turn.agent if _ra_state else None + if running_agent is _AGENT_PENDING_SENTINEL: + # Agent is being set up but not ready yet. + if event.get_command() == "stop": + # Force-clean the sentinel so the session is unlocked. + self._release_running_agent_state(_quick_key) + logger.info("HARD STOP (pending) for session %s — sentinel cleared", _quick_key) + return EphemeralReply("⚡ Force-stopped. The agent was still starting — session unlocked.") + # Queue the message so it will be picked up after the + # agent starts. + adapter = self._adapter_for_source(source) + if adapter: + merge_pending_message_event( + adapter._pending_messages, + _quick_key, + event, + merge_text=True, + ) + return None + if self._draining: + queue_during_drain = self._queue_during_drain_enabled( + effective_busy_input_mode + ) + if queue_during_drain: + self._queue_or_replace_pending_event(_quick_key, event) + return ( + f"⏳ Gateway {self._status_action_gerund()} — queued for the next turn after it comes back." + if queue_during_drain + else f"⏳ Gateway is {self._status_action_gerund()} and is not accepting another turn right now." + ) + if effective_busy_input_mode == "queue": + logger.debug("PRIORITY queue follow-up for session %s", _quick_key) + self._queue_or_replace_pending_event(_quick_key, event) + return None + if effective_busy_input_mode == "steer": + # Steer mode: inject text into the running agent mid-run via + # agent.steer(). Falls back to queue semantics if the payload + # is empty, the agent lacks steer(), or steer() rejects. + steer_text = (event.text or "").strip() + steered = False + if ( + event.message_type == MessageType.TEXT + and not event.media_urls + and not event.media_types + and steer_text + and hasattr(running_agent, "steer") + ): + try: + steered = bool(running_agent.steer(steer_text)) + except Exception as exc: + logger.warning("PRIORITY steer failed for session %s: %s", _quick_key, exc) + steered = False + if steered: + logger.debug("PRIORITY steer for session %s", _quick_key) + return None + logger.debug("PRIORITY steer-fallback-to-queue for session %s", _quick_key) + self._queue_or_replace_pending_event(_quick_key, event) + return None + # Subagent protection (PRIORITY path). Same rationale as + # ``_handle_active_session_busy_message``: an interrupt cascades through + # ``_active_children`` and aborts in-flight delegate_task work, so demote to queue + # semantics while subagents run. /stop reached its handler above — still an escape hatch. + if self._agent_has_active_subagents(running_agent): + logger.info( + "PRIORITY interrupt demoted to queue for session %s " + "because the running agent has active subagents (#30170)", + _quick_key, + ) + self._queue_or_replace_pending_event(_quick_key, event) + return None + # Compression protection (PRIORITY path), as in ``_handle_active_session_busy_message``: + # an interrupt would start a new turn on the pre-rotation parent while compression + # rotates the id away, forking orphaned siblings. Demote to queue until rotation lands. + if await self._session_has_compression_in_flight(_quick_key): + logger.info( + "PRIORITY interrupt demoted to queue for session %s " + "because context compression is in flight (#56391)", + _quick_key, + ) + self._queue_or_replace_pending_event(_quick_key, event) + return None + # Text-only corrections redirect the live turn (preserving displayed context) when the + # runtime supports it; media/voice and older runtimes use the interrupt path below. + if ( + event.message_type == MessageType.TEXT + and not event.media_urls + and not event.media_types + and getattr(running_agent, "_supports_active_turn_redirect", False) + is True + and hasattr(running_agent, "redirect") + ): + try: + if running_agent.redirect((event.text or "").strip()): + logger.debug("PRIORITY redirect for session %s", _quick_key) + return None + except Exception as exc: + logger.warning( + "PRIORITY redirect failed for session %s: %s", + _quick_key, + exc, + ) + logger.debug("PRIORITY interrupt for session %s", _quick_key) + _interrupt_text = event.text + _media_urls = getattr(event, "media_urls", None) or [] + if self._pending_event_audio_paths(event): + _interrupt_text, _ = await self._transcribe_and_echo_pending_voice( + event, + self._adapter_for_source(source), + source, + event.text or "", + log_context="Voice-priority-interrupt", + ) + elif not _interrupt_text and _media_urls: + _interrupt_text = _build_media_placeholder(event) + running_agent.interrupt(_interrupt_text) + # The interrupt message is delivered via adapter._pending_messages (read by _run_agent); + # don't also buffer it on self — that copy was never consumed and grew unbounded. + return None + + async def _hm_resolve_command( + self, event: "MessageEvent", source: SessionSource, _quick_key: str + ) -> Tuple[bool, Optional[str], Optional[str], Optional[str]]: + """Resolve the slash command (aliases, access gate, ``pre_command`` + ``command:`` hooks). + + Returns ``(handled, result, command, canonical)``; when ``handled`` the caller returns + ``result`` as-is (it may legitimately be None). + """ + # Check for commands + command = event.get_command() + + from hermes_cli.commands import ( + is_gateway_known_command, + resolve_command as _resolve_cmd, + ) + + # Resolve aliases to canonical name so dispatch and hook names + # don't depend on the exact alias the user typed. + _cmd_def = _resolve_cmd(command) if command else None + canonical = _cmd_def.name if _cmd_def else command + + # Expand alias quick commands before built-in dispatch so targets like /model openai/gpt-5.5 + # --provider openrouter reach the /model handler. Preserve built-in precedence; aliases only + # need early handling when the typed command is not already known. + if command and _cmd_def is None: + if isinstance(self.config, dict): + quick_commands = self.config.get("quick_commands", {}) or {} + else: + quick_commands = getattr(self.config, "quick_commands", {}) or {} + if isinstance(quick_commands, dict) and command in quick_commands: + qcmd = quick_commands[command] + if qcmd.get("type") == "alias": + target = (qcmd.get("target") or "").strip() + if target: + target = target if target.startswith("/") else f"/{target}" + target_command = target.lstrip("/") + user_args = event.get_command_args().strip() + event.text = f"{target} {user_args}".strip() + command = target_command.split()[0] if target_command else target_command + _cmd_def = _resolve_cmd(command) if command else None + canonical = _cmd_def.name if _cmd_def else command + + # Per-platform slash command access control. Only kicks in when the operator has set + # ``allow_admin_from`` for the source's scope (DM vs group). When unset → backward-compat: + # every allowed user can run every command. When set → non-admins get only + # ``user_allowed_commands`` plus the /help, /whoami floor. Plain chat is never gated. + if command and canonical and is_gateway_known_command(canonical): + _denied = self._check_slash_access(source, canonical) + if _denied is not None: + return True, _denied, command, canonical + + # pre_command observer hook (returns ignored) fires for every recognized slash command + # BEFORE core handling, mirroring cli.py. The running-agent intercept path above (/stop, + # /approve, busy_policy) deliberately does NOT fire it — a slow or hostile plugin must not + # interfere with the operator's escape hatches for a live agent. + if command and is_gateway_known_command(canonical): + try: + from hermes_cli.plugins import fire_pre_command_hook + fire_pre_command_hook( + surface="gateway", + command=str(canonical), + alias_used=str(command), + args_raw=event.get_command_args().strip(), + session_key=_quick_key, + platform=source.platform.value if source.platform else "", + ) + except Exception as _pre_cmd_err: + logger.debug( + "pre_command hook dispatch failed (non-fatal): %s", + _pre_cmd_err, + ) + + # Fire ``command:`` for any recognized slash command (built-in or plugin). + # Handlers may return ``{"decision": "deny" | "handled" | "rewrite", ...}`` to intercept + # dispatch; handlers returning nothing behave as plain observers. + if command and is_gateway_known_command(canonical): + raw_args = event.get_command_args().strip() + hook_ctx = { + "platform": source.platform.value if source.platform else "", + "user_id": source.user_id, + "command": canonical, + "raw_command": command, + "args": raw_args, + "raw_args": raw_args, + } + try: + hook_results = await self.hooks.emit_collect( + f"command:{canonical}", hook_ctx + ) + except Exception as _hook_err: + logger.debug( + "command:%s hook dispatch failed (non-fatal): %s", + canonical, _hook_err, + ) + hook_results = [] + + for hook_result in hook_results: + if not isinstance(hook_result, dict): + continue + decision = str(hook_result.get("decision", "")).strip().lower() + if not decision or decision == "allow": + continue + if decision == "deny": + message = hook_result.get("message") + if isinstance(message, str) and message: + return True, message, command, canonical + return True, f"Command `/{command}` was blocked by a hook.", command, canonical + if decision == "handled": + message = hook_result.get("message") + return True, message if isinstance(message, str) and message else None, command, canonical + if decision == "rewrite": + new_command = str( + hook_result.get("command_name", "") + ).strip().lstrip("/") + if not new_command: + continue + new_args = str(hook_result.get("raw_args", "")).strip() + event.text = f"/{new_command} {new_args}".strip() + command = event.get_command() + _cmd_def = _resolve_cmd(command) if command else None + canonical = _cmd_def.name if _cmd_def else command + break + + return False, None, command, canonical + + async def _hm_dispatch_canonical_command( + self, + event: "MessageEvent", + source: SessionSource, + _quick_key: str, + canonical: Optional[str], + ) -> Tuple[bool, Optional[str]]: + """Dispatch built-in idle-path commands (plain handlers, /new, /learn, /plan, /moa, ...). + + Returns ``(handled, result)``. Prompt-rewriting commands (/learn, /plan, /init, /steer, + /moa, ...) mutate ``event.text`` and return ``(False, None)`` to fall through to the agent. + """ + plain_handler = ( + self._gateway_plain_command_handlers().get(canonical) + or self._gateway_idle_command_handlers().get(canonical) + ) + if plain_handler is not None: + return True, await plain_handler(event) + + if canonical == "new": + if await asyncio.to_thread(self._is_telegram_topic_root_lobby, source): + return True, self._telegram_topic_root_new_message() + async def _do_reset(): + return await self._handle_reset_command(event) + return True, await self._maybe_confirm_destructive_slash( + event=event, + command="new", + title="/new", + detail=( + "This starts a fresh session and discards the current " + "conversation history." + ), + execute=_do_reset, + ) + + if canonical == "start": + logger.info("Ignoring /start platform ping for session %s", _quick_key) + return True, "" + + if canonical == "egress": + from hermes_cli.proxy_cli import format_status_text + + return True, format_status_text() + + if canonical == "learn": + # Open-ended: rewrite the turn to a standards-guided prompt and fall through to normal + # agent processing. Mirrors the /blueprint fall-through so role alternation is + # preserved. No engine, works on any backend. + from agent.learn_prompt import build_learn_prompt + + _learn_req = event.get_command_args().strip() + _ack = ( + "Learning a skill from what you described…" + if _learn_req + else "Learning a skill from this conversation…" + ) + await self._send_command_ack(source, _ack, "learn") + try: + event.text = build_learn_prompt(_learn_req) + # fall through to agent processing + except Exception: + return True, "Could not start /learn — please try again." + + if canonical == "plan": + # /plan: rewrite the turn to the plan-mode prompt and fall through to normal agent + # processing (the /learn fall-through keeps role alternation). Works on any backend. + from agent.plan_prompt import build_plan_prompt + + _plan_task = event.get_command_args().strip() + _ack = ( + f"Planning: {_plan_task[:80]}{'…' if len(_plan_task) > 80 else ''}" + if _plan_task + else "Planning from this conversation's context…" + ) + await self._send_command_ack(source, _ack, "plan") + try: + event.text = build_plan_prompt(_plan_task) + # fall through to agent processing + except Exception: + return True, "Could not start /plan — please try again." + + if canonical == "init": + # /init: rewrite the turn to a guidance-laden prompt and fall through to normal agent + # processing (the /learn fall-through keeps role alternation). Works on any backend. + from hermes_cli.init_command import build_init_prompt_for_cwd + + _init_notes = event.get_command_args().strip() + try: + _init_prompt = build_init_prompt_for_cwd(extra=_init_notes) + except Exception: + return True, "Could not start /init — please try again." + _ack = ( + "Updating AGENTS.md from a project scan…" + if "UPDATE the existing AGENTS.md" in _init_prompt + else "Generating AGENTS.md from a project scan…" + ) + await self._send_command_ack(source, _ack, "init") + event.text = _init_prompt + # fall through to agent processing + + if canonical == "blueprint": + _blueprint_result = await self._handle_blueprint_command(event) + _blueprint_seed = getattr(_blueprint_result, "agent_seed", None) + if _blueprint_seed: + # Blueprint matched — rewrite the turn to the seed and fall through to + # _handle_message_with_agent so the agent collects each slot value conversationally, + # then calls the cronjob tool (the /steer fall-through pattern). + _ack = getattr(_blueprint_result, "text", "") or "" + if _ack: + await self._send_command_ack(source, _ack, "blueprint") + try: + event.text = _blueprint_seed + except Exception: + return True, getattr(_blueprint_result, "text", "") or None + else: + return True, getattr(_blueprint_result, "text", "") or None + + if canonical == "undo": + async def _do_undo(): + return await self._handle_undo_command(event) + _undo_n = 1 + _undo_raw = event.get_command_args().strip() + if _undo_raw: + try: + _undo_n = max(1, int(_undo_raw.split()[0])) + except (ValueError, IndexError): + _undo_n = 1 + _undo_detail = ( + "This removes the last user/assistant exchange from history." + if _undo_n == 1 + else f"This removes the last {_undo_n} user turns from history." + ) + return True, await self._maybe_confirm_destructive_slash( + event=event, + command="undo", + title="/undo", + detail=_undo_detail, + execute=_do_undo, + ) + + if canonical == "queue": + queue_payload = event.get_command_args().strip() + if not queue_payload: + return True, "Usage: /queue " + with suppress(Exception): + event.text = queue_payload + + if canonical == "steer": + # No active agent — /steer has nothing to inject into. Strip the prefix so downstream + # treats it as a normal user message; an empty payload surfaces the usage hint. + steer_payload = event.get_command_args().strip() + if not steer_payload: + return True, "Usage: /steer (no agent is running; sending as a normal message)" + with suppress(Exception): + event.text = steer_payload + # Do NOT return — fall through to _handle_message_with_agent at the end of this function + # so the rewritten text is sent to the agent as a regular user turn. + + if canonical == "moa": + # /moa is one-shot sugar only: run a single prompt through the default MoA preset, then + # restore the prior model. To *switch* to a MoA preset for the session, pick it from the + # model picker (MoA presets surface as a virtual "Mixture of Agents" provider). + from hermes_cli.moa_config import ( + moa_usage, + normalize_moa_config, + ) + from hermes_cli.config import load_config + + moa_payload = event.get_command_args().strip() + if not moa_payload: + return True, moa_usage() + try: + cfg = load_config() + moa_cfg = normalize_moa_config(cfg.get("moa") if isinstance(cfg, dict) else {}) + except Exception: + moa_cfg = normalize_moa_config({}) + preset = moa_cfg["default_preset"] + try: + event.text = moa_payload + _moa_state = self._session_state(_quick_key) + event._moa_restore_override = _moa_state.conversation.model_override + _moa_state.conversation.model_override = { + "provider": "moa", + "model": preset, + "base_url": "moa://local", + "api_key": "moa-virtual-provider", + "api_mode": "chat_completions", + } + self._evict_cached_agent(_quick_key) + event._moa_disable_after_turn = True + except Exception: + return True, "Failed to prepare MoA turn." + + return False, None + + async def _hm_dispatch_quick_and_plugin_commands( + self, event: "MessageEvent", source: SessionSource, command: Optional[str] + ) -> Tuple[bool, Optional[str], Optional[str]]: + """Drain gate, user-defined quick commands (exec/alias) and plugin slash commands. + + Returns ``(handled, result, command)`` — an alias quick command rewrites ``command``. + """ + if self._draining: + return True, f"⏳ Gateway is {self._status_action_gerund()} and is not accepting new work right now.", command + + # User-defined quick commands (bypass agent loop, no LLM call) + if command: + if isinstance(self.config, dict): + quick_commands = self.config.get("quick_commands", {}) or {} + else: + quick_commands = getattr(self.config, "quick_commands", {}) or {} + if not isinstance(quick_commands, dict): + quick_commands = {} + if command in quick_commands: + # Quick commands are slash capabilities too — and type:exec ones run a shell command + # in the gateway process. The early gate only fires for registry-known commands and + # quick commands are never in the registry, so apply the same admin/user policy to + # the raw typed name here so non-admins can't invoke admin-only quick commands. + _denied = self._check_slash_access(source, command) + if _denied is not None: + return True, _denied, command + qcmd = quick_commands[command] + if qcmd.get("type") == "exec": + exec_cmd = qcmd.get("command", "") + if exec_cmd: + try: + # Sanitize env to prevent credential leakage — quick commands run in the + # gateway process, which has all API keys in os.environ. + from tools.environments.local import build_subprocess_env + sanitized_env = build_subprocess_env() + proc = await asyncio.create_subprocess_shell( + exec_cmd, + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + env=sanitized_env, + ) + stdout, stderr = await asyncio.wait_for(proc.communicate(), timeout=30) + output = (stdout or stderr).decode().strip() + # Redact any remaining sensitive patterns in output + if output: + from agent.redact import redact_sensitive_text + output = redact_sensitive_text(output) + return True, output if output else "Command returned no output.", command + except asyncio.TimeoutError: + return True, "Quick command timed out (30s).", command + except Exception as e: + return True, f"Quick command error: {e}", command + else: + return True, f"Quick command '/{command}' has no command defined.", command + elif qcmd.get("type") == "alias": + target = (qcmd.get("target") or "").strip() + if target: + target = target if target.startswith("/") else f"/{target}" + target_command = target.lstrip("/") + user_args = event.get_command_args().strip() + event.text = f"{target} {user_args}".strip() + command = target_command.split()[0] if target_command else target_command + # Fall through to normal command dispatch below + else: + return True, f"Quick command '/{command}' has no target defined.", command + else: + return True, f"Quick command '/{command}' has unsupported type (supported: 'exec', 'alias').", command + + # Plugin-registered slash commands + if command: + try: + from hermes_cli.plugins import get_plugin_command_handler + # Normalize underscores to hyphens so Telegram's underscored autocomplete form + # matches plugin commands registered with hyphens (see _build_telegram_menu). + plugin_handler = get_plugin_command_handler(command.replace("_", "-")) + if plugin_handler: + user_args = event.get_command_args().strip() + result = plugin_handler(user_args) + if asyncio.iscoroutine(result): + result = await result + return True, str(result) if result else None, command + except Exception as e: + logger.warning("Plugin command dispatch failed: %s", e) + + return False, None, command + + def _hm_skill_slash_rewrite( + self, + event: "MessageEvent", + source: SessionSource, + _quick_key: str, + command: Optional[str], + ) -> Optional[str]: + """Rewrite ``/`` / ``/`` invocations into the skill prompt on ``event.text``. + + Returns a reply string when the command is disabled/unknown/failed, else None. + """ + from gateway.run import _check_unavailable_skill + from hermes_cli.commands import GATEWAY_KNOWN_COMMANDS + + # Skill slash commands: /skill-name loads the skill and sends to agent. + # resolve_skill_command_key() handles the Telegram underscore/hyphen round-trip so + # /claude_code from Telegram autocomplete still resolves to the claude-code skill. + if command: + # Skill bundles take precedence over individual skill commands — + # / loads multiple skills at once. Mirrors CLI dispatch. + _bundle_handled = False + try: + from agent.skill_bundles import ( + build_bundle_invocation_message, + resolve_bundle_command_key, + ) + bundle_key = resolve_bundle_command_key(command) + if bundle_key is not None: + user_instruction = event.get_command_args().strip() + # Pass the platform explicitly: bundle skill loading bypasses + # get_skill_commands()' scan-time disabled filter, and the gateway serves + # multiple platforms in one process, so env-var platform resolution can't be + # trusted here. + _bundle_plat = source.platform.value if source.platform else None + bundle_result = build_bundle_invocation_message( + bundle_key, user_instruction, task_id=_quick_key, + platform=_bundle_plat, + ) + if bundle_result: + msg, _loaded, missing = bundle_result + event.text = msg + _bundle_handled = True + if missing: + logger.info( + "Bundle %s skipped missing skills: %s", + bundle_key, ", ".join(missing), + ) + # Fall through to normal message processing with bundle content + except Exception as exc: + logger.warning("Bundle dispatch failed: %s", exc) + + if command and not locals().get("_bundle_handled", False): + try: + from agent.skill_commands import ( + get_skill_commands, + build_skill_invocation_message, + resolve_skill_command_key, + ) + skill_cmds = get_skill_commands() + cmd_key = resolve_skill_command_key(command) + if cmd_key is not None: + # Check per-platform disabled status before executing. get_skill_commands() only + # applies the *global* disabled list at scan time; per-platform overrides need + # checking here because the cache is process-global across platforms. + _skill_name = skill_cmds[cmd_key].get("name", "") + _plat = source.platform.value if source.platform else None + if _plat and _skill_name: + from agent.skill_utils import get_disabled_skill_names as _get_plat_disabled + if _skill_name in _get_plat_disabled(platform=_plat): + return ( + f"The **{_skill_name}** skill is disabled for {_plat}.\n" + f"Enable it with: `hermes skills config`" + ) + user_instruction = event.get_command_args().strip() + # Stacked slash-skill invocations: `/skill-a /skill-b do XYZ` loads every + # leading skill (up to 5), not just the first. Mirrors CLI. + try: + from agent.skill_commands import ( + build_stacked_skill_invocation_message as _build_stacked, + split_stacked_skill_commands, + ) + extra_keys, stacked_instruction = ( + split_stacked_skill_commands(user_instruction) + ) + except Exception: + _build_stacked = None + extra_keys, stacked_instruction = [], user_instruction + if extra_keys and _plat: + # split_stacked_skill_commands() only resolves that each extra token is a + # KNOWN skill command — like get_skill_commands() itself, it has no per- + # platform view. Re-check every stacked skill against the same disabled list, + # or a skill disabled for this platform still gets loaded via the stack. + from agent.skill_utils import get_disabled_skill_names as _get_plat_disabled + _plat_disabled = _get_plat_disabled(platform=_plat) + _disabled_extra = [ + skill_cmds.get(k, {}).get("name", "") + for k in extra_keys + if skill_cmds.get(k, {}).get("name", "") in _plat_disabled + ] + if _disabled_extra: + return ( + f"The **{', '.join(_disabled_extra)}** skill(s) in this " + f"stacked invocation are disabled for {_plat}.\n" + f"Enable them with: `hermes skills config`" + ) + if extra_keys and _build_stacked is not None: + stacked_result = _build_stacked( + [cmd_key, *extra_keys], + stacked_instruction, + task_id=_quick_key, + ) + if stacked_result: + msg, _loaded, _missing = stacked_result + event.text = msg + # Fall through to normal message processing + else: + return f"Failed to load stacked skills for /{command}." + else: + msg = build_skill_invocation_message( + cmd_key, user_instruction, task_id=_quick_key + ) + if msg: + event.text = msg + # Fall through to normal message processing with skill content + else: + # Not an active skill — check if it's a known-but-disabled or + # uninstalled skill and give actionable guidance. + _unavail_msg = _check_unavailable_skill(command) + if _unavail_msg: + return _unavail_msg + # Genuinely unrecognized /command (not built-in/plugin/skill/known-inactive): + # warn instead of forwarding to the LLM as free text (it invents tool calls). + # Normalize to hyphenated form first: the quick-command block may have set an + # alias target, so _cmd_def can be stale. + if command.replace("_", "-") not in GATEWAY_KNOWN_COMMANDS: + logger.warning( + "Unrecognized slash command /%s from %s — " + "replying with unknown-command notice", + command, + source.platform.value if source.platform else "?", + ) + return ( + f"Unknown command `/{command}`. " + f"Type /commands to see what's available, " + f"or resend without the leading slash to send " + f"as a regular message." + ) + except Exception as e: + logger.debug("Skill command check failed (non-fatal): %s", e) + return None + + async def _handle_message(self, event: MessageEvent) -> Optional[str]: + """Handle an incoming message from any platform. + + Pipeline: auth → command check → running-agent interrupt → get/create session → build + context → run agent → return response. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + _admitted = await self._hm_admit_event(event) + if _admitted is None: + return None + event, source, is_internal = _admitted + + _paused_notice = self._hm_estop_gate(event, source, is_internal) + if _paused_notice is not None: + return _paused_notice + + # Replies owned by in-flight work: pending /update prompt, clarify, slash-confirm. + _quick_key = self._session_key_for_source(source) + allow_gateway_control = event.allow_gateway_control + _update_reply = self._hm_update_prompt_reply(event, _quick_key, allow_gateway_control) + if _update_reply is not None: + return _update_reply + _clarify_reply = await self._hm_clarify_reply( + event, source, _quick_key, allow_gateway_control + ) + if _clarify_reply is not None: + return _clarify_reply + _confirm_reply = await self._hm_slash_confirm_reply( + event, _quick_key, allow_gateway_control + ) + if _confirm_reply is not None: + return _confirm_reply + + self._hm_evict_stale_running_agent(_quick_key) + if self._is_session_running(_quick_key): + return await self._hm_handle_running_session_message(event, source, _quick_key) + + # Idle path: resolve + dispatch slash commands; rewriting commands fall through to the agent. + _handled, _result, command, canonical = await self._hm_resolve_command( + event, source, _quick_key + ) + if _handled: + return _result + _handled, _result = await self._hm_dispatch_canonical_command( + event, source, _quick_key, canonical + ) + if _handled: + return _result + _handled, _result, command = await self._hm_dispatch_quick_and_plugin_commands( + event, source, command + ) + if _handled: + return _result + _skill_reply = self._hm_skill_slash_rewrite(event, source, _quick_key, command) + if _skill_reply is not None: + return _skill_reply + + # Pending exec approvals go through /approve and /deny only — no bare-text matching, or a + # conversational "yes" would execute a dangerous command. + + if not is_internal and await asyncio.to_thread( + self._is_telegram_topic_root_lobby, source + ): + # Debounce the lobby reminder so a user who forgets about + # topic mode and fires ten prompts doesn't get ten copies. + if self._should_send_telegram_lobby_reminder(source): + return self._telegram_topic_root_lobby_message() + return None + + # ── External-drain new-turn gate ───────────────────────────── + # When NAS engaged an external drain (.drain_request.json, seen by _drain_control_watcher), + # refuse to START new turns so the in-flight set can only fall to zero (stop accepting + # FIRST, then NAS polls active_agents==0). Internal/system events bypass; reversible. + if self._external_drain_active and not is_internal: + logger.info( + "Refusing new turn for session %s — external drain active.", + _quick_key, + ) + return ( + "⏳ This agent is draining for a maintenance action and isn't " + "accepting new turns right now. It'll be back in a moment — " + "please resend shortly." + ) + + # ── Claim this session before any await ─────────────────────── + # Many awaits sit between here and _run_agent registering the real AIAgent; without this + # sentinel a second message during any of them passes the "already running" guard and spins + # up a duplicate agent for the same session, corrupting the transcript. + _active_session_lease, _limit_message = self._claim_active_session_slot( + _quick_key, + source, + ) + if _limit_message is not None: + logger.info( + "Rejecting new active session %s: max_concurrent_sessions reached", + _quick_key, + ) + return _limit_message + + # ── FIFO orphan rescue ─────────────────────────────────────── + # A session that went idle with a populated overflow (post-turn drain never promoted, e.g. a + # compression-demoted follow-up) silently orphaned those events. Re-stage them FIFO and + # enqueue this event behind them. Skipped for control commands and internal events. + try: + _orphan_adapter = self._adapter_for_source(source) + if ( + _orphan_adapter is not None + and not bool(getattr(event, "internal", False)) + and not event.get_command() + ): + _rescued = self._rescue_orphaned_overflow( + _quick_key, _orphan_adapter + ) + if _rescued is not None: + # The oldest orphan runs as THIS turn. Park the incoming event behind the rest of + # the chain: into the slot when the chain was a single orphan (post-turn drain + # picks it up), otherwise into overflow behind the already-staged next orphan. + self._enqueue_fifo(_quick_key, event, _orphan_adapter) + event = _rescued + # Same session key by construction; carry the orphan's own source so reply + # anchors / thread metadata point at the message actually being answered. + _rescued_source = getattr(_rescued, "source", None) + if _rescued_source is not None: + source = _rescued_source + is_internal = bool(getattr(_rescued, "internal", False)) + except Exception: + logger.debug( + "FIFO orphan rescue pre-claim failed for %s", + _quick_key, + exc_info=True, + ) + + _claim_state = self._session_state(_quick_key) + if _active_session_lease is not None: + _claim_state.turn.lease = _active_session_lease + _claim_state.turn.agent = _AGENT_PENDING_SENTINEL + _claim_state.turn.started_ts = time.time() + self._persist_active_agents() + _run_generation = self._begin_session_run_generation(_quick_key) + + try: + try: + _agent_result = await self._handle_message_with_agent( + event, source, _quick_key, _run_generation + ) + except TurnLeaseTimeoutError as exc: + # A rejected message, not a completed turn: return before the /goal judge below so + # it cannot consume the resend notice and enqueue a synthetic continuation loop. + logger.error( + "Rejecting turn for routing key %s on session %s after " + "turn-lease timeout; transcript load was not started and " + "the user must resend", + _quick_key, + exc.session_id, + ) + return ( + "⏳ Another turn is still running on this session. To " + "protect the transcript, this message was not processed. " + "Wait for the active turn to finish, then resend it." + ) + try: + await self._run_post_turn_hooks( + agent_result=_agent_result, + source=source, + is_internal=is_internal, + event=event, + ) + except Exception as _goal_exc: + logger.debug("post-turn hook failed: %s", _goal_exc) + return _agent_result + finally: + # MoA one-shot restore must run on EVERY exit path: the restore data lives on the + # per-turn event, so a restore in the try block is skipped when the handler raises and + # the override leaks permanently; finally covers success, exception and interrupt. + self._restore_moa_one_shot(event, _quick_key) + self._restore_pending_one_turn_model_override(_quick_key) + # Normal completion/exception/interrupt clears this durable marker; SIGKILL/OOM skips + # finally, leaving it for the next unclean startup's recovery pass. + await self._clear_durable_active_turn(event) + # Unconditional release covers every exit path: _release_running_agent_state is idempotent + # and, without a run_generation guard, clears the slot whichever generation holds it. This + # evicts the zombie left when session_reset bumps the generation mid-flight (gen-N's + # guarded release in _run_agent returns False; a sentinel-only check would lock forever). + self._release_running_agent_state(_quick_key) + # Turn lease: release THIS turn's token — keyed by (routing key, run generation) so this + # unwind can only free the lease its own turn acquired, never a newer turn's. + self._release_turn_lease(_quick_key, _run_generation) + + def _restore_moa_one_shot(self, event: "MessageEvent", quick_key: str) -> None: + """Revert a ``/moa `` one-shot model override after its turn. + + Called from the message-handling ``finally`` so it fires on success, error or interrupt. + No-op unless ``event._moa_disable_after_turn``; ``_moa_restore_override`` holds the prior + per-session override (``None`` = clear the MoA override outright). + """ + if not getattr(event, "_moa_disable_after_turn", False): + return + try: + _restore = getattr(event, "_moa_restore_override", None) + self._session_state(quick_key).conversation.model_override = _restore + self._evict_cached_agent(quick_key) + except Exception: + pass + + def _restore_pending_one_turn_model_override(self, session_key: str) -> None: + """Restore a per-session model override after ``/model --once`` runs.""" + if not session_key: + return + try: + _otr_state = self._peek_session_state(session_key) + snapshot = _otr_state.conversation.one_turn_restore if _otr_state else None + if _otr_state is not None: + _otr_state.conversation.one_turn_restore = None + if not snapshot: + return + self._restore_session_model_override(session_key, snapshot) + except Exception: + logger.debug("Failed to restore one-turn model override", exc_info=True) + + def _prefix_inbound_sender_context(self, event: MessageEvent, source: SessionSource, message_text: str) -> str: + """Attribute the sender in shared multi-user sessions and prepend history-backfill channel context.""" + _group_sessions_per_user = getattr(self.config, "group_sessions_per_user", True) + _thread_sessions_per_user = getattr(self.config, "thread_sessions_per_user", False) + _is_shared_multi_user = is_shared_multi_user_session( + source, + group_sessions_per_user=_group_sessions_per_user, + thread_sessions_per_user=_thread_sessions_per_user, + ) + if _is_shared_multi_user and source.user_name: + # source.user_name is the platform display name — attacker-influenceable on any + # platform that lets participants set their own name. Neutralize newlines/control chars + # before interpolating it into every message, or a hostile name can masquerade as a + # fake markdown section (mirrors build_session_context_prompt's treatment). + _safe_user_name = neutralize_untrusted_inline_text(source.user_name) + # On Slack, expose the current author's verifiable user ID next to the display name: + # "mention me again" requests need a trusted `<@U...>` target for the CURRENT speaker — + # display names are ambiguous and historical mentions may point at someone else. The + # user_id comes from the Slack event envelope (not user-editable), so no neutralization. + if source.platform == Platform.SLACK and source.user_id: + _safe_user_name = ( + f"{_safe_user_name} | Slack user <@{source.user_id}>" + ) + message_text = f"[{_safe_user_name}] {message_text}" + + # Prepend history-backfill channel context after the sender-prefix so the prefix applies + # only to the trigger message, not the backfill block. + if getattr(event, "channel_context", None): + message_text = f"{event.channel_context}\n\n[New message]\n{message_text}" + return message_text + + @staticmethod + def _classify_inbound_media( + event: MessageEvent, pending_stt_prepared: bool + ) -> Tuple[list, list, list, list]: + """Split ``event.media_urls`` into (image, STT-voice, audio-file, video) paths.""" + from gateway.run import _event_media_is_audio, _event_media_is_image, _event_media_is_stt_input + image_paths: list[str] = [] + audio_paths: list[str] = [] + audio_file_paths: list[str] = [] + video_paths: list[str] = [] + + if event.media_urls: + for i, path in enumerate(event.media_urls): + mtype = event.media_types[i] if i < len(event.media_types) else "" + # Classify images per-attachment: trust this attachment's own MIME, and only honour + # the message-level PHOTO type when the per-attachment MIME is unknown. Otherwise a + # document sent alongside an image gets mis-routed as an image and the provider 400s. + if _event_media_is_image(event, i): + image_paths.append(path) + # MessageType.AUDIO = audio file attachment (e.g. .mp3, .m4a) — never STT. + # Mixed DOCUMENT events also preserve audio as a file path instead of + # dropping it or treating it as a voice note. + if _event_media_is_audio(event, i): + if event.message_type in {MessageType.AUDIO, MessageType.DOCUMENT}: + audio_file_paths.append(path) + elif not pending_stt_prepared and _event_media_is_stt_input(event, i): + audio_paths.append(path) + if mtype.startswith("video/") or (not mtype and event.message_type == MessageType.VIDEO): + video_paths.append(path) + return image_paths, audio_paths, audio_file_paths, video_paths + + async def _enrich_inbound_images( + self, source: SessionSource, session_key: str, message_text: str, image_paths: list[str] + ) -> str: + # Decide routing: native (attach pixels) vs text (vision_analyze pre-run + prepend + # description). See agent/image_routing.py. Offloaded to a thread: the decision does + # blocking network I/O (models.dev fetch on cache miss, Ollama /api/show probe) whose + # timeout would otherwise stall the whole gateway event loop. + _img_mode = await asyncio.to_thread( + self._decide_image_input_mode, + source=source, + session_key=session_key, + ) + if _img_mode == "native": + # Defer attachment to the run_conversation call site. + self._session_state( + session_key + ).persistent.native_image_paths = list(image_paths) + logger.info( + "Image routing: native (model supports vision). %d image(s) will be attached inline.", + len(image_paths), + ) + else: + logger.info( + "Image routing: text (mode=%s). Pre-analyzing %d image(s) via vision_analyze.", + _img_mode, len(image_paths), + ) + # Vision enrichment runs before AIAgent.run_conversation(), + # so bind this session's resolved runtime explicitly rather + # than consulting process-global compatibility mirrors. + vision_runtime = None + try: + turn_model, runtime_kwargs = self._resolve_session_agent_runtime( + source=source, + session_key=session_key, + ) + vision_runtime = dict(runtime_kwargs or {}) + vision_runtime["model"] = turn_model + except Exception: + logger.debug( + "vision enrichment: session runtime resolution failed", + exc_info=True, + ) + + from agent.auxiliary_client import scoped_runtime_main + + with scoped_runtime_main(vision_runtime): + message_text = await self._enrich_message_with_vision( + message_text, + image_paths, + ) + return message_text + + async def _enrich_inbound_voice( + self, event: MessageEvent, source: SessionSource, message_text: str, audio_paths: list[str] + ) -> str: + message_text, _successful_transcripts = await self._enrich_message_with_transcription( + message_text, + audio_paths, + ) + # Echo each successful transcript back to the user immediately when configured. Lets + # users verify STT quality in real-time, while allowing quiet STT for users who only + # want the agent to receive the transcription. + if _successful_transcripts and self._should_echo_stt_transcripts(): + _echo_adapter = self._adapter_for_source(source) + _echo_meta = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) + if _echo_adapter: + for _tx in _successful_transcripts: + try: + await _echo_adapter.send( + source.chat_id, + f'🎙️ "{_tx}"', + metadata=_echo_meta, + ) + except Exception as _echo_exc: + logger.debug( + "Transcript echo failed (non-fatal): %s", _echo_exc, + ) + # On transcription failure, do NOT send a hardcoded notice here: that bypassed the + # LLM and produced two replies (one pre-canned, TTS'd in the wrong language). + # Enrichment leaves a single neutral marker so the LLM gives one localized reply. + return message_text + + @staticmethod + def _prepend_inbound_media_file_notes(message_text: str, audio_file_paths: list[str], video_paths: list[str]) -> str: + if audio_file_paths: + from tools.credential_files import to_agent_visible_cache_path as _to_agent_path + for _apath in audio_file_paths: + _basename = os.path.basename(_apath) + _parts = _basename.split("_", 2) + _display = _parts[2] if len(_parts) >= 3 else _basename + _display = re.sub(r'[^\w.\- ]', '_', _display) + _agent_path = _to_agent_path(_apath) + _note = ( + f"[The user sent an audio file attachment: '{_display}'. " + f"It is saved at: {_agent_path}. " + f"Its content is not inlined here. If the user's request involves " + f"what the audio contains, transcribe or process it yourself — for " + f"example by passing the path to a transcription or media tool — " + f"instead of asking the user to describe it. Only ask what to do " + f"with it if their intent is genuinely unclear.]" + ) + message_text = f"{_note}\n\n{message_text}" + + if video_paths: + from tools.credential_files import to_agent_visible_cache_path as _to_agent_path + for _vpath in video_paths: + _basename = os.path.basename(_vpath) + _parts = _basename.split("_", 2) + _display = _parts[2] if len(_parts) >= 3 else _basename + _display = re.sub(r'[^\w.\- ]', '_', _display) + _agent_path = _to_agent_path(_vpath) + _note = ( + f"[The user sent a video attachment: '{_display}'. " + f"It is saved at: {_agent_path}. " + f"Its content is not inlined here. If the user's request involves " + f"what the video contains, inspect or process it yourself — for " + f"example by passing the path to a video analysis or media tool — " + f"instead of asking the user to describe it. Only ask what to do " + f"with it if their intent is genuinely unclear.]" + ) + message_text = f"{_note}\n\n{message_text}" + return message_text + + @staticmethod + def _prepend_inbound_document_notes(event: MessageEvent, message_text: str) -> str: + from gateway.run import ( + _build_document_context_note, + _event_media_is_audio, + _event_media_is_image, + _event_media_is_video, + ) + if event.media_urls: + import mimetypes as _mimetypes + from tools.credential_files import to_agent_visible_cache_path + + _TEXT_EXTENSIONS = {".txt", ".md", ".csv", ".log", ".json", ".xml", ".yaml", ".yml", ".toml", ".ini", ".cfg"} + for i, path in enumerate(event.media_urls): + # Per-attachment document handling: skip anything already routed as image/audio/video + # above; only genuine non-media files get a context note. A document mixed into a + # PHOTO/VOICE message (message-level type != DOCUMENT) thus still reaches the agent. + if ( + _event_media_is_image(event, i) + or _event_media_is_audio(event, i) + or _event_media_is_video(event, i) + ): + continue + mtype = event.media_types[i] if i < len(event.media_types) else "" + if mtype in {"", "application/octet-stream"}: + _ext = os.path.splitext(path)[1].lower() + if _ext in _TEXT_EXTENSIONS: + mtype = "text/plain" + else: + guessed, _ = _mimetypes.guess_type(path) + mtype = guessed or "application/octet-stream" + # Any accepted file gets a path-pointing context note — we accept + # all file types now, so a non-text/non-application MIME (font/*, + # model/*, etc.) must still tell the agent the file exists. + + basename = os.path.basename(path) + parts = basename.split("_", 2) + display_name = parts[2] if len(parts) >= 3 else basename + display_name = re.sub(r'[^\w.\- ]', '_', display_name) + + # Translate host cache path to in-container path if running under Docker backend. + # This ensures the agent receives a path it can open inside its sandbox, as the + # cache directories are auto-mounted at /root/.hermes/cache/* by get_cache_directory_mounts(). + agent_path = to_agent_visible_cache_path(path) + + inline_flags = getattr(event, "media_text_inlined", None) or [] + inline_flag = inline_flags[i] if i < len(inline_flags) else None + context_note = _build_document_context_note( + display_name, + agent_path, + mtype, + content_inlined=inline_flag is not False, + ) + message_text = f"{context_note}\n\n{message_text}" + return message_text + + @staticmethod + def _prepend_inbound_reply_context(event: MessageEvent, source: SessionSource, message_text: str) -> str: + # Discord: surface the triggering message id per-turn on the user message rather than in the + # cached system prompt. message_id changes every turn, so baking it into + # build_session_context_prompt() would bust the agent-cache signature and rebuild the + # AIAgent every message (destroying prompt caching). + if ( + source is not None + and getattr(source, "platform", None) == Platform.DISCORD + and getattr(event, "message_id", None) + ): + from gateway.session import _discord_tools_loaded as _disc_tools_loaded + if _disc_tools_loaded(): + message_text = ( + f"[Triggering message id: `{event.message_id}` — use as " + f"`message_id` for reply/react/pin via the discord tools.]\n\n" + f"{message_text}" + ) + + if getattr(event, "reply_to_text", None) and event.reply_to_message_id: + # Always inject the reply-to pointer — even when the quoted text already appears in + # history. The prefix isn't deduplication, it's disambiguation: it tells the agent + # *which* prior message the user is referencing. Token overhead is minimal. + reply_snippet = event.reply_to_text[:500] + if getattr(event, "reply_to_is_own_message", False): + message_text = ( + f'[Replying to your previous message: "{reply_snippet}"]\n\n' + f"{message_text}" + ) + else: + message_text = f'[Replying to: "{reply_snippet}"]\n\n{message_text}' + return message_text + + async def _expand_inbound_context_references( + self, source: SessionSource, session_key: str, message_text: str + ) -> Optional[str]: + """Expand ``@`` context references; returns None when the injection was refused (user notified).""" + from gateway.run import _load_gateway_config + try: + from agent.context_references import preprocess_context_references_async + from agent.model_metadata import get_model_context_length_async + + try: + from tools.terminal_scope import terminal_env as _ts_env + except ImportError: + _msg_cwd = os.environ.get("TERMINAL_CWD", os.path.expanduser("~")) + else: + _msg_cwd = _ts_env("TERMINAL_CWD", os.path.expanduser("~")) + _msg_config_ctx = None + _msg_cfg = None + _msg_model_cfg = {} + _msg_custom_providers = [] + try: + _msg_cfg = _load_gateway_config() + _msg_model_cfg = _msg_cfg.get("model", {}) + if isinstance(_msg_model_cfg, dict): + _msg_raw_ctx = _msg_model_cfg.get("context_length") + if _msg_raw_ctx is not None: + _msg_config_ctx = int(_msg_raw_ctx) + try: + from hermes_cli.config import get_compatible_custom_providers + + _msg_custom_providers = get_compatible_custom_providers(_msg_cfg) + except Exception: + _msg_custom_providers = _msg_cfg.get("custom_providers") or [] + except Exception: + pass + # Resolve the session's actual model/provider/base_url as the hygiene compression + # block does; GatewayRunner has no self._model/self._base_url (AttributeError, + # silently caught below). + _msg_model, _msg_runtime = self._resolve_session_agent_runtime( + source=source, + session_key=session_key, + user_config=_msg_cfg, + ) + _msg_base_url = _msg_runtime.get("base_url") or "" + # A global model.context_length belongs to the configured + # model, not a session /model or channel override. Prefer a + # matching per-custom-provider model limit when available. + _msg_configured_model = ( + _msg_model_cfg.get("default") or _msg_model_cfg.get("model") + if isinstance(_msg_model_cfg, dict) + else _msg_model_cfg + ) + if _msg_model != _msg_configured_model: + _msg_config_ctx = None + if _msg_config_ctx is not None and isinstance(_msg_model_cfg, dict): + try: + from hermes_cli.route_identity import should_clear_context_pin_async + + if await should_clear_context_pin_async( + None, # model match already checked above + None, + _msg_model_cfg.get("base_url"), + _msg_base_url, + _msg_model_cfg.get("provider"), + _msg_runtime.get("provider"), + ): + _msg_config_ctx = None + except Exception: + _msg_config_ctx = None + if _msg_custom_providers and _msg_base_url: + try: + from hermes_cli.config import get_custom_provider_context_length + + _msg_custom_ctx = get_custom_provider_context_length( + model=_msg_model, + base_url=_msg_base_url, + custom_providers=_msg_custom_providers, + ) + if _msg_custom_ctx: + _msg_config_ctx = _msg_custom_ctx + except Exception: + pass + _msg_ctx_len = await get_model_context_length_async( + _msg_model, + base_url=_msg_base_url, + api_key=_msg_runtime.get("api_key") or "", + config_context_length=_msg_config_ctx, + provider=_msg_runtime.get("provider") or "", + custom_providers=_msg_custom_providers, + ) + _ctx_result = await preprocess_context_references_async( + message_text, + cwd=_msg_cwd, + context_length=_msg_ctx_len, + allowed_root=_msg_cwd, + ) + if _ctx_result.blocked: + _adapter = self._adapter_for_source(source) + if _adapter: + await _adapter.send( + source.chat_id, + "\n".join(_ctx_result.warnings) or "Context injection refused.", + ) + return None + if _ctx_result.expanded: + message_text = _ctx_result.message + except Exception as exc: + logger.warning("@ context reference expansion failed: %s", exc) + logger.debug("@ context reference expansion failure detail", exc_info=True) + return message_text + + async def _prepare_inbound_message_text( + self, + *, + event: MessageEvent, + source: SessionSource, + history: List[Dict[str, Any]], + session_key: Optional[str] = None, + ) -> Optional[str]: + """Prepare inbound event text for the agent. + + Shared by the normal inbound and queued follow-up paths so attribution, image enrichment, + STT, document notes, reply context and @ references behave the same. Side effect: buffers + per-session native image paths when the model supports native vision; the caller consumes + that buffer at ``run_conversation``. Empty list means the text vision path already ran. + """ + history = history or [] + _pending_stt_prepared = hasattr(event, "_gateway_pending_stt_text") + message_text = ( + getattr(event, "_gateway_pending_stt_text", None) + if _pending_stt_prepared + else event.text + ) or "" + # Prefer the caller's resolved session key so this write key matches the consume key at the + # run_conversation site; derive it here only for tests and legacy standalone callers. + session_key = session_key or self._session_key_for_source(source) + # Reset only this session's per-call buffer; other sessions may be + # concurrently preparing multimodal turns on the same runner. + self._consume_pending_native_image_paths(session_key) + + message_text = self._prefix_inbound_sender_context(event, source, message_text) + image_paths, audio_paths, audio_file_paths, video_paths = self._classify_inbound_media( + event, _pending_stt_prepared + ) + if image_paths: + message_text = await self._enrich_inbound_images(source, session_key, message_text, image_paths) + if audio_paths: + message_text = await self._enrich_inbound_voice(event, source, message_text, audio_paths) + message_text = self._prepend_inbound_media_file_notes(message_text, audio_file_paths, video_paths) + message_text = self._prepend_inbound_document_notes(event, message_text) + message_text = self._prepend_inbound_reply_context(event, source, message_text) + if "@" in message_text: + message_text = await self._expand_inbound_context_references(source, session_key, message_text) + if message_text is None: + return None + + return message_text + + async def _prepare_profile_scoped_inbound_message_text( + self, + *, + event: MessageEvent, + source: SessionSource, + history: List[Dict[str, Any]], + session_key: Optional[str] = None, + ) -> Optional[str]: + """Run inbound preprocessing under the routed profile when multiplexed.""" + from gateway.run import _async_profile_runtime_scope + if getattr(getattr(self, "config", None), "multiplex_profiles", False): + async with _async_profile_runtime_scope( + self._resolve_profile_home_for_source(source) + ): + return await self._prepare_inbound_message_text( + event=event, + source=source, + history=history, + session_key=session_key, + ) + return await self._prepare_inbound_message_text( + event=event, + source=source, + history=history, + session_key=session_key, + ) + + async def _prepare_clarify_reply_text(self, event) -> str: + """Return raw text or successful voice transcripts for a clarify reply.""" + if not self._pending_event_audio_paths(event): + return (event.text or "").strip() + + _, successful_transcripts = await self._transcribe_pending_audio_event_once( + event, "", + ) + return "\n\n".join( + transcript.strip() + for transcript in successful_transcripts + if transcript.strip() + ) + + def _consume_pending_native_image_paths(self, session_key: str) -> List[str]: + state = self._peek_session_state(session_key) + if state is None or not state.persistent.native_image_paths: + return [] + paths = list(state.persistent.native_image_paths) + state.persistent.native_image_paths = [] + return paths + + async def _mark_durable_active_turn( + self, + event: "MessageEvent", + session_key: str, + ) -> bool: + """Persist the exact resolved routing key for this running turn.""" + try: + token = await self.async_session_store.mark_turn_active(session_key) + except Exception as exc: + logger.warning( + "Could not persist active-turn marker for %s: %s", + session_key, + exc, + ) + return False + if not token: + return False + # Private event attributes are process-local ownership state. Keep the + # token out of public metadata, transcripts, and platform payloads. + setattr(event, "_gateway_active_turn_session_key", session_key) + setattr(event, "_gateway_active_turn_token", token) + return True + + async def _clear_durable_active_turn(self, event: "MessageEvent") -> bool: + """Best-effort CAS clear of the marker owned by *event*.""" + session_key = getattr(event, "_gateway_active_turn_session_key", None) + token = getattr(event, "_gateway_active_turn_token", None) + try: + if not session_key or not token: + return False + last_error: Optional[Exception] = None + for attempt in range(1, 4): + try: + return bool( + await self.async_session_store.clear_turn_active( + session_key, token + ) + ) + except Exception as exc: + last_error = exc + if attempt < 3: + logger.debug( + "Retrying active-turn marker cleanup for %s (%d/3): %s", + session_key, + attempt, + exc, + ) + # Never let marker cleanup block agent/lease release; a stale marker is bounded by the + # agent timeout and the clean-start orphan-marker discard path. + logger.warning( + "Could not clear active-turn marker for %s after 3 attempts: %s", + session_key, + last_error, + ) + return False + finally: + for attr in ( + "_gateway_active_turn_session_key", + "_gateway_active_turn_token", + ): + with suppress(AttributeError): + delattr(event, attr) + + def _install_plugin_message_injector(self) -> None: + """Publish this live gateway's plugin message scheduler.""" + from hermes_cli.plugins import get_plugin_manager + + get_plugin_manager().set_gateway_message_injector( + self, + self._schedule_plugin_message_injection, + ) + + def _clear_plugin_message_injector(self) -> None: + """Remove this runner's scheduler without clobbering a newer owner.""" + from hermes_cli.plugins import get_plugin_manager + + get_plugin_manager().clear_gateway_message_injector(self) + + def _schedule_plugin_message_injection( + self, + *, + session_key: str, + content: str, + plugin_id: str, + ) -> bool: + """Schedule a plugin-triggered turn on the live gateway loop.""" + from gateway.run import safe_schedule_threadsafe + loop = getattr(self, "_gateway_loop", None) + if not getattr(self, "_running", False) or loop is None or loop.is_closed(): + return False + + coro = self._dispatch_plugin_message_injection( + session_key=session_key, + content=content, + plugin_id=plugin_id, + ) + try: + current_loop = asyncio.get_running_loop() + except RuntimeError: + current_loop = None + + if current_loop is loop: + try: + future = loop.create_task(coro) + except Exception: + coro.close() + logger.warning( + "Plugin message injection scheduling failed", + exc_info=True, + ) + return False + self._background_tasks.add(future) + future.add_done_callback(self._background_tasks.discard) + else: + future = safe_schedule_threadsafe( + coro, + loop, + logger=logger, + log_message="Plugin message injection scheduling failed", + log_level=logging.WARNING, + ) + if future is None: + return False + + def _log_result(completed) -> None: + try: + accepted = completed.result() + except (asyncio.CancelledError, concurrent.futures.CancelledError): + return + except Exception: + logger.warning( + "Plugin message injection failed: plugin=%s session=%s", + plugin_id, + session_key, + exc_info=True, + ) + return + if not accepted: + logger.warning( + "Plugin message injection was not routed: plugin=%s session=%s", + plugin_id, + session_key, + ) + + future.add_done_callback(_log_result) + return True + + async def _dispatch_plugin_message_injection( + self, + *, + session_key: str, + content: str, + plugin_id: str, + ) -> bool: + """Route a plugin-triggered turn through the session's live adapter.""" + if not getattr(self, "_running", False) or getattr(self, "_draining", False): + return False + + entry = await self.async_session_store.lookup_by_session_key(session_key) + if entry is None or entry.origin is None: + return False + if not getattr(self, "_running", False) or getattr(self, "_draining", False): + return False + + source = dataclasses.replace(entry.origin) + try: + if not self._is_user_authorized( + source, + allow_adapter_delegation=False, + ): + logger.warning( + "Plugin message injection denied by current gateway authorization: " + "plugin=%s session=%s", + plugin_id, + session_key, + ) + return False + except Exception: + logger.warning( + "Plugin message injection authorization check failed: " + "plugin=%s session=%s", + plugin_id, + session_key, + exc_info=True, + ) + return False + + adapter = self._adapter_for_source(source) + if adapter is None: + return False + + event = MessageEvent( + text=content, + message_type=MessageType.TEXT, + source=source, + internal=True, + allow_gateway_control=False, + metadata={ + "hermes_plugin_id": plugin_id, + "hermes_plugin_injection": True, + "gateway_session_key": session_key, + "gateway_session_id": entry.session_id, + "gateway_session_strict": True, + }, + ) + await adapter.handle_message(event) + logger.info( + "Plugin message injection dispatched: plugin=%s session=%s session_id=%s", + plugin_id, + session_key, + entry.session_id, + ) + return True + + def _decide_image_input_mode( + self, + *, + source: Optional[SessionSource] = None, + session_key: Optional[str] = None, + user_config: Optional[dict] = None, + provider: Optional[str] = None, + model: Optional[str] = None, + ) -> str: + """Resolve image-input routing for the effective model this turn. + + Returns ``"native"`` (attach pixels on the user turn) or ``"text"`` (pre-analyze with + vision_analyze and prepend the description); see agent/image_routing.py. Gateway sessions + can carry /model overrides and image preprocessing runs before AIAgent sets the + auxiliary_client runtime globals, so resolve the per-session runtime bundle the upcoming + turn will use, not just the persisted default. + """ + try: + from agent.image_routing import decide_image_input_mode + from agent.auxiliary_client import _read_main_model, _read_main_provider + from hermes_cli.config import load_config + + cfg = user_config if isinstance(user_config, dict) else load_config() + resolved_provider = (provider or "").strip() + resolved_model = (model or "").strip() + resolved_requested_provider = "" + + needs_session_runtime = not resolved_provider or not resolved_model + has_session_identity = source is not None or session_key + if needs_session_runtime and has_session_identity: + try: + turn_model, runtime_kwargs = self._resolve_session_agent_runtime( + source=source, + session_key=session_key, + user_config=cfg, + ) + if not resolved_model and isinstance(turn_model, str): + resolved_model = turn_model.strip() + runtime_provider = runtime_kwargs.get("provider") if isinstance(runtime_kwargs, dict) else None + runtime_requested_provider = ( + runtime_kwargs.get("requested_provider") + if isinstance(runtime_kwargs, dict) + else None + ) + if not resolved_provider and isinstance(runtime_provider, str): + resolved_provider = runtime_provider.strip() + if isinstance(runtime_requested_provider, str): + resolved_requested_provider = runtime_requested_provider.strip() + except Exception as exc: + logger.debug( + "image_routing: session runtime resolution failed, falling back to config — %s", + exc, + ) + + if not resolved_provider: + resolved_provider = _read_main_provider() + if not resolved_model: + resolved_model = _read_main_model() + + return decide_image_input_mode( + resolved_provider, + resolved_model, + cfg, + requested_provider=resolved_requested_provider, + ) + except Exception as exc: + logger.debug("image_routing: decision failed, falling back to text — %s", exc) + return "text" + + async def _enrich_message_with_vision( + self, + user_text: str, + image_paths: List[str], + ) -> str: + """Auto-analyze user-attached images with the vision tool and prepend the descriptions to + the message text. + + Description *and* local cache path are injected so the model understands the image without + a tool call and can re-examine it with vision_analyze. Returns the enriched message string. + """ + from tools.vision_tools import vision_analyze_tool + from agent.memory_manager import sanitize_context + + analysis_prompt = ( + "Concisely describe this image in 2-4 sentences " + "(~200 Chinese characters or ~150 English words). " + "Cover the main subject, key visible text/data/code, and overall context. " + "If it is a chart, diagram, or scientific figure, include the important " + "labels, legend, and key values. Skip decorative details." + ) + + enriched_parts = [] + for path in image_paths: + try: + logger.debug("Auto-analyzing user image: %s", path) + result_json = await vision_analyze_tool( + image_url=path, + user_prompt=analysis_prompt, + ) + result = json.loads(result_json) + if result.get("success"): + description = result.get("analysis", "") + description = sanitize_context(description) + enriched_parts.append( + f"[The user sent an image~ Here's what I can see:\n{description}]\n" + f"[If you need a closer look, use vision_analyze with " + f"image_url: {path} ~]" + ) + else: + enriched_parts.append( + "[The user sent an image but I couldn't quite see it " + "this time (>_<) You can try looking at it yourself " + f"with vision_analyze using image_url: {path}]" + ) + except Exception as e: + logger.error("Vision auto-analysis error: %s", e) + enriched_parts.append( + f"[The user sent an image but something went wrong when I " + f"tried to look at it~ You can try examining it yourself " + f"with vision_analyze using image_url: {path}]" + ) + + # Combine: vision descriptions first, then the user's original text + if enriched_parts: + prefix = "\n\n".join(enriched_parts) + if user_text: + return f"{prefix}\n\n{user_text}" + return prefix + return user_text + + async def _enrich_message_with_transcription( + self, + user_text: str, + audio_paths: List[str], + ) -> tuple[str, List[str]]: + """Auto-transcribe user voice/audio messages using the configured STT provider and prepend + the transcript to the message text. + + Returns ``(enriched_text, successful_transcripts)``: the message with transcription + wrappers prepended, and the raw transcripts of successfully transcribed clips in input + order (empty if every clip failed or STT is disabled) so callers can echo them back to + the user before the agent loop. + """ + from gateway.run import _probe_audio_duration + seen = set() + audio_paths = [p for p in audio_paths if p not in seen and not seen.add(p)] + if not getattr(self.config, "stt_enabled", True): + notes = [] + for path in audio_paths: + abs_path = os.path.abspath(path) + duration_str = await _probe_audio_duration(abs_path) + if duration_str: + notes.append( + f"[The user sent a voice message: {abs_path} (duration: {duration_str})]" + ) + else: + notes.append(f"[The user sent a voice message: {abs_path}]") + if not notes: + return user_text, [] + prefix = "\n\n".join(notes) + _placeholder = "(The user sent a message with no text content)" + if user_text and user_text.strip() == _placeholder: + return prefix, [] + if user_text: + return f"{prefix}\n\n{user_text}", [] + return prefix, [] + + try: + from tools.transcription_tools import ( + transcribe_audio, + transcribe_audio_local_fallback, + ) + except ModuleNotFoundError as e: + logger.error("Transcription module unavailable: %s", e) + unavailable_note = "[voice message could not be transcribed]" + _placeholder = "(The user sent a message with no text content)" + if user_text and user_text.strip() == _placeholder: + return unavailable_note, [] + if user_text: + return f"{unavailable_note}\n\n{user_text}", [] + return unavailable_note, [] + + enriched_parts = [] + successful_transcripts: List[str] = [] + for path in audio_paths: + try: + logger.debug("Transcribing user voice: %s", path) + result = await asyncio.to_thread( + transcribe_audio, path, None, "gateway", + ) + if not result.get("success"): + fallback = await asyncio.to_thread( + transcribe_audio_local_fallback, + path, + ) + if fallback.get("success"): + logger.info( + "Configured STT failed for %s; recovered with local STT", + path, + ) + result = fallback + if result["success"]: + transcript = result["transcript"] + # STT may return success=True with an empty/whitespace transcript (silence, cut-off); + # empty quotes make the agent reply to nothing and can loop, so emit a sentinel note. + if not (transcript or "").strip(): + enriched_parts.append( + "[The user sent a voice message but it came through " + "empty or inaudible — speech-to-text returned no " + "words. Do not guess at the content; ask the user " + "to resend or type it out.]" + ) + continue + successful_transcripts.append(transcript) + # Pass the transcript as a plain quoted line: a "The user sent a voice message..." + # wrapper read as a meta-instruction and made the LLM comment on voice mode instead. + enriched_parts.append(f'"{transcript}"') + else: + error = result.get("error", "unknown error") + # All failure branches: one minimal neutral marker. Never mention "no STT provider", + # setup steps, or a DM sent — persisted in history they poison later turns (the model + # keeps volunteering STT-setup advice). Cause is logged for operators, not the prompt. + logger.info("Voice transcription failed for %s: %s", path, error) + from tools.credential_files import to_agent_visible_cache_path + + agent_path = to_agent_visible_cache_path(os.path.abspath(path)) + enriched_parts.append( + "[voice message could not be transcribed automatically; " + f"the audio is available at: {agent_path}]" + ) + except Exception as e: + logger.error("Transcription error: %s", e) + from tools.credential_files import to_agent_visible_cache_path + + agent_path = to_agent_visible_cache_path(os.path.abspath(path)) + enriched_parts.append( + "[voice message could not be transcribed automatically; " + f"the audio is available at: {agent_path}]" + ) + + if enriched_parts: + prefix = "\n\n".join(enriched_parts) + # Strip the empty-content placeholder from the Discord adapter + # when we successfully transcribed the audio — it's redundant. + _placeholder = "(The user sent a message with no text content)" + if user_text and user_text.strip() == _placeholder: + return prefix, successful_transcripts + if user_text: + return f"{prefix}\n\n{user_text}", successful_transcripts + return prefix, successful_transcripts + return user_text, successful_transcripts + + def _pending_event_audio_paths(self, event) -> List[str]: + """Return STT-eligible paths from a pending voice message.""" + from gateway.run import _event_media_is_stt_input + audio_paths: List[str] = [] + media_urls = getattr(event, "media_urls", None) or [] + for i, path in enumerate(media_urls): + if _event_media_is_stt_input(event, i): + audio_paths.append(path) + return audio_paths + + async def _transcribe_pending_audio_event_once( + self, + event, + user_text: Optional[str] = None, + ) -> tuple[str | None, List[str]]: + """Transcribe a pending audio event once and cache the result on the event. + + The interrupt monitor and the pending-drain path both need the transcript; caching keeps + it to one STT call and one transcript echo per platform message. + """ + if hasattr(event, "_gateway_pending_stt_text"): + cached_text = getattr(event, "_gateway_pending_stt_text") + cached_transcripts = getattr(event, "_gateway_pending_stt_transcripts", []) or [] + return cached_text, list(cached_transcripts) + + audio_paths = self._pending_event_audio_paths(event) + if not audio_paths: + return user_text if user_text is not None else (getattr(event, "text", None) or None), [] + + text = user_text if user_text is not None else (getattr(event, "text", "") or "") + enriched_text, successful_transcripts = await self._enrich_message_with_transcription( + text, + audio_paths, + ) + setattr(event, "_gateway_pending_stt_text", enriched_text) + setattr(event, "_gateway_pending_stt_transcripts", list(successful_transcripts)) + return enriched_text, successful_transcripts + + async def _echo_pending_stt_transcripts_once( + self, + event, + adapter, + source, + transcripts: List[str], + *, + metadata=None, + log_context: str = "Transcript", + ) -> None: + """Echo pending-event STT transcripts to the chat at most once. + + Tracked as a COUNT (not a set — identical transcripts are distinct deliveries): + ``merge_pending_message_event`` can append a second voice note and invalidate the cache, + and the re-run returns earlier transcripts as a prefix, so only the unsent tail is echoed. + """ + if ( + not transcripts + or not self._should_echo_stt_transcripts() + or adapter is None + ): + return + already_echoed = int(getattr(event, "_gateway_pending_stt_echoed", 0) or 0) + unsent = transcripts[already_echoed:] + setattr(event, "_gateway_pending_stt_echoed", already_echoed + len(unsent)) + for tx in unsent: + try: + await adapter.send( + source.chat_id, + f'🎙️ "{tx}"', + metadata=metadata, + ) + except Exception as echo_exc: + logger.debug("%s echo failed (non-fatal): %s", log_context, echo_exc) + + async def _transcribe_and_echo_pending_voice( + self, + event, + adapter, + source, + text: str, + *, + log_context: str, + metadata=_UNSET, + ) -> tuple[str, List[str]]: + """Transcribe a pending voice event and echo transcripts once. + + Returns ``(enriched_text, transcripts)`` for ``agent.interrupt()`` or the pending-drain + flow; ``(text, [])`` unchanged when there is no STT-eligible media (caller owns the + ``_build_media_placeholder`` fallback for empty ``text`` with non-audio media). + """ + if not self._pending_event_audio_paths(event): + return text, [] + try: + enriched_text, transcripts = await self._transcribe_pending_audio_event_once( + event, + text, + ) + echo_meta = self._thread_metadata_for_source( + source, + self._reply_anchor_for_event(event), + ) if metadata is _UNSET else metadata + await self._echo_pending_stt_transcripts_once( + event, + adapter, + source, + transcripts, + metadata=echo_meta, + log_context=log_context, + ) + return enriched_text or text, transcripts + except Exception as trans_exc: + logger.warning("%s transcription failed: %s", log_context, trans_exc) + return text, [] diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py new file mode 100644 index 0000000000..a20a93a08c --- /dev/null +++ b/gateway/run_notifications.py @@ -0,0 +1,2098 @@ +"""Process/completion/update notifications, media delivery and async-delegation delivery methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import json +import time +from contextlib import suppress +from gateway.config import Platform, _BUILTIN_PLATFORM_VALUES +from gateway.platforms.base import MessageEvent, MessageType +from gateway.session import SessionEntry, SessionSource +from pathlib import Path +from typing import Any, Dict, Optional, cast + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayNotificationsMixin: + """Process/completion/update notifications, media delivery and async-delegation delivery methods for GatewayRunner.""" + + async def _deliver_platform_notice(self, source, content: str) -> None: + """Deliver a setup/operational notice using platform-specific privacy rules.""" + from gateway.run import _is_slack_ignored_channel + adapter = self._adapter_for_source(source) + if not adapter: + return + + config = getattr(self, "config", None) + if ( + config + and getattr(source, "platform", None) == Platform.SLACK + and _is_slack_ignored_channel(config, getattr(source, "chat_id", None)) + ): + logger.info( + "Skipping Slack platform notice for configured ignored channel %s", + getattr(source, "chat_id", None), + ) + return + + notice_delivery = "public" + if config and hasattr(config, "get_notice_delivery"): + notice_delivery = config.get_notice_delivery(source.platform) + + metadata = self._thread_metadata_for_source(source) + if notice_delivery == "private" and getattr(source, "user_id", None): + try: + result = await adapter.send_private_notice( + source.chat_id, + source.user_id, + content, + metadata=metadata, + ) + if getattr(result, "success", False): + return + except Exception: + logger.debug( + "[%s] send_private_notice failed, falling back to public", + getattr(source, "platform", "?"), + exc_info=True, + ) + + await adapter.send(source.chat_id, content, metadata=metadata) + + async def _resolve_async_delegation_session( + self, + session_entry: SessionEntry, + pinned_session_id: str, + ) -> Optional[SessionEntry]: + """Resolve an async completion to its verified owning gateway session. + + Follow compression-rotation lineage (parent row ended, child continues), but never let a + late completion override an unrelated /new or restored route. Unknown ownership fails + closed; the result stays in the delegation records. + """ + from gateway.run import _USER_BOUNDARY_END_REASONS + session_db = cast(Any, self._session_db) + if session_db is None: + logger.warning( + "Async-delegation completion has no session database; " + "dropping injection (#55578 fail-closed)." + ) + return None + + pinned_row = None + try: + pinned_row = await session_db.get_session(pinned_session_id) + except Exception: + logger.debug( + "Async-delegation parent lookup failed for %s", + pinned_session_id, + exc_info=True, + ) + + if pinned_row is None: + logger.warning( + "Async-delegation completion has unknown spawning session %s; " + "dropping injection (#55578 fail-closed).", + pinned_session_id, + ) + return None + + target_session_id = pinned_session_id + follows_compression = False + if pinned_row.get("ended_at"): + _end_reason = str(pinned_row.get("end_reason") or "") + if _end_reason in _USER_BOUNDARY_END_REASONS: + logger.warning( + "Async-delegation completion pinned to user-closed session %s " + "(end_reason=%r); dropping injection instead of resurrecting it " + "(#55578 fail-closed).", + pinned_session_id, + _end_reason, + ) + return None + if _end_reason != "compression": + # Idle/timeout/lifecycle end (scale-to-zero norm): the chat route is still valid and + # ``session_entry`` is its current session, so deliver here rather than drop — otherwise + # the row is acked at adapter acceptance then silently lost. + logger.info( + "Async-delegation completion pinned to %s-ended session %s; " + "retargeting to the chat's current session %s.", + _end_reason or "idle", + pinned_session_id, + session_entry.session_id, + ) + return session_entry + + follows_compression = True + try: + target_session_id = await session_db.get_compression_tip( + pinned_session_id + ) + except Exception: + logger.debug( + "Async-delegation compression-tip lookup failed for %s", + pinned_session_id, + exc_info=True, + ) + target_session_id = None + + if not target_session_id or target_session_id == pinned_session_id: + logger.warning( + "Async-delegation completion pinned to compressed session %s " + "without a continuation; dropping injection.", + pinned_session_id, + ) + return None + + try: + tip_row = await session_db.get_session(target_session_id) + except Exception: + tip_row = None + if tip_row is None or tip_row.get("ended_at"): + logger.warning( + "Async-delegation compression continuation %s is %s; " + "dropping injection.", + target_session_id, + "unknown" if tip_row is None else "ended", + ) + return None + + route_owns_lineage = session_entry.session_id in { + pinned_session_id, + target_session_id, + } + if not route_owns_lineage: + # A long-running delegation may survive multiple compression + # rotations. Accept an intermediate stale route only when its + # own verified compression tip is the same live target. + try: + route_row = await session_db.get_session(session_entry.session_id) + route_tip = ( + await session_db.get_compression_tip(session_entry.session_id) + if route_row is not None + and route_row.get("ended_at") + and route_row.get("end_reason") == "compression" + else None + ) + except Exception: + route_tip = None + route_owns_lineage = route_tip == target_session_id + + if not route_owns_lineage: + logger.warning( + "Async-delegation completion for compression lineage %s -> %s " + "does not own current route %s; dropping injection.", + pinned_session_id, + target_session_id, + session_entry.session_id, + ) + return None + + if target_session_id == session_entry.session_id: + return session_entry + + prior_session_id = session_entry.session_id + if follows_compression: + switched = await self.async_session_store.advance_compression_session( + session_entry.session_key, + prior_session_id, + target_session_id, + ) + else: + switched = await self.async_session_store.switch_session( + session_entry.session_key, + target_session_id, + ) + if switched is None: + logger.warning( + "Async-delegation completion could not bind routing key %s to " + "owning session %s; dropping injection.", + session_entry.session_key, + target_session_id, + ) + return None + + logger.info( + "Pinned async-delegation completion to owning session %s " + "(was %s) for routing key %s (#57498)", + target_session_id, + prior_session_id, + session_entry.session_key, + ) + return switched + + async def _deliver_media_from_response( + self, + response: str, + event: MessageEvent, + adapter, + thread_metadata: Optional[Dict[str, Any]] = None, + ) -> None: + """Extract explicit MEDIA: tags from an already-streamed response and deliver them. + + The text is already delivered; this only handles file attachments the normal + _process_message_background path would have caught. Unlike the non-streaming path in + ``gateway/platforms/base.py`` this rescan is EXPLICIT-ONLY: a bare local path in a + streamed reply was either shown as text or is stale inspected content, and promoting it + sent files the model never asked to deliver. + """ + from pathlib import Path + from urllib.parse import quote as _quote + + try: + # Capture [[as_document]] before extract_media strips it, so the dispatch partition + # below can route image-extension files through send_document (preserving bytes) instead + # of send_multiple_images (Telegram sendPhoto recompresses to ~1280px). + force_document_attachments = "[[as_document]]" in response + + from gateway.platforms.base import BasePlatformAdapter, should_send_media_as_audio + + media_files, cleaned = adapter.extract_media(response) + media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files) + # Do NOT deduplicate explicit MEDIA tags against prior turns here. This rescan is + # already EXPLICIT-ONLY (see docstring): a MEDIA: directive in the final streamed reply + # is the model deliberately attaching a file — including a user-requested resend. Stale + # auto-appended tags are deduped upstream (_collect_auto_append_media_tags). Strip image + # URLs for parity with the non-streaming chain, but do NOT run extract_local_files here. + adapter.extract_images(cleaned) + + _thread_meta = ( + dict(thread_metadata) + if thread_metadata is not None + else self._thread_metadata_for_source( + event.source, + self._reply_anchor_for_event(event), + ) + ) + + _VIDEO_EXTS = {'.mp4', '.mov', '.avi', '.mkv', '.webm', '.3gp'} + _IMAGE_EXTS = {'.jpg', '.jpeg', '.png', '.webp', '.gif'} + + # Partition out images so they can be sent as a single batch (e.g. Signal's multi- + # attachment RPC). When [[as_document]] was set, image-extension files skip the photo + # path and route to send_document below — preserving original bytes. + image_paths: list = [] + non_image_media: list = [] + for media_path, is_voice in media_files: + ext = Path(media_path).suffix.lower() + if (ext in _IMAGE_EXTS + and not is_voice + and not force_document_attachments): + image_paths.append(media_path) + else: + non_image_media.append((media_path, is_voice)) + + if image_paths: + try: + images = [(f"file://{_quote(p)}", "") for p in image_paths] + await adapter.send_multiple_images( + chat_id=event.source.chat_id, + images=images, + metadata=_thread_meta, + ) + except Exception as e: + logger.warning("[%s] Post-stream image batch delivery failed: %s", adapter.name, e) + + for media_path, is_voice in non_image_media: + try: + ext = Path(media_path).suffix.lower() + if should_send_media_as_audio(event.source.platform, ext, is_voice=is_voice): + await adapter.send_voice( + chat_id=event.source.chat_id, + audio_path=media_path, + metadata=_thread_meta, + is_voice=is_voice, + ) + elif ext in _VIDEO_EXTS: + await adapter.send_video( + chat_id=event.source.chat_id, + video_path=media_path, + metadata=_thread_meta, + ) + else: + await adapter.send_document( + chat_id=event.source.chat_id, + file_path=media_path, + metadata=_thread_meta, + ) + except Exception as e: + logger.warning("[%s] Post-stream media delivery failed: %s", adapter.name, e) + + except Exception as e: + logger.warning("Post-stream media extraction failed: %s", e) + + async def _deliver_queued_first_response( + self, + response: str, + source: SessionSource, + adapter, + metadata: Optional[Dict[str, Any]] = None, + event_message_id: Optional[str] = None, + text_already_delivered: bool = False, + deliver_media: bool = True, + stream_consumer=None, + ) -> None: + """Deliver a queued response using the normal text+attachment split.""" + from gateway.run import _strip_response_attachments_for_direct_send + if not text_already_delivered: + text_content = _strip_response_attachments_for_direct_send(response, adapter) + if text_content: + # Reconcile-by-edit first (live finding, 2026-08-16 canary): when the stream + # consumer delivered/sealed a message but its recorded payload didn't confirm the + # final (post-stream mutation), plain-sending here creates the duplicate — the + # sealed message already carries most of the answer. + _reconciled = False + _sc_msg_id = getattr(stream_consumer, "message_id", None) + if ( + _sc_msg_id + and _sc_msg_id != "__no_edit__" + and not getattr(stream_consumer, "_turn_split_delivery", False) + ): + try: + _edit_res = await adapter.edit_message( + chat_id=source.chat_id, + message_id=_sc_msg_id, + content=text_content, + finalize=True, + ) + if getattr(_edit_res, "success", False): + _reconciled = True + logger.info( + "Queued-lane final reconciled by editing message %s in place (no duplicate send).", + _sc_msg_id, + ) + except Exception as _qe: + logger.debug( + "Queued-lane reconcile edit failed (%s); falling back to send.", + _qe, + ) + if not _reconciled: + await adapter.send( + source.chat_id, + text_content, + metadata=metadata, + ) + + # Failed turns still deliver their (normalized failure) text above, but must not upload + # attachments as if the turn succeeded — mirrors the ``not agent_result.get("failed")`` + # guard on the completed-turn delivery path. + if not deliver_media: + return + + synthetic_event = MessageEvent( + text="", + source=source, + message_id=event_message_id, + ) + await self._deliver_media_from_response( + response, + synthetic_event, + adapter, + thread_metadata=metadata, + ) + + def _schedule_update_notification_watch(self) -> None: + """Ensure a background task is watching for update completion.""" + existing_task = getattr(self, "_update_notification_task", None) + if existing_task and not existing_task.done(): + return + + try: + self._update_notification_task = asyncio.create_task( + self._watch_update_progress() + ) + except RuntimeError: + logger.debug("Skipping update notification watcher: no running event loop") + + async def _watch_update_progress( + self, + poll_interval: float = 2.0, + stream_interval: float = 4.0, + timeout: float = 1800.0, + ) -> None: + """Watch ``hermes update --gateway``, streaming output + forwarding prompts. + + Polls ``.update_output.txt`` for new content and sends chunks to the user periodically; + detects ``.update_prompt.json`` (written when the update process needs input) and forwards it. + """ + from gateway.run import _hermes_home, _non_conversational_metadata + pending_path = _hermes_home / ".update_pending.json" + claimed_path = _hermes_home / ".update_pending.claimed.json" + output_path = _hermes_home / ".update_output.txt" + exit_code_path = _hermes_home / ".update_exit_code" + prompt_path = _hermes_home / ".update_prompt.json" + + loop = asyncio.get_running_loop() + deadline = loop.time() + timeout + + # Resolve the adapter and chat_id for sending messages + adapter = None + chat_id = None + session_key = None + metadata = None + for path in (claimed_path, pending_path): + if path.exists(): + try: + pending = json.loads(path.read_text(encoding="utf-8")) + platform_str = pending.get("platform") + chat_id = pending.get("chat_id") + chat_type = pending.get("chat_type") + session_key = pending.get("session_key") + thread_id = pending.get("thread_id") + message_id = pending.get("message_id") + if platform_str and chat_id: + platform = Platform(platform_str) + adapter = self.adapters.get(platform) + metadata = self._thread_metadata_for_target( + platform, + chat_id, + thread_id, + chat_type=chat_type, + reply_to_message_id=message_id, + adapter=adapter, + ) + # Fallback session key if not stored (old pending files) + if not session_key: + session_key = f"{platform_str}:{chat_id}" + break + except Exception: + pass + + if not adapter or not chat_id: + logger.warning("Update watcher: cannot resolve adapter/chat_id, falling back to completion-only") + # Completion-only fallback: wait for the exit code, then keep polling until + # _send_update_notification actually delivers (True) — it re-resolves the adapter each + # call and returns False (markers kept) while the platform is still reconnecting. + while (pending_path.exists() or claimed_path.exists()) and loop.time() < deadline: + if exit_code_path.exists() and await self._send_update_notification(): + return + await asyncio.sleep(poll_interval) + if (pending_path.exists() or claimed_path.exists()) and not exit_code_path.exists(): + exit_code_path.write_text("124", encoding="utf-8") + await self._send_update_notification() + return + + def _strip_ansi(text: str) -> str: + from tools.ansi_strip import strip_ansi + return strip_ansi(text) + + def _read_output_since(path: Path, offset: int) -> tuple[str, int]: + """Read update output defensively; logs may contain invalid UTF-8.""" + try: + data = path.read_bytes() + except OSError: + return "", offset + if len(data) <= offset: + return "", len(data) + return data[offset:].decode("utf-8", errors="replace"), len(data) + + bytes_sent = 0 + last_stream_time = loop.time() + buffer = "" + + async def _flush_buffer() -> None: + """Send buffered output to the user.""" + nonlocal buffer, last_stream_time + if not buffer.strip(): + buffer = "" + return + # Chunk to fit message limits (Telegram: 4096, others: generous) + clean = _strip_ansi(buffer).strip() + buffer = "" + last_stream_time = loop.time() + if not clean: + return + # Split into chunks if too long + max_chunk = 3500 + chunks = [clean[i:i + max_chunk] for i in range(0, len(clean), max_chunk)] + for chunk in chunks: + try: + await adapter.send( + chat_id, + f"```\n{chunk}\n```", + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + except Exception as e: + logger.debug("Update stream send failed: %s", e) + + while loop.time() < deadline: + # Check for completion + if exit_code_path.exists(): + # Read any remaining output + if output_path.exists(): + try: + chunk, bytes_sent = _read_output_since(output_path, bytes_sent) + if chunk: + buffer += chunk + except OSError: + pass + await _flush_buffer() + + # Send final status + try: + exit_code_raw = exit_code_path.read_text(encoding="utf-8").strip() or "1" + exit_code = int(exit_code_raw) + if exit_code == 0: + await adapter.send( + chat_id, + "✅ Hermes update finished.", + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + else: + await adapter.send( + chat_id, + "❌ Hermes update failed (exit code {}).".format(exit_code), + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + logger.info("Update finished (exit=%s), notified %s", exit_code, session_key) + except Exception as e: + logger.warning("Update final notification failed: %s", e) + + # Cleanup + for p in (pending_path, claimed_path, output_path, + exit_code_path, prompt_path): + p.unlink(missing_ok=True) + (_hermes_home / ".update_response").unlink(missing_ok=True) + _up_done = self._peek_session_state(session_key) + if _up_done is not None: + _up_done.persistent.update_prompt_pending = False + return + + # Check for new output + if output_path.exists(): + try: + chunk, bytes_sent = _read_output_since(output_path, bytes_sent) + if chunk: + buffer += chunk + except OSError: + pass + + # Flush buffer periodically + if buffer.strip() and (loop.time() - last_stream_time) >= stream_interval: + await _flush_buffer() + + # Check for prompts — only forward if we haven't already sent one that's still awaiting + # a response. Without this guard the watcher would re-read the same .update_prompt.json + # every poll cycle and spam the user with duplicate prompt messages. + _up_pending_state = ( + self._peek_session_state(session_key) if session_key else None + ) + if (prompt_path.exists() and session_key + and not ( + _up_pending_state is not None + and _up_pending_state.persistent.update_prompt_pending + )): + try: + prompt_data = json.loads(prompt_path.read_text(encoding="utf-8")) + prompt_text = prompt_data.get("prompt", "") + default = prompt_data.get("default", "") + if prompt_text: + # Flush any buffered output first so the user sees + # context before the prompt + await _flush_buffer() + # Try platform-native buttons first (Discord, Telegram) + sent_buttons = False + if getattr(type(adapter), "send_update_prompt", None) is not None: + try: + await adapter.send_update_prompt( + chat_id=chat_id, + prompt=prompt_text, + default=default, + session_key=session_key, + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + sent_buttons = True + except Exception as btn_err: + logger.debug("Button-based update prompt failed: %s", btn_err) + if not sent_buttons: + default_hint = f" (default: {default})" if default else "" + _p = getattr(adapter, "typed_command_prefix", "/") + await adapter.send( + chat_id, + f"⚕ **Update needs your input:**\n\n" + f"{prompt_text}{default_hint}\n\n" + f"Reply `{_p}approve` (yes) or `{_p}deny` (no), " + f"or type your answer directly.", + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + # Keep the prompt marker on disk until the user answers so a watcher after a + # mid-prompt gateway restart can recover by re-forwarding it. + self._session_state( + session_key + ).persistent.update_prompt_pending = True + # .update_response to continue — it doesn't re-check + logger.info("Forwarded update prompt to %s: %s", session_key, prompt_text[:80]) + except (json.JSONDecodeError, OSError) as e: + logger.debug("Failed to read update prompt: %s", e) + + await asyncio.sleep(poll_interval) + + # Timeout + if not exit_code_path.exists(): + logger.warning("Update watcher timed out after %.0fs", timeout) + exit_code_path.write_text("124", encoding="utf-8") + await _flush_buffer() + with suppress(Exception): + await adapter.send( + chat_id, + "❌ Hermes update timed out after 30 minutes.", + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + for p in (pending_path, claimed_path, output_path, + exit_code_path, prompt_path): + p.unlink(missing_ok=True) + (_hermes_home / ".update_response").unlink(missing_ok=True) + _up_timeout_state = self._peek_session_state(session_key) + if _up_timeout_state is not None: + _up_timeout_state.persistent.update_prompt_pending = False + + async def _send_update_notification(self) -> bool: + """If an update finished, notify the user. + + False while the update is still running (caller may retry); True after a definitive send/skip. + """ + from gateway.run import _hermes_home, _non_conversational_metadata + pending_path = _hermes_home / ".update_pending.json" + claimed_path = _hermes_home / ".update_pending.claimed.json" + output_path = _hermes_home / ".update_output.txt" + exit_code_path = _hermes_home / ".update_exit_code" + + if not pending_path.exists() and not claimed_path.exists(): + return False + + cleanup = True + active_pending_path = claimed_path + try: + if pending_path.exists(): + try: + pending_path.replace(claimed_path) + except FileNotFoundError: + if not claimed_path.exists(): + return True + elif not claimed_path.exists(): + return True + + pending = json.loads(claimed_path.read_text(encoding="utf-8")) + platform_str = pending.get("platform") + chat_id = pending.get("chat_id") + chat_type = pending.get("chat_type") + thread_id = pending.get("thread_id") + message_id = pending.get("message_id") + + if not exit_code_path.exists(): + logger.info("Update notification deferred: update still running") + cleanup = False + active_pending_path = pending_path + claimed_path.replace(pending_path) + return False + + exit_code_raw = exit_code_path.read_text(encoding="utf-8").strip() or "1" + exit_code = int(exit_code_raw) + + # Read the captured update output + output = "" + if output_path.exists(): + output = output_path.read_bytes().decode("utf-8", errors="replace") + + # Resolve adapter + platform = Platform(platform_str) + adapter = self.adapters.get(platform) + + if not adapter and chat_id: + # The update finished, but the target platform has not reconnected yet (common right + # after the restart that `hermes update` triggers). Treating "adapter missing" as a + # definitive skip would delete the markers and silently lose the notification; + # preserve them so a later retry (watcher poll or next startup) can deliver it. + logger.info( + "Update notification deferred: %s adapter not connected yet", + platform_str, + ) + cleanup = False + active_pending_path = pending_path + claimed_path.replace(pending_path) + return False + + if adapter and chat_id: + metadata = self._thread_metadata_for_target( + platform, + chat_id, + thread_id, + chat_type=chat_type, + reply_to_message_id=message_id, + adapter=adapter, + ) + # Strip ANSI escape codes for clean display + from tools.ansi_strip import strip_ansi + output = strip_ansi(output).strip() + if output: + if len(output) > 3500: + output = "…" + output[-3500:] + if exit_code == 0: + msg = f"✅ Hermes update finished.\n\n```\n{output}\n```" + else: + msg = f"❌ Hermes update failed.\n\n```\n{output}\n```" + elif exit_code == 0: + msg = "✅ Hermes update finished successfully." + else: + msg = "❌ Hermes update failed. Check the gateway logs or run `hermes update` manually for details." + await adapter.send( + chat_id, + msg, + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + logger.info( + "Sent post-update notification to %s:%s (exit=%s)", + platform_str, + chat_id, + exit_code, + ) + except Exception as e: + logger.warning("Post-update notification failed: %s", e) + finally: + if cleanup: + active_pending_path.unlink(missing_ok=True) + claimed_path.unlink(missing_ok=True) + output_path.unlink(missing_ok=True) + exit_code_path.unlink(missing_ok=True) + + return True + + async def _send_restart_notification(self) -> Optional[tuple[str, str, Optional[str]]]: + """Notify the chat that initiated /restart that the gateway is back.""" + from gateway.run import _hermes_home, _non_conversational_metadata, resolve_delivery_transport + notify_path = _hermes_home / ".restart_notify.json" + if not notify_path.exists(): + return None + + try: + data = json.loads(notify_path.read_text(encoding="utf-8")) + platform_str = data.get("platform") + chat_id = data.get("chat_id") + chat_type = data.get("chat_type") + thread_id = data.get("thread_id") + message_id = data.get("message_id") + + if not platform_str or not chat_id: + return None + + platform = Platform(platform_str) + transport = resolve_delivery_transport(platform, self.config, self.adapters) + if transport is None: + logger.debug( + "Restart notification skipped: no live transport for %s", + platform_str, + ) + return None + + platform_cfg = self.config.platforms.get(platform) + if platform_cfg is not None and not platform_cfg.gateway_restart_notification: + logger.info( + "Restart notification suppressed: %s has gateway_restart_notification=false", + platform_str, + ) + return None + + metadata = self._thread_metadata_for_target( + platform, + chat_id, + thread_id, + chat_type=chat_type, + reply_to_message_id=message_id, + adapter=transport.adapter, + ) + if data.get("delivered_via_upstream_relay") is True: + metadata = dict(metadata or {}) + if data.get("user_id"): + metadata["user_id"] = str(data["user_id"]) + if data.get("scope_id"): + metadata["scope_id"] = str(data["scope_id"]) + result = await transport.send( + platform, + str(chat_id), + "♻ Gateway restarted successfully. Your session continues.", + metadata=_non_conversational_metadata(metadata, platform=platform), + ) + # adapter.send() catches provider errors (e.g. "Chat not found") and returns + # SendResult(success=False) rather than raising, so inspect the result before claiming + # success — otherwise the log line hides real delivery failures. + if result is not None and getattr(result, "success", True) is False: + logger.warning( + "Restart notification to %s:%s was not delivered: %s", + platform_str, + chat_id, + getattr(result, "error", "send returned success=False"), + ) + return None + + logger.info( + "Sent restart notification to %s:%s", + platform_str, + chat_id, + ) + return str(platform_str), str(chat_id), str(thread_id) if thread_id else None + except Exception as e: + logger.warning("Restart notification failed: %s", e) + return None + finally: + notify_path.unlink(missing_ok=True) + + async def _send_home_channel_startup_notifications( + self, + *, + skip_targets: Optional[set[tuple[str, str, Optional[str]]]] = None, + ) -> set[tuple[str, str, Optional[str]]]: + """Notify configured home channels that the gateway is back online. + + The notification is best-effort and sent once per connected platform + home channel. ``skip_targets`` lets startup avoid duplicate messages + when a more specific restart notification is queued for the same chat. + """ + from gateway.run import _non_conversational_metadata, resolve_delivery_transport + delivered: set[tuple[str, str, Optional[str]]] = set() + skipped = skip_targets or set() + message = "♻️ Gateway online — Hermes is back and ready." + + for platform, platform_cfg in self.config.platforms.items(): + home = platform_cfg.home_channel + if not home or not home.chat_id: + continue + + transport = resolve_delivery_transport(platform, self.config, self.adapters) + if transport is None: + continue + + if not platform_cfg.gateway_restart_notification: + logger.info( + "Home-channel startup notification suppressed: %s has gateway_restart_notification=false", + platform.value, + ) + continue + + target = (platform.value, str(home.chat_id), str(home.thread_id) if home.thread_id else None) + if target in skipped or target in delivered: + continue + + try: + metadata = self._thread_metadata_for_target( + platform, + home.chat_id, + home.thread_id, + adapter=transport.adapter, + ) + if transport.is_relay: + metadata = dict(metadata or {}) + if home.user_id: + metadata["user_id"] = home.user_id + if home.scope_id: + metadata["scope_id"] = home.scope_id + send_metadata = _non_conversational_metadata(metadata, platform=platform) + if send_metadata is not None or transport.is_relay: + result = await transport.send( + platform, + str(home.chat_id), + message, + metadata=send_metadata, + ) + else: + result = await transport.adapter.send(str(home.chat_id), message) + if result is not None and getattr(result, "success", True) is False: + logger.warning( + "Home-channel startup notification failed for %s:%s: %s", + platform.value, + home.chat_id, + getattr(result, "error", "send returned success=False"), + ) + continue + + delivered.add(target) + logger.info( + "Sent home-channel startup notification to %s:%s", + platform.value, + home.chat_id, + ) + except Exception as exc: + logger.warning( + "Home-channel startup notification failed for %s:%s: %s", + platform.value, + home.chat_id, + exc, + ) + + return delivered + + async def _send_session_db_warning_notifications(self) -> None: + """Broadcast a state.db failure warning to all home channels. + + When SessionDB init fails at gateway startup, messages may flow but nothing is persisted + — /resume, /history, and session_search all silently break. Best-effort: failures are + logged, not raised. + """ + from gateway.run import _non_conversational_metadata, resolve_delivery_transport + error = getattr(self, "_session_db_init_error", None) + if not error: + return + + from hermes_state import classify_persistence_error, format_session_db_unavailable + + cause = classify_persistence_error(error) + hint = format_session_db_unavailable() + if cause == "corrupt": + message = ( + "⚠️ Session database corruption detected. Messages may not be " + "persisted. Recovery options:\n" + "1. Run `hermes doctor --fix`\n" + "2. Salvage with: sqlite3 ~/.hermes/state.db \".recover\" " + "(then replace state.db)\n" + "3. Restore from a backup in ~/.hermes/backups/\n" + "Run `hermes doctor` for sanitized diagnostics." + ) + else: + message = ( + f"⚠️ Session database unavailable — messages may not be persisted. " + f"{hint}\n" + f"Run `hermes doctor` for diagnostics." + ) + + logger.warning( + "Broadcasting state.db failure warning to home channels: %s", error + ) + + for platform, platform_cfg in self.config.platforms.items(): + home = platform_cfg.home_channel + if not home or not home.chat_id: + continue + transport = resolve_delivery_transport(platform, self.config, self.adapters) + if transport is None: + continue + try: + metadata = self._thread_metadata_for_target( + platform, + home.chat_id, + home.thread_id, + adapter=transport.adapter, + ) + if transport.is_relay: + metadata = dict(metadata or {}) + if home.user_id: + metadata["user_id"] = home.user_id + if home.scope_id: + metadata["scope_id"] = home.scope_id + send_metadata = _non_conversational_metadata(metadata, platform=platform) + if send_metadata is not None or transport.is_relay: + result = await transport.send( + platform, + str(home.chat_id), + message, + metadata=send_metadata, + ) + else: + result = await transport.adapter.send(str(home.chat_id), message) + if result is not None and getattr(result, "success", True) is False: + logger.warning( + "state.db warning notification failed for %s:%s: %s", + platform.value, + home.chat_id, + getattr(result, "error", "send returned success=False"), + ) + except Exception as exc: + logger.warning( + "state.db warning notification failed for %s:%s: %s", + platform.value, + home.chat_id, + exc, + ) + + def _build_process_event_source(self, evt: dict): + """Resolve the canonical source for a synthetic background-process event. + + Prefer the persisted session-store origin; the active foreground event causes cross-topic bleed. + """ + from gateway.run import _parse_session_key + from gateway.session import SessionSource + + session_key = str(evt.get("session_key") or "").strip() + derived_platform = "" + derived_chat_type = "" + derived_chat_id = "" + + if session_key: + try: + self.session_store._ensure_loaded() + entry = self.session_store._entries.get(session_key) + if entry and getattr(entry, "origin", None): + return entry.origin + except Exception as exc: + logger.debug( + "Synthetic process-event session-store lookup failed for %s: %s", + session_key, + exc, + ) + + cached_source = self._get_cached_session_source(session_key) + if cached_source is not None: + return cached_source + + _parsed = _parse_session_key(session_key) + if _parsed: + derived_platform = _parsed["platform"] + derived_chat_type = _parsed["chat_type"] + derived_chat_id = _parsed["chat_id"] + + platform_name = str(evt.get("platform") or derived_platform or "").strip().lower() + chat_type = str(evt.get("chat_type") or derived_chat_type or "").strip().lower() + chat_id = str(evt.get("chat_id") or derived_chat_id or "").strip() + if not platform_name or not chat_type or not chat_id: + logger.warning( + "Synthetic event source unresolvable: " + "session_key=%r platform=%r chat_type=%r chat_id=%r " + "evt_type=%s", + session_key, platform_name, chat_type, chat_id, + evt.get("type", "?"), + ) + return None + + try: + platform = Platform(platform_name) + # Reject arbitrary strings (dynamic pseudo-members): built-ins are always valid, plugin + # platforms must be registered in the platform registry. + if platform.value not in _BUILTIN_PLATFORM_VALUES: + try: + from gateway.platform_registry import platform_registry + if not platform_registry.is_registered(platform.value): + raise ValueError(platform_name) + except Exception: + raise ValueError(platform_name) + except Exception: + logger.warning( + "Synthetic process event has invalid platform metadata: %r", + platform_name, + ) + return None + + scope_id = str(evt.get("scope_id") or "").strip() or None + if scope_id is None and chat_type not in ("dm", "thread"): + # Reconstructed (non-persisted) source for a scoped chat with no scope discriminator: a + # relay connector's fail-closed tenant guard may decline the reply unless user_id resolves it + # (resolveByUser). Don't fail — DMs/author-bound chats still route and native adapters need + # no scope_id — but warn so a post-restart egress decline isn't silent. + logger.warning( + "Synthetic event source for %s chat=%s (%s) reconstructed " + "without scope_id; scoped relay egress may be declined by " + "the connector's tenant guard (user_id fallback only).", + platform_name, chat_id, chat_type, + ) + return SessionSource( + platform=platform, + chat_id=chat_id, + chat_type=chat_type, + thread_id=str(evt.get("thread_id") or "").strip() or None, + user_id=str(evt.get("user_id") or "").strip() or None, + user_name=str(evt.get("user_name") or "").strip() or None, + scope_id=scope_id, + ) + + async def _drain_watch_notifications(self, completion_queue) -> None: + """Consume queued watch events and inject them when notifications are enabled. + + The queue is ALWAYS drained (so watch events don't rot or requeue-spin) but injection is + skipped entirely when ``display.background_process_notifications`` is ``off``. + """ + from gateway.run import _drain_gateway_watch_events, _format_gateway_process_notification + watch_events = _drain_gateway_watch_events(completion_queue) + if self._load_background_notifications_mode() == "off": + return + + for evt in watch_events: + synth_text = _format_gateway_process_notification(evt) + if not synth_text: + continue + try: + await self._inject_watch_notification(synth_text, evt) + except Exception as exc: + logger.error("Watch notification injection error: %s", exc) + + async def _inject_watch_notification( + self, synth_text: str, evt: dict, + ) -> Optional[bool]: + """Inject a watch/completion notification as a synthetic message event. + + Routing comes from the queued event, never the active foreground message. Returns + ``True`` on adapter acceptance, ``False`` on retryable adapter failure, ``None`` with no + gateway route. Not transactional: a crash after acceptance can replay (at-least-once). + """ + from gateway.run import _parse_session_key, resolve_delivery_transport + source = await asyncio.to_thread(self._build_process_event_source, evt) + if not source: + # API-server sessions bind the RAW X-Hermes-Session-Id key (_bind_api_server_session), not a + # structured ``agent:main:...`` key, so _build_process_event_source returned None above. + raw_sid = str(evt.get("origin_session_id") or "").strip() + if not raw_sid: + _sk = str(evt.get("session_key") or "").strip() + if _sk and _parse_session_key(_sk) is None: + raw_sid = _sk + if raw_sid: + adapter = self.adapters.get(Platform.API_SERVER) + from gateway.wake import ( + adapter_supports_push, + deliver_wake, + persist_delegation_delivery, + ) + if adapter is not None and not adapter_supports_push(adapter): + if evt.get("type") == "async_delegation": + # after the parent turn's event.complete the CLIENT owns the next turn on + # this stateless surface. Persist the completion as a durable delivery row — + # never self-post it as a new role=user prompt. + try: + logger.info( + "Async delegation completion — persisting " + "delivery row for api_server session %s " + "(no wake turn)", + raw_sid, + ) + await persist_delegation_delivery( + adapter, text=synth_text, + session_id=raw_sid, evt=evt, + ) + return True + except Exception as e: + logger.warning( + "Async delegation delivery persist failed " + "for session %s: %s", + raw_sid, e, + ) + return False + try: + logger.info( + "Watch pattern notification — waking api_server " + "session %s via self-post", + raw_sid, + ) + await deliver_wake(adapter, text=synth_text, session_id=raw_sid) + return True + except Exception as e: + logger.warning( + "Watch notification self-post wake failed for " + "session %s: %s", + raw_sid, e, + ) + return False + logger.warning( + "Dropping watch notification for raw session %s: no " + "api_server adapter to self-post through", + raw_sid, + ) + return None + logger.warning( + "Dropping watch notification with no routing metadata for process %s", + evt.get("session_id", "unknown"), + ) + return None + platform_name = source.platform.value if hasattr(source.platform, "value") else str(source.platform) + # Alias-aware resolution (relay-plane): one adapter under Platform.RELAY fronts N logical + # platforms, so a literal ``p.value == platform_name`` scan misses "slack" and drops the + # completion as "no gateway route". Use the shared transport resolver — native adapter wins; + # relay is eligible only when it advertises fronting the logical platform. + adapter = None + try: + _platform_enum = Platform(platform_name) + except (ValueError, KeyError): + _platform_enum = None + if _platform_enum is not None: + try: + _transport = resolve_delivery_transport( + _platform_enum, self.config, self.adapters, + ) + except Exception: + _transport = None + if _transport is not None: + adapter = _transport.adapter + if adapter is None: + # Legacy literal scan — still correct for native adapters; keeps minimal runner stubs (tests) + # and exotic platform strings working when the resolver can't run. + for p, a in self.adapters.items(): + if p.value == platform_name: + adapter = a + break + if not adapter: + return None + from gateway.wake import adapter_supports_push as _wake_push_ok + if not _wake_push_ok(adapter): + # Non-push adapter (api_server) resolved WITH routing metadata: its chat_id is the raw session + # id (_bind_api_server_session binds chat_id = session_id), so handle_message would run the + # wake under a build_session_key() key that never matches the raw session — self-post. + from gateway.wake import deliver_wake, persist_delegation_delivery + raw_sid = str(evt.get("origin_session_id") or "").strip() or str(source.chat_id or "") + if evt.get("type") == "async_delegation": + # Same client-owns-the-turn rule as the raw-key branch above: persist the completion as a + # delivery row, never self-post it as a new role=user prompt. + try: + logger.info( + "Async delegation completion — persisting delivery " + "row for api_server session %s (no wake turn)", + raw_sid, + ) + await persist_delegation_delivery( + adapter, text=synth_text, session_id=raw_sid, evt=evt, + ) + return True + except Exception as e: + logger.warning( + "Async delegation delivery persist failed for " + "session %s: %s", + raw_sid, e, + ) + return False + try: + logger.info( + "Watch pattern notification — waking api_server session " + "%s via self-post", + raw_sid, + ) + await deliver_wake(adapter, text=synth_text, session_id=raw_sid) + return True + except Exception as e: + logger.warning( + "Watch notification self-post wake failed for session " + "%s: %s", + raw_sid, e, + ) + return False + try: + metadata = {} + parent_session_id = str(evt.get("parent_session_id") or "").strip() + if parent_session_id: + metadata["gateway_session_id"] = parent_session_id + synth_event = MessageEvent( + text=synth_text, + message_type=MessageType.TEXT, + source=source, + internal=True, + message_id=str(evt.get("message_id") or "").strip() or None, + metadata=metadata, + ) + logger.info( + "Watch pattern notification — injecting for %s chat=%s thread=%s", + platform_name, + source.chat_id, + source.thread_id, + ) + # Relay-plane egress priming: a synthetic turn injected right after a restart reaches a relay + # adapter whose per-chat routing caches are cold (they warm only on inbound), so its replies + # egress without tenant discriminators and the connector's fail-closed guard declines them. + _prime = getattr(adapter, "prime_routing_cache", None) + if callable(_prime): + _prime(synth_event) + await adapter.handle_message(synth_event) + return True + except Exception as e: + logger.error("Watch notification injection error: %s", e) + return False + + @staticmethod + def _completion_delivery_identity(evt: dict) -> Optional[tuple[str, str, object]]: + """Return a producer-stable identity when one is available. + + Delegation UUIDs identify one producer completion. Process session IDs include the + persisted spawn epoch so a reused ID is a distinct incarnation; legacy events without + ``started_at`` are delivered undeduplicated rather than risk suppressing a real completion. + """ + evt_type = str(evt.get("type") or "") + if evt_type == "async_delegation": + producer_id = str(evt.get("delegation_id") or "") + return (evt_type, producer_id, "") if producer_id else None + if evt_type == "completion": + producer_id = str(evt.get("session_id") or "") + started_at = evt.get("started_at") + if producer_id and started_at is not None: + return (evt_type, producer_id, started_at) + return None + + async def _classify_completion_target(self, parent_session_id: str) -> str: + """Classify an async-completion delivery target before adapter acceptance. + + - ``"deliver"``: spawning session live (or compression-rotated with a live continuation); + proves deliverability only, the resolver still retargets. + - ``"terminal"``: parent gone for good (unknown / explicit user boundary like /new); drop + the durable row rather than falsely ack or replay forever. + - ``"retry"``: transient uncertainty (DB unavailable, rotation mid-flight); release the + claim for a later consumer; the attempt cap bounds churn. + """ + from gateway.run import _USER_BOUNDARY_END_REASONS + session_db = getattr(self, "_session_db", None) + if session_db is None: + return "retry" + try: + parent = await session_db.get_session(parent_session_id) + except Exception: + logger.debug( + "Async-completion pre-flight parent lookup failed for %s", + parent_session_id, exc_info=True, + ) + return "retry" + if parent is None: + return "terminal" + if not parent.get("ended_at"): + return "deliver" + end_reason = str(parent.get("end_reason") or "") + if end_reason != "compression": + # An ended parent is unreachable only when the USER closed the thread of work (/new -> + # session_reset / new_session, user_exit, session_switch). Idle/timeout ends are normal on + # scale-to-zero relays — the chat stays routable and the resolver retargets, so dropping loses + # finished work. Boundary set shared with the resolver (_USER_BOUNDARY_END_REASONS): no drift. + if end_reason in _USER_BOUNDARY_END_REASONS: + return "terminal" + return "deliver" + try: + tip_session_id = await session_db.get_compression_tip(parent_session_id) + if not tip_session_id or tip_session_id == parent_session_id: + # Rotation caught mid-flight: parent is compression-ended but + # its continuation isn't visible yet. Retry, don't drop. + return "retry" + tip = await session_db.get_session(tip_session_id) + except Exception: + logger.debug( + "Async-completion pre-flight tip lookup failed for %s", + parent_session_id, exc_info=True, + ) + return "retry" + if tip is None or tip.get("ended_at"): + return "retry" + return "deliver" + + async def _deliver_completion_notification( + self, synth_text: str, evt: dict, + ) -> Optional[bool]: + """Deliver once per live gateway, or return False for a retry. + + ``True``: adapter accepted; ``False``: injection failed, claim released for retry; ``None``: + another same-lifecycle caller owns/delivered it, or no route. No cross-process exactly-once. + """ + identity = self._completion_delivery_identity(evt) + durable_claim_id = "" + durable_delegation_id = "" + if evt.get("type") == "async_delegation": + durable_delegation_id = str(evt.get("delegation_id") or "") + if durable_delegation_id: + try: + from tools.async_delegation import claim_completion_delivery + + durable_claim_id = f"gateway:{id(self)}:{__import__('uuid').uuid4().hex}" + if not claim_completion_delivery( + durable_delegation_id, durable_claim_id, + ): + return None + except Exception as exc: + logger.warning( + "Could not claim durable async completion %s: %s", + durable_delegation_id, exc, + ) + return False + parent_session_id = str(evt.get("parent_session_id") or "").strip() + if parent_session_id: + # Adapter acceptance is not proof of delivery: the inner resolver can still fail closed + # inside the pipeline after acceptance, falsely acking the durable row as delivered. + # Verify the target before acceptance so drops get an honest durable disposition. + verdict = await self._classify_completion_target(parent_session_id) + if verdict == "terminal": + logger.warning( + "Async delegation %s targets permanently-gone session %s; " + "terminally dropping delivery (result remains in the " + "delegation records).", + durable_delegation_id or "", parent_session_id, + ) + if durable_claim_id: + try: + from tools.async_delegation import drop_completion_delivery + + drop_completion_delivery( + durable_delegation_id, durable_claim_id, + ) + except Exception: + logger.debug( + "Could not drop durable completion claim", + exc_info=True, + ) + return None + if verdict == "retry": + if durable_claim_id: + try: + from tools.async_delegation import release_completion_delivery + + release_completion_delivery( + durable_delegation_id, durable_claim_id, + ) + except Exception: + logger.debug( + "Could not release durable completion claim", + exc_info=True, + ) + return False + elif evt.get("type") == "completion": + # Background completions carry only session_key, so after /new the OLD session's + # notification would land in the chat's NEW session. Stamped events get the same + # pre-flight as async delegations (_classify_completion_target); unstamped ones deliver. + parent_session_id = str(evt.get("parent_session_id") or "").strip() + if parent_session_id: + verdict = await self._classify_completion_target(parent_session_id) + if verdict == "terminal": + logger.warning( + "Background process %s completion targets " + "permanently-gone session %s (user boundary such as " + "/new); dropping notification (output remains " + "available via process(action='log')).", + evt.get("session_id") or "", parent_session_id, + ) + return None + if verdict == "retry": + # Transient uncertainty (session DB down / compression rotation mid-flight): tell the + # watcher to re-poll and retry rather than drop or misroute the result. + return False + if identity is not None: + with self._completion_delivery_lock: + if ( + identity in self._completion_deliveries_inflight + or identity in self._completion_deliveries_delivered + ): + return None + self._completion_deliveries_inflight.add(identity) + + accepted = False + try: + injection_result = await self._inject_watch_notification(synth_text, evt) + if injection_result is not True: + return injection_result + accepted = True + + if identity is not None: + with self._completion_delivery_lock: + self._completion_deliveries_inflight.discard(identity) + self._completion_deliveries_delivered[identity] = None + while ( + len(self._completion_deliveries_delivered) + > self._completion_delivery_retention + ): + self._completion_deliveries_delivered.popitem(last=False) + + # When the durable async-delegation producer branch is present, its SQLite row is the + # authoritative replay state — ack it after adapter acceptance; no parallel ledger here. + if durable_claim_id: + try: + from tools.async_delegation import complete_completion_delivery + + complete_completion_delivery( + durable_delegation_id, durable_claim_id, + ) + except Exception as exc: + logger.warning( + "Could not acknowledge durable async completion %s: %s", + durable_delegation_id, exc, + ) + return True + finally: + if identity is not None and not accepted: + with self._completion_delivery_lock: + self._completion_deliveries_inflight.discard(identity) + if durable_claim_id and not accepted: + try: + from tools.async_delegation import release_completion_delivery + + release_completion_delivery( + durable_delegation_id, durable_claim_id, + ) + except Exception: + logger.debug("Could not release durable completion claim", exc_info=True) + + @staticmethod + def _completion_notification_batch_key(evt: dict) -> tuple[str, ...]: + """Return a routing-complete key for short-window process fan-in.""" + return tuple(str(evt.get(field) or "") for field in ( + "session_key", + "platform", + "chat_type", + "chat_id", + "thread_id", + "user_id", + )) + + @staticmethod + def _format_coalesced_process_completions(entries: list[tuple[str, dict, asyncio.Future]]) -> str: + """Build one bounded synthetic event from several redacted completions.""" + from gateway.run import _redact_gateway_user_facing_secrets + lines = [ + f"[IMPORTANT: {len(entries)} background processes completed for this session.", + "Treat these results as one completion batch and send at most one " + "consolidated user-facing response.", + ] + shown = entries[:10] + for _text, evt, _future in shown: + session_id = str(evt.get("session_id") or "unknown") + exit_code = evt.get("exit_code") + reason = str(evt.get("completion_reason") or "exited") + # Completion output normally passes the terminal redactor at the producer seam, but that is + # configurable and this is user-facing, so keep the unconditional gateway floor. Redact + # BEFORE slicing: truncating first can leave a credential fragment the patterns miss. + output = _redact_gateway_user_facing_secrets( + str(evt.get("output") or "") + ).strip() + if len(output) > 800: + output = f"[… truncated …]\n{output[-800:]}" + lines.append( + f"\n- {session_id}: exit_code={exit_code}, reason={reason}" + ) + if output: + lines.append(output) + omitted = len(entries) - len(shown) + if omitted: + lines.append( + f"\n- … and {omitted} more completion(s); inspect them with " + "the process tool if they affect the conclusion." + ) + lines.append( + "If a result does not change the current conclusion, absorb it silently.]" + ) + return "\n".join(lines) + + def _record_coalesced_completion_siblings(self, events: list[dict]) -> None: + """Extend a successful primary delivery claim to its batched siblings.""" + with self._completion_delivery_lock: + for evt in events: + identity = self._completion_delivery_identity(evt) + if identity is None: + continue + self._completion_deliveries_inflight.discard(identity) + self._completion_deliveries_delivered[identity] = None + while ( + len(self._completion_deliveries_delivered) + > self._completion_delivery_retention + ): + self._completion_deliveries_delivered.popitem(last=False) + + async def _flush_process_completion_batch(self, key: tuple[str, ...]) -> None: + """Deliver one short-window completion batch and resolve its waiters.""" + current_task = asyncio.current_task() + entries: list[tuple[str, dict, asyncio.Future]] = [] + delivered: Optional[bool] = False + try: + await asyncio.sleep(self._completion_notification_batch_window) + entries = self._completion_notification_batches.pop(key, []) + # Detach before adapter delivery. A completion that arrives while + # this batch is in flight must be able to schedule the next flush. + if self._completion_notification_batch_tasks.get(key) is current_task: + self._completion_notification_batch_tasks.pop(key, None) + if not entries: + return + if len(entries) == 1: + synth_text = entries[0][0] + else: + synth_text = self._format_coalesced_process_completions(entries) + + # A duplicate primary can legitimately return None from the lifecycle dedupe seam; try the + # next batch identity so a fresh sibling is never discarded with that duplicate. + delivered = None + for _text, candidate_evt, _future in entries: + delivered = await self._deliver_completion_notification( + synth_text, candidate_evt, + ) + if delivered is not None: + break + if delivered is True and len(entries) > 1: + self._record_coalesced_completion_siblings( + [evt for _text, evt, _future in entries] + ) + except asyncio.CancelledError: + # Shutdown may cancel us mid fan-in or while adapter delivery is blocked: recover entries not + # yet detached and resolve every waiter as retryable before adapters are torn down. + delivered = False + if not entries: + entries = self._completion_notification_batches.pop(key, []) + raise + except Exception: + logger.exception("Coalesced process completion delivery failed") + delivered = False + finally: + # Never strand watcher futures when formatting, delivery, or cancellation interrupts a batch: + # False follows the existing watcher retry path; None remains the ordinary dedupe result. + for _text, _evt, future in entries: + if not future.done(): + future.set_result(delivered) + # Do not remove a newer flush task that reused the same route key. + if self._completion_notification_batch_tasks.get(key) is current_task: + self._completion_notification_batch_tasks.pop(key, None) + + async def _cancel_process_completion_batch_tasks(self) -> None: + """Settle pending completion batches before adapter teardown.""" + self._completion_notification_batches_stopping = True + tasks = { + task + for task in getattr( + self, "_completion_notification_batch_flush_tasks", set() + ) + if not task.done() + } + for task in tasks: + task.cancel() + if tasks: + await asyncio.gather(*tasks, return_exceptions=True) + + # Defensive cleanup for an orphaned queue with no live flush task. + batches = getattr(self, "_completion_notification_batches", {}) + for entries in batches.values(): + for _text, _evt, future in entries: + if not future.done(): + future.set_result(False) + batches.clear() + getattr(self, "_completion_notification_batch_tasks", {}).clear() + getattr(self, "_completion_notification_batch_flush_tasks", set()).clear() + + async def _enqueue_process_completion_notification( + self, synth_text: str, evt: dict, + ) -> Optional[bool]: + """Fan in concurrent process completions that share one conversation.""" + # Some unit tests construct GatewayRunner with object.__new__. Keep the + # batching seam lazy so those focused lifecycle tests remain valid. + if not hasattr(self, "_completion_notification_batches"): + self._completion_notification_batches = {} + if not hasattr(self, "_completion_notification_batch_tasks"): + self._completion_notification_batch_tasks = {} + if not hasattr(self, "_completion_notification_batch_flush_tasks"): + self._completion_notification_batch_flush_tasks = set() + if not hasattr(self, "_completion_notification_batch_window"): + self._completion_notification_batch_window = 0.1 + if not hasattr(self, "_completion_notification_batches_stopping"): + self._completion_notification_batches_stopping = False + + if self._completion_notification_batches_stopping: + return False + + key = self._completion_notification_batch_key(evt) + future = asyncio.get_running_loop().create_future() + self._completion_notification_batches.setdefault(key, []).append( + (synth_text, evt, future) + ) + if key not in self._completion_notification_batch_tasks: + task = asyncio.create_task( + self._flush_process_completion_batch(key) + ) + self._completion_notification_batch_tasks[key] = task + # Keep the flush alive under the gateway's normal lifecycle accounting; runners built via + # object.__new__ (focused tests) lazily receive the same ownership set. + if not hasattr(self, "_background_tasks"): + self._background_tasks = set() + self._background_tasks.add(task) + self._completion_notification_batch_flush_tasks.add(task) + task.add_done_callback(self._background_tasks.discard) + task.add_done_callback( + self._completion_notification_batch_flush_tasks.discard + ) + return await future + + def _enrich_async_delegation_routing(self, evt: dict) -> None: + """Fill platform/chat_id/thread_id/chat_type on an async-delegation event. + + Such events only carry ``session_key`` (the daemon worker lacks per-message routing + metadata). Best-effort: a CLI-origin event (empty session_key) is left as-is and won't route. + """ + from gateway.run import _parse_session_key + if evt.get("platform"): + return # already enriched + parsed = _parse_session_key(evt.get("session_key", "") or "") + if not parsed: + return + evt["platform"] = parsed.get("platform", "") + evt["chat_type"] = parsed.get("chat_type", "") + evt["chat_id"] = parsed.get("chat_id", "") + if parsed.get("thread_id"): + evt["thread_id"] = parsed["thread_id"] + + @staticmethod + def _async_delegation_group_key(evt: dict) -> tuple[str, ...]: + """Return the async-completion coalescing key: originating session, parent session, route.""" + return tuple(str(evt.get(field) or "") for field in ( + "session_key", + "parent_session_id", + "platform", + "chat_type", + "chat_id", + "thread_id", + "user_id", + )) + + @staticmethod + def _format_coalesced_async_delegations(blocks: list[str]) -> str: + """Join per-delegation formatted blocks into one consolidated turn.""" + header = ( + f"[IMPORTANT: {len(blocks)} background subagent delegations " + "completed for this session. Treat these results as one " + "completion batch and send at most one consolidated user-facing " + "response. If a result does not change the current conclusion, " + "absorb it silently.]" + ) + return "\n\n".join([header, *blocks]) + + async def _deliver_async_delegation_group( + self, group: list[dict], + ) -> Optional[bool]: + """Deliver a same-session batch of async completions as ONE turn. + + Single-event groups ride the per-event path. Multi-event groups deliver the primary via + ``_deliver_completion_notification`` with consolidated text of every sibling THIS runner + claimed; sibling claims are acked only after adapter acceptance, and siblings claimed by + another consumer are excluded (no double delivery). Returns True after acceptance, False + to requeue the group, None when nothing is deliverable here (retry siblings requeued). + """ + from gateway.run import _format_gateway_process_notification + from tools.process_registry import process_registry as _pr + + deliverable: list[tuple[dict, str]] = [] + for evt in group: + synth_text = _format_gateway_process_notification(evt) + if not synth_text: + continue + identity = self._completion_delivery_identity(evt) + if identity is not None: + with self._completion_delivery_lock: + if ( + identity in self._completion_deliveries_inflight + or identity in self._completion_deliveries_delivered + ): + continue + deliverable.append((evt, synth_text)) + + if not deliverable: + return None + if len(deliverable) == 1: + evt, synth_text = deliverable[0] + return await self._deliver_completion_notification(synth_text, evt) + + from tools.async_delegation import ( + claim_event_delivery, + complete_event_delivery, + release_event_delivery, + ) + + primary_evt, primary_text = deliverable[0] + blocks = [primary_text] + siblings: list[tuple[dict, str]] = [] + for evt, synth_text in deliverable[1:]: + claim_id = claim_event_delivery(evt, f"gateway-batch:{id(self)}") + if claim_id is None: + # Another consumer owns this row's delivery; keep its result + # out of our consolidated text so it is never double-injected. + continue + siblings.append((evt, claim_id)) + blocks.append(synth_text) + + if not siblings: + return await self._deliver_completion_notification( + primary_text, primary_evt, + ) + + consolidated = self._format_coalesced_async_delegations(blocks) + delivered: Optional[bool] = False + try: + delivered = await self._deliver_completion_notification( + consolidated, primary_evt, + ) + finally: + if delivered is True: + for evt, claim_id in siblings: + try: + complete_event_delivery(evt, claim_id) + except Exception: + logger.debug( + "Could not acknowledge coalesced durable completion", + exc_info=True, + ) + self._record_coalesced_completion_siblings( + [evt for evt, _claim_id in siblings] + ) + else: + # Not delivered — release every sibling claim so a retry or another consumer can claim it, + # honestly leaving the durable rows pending. + for evt, claim_id in siblings: + try: + release_event_delivery(evt, claim_id) + except Exception: + logger.debug( + "Could not release coalesced durable claim", + exc_info=True, + ) + if delivered is None: + # The primary was dropped/owned elsewhere but the siblings + # still need delivery — requeue just them for the next tick. + for evt, _claim_id in siblings: + _pr.completion_queue.put(evt) + return delivered + + async def _async_delegation_watcher(self, interval: float = 2.0) -> None: + """Drain async-delegation completions and inject them as new turns (IDLE case). + + Background subagents run on the daemon executor with no per-process watcher, so their + completions would otherwise only be seen by the post-turn drain. Ignores non-async events. + """ + await asyncio.sleep(3) # let platforms finish connecting + from tools.process_registry import process_registry as _pr + while self._running: + try: + # Peek for async-delegation events only; watch/completion events belong to other drains, + # so requeue anything that isn't ours. + requeue = [] + async_events = [] + while not _pr.completion_queue.empty(): + try: + evt = _pr.completion_queue.get_nowait() + except Exception: + break + if evt.get("type") == "async_delegation": + async_events.append(evt) + else: + requeue.append(evt) + for evt in requeue: + _pr.completion_queue.put(evt) + # A same-tick drain often carries several completions for the SAME session (a fan-out + # finishing together); delivering each individually floods it with N synthetic turns. + # Group by full gateway route + parent session: one consolidated turn per group. + groups: dict[tuple[str, ...], list[dict]] = {} + group_order: list[tuple[str, ...]] = [] + for evt in async_events: + self._enrich_async_delegation_routing(evt) + key = self._async_delegation_group_key(evt) + if key not in groups: + groups[key] = [] + group_order.append(key) + groups[key].append(evt) + for key in group_order: + group = groups[key] + try: + delivered = await self._deliver_async_delegation_group(group) + if delivered is False: + for evt in group: + _pr.completion_queue.put(evt) + except Exception as e: + for evt in group: + _pr.completion_queue.put(evt) + logger.error("Async delegation injection error: %s", e) + except Exception as e: + logger.debug("Async delegation watcher error: %s", e) + await asyncio.sleep(interval) + + async def _run_process_watcher(self, watcher: dict) -> None: + """Periodically check a background process and push updates to the user. + + Runs as an asyncio task. Stays silent when nothing changed. Auto-removes when the process + exits or is killed. Notification mode (``display.background_process_notifications``): + concise (default, one-line; failures append output tail) / all (running updates + final + raw output) / result (final raw only) / error (final raw only if exit != 0) / off. + """ + from gateway.run import ( + _format_concise_process_notification, + _non_conversational_metadata, + _redact_gateway_user_facing_secrets, + ) + from tools.process_registry import process_registry + + session_id = watcher["session_id"] + interval = watcher["check_interval"] + session_key = watcher.get("session_key", "") + platform_name = watcher.get("platform", "") + chat_id = watcher.get("chat_id", "") + thread_id = watcher.get("thread_id", "") + user_id = watcher.get("user_id", "") + user_name = watcher.get("user_name", "") + message_id = str(watcher.get("message_id") or "").strip() or None + agent_notify = watcher.get("notify_on_complete", False) + notify_mode = self._load_background_notifications_mode() + + logger.debug("Process watcher started: %s (every %ss, notify=%s, agent_notify=%s)", + session_id, interval, notify_mode, agent_notify) + + if notify_mode == "off" and not agent_notify: + # Still wait for the process to exit so we can log it, but don't + # push any messages to the user. + while True: + await asyncio.sleep(interval) + session = process_registry.get(session_id) + if session is None or session.exited: + break + logger.debug("Process watcher ended (silent): %s", session_id) + return + + last_output_len = 0 + while True: + await asyncio.sleep(interval) + + session = process_registry.get(session_id) + if session is None: + break + + current_output_len = len(session.output_buffer) + has_new_output = current_output_len > last_output_len + last_output_len = current_output_len + + if session.exited: + # Agent-triggered completion: inject a synthetic message unless the agent already consumed + # the result via wait/log. poll() is read-only and deliberately does NOT mark consumed — + # a status check must not suppress this delivery turn. + from tools.process_registry import format_process_notification, process_registry as _pr_check + if agent_notify and not _pr_check.is_completion_consumed(session_id): + from agent.redact import redact_terminal_output + from tools.ansi_strip import strip_ansi + _command = getattr(session, "command", "") or "" + _raw = strip_ansi(session.output_buffer) if session.output_buffer else "" + _raw = redact_terminal_output(_raw, _command) + _command = _redact_gateway_user_facing_secrets(_command) + # Truncate on line boundaries (never start mid-line): keep the last ~2000 chars + # snapped to the preceding newline, prepending a marker when output was cut. + _LIMIT = 2000 + if len(_raw) > _LIMIT: + _tail = _raw[-_LIMIT:] + _nl = _tail.find("\n") + _tail = _tail[_nl + 1:] if _nl != -1 else _tail + _out = f"[… output truncated — showing last {len(_tail)} chars]\n{_tail}" + else: + _out = _raw + _out = _redact_gateway_user_facing_secrets(_out) + completion_evt = { + "type": "completion", + "session_id": session_id, + "session_key": session_key, + "platform": platform_name, + "chat_type": watcher.get("chat_type", ""), + "chat_id": chat_id, + "thread_id": thread_id, + "user_id": user_id, + "user_name": user_name, + "message_id": message_id, + "started_at": getattr(session, "started_at", None), + "command": _command, + "exit_code": session.exit_code, + "completion_reason": getattr(session, "completion_reason", "exited"), + "termination_source": getattr(session, "termination_source", ""), + "output": _out, + # Spawning conversation's session-db id (stamped in terminal_tool); lets delivery + # pre-flight drop this completion if the user closed that session (/new) first. + "parent_session_id": ( + watcher.get("parent_session_id") + or getattr(session, "parent_session_id", "") + or "" + ), + } + synth_text = format_process_notification(completion_evt) + if not synth_text: + break + delivered = await self._enqueue_process_completion_notification( + synth_text, completion_evt, + ) + if delivered is False: + # The process remains terminal; retry after failed + # adapter injection instead of suppressing the result. + continue + break + + # Normal text-only notification. Skip when the agent already consumed this completion via + # wait/log (output returned inline) — the raw "finished" message would be a duplicate. + # The agent_notify skip FALLS THROUGH here, hence this check. poll() is read-only. + if _pr_check.is_completion_consumed(session_id): + logger.debug( + "Process watcher: completion for %s already consumed " + "via wait/log — skipping raw notification (#65379)", + session_id, + ) + break + # Decide whether to notify based on mode + should_notify = ( + notify_mode in {"concise", "all", "result"} + or (notify_mode == "error" and session.exit_code not in {0, None}) + ) + if should_notify: + new_output = session.output_buffer[-1000:] if session.output_buffer else "" + if new_output: + from agent.redact import redact_terminal_output + new_output = redact_terminal_output( + new_output, getattr(session, "command", "") or "" + ) + # redact_terminal_output() is unforced, so it returns raw text when + # security.redact_secrets is off. This goes straight to the platform + # adapter, so it needs the same unconditional floor as agent-notify. + new_output = _redact_gateway_user_facing_secrets(new_output) + if notify_mode == "concise": + _cmd_disp = _redact_gateway_user_facing_secrets( + getattr(session, "command", "") or "" + ) + _started = getattr(session, "started_at", None) + _dur = None + if isinstance(_started, (int, float)): + _dur = max(0.0, time.time() - _started) + message_text = _format_concise_process_notification( + session_id, + _cmd_disp, + session.exit_code, + new_output, + duration_seconds=_dur, + ) + else: + message_text = ( + f"[Background process {session_id} finished with exit code {session.exit_code}~ " + f"Here's the final output:\n{new_output}]" + ) + adapter = None + for p, a in self.adapters.items(): + if p.value == platform_name: + adapter = a + break + if adapter and chat_id: + try: + send_meta = {"thread_id": thread_id} if thread_id else None + await adapter.send( + chat_id, + message_text, + metadata=_non_conversational_metadata(send_meta, platform=platform_name), + ) + except Exception as e: + logger.error("Watcher delivery error: %s", e) + break + + elif has_new_output and notify_mode == "all" and not agent_notify: + # New output available -- deliver status update (only in "all" mode) + # Skip periodic updates for agent_notify watchers (they only care about completion) + new_output = session.output_buffer[-500:] if session.output_buffer else "" + if new_output: + from agent.redact import redact_terminal_output + new_output = redact_terminal_output( + new_output, getattr(session, "command", "") or "" + ) + new_output = _redact_gateway_user_facing_secrets(new_output) + message_text = ( + f"[Background process {session_id} is still running~ " + f"New output:\n{new_output}]" + ) + adapter = None + for p, a in self.adapters.items(): + if p.value == platform_name: + adapter = a + break + if adapter and chat_id: + try: + send_meta = {"thread_id": thread_id} if thread_id else None + await adapter.send( + chat_id, + message_text, + metadata=_non_conversational_metadata(send_meta, platform=platform_name), + ) + except Exception as e: + logger.error("Watcher delivery error: %s", e) + + logger.debug("Process watcher ended: %s", session_id) diff --git a/gateway/run_shutdown.py b/gateway/run_shutdown.py new file mode 100644 index 0000000000..b37646bbfb --- /dev/null +++ b/gateway/run_shutdown.py @@ -0,0 +1,2348 @@ +"""Stop/drain/restart, scale-to-zero and active-work accounting methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import os +import shlex +import sys +import threading +import time +from contextlib import suppress +from gateway.config import Platform +from gateway.restart import ( + DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, + GATEWAY_SERVICE_RESTART_EXIT_CODE, + resolve_cron_drain_budget, +) +from gateway.run_common import _UNSET +from gateway.shutdown_watchdog import arm_shutdown_watchdog, resolve_shutdown_watchdog_delay +from pathlib import Path +from typing import Any, Callable, Dict, Optional, Tuple + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayShutdownMixin: + """Stop/drain/restart, scale-to-zero and active-work accounting methods for GatewayRunner.""" + + def _active_work_count(self) -> int: + """All agent work the gateway must expose and drain as one total.""" + return ( + self._running_agent_count() + + self._active_cron_job_count() + + self._active_api_run_count() + + self._active_deferred_agent_worker_count() + ) + + def _active_cron_job_count(self) -> int: + """Count of cron jobs currently executing (``cron.scheduler._running_job_ids``). + + Cron jobs run on the scheduler's own thread pool, outside ``self._running_agents`` which + every OTHER active-work check reads; without this the shutdown drain can kill a cron job's + tool subprocess mid-run. Best-effort: returns 0 if the cron module can't be imported. + """ + try: + from cron.scheduler import get_running_job_ids + return len(get_running_job_ids()) + except Exception: + return 0 + + def _active_api_run_count(self) -> int: + """Count API-server work that is outside ``_running_agents``. + + Only the primary API server owns the HTTP listener (secondary multiplex profiles cannot + bind a port), so only the primary registry is a source of this work. + """ + try: + adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) + helper = getattr(adapter, "active_agent_work_count", None) + return max(0, int(helper())) if callable(helper) else 0 + except Exception: + return 0 + + def _interrupt_api_server_runs(self, reason: str) -> int: + """Interrupt API-server agents that are not in ``_running_agents``. + + Counterpart of ``_active_api_run_count()``: must reach the same agents when the drain times + out. Duck-typed so an adapter (or test double) without the hook is skipped, not raised on. + """ + try: + adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) + helper = getattr(adapter, "interrupt_active_runs", None) + return max(0, int(helper(reason))) if callable(helper) else 0 + except Exception as exc: + logger.debug("Failed interrupting api_server runs during shutdown: %s", exc) + return 0 + + def _active_deferred_agent_worker_count(self) -> int: + """Count executor workers that outlived their owning gateway turn. + + A timed-out hygiene compression keeps running in its executor thread. + Some paths defer agent cleanup; the live Codex path keeps its cached + agent. In both cases the turn can finish before the worker does, so + ``_running_agents`` no longer represents it. Count the worker itself. + """ + workers = getattr(self, "_deferred_agent_workers", None) + if not isinstance(workers, dict): + return 0 + return sum(1 for future in list(workers) if not future.done()) + + def _track_deferred_agent_worker( + self, + future: asyncio.Future, + agent: Any, + ) -> None: + """Expose an executor worker to drain/interrupt until it really exits.""" + workers = getattr(self, "_deferred_agent_workers", None) + if workers is None: + workers = {} + self._deferred_agent_workers = workers + workers[future] = agent + + def _discard_worker(done_future: asyncio.Future) -> None: + workers.pop(done_future, None) + # Some tracked workers intentionally outlive the coroutine that + # started them and therefore have no later waiter. Consume their + # terminal exception so asyncio does not emit an unhandled-future + # warning after the worker eventually unwinds (#98973). + if not done_future.cancelled(): + try: + done_future.exception() + except Exception: + pass + + future.add_done_callback(_discard_worker) + + def _interrupt_deferred_agent_workers(self, reason: str) -> int: + """Request cancellation of detached executor-backed agent work.""" + from gateway.run import request_hard_interrupt + workers = getattr(self, "_deferred_agent_workers", None) + if not isinstance(workers, dict): + return 0 + interrupted = 0 + seen: set[int] = set() + for future, agent in list(workers.items()): + if future.done() or agent is None or id(agent) in seen: + continue + seen.add(id(agent)) + try: + request_hard_interrupt(agent, reason) + interrupted += 1 + except Exception as exc: + logger.debug( + "Failed interrupting deferred agent worker during shutdown: %s", + exc, + ) + return interrupted + + def _scale_to_zero_has_live_background_work(self) -> bool: + """Live background work that must block a suspend. + + Backgrounded delegate_task / kanban / terminal(background=true) are NOT counted by + _running_agent_count() but suspending loses them; checks tracked tasks + process registry + + pending completion watchers. PERMANENT supervised watchers (_hermes_supervised_watcher) are + excluded — they live for the whole process (including the scale-to-zero watcher itself), so + counting them would make this True forever and the gateway could never go dormant. + """ + if any( + not t.done() and not getattr(t, "_hermes_supervised_watcher", False) + for t in self._background_tasks + ): + return True + try: + from tools.async_delegation import active_count + + if active_count() > 0: + return True + except Exception: # noqa: BLE001 - never let the idle check raise + logger.debug("scale-to-zero async-delegation check failed", exc_info=True) + try: + from tools.process_registry import process_registry + + if process_registry.has_any_active(): + return True + if process_registry.pending_watchers: + return True + except Exception: # noqa: BLE001 - never let the idle check raise + logger.debug("scale-to-zero bg-work check failed", exc_info=True) + return False + + def _scale_to_zero_idle_timeout_seconds(self) -> float: + from gateway.run import _load_gateway_config + from gateway.scale_to_zero import parse_idle_timeout_seconds + + raw = None + try: + user_cfg = _load_gateway_config() + gw = user_cfg.get("gateway") if isinstance(user_cfg, dict) else None + stz = gw.get("scale_to_zero") if isinstance(gw, dict) else None + if isinstance(stz, dict): + raw = stz.get("idle_timeout_minutes") + except Exception: # noqa: BLE001 + raw = None + return parse_idle_timeout_seconds(raw) + + def _restart_loop_guard_config(self) -> tuple: + """Return ``(max_restarts, window_seconds, max_gap_seconds)`` for the auto-resume + restart-loop breaker, from ``gateway.restart_loop_guard`` with module defaults as fallback. + + ``max_restarts <= 0`` disables the breaker. ``max_gap_seconds`` is the longest spacing + between consecutive restart-interrupted boots that still counts as the same loop, so a + crash cycle slower than ``window_seconds`` stays visible. + """ + from gateway.run import _load_gateway_config + from gateway import restart_loop_guard as _rlg + + max_restarts = _rlg.DEFAULT_MAX_RESTARTS + window_seconds = _rlg.DEFAULT_WINDOW_SECONDS + max_gap_seconds = _rlg.DEFAULT_MAX_GAP_SECONDS + try: + user_cfg = _load_gateway_config() + gw = user_cfg.get("gateway") if isinstance(user_cfg, dict) else None + rlg = gw.get("restart_loop_guard") if isinstance(gw, dict) else None + if isinstance(rlg, dict): + if isinstance(rlg.get("max_restarts"), int): + max_restarts = rlg["max_restarts"] + if isinstance(rlg.get("window_seconds"), int) and rlg["window_seconds"] > 0: + window_seconds = rlg["window_seconds"] + if ( + isinstance(rlg.get("max_gap_seconds"), int) + and rlg["max_gap_seconds"] > 0 + ): + max_gap_seconds = rlg["max_gap_seconds"] + except Exception: # noqa: BLE001 + pass + return max_restarts, window_seconds, max_gap_seconds + + def _scale_to_zero_active_messaging_platforms(self) -> list: + """ENABLED platforms that count for the relay-only arm gate. + + Two load-bearing filters: enabled only (config.platforms is pre-seeded with disabled + placeholders for the whole catalog) and MESSAGING only (the api_server is a loopback listener + force-enabled on every hosted container with no outbound socket; counting it silently + disarmed the feature everywhere). Mirrors the non-messaging exclusion in _connect_platforms. + """ + if not self.config: + return [] + non_messaging = {Platform.LOCAL, Platform.API_SERVER, Platform.WEBHOOK} + try: + return [ + p + for p, pc in self.config.platforms.items() + if getattr(pc, "enabled", False) and p not in non_messaging + ] + except Exception: # noqa: BLE001 + return [] + + def _scale_to_zero_should_arm(self) -> bool: + """Whether to start the idle watcher (D1/D11/§3.4(1)).""" + from gateway.relay import relay_wake_url + from gateway.scale_to_zero import ( + messaging_is_relay_only_or_absent, + scale_to_zero_enabled, + should_arm, + ) + + platforms = self._scale_to_zero_active_messaging_platforms() + try: + wake_url = relay_wake_url() + except Exception: # noqa: BLE001 + wake_url = None + return should_arm( + enabled=scale_to_zero_enabled(), + relay_only_or_absent=messaging_is_relay_only_or_absent(platforms), + wake_url=wake_url, + ) + + def _log_scale_to_zero_not_armed_reason(self) -> None: + """Log why the idle watcher did NOT arm — but only for an OPTED-IN instance. + + A non-opted instance (no HERMES_SCALE_TO_ZERO stamp) not arming is normal and stays silent; + with the stamp set, the surprise earns one INFO line so the answer is a log grep. + """ + from gateway.relay import relay_wake_url + from gateway.scale_to_zero import ( + messaging_is_relay_only_or_absent, + scale_to_zero_enabled, + ) + + try: + enabled = scale_to_zero_enabled() + if not enabled: + return # not opted in — normal, stay quiet + active = [ + getattr(p, "value", p) + for p in self._scale_to_zero_active_messaging_platforms() + ] + relay_only = messaging_is_relay_only_or_absent(active) + try: + wake_url = relay_wake_url() + except Exception: # noqa: BLE001 + wake_url = None + logger.info( + "scale-to-zero: NOT armed despite opt-in — " + "relay_only_or_absent=%s (enabled platforms=%s), wake_url=%s. " + "Need relay-only messaging + a registered wake URL.", + relay_only, + active or "none", + "set" if wake_url else "MISSING", + ) + except Exception: # noqa: BLE001 - diagnostics must never block startup + logger.debug("scale-to-zero: not-armed reason logging failed", exc_info=True) + + def _scale_to_zero_is_idle(self) -> bool: + from gateway.scale_to_zero import is_idle + + # The FULL work aggregate, not _running_agent_count(): cron jobs and API-server runs live + # outside _running_agents, so counting agents alone let a suspend land mid-cron-job. + # Fail-AWAKE accounting: the shutdown-drain counters swallow exceptions to 0, which is fine + # for a drain but unsafe for a suspend predicate (a transient read failure would look idle). + # Here an unreadable source counts as work (sentinel 1) so the machine stays awake. + try: + from cron.scheduler import get_running_job_ids + + cron_count = len(get_running_job_ids()) + except Exception: # noqa: BLE001 - unreadable source => assume busy + logger.debug("scale-to-zero: cron work count unreadable — staying awake", exc_info=True) + cron_count = 1 + try: + adapter = getattr(self, "adapters", {}).get(Platform.API_SERVER) + helper = getattr(adapter, "active_agent_work_count", None) + api_count = max(0, int(helper())) if callable(helper) else 0 + except Exception: # noqa: BLE001 - unreadable source => assume busy + logger.debug("scale-to-zero: api work count unreadable — staying awake", exc_info=True) + api_count = 1 + # An attached dashboard/desktop/TUI client is inbound activity too; it lives in the DASHBOARD + # process and reaches us as a file mtime refreshed on every WS frame (gateway/scale_to_zero.py). + # Folded into the inbound clock rather than a conjunct: same idle_timeout grace after + # disconnect as a chat message, and a lingering marker cannot pin the box. + last_inbound = self._last_inbound_at + try: + from gateway.scale_to_zero import dashboard_client_last_seen + + seen = dashboard_client_last_seen() + except Exception: # noqa: BLE001 - unreadable source => assume busy + logger.debug("scale-to-zero: dashboard heartbeat unreadable — staying awake", exc_info=True) + seen = time.time() + if seen is not None and seen > last_inbound: + last_inbound = seen + return is_idle( + active_work_count=self._running_agent_count() + cron_count + api_count, + seconds_since_last_inbound=time.time() - last_inbound, + idle_timeout_seconds=self._scale_to_zero_idle_timeout_seconds(), + has_live_background_work=self._scale_to_zero_has_live_background_work(), + ) + + def _scale_to_zero_note_real_inbound(self) -> None: + """Stamp real inbound and restore lifecycle after a dormant wake. + + Dormancy marks status `draining` but is not the stop/restart drain: the process stays alive + and should present as running once real traffic wakes it. Internal completion/replay events + deliberately do not call this, so they don't keep an idle gateway awake. + """ + self._last_inbound_at = time.time() + if getattr(self, "_scale_to_zero_cooldown_until", 0.0) > 0: + try: + self._update_runtime_status("running") + except Exception: # noqa: BLE001 - status restoration is best-effort + logger.debug("scale-to-zero: status restore failed", exc_info=True) + self._scale_to_zero_cooldown_until = 0.0 + + def _relay_adapter_for_dormancy(self): + """Return the connected RELAY adapter, if any (the one go_dormant targets).""" + try: + from gateway.platforms.base import Platform + except Exception: # noqa: BLE001 + return None + return self.adapters.get(Platform.RELAY) + + async def _scale_to_zero_watcher(self, interval: float = 30.0) -> None: + """Watch for idle, drive the relay dormant, then self-suspend the machine. + + Armed ONLY via _scale_to_zero_should_arm() (HERMES_SCALE_TO_ZERO stamp + relay-only/absent + messaging + wakeUrl). On sustained idle: mark status `draining` (NOT _running=False), relay + adapter.go_dormant() (supervisor-preserving socket close, NOT disconnect()), NO + mark_resume_pending (suspend preserves RAM), THEN suspend via the local flaps socket. The + gateway owns the suspend because Fly autostop sees only INBOUND connections and would freeze + mid-job or before the relay flip (machines run autostop:"off"); autostart stays platform-side. + A re-arm cooldown keeps a wake's drained backlog from being re-quiesced. Off-Fly (no flaps + socket) the watcher does not quiesce at all. + """ + await asyncio.sleep(min(interval, 30.0)) # let startup settle + while self._running: + try: + await asyncio.sleep(interval) + if not self._running: + return + if time.time() < self._scale_to_zero_cooldown_until: + continue + if not self._scale_to_zero_is_idle(): + continue + adapter = self._relay_adapter_for_dormancy() + if adapter is None: + continue + go_dormant = getattr(adapter, "go_dormant", None) + if not callable(go_dormant): + continue + # Quiesce only when a suspend can follow. Off-Fly the platform owns the freeze and + # go_dormant()'s socket close arms the reconnect supervisor (re-dial ~1.4s, unflipped + # at freeze, inbound dropped not buffered); stay connected, orphan detection adopts it. + from gateway.scale_to_zero import self_suspend_available + + if not self_suspend_available(): + if not self._scale_to_zero_no_suspend_logged: + self._scale_to_zero_no_suspend_logged = True + logger.info( + "scale-to-zero: idle, but this platform suspends on " + "its own timer (no in-machine suspend API); staying " + "connected rather than quiescing" + ) + continue + logger.info( + "scale-to-zero: gateway idle for >= %.0fs — going dormant " + "(relay buffered, socket closed) then self-suspending", + self._scale_to_zero_idle_timeout_seconds(), + ) + try: + self._update_runtime_status("draining") + except Exception: # noqa: BLE001 - status is best-effort + logger.debug("scale-to-zero: status mark failed", exc_info=True) + dormant_ok = True + try: + result = go_dormant() + if asyncio.iscoroutine(result): + await result + except Exception: # noqa: BLE001 - dormancy is best-effort + dormant_ok = False + logger.debug("scale-to-zero: go_dormant failed", exc_info=True) + # After a wake the drained inbound updates _last_inbound_at; give it a window so we + # don't immediately re-go-dormant on the same idle reading before traffic lands. + self._scale_to_zero_cooldown_until = time.time() + max(interval, 60.0) + # Self-suspend ONLY after a clean quiesce: the relay flip (buffered delivery + wake + # poke armed) must be set before the freeze, or inbound black-holes while we sleep. + # Re-check idle one last time — inbound may have landed during the quiesce await. + if not dormant_ok: + continue + if not self._scale_to_zero_is_idle(): + logger.info( + "scale-to-zero: inbound arrived during quiesce — skipping suspend" + ) + continue + await self._scale_to_zero_self_suspend() + except asyncio.CancelledError: + raise + except Exception: # noqa: BLE001 - the watcher must never crash the gateway + logger.debug("scale-to-zero watcher iteration error", exc_info=True) + + async def _scale_to_zero_self_suspend(self) -> None: + """Suspend this Fly machine via the local flaps socket (fail-awake). + + Blocking unix-socket call runs in a worker thread so the loop stays live until the kernel + freeze; nothing meaningful runs until wake. Off-Fly this is a silent no-op. + """ + from gateway.scale_to_zero import self_suspend_available, suspend_self + + try: + if not self_suspend_available(): + logger.debug( + "scale-to-zero: flaps socket / machine identity absent — " + "dormant without platform suspend" + ) + return + accepted = await asyncio.to_thread(suspend_self) + if not accepted: + logger.warning( + "scale-to-zero: self-suspend not accepted — machine stays " + "awake (fail-awake); will retry on the next idle window" + ) + except Exception: # noqa: BLE001 - suspend is best-effort, never crash + logger.debug("scale-to-zero: self-suspend failed", exc_info=True) + + # ------------------------------------------------------------------ + # External drain control (NAS-driven quiesce-without-restart). The dashboard's + # begin/cancel-drain endpoint writes/removes the ``.drain_request.json`` marker + # (gateway/drain_control.py); this watcher flips the gateway between accepting and refusing + # NEW turns WITHOUT exiting. Reversible: NAS begins drain, polls /api/status until + # active_agents hits 0, acts; on cancel/abort the marker is removed and turns resume. + # ------------------------------------------------------------------ + def _enter_external_drain(self) -> None: + """Begin external drain: refuse NEW turns (in-flight ones are NOT interrupted), flip state. + + Idempotent: re-entry only re-writes status. + """ + if self._external_drain_active: + return + self._external_drain_active = True + logger.info( + "External drain ENGAGED (.drain_request.json present) — refusing " + "new turns; %d in-flight turn(s) will finish. Process stays up.", + self._active_work_count(), + ) + # Flip persisted lifecycle state so /api/status.gateway_busy / gateway_drainable track the + # drain; active_agents is preserved (read-merge keeps the live count), only state changes. + self._update_runtime_status("draining") + + def _exit_external_drain(self) -> None: + """Cancel external drain: revert state, re-accept new turns. + + Idempotent. Reverts to ``running`` only when actually mid-drain AND not shutting down — + a real shutdown ``_draining`` must win; never resurrect a stopping gateway. + """ + if not self._external_drain_active: + return + self._external_drain_active = False + if self._draining or not self._running: + # A shutdown drain is in progress / the loop has stopped — do not + # clobber the terminal state back to running. + logger.info( + "External drain marker cleared during shutdown — not reverting " + "to running (shutdown takes precedence)." + ) + return + logger.info( + "External drain RELEASED (.drain_request.json removed) — " + "re-accepting new turns; gateway_state -> running." + ) + self._update_runtime_status("running") + + async def _drain_control_watcher(self, interval: float = 1.0) -> None: + """Background task: reconcile gateway accept-state with the drain marker. + + Polls ``.drain_request.json`` (presence-based) at 1s: present -> enter drain, absent -> exit; + reconciles once at startup. A marker from a PRIOR instantiation epoch (survived a machine + restart) is treated as absent. Best-effort: tick errors are logged and the loop continues. + """ + from gateway.drain_control import drain_requested + + while self._running: + try: + # drain_requested() does a synchronous read_text() on the marker file: at 1s cadence + # that is a blocking disk read on the event loop ~86k times/day, and under host I/O + # pressure one read can stall 30s+ and take every platform heartbeat down. Off-thread it. + if await asyncio.to_thread(drain_requested): + self._enter_external_drain() + # API and cron work live outside messaging's _running_agents map; refresh the + # aggregate while an external caller polls this reversible drain state. + self._persist_active_agents() + else: + self._exit_external_drain() + except asyncio.CancelledError: + raise + except Exception as exc: + logger.debug("Drain-control watcher tick error: %s", exc, exc_info=True) + await asyncio.sleep(interval) + + def _update_platform_runtime_status( + self, + platform: str, + *, + platform_state: Optional[str] = None, + error_code: Optional[str] = None, + error_message: Optional[str] = None, + needs_attention: Optional[bool] = None, + retrying_since: Any = _UNSET, + ) -> None: + try: + from gateway.status import write_runtime_status + extra: Dict[str, Any] = {} + if needs_attention is not None: + extra["needs_attention"] = needs_attention + if retrying_since is not _UNSET: + extra["retrying_since"] = retrying_since + write_runtime_status( + platform=platform, + platform_state=platform_state, + error_code=error_code, + error_message=error_message, + **extra, + ) + except Exception: + pass + + # ------------------------------------------------------------------ + # Per-platform circuit breaker (pause/resume): reconnect watcher + /platform pause|resume. + # ------------------------------------------------------------------ + def _pause_failed_platform(self, platform, *, reason: str = "") -> None: + """Mark a queued platform as paused — stays in ``_failed_platforms`` but the reconnect + watcher stops hammering it. + + Manual (``/platform pause ``) only: the watcher never auto-pauses — retryable failures + keep retrying at the backoff cap so a transient outage self-heals. + """ + info = getattr(self, "_failed_platforms", {}).get(platform) + if info is None: + return + if info.get("paused"): + return + info["paused"] = True + info["pause_reason"] = reason or "auto-paused after repeated failures" + # Push next_retry far enough out that even if "paused" is missed + # by a stale code path, the watcher won't fire on it. + info["next_retry"] = float("inf") + with suppress(Exception): + self._update_platform_runtime_status( + platform.value, + platform_state="paused", + error_code=None, + error_message=info["pause_reason"], + ) + logger.warning( + "%s paused after %d consecutive failures (%s) — " + "fix the underlying issue then run `/platform resume %s` " + "to retry, or `hermes gateway restart` to restart the gateway.", + platform.value, info.get("attempts", 0), + info["pause_reason"], platform.value, + ) + + def _resume_paused_platform(self, platform) -> bool: + """Unpause a platform — reset its attempt counter and schedule an + immediate retry. Returns True if the platform was paused and is + now queued; False if it wasn't paused (or wasn't in the queue). + """ + info = getattr(self, "_failed_platforms", {}).get(platform) + if info is None: + return False + if not info.get("paused"): + return False + info["paused"] = False + info.pop("pause_reason", None) + info["attempts"] = 0 + info["next_retry"] = time.monotonic() # retry on next watcher tick + with suppress(Exception): + self._update_platform_runtime_status( + platform.value, + platform_state="retrying", + error_code=None, + error_message=None, + ) + logger.info("%s resumed — retrying on next watcher tick", platform.value) + return True + + async def _drain_active_agents( + self, timeout: float, cron_timeout: Optional[float] = None + ) -> tuple[Dict[str, Any], bool]: + snapshot = self._snapshot_running_agents() + last_active_count = self._running_agent_count() + last_cron_count = self._active_cron_job_count() + last_api_count = self._active_api_run_count() + last_deferred_count = self._active_deferred_agent_worker_count() + last_status_at = 0.0 + + def _maybe_update_status(force: bool = False) -> None: + nonlocal last_active_count, last_cron_count, last_api_count + nonlocal last_deferred_count, last_status_at + now = asyncio.get_running_loop().time() + active_count = self._running_agent_count() + cron_count = self._active_cron_job_count() + api_count = self._active_api_run_count() + deferred_count = self._active_deferred_agent_worker_count() + if ( + force + or active_count != last_active_count + or cron_count != last_cron_count + or api_count != last_api_count + or deferred_count != last_deferred_count + or (now - last_status_at) >= 1.0 + ): + self._update_runtime_status("draining") + last_active_count = active_count + last_cron_count = cron_count + last_api_count = api_count + last_deferred_count = deferred_count + last_status_at = now + + # Cron jobs run on the scheduler's pool, outside ``self._running_agents`` — fold their in-flight + # count into this wait, or a cron job's tool work is killed without warning once it's the only + # active thing running. API-server/desk sessions and detached deferred workers share the gap. + if ( + not self._running_agents + and last_cron_count == 0 + and last_api_count == 0 + and last_deferred_count == 0 + ): + _maybe_update_status(force=True) + return snapshot, False + + _maybe_update_status(force=True) + + # Cron drains on its own deadline: ``timeout`` (``restart_drain_timeout``) defaults to 0 since + # an interrupted chat turn is announced and resumable, while a cron run killed mid-flight is a + # permanent failure nobody is waiting on. One shared budget would kill cron after 0.00s. + loop = asyncio.get_running_loop() + started = loop.time() + deadline = started + timeout + cron_deadline = started + (timeout if cron_timeout is None else cron_timeout) + + def _still_draining() -> bool: + now = loop.time() + if ( + len(self._running_agents) + or self._active_api_run_count() + or self._active_deferred_agent_worker_count() + ) and now < deadline: + return True + return bool(self._active_cron_job_count()) and now < cron_deadline + + # Both budgets at 0 leave this loop unentered ("interrupt immediately") as an expired deadline, + # not a special case, so timed_out below is always computed from real state. + while _still_draining(): + _maybe_update_status() + await asyncio.sleep(0.1) + timed_out = ( + bool(len(self._running_agents)) + or bool(self._active_cron_job_count()) + or bool(self._active_api_run_count()) + or bool(self._active_deferred_agent_worker_count()) + ) + _maybe_update_status(force=True) + return snapshot, timed_out + + def _interrupt_running_agents(self, reason: str) -> None: + from gateway.run import _AGENT_PENDING_SENTINEL, request_hard_interrupt + for session_key, agent in list(self._running_agents.items()): + if agent is _AGENT_PENDING_SENTINEL: + continue + try: + request_hard_interrupt(agent, reason) + logger.debug("Interrupted running agent for session %s during shutdown", session_key) + except Exception as e: + logger.debug("Failed interrupting agent during shutdown: %s", e) + # API-server / desk turns are adapter-owned and never enter _running_agents, so the loop above + # cannot see them even though _drain_active_agents() waited for them. + interrupted_api = self._interrupt_api_server_runs(reason) + if interrupted_api: + logger.debug("Interrupted %d api_server run(s) during shutdown", interrupted_api) + interrupted_deferred = self._interrupt_deferred_agent_workers(reason) + if interrupted_deferred: + logger.debug( + "Interrupted %d deferred agent worker(s) during shutdown", + interrupted_deferred, + ) + + async def _notify_interrupted_cron_jobs(self, job_ids) -> int: + """Tell the owner of each just-interrupted cron job that its run died. + + The cron worker can't: its thread reaches ``_deliver_result`` after teardown closed the + transport. Must run post-interrupt while adapters are still connected (the window + ``_notify_active_sessions_of_shutdown`` uses, which is blind to cron work). Best-effort: every + failure is swallowed so a wedged adapter can't extend shutdown. Returns notices sent. + """ + if not job_ids: + return 0 + try: + from cron.jobs import get_job + from cron.scheduler import _resolve_delivery_targets + except Exception as e: + logger.debug("Cron interrupt notification unavailable: %s", e) + return 0 + + action = "restarting" if self._restart_requested else "shutting down" + notified: set = set() + for job_id in job_ids: + try: + job = get_job(job_id) + if not job: + continue + # deliver=local jobs, and deliver=origin jobs with no resolvable origin, resolve to zero + # targets and must stay silent rather than fall back to a home channel. Interrupted + # notices are failure-category engine status, so they honor failure_deliver. + targets = _resolve_delivery_targets(job, for_failure=True) + except Exception as e: + logger.debug("Cron interrupt targets unresolved for %s: %s", job_id, e) + continue + if not targets: + continue + + msg = ( + f"⚠️ Cron job '{job.get('name') or job_id}' was interrupted — " + f"the gateway is {action} and killed the run before it " + "finished. No result was produced for this run." + ) + for target in targets: + try: + platform = Platform(str(target.get("platform", "")).lower()) + except Exception: + continue + adapter = self.adapters.get(platform) + if adapter is None: + continue + platform_cfg = self.config.platforms.get(platform) + if platform_cfg is not None and not platform_cfg.gateway_restart_notification: + continue + + chat_id = str(target.get("chat_id")) + thread_id = target.get("thread_id") + dedup_key = ( + job_id, + platform.value, + chat_id, + str(thread_id) if thread_id else None, + ) + if dedup_key in notified: + continue + try: + metadata = self._thread_metadata_for_target( + platform, chat_id, thread_id, adapter=adapter + ) + result = await adapter.send(chat_id, msg, metadata=metadata) + if result is not None and getattr(result, "success", True) is False: + logger.debug( + "Cron interrupt notice to %s:%s failed: %s", + platform.value, chat_id, + getattr(result, "error", "send returned success=False"), + ) + continue + notified.add(dedup_key) + except Exception as e: + logger.debug( + "Cron interrupt notice to %s:%s raised: %s", + platform.value, chat_id, e, + ) + if notified: + logger.info( + "Shutdown: delivered %d interrupted-cron-job notice(s)", + len(notified), + ) + return len(notified) + + async def _notify_active_sessions_of_shutdown(self) -> None: + """Send shutdown/restart notifications to active chats and home channels. + + Called at the start of stop() while adapters are connected; send failures never block shutdown. + """ + from gateway.run import _parse_session_key + active = self._snapshot_running_agents() + restart_source = self._restart_command_source if self._restart_requested else None + + action = "restarting" if self._restart_requested else "shutting down" + hint = ( + "Your current task will be interrupted. " + "Send any message after restart and I'll try to resume where you left off." + if self._restart_requested + else "Your current task will be interrupted." + ) + msg = f"⚠️ Gateway {action} — {hint}" + + notified: set[tuple[str, str, Optional[str]]] = set() + for session_key in active: + source = None + try: + if getattr(self, "session_store", None) is not None: + await self.async_session_store._ensure_loaded() + entry = self.session_store._entries.get(session_key) + source = getattr(entry, "origin", None) if entry else None + except Exception as e: + logger.debug( + "Failed to load session origin for shutdown notification %s: %s", + session_key, + e, + ) + + if source is None: + source = self._get_cached_session_source(session_key) + + if source is not None: + platform_str = source.platform.value + chat_id = str(source.chat_id) + thread_id = source.thread_id + else: + # Fall back to parsing the session key when no persisted + # origin is available (legacy sessions/tests). + _parsed = _parse_session_key(session_key) + if not _parsed: + continue + platform_str = _parsed["platform"] + chat_id = _parsed["chat_id"] + thread_id = _parsed.get("thread_id") + + # Dedupe only identical targets: thread/topic platforms share a parent chat yet route to + # distinct destinations via metadata. + dedup_key = (platform_str, chat_id, str(thread_id) if thread_id else None) + if dedup_key in notified: + continue + + try: + platform = Platform(platform_str) + adapter = self.adapters.get(platform) + if not adapter: + continue + + platform_cfg = self.config.platforms.get(platform) + if platform_cfg is not None and not platform_cfg.gateway_restart_notification: + logger.info( + "Shutdown notification suppressed for active session: %s has gateway_restart_notification=false", + platform_str, + ) + continue + + reply_to_message_id = getattr(source, "message_id", None) if source is not None else None + if reply_to_message_id is None and restart_source is not None: + try: + restart_platform = restart_source.platform.value + restart_chat_id = str(restart_source.chat_id) + restart_thread_id = str(restart_source.thread_id) if restart_source.thread_id else None + if (restart_platform, restart_chat_id, restart_thread_id) == dedup_key: + reply_to_message_id = getattr(restart_source, "message_id", None) + except Exception: + pass + + metadata = self._thread_metadata_for_target( + platform, + chat_id, + thread_id, + chat_type=getattr(source, "chat_type", None) if source is not None else None, + reply_to_message_id=reply_to_message_id, + adapter=adapter, + ) + + result = await adapter.send(chat_id, msg, metadata=metadata) + if result is not None and getattr(result, "success", True) is False: + logger.debug( + "Failed to send shutdown notification to %s:%s: %s", + platform_str, + chat_id, + getattr(result, "error", "send returned success=False"), + ) + continue + + notified.add(dedup_key) + logger.info( + "Sent shutdown notification to active chat %s:%s", + platform_str, chat_id, + ) + except Exception as e: + logger.debug( + "Failed to send shutdown notification to %s:%s: %s", + platform_str, chat_id, e, + ) + + if self._restart_requested and restart_source is not None: + logger.debug("Skipping home-channel shutdown notifications for in-chat restart") + return + + # Suppress ONLY the home-channel broadcast when the drain asked to be quiet (e.g. routine + # auto-update on an always-on fleet). Per-session interrupt pings above are NOT gated: empty by + # construction on a drained shutdown, and useful ("task cut off, message me to resume") on a + # force-interrupt. Honoured only for a CURRENT-epoch marker (staleness check inside + # drain_notification_suppressed), so an orphaned marker can't silence a fresh gateway. + try: + from gateway.drain_control import drain_notification_suppressed + if drain_notification_suppressed(): + logger.info( + "Home-channel shutdown broadcast suppressed by drain marker " + "(suppress_notification=true)" + ) + return + except Exception as e: + # Never let the suppression check block the shutdown broadcast — + # fail toward the louder, more-visible behaviour. + logger.debug("drain_notification_suppressed check failed: %s", e) + + # Snapshot adapters: adapter.send() can hit a fatal path (_handle_fatal) that pops the adapter + # from self.adapters -> ``RuntimeError: dictionary changed size during iteration``. + for platform, adapter in list(self.adapters.items()): + home = self.config.get_home_channel(platform) + if not home or not home.chat_id: + continue + + platform_cfg = self.config.platforms.get(platform) + if platform_cfg is not None and not platform_cfg.gateway_restart_notification: + logger.info( + "Shutdown notification suppressed for home channel: %s has gateway_restart_notification=false", + platform.value, + ) + continue + + dedup_key = (platform.value, str(home.chat_id), str(home.thread_id) if home.thread_id else None) + if dedup_key in notified: + continue + + try: + metadata = self._thread_metadata_for_target( + platform, + home.chat_id, + home.thread_id, + adapter=adapter, + ) + if metadata: + result = await adapter.send(str(home.chat_id), msg, metadata=metadata) + else: + result = await adapter.send(str(home.chat_id), msg) + if result is not None and getattr(result, "success", True) is False: + logger.debug( + "Failed to send shutdown notification to home channel %s:%s: %s", + platform.value, + home.chat_id, + getattr(result, "error", "send returned success=False"), + ) + continue + + notified.add(dedup_key) + logger.info( + "Sent shutdown notification to home channel %s:%s", + platform.value, + home.chat_id, + ) + except Exception as e: + logger.debug( + "Failed to send shutdown notification to home channel %s:%s: %s", + platform.value, + home.chat_id, + e, + ) + + async def _finalize_shutdown_agents(self, active_agents: Dict[str, Any]) -> None: + for agent in active_agents.values(): + # Persist in-flight transcripts before teardown: a force-interrupted agent may never reach + # finalize_turn (the only mid-turn flush), so its tool rounds would vanish from + # load_transcript() on resume (resume already tolerates a pending-tool-result tail). The + # flush is idempotent (identity-tracked); gracefully finished agents re-flush nothing. + try: + _flush = getattr(agent, "_flush_messages_to_session_db", None) + _session_messages = getattr(agent, "_session_messages", None) + if callable(_flush) and isinstance(_session_messages, list) and _session_messages: + # Strip empty-response retry scaffolding from the tail first (as ``_persist_session`` + # does) so a resumed turn doesn't replay synthetic recovery nudges. + _strip = getattr( + agent, "_drop_trailing_empty_response_scaffolding", None + ) + if callable(_strip): + with suppress(Exception): + _strip(_session_messages) + try: + _flush(_session_messages) + except Exception as _flush_err: + # Transcript could not be persisted (e.g. FTS/SQLite index corruption). A log + # line alone loses the conversation at exit, so dump the live history to an + # external JSON recovery snapshot. Non-fatal: shutdown never blocks on a backup. + logger.warning( + "Shutdown transcript flush failed (%s); preserving " + "%d in-memory message(s) to recovery snapshot", + _flush_err, + len(_session_messages), + ) + from gateway.shutdown_flush import flush_agent_history_to_file + flush_agent_history_to_file( + getattr(agent, "session_id", None), + _session_messages, + ) + except Exception as _e: + logger.debug("Shutdown transcript flush failed: %s", _e) + # Off-loop + bounded: plugin on_session_finalize hooks can do arbitrary synchronous work + # (e.g. a full-session trace export) — same hang class as the memory provider below. + await self._finalize_session_off_loop( + session_id=getattr(agent, "session_id", None), + platform="gateway", + reason="shutdown", + ) + # Off-loop + bounded: a wedged memory provider here used to hang + # the whole shutdown so SIGTERM never completed (#53175). + await self._cleanup_agent_resources_off_loop( + agent, context="shutdown finalize" + ) + + def _should_emit_long_running_notification( + self, + session_key: Optional[str], + agent: Any, + executor_task: Optional[Any], + ) -> bool: + """Only emit the heartbeat while this task still owns the live run. + + Stop once the executor finishes, the agent is gone, or the session key was rebound (e.g. + ``/new`` mid-run) — else a stale ``running: delegate_task`` heartbeat outlives its run. + """ + if agent is None: + return False + if executor_task is not None and executor_task.done(): + return False + if session_key: + _hb_state = self._peek_session_state(session_key) + if (_hb_state.turn.agent if _hb_state else None) is not agent: + return False + return True + + def _defer_agent_cleanup_until_future_done( + self, + future: asyncio.Future, + agent: Any, + *, + context: str, + ) -> None: + """Clean up ``agent`` only after its executor future has finished. + + A timed-out executor call keeps running in its worker thread; closing the agent first can + tear down clients it still uses, so hold a strong task ref and await the real future. + """ + + async def _cleanup_when_done() -> None: + try: + await asyncio.shield(future) + except asyncio.CancelledError: + # Loop shutdown can cancel this waiter while the executor still + # runs. Never turn that cancellation into premature cleanup. + return + except Exception as exc: + logger.debug( + "Deferred agent worker%s finished with an error: %s", + f" ({context})" if context else "", + exc, + ) + await self._cleanup_agent_resources_off_loop(agent, context=context) + + self._track_deferred_agent_worker(future, agent) + + task = asyncio.create_task(_cleanup_when_done()) + tasks = getattr(self, "_deferred_agent_cleanup_tasks", None) + if tasks is None: + tasks = set() + self._deferred_agent_cleanup_tasks = tasks + tasks.add(task) + task.add_done_callback(tasks.discard) + + async def _finalize_session_off_loop( + self, + *, + session_id: Any, + platform: str, + reason: str, + **extra: Any, + ) -> None: + """Run hermes_cli.lifecycle.finalize_session off the event loop, bounded. + + On timeout the worker thread is left to finish (or leak) and the caller proceeds. + """ + + def _call() -> None: + from hermes_cli.lifecycle import finalize_session + + finalize_session( + session_id=session_id, + platform=platform, + reason=reason, + **extra, + ) + + try: + await asyncio.wait_for( + self._run_in_executor_with_context(_call), + timeout=self._FINALIZE_TIMEOUT_S, + ) + except asyncio.TimeoutError: + logger.warning( + "Session finalize hooks (%s, reason=%s) exceeded %ss; " + "proceeding without blocking the event loop (the worker " + "thread is left to finish on its own).", + session_id, + reason, + self._FINALIZE_TIMEOUT_S, + ) + except Exception as finalize_exc: + logger.debug( + "Session finalize hooks (%s, reason=%s) failed: %s", + session_id, + reason, + finalize_exc, + ) + + async def _cleanup_agent_resources_off_loop( + self, agent: Any, *, context: str = "" + ) -> None: + """Run _cleanup_agent_resources in a worker thread with a bounded wait. + + On timeout the worker thread is left to finish (or leak) and the caller proceeds, as /new does. + """ + if agent is None: + return + if context.startswith("shutdown") or context == "session expiry": + with suppress(Exception): + agent._end_session_on_close = False + try: + await asyncio.wait_for( + self._run_in_executor_with_context( + self._cleanup_agent_resources, agent + ), + timeout=self._CLEANUP_TIMEOUT_S, + ) + except asyncio.TimeoutError: + logger.warning( + "Agent resource cleanup%s exceeded %ss; proceeding without " + "blocking the event loop (the worker thread is left to finish " + "on its own). (#53175)", + f" ({context})" if context else "", + self._CLEANUP_TIMEOUT_S, + ) + except Exception as cleanup_exc: + logger.warning( + "Agent resource cleanup%s failed: %s (#53175)", + f" ({context})" if context else "", + cleanup_exc, + ) + + def _cleanup_agent_resources(self, agent: Any) -> None: + """Best-effort cleanup for temporary or cached agent instances.""" + if agent is None: + return + try: + if hasattr(agent, "shutdown_memory_provider"): + # Drain queued memory writes BEFORE tearing the provider down: shutdown_all() gives + # the serialized memory worker only ~5s and cancels the rest, so a /reset or rotation + # could drop handed-off writes and the next session loads stale memory. Bounded head + # start via the manager's own barrier (mirrors CLI exit); a failure never blocks teardown. + _mm = getattr(agent, "_memory_manager", None) + if _mm is not None and hasattr(_mm, "flush_pending"): + with suppress(Exception): + _mm.flush_pending(timeout=10) + # Pass the real transcript so ``on_session_end`` hooks don't see the empty default. + # ``_session_messages`` may be absent on ``object.__new__`` test stubs, hence getattr. + session_messages = getattr(agent, "_session_messages", None) + if isinstance(session_messages, list): + agent.shutdown_memory_provider(session_messages) + else: + agent.shutdown_memory_provider() + except Exception: + pass + # Close tool resources (sandboxes, browser daemons, background processes, httpx clients). + try: + if hasattr(agent, "close"): + agent.close() + except Exception: + pass + # Auxiliary async clients live in a process-global cache created from worker threads; drop + # entries whose event loop is dead so httpx transports don't accumulate across turns. + try: + from agent.auxiliary_client import cleanup_stale_async_clients + cleanup_stale_async_clients() + except Exception: + pass + + def _increment_restart_failure_counts(self, active_session_keys: set) -> None: + """Increment restart-failure counters for sessions active at shutdown. + + Persists to a JSON file so counters survive across restarts. Sessions NOT in + active_session_keys are removed (they completed successfully, so the loop is broken). + """ + from gateway.run import _hermes_home, atomic_json_write + import json + + path = _hermes_home / self._STUCK_LOOP_FILE + try: + counts = json.loads(path.read_text(encoding="utf-8")) if path.exists() else {} + except Exception: + counts = {} + + # Increment active sessions, remove inactive ones (loop broken) + new_counts = {} + for key in active_session_keys: + new_counts[key] = counts.get(key, 0) + 1 + # Keep any entries that are still above 0 even if not active now + # (they might become active again next restart) + + with suppress(Exception): + atomic_json_write(path, new_counts, indent=None) + + def _suspend_stuck_loop_sessions(self) -> int: + """Suspend sessions active across too many restarts (load → stuck → restart loop). + + Runs at startup AFTER suspend_recently_active(). Returns the number suspended. + """ + from gateway.run import _hermes_home + import json + + path = _hermes_home / self._STUCK_LOOP_FILE + if not path.exists(): + return 0 + + try: + counts = json.loads(path.read_text(encoding="utf-8")) + except Exception: + return 0 + + suspended = 0 + stuck_keys = [k for k, v in counts.items() if v >= self._STUCK_LOOP_THRESHOLD] + + for session_key in stuck_keys: + try: + entry = self.session_store._entries.get(session_key) + if entry and not entry.suspended: + entry.suspended = True + suspended += 1 + logger.warning( + "Auto-suspended stuck session %s (active across %d " + "consecutive restarts — likely a stuck loop)", + session_key, counts[session_key], + ) + except Exception: + pass + + if suspended: + with suppress(Exception): + self.session_store._save() + + # Clear the file — counters start fresh after suspension + with suppress(Exception): + path.unlink(missing_ok=True) + + return suspended + + async def _clear_restart_failure_count(self, session_key: str) -> None: + """Clear a completed session's restart-failure counter off-loop (atomic_json_write fsyncs).""" + from gateway.run import _hermes_home, atomic_json_write + import json + + path = _hermes_home / self._STUCK_LOOP_FILE + if not path.exists(): + return + try: + counts = json.loads(path.read_text(encoding="utf-8")) + if session_key in counts: + del counts[session_key] + if counts: + await asyncio.to_thread(atomic_json_write, path, counts, indent=None) + else: + path.unlink(missing_ok=True) + except Exception: + pass + + async def _launch_detached_restart_command(self) -> None: + from gateway.run import _resolve_hermes_bin + import shutil + import subprocess + + hermes_cmd = _resolve_hermes_bin() + if not hermes_cmd: + logger.error("Could not locate hermes binary for detached /restart") + return + if self._detached_restart_helper_started: + return + self._detached_restart_helper_started = True + + current_pid = os.getpid() + restart_after_s = max(float(getattr(self, "_restart_drain_timeout", 0.0) or 0.0) + 5.0, 5.0) + + # On Windows there's no bash/setsid chain — spawn a tiny Python watcher directly via + # sys.executable instead. + if sys.platform == "win32": + import textwrap + from hermes_cli._subprocess_compat import ( + windows_detach_flags_without_breakaway, + windows_detach_popen_kwargs, + ) + + cmd_argv = [*hermes_cmd, "gateway", "restart"] + watcher = textwrap.dedent( + """ + import os, subprocess, sys, time + from hermes_cli._subprocess_compat import windows_detach_flags_without_breakaway + pid = int(sys.argv[1]) + restart_after_s = float(sys.argv[2]) + cmd = sys.argv[3:] + deadline = time.monotonic() + restart_after_s + + def _alive(p): + # On Windows, os.kill(pid, 0) is NOT a no-op — it maps to + # GenerateConsoleCtrlEvent(0, pid) (bpo-14484). Use the + # Win32 handle-based existence check instead. + if os.name == 'nt': + import ctypes + k32 = ctypes.windll.kernel32 + k32.OpenProcess.restype = ctypes.c_void_p + k32.WaitForSingleObject.restype = ctypes.c_uint + k32.GetLastError.restype = ctypes.c_uint + h = k32.OpenProcess(0x1000 | 0x100000, False, int(p)) + if not h: + return k32.GetLastError() != 87 + try: + return k32.WaitForSingleObject(h, 0) == 0x102 + finally: + k32.CloseHandle(h) + try: + os.kill(int(p), 0) + return True + except ProcessLookupError: + return False + except PermissionError: + return True + except OSError: + return False + + while time.monotonic() < deadline: + if not _alive(pid): + break + time.sleep(0.2) + subprocess.Popen( + cmd, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + creationflags=windows_detach_flags_without_breakaway(), + ) + """ + ).strip() + from tools.environments.local import build_subprocess_env + watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) + # The watcher must not inherit the gateway marker, else `hermes gateway restart` refuses to + # run (self-restart loop guard) and the gateway stays stopped. + watcher_env.pop("_HERMES_GATEWAY", None) + project_root = Path(__file__).resolve().parent.parent + # Console python under CREATE_NO_WINDOW owns one hidden console inherited by the restart + # child, so nothing flashes. Do NOT swap in pythonw.exe — a console-less watcher forces + # every console-subsystem descendant to allocate a visible conhost. + watcher_python = sys.executable + venv_dir = Path(watcher_env.get("VIRTUAL_ENV") or project_root / "venv") + site_packages = venv_dir / "Lib" / "site-packages" + if site_packages.exists(): + watcher_env["VIRTUAL_ENV"] = str(venv_dir) + pythonpath = [str(project_root), str(site_packages)] + if watcher_env.get("PYTHONPATH"): + pythonpath.append(watcher_env["PYTHONPATH"]) + watcher_env["PYTHONPATH"] = os.pathsep.join(dict.fromkeys(pythonpath)) + watcher_argv = [ + watcher_python, + "-c", + watcher, + str(current_pid), + str(restart_after_s), + *cmd_argv, + ] + # The watcher must break away from any job object the parent CLI lives in (Desktop + # wrappers, Windows Terminal, schtasks), else it is reaped when the CLI exits and the + # gateway never respawns. windows_detach_popen_kwargs() sets CREATE_BREAKAWAY_FROM_JOB, + # but a job without JOB_OBJECT_LIMIT_BREAKAWAY_OK rejects it (ERROR_ACCESS_DENIED as + # OSError); retry once without the bit, preserving argv and the scrubbed watcher_env. + try: + subprocess.Popen( + watcher_argv, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=watcher_env, + **windows_detach_popen_kwargs(), + ) + except OSError: + try: + subprocess.Popen( + watcher_argv, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=watcher_env, + creationflags=windows_detach_flags_without_breakaway(), + ) + except OSError as exc: + # Both spawns failed. Log only the interpreter basename and numeric errno — never + # argv, env, watcher source, or str(exc) (may carry a full path) — and return. + winerror = getattr(exc, "winerror", None) + error_code = winerror if winerror is not None else exc.errno + error_field = "winerror" if winerror is not None else "errno" + logger.warning( + "Detached restart watcher was not started after the " + "no-breakaway retry (%s; %s=%r). The gateway will not " + "be respawned by this restart attempt.", + os.path.basename(watcher_python), + error_field, + error_code, + ) + return + + cmd = " ".join(shlex.quote(part) for part in hermes_cmd) + shell_cmd = ( + f"deadline=$(( $(date +%s) + {int(restart_after_s)} )); " + f"while kill -0 {current_pid} 2>/dev/null && [ $(date +%s) -lt $deadline ]; do sleep 0.2; done; " + f"{cmd} gateway restart" + ) + # Same marker scrub as the Windows watcher: an inherited _HERMES_GATEWAY=1 makes the CLI's + # self-restart loop guard refuse silently (DEVNULL), so the gateway stops and never comes back. + from tools.environments.local import build_subprocess_env + watcher_env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=True) + watcher_env.pop("_HERMES_GATEWAY", None) + setsid_bin = shutil.which("setsid") + if setsid_bin: + subprocess.Popen( + [setsid_bin, "bash", "-lc", shell_cmd], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=watcher_env, + start_new_session=True, + ) + else: + subprocess.Popen( + ["bash", "-lc", shell_cmd], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=watcher_env, + start_new_session=True, + ) + + def _wedged_agent_count(self) -> int: + """Count running chat agents already past the inactivity timeout. + + No activity (API bytes, tool progress) for ``agent.gateway_timeout`` = wedged (the turn reaper's + threshold). Returns 0 when the timeout is disabled (the after-turn cap still bounds the wait). + Cron/API-server work has no activity clock and pending sentinels are brand-new, so neither + counts. Fail-open per agent: an unreadable activity summary means "not wedged". + """ + from gateway.run import _AGENT_PENDING_SENTINEL, _float_env + timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) + if timeout <= 0: + return 0 + wedged = 0 + for agent in list((getattr(self, "_running_agents", None) or {}).values()): + if agent is None or agent is _AGENT_PENDING_SENTINEL: + continue + summary_fn = getattr(agent, "get_activity_summary", None) + if not callable(summary_fn): + continue + try: + summary = summary_fn() + if not isinstance(summary, dict): + continue + idle = float(summary.get("seconds_since_activity", 0.0)) + except Exception: + continue + if idle >= timeout: + wedged += 1 + return wedged + + def _awaitable_work_count(self) -> int: + """Active work minus wedged turns — what the restart wait waits on.""" + return max(0, self._active_work_count() - self._wedged_agent_count()) + + async def _await_active_work_before_restart(self) -> bool: + """Wait for in-flight work to finish before entering ``stop()``. + + Calling ``stop()`` immediately would fold the requesting turn into the drain set and + force-interrupt it at ``restart_drain_timeout``; instead refuse new turns, wait for active + agents/cron/api work to reach zero, then ``stop()`` an idle gateway. Wedged turns + (``_wedged_agent_count``) are excluded — restart is the remedy, so ``stop()``'s drain + interrupts them. Returns True when drained to zero, False when the safety cap elapsed or + only wedged work remains (caller proceeds to ``stop()``). + """ + active = self._active_work_count() + if active <= 0: + return True + + awaitable = self._awaitable_work_count() + if awaitable <= 0: + logger.warning( + "Restart requested with %d active work unit(s), all wedged " + "past the inactivity timeout; skipping the after-turn wait " + "and proceeding to stop()/drain which will interrupt them", + active, + ) + return False + + timeout = float(getattr(self, "_restart_after_turn_timeout", 0.0) or 0.0) + if timeout <= 0: + logger.info( + "Restart requested with %d active work unit(s); " + "restart_after_turn_timeout=0 — entering stop()/drain immediately", + active, + ) + return False + + logger.info( + "Restart requested with %d active work unit(s); " + "deferring stop() until they finish (cap=%.0fs) so in-flight " + "turns are not amputated (#77184)", + active, + timeout, + ) + with suppress(Exception): + self._update_runtime_status("draining") + + loop = asyncio.get_running_loop() + deadline = loop.time() + timeout + last_status_at = 0.0 + while self._awaitable_work_count() > 0: + now = loop.time() + if now >= deadline: + logger.warning( + "Restart after-turn wait timed out after %.0fs with %d " + "still active; proceeding to stop()/drain which may " + "interrupt remaining work (#77184)", + timeout, + self._active_work_count(), + ) + return False + if (now - last_status_at) >= 30.0: + logger.info( + "Restart deferred: waiting on %d active work unit(s) " + "(%d wedged and excluded; %.0fs remaining before force drain)", + self._awaitable_work_count(), + self._wedged_agent_count(), + deadline - now, + ) + with suppress(Exception): + self._update_runtime_status("draining") + last_status_at = now + await asyncio.sleep(0.1) + + if self._active_work_count() > 0: + logger.warning( + "Restart deferred wait: %d wedged work unit(s) remain; " + "proceeding to stop()/drain which will interrupt them", + self._active_work_count(), + ) + return False + + logger.info( + "Restart deferred wait complete — active work drained; " + "proceeding to stop()" + ) + return True + + def request_restart(self, *, detached: bool = False, via_service: bool = False) -> bool: + if self._restart_task_started: + return False + self._restart_requested = True + self._restart_detached = detached + self._restart_via_service = via_service + self._restart_task_started = True + # Refuse new turns while in-flight work finishes. Keep ``_running`` True so adapters stay + # connected and the active turn can still deliver its final response. + self._draining = True + + async def _run_restart() -> None: + await self._await_active_work_before_restart() + # Launch the detached helper only AFTER the after-turn wait: its drain_timeout+5 deadline + # covers stop() teardown; earlier it would fire the restart mid-turn. + if detached: + try: + await self._launch_detached_restart_command() + except Exception as e: + logger.error("Failed to launch detached gateway restart helper: %s", e) + await asyncio.sleep(0.05) + await self.stop(restart=True, detached_restart=detached, service_restart=via_service) + + # Do NOT add _run_restart to _background_tasks: _stop_impl cancels every entry there, which + # would cancel it while awaiting _stop_task and propagate CancelledError into _stop_impl, + # skipping _shutdown_event.set() / _exit_code = 75. Keep a strong ref in self._restart_task. + self._restart_task = asyncio.create_task(_run_restart()) + return True + + def _start_systemd_watchdog(self) -> bool: + """Start sd_notify only after a configured gateway is truly running.""" + if not self._running or self.config.systemd_watchdog_seconds <= 0: + return False + if self._systemd_watchdog is not None: + return True + + from gateway.systemd_notify import SystemdWatchdog + + watchdog = SystemdWatchdog(config_enabled=True) + if not watchdog.start(): + return False + self._systemd_watchdog = watchdog + watchdog.ready("Hermes Gateway running") + return True + + async def _stop_systemd_watchdog(self) -> None: + """Stop heartbeats before any potentially long shutdown drain.""" + watchdog = self._systemd_watchdog + if watchdog is None: + return + self._systemd_watchdog = None + await watchdog.stop() + + @staticmethod + def _stop_kill_tool_subprocesses(phase: str) -> list: + """Kill tool subprocesses + tear down terminal envs + browsers. + + Returns the cron job IDs marked interrupted so the caller can notify owners while + adapters are still up. Called twice: eagerly after a drain timeout forces interrupt + (reclaim children before systemd SIGKILLs) and as a final catch-all in _stop_impl(). + Best-effort; exceptions swallowed so one subsystem cannot block the rest. + """ + try: + from tools.process_registry import process_registry + _killed = process_registry.kill_all() + if _killed: + logger.info( + "Shutdown (%s): killed %d tool subprocess(es)", + phase, _killed, + ) + except Exception as _e: + logger.debug("process_registry.kill_all (%s) error: %s", phase, _e) + _marked_cron_jobs: list = [] + try: + # kill_all() is a global sweep, so any cron job dispatched right now lost its tool + # subprocess; its agent thread may still emit a plausible response from truncated + # output. Mark the run interrupted so it can never be reported as success. + from cron.scheduler import mark_running_jobs_interrupted + _interrupted = _marked_cron_jobs = mark_running_jobs_interrupted( + f"Gateway shutdown ({phase}) killed the job's tool " + "subprocess before the run finished." + ) + if _interrupted: + logger.warning( + "Shutdown (%s): marked %d in-flight cron job(s) interrupted: %s", + phase, len(_interrupted), ", ".join(_interrupted), + ) + except Exception as _e: + logger.debug("mark_running_jobs_interrupted (%s) error: %s", phase, _e) + try: + from tools.async_delegation import interrupt_all as _interrupt_async + _async_n = _interrupt_async(reason=f"gateway shutdown ({phase})") + if _async_n: + logger.info( + "Shutdown (%s): interrupted %d background delegation(s)", + phase, _async_n, + ) + except Exception as _e: + logger.debug("async interrupt_all (%s) error: %s", phase, _e) + try: + from tools.terminal_tool import cleanup_all_environments + cleanup_all_environments() + except Exception as _e: + logger.debug("cleanup_all_environments (%s) error: %s", phase, _e) + try: + from tools.browser_tool import cleanup_all_browsers + cleanup_all_browsers() + except Exception as _e: + logger.debug("cleanup_all_browsers (%s) error: %s", phase, _e) + return _marked_cron_jobs + + async def _stop_begin_teardown( + self, _stop_started_at_box: dict + ) -> Tuple[Callable[[], int], Callable[[], float]]: + """Flag teardown, stop room worker/watchdog, notify sessions. Returns the phase clocks.""" + # Shutdown-path tests and third-party runner doubles may only + # implement the older drain-count surface. + _deferred_worker_count = getattr( + self, + "_active_deferred_agent_worker_count", + lambda: 0, + ) + logger.info( + "Stopping gateway%s...", + " for restart" if self._restart_requested else "", + ) + _stop_started_at = time.monotonic() + _stop_started_at_box["t"] = _stop_started_at + + def _phase_elapsed() -> float: + return time.monotonic() - _stop_started_at + + self._running = False + self._clear_plugin_message_injector() + self._draining = True + + stop_room_worker = getattr(self, "_stop_hosted_room_worker", None) + if callable(stop_room_worker): + try: + stopped = await stop_room_worker(timeout=5.0) + if not stopped: + logger.warning( + "Group Chat worker is still settling durable work; " + "the next gateway start will recover it" + ) + except Exception: + logger.warning( + "Group Chat worker could not stop cleanly; the next gateway " + "start will recover durable work", + exc_info=True, + ) + + stop_watchdog = getattr(self, "_stop_systemd_watchdog", None) + if callable(stop_watchdog): + await stop_watchdog() + + await self._cancel_secondary_profile_reconnect_tasks() + + # Notify all chats with active agents BEFORE draining. + # Adapters are still connected here, so messages can be sent. + await self._notify_active_sessions_of_shutdown() + logger.info( + "Shutdown phase: notify_active_sessions done at +%.2fs", + _phase_elapsed(), + ) + return _deferred_worker_count, _phase_elapsed + + async def _stop_drain_active_work( + self, + timeout: float, + _deferred_worker_count: Callable[[], int], + _phase_elapsed: Callable[[], float], + ) -> Tuple[dict, bool, float]: + """Pre-mark resume_pending, drain agents/cron/API work. Returns (active_agents, timed_out, drain_elapsed).""" + from gateway.run import _AGENT_PENDING_SENTINEL + # Pre-mark sessions resume_pending BEFORE the drain wait: if the service manager kills + # the process mid-drain, the durable marker already lets the next boot recover them. + _pre_drain_keys: list[str] = [] + for _sk, _agent in list(self._running_agents.items()): + if _agent is _AGENT_PENDING_SENTINEL: + continue + try: + await self.async_session_store.mark_resume_pending( + _sk, + "restart_timeout" if self._restart_requested else "shutdown_timeout", + ) + _pre_drain_keys.append(_sk) + except Exception as _e: + logger.debug("pre-drain mark_resume_pending failed for %s: %s", _sk, _e) + + _cron_at_start = self._active_cron_job_count() + _api_at_start = self._active_api_run_count() + _deferred_at_start = _deferred_worker_count() + # In-flight cron work gets its own floor, clamped to the watchdog leash so the extra + # wait never costs the post-drain cleanup window. getattr-guard: shutdown-path tests + # drive _stop_impl_body from bare doubles (not GatewayRunner) lacking the class default. + _cron_drain_cfg = getattr( + self, "_cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT + ) + _cron_timeout = resolve_cron_drain_budget( + timeout, + _cron_drain_cfg, + watchdog_delay=resolve_shutdown_watchdog_delay(timeout), + elapsed=_phase_elapsed(), + ) + if _cron_at_start and _cron_timeout > timeout: + logger.info( + "Shutdown drain: %d in-flight cron job(s) — waiting up to " + "%.0fs for them (cron_drain_timeout=%.0fs, " + "restart_drain_timeout=%.0fs)", + _cron_at_start, + _cron_timeout, + _cron_drain_cfg, + timeout, + ) + _drain_started_at = time.monotonic() + active_agents, timed_out = await self._drain_active_agents( + timeout, _cron_timeout + ) + _drain_elapsed = time.monotonic() - _drain_started_at + logger.info( + "Shutdown phase: drain done at +%.2fs (drain took %.2fs, " + "timed_out=%s, active_at_start=%d, active_now=%d, " + "cron_at_start=%d, cron_now=%d, " + "api_at_start=%d, api_now=%d, " + "deferred_at_start=%d, deferred_now=%d)", + _phase_elapsed(), + _drain_elapsed, + timed_out, + len(active_agents), + self._running_agent_count(), + _cron_at_start, + self._active_cron_job_count(), + _api_at_start, + self._active_api_run_count(), + _deferred_at_start, + _deferred_worker_count(), + ) + + if not timed_out: + # Graceful drain: clear the pre-drain resume_pending markers so sessions that + # finished during the drain window don't carry a stale flag. + for _sk in _pre_drain_keys: + if _sk not in self._running_agents: + try: + await self.async_session_store.clear_resume_pending(_sk) + except Exception as _e: + logger.debug( + "clear_resume_pending after drain failed for %s: %s", + _sk, _e, + ) + return active_agents, timed_out, _drain_elapsed + + async def _stop_interrupt_remaining_work( + self, + _drain_elapsed: float, + _deferred_worker_count: Callable[[], int], + _phase_elapsed: Callable[[], float], + ) -> None: + """Drain timed out: mark resume_pending, interrupt, settle, kill tool subprocesses, notify cron.""" + from gateway.run import ( + GatewayRunner, + _AGENT_PENDING_SENTINEL, + _INTERRUPT_REASON_GATEWAY_RESTART, + _INTERRUPT_REASON_GATEWAY_SHUTDOWN, + ) + logger.warning( + "Gateway drain timed out after %.1fs with %d active agent(s), " + "%d in-flight cron job(s), %d api_server run(s), and " + "%d deferred agent worker(s); " + "interrupting remaining work.", + _drain_elapsed, + self._running_agent_count(), + self._active_cron_job_count(), + self._active_api_run_count(), + _deferred_worker_count(), + ) + # Mark forcibly-interrupted sessions resume_pending BEFORE interrupting, so the next + # message on the same session_key auto-resumes instead of being converted to a fresh + # session by suspend_recently_active(). Genuinely stuck sessions still escalate via + # ``.restart_failure_counts`` (threshold 3), which sets ``suspended=True`` and wins. + # + # Iterate self._running_agents (current), not the drain-start snapshot: sessions that + # finished cleanly during the drain would otherwise get a stray interruption note. + # Skip pending sentinels as _interrupt_running_agents() does — nothing has started. + _resume_reason = ( + "restart_timeout" if self._restart_requested else "shutdown_timeout" + ) + for _sk, _agent in list(self._running_agents.items()): + if _agent is _AGENT_PENDING_SENTINEL: + continue + try: + await self.async_session_store.mark_resume_pending(_sk, _resume_reason) + except Exception as _e: + logger.debug( + "mark_resume_pending failed for %s: %s", + _sk, _e, + ) + self._interrupt_running_agents( + _INTERRUPT_REASON_GATEWAY_RESTART if self._restart_requested else _INTERRUPT_REASON_GATEWAY_SHUTDOWN + ) + interrupt_grace_timeout = ( + GatewayRunner._post_interrupt_grace_timeout(self) + ) + interrupt_deadline = ( + asyncio.get_running_loop().time() + interrupt_grace_timeout + ) + logger.info( + "Shutdown phase: allowing %.1fs for interrupted agents to unwind", + interrupt_grace_timeout, + ) + # Wait on API-server work too: the interrupt is cooperative, and without this the + # settle window closes as soon as _running_agents is empty, so an API turn just asked + # to stop has its tool subprocesses killed below before it can unwind. + while ( + self._running_agents + or self._active_api_run_count() + or _deferred_worker_count() + ) and asyncio.get_running_loop().time() < interrupt_deadline: + self._update_runtime_status("draining") + await asyncio.sleep(0.1) + + # The interrupt fires once, but work can materialize AFTER it: a /v1/runs task enters + # _active_run_agents only when _create_agent returns, and a _AGENT_PENDING_SENTINEL + # entry is promoted by track_agent() on its own schedule. Re-signal anything still + # live so it gets a cooperative interrupt instead of a bare tool-subprocess kill. + if ( + self._running_agents + or self._active_api_run_count() + or _deferred_worker_count() + ): + self._interrupt_running_agents( + _INTERRUPT_REASON_GATEWAY_RESTART + if self._restart_requested + else _INTERRUPT_REASON_GATEWAY_SHUTDOWN + ) + logger.debug( + "Re-signaled interrupt for work still live at settle-window exit" + ) + + # Kill lingering tool subprocesses NOW, before adapter disconnect / DB close: under + # systemd (TimeoutStopSec ≈ drain_timeout + headroom) deferring risks the cgroup + # SIGKILL reaping orphaned children instead of us. The final catch-all still runs. + _interrupted_cron_jobs = GatewayRunner._stop_kill_tool_subprocesses("post-interrupt") + logger.info( + "Shutdown phase: post-interrupt tool kill done at +%.2fs", + _phase_elapsed(), + ) + # Last window where the transport is still up. The cron worker whose run we just + # killed will try to deliver its own "interrupted" notice, but it gets there after + # the adapter teardown below and the message is lost. + try: + await self._notify_interrupted_cron_jobs(_interrupted_cron_jobs) + except Exception as _e: + logger.debug("Cron interrupt notification failed: %s", _e) + logger.info( + "Shutdown phase: cron interrupt notices done at +%.2fs", + _phase_elapsed(), + ) + + async def _stop_finalize_agents_and_adapters( + self, active_agents: dict, _phase_elapsed: Callable[[], float] + ) -> None: + """Detached restart launch, agent finalization, idle-cache cleanup, adapter teardown.""" + if self._restart_requested and self._restart_detached: + try: + await self._launch_detached_restart_command() + except Exception as e: + logger.error("Failed to launch detached gateway restart: %s", e) + + await self._finalize_shutdown_agents(active_agents) + + # Also shut down memory providers on idle cached agents. _finalize_shutdown_agents only + # handles agents that were mid-turn at drain time; the _agent_cache may still hold idle + # agents whose MemoryProviders never received on_session_end(). + _cache_lock = getattr(self, "_agent_cache_lock", None) + _cache = getattr(self, "_agent_cache", None) + if _cache_lock is not None and _cache is not None: + with _cache_lock: + _idle_agents = list(_cache.values()) + _cache.clear() + for _entry in _idle_agents: + _agent = ( + _entry[0] if isinstance(_entry, tuple) else _entry + ) + # Bounded + off-loop so a wedged memory provider can't hang shutdown forever + # (this path is why SIGTERM once failed to kill the process). + await self._cleanup_agent_resources_off_loop( + _agent, context="shutdown idle-cache" + ) + + # Completion flush tasks can be sleeping in their fan-in window or blocked in adapter + # delivery. Cancel and await them while adapters are still alive so every watcher + # receives a retryable result before platform teardown begins. + cancel_completion_batches = getattr( + self, "_cancel_process_completion_batch_tasks", None + ) + if cancel_completion_batches is not None: + await cancel_completion_batches() + + for platform, adapter in list(self.adapters.items()): + await self._bounded_adapter_teardown(adapter, platform) + + # Disconnect secondary-profile adapters (multiplex mode). + for _prof, _amap in list(getattr(self, "_profile_adapters", {}).items()): + for platform, adapter in list(_amap.items()): + await self._bounded_adapter_teardown( + adapter, platform, profile=_prof + ) + _amap.clear() + if hasattr(self, "_profile_adapters"): + self._profile_adapters.clear() + logger.info( + "Shutdown phase: all adapters disconnected at +%.2fs", + _phase_elapsed(), + ) + + def _stop_release_runtime_state(self, _phase_elapsed: Callable[[], float]) -> None: + """Cancel background tasks, flush pending messages, clear per-session state, final tool kill.""" + from gateway.run import GatewayRunner + for _task in list(self._background_tasks): + if _task is self._stop_task: + continue + if _task is self._restart_task: + # The restart orchestration task is awaiting _stop_task right now; cancelling it + # would propagate CancelledError into this _stop_impl and skip + # _shutdown_event.set() / _exit_code = 75. It self-terminates anyway. + continue + _task.cancel() + self._background_tasks.clear() + + self.adapters.clear() + for _session_key in list(self._running_agents): + self._release_running_agent_state(_session_key) + # Flush pending messages to disk before clearing: under FTS5 corruption the in-memory + # pending text is the only surviving copy; clearing unflushed loses it permanently. + try: + from gateway.shutdown_flush import flush_pending_to_file + flush_pending_to_file(dict(self._pending_messages), reason="shutdown") + except Exception: + pass + # The FIFO tail lives in SessionState.conversation.queued_events, not the slot dict + # above — flush it too or every follow-up parked in overflow at restart time is lost. + try: + from gateway.shutdown_flush import flush_overflow_to_file + flush_overflow_to_file( + { + _k: list(_v) + for _k, _v in dict(getattr(self, "_queued_events", None) or {}).items() + if _v + }, + reason="shutdown", + ) + except Exception: + pass + # On the real runner these are live SessionState views whose clear() resets one field + # per session — never a wholesale dict swap, so a concurrent writer on another session + # can't lose its entry. Test fakes borrowing _stop_impl keep plain dicts. + self._running_agents.clear() + self._running_agents_ts.clear() + if hasattr(self, "_active_session_leases"): + self._active_session_leases.clear() + self._pending_messages.clear() + self._pending_approvals.clear() + if hasattr(self, '_busy_ack_ts'): + self._busy_ack_ts.clear() + self._shutdown_event.set() + + # Global catch-all subprocess kill (safe to repeat): covers the graceful path and + # anything respawned since the drain-timeout path's post-interrupt kill. + GatewayRunner._stop_kill_tool_subprocesses("final-cleanup") + logger.info( + "Shutdown phase: final-cleanup tool kill done at +%.2fs", + _phase_elapsed(), + ) + + # Reap the process-global auxiliary-client cache once at the end of teardown. Per-turn + # cleanup misses clients bound to worker-thread loops that died with their executor + # (cron ticks); without this sweep async httpx transports accumulate until EMFILE. + try: + from agent.auxiliary_client import shutdown_cached_clients + shutdown_cached_clients() + except Exception as _e: + logger.debug("shutdown_cached_clients error: %s", _e) + + def _stop_quiesce_and_close_session_dbs( + self, timeout: float, _phase_elapsed: Callable[[], float] + ) -> None: + """Quiesce the executor, then close SessionDB handles only if no worker is still live.""" + from gateway.run import GatewayRunner, _EXECUTOR_QUIESCE_TIMEOUT + # Quiesce the gateway thread pool BEFORE the session databases are closed. Running it + # after the close left two holes: (a) ``_executor_closing`` was still False, so any + # coroutine reaching ``_run_in_executor_with_context`` minted a fresh pool and ran more + # blocking DB work against just-closed handles; (b) cancelling ``self._background_tasks`` + # does not stop a ``run_in_executor`` future that already started — the task dies, the + # worker keeps writing. Either way a write lands after ``SessionDB.close()`` has + # checkpointed the WAL and let SQLite unlink the sidecar; the late write silently + # reopens the handle and mints a fresh WAL generation behind that checkpoint, so + # teardown checkpoints the same file twice from an unaccounted connection + # (close-time page-write corruption / split WAL generation). + # The wait is bounded and clamped to what is left of the shutdown watchdog leash + # (minus a second for the close itself), so a stuck worker can never cost us the + # post-close cleanup window. + _exec_quiesce_budget = max( + 0.0, + min( + _EXECUTOR_QUIESCE_TIMEOUT, + resolve_shutdown_watchdog_delay(timeout) + - _phase_elapsed() + - 1.0, + ), + ) + _exec_live = GatewayRunner._shutdown_executor( + self, drain_timeout=_exec_quiesce_budget + ) + if _exec_live: + # A live worker can still be mid-write against a SessionDB + # handle. Checkpointing/closing it now is exactly the + # sequence that produced the wrong-page-number corruption in + # #101093, so the close path below is skipped entirely + # rather than raced — the handle is left open for SQLite to + # recover from its own WAL on the next open, which is a + # transient "database is locked" on an immediate --replace + # at worst, not a corrupt file. + logger.warning( + "Shutdown phase: %d executor worker(s) still running after " + "a %.2fs quiesce — skipping the SessionDB close/checkpoint " + "to avoid racing a live write (#101093); handles are left " + "open for SQLite to recover on next open", + _exec_live, + _exec_quiesce_budget, + ) + else: + logger.info( + "Shutdown phase: executor quiesced at +%.2fs", + _phase_elapsed(), + ) + + # Close SQLite session DBs so the WAL lock is released; otherwise --replace leaves the old + # connection holding it until exit and the new gateway gets 'database is locked'. + # ``_session_db`` is an AsyncSessionDB facade — unwrap; ``session_store`` holds ``_db``. + _self_db = getattr(self, "_session_db", None) + _self_db = getattr(_self_db, "_db", _self_db) + for _db in (_self_db, getattr(getattr(self, "session_store", None), "_db", None)): + if _db is None or not hasattr(_db, "close"): + continue + try: + _db.close() + except Exception as _e: + logger.debug("SessionDB close error: %s", _e) + # A multiplexed session_store caches one SessionDB per profile; ``_db`` above only covered + # the root scope. Sweep the rest so secondary WAL locks are released before --replace. + _sweep = getattr( + getattr(self, "session_store", None), "close_all_db_handles", None + ) + if _sweep is not None: + try: + _sweep() + except Exception as _e: + logger.debug("SessionDB handle sweep error: %s", _e) + # Same sweep for the runner's own per-profile session_search + # handles (slash commands resolve them under profile scopes). + try: + GatewayRunner.close_all_session_db_handles(self) + except Exception as _e: + logger.debug("Runner SessionDB handle sweep error: %s", _e) + # Final sweep: close shared SessionDB instances still held by the process-wide registry + # (tools, cron, mirror, etc. opened via get_shared_session_db but not released above). + try: + from hermes_state import close_shared_session_dbs + closed = close_shared_session_dbs() + if closed: + logger.debug("Closed %d shared SessionDB instance(s) at shutdown", closed) + except Exception as _e: + logger.debug("Shared SessionDB close error: %s", _e) + logger.info( + "Shutdown phase: SessionDB close done at +%.2fs", + _phase_elapsed(), + ) + + def _stop_persist_exit_state( + self, timed_out: bool, active_agents: dict, _phase_elapsed: Callable[[], float] + ) -> None: + """PID/lock release, clean-shutdown marker, restart markers, terminal runtime status.""" + from gateway.run import ( + _hermes_home, + _planned_restart_notification_path, + _shutdown_gateway_health_export, + atomic_json_write, + ) + from gateway.status import remove_pid_file, release_gateway_runtime_lock + remove_pid_file() + release_gateway_runtime_lock() + + # Clean-shutdown marker: suspend_recently_active() need only run after unexpected exits. + # If the drain timed out and agents were force-interrupted, sessions may be half-finished + # — skip the marker so the next startup suspends them. + if not timed_out: + with suppress(Exception): + (_hermes_home / ".clean_shutdown").touch() + else: + logger.info( + "Skipping .clean_shutdown marker — drain timed out with " + "interrupted agents; next startup will suspend recently " + "active sessions." + ) + + # Stuck-loop detection: the counter increments for sessions active at each restart; at + # the threshold (3 consecutive) the next startup auto-suspends the session. + if active_agents: + self._increment_restart_failure_counts(set(active_agents.keys())) + + if self._restart_requested and self._restart_command_source is None: + try: + atomic_json_write( + _planned_restart_notification_path(), + { + "requested_at": time.time(), + "via_service": bool(self._restart_via_service), + "detached": bool(self._restart_detached), + }, + indent=None, + ) + except Exception as e: + logger.debug("Failed to write planned restart notification marker: %s", e) + + if self._restart_requested and self._restart_via_service: + # Service manager owns restarts: exit 75 + ``RestartForceExitStatus=75`` has systemd + # replace this process without a second helper racing the unit's stop/start job. + self._exit_code = GATEWAY_SERVICE_RESTART_EXIT_CODE + self._exit_reason = self._exit_reason or "Gateway restart requested" + + self._draining = False + # Persist terminal gateway_state: "stopped" by default, but "running" on an UNEXPECTED + # external signal (s6 SIGTERM on docker restart, OOM-kill, kill) — container_boot.py + # only auto-starts gateways last seen "running", so "stopped"/"draining" after a routine + # recreate would leave channels dark. Operator stops write a planned-stop marker BEFORE + # signalling and persist "stopped"; a restart also persists "stopped". + if getattr(self, "_signal_initiated_shutdown", False) and not self._restart_requested: + logger.info( + "Gateway stopped by an unexpected signal — persisting " + "gateway_state=running so container_boot auto-starts on " + "the next boot (issue #42675)" + ) + self._update_runtime_status("running", self._exit_reason) + else: + self._update_runtime_status("stopped", self._exit_reason) + _shutdown_gateway_health_export(self) + logger.info("Gateway stopped (total teardown %.2fs)", _phase_elapsed()) + + async def stop( + self, + *, + restart: bool = False, + detached_restart: bool = False, + service_restart: bool = False, + ) -> None: + """Stop the gateway and disconnect all adapters.""" + from gateway.run import GatewayRunner + # getattr-guard: shutdown-path tests build bare runners via + # object.__new__ that lack the liveness-guard machinery. + _stop_guards = getattr(self, "_stop_loop_liveness_guards", None) + if callable(_stop_guards): + _stop_guards() + if restart: + self._restart_requested = True + self._restart_detached = detached_restart + self._restart_via_service = service_restart + if self._stop_task is not None: + await self._stop_task + return + + async def _stop_impl() -> None: + # Thread-based shutdown watchdog: asyncio timeouts cannot recover a frozen loop. Arm a + # plain OS thread at the start of stop(); if teardown never finishes within drain+grace + # it dumps faulthandler stacks and os._exit so KeepAlive/systemd can revive. Skipped + # under pytest so stop()-driving tests don't get a delayed hard-exit in the worker. + _watchdog_done = threading.Event() + self._shutdown_watchdog_done = _watchdog_done + _stop_started_at_box: dict[str, float] = {} + + def _shutdown_watchdog_snapshot() -> dict: + started = _stop_started_at_box.get("t") + return { + "restart_requested": bool(self._restart_requested), + "draining": bool(self._draining), + "running": bool(self._running), + "active_agents": self._running_agent_count(), + "active_cron_jobs": self._active_cron_job_count(), + "active_api_runs": self._active_api_run_count(), + "active_deferred_agent_workers": getattr( + self, + "_active_deferred_agent_worker_count", + lambda: 0, + )(), + "restart_drain_timeout": self._restart_drain_timeout, + "watchdog_delay_s": resolve_shutdown_watchdog_delay( + self._restart_drain_timeout + ), + "phase_elapsed_s": ( + time.monotonic() - started if started is not None else None + ), + } + + if not os.environ.get("PYTEST_CURRENT_TEST"): + arm_shutdown_watchdog( + resolve_shutdown_watchdog_delay(self._restart_drain_timeout), + done_event=_watchdog_done, + snapshot_fn=_shutdown_watchdog_snapshot, + exit_code=1, + ) + + try: + await _stop_impl_body(_stop_started_at_box) + finally: + _watchdog_done.set() + + async def _stop_impl_body(_stop_started_at_box) -> None: + _deferred_worker_count, _phase_elapsed = await GatewayRunner._stop_begin_teardown( + self, _stop_started_at_box + ) + + timeout = self._restart_drain_timeout + active_agents, timed_out, _drain_elapsed = await GatewayRunner._stop_drain_active_work( + self, timeout, _deferred_worker_count, _phase_elapsed + ) + + if timed_out: + await GatewayRunner._stop_interrupt_remaining_work( + self, _drain_elapsed, _deferred_worker_count, _phase_elapsed + ) + + await GatewayRunner._stop_finalize_agents_and_adapters( + self, active_agents, _phase_elapsed + ) + GatewayRunner._stop_release_runtime_state(self, _phase_elapsed) + GatewayRunner._stop_quiesce_and_close_session_dbs(self, timeout, _phase_elapsed) + GatewayRunner._stop_persist_exit_state(self, timed_out, active_agents, _phase_elapsed) + + self._stop_task = asyncio.create_task(_stop_impl()) + await self._stop_task + + async def wait_for_shutdown(self) -> None: + """Wait for shutdown signal.""" + await self._shutdown_event.wait() diff --git a/gateway/run_startup.py b/gateway/run_startup.py new file mode 100644 index 0000000000..bb1b4244c2 --- /dev/null +++ b/gateway/run_startup.py @@ -0,0 +1,2071 @@ +"""Startup sequence, resume/restore and handoff methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import faulthandler +import os +import signal +import time +from contextlib import suppress +from datetime import datetime +from gateway.config import Platform +from gateway.delivery import looks_like_telegram_private_chat_id +from gateway.platforms.base import BasePlatformAdapter, MessageEvent, MessageType +from gateway.restart import ( + DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, + GATEWAY_FATAL_CONFIG_EXIT_CODE, + is_global_startup_conflict, +) +from gateway.shutdown_watchdog import ( + DEFAULT_HEARTBEAT_INTERVAL_S, + DEFAULT_LOOP_WATCHDOG_INTERVAL_S, + DEFAULT_LOOP_WATCHDOG_MAX_STRIKES, + DEFAULT_LOOP_WATCHDOG_TIMEOUT_S, + loop_heartbeat_forever, +) +from typing import Any, Dict, Optional, Tuple + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayStartupMixin: + """Startup sequence, resume/restore and handoff methods for GatewayRunner.""" + + async def _run_startup_resume_event( + self, + adapter: BasePlatformAdapter, + event: MessageEvent, + session_key: str, + ) -> None: + """Dispatch one synthetic startup resume and wait for its agent turn. + + Inbound messages stay queued until the resumed turn finishes, else a user message can race it. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + try: + await adapter.handle_message(event) + session_tasks = getattr(adapter, "_session_tasks", {}) + task = session_tasks.get(session_key) if isinstance(session_tasks, dict) else None + if task is not None: + await asyncio.shield(task) + finally: + # The runner slot was pre-claimed before this task spawned; release it if handle_message + # raises before _handle_message takes ownership, else the real run's cleanup owns it. + _pre_state = self._peek_session_state(session_key) + if (_pre_state.turn.agent if _pre_state else None) is _AGENT_PENDING_SENTINEL: + self._release_running_agent_state(session_key) + + def _queue_startup_restore_event(self, event: MessageEvent) -> None: + queue = getattr(self, "_startup_restore_queue", None) + if queue is None: + queue = [] + self._startup_restore_queue = queue + queue.append(event) + try: + source = event.source + logger.info( + "Queued inbound message during gateway startup restore: platform=%s chat=%s", + source.platform.value if source and source.platform else "unknown", + source.chat_id if source else "unknown", + ) + except Exception: + pass + + async def _drain_startup_restore_queue(self) -> int: + """Replay inbound messages queued while startup auto-resume ran.""" + drained = 0 + queue = getattr(self, "_startup_restore_queue", None) + if queue is None: + return 0 + while queue: + event = queue.pop(0) + source = getattr(event, "source", None) + adapter = self._adapter_for_source(source) + if adapter is None: + logger.debug( + "Dropping startup-restore queued message: adapter unavailable for %s", + getattr(getattr(source, "platform", None), "value", None), + ) + continue + # Mark this replay so _handle_message does not queue it again while + # the restore gate remains closed for any fresh inbound arrivals. + with suppress(Exception): + setattr(event, "_hermes_startup_restore_replay", True) + await adapter.handle_message(event) + drained += 1 + return drained + + def _start_startup_warmup(self) -> None: + """Kick off the boot turn-machinery warm-up in the background. + + Called from ``start()`` right after the startup-restore gate closes so the warm-up overlaps + the network-bound platform connects; ``_finish_startup_restore`` awaits it (bounded). + """ + from gateway.run import _startup_warmup_timeout_secs + timeout = _startup_warmup_timeout_secs() + if timeout <= 0: + self._startup_warmup_task = None + return + self._startup_warmup_task = asyncio.ensure_future( + self._warm_turn_prerequisites() + ) + + async def _warm_turn_prerequisites(self) -> None: + """Initialize turn machinery on an executor thread before the gate opens. + + Never raises: a failed warm-up degrades to lazy init and must not block startup. + """ + from gateway.run import _warm_turn_machinery_sync + try: + loop = asyncio.get_running_loop() + t0 = time.monotonic() + tool_count = await loop.run_in_executor(None, _warm_turn_machinery_sync) + logger.info( + "Turn machinery warmed in %.1fs (%d tool schema(s) materialized)", + time.monotonic() - t0, + tool_count, + ) + except Exception: + logger.warning( + "Turn-machinery warm-up failed; first inbound turn will " + "initialize lazily", + exc_info=True, + ) + + async def _await_startup_warmup(self) -> None: + """Bounded wait for the boot warm-up before the inbound gate opens. + + On timeout the gate opens anyway (availability outranks prompt completeness for a WEDGED + init) and the warm-up continues in the background; a late failure is still logged. + """ + from gateway.run import GatewayRunner, _startup_warmup_timeout_secs + task = getattr(self, "_startup_warmup_task", None) + if task is None or task.done(): + return + timeout = _startup_warmup_timeout_secs() + if timeout <= 0: + return + done, pending = await asyncio.wait({task}, timeout=timeout) + if pending: + logger.warning( + "Turn-machinery warm-up still running after %.0fs; opening " + "inbound gate anyway — the first turn may see lazily " + "initialized machinery (#99373). Warm-up continues in the " + "background.", + timeout, + ) + task.add_done_callback( + lambda t: GatewayRunner._log_late_background_failure( + t, + "boot turn-machinery warm-up failed after gate release", + level=logging.DEBUG, + ) + ) + + async def _finish_startup_restore(self) -> None: + """Wait (BOUNDED) for startup auto-resume, then release + drain inbound. + + Bounded by ``_startup_restore_drain_timeout_secs`` so one pathological boot-resume turn + cannot hold the gate shut for every channel; on timeout the gate opens and resume turns + finish in the background (NOT cancelled). Safe because ``_schedule_resume_pending_sessions`` + claims each ``_running_agents`` slot SYNCHRONOUSLY first, so drained inbound queues behind. + """ + from gateway.run import _startup_restore_drain_timeout_secs + tasks = list(getattr(self, "_startup_restore_tasks", []) or []) + if tasks: + timeout = _startup_restore_drain_timeout_secs() + if timeout > 0: + # asyncio.wait (unlike wait_for / gather+timeout) does NOT cancel pending tasks on + # timeout — the slow resume turn keeps running in the background. + done, pending = await asyncio.wait(tasks, timeout=timeout) + if pending: + logger.warning( + "Startup-restore gate released after %.0fs with %d boot " + "auto-resume turn(s) still running; draining inbound " + "queue now (resume slots already claimed, so no " + "duplicate agents). Slow turn(s) continue in the " + "background.", + timeout, + len(pending), + ) + # These tasks outlive the gate. Their normal done-callback only discards them + # from _background_tasks, so a LATER failure would be silently swallowed. + for task in pending: + task.add_done_callback(self._log_background_resume_result) + else: + # Non-positive timeout => opt out of the bound (historical + # "wait forever" behaviour). + await asyncio.gather(*tasks, return_exceptions=True) + done = set(tasks) + for task in done: + if task.cancelled(): + continue + exc = task.exception() + if exc is not None: + logger.debug( + "startup auto-resume task failed", + exc_info=(type(exc), exc, exc.__traceback__), + ) + self._startup_restore_tasks = [] + # Warm the turn machinery BEFORE the queue drains: replayed (and + # fresh) inbound turns must not build skeleton prompts (#99373). + await self._await_startup_warmup() + drained = await self._drain_startup_restore_queue() + self._startup_restore_in_progress = False + if drained: + logger.info("Drained %d inbound message(s) queued during startup restore", drained) + + @staticmethod + def _log_background_resume_result(task: "asyncio.Task") -> None: + """Done-callback for a boot-resume turn that outlived the startup-restore gate.""" + from gateway.run import GatewayRunner + GatewayRunner._log_late_background_failure( + task, + "background startup auto-resume task failed after gate release", + level=logging.DEBUG, + ) + + @staticmethod + def _log_late_background_failure( + task: "asyncio.Task", message: str, *, level: int = logging.WARNING + ) -> None: + """Shared done-callback body for boot-path tasks that outlive the startup-restore gate: + surface a late failure otherwise swallowed once the task leaves ``_background_tasks``. + Cancellation (shutdown) is expected, not an error.""" + if task.cancelled(): + return + exc = task.exception() + if exc is not None: + logger.log( + level, + message, + exc_info=(type(exc), exc, exc.__traceback__), + ) + + async def _await_startup_boot_sends( + self, + *, + planned_restart_notification_pending: bool, + ) -> None: + """Run boot-path sends without letting them pin the inbound restore gate. + + Awaiting ``_send_restart_notification`` / ``_redeliver_pending_obligations`` inline before + the gate releases lets one Telegram flood-control sleep freeze inbound on every platform. + Same bounded ``asyncio.wait`` as the resume gate: on timeout return and let the sends finish + in the background (not cancelled). The ledger claim + ``resume_pending`` clear run INLINE + before the send task exists: bounded DB work, and deferring it let a hung notification + expire the gate with zero rows claimed, so answered turns were replayed AND redelivered. + """ + from gateway.run import _clear_planned_restart_notification, _startup_restore_drain_timeout_secs + claimed = await self._claim_pending_obligations() + + async def _boot_sends() -> None: + await self._send_restart_notification() + if planned_restart_notification_pending: + try: + await self._send_home_channel_startup_notifications( + skip_targets=None, + ) + finally: + _clear_planned_restart_notification() + await self._redeliver_claimed_obligations(claimed) + + boot_task = asyncio.create_task(_boot_sends()) + timeout = _startup_restore_drain_timeout_secs() + if timeout > 0: + _done, pending = await asyncio.wait({boot_task}, timeout=timeout) + if pending: + logger.warning( + "Boot-path sends still running after %.0fs; releasing " + "inbound gate so other platforms are not frozen. " + "Restart notification / obligation redelivery continue " + "in the background.", + timeout, + ) + boot_task.add_done_callback(self._log_background_boot_send_result) + tasks = getattr(self, "_background_tasks", None) + if tasks is None: + self._background_tasks = set() + tasks = self._background_tasks + tasks.add(boot_task) + boot_task.add_done_callback(tasks.discard) + else: + await boot_task + + @staticmethod + def _log_background_boot_send_result(task: "asyncio.Task") -> None: + """Done-callback for boot-path sends that outlived the restore gate.""" + from gateway.run import GatewayRunner + GatewayRunner._log_late_background_failure( + task, "background boot-path send failed after gate release: see traceback" + ) + + async def _clear_resume_pending_for_claimed_obligations( + self, claimed: list, *, require_success: bool = False + ) -> list: + """Clear resume flags and return rows safe to redeliver. + + Startup recovery stays best-effort. Runtime reconnect recovery is stricter: if the + session-store write fails the response must not be sent, or the turn could be resumed too. + """ + sendable = [] + for row in claimed: + session_key = row.get("session_key") or "" + if not session_key: + sendable.append(row) + continue + try: + await self.async_session_store.clear_resume_pending(session_key) + except Exception: + logger.debug( + "clear_resume_pending failed for %s", session_key, + exc_info=True, + ) + if not require_success: + sendable.append(row) + else: + sendable.append(row) + return sendable + + async def _claim_pending_obligations(self) -> list: + """Claim recoverable delivery-ledger rows and clear their ``resume_pending`` flags. + + Pure DB work, no sends. Must run INLINE at startup BEFORE ``_schedule_resume_pending_sessions`` + and before the abandonable boot-send task exists: these sessions already produced their + answer, so clearing ``resume_pending`` here stops the resume path from re-running (and + re-paying for) the turn however long the sends take. Rows that were mid-send or previously + rejected carry a visible recovered-reply marker so a possible duplicate is labeled, never + silent (gateway/delivery_ledger.py). Returns the claimed rows for redelivery. + """ + try: + from gateway.delivery_ledger import ( + ledger_enabled, + sweep_recoverable, + ) + + if not await asyncio.to_thread(ledger_enabled): + return [] + # Only claim rows whose exact transport owner is connected this boot. A multiplexed + # gateway can host several bot identities for one platform; platform-only filtering + # would spend a disconnected bot's retry budget merely because another bot is online. + _profile_adapters = getattr(self, "_profile_adapters", None) or {} + _deliverable_targets = { + (getattr(p, "value", str(p)), "default") for p in self.adapters + } + # Legacy rows predate adapter_profile. They are unambiguous only in a non-multiplexed + # gateway; fail closed when multiple bot identities share the process. + if not _profile_adapters: + _deliverable_targets.update( + (getattr(p, "value", str(p)), None) for p in self.adapters + ) + for _profile, _adapters in _profile_adapters.items(): + _deliverable_targets.update( + (getattr(p, "value", str(p)), _profile) for p in _adapters + ) + _deliverable = {platform for platform, _ in _deliverable_targets} + claimed = await asyncio.to_thread( + sweep_recoverable, + None, + deliverable_platforms=_deliverable, + deliverable_targets=_deliverable_targets, + ) + except Exception: + logger.debug("delivery ledger sweep failed", exc_info=True) + return [] + if not claimed: + return [] + + # Clear resume_pending for EVERY claimed row before any send: claiming already spent one + # redelivery attempt and the answer is in the ledger, so the resume path must never re-run. + await self._clear_resume_pending_for_claimed_obligations(claimed) + return claimed + + async def _redeliver_claimed_obligations(self, claimed: list) -> int: + """Redeliver final responses for rows claimed by :meth:`_claim_pending_obligations`. + + Network half of the split: runs inside the bounded boot-send task, so a flood-limited send + can be abandoned by the restore gate without reopening the turn-replay window. Returns count. + """ + if not claimed: + return 0 + try: + from gateway.delivery_ledger import ( + RECOVERED_MARKER, + mark_delivered, + mark_failed, + release_runtime_claim, + ) + except Exception: + logger.debug("delivery ledger import failed", exc_info=True) + return 0 + + redelivered = 0 + for row in claimed: + try: + platform = Platform(row["platform"]) + except Exception: + logger.debug( + "obligation %s: unknown platform %r", + row["obligation_id"], row.get("platform"), + ) + continue + if "profile" in row: + adapter = self._authorization_adapter( + platform, row.get("profile") + ) + else: + # Startup rows preserve the historical default-adapter route. + adapter = self.adapters.get(platform) + if adapter is None: + # Runtime claims have not reached a transport yet. If the + # reconnect vanished before dispatch, release the claim without + # spending an attempt so the next reconnect can retry it. + if row.get("runtime_recovery"): + try: + await asyncio.to_thread( + release_runtime_claim, + row["obligation_id"], + "send_path_degraded", + ) + except Exception: + logger.debug( + "failed to release undispatched runtime obligation %s", + row["obligation_id"], + exc_info=True, + ) + # Startup claims preserve their historical state; attempts cap + # + stale cutoff bound later retries. + continue + content = row["content"] + if row.get("needs_marker"): + content = row.get("marker", RECOVERED_MARKER) + content + metadata = ( + {"thread_id": row["thread_id"]} if row.get("thread_id") else None + ) + + try: + result = await adapter.send( + chat_id=row["chat_id"], + content=content, + metadata=metadata, + ) + except Exception as send_err: + logger.warning( + "obligation %s: redelivery send raised: %s", + row["obligation_id"], send_err, + ) + result = None + try: + if result is not None and getattr(result, "success", False): + await asyncio.to_thread(mark_delivered, row["obligation_id"]) + redelivered += 1 + logger.info( + "Redelivered recovered final response to %s:%s " + "(obligation %s, attempt %d)", + row["platform"], row["chat_id"], + row["obligation_id"], row["attempts"], + ) + else: + await asyncio.to_thread( + mark_failed, + row["obligation_id"], + str(getattr(result, "error", "") or "send failed"), + ) + except Exception: + logger.debug("delivery ledger update failed", exc_info=True) + return redelivered + + async def _redeliver_pending_obligations(self) -> int: + """Claim + redeliver in one call (:meth:`_claim_pending_obligations` then + :meth:`_redeliver_claimed_obligations`). Stable public shape for tests/external callers; + the startup path calls the halves separately so the DB half runs inline before the + abandonable send task. + """ + return await self._redeliver_claimed_obligations( + await self._claim_pending_obligations() + ) + + async def _redeliver_failed_obligations_for_platform( + self, + platform: Platform, + *, + profile: Optional[str] = None, + ) -> int: + """Replay one adapter identity's transient failures after reconnect. + + The startup sweep cannot claim live-owner rows, and an adapter can reconnect without the + process exiting, so ``send_path_degraded`` responses would otherwise stay failed until the + next restart. Claim/clear/send are best-effort and reuse the startup redelivery contract. + """ + try: + from gateway.delivery_ledger import ( + ledger_enabled, + release_runtime_claim, + sweep_failed_for_runtime, + ) + + if not await asyncio.to_thread(ledger_enabled): + return 0 + claimed = await asyncio.to_thread( + sweep_failed_for_runtime, + platform.value, + profile=profile, + ) + except Exception: + logger.debug( + "runtime delivery ledger sweep failed after %s reconnect", + platform.value, + exc_info=True, + ) + return 0 + if not claimed: + return 0 + + # Clear before any send so the reconnect path cannot both redeliver an + # already-produced answer and schedule the same agent turn for resume. + sendable = await self._clear_resume_pending_for_claimed_obligations( + claimed, require_success=True + ) + sendable_ids = {row["obligation_id"] for row in sendable} + for row in claimed: + if row["obligation_id"] in sendable_ids: + continue + try: + await asyncio.to_thread( + release_runtime_claim, + row["obligation_id"], + "send_path_degraded", + ) + except Exception: + logger.debug( + "failed to release runtime delivery claim %s", + row["obligation_id"], + exc_info=True, + ) + return await self._redeliver_claimed_obligations(sendable) + + def _schedule_resume_pending_sessions(self, platform=None) -> int: + """Auto-continue fresh restart-interrupted sessions after startup. + + Synthesizes the next turn once adapters are back online; the event text is empty so the + existing ``_is_resume_pending`` injection path owns the recovery wording. Sessions whose + adapter is not in ``self.adapters`` stay ``resume_pending`` for the reconnect watcher, which + re-calls this scoped to that ``platform`` (a reconnecting platform never touches another's + recoveries); sessions with a running agent are skipped so none is resumed twice. + """ + from gateway.run import _AGENT_PENDING_SENTINEL, _auto_continue_freshness_window + window = _auto_continue_freshness_window() + try: + with self.session_store._lock: # noqa: SLF001 — snapshot under lock + self.session_store._ensure_loaded_locked() # noqa: SLF001 + candidates = [ + entry for entry in self.session_store._entries.values() # noqa: SLF001 + if entry.resume_pending + and not entry.suspended + and entry.origin is not None + and entry.resume_reason in self._AUTO_RESUME_REASONS + and (platform is None or entry.origin.platform == platform) + ] + except Exception as exc: + logger.warning("Failed to enumerate resume-pending sessions: %s", exc) + return 0 + + # Defense-3: break the SIGTERM-respawn loop. Only count this boot when there are restart- + # interrupted sessions to resume — a clean boot must not accrue toward the breaker. If too + # many such boots hit the window, skip auto-resume for THIS boot only: the gateway still + # serves inbound; the session stays resume_pending so a real user message can continue it. + if candidates: + try: + from gateway import restart_loop_guard as _rlg + + _max_restarts, _window, _max_gap = self._restart_loop_guard_config() + if _rlg.check_and_record( + _max_restarts, _window, max_gap_seconds=_max_gap + ): + return 0 + except Exception as exc: # noqa: BLE001 — breaker must fail OPEN + logger.debug("Restart-loop guard check skipped: %s", exc) + + now = datetime.now() + scheduled = 0 + for entry in candidates: + marker = entry.last_resume_marked_at or entry.updated_at + if marker is not None and (now - marker).total_seconds() > window: + continue + + # Already being resumed (e.g. scheduled at startup and still + # in-flight) — don't synthesize a second continuation turn. + if self._is_session_running(entry.session_key): + continue + + source = entry.origin + adapter = self._adapter_for_source(source) + if adapter is None: + logger.debug( + "Skipping auto-resume for %s: adapter not ready for %s", + entry.session_key, + getattr(source.platform, "value", source.platform), + ) + continue + + # Validate the session owner against the current allowlist before auto-resuming: a + # session created before the allowlist existed (or whose owner was since removed) must + # not silently receive a full agent response just because it carries a resume marker. + try: + if not self._is_user_authorized(source): + logger.warning( + "Skipping auto-resume for %s: session owner is no " + "longer authorized under the current allowlist", + entry.session_key, + ) + continue + except Exception as exc: + logger.warning( + "Skipping auto-resume for %s: authorization check failed: %s", + entry.session_key, exc, + ) + continue + + # Claim the session slot *before* spawning the task so an inbound message arriving + # between task creation and the task's first await (where _process_message_background + # sets the real sentinel) sees the slot occupied and queues, not a duplicate AIAgent. + _resume_state = self._session_state(entry.session_key) + _resume_state.turn.agent = _AGENT_PENDING_SENTINEL + _resume_state.turn.started_ts = time.time() + self._persist_active_agents() + + # Empty-text internal event: the _is_resume_pending branch in _handle_message_with_agent + # prepends the reason-aware system note before the turn runs. + event = MessageEvent( + text="", + message_type=MessageType.TEXT, + source=source, + internal=True, + ) + task = asyncio.create_task( + self._run_startup_resume_event(adapter, event, entry.session_key) + ) + self._background_tasks.add(task) + task.add_done_callback(self._background_tasks.discard) + if getattr(self, "_startup_restore_in_progress", False): + tasks = getattr(self, "_startup_restore_tasks", None) + if tasks is None: + tasks = [] + self._startup_restore_tasks = tasks + tasks.append(task) + scheduled += 1 + if scheduled: + logger.info( + "Scheduled auto-resume for %d restart-interrupted session(s)", + scheduled, + ) + return scheduled + + def _startup_should_abort(self) -> bool: + return ( + self._restart_requested + or self._draining + or self._shutdown_event.is_set() + ) + + async def _abort_startup_if_shutdown_requested( + self, + adapter: Optional[BasePlatformAdapter] = None, + platform: Optional[Platform] = None, + ) -> bool: + """Clean up and exit startup when restart/shutdown begins mid-startup.""" + if not self._startup_should_abort(): + return False + if adapter is not None and platform is not None: + try: + await adapter.cancel_background_tasks() + except Exception as e: + logger.debug("✗ %s background-task cancel error: %s", platform.value, e) + await self._safe_adapter_disconnect(adapter, platform) + stop_task = self._stop_task + current_task = asyncio.current_task() + if stop_task is not None and stop_task is not current_task: + await stop_task + elif not self._shutdown_event.is_set(): + await self.stop( + restart=self._restart_requested, + detached_restart=self._restart_detached, + service_restart=self._restart_via_service, + ) + return True + + def _start_loop_liveness_guards(self, loop: asyncio.AbstractEventLoop) -> None: + """Arm the selector floor and out-of-loop watchdog before adapters. + + Disabled entirely with ``gateway.loop_watchdog: false`` in config.yaml (config-only knob). + """ + from gateway.run import _arm_loop_floor_timer, start_loop_liveness_watchdog + config = getattr(self, "config", None) + if config is not None and not getattr(config, "loop_watchdog", True): + return + if getattr(self, "_loop_floor_timer_handle", None) is None: + try: + self._loop_floor_timer_handle = _arm_loop_floor_timer(loop) + except Exception: + logger.debug("Failed to arm gateway loop floor timer", exc_info=True) + + watchdog = getattr(self, "_loop_liveness_watchdog", None) + if watchdog is None or not watchdog.is_alive(): + try: + # getattr defaults cover the config=None / bare-object test path; config-loaded + # values are already validated+clamped by GatewayConfig.from_dict; no re-clamping. + interval = getattr( + config, + "loop_watchdog_probe_interval_s", + DEFAULT_LOOP_WATCHDOG_INTERVAL_S, + ) + timeout = getattr( + config, + "loop_watchdog_probe_timeout_s", + DEFAULT_LOOP_WATCHDOG_TIMEOUT_S, + ) + strikes = getattr( + config, + "loop_watchdog_max_strikes", + DEFAULT_LOOP_WATCHDOG_MAX_STRIKES, + ) + self._loop_liveness_watchdog = start_loop_liveness_watchdog( + loop, + probe_interval=float(interval), + probe_timeout=float(timeout), + max_strikes=int(strikes), + ) + except Exception: + logger.debug("Failed to start gateway loop liveness watchdog", exc_info=True) + + def _stop_loop_liveness_guards(self) -> None: + """Disarm lifetime liveness guards before shutdown can load the loop.""" + watchdog = getattr(self, "_loop_liveness_watchdog", None) + self._loop_liveness_watchdog = None + if watchdog is not None: + try: + watchdog.stop() + except Exception: + logger.debug("Failed to stop gateway loop liveness watchdog", exc_info=True) + + floor_timer = getattr(self, "_loop_floor_timer_handle", None) + self._loop_floor_timer_handle = None + if floor_timer is not None: + try: + floor_timer.cancel() + except Exception: + logger.debug("Failed to cancel gateway loop floor timer", exc_info=True) + + # Also disarm the heartbeat writer task: once shutdown starts loading the loop, a heartbeat + # that keeps refreshing the file makes a draining gateway look healthy to external probes. + heartbeat = getattr(self, "_loop_heartbeat_task", None) + self._loop_heartbeat_task = None + if heartbeat is not None: + try: + heartbeat.cancel() + except Exception: + logger.debug("Failed to cancel gateway loop heartbeat task", exc_info=True) + + async def _consume_clean_shutdown_marker(self, marker_path) -> int: + """Discard orphan turn markers before consuming a clean-exit receipt. + + If persistence or marker removal fails, startup must fail closed: continuing with the old + receipt would let a later unclean exit masquerade as clean and discard interrupted turns. + """ + discarded = await self.async_session_store.discard_active_turn_markers() + marker_path.unlink() + return discarded + + async def _recover_unclean_sessions(self) -> tuple[int, int]: + """Recover exact active turns, then run the legacy recency fallback.""" + from gateway.run import _float_env + exact = 0 + fallback = 0 + try: + agent_timeout = max(1.0, _float_env("HERMES_AGENT_TIMEOUT", 1800)) + marker_max_age = max(60 * 60, int(agent_timeout * 2)) + exact = await self.async_session_store.recover_interrupted_turns( + max_age_seconds=marker_max_age + ) + except Exception as exc: + logger.warning("Exact active-turn recovery on startup failed: %s", exc) + try: + fallback = await self.async_session_store.suspend_recently_active( + max_age_seconds=120 + ) + except Exception as exc: + logger.warning("Legacy session recovery on startup failed: %s", exc) + return exact, fallback + + @staticmethod + def _start_hosted_room_worker_sync(): + """Start the local Group Chat worker without importing the dashboard.""" + + import tui_gateway.server # noqa: F401 + from tui_gateway import methods_groups + + service = methods_groups.get_hosted_room_service() + if service is None: + service = methods_groups.start_hosted_room_service() + if service is None: + raise RuntimeError("Group Chat worker has no bound session backend") + status = service.runtime.status() + if not status.get("running") or status.get("stopping"): + raise RuntimeError("Group Chat worker did not start") + return service + + async def _ensure_hosted_room_worker(self): + return await asyncio.to_thread(self._start_hosted_room_worker_sync) + + async def _hosted_room_worker_watcher(self, interval: float = 1.0) -> None: + """Keep the room worker alive for the messaging gateway lifetime.""" + + while self._running: + await self._ensure_hosted_room_worker() + await asyncio.sleep(interval) + + async def _stop_hosted_room_worker(self, timeout: float = 5.0) -> bool: + """Pause room execution durably without interrupting accepted turns.""" + + from tui_gateway import methods_groups + + return await asyncio.to_thread( + methods_groups.stop_hosted_room_service, + timeout=timeout, + ) + + def _start_loop_heartbeat_task(self) -> None: + """Start the loop-liveness heartbeat task, idempotent. + + An asyncio task so a frozen loop stops refreshing ``state/gateway.heartbeat``; cancelled + with the other background tasks in stop(). Best-effort — must never abort startup. + """ + try: + _existing_hb = getattr(self, "_loop_heartbeat_task", None) + if _existing_hb is not None and not _existing_hb.done(): + return + self._loop_heartbeat_task = asyncio.create_task( + loop_heartbeat_forever( + interval_s=DEFAULT_HEARTBEAT_INTERVAL_S, + start_time=getattr(self, "_gateway_started_at", 0.0), + ) + ) + # PERMANENT for the process lifetime, same as a _spawn_supervised watcher — tag it so + # _scale_to_zero_has_live_background_work() doesn't treat an armed, otherwise-idle + # gateway as busy forever. + self._loop_heartbeat_task._hermes_supervised_watcher = True # type: ignore[attr-defined] + _bg = getattr(self, "_background_tasks", None) + if _bg is not None: + _bg.add(self._loop_heartbeat_task) + self._loop_heartbeat_task.add_done_callback(_bg.discard) + except Exception: + logger.debug("Failed to start gateway loop heartbeat", exc_info=True) + + def _start_install_faulthandler(self) -> None: + """Enable faulthandler (stderr or a log file) plus the SIGUSR2 stack-dump hook.""" + from gateway.run import get_hermes_home + # Enable faulthandler for stack dumps on freezes/crashes. Falls back to a log file when + # sys.stderr is None (Windows VBS / pythonw / detached service) — otherwise the gateway + # would die here and take every adapter offline. + try: + faulthandler.enable() + except (RuntimeError, ValueError, OSError): + try: + _fh_log_dir = getattr(self.config, "log_dir", None) or os.path.join( + str(get_hermes_home()), + "logs", + ) + os.makedirs(_fh_log_dir, exist_ok=True) + _fh_enable_path = os.path.join(_fh_log_dir, "gateway_faulthandler.log") + _fh_enable_file = open(_fh_enable_path, "a", encoding="utf-8") + faulthandler.enable(file=_fh_enable_file, all_threads=True) + except Exception: + logger.debug("faulthandler.enable() unavailable", exc_info=True) + # Also dump stacks to a rotating file for off-line analysis under a service manager that + # doesn't capture stderr. faulthandler.register()/SIGUSR2 are POSIX-only: skip the signal- + # triggered file dump on Windows (faulthandler.enable() above still covers fatal errors). + _sigusr2 = getattr(signal, "SIGUSR2", None) + if _sigusr2 is not None and hasattr(faulthandler, "register"): + try: + _log_dir = getattr(self.config, "log_dir", None) or os.path.join( + str(get_hermes_home()), + "logs", + ) + _faulthandler_path = os.path.join(_log_dir, "gateway_faulthandler.log") + os.makedirs(_log_dir, exist_ok=True) + _fh = open(_faulthandler_path, "a", encoding="utf-8") + faulthandler.register( + _sigusr2, + file=_fh, + all_threads=True, + chain=True, + ) + except Exception: + logger.debug("Could not set up faulthandler file logging", exc_info=True) + + def _start_log_startup_environment(self) -> None: + """Bind the gateway loop, disarm the startup watchdog, and log the startup environment.""" + try: + self._gateway_loop = asyncio.get_running_loop() + except RuntimeError: + self._gateway_loop = None + if self._gateway_loop is not None: + self._start_loop_liveness_guards(self._gateway_loop) + # Loop confirmed live: the startup-liveness watchdog is done and the loop-liveness + # watchdog (armed above) takes over. Disarm even when loop guards are config-disabled — + # the startup watchdog covers only the pre-loop window. Deliberately inside this branch: + # if the loop isn't live, startup has NOT reached the milestone and it must stay armed. + try: + from gateway.startup_watchdog import disarm_startup_watchdog + + disarm_startup_watchdog() + except Exception: + logger.debug("Startup watchdog disarm failed", exc_info=True) + logger.info("Session storage: %s", self.config.sessions_dir) + + # Sanity-check that systemd's TimeoutStopSec covers our drain window: a unit file from + # before a hermes-agent upgrade (no ``hermes setup`` re-run) may encode the old default, + # so SIGKILL hits mid-drain and looks like a phantom kill in the journal. Never raises. + try: + from gateway.shutdown_forensics import check_systemd_timing_alignment + _alignment = check_systemd_timing_alignment( + self._restart_drain_timeout, + getattr(self, "_cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT), + ) + if _alignment is not None and _alignment.get("mismatch"): + logger.warning( + "Stale systemd unit detected: %s has TimeoutStopSec=%.0fs but " + "drain_timeout=%.0fs cron_drain_timeout=%.0fs (expected >=%.0fs). " + "systemd may SIGKILL the gateway mid-drain. Run " + "`hermes gateway install --force` to regenerate the unit, or " + "shorten agent.restart_drain_timeout / agent.cron_drain_timeout.", + _alignment.get("unit", "(unknown)"), + _alignment["timeout_stop_sec"], + _alignment["drain_timeout"], + _alignment.get( + "cron_drain_timeout", DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT + ), + _alignment["expected_min"], + ) + except Exception as _e: + logger.debug("check_systemd_timing_alignment failed: %s", _e) + # Log the resolved max_iterations budget so operators can verify the config.yaml → env + # bridge at a glance (instead of silently running at a stale .env value for weeks). + try: + _effective_max_iter = int(os.getenv("HERMES_MAX_ITERATIONS", "500")) + logger.info( + "Agent budget: max_iterations=%d (agent.max_turns from config.yaml, " + "or HERMES_MAX_ITERATIONS from .env, or default 500)", + _effective_max_iter, + ) + except Exception: + pass + # Redaction is ON by default; warn prominently when an operator has explicitly opted out so + # the downgrade isn't forgotten. The redactor snapshots its state at import time, so this + # log line is the source of truth for the process lifetime. + try: + _redact_raw = os.getenv("HERMES_REDACT_SECRETS", "true") + _redact_on = _redact_raw.lower() in {"1", "true", "yes", "on"} + if _redact_on: + logger.info( + "Secret redaction: ENABLED (tool output, logs, and chat " + "responses are scrubbed before delivery)" + ) + else: + logger.warning( + "Secret redaction: DISABLED (HERMES_REDACT_SECRETS=%s). " + "API keys and tokens may appear verbatim in chat output, " + "session JSONs, and logs. Set security.redact_secrets: true " + "in config.yaml to re-enable.", + _redact_raw, + ) + except Exception: + pass + try: + from hermes_cli.profiles import get_active_profile_name + _profile = get_active_profile_name() + if _profile and _profile != "default": + logger.info("Active profile: %s", _profile) + except Exception: + pass + try: + from gateway.status import write_runtime_status + write_runtime_status( + gateway_state="starting", + exit_reason=None, + clear_profile_platforms=True, + ) + except Exception: + pass + try: + from hermes_cli.config import load_config + from agent.monitoring.gateway_health_export import start_gateway_health_export + self._gateway_health_export_runtime = start_gateway_health_export(load_config()) + if getattr(self._gateway_health_export_runtime, "enabled", False): + logger.info("Gateway health OTLP export: enabled") + except Exception: + logger.debug("gateway health OTLP export startup failed", exc_info=True) + + # Log any active supply-chain security advisories. Deliberately does NOT block startup or + # surface inline to users — only the operator can act (uninstall, rotate credentials). + try: + from hermes_cli.security_advisories import ( + detect_compromised, + gateway_log_message, + ) + _adv_hits = detect_compromised() + _adv_msg = gateway_log_message(_adv_hits) + if _adv_msg: + logger.warning("%s", _adv_msg) + logger.warning( + "Run `hermes doctor` on the gateway host for full " + "remediation steps." + ) + except Exception: + logger.debug( + "security advisory check failed at gateway startup", + exc_info=True, + ) + + def _start_check_access_policy(self) -> bool: + """Warn about missing allowlists; return True when startup must be refused.""" + from gateway.run import ( + _OWN_POLICY_OPEN_ENV, + _own_policy_open_startup_violation, + _write_runtime_status_quiet, + ) + # Warn if no user allowlists are configured and open access is not opted in + _builtin_allowed_vars = ( + "TELEGRAM_ALLOWED_USERS", "DISCORD_ALLOWED_USERS", + "WHATSAPP_ALLOWED_USERS", "WHATSAPP_CLOUD_ALLOWED_USERS", + "SLACK_ALLOWED_USERS", + "SIGNAL_ALLOWED_USERS", "SIGNAL_GROUP_ALLOWED_USERS", + "TELEGRAM_GROUP_ALLOWED_USERS", + "TELEGRAM_GROUP_ALLOWED_CHATS", + "EMAIL_ALLOWED_USERS", + "SMS_ALLOWED_USERS", "MATTERMOST_ALLOWED_USERS", + "MATRIX_ALLOWED_USERS", "DINGTALK_ALLOWED_USERS", + "FEISHU_ALLOWED_USERS", + "WECOM_ALLOWED_USERS", + "WECOM_CALLBACK_ALLOWED_USERS", + "WEIXIN_ALLOWED_USERS", + "BLUEBUBBLES_ALLOWED_USERS", + "QQ_ALLOWED_USERS", + "YUANBAO_ALLOWED_USERS", + "GATEWAY_ALLOWED_USERS", + ) + _builtin_allow_all_vars = ( + "TELEGRAM_ALLOW_ALL_USERS", "DISCORD_ALLOW_ALL_USERS", + "WHATSAPP_ALLOW_ALL_USERS", "WHATSAPP_CLOUD_ALLOW_ALL_USERS", + "SLACK_ALLOW_ALL_USERS", + "SIGNAL_ALLOW_ALL_USERS", "EMAIL_ALLOW_ALL_USERS", + "SMS_ALLOW_ALL_USERS", "MATTERMOST_ALLOW_ALL_USERS", + "MATRIX_ALLOW_ALL_USERS", "DINGTALK_ALLOW_ALL_USERS", + "FEISHU_ALLOW_ALL_USERS", + "WECOM_ALLOW_ALL_USERS", + "WECOM_CALLBACK_ALLOW_ALL_USERS", + "WEIXIN_ALLOW_ALL_USERS", + "BLUEBUBBLES_ALLOW_ALL_USERS", + "QQ_ALLOW_ALL_USERS", + "YUANBAO_ALLOW_ALL_USERS", + ) + # Also pick up plugin-registered platforms — each entry can declare its own + # allowed_users_env / allow_all_env, so the warning stays accurate as plugins (IRC) arrive. + _plugin_allowed_vars: tuple = () + _plugin_allow_all_vars: tuple = () + try: + from gateway.platform_registry import platform_registry + _plugin_allowed_vars = tuple( + e.allowed_users_env for e in platform_registry.plugin_entries() + if e.allowed_users_env + ) + _plugin_allow_all_vars = tuple( + e.allow_all_env for e in platform_registry.plugin_entries() + if e.allow_all_env + ) + except Exception: + pass + _any_allowlist = any( + os.getenv(v) for v in _builtin_allowed_vars + _plugin_allowed_vars + ) + _allow_all = os.getenv("GATEWAY_ALLOW_ALL_USERS", "").lower() in {"true", "1", "yes"} or any( + os.getenv(v, "").lower() in {"true", "1", "yes"} + for v in _builtin_allow_all_vars + _plugin_allow_all_vars + ) + if not _any_allowlist and not _allow_all: + logger.warning( + "No env user allowlists configured. Messaging platforms default to " + "pairing/allowlist policies and will deny unknown senders unless you " + "configure platform allowlists (e.g., TELEGRAM_ALLOWED_USERS=your_id) " + "or explicitly opt in with GATEWAY_ALLOW_ALL_USERS=true plus " + "dm_policy/group_policy: open on the platform." + ) + + reason = _own_policy_open_startup_violation(self.config) + if reason: + platform_value = reason.split(":", 1)[0] + allow_all_env = None + for platform, open_env in _OWN_POLICY_OPEN_ENV.items(): + if platform.value == platform_value: + allow_all_env = open_env[2] + break + logger.error( + "Refusing to start: %s has dm_policy/group_policy set to 'open' " + "but neither GATEWAY_ALLOW_ALL_USERS nor %s is enabled.", + platform_value, + allow_all_env or "a platform allow-all flag", + ) + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) + self._request_clean_exit(reason) + return True + return False + + async def _start_recover_previous_run(self) -> None: + """Plugins, relay, hooks, then crash/clean-exit recovery of processes and sessions.""" + from gateway.run import _hermes_home + # Discover Python plugins before shell hooks so plugin block decisions take precedence in + # tie cases. Explicit here because the gateway lazily imports run_agent per request, so + # the discover_plugins() side-effect in model_tools.py is NOT guaranteed to have run yet. + try: + from hermes_cli.plugins import discover_plugins + discover_plugins() + except Exception: + logger.warning( + "plugin discovery failed at gateway startup", exc_info=True, + ) + + # Register the generic relay adapter only if GATEWAY_RELAY_URL / gateway.relay_url is set. + # No URL -> no-op, so direct/single-tenant deployments are unaffected. + try: + from gateway.relay import ( + register_relay_adapter, + relay_url, + self_provision_relay, + send_relay_policy, + ) + + # Boot-time relay self-provision: resolve the agent's NAS token -> POST /relay/provision + # -> set GATEWAY_RELAY_* in os.environ BEFORE registration reads them. Never raises. + self_provision_relay() + + if register_relay_adapter(): + logger.info("relay adapter registered (connector at %s)", relay_url()) + # Declare this gateway's relevance policy (mention-gating / free-response / allow- + # bots) to the connector so the SAME behavior governs relay delivery (Phase 6 Unit + # ζ). Runs after the secret is resolved; never raises, never blocks boot. + send_relay_policy() + except Exception: + logger.warning( + "relay adapter registration failed at gateway startup", exc_info=True, + ) + + # Register declarative shell hooks from cli-config.yaml. Gateway has no TTY, so consent must + # come from --accept-hooks, HERMES_ACCEPT_HOOKS, or hooks_auto_accept: true; pass + # accept_hooks=False and let register_from_config resolve env + config. Never blocks startup. + try: + from hermes_cli.config import load_config + from agent.shell_hooks import register_from_config + _hooks_cfg = load_config() + register_from_config(_hooks_cfg, accept_hooks=False) + + from agent.outbound_webhooks import ( + register_from_config as register_outbound_webhooks, + ) + register_outbound_webhooks(_hooks_cfg) + except Exception: + logger.debug( + "shell-hook registration failed at gateway startup", + exc_info=True, + ) + + # Discover and load event hooks + self.hooks.discover_and_load() + + # Recover background processes from checkpoint (crash recovery) + try: + from tools.process_registry import process_registry + recovered = process_registry.recover_from_checkpoint() + if recovered: + logger.info("Recovered %s background process(es) from previous run", recovered) + except Exception as e: + logger.warning("Process checkpoint recovery: %s", e) + + # Recover sessions active when the gateway last exited. Exact durable turn markers cover + # long-running work; the 120s recency heuristic remains as a fallback for turns from older + # versions without markers. SKIP after a clean shutdown — the previous process already drained. + _clean_marker = _hermes_home / ".clean_shutdown" + if _clean_marker.exists(): + logger.info("Previous gateway exited cleanly — skipping session suspension") + try: + discarded = await self._consume_clean_shutdown_marker(_clean_marker) + except Exception as exc: + logger.error( + "Clean-start marker cleanup failed; refusing startup so the " + "clean-exit receipt cannot mask a later unclean exit: %s", + exc, + ) + raise RuntimeError("clean-start recovery cleanup failed") from exc + if discarded: + logger.info( + "Discarded %d orphan active-turn marker(s) after clean shutdown", + discarded, + ) + else: + exact, fallback = await self._recover_unclean_sessions() + recovered = exact + fallback + if recovered: + logger.info( + "Marked %d in-flight session(s) as resumable from previous run " + "(%d exact, %d legacy)", + recovered, + exact, + fallback, + ) + + # Stuck-loop detection: a session active across 3+ consecutive restarts is probably looping + # (its history keeps hanging the agent); auto-suspend so the next message starts clean. + try: + stuck = self._suspend_stuck_loop_sessions() + if stuck: + logger.warning("Auto-suspended %d stuck-loop session(s)", stuck) + except Exception as e: + logger.debug("Stuck-loop detection failed: %s", e) + + async def _start_prefilter_platforms(self) -> Tuple[bool, int, list, list]: + """Create + wire an adapter per enabled platform (serial pre-filter, no connects). + + Returns (aborted, enabled_platform_count, multiplex_skipped_platforms, pending_connects). + """ + from gateway.run import _platform_has_bot_credential + enabled_platform_count = 0 + _multiplex_on = bool(getattr(self.config, "multiplex_profiles", False)) + _multiplex_skipped_platforms: list[Platform] = [] + # Initialize and connect each configured platform. connect() calls run concurrently so one + # slow/failing platform (e.g. Telegram behind a dead proxy) cannot delay the others by a + # full timeout window; the cheap serial pre-filter and per-platform timeouts are unchanged. + _pending_connects = [] # (platform, platform_config, adapter) + for platform, platform_config in self.config.platforms.items(): + if await self._abort_startup_if_shutdown_requested(): + return True, enabled_platform_count, _multiplex_skipped_platforms, _pending_connects + if not platform_config.enabled: + continue + # Under multiplexing, a platform may be enabled on the default profile's config.yaml + # while its bot token lives only in a secondary profile's .env. Starting the primary with + # an empty token fails at once and queues a reconnect loop that can never heal; the + # secondary starts its own adapter with the real token, so skip the empty primary. + if _multiplex_on and not _platform_has_bot_credential(platform, platform_config): + logger.info( + "Skipping %s on default profile: no bot credential in this " + "profile's secrets. Secondary multiplexed profiles that " + "provide the token will still connect.", + platform.value, + ) + _multiplex_skipped_platforms.append(platform) + continue + enabled_platform_count += 1 + + adapter = self._create_adapter(platform, platform_config) + if not adapter: + # Distinguish between missing builtin deps and missing plugin + _pval = platform.value + _builtin_names = {m.value for m in Platform.__members__.values()} + if _pval not in _builtin_names: + logger.warning( + "No adapter for '%s' -- is the plugin installed? " + "(platform is enabled in config.yaml but no plugin registered it)", + _pval, + ) + else: + logger.warning("No adapter available for %s", _pval) + continue + + # Set up message + fatal error handlers. Under multiplexing the default profile needs + # the same whole-handler runtime scope as a secondary profile: authorization and prompt + # rendering both run before the narrower agent-turn scope is installed. + adapter.set_message_handler(self._primary_message_handler()) + adapter.set_fatal_error_handler(self._handle_adapter_fatal_error) + adapter.set_session_store(self.session_store) + adapter.set_busy_session_handler(self._handle_active_session_busy_message) + _set_reaction = getattr(adapter, "set_reaction_handler", None) + if callable(_set_reaction): + _set_reaction(self._handle_reaction_event) + adapter.set_topic_recovery_fn(self._recover_telegram_topic_thread_id) + adapter.set_authorization_check(self._make_adapter_auth_check(adapter.platform)) + adapter.set_platform_event_handler(self._primary_platform_event_handler()) + adapter._busy_text_mode = self._busy_text_mode + _pending_connects.append((platform, platform_config, adapter)) + return False, enabled_platform_count, _multiplex_skipped_platforms, _pending_connects + + async def _start_connect_pending(self, _pending_connects: list) -> Optional[list]: + """Connect the pre-filtered adapters concurrently. + + Returns the raw per-platform results, or None when a restart/shutdown aborted startup + mid-connect (adapters already torn down). + """ + async def _connect_one_startup(p, p_cfg, adp): + """Connect a single platform; never let one block the others (#83791).""" + if await self._abort_startup_if_shutdown_requested(adp, p): + return (p, adp, p_cfg, "aborted", None) + logger.info("Connecting to %s...", p.value) + self._update_platform_runtime_status( + p.value, platform_state="connecting", error_code=None, error_message=None, + ) + try: + ok = await self._connect_initial_adapter_with_timeout(adp, p) + except Exception as _exc: # noqa: BLE001 - surfaced below as a retryable error + return (p, adp, p_cfg, "exception", _exc) + return (p, adp, p_cfg, "ok" if ok else "failed", None) + + if _pending_connects: + # Abort-aware concurrent wait (parity with the serial loop's between-platforms check): a + # restart/shutdown requested mid-connect must cancel still-pending connects, clean up the + # ones already completed, and abort startup. + _task_map: dict = {} + for (p, c, a) in _pending_connects: + _t = asyncio.ensure_future(_connect_one_startup(p, c, a)) + _task_map[_t] = (p, c, a) + _pending_tasks = set(_task_map) + _abort_mid_connect = False + while _pending_tasks: + _done, _pending_tasks = await asyncio.wait( + _pending_tasks, timeout=0.05 + ) + if _pending_tasks and self._startup_should_abort(): + _abort_mid_connect = True + break + if _abort_mid_connect: + # Cancel and fully settle the in-flight connects FIRST, so a completed adapter's + # disconnect cannot unblock a sibling's connect() before the sibling is cancelled. + for _t in _pending_tasks: + _t.cancel() + await asyncio.gather(*_pending_tasks, return_exceptions=True) + for _t in _pending_tasks: + _p, _c, _a = _task_map[_t] + try: + await _a.cancel_background_tasks() + except Exception as e: + logger.debug( + "✗ %s background-task cancel error: %s", _p.value, e + ) + await self._safe_adapter_disconnect(_a, _p) + # Tear down adapters whose connect already succeeded — they + # were never registered, so stop() won't reach them. + for _t, (_p, _c, _a) in _task_map.items(): + if _t in _pending_tasks or _t.cancelled(): + continue + _res = _t.exception() is None and _t.result() or None + if _res and _res[3] == "ok": + try: + await _a.cancel_background_tasks() + except Exception as e: + logger.debug( + "✗ %s background-task cancel error: %s", + _p.value, e, + ) + await self._safe_adapter_disconnect(_a, _p) + await self._abort_startup_if_shutdown_requested() + return None + _raw = [ + _t.exception() or _t.result() for _t in _task_map + ] + else: + _raw = [] + return _raw + + async def _start_aggregate_connect_results( + self, + _raw: list, + startup_retryable_errors: list, + startup_nonretryable_errors: list, + ) -> int: + """Apply connect outcomes to shared state; returns the connected adapter count.""" + connected_count = 0 + # Aggregate results single-threaded so shared state (self.adapters, self._failed_platforms, + # the error lists, connected_count) is mutated exactly as the original serial loop did -- + # only the connect() wall-clock overlap changed. + for _item in _raw: + if isinstance(_item, Exception): + # Unexpected escape from _connect_one_startup (shouldn't happen); + # log and skip rather than aborting the whole startup. + logger.error("Unexpected startup connect error: %s", _item) + continue + platform, adapter, platform_config, outcome, exc = _item + if outcome == "aborted": + continue + if outcome == "exception": + logger.error("\u2717 %s error: %s", platform.value, exc) + # Same defensive cleanup path for exceptions -- an adapter that raised mid-connect + # may still have a live aiohttp.ClientSession or child subprocess. + await self._safe_adapter_disconnect(adapter, platform) + self._update_platform_runtime_status( + platform.value, platform_state="retrying", error_code=None, error_message=str(exc), + ) + startup_retryable_errors.append(f"{platform.value}: {exc}") + # Unexpected exceptions are typically transient -- queue for retry + self._failed_platforms[platform] = { + "config": platform_config, + "attempts": 1, + "next_retry": time.monotonic() + 30, + "queued_at": time.monotonic(), + "credential_claim": self._adapter_credential_claim(platform, adapter), + "listener_claim": self._adapter_listener_claim(platform, adapter), + } + continue + if outcome == "ok": + self.adapters[platform] = adapter + self._sync_voice_mode_state_to_adapter(adapter) + # Wire voice input callback at connect time so voice + # transcription is forwarded without requiring /voice join. + self._bind_voice_input_callback(adapter) + connected_count += 1 + self._update_platform_runtime_status( + platform.value, platform_state="connected", error_code=None, error_message=None, + ) + logger.info("\u2713 %s connected", platform.value) + else: # outcome == "failed" + logger.warning("\u2717 %s failed to connect", platform.value) + # Defensive cleanup: a failed connect() may have allocated resources + # (aiohttp.ClientSession, poll tasks, bridge subprocesses) before giving up. + await self._safe_adapter_disconnect(adapter, platform) + if adapter.has_fatal_error: + # A live foreign holder of this bot token is a single-writer ownership conflict, + # not a blip — ``_acquire_platform_lock`` emits it retryable only so a MID-RUN + # reconnect can recover. At startup route it non-retryable: with nothing connected + # the gateway exits 78 instead of sitting alive and deaf in the retry queue. + _retryable = adapter.fatal_error_retryable and not ( + is_global_startup_conflict(adapter.fatal_error_code) + ) + self._update_platform_runtime_status( + platform.value, + platform_state="retrying" if _retryable else "fatal", + error_code=adapter.fatal_error_code, + error_message=adapter.fatal_error_message, + ) + target = ( + startup_retryable_errors + if _retryable + else startup_nonretryable_errors + ) + target.append(f"{platform.value}: {adapter.fatal_error_message}") + # Queue for reconnection if the error is retryable + if _retryable: + self._failed_platforms[platform] = { + "config": platform_config, + "attempts": 1, + "next_retry": time.monotonic() + 30, + "credential_claim": self._adapter_credential_claim(platform, adapter), + "listener_claim": self._adapter_listener_claim(platform, adapter), + } + else: + self._update_platform_runtime_status( + platform.value, platform_state="retrying", error_code=None, error_message="failed to connect", + ) + startup_retryable_errors.append(f"{platform.value}: failed to connect") + # No fatal error info means likely a transient issue -- queue for retry + self._failed_platforms[platform] = { + "config": platform_config, + "attempts": 1, + "next_retry": time.monotonic() + 30, + "queued_at": time.monotonic(), + "credential_claim": self._adapter_credential_claim(platform, adapter), + "listener_claim": self._adapter_listener_claim(platform, adapter), + } + return connected_count + + async def _start_secondary_profiles( + self, connected_count: int, _multiplex_skipped_platforms: list + ) -> Tuple[bool, int]: + """Bring up multiplexed secondary-profile adapters. Returns (aborted, connected_count).""" + from gateway.run import MultiplexConfigError, _write_runtime_status_quiet + # Multi-profile multiplexing: bring up adapters for every OTHER profile this gateway serves. + # Each profile's adapters connect under that profile's home + credential scope and stamp + # their inbound events with the profile so the agent turn resolves correctly. + try: + _secondary_connected = await self._start_secondary_profile_adapters() + connected_count += _secondary_connected + except MultiplexConfigError as e: + # Invalid multiplexer config — abort startup cleanly so the operator + # fixes config.yaml rather than running a half-wired gateway. + reason = str(e) + logger.error("Gateway multiplexer config error: %s", reason) + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) + self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE + self._request_clean_exit(reason) + self._startup_restore_in_progress = False + return True, connected_count + except Exception as e: + logger.error("Secondary-profile adapter startup failed: %s", e, exc_info=True) + finally: + # Startup authority is one phase, not a persistent runner mode. + # From this point onward every adapter retry is non-evicting. + self._platform_lock_takeover_on_start = False + + # A platform skipped on the primary for a missing credential should have been picked up by + # a secondary profile owning the token. If none did, it is enabled in config.yaml yet + # silently unserved — surface it loudly instead of leaving a quiet dead channel. + for _skipped in _multiplex_skipped_platforms: + _served_by_secondary = any( + _skipped in _profile_map + for _profile_map in self._profile_adapters.values() + ) + if not _served_by_secondary: + logger.warning( + "%s is enabled but no profile (default or secondary) " + "provided a bot credential for it — the platform is not " + "being served. Add its token to the profile that should " + "own it, or disable the platform.", + _skipped.value, + ) + return False, connected_count + + def _start_handle_no_connections( + self, + connected_count: int, + enabled_platform_count: int, + startup_retryable_errors: list, + startup_nonretryable_errors: list, + ) -> bool: + """Log/degrade when nothing connected; return True when startup must exit.""" + from gateway.run import _write_runtime_status_quiet + if connected_count == 0: + if startup_nonretryable_errors and not startup_retryable_errors: + reason = "; ".join(startup_nonretryable_errors) + logger.error("Gateway hit a non-retryable startup conflict: %s", reason) + _write_runtime_status_quiet(gateway_state="startup_failed", exit_reason=reason) + self._exit_code = GATEWAY_FATAL_CONFIG_EXIT_CODE + self._request_clean_exit(reason) + self._startup_restore_in_progress = False + return True + if startup_nonretryable_errors: + # Mixed failure mode: some platforms fatally misconfigured (e.g. WhatsApp never + # paired), others merely transient (e.g. Telegram TimedOut). Exiting 78 here would + # let exit-78 supervisors take the gateway PERMANENTLY down over a network blip and + # deny the retryable ones their retry. Log the fatal side loudly, then fall through to + # the degraded/retry path: the watcher recovers the retryable; the rest stay parked. + logger.error( + "%d platform(s) fatally misconfigured and parked: %s. " + "Staying alive so retryable platforms can recover.", + len(startup_nonretryable_errors), + "; ".join(startup_nonretryable_errors), + ) + if enabled_platform_count > 0: + if startup_retryable_errors: + # All enabled platforms hit retryable failures (network blip, bridge not paired, + # npm install timeout...). Keep the gateway alive so cron jobs still run and the + # reconnect watcher can recover the platforms once the cause is fixed; exiting + # here would turn one misconfigured platform into an infinite systemd restart loop. + reason = "; ".join(startup_retryable_errors) + logger.warning( + "Gateway started with no connected platforms — " + "%d platform(s) queued for retry: %s", + len(self._failed_platforms), reason, + ) + try: + from gateway.status import write_runtime_status + write_runtime_status( + gateway_state="degraded", + exit_reason=None, + ) + except Exception: + pass + # Fall through to the normal "running" state — reconnect watcher takes it from here. + # All enabled platforms had no adapter (missing library or credentials). Fleet nodes + # share one config.yaml but hold credentials for only a subset of platforms, so + # degrade gracefully and let cron jobs run. + logger.warning( + "No adapter could be created for any of the %d configured platform(s). " + "Check that required dependencies are installed and credentials are set. " + "Gateway will continue for cron job execution.", + enabled_platform_count, + ) + else: + logger.warning("No messaging platforms enabled.") + logger.info("Gateway will continue running for cron job execution.") + return False + + async def _start_finish_wiring(self, connected_count: int) -> None: + """Post-connect wiring: room worker, heartbeat, hooks, notifications, restore, watchers.""" + from gateway.run import ( + _hermes_home, + _planned_restart_notification_pending, + _restart_notification_pending, + ) + try: + await self._ensure_hosted_room_worker() + except Exception: + logger.error( + "Group Chat worker failed to start; mutating Group Chat commands " + "will fail closed until supervision recovers it", + exc_info=True, + ) + self._spawn_supervised( + self._hosted_room_worker_watcher, + "hosted_room_worker", + ) + + self._start_loop_heartbeat_task() + + # Emit gateway:startup hook + hook_count = len(self.hooks.loaded_hooks) + if hook_count: + logger.info("%s hook(s) loaded", hook_count) + await self.hooks.emit("gateway:startup", { + "platforms": [p.value for p in self.adapters], + }) + + if connected_count > 0: + logger.info("Gateway running with %s platform(s)", connected_count) + + # Build initial channel directory for send_message name resolution + try: + from gateway.channel_directory import build_channel_directory + directory = await build_channel_directory(self.adapters) + ch_count = sum(len(chs) for chs in directory.get("platforms", {}).values()) + logger.info("Channel directory built: %d target(s)", ch_count) + except Exception as e: + logger.warning("Channel directory build failed: %s", e) + + # Check if we're restarting after a /update command. If the update is + # still running, keep watching so we notify once it actually finishes. + notified = await self._send_update_notification() + if not notified and any( + path.exists() + for path in ( + _hermes_home / ".update_pending.json", + _hermes_home / ".update_pending.claimed.json", + ) + ): + self._schedule_update_notification_watch() + + # Give freshly connected adapters a brief moment to settle before sending restart/startup + # lifecycle messages; in practice this helps Discord thread deliveries after reconnect. + if connected_count > 0: + await asyncio.sleep(1.0) + + # Notify the chat that initiated /restart that the gateway is back. + chat_restart_notification_pending = _restart_notification_pending() + planned_restart_notification_pending = _planned_restart_notification_pending() + # Capture, before _send_restart_notification() unlinks the marker, whether this process + # booted from a chat-originated /restart. One-shot signal for the /restart redelivery + # guard (_is_stale_restart_redelivery): a missing dedup marker only suppresses a /restart + # when we KNOW we just came out of a restart cycle. + if chat_restart_notification_pending: + self._booted_from_restart = True + # Restart notification, home-channel startup notice, and obligation redelivery all call + # adapter.send(). Those sends must not pin the inbound restore gate — a Telegram flood- + # control sleep on this path froze every platform for the full penalty. + await self._await_startup_boot_sends( + planned_restart_notification_pending=planned_restart_notification_pending, + ) + + # Auto-continue fresh sessions interrupted by the previous restart/shutdown. resume_pending + # is cleared by the normal successful-turn path, so a failed auto-resume stays visible on the + # next user message. _await_startup_boot_sends already cleared sessions answered in the ledger. + self._schedule_resume_pending_sessions() + await self._finish_startup_restore() + + # Surface state.db init failures to the user's messaging platforms + # so they know persistence is broken before losing data (#88235). + await self._send_session_db_warning_notifications() + + # Drain any recovered process watchers (from crash recovery checkpoint) + try: + from tools.process_registry import process_registry + # Detach the current batch atomically: reassigning to a fresh list takes ownership of + # exactly the watchers present now, so any watcher appended concurrently during the + # yield below isn't silently dropped by a clear() on the shared list. + watchers = process_registry.pending_watchers + process_registry.pending_watchers = [] + # Process in batches of 100 with event-loop yield points to avoid + # O(n^2) event-loop blocking when recovering thousands of watchers. + for i, watcher in enumerate(watchers): + self._spawn_supervised( + lambda w=watcher: self._run_process_watcher(w), + f"process_watcher:{watcher.get('session_id')}", + restart=False, + ) + logger.info("Resumed watcher for recovered process %s", watcher.get("session_id")) + if i % 100 == 99: + await asyncio.sleep(0) + except Exception as e: + logger.error("Recovered watcher setup error: %s", e) + + def _start_spawn_background_watchers(self) -> None: + """Spawn the long-lived supervised background watchers.""" + # Start background session expiry watcher to finalize expired sessions + self._spawn_supervised(self._session_expiry_watcher, "session_expiry_watcher") + + # Keep the /model picker's remote catalogs (curated manifest, OpenRouter live list, Nous + # Portal recommendations) warm on disk so a delisted or newly-published model reaches the + # picker within one TTL window (model_catalog.ttl_minutes, default 20) without a cold open. + self._spawn_supervised(self._model_catalog_refresh_watcher, "model_catalog_refresh_watcher") + + # Stall watchdog: pending inbound + stale agent activity → warn user + # to /new (does not kill the turn; see agent.session_stall_timeout). + self._spawn_supervised(self._session_stall_watcher, "session_stall_watcher") + + # Start the kanban notifier — each gateway delivers events for subscriptions owned by the + # profiles whose adapters it hosts, even when another gateway owns the single dispatcher. + self._spawn_supervised(self._kanban_notifier_watcher, "kanban_notifier_watcher") + + # Start background kanban dispatcher — spawns workers for ready tasks. Gated by + # `kanban.dispatch_in_gateway` (default True). When false, users run `hermes kanban daemon` + # externally or simply don't use kanban; this loop becomes a no-op. + self._spawn_supervised(self._kanban_dispatcher_watcher, "kanban_dispatcher_watcher") + + # Start background reconnection watcher for platforms that failed at startup + if self._failed_platforms: + logger.info( + "Starting reconnection watcher for %d failed platform(s): %s", + len(self._failed_platforms), + ", ".join(p.value for p in self._failed_platforms), + ) + # Track the reconnect watcher task so _ensure_reconnect_watcher_running can detect death + # and respawn it. Spawned via _spawn_supervised so an exception escaping the watcher's OUTER + # loop is caught, logged, and restarted with backoff instead of silently killing it (else a + # platform already queued in _failed_platforms stays stranded: the ensure hook only runs on + # a NEW fatal-error arrival). ``on_spawn`` keeps ``_reconnect_watcher_task`` on the CURRENT + # live task across backoff respawns so a superseded handle never looks like a dead watcher. + self._spawn_reconnect_watcher() + + # Start background handoff watcher — picks up CLI sessions marked handoff_state='pending' in + # state.db and re-binds them to the destination platform's home channel, then forges a + # synthetic user turn so the agent kicks off the new chat. + self._spawn_supervised(self._handoff_watcher, "handoff_watcher") + + # Async-delegation watcher: drains delegate_task(background=true) completions and injects + # each result into its originating session as a new turn (covers the idle, no-turn case). + self._spawn_supervised(self._async_delegation_watcher, "async_delegation_watcher") + + # /loop wakeup watcher: scans persisted loops (SessionDB loop:* rows) and injects due + # wakeup prompts into their originating chats while the session is idle. + self._spawn_supervised(self._loop_wakeup_watcher, "loop_wakeup_watcher") + + # Start the scale-to-zero idle watcher ONLY when opted in (HERMES_SCALE_TO_ZERO stamp), + # messaging is relay-only/absent, and a wakeUrl is registered. When armed it drives the relay + # dormant on sustained idle, then suspends via flaps — Fly autostop is inbound-only, job-blind. + try: + if self._scale_to_zero_should_arm(): + logger.info( + "scale-to-zero: armed (idle timeout %.0fs) — watching for idle", + self._scale_to_zero_idle_timeout_seconds(), + ) + self._spawn_supervised(self._scale_to_zero_watcher, "scale_to_zero_watcher") + else: + # Surface WHY an OPTED-IN instance didn't arm (non-opted not arming is normal — + # stay silent); otherwise a failed arm is invisible and needs a box-dive. + self._log_scale_to_zero_not_armed_reason() + except Exception: # noqa: BLE001 - arming must never block startup + logger.debug("scale-to-zero: arm check failed at startup", exc_info=True) + + # Drain-control watcher: reconciles the gateway's new-turn accept-state with the external + # ``.drain_request.json`` marker the dashboard begin/cancel-drain endpoint writes. A marker + # from a prior instantiation (durable-volume restart) is ignored via its epoch. + self._spawn_supervised(self._drain_control_watcher, "drain_control_watcher") + + async def start(self) -> bool: + """Start the gateway and all configured platform adapters. + + Returns True if at least one adapter connected successfully. + """ + logger.info("Starting Hermes Gateway...") + self._start_install_faulthandler() + self._start_log_startup_environment() + if await self._abort_startup_if_shutdown_requested(): + return True + if self._start_check_access_policy(): + return True + await self._start_recover_previous_run() + + # Serialize startup restore against inbound dispatch: adapters can receive messages as soon + # as they connect, but restart-interrupted sessions are not auto-resumed until all startup + # wiring below completes, so inbound queues until every synthetic resume turn has finished. + self._startup_restore_in_progress = True + self._startup_restore_queue = [] + self._startup_restore_tasks = [] + # Fresh-boot readiness: with no resume_pending sessions the gate opens almost immediately + # while the turn machinery is still cold, so a message in that window got a skeleton system + # prompt. Warm NOW to overlap the connects below; _finish_startup_restore awaits it (bounded). + self._start_startup_warmup() + + startup_nonretryable_errors: list[str] = [] + startup_retryable_errors: list[str] = [] + ( + _aborted, + enabled_platform_count, + _multiplex_skipped_platforms, + _pending_connects, + ) = await self._start_prefilter_platforms() + if _aborted: + return True + + if await self._abort_startup_if_shutdown_requested(): + return True + _raw = await self._start_connect_pending(_pending_connects) + if _raw is None: + return True + connected_count = await self._start_aggregate_connect_results( + _raw, startup_retryable_errors, startup_nonretryable_errors + ) + + if await self._abort_startup_if_shutdown_requested(): + return True + _aborted, connected_count = await self._start_secondary_profiles( + connected_count, _multiplex_skipped_platforms + ) + if _aborted: + return True + if self._start_handle_no_connections( + connected_count, + enabled_platform_count, + startup_retryable_errors, + startup_nonretryable_errors, + ): + return True + + # Update delivery router with adapters + if await self._abort_startup_if_shutdown_requested(): + return True + self.delivery_router.adapters = self.adapters + self._wire_teams_pipeline_runtime() + + self._running = True + self._install_plugin_message_injector() + self._update_runtime_status("running") + await self._start_finish_wiring(connected_count) + self._start_spawn_background_watchers() + + logger.info("Press Ctrl+C to stop") + + return True + + async def _process_handoff( + self, row: Dict[str, Any], profile_name: Optional[str] = None, + ) -> None: + """Execute one handoff row. Raises on failure (caller marks failed). + + ``profile_name`` (``None`` = root) is the profile whose store queued this handoff. Under + multiplex it is load-bearing: ``self.adapters``/``self.config`` are the primary's (secondaries + live in ``_profile_adapters``), and the session key must be namespaced ``agent::...`` + or it binds a key nobody reads. Passing the name beats re-deriving it from the contextvar. + """ + from gateway.run import load_gateway_config, resolve_delivery_transport + from gateway.config import Platform + from gateway.session import SessionSource, build_session_key + from gateway.platforms.base import MessageEvent + + cli_session_id = row["id"] + platform_name = (row.get("handoff_platform") or "").strip().lower() + if not platform_name: + raise RuntimeError("handoff_platform is empty") + + # Resolve platform enum + try: + platform = Platform(platform_name) + except (ValueError, KeyError): + raise RuntimeError(f"unknown platform '{platform_name}'") + + # Resolve the config + adapter map for the profile that queued this handoff; single-profile + # gateways (or a default-profile handoff) fall back to self.config/self.adapters. + handoff_config = self.config + handoff_adapters = self.adapters + if profile_name and profile_name != "default": + secondary = (self._profile_adapters or {}).get(profile_name) + if not secondary: + raise RuntimeError( + f"profile '{profile_name}' has no live adapters in this gateway" + ) + handoff_adapters = secondary + # The watcher already entered _profile_runtime_scope, so a fresh load resolves THIS + # profile's config. Fail closed — self.config would deliver to the WRONG chat. + try: + handoff_config = load_gateway_config() + except Exception as exc: + logger.error( + "Handoff: could not load config for profile %s; " + "failing the handoff instead of delivering via the " + "primary's config", + profile_name, exc_info=True, + ) + raise RuntimeError( + f"could not load config for profile '{profile_name}': {exc}" + ) from exc + + # Adapter must be live. A relay-fronted gateway registers ONE adapter under Platform.RELAY + # fronting N logical platforms, so a literal adapters.get(discord) misses a deliverable + # platform; resolve_delivery_transport is the alias-aware resolver (native adapter wins). + transport = resolve_delivery_transport(platform, handoff_config, handoff_adapters) + if not transport: + raise RuntimeError( + f"platform '{platform_name}' is not active in this gateway" + ) + adapter = transport.adapter + + # Home channel must be configured + home = handoff_config.get_home_channel(platform) + if not home or not home.chat_id: + raise RuntimeError( + f"no home channel configured for {platform_name}; " + f"run /sethome on the desired chat first" + ) + + cli_title = row.get("title") or cli_session_id[:8] + + # Create a fresh thread on the destination so the handoff has its own scrollback. Adapter + # returns None if threading is unsupported (Matrix/WhatsApp/Signal/SMS) or creation failed. + thread_name = f"Hermes — {cli_title}" + try: + new_thread_id = await adapter.create_handoff_thread( + str(home.chat_id), thread_name, + ) + except Exception as exc: + logger.debug( + "Handoff: create_handoff_thread raised on %s: %s", + platform_name, exc, exc_info=True, + ) + new_thread_id = None + + effective_thread_id = new_thread_id or ( + str(home.thread_id) if home.thread_id else None + ) + + # Telegram private-chat DM topics are shaped differently from group/forum threads by the + # inbound adapter: a handoff-created topic in a positive chat_id must use the DM-topic source + # shape, or the synthetic turn binds a `thread` key while real replies arrive on a `dm` key. + home_chat_id = str(home.chat_id) + is_telegram_private_chat = ( + platform == Platform.TELEGRAM + and looks_like_telegram_private_chat_id(home_chat_id) + ) + + if new_thread_id and not is_telegram_private_chat: + dest_chat_type = "thread" + dest_user_id = "system:handoff" + else: + # No thread — assume DM-style. For Telegram private-chat topics use the real user id + # (== chat_id) so topic-mode checks and binding persistence match later inbound turns. + dest_chat_type = "dm" + dest_user_id = home_chat_id if is_telegram_private_chat else "system:handoff" + + # Discord (unlike Slack/Telegram) builds in-thread messages with ``chat_id == thread id``, + # so key on the thread's OWN id; keying on the parent would make the next reply spawn anew. + if platform == Platform.DISCORD and dest_chat_type == "thread" and effective_thread_id: + dest_chat_id = str(effective_thread_id) + else: + dest_chat_id = home_chat_id + dest_source = SessionSource( + platform=platform, + chat_id=dest_chat_id, + chat_name=home.name, + chat_type=dest_chat_type, + user_id=dest_user_id, + user_name="Handoff", + thread_id=effective_thread_id, + profile=profile_name, + ) + + # Build the session_key with the adapters' own rules so switch_session hits the right entry. + # Thread keys omit user_id (thread_sessions_per_user default) so the next message shares it. + platform_cfg = handoff_config.platforms.get(platform) + extra = platform_cfg.extra if platform_cfg else {} + # Namespace the key to the queuing profile: a multiplexed gateway would otherwise build + # ``agent:main:...`` while the profile's adapter routes inbound on ``agent::...``. + # The resolver is only the root fallback (None when multiplexing is off; old key unchanged). + # The isinstance check is load-bearing: a Mock store returns a truthy MagicMock. + handoff_profile = profile_name if (profile_name and profile_name != "default") else None + if handoff_profile is None: + try: + store = getattr(self.async_session_store, "_store", self.async_session_store) + resolver = getattr(store, "_resolve_profile_for_key", None) + if callable(resolver): + resolved = resolver(dest_source) + if isinstance(resolved, str) and resolved.strip(): + handoff_profile = resolved + except Exception: + logger.debug("Handoff: could not resolve profile namespace", exc_info=True) + session_key = build_session_key( + dest_source, + group_sessions_per_user=extra.get("group_sessions_per_user", True), + thread_sessions_per_user=extra.get("thread_sessions_per_user", False), + profile=handoff_profile, + ) + + # Ensure a session_store entry exists for this key (get_or_create_session creates one for a + # never-used home channel); switch_session then re-points it. + await self.async_session_store.get_or_create_session(dest_source) + + # Re-bind the destination key to the CLI session_id: switch_session ends the prior session + # in SQLite and reopens the CLI session under the new key; its transcript is now active. + switched = await self.async_session_store.switch_session(session_key, cli_session_id) + if switched is None: + raise RuntimeError( + f"could not switch session key {session_key} → {cli_session_id}" + ) + + # Evict any cached AIAgent for this session_key so the next dispatch + # rebuilds it against the CLI session_id (mirrors /resume / /branch). + self._evict_cached_agent(session_key) + + # Cancel any in-flight running-agent state for the destination key + # so the synthetic turn isn't queued behind a stale running flag. + self._release_running_agent_state(session_key) + + synthetic_text = ( + f"[Session was just handed off from CLI (\"{cli_title}\") to this " + f"channel. The full prior conversation history is loaded above. " + f"Briefly confirm you're working here and summarize what we were " + f"working on, so the user can continue from this device.]" + ) + + synthetic_event = MessageEvent( + text=synthetic_text, + source=dest_source, + internal=True, + ) + + logger.info( + "Handoff: dispatching synthetic turn for CLI session %s → %s " + "(home=%s, thread=%s, session_key=%s)", + cli_session_id, platform_name, home.chat_id, effective_thread_id, + session_key, + ) + + # Dispatch through the runner directly: adapter.handle_message would spawn a background task + # and lose error visibility; inline _handle_message keeps success/failure observable. + response_text = await self._handle_message(synthetic_event) + if not response_text: + # Streaming may have already delivered the response inline. + # Either way, agent ran without raising — count as success. + return + + # Send the reply to the new thread if we created one, else the configured home channel + # (which may carry a thread_id). Use the resolved transport (not adapter.send) so a + # relay-fronted logical platform is stamped on the outbound frame (send_for_platform). + send_metadata: Dict[str, Any] = {} + if effective_thread_id: + send_metadata["thread_id"] = effective_thread_id + try: + result = await transport.send( + platform, + str(home.chat_id), + response_text, + send_metadata or None, + ) + except Exception as exc: + raise RuntimeError(f"adapter.send failed: {exc}") from exc + + if not getattr(result, "success", True): + err = getattr(result, "error", "send returned success=False") + raise RuntimeError(f"adapter.send failed: {err}") diff --git a/gateway/run_topics.py b/gateway/run_topics.py new file mode 100644 index 0000000000..9ee08c773a --- /dev/null +++ b/gateway/run_topics.py @@ -0,0 +1,878 @@ +"""Telegram forum-topic and Discord auto-thread binding/rename methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import dataclasses +import re +from agent.compaction_display import project_compaction_message_for_display +from agent.i18n import t +from gateway.config import Platform +from gateway.platforms.base import MessageEvent, _prefix_within_utf16_limit, utf16_len +from gateway.session import SessionSource +from pathlib import Path +from typing import Optional, Tuple + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayTopicThreadsMixin: + """Telegram forum-topic and Discord auto-thread binding/rename methods for GatewayRunner.""" + + @staticmethod + def _telegram_topic_profile_name(source: SessionSource) -> str: + """Profile namespace for Telegram topic-mode rows. + + Use the profile stamped on the routed event (``source.profile``), never the process-global + active profile — under multiplex that mis-attributes topic state across bots sharing state.db. + """ + name = str(getattr(source, "profile", None) or "").strip() + return name if name else "default" + + def _telegram_topic_mode_enabled(self, source: SessionSource) -> bool: + """Return whether Telegram DM topic mode is active for this chat.""" + if source.platform != Platform.TELEGRAM or source.chat_type != "dm": + return False + session_db = getattr(self, "_session_db", None) + if session_db is None: + return False + # Runs off-loop (always via asyncio.to_thread); use the sync handle. + session_db = getattr(session_db, "_db", session_db) + try: + raw = session_db.is_telegram_topic_mode_enabled( + chat_id=str(source.chat_id), + user_id=str(source.user_id), + profile_name=self._telegram_topic_profile_name(source), + ) + except Exception: + logger.debug("Failed to read Telegram topic mode state", exc_info=True) + return False + # Only a real True from the SessionDB enables topic mode; anything else (including MagicMock + # from test fixtures that didn't opt in) means off for this chat. + return raw is True + + def _is_telegram_topic_root_lobby(self, source: SessionSource) -> bool: + """True for the main Telegram DM (or General topic) when topic mode has made it a lobby.""" + if source.platform != Platform.TELEGRAM or source.chat_type != "dm": + return False + if not self._telegram_topic_mode_enabled(source): + return False + tid = str(source.thread_id or "") + return tid in self._TELEGRAM_GENERAL_TOPIC_IDS + + def _is_telegram_topic_lane(self, source: SessionSource) -> bool: + """True for a user-created Telegram private-chat topic lane.""" + if source.platform != Platform.TELEGRAM or source.chat_type != "dm": + return False + if not self._telegram_topic_mode_enabled(source): + return False + tid = str(source.thread_id or "") + return bool(tid) and tid not in self._TELEGRAM_GENERAL_TOPIC_IDS + + def _telegram_topic_cooldown_key(self, source: SessionSource) -> Optional[str]: + """Cooldown key for topic-mode cooldowns: (profile, chat_id). + + Profiles sharing a Telegram private chat_id under multiplex must not + suppress each other's lobby reminders / capability hints (#76423). + """ + chat_id = str(source.chat_id or "") + if not chat_id: + return None + return f"{self._telegram_topic_profile_name(source)}:{chat_id}" + + def _should_send_telegram_lobby_reminder(self, source: SessionSource) -> bool: + """Rate-limit root-DM lobby reminders to one per cooldown window, not one per prompt typed.""" + if not hasattr(self, "_telegram_lobby_reminder_ts"): + self._telegram_lobby_reminder_ts = {} + key = self._telegram_topic_cooldown_key(source) + if not key: + return True + import time as _time + now = _time.monotonic() + last = self._telegram_lobby_reminder_ts.get(key, 0.0) + if now - last < self._TELEGRAM_LOBBY_REMINDER_COOLDOWN_S: + return False + self._telegram_lobby_reminder_ts[key] = now + return True + + def _telegram_topic_root_lobby_message(self) -> str: + return ( + "This main chat is reserved for system commands.\n\n" + "To start a new Hermes chat, open the All Messages topic at the top " + "of this bot interface and send any message there. Telegram will " + "create a new topic for that message; each topic works as an " + "independent Hermes session." + ) + + def _telegram_topic_root_new_message(self) -> str: + return ( + "To start a new parallel Hermes chat, open the All Messages topic " + "at the top of this bot interface and send any message there. " + "Telegram will create a new topic for it.\n\n" + "Each topic is an independent Hermes session. Use /new inside an " + "existing topic only if you want to replace that topic's current session." + ) + + def _telegram_topic_new_header(self, source: SessionSource) -> Optional[str]: + if not self._is_telegram_topic_lane(source): + return None + return ( + "Started a new Hermes session in this topic.\n\n" + "Tip: for parallel work, open All Messages and send a message there " + "to create a separate topic instead of using /new here. /new replaces " + "the session attached to the current topic." + ) + + def _record_telegram_topic_binding( + self, + source: SessionSource, + session_entry, + ) -> None: + """Persist the Telegram topic -> Hermes session binding for topic lanes.""" + session_db = getattr(self, "_session_db", None) + if session_db is None or not source.chat_id or not source.thread_id: + return + # Runs off-loop (always via asyncio.to_thread); use the sync handle. + session_db = getattr(session_db, "_db", session_db) + session_db.bind_telegram_topic( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + user_id=str(source.user_id or ""), + session_key=session_entry.session_key, + session_id=session_entry.session_id, + profile_name=self._telegram_topic_profile_name(source), + ) + + def _sync_telegram_topic_binding( + self, + source: SessionSource, + session_entry, + *, + reason: str, + ) -> None: + """Update the topic binding to point at ``session_entry.session_id``. + + Topic lanes persist (chat_id, thread_id) -> session_id so reopening a topic resumes the + right session. When compression rotates the id mid-turn a stale binding reloads the + oversized parent next message, retriggering preflight compression — sometimes in a loop. + """ + if not self._is_telegram_topic_lane(source): + return + try: + self._record_telegram_topic_binding(source, session_entry) + except Exception: + logger.debug( + "telegram topic binding refresh failed (%s)", reason, exc_info=True, + ) + + def _recover_telegram_topic_thread_id( + self, + source: SessionSource, + ) -> Optional[str]: + """Pin DM-topic routing to the user's last-active topic. + + Telegram can omit ``message_thread_id`` or surface General (``1``) for topic-mode DM + replies; in those lobby-shaped cases keep the conversation on the user's most-recent bound + topic. Do not rewrite a non-lobby, previously-unbound thread id: a brand-new DM topic is + also "unknown" until its first inbound message is recorded, and rewriting would send its + answer into an older lane. Returns None to leave the source alone. + """ + if ( + source.platform != Platform.TELEGRAM + or source.chat_type != "dm" + or not source.chat_id + or not source.user_id + or not self._telegram_topic_mode_enabled(source) + ): + return None + inbound = str(source.thread_id or "") + is_lobby = not inbound or inbound in self._TELEGRAM_GENERAL_TOPIC_IDS + if not is_lobby: + # A non-lobby, unknown thread_id is likely the first message of a new Telegram DM topic: + # preserve it to be recorded as a new lane below rather than hijack the latest binding. + return None + session_db = getattr(self, "_session_db", None) + if session_db is None: + return None + # Runs off-loop (always via asyncio.to_thread); use the sync handle. + session_db = getattr(session_db, "_db", session_db) + try: + bindings = session_db.list_telegram_topic_bindings_for_chat( + chat_id=str(source.chat_id), + profile_name=self._telegram_topic_profile_name(source), + ) + except Exception: + logger.debug("topic-recover: read failed", exc_info=True) + return None + if not bindings: + return None + user_id = str(source.user_id) + for b in bindings: # newest-first + if str(b.get("user_id") or "") == user_id: + recovered = str(b.get("thread_id") or "") + if recovered and recovered != inbound: + return recovered + return None + return None + + async def _get_telegram_topic_capabilities(self, source: SessionSource) -> dict: + """Read Telegram private-topic capability flags via Bot API getMe.""" + adapter = self._adapter_for_source(source) + bot = getattr(adapter, "_bot", None) + if bot is None or not hasattr(bot, "get_me"): + return {"checked": False} + try: + me = await bot.get_me() + except Exception: + logger.debug("Failed to fetch Telegram getMe topic capabilities", exc_info=True) + return {"checked": False} + + def _field(name: str): + if hasattr(me, name): + return getattr(me, name) + api_kwargs = getattr(me, "api_kwargs", None) + if isinstance(api_kwargs, dict) and name in api_kwargs: + return api_kwargs.get(name) + if isinstance(me, dict): + return me.get(name) + return None + + return { + "checked": True, + "has_topics_enabled": _field("has_topics_enabled"), + "allows_users_to_create_topics": _field("allows_users_to_create_topics"), + } + + async def _ensure_telegram_system_topic(self, source: SessionSource) -> None: + """Create/pin the managed System topic after /topic activation when possible.""" + adapter = self._adapter_for_source(source) + if adapter is None or not source.chat_id: + return + + thread_id = None + create_topic = getattr(adapter, "_create_dm_topic", None) + if callable(create_topic): + try: + thread_id = await create_topic(int(source.chat_id), "System") + except Exception: + logger.debug("Failed to create Telegram System topic", exc_info=True) + if not thread_id: + return + + message_id = None + try: + send_result = await adapter.send( + source.chat_id, + "System topic for Hermes commands and status.", + metadata={"thread_id": str(thread_id)}, + ) + message_id = getattr(send_result, "message_id", None) + except Exception: + logger.debug("Failed to send Telegram System topic intro", exc_info=True) + if not message_id: + return + + bot = getattr(adapter, "_bot", None) + if bot is None or not hasattr(bot, "pin_chat_message"): + return + try: + await bot.pin_chat_message( + chat_id=int(source.chat_id), + message_id=int(message_id), + disable_notification=True, + ) + except Exception: + logger.debug("Failed to pin Telegram System topic intro", exc_info=True) + + async def _send_telegram_topic_setup_image(self, source: SessionSource) -> None: + """Send the bundled BotFather Threads Settings screenshot when available.""" + adapter = self._adapter_for_source(source) + if adapter is None or not source.chat_id or not hasattr(adapter, "send_image_file"): + return + image_path = Path(__file__).resolve().parent / "assets" / "telegram-botfather-threads-settings.jpg" + if not image_path.exists(): + return + try: + await adapter.send_image_file( + chat_id=source.chat_id, + image_path=str(image_path), + caption="BotFather → Bot Settings → Threads Settings", + metadata={"thread_id": str(source.thread_id)} if source.thread_id else None, + ) + except Exception: + logger.debug("Failed to send Telegram topic setup image", exc_info=True) + + def _sanitize_telegram_topic_title(self, title: str) -> str: + """Return a Bot API-safe forum topic name from a generated session title.""" + cleaned = re.sub(r"\s+", " ", str(title or "")).strip() + if not cleaned: + return "Hermes Chat" + # Telegram forum topic names are short (currently 1-128 chars). Keep + # extra room for multi-byte titles and avoid trailing ellipsis churn. + if len(cleaned) > 120: + cleaned = cleaned[:117].rstrip() + "..." + return cleaned + + def _is_discord_auto_thread_lane(self, source: SessionSource) -> bool: + """Return True only for Discord threads Hermes just auto-created.""" + return ( + source.platform == Platform.DISCORD + and source.chat_type == "thread" + and bool(getattr(source, "auto_thread_created", False)) + and bool(source.thread_id) + and bool(getattr(source, "auto_thread_initial_name", None)) + ) + + def _is_relay_discord_channel_lane(self, source: SessionSource) -> bool: + """Shape-only check: a relay-delivered Discord CHANNEL event whose + reply the connector MAY auto-thread (title-turn registration gate). + + Deliberately does NOT consult the send-result cache: at registration + time (before delivery) the feedback can't exist yet. The rename lane + polls the cache at fire time instead.""" + return ( + source.platform == Platform.DISCORD + and bool(source.chat_id) + and not source.thread_id + and source.chat_type in ("group", "channel") + and getattr(source, "delivered_via_upstream_relay", False) is True + ) + + def _relay_auto_thread_info( + self, source: SessionSource + ) -> Optional[Tuple[str, str]]: + """(thread_id, initial_name) when the RELAY connector auto-threaded our reply to this + source's chat — the title-turn sibling of _is_discord_auto_thread_lane. + + The marker check only matches events ARRIVING IN an auto-created thread (turn 2+); the + auto-title fires on the FIRST exchange, whose source is the PARENT channel event with no + markers. Preferred: the connector's ``prospective_thread_id`` stamp (anchor message id == + the thread it will create) — per-message, so it names the EXACT thread even when several + auto-threads spawn from one channel; the connector's created-name guard enforces + no-clobber. Fallback: the per-chat send-result thread_id/auto_thread_name cache (older + connectors), which only ever renamed the FIRST thread. + """ + from gateway.run import _as_thread_info + if source.platform != Platform.DISCORD or not source.chat_id: + return None + if not getattr(source, "delivered_via_upstream_relay", False): + return None + prospective = getattr(source, "prospective_thread_id", None) + if prospective: + # Deterministic per-thread identity; the empty initial-name marker + # signals the caller to rely on the connector-side no-clobber guard. + return (str(prospective), "") + adapter = self._adapter_for_source(source) + info_fn = getattr(adapter, "auto_thread_info_for_chat", None) + if not callable(info_fn): + return None + try: + return _as_thread_info(info_fn(str(source.chat_id))) + except Exception: + return None + + async def _await_relay_auto_thread_info( + self, source: SessionSource + ) -> Optional[Tuple[str, str]]: + """``_relay_auto_thread_info``, waited out until this turn delivers. + + The legacy send-result path can only answer once the reply is sent, and the caller asks + at title time — one turn early. The adapter answers on the send either way, so the + timeout is only a backstop for a turn that never sends at all; the turn's own inactivity + limit is exactly how long that turn could still be alive. + """ + from gateway.run import _as_thread_info, _float_env + # The connector-stamped prospective id is known at ingest, so most + # sessions answer here and never wait at all. + known = self._relay_auto_thread_info(source) + if known is not None: + return known + adapter = self._adapter_for_source(source) + wait_fn = getattr(adapter, "wait_for_auto_thread_info", None) + if not callable(wait_fn) or not source.chat_id: + return None + # 0 means the operator disabled the turn limit; the backstop still needs one. + timeout = _float_env("HERMES_AGENT_TIMEOUT", 1800) or 1800 + try: + return _as_thread_info(await wait_fn(str(source.chat_id), timeout)) + except Exception: + return None + + def _sanitize_discord_thread_title(self, title: str) -> str: + """Return a Discord-safe semantic thread title from a session title. + + Discord thread names are capped at 100 characters measured in UTF-16 code units (emoji + count double), so truncate with the UTF-16 helpers rather than Python code-point slices. + """ + cleaned = re.sub(r"\s+", " ", str(title or "")).strip() + if not cleaned: + return "Hermes Chat" + if utf16_len(cleaned) > 80: + cleaned = _prefix_within_utf16_limit(cleaned, 77).rstrip() + "..." + return cleaned + + async def _rename_discord_auto_thread_for_session_title( + self, + source: SessionSource, + session_id: str, + title: str, + relay_info: Optional[Tuple[str, str]] = None, + ) -> None: + """Best-effort semantic rename of a newly auto-created Discord thread. + + ``relay_info`` is the (thread_id, initial_name) pair from the relay connector's send- + result feedback — supplied on the title turn, where the source is the parent-channel + event and carries no auto-thread markers (see _relay_auto_thread_info). + """ + if relay_info is None and not await asyncio.to_thread( + self._is_discord_auto_thread_lane, source + ): + # Relay title turn with no feedback captured at schedule time: the title comes off the + # user's opening message, so it beats the delivery that produces the connector's send- + # result feedback (thread_id + initial name) by the whole length of the turn. + if not self._is_relay_discord_channel_lane(source): + return + relay_info = await self._await_relay_auto_thread_info(source) + if relay_info is None: + # True miss: the connector did not auto-thread this reply + # (policy off, DM, already-threaded, or send failed). + return + adapter = self._adapter_for_source(source) if getattr(self, "adapters", None) else None + if adapter is None: + return + rename_thread = getattr(adapter, "rename_thread", None) + if rename_thread is None: + return + target_thread_id = relay_info[0] if relay_info else str(source.thread_id) + # Relay lane (relay_info present): ask the CONNECTOR to enforce the no-clobber guard from + # its own created-name memory — the gateway can't reliably reproduce the thread's initial + # name byte-for-byte (normalization drift silently declined every rename before this). + use_connector_guard = relay_info is not None + guard_name = ( + None + if use_connector_guard + else getattr(source, "auto_thread_initial_name", None) + ) + thread_name = self._sanitize_discord_thread_title(title) + # Relay lane only: the connector's egress guard resolves the owning tenant from the + # outbound scope_id/user_id caches, keyed by the PARENT channel chat_id (learned at + # inbound), not the thread id. rename_thread defaults chat_id to the thread id, so the + # lookup misses and the connector declines; pass the parent channel id (the relay source's + # chat_id). Native lane needs nothing: its source IS the thread, direct Discord API. + parent_chat_id = ( + str(source.chat_id) if use_connector_guard and source.chat_id else None + ) + logger.info( + "discord auto-thread rename: thread=%s lane=%s new_title=%r", + target_thread_id, + "relay" if use_connector_guard else "native", + thread_name, + ) + rename_kwargs = ( + { + "prefer_connector_created": True, + "parent_chat_id": parent_chat_id, + } + if use_connector_guard + else {"only_if_current_name": guard_name} + ) + try: + renamed = await rename_thread( + target_thread_id, + thread_name, + **rename_kwargs, + ) + logger.info( + "discord auto-thread rename result: thread=%s applied=%s", + target_thread_id, + bool(renamed), + ) + except TypeError: + logger.warning( + "Discord semantic thread rename raised TypeError (adapter=%s)", + type(adapter).__name__, + exc_info=True, + ) + except Exception: + logger.debug("Failed to rename Discord auto-thread for generated session title", exc_info=True) + + def _schedule_rename_from_title_thread(self, source: SessionSource, make_coro, label: str) -> None: + """Schedule a best-effort rename coroutine onto the gateway loop from the auto-title thread. + + The source is copied so the background thread never shares the live dataclass with the + loop; failures are logged at debug and never propagate.""" + from gateway.run import safe_schedule_threadsafe + try: + loop = asyncio.get_running_loop() + except RuntimeError: + loop = getattr(self, "_gateway_loop", None) + if loop is None or loop.is_closed(): + return + try: + copied_source = dataclasses.replace(source) + except Exception: + copied_source = source + future = safe_schedule_threadsafe( + make_coro(copied_source), + loop, + logger=logger, + log_message=f"{label} failed to schedule", + ) + if future is None: + return + + def _log_rename_failure(fut) -> None: + try: + fut.result() + except Exception: + logger.debug("%s failed", label, exc_info=True) + + future.add_done_callback(_log_rename_failure) + + def _schedule_discord_semantic_thread_rename( + self, + source: SessionSource, + session_id: str, + title: str, + ) -> None: + """Schedule Discord auto-thread rename from the auto-title background thread.""" + relay_info = None + if not title: + return + if not self._is_discord_auto_thread_lane(source): + # Relay title turn: the source is the PARENT channel event (thread didn't exist at + # ingest, no auto-thread markers). The connector's send-result feedback says where the + # reply landed, but the auto-title races that delivery, so a cache miss HERE is not a + # verdict. Schedule whenever the SHAPE matches; the async rename lane polls the cache + # (bounded wait) and no-ops on a true miss. + relay_info = self._relay_auto_thread_info(source) + if relay_info is None and not self._is_relay_discord_channel_lane( + source + ): + return + self._schedule_rename_from_title_thread( + source, + lambda copied: self._rename_discord_auto_thread_for_session_title( + copied, session_id, title, relay_info=relay_info + ), + "Discord semantic thread rename", + ) + + async def _rename_telegram_topic_for_session_title( + self, + source: SessionSource, + session_id: str, + title: str, + ) -> None: + """Best-effort rename of a Telegram DM topic when Hermes auto-titles a session.""" + if not await asyncio.to_thread(self._is_telegram_topic_lane, source) or not source.chat_id or not source.thread_id: + return + + # extra.disable_topic_auto_rename lets the operator disable per-topic auto-rename entirely, + # e.g. user-managed topics (ad-hoc Threaded Mode) that auto-rename would keep overwriting. + if self._telegram_topic_auto_rename_disabled(source): + return + + # Skip rename when the topic is operator-declared via extra.dm_topics. Those topics have + # fixed names chosen by the operator (plus optional skill binding); auto-renaming would + # silently mutate operator config. Check the class, not the instance — getattr() on a + # MagicMock auto-creates attributes, so an instance hasattr() is True for every test double. + adapter = self._adapter_for_source(source) + if adapter is not None: + get_info = getattr(type(adapter), "_get_dm_topic_info", None) + if callable(get_info): + try: + operator_topic = get_info(adapter, str(source.chat_id), str(source.thread_id)) + except Exception: + operator_topic = None + # Only treat dict-shaped returns as operator-declared; a + # bare MagicMock or other sentinel shouldn't count. + if isinstance(operator_topic, dict): + return + + session_db = getattr(self, "_session_db", None) + if session_db is not None: + try: + binding = await session_db.get_telegram_topic_binding( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + profile_name=self._telegram_topic_profile_name(source), + ) + if binding and str(binding.get("session_id") or "") != str(session_id): + return + except Exception: + logger.debug("Failed to verify Telegram topic binding before rename", exc_info=True) + return + + if adapter is None: + return + topic_name = self._sanitize_telegram_topic_title(title) + try: + rename_topic = getattr(adapter, "rename_dm_topic", None) + if rename_topic is not None: + await rename_topic( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + name=topic_name, + ) + return + + bot = getattr(adapter, "_bot", None) + edit_forum_topic = getattr(bot, "edit_forum_topic", None) if bot is not None else None + if edit_forum_topic is None: + edit_forum_topic = getattr(bot, "editForumTopic", None) if bot is not None else None + if edit_forum_topic is None: + return + try: + await edit_forum_topic( + chat_id=int(source.chat_id), + message_thread_id=int(source.thread_id), + name=topic_name, + ) + except (TypeError, ValueError): + await edit_forum_topic( + chat_id=source.chat_id, + message_thread_id=source.thread_id, + name=topic_name, + ) + except Exception: + logger.debug("Failed to rename Telegram topic for auto-generated title", exc_info=True) + + def _telegram_topic_auto_rename_disabled(self, source: SessionSource) -> bool: + """Return True when operator disabled per-topic auto-rename for this Telegram chat. + + ``gateway.platforms.telegram.extra.disable_topic_auto_rename``; default False (auto-rename on). + """ + platform_cfg = ( + self.config.platforms.get(source.platform) + if getattr(self, "config", None) and getattr(self.config, "platforms", None) + else None + ) + if platform_cfg is None: + return False + extra = getattr(platform_cfg, "extra", None) or {} + value = extra.get("disable_topic_auto_rename") + if value is None: + return False + if isinstance(value, bool): + return value + if isinstance(value, str): + return value.strip().lower() in {"1", "true", "yes", "on"} + return bool(value) + + def _schedule_telegram_topic_title_rename( + self, + source: SessionSource, + session_id: str, + title: str, + ) -> None: + """Schedule a topic rename from the auto-title background thread.""" + if not title or not self._is_telegram_topic_lane(source): + return + if self._telegram_topic_auto_rename_disabled(source): + return + self._schedule_rename_from_title_thread( + source, + lambda copied: self._rename_telegram_topic_for_session_title(copied, session_id, title), + "Telegram topic title rename", + ) + + def _should_send_telegram_capability_hint(self, source: SessionSource) -> bool: + """Rate-limit the BotFather Threads Settings screenshot. + + Repeated /topic while Threads Settings are still off must not re-upload it every time. + """ + if not hasattr(self, "_telegram_capability_hint_ts"): + self._telegram_capability_hint_ts = {} + key = self._telegram_topic_cooldown_key(source) + if not key: + return True + import time as _time + now = _time.monotonic() + last = self._telegram_capability_hint_ts.get(key, 0.0) + if now - last < self._TELEGRAM_CAPABILITY_HINT_COOLDOWN_S: + return False + self._telegram_capability_hint_ts[key] = now + return True + + def _telegram_topic_help_text(self) -> str: + return ( + "/topic — enable multi-session DM mode (one bot, many parallel chats)\n" + "\n" + "Usage:\n" + " /topic Enable topic mode, or show status if already on\n" + " /topic help Show this message\n" + " /topic off Disable topic mode and clear topic bindings\n" + " /topic Inside a topic: restore a previous session by ID\n" + "\n" + "How it works:\n" + "1. Run /topic once in this DM — Hermes checks BotFather Threads\n" + " Settings are enabled and flips on multi-session mode.\n" + "2. Tap All Messages at the top of the bot and send any message.\n" + " Telegram creates a new topic for that message; each topic is\n" + " an independent Hermes session (fresh history, fresh context).\n" + "3. The root DM becomes a system lobby — send /topic, /status,\n" + " /help, /usage there. Normal prompts go in a topic.\n" + "4. /new inside a topic resets just that topic's session.\n" + "5. /topic inside a topic restores an old session into it." + ) + + async def _disable_telegram_topic_mode_for_chat(self, source: SessionSource) -> str: + """Cleanly disable topic mode for a chat via /topic off.""" + if not self._session_db: + from hermes_state import format_session_db_unavailable + return format_session_db_unavailable(prefix=t("gateway.shared.session_db_unavailable_prefix")) + chat_id = str(source.chat_id or "") + if not chat_id: + return "Could not determine chat ID." + # No-op if never enabled. + try: + currently_enabled = await self._session_db.is_telegram_topic_mode_enabled( + chat_id=chat_id, + user_id=str(source.user_id or ""), + profile_name=self._telegram_topic_profile_name(source), + ) + except Exception: + currently_enabled = False + if not currently_enabled: + return "Multi-session topic mode is not currently enabled for this chat." + try: + await self._session_db.disable_telegram_topic_mode( + chat_id=chat_id, + profile_name=self._telegram_topic_profile_name(source), + ) + except Exception as exc: + logger.exception("Failed to disable Telegram topic mode") + return f"Failed to disable topic mode: {exc}" + # Reset per-profile+chat debounce state so the user doesn't see a + # stale cooldown on the next activation (issue #76423). + cooldown_key = self._telegram_topic_cooldown_key(source) + if cooldown_key: + for attr in ("_telegram_lobby_reminder_ts", "_telegram_capability_hint_ts"): + store = getattr(self, attr, None) + if isinstance(store, dict): + store.pop(cooldown_key, None) + return ( + "Multi-session topic mode is now OFF for this chat.\n\n" + "Existing topics in Telegram aren't removed — they'll just stop " + "being gated as independent sessions. The root DM works as a " + "normal Hermes chat again. Run /topic to re-enable later." + ) + + async def _telegram_topic_root_status_message(self, source: SessionSource) -> str: + lines = [ + "Telegram multi-session topics are enabled.", + "", + "To create a new Hermes chat, open All Messages at the top of this " + "bot interface and send any message there. Telegram will create a " + "new topic for it.", + "", + ] + try: + sessions = await self._session_db.list_unlinked_telegram_sessions_for_user( + chat_id=str(source.chat_id), + user_id=str(source.user_id), + profile_name=self._telegram_topic_profile_name(source), + limit=10, + ) + except Exception: + logger.debug("Failed to list unlinked Telegram sessions", exc_info=True) + sessions = [] + + if sessions: + lines.append("Previous unlinked sessions:") + for session in sessions: + session_id = str(session.get("id") or "") + title = str(session.get("title") or "Untitled session") + preview = str(session.get("preview") or "").strip() + line = f"- {title} — `{session_id}`" + if preview: + line += f" — {preview}" + lines.append(line) + lines.extend([ + "", + "To restore one:", + "1. Create or open a topic. To create a new one, open All Messages and send any message there.", + "2. Send /topic inside that topic.", + f"Example: Send /topic {sessions[0].get('id')} inside a topic.", + ]) + else: + lines.extend([ + "No previous unlinked Telegram sessions found.", + "", + "To restore a previous session later:", + "1. Create or open a topic. To create a new one, open All Messages and send any message there.", + "2. Send /topic inside that topic.", + ]) + return "\n".join(lines) + + async def _restore_telegram_topic_session(self, event: MessageEvent, raw_session_id: str) -> str: + """Restore an existing Telegram-owned Hermes session into this topic.""" + source = event.source + session_id = await self._session_db.resolve_session_id(raw_session_id.strip()) + if not session_id: + return f"Session not found: {raw_session_id.strip()}" + + session = await self._session_db.get_session(session_id) + if not session: + return f"Session not found: {raw_session_id.strip()}" + if str(session.get("source") or "") != "telegram": + return "That session is not a Telegram session and cannot be restored into this topic." + if str(session.get("user_id") or "") != str(source.user_id): + return "That session does not belong to this Telegram user." + + linked = await self._session_db.is_telegram_session_linked_to_topic(session_id=session_id) + topic_profile = self._telegram_topic_profile_name(source) + current_binding = await self._session_db.get_telegram_topic_binding( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + profile_name=topic_profile, + ) + if linked: + if not current_binding or current_binding.get("session_id") != session_id: + return "That session is already linked to another Telegram topic." + + session_key = self._session_key_for_source(source) + try: + await self._session_db.bind_telegram_topic( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + user_id=str(source.user_id), + session_key=session_key, + session_id=session_id, + managed_mode="restored", + profile_name=topic_profile, + ) + except ValueError as exc: + if "already linked" in str(exc): + return "That session is already linked to another Telegram topic." + raise + + title = await self._session_db.get_session_title(session_id) or session_id + last_assistant = None + try: + for message in reversed(await self._session_db.get_messages(session_id)): + if message.get("role") != "assistant": + continue + projected = project_compaction_message_for_display(message) + if projected is not None and projected.get("content"): + last_assistant = str(projected.get("content")) + break + except Exception: + last_assistant = None + + response = f"Session restored: {title}" + if last_assistant: + response += f"\n\nLast Hermes message:\n{last_assistant}" + return response diff --git a/gateway/run_turn.py b/gateway/run_turn.py new file mode 100644 index 0000000000..abdc10fa12 --- /dev/null +++ b/gateway/run_turn.py @@ -0,0 +1,5937 @@ +"""Agent-turn execution (_handle_message_with_agent, _run_agent*, proxy path, background tasks, MCP reload) for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import dataclasses +import inspect +import json +import os +import queue +import threading +import time +from agent.i18n import t +from contextlib import suppress +from contextvars import copy_context +from gateway.config import Platform +from gateway.media_repair import repair_explicit_computer_use_media_paths +from gateway.platforms.base import BasePlatformAdapter, MessageEvent +from gateway.session import ( + SessionSource, + TranscriptReadError, + _session_key_namespace, + build_channel_continuity_note, + build_session_context, +) +from gateway.turn_context import TurnContext +from gateway.turn_lease import DEFAULT_LEASE_WAIT, TurnLeaseTimeoutError +from hermes_constants import get_hermes_home_override +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional, Tuple +from utils import base_url_hostname + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayTurnMixin: + """Agent-turn execution (_handle_message_with_agent, _run_agent*, proxy path, background tasks, MCP reload) for GatewayRunner.""" + + def _resolve_session_agent_runtime( + self, + *, + source: Optional[SessionSource] = None, + session_key: Optional[str] = None, + user_config: Optional[dict] = None, + ) -> tuple[str, dict]: + """Resolve model/runtime for a session. + + Priority (highest first): session ``/model`` → ``channel_overrides`` → global config/env + (``_resolve_gateway_model(user_config)`` and default provider resolution). + """ + from gateway.run import ( + _credential_pool_for_provider, + _get_channel_override, + _resolve_gateway_model, + _resolve_runtime_agent_kwargs, + _resolve_runtime_agent_kwargs_for_provider, + ) + resolved_session_key = self._resolve_session_key_or_none(source, session_key) + + model = _resolve_gateway_model(user_config) + if resolved_session_key: + self._rehydrate_session_model_override(resolved_session_key) + _override_state = ( + self._peek_session_state(resolved_session_key) + if resolved_session_key + else None + ) + override = ( + _override_state.conversation.model_override if _override_state else None + ) + if override: + override_model = override.get("model", model) + override_runtime = { + "provider": override.get("provider"), + "requested_provider": override.get("requested_provider"), + "api_key": override.get("api_key"), + "base_url": override.get("base_url"), + "api_mode": override.get("api_mode"), + "max_tokens": override.get("max_tokens"), + "credential_pool": override.get("credential_pool"), + "request_overrides": override.get("request_overrides"), + "capabilities": dict(override.get("capabilities") or {}), + } + if override_runtime.get("api_key"): + if override_runtime.get("credential_pool") is None: + override_runtime["credential_pool"] = _credential_pool_for_provider( + override.get("provider") + ) + logger.debug( + "Session model override (fast): session=%s config_model=%s -> override_model=%s provider=%s", + resolved_session_key or "", model, override_model, + override_runtime.get("provider"), + ) + return override_model, override_runtime + # Override exists but has no api_key — fall through to env-based + # resolution and apply model/provider from the override on top. + logger.debug( + "Session model override (no api_key, fallback): session=%s config_model=%s override_model=%s", + resolved_session_key or "", model, override_model, + ) + else: + logger.debug( + "No session model override: session=%s config_model=%s override_keys=%s", + resolved_session_key or "", model, + [ + _key + for _key, _st in list(self._sessions_map().items()) + if _st.conversation.model_override is not None + ][:5] or "[]", + ) + + runtime_kwargs = _resolve_runtime_agent_kwargs() + runtime_model = runtime_kwargs.pop("model", None) + if runtime_model: + logger.info( + "Runtime provider supplied explicit model override: %s -> %s", + model, + runtime_model, + ) + model = runtime_model + + cfg = getattr(self, "config", None) + if cfg and source is not None: + chat_id = str(source.chat_id) if source.chat_id else "" + thread_id = ( + str(source.thread_id) if getattr(source, "thread_id", None) else None + ) + parent_id = ( + str(source.parent_chat_id) + if getattr(source, "parent_chat_id", None) + else None + ) + ch = _get_channel_override( + cfg, + source.platform, + chat_id, + thread_id=thread_id, + parent_id=parent_id, + ) + if ch: + if ch.model: + model = ch.model + if ch.provider: + runtime_kwargs = _resolve_runtime_agent_kwargs_for_provider( + ch.provider + ) + ch_runtime_model = runtime_kwargs.pop("model", None) + # Only adopt the provider's bundled model when the override + # did not specify an explicit model. + if ch_runtime_model and not ch.model: + model = ch_runtime_model + + if override and resolved_session_key: + model, runtime_kwargs = self._apply_session_model_override( + resolved_session_key, model, runtime_kwargs + ) + + # No model.default but a provider resolved (e.g. `hermes auth add openai-codex` without + # `hermes model`): fall back to the provider's first catalog model so the API call has one. + if not model and runtime_kwargs.get("provider"): + try: + from hermes_cli.models import get_default_model_for_provider + model = get_default_model_for_provider(runtime_kwargs["provider"]) + if model: + logger.info( + "No model configured — defaulting to %s for provider %s", + model, runtime_kwargs["provider"], + ) + except Exception: + pass + + # Final safety net: if resolution still produced an empty model (e.g. a transient config-cache + # miss on a post-interrupt recovery turn), reuse the last model resolved for this session, + # else the most recent process-wide — model="" makes every API call fail HTTP 400 and the + # session goes silent. ``getattr`` guards bare test runners built via ``object.__new__``. + if not model: + _lr_state = ( + self._peek_session_state(resolved_session_key) + if resolved_session_key + else None + ) + _lr_star = self._peek_session_state("*") + _recovered = ( + (_lr_state.conversation.last_resolved_model if _lr_state else "") + or (_lr_star.conversation.last_resolved_model if _lr_star else "") + ) + if _recovered: + logger.warning( + "Empty model resolved for session=%s — recovering " + "last-known-good model %s (config read likely returned " + "empty; see #35314)", + resolved_session_key or "", _recovered, + ) + model = _recovered + elif model: + # Cache the good resolution for future recovery turns. + if resolved_session_key: + self._session_state( + resolved_session_key + ).conversation.last_resolved_model = model + self._session_state("*").conversation.last_resolved_model = model + + return model, runtime_kwargs + + def _resolve_turn_agent_config(self, user_message: str, model: str, runtime_kwargs: dict) -> dict: + """Build the effective model/runtime config for a single turn. + + Always uses the session's primary model/provider. If `/fast` is enabled and the model + supports it, attach `request_overrides` for priority processing. Per-provider + ``request_overrides`` from ``resolve_runtime_provider`` (e.g. ``custom_providers`` + ``extra_body``) are merged *under* the fast-mode overrides so they still reach the model. + """ + from gateway.run import _deep_merge_request_overrides + from hermes_cli.models import resolve_fast_mode_overrides + + runtime = { + "api_key": runtime_kwargs.get("api_key"), + "base_url": runtime_kwargs.get("base_url"), + "provider": runtime_kwargs.get("provider"), + "requested_provider": runtime_kwargs.get("requested_provider"), + "api_mode": runtime_kwargs.get("api_mode"), + "command": runtime_kwargs.get("command"), + "args": list(runtime_kwargs.get("args") or []), + "credential_pool": runtime_kwargs.get("credential_pool"), + "max_tokens": runtime_kwargs.get("max_tokens"), + "capabilities": dict(runtime_kwargs.get("capabilities") or {}), + } + base_request_overrides = dict(runtime_kwargs.get("request_overrides") or {}) + route = { + "model": model, + "runtime": runtime, + "signature": ( + model, + runtime["provider"], + runtime["requested_provider"], + runtime["base_url"], + runtime["api_mode"], + runtime["command"], + tuple(runtime["args"]), + ), + } + + # Provider-level request_overrides (e.g. a custom_providers extra_body) resolved upstream by + # resolve_runtime_provider(). + service_tier = getattr(self, "_service_tier", None) + if service_tier != "priority": + # None (normal) or auto/cold — the bounded window is applied per + # request by agent.fast_mode, not pinned into request_overrides. + route["request_overrides"] = base_request_overrides + return route + + try: + overrides = resolve_fast_mode_overrides( + route["model"], + provider=runtime["provider"], + base_url=runtime["base_url"], + ) + except Exception: + overrides = None + # Fast-mode overrides (service_tier / speed) are top-level keys and do + # not collide with extra_body; deep-merge them over the provider overrides. + route["request_overrides"] = _deep_merge_request_overrides( + base_request_overrides, + overrides or {}, + ) + return route + + def _sync_session_model_from_agent(self, session_id: str, agent: Any) -> None: + """Persist the runtime model/provider actually used by a gateway turn. + + Provider fallback can switch ``agent.model``/``agent.provider`` after the session row was + created; keep the DB metadata in sync so session lists and tooling report the backend that + actually answered. Runs in the ``run_sync`` closure (executor thread), so it uses the sync + ``SessionDB`` (``_db``) directly rather than the AsyncSessionDB forwarder. + """ + if not session_id or agent is None or self._session_db is None: + return + model = getattr(agent, "model", None) + if not model: + return + runtime = { + "provider": getattr(agent, "provider", None), + "base_url": getattr(agent, "base_url", None), + "api_mode": getattr(agent, "api_mode", None), + "fallback_active": bool(getattr(agent, "_fallback_activated", False)), + } + runtime = {k: v for k, v in runtime.items() if v not in (None, "")} + + try: + db = self._session_db._db + row = db.get_session(session_id) + if not row: + return + current_model = row.get("model") + raw_config = row.get("model_config") + try: + config = json.loads(raw_config) if raw_config else {} + except Exception: + config = {} + if not isinstance(config, dict): + config = {} + gateway_runtime = dict(config.get("gateway_runtime") or {}) + if current_model == model and all( + gateway_runtime.get(k) == v for k, v in runtime.items() + ): + return + config["gateway_runtime"] = runtime + db.update_session_meta(session_id, json.dumps(config), model=model) + except Exception: + logger.debug("Failed to sync gateway session model metadata", exc_info=True) + + async def _hmwa_resolve_session(self, event, source): + """Resolve ``source`` to its session entry (topic recovery, internal-route guards, Telegram + topic-binding heal). Returns ``(source, session_entry, session_key)`` or ``None`` to drop + the event.""" + # Get or create session Topic-mode DMs: rewrite a stale/foreign thread_id to the user's + # last-active topic so a cross-topic Reply or stripped plain reply doesn't fragment the + # conversation across sessions. + recovered = await asyncio.to_thread(self._recover_telegram_topic_thread_id, source) + if recovered is not None: + logger.info( + "telegram topic recovery: chat=%s user=%s %r -> %s", + source.chat_id, source.user_id, source.thread_id, recovered, + ) + source = dataclasses.replace(source, thread_id=recovered) + with suppress(Exception): + event.source = source + + event_metadata = getattr(event, "metadata", None) or {} + expected_session_key = str( + event_metadata.get("gateway_session_key") or "" + ).strip() + if expected_session_key: + derived_session_key = self._session_key_for_source(source) + if derived_session_key != expected_session_key: + logger.warning( + "Dropping internally routed event after route recovery: " + "expected session=%s derived=%s", + expected_session_key, + derived_session_key, + ) + return + + strict_session = bool(event_metadata.get("gateway_session_strict")) + pinned_session_id = str(event_metadata.get("gateway_session_id") or "").strip() + if strict_session: + session_entry = await self.async_session_store.lookup_by_session_key( + expected_session_key + ) + if ( + session_entry is None + or not pinned_session_id + or session_entry.session_id != pinned_session_id + ): + logger.warning( + "Dropping internally routed event: expected session id=%s is no " + "longer current for key=%s", + pinned_session_id or "missing", + expected_session_key or "missing", + ) + return + else: + # Internal wakes must observe reset policy without becoming user activity themselves. + # Otherwise periodic Kanban/process notifications keep the stable routing key alive + # across every daily/idle boundary. + session_entry = await self.async_session_store.get_or_create_session( + source, + touch_activity=not bool(getattr(event, "internal", False)), + ) + session_key = session_entry.session_key + if not strict_session and pinned_session_id: + resolved_entry = await self._resolve_async_delegation_session( + session_entry, + pinned_session_id, + ) + if resolved_entry is None: + return + session_entry = resolved_entry + self._cache_session_source(session_key, source) + if await asyncio.to_thread(self._is_telegram_topic_lane, source): + try: + binding = (await self._session_db.get_telegram_topic_binding( + chat_id=str(source.chat_id), + thread_id=str(source.thread_id), + profile_name=self._telegram_topic_profile_name(source), + )) if self._session_db else None + except Exception: + logger.debug("Failed to read Telegram topic binding", exc_info=True) + binding = None + if binding: + bound_session_id = str(binding.get("session_id") or "") + # Heal bindings that point at a pre-compression parent: walk the compression- + # continuation chain forward to its tip so the next message resumes the compressed + # child instead of reloading the oversized parent transcript. + if bound_session_id and self._session_db is not None: + try: + canonical_session_id = await self._session_db.get_compression_tip( + bound_session_id, + ) + except Exception: + logger.debug( + "compression-tip lookup failed for %s", + bound_session_id, exc_info=True, + ) + canonical_session_id = bound_session_id + if ( + canonical_session_id + and canonical_session_id != bound_session_id + ): + bound_session_id = canonical_session_id + if bound_session_id and bound_session_id != session_entry.session_id: + # Route the override through SessionStore so the session_key → session_id + # mapping is persisted to disk and the previous lane session is ended cleanly. + # Mutating session_entry in place created a split-brain: the JSON index pointed + # at one id while downstream code used another. + switched = await self.async_session_store.switch_session(session_key, bound_session_id) + if switched is not None: + session_entry = switched + # If the stored binding pointed at a parent, rewrite it to the + # canonical descendant now that we've followed the chain. + if ( + bound_session_id + and bound_session_id != str(binding.get("session_id") or "") + ): + await asyncio.to_thread( + self._sync_telegram_topic_binding, + source, session_entry, reason="compression-tip-walk", + ) + else: + try: + await asyncio.to_thread(self._record_telegram_topic_binding, source, session_entry) + except Exception: + logger.debug("Failed to record Telegram topic binding", exc_info=True) + return source, session_entry, session_key + + async def _hmwa_open_session(self, session_entry, session_key, source): + """Consume auto-reset / fresh-reset flags and emit ``session:start`` for new sessions. + Returns ``(_was_auto_reset, _is_new_session)``.""" + # Capture and consume was_auto_reset immediately so it cannot re-fire on later messages and + # wipe model/reasoning overrides set between turns. + _was_auto_reset = getattr(session_entry, "was_auto_reset", False) + if _was_auto_reset: + # Auto-reset is a full conversation boundary: one funnel call clears every conversation- + # scoped per-session dict so the fresh session inherits no model/reasoning overrides, no + # queued "/model switched" note and no stale resolved-model cache. + self._clear_conversation_scope(session_key, reason="auto_reset") + # Evict the cached agent: the cache is keyed on the stable session_key, so an auto-reset + # would otherwise reuse the old agent and leak context_compressor._previous_summary + # (prior history) into new compaction summaries. + self._evict_cached_agent(session_key) + session_entry.was_auto_reset = False + + # Emit session:start for new or auto-reset sessions + _is_new_session = ( + session_entry.created_at == session_entry.updated_at + or _was_auto_reset + or getattr(session_entry, "is_fresh_reset", False) + ) + # Consume the is_fresh_reset flag immediately so it doesn't leak + # onto subsequent messages in the same session (issue #6508). + if getattr(session_entry, "is_fresh_reset", False): + session_entry.is_fresh_reset = False + if _is_new_session: + await self.hooks.emit("session:start", { + "platform": source.platform.value if source.platform else "", + "user_id": source.user_id, + "session_id": session_entry.session_id, + "session_key": session_key, + }) + return _was_auto_reset, _is_new_session + + async def _hmwa_deliver_auto_reset_notice(self, session_entry, source, turn_sidecar_notes): + """Stage the auto-reset sidecar note for the agent and notify the user (policy-gated).""" + from gateway.run import _AUTO_RESET_CONTEXT_NOTES, _auto_reset_reason_text + reset_reason = getattr(session_entry, 'auto_reset_reason', None) or 'idle' + context_note = _AUTO_RESET_CONTEXT_NOTES.get(reset_reason, _AUTO_RESET_CONTEXT_NOTES["idle"]) + # Slack/Discord channels/threads are long-lived: point the agent at the specific prior + # same-channel session so it recalls that context via session_search instead of an + # unrelated recent session. Deterministic — no extra API/DB calls. + try: + continuity_note = build_channel_continuity_note(session_entry, source) + except Exception: + continuity_note = None + if continuity_note: + context_note = context_note + "\n\n" + continuity_note + turn_sidecar_notes.append(context_note) + + # Notify the user about the reset unless notifications are disabled in config, the + # platform is excluded (e.g. api_server, webhook), or the expired session had no + # activity. + try: + policy = self.session_store.config.get_reset_policy( + platform=source.platform, + session_type=getattr(source, 'chat_type', 'dm'), + ) + platform_name = source.platform.value if source.platform else "" + had_activity = getattr(session_entry, 'reset_had_activity', False) + # Suspended and restart-recovery-expired sessions always notify regardless of + # policy.notify — the user had an active session that was silently replaced, so they + # need to know they can /resume it. Idle/daily resets respect the policy flag. + should_notify = reset_reason in {"suspended", "resume_pending_expired"} or ( + policy.notify + and had_activity + and platform_name not in policy.notify_exclude_platforms + ) + if should_notify: + adapter = self._adapter_for_source(source) + if adapter: + reason_text = _auto_reset_reason_text(reset_reason, policy) + notice = ( + f"◐ Session automatically reset ({reason_text}). " + f"Conversation history cleared.\n" + f"Use /resume to browse and restore a previous session.\n" + f"Adjust reset timing in config.yaml under session_reset." + ) + try: + session_info = await asyncio.to_thread( + self._reset_notice_session_info, source + ) + if session_info: + notice = f"{notice}\n\n{session_info}" + except Exception: + pass + await adapter.send( + source.chat_id, notice, + metadata=self._thread_metadata_for_source(source), + ) + except Exception as e: + logger.debug("Auto-reset notification failed (non-fatal): %s", e) + + # was_auto_reset is already consumed in the cleanup block above + # (single source of truth); only the reset reason needs clearing here. + session_entry.auto_reset_reason = None + + def _hmwa_auto_load_skills(self, event, _auto, _quick_key, session_key): + """Prepend topic/channel-bound skill payload(s) to ``event.text`` on a new session.""" + _skill_names = [_auto] if isinstance(_auto, str) else list(_auto) + try: + from agent.skill_commands import _load_skill_payload, _build_skill_message + _combined_parts: list[str] = [] + _loaded_names: list[str] = [] + for _sname in _skill_names: + _loaded = _load_skill_payload(_sname, task_id=_quick_key) + if _loaded: + _loaded_skill, _skill_dir, _display_name = _loaded + _note = ( + f'[IMPORTANT: The "{_display_name}" skill is auto-loaded. ' + f"Follow its instructions for this session.]" + ) + _part = _build_skill_message(_loaded_skill, _skill_dir, _note) + if _part: + _combined_parts.append(_part) + _loaded_names.append(_sname) + else: + logger.warning("[Gateway] Auto-skill '%s' not found", _sname) + if _combined_parts: + # Append the user's original text after all skill payloads + _combined_parts.append(event.text) + event.text = "\n\n".join(_combined_parts) + logger.info( + "[Gateway] Auto-loaded skill(s) %s for session %s", + _loaded_names, session_key, + ) + except Exception as e: + logger.warning("[Gateway] Failed to auto-load skill(s) %s: %s", _skill_names, e) + + async def _hmwa_acquire_turn_lease(self, _quick_key, run_generation, session_entry, _session_env_tokens): + # ── Turn lease: session resolution is FINAL here. Serialize [load history → run → flush] + # per resolved SESSION_ID: another routing key mapped to the same session_id waits for the + # prior flush instead of loading a stale base and interleaving writes (same-key messages + # never reach here mid-turn thanks to adapter + runner guards, so the lock is otherwise + # uncontended). Fail-closed on timeout: never enter the transcript region without a lease; + # outer dispatch returns a bounded resend notice. Released in _handle_message's finally + # (_release_turn_lease), granted per (routing key, run generation) so a stale unwind can't + # release a newer turn's. + from gateway.run import _float_env + _lease_registry = getattr(self, "_turn_leases", None) + if _lease_registry is not None: + try: + _lease_token = await _lease_registry.acquire( + session_entry.session_id, + owner_key=_quick_key, + generation=run_generation, + timeout=_float_env( + "HERMES_TURN_LEASE_TIMEOUT", DEFAULT_LEASE_WAIT + ), + ) + except TurnLeaseTimeoutError: + # The broad session-context cleanup finally starts later in this + # method. Restore the tokens here before propagating the rejection + # to outer dispatch, or this early exit leaks task-local identity. + self._clear_session_env(_session_env_tokens) + raise + if _lease_token is not None: + _lease_state = self._session_state(_quick_key).turn + _lease_state.lease_token = _lease_token + _lease_state.lease_generation = run_generation + + async def _hmwa_hygiene_settings(self, source, session_key): + """Resolve model/provider/context-length + hygiene knobs for the pre-agent compression + safety net (fail-soft: any config/runtime error keeps the defaults).""" + from gateway.run import _load_gateway_config + # Read model + compression config. Hygiene threshold is intentionally HIGHER than the + # agent's own compressor (0.85 vs 0.50): it is a pre-agent safety net for sessions that + # grew between turns; at 0.50 it compressed prematurely on every turn in long sessions. + _hyg_model = "anthropic/claude-sonnet-4.6" + _hyg_threshold_pct = 0.85 + _hyg_compression_enabled = True + _hyg_hard_msg_limit = 5000 + _hyg_timeout_seconds = 30.0 + _hyg_total_ceiling_seconds = 600.0 + # Max wall-clock the user's TURN is held waiting on hygiene compression before the + # gateway stops waiting and proceeds on the uncompressed transcript. The compressor keeps + # running detached; its commit is fenced (revoke_commit_admission) so a stale result can + # never clobber later turns. Kept well below transport idle-timeouts (Telegram ~30s). + _hyg_max_turn_hold_seconds = 10.0 + _hyg_failure_cooldown_seconds = 300.0 + _hyg_config_context_length = None + _hyg_provider = None + _hyg_base_url = None + _hyg_api_key = None + _hyg_configured_model = None + _hyg_configured_provider = None + _hyg_configured_base_url = None + _hyg_data = {} + try: + _hyg_data = _load_gateway_config() + if _hyg_data: + # Resolve model name (same logic as run_sync) + _model_cfg = _hyg_data.get("model", {}) + if isinstance(_model_cfg, str): + _hyg_model = _model_cfg + elif isinstance(_model_cfg, dict): + _hyg_model = _model_cfg.get("default") or _model_cfg.get("model") or _hyg_model + # Read explicit context_length override from model config + # (same as run_agent.py lines 995-1005) + _raw_ctx = _model_cfg.get("context_length") + if _raw_ctx is not None: + with suppress(TypeError, ValueError): + _hyg_config_context_length = int(_raw_ctx) + # Read provider for accurate context detection + _hyg_provider = _model_cfg.get("provider") or None + _hyg_base_url = _model_cfg.get("base_url") or None + + # Only the enabled flag is read; hygiene's threshold is deliberately separate + # from the agent's compression.threshold (hygiene runs higher). + _comp_cfg = _hyg_data.get("compression", {}) + if isinstance(_comp_cfg, dict): + _hyg_compression_enabled = str( + _comp_cfg.get("enabled", True) + ).lower() in {"true", "1", "yes"} + _raw_hard_limit = _comp_cfg.get("hygiene_hard_message_limit") + if _raw_hard_limit is not None: + try: + _parsed = int(_raw_hard_limit) + if _parsed > 0: + _hyg_hard_msg_limit = _parsed + except (TypeError, ValueError): + pass + _raw_timeout = _comp_cfg.get("hygiene_timeout_seconds") + if _raw_timeout is not None: + try: + _parsed = float(_raw_timeout) + if _parsed > 0: + _hyg_timeout_seconds = _parsed + except (TypeError, ValueError): + pass + _raw_ceiling = _comp_cfg.get("hygiene_total_ceiling_seconds") + if _raw_ceiling is not None: + try: + _parsed = float(_raw_ceiling) + if _parsed > 0: + _hyg_total_ceiling_seconds = _parsed + except (TypeError, ValueError): + pass + # The ceiling can never be tighter than one idle + # window, or the extension loop would be dead code. + _hyg_total_ceiling_seconds = max( + _hyg_total_ceiling_seconds, _hyg_timeout_seconds, + ) + _raw_turn_hold = _comp_cfg.get("hygiene_max_turn_hold_seconds") + if _raw_turn_hold is not None: + try: + _parsed = float(_raw_turn_hold) + if _parsed > 0: + _hyg_max_turn_hold_seconds = _parsed + except (TypeError, ValueError): + pass + _raw_cooldown = _comp_cfg.get("hygiene_failure_cooldown_seconds") + if _raw_cooldown is not None: + try: + _parsed = float(_raw_cooldown) + if _parsed >= 0: + _hyg_failure_cooldown_seconds = _parsed + except (TypeError, ValueError): + pass + + _hyg_configured_model = _hyg_model + _hyg_configured_provider = _hyg_provider + _hyg_configured_base_url = _hyg_base_url + + try: + _hyg_model, _hyg_runtime = self._resolve_session_agent_runtime( + source=source, + session_key=session_key, + user_config=_hyg_data if isinstance(_hyg_data, dict) else None, + ) + _hyg_provider = _hyg_runtime.get("provider") or _hyg_provider + _hyg_base_url = _hyg_runtime.get("base_url") or _hyg_base_url + _hyg_api_key = _hyg_runtime.get("api_key") or _hyg_api_key + except Exception: + pass + + if _hyg_config_context_length is not None: + try: + from hermes_cli.route_identity import should_clear_context_pin_async + + if await should_clear_context_pin_async( + _hyg_configured_model, + _hyg_model, + _hyg_configured_base_url, + _hyg_base_url, + _hyg_configured_provider, + _hyg_provider, + ): + _hyg_config_context_length = None + except Exception: + _hyg_config_context_length = None + + # custom_providers per-model context_length fallback (as in run_agent.py); must run + # after runtime resolution so _hyg_base_url is set. + if _hyg_config_context_length is None and _hyg_base_url: + try: + try: + from hermes_cli.config import ( + get_compatible_custom_providers as _gw_gcp, + get_custom_provider_context_length as _gw_gccl, + ) + _hyg_custom_providers = _gw_gcp(_hyg_data) + except Exception: + _hyg_custom_providers = _hyg_data.get("custom_providers") + if not isinstance(_hyg_custom_providers, list): + _hyg_custom_providers = [] + _hyg_custom_ctx = _gw_gccl( + model=_hyg_model, + base_url=_hyg_base_url, + custom_providers=_hyg_custom_providers, + ) + if _hyg_custom_ctx: + _hyg_config_context_length = int(_hyg_custom_ctx) + except (TypeError, ValueError): + pass + except Exception: + pass + return self._HygieneSettings( + model=_hyg_model, + threshold_pct=_hyg_threshold_pct, + compression_enabled=_hyg_compression_enabled, + hard_msg_limit=_hyg_hard_msg_limit, + timeout_seconds=_hyg_timeout_seconds, + total_ceiling_seconds=_hyg_total_ceiling_seconds, + max_turn_hold_seconds=_hyg_max_turn_hold_seconds, + failure_cooldown_seconds=_hyg_failure_cooldown_seconds, + config_context_length=_hyg_config_context_length, + provider=_hyg_provider, + base_url=_hyg_base_url, + api_key=_hyg_api_key, + data=_hyg_data, + ) + + async def _hmwa_hygiene_plan(self, hs, history, session_entry, session_key): + """Decide whether hygiene compression fires this turn (token/message thresholds, DB-backed + failure cooldown, in-flight compression). Returns + ``(_needs_compress, _approx_tokens, _msg_count, _warn_token_threshold)``.""" + from agent.model_metadata import ( + estimate_messages_tokens_rough, + get_model_context_length_async, + ) + + _hyg_model = hs.model + _hyg_threshold_pct = hs.threshold_pct + _hyg_hard_msg_limit = hs.hard_msg_limit + _hyg_config_context_length = hs.config_context_length + _hyg_provider = hs.provider + _hyg_base_url = hs.base_url + _hyg_api_key = hs.api_key + + _hyg_context_length = await get_model_context_length_async( + _hyg_model, + base_url=_hyg_base_url or "", + api_key=_hyg_api_key or "", + config_context_length=_hyg_config_context_length, + provider=_hyg_provider or "", + ) + _compress_token_threshold = int( + _hyg_context_length * _hyg_threshold_pct + ) + _warn_token_threshold = int(_hyg_context_length * 0.95) + + _msg_count = len(history) + + # Prefer actual API-reported tokens from the last turn + # (stored in session entry) over the rough char-based estimate. + _stored_tokens = session_entry.last_prompt_tokens + if _stored_tokens > 0: + _approx_tokens = _stored_tokens + _token_source = "actual" + else: + _approx_tokens = estimate_messages_tokens_rough(history) + _token_source = "estimated" + # Rough estimates run 30-50% high on code/JSON-heavy sessions, which only makes + # hygiene fire early (safe). Do NOT compensate with a threshold multiplier: 85% + # * 1.4 = 119% of context kept hygiene from ever firing for ~200K models. + + # Hard safety valve: force compression at an extreme message count regardless of token + # estimates, breaking the spiral where API disconnects prevent token data → no + # compression → more disconnects. Default 5000 sits clear of legitimate 1M+ context + # sessions (those compress on tokens). Config: compression.hygiene_hard_message_limit. + _HARD_MSG_LIMIT = _hyg_hard_msg_limit + _needs_compress = ( + _approx_tokens >= _compress_token_threshold + or _msg_count >= _HARD_MSG_LIMIT + ) + + if _needs_compress: + # Use the persistent DB-backed cooldown (same as the in-conversation compression + # path in context_compressor.py) so the cooldown survives gateway restarts. The + # in-memory dict reset on every restart, re-triggering the same failing + # compression and wedging session storage. + _session_db = getattr(self, "_session_db", None) + if _session_db is not None: + _session_db = getattr(_session_db, "_db", _session_db) + _getter = getattr(_session_db, "get_compression_failure_cooldown", None) + if _getter is not None: + try: + _cooldown_state = _getter(session_entry.session_id) + except Exception: + _cooldown_state = None + if _cooldown_state and _cooldown_state.get("remaining_seconds", 0) > 0: + logger.info( + "Session hygiene: skipping compression for %s; " + "previous failure cooldown active for %.1fs", + session_entry.session_id, + _cooldown_state["remaining_seconds"], + ) + _needs_compress = False + + if _needs_compress and await self._session_has_compression_in_flight( + session_key + ): + # A prior hygiene/agent compression still holds the durable lock (typically a + # shielded worker left behind by /stop or /restart). Starting another attempt + # would wait up to the 600s ceiling behind a commit the fence will refuse, while + # inbound messages demote to queue. + logger.info( + "Session hygiene: skipping compression for %s; " + "another compression is already in flight", + session_entry.session_id, + ) + _needs_compress = False + + if _needs_compress: + logger.info( + "Session hygiene: %s messages, ~%s tokens (%s) — auto-compressing " + "(threshold: %s%% of %s = %s tokens)", + _msg_count, f"{_approx_tokens:,}", _token_source, + int(_hyg_threshold_pct * 100), + f"{_hyg_context_length:,}", + f"{_compress_token_threshold:,}", + ) + return _needs_compress, _approx_tokens, _msg_count, _warn_token_threshold + + async def _hmwa_hygiene_wait_for_summary(self, attempt, hs, session_entry): + """Progress-aware inline wait for the detached hygiene compressor. Returns the compressed + transcript; raises ``HygieneTurnHoldExceeded`` (turn-hold budget) or + ``asyncio.TimeoutError`` (idle/ceiling/fence cancel) for the caller's handlers.""" + from gateway.run import HygieneTurnHoldExceeded, hygiene_wait_should_extend + _hyg_commit_fence = attempt.commit_fence + _hyg_future = attempt.future + _hyg_wait_started = attempt.wait_started + _hyg_timeout_seconds = hs.timeout_seconds + _hyg_total_ceiling_seconds = hs.total_ceiling_seconds + _hyg_max_turn_hold_seconds = hs.max_turn_hold_seconds + + # Progress-aware wait: the timeout is an INACTIVITY budget — + # the worker ticks the fence per streamed token, so a slow but + # still-generating model extends the deadline. A hard ceiling + # bounds the total so a trickle stream can't hold the turn. + while True: + if _hyg_commit_fence.is_cancelled: + raise asyncio.TimeoutError + # Charge the idle budget from the LAST PROGRESS event, + # not from the start of this wait slice — otherwise + # silence can approach 2x the configured timeout. + _hyg_waited = ( + time.monotonic() - _hyg_wait_started + ) + _slice = min( + max( + _hyg_timeout_seconds + - _hyg_commit_fence.seconds_since_progress(), + 0.005, + ), + max( + _hyg_total_ceiling_seconds + - _hyg_waited, + 0.005, + ), + ) + # Bounded turn-hold: cap this slice at the remaining + # turn-hold budget so it is re-checked against + # _hyg_max_turn_hold_seconds at least that often — + # otherwise a continuously-streaming worker keeps the + # slice large and holds the turn until the ceiling. + _turn_hold_remaining = ( + _hyg_max_turn_hold_seconds + - (time.monotonic() - _hyg_wait_started) + ) + if _turn_hold_remaining <= 0: + # Budget exhausted: force an immediate timeout so + # the abandonment path below runs. + _slice = 0.005 + else: + _slice = min( + _slice, + max(_turn_hold_remaining, 0.005), + ) + # Re-check the fence on a short poll so a + # /stop or /restart cancel is not stuck + # behind a full idle window (#96953). + _idle_left = max( + _hyg_timeout_seconds + - _hyg_commit_fence.seconds_since_progress(), + 0.005, + ) + _slice = min(_slice, 0.25) + try: + _compressed, _ = await asyncio.wait_for( + asyncio.shield(_hyg_future), + timeout=_slice, + ) + break + except asyncio.TimeoutError: + if _hyg_commit_fence.is_cancelled: + raise + _hyg_waited = time.monotonic() - _hyg_wait_started + _idle = _hyg_commit_fence.seconds_since_progress() + # Bounded turn-hold: never hold the user's TURN past + # _hyg_max_turn_hold_seconds even if the summary + # model is still streaming; fall through to the + # timeout path, which revokes commit admission and + # proceeds on the uncompressed transcript, so the + # wire never trips a transport idle-timeout. + if ( + _hyg_waited + >= _hyg_max_turn_hold_seconds + ): + logger.info( + "Session hygiene compression for " + "session %s exceeded the turn-hold " + "budget (%.1fs >= %.1fs) — " + "abandoning inline wait, proceeding " + "without compression this turn", + session_entry.session_id, + _hyg_waited, + _hyg_max_turn_hold_seconds, + ) + raise HygieneTurnHoldExceeded( + f"turn-hold budget {_hyg_max_turn_hold_seconds:.1f}s " + f"elapsed after {_hyg_waited:.1f}s" + ) + if hygiene_wait_should_extend( + idle=_idle, + timeout=_hyg_timeout_seconds, + waited=_hyg_waited, + ceiling=_hyg_total_ceiling_seconds, + fence_cancelled=_hyg_commit_fence.is_cancelled, + ): + if _slice >= _idle_left - 1e-9: + logger.info( + "Session hygiene compression for " + "session %s still streaming after " + "%.0fs (last progress %.1fs ago) — " + "extending wait (ceiling %.0fs)", + session_entry.session_id, + _hyg_waited, _idle, + _hyg_total_ceiling_seconds, + ) + continue + raise + return _compressed + + async def _hmwa_hygiene_on_turn_hold(self, attempt, hs, session_entry, session_key, source): + """``except HygieneTurnHoldExceeded`` body: keep or cancel the worker's commit admission, + notify the user, and re-raise; returns the compressed transcript only when the worker + was already committing.""" + from gateway.run import ( + _HYGIENE_TURNHOLD_RETRY_SECONDS, + _record_hygiene_cooldown, + _reset_hygiene_failure_streak, + _stamp_hygiene_compression_provenance, + ) + _hyg_agent = attempt.agent + _hyg_meta = attempt.meta + _hyg_commit_fence = attempt.commit_fence + _hyg_future = attempt.future + _hyg_wait_started = attempt.wait_started + _hyg_max_turn_hold_seconds = hs.max_turn_hold_seconds + + # Turn-hold expiry is an availability boundary, not a + # failure: the compressor is healthy and still streaming; we + # just can't hold the turn longer. Share the safe mechanics + # (fence, release, defer, proceed uncompressed) with + # distinct provenance / user message and NO failure-cooldown + # increment. Decouple the TURN from the COMPRESSION: when + # the worker's commit is watermark-fenced (rows appended + # after compression start, this turn included, survive its + # late commit as cloned concurrent tail) the attempt KEEPS + # commit admission — the turn proceeds uncompressed NOW and + # the summary is adopted at the worker's own fenced commit + # (archive_and_compact / rotation publish); always + # cancelling burned every attempt for thinking summary + # models whose reasoning prefix alone exceeds the hold. If + # NOT watermark-fenced (no session_db, capture failed, + # legacy lock API) a late commit could clobber newer turns, + # so cancel. + _hyg_keep_admission = bool( + getattr( + _hyg_commit_fence, + "commit_watermark_fenced", + False, + ) + ) and not _hyg_commit_fence.is_cancelled + if _hyg_keep_admission: + self._defer_agent_cleanup_until_future_done( + _hyg_future, + _hyg_agent, + context="session hygiene turn-hold", + ) + attempt.cleanup_deferred = True + # NO retry-after here: the attempt is still running + # toward a real commit, and the flat 60s retry-after + # would also block the agent-side preflight compressor + # (same-session cooldown). Re-attempt spacing comes from + # the durable compression lock instead: the next turn's + # hygiene pre-check skips while this worker's lease is + # held (_session_has_compression_in_flight). The + # done-callback below records the flat retry-after ONLY + # if the worker ends without committing anything. + _hyg_deferred_sid = session_entry.session_id + _hyg_deferred_key = session_key + _hyg_deferred_agent = _hyg_agent + + def _hyg_adopt_or_space_retry( + _fut, + _gw=self, + _sid=_hyg_deferred_sid, + _skey=_hyg_deferred_key, + _agent=_hyg_deferred_agent, + ): + try: + _exc = _fut.exception() + except ( + asyncio.CancelledError, + Exception, + ): + _exc = None + _committed = False + else: + _committed = _exc is None and ( + bool( + getattr( + _agent, + "_last_compaction_in_place", + False, + ) + ) + or getattr( + _agent, "session_id", _sid + ) + != _sid + ) + if _committed: + logger.info( + "Session hygiene compression for " + "session %s finished after the " + "turn-hold was released — summary " + "adopted at the watermark-fenced " + "commit boundary (#97963)", + _sid, + ) + try: + _reset_hygiene_failure_streak( + _gw, _skey + ) + except Exception as _rs_err: + logger.debug( + "hygiene streak reset after " + "deferred adoption failed: %s", + _rs_err, + ) + else: + # Nothing to adopt (summary failed, fence + # refused the commit, or the attempt was + # superseded). Record flat spacing so sustained + # traffic does not spawn and abandon a fresh + # compressor every turn; non-escalating — the + # failure streak must not advance for a deferral. + _record_hygiene_cooldown( + _gw, _sid, + _HYGIENE_TURNHOLD_RETRY_SECONDS, + "hygiene compression deferred: " + "turn-hold budget expired and the " + "detached attempt did not commit", + ) + + _hyg_future.add_done_callback( + _hyg_adopt_or_space_retry + ) + from agent.session_activity import ( + ActivityProvenance, + ) + _stamp_hygiene_compression_provenance( + _hyg_agent, + "session hygiene compression turn-hold", + ActivityProvenance.AGENT_COMPRESSION_TURNHOLD, + "hygiene compression turn-hold " + "activity stamp failed", + ) + logger.info( + "Session hygiene compression for session %s " + "exceeded turn-hold budget (%.1fs); " + "proceeding without compression this turn — " + "the watermark-fenced worker keeps its " + "commit admission and the summary will be " + "adopted when it finishes", + session_entry.session_id, + time.monotonic() - _hyg_wait_started, + ) + _turnhold_msg = t( + "gateway.compress.turnhold_deferred" + ) + try: + _adapter = self._adapter_for_source(source) + if _adapter and source.chat_id: + await _adapter.send( + source.chat_id, + _turnhold_msg, + metadata=_hyg_meta, + ) + except Exception as _werr: + logger.warning( + "Failed to deliver compression-turnhold " + "notice to user: %s", + _werr, + ) + raise + _cancelled = None + while _cancelled is None: + if _hyg_commit_fence.commit_in_flight: + _cancelled = False + break + _cancelled = ( + _hyg_commit_fence.try_cancel_before_commit() + ) + if _cancelled is None: + await asyncio.sleep(0.025) + if not _cancelled: + # NOTE: bounded overshoot by design: the turn can be held + # past _hyg_max_turn_hold_seconds by up to the commit + # duration. Aborting mid-commit would corrupt the + # message-store transaction — the overshoot is the + # cheaper failure mode. Do NOT "fix" this into a + # mid-commit cancellation. + _compressed, _ = await _hyg_future + else: + _hyg_commit_fence.release_cancelled_compression_lock() + self._defer_agent_cleanup_until_future_done( + _hyg_future, + _hyg_agent, + context="session hygiene turn-hold", + ) + attempt.cleanup_deferred = True + # Short, NON-escalating retry-after. Without it every + # turn re-spawns a compressor, holds it for the turn-hold + # budget and cancels it — token burn that never commits. + # Deliberately NOT _hygiene_cooldown_for_failure: the + # compressor is healthy, so the failure streak must not + # advance; only flat retry spacing is recorded. + _record_hygiene_cooldown( + self, session_entry.session_id, + _HYGIENE_TURNHOLD_RETRY_SECONDS, + "hygiene compression deferred: " + "turn-hold budget expired while the " + "summary was still streaming", + ) + from agent.session_activity import ( + ActivityProvenance, + ) + _stamp_hygiene_compression_provenance( + _hyg_agent, + "session hygiene compression turn-hold", + ActivityProvenance.AGENT_COMPRESSION_TURNHOLD, + "hygiene compression turn-hold " + "activity stamp failed", + ) + logger.info( + "Session hygiene compression for session %s " + "exceeded turn-hold budget (%.1fs); " + "proceeding without compression this turn", + session_entry.session_id, + time.monotonic() - _hyg_wait_started, + ) + _turnhold_msg = t( + "gateway.compress.turnhold_deferred" + ) + try: + _adapter = self._adapter_for_source(source) + if _adapter and source.chat_id: + await _adapter.send( + source.chat_id, + _turnhold_msg, + metadata=_hyg_meta, + ) + except Exception as _werr: + logger.warning( + "Failed to deliver compression-turnhold " + "notice to user: %s", + _werr, + ) + raise + return _compressed + + async def _hmwa_hygiene_on_timeout(self, attempt, hs, session_entry, session_key, source): + """``except asyncio.TimeoutError`` body: cancel at the commit fence, record the failure + cooldown, warn the user, and re-raise; returns the compressed transcript only when the + worker crossed the commit boundary first.""" + from gateway.run import ( + _hygiene_compression_timeout_message, + _hygiene_cooldown_for_failure, + _record_hygiene_cooldown, + _stamp_hygiene_compression_provenance, + ) + _hyg_agent = attempt.agent + _hyg_meta = attempt.meta + _hyg_commit_fence = attempt.commit_fence + _hyg_future = attempt.future + _hyg_wait_started = attempt.wait_started + _hyg_timeout_seconds = hs.timeout_seconds + _hyg_total_ceiling_seconds = hs.total_ceiling_seconds + _hyg_failure_cooldown_seconds = hs.failure_cooldown_seconds + + _hyg_waited = time.monotonic() - _hyg_wait_started + _hyg_total_exhausted = ( + _hyg_waited >= _hyg_total_ceiling_seconds + or _hyg_commit_fence.deadline_exceeded + ) + if _hyg_total_exhausted: + # The worker cooperatively checks this deadline between + # digest calls. Keep its lease until it exits so an + # unchanged session cannot overlap a retry. + _hyg_commit_fence.retain_compression_lock_until_worker_done() + # Capture fence state BEFORE try_cancel — that call itself + # sets is_cancelled, which would mis-label a genuine idle + # timeout as a fence cancel. + _hyg_fence_cancelled = ( + _hyg_commit_fence.is_cancelled + ) + _cancelled = None + while _cancelled is None: + # #76354 F1: a hung commit retains the fence lock; the + # lock-free phase marker keeps this loop from spinning + # forever while the commit blocks. + if _hyg_commit_fence.commit_in_flight: + _cancelled = False + break + _cancelled = ( + _hyg_commit_fence.try_cancel_before_commit() + ) + if _cancelled is None: + # Round-2 #5: transient lock-setup windows ride + # write patience for seconds; 25ms keeps sub-tick + # latency without 1kHz spin. + await asyncio.sleep(0.025) + if not _cancelled: + # The worker crossed the commit boundary just before the + # timeout; the fence poll waited for it to finish, so + # consume the result instead of treating a successful + # compaction as a timeout. + _compressed, _ = await _hyg_future + else: + # Release an inactivity-timed-out worker's holder- + # qualified lease promptly. Total-ceiling attempts + # retained it above, so this is a no-op for them. + _hyg_commit_fence.release_cancelled_compression_lock() + self._defer_agent_cleanup_until_future_done( + _hyg_future, + _hyg_agent, + context="session hygiene timeout", + ) + attempt.cleanup_deferred = True + _hyg_timeout_error = ( + "session hygiene compression " + "cancelled at commit fence" + if _hyg_fence_cancelled + else ( + "session hygiene compression " + "timed out with no output from " + "the summary model" + ) + ) + if _hyg_failure_cooldown_seconds >= 0: + _hyg_cooldown = await asyncio.to_thread( + _hygiene_cooldown_for_failure, + self, + session_key, + _hyg_failure_cooldown_seconds, + ) + _timeout_reason = ( + _hyg_timeout_error + if _hyg_fence_cancelled + else ( + "session hygiene compression total " + "ceiling exhausted" + if _hyg_total_exhausted + else "session hygiene compression " + "timed out with no output from the " + "summary model" + ) + ) + _record_hygiene_cooldown( + self, session_entry.session_id, + _hyg_cooldown, + _timeout_reason, + ) + from agent.session_activity import ( + ActivityProvenance, + ) + _stamp_hygiene_compression_provenance( + _hyg_agent, + ( + "session hygiene compression " + "cancelled at commit fence" + if _hyg_fence_cancelled + else "session hygiene compression timed out" + ), + ActivityProvenance.AGENT_COMPRESSION_TIMEOUT, + "hygiene compression timeout " + "activity stamp failed", + ) + if _hyg_fence_cancelled: + logger.warning( + "Session hygiene compression for " + "session %s was cancelled at the " + "commit fence; continuing without " + "compression", + session_entry.session_id, + ) + else: + _hyg_elapsed = ( + time.monotonic() - _hyg_wait_started + ) + if _hyg_total_exhausted: + logger.warning( + "Session hygiene compression for session %s " + "reached its total ceiling after %.1fs " + "(progress observed=%s); continuing without " + "compression", + session_entry.session_id, + _hyg_elapsed, + _hyg_commit_fence.progress_observed, + ) + else: + logger.warning( + "Session hygiene compression for session %s " + "made no progress for %.1fs (total wait " + "%.1fs, ceiling %.1fs); continuing without " + "compression", + session_entry.session_id, + _hyg_commit_fence.seconds_since_progress(), + _hyg_elapsed, + _hyg_total_ceiling_seconds, + ) + _timeout_msg = ( + _hygiene_compression_timeout_message( + total_exhausted=_hyg_total_exhausted, + elapsed=_hyg_elapsed, + idle_timeout=_hyg_timeout_seconds, + progress_observed=( + _hyg_commit_fence.progress_observed + ), + ) + ) + try: + _adapter = self._adapter_for_source(source) + if _adapter and source.chat_id: + await _adapter.send( + source.chat_id, + _timeout_msg, + metadata=_hyg_meta, + ) + except Exception as _werr: + logger.warning( + "Failed to deliver compression-timeout " + "warning to user: %s", + _werr, + ) + raise + return _compressed + + def _hmwa_hygiene_on_unwind(self, attempt, hs, session_entry, session_key): + """``except BaseException`` body (caller re-raises): revoke commit admission and record a + cooldown so the next turn does not immediately re-arm hygiene.""" + from gateway.run import _hygiene_cooldown_for_failure, _record_hygiene_cooldown + _hyg_agent = attempt.agent + _hyg_commit_fence = attempt.commit_fence + _hyg_future = attempt.future + _hyg_failure_cooldown_seconds = hs.failure_cooldown_seconds + + # #76354 F2: non-timeout unwind while the detached hygiene + # worker may still run — KeyboardInterrupt, task + # cancellation, or any unexpected error. Revoke commit + # admission (and release the worker's durable lease) BEFORE + # the host unwinds so the worker can never commit later. + _hyg_commit_fence.revoke_commit_admission() + if not attempt.cleanup_deferred: + self._defer_agent_cleanup_until_future_done( + _hyg_future, + _hyg_agent, + context="session hygiene unwind", + ) + attempt.cleanup_deferred = True + # restart drain / task cancel must record a cooldown, or the + # next turn immediately re-arms hygiene and waits up to 600s + # behind a fence that would refuse the commit again. + if _hyg_failure_cooldown_seconds >= 0: + try: + _hyg_cooldown = _hygiene_cooldown_for_failure( + self, + session_key, + _hyg_failure_cooldown_seconds, + ) + _record_hygiene_cooldown( + self, session_entry.session_id, + _hyg_cooldown, + "session hygiene compression " + "cancelled at commit fence", + ) + except Exception as _cd_err: + logger.debug( + "hygiene unwind cooldown " + "record failed: %s", + _cd_err, + ) + + async def _hmwa_hygiene_apply_result( + self, attempt, hs, _compressed, history, *, + _approx_tokens, _msg_count, _warn_token_threshold, + session_entry, session_key, source, _quick_key, run_generation, + ): + """Adopt a finished hygiene compression (rotation / in-place / refused), rebind the session + + turn lease, record streak/cooldown, and warn the user on abort. Publishes the + (possibly replaced) transcript on ``attempt.history``.""" + from gateway.run import ( + _hygiene_cooldown_for_failure, + _record_hygiene_cooldown, + _reset_hygiene_failure_streak, + _stamp_hygiene_compression_provenance, + hygiene_compaction_recovered, + ) + from agent.model_metadata import estimate_messages_tokens_rough + + _hyg_agent = attempt.agent + _hyg_meta = attempt.meta + _hyg_commit_fence = attempt.commit_fence + _hyg_failure_cooldown_seconds = hs.failure_cooldown_seconds + + # _compress_context ends the old session and creates a new + # session_id. Write compressed messages into the NEW session so + # the old transcript stays intact and searchable. + _hyg_new_sid = _hyg_agent.session_id + _hyg_rotated = _hyg_new_sid != session_entry.session_id + _hyg_in_place = bool( + getattr(_hyg_agent, "_last_compaction_in_place", False) + ) + # Anti-growth guard: refuse a compression that did not shrink + # the transcript (observed: 427K -> 598K). Compare like-for-like + # rough estimates. + _hyg_in_toks = estimate_messages_tokens_rough(history) + _hyg_out_toks = estimate_messages_tokens_rough(_compressed) + if _hyg_rotated and _hyg_out_toks > _hyg_in_toks: + logger.warning( + "Gateway hygiene compression for session %s " + "would grow transcript (~%s -> ~%s tokens); " + "keeping the original transcript unchanged", + session_entry.session_id, + f"{_hyg_in_toks:,}", + f"{_hyg_out_toks:,}", + ) + _hyg_rotated = False + _compressed = history + # Rewrite the transcript only when rotation produced a NEW + # session id. In-place compaction needs none: + # archive_and_compact() already soft-archived the previous + # active rows and inserted the compacted set, and + # rewrite_transcript() would run + # replace_messages(active_only=False) and DELETE the archived + # turns. Likewise a summary with neither rotation nor a + # completed archive_and_compact() (unchanged session_id) signals + # FAILURE; an unconditional rewrite would replace the originals + # with only the summary (permanent data loss). + # Write-before-repoint (mirrors manual /compress): if + # session_entry were repointed to the child SID and + # rewrite_transcript then failed (lock/ENOSPC), the live entry + # would reference an empty session — the conversation silently + # vanishes. Persist the child transcript first, then rebind. + if _hyg_rotated: + if not await self.async_session_store.rewrite_transcript( + _hyg_new_sid, _compressed + ): + logger.error( + "Session hygiene: failed to persist " + "compressed transcript for rotated " + "session %s → %s; keeping the live " + "entry on the original session so the " + "conversation is not dropped", + session_entry.session_id, + _hyg_new_sid, + ) + # Fail closed: treat like no rotation. + _hyg_rotated = False + _hyg_in_place = False + else: + session_entry.session_id = _hyg_new_sid + # The held turn lease follows the rotation so an alias + # key resolving the fresh child still serializes against + # this turn. + self._rebind_turn_lease( + _quick_key, run_generation, _hyg_new_sid + ) + await self.async_session_store._save() + await asyncio.to_thread( + self._sync_telegram_topic_binding, + source, session_entry, + reason="hygiene-compression", + ) + + if _hyg_rotated: + # Reset stored token count — transcript rewritten + session_entry.last_prompt_tokens = 0 + attempt.history = _compressed + _new_count = len(_compressed) + _new_tokens = estimate_messages_tokens_rough( + _compressed + ) + elif _hyg_in_place: + # archive_and_compact() already persisted the + # compacted transcript inside _compress_context. + # Reset counts to match the new active set. + session_entry.last_prompt_tokens = 0 + attempt.history = _compressed + _new_count = len(_compressed) + _new_tokens = estimate_messages_tokens_rough( + _compressed + ) + else: + # No rewrite happened — the transcript is unchanged, so the + # post-compression counts equal the pre-compression ones. + _new_count = _msg_count + _new_tokens = _approx_tokens + logger.warning( + "Gateway hygiene compression for session %s " + "did not rotate or compact in place " + "(no session_db on the hygiene agent) — " + "preserving the original transcript instead " + "of overwriting it with the summary (#21301).", + session_entry.session_id, + ) + + logger.info( + "Session hygiene: compressed %s → %s msgs, " + "~%s → ~%s tokens", + _msg_count, _new_count, + f"{_approx_tokens:,}", f"{_new_tokens:,}", + ) + + if _new_tokens >= _warn_token_threshold: + logger.warning( + "Session hygiene: still ~%s tokens after " + "compression", + f"{_new_tokens:,}", + ) + + # Summary failure aborts the compressor entirely (messages + # unchanged, nothing dropped). Warn the gateway user visibly + # — agent.log is invisible on TG/Discord/etc. — so they know + # the chat is "frozen" at this size and can /compress to + # retry or /reset to start fresh. + _comp = getattr(_hyg_agent, "context_compressor", None) + _hyg_aborted = _comp is not None and getattr( + _comp, "_last_compress_aborted", False + ) + # Fence-cancelled _compress_context returns the original + # transcript with _last_compress_aborted still False + # (failure_class=commit_fence_cancelled, chunk_count=0). Treat + # that no-op as an abort so hygiene records a cooldown instead + # of retrying into the 600s wait. A successful rotate/in-place + # commit is not an abort even if a later invalidation flipped + # the fence. + _hyg_fence_cancelled = bool( + _hyg_commit_fence.is_cancelled + and not _hyg_rotated + and not _hyg_in_place + ) + if _hyg_fence_cancelled: + _hyg_aborted = True + if not _hyg_aborted: + # Recovery decision lives in the unit-tested predicate: the + # degenerate "neither rotated nor compacted in place" path + # sets both flags False and reuses the pre-compression + # counts, so a numbers-only check would read a no-op as + # success and clear the streak. + if hygiene_compaction_recovered( + aborted=_hyg_aborted, + rotated=_hyg_rotated, + in_place=_hyg_in_place, + msg_count=_msg_count, + new_count=_new_count, + approx_tokens=_approx_tokens, + new_tokens=_new_tokens, + ): + await asyncio.to_thread( + _reset_hygiene_failure_streak, + self, + session_key, + ) + if _hyg_aborted: + if _hyg_failure_cooldown_seconds >= 0: + _hyg_cooldown = await asyncio.to_thread( + _hygiene_cooldown_for_failure, + self, + session_key, + _hyg_failure_cooldown_seconds, + ) + _record_hygiene_cooldown( + self, session_entry.session_id, + _hyg_cooldown, + ( + "session hygiene compression " + "cancelled at commit fence" + if _hyg_fence_cancelled + else getattr( + _comp, "_last_summary_error", None + ) + ), + ) + from agent.session_activity import ( + ActivityProvenance, + ) + _stamp_hygiene_compression_provenance( + _hyg_agent, + "session hygiene compression aborted", + ActivityProvenance.AGENT_COMPRESSION_COOLDOWN, + "hygiene compression abort " + "activity stamp failed", + ) + if not _hyg_fence_cancelled: + _err = getattr(_comp, "_last_summary_error", None) or "unknown error" + # Force-redact: provider exception text may contain + # credentials and this message reaches gateway users. + from agent.redact import redact_sensitive_text + _err = redact_sensitive_text(_err, force=True) + _warn_msg = ( + "⚠️ Context compression aborted " + f"({_err}). No messages were dropped — " + "conversation is unchanged. Run /compress " + "to retry, /reset for a clean session, or " + "check your auxiliary.compression model " + "configuration." + ) + try: + _adapter = self._adapter_for_source(source) + if _adapter and source.chat_id: + await _adapter.send(source.chat_id, _warn_msg, metadata=_hyg_meta) + except Exception as _werr: + logger.warning( + "Failed to deliver compression-failure warning to user: %s", + _werr, + ) + # Separately: if the user's CONFIGURED aux model failed and we + # recovered by falling back to the main model, tell them — a + # misconfigured auxiliary.compression.model is something only + # they can fix, and silent recovery would hide it. + elif _comp is not None and getattr(_comp, "_last_aux_model_failure_model", None): + _aux_model = getattr(_comp, "_last_aux_model_failure_model", "") + _aux_err = getattr(_comp, "_last_aux_model_failure_error", None) or "unknown error" + _aux_msg = ( + f"ℹ️ Configured compression model `{_aux_model}` " + f"failed ({_aux_err}). Recovered using your main " + "model — context is intact — but you may want to " + "check `auxiliary.compression.model` in config.yaml." + ) + try: + _adapter = self._adapter_for_source(source) + if _adapter and source.chat_id: + await _adapter.send(source.chat_id, _aux_msg, metadata=_hyg_meta) + except Exception as _werr: + logger.warning( + "Failed to deliver aux-model-fallback notice to user: %s", + _werr, + ) + + async def _hmwa_run_session_hygiene( + self, event, source, session_entry, session_key, history, _quick_key, run_generation, + ): + # Session hygiene: auto-compress pathologically large transcripts before the agent starts so + # oversized histories don't cause repeated truncation/context failures. Token source: the + # API's prompt_tokens from the last turn (session_entry.last_prompt_tokens), else a char/4 + # estimate (30-50% high on code-heavy sessions, so hygiene merely fires a bit early). + from gateway.run import ( + HygieneTurnHoldExceeded, + _GATEWAY_HYGIENE_PLATFORM, + _seed_hygiene_system_prompt, + run_codex_hygiene_compaction, + ) + if not history or len(history) < 4: + return history + + hs = await self._hmwa_hygiene_settings(source, session_key) + if not hs.compression_enabled: + return history + _needs_compress, _approx_tokens, _msg_count, _warn_token_threshold = ( + await self._hmwa_hygiene_plan(hs, history, session_entry, session_key) + ) + if not _needs_compress: + return history + + _hyg_total_ceiling_seconds = hs.total_ceiling_seconds + _hyg_failure_cooldown_seconds = hs.failure_cooldown_seconds + _hyg_data = hs.data + + _hyg_meta = self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)) + + try: + from agent.conversation_compression import CompressionCommitFence + from run_agent import AIAgent + + _hyg_model, _hyg_runtime = self._resolve_session_agent_runtime( + source=source, + session_key=session_key, + user_config=_hyg_data if isinstance(_hyg_data, dict) else None, + ) + _hyg_api_mode = str( + _hyg_runtime.get("api_mode") or "" + ).lower() + if _hyg_api_mode == "codex_app_server": + # codex app-server runtime: the real context is the server-side thread, + # not the transcript mirror. The detached-agent block below would only + # rewrite the mirror and its finally-eviction would destroy the live + # thread (next turn starts blank). Use the cached agent's + # thread/compact/start and KEEP it cached. + _hyg_codex_auto = "native" + _hyg_comp_cfg = ( + _hyg_data.get("compression") + if isinstance(_hyg_data, dict) + else None + ) + if isinstance(_hyg_comp_cfg, dict): + _hyg_codex_auto = str( + _hyg_comp_cfg.get( + "codex_app_server_auto", "native" + ) + or "native" + ) + _hyg_codex_outcome = await run_codex_hygiene_compaction( + self, + session_key, + session_entry.session_id, + auto_mode=_hyg_codex_auto, + history=history, + approx_tokens=_approx_tokens, + timeout_seconds=_hyg_total_ceiling_seconds, + failure_cooldown_seconds=_hyg_failure_cooldown_seconds, + ) + logger.info( + "Session hygiene (codex app-server): %s " + "(session=%s, mode=%s, ~%s tokens)", + _hyg_codex_outcome, + session_entry.session_id, + _hyg_codex_auto, + f"{_approx_tokens:,}", + ) + elif _hyg_runtime.get("api_key"): + # Pass the FULL transcript (tool results included), matching the agent + # loop: filtering to user/assistant starved the compressor — tool results + # are the bulk of context and short histories tripped the + # protect-first/last early-return so nothing compressed. + _hyg_msgs = [ + m for m in history + if m.get("role") in {"user", "assistant", "tool"} + ] + + if len(_hyg_msgs) >= 4: + try: + _hyg_session_row = await self._session_db.get_session( + session_entry.session_id + ) + except Exception as exc: + _hyg_session_row = None + logger.warning( + "Session hygiene could not restore the system " + "prompt for session %s: %s. Preserving an empty " + "prompt so the live turn rebuilds it with its " + "configured providers.", + session_entry.session_id, + exc, + exc_info=True, + ) + _hyg_session_db = getattr(self._session_db, "_db", self._session_db) + # Hygiene is the same lossy rewrite as normal compression: when + # compression.checkpoint_required is on, load the memory provider so + # the checkpoint exists before any mutation; otherwise keep the + # fast path (no provider init, no best-effort hook). + from hermes_cli.config import load_config as _load_cfg + from utils import is_truthy_value as _is_truthy + + _hyg_checkpoint_required = _is_truthy( + ((_load_cfg() or {}).get("compression") or {}).get( + "checkpoint_required" + ), + default=False, + ) + _hyg_agent = AIAgent( + **_hyg_runtime, + model=_hyg_model, + max_iterations=4, + quiet_mode=True, + skip_memory=not _hyg_checkpoint_required, + enabled_toolsets=["memory"], + session_id=session_entry.session_id, + session_db=_hyg_session_db, + ) + _seed_hygiene_system_prompt( + _hyg_agent, + _hyg_session_row, + ) + # If compression must rebuild instead of retaining + # the cached prompt, make the persisted result + # deliberately stale for every real gateway surface. + _hyg_agent.platform = _GATEWAY_HYGIENE_PLATFORM + attempt = self._HygieneAttempt(agent=_hyg_agent, meta=_hyg_meta, history=history) + try: + # Hygiene runs before the turn and owns the session binding, so + # prefer in-place compaction: archive old rows under the same + # session id rather than minting a continuation child that must + # be published back to SessionStore/topic bindings. Without a + # SessionDB this stays False and the guard below preserves it. + _hyg_agent.compression_in_place = True + _bind_hyg_state = getattr( + getattr(_hyg_agent, "context_compressor", None), + "bind_session_state", + None, + ) + if callable(_bind_hyg_state): + _bind_hyg_state( + _hyg_session_db, + session_entry.session_id, + ) + # It must never finalize on close() — close() + # would end the live gateway session row. + _hyg_agent._end_session_on_close = False + _hyg_agent._print_fn = lambda *a, **kw: None + + loop = asyncio.get_running_loop() + _hyg_commit_fence = CompressionCommitFence( + total_ceiling_seconds=_hyg_total_ceiling_seconds + ) + # Default executor (NOT self._get_executor): a fence-cancelled + # hung summary must never occupy an agent-work slot. But it MUST + # run in the caller's contextvars: under multiplex_profiles the + # secret scope / HERMES_HOME live in ContextVars, and an empty + # Context makes get_secret() fail closed → lossy truncation. + _hyg_future = loop.run_in_executor( + None, + copy_context().run, + lambda: _hyg_agent._compress_context( + _hyg_msgs, "", + approx_tokens=_approx_tokens, + commit_fence=_hyg_commit_fence, + ), + ) + attempt.commit_fence = _hyg_commit_fence + attempt.future = _hyg_future + attempt.wait_started = time.monotonic() + try: + _compressed = await self._hmwa_hygiene_wait_for_summary( + attempt, hs, session_entry, + ) + except HygieneTurnHoldExceeded: + _compressed = await self._hmwa_hygiene_on_turn_hold( + attempt, hs, session_entry, session_key, source, + ) + except asyncio.TimeoutError: + _compressed = await self._hmwa_hygiene_on_timeout( + attempt, hs, session_entry, session_key, source, + ) + except BaseException: + self._hmwa_hygiene_on_unwind(attempt, hs, session_entry, session_key) + raise + + await self._hmwa_hygiene_apply_result( + attempt, hs, _compressed, history, + _approx_tokens=_approx_tokens, + _msg_count=_msg_count, + _warn_token_threshold=_warn_token_threshold, + session_entry=session_entry, + session_key=session_key, + source=source, + _quick_key=_quick_key, + run_generation=run_generation, + ) + finally: + history = attempt.history + # Evict the cached agent so the next turn rebuilds its system + # prompt from current SOUL.md, memory, and skills. + self._evict_cached_agent(session_key) + if not attempt.cleanup_deferred: + await self._cleanup_agent_resources_off_loop( + _hyg_agent, context="session hygiene" + ) + except HygieneTurnHoldExceeded: + # Availability boundary, not a failure — already logged at INFO by the turn- + # hold handler. Must not hit the generic "auto-compress failed" warning + # below: that log made thinking-model deployments read as permanently broken. + pass + except Exception as e: + logger.warning( + "Session hygiene auto-compress failed: %s", e + ) + return history + + async def _hmwa_first_contact_notes(self, source, history, turn_sidecar_notes): + """First-ever-message onboarding note + one-time 'no home channel' prompt (both only when + the session has no history).""" + from gateway.run import _hermes_home, _home_target_env_var, _load_gateway_config + # First-message onboarding -- only on the very first interaction ever. Delivered on the + # current user message (sidecar), NOT the ephemeral system prompt: present-on-turn-1/absent- + # on-turn-2 was a guaranteed system-prompt diff and agent rebuild. + if not history and not await self.async_session_store.has_any_sessions(): + # Default first-contact note: a brief self-introduction. + _intro_note = ( + "[System note: This is the user's very first message ever. " + "Briefly introduce yourself and mention that /help shows available commands. " + "Keep the introduction concise -- one or two sentences max.]" + ) + # Opt-in structured profile-build path: when enabled (default "ask") and not yet offered + # on this install, swap the plain intro for a consent-gated directive that offers to + # build a user profile and persists confirmed facts via memory(target="user"). Fires at + # most once (onboarding.seen flag); onboarding.profile_build: off in config.yaml + # disables it. + try: + from agent.onboarding import ( + PROFILE_BUILD_FLAG, + is_seen, + mark_seen, + profile_build_directive, + profile_build_mode, + ) + _onb_cfg = _load_gateway_config() + if ( + profile_build_mode(_onb_cfg) == "ask" + and not is_seen(_onb_cfg, PROFILE_BUILD_FLAG) + ): + turn_sidecar_notes.append(profile_build_directive().strip()) + mark_seen(_hermes_home / "config.yaml", PROFILE_BUILD_FLAG) + else: + turn_sidecar_notes.append(_intro_note) + except Exception as _pb_err: + logger.debug( + "Profile-build onboarding directive failed, using plain intro: %s", + _pb_err, + ) + turn_sidecar_notes.append(_intro_note) + + # One-time prompt if no home channel is set for this platform + # Skip for webhooks - they deliver directly to configured targets (github_comment, etc.) + if not history and source.platform and source.platform != Platform.LOCAL and source.platform != Platform.WEBHOOK: + platform_name = source.platform.value + env_key = _home_target_env_var(platform_name) + # Multiplex: home channel may live only in the profile secret + # scope / PlatformConfig, not process os.environ. + home_env = "" + try: + from agent.secret_scope import get_secret + + home_env = (get_secret(env_key) or "").strip() if env_key else "" + except Exception: + home_env = "" + if not home_env: + home_env = (os.getenv(env_key) or "").strip() if env_key else "" + # Also honor in-memory / yaml home_channel on this platform. + try: + if not home_env and self.config.get_home_channel(source.platform): + home_env = "set" + except Exception: + pass + # Secondary-profile platforms (e.g. Slack on yolo) may only exist + # under that profile's loaded config — check after scope install. + if not home_env: + try: + from gateway.config import load_gateway_config as _lgc + prof = (getattr(source, "profile", None) or "").strip() + if prof and prof != "default": + # Already inside profile scope for secondary handlers; + # re-read live config for home_channel. + _pcfg = _lgc() + if _pcfg.get_home_channel(source.platform): + home_env = "set" + except Exception: + pass + if not home_env: + # Slack dispatches all Hermes commands through a single + # parent slash command `/hermes`; bare `/sethome` is not + # registered and would fail with "app did not respond". + sethome_cmd = ( + "/hermes sethome" + if source.platform == Platform.SLACK + else "/sethome" + ) + notice = ( + f"📬 No home channel is set for {platform_name.title()}. " + f"A home channel is where Hermes delivers cron job results " + f"and cross-platform messages.\n\n" + f"Type {sethome_cmd} to make this chat your home channel, " + f"or ignore to skip." + ) + await self._deliver_platform_notice(source, notice) + + def _hmwa_apply_message_timestamp(self, event, message_text): + # Capture the platform event time as message metadata and keep the persisted transcript + # clean (strip any leading timestamp prefix). This runs regardless of the toggle so storage + # stays clean and the send-time is preserved. Only the in-context RENDER (the prefix the + # model sees) is gated behind gateway.message_timestamps.enabled — default OFF. + from gateway.run import _load_gateway_config, _message_timestamps_enabled + persist_user_message = None + persist_user_timestamp = None + try: + from hermes_time import get_timezone as _get_evt_tz + from gateway.message_timestamps import ( + coerce_message_timestamp as _coerce_msg_ts, + render_user_content_with_timestamp as _render_msg_ts, + strip_leading_message_timestamps as _strip_msg_ts, + ) + _evt_tz = _get_evt_tz() + _evt_ts = getattr(event, "timestamp", None) + if message_text and isinstance(message_text, str): + _clean_message_text, _embedded_ts = _strip_msg_ts( + message_text, tz=_evt_tz) + persist_user_message = _clean_message_text + _event_epoch = _coerce_msg_ts(_evt_ts, tz=_evt_tz) + persist_user_timestamp = ( + _event_epoch if _event_epoch is not None else _embedded_ts + ) + if _message_timestamps_enabled(_load_gateway_config()): + message_text = _render_msg_ts( + _clean_message_text, + persist_user_timestamp, + tz=_evt_tz, + ) + else: + # Toggle off: model sees the clean message; the timestamp + # is still stored as metadata for later opt-in. + message_text = _clean_message_text + except Exception as _ts_err: + logger.debug("Message timestamp injection failed (non-fatal): %s", _ts_err) + return message_text, persist_user_message, persist_user_timestamp + + async def _hmwa_stop_typing_for_turn(self, event, source): + """Stop the typing indicator (never raises). Slack AI status is scoped to a thread/ + workspace, so preserve the routing metadata used by the response delivery path.""" + try: + _typing_adapter = self._adapter_for_source(source) + _stop_with_metadata = getattr( + type(_typing_adapter), "_stop_typing_with_metadata", None + ) + _stop_typing = getattr(type(_typing_adapter), "stop_typing", None) + if _typing_adapter and callable(_stop_with_metadata): + await _typing_adapter._stop_typing_with_metadata( + source.chat_id, + self._thread_metadata_for_source( + source, self._reply_anchor_for_event(event) + ), + ) + elif _typing_adapter and callable(_stop_typing): + await _typing_adapter.stop_typing(source.chat_id) + except Exception: + pass + + async def _hmwa_shape_agent_response( + self, agent_result, source, history, session_entry, session_key, + _quick_key, run_generation, _run_start_session_id, _platform_name, _msg_start_time, + ): + """Turn the raw agent result into the outbound text: sentinel/silence handling, response + logging, resume-pending clear, empty-response normalization, and identity-guarded + post-compression session_id propagation. Returns + ``(response, _intentional_silence, agent_messages)``.""" + from gateway.run import ( + _is_gateway_hidden_reasoning_incomplete_turn, + _normalize_empty_agent_response, + _sanitize_gateway_final_response, + _should_clear_resume_pending_after_turn, + ) + response = agent_result.get("final_response") or "" + # Hidden-reasoning-only retry exhaustion: the loop's sentinel text ("Codex response + # remained incomplete after 3 continuation attempts") doubles as final_response, so it + # would be delivered verbatim into the channel — where peer agents can ingest it as a + # completed assistant turn. + if _is_gateway_hidden_reasoning_incomplete_turn(agent_result): + response = "" + try: + from gateway.response_filters import is_intentional_silence_agent_result + _intentional_silence = is_intentional_silence_agent_result( + agent_result, response, + ) + except Exception: + _intentional_silence = False + + # Convert the agent's internal "(empty)" sentinel into a user-friendly message. + # "(empty)" means the model failed to produce visible content after exhausting all + # retries (nudge, prefill, empty-retry, fallback). + if response == "(empty)" and not _intentional_silence: + response = ( + "⚠️ The model returned no response after processing tool " + "results. This can happen with some models — try again or " + "rephrase your question." + ) + agent_messages = agent_result.get("messages", []) + _response_time = time.time() - _msg_start_time + _api_calls = agent_result.get("api_calls", 0) + _resp_len = len(response) + logger.info( + "response ready: platform=%s chat=%s time=%.1fs api_calls=%d response=%d chars", + _platform_name, source.chat_id or "unknown", + _response_time, _api_calls, _resp_len, + ) + + # The cross-process cache-coherence re-baseline (_refresh_agent_cache_message_count) is + # deferred until AFTER the transcript persistence block below: it must include the + # first-turn `session_meta` marker row and the compression session_id swap. + + # Successful turn: clear the stuck-loop counter (it only accumulates across CONSECUTIVE + # restarts where the session never completed) and resume_pending (set by drain-timeout + # shutdown) so later messages don't get the restart-interruption system note. + if session_key and _should_clear_resume_pending_after_turn(agent_result): + await self._clear_restart_failure_count(session_key) + try: + await self.async_session_store.clear_resume_pending(session_key) + except Exception as _e: + logger.debug( + "clear_resume_pending failed for %s: %s", + session_key, _e, + ) + + # Normalize empty responses: surface errors, partial failures, and + # the case where agent did work but returned no text. Fix for #18765. + if not _intentional_silence: + response = _normalize_empty_agent_response( + agent_result, response, history_len=len(history), + ) + response = _sanitize_gateway_final_response(source.platform, response) + + # Ordering contract: the agent thread already updated the contextvar in + # conversation_compression.py; propagate to SessionEntry + _save(). + if agent_result.get("session_id") and agent_result["session_id"] != session_entry.session_id: + if session_entry.session_id == _run_start_session_id: + session_entry.session_id = agent_result["session_id"] + # The held turn lease follows the rotation: the transcript persistence below + # writes to the NEW id, so the serialization boundary must move with it or an + # alias key resolving the fresh child could interleave. + self._rebind_turn_lease( + _quick_key, run_generation, session_entry.session_id + ) + await self.async_session_store._save() + await self.async_session_store._record_gateway_session_peer( + session_entry.session_id, + session_key, + source, + ) + await asyncio.to_thread( + self._sync_telegram_topic_binding, + source, session_entry, reason="agent-result-compression", + ) + else: + logger.info( + "Skipping agent-result session split sync for %s because " + "the session binding moved from %s to %s before " + "compression finished", + session_key or "?", + _run_start_session_id, + session_entry.session_id, + ) + return response, _intentional_silence, agent_messages + + def _hmwa_prepend_reasoning(self, agent_result, response, source, _intentional_silence): + # Prepend reasoning if display is enabled (per-platform). Mattermost requires explicit + # opt-in because this is scratch text, not ordinary final-answer content. + from gateway.run import _load_gateway_config, _platform_config_key, _resolve_gateway_display_bool + try: + _show_reasoning_effective = _resolve_gateway_display_bool( + _load_gateway_config(), + _platform_config_key(source.platform), + "show_reasoning", + default=bool(getattr(self, "_show_reasoning", False)), + platform=source.platform, + require_platform_override_for={Platform.MATTERMOST}, + ) + except Exception: + _show_reasoning_effective = ( + False + if source.platform == Platform.MATTERMOST + else getattr(self, "_show_reasoning", False) + ) + if _show_reasoning_effective and response and not _intentional_silence: + last_reasoning = agent_result.get("last_reasoning") + if last_reasoning: + from gateway.stream_consumer import escape_code_fences_for_display + # Collapse long reasoning to keep messages readable + lines = last_reasoning.strip().splitlines() + if len(lines) > 15: + display_reasoning = "\n".join(lines[:15]) + display_reasoning += f"\n_... ({len(lines) - 15} more lines)_" + else: + display_reasoning = last_reasoning.strip() + # Render style is per-platform: Discord defaults to "-# " subtext (native small + # grey metadata text); other platforms keep the fenced code block. + try: + from gateway.display_config import resolve_display_setting + _reasoning_style = resolve_display_setting( + _load_gateway_config(), + _platform_config_key(source.platform), + "reasoning_style", + "code", + ) + except Exception: + _reasoning_style = "code" + if _reasoning_style == "subtext": + _quoted = "\n".join( + f"-# {ln}" if ln else "-#" for ln in display_reasoning.splitlines() + ) + response = f"-# 💭 Reasoning\n{_quoted}\n\n{response}" + elif _reasoning_style == "blockquote": + _quoted = "\n".join( + f"> {ln}" if ln else ">" for ln in display_reasoning.splitlines() + ) + response = f"> 💭 **Reasoning:**\n{_quoted}\n\n{response}" + else: + # Escape ``` inside reasoning so inner fences don't + # break the outer code block used to render it. + display_reasoning = escape_code_fences_for_display(display_reasoning) + response = f"💭 **Reasoning:**\n```\n{display_reasoning}\n```\n\n{response}" + return response + + def _hmwa_runtime_footer_line(self, agent_result, source, _turn_seconds): + # Runtime-metadata footer — only on the FINAL message of the turn. Off by default + # (display.runtime_footer.enabled=false). When streaming already delivered the body, we + # can't mutate the sent text, so we fire a separate trailing send below. + from gateway.run import _load_gateway_config, _platform_config_key, _terminal_scope_cwd + _footer_line = "" + try: + from gateway.runtime_footer import build_footer_line as _bfl + _footer_line = _bfl( + user_config=_load_gateway_config(), + platform_key=_platform_config_key(source.platform), + model=agent_result.get("model"), + context_tokens=agent_result.get("last_prompt_tokens", 0) or 0, + context_length=agent_result.get("context_length") or None, + cwd=_terminal_scope_cwd(""), + turn_seconds=_turn_seconds, + ) + except Exception as _footer_err: + logger.debug("runtime_footer build failed: %s", _footer_err) + _footer_line = "" + return _footer_line + + async def _hmwa_post_turn_hooks(self, hook_ctx, agent_result, response): + """agent:end hook, process-watcher scheduling, and watch-notification drain.""" + # Emit agent:end hook + await self.hooks.emit("agent:end", { + **hook_ctx, + "response": (response or "")[:500], + "model": agent_result.get("model", ""), + "provider": agent_result.get("provider", ""), + }) + + # Check for pending process watchers (check_interval on background processes) + try: + from tools.process_registry import process_registry + # Detach the current batch atomically (see crash-recovery drain + # above): reassign to a fresh list so a watcher appended by a + # concurrent session during the yield isn't dropped by clear(). + watchers = process_registry.pending_watchers + process_registry.pending_watchers = [] + for i, watcher in enumerate(watchers): + asyncio.create_task(self._run_process_watcher(watcher)) + if i % 100 == 99: + await asyncio.sleep(0) + except Exception as e: + logger.error("Process watcher setup error: %s", e) + + # Drain watch notifications that arrived during the run. The queue also carries process + # completions (handled by the per-process watcher task above) and async-delegation + # completions (owned by _async_delegation_watcher, the single consumer for idle and + # post-turn cases) — inject only watch-type events and leave the rest on the queue. + try: + from tools.process_registry import process_registry as _pr + await self._drain_watch_notifications(_pr.completion_queue) + except Exception as e: + logger.debug("Watch queue drain error: %s", e) + + def _hmwa_classify_turn_failure(self, agent_result, history, session_entry): + """Classify a finished turn for transcript persistence. Returns + ``(agent_failed_early, hidden_reasoning_incomplete, is_context_overflow_failure)``.""" + from gateway.run import _is_gateway_hidden_reasoning_incomplete_turn + # Persist the full agent loop (tool calls, results, reasoning) so sessions resume with + # full context. IMPORTANT: on context-overflow failures (compression exhausted, generic + # 400 on large sessions) do NOT persist the user message — it would grow the session and + # reproduce the failure forever. Transient failures (429, timeout, connection error, + # 5xx) are different: the session is not oversized and dropping the user turn causes + # severe context loss on retry, so persist it. + agent_failed_early = bool(agent_result.get("failed")) + hidden_reasoning_incomplete = _is_gateway_hidden_reasoning_incomplete_turn( + agent_result + ) + _err_str_for_classify = str(agent_result.get("error", "")).lower() + # Use specific multi-word phrases (not bare "exceed"/"token") to avoid false positives + # on transient errors such as "rate limit exceeded"; matches run_agent.py's classifier. + is_context_overflow_failure = agent_failed_early and ( + bool(agent_result.get("compression_exhausted")) + or any(p in _err_str_for_classify for p in ( + "context length", "context size", "context window", + "maximum context", "token limit", "too many tokens", + "reduce the length", "exceeds the limit", + "request entity too large", "prompt is too long", + "payload too large", "input is too long", + )) + or ("400" in _err_str_for_classify and len(history) > 50) + ) + if is_context_overflow_failure: + logger.info( + "Skipping transcript persistence for context-overflow " + "failure in session %s to prevent session growth loop.", + session_entry.session_id, + ) + elif agent_failed_early: + logger.info( + "Transient agent failure in session %s — persisting user " + "message so conversation context is preserved on retry.", + session_entry.session_id, + ) + elif hidden_reasoning_incomplete: + logger.warning( + "Suppressing hidden-reasoning-only incomplete gateway turn " + "for session %s: %s", + session_entry.session_id, + agent_result.get("error", "processing incomplete"), + ) + return agent_failed_early, hidden_reasoning_incomplete, is_context_overflow_failure + + async def _hmwa_compression_exhaustion_reset( + self, agent_result, response, session_entry, session_key, source, + ): + """Auto-reset a permanently oversized session (never on a lock-contended defer). Returns + ``(response, session_entry)``.""" + # Compression exhausted = permanently too large: auto-reset so the next message starts + # fresh instead of replaying the oversized context forever. A lock-contended defer is + # the OPPOSITE case (a concurrent path holds the lock and is shrinking it): never wipe + # for that. + if agent_result.get("compression_deferred"): + logger.info( + "Compression deferred for session %s — the compression " + "lock is held by a concurrent compressor. Keeping the " + "session intact; the next message retries normally.", + session_entry.session_id if session_entry else "?", + ) + elif agent_result.get("compression_exhausted") and session_entry and session_key: + logger.info( + "Auto-resetting session %s after compression exhaustion.", + session_entry.session_id, + ) + new_entry = await self.async_session_store.reset_session(session_key) + self._evict_cached_agent(session_key) + # Conversation boundary: one funnel call clears every conversation-scoped + # per-session dict (see _CONVERSATION_SCOPED_STATE). + self._clear_conversation_scope( + session_key, reason="compression_exhausted_reset" + ) + if new_entry is not None: + # Re-point the Telegram topic binding at the fresh session: compression rotated + # session_entry.session_id to the bloated child earlier this turn and that _sync + # also rewrote the (chat_id, thread_id) binding. Without a re-sync the + # binding-heal walk switches the next inbound message back onto the child and + # re-triggers exhaustion forever. No-op on non-topic lanes. + session_entry = new_entry + await asyncio.to_thread( + self._sync_telegram_topic_binding, + source, session_entry, reason="compression-exhausted-reset", + ) + response = (response or "") + ( + "\n\n🔄 Session auto-reset — the conversation exceeded the " + "maximum context size and could not be compressed further. " + "Your next message will start a fresh session." + ) + return response, session_entry + + async def _hmwa_persist_turn_transcript( + self, *, event, source, session_entry, session_key, agent_result, agent_messages, + history, response, message_text, persist_user_message, persist_user_timestamp, + persist_user_display_kind, agent_failed_early, hidden_reasoning_incomplete, + is_context_overflow_failure, + ): + """Persist this turn to the transcript (session_meta on first turn, user-only on transient + failure, nothing on context overflow), update last_prompt_tokens, and re-baseline the + cached agent's message count.""" + from gateway.run import _resolve_gateway_model + ts = time.time() # Unix epoch float — consistent with DB storage + + # Fresh session (no history): write the full tool definitions as the first entry so the + # transcript is self-describing — the same dicts sent as tools=[...] in the API request. + if is_context_overflow_failure: + pass # Skip all transcript writes — don't grow a broken session + elif not history: + tool_defs = agent_result.get("tools", []) + await self.async_session_store.append_to_transcript( + session_entry.session_id, + { + "role": "session_meta", + "tools": tool_defs or [], + "model": _resolve_gateway_model(), + "platform": source.platform.value if source.platform else "", + "timestamp": ts, + } + ) + + # The agent already persisted these via _flush_messages_to_session_db(); skip the DB + # write to avoid duplicates. Holds for the codex app-server runtime too (it flushes its + # own projected messages before returning and reports agent_persisted=True). Reading the + # flag (default = self._session_db is not None) keeps the contract explicit; a + # non-persisting runtime opts in via False. + agent_persisted = agent_result.get("agent_persisted", self._session_db is not None) + + # Only the NEW messages from this turn: use history_offset (what the agent saw), not + # len(history), which counts session_meta entries stripped before the agent saw them. + if is_context_overflow_failure: + pass # handled above — skip all transcript writes + elif agent_failed_early or hidden_reasoning_incomplete: + # Transient failure (429/timeout/5xx): persist only the user message so the next + # message can load a transcript that reflects what was said. Skip the assistant + # error text since it's a gateway-generated hint, not model output. Hidden-reasoning + # incomplete turns follow the same rule so peer-agent channels don't ingest them. + _user_entry = { + "role": "user", + "content": ( + persist_user_message + if persist_user_message is not None + else message_text + ), + "timestamp": ( + persist_user_timestamp + if persist_user_timestamp is not None + else ts + ), + } + if persist_user_display_kind: + _user_entry["display_kind"] = persist_user_display_kind + if event.message_id: + _user_entry["message_id"] = str(event.message_id) + # Dedupe: skip if this platform message_id is already in the transcript (prevents + # duplicate user turns on Telegram retries after transient failures). + _skip_persist = ( + event.message_id + and await self.async_session_store.has_platform_message_id( + session_entry.session_id, str(event.message_id) + ) + ) + if _skip_persist: + logger.info( + "Skipping duplicate user turn " + "(message_id=%s) in session %s", + event.message_id, session_entry.session_id, + ) + else: + await self.async_session_store.append_to_transcript( + session_entry.session_id, + _user_entry, + skip_db=agent_persisted, + ) + else: + history_len = agent_result.get("history_offset", len(history)) + new_messages = agent_messages[history_len:] if len(agent_messages) > history_len else [] + + # If no new messages found (edge case), fall back to simple user/assistant + if not new_messages: + _user_entry = { + "role": "user", + "content": ( + persist_user_message + if persist_user_message is not None + else message_text + ), + "timestamp": ( + persist_user_timestamp + if persist_user_timestamp is not None + else ts + ), + } + if persist_user_display_kind: + _user_entry["display_kind"] = persist_user_display_kind + if event.message_id: + _user_entry["message_id"] = str(event.message_id) + await self.async_session_store.append_to_transcript( + session_entry.session_id, + _user_entry, + skip_db=agent_persisted, + ) + if response: + await self.async_session_store.append_to_transcript( + session_entry.session_id, + {"role": "assistant", "content": response, "timestamp": ts}, + skip_db=agent_persisted, + ) + else: + # Attach the inbound platform message_id to the first user entry written this + # turn so platform-level quote-resolution (e.g. Yuanbao QuoteContextMiddleware's + # transcript fallback) can find earlier @bot messages by their original id. + _user_msg_id_attached = False + for msg in new_messages: + # Skip system messages (they're rebuilt each run) + if msg.get("role") == "system": + continue + # Add timestamp to each message for debugging + entry = {**msg, "timestamp": ts} + if ( + not _user_msg_id_attached + and msg.get("role") == "user" + and event.message_id + and "message_id" not in entry + ): + entry["message_id"] = str(event.message_id) + _user_msg_id_attached = True + await self.async_session_store.append_to_transcript( + session_entry.session_id, entry, + skip_db=agent_persisted, + ) + + # The agent persists token counts and model itself; keep only last_prompt_tokens here + # for context-window tracking and compression decisions. + await self.async_session_store.update_session( + session_entry.session_key, + last_prompt_tokens=agent_result.get("last_prompt_tokens", 0), + touch_activity=not bool(getattr(event, "internal", False)), + ) + + # Re-baseline the cached agent's message_count snapshot now that ALL of this turn's + # transcript writes are done (flushed rows AND the first-turn `session_meta` marker). + # The cross-process coherence guard snapshots at agent-BUILD time and never refreshes + # on reuse, so our own writes would trigger a rebuild next turn (destroying prompt + # caching). MUST run after the session_meta append (that row bumps the count too). + await self._refresh_agent_cache_message_count( + session_key, session_entry.session_id + ) + + async def _hmwa_deliver_turn_response( + self, event, source, session_entry, session_key, run_generation, + agent_result, agent_messages, response, _footer_line, _intentional_silence, + ): + """Final delivery decisions: intentional silence, voice reply, streamed-turn media/footer. + Returns the text for the adapter to send, or ``None`` when already delivered.""" + # Intentional silence is a delivery decision, not a transcript mutation: the [SILENT] + # assistant turn stays persisted so later turns keep user/assistant alternation; only + # the outbound delivery is suppressed. + if _intentional_silence: + logger.info( + "Suppressing intentional silence marker for session %s", + session_entry.session_id, + ) + response = "" + + # Auto voice reply: send TTS audio before the text response + _already_sent = bool(agent_result.get("already_sent")) + # Skip when streaming TTS already delivered audio for this turn (#60671). + _stts_adapter = self._adapter_for_source(source) + _streaming_tts_done = ( + _stts_adapter is not None + and bool(getattr(_stts_adapter, "_streaming_tts_turn_completed", lambda *_a, **_k: False)(session_key, run_generation)) + ) + if ( + not _streaming_tts_done + and self._should_send_voice_reply(event, response, agent_messages, already_sent=_already_sent) + ): + await self._send_voice_reply(event, response) + + # Streamed responses still need MEDIA: files delivered before returning None (chunks + # carry the tags verbatim and post-processing is skipped when already_sent). Never skip + # when the agent failed: the error text is new content streaming didn't show. + if agent_result.get("already_sent") and not agent_result.get("failed"): + if response: + _media_adapter = self._adapter_for_source(source) + if _media_adapter: + await self._deliver_media_from_response( + response, event, _media_adapter, + ) + # Streaming already delivered the body text, but the footer was intentionally held + # back (see the `not already_sent` gate above). + if _footer_line: + try: + _foot_adapter = self._adapter_for_source(source) + if _foot_adapter: + await _foot_adapter.send( + source.chat_id, + _footer_line, + metadata=self._thread_metadata_for_source(source, self._reply_anchor_for_event(event)), + ) + except Exception as _e: + logger.debug("trailing footer send failed: %s", _e) + # This branch returns None so the adapter does not send the body twice. /loop and + # /goal hooks in _handle_message read the return value, so stash the delivered text + # on the event or those hooks never run and a /loop tick stays awaiting. + with suppress(Exception): + event._streamed_final_response = str(response or "") + return None + + return response + + async def _hmwa_agent_error_reply( + self, e, event, source, session_entry, session_key, history, message_text, + persist_user_message, persist_user_timestamp, persist_user_display_kind, + ): + """``except Exception`` body of the agent turn: stop typing, log, persist the inbound user + turn once, and build the sanitized user-facing error reply.""" + # Stop typing indicator on error too, retaining Slack thread/workspace + # routing so a failed turn cannot leave its status visible. + await self._hmwa_stop_typing_for_turn(event, source) + logger.exception("Agent error in session %s", session_key) + # Crash-resilience for failures before AIAgent enters run_conversation() (e.g. provider/ + # httpx client init): the agent can't persist the inbound turn there, so append the user + # message here once; if the agent already reached turn-start persistence the latest user + # row matches and we skip the duplicate. + try: + if message_text is not None and session_entry is not None: + _already_persisted = False + try: + _recent_transcript = await self.async_session_store.load_transcript(session_entry.session_id) + except Exception: + _recent_transcript = [] + for _msg in reversed(_recent_transcript[-10:]): + if _msg.get("role") == "user": + _expected_user_content = ( + persist_user_message + if persist_user_message is not None + else message_text + ) + _already_persisted = (_msg.get("content") == _expected_user_content) + break + if not _already_persisted: + _user_entry = { + "role": "user", + "content": ( + persist_user_message + if persist_user_message is not None + else message_text + ), + "timestamp": ( + persist_user_timestamp + if persist_user_timestamp is not None + else time.time() + ), + } + if persist_user_display_kind: + _user_entry["display_kind"] = persist_user_display_kind + if getattr(event, "message_id", None): + _user_entry["message_id"] = str(event.message_id) + await self.async_session_store.append_to_transcript( + session_entry.session_id, + _user_entry, + ) + except Exception: + logger.debug("Failed to persist inbound user message after agent exception", exc_info=True) + # Log full details server-side only; never expose raw exception + # types or messages to end users (info-leakage risk). + status_hint = "" + status_code = getattr(e, "status_code", None) + _hist_len = len(history) + if status_code == 401: + status_hint = " Check your API key or run `claude /login` to refresh OAuth credentials." + elif status_code == 402: + status_hint = " Your API balance or quota is exhausted. Check your provider dashboard." + elif status_code == 429: + # Check if this is a plan usage limit (resets on a schedule) vs a transient rate limit + _err_body = getattr(e, "response", None) + _err_json = {} + try: + if _err_body is not None: + _err_json = _err_body.json().get("error", {}) + if not isinstance(_err_json, dict): + _err_json = {} + except Exception: + pass + if _err_json.get("type") == "usage_limit_reached": + _resets_in = _err_json.get("resets_in_seconds") + if _resets_in and _resets_in > 0: + import math + _hours = math.ceil(_resets_in / 3600) + status_hint = f" Your plan's usage limit has been reached. It resets in ~{_hours}h." + else: + status_hint = " Your plan's usage limit has been reached. Please wait until it resets." + else: + status_hint = " You are being rate-limited. Please wait a moment and try again." + elif status_code == 529: + status_hint = " The API is temporarily overloaded. Please try again shortly." + elif status_code in {400, 500}: + # 400 on a large session is context overflow; 500 on a large session often means the + # payload is too large for the API — treat it the same way. + if _hist_len > 50: + return ( + "⚠️ Session too large for the model's context window.\n" + "Use /compact to compress the conversation, or " + "/reset to start fresh." + ) + elif status_code == 400: + status_hint = " The request was rejected by the API." + return ( + f"Sorry, I encountered an unexpected error.{status_hint}\n" + "Try again or use /reset to start a fresh session." + ) + + async def _handle_message_with_agent(self, event, source, _quick_key: str, run_generation: int): + """Inner handler that runs under the _running_agents sentinel guard.""" + from gateway.run import _load_gateway_config + _msg_start_time = time.time() + _platform_name = source.platform.value if hasattr(source.platform, "value") else str(source.platform) + _msg_preview = (event.text or "")[:80].replace("\n", " ") + _reply_id = getattr(event, "reply_to_message_id", None) + _reply_txt = (getattr(event, "reply_to_text", None) or "")[:80].replace("\n", " ") + logger.info( + "inbound message: platform=%s user=%s chat=%s msg=%r reply_to_id=%s reply_to_text=%r", + _platform_name, source.user_name or source.user_id or "unknown", + source.chat_id or "unknown", _msg_preview, _reply_id, _reply_txt, + ) + + resolved = await self._hmwa_resolve_session(event, source) + if resolved is None: + return + source, session_entry, session_key = resolved + _was_auto_reset, _is_new_session = await self._hmwa_open_session( + session_entry, session_key, source + ) + + # Build session context + context = build_session_context(source, self.config, session_entry) + + # Set session context variables for tools (task-local, concurrency-safe) + _session_env_tokens = self._set_session_env(context) + + # Read privacy.redact_pii from config (re-read per message) + _redact_pii = False + # Synthetic self-injected turns (batch completions, watch notifications, resume wake-ups) + # arrive as MessageEvent(internal=True). Persist with display_kind="internal_notification" + # so UIs render timeline notices, not user bubbles. display_kind is a DB-only sidecar + # stripped from every provider-bound payload; role/content untouched. + persist_user_display_kind = ( + "internal_notification" if getattr(event, "internal", False) else None + ) + try: + _pcfg = _load_gateway_config() + _redact_pii = bool((_pcfg.get("privacy") or {}).get("redact_pii", False)) + except Exception: + pass + + # Build the context prompt. The render is pinned per session, keyed by a hash of the exact + # renderer inputs (_ephemeral_change_key): a hit reuses the pinned bytes so the system prompt + # cannot drift turn-over-turn; a miss (thread rename, /sethome, redact_pii flip) re-renders. + context_prompt = self._pinned_session_context_prompt( + context, _redact_pii, session_key + ) + + # Per-turn must-deliver notes ride the user message via the api_content sidecar (staged + # below, consumed in run_sync → build_turn_context), NOT context_prompt: appending them to + # the ephemeral system prompt guaranteed a turn1→turn2 diff and a full agent rebuild. + turn_sidecar_notes: List[str] = [] + + # If the previous session expired and was auto-reset, deliver a notice + # so the agent knows this is a fresh conversation (not an intentional /reset). + if _was_auto_reset: + await self._hmwa_deliver_auto_reset_notice(session_entry, source, turn_sidecar_notes) + + # Auto-load skill(s) for topic/channel bindings (Telegram DM Topics, Discord + # channel_skill_bindings). Supports a single name or ordered list. Only inject on NEW + # sessions — ongoing conversations already carry the skill content in their history. + _auto = getattr(event, "auto_skill", None) + if _is_new_session and _auto: + self._hmwa_auto_load_skills(event, _auto, _quick_key, session_key) + + # Turn lease: session resolution is FINAL here (see _hmwa_acquire_turn_lease). + await self._hmwa_acquire_turn_lease( + _quick_key, run_generation, session_entry, _session_env_tokens + ) + + # A turn only becomes durable recovery work after it owns (or has explicitly degraded past) + # the per-session lease. Marking before the await above would falsely recover an alias- + # routed message that never began processing if the gateway died while it was still waiting. + await self._mark_durable_active_turn(event, session_entry.session_key) + + # Load conversation history from transcript. An unreadable canonical store is not an empty + # conversation: stop before the agent can invent continuity from a plausible-looking []. + # This return happens before the broad cleanup finally below, so restore task-local context + # here; the outer dispatch still clears the durable marker and turn lease. + try: + history = await self.async_session_store.load_transcript( + session_entry.session_id + ) + except TranscriptReadError: + self._clear_session_env(_session_env_tokens) + return ( + "⚠️ This session's history is temporarily unavailable, so " + "this message was not processed. Ask the operator to inspect " + "state.db, then resend after it is healthy. Use /reset only " + "if you intentionally want to start a new conversation." + ) + + # Session hygiene: auto-compress pathologically large transcripts before the agent starts. + history = await self._hmwa_run_session_hygiene( + event, source, session_entry, session_key, history, _quick_key, run_generation, + ) + + await self._hmwa_first_contact_notes(source, history, turn_sidecar_notes) + + # Voice channel awareness: deliver voice channel state (who is present / speaking) on the + # user message, ONLY when changed since the previous turn. It differs almost every turn, and + # in the ephemeral system prompt it forced a full agent rebuild + prompt-cache re-key per + # message; the system prompt carries a static pointer line instead (gateway/session.py). + _vc_note = self._voice_channel_sidecar_note(event, source, session_key) + if _vc_note: + turn_sidecar_notes.append(_vc_note) + + # Auto-analyze user images: run the vision tool eagerly so the model always gets a text + # description plus the local path for re-examination via vision_analyze. Filter to image + # media_type so documents/audio in the same message are not sent to the vision tool. + message_text = await self._prepare_profile_scoped_inbound_message_text( + event=event, + source=source, + history=history, + session_key=session_key, + ) + if message_text is None: + return + + # Capture the platform event time as message metadata and keep the persisted transcript + # clean; only the in-context render is gated behind gateway.message_timestamps.enabled. + message_text, persist_user_message, persist_user_timestamp = ( + self._hmwa_apply_message_timestamp(event, message_text) + ) + + # Stage this turn's must-deliver notes (one-shot; consumed in run_sync) AFTER the + # message_text early-out so an aborted turn cannot leak its notes into the next turn. + if turn_sidecar_notes and session_key: + self._set_pending_turn_sidecar_notes(session_key, turn_sidecar_notes) + + # Bind this run generation to the adapter's active-session event so deferred post-delivery + # callbacks can be released by the same run that registered them. + self._bind_adapter_run_generation( + self._adapter_for_source(source), + session_key, + run_generation, + ) + + try: + # Emit agent:start hook + hook_ctx = { + "platform": source.platform.value if source.platform else "", + "user_id": source.user_id, + "chat_id": source.chat_id or "", + "thread_id": str(getattr(source, "thread_id", None)) if getattr(source, "thread_id", None) else "", + "chat_type": getattr(source, "chat_type", "") or "", + "session_id": session_entry.session_id, + "message": message_text[:500], + } + await self.hooks.emit("agent:start", hook_ctx) + + # Run the agent. Capture the session id that this run was launched against so post-run + # compression publication can be identity-guarded below; a /new or another lifecycle + # transition may move session_entry.session_id while the old run is still unwinding. + _run_start_session_id = session_entry.session_id + _turn_started_monotonic = time.monotonic() + agent_result = await self._run_agent( + message=message_text, + context_prompt=context_prompt, + history=history, + source=source, + session_id=_run_start_session_id, + session_key=session_key, + run_generation=run_generation, + event_message_id=self._reply_anchor_for_event(event), + inbound_message_id=( + str(event.message_id) if event.message_id else None + ), + channel_prompt=event.channel_prompt, + moa_config=getattr(event, "_moa_config", None), + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + message_type=event.message_type, + ) + _turn_seconds = time.monotonic() - _turn_started_monotonic + + await self._hmwa_stop_typing_for_turn(event, source) + + if not self._is_session_run_current(_quick_key, run_generation): + logger.info( + "Discarding stale agent result for %s — generation %d is no longer current", + _quick_key or "?", + run_generation, + ) + _stale_adapter = self._adapter_for_source(source) + if getattr(type(_stale_adapter), "pop_post_delivery_callback", None) is not None: + _stale_adapter.pop_post_delivery_callback( + _quick_key, + generation=run_generation, + ) + elif _stale_adapter and hasattr(_stale_adapter, "_post_delivery_callbacks"): + _stale_adapter._post_delivery_callbacks.pop(_quick_key, None) + return None + + response, _intentional_silence, agent_messages = await self._hmwa_shape_agent_response( + agent_result, source, history, session_entry, session_key, + _quick_key, run_generation, _run_start_session_id, _platform_name, _msg_start_time, + ) + response = self._hmwa_prepend_reasoning(agent_result, response, source, _intentional_silence) + _footer_line = self._hmwa_runtime_footer_line(agent_result, source, _turn_seconds) + if _footer_line and response and not agent_result.get("already_sent") and not _intentional_silence: + response = f"{response}\n\n{_footer_line}" + await self._hmwa_post_turn_hooks(hook_ctx, agent_result, response) + + agent_failed_early, hidden_reasoning_incomplete, is_context_overflow_failure = ( + self._hmwa_classify_turn_failure(agent_result, history, session_entry) + ) + response, session_entry = await self._hmwa_compression_exhaustion_reset( + agent_result, response, session_entry, session_key, source, + ) + await self._hmwa_persist_turn_transcript( + event=event, + source=source, + session_entry=session_entry, + session_key=session_key, + agent_result=agent_result, + agent_messages=agent_messages, + history=history, + response=response, + message_text=message_text, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + agent_failed_early=agent_failed_early, + hidden_reasoning_incomplete=hidden_reasoning_incomplete, + is_context_overflow_failure=is_context_overflow_failure, + ) + return await self._hmwa_deliver_turn_response( + event, source, session_entry, session_key, run_generation, + agent_result, agent_messages, response, _footer_line, _intentional_silence, + ) + + except Exception as e: + return await self._hmwa_agent_error_reply( + e, event, source, session_entry, session_key, history, message_text, + persist_user_message, persist_user_timestamp, persist_user_display_kind, + ) + finally: + # Restore session context variables to their pre-handler state + self._clear_session_env(_session_env_tokens) + + def _reset_notice_session_info(self, source: SessionSource) -> str: + """Session-info block for the auto-reset notice, profile-scoped. + + Under multiplexing, resolve model/provider/context inside the profile serving ``source`` + (mirrors ``_run_agent``'s gating) or the banner advertises the base config's model. Call + via ``asyncio.to_thread``: resolution can block (credential refresh, context-length + probes), and the scope is entered here so contextvars behave in the worker thread. + """ + from gateway.run import _profile_runtime_scope + if getattr(getattr(self, "config", None), "multiplex_profiles", False): + with _profile_runtime_scope(self._resolve_profile_home_for_source(source)): + return self._format_session_info() + return self._format_session_info() + + def _format_session_info(self) -> str: + """Resolve current model config and return a formatted info block. + + Surfaces model, provider, context length, and endpoint so gateway users can immediately + see if context detection went wrong (e.g. local models falling to the 128K default). + """ + from gateway.run import _resolve_gateway_model_context + resolved = _resolve_gateway_model_context() + model = resolved.model + provider = resolved.provider + base_url = resolved.base_url + context_length = resolved.context_length + + # Format context source hint + if resolved.context_source == "config": + ctx_source = "config" + elif resolved.context_source == "default": + ctx_source = "default — set model.context_length in config to override" + else: + ctx_source = "detected" + + # Format context length for display + if context_length >= 1_000_000: + ctx_display = f"{context_length / 1_000_000:.1f}M" + elif context_length >= 1_000: + ctx_display = f"{context_length // 1_000}K" + else: + ctx_display = str(context_length) + + lines = [ + f"◆ Model: `{model}`", + f"◆ Provider: {provider or 'openrouter'}", + f"◆ Context: {ctx_display} tokens ({ctx_source})", + ] + + # Show endpoint for local/custom setups + if base_url and base_url_hostname(base_url) in ("localhost", "127.0.0.1", "0.0.0.0"): + lines.append(f"◆ Endpoint: {base_url}") + + return "\n".join(lines) + + async def _run_background_task( + self, + prompt: str, + source: "SessionSource", + task_id: str, + event_message_id: Optional[str] = None, + media_urls: Optional[List[str]] = None, + media_types: Optional[List[str]] = None, + ) -> None: + """Profile-scoping wrapper around the background agent task. + + When multiplexing is active, resolve the inbound source's profile and run the whole task + inside ``_profile_runtime_scope`` so credentials resolve from that profile's secret + scope. Mirrors the pattern in ``_run_agent``. + """ + from gateway.run import _profile_runtime_scope + if not getattr(getattr(self, "config", None), "multiplex_profiles", False): + return await self._run_background_task_inner( + prompt, source, task_id, event_message_id, media_urls, media_types, + ) + + profile_home = self._resolve_profile_home_for_source(source) + with _profile_runtime_scope(profile_home): + return await self._run_background_task_inner( + prompt, source, task_id, event_message_id, media_urls, media_types, + ) + + def _resolve_enabled_toolsets_for_source( + self, + user_config: dict, + source: "SessionSource", + platform_key: str, + ) -> list: + """Resolve enabled toolsets for an agent run, honoring per-source overrides. + + An adapter ``toolsets_for_source()`` override (e.g. per-route webhook toolsets) is + validated through the SAME ``_get_platform_tools`` path as normal platform config, so + unknown and platform-restricted toolsets are dropped rather than trusted. Absent an + override, falls back to ``platform_toolsets.``. + """ + from hermes_cli.tools_config import _get_platform_tools + + override = None + try: + adapter = self._adapter_for_source(source) + if adapter is not None: + override = adapter.toolsets_for_source(source) + except Exception: + override = None + + if override and isinstance(override, list): + cfg = dict(user_config) + pts = dict(cfg.get("platform_toolsets") or {}) + pts[platform_key] = [str(t) for t in override] + cfg["platform_toolsets"] = pts + return sorted(_get_platform_tools(cfg, platform_key)) + + return sorted(_get_platform_tools(user_config, platform_key)) + + async def _run_background_task_inner( + self, + prompt: str, + source: "SessionSource", + task_id: str, + event_message_id: Optional[str] = None, + media_urls: Optional[List[str]] = None, + media_types: Optional[List[str]] = None, + ) -> None: + """Execute a background agent task and deliver the result to the chat.""" + from gateway.run import ( + _checkpoint_agent_kwargs, + _current_max_iterations, + _load_gateway_config, + _platform_config_key, + ) + from run_agent import AIAgent + + media_urls = media_urls or [] + media_types = media_types or [] + + adapter = self._adapter_for_source(source) + if not adapter: + logger.warning("No adapter for platform %s in background task %s", source.platform, task_id) + return + + _thread_metadata = self._thread_metadata_for_source(source, event_message_id) + + try: + user_config = _load_gateway_config() + model, runtime_kwargs = self._resolve_session_agent_runtime( + source=source, + user_config=user_config, + ) + if not runtime_kwargs.get("api_key"): + await adapter.send( + source.chat_id, + f"❌ Background task {task_id} failed: no provider credentials configured.", + metadata=_thread_metadata, + ) + return + + platform_key = _platform_config_key(source.platform) + + enabled_toolsets = self._resolve_enabled_toolsets_for_source( + user_config, source, platform_key + ) + agent_cfg = user_config.get("agent") or {} + from agent.skill_utils import parse_config_string_list + + disabled_toolsets = parse_config_string_list(agent_cfg.get("disabled_toolsets")) or None + + pr = self._provider_routing + max_iterations = _current_max_iterations() + reasoning_config = self._resolve_session_reasoning_config( + source=source, model=model + ) + self._reasoning_config = reasoning_config + self._service_tier = self._resolve_session_service_tier(source=source) + turn_route = self._resolve_turn_agent_config(prompt, model, runtime_kwargs) + + # Enrich the prompt with image descriptions so the background + # agent can see user-attached images (same as the main flow). + enriched_prompt = prompt + if media_urls: + image_paths = [] + for i, path in enumerate(media_urls): + mtype = media_types[i] if i < len(media_types) else "" + if mtype.startswith("image/"): + image_paths.append(path) + if image_paths: + try: + enriched_prompt = await self._enrich_message_with_vision( + prompt, image_paths, + ) + except Exception as e: + logger.warning("Background task vision enrichment failed: %s", e) + + def run_sync(): + agent = AIAgent( + model=turn_route["model"], + **turn_route["runtime"], + **_checkpoint_agent_kwargs(user_config), + max_iterations=max_iterations, + quiet_mode=True, + verbose_logging=False, + enabled_toolsets=enabled_toolsets, + disabled_toolsets=disabled_toolsets, + reasoning_config=reasoning_config, + service_tier=self._service_tier, + request_overrides=turn_route.get("request_overrides"), + providers_allowed=pr.get("only"), + providers_ignored=pr.get("ignore"), + providers_order=pr.get("order"), + provider_sort=pr.get("sort"), + provider_require_parameters=pr.get("require_parameters", False), + provider_data_collection=pr.get("data_collection"), + session_id=task_id, + platform=platform_key, + user_id=source.user_id, + user_id_alt=source.user_id_alt, + user_name=source.user_name, + chat_id=source.chat_id, + chat_name=source.chat_name, + chat_type=source.chat_type, + thread_id=source.thread_id, + session_db=getattr(self._session_db, "_db", self._session_db), + # Reload from disk — do not reuse the startup snapshot (#60955). + fallback_model=self._refresh_fallback_model(), + ) + try: + return agent.run_conversation( + user_message=enriched_prompt, + task_id=task_id, + ) + finally: + self._cleanup_agent_resources(agent) + + result = await self._run_in_executor_with_context(run_sync) + + response = result.get("final_response", "") if result else "" + if not response and result and result.get("error"): + response = f"Error: {result['error']}" + + # Background tasks start a fresh conversation, so history_offset=0: every message in the + # run belongs to this turn. Mirrors the repair on the main turn path. + if response: + response = repair_explicit_computer_use_media_paths( + response, + result.get("messages", []), + ) + + # Extract media files from the response + if response: + media_files, response = adapter.extract_media(response) + from gateway.platforms.base import BasePlatformAdapter + media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files) + images, text_content = adapter.extract_images(response) + + preview = prompt[:60] + ("..." if len(prompt) > 60 else "") + header = f'✅ Background task complete\nPrompt: "{preview}"\n\n' + + if text_content: + await adapter.send( + chat_id=source.chat_id, + content=header + text_content, + metadata=_thread_metadata, + ) + elif not images and not media_files: + await adapter.send( + chat_id=source.chat_id, + content=header + "(No response generated)", + metadata=_thread_metadata, + ) + + # Send extracted images + for image_url, alt_text in (images or []): + with suppress(Exception): + await adapter.send_image( + chat_id=source.chat_id, + image_url=image_url, + caption=alt_text, + metadata=_thread_metadata, + ) + + # Route each media file by type so a TTS clip arrives as a voice bubble and a clip + # as a video rather than a generic document. Mirrors the streaming + kanban paths. + from gateway.platforms.base import ( + should_send_media_as_audio as _should_send_media_as_audio, + ) + _IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp"} + _VIDEO_EXTS = {".mp4", ".mov", ".avi", ".mkv", ".webm", ".3gp"} + for media_path, _is_voice in (media_files or []): + _ext = os.path.splitext(media_path)[1].lower() + try: + if _should_send_media_as_audio(source.platform, _ext, _is_voice): + await adapter.send_voice( + chat_id=source.chat_id, + audio_path=media_path, + metadata=_thread_metadata, + is_voice=_is_voice, + ) + elif _ext in _VIDEO_EXTS: + await adapter.send_video( + chat_id=source.chat_id, + video_path=media_path, + metadata=_thread_metadata, + ) + elif _ext in _IMAGE_EXTS: + await adapter.send_image_file( + chat_id=source.chat_id, + image_path=media_path, + metadata=_thread_metadata, + ) + else: + await adapter.send_document( + chat_id=source.chat_id, + file_path=media_path, + metadata=_thread_metadata, + ) + except Exception: + pass + else: + preview = prompt[:60] + ("..." if len(prompt) > 60 else "") + await adapter.send( + chat_id=source.chat_id, + content=f'✅ Background task complete\nPrompt: "{preview}"\n\n(No response generated)', + metadata=_thread_metadata, + ) + + except Exception as e: + logger.exception("Background task %s failed", task_id) + with suppress(Exception): + await adapter.send( + chat_id=source.chat_id, + content=f"❌ Background task {task_id} failed: {e}", + metadata=_thread_metadata, + ) + + async def _execute_mcp_reload(self, event: MessageEvent) -> str: + """Actually disconnect, reconnect, and notify MCP tool changes. + + Split out so the confirmation wrapper can invoke the same path for button, text reply, + or disabled confirm gate. Under multiplex the reload runs inside the requesting profile's + runtime scope (entered here when the caller did not) and only that profile's servers are + torn down and rediscovered. + """ + from gateway.run import _profile_runtime_scope + multiplex = bool(getattr(self.config, "multiplex_profiles", False)) + if multiplex and not get_hermes_home_override(): + profile_home = self._resolve_profile_home_for_source(event.source) + with _profile_runtime_scope(Path(profile_home)): + return await self._execute_mcp_reload(event) + try: + from tools.mcp_tool import shutdown_mcp_servers, discover_mcp_tools, _servers, _lock + from tools.mcp_tool import _server_scope_keys, reprobe_tool_availability + from tools.registry import registry + + reload_scope = registry.current_scope_key() if multiplex else None + + def _scoped_server_names() -> set: + with _lock: + return { + name for name in _servers + if reload_scope is None or _server_scope_keys.get(name) == reload_scope + } + + # Capture old server names before shutdown + old_servers = _scoped_server_names() + + # Read new config before shutting down, so we know what will be added/removed + # Shutdown existing connections + await self._run_in_executor_with_context( + lambda: shutdown_mcp_servers(scope=reload_scope) + ) + # Explicit reload also re-probes tool availability (check_fn). + reprobe_tool_availability() + + # Reconnect by discovering tools (reads config.yaml fresh) + new_tools = await self._run_in_executor_with_context(discover_mcp_tools) + + # Compute what changed + connected_servers = _scoped_server_names() + if reload_scope is not None: + from tools.mcp_tool import _mcp_tool_server_names + + with _lock: + new_tools = [ + n for n in new_tools + if _mcp_tool_server_names.get(n) in connected_servers + ] + + added = connected_servers - old_servers + removed = old_servers - connected_servers + reconnected = connected_servers & old_servers + + lines = [t("gateway.reload_mcp.header")] + if reconnected: + lines.append(t("gateway.reload_mcp.reconnected", names=", ".join(sorted(reconnected)))) + if added: + lines.append(t("gateway.reload_mcp.added", names=", ".join(sorted(added)))) + if removed: + lines.append(t("gateway.reload_mcp.removed", names=", ".join(sorted(removed)))) + if not connected_servers: + lines.append(t("gateway.reload_mcp.none_connected")) + else: + lines.append(t("gateway.reload_mcp.tools_available", tools=len(new_tools), servers=len(connected_servers))) + + # Refresh cached agents so existing sessions see new MCP tools on their next turn — + # without this, the user has to `/new` (which discards conversation history) to pick up + # tools from a server that was just added or reconnected. + try: + from tools.mcp_tool import refresh_agent_mcp_tools + _cache = getattr(self, "_agent_cache", None) + _cache_lock = getattr(self, "_agent_cache_lock", None) + if _cache_lock is not None and _cache: + # Multiplex: only this profile's sessions; rebuilding another profile's agent in + # this scope would hand it this profile's tool registry. + _ns_prefix = ( + _session_key_namespace(event.source.profile) + ":" + if multiplex else None + ) + with _cache_lock: + for _sess_key, _entry in list(_cache.items()): + if _ns_prefix and not str(_sess_key).startswith(_ns_prefix): + continue + try: + _agent = _entry[0] if isinstance(_entry, tuple) else _entry + except Exception: + continue + if _agent is None: + continue + # Preserve each cached agent's build-time toolset selection EXACTLY: a + # gateway session built with a restricted enabled_toolsets (e.g. + # ["safe"]) must NOT silently gain tools after a reload. Unlike the + # CLI/TUI /reload-mcp (one user re-applying their own config), gateway + # agents are per-session and may be deliberately locked down. + refresh_agent_mcp_tools(_agent, quiet_mode=True) + except Exception as _exc: + logger.debug( + "Failed to update cached agent tools after MCP reload: %s", + _exc, + ) + + # Inject a message at the END of the session history so the model knows tools changed + # next turn; appending after all existing messages preserves the prompt-cache prefix. + change_parts = [] + if added: + change_parts.append(f"Added servers: {', '.join(sorted(added))}") + if removed: + change_parts.append(f"Removed servers: {', '.join(sorted(removed))}") + if reconnected: + change_parts.append(f"Reconnected servers: {', '.join(sorted(reconnected))}") + tool_summary = f"{len(new_tools)} MCP tool(s) now available" if new_tools else "No MCP tools available" + change_detail = ". ".join(change_parts) + ". " if change_parts else "" + reload_msg = { + "role": "user", + "content": f"[IMPORTANT: MCP servers have been reloaded. {change_detail}{tool_summary}. The tool list for this conversation has been updated accordingly.]", + } + try: + session_entry = await self.async_session_store.get_or_create_session(event.source) + await self.async_session_store.append_to_transcript( + session_entry.session_id, reload_msg + ) + except Exception: + pass # Best-effort; don't fail the reload over a transcript write + + return "\n".join(lines) + + except Exception as e: + logger.warning("MCP reload failed: %s", e) + return t("gateway.reload_mcp.failed", error=e) + + def _get_proxy_url(self) -> Optional[str]: + """Return the proxy URL if proxy mode is configured, else None. + + GATEWAY_PROXY_URL env var (Docker-friendly) wins over ``gateway.proxy_url`` in config.yaml. + """ + from gateway.run import _load_gateway_config + url = os.getenv("GATEWAY_PROXY_URL", "").strip() + if url: + return url.rstrip("/") + cfg = _load_gateway_config() + url = (cfg.get("gateway") or {}).get("proxy_url") + url = (url or "").strip() + if url: + return url.rstrip("/") + return None + + def _build_stream_consumer_config( + self, + source: "SessionSource", + scfg: Any, + adapter: Any, + *, + on_missing_cursor: str, + ) -> "tuple[Any, Optional[Callable[[], None]]]": + """Build the shared ``StreamConsumerConfig`` and optional Telegram pause-typing closure. + + ``on_missing_cursor`` handles adapters with ``SUPPORTS_MESSAGE_EDITING = False``: + ``"fallback"`` (proxy path) streams with an empty cursor; ``"raise"`` (in-process path) + raises ``RuntimeError`` so the caller's ``except`` skips streaming entirely. Returns + ``(consumer_cfg, pause_typing_before_finalize)``. + """ + from gateway.stream_consumer import StreamConsumerConfig + + _pause_typing_before_finalize = None + if source.platform == Platform.TELEGRAM and hasattr(adapter, "pause_typing_for_chat"): + def _pause_typing_before_finalize( + _adapter=adapter, + _chat_id=source.chat_id, + ) -> None: + _adapter.pause_typing_for_chat(_chat_id) + # Platforms that can't edit sent messages (e.g. QQ, WeChat) skip streaming entirely: the + # partial first message could never be updated, yielding duplicates (partial + final). + _adapter_supports_edit = getattr(adapter, "SUPPORTS_MESSAGE_EDITING", True) + # Adapters that can't edit but have a native-streaming transport (e.g. WeCom msgtype "stream" + # via send_stream_frame) pass the gate — the consumer's native branch delivers the full turn. + _adapter_supports_native_stream = bool(getattr( + adapter, "SUPPORTS_NATIVE_STREAMING", False, + )) + if ( + not _adapter_supports_edit + and not _adapter_supports_native_stream + and on_missing_cursor == "raise" + ): + raise RuntimeError("skip streaming for non-editable platform") + _effective_cursor = scfg.cursor if _adapter_supports_edit else "" + # Some Matrix clients render the streaming cursor as a visible tofu/white-box artifact: keep + # streaming text on Matrix, but suppress the cursor. + _buffer_only = False + if source.platform == Platform.MATRIX: + _effective_cursor = "" + _buffer_only = True + # Fresh-final applies to Telegram only — other platforms edit in place cheaply (Discord, Slack) + # or lack the edit-timestamp-stays-stale problem. + _fresh_final_secs = ( + float(getattr(scfg, "fresh_final_after_seconds", 0.0) or 0.0) + if source.platform == Platform.TELEGRAM + else 0.0 + ) + _consumer_cfg = StreamConsumerConfig( + edit_interval=scfg.edit_interval, + buffer_threshold=scfg.buffer_threshold, + cursor=_effective_cursor, + buffer_only=_buffer_only, + fresh_final_after_seconds=_fresh_final_secs, + transport=scfg.transport or "edit", + chat_type=getattr(source, "chat_type", "") or "", + ) + return _consumer_cfg, _pause_typing_before_finalize + + async def _run_agent_via_proxy( + self, + message: str, + context_prompt: str, + history: List[Dict[str, Any]], + source: "SessionSource", + session_id: str, + session_key: str = None, + run_generation: Optional[int] = None, + event_message_id: Optional[str] = None, + ) -> Dict[str, Any]: + """Forward the message to a remote Hermes API server instead of running a local AIAgent. + + This lets a Docker container handle Matrix E2EE while the actual agent runs on the host + with full access to local files, memory, skills, and a unified session store. + """ + from gateway.run import ( + _GATEWAY_PROXY_SSE_BUFFER_MAX_CHARS, + _load_gateway_config, + _platform_config_key, + ) + try: + from aiohttp import ClientSession as _AioClientSession, ClientTimeout + except ImportError: + return { + "final_response": "⚠️ Proxy mode requires aiohttp. Install with: pip install aiohttp", + "messages": [], + "api_calls": 0, + "tools": [], + } + + proxy_url = self._get_proxy_url() + if not proxy_url: + return { + "final_response": "⚠️ Proxy URL not configured (GATEWAY_PROXY_URL or gateway.proxy_url)", + "messages": [], + "api_calls": 0, + "tools": [], + } + + # Scope-aware read: the proxy key is a per-profile credential; under multiplex honor the + # installed scope's verdict (Slack pattern for the unscoped default-profile loop). + try: + from agent.secret_scope import UnscopedSecretError, get_secret + + try: + proxy_key = (get_secret("GATEWAY_PROXY_KEY") or "").strip() + except UnscopedSecretError: + proxy_key = os.getenv("GATEWAY_PROXY_KEY", "").strip() + except Exception: + proxy_key = os.getenv("GATEWAY_PROXY_KEY", "").strip() + + def _run_still_current() -> bool: + if run_generation is None or not session_key: + return True + return self._is_session_run_current(session_key, run_generation) + + # Build messages in OpenAI chat format. The remote api_server keeps continuity via + # X-Hermes-Session-Id and loads its own history, so send only the current message; if the + # remote has no history yet, include a compact text-only local history (remote replays tools). + api_messages: List[Dict[str, str]] = [] + + if context_prompt: + api_messages.append({"role": "system", "content": context_prompt}) + + for msg in history: + role = msg.get("role") + content = msg.get("content") + if role in {"user", "assistant"} and content: + api_messages.append({"role": role, "content": content}) + + api_messages.append({"role": "user", "content": message}) + + # HTTP headers --------------------------------------------------- + headers: Dict[str, str] = {"Content-Type": "application/json"} + if proxy_key: + headers["Authorization"] = f"Bearer {proxy_key}" + if session_id: + headers["X-Hermes-Session-Id"] = session_id + + body = { + "model": "hermes-agent", + "messages": api_messages, + "stream": True, + } + + # Set up platform streaming if available ------------------------- + _stream_consumer = None + _scfg = getattr(getattr(self, "config", None), "streaming", None) + if _scfg is None: + from gateway.config import StreamingConfig + _scfg = StreamingConfig() + + platform_key = _platform_config_key(source.platform) + user_config = _load_gateway_config() + from gateway.display_config import resolve_display_setting + _plat_streaming = resolve_display_setting( + user_config, platform_key, "streaming" + ) + _streaming_enabled = ( + _scfg.enabled and _scfg.transport != "off" + if _plat_streaming is None + else bool(_plat_streaming) + ) + + _thread_metadata: Optional[Dict[str, Any]] = self._thread_metadata_for_source(source, event_message_id) + + if _streaming_enabled: + try: + from gateway.stream_consumer import GatewayStreamConsumer + _adapter = self._adapter_for_source(source) + if _adapter: + _consumer_cfg, _pause_typing_before_finalize = ( + self._build_stream_consumer_config( + source, _scfg, _adapter, + on_missing_cursor="fallback", + ) + ) + _stream_consumer = GatewayStreamConsumer( + adapter=_adapter, + chat_id=source.chat_id, + config=_consumer_cfg, + metadata=_thread_metadata, + on_before_finalize=_pause_typing_before_finalize, + initial_reply_to_id=event_message_id, + run_still_current=_run_still_current, + ) + except Exception as _sc_err: + logger.debug("Proxy: could not set up stream consumer: %s", _sc_err) + + # Run the stream consumer task in the background + stream_task = None + if _stream_consumer: + stream_task = asyncio.create_task(_stream_consumer.run()) + + # Send typing indicator + _adapter = self._adapter_for_source(source) + if _adapter: + with suppress(Exception): + await _adapter.send_typing(source.chat_id, metadata=_thread_metadata) + + # Make the HTTP request with SSE streaming ----------------------- + full_response = "" + _start = time.time() + + try: + _timeout = ClientTimeout(total=0, sock_read=1800) + async with _AioClientSession(timeout=_timeout) as session: + async with session.post( + f"{proxy_url}/v1/chat/completions", + json=body, + headers=headers, + ) as resp: + if resp.status != 200: + error_text = await resp.text() + logger.warning( + "Proxy error (%d) from %s: %s", + resp.status, proxy_url, error_text[:500], + ) + return { + "final_response": f"⚠️ Proxy error ({resp.status}): {error_text[:300]}", + "messages": [], + "api_calls": 0, + "tools": [], + } + + # Parse SSE stream + buffer = "" + async for chunk in resp.content.iter_any(): + if not _run_still_current(): + logger.info( + "Discarding stale proxy stream for %s — generation %d is no longer current", + session_key or "?", + run_generation or 0, + ) + return { + "final_response": "", + "messages": [], + "api_calls": 0, + "tools": [], + "history_offset": len(history), + "session_id": session_id, + "response_previewed": False, + } + text = chunk.decode("utf-8", errors="replace") + buffer += text + + # Process complete SSE lines + while "\n" in buffer: + line, buffer = buffer.split("\n", 1) + line = line.strip() + if not line: + continue + if line.startswith("data: "): + data = line[6:] + if data.strip() == "[DONE]": + break + try: + obj = json.loads(data) + choices = obj.get("choices", []) + if choices: + delta = choices[0].get("delta", {}) + content = delta.get("content", "") + if content: + full_response += content + if _stream_consumer: + _stream_consumer.on_delta(content) + except json.JSONDecodeError: + pass + if len(buffer) > _GATEWAY_PROXY_SSE_BUFFER_MAX_CHARS: + raise ValueError( + "Proxy SSE stream exceeded max buffer size without a line boundary" + ) + + except asyncio.CancelledError: + raise + except Exception as e: + logger.error("Proxy connection error to %s: %s", proxy_url, e) + if not full_response: + return { + "final_response": f"⚠️ Proxy connection error: {e}", + "messages": [], + "api_calls": 0, + "tools": [], + } + # Partial response — return what we got + finally: + # Finalize stream consumer + if _stream_consumer: + _stream_consumer.finish() + if stream_task: + try: + await asyncio.wait_for(stream_task, timeout=5.0) + except (asyncio.TimeoutError, asyncio.CancelledError): + stream_task.cancel() + + _elapsed = time.time() - _start + if not _run_still_current(): + logger.info( + "Discarding stale proxy result for %s — generation %d is no longer current", + session_key or "?", + run_generation or 0, + ) + return { + "final_response": "", + "messages": [], + "api_calls": 0, + "tools": [], + "history_offset": len(history), + "session_id": session_id, + "response_previewed": False, + } + logger.info( + "proxy response: url=%s session=%s time=%.1fs response=%d chars", + proxy_url, (session_id or "")[:20], _elapsed, len(full_response), + ) + + return { + "final_response": full_response or "(No response from remote agent)", + "messages": [ + {"role": "user", "content": message}, + {"role": "assistant", "content": full_response}, + ], + "api_calls": 1, + "tools": [], + "history_offset": len(history), + "session_id": session_id, + "response_previewed": _stream_consumer is not None and bool(full_response), + } + + async def _run_agent( + self, + message: str, + context_prompt: str, + history: List[Dict[str, Any]], + source: SessionSource, + session_id: str, + session_key: str = None, + run_generation: Optional[int] = None, + _interrupt_depth: int = 0, + event_message_id: Optional[str] = None, + inbound_message_id: Optional[str] = None, + channel_prompt: Optional[str] = None, + moa_config: Optional[dict] = None, + persist_user_message: Optional[Any] = None, + persist_user_timestamp: Optional[float] = None, + persist_user_display_kind: Optional[str] = None, + message_type: Optional[str] = None, + ) -> Dict[str, Any]: + """Profile-scoping wrapper around the agent run. + + Under multiplexing, run the turn inside ``_profile_runtime_scope`` so config/skills/memory + resolve to the source profile's home AND credentials come from its secret scope (never + process-global ``os.environ``). Transparent pass-through when multiplexing is off. + """ + from gateway.run import _profile_runtime_scope + if not getattr(getattr(self, "config", None), "multiplex_profiles", False): + return await self._run_agent_inner( + message, context_prompt, history, source, session_id, + session_key=session_key, run_generation=run_generation, + _interrupt_depth=_interrupt_depth, event_message_id=event_message_id, + inbound_message_id=inbound_message_id, + channel_prompt=channel_prompt, moa_config=moa_config, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + message_type=message_type, + ) + + profile_home = self._resolve_profile_home_for_source(source) + with _profile_runtime_scope(profile_home): + return await self._run_agent_inner( + message, context_prompt, history, source, session_id, + session_key=session_key, run_generation=run_generation, + _interrupt_depth=_interrupt_depth, event_message_id=event_message_id, + inbound_message_id=inbound_message_id, + channel_prompt=channel_prompt, moa_config=moa_config, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + message_type=message_type, + ) + + def _run_agent_display_settings(self, source: SessionSource) -> "GatewayRunner._RunAgentDisplay": + """Resolve per-platform display, progress, status and streaming-surface settings for a turn.""" + from gateway.run import ( + _gateway_platform_value, + _has_platform_display_override, + _load_gateway_config, + _platform_config_key, + ) + user_config = _load_gateway_config() + platform_key = _platform_config_key(source.platform) + + enabled_toolsets = self._resolve_enabled_toolsets_for_source( + user_config, source, platform_key + ) + agent_cfg_local = user_config.get("agent") or {} + from agent.skill_utils import parse_config_string_list + + disabled_toolsets = parse_config_string_list(agent_cfg_local.get("disabled_toolsets")) or None + + display_config = user_config.get("display", {}) + if not isinstance(display_config, dict): + display_config = {} + + # Per-platform display settings via display_config: display.platforms.., then + # display. global, then built-in platform defaults. + from gateway.display_config import resolve_display_setting + + # Apply tool preview length config (0 = no limit) + try: + from agent.display import set_tool_preview_max_len + _tpl = resolve_display_setting(user_config, platform_key, "tool_preview_length", 0) + set_tool_preview_max_len(int(_tpl) if _tpl else 0) + except Exception: + pass + + # Apply friendly tool labels config (default on) — per-platform aware + try: + from agent.display import set_friendly_tool_labels + _ftl = resolve_display_setting(user_config, platform_key, "friendly_tool_labels", True) + set_friendly_tool_labels(bool(_ftl)) + except Exception: + pass + + # Tool progress mode — resolved per-platform with env var fallback + _resolved_tp = resolve_display_setting(user_config, platform_key, "tool_progress") + _env_tp = os.getenv("HERMES_TOOL_PROGRESS_MODE") + _display_cfg = display_config if isinstance(display_config, dict) else {} + _platforms_cfg = _display_cfg.get("platforms") or {} + _platform_cfg = _platforms_cfg.get(platform_key) or {} + _legacy_tp_overrides = _display_cfg.get("tool_progress_overrides") or {} + _tool_progress_configured = ( + "tool_progress" in _display_cfg + or ( + isinstance(_platform_cfg, dict) + and "tool_progress" in _platform_cfg + ) + or ( + isinstance(_legacy_tp_overrides, dict) + and platform_key in _legacy_tp_overrides + ) + ) + progress_mode = ( + _env_tp + if _env_tp and not _tool_progress_configured + else (_resolved_tp or _env_tp or "all") + ) + # Tool progress grouping: "accumulate" (edit one bubble) or "separate" (one msg per tool) + progress_grouping = resolve_display_setting(user_config, platform_key, "tool_progress_grouping") or "accumulate" + from gateway.status_phrases import choose_status_phrase, resolve_status_phrase_catalog + _generic_status_recent: List[str] = [] + _generic_status_catalog = resolve_status_phrase_catalog(user_config, platform_key) + + def _display_surface_mode( + setting: str, + *, + default: bool = False, + require_platform_override_for: set[Any] | None = None, + allow_generic: bool = False, + ) -> str: + """Return off|raw|generic for a gateway visibility surface.""" + if require_platform_override_for: + current_platform = _gateway_platform_value(source.platform) + platform_only = { + _gateway_platform_value(item) + for item in require_platform_override_for + } + if ( + current_platform in platform_only + and not _has_platform_display_override(user_config, platform_key, setting) + ): + return "off" + value = resolve_display_setting(user_config, platform_key, setting, default) + if isinstance(value, str) and value.strip().lower() == "generic": + return "generic" if allow_generic else "off" + return "raw" if bool(value) else "off" + + def _generic_status_phrase(kind: str, *, tool_name: str | None = None, preview: str | None = None, args: Any = None) -> str: + try: + return choose_status_phrase( + kind, + tool_name=tool_name, + preview=preview, + args=args, + recent=_generic_status_recent, + catalog=_generic_status_catalog, + ) + except Exception as _phrase_err: + logger.debug("generic status phrase selection failed: %s", _phrase_err) + return "still on it" if kind in {"heartbeat", "waiting", "long_running", "status"} else "one sec" + # Disable tool progress for webhooks - they don't support message editing, + # so each progress line would be sent as a separate message. + from gateway.config import Platform + tool_progress_enabled = progress_mode not in {"off", "log"} and source.platform != Platform.WEBHOOK + # Live working-state status for text-rendering typing indicators (Slack's assistant status + # line). Independent of tool_progress (Slack defaults it off; the status line is ephemeral). + # Rides the existing _keep_typing refresh — the callback only stores a phrase, no extra calls. + _live_status_mode = resolve_display_setting( + user_config, platform_key, "live_status", "full" + ) + _live_status_adapter = self._adapter_for_source(source) + if not getattr(_live_status_adapter, "supports_status_text", False): + _live_status_adapter = None + if _live_status_mode == "off": + _live_status_adapter = None + # "log" mode: tool calls are written to ~/.hermes/logs/tool_calls.log + # instead of the chat (#3459 / #3458). Gateway-only by design. + log_mode_enabled = progress_mode == "log" and source.platform != Platform.WEBHOOK + log_queue: "queue.Queue | None" = queue.Queue() if log_mode_enabled else None + # Natural assistant status messages are independent from tool progress and token streaming: + # tool_progress can stay quiet while users opt into concise mid-turn updates. + interim_assistant_messages_mode = _display_surface_mode( + "interim_assistant_messages", + default=True, + require_platform_override_for={Platform.MATTERMOST}, + ) + interim_assistant_messages_enabled = ( + source.platform != Platform.WEBHOOK + and interim_assistant_messages_mode != "off" + ) + # thinking_progress is independent — if enabled, we need the progress queue even when + # tool_progress is off (thinking relay uses same infra). Mattermost requires a per-platform + # opt-in: global scratch-text display is too easy to leak into busy public threads. + _thinking_mode = _display_surface_mode( + "thinking_progress", + default=False, + require_platform_override_for={Platform.MATTERMOST}, + ) + _thinking_enabled = _thinking_mode != "off" + # Slack-native task cards: with the Slack adapter's opt-in, tool progress renders as native + # plan/task cards via chat.startStream, so the progress queue is needed even though Slack keeps + # text tool_progress off by default (requiring both flags would silently disable the feature). + _progress_adapter_for_native = self._adapter_for_source(source) + _native_slack_task_cards = False + if ( + source.platform == Platform.SLACK + and _progress_adapter_for_native is not None + and hasattr(_progress_adapter_for_native, "native_task_cards_enabled") + ): + try: + _native_slack_task_cards = bool( + _progress_adapter_for_native.native_task_cards_enabled() + ) + except Exception: + logger.debug("Slack native task-card config check failed", exc_info=True) + needs_progress_queue = ( + tool_progress_enabled or _thinking_enabled or _native_slack_task_cards + ) + return self._RunAgentDisplay( + user_config=user_config, + platform_key=platform_key, + enabled_toolsets=enabled_toolsets, + disabled_toolsets=disabled_toolsets, + resolve_display_setting=resolve_display_setting, + progress_mode=progress_mode, + progress_grouping=progress_grouping, + _display_surface_mode=_display_surface_mode, + tool_progress_enabled=tool_progress_enabled, + _live_status_mode=_live_status_mode, + _live_status_adapter=_live_status_adapter, + log_mode_enabled=log_mode_enabled, + log_queue=log_queue, + interim_assistant_messages_enabled=interim_assistant_messages_enabled, + _thinking_enabled=_thinking_enabled, + _native_slack_task_cards=_native_slack_task_cards, + needs_progress_queue=needs_progress_queue, + _generic_status_phrase=_generic_status_phrase, + ) + + def _run_agent_build_turn_context( + self, + disp: "GatewayRunner._RunAgentDisplay", + AIAgent: Any, + *, + message: str, + context_prompt: str, + history: List[Dict[str, Any]], + source: SessionSource, + session_id: str, + session_key: Optional[str], + run_generation: Optional[int], + _interrupt_depth: int, + event_message_id: Optional[str], + inbound_message_id: Optional[str], + channel_prompt: Optional[str], + moa_config: Optional[dict], + persist_user_message: Optional[Any], + persist_user_timestamp: Optional[float], + persist_user_display_kind: Optional[str], + ) -> Tuple[TurnContext, TurnRunner, Any]: + """Build the progress queues / holders, the ``TurnContext`` and its ``TurnRunner``. + + Returns ``(turn_ctx, turn_runner, cleanup_adapter)``; the progress-bubble cleanup flags travel + on ``turn_ctx._cleanup_progress`` / ``turn_ctx._cleanup_msg_ids``. + """ + from gateway.run import TurnRunner + def _run_still_current() -> bool: + if run_generation is None or not session_key: + return True + return self._is_session_run_current(session_key, run_generation) + + # Queue for progress messages (thread-safe) + progress_queue = queue.Queue() if disp.needs_progress_queue else None + last_tool = [None] # Mutable container for tracking in closure + last_progress_msg = [None] # Track last message for dedup + repeat_count = [0] # How many times the same message repeated + # True when the previous progress line was a terminal fenced code block — consecutive terminal + # calls then drop the repeated "💻 terminal" header and render back-to-back blocks. + last_was_terminal_block = [False] + + # Discord voice "verbal ack before tool calls": with the continuous mixer installed + # (discord.voice_fx.enabled), speak a short phrase over the idle bed on the FIRST tool call of + # the turn (from tool_start_callback, independent of the tool-progress text gate); once per turn. + _voice_ack_fired = [False] + _voice_ack_guild: List[Optional[int]] = [None] + if source.platform == Platform.DISCORD: + _va = self.adapters.get(Platform.DISCORD) + # source.chat_id is the linked text channel; resolve the guild whose + # voice connection is bound to it (mirrors DiscordAdapter.play_tts). + _vtc = getattr(_va, "_voice_text_channels", None) + if isinstance(_vtc, dict) and hasattr(_va, "voice_mixer_active"): + for _gid, _tc in _vtc.items(): + if str(_tc) == str(source.chat_id) and _va.voice_mixer_active(_gid): + _voice_ack_guild[0] = _gid + break + _voice_ack_loop = asyncio.get_running_loop() + + # voice_ack_callback extracted to TurnRunner.voice_ack_callback + # (published onto turn_ctx after the runner is constructed below). + + # Auto-cleanup of temporary progress bubbles (Telegram + any adapter that implements + # ``delete_message``). Failed runs skip cleanup so the bubbles remain as breadcrumbs. + _cleanup_progress = bool( + disp.resolve_display_setting(disp.user_config, disp.platform_key, "cleanup_progress") + ) + _cleanup_adapter = self._adapter_for_source(source) if _cleanup_progress else None + # getattr, not attribute access — same duck-typed-adapter guard as the edit_message check in + # send_progress_messages: a fake adapter without delete_message means "can't delete", not a crash. + _cleanup_delete = getattr(type(_cleanup_adapter), "delete_message", None) if _cleanup_adapter is not None else None + if _cleanup_adapter is not None and ( + _cleanup_delete is None + or _cleanup_delete is BasePlatformAdapter.delete_message + ): + # Adapter doesn't support deletion — silently disable. + _cleanup_progress = False + _cleanup_adapter = None + _cleanup_msg_ids: List[str] = [] + # First-touch onboarding latch: fires at most once per run, even if + # several tools exceed the threshold. + long_tool_hint_fired = [False] + _LONG_TOOL_THRESHOLD_S = 30.0 + + turn_ctx = TurnContext( + source=source, + _run_still_current=_run_still_current, + _live_status_adapter=disp._live_status_adapter, + _live_status_mode=disp._live_status_mode, + _thinking_enabled=disp._thinking_enabled, + progress_mode=disp.progress_mode, + progress_grouping=disp.progress_grouping, + tool_progress_enabled=disp.tool_progress_enabled, + progress_queue=progress_queue, + log_queue=disp.log_queue, + last_progress_msg=last_progress_msg, + last_tool=last_tool, + last_was_terminal_block=last_was_terminal_block, + repeat_count=repeat_count, + long_tool_hint_fired=long_tool_hint_fired, + _LONG_TOOL_THRESHOLD_S=_LONG_TOOL_THRESHOLD_S, + _cleanup_progress=_cleanup_progress, + _cleanup_msg_ids=_cleanup_msg_ids, + message=message, + AIAgent=AIAgent, + resolve_display_setting=disp.resolve_display_setting, + user_config=disp.user_config, + enabled_toolsets=disp.enabled_toolsets, + disabled_toolsets=disp.disabled_toolsets, + log_mode_enabled=disp.log_mode_enabled, + interim_assistant_messages_enabled=disp.interim_assistant_messages_enabled, + needs_progress_queue=disp.needs_progress_queue, + _native_slack_task_cards=disp._native_slack_task_cards, + _voice_ack_fired=_voice_ack_fired, + _voice_ack_guild=_voice_ack_guild, + _voice_ack_loop=_voice_ack_loop, + history=history, + context_prompt=context_prompt, + channel_prompt=channel_prompt, + session_id=session_id, + session_key=session_key, + run_generation=run_generation, + _interrupt_depth=_interrupt_depth, + event_message_id=event_message_id, + inbound_message_id=inbound_message_id, + moa_config=moa_config, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + ) + turn_runner = TurnRunner(self, turn_ctx) + # Callback invoked by agent on tool lifecycle events — extracted to + # TurnRunner.progress_callback (bound method, same signature). + turn_ctx.progress_callback = turn_runner.progress_callback + turn_ctx.voice_ack_callback = turn_runner.voice_ack_callback + turn_ctx.native_tool_start_callback = turn_runner.combined_tool_start_callback + turn_ctx.native_tool_complete_callback = ( + turn_runner.native_tool_complete_callback + ) + return turn_ctx, turn_runner, _cleanup_adapter + + def _run_agent_progress_threading( + self, + source: SessionSource, + event_message_id: Optional[str], + _native_slack_task_cards: bool, + ) -> Tuple[Optional[dict], Optional[str], Any, Optional[str]]: + """Resolve where progress bubbles are threaded. + + Returns ``(progress_metadata, progress_reply_to, progress_thread_id, relay_prospective_thread_id)``. + """ + from gateway.run import _non_conversational_metadata, _resolve_progress_thread_id + # Background task accumulating tool lines into one edited progress message. Threading metadata + # is platform-specific: Slack DM threading needs the event_message_id fallback; Telegram forum + # topics use message_thread_id and Hermes-created private DM topic lanes need thread metadata + # plus a reply anchor; Feishu only honors reply_in_thread on a reply, so topic progress replies + # to the triggering event; others use explicit source.thread_id only. Slack honours + # reply_in_thread=false: don't synthesise a thread for progress, or every later reply inherits it. + _progress_reply_in_thread = True + if source.platform == Platform.SLACK: + _slack_adapter_for_progress = self._adapter_for_source(source) + if _slack_adapter_for_progress is not None: + try: + # Relay lane: adapter owns mode resolution (nested platforms.relay.extra.slack subset, + # flat-key fallback). Native lane: read the flat extra as before. + _mode_fn = getattr( + _slack_adapter_for_progress, + "_effective_reply_in_thread", + None, + ) + if callable(_mode_fn): + _progress_reply_in_thread = bool(_mode_fn()) + else: + _progress_reply_in_thread = bool( + _slack_adapter_for_progress.config.extra.get( + "reply_in_thread", True + ) + ) + except Exception: + _progress_reply_in_thread = True + elif str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz": + # Buzz honours the same opt-out (reply_to_mode: off / extra.reply_in_thread: false): when the + # user asked for flat channel replies, progress must not synthesise a thread either. + _buzz_adapter_for_progress = self._adapter_for_source(source) + if _buzz_adapter_for_progress is not None: + try: + _progress_reply_in_thread = ( + getattr(_buzz_adapter_for_progress, "_reply_to_mode", "first") + != "off" + ) + except Exception: + _progress_reply_in_thread = True + _progress_thread_id = _resolve_progress_thread_id( + source.platform, source.thread_id, event_message_id, + reply_in_thread=_progress_reply_in_thread, + ) + # Relay Discord auto-thread lane: a channel-initiating message has no thread_id at ingest + # (thread is born on the connector's FIRST send). The connector stamps prospective_thread_id + # (anchor id == the thread it will create); carry it as reply_to on the progress send so + # bubbles route into the SAME auto-thread instead of landing flat in the parent channel. + _relay_prospective_thread_id = ( + str(getattr(source, "prospective_thread_id", None)) + if source.platform == Platform.DISCORD + and getattr(source, "delivered_via_upstream_relay", False) + and getattr(source, "prospective_thread_id", None) + and not source.thread_id + else None + ) + _progress_metadata = ( + self._thread_metadata_for_source(source, event_message_id) + if _progress_thread_id == source.thread_id + else self._thread_metadata_for_target( + source.platform, + source.chat_id, + _progress_thread_id, + chat_type=getattr(source, "chat_type", None), + reply_to_message_id=event_message_id, + ) + ) if _progress_thread_id else None + if _progress_metadata is None and _relay_prospective_thread_id: + # No real thread yet, but the connector will auto-thread on the + # reply anchor; carry it so progress joins that thread. + _progress_metadata = {"reply_to_message_id": event_message_id} + _progress_metadata = _non_conversational_metadata(_progress_metadata, platform=source.platform) + if _native_slack_task_cards: + # chat.startStream in channels requires the recipient team/user + # pair; harmless extras elsewhere, so stamp them whenever known. + _progress_metadata = dict(_progress_metadata or {}) + if source.scope_id: + _progress_metadata.setdefault("recipient_team_id", source.scope_id) + _progress_metadata.setdefault("slack_team_id", source.scope_id) + if source.user_id: + _progress_metadata.setdefault("recipient_user_id", source.user_id) + _progress_reply_to = ( + event_message_id + if ( + source.platform in (Platform.FEISHU, Platform.MATTERMOST) + and source.thread_id + and event_message_id + ) + or ( + # Buzz has no native thread_id; threading is always via reply-to the triggering event id + # (channel clutter otherwise); skipped when the user opted out of threaded replies. + str(getattr(source.platform, "value", source.platform) or "").lower() == "buzz" + and event_message_id + and _progress_reply_in_thread + ) + or _relay_prospective_thread_id + else None + ) + return _progress_metadata, _progress_reply_to, _progress_thread_id, _relay_prospective_thread_id + + async def _run_agent_write_tool_log(self, log_queue: Any) -> None: + """Drain log_queue and append tool-call lines to tool_calls.log (tool_progress=log). + + RotatingFileHandler (5MB × 3) bounds the log; RedactingFormatter keeps secrets off disk. + """ + from gateway.run import _hermes_home + if log_queue is None: + return + from logging.handlers import RotatingFileHandler + + from agent.redact import RedactingFormatter + + log_dir = _hermes_home / "logs" + log_dir.mkdir(parents=True, exist_ok=True) + file_handler = RotatingFileHandler( + log_dir / "tool_calls.log", + maxBytes=5 * 1024 * 1024, + backupCount=3, + encoding="utf-8", + ) + file_handler.setFormatter(RedactingFormatter("%(message)s")) + tool_logger = logging.getLogger(f"hermes.tool_calls.{id(log_queue)}") + tool_logger.setLevel(logging.INFO) + tool_logger.propagate = False + tool_logger.addHandler(file_handler) + try: + while True: + try: + tool_logger.info("%s", log_queue.get_nowait()) + except queue.Empty: + await asyncio.sleep(0.3) + except Exception as e: + logger.error("write_tool_log error: %s", e) + await asyncio.sleep(1) + except asyncio.CancelledError: + pass + finally: + # Drain remaining entries before closing so late tool calls + # from the final iteration aren't lost. + while True: + try: + tool_logger.info("%s", log_queue.get_nowait()) + except queue.Empty: + break + except Exception: + break + tool_logger.removeHandler(file_handler) + try: + file_handler.flush() + file_handler.close() + except Exception: + pass + + def _run_agent_status_thread_metadata( + self, + source: SessionSource, + event_message_id: Optional[str], + _progress_thread_id: Any, + _relay_prospective_thread_id: Optional[str], + ) -> Optional[Dict[str, Any]]: + """Thread metadata for status / approval / stream sends (Feishu carries the reply anchor).""" + if source.platform == Platform.FEISHU and source.thread_id and event_message_id: + # Feishu topics only keep messages inside the topic when they are sent via the reply API + # with reply_in_thread=true. Status/approval/stream paths usually only get metadata, so + # carry the triggering message id as a Feishu-specific fallback. + _status_thread_metadata: Optional[Dict[str, Any]] = { + "thread_id": _progress_thread_id, + "reply_to_message_id": event_message_id, + } + else: + _status_thread_metadata = ( + self._thread_metadata_for_source(source, event_message_id) + if _progress_thread_id == source.thread_id + else self._thread_metadata_for_target( + source.platform, + source.chat_id, + _progress_thread_id, + chat_type=getattr(source, "chat_type", None), + reply_to_message_id=event_message_id, + ) + ) if _progress_thread_id else None + if _status_thread_metadata is None and _relay_prospective_thread_id: + # Relay Discord auto-thread lane (see _progress_metadata): carry the reply anchor so + # status/interim bubbles route into the same connector-created thread as the final reply. + _status_thread_metadata = { + "reply_to_message_id": event_message_id + } + return _status_thread_metadata + + def _run_agent_start_streaming_tts( + self, + source: SessionSource, + message_type: Optional[str], + _status_thread_metadata: Optional[Dict[str, Any]], + streaming_tts_consumer_holder: list, + ) -> None: + # Streaming TTS consumer setup. Created on the gateway event-loop thread (here), NOT inside + # run_sync's executor worker: the outer interrupt / finalisation paths reference the consumer + # via ``streaming_tts_consumer_holder[0]`` and would hit a cross-scope NameError. + _stts_adapter = self._adapter_for_source(source) + _is_voice_input = ( + message_type is not None + and str(getattr(message_type, "value", message_type)).lower() == "voice" + ) + if ( + _stts_adapter is not None + and _is_voice_input + and _stts_adapter._should_auto_tts_for_chat(source.chat_id) + ): + try: + from gateway.streaming_tts_consumer import StreamingTTSConsumer + from tools.tts_tool import _load_tts_config + _tts_cfg = _load_tts_config() + _gateway_loop = self._gateway_loop or asyncio.get_event_loop() + _stts_consumer = StreamingTTSConsumer( + adapter=_stts_adapter, + chat_id=source.chat_id, + tts_config=_tts_cfg, + loop=_gateway_loop, + metadata=_status_thread_metadata, + ) + if _stts_consumer.active: + streaming_tts_consumer_holder[0] = _stts_consumer + _stts_consumer.start() + # else: consumer inactive (no streaming provider) — leave + # the holder as None so the whole-file fallback path runs. + except Exception as _stts_err: + logger.debug("Could not set up streaming TTS consumer: %s", _stts_err) + + async def _run_agent_stream_consumer_task(self, stream_consumer_holder: list) -> None: + """Wait for the stream consumer to be created, then run it.""" + for _ in range(200): # Up to 10s wait + if stream_consumer_holder[0] is not None: + await stream_consumer_holder[0].run() + return + await asyncio.sleep(0.05) + + async def _run_agent_track_agent( + self, + session_key: Optional[str], + run_generation: Optional[int], + agent_holder: list, + ) -> None: + """Track this agent as running for the session (interrupt support) once it is created.""" + # Wait for agent to be created + while agent_holder[0] is None: + await asyncio.sleep(0.05) + if not session_key: + return + # Only promote the sentinel to the real agent if this run is still current. If /stop or + # /new bumped the generation while we were spinning up, leave the newer run's slot alone + # — we'll be discarded by the stale-result check in _handle_message_with_agent. + if run_generation is not None and not self._is_session_run_current( + session_key, run_generation + ): + logger.info( + "Skipping stale agent promotion for %s — generation %s is no longer current", + session_key or "", + run_generation, + ) + return + self._session_state(session_key).turn.agent = agent_holder[0] + if self._draining: + self._update_runtime_status("draining") + + async def _run_agent_monitor_for_interrupt( + self, + source: SessionSource, + session_key: Optional[str], + agent_holder: list, + _interrupt_detected: "asyncio.Event", + streaming_tts_consumer_holder: list, + ) -> None: + # Monitor adapter interrupts (new messages). PRIMARY interrupt path for regular text: Level 1 + # (base.py) catches them before _handle_message(), so the Level 2 running_agent.interrupt() path + # never fires. The inactivity poll loop has a BACKUP check in case this task dies silently. + from gateway.run import _build_media_placeholder + if not session_key: + return + + while True: + await asyncio.sleep(0.2) # Check every 200ms + try: + # Re-resolve adapter each iteration so reconnects don't + # leave us holding a stale reference. + _adapter = self._adapter_for_source(source) + if not _adapter: + continue + # Must use session_key (build_session_key output), NOT source.chat_id: the adapter + # stores interrupt events under the full session key. + if hasattr(_adapter, 'has_pending_interrupt') and _adapter.has_pending_interrupt(session_key): + agent = agent_holder[0] + if agent: + # Peek WITHOUT consuming: the message must stay in _pending_messages for the + # post-run _dequeue_pending_event() (full MessageEvent + media). Popping here + # races: the agent may finish before checking _interrupt_requested, losing it. + _peek_event = _adapter._pending_messages.get(session_key) + pending_text = None + if _peek_event is not None: + pending_text = _peek_event.text or "" + # Transcribe audio BEFORE signaling the agent, so voice messages interrupt + # with the real transcript, not an empty string / file-path placeholder. + _media_urls = getattr(_peek_event, "media_urls", None) or [] + if self._pending_event_audio_paths(_peek_event): + pending_text, _ = await self._transcribe_and_echo_pending_voice( + _peek_event, + _adapter, + source, + pending_text, + log_context="Voice-interrupt", + metadata={"thread_id": source.thread_id} if source.thread_id else None, + ) + elif not pending_text and _media_urls: + pending_text = _build_media_placeholder(_peek_event) + logger.debug("Interrupt detected from adapter, signaling agent...") + agent.interrupt(pending_text) + _interrupt_detected.set() + # Abort streaming TTS on barge-in (#60671). + _stts = streaming_tts_consumer_holder[0] + if _stts is not None: + _stts.abort("barge-in") + break + except asyncio.CancelledError: + raise + except Exception as _mon_err: + logger.debug("monitor_for_interrupt error (will retry): %s", _mon_err) + + @staticmethod + def _run_agent_stream_confirmed_final_delivery( + consumer, + final_text: str, + *, + previewed: bool = False, + ) -> bool: + """Return True only when the actual final reply reached the user.""" + if consumer is None: + return False + if getattr(consumer, "final_response_sent", False): + # A successful finalize call is not proof the *content* was final: the edit may carry + # only the last preview snapshot. Reconcile against the recorded turn-final payload: + # only a demonstrable mismatch (False, incl. payload-less split delivery) overrides + # the flag; None keeps legacy trust so timeout dedup isn't regressed. + matcher = getattr(consumer, "delivered_final_matches", None) + if callable(matcher): + try: + if matcher(final_text) is False: + return False + except Exception: + pass + return True + if previewed: + has_delivered_text = getattr(consumer, "has_delivered_text", None) + if callable(has_delivered_text): + try: + return bool(has_delivered_text(final_text)) + except Exception: + return False + return False + + def _run_agent_start_turn_worker( + self, + turn_ctx: TurnContext, + run_sync: Callable[[], Any], + agent_holder: list, + session_id: str, + session_key: Optional[str], + run_generation: Optional[int], + ) -> "GatewayRunner._RunAgentWorker": + """Schedule ``run_sync`` on the executor plus the inactivity watchdog thread.""" + from gateway.run import _float_env, _watch_gateway_turn_inactivity + # Thread pool so we don't block. *Inactivity* timeout, not wall-clock: the agent may run for + # hours while actively calling tools / streaming, but a hung API call or stuck tool is killed. + # agent.gateway_timeout / HERMES_AGENT_TIMEOUT (env wins); default 1800s; 0 = unlimited. + _agent_timeout_raw = _float_env("HERMES_AGENT_TIMEOUT", 1800) + _agent_timeout = _agent_timeout_raw if _agent_timeout_raw > 0 else None + _agent_warning_raw = _float_env("HERMES_AGENT_TIMEOUT_WARNING", 900) + _agent_warning = _agent_warning_raw if _agent_warning_raw > 0 else None + + # A background=true process intentionally survives a successful turn, so capture + # existing IDs and reap only children created by THIS turn if it times out. The daemon + # watchdog is independent of asyncio: cgroup memory reclaim can starve the loop that + # runs the normal timeout poll, and cleanup must not wait for the loop to recover. + from tools.process_registry import process_registry + + _turn_task_id = session_id or "" + _turn_process_baseline = process_registry.snapshot_running_ids(_turn_task_id) + turn_ctx.process_task_id = _turn_task_id + turn_ctx.process_baseline = _turn_process_baseline + _turn_worker_done = threading.Event() + _turn_timeout_fired = threading.Event() + _turn_cleanup_lock = threading.Lock() + # task_id is session-scoped, not turn-scoped: gate the eventual reap on this exact claim still + # being current, so a replacement turn on the same session that starts before the watchdog + # fires doesn't get its own fresh process killed by this turn's stale baseline. + _turn_run_generation = run_generation + _turn_is_current = ( + (lambda: self._is_session_run_current(session_key, _turn_run_generation)) + if _turn_run_generation is not None + else (lambda: True) + ) + + def _run_sync_with_timeout_lifecycle(): + try: + return run_sync() + finally: + _turn_worker_done.set() + # `.turn.agent` is only reset to _AGENT_PENDING_SENTINEL when the *next* turn is + # claimed, so this agent stays reachable from _interrupt_and_clear_session() + # until then. Clearing ownership markers the instant our worker finishes means a + # /stop on the finished turn no longer reaps background work it left running. + _finished_agent = agent_holder[0] if agent_holder else None + if _finished_agent is not None: + _finished_agent._gateway_turn_process_task_id = "" + _finished_agent._gateway_turn_process_baseline = frozenset() + + if _agent_timeout is not None: + threading.Thread( + target=_watch_gateway_turn_inactivity, + kwargs={ + "agent_holder": agent_holder, + "task_id": _turn_task_id, + "process_baseline": _turn_process_baseline, + "timeout": _agent_timeout, + "worker_done": _turn_worker_done, + "timeout_fired": _turn_timeout_fired, + "cleanup_lock": _turn_cleanup_lock, + "poll_interval": 5.0, + "is_still_current": _turn_is_current, + }, + name=f"gateway-turn-watchdog-{_turn_task_id[:12]}", + daemon=True, + ).start() + _executor_task = asyncio.ensure_future( + self._run_in_executor_with_context(_run_sync_with_timeout_lifecycle) + ) + return self._RunAgentWorker( + executor_task=_executor_task, + agent_timeout=_agent_timeout, + agent_warning=_agent_warning, + task_id=_turn_task_id, + process_baseline=_turn_process_baseline, + worker_done=_turn_worker_done, + timeout_fired=_turn_timeout_fired, + cleanup_lock=_turn_cleanup_lock, + is_current=_turn_is_current, + ) + + async def _run_agent_backup_interrupt_check( + self, + source: SessionSource, + session_key: Optional[str], + agent_holder: list, + _interrupt_detected: "asyncio.Event", + interrupt_monitor: "asyncio.Task", + streaming_tts_consumer_holder: list, + ) -> None: + """Backup interrupt check: if the monitor task died or missed the interrupt, catch it here.""" + from gateway.run import _build_media_placeholder + if not _interrupt_detected.is_set() and session_key: + _backup_adapter = self._adapter_for_source(source) + _backup_agent = agent_holder[0] + if (_backup_adapter and _backup_agent + and hasattr(_backup_adapter, 'has_pending_interrupt') + and _backup_adapter.has_pending_interrupt(session_key)): + _bp_event = _backup_adapter._pending_messages.get(session_key) + _bp_text = _bp_event.text if _bp_event else None + if _bp_event is not None: + _bp_media_urls = getattr(_bp_event, "media_urls", None) or [] + if self._pending_event_audio_paths(_bp_event): + _bp_text, _ = await self._transcribe_and_echo_pending_voice( + _bp_event, + _backup_adapter, + source, + _bp_text or "", + log_context="Voice-backup-interrupt", + metadata={"thread_id": source.thread_id} if source.thread_id else None, + ) + elif not _bp_text and _bp_media_urls: + _bp_text = _build_media_placeholder(_bp_event) + logger.info( + "Backup interrupt detected for session %s " + "(monitor task state: %s)", + session_key, + "done" if interrupt_monitor.done() else "running", + ) + _backup_agent.interrupt(_bp_text) + _interrupt_detected.set() + # Abort streaming TTS on barge-in (#60671). + _stts = streaming_tts_consumer_holder[0] + if _stts is not None: + _stts.abort("barge-in") + + async def _run_agent_await_turn_worker( + self, + worker: "GatewayRunner._RunAgentWorker", + *, + source: SessionSource, + session_key: Optional[str], + agent_holder: list, + result_holder: list, + tools_holder: list, + _interrupt_detected: "asyncio.Event", + interrupt_monitor: "asyncio.Task", + streaming_tts_consumer_holder: list, + _status_thread_metadata: Optional[Dict[str, Any]], + ) -> Any: + """Poll the executor future (inactivity timeout + backup interrupt checks); return its result. + + On inactivity timeout the result is a synthetic failed run dict carrying the diagnostic. + """ + from gateway.run import ( + _INTERRUPT_REASON_TIMEOUT, + _abandon_timed_out_gateway_turn, + _interim_metadata, + request_hard_interrupt, + ) + _warning_fired = False + _inactivity_timeout = False + _POLL_INTERVAL = 5.0 + + if worker.agent_timeout is None: + # Unlimited — still poll periodically for backup interrupt + # detection in case monitor_for_interrupt() silently died. + response = None + while True: + done, _ = await asyncio.wait( + {worker.executor_task}, timeout=_POLL_INTERVAL + ) + if done: + response = worker.executor_task.result() + break + # Backup interrupt check: if the monitor task died or + # missed the interrupt, catch it here. + await self._run_agent_backup_interrupt_check( + source, + session_key, + agent_holder, + _interrupt_detected, + interrupt_monitor, + streaming_tts_consumer_holder, + ) + + else: + # Poll the agent's built-in activity tracker (updated by _touch_activity() on every tool + # call, API call, and stream delta) every few seconds. + response = None + while True: + done, _ = await asyncio.wait( + {worker.executor_task}, timeout=_POLL_INTERVAL + ) + if done: + # Prefer the real result when the worker finished even if the watchdog fired in + # the same window: the completed run already persisted its reply, so the "agent + # inactive" diagnostic would contradict the stored transcript. + response = worker.executor_task.result() + break + if worker.timeout_fired.is_set(): + _inactivity_timeout = True + break + # Agent still running — check inactivity. + _agent_ref = agent_holder[0] + _idle_secs = 0.0 + if _agent_ref and hasattr(_agent_ref, "get_activity_summary"): + try: + _act = _agent_ref.get_activity_summary() + _idle_secs = _act.get("seconds_since_activity", 0.0) + except Exception: + pass + # Staged warning: fire once before escalating to full timeout. + if (not _warning_fired and worker.agent_warning is not None + and _idle_secs >= worker.agent_warning): + _warning_fired = True + _warn_adapter = self._adapter_for_source(source) + if _warn_adapter: + _elapsed_warn = int(worker.agent_warning // 60) or 1 + _remaining_mins = int((worker.agent_timeout - worker.agent_warning) // 60) or 1 + try: + await _warn_adapter.send( + source.chat_id, + f"⚠️ No activity for {_elapsed_warn} min. " + f"If the agent does not respond soon, it will " + f"be timed out in {_remaining_mins} min. " + f"You can continue waiting or use /reset.", + metadata=_interim_metadata(_status_thread_metadata), + ) + except Exception as _warn_err: + logger.debug("Inactivity warning send error: %s", _warn_err) + if _idle_secs >= worker.agent_timeout: + _inactivity_timeout = True + threading.Thread( + target=_abandon_timed_out_gateway_turn, + kwargs={ + "agent_holder": agent_holder, + "task_id": worker.task_id, + "process_baseline": worker.process_baseline, + "worker_done": worker.worker_done, + "timeout_fired": worker.timeout_fired, + "cleanup_lock": worker.cleanup_lock, + "is_still_current": worker.is_current, + }, + name=f"gateway-turn-reaper-{worker.task_id[:12]}", + daemon=True, + ).start() + break + # Backup interrupt check (same as unlimited path). + await self._run_agent_backup_interrupt_check( + source, + session_key, + agent_holder, + _interrupt_detected, + interrupt_monitor, + streaming_tts_consumer_holder, + ) + + if _inactivity_timeout: + # Build a diagnostic summary from the agent's activity tracker. + _timed_out_agent = agent_holder[0] + _activity = {} + if _timed_out_agent and hasattr(_timed_out_agent, "get_activity_summary"): + with suppress(Exception): + _activity = _timed_out_agent.get_activity_summary() + + _last_desc = _activity.get("last_activity_desc", "unknown") + _secs_ago = _activity.get("seconds_since_activity", 0) + _cur_tool = _activity.get("current_tool") + _iter_n = _activity.get("api_call_count", 0) + _iter_max = _activity.get("max_iterations", 0) + + logger.error( + "Agent idle for %.0fs (timeout %.0fs) in session %s " + "| last_activity=%s | iteration=%s/%s | tool=%s", + _secs_ago, worker.agent_timeout, session_key, + _last_desc, _iter_n, _iter_max, + _cur_tool or "none", + ) + + # Interrupt the agent if it's still running so the thread + # pool worker is freed. + if _timed_out_agent: + request_hard_interrupt(_timed_out_agent, _INTERRUPT_REASON_TIMEOUT) + + _timeout_mins = int(worker.agent_timeout // 60) or 1 + + # Construct a user-facing message with diagnostic context. + _diag_lines = [ + f"⏱️ Agent inactive for {_timeout_mins} min — no tool calls " + f"or API responses." + ] + if _cur_tool: + _diag_lines.append( + f"The agent appears stuck on tool `{_cur_tool}` " + f"({_secs_ago:.0f}s since last activity, " + f"iteration {_iter_n}/{_iter_max})." + ) + else: + _diag_lines.append( + f"Last activity: {_last_desc} ({_secs_ago:.0f}s ago, " + f"iteration {_iter_n}/{_iter_max}). " + "The agent may have been waiting on an API response." + ) + _diag_lines.append( + "To increase the limit, set agent.gateway_timeout in config.yaml " + "(value in seconds, 0 = no limit) and restart the gateway.\n" + "Try again, or use /reset to start fresh." + ) + + response = { + "final_response": "\n".join(_diag_lines), + "messages": result_holder[0].get("messages", []) if result_holder[0] else [], + "api_calls": _iter_n, + "tools": tools_holder[0] or [], + "history_offset": 0, + "failed": True, + } + return response + + def _run_agent_evict_on_fallback( + self, session_key: Optional[str], agent_holder: list, result_holder: list, + ) -> None: + # Persist fallback-model switches so /model shows the actually-active model. Skip + # eviction when the run failed — evicting forces MCP reinit on the next message for no + # benefit (bad model → fallback → evict → recreate → same 400 loop burning CPU). + from gateway.run import _resolve_gateway_model + _agent = agent_holder[0] + _result_for_fb = result_holder[0] + _run_failed = _result_for_fb.get("failed") if _result_for_fb else False + if _agent is not None and hasattr(_agent, 'model') and not _run_failed: + _cfg_model = _resolve_gateway_model() + # Normalize _cfg_model as AIAgent.__init__ does so a vendor-prefixed config value + # matches the agent's stripped model on native providers — otherwise the cached agent + # is evicted every turn, destroying prompt caching. Aggregators keep the vendor slug. + try: + from hermes_cli.model_normalize import ( + _AGGREGATOR_PROVIDERS, + normalize_model_for_provider, + ) + _agent_provider = getattr(_agent, 'provider', '') or '' + if _agent_provider and _agent_provider not in _AGGREGATOR_PROVIDERS: + _cfg_model = normalize_model_for_provider(_cfg_model, _agent_provider) + except Exception: + pass + if _agent.model != _cfg_model and not self._is_intentional_model_switch(session_key, _agent.model): + # Fallback activated on a successful run — evict cached + # agent so the next message retries the primary model. + self._evict_cached_agent(session_key) + + async def _run_agent_finalize_streaming_tts( + self, + streaming_tts_consumer_holder: list, + adapter: Any, + session_key: Optional[str], + run_generation: Optional[int], + ) -> None: + # Finalize the streaming-TTS consumer. finish() runs on the outer event-loop thread so + # early returns from run_sync are also finalised. wait_complete() drains queued audio; + # on timeout abort unconditionally — if audio was audible keep suppression (no replay + # from the start); if not, the whole-file fallback is permitted. + _stts = streaming_tts_consumer_holder[0] + if _stts is not None: + _stts.finish() + try: + await _stts.wait_complete(timeout=10.0) + except Exception as _stts_done_err: + logger.debug("streaming TTS wait_complete error: %s", _stts_done_err) + if not _stts.done: + # Timeout before or after audible audio: abort to free the consumer task. Audible + # streams retain suppression; silent streams stay eligible for whole-file fallback. + _stts.abort("streaming TTS finalisation timeout") + await _stts.wait_complete(timeout=2.0) + if _stts.suppress_whole_file and adapter is not None: + _mark_turn = getattr(adapter, "_mark_streaming_tts_completed_turn", None) + if callable(_mark_turn): + _mark_turn(session_key, run_generation) + + async def _run_agent_drain_pending( + self, + result: Any, + adapter: Any, + source: SessionSource, + session_key: Optional[str], + ) -> Tuple[Any, Optional[str]]: + """Dequeue the adapter's pending / interrupt / leftover-steer follow-up. + + Returns ``(pending_event, pending)``. + """ + from gateway.run import ( + _build_media_placeholder, + _dequeue_pending_event, + _is_control_interrupt_message, + ) + # Get pending message from adapter. + # Use session_key (not source.chat_id) to match adapter's storage keys. + pending_event = None + pending = None + if result and adapter and session_key: + pending_event = _dequeue_pending_event(adapter, session_key) + # /queue overflow: after consuming the adapter's "next-up" slot, promote the next + # queued event into it so the recursive run's drain will see it. Keeping the slot + # occupied for the whole FIFO chain preserves order and makes a mid-chain /queue + # route to overflow instead of jumping the queue. + pending_event = self._promote_queued_event(session_key, adapter, pending_event) + if result.get("interrupted") and not pending_event and result.get("interrupt_message"): + interrupt_message = result.get("interrupt_message") + if _is_control_interrupt_message(interrupt_message): + logger.info( + "Ignoring control interrupt message for session %s: %s", + session_key or "?", + interrupt_message, + ) + else: + pending = interrupt_message + elif pending_event: + # Transcribe audio on the dequeued event BEFORE it becomes the next user turn, so + # queued/interrupting voice messages drain with the real transcript, not a file path. + _pending_text = pending_event.text or "" + _media_urls = getattr(pending_event, "media_urls", None) or [] + if self._pending_event_audio_paths(pending_event): + pending, _ = await self._transcribe_and_echo_pending_voice( + pending_event, + adapter, + source, + _pending_text, + log_context="Voice-drain", + metadata={"thread_id": source.thread_id} if source.thread_id else None, + ) + if not pending: + pending = _build_media_placeholder(pending_event) + else: + pending = _pending_text or _build_media_placeholder(pending_event) + if pending: + logger.debug("Processing queued message after agent completion: '%s...'", pending[:40]) + + # Leftover /steer: a steer arriving after the last tool batch (e.g. during the final API + # call) comes back in result["pending_steer"]; deliver it as the next user turn, not drop it. + if result and not pending and not pending_event: + _leftover_steer = result.get("pending_steer") + if _leftover_steer: + pending = _leftover_steer + logger.debug("Delivering leftover /steer as next turn: '%s...'", pending[:40]) + + # Safety net: if the pending text is a slash command (e.g. "/stop", "/new"), discard it + # — commands should never be passed to the agent as user input. + if pending and pending.strip().startswith("/"): + _pending_parts = pending.strip().split(None, 1) + _pending_cmd_word = _pending_parts[0][1:].lower() if _pending_parts else "" + if _pending_cmd_word: + try: + from hermes_cli.commands import resolve_command as _rc_pending + if _rc_pending(_pending_cmd_word): + logger.info( + "Discarding command '/%s' from pending queue — " + "commands must not be passed as agent input", + _pending_cmd_word, + ) + pending_event = None + pending = None + except Exception: + pass + + if self._draining and (pending_event or pending): + logger.info( + "Discarding pending follow-up for session %s during gateway %s", + session_key or "?", + self._status_action_label(), + ) + pending_event = None + pending = None + return pending_event, pending + + async def _run_agent_deliver_first_response( + self, + *, + source: SessionSource, + adapter: Any, + session_key: Optional[str], + run_generation: Optional[int], + event_message_id: Optional[str], + response: Any, + result: Any, + stream_consumer_holder: list, + stream_task: Any, + _status_thread_metadata: Optional[Dict[str, Any]], + ) -> None: + # Queued message after normal completion: deliver the first response before the + # queued follow-up, unless streaming already delivered it. + _sc = stream_consumer_holder[0] + if _sc and stream_task: + try: + await asyncio.wait_for(stream_task, timeout=5.0) + except (asyncio.TimeoutError, asyncio.CancelledError): + stream_task.cancel() + with suppress(asyncio.CancelledError): + await stream_task + except Exception as e: + logger.debug("Stream consumer wait before queued message failed: %s", e) + # The queued branch needs raw ``result`` for interruption, history, and + # recursion state, but delivery must use the finalized task result — it carries + # empty/failure normalization and final-response processing from _run_agent_task. + _delivery_result = response if isinstance(response, dict) else (result or {}) + _previewed = bool(_delivery_result.get("response_previewed")) + first_response = _delivery_result.get("final_response", "") + _already_streamed = self._run_agent_stream_confirmed_final_delivery( + _sc, + first_response, + previewed=_previewed, + ) + # Same predicate as the normal completed-turn path: this direct queued-send branch + # predates intentional-silence filtering and would leak the literal marker. + try: + from gateway.response_filters import is_intentional_silence_agent_result + _intentional_silence = is_intentional_silence_agent_result( + _delivery_result, first_response, + ) + except Exception: + _intentional_silence = False + if _intentional_silence: + logger.info( + "Queued follow-up for session %s: suppressing intentional silence marker before continuing.", + session_key or "?", + ) + elif first_response: + try: + if _already_streamed: + logger.info( + "Queued follow-up for session %s: final text delivery confirmed; delivering explicit media before continuing.", + session_key or "?", + ) + else: + logger.info( + "Queued follow-up for session %s: final stream delivery not confirmed; sending first response before continuing.", + session_key or "?", + ) + await self._deliver_queued_first_response( + first_response, + source=source, + adapter=adapter, + metadata=_status_thread_metadata, + event_message_id=event_message_id, + text_already_delivered=_already_streamed, + deliver_media=not _delivery_result.get("failed"), + stream_consumer=_sc, + ) + except Exception as e: + logger.warning("Failed to send first response before queued message: %s", e) + # Release deferred bg-review notifications now that the first response is delivered: + # pop from the adapter's callback dict (no double-fire in base.py's finally) and call. + if getattr(type(adapter), "pop_post_delivery_callback", None) is not None: + _bg_cb = adapter.pop_post_delivery_callback( + session_key, + generation=run_generation, + ) + if callable(_bg_cb): + try: + _bg_result = _bg_cb() + if inspect.isawaitable(_bg_result): + await _bg_result + except Exception: + pass + elif adapter and hasattr(adapter, "_post_delivery_callbacks"): + _bg_cb = adapter._post_delivery_callbacks.pop(session_key, None) + if callable(_bg_cb): + try: + _bg_result = _bg_cb() + if inspect.isawaitable(_bg_result): + await _bg_result + except Exception: + pass + + async def _run_agent_queued_followup( + self, + *, + source: SessionSource, + adapter: Any, + session_id: str, + session_key: Optional[str], + run_generation: Optional[int], + _interrupt_depth: int, + event_message_id: Optional[str], + context_prompt: str, + history: List[Dict[str, Any]], + pending: Optional[str], + pending_event: Any, + response: Any, + result: Any, + result_holder: list, + stream_consumer_holder: list, + stream_task: Any, + _status_thread_metadata: Optional[Dict[str, Any]], + ) -> Any: + """Run the queued / interrupting follow-up as the next turn (recursive ``_run_agent``).""" + from gateway.run import _preserve_queued_followup_history_offset, merge_pending_message_event + logger.debug("Processing pending message: '%s...'", pending[:40]) + + # Clear the adapter's interrupt event so the next _run_agent call doesn't re-trigger the + # interrupt before the new agent's first API call (infinite loop otherwise). + if adapter and hasattr(adapter, '_active_sessions') and session_key and session_key in adapter._active_sessions: + adapter._active_sessions[session_key].clear() + + # Cap recursion depth to prevent resource exhaustion when the + # user sends multiple messages while the agent keeps failing. (#816) + if _interrupt_depth >= self._MAX_INTERRUPT_DEPTH: + logger.warning( + "Interrupt recursion depth %d reached for session %s — " + "queueing message instead of recursing.", + _interrupt_depth, session_key, + ) + adapter = self._adapter_for_source(source) + if adapter and pending_event: + merge_pending_message_event(adapter._pending_messages, session_key, pending_event) + elif adapter and hasattr(adapter, 'queue_message'): + adapter.queue_message(session_key, pending) + return result_holder[0] or {"final_response": response, "messages": history} + + was_interrupted = result.get("interrupted") + if not was_interrupted: + await self._run_agent_deliver_first_response( + source=source, + adapter=adapter, + session_key=session_key, + run_generation=run_generation, + event_message_id=event_message_id, + response=response, + result=result, + stream_consumer_holder=stream_consumer_holder, + stream_task=stream_task, + _status_thread_metadata=_status_thread_metadata, + ) + # else: interrupted — discard the response ("Operation interrupted." is noise; the user + # knows they sent a new message). + + updated_history = result.get("messages", history) + next_source = source + next_message = pending + next_message_id = None + next_channel_prompt = None + next_session_key = session_key + # Carry the pending event's message_type into the recursive call so queued voice turns + # can stream TTS and re-mark the generation for the final delivered turn. + next_message_type = None + if pending_event is not None: + next_source = getattr(pending_event, "source", None) or source + if self._is_goal_continuation_event(pending_event) and not self._goal_still_active_for_session(session_id): + logger.info( + "Discarding stale goal continuation for session %s — goal is no longer active", + session_key or "?", + ) + return result + # Resolve the follow-up's session key BEFORE preparing the inbound text: + # _prepare_inbound_message_text buffers native image paths under the key given, and + # the recursive _run_agent consumes them under next_session_key — mismatch drops them. + try: + next_session_key = self._session_key_for_source(next_source) + except Exception: + logger.debug( + "Queued follow-up session-key resolution failed; reusing %s", + session_key or "?", + exc_info=True, + ) + next_message = await self._prepare_profile_scoped_inbound_message_text( + event=pending_event, + source=next_source, + history=updated_history, + session_key=next_session_key, + ) + if next_message is None: + return result + next_message_id = self._reply_anchor_for_event(pending_event) + next_channel_prompt = getattr(pending_event, "channel_prompt", None) + next_message_type = getattr(pending_event, "message_type", None) + + # Clear the prior logical turn's completed streaming marker so the recursive turn's + # streaming TTS isn't suppressed by that completion. + _clear_adapter = self._adapter_for_source(source) + if _clear_adapter is not None and session_key and run_generation is not None: + _completed_turns = getattr(_clear_adapter, "_streaming_tts_completed_turns", None) + if _completed_turns is not None: + _prior_key = getattr(_clear_adapter, "_streaming_tts_turn_key", None) + if callable(_prior_key): + _pk = _prior_key(session_key, run_generation) + if _pk: + _completed_turns.discard(_pk) + + # Restart the typing indicator for the follow-up turn; the outer + # _process_message_background typing task is alive but may be stale. + _followup_adapter = self._adapter_for_source(source) + if _followup_adapter: + with suppress(Exception): + await _followup_adapter.send_typing( + source.chat_id, + metadata=_status_thread_metadata, + ) + + # Re-baseline the cached agent's message_count before recursing into the /queue follow-up: + # the coherence guard would otherwise rebuild on OUR OWN flushed rows and destroy the + # prompt-cache prefix; _handle_message_with_agent re-baselines only after the chain ends. + await self._refresh_agent_cache_message_count(session_key, session_id) + + followup_result = await self._run_agent( + message=next_message, + context_prompt=context_prompt, + history=updated_history, + source=next_source, + session_id=session_id, + session_key=next_session_key, + run_generation=run_generation, + _interrupt_depth=_interrupt_depth + 1, + event_message_id=next_message_id, + channel_prompt=next_channel_prompt, + message_type=next_message_type, + ) + return _preserve_queued_followup_history_offset(result, followup_result) + + async def _run_agent_cleanup_turn_tasks( + self, + *, + progress_task: Any, + log_task: Any, + interrupt_monitor: "asyncio.Task", + _notify_task: "asyncio.Task", + tracking_task: "asyncio.Task", + stream_task: Any, + stream_consumer_holder: list, + streaming_tts_consumer_holder: list, + session_key: Optional[str], + run_generation: Optional[int], + ) -> None: + """``finally`` half of a turn: cancel background tasks, flush stream, release the session slot.""" + # Stop progress sender, interrupt monitor, and notification task + if progress_task: + progress_task.cancel() + if log_task: + log_task.cancel() + interrupt_monitor.cancel() + _notify_task.cancel() + + # Wait for stream consumer to finish its final edit + if stream_task: + # If the agent never created a stream consumer (non-streaming path, or a test stub + # returning synchronously) there is nothing to flush — cancel now instead of waiting + # out the 5s timeout polling for a consumer that will never arrive. + _has_stream_consumer = ( + stream_consumer_holder + and stream_consumer_holder[0] is not None + ) + if not _has_stream_consumer: + stream_task.cancel() + with suppress(asyncio.CancelledError): + await stream_task + else: + try: + await asyncio.wait_for(stream_task, timeout=5.0) + except (asyncio.TimeoutError, asyncio.CancelledError): + stream_task.cancel() + with suppress(asyncio.CancelledError): + await stream_task + + # Unconditional abort + bounded wait for the streaming-TTS consumer: covers cancellation / + # exception paths where the normal finalisation block was skipped. + _stts_finally = streaming_tts_consumer_holder[0] + if _stts_finally is not None and not _stts_finally.done: + _stts_finally.abort("cleanup") + with suppress(Exception): + await _stts_finally.wait_complete(timeout=2.0) + + # Clean up tracking + tracking_task.cancel() + if session_key: + # Release the slot only if this run's generation still owns it: a /stop or /new that + # bumped the generation while we unwound already installed its own state; keep it. + self._release_running_agent_state( + session_key, run_generation=run_generation + ) + if self._draining: + self._update_runtime_status("draining") + + # Wait for cancelled tasks + for task in [progress_task, log_task, interrupt_monitor, tracking_task, _notify_task]: + if task: + try: + await task + except asyncio.CancelledError: + pass + except Exception: + # A background task that died of a non-cancellation error (transport drop in + # a progress/card publish) must not abort the cleanup path — everything + # after this loop (final-delivery bookkeeping) still runs (review B7). + logger.debug( + "background turn task failed during cleanup", + exc_info=True, + ) + + async def _run_agent_mark_streamed_delivery( + self, + response: Any, + stream_consumer_holder: list, + source: SessionSource, + session_key: Optional[str], + ) -> None: + # If streaming already delivered the response, skip the caller's send() — but never when the + # agent failed (the error is unseen content) or on "(empty)": interim text ("Let me search…") + # set already_sent but is NOT the final answer; suppressing would leave the user with silence. + _sc = stream_consumer_holder[0] + if isinstance(response, dict) and not response.get("failed"): + _final = response.get("final_response") or "" + _is_empty_sentinel = not _final or _final == "(empty)" + # response_previewed means interim_assistant_callback already saw the final text, but only + # suppress the send if that exact text was delivered — unrelated commentary/progress isn't it. + _previewed = bool(response.get("response_previewed")) + _content_delivered = bool( + _sc and getattr(_sc, "final_content_delivered", False) + ) + # A *successful* finalize edit can still carry only the last preview snapshot, and both + # suppression flags reflect call success, not content. Reconcile against the recorded + # turn-final payload: on mismatch (False, incl. payload-less split delivery) neither flag + # may suppress the final send; None (no record) keeps legacy trust. + _stale_finalized = False + if _content_delivered and not _is_empty_sentinel: + _matcher = getattr(_sc, "delivered_final_matches", None) + if callable(_matcher): + try: + _stale_finalized = _matcher(_final) is False + except Exception: + _stale_finalized = False + if _stale_finalized: + _content_delivered = False + # Plugin hooks (e.g. transform_llm_output) may append content after streaming finished — when + # transformed, always send the final version so the appended content reaches the client. + _transformed = bool(response.get("response_transformed")) + # Suppress the normal send only when the actual final reply reached the user (streamed, or + # interim preview of that *exact* text); commentary shown during a compression/split isn't it. + _streamed = self._run_agent_stream_confirmed_final_delivery( + _sc, + _final, + previewed=_previewed, + ) + if not _is_empty_sentinel and not _transformed and (_streamed or _content_delivered): + logger.info( + "Suppressing normal final send for session %s: final delivery already confirmed (streamed=%s previewed=%s content_delivered=%s).", + session_key or "?", + _streamed, + _previewed, + _content_delivered, + ) + response["already_sent"] = True + elif not _is_empty_sentinel and not _transformed and _stale_finalized and _sc is not None: + # Stale finalize: the streamed message holds only the last preview snapshot. Edit it + # up to the complete response; on edit failure leave already_sent unset so the normal + # send delivers. Not for split delivery: message_id is only the LAST chunk, so editing + # it would repeat every sealed head chunk — fall through to the normal send. + _sc_msg_id = _sc.message_id + _sc_adapter = getattr(_sc, "adapter", None) + if getattr(_sc, "_turn_split_delivery", False): + logger.info( + "Stale streamed finalize detected for session %s on a multi-message split; skipping the in-place reconciliation edit and delivering the complete response via normal final send (#78541).", + session_key or "?", + ) + elif _sc_msg_id and _sc_msg_id != "__no_edit__" and _sc_adapter is not None: + try: + _reconcile_res = await _sc_adapter.edit_message( + chat_id=source.chat_id, + message_id=_sc_msg_id, + content=_final, + finalize=True, + ) + if getattr(_reconcile_res, "success", True): + response["already_sent"] = True + logger.info( + "Reconciled stale streamed finalize for session %s: edited message %s with the complete response (#71643).", + session_key or "?", _sc_msg_id, + ) + else: + logger.warning( + "Stale-finalize reconciliation edit failed for session %s (%s); sending complete response via normal final send.", + session_key or "?", + getattr(_reconcile_res, "error", None), + ) + except Exception as _edit_err: + logger.warning( + "Stale-finalize reconciliation edit failed for session %s: %s; sending complete response via normal final send.", + session_key or "?", _edit_err, + ) + else: + logger.info( + "Stale streamed finalize detected for session %s with no editable message; delivering complete response via normal final send (#71643).", + session_key or "?", + ) + elif not _is_empty_sentinel and _transformed and _sc is not None: + # Plugin hooks transformed the response after streaming — edit the + # existing streamed message instead of sending a duplicate. + _sc_msg_id = _sc.message_id + if _sc_msg_id: + try: + await _sc.adapter.edit_message( + chat_id=source.chat_id, + message_id=_sc_msg_id, + content=response["final_response"], + finalize=True, + ) + response["already_sent"] = True + logger.info( + "Edited streamed message %s for session %s to include plugin-transformed content.", + _sc_msg_id, session_key or "?", + ) + except Exception as _edit_err: + logger.warning( + "Failed to edit streamed message for session %s: %s", + session_key or "?", _edit_err, + ) + elif _sc is not None and not _is_empty_sentinel: + # DUPLICATE-RISK DIAGNOSTIC: a stream consumer existed for this turn but suppression + # did NOT fire, so the gateway's normal final-send is about to run. Log the decision + # inputs so a recurrence can be pinned to "signal never set" vs "ack-pending race". + logger.warning( + "Normal final-send NOT suppressed despite active stream " + "consumer for session %s: streamed=%s previewed=%s " + "content_delivered=%s transformed=%s final_len=%d — " + "possible duplicate send (see wecom ack-timeout RCA).", + session_key or "?", + _streamed, + _previewed, + _content_delivered, + _transformed, + len(_final), + ) + + def _run_agent_schedule_bubble_cleanup( + self, + response: Any, + _cleanup_progress: bool, + _cleanup_adapter: Any, + _cleanup_msg_ids: List[str], + source: SessionSource, + session_key: Optional[str], + run_generation: Optional[int], + ) -> None: + # Schedule deletion of tracked temporary progress bubbles after the final response lands; failed + # runs keep them as breadcrumbs. Only on adapters with ``delete_message``; failures swallowed. + from gateway.run import safe_schedule_threadsafe + if ( + _cleanup_progress + and _cleanup_adapter is not None + and _cleanup_msg_ids + and session_key + and isinstance(response, dict) + and not response.get("failed") + and hasattr(_cleanup_adapter, "register_post_delivery_callback") + ): + _ids_snapshot = list(_cleanup_msg_ids) + _chat_id_snapshot = source.chat_id + _adapter_snapshot = _cleanup_adapter + _loop_snapshot = asyncio.get_running_loop() + + def _cleanup_temp_bubbles() -> None: + async def _delete_all() -> None: + for _mid in _ids_snapshot: + with suppress(Exception): + await _adapter_snapshot.delete_message( + _chat_id_snapshot, _mid + ) + with suppress(Exception): + safe_schedule_threadsafe( + _delete_all(), _loop_snapshot, + logger=logger, + log_message="Temp bubble cleanup scheduling error", + ) + + try: + _cleanup_adapter.register_post_delivery_callback( + session_key, + _cleanup_temp_bubbles, + generation=run_generation, + ) + except Exception as _rpe: + logger.debug("Post-delivery cleanup registration failed: %s", _rpe) + + def _run_agent_bind_turn_wiring( + self, + turn_ctx: TurnContext, + turn_runner: TurnRunner, + source: SessionSource, + event_message_id: Optional[str], + _progress_metadata: Optional[dict], + _progress_reply_to: Optional[str], + _progress_thread_id: Any, + _relay_prospective_thread_id: Optional[str], + ) -> Optional[Dict[str, Any]]: + """Publish progress metadata, result holders and the sync→async bridges onto ``turn_ctx``. + + Returns ``_status_thread_metadata``; the holders are read back via ``turn_ctx.*_holder``. + """ + # Extracted to TurnRunner.send_progress_messages; the threading metadata above is published + # onto the shared TurnContext where the original closure's captured locals were bound. + turn_ctx._progress_metadata = _progress_metadata + turn_ctx._progress_reply_to = _progress_reply_to + + # We need to share the agent instance for interrupt support + agent_holder = [None] # Mutable container for the agent instance + turn_ctx.agent_holder = agent_holder + result_holder = [None] # Mutable container for the result + tools_holder = [None] # Mutable container for the tool definitions + stream_consumer_holder = [None] # Mutable container for stream consumer + # streaming PCM audio consumer. Created on the gateway event-loop thread (NOT in run_sync's + # executor worker) so outer finalisation / interrupt paths can reference it without a NameError. + streaming_tts_consumer_holder: list = [None] + turn_ctx.result_holder = result_holder + turn_ctx.tools_holder = tools_holder + turn_ctx.stream_consumer_holder = stream_consumer_holder + turn_ctx.streaming_tts_consumer_holder = streaming_tts_consumer_holder + + # Bridge sync step_callback → async hooks.emit for agent:step events + _loop_for_step = asyncio.get_running_loop() + _hooks_ref = self.hooks + + # Bridge extracted to TurnRunner._step_callback_sync; the loop and + # hooks refs bound just above are published at their original site. + turn_ctx._loop_for_step = _loop_for_step + turn_ctx._hooks_ref = _hooks_ref + turn_ctx._step_callback_sync = turn_runner._step_callback_sync + + # Bridge sync event_callback → async hooks.emit for lifecycle events (e.g. session:compress + # after a compression split); extracted to TurnRunner._event_callback_sync. + turn_ctx._event_callback_sync = turn_runner._event_callback_sync + + # Bridge sync status_callback → async adapter.send for context pressure + _status_adapter = self._adapter_for_source(source) + _status_chat_id = source.chat_id + _status_thread_metadata = self._run_agent_status_thread_metadata( + source, event_message_id, _progress_thread_id, _relay_prospective_thread_id, + ) + + # Bridge extracted to TurnRunner._status_callback_sync; publish the status wiring computed + # above onto the shared TurnContext at the exact original binding site. + turn_ctx._status_adapter = _status_adapter + turn_ctx._status_chat_id = _status_chat_id + turn_ctx._status_thread_metadata = _status_thread_metadata + turn_ctx._status_callback_sync = turn_runner._status_callback_sync + return _status_thread_metadata + + async def _run_agent_notify_long_running( + self, + disp: "GatewayRunner._RunAgentDisplay", + *, + source: SessionSource, + session_key: Optional[str], + agent_holder: list, + _executor_task_holder: list, + _NOTIFY_INTERVAL: Optional[float], + _long_running_mode: str, + _notify_start: float, + _status_thread_metadata: Optional[Dict[str, Any]], + _cleanup_progress: bool, + _cleanup_msg_ids: List[str], + ) -> None: + """Periodic \"still working\" heartbeat (edited in place where the adapter supports it). + + ``_executor_task_holder[0]`` is populated once the executor future exists; tolerate the + brief window before then (it reads as None). + """ + from gateway.run import _interim_metadata, _non_conversational_metadata + if _NOTIFY_INTERVAL is None: + return # Notifications disabled (gateway_notify_interval: 0) + _notify_adapter = self._adapter_for_source(source) + if not _notify_adapter: + return + # Track the heartbeat message id to edit in place where supported (Telegram, Discord, + # Slack, ...) instead of a new "Still working" bubble every interval. + _heartbeat_msg_id: Optional[str] = None + while True: + await asyncio.sleep(_NOTIFY_INTERVAL) + # Stop heartbeating once this run no longer owns the session slot or the executor has + # finished, else a stale "running: delegate_task" bubble outlives its run. _executor_task + # is bound just after this task is scheduled; tolerate the brief window before then. + _exec_ref = _executor_task_holder[0] + if not self._should_emit_long_running_notification( + session_key, agent_holder[0], _exec_ref + ): + break + _elapsed_mins = int((time.time() - _notify_start) // 60) + # Default heartbeat is terse (elapsed + current tool); the verbose iteration counter is + # gated on busy_ack_detail so users can opt in per platform. + _agent_ref = agent_holder[0] + _status_detail = "" + _want_iteration_detail = bool( + disp.resolve_display_setting( + disp.user_config, + disp.platform_key, + "busy_ack_detail", + True, + ) + ) + if _agent_ref and hasattr(_agent_ref, "get_activity_summary"): + try: + _a = _agent_ref.get_activity_summary() + _parts = [] + if _want_iteration_detail: + _parts.append( + f"iteration {_a['api_call_count']}/{_a['max_iterations']}" + ) + _action = _a.get("current_tool") or _a.get("last_activity_desc") + if _action: + _parts.append(str(_action)) + if _parts: + _status_detail = " — " + ", ".join(_parts) + except Exception: + pass + _heartbeat_text = ( + disp._generic_status_phrase("status") + if _long_running_mode == "generic" + else f"⏳ Working — {_elapsed_mins} min{_status_detail}" + ) + try: + _notify_res = None + if _heartbeat_msg_id: + try: + _notify_res = await _notify_adapter.edit_message( + source.chat_id, + _heartbeat_msg_id, + _heartbeat_text, + ) + except Exception as _ee: + logger.debug("Heartbeat edit failed: %s", _ee) + _notify_res = None + if not (_notify_res and getattr(_notify_res, "success", False)): + _notify_res = await _notify_adapter.send( + source.chat_id, + _heartbeat_text, + metadata=_interim_metadata(_non_conversational_metadata(_status_thread_metadata, platform=source.platform)), + ) + if getattr(_notify_res, "success", False) and getattr( + _notify_res, "message_id", None + ): + _heartbeat_msg_id = str(_notify_res.message_id) + if _cleanup_progress: + _cleanup_msg_ids.append(_heartbeat_msg_id) + except Exception as _ne: + logger.debug("Long-running notification error: %s", _ne) + + async def _run_agent_inner( + self, + message: str, + context_prompt: str, + history: List[Dict[str, Any]], + source: SessionSource, + session_id: str, + session_key: str = None, + run_generation: Optional[int] = None, + _interrupt_depth: int = 0, + event_message_id: Optional[str] = None, + inbound_message_id: Optional[str] = None, + channel_prompt: Optional[str] = None, + moa_config: Optional[dict] = None, + persist_user_message: Optional[Any] = None, + persist_user_timestamp: Optional[float] = None, + persist_user_display_kind: Optional[str] = None, + message_type: Optional[str] = None, + ) -> Dict[str, Any]: + """Run the agent; returns the full run_conversation result dict. + + Keys: "final_response", "messages", "api_calls", "completed". + """ + from gateway.run import _float_env + # ---- Proxy mode: delegate to remote API server ---- + if self._get_proxy_url(): + return await self._run_agent_via_proxy( + message=message, + context_prompt=context_prompt, + history=history, + source=source, + session_id=session_id, + session_key=session_key, + run_generation=run_generation, + event_message_id=event_message_id, + ) + + from run_agent import AIAgent + + disp = self._run_agent_display_settings(source) + _display_surface_mode = disp._display_surface_mode + needs_progress_queue = disp.needs_progress_queue + log_mode_enabled = disp.log_mode_enabled + log_queue = disp.log_queue + + turn_ctx, turn_runner, _cleanup_adapter = self._run_agent_build_turn_context( + disp, + AIAgent, + message=message, + context_prompt=context_prompt, + history=history, + source=source, + session_id=session_id, + session_key=session_key, + run_generation=run_generation, + _interrupt_depth=_interrupt_depth, + event_message_id=event_message_id, + inbound_message_id=inbound_message_id, + channel_prompt=channel_prompt, + moa_config=moa_config, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + ) + _cleanup_progress = turn_ctx._cleanup_progress + _cleanup_msg_ids = turn_ctx._cleanup_msg_ids + + ( + _progress_metadata, + _progress_reply_to, + _progress_thread_id, + _relay_prospective_thread_id, + ) = self._run_agent_progress_threading(source, event_message_id, disp._native_slack_task_cards) + + _status_thread_metadata = self._run_agent_bind_turn_wiring( + turn_ctx, + turn_runner, + source, + event_message_id, + _progress_metadata, + _progress_reply_to, + _progress_thread_id, + _relay_prospective_thread_id, + ) + send_progress_messages = turn_runner.send_progress_messages + agent_holder = turn_ctx.agent_holder + result_holder = turn_ctx.result_holder + tools_holder = turn_ctx.tools_holder + stream_consumer_holder = turn_ctx.stream_consumer_holder + streaming_tts_consumer_holder = turn_ctx.streaming_tts_consumer_holder + + self._run_agent_start_streaming_tts( + source, message_type, _status_thread_metadata, streaming_tts_consumer_holder, + ) + + # run_sync extracted to TurnRunner.run_sync (bound method; executor call unchanged). Its + # closed-over locals travel on turn_ctx; `nonlocal message` rebinds became ctx.message writes. + run_sync = turn_runner.run_sync + + # Start the progress sender if enabled. Gate on needs_progress_queue (tool_progress OR + # thinking_progress), not tool_progress alone: the sender drains BOTH tool-progress lines and + # _thinking scratch bubbles — a tool_progress-only gate left thinking-only queues never drained. + progress_task = None + if needs_progress_queue: + progress_task = asyncio.create_task(send_progress_messages()) + + # Start the tool-call log writer when tool_progress == "log". + log_task = None + if log_mode_enabled: + log_task = asyncio.create_task(self._run_agent_write_tool_log(log_queue)) + + # Start stream consumer task — polls for consumer creation since it + # happens inside run_sync (thread pool) after the agent is constructed. + stream_task = None + stream_task = asyncio.create_task(self._run_agent_stream_consumer_task(stream_consumer_holder)) + + # Track this agent as running for this session (for interrupt support) + # We do this in a callback after the agent is created + tracking_task = asyncio.create_task( + self._run_agent_track_agent(session_key, run_generation, agent_holder) + ) + + _interrupt_detected = asyncio.Event() # shared with backup check + interrupt_monitor = asyncio.create_task( + self._run_agent_monitor_for_interrupt( + source, + session_key, + agent_holder, + _interrupt_detected, + streaming_tts_consumer_holder, + ) + ) + + # Periodic "still working" notifications so the user knows the agent hasn't died. Config: + # agent.gateway_notify_interval or HERMES_AGENT_NOTIFY_INTERVAL env; default 180s. + _NOTIFY_INTERVAL_RAW = _float_env("HERMES_AGENT_NOTIFY_INTERVAL", 180) + _NOTIFY_INTERVAL = _NOTIFY_INTERVAL_RAW if _NOTIFY_INTERVAL_RAW > 0 else None + _long_running_mode = _display_surface_mode( + "long_running_notifications", + default=True, + allow_generic=True, + ) + if _long_running_mode == "off": + _NOTIFY_INTERVAL = None + _notify_start = time.time() + _executor_task_holder: list = [None] # bound once the executor future exists (see below) + _notify_task = asyncio.create_task( + self._run_agent_notify_long_running( + disp, + source=source, + session_key=session_key, + agent_holder=agent_holder, + _executor_task_holder=_executor_task_holder, + _NOTIFY_INTERVAL=_NOTIFY_INTERVAL, + _long_running_mode=_long_running_mode, + _notify_start=_notify_start, + _status_thread_metadata=_status_thread_metadata, + _cleanup_progress=_cleanup_progress, + _cleanup_msg_ids=_cleanup_msg_ids, + ) + ) + + try: + worker = self._run_agent_start_turn_worker( + turn_ctx, run_sync, agent_holder, session_id, session_key, run_generation, + ) + _executor_task_holder[0] = worker.executor_task # read late by _notify_long_running + response = await self._run_agent_await_turn_worker( + worker, + source=source, + session_key=session_key, + agent_holder=agent_holder, + result_holder=result_holder, + tools_holder=tools_holder, + _interrupt_detected=_interrupt_detected, + interrupt_monitor=interrupt_monitor, + streaming_tts_consumer_holder=streaming_tts_consumer_holder, + _status_thread_metadata=_status_thread_metadata, + ) + + self._run_agent_evict_on_fallback(session_key, agent_holder, result_holder) + + # Check if we were interrupted OR have a queued message (/queue). + result = result_holder[0] + adapter = self._adapter_for_source(source) + + await self._run_agent_finalize_streaming_tts( + streaming_tts_consumer_holder, adapter, session_key, run_generation, + ) + + pending_event, pending = await self._run_agent_drain_pending( + result, adapter, source, session_key, + ) + + if pending_event or pending: + return await self._run_agent_queued_followup( + source=source, + adapter=adapter, + session_id=session_id, + session_key=session_key, + run_generation=run_generation, + _interrupt_depth=_interrupt_depth, + event_message_id=event_message_id, + context_prompt=context_prompt, + history=history, + pending=pending, + pending_event=pending_event, + response=response, + result=result, + result_holder=result_holder, + stream_consumer_holder=stream_consumer_holder, + stream_task=stream_task, + _status_thread_metadata=_status_thread_metadata, + ) + finally: + await self._run_agent_cleanup_turn_tasks( + progress_task=progress_task, + log_task=log_task, + interrupt_monitor=interrupt_monitor, + _notify_task=_notify_task, + tracking_task=tracking_task, + stream_task=stream_task, + stream_consumer_holder=stream_consumer_holder, + streaming_tts_consumer_holder=streaming_tts_consumer_holder, + session_key=session_key, + run_generation=run_generation, + ) + + await self._run_agent_mark_streamed_delivery( + response, stream_consumer_holder, source, session_key, + ) + self._run_agent_schedule_bubble_cleanup( + response, + _cleanup_progress, + _cleanup_adapter, + _cleanup_msg_ids, + source, + session_key, + run_generation, + ) + + return response diff --git a/gateway/run_turn_runner.py b/gateway/run_turn_runner.py new file mode 100644 index 0000000000..edea61702c --- /dev/null +++ b/gateway/run_turn_runner.py @@ -0,0 +1,2402 @@ +"""Per-turn callback runner (progress/status/voice/run_sync) for the gateway agent turn. + +Split out of ``gateway/run.py``; ``TurnRunner`` owns the per-turn callbacks/closures ``GatewayRunner._run_agent_inner`` binds. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import dataclasses +import inspect +import json +import queue +import re +import threading +import time +from agent.replay_cleanup import strip_stale_dangerous_confirmations +from contextlib import suppress +from datetime import datetime +from gateway.config import Platform +from gateway.media_repair import repair_explicit_computer_use_media_paths +from gateway.platforms.base import BasePlatformAdapter +from gateway.turn_context import TurnContext +from hermes_cli.config import cfg_get +from typing import Any, Dict, List, Optional +from utils import is_truthy_value + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class TurnRunner: + """Per-turn collaborator carrying ``GatewayRunner._run_agent_inner``'s tool-progress callbacks. + + Module-global references (logger, cfg_get, BasePlatformAdapter, ...) resolve in this module. + """ + + def __init__(self, runner: "GatewayRunner", ctx: TurnContext) -> None: + self._runner = runner + self._ctx = ctx + + def progress_callback(self, event_type: str, tool_name: str = None, preview: str = None, args: dict = None, **kwargs): + """Callback invoked by agent on tool lifecycle events.""" + from gateway.run import _hermes_home, _load_gateway_config, safe_schedule_threadsafe + ctx = self._ctx + # Failed subagent → one clean user-facing notice, handled FIRST, before every progress-queue + # gate: platforms with tool_progress off must still hear about a dead delegation. Only + # terminal failure statuses render (same notice rail as credit warnings); success/interrupt + # stay quiet. + if event_type == "subagent.complete": + _sub_status = kwargs.get("status") + try: + from tools.delegate_tool import ( + SUBAGENT_FAILURE_STATUSES, + format_subagent_failure_line, + ) + if _sub_status in SUBAGENT_FAILURE_STATUSES and ctx._run_still_current(): + _line = format_subagent_failure_line( + kwargs.get("goal"), + _sub_status, + error=kwargs.get("summary") or preview, + duration_seconds=kwargs.get("duration_seconds"), + ) + safe_schedule_threadsafe( + self._runner._deliver_platform_notice(ctx.source, _line), + ctx._loop_for_step, + logger=logger, + log_message="subagent failure notice scheduling error", + ) + except Exception: + logger.debug("subagent failure notice failed", exc_info=True) + return + # Live status line (Slack assistant status): stash the tool phrase on the adapter; the + # _keep_typing refresh renders it. Plain dict write, safe from the sync worker thread. + if ( + ctx._live_status_adapter is not None + and ctx._live_status_mode != "off" + and tool_name != "_thinking" + ): + try: + if event_type == "tool.started" and tool_name and ctx._run_still_current(): + from agent.display import build_status_phrase + _phrase = build_status_phrase( + tool_name, + args if ctx._live_status_mode == "full" else None, + ) + ctx._live_status_adapter.set_status_text(ctx.source.chat_id, _phrase) + elif event_type == "tool.completed": + # Between tools the model is genuinely "thinking" + # again — revert to the static default. + ctx._live_status_adapter.set_status_text(ctx.source.chat_id, None) + except Exception as _ls_err: + logger.debug("live status update failed: %s", _ls_err) + # "log" mode: append tool.started lines to the log queue, silent in chat. Handled before + # the progress_queue guard because log mode runs without a chat progress queue. + if ctx.log_queue is not None: + if event_type == "tool.started" and tool_name and tool_name != "_thinking": + ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + preview_str = f' "{preview}"' if preview else "" + ctx.log_queue.put(f"{ts} {tool_name}:{preview_str}".rstrip()) + if not ctx.progress_queue: + return + if not ctx.progress_queue or not ctx._run_still_current(): + return + + # First-touch onboarding: the first time a tool exceeds _LONG_TOOL_THRESHOLD_S while + # streaming every tool (progress_mode == "all"), append a one-time /verbose hint. + if event_type == "tool.completed" and not ctx.long_tool_hint_fired[0]: + try: + duration = kwargs.get("duration") or 0 + if duration >= ctx._LONG_TOOL_THRESHOLD_S and ctx.progress_mode == "all": + from agent.onboarding import ( + TOOL_PROGRESS_FLAG, + is_seen, + mark_seen, + tool_progress_hint_gateway, + ) + _cfg = _load_gateway_config() + gate_on = is_truthy_value( + cfg_get(_cfg, "display", "tool_progress_command"), + default=False, + ) + if gate_on and not is_seen(_cfg, TOOL_PROGRESS_FLAG): + ctx.long_tool_hint_fired[0] = True + ctx.progress_queue.put(tool_progress_hint_gateway()) + mark_seen(_hermes_home / "config.yaml", TOOL_PROGRESS_FLAG) + except Exception as _hint_err: + logger.debug("tool-progress onboarding hint failed: %s", _hint_err) + return + + # "_thinking" is assistant scratch text between tool calls. It is never ordinary tool + # progress: only relay it when the platform explicitly opted into thinking_progress. + if event_type == "_thinking" or tool_name == "_thinking": + if not ctx._thinking_enabled: + return + thinking_text = preview if tool_name == "_thinking" else tool_name + msg = f"💬 {thinking_text}" if thinking_text else None + if msg: + ctx.progress_queue.put(msg) + return + + # Native task cards consume the ID-bearing tool_start/tool_complete callbacks instead; + # name-correlated text events would duplicate cards and mispair concurrent same-tool calls. + if ctx._native_slack_task_cards and event_type in { + "tool.started", + "tool.completed", + }: + return + + # If tool_progress is off, only _thinking passes through (above). + # Regular tool calls are suppressed. + if not ctx.tool_progress_enabled: + return + + # Only act on tool.started events (ignore tool.completed, reasoning.available, etc.) + if event_type not in {"tool.started",}: + return + + # Never render a progress bubble for clarify: send_clarify IS the user-facing rendering, so + # a bubble is duplication, and verbose mode would dump the raw tool-call args JSON, which + # (progress queue drains on a background task) lands right under the rendered prompt. + if tool_name == "clarify": + return + + # Suppress tool-progress bubbles once the user sent `stop`: N parallel tool calls fire N + # "tool.started" events before the interrupt check, so a late `stop` would still render + # all N bubbles. (agent_holder[0] is the shared agent handle across nested scopes.) + try: + _agent_for_interrupt = ctx.agent_holder[0] if ctx.agent_holder else None + if _agent_for_interrupt is not None and getattr( + _agent_for_interrupt, "is_interrupted", False + ): + return + except Exception: + pass + + # "new" mode: only report when tool changes + if ctx.progress_mode == "new" and tool_name == ctx.last_tool[0]: + return + ctx.last_tool[0] = tool_name + + # Build progress message with primary argument preview + from agent.display import get_tool_emoji + emoji = get_tool_emoji(tool_name, default="⚙️") + + # Markdown platforms (``supports_code_blocks``) fence terminal commands; plain-text ones + # keep the compact `terminal: "cmd…"` line. No language tag: Slack mrkdwn renders it as a + # literal first code line. Verbose shows the FULL command; "all"/"new" fence but truncate to + # one line capped at ``tool_preview_length`` (default 40), the non-terminal preview budget. + _code_block_full = None + _code_block_short = None + try: + _progress_adapter = self._runner._adapter_for_source(ctx.source) + except Exception: + _progress_adapter = None + if ( + getattr(_progress_adapter, "supports_code_blocks", False) + and tool_name == "terminal" + and isinstance(args, dict) + and isinstance(args.get("command"), str) + and args["command"].strip() + ): + from agent.display import get_tool_preview_max_len + _cmd_full = args["command"].rstrip() + # Consecutive terminal calls drop the repeated "💻 terminal" header so back-to-back + # commands render as adjacent code blocks under one header. + _block_header = ( + "" if ctx.last_was_terminal_block[0] else f"{emoji} {tool_name}\n" + ) + _code_block_full = f"{_block_header}```\n{_cmd_full}\n```" + # Single-line, capped preview for non-verbose modes. + _pl = get_tool_preview_max_len() + _cap = _pl if _pl > 0 else 40 + _lines = _cmd_full.splitlines() + _cmd_short = _lines[0] if _lines else _cmd_full + _multiline = len(_lines) > 1 + if len(_cmd_short) > _cap: + _cmd_short = _cmd_short[:_cap - 3] + "..." + elif _multiline: + _cmd_short = _cmd_short + " ..." + _code_block_short = f"{_block_header}```\n{_cmd_short}\n```" + + # Verbose mode: show detailed arguments, respects tool_preview_length + if ctx.progress_mode == "verbose": + if _code_block_full is not None: + ctx.last_was_terminal_block[0] = True + ctx.progress_queue.put(_code_block_full) + return + ctx.last_was_terminal_block[0] = False + if args: + from agent.display import get_tool_preview_max_len + _pl = get_tool_preview_max_len() + args_str = json.dumps(args, ensure_ascii=False, default=str) + # tool_preview_length 0 (default) = no truncation in verbose mode; the user asked + # for full detail and platform message-length limits handle the rest. + if _pl > 0 and len(args_str) > _pl: + args_str = args_str[:_pl - 3] + "..." + msg = f"{emoji} {tool_name}({list(args.keys())})\n{args_str}" + elif preview: + msg = f"{emoji} {tool_name}: \"{preview}\"" + else: + msg = f"{emoji} {tool_name}..." + ctx.progress_queue.put(msg) + return + + # "all" / "new" modes: short preview capped by tool_preview_length (default 40; gateway + # messages persist, unlike CLI spinners). Markdown terminal commands use the fence above. + if _code_block_short is not None: + msg = _code_block_short + ctx.last_was_terminal_block[0] = True + elif preview: + from agent.display import ( + get_tool_preview_max_len, + get_tool_verb, + prepare_tool_preview, + tool_verb_connector, + verb_drops_preview, + ) + _pl = get_tool_preview_max_len() + _cap = _pl if _pl > 0 else 40 + _prepared_preview = prepare_tool_preview( + tool_name, + args, + fallback=preview, + max_len=_cap, + ) + if _progress_adapter is not None: + preview = _progress_adapter.format_tool_preview(_prepared_preview) + else: + preview = _prepared_preview.text + # Friendly labels: human-phrased line for built-in tools ("🔍 Searching the web for ...") + # by prefixing the verb onto the computed preview, so the command/url/query is kept. + _verb = get_tool_verb(tool_name) + if _verb: + if verb_drops_preview(tool_name): + msg = f"{emoji} {_verb}" + else: + msg = f"{emoji} {_verb}{tool_verb_connector(tool_name)}{preview}" + else: + msg = f"{emoji} {tool_name}: \"{preview}\"" + ctx.last_was_terminal_block[0] = False + else: + msg = f"{emoji} {tool_name}..." + ctx.last_was_terminal_block[0] = False + + # Dedup consecutive identical progress messages (common with execute_code: same + # boilerplate imports → identical previews). + if msg == ctx.last_progress_msg[0]: + ctx.repeat_count[0] += 1 + # Native-stream-progress routing: dedup updates the last line + # in the overlay rather than sending a queue signal. + _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None + if _sc is not None and getattr(_sc, "accepts_tool_progress", False): + # Replace the last progress line with the dedup version + _sc.on_tool_progress(f"{msg} (×{ctx.repeat_count[0] + 1})") + return + # Update the last line in progress_lines with a counter + # via a special "dedup" queue message. + ctx.progress_queue.put(("__dedup__", msg, ctx.repeat_count[0])) + return + ctx.last_progress_msg[0] = msg + ctx.repeat_count[0] = 0 + + # If the stream consumer is active with native streaming, inject progress into the stream + # bubble instead of the separate progress queue. + _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None + if _sc is not None and getattr(_sc, "accepts_tool_progress", False): + _sc.on_tool_progress(msg) + return + + ctx.progress_queue.put(msg) + + async def _send_native_task_card_progress(self, adapter) -> None: + """Drain the progress queue into Slack-native plan/task cards. + + On any native failure, fall back to an editable in-thread message so progress stays live. + """ + ctx = self._ctx + tasks: Dict[str, Dict[str, str]] = {} + task_order: List[str] = [] + fallback_msg_id: Optional[str] = None + native_failed = False + anonymous_seq = 0 + + def _compact(value: Any, limit: int = 120) -> str: + text = re.sub(r"\s+", " ", str(value or "")).strip() + if len(text) <= limit: + return text + return text[: limit - 3].rstrip() + "..." + + def _visible_tasks() -> List[Dict[str, str]]: + return [tasks[task_id] for task_id in task_order[-8:]] + + def _fallback_text() -> str: + labels = { + "in_progress": "running", + "complete": "complete", + "error": "error", + } + lines = [ + f"- {task['title']} - {labels.get(task['status'], task['status'])}" + for task in _visible_tasks() + ] + return "Hermes is working\n" + "\n".join(lines) + + def _apply_native_event(raw: Any) -> bool: + nonlocal anonymous_seq + if not isinstance(raw, dict): + return False + event_type = raw.get("type") + if event_type not in {"tool.started", "tool.completed"}: + return False + call_id = str(raw.get("tool_call_id") or "") + if not call_id: + anonymous_seq += 1 + call_id = f"anonymous_{anonymous_seq}" + tool_name = str(raw.get("tool_name") or "tool") + + if event_type == "tool.started": + title = tool_name + preview = _compact(raw.get("preview"), 64) + if preview: + title = f"{tool_name} - {preview}" + if call_id not in tasks: + task_order.append(call_id) + tasks[call_id] = { + "id": call_id, + "title": _compact(title), + "status": "in_progress", + } + return True + + task = tasks.get(call_id) + if task is None: + # Completion-only events are rare but valid on some runtimes; keep their real ID + # instead of guessing a same-name pending call. + task = { + "id": call_id, + "title": _compact(tool_name), + "status": "in_progress", + } + tasks[call_id] = task + task_order.append(call_id) + task["status"] = "error" if raw.get("is_error") else "complete" + return True + + async def _send_or_edit_fallback() -> None: + nonlocal fallback_msg_id + text = _fallback_text() + if fallback_msg_id: + result = await adapter.edit_message( + chat_id=ctx.source.chat_id, + message_id=fallback_msg_id, + content=text, + metadata=ctx._progress_metadata, + ) + if getattr(result, "success", False): + return + result = await adapter.send( + chat_id=ctx.source.chat_id, + content=text, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + if getattr(result, "success", False) and getattr( + result, "message_id", None + ): + fallback_msg_id = str(result.message_id) + if ctx._cleanup_progress: + ctx._cleanup_msg_ids.append(fallback_msg_id) + + async def _publish_native_progress() -> None: + nonlocal native_failed + if not tasks: + return + if not native_failed: + result = await adapter.send_native_task_card_progress( + chat_id=ctx.source.chat_id, + tasks=_visible_tasks(), + title="Hermes is working", + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + fallback_text=_fallback_text(), + ) + if getattr(result, "success", False): + return + native_failed = True + logger.warning( + "Slack native task-card progress failed; falling back " + "to an editable text update: %s", + getattr(result, "error", "unknown error"), + ) + # Once the native rail fails, every later lifecycle event + # edits the same fallback message so progress remains live. + await _send_or_edit_fallback() + + def _drain_native_queue() -> bool: + changed = False + while True: + try: + changed = _apply_native_event( + ctx.progress_queue.get_nowait() + ) or changed + except queue.Empty: + return changed + except Exception: + logger.debug( + "Slack native progress queue drain failed", + exc_info=True, + ) + return changed + + def _agent_interrupted() -> bool: + try: + _agent = ctx.agent_holder[0] if ctx.agent_holder else None + return bool( + _agent is not None and getattr(_agent, "is_interrupted", False) + ) + except Exception: + return False + + try: + while True: + if not ctx._run_still_current(): + return + try: + raw = ctx.progress_queue.get_nowait() + except queue.Empty: + await asyncio.sleep(0.1) + continue + + if _agent_interrupted(): + continue + + if _apply_native_event(raw): + await _publish_native_progress() + except asyncio.CancelledError: + if _drain_native_queue() and ctx._run_still_current(): + if not _agent_interrupted(): + await _publish_native_progress() + return + finally: + if hasattr(adapter, "stop_native_task_card_progress"): + # Best-effort on the turn-cleanup path: an escaping transport exception would skip + # final-delivery logic (cleanup awaits catch only CancelledError). + try: + await adapter.stop_native_task_card_progress( + ctx.source.chat_id, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + except asyncio.CancelledError: + raise + except Exception: + logger.debug( + "task-card stop failed during turn cleanup", + exc_info=True, + ) + + @dataclasses.dataclass + class _ProgressEditState: + """Mutable editable-bubble state shared by ``send_progress_messages`` and its helpers.""" + adapter: Any + progress_lines: list + progress_msg_id: Any + can_edit: bool + _progress_len_fn: Any + _PROGRESS_TEXT_LIMIT: int + _edit_accepts_metadata: bool + + def _progress_edit_state(self, adapter) -> "TurnRunner._ProgressEditState": + ctx = self._ctx + progress_lines = [] # Accumulated tool lines for the CURRENT editable bubble + progress_msg_id = None # ID of the current progress message to edit + can_edit = ctx.progress_grouping != "separate" # "separate" = one message per tool (pre-v0.9 behavior) + + _progress_len_fn = ( + adapter.message_len_fn + if isinstance(adapter, BasePlatformAdapter) + else len + ) + try: + _raw_progress_limit = int(getattr(adapter, "MAX_MESSAGE_LENGTH", 4000) or 4000) + except Exception: + _raw_progress_limit = 4000 + # Per-chat resolution (relay adapter fronting N platforms): cap and length unit follow the + # chat's underlying platform; native adapters return their scalar/property unchanged. + if isinstance(adapter, BasePlatformAdapter): + try: + _raw_progress_limit = int( + adapter.max_message_length_for_chat(ctx.source.chat_id) or 4000 + ) + _progress_len_fn = adapter.message_len_fn_for_chat(ctx.source.chat_id) + except Exception: + pass + # Leave a little room for platform quirks / formatting. For tiny + # test adapters keep the limit usable instead of clamping to 500+. + _PROGRESS_TEXT_LIMIT = max( + 1, + _raw_progress_limit - (64 if _raw_progress_limit > 128 else 0), + ) + + # Detect whether the adapter's edit_message accepts metadata so + # overflow edits preserve Telegram topic/thread routing (#27487). + _edit_accepts_metadata = False + if ctx._progress_metadata: + try: + _edit_params = inspect.signature(adapter.edit_message).parameters + _edit_accepts_metadata = ( + "metadata" in _edit_params + or any( + param.kind is inspect.Parameter.VAR_KEYWORD + for param in _edit_params.values() + ) + ) + except (TypeError, ValueError): + _edit_accepts_metadata = False + return self._ProgressEditState( + adapter=adapter, + progress_lines=progress_lines, + progress_msg_id=progress_msg_id, + can_edit=can_edit, + _progress_len_fn=_progress_len_fn, + _PROGRESS_TEXT_LIMIT=_PROGRESS_TEXT_LIMIT, + _edit_accepts_metadata=_edit_accepts_metadata, + ) + + async def _edit_progress_message(self, st, message_id: str, content: str): + ctx = self._ctx + kwargs = { + "chat_id": ctx.source.chat_id, + "message_id": message_id, + "content": content, + } + if getattr(st.adapter, "REQUIRES_EDIT_FINALIZE", False): + kwargs["finalize"] = True + if st._edit_accepts_metadata: + kwargs["metadata"] = ctx._progress_metadata + return await st.adapter.edit_message(**kwargs) + + def _progress_text(self, lines: list) -> str: + return "\n".join(str(line) for line in lines) + + def _split_progress_groups(self, st, lines: list) -> list[list]: + """Partition progress lines into platform-sized editable bubbles.""" + groups: list[list] = [] + current: list = [] + for line in lines: + candidate = current + [line] + if current and st._progress_len_fn(self._progress_text(candidate)) > st._PROGRESS_TEXT_LIMIT: + groups.append(current) + current = [line] + else: + current = candidate + if current: + groups.append(current) + return groups + + def _track_progress_result(self, result) -> None: + ctx = self._ctx + if ( + ctx._cleanup_progress + and getattr(result, "success", False) + and getattr(result, "message_id", None) + ): + ctx._cleanup_msg_ids.append(str(result.message_id)) + + async def _send_progress_text(self, st, text: str): + ctx = self._ctx + result = await st.adapter.send( + chat_id=ctx.source.chat_id, + content=text, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + self._track_progress_result(result) + return result + + async def _roll_progress_overflow_if_needed(self, st) -> bool: + """Start fresh editable progress bubbles before a bubble exceeds limit. + + Returns True when it delivered/split the buffer or a transient edit failure left it + intact for retry — either way the caller skips the normal send/edit path this tick. + """ + if not st.progress_lines or not st.can_edit: + return False + groups = self._split_progress_groups(st, st.progress_lines) + if len(groups) <= 1: + return False + + first_text = self._progress_text(groups[0]) + if st.progress_msg_id is not None: + result = await self._edit_progress_message(st, st.progress_msg_id, first_text) + if not result.success: + if getattr(result, "retryable", False): + logger.debug( + "[%s] Transient overflow edit failure — keeping can_edit=True", + st.adapter.name, + ) + return True + st.can_edit = False + # Fall back to the existing non-edit behavior below. + return False + else: + result = await self._send_progress_text(st, first_text) + if result.success and result.message_id: + st.progress_msg_id = result.message_id + + for group in groups[1:]: + result = await self._send_progress_text(st, self._progress_text(group)) + if result.success and result.message_id: + st.progress_msg_id = result.message_id + + # The newest continuation is the only mutable bubble: keep just its lines so later + # edits update it instead of replaying the full transcript into new messages. + st.progress_lines = groups[-1] + return True + + async def _drain_progress_on_cancel(self, st) -> None: + ctx = self._ctx + # Drain remaining queued messages + while not ctx.progress_queue.empty(): + try: + raw = ctx.progress_queue.get_nowait() + if isinstance(raw, tuple) and len(raw) == 3 and raw[0] == "__dedup__": + _, base_msg, count = raw + if st.progress_lines: + st.progress_lines[-1] = f"{base_msg} (×{count + 1})" + await self._roll_progress_overflow_if_needed(st) + elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": + # Content-bubble marker during drain: close the current progress bubble + # and start a fresh one for tool lines that arrived after. + await self._roll_progress_overflow_if_needed(st) + if st.can_edit and st.progress_lines and st.progress_msg_id: + _pending_text = self._progress_text(st.progress_lines) + with suppress(Exception): + await self._edit_progress_message(st, st.progress_msg_id, _pending_text) + st.progress_msg_id = None + st.progress_lines = [] + ctx.last_progress_msg[0] = None + ctx.repeat_count[0] = 0 + else: + st.progress_lines.append(raw) + await self._roll_progress_overflow_if_needed(st) + except Exception: + break + # Final edit with all remaining tools (only if editing works) + if st.can_edit and st.progress_lines and st.progress_msg_id: + await self._roll_progress_overflow_if_needed(st) + if st.can_edit and st.progress_lines and st.progress_msg_id: + full_text = self._progress_text(st.progress_lines) + with suppress(Exception): + await self._edit_progress_message(st, st.progress_msg_id, full_text) + + async def send_progress_messages(self): + ctx = self._ctx + if not ctx.progress_queue: + return + + adapter = self._runner._adapter_for_source(ctx.source) + if not adapter: + return + + if ctx._native_slack_task_cards and hasattr( + adapter, "send_native_task_card_progress" + ): + await self._send_native_task_card_progress(adapter) + return + + # Skip tool progress for platforms that can't edit messages (e.g. iMessage/BlueBubbles): + # each update would be a separate bubble. getattr, not attribute access: duck-typed + # adapters (test fakes, minimal plugins) may lack edit_message — treated as "can't edit". + _adapter_edit = getattr(type(adapter), "edit_message", None) + if _adapter_edit is None or _adapter_edit is BasePlatformAdapter.edit_message: + while not ctx.progress_queue.empty(): + try: + ctx.progress_queue.get_nowait() + except Exception: + break + return + + st = self._progress_edit_state(adapter) + _last_edit_ts = 0.0 # Throttle edits to avoid Telegram flood control + _PROGRESS_EDIT_INTERVAL = 1.5 # Minimum seconds between edits + + while True: + try: + if not ctx._run_still_current(): + while not ctx.progress_queue.empty(): + try: + ctx.progress_queue.get_nowait() + except Exception: + break + return + + raw = ctx.progress_queue.get_nowait() + + # Drain silently when interrupted: events queued in the window between tool parse + # and interrupt processing should not render as bubbles. + try: + _agent_for_interrupt = ctx.agent_holder[0] if ctx.agent_holder else None + if _agent_for_interrupt is not None and getattr( + _agent_for_interrupt, "is_interrupted", False + ): + # Drop this event and continue draining. + await asyncio.sleep(0) + continue + except Exception: + pass + + # Handle dedup messages: update last line with repeat counter + if isinstance(raw, tuple) and len(raw) == 3 and raw[0] == "__dedup__": + _, base_msg, count = raw + if st.progress_lines: + st.progress_lines[-1] = f"{base_msg} (×{count + 1})" + msg = st.progress_lines[-1] if st.progress_lines else base_msg + elif isinstance(raw, tuple) and len(raw) >= 1 and raw[0] == "__reset__": + # Content bubble landed — close the tool-progress bubble so the next tool starts + # fresh below it; else tool edits hit the ORIGINAL message above (out of order). + st.progress_msg_id = None + st.progress_lines = [] + ctx.last_progress_msg[0] = None + ctx.repeat_count[0] = 0 + continue + else: + msg = raw + st.progress_lines.append(msg) + + if await self._roll_progress_overflow_if_needed(st): + _last_edit_ts = time.monotonic() + await asyncio.sleep(0.3) + if ctx._run_still_current(): + await st.adapter.send_typing(ctx.source.chat_id, metadata=ctx._progress_metadata) + continue + + # Throttle edits: batch rapid tool updates into fewer API calls to avoid Telegram + # flood control (grammY pattern: proactively rate-limit rather than react to 429s). + _now = time.monotonic() + _remaining = _PROGRESS_EDIT_INTERVAL - (_now - _last_edit_ts) + if _remaining > 0: + # Wait out the throttle interval, then loop back to drain any further queued + # messages before sending a single batched edit. + await asyncio.sleep(_remaining) + continue + + if not ctx._run_still_current(): + return + + if st.can_edit and st.progress_msg_id is not None: + # Try to edit the existing progress message + full_text = "\n".join(st.progress_lines) + result = await self._edit_progress_message(st, st.progress_msg_id, full_text) + if not result.success: + _err = (getattr(result, "error", "") or "").lower() + # Transient network errors (ConnectError, timeouts) must not disable editing; + # only permanent failures (flood, not found, permissions) set can_edit = False. + if getattr(result, "retryable", False): + logger.debug( + "[%s] Transient edit failure — keeping can_edit=True", + st.adapter.name, + ) + continue + if "flood" in _err or "retry after" in _err: + # Flood control hit — backoff but keep editing. + # Only disable edits for non-recoverable errors. + logger.info( + "[%s] Progress edit flood control, backing off", + st.adapter.name, + ) + _last_edit_ts = time.monotonic() + else: + st.can_edit = False + _flood_result = await st.adapter.send( + chat_id=ctx.source.chat_id, + content=msg, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + if ( + ctx._cleanup_progress + and getattr(_flood_result, "success", False) + and getattr(_flood_result, "message_id", None) + ): + ctx._cleanup_msg_ids.append(str(_flood_result.message_id)) + else: + if st.can_edit: + # First tool: send all accumulated text as new message + full_text = "\n".join(st.progress_lines) + result = await st.adapter.send( + chat_id=ctx.source.chat_id, + content=full_text, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + else: + # Editing unsupported: send just this line + result = await st.adapter.send( + chat_id=ctx.source.chat_id, + content=msg, + reply_to=ctx._progress_reply_to, + metadata=ctx._progress_metadata, + ) + if result.success and result.message_id: + st.progress_msg_id = result.message_id + if ctx._cleanup_progress: + ctx._cleanup_msg_ids.append(str(result.message_id)) + + _last_edit_ts = time.monotonic() + + # Restore typing indicator + await asyncio.sleep(0.3) + if ctx._run_still_current(): + await st.adapter.send_typing(ctx.source.chat_id, metadata=ctx._progress_metadata) + + except queue.Empty: + await asyncio.sleep(0.3) + except asyncio.CancelledError: + await self._drain_progress_on_cancel(st) + return + except Exception as e: + logger.error("Progress message error: %s", e) + await asyncio.sleep(1) + + def voice_ack_callback(self, call_id, tool_name, args): + """tool_start_callback: speak a one-time ack in the voice channel.""" + from gateway.run import safe_schedule_threadsafe + ctx = self._ctx + if ctx._voice_ack_fired[0] or ctx._voice_ack_guild[0] is None: + return + if not ctx._run_still_current(): + return + ctx._voice_ack_fired[0] = True + _adapter = self._runner.adapters.get(Platform.DISCORD) + if _adapter is None or not hasattr(_adapter, "play_ack_in_voice"): + return + try: + safe_schedule_threadsafe( + _adapter.play_ack_in_voice(ctx._voice_ack_guild[0]), + ctx._voice_ack_loop, + logger=logger, + log_message="voice ack scheduling error", + ) + except Exception as _ack_err: + logger.debug("voice ack schedule failed: %s", _ack_err) + + # ── Slack-native task cards: ID-bearing lifecycle callbacks ── ride agent.tool_start_callback / + # agent.tool_complete_callback so start/completion correlate by the REAL tool-call id; the + # name-correlated progress_callback text events would duplicate cards and mispair concurrent calls. + + def native_tool_start_callback(self, call_id, tool_name, args): + """Queue an ID-correlated native progress start from the agent thread.""" + ctx = self._ctx + if not ctx.progress_queue or not ctx._run_still_current(): + return + try: + _agent = ctx.agent_holder[0] if ctx.agent_holder else None + if _agent is not None and getattr(_agent, "is_interrupted", False): + return + except Exception: + pass + from agent.display import build_tool_preview + + ctx.progress_queue.put( + { + "type": "tool.started", + "tool_call_id": str(call_id or ""), + "tool_name": str(tool_name or "tool"), + "preview": build_tool_preview( + str(tool_name or "tool"), args or {}, max_len=64 + ) + or "", + } + ) + + def native_tool_complete_callback(self, call_id, tool_name, args, result): + """Queue the matching native completion using the real tool-call ID.""" + ctx = self._ctx + if not ctx.progress_queue or not ctx._run_still_current(): + return + try: + _agent = ctx.agent_holder[0] if ctx.agent_holder else None + if _agent is not None and getattr(_agent, "is_interrupted", False): + return + except Exception: + pass + from agent.display import _detect_tool_failure + + is_error, _ = _detect_tool_failure(str(tool_name or "tool"), result) + ctx.progress_queue.put( + { + "type": "tool.completed", + "tool_call_id": str(call_id or ""), + "tool_name": str(tool_name or "tool"), + "is_error": bool(is_error), + } + ) + + def combined_tool_start_callback(self, call_id, tool_name, args): + """Compose the voice ack + native task-card start consumers.""" + ctx = self._ctx + if ctx._voice_ack_guild[0] is not None: + self.voice_ack_callback(call_id, tool_name, args) + if ctx._native_slack_task_cards: + self.native_tool_start_callback(call_id, tool_name, args) + + def _step_callback_sync(self, iteration: int, prev_tools: list) -> None: + from gateway.run import safe_schedule_threadsafe + ctx = self._ctx + if not ctx._run_still_current(): + return + # prev_tools may be list[str] or list[dict] with "name"/"result" keys. Normalise so + # "tool_names" stays backward-compatible for user hooks that do ', '.join(tool_names). + _names: list[str] = [] + for _t in (prev_tools or []): + if isinstance(_t, dict): + _names.append(_t.get("name") or "") + else: + _names.append(str(_t)) + safe_schedule_threadsafe( + ctx._hooks_ref.emit("agent:step", { + "platform": ctx.source.platform.value if ctx.source.platform else "", + "user_id": ctx.source.user_id, + "session_id": ctx.session_id, + "iteration": iteration, + "tool_names": _names, + "tools": prev_tools, + }), + ctx._loop_for_step, + logger=logger, + log_message="agent:step hook scheduling error", + ) + + def _event_callback_sync(self, event_type: str, context: dict) -> None: + ctx = self._ctx + try: + asyncio.run_coroutine_threadsafe( + ctx._hooks_ref.emit(event_type, context), + ctx._loop_for_step, + ) + except Exception as _e: + logger.debug("event_callback hook error: %s", _e) + + def _attach_session_title_callback(self, agent, ctx) -> None: + """Wire the platform thread-rename lane onto the agent as `_on_session_title`. + + The titler runs in the turn prologue, so attach before the run, not after it. + """ + try: + # Gateway auto-title failures are not user-actionable, so never surface them as messages; + # overriding the failure sink keeps CLI on _emit_auxiliary_failure while gateway logs debug. + def _title_failure_cb(task: str, exc: BaseException) -> None: + logger.debug( + "Gateway auto-title failure suppressed (not user-visible): %s: %s", + task, exc, + ) + + agent._title_failure_callback = _title_failure_cb + + session_id = getattr(agent, "session_id", None) + source = ctx.source + + # Both lanes spend a rate-limited platform call per title, so they use the model's title + # only (TitleCallback); renaming twice burns Discord's 2-per-10-min budget on a throwaway. + if self._runner._is_telegram_topic_lane(source): + agent._on_session_title = lambda title, title_source: ( + title_source == "llm" + and self._runner._schedule_telegram_topic_title_rename( + source, session_id, title, + ) + ) + elif self._runner._is_discord_auto_thread_lane(source) or ( + self._runner._is_relay_discord_channel_lane(source) + ): + # Relay note: the second predicate is shape-only (relay Discord channel event). + # Whether the connector auto-threaded our reply is only knowable AFTER delivery, so + # the callback must be registered eagerly and the rename lane does the cache lookup + # at fire time — gating registration on the cache read meant it never registered. + agent._on_session_title = lambda title, title_source: ( + title_source == "llm" + and self._runner._schedule_discord_semantic_thread_rename( + source, session_id, title, + ) + ) + except Exception: + logger.debug("Failed to attach session title callback", exc_info=True) + + def _status_callback_sync(self, event_type: str, message: str) -> None: + from gateway.run import ( + _prepare_gateway_status_message, + _redact_gateway_user_facing_secrets, + _send_or_update_status_coro, + safe_schedule_threadsafe, + ) + ctx = self._ctx + if not ctx._status_adapter or not ctx._run_still_current(): + return + prepared_message = _prepare_gateway_status_message( + ctx.source.platform, + event_type, + message, + ) + if prepared_message is None: + logger.debug( + "status_callback suppressed for %s/%s: %s", + ctx.source.platform.value if ctx.source.platform else "unknown", + event_type, + _redact_gateway_user_facing_secrets(str(message or ""))[:160], + ) + return + _fut = safe_schedule_threadsafe( + _send_or_update_status_coro(ctx._status_adapter, ctx._status_chat_id, event_type, prepared_message, ctx._status_thread_metadata), + ctx._loop_for_step, + logger=logger, + log_message=f"status_callback ({event_type}) scheduling error", + ) + if _fut is None: + return + if ctx._cleanup_progress: + def _track_status_id(fut) -> None: + try: + res = fut.result() + except Exception: + return + mid = getattr(res, "message_id", None) + if getattr(res, "success", False) and mid: + ctx._cleanup_msg_ids.append(str(mid)) + _fut.add_done_callback(_track_status_id) + + def _setup_stream_consumer(self, platform_key): + from gateway.run import safe_schedule_threadsafe + ctx = self._ctx + # Set up stream consumer for token streaming or interim commentary. + _stream_consumer = None + _stream_delta_cb = None + # streaming TTS consumer is created on the outer event-loop thread before run_sync launches. + # run_sync only reads it via ``streaming_tts_consumer_holder[0]`` for delta callback wiring. + _stts_consumer_ref = ctx.streaming_tts_consumer_holder[0] + _scfg = getattr(getattr(self._runner, 'config', None), 'streaming', None) + if _scfg is None: + from gateway.config import StreamingConfig + _scfg = StreamingConfig() + + # Per-platform streaming gate: display.platforms..streaming can disable streaming + # for specific platforms even when the global streaming config is enabled. + _plat_streaming = ctx.resolve_display_setting( + ctx.user_config, platform_key, "streaming" + ) + # None = no per-platform override → follow global config + _streaming_enabled = ( + _scfg.enabled and _scfg.transport != "off" + if _plat_streaming is None + else bool(_plat_streaming) + ) + _want_stream_deltas = _streaming_enabled + _want_interim_messages = ctx.interim_assistant_messages_enabled + _want_interim_consumer = _want_interim_messages + if _want_stream_deltas or _want_interim_consumer: + try: + from gateway.stream_consumer import GatewayStreamConsumer + _adapter = self._runner._adapter_for_source(ctx.source) + if _adapter: + _consumer_cfg, _pause_typing_before_finalize = ( + self._runner._build_stream_consumer_config( + ctx.source, _scfg, _adapter, + on_missing_cursor="raise", + ) + ) + _stream_consumer = GatewayStreamConsumer( + adapter=_adapter, + chat_id=ctx.source.chat_id, + config=_consumer_cfg, + metadata=ctx._status_thread_metadata, + on_new_message=( + (lambda: ctx.progress_queue.put(("__reset__",))) + if ctx.progress_queue is not None + else None + ), + on_before_finalize=_pause_typing_before_finalize, + initial_reply_to_id=ctx.event_message_id, + run_still_current=ctx._run_still_current, + ) + if _want_stream_deltas: + def _stream_delta_cb(text: str) -> None: + if ctx._run_still_current(): + _stream_consumer.on_delta(text) + # Tee to the streaming-TTS consumer (#60671). + if _stts_consumer_ref is not None: + _stts_consumer_ref.on_delta(text) + ctx.stream_consumer_holder[0] = _stream_consumer + except Exception as _sc_err: + logger.debug("Could not set up stream consumer: %s", _sc_err) + + # Text streaming off but streaming TTS active: install a TTS-only delta callback so the + # consumer still receives LLM deltas for audio synthesis. + if _stream_delta_cb is None and _stts_consumer_ref is not None: + def _stream_delta_cb(text: str) -> None: + if ctx._run_still_current(): + _stts_consumer_ref.on_delta(text) + + def _interim_assistant_cb(text: str, *, already_streamed: bool = False) -> None: + if not ctx._run_still_current(): + return + display_text = text + if _stream_consumer is not None: + if already_streamed: + _stream_consumer.on_segment_break() + else: + _stream_consumer.on_commentary(display_text) + return + if already_streamed or not ctx._status_adapter or not str(display_text or "").strip(): + return + safe_schedule_threadsafe( + ctx._status_adapter.send( + ctx._status_chat_id, + display_text, + metadata=ctx._status_thread_metadata, + ), + ctx._loop_for_step, + logger=logger, + log_message="interim_assistant_callback scheduling error", + ) + return _stream_consumer, _stream_delta_cb, _interim_assistant_cb, _want_interim_messages + + def _resolve_turn_agent( + self, turn_route, platform_key, combined_ephemeral, max_iterations, reasoning_config, pr, + ): + from gateway.run import _AGENT_PENDING_SENTINEL, _checkpoint_agent_kwargs + ctx = self._ctx + # Per-platform skip_context_files — messaging platforms can opt out of filesystem-heavy + # context-file discovery (SOUL.md, AGENTS.md, .cursorrules) to cut AIAgent build latency. + _platforms_gw_cfg = (ctx.user_config.get("gateway") or {}).get("platforms") or {} + # ``hermes gateway setup`` writes ``gateway.platforms`` as a LIST of enabled platform names, + # not a dict; treat any non-dict shape as "no per-platform overrides" rather than crashing. + if not isinstance(_platforms_gw_cfg, dict): + _platforms_gw_cfg = {} + _plat_gw_cfg = _platforms_gw_cfg.get(platform_key) or {} + _skip_context = _plat_gw_cfg.get("skip_context_files") + skip_context_files = bool(_skip_context) if _skip_context is not None else False + + # Agent cache: reuse this session's previous AIAgent to preserve the frozen system prompt + # and tool schemas for prompt cache hits. + _sig = self._runner._agent_config_signature( + turn_route["model"], + turn_route["runtime"], + ctx.enabled_toolsets, + combined_ephemeral, + cache_keys=self._runner._extract_cache_busting_config(ctx.user_config), + user_id=getattr(ctx.source, "user_id", None), + user_id_alt=getattr(ctx.source, "user_id_alt", None), + skip_context_files=skip_context_files, + ) + agent = None + reused_cached_agent = False + _cache_lock = getattr(self._runner, "_agent_cache_lock", None) + _cache = getattr(self._runner, "_agent_cache", None) + + # Peek at the cached entry's snapshot session_id so we can check, OUTSIDE the cache lock, + # whether it is a DEAD session in state.db. "cached sid != current sid" normally means an + # intentional switch (reuse the agent), but the routing-key self-heal yields the same shape + # with an agent bound to a DEAD session; reusing it re-binds the dead sid and loops. + _peek_cached_sid = None + if _cache_lock and _cache is not None: + with _cache_lock: + _peek_entry = _cache.get(ctx.session_key) + if _peek_entry and len(_peek_entry) > 3: + _peek_cached_sid = _peek_entry[3] + _cached_sid_is_dead = False + if ( + _peek_cached_sid is not None + and ctx.session_id is not None + and _peek_cached_sid != ctx.session_id + ): + try: + _cached_sid_is_dead = self._runner.session_store._is_session_ended_in_db( + _peek_cached_sid + ) + except Exception: + _cached_sid_is_dead = False + + # Cross-process write guard: another process (e.g. hermes dashboard) appending to the same + # SessionDB session makes the cached agent's transcript stale. On message_count mismatch vs + # the count recorded at cache time, invalidate so a fresh agent re-reads from disk. + _current_msg_count = None + if self._runner._session_db is not None and ctx.session_id: + try: + # run_sync is off-loop (executor); sync DB is fine. + _sess_row = self._runner._session_db._db.get_session(ctx.session_id) + if _sess_row: + _current_msg_count = _sess_row.get("message_count", 0) + except Exception: + pass + + _xproc_evicted_agent = None + if _cache_lock and _cache is not None: + with _cache_lock: + cached = _cache.get(ctx.session_key) + if cached and cached[1] == _sig: + # cached[2] is the message_count at cache time; stale when a second process + # appended rows. cached[3] (when present) is the session_id the snapshot was + # taken for — used to skip the guard when the active session_id differs. + _cached_mc = cached[2] if len(cached) > 2 else None + _cached_sid = cached[3] if len(cached) > 3 else None + # Snapshot from a different session_id (same session_key, other conversation): the + # counts track DIFFERENT DB rows, so the comparison is meaningless. REUSE the cached + # agent rather than rebuild and bust the prompt cache on every session switch. + _session_id_mismatch = ( + _cached_sid is not None + and ctx.session_id is not None + and _cached_sid != ctx.session_id + ) + # Re-validate the OUTSIDE-lock dead-session peek against the tuple read under THIS + # lock: the entry may have been replaced between peek and acquisition, and a stale + # "dead" verdict must never be applied to a different (possibly live) cached agent. + _stale_dead_sid_reuse = ( + _session_id_mismatch + and _cached_sid_is_dead + and _cached_sid == _peek_cached_sid + ) + if _stale_dead_sid_reuse: + # The routing key was just self-healed away from a session state.db marked + # ended, but this cached AIAgent still belongs to that DEAD session_id. + # Reusing it would re-bind the dead sid and undo the self-heal; rebuild fresh. + logger.info( + "Agent cache invalidated for session %s: " + "cached agent's session_id %s is ended in " + "state.db (stale self-heal artifact, " + "#54878 x #54947) — discarding instead of " + "reusing across the routing recovery", + ctx.session_key, _cached_sid, + ) + evicted = self._runner._agent_cache.pop(ctx.session_key, None) + _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None + if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: + # Same deferred-cleanup rationale as the cross-process branch below: don't + # block the event loop / cache lock on memory-provider or socket teardown. + _xproc_evicted_agent = _ev_agent + elif ( + not _session_id_mismatch + and _cached_mc is not None + and _current_msg_count is not None + and _current_msg_count != _cached_mc + ): + # Cross-process write detected — discard stale + # agent so it rebuilds from fresh DB transcript. + logger.info( + "Agent cache invalidated for session %s: " + "message_count changed (%s -> %s), " + "possible cross-process write", + ctx.session_key, _cached_mc, _current_msg_count, + ) + evicted = self._runner._agent_cache.pop(ctx.session_key, None) + _ev_agent = evicted[0] if isinstance(evicted, tuple) and evicted else None + if _ev_agent and _ev_agent is not _AGENT_PENDING_SENTINEL: + # Defer cleanup until AFTER the lock is released: release_clients can + # block on memory-provider/socket teardown, stalling the event loop while + # the idle sweeper waits on this lock (blocking Discord heartbeats). The + # session rebuilds a fresh agent below, so use the SOFT release that keeps + # its terminal sandbox / browser / bg processes for the new agent to + # inherit — mirrors _evict_cached_agent / idle-sweep. + _xproc_evicted_agent = _ev_agent + else: + agent = cached[0] + # Refresh LRU order so the cap enforcement evicts + # truly-oldest entries, not the one we just used. + if hasattr(_cache, "move_to_end"): + with suppress(KeyError): + _cache.move_to_end(ctx.session_key) + self._runner._init_cached_agent_for_turn(agent, ctx._interrupt_depth) + # Refresh agent max_iterations from current config + # (cached agent may have been created with old config) + agent.max_iterations = max_iterations + logger.debug("Reusing cached agent for session %s", ctx.session_key) + reused_cached_agent = True + + # Lock released — refresh the reused agent's fallback chain from disk OUTSIDE the cache lock + # (disk I/O under the lock stalls the idle-sweep watcher and Discord heartbeats). A chain + # configured after caching must reach the next turn; per-session serialization keeps it safe. + if reused_cached_agent and agent is not None: + self._runner._apply_fallback_chain_to_agent( + agent, self._runner._refresh_fallback_model(), + ) + + # Lock released — schedule cleanup of any cross-process-evicted agent on a daemon thread so + # memory-provider/socket teardown never blocks the gateway loop or the expiry watcher's lock. + if _xproc_evicted_agent is not None: + try: + threading.Thread( + target=self._runner._release_evicted_agent_soft, + args=(_xproc_evicted_agent,), + daemon=True, + name=f"agent-xproc-evict-{str(ctx.session_key)[:24]}", + ).start() + except Exception: + # Interpreter shutdown or thread-spawn failure — release + # inline as a best-effort fallback. + with suppress(Exception): + self._runner._release_evicted_agent_soft(_xproc_evicted_agent) + + if agent is None: + # Config changed or first message — create fresh agent + agent = ctx.AIAgent( + model=turn_route["model"], + **turn_route["runtime"], + **_checkpoint_agent_kwargs(ctx.user_config), + max_iterations=max_iterations, + quiet_mode=True, + verbose_logging=False, + enabled_toolsets=ctx.enabled_toolsets, + disabled_toolsets=ctx.disabled_toolsets, + ephemeral_system_prompt=combined_ephemeral or None, + prefill_messages=self._runner._prefill_messages or None, + reasoning_config=reasoning_config, + service_tier=self._runner._service_tier, + request_overrides=turn_route.get("request_overrides"), + providers_allowed=pr.get("only"), + providers_ignored=pr.get("ignore"), + providers_order=pr.get("order"), + provider_sort=pr.get("sort"), + provider_require_parameters=pr.get("require_parameters", False), + provider_data_collection=pr.get("data_collection"), + session_id=ctx.session_id, + platform=platform_key, + user_id=ctx.source.user_id, + user_id_alt=ctx.source.user_id_alt, + user_name=ctx.source.user_name, + chat_id=ctx.source.chat_id, + chat_name=ctx.source.chat_name, + chat_type=ctx.source.chat_type, + thread_id=ctx.source.thread_id, + gateway_session_key=ctx.session_key, + session_db=getattr(self._runner._session_db, "_db", self._runner._session_db), + # Reload from disk — do not reuse the startup snapshot (#60955). + fallback_model=self._runner._refresh_fallback_model(), + skip_context_files=skip_context_files, + # Keep the persona even with minimal context: soul identity is + # a single small file, not part of the expensive walk. + load_soul_identity=True, + ) + if _cache_lock and _cache is not None: + with _cache_lock: + # Record the snapshot's session_id with message_count so the cross-process guard can + # skip the meaningless count comparison if the active session_id later switches. + _cache[ctx.session_key] = ( + agent, _sig, _current_msg_count, ctx.session_id, + ) + self._runner._enforce_agent_cache_cap() + logger.debug("Created new agent for session %s (sig=%s)", ctx.session_key, _sig) + return agent, reused_cached_agent + + def _wire_turn_agent_callbacks( + self, agent, turn_route, reasoning_config, + _stream_delta_cb, _interim_assistant_cb, _want_interim_messages, + ): + from gateway.run import ( + _interim_metadata, + _non_conversational_metadata, + render_notice_line, + safe_schedule_threadsafe, + ) + ctx = self._ctx + # Per-message state — callbacks and reasoning config change every turn, so they aren't baked + # into the cached agent. The progress callback is ALWAYS attached (never gated to None): its + # body gates each event class, and subagent-failure notices must fire even with + # tool_progress/thinking off — a None gate made dead subagents vanish silently. + agent.tool_progress_callback = ctx.progress_callback + # Compose ID-bearing lifecycle consumers: Discord's one-time voice ack and Slack's task cards + # both ride the authoritative start callback, so neither infers identity from tool names. + _combined_start_cb = ctx.native_tool_start_callback or ctx.voice_ack_callback + agent.tool_start_callback = ( + _combined_start_cb + if ( + ctx._voice_ack_guild[0] is not None + or ctx._native_slack_task_cards + ) + else None + ) + agent.tool_complete_callback = ( + ctx.native_tool_complete_callback + if ctx._native_slack_task_cards + and ctx.native_tool_complete_callback is not None + else None + ) + agent.step_callback = ctx._step_callback_sync if ctx._hooks_ref.loaded_hooks else None + agent.stream_delta_callback = _stream_delta_cb + agent.interim_assistant_callback = _interim_assistant_cb if _want_interim_messages else None + agent.status_callback = ctx._status_callback_sync + # Credits / out-of-band notices (usage bands, depletion, restored) fire from the agent's sync + # worker thread, so hop onto the gateway loop via safe_schedule_threadsafe. Fired-once latch + # lives on the cached agent (no per-turn re-nag); clear is a no-op — sends can't be retracted. + def _notice_callback_sync(notice) -> None: + if not ctx._status_adapter or not ctx._run_still_current(): + return + try: + line = render_notice_line(notice) + except Exception: + logger.debug("render_notice_line failed", exc_info=True) + return + if not line: + return + safe_schedule_threadsafe( + self._runner._deliver_platform_notice(ctx.source, line), + ctx._loop_for_step, + logger=logger, + log_message="notice_callback delivery scheduling error", + ) + + agent.notice_callback = _notice_callback_sync + agent.notice_clear_callback = None + agent.event_callback = ctx._event_callback_sync + agent.reasoning_config = reasoning_config + agent.service_tier = self._runner._service_tier + # Merge, never overwrite: init-time request overrides (e.g. a custom provider's extra_body + # merged at agent construction) must survive every reused-agent turn. Drop only the PREVIOUS + # turn's routing overrides before layering this turn's, so stale per-turn values never linger. + request_overrides = dict(getattr(agent, "request_overrides", {}) or {}) + previous_turn_overrides = dict( + getattr(agent, "_gateway_turn_request_overrides", {}) or {} + ) + for key, value in previous_turn_overrides.items(): + if request_overrides.get(key) == value: + request_overrides.pop(key, None) + turn_request_overrides = dict(turn_route.get("request_overrides") or {}) + request_overrides.update(turn_request_overrides) + agent.request_overrides = request_overrides + agent._gateway_turn_request_overrides = turn_request_overrides + # Must-deliver notes for THIS turn ride the current user message (api_content sidecar), never + # the system prompt: staged by _handle_message_with_agent (auto-reset, first-contact intro, + # voice-channel change). Assigned unconditionally so a reused agent never replays a stale note. + agent._gateway_turn_context_notes = "\n\n".join( + self._runner._consume_pending_turn_sidecar_notes(ctx.session_key) + ) + + _bg_review_release = threading.Event() + _bg_review_pending: list[str] = [] + _bg_review_pending_lock = threading.Lock() + + def _deliver_bg_review_message(message: str) -> None: + if not ctx._status_adapter or not ctx._run_still_current(): + return + safe_schedule_threadsafe( + ctx._status_adapter.send( + ctx._status_chat_id, + message, + metadata=_interim_metadata(_non_conversational_metadata(ctx._status_thread_metadata, platform=ctx.source.platform)), + ), + ctx._loop_for_step, + logger=logger, + log_message="background_review_callback scheduling error", + ) + + def _release_bg_review_messages() -> None: + _bg_review_release.set() + with _bg_review_pending_lock: + pending = list(_bg_review_pending) + _bg_review_pending.clear() + for queued in pending: + _deliver_bg_review_message(queued) + + # Background review delivery — send "💾 Memory updated" etc. to user + def _bg_review_send(message: str) -> None: + if not ctx._status_adapter or not ctx._run_still_current(): + return + if not _bg_review_release.is_set(): + with _bg_review_pending_lock: + if not _bg_review_release.is_set(): + _bg_review_pending.append(message) + return + _deliver_bg_review_message(message) + + agent.background_review_callback = _bg_review_send + # Register the release hook on the adapter so base.py's finally + # block can fire it after delivering the main response. + if ctx._status_adapter and ctx.session_key: + if getattr(type(ctx._status_adapter), "register_post_delivery_callback", None) is not None: + ctx._status_adapter.register_post_delivery_callback( + ctx.session_key, + _release_bg_review_messages, + generation=ctx.run_generation, + ) + else: + _pdc = getattr(ctx._status_adapter, "_post_delivery_callbacks", None) + if _pdc is not None: + _pdc[ctx.session_key] = _release_bg_review_messages + # Memory update notifications in chat. Config: display.memory_notifications + # off — no chat notification (still logged to stdout) + # on — generic "💾 Memory updated" (default) + # verbose — content preview: "💾 Memory ➕ Hermes Repo..." + _mem_notif = ctx.user_config.get("display", {}).get("memory_notifications") + if isinstance(_mem_notif, bool): + _mem_notif = "on" if _mem_notif else "off" + agent.memory_notifications = str(_mem_notif).lower() if _mem_notif else "on" + + agent.clarify_callback = self._clarify_callback_sync + + # Show assistant thinking between tool calls — independent of tool_progress mode. Mattermost + # needs an explicit per-platform opt-in so global scratch-text doesn't leak into threads. + agent.thinking_progress = ctx._thinking_enabled + # Store agent reference for interrupt support + ctx.agent_holder[0] = agent + # Wire the platform thread-rename lane onto the agent: the titler fires from the turn prologue, + # not after the response, so titles are pushed the moment they land. + self._attach_session_title_callback(agent, ctx) + # Publish turn ownership for explicit /stop, /new, disconnect, and shutdown interrupts. + # Older session processes are outside this baseline and remain alive. + agent._gateway_turn_process_task_id = ctx.process_task_id + agent._gateway_turn_process_baseline = ctx.process_baseline + # Capture the full tool definitions for transcript logging + ctx.tools_holder[0] = agent.tools if hasattr(agent, 'tools') else None + + # ------------------------------------------------------------------ + # Shared native-stream boundary close: for native-streaming platforms (e.g. WeCom), an + # interrupting interaction (approval or clarify prompt) must finalize the current stream + # and disable native streaming first, or post-interaction output keeps updating the OLD + # bubble above the prompt. Runs on the agent thread; the consumer serializes via its queue. + def _close_native_stream_boundary( + self, _reason: str, _placeholder: str | None = None, _reopen: bool = False, + ) -> bool: + ctx = self._ctx + _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None + if not (_sc and getattr(_sc, "_use_native_streaming", False)): + return True + _cancelled_flag = None + try: + _boundary_result = _sc.close_for_approval_prompt( + _placeholder, reason=_reason, reopen=_reopen, + ) + # Returns (future, cancelled_flag) or just a future. + if isinstance(_boundary_result, tuple): + _boundary_future, _cancelled_flag = _boundary_result + else: + _boundary_future = _boundary_result + if hasattr(_boundary_future, "result"): + _ok = _boundary_future.result(timeout=10) + if not _ok: + logger.warning( + "%s boundary failed to close stream properly — " + "prompt may still appear in typing bubble", _reason, + ) + return bool(_ok) + return True + except (TimeoutError, Exception) as _boundary_err: + if _cancelled_flag is not None: + _cancelled_flag["cancelled"] = True + logger.warning( + "%s boundary timed out or failed: %s", _reason, _boundary_err, + ) + return False + + # ------------------------------------------------------------------ + # Clarify callback: present a clarify prompt and block on a response. Runs on the agent's + # worker thread (clarify_tool's synchronous contract): schedules the adapter's send_clarify + # on the gateway loop, then blocks on the primitive's threading.Event with a timeout. + # Returns the response string, or a sentinel explaining no response arrived. + # ------------------------------------------------------------------ + def _clarify_callback_sync(self, question: str, choices, multi_select: bool = False) -> str: + from gateway.run import _clarify_send_then_wait, safe_schedule_threadsafe + ctx = self._ctx + from tools import clarify_gateway as _clarify_mod + import uuid as _uuid + + if not ctx._status_adapter: + return "" + + clarify_id = _uuid.uuid4().hex[:10] + _clarify_mod.register( + clarify_id=clarify_id, + session_key=ctx.session_key or "", + question=question, + choices=list(choices) if choices else None, + multi_select=bool(multi_select), + ) + + # WeCom native streaming: finalize the current stream before the clarify prompt so the + # post-answer output opens a fresh bubble below the question ("气泡割裂" otherwise). Unlike + # approval, clarify passes reopen=True so the continuation re-opens a native stream; if + # the re-seed fails the consumer degrades to send() automatically. + self._close_native_stream_boundary( + "Clarify", "💬 等待你的选择...", _reopen=True, + ) + + # Pause typing — as with approval, a "thinking..." status must not obscure the prompt or + # block an "Other" reply on platforms that disable input while typing (Slack Assistant). + with suppress(Exception): + ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) + + # Ordering barrier: flush buffered assistant prose to the platform BEFORE sending the + # poll, which goes out on a separate agent-thread-blocking path and would otherwise + # render ABOVE its own explanation. Best-effort + short timeout so the agent thread + # never hangs if the consumer task isn't running. + try: + _sc = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None + _flush = getattr(_sc, "flush_pending_sync", None) + if callable(_flush): + _flush(timeout=3.0) + except Exception: + logger.debug( + "Stream-consumer flush before clarify prompt failed", + exc_info=True, + ) + + fut = safe_schedule_threadsafe( + ctx._status_adapter.send_clarify( + chat_id=ctx._status_chat_id, + question=question, + choices=list(choices) if choices else None, + clarify_id=clarify_id, + session_key=ctx.session_key or "", + metadata=ctx._status_thread_metadata, + ), + ctx._loop_for_step, + logger=logger, + log_message="Clarify send failed to schedule", + ) + # Boundary rule (see _approval_send_outcome): a send timeout is AMBIGUOUS — the card may + # have posted with a late ack. Only a definitive failure tears down the registration; + # ambiguous falls through to the bounded wait so a late reply resolves. + _clarify_response = _clarify_send_then_wait( + fut, + clarify_id=clarify_id, + session_key=ctx.session_key or "", + clarify_mod=_clarify_mod, + ) + # Only re-arm typing when the user actually answered — the undeliverable sentinel and the + # timeout/cancellation strings start with '[' and must pass through untouched. + if not ( + isinstance(_clarify_response, str) + and _clarify_response.startswith("[") + ): + # User answered: reopen typing IMMEDIATELY, not on the LLM's first post-answer token + # (native streaming otherwise re-seeds lazily on the first delta: ~48s of dead air). + # request_reopen_seed is a no-op outside the reopen-pending native state; always safe. + _sc_reopen = ctx.stream_consumer_holder[0] if ctx.stream_consumer_holder else None + if _sc_reopen is not None: + try: + _sc_reopen.request_reopen_seed() + except Exception: + logger.debug( + "request_reopen_seed after clarify answer failed", + exc_info=True, + ) + try: + ctx._status_adapter.resume_typing_for_chat(ctx._status_chat_id) + except Exception: + logger.debug( + "resume_typing_for_chat after clarify answer failed", + exc_info=True, + ) + return _clarify_response + + def _load_turn_history(self, agent, reused_cached_agent): + from gateway.run import ( + _build_gateway_agent_history, + _collect_history_media_paths, + _message_timestamps_enabled, + _select_cached_agent_history, + ) + ctx = self._ctx + # Convert history to agent format. Transcript path: {role, content, timestamp} dicts — strip + # timestamps. Interrupt path (agent result["messages"]): full agent messages with + # tool_calls/tool_call_id/reasoning — pass through intact so the API sees valid assistant→tool + # sequences (dropping tool_calls causes 500s). Telegram observed group context: observed=True + # rows are withheld from replayable history and attached to the current addressed message as + # API-only context, so persisted history stores only the real addressed user turn. + agent_history, observed_group_context = _build_gateway_agent_history( + ctx.history, + channel_prompt=ctx.channel_prompt, + inject_timestamps=_message_timestamps_enabled(ctx.user_config), + ) + + # FTS write-corruption guard: if persistence failed silently via corrupt FTS triggers, the + # reloaded transcript is stale/empty while the SAME cached agent still holds the full live + # conversation in `_session_messages`; replacing it causes same-session amnesia. Only for + # a reused agent bound to this exact session_id. + if reused_cached_agent and getattr(agent, "session_id", None) == ctx.session_id: + _selected = _select_cached_agent_history( + agent_history, getattr(agent, "_session_messages", None) + ) + if _selected is not agent_history: + logger.warning( + "Persisted transcript lagged live cached history for " + "session %s (disk=%d, memory=%d); preserving live " + "conversation context (possible FTS write corruption)", + ctx.session_key, len(agent_history), len(_selected), + ) + # The live in-memory history bypassed the _build_gateway_agent_history cleanup above — + # re-apply the stale-confirmation expiry so a dangerous confirmation can't slip through. + agent_history = strip_stale_dangerous_confirmations( + _selected, now=time.time() + ) + + # Collect MEDIA paths already in history to exclude them from this turn's extraction. + # Compression-safe: even if the message list shrinks, we know which paths are old. + _history_media_paths: set = _collect_history_media_paths(agent_history) + return agent_history, observed_group_context, _history_media_paths + + def _approval_notify_sync(self, approval_data: dict) -> None: + """Send the approval request to the user from the agent thread. + + Uses the adapter's interactive button approvals (e.g. ``send_exec_approval``) when + available, else a plain text message with ``/approve`` instructions. + """ + from gateway.run import ( + _approval_send_outcome, + _format_exec_approval_fallback, + _interim_metadata, + _redact_approval_command, + safe_schedule_threadsafe, + ) + ctx = self._ctx + # Pause typing while awaiting approval: Slack's assistant_threads_setStatus disables the + # compose box, so the user can't type /approve while "is thinking..." shows. The approval + # send auto-clears it; pausing stops _keep_typing re-setting it. Resumed in approve/deny. + ctx._status_adapter.pause_typing_for_chat(ctx._status_chat_id) + + # WeCom native streaming: ask the stream consumer to close the current stream before the + # approval prompt — via the consumer's queue, so it serializes with pending deltas. + self._close_native_stream_boundary("Approval") + + cmd = approval_data.get("command", "") + desc = approval_data.get("description", "dangerous command") + + # Redact credentials from the command before display — Tirith's findings are already + # redacted, but the raw command string still leaks secrets to the chat platform. Done + # here so BOTH the button-based and plain-text fallback paths use the redacted value. + cmd = _redact_approval_command(cmd) + + # Prefer button-based approval when the adapter supports it. Check the *class*, not the + # instance — avoids false positives from MagicMock auto-attribute creation in tests. + if getattr(type(ctx._status_adapter), "send_exec_approval", None) is not None: + try: + _approval_fut = safe_schedule_threadsafe( + ctx._status_adapter.send_exec_approval( + chat_id=ctx._status_chat_id, + command=cmd, + session_key=ctx.session_key or "", + description=desc, + metadata=ctx._status_thread_metadata, + allow_permanent=approval_data.get("allow_permanent", True), + allow_session=approval_data.get("allow_session", True), + smart_denied=approval_data.get("smart_denied", False), + ), + ctx._loop_for_step, + logger=logger, + log_message="send_exec_approval scheduling error", + ) + if _approval_fut is None: + raise RuntimeError("send_exec_approval: loop unavailable") + _outcome = _approval_send_outcome(_approval_fut, timeout=15) + if _outcome == "sent": + return + if _outcome == "ambiguous": + # Timeout ≠ failure: the card may have posted with a late ack (slow API or + # backpressure). The prompt registration stays alive so a tap still resolves; + # re-sending made duplicate cards + orphaned "/approve: nothing pending". Skip. + logger.warning( + "Button-based approval send timed out — treating " + "as possibly-delivered (no re-send; the prompt " + "stays armed for a late tap)" + ) + return + logger.warning( + "Button-based approval failed (send returned error), falling back to text" + ) + except Exception as _e: + logger.warning( + "Button-based approval failed, falling back to text: %s", _e + ) + + # Fallback: plain-text approval prompt with the adapter's typed prefix (e.g. `!approve`) — + # typed "/" is blocked in Slack threads and reserved by Matrix clients. + _p = getattr(ctx._status_adapter, "typed_command_prefix", "/") + msg = _format_exec_approval_fallback( + cmd, + desc, + _p, + allow_permanent=approval_data.get("allow_permanent", True), + allow_session=approval_data.get("allow_session", True), + smart_denied=approval_data.get("smart_denied", False), + ) + try: + # Mark as approval prompt so WeCom routes through control lane + _approval_metadata = dict(ctx._status_thread_metadata or {}) + _approval_metadata["is_approval_prompt"] = True + + _approval_send_fut = safe_schedule_threadsafe( + ctx._status_adapter.send( + ctx._status_chat_id, + msg, + metadata=_interim_metadata(_approval_metadata), + ), + ctx._loop_for_step, + logger=logger, + log_message="Approval text-send scheduling error", + ) + if _approval_send_fut is not None: + _approval_send_fut.result(timeout=15) + except Exception as _e: + logger.error("Failed to send approval request: %s", _e) + + def _prepare_turn_message(self, agent_history): + from gateway.run import ( + _auto_continue_freshness_window, + _is_fresh_gateway_interruption, + _last_transcript_timestamp, + _prepare_resume_pending_message, + build_resume_recovery_note, + ) + ctx = self._ctx + # Keep real user text separate from API-only recovery guidance: if an auto-continue note is + # prepended below, persist the original so stale guidance never replays as user text. + _persist_user_message_override: Optional[Any] = ctx.persist_user_message + _persist_user_timestamp_override: Optional[float] = ctx.persist_user_timestamp + + # Prepend pending model switch note so the model knows about the switch + _pending_notes = getattr(self._runner, '_pending_model_notes', {}) + _msn = _pending_notes.pop(ctx.session_key, None) if ctx.session_key else None + if _msn: + ctx.message = _msn + "\n\n" + ctx.message + + # Auto-continue: history ending with a tool result means the previous turn was cut off + # (restart, crash, SIGTERM) — prepend a system note so the model finishes the pending tool + # results first. Session-level resume_pending (drain-timeout shutdown) uses stronger + # reason-aware wording that subsumes this case. Both gate on the age of ``history[-1]`` (not + # agent_history, which stripped ``timestamp`` off tool rows); rows without one are fresh. + _freshness_window = _auto_continue_freshness_window() + _interruption_is_fresh = _is_fresh_gateway_interruption( + _last_transcript_timestamp(ctx.history), + window_secs=_freshness_window, + ) + + _resume_entry = None + if ctx.session_key: + try: + _resume_entry = self._runner.session_store._entries.get(ctx.session_key) + except Exception: + _resume_entry = None + + # resume_pending freshness also uses the restart watchdog's ``last_resume_marked_at`` (the + # true interruption stamp): the transcript clock (_interruption_is_fresh) can be hours older + # for an active thread, so gating on it alone drops the recovery note — and the startup + # auto-resume turn has empty text, so the model gets a blank user message. Fresh if EITHER is. + _resume_mark_is_fresh = False + if _resume_entry is not None and getattr(_resume_entry, "resume_pending", False): + _resume_mark_is_fresh = _is_fresh_gateway_interruption( + getattr(_resume_entry, "last_resume_marked_at", None), + window_secs=_freshness_window, + ) + _is_resume_pending = bool( + _resume_entry is not None + and getattr(_resume_entry, "resume_pending", False) + and (_interruption_is_fresh or _resume_mark_is_fresh) + ) + _has_fresh_tool_tail = bool( + agent_history + and agent_history[-1].get("role") == "tool" + and _interruption_is_fresh + ) + + if _is_resume_pending: + _reason = getattr(_resume_entry, "resume_reason", None) or "restart_timeout" + # Empty message = the startup auto-resume turn from _schedule_resume_pending_sessions; + # there is no NEW user message. Interactive platforms report the restore and ask what + # next; event platforms (webhook, API server) continue the work — nobody is present to + # answer, and an acknowledgement would silently abandon the task. + _resume_adapter = self._runner._adapter_for_source(ctx.source) + _interactive_resume = bool( + getattr(_resume_adapter, "interactive_resume", True) + ) + ctx.message, _persist_user_message_override = _prepare_resume_pending_message( + _reason, ctx.message, interactive=_interactive_resume, + ) + elif _has_fresh_tool_tail: + _persist_user_message_override = ctx.message + ctx.message = ( + "[System note: A new message has arrived. The conversation " + "history contains pending tool outputs from an interrupted turn. " + "IGNORE those pending results. Address the user's NEW message " + "below FIRST. Do NOT re-execute old tool calls from the history.]\n\n" + + ctx.message + ) + + # Consume one-shot /reload-skills note (same queue pattern as CLI): prepend to the NEXT user + # message, then clear. Nothing hit the transcript out-of-band, so alternation stays intact. + _pending_notes = getattr(self._runner, "_pending_skills_reload_notes", None) + if _pending_notes and ctx.session_key and ctx.session_key in _pending_notes: + _srn = _pending_notes.pop(ctx.session_key, None) + if _srn: + ctx.message = _srn + "\n\n" + ctx.message + + # Safety net: a startup auto-resume event carries empty text and relies on the resume_pending + # branch above for the recovery note. If it did not fire (freshness signals disagreed, marker + # cleared before dispatch) we must NOT hand the model a blank user turn. Restricted to + # resume_pending sessions so legitimately empty turns (caption-less image) are untouched. + if ( + isinstance(ctx.message, str) + and not ctx.message.strip() + and _resume_entry is not None + and getattr(_resume_entry, "resume_pending", False) + ): + _sn_reason = ( + getattr(_resume_entry, "resume_reason", None) or "restart_timeout" + ) + _sn_adapter = self._runner._adapter_for_source(ctx.source) + ctx.message = build_resume_recovery_note( + _sn_reason, + "", + interactive=bool( + getattr(_sn_adapter, "interactive_resume", True) + ), + ) + return _persist_user_message_override, _persist_user_timestamp_override + + def _run_conversation_with_approval( + self, agent, agent_history, observed_group_context, + _persist_user_message_override, _persist_user_timestamp_override, + ): + from gateway.run import _wrap_current_message_with_observed_context + ctx = self._ctx + # Per-session gateway approval callback: dangerous-command approval blocks the agent thread + # (mirrors CLI input()); the callback bridges sync→async to send the request immediately. + from tools.approval import ( + register_gateway_notify, + reset_current_session_key, + set_current_session_key, + unregister_gateway_notify, + ) + + _approval_session_key = ctx.session_key or "" + _approval_session_token = set_current_session_key(_approval_session_key) + register_gateway_notify(_approval_session_key, self._approval_notify_sync) + try: + # If _prepare_inbound_message_text buffered image paths for native attachment, wrap the + # user turn as an OpenAI-style multimodal content list. Consume-and-clear so subsequent + # turns on the same runner instance don't re-attach stale images. + _native_imgs = self._runner._consume_pending_native_image_paths(ctx.session_key) + if _native_imgs: + try: + from agent.image_routing import build_native_content_parts + _parts, _skipped = build_native_content_parts( + ctx.message, + _native_imgs, + ) + if _skipped: + logger.warning( + "Native image attachment: skipped %d unreadable path(s): %s", + len(_skipped), _skipped, + ) + if any(p.get("type") == "image_url" for p in _parts): + _run_message: Any = _parts + else: + # All images failed to read — fall back to plain text. + _run_message = ctx.message + except Exception as _img_exc: + logger.warning( + "Native image attachment failed, falling back to text: %s", + _img_exc, + ) + _run_message = ctx.message + else: + _run_message = ctx.message + + _api_run_message = _wrap_current_message_with_observed_context( + _run_message, + observed_group_context, + ) + _conversation_kwargs = { + "conversation_history": agent_history, + "task_id": ctx.session_id, + } + if _persist_user_message_override is not None: + _conversation_kwargs["persist_user_message"] = _persist_user_message_override + elif observed_group_context: + _conversation_kwargs["persist_user_message"] = ctx.message + if ctx.persist_user_display_kind: + # Internal self-injected turn: type the persisted user row at turn start so UIs + # render it as a timeline notice, not a user bubble. Role/content are untouched and + # the key is stripped from provider-bound payloads in conversation_loop. + _conversation_kwargs["persist_user_display_kind"] = ( + ctx.persist_user_display_kind + ) + if ctx.moa_config is not None: + _conversation_kwargs["moa_config"] = ctx.moa_config + if _persist_user_timestamp_override is not None: + _conversation_kwargs["persist_user_timestamp"] = _persist_user_timestamp_override + # Thread the platform-side inbound message id onto the persisted user turn so a turn + # interrupted by a restart is recorded WITH its id — drain-window recovery dedups on + # has_platform_message_id. Uses the raw inbound id, NOT event_message_id (reply anchor). + if ctx.inbound_message_id is not None: + _conversation_kwargs["persist_user_platform_id"] = str(ctx.inbound_message_id) + result = agent.run_conversation(_api_run_message, **_conversation_kwargs) + finally: + unregister_gateway_notify(_approval_session_key) + # Cancel any pending clarify entries so blocked agent threads don't hang past the end of + # the run (interrupt, completion, gateway shutdown). Idempotent. + try: + from tools.clarify_gateway import clear_session as _clear_clarify_session + _clear_clarify_session(_approval_session_key) + except Exception: + pass + reset_current_session_key(_approval_session_token) + return result + + def _finish_stream_consumer(self, result, agent_history, _stream_consumer): + ctx = self._ctx + # Canonicalize a model-emitted computer-use screenshot path at the common result boundary: the + # streaming finalizer below and the non-streaming delivery path must see the same response; + # repairing only in later media scanning leaves streaming a mangled path + rejected attachment. + if isinstance(result, dict): + _result_final = result.get("final_response") + if isinstance(_result_final, str): + result["final_response"] = repair_explicit_computer_use_media_paths( + _result_final, + result.get("messages", []), + history_offset=len(agent_history), + ) + + ctx.result_holder[0] = result + + # Signal the stream consumer that the agent is done, passing final_response as the + # authoritative finalize payload: it includes post-stream augmentation (verifier footer, + # explainer) the accumulator never saw, so the seal delivers the TRUE final with no + # corrective send. Failed turns pass nothing — error text goes via the normal path. + if _stream_consumer is not None: + _final_for_stream = None + # Adopt ONLY a genuinely completed final: interrupt paths return {interrupted: True, + # completed: False} with a DIAGNOSTIC final_response and no failed key — adopting it + # would seal the streamed partial answer over with the diagnostic AND make + # delivered_final_matches reconcile, suppressing the gateway's own error delivery. + if ( + isinstance(result, dict) + and not result.get("failed") + and not result.get("interrupted") + and result.get("completed") is not False + ): + _fr = result.get("final_response") + if isinstance(_fr, str) and _fr.strip() and _fr != "(empty)": + _final_for_stream = _fr + if _final_for_stream is not None: + # Duck-type safe: test doubles / older consumers may expose a zero-arg finish(). The + # payload is an optimization, not a requirement — fall back to the bare signal. + try: + _stream_consumer.finish(_final_for_stream) + except TypeError: + _stream_consumer.finish() + else: + _stream_consumer.finish() + + def _sync_session_after_run(self, agent_history): + ctx = self._ctx + # Sync session_id right after run_conversation(): compression can rotate before a follow-up + # model call fails, and the failure return below must still point at the compressed child. + agent = ctx.agent_holder[0] + _session_was_split = False + # In-place compaction (compression.in_place) compacts the transcript WITHOUT rotating the id, + # so the id-change diff below can't see it. compress_context() sets this flag on the agent; the + # gateway re-baselines (history_offset=0 + JSONL rewrite) as for a split despite unchanged id. + _compacted_in_place = bool(getattr(agent, "_last_compaction_in_place", False)) if agent else False + agent_session_id = getattr(agent, 'session_id', ctx.session_id) if agent else ctx.session_id + if agent and ctx.session_key and agent_session_id != ctx.session_id: + _session_was_split = True + logger.info( + "Session split detected: %s → %s (compression)", + ctx.session_id, agent_session_id, + ) + entry = self._runner.session_store._entries.get(ctx.session_key) + _session_split_entry_persisted = False + if entry: + entry_session_id = getattr(entry, "session_id", None) + if not ctx._run_still_current(): + logger.info( + "Skipping session split sync for stale run %s — " + "generation %s is no longer current", + ctx.session_key or "?", + ctx.run_generation, + ) + elif entry_session_id == agent_session_id: + _session_split_entry_persisted = True + elif entry_session_id != ctx.session_id: + logger.info( + "Skipping session split sync for %s because the " + "session binding moved from %s to %s before " + "compression finished", + ctx.session_key or "?", + ctx.session_id, + entry_session_id, + ) + else: + entry.session_id = agent_session_id + self._runner.session_store._save() + self._runner.session_store._record_gateway_session_peer( + agent_session_id, + ctx.session_key, + ctx.source, + ) + _session_split_entry_persisted = True + + # Telegram DM whose source.thread_id was lost in the session split (synthetic/recovered + # event): restore it from the binding so _thread_metadata_for_source yields the right + # message_thread_id instead of the General thread (non-fatal). Only after this run + # published its split — a stale /stop→/new predecessor must not mutate routing state. + if _session_split_entry_persisted and ( + getattr(ctx.source, "platform", None) == Platform.TELEGRAM + and getattr(ctx.source, "chat_type", None) == "dm" + and getattr(ctx.source, "thread_id", None) is None + and self._runner._session_db is not None + ): + try: + # run_sync is off-loop (executor); sync DB is fine. + _binding = self._runner._session_db._db.get_telegram_topic_binding_by_session( + session_id=agent_session_id, + ) + if _binding and _binding.get("thread_id"): + ctx.source.thread_id = str(_binding["thread_id"]) + logger.debug( + "Restored source.thread_id=%s from binding after session split %s → %s", + ctx.source.thread_id, + ctx.session_id, + agent_session_id, + ) + except Exception: + logger.debug( + "Failed to restore thread_id from binding after session split", + exc_info=True, + ) + if _session_split_entry_persisted: + self._runner._sync_telegram_topic_binding( + ctx.source, entry, reason="agent-run-compression", + ) + + effective_session_id = agent_session_id + self._runner._sync_session_model_from_agent(effective_session_id, agent) + # history_offset=0 whenever the agent's message list lost the original history prefix: rotation + # (split) OR in-place compaction. Either way the returned `messages` is the compacted set, so + # persist all of it; slicing past the pre-compaction length would drop everything. + _effective_history_offset = ( + 0 if (_session_was_split or _compacted_in_place) else len(agent_history) + ) + return _compacted_in_place, effective_session_id, _effective_history_offset + + def run_sync(self): + from gateway.run import ( + _collect_auto_append_media_tags, + _current_max_iterations, + _normalize_empty_agent_response, + _sanitize_gateway_final_response, + ) + ctx = self._ctx + # As a method the turn message lives on the shared TurnContext: every rebind writes + # `ctx.message`, so the outer `_run_agent_inner` body sees the update as via the closure cell. + + # session_key propagates via contextvars (_set_session_env / set_current_session_key): + # concurrency-safe and inherited by tool worker threads. Deliberately do NOT write + # os.environ["HERMES_SESSION_KEY"]: it is process-global, so concurrent sessions would clobber + # each other and a tool thread with an unset contextvar would read the wrong key, misrouting + # approvals. Only the TUI slash-worker subprocess exports the env var (from its own argv). + + # Map platform enum to the platform hint key the agent understands. + # Platform.LOCAL ("local") maps to "cli"; others pass through as-is. + platform_key = "cli" if ctx.source.platform == Platform.LOCAL else ctx.source.platform.value + + # Combine platform context, YAML channel_prompts hint for this chat, channel_overrides + # system_prompt (or global ephemeral), and the gateway ephemeral prompt. + combined_ephemeral = ctx.context_prompt or "" + event_channel_prompt = (ctx.channel_prompt or "").strip() + if event_channel_prompt: + combined_ephemeral = (combined_ephemeral + "\n\n" + event_channel_prompt).strip() + cfg_channel_prompt = self._runner._get_system_prompt_for_channel( + ctx.source.platform, + ctx.source.chat_id or "", + thread_id=getattr(ctx.source, "thread_id", None), + parent_id=getattr(ctx.source, "parent_chat_id", None), + ) + if cfg_channel_prompt: + combined_ephemeral = (combined_ephemeral + "\n\n" + cfg_channel_prompt).strip() + + max_iterations = _current_max_iterations() + + try: + model, runtime_kwargs = self._runner._resolve_session_agent_runtime( + source=ctx.source, + session_key=ctx.session_key, + user_config=ctx.user_config, + ) + logger.debug( + "run_agent resolved: model=%s provider=%s session=%s", + model, runtime_kwargs.get("provider"), ctx.session_key or "", + ) + except Exception as exc: + return { + "final_response": f"⚠️ Provider authentication failed: {exc}", + "messages": [], + "api_calls": 0, + "tools": [], + } + + pr = self._runner._provider_routing + reasoning_config = self._runner._resolve_session_reasoning_config( + source=ctx.source, + session_key=ctx.session_key, + model=model, + ) + self._runner._reasoning_config = reasoning_config + self._runner._service_tier = self._runner._resolve_session_service_tier( + source=ctx.source, session_key=ctx.session_key + ) + ( + _stream_consumer, + _stream_delta_cb, + _interim_assistant_cb, + _want_interim_messages, + ) = self._setup_stream_consumer(platform_key) + + turn_route = self._runner._resolve_turn_agent_config(ctx.message, model, runtime_kwargs) + agent, reused_cached_agent = self._resolve_turn_agent( + turn_route, platform_key, combined_ephemeral, max_iterations, reasoning_config, pr, + ) + self._wire_turn_agent_callbacks( + agent, turn_route, reasoning_config, + _stream_delta_cb, _interim_assistant_cb, _want_interim_messages, + ) + agent_history, observed_group_context, _history_media_paths = ( + self._load_turn_history(agent, reused_cached_agent) + ) + _persist_user_message_override, _persist_user_timestamp_override = ( + self._prepare_turn_message(agent_history) + ) + result = self._run_conversation_with_approval( + agent, agent_history, observed_group_context, + _persist_user_message_override, _persist_user_timestamp_override, + ) + self._finish_stream_consumer(result, agent_history, _stream_consumer) + + # Signal the streaming-TTS consumer that the agent is done. finish() runs on the outer + # event-loop thread after the executor returns, so early run_sync returns are also finalised. + + # Return final response, or a message if something went wrong + final_response = result.get("final_response") + + # Extract actual token counts from the agent instance used for this run + _last_prompt_toks = 0 + _input_toks = 0 + _output_toks = 0 + _context_length = 0 + _agent = ctx.agent_holder[0] + if _agent and hasattr(_agent, "context_compressor"): + _last_prompt_toks = getattr(_agent.context_compressor, "last_prompt_tokens", 0) + _input_toks = getattr(_agent, "session_prompt_tokens", 0) + _output_toks = getattr(_agent, "session_completion_tokens", 0) + _context_length = getattr(_agent.context_compressor, "context_length", 0) or 0 + _resolved_model = getattr(_agent, "model", None) if _agent else None + + _compacted_in_place, effective_session_id, _effective_history_offset = ( + self._sync_session_after_run(agent_history) + ) + + if not final_response: + final_response = _normalize_empty_agent_response( + result, final_response or "", history_len=len(agent_history), + ) + final_response = _sanitize_gateway_final_response(ctx.source.platform, final_response) + if not final_response: + final_response = f"⚠️ {result['error']}" if result.get("error") else "" + return { + "final_response": final_response, + "messages": result.get("messages", []), + "api_calls": result.get("api_calls", 0), + "failed": result.get("failed", False), + # Sibling of the non-empty-response return below: the classifier's failure_reason + # must survive the empty-response path too, or downstream consumers (TUI billing, + # transient-failure persistence) lose the structured reason when no text was produced. + "failure_reason": result.get("failure_reason"), + "partial": result.get("partial", False), + "completed": result.get("completed"), + "interrupted": result.get("interrupted", False), + "interrupt_message": result.get("interrupt_message"), + "error": result.get("error"), + "compression_exhausted": result.get("compression_exhausted", False), + "compression_deferred": result.get("compression_deferred", False), + "tools": ctx.tools_holder[0] or [], + "history_offset": _effective_history_offset, + "compacted_in_place": _compacted_in_place, + "session_id": effective_session_id, + "last_prompt_tokens": _last_prompt_toks, + "input_tokens": _input_toks, + "output_tokens": _output_toks, + "model": _resolved_model, + "context_length": _context_length, + } + + # Append MEDIA: tags from tool results (e.g. TTS) that the model's final text omits, so + # extract_media() delivers each file once. Scope to THIS turn (slice at ``len(agent_history)``) + # so a stale MEDIA: path from an earlier turn doesn't ride a later text-only reply; dedup + # against _history_media_paths is the secondary guard — and the sole one on the fallback + # branch when mid-run compression shrank the list below the history length. + if "MEDIA:" not in final_response: + media_tags, has_voice_directive = _collect_auto_append_media_tags( + result.get("messages", []), + history_offset=len(agent_history), + history_media_paths=_history_media_paths, + ) + + if media_tags: + seen = set() + unique_tags = [] + for tag in media_tags: + if tag not in seen: + seen.add(tag) + unique_tags.append(tag) + if has_voice_directive: + unique_tags.insert(0, "[[audio_as_voice]]") + final_response = final_response + "\n" + "\n".join(unique_tags) + + # Auto-titling runs at TURN START (agent/turn_context.py) from the user's message alone, so a + # failed/interrupted turn is still titled. Thread-rename callbacks are attached as + # `_on_session_title` before the run because the titler fires from the turn prologue. + + return { + "final_response": final_response, + "last_reasoning": result.get("last_reasoning"), + "messages": ctx.result_holder[0].get("messages", []) if ctx.result_holder[0] else [], + "api_calls": ctx.result_holder[0].get("api_calls", 0) if ctx.result_holder[0] else 0, + "failed": ctx.result_holder[0].get("failed", False) if ctx.result_holder[0] else False, + "failure_reason": ( + ctx.result_holder[0].get("failure_reason") if ctx.result_holder[0] else None + ), + "completed": ctx.result_holder[0].get("completed") if ctx.result_holder[0] else None, + "interrupted": ctx.result_holder[0].get("interrupted", False) if ctx.result_holder[0] else False, + "partial": ctx.result_holder[0].get("partial", False) if ctx.result_holder[0] else False, + "error": ctx.result_holder[0].get("error") if ctx.result_holder[0] else None, + "interrupt_message": ctx.result_holder[0].get("interrupt_message") if ctx.result_holder[0] else None, + "compression_exhausted": ( + ctx.result_holder[0].get("compression_exhausted", False) + if ctx.result_holder[0] else False + ), + # Soft lock-contention defer: distinct from compression_exhausted so the gateway never + # auto-resets a session that a concurrent compressor is about to shrink. + "compression_deferred": ( + ctx.result_holder[0].get("compression_deferred", False) + if ctx.result_holder[0] else False + ), + "tools": ctx.tools_holder[0] or [], + "history_offset": _effective_history_offset, + "compacted_in_place": _compacted_in_place, + "last_prompt_tokens": _last_prompt_toks, + "input_tokens": _input_toks, + "output_tokens": _output_toks, + "model": _resolved_model, + "context_length": _context_length, + "session_id": effective_session_id, + "response_previewed": result.get("response_previewed", False), + "response_transformed": result.get("response_transformed", False), + # Pass through agent_persisted so the persistence block above can tell whether the codex + # app-server path self-persisted (it didn't — see codex_runtime.py); default True keeps the + # skip-db behaviour for the standard runtime. + "agent_persisted": (ctx.result_holder[0].get("agent_persisted", True) if ctx.result_holder[0] else True), + } diff --git a/gateway/run_voice.py b/gateway/run_voice.py new file mode 100644 index 0000000000..2157c92912 --- /dev/null +++ b/gateway/run_voice.py @@ -0,0 +1,557 @@ +"""Voice-channel / auto-TTS methods for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import functools +import json +import os +import re +import sys +import time +from contextlib import suppress +from gateway.config import Platform +from gateway.platforms.base import MessageEvent, MessageType, build_auto_tts_output_path +from gateway.session import SessionSource +from typing import Any, Awaitable, Callable, Dict, List, Optional, cast + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewayVoiceMixin: + """Voice-channel / auto-TTS methods for GatewayRunner.""" + + def _voice_key( + self, platform: Platform, chat_id: str, profile: Optional[str] = None + ) -> str: + """Return a platform-namespaced key for voice mode state. + + Under multiplexing the key is ``::`` (profile whose bot speaks); + the default profile keeps ``:`` so persisted state stays valid. Otherwise + two bots in one Discord channel share a key and one profile's ``/voice`` flips the other's. + """ + base = f"{platform.value}:{chat_id}" + profile = profile.strip() if isinstance(profile, str) else "" + if not profile or profile == "default": + return base + return f"{profile}:{base}" + + def _voice_key_for_source(self, source: SessionSource) -> str: + """Voice-state key for an inbound source, namespaced by its transport owner. + + Voice mode belongs to the (bot, chat) pair, so the namespace is the profile that OWNS the + receiving adapter (matching ``_sync_voice_mode_state_to_adapter``), not the routed profile. + """ + return self._voice_key( + source.platform, + source.chat_id, + profile=self._adapter_profile_for_source(source), + ) + + def _bind_voice_input_callback(self, adapter) -> None: + """Route voice transcripts back through the adapter that captured them.""" + if hasattr(adapter, "_voice_input_callback"): + adapter._voice_input_callback = functools.partial( + self._handle_voice_channel_input, adapter=adapter + ) + + def _load_voice_modes(self) -> Dict[str, str]: + try: + data = json.loads(self._VOICE_MODE_PATH.read_text(encoding="utf-8")) + except (FileNotFoundError, json.JSONDecodeError, OSError): + return {} + + if not isinstance(data, dict): + return {} + + valid_modes = {"off", "voice_only", "all"} + result = {} + for chat_id, mode in data.items(): + if mode not in valid_modes: + continue + key = str(chat_id) + # Skip legacy unprefixed keys (warn and skip) + if ":" not in key: + logger.warning( + "Skipping legacy unprefixed voice mode key %r during migration. " + "Re-enable voice mode on that chat to rebuild the prefixed key.", + key, + ) + continue + result[key] = mode + return result + + def _save_voice_modes(self) -> None: + try: + self._VOICE_MODE_PATH.parent.mkdir(parents=True, exist_ok=True) + self._VOICE_MODE_PATH.write_text( + json.dumps(self._voice_mode, indent=2), encoding="utf-8" + ) + except OSError as e: + logger.warning("Failed to save voice modes: %s", e) + + @staticmethod + def _toggle_adapter_auto_tts_set(adapter, chat_id: str, on: bool, *, add_to: str, clear_from: str) -> None: + """Add/discard ``chat_id`` in the adapter's ``add_to`` set; adding also clears it from ``clear_from``. + + ``/voice off`` and an explicit ``/voice on``/``/voice tts`` are hard overrides of each other.""" + target = getattr(adapter, add_to, None) + if not isinstance(target, set): + return + if on: + target.add(chat_id) + other = getattr(adapter, clear_from, None) + if isinstance(other, set): + other.discard(chat_id) + else: + target.discard(chat_id) + + def _set_adapter_auto_tts_disabled(self, adapter, chat_id: str, disabled: bool) -> None: + """Update an adapter's in-memory auto-TTS suppression set if present.""" + self._toggle_adapter_auto_tts_set( + adapter, chat_id, disabled, add_to="_auto_tts_disabled_chats", clear_from="_auto_tts_enabled_chats" + ) + + def _set_adapter_auto_tts_enabled(self, adapter, chat_id: str, enabled: bool) -> None: + """Update an adapter's per-chat auto-TTS opt-in set (auto-TTS even when ``voice.auto_tts`` is False).""" + self._toggle_adapter_auto_tts_set( + adapter, chat_id, enabled, add_to="_auto_tts_enabled_chats", clear_from="_auto_tts_disabled_chats" + ) + + def _sync_voice_mode_state_to_adapter(self, adapter) -> None: + """Restore persisted /voice state into a live platform adapter. + + Sets ``_auto_tts_default`` (from ``voice.auto_tts``) and, from ``self._voice_mode``, + ``_auto_tts_enabled_chats`` (modes ``voice_only``/``all``) and ``_auto_tts_disabled_chats`` + (mode ``off``). + """ + platform = getattr(adapter, "platform", None) + if not isinstance(platform, Platform): + return + + disabled_chats = getattr(adapter, "_auto_tts_disabled_chats", None) + enabled_chats = getattr(adapter, "_auto_tts_enabled_chats", None) + if not isinstance(disabled_chats, set) and not isinstance(enabled_chats, set): + return + + # Push the global voice.auto_tts default (config.yaml) onto the adapter. + # Lazy import to avoid adding a module-level dep from gateway → hermes_cli. + try: + from hermes_cli.config import load_config as _load_full_config + _full_cfg = _load_full_config() + _auto_tts_default = bool( + (_full_cfg.get("voice") or {}).get("auto_tts", False) + ) + except Exception: + _auto_tts_default = False + if hasattr(adapter, "_auto_tts_default"): + adapter._auto_tts_default = _auto_tts_default + + prefix = self._voice_key(platform, "", profile=getattr(adapter, "_owner_profile", None)) + if isinstance(disabled_chats, set): + disabled_chats.clear() + disabled_chats.update( + key[len(prefix):] for key, mode in self._voice_mode.items() + if mode == "off" and key.startswith(prefix) + ) + if isinstance(enabled_chats, set): + enabled_chats.clear() + enabled_chats.update( + key[len(prefix):] for key, mode in self._voice_mode.items() + if mode in {"voice_only", "all"} and key.startswith(prefix) + ) + + @staticmethod + def _get_guild_id(event: MessageEvent) -> Optional[int]: + """Extract Discord guild_id from the raw message object.""" + raw = getattr(event, "raw_message", None) + if raw is None: + return None + # Slash command interaction + if hasattr(raw, "guild_id") and raw.guild_id: + return int(raw.guild_id) + # Regular message + if hasattr(raw, "guild") and raw.guild: + return raw.guild.id + return None + + async def _handle_voice_channel_join(self, event: MessageEvent) -> str: + """Join the user's current Discord voice channel.""" + adapter = self._adapter_for_source(event.source) + if not hasattr(adapter, "join_voice_channel"): + return "Voice channels are not supported on this platform." + + guild_id = self._get_guild_id(event) + if not guild_id: + return "This command only works in a Discord server." + + voice_channel = await adapter.get_user_voice_channel( + guild_id, event.source.user_id + ) + if not voice_channel: + return "You need to be in a voice channel first." + + # Wire callbacks BEFORE join so voice input arriving immediately + # after connection is not lost. + self._bind_voice_input_callback(adapter) + voice_profile = self._adapter_profile_for_source(event.source) + if hasattr(adapter, "_on_voice_disconnect"): + adapter._on_voice_disconnect = functools.partial( + self._handle_voice_timeout_cleanup, adapter=adapter + ) + # Let the adapter's inactivity timer see the live voice-reply mode so it + # doesn't disconnect a deliberately text-only (/voice off) session. + if hasattr(adapter, "_voice_mode_getter"): + adapter._voice_mode_getter = lambda chat_id: self._voice_mode.get( + self._voice_key(Platform.DISCORD, str(chat_id), profile=voice_profile), + "off", + ) + + try: + success = await adapter.join_voice_channel(voice_channel) + except Exception as e: + logger.warning("Failed to join voice channel: %s", e) + adapter._voice_input_callback = None + err_lower = str(e).lower() + if "pynacl" in err_lower or "nacl" in err_lower or "davey" in err_lower: + return ( + "Voice dependencies are missing (PyNaCl / davey). " + f"Install with: `{sys.executable} -m pip install PyNaCl`" + ) + return f"Failed to join voice channel: {e}" + + if success: + adapter._voice_text_channels[guild_id] = int(event.source.chat_id) + if hasattr(adapter, "_voice_sources"): + adapter._voice_sources[guild_id] = event.source.to_dict() + self._voice_mode[self._voice_key_for_source(event.source)] = "all" + self._save_voice_modes() + self._set_adapter_auto_tts_enabled(adapter, event.source.chat_id, enabled=True) + return ( + f"Joined voice channel **{voice_channel.name}**.\n" + f"I'll speak my replies and listen to you. Use /voice leave to disconnect." + ) + # Join failed — clear callback + adapter._voice_input_callback = None + return "Failed to join voice channel. Check bot permissions (Connect + Speak)." + + async def _handle_voice_channel_leave(self, event: MessageEvent) -> str: + """Leave the Discord voice channel.""" + adapter = self._adapter_for_source(event.source) + guild_id = self._get_guild_id(event) + + if not guild_id or not hasattr(adapter, "leave_voice_channel"): + return "Not in a voice channel." + + if not hasattr(adapter, "is_in_voice_channel") or not adapter.is_in_voice_channel(guild_id): + return "Not in a voice channel." + + try: + await adapter.leave_voice_channel(guild_id) + except Exception as e: + logger.warning("Error leaving voice channel: %s", e) + # Always clean up state even if leave raised an exception + self._voice_mode[self._voice_key_for_source(event.source)] = "off" + self._save_voice_modes() + self._set_adapter_auto_tts_disabled(adapter, event.source.chat_id, disabled=True) + if hasattr(adapter, "_voice_input_callback"): + adapter._voice_input_callback = None + return "Left voice channel." + + def _handle_voice_timeout_cleanup(self, chat_id: str, *, adapter=None) -> None: + """Called by the adapter when a voice channel times out. + + Cleans up runner-side voice_mode state that the adapter cannot reach. ``adapter`` is the + Discord adapter that timed out (bound at join time); under multiplexing that is a + specific profile's bot, not necessarily ``self.adapters[DISCORD]``. + """ + if adapter is None: + adapter = self.adapters.get(Platform.DISCORD) + profile = getattr(adapter, "_owner_profile", None) + self._voice_mode[self._voice_key(Platform.DISCORD, chat_id, profile=profile)] = "off" + self._save_voice_modes() + self._set_adapter_auto_tts_disabled(adapter, chat_id, disabled=True) + + def _is_duplicate_voice_transcript(self, guild_id: int, user_id: int, transcript: str) -> bool: + """Suppress repeated STT outputs for the same recent utterance. + + Voice capture can occasionally emit the same utterance twice a few seconds apart, which + creates a second queued agent run and overlapping spoken replies. + """ + from difflib import SequenceMatcher + + normalized = re.sub(r"\s+", " ", transcript).strip().lower() + normalized = re.sub(r"[^\w\s]", "", normalized) + if not normalized: + return False + + now = time.monotonic() + window_seconds = 12.0 + key = (guild_id, user_id) + recent_store = getattr(self, "_recent_voice_transcripts", None) + if not isinstance(recent_store, dict): + recent_store = {} + self._recent_voice_transcripts = recent_store + recent = [ + (ts, txt) + for ts, txt in recent_store.get(key, []) + if now - ts <= window_seconds + ] + + for _, prior in recent: + if prior == normalized: + recent_store[key] = recent + return True + if len(prior) >= 16 and len(normalized) >= 16: + if SequenceMatcher(None, prior, normalized).ratio() >= 0.95: + recent_store[key] = recent + return True + + recent.append((now, normalized)) + recent_store[key] = recent[-5:] + return False + + async def _handle_voice_channel_input( + self, guild_id: int, user_id: int, transcript: str, *, adapter=None + ): + """Handle transcribed voice from a user in a voice channel. + + ``adapter`` is the Discord adapter that captured the audio (bound via + ``_bind_voice_input_callback``); under multiplexing each profile's bot must dispatch + through its own adapter, never the default profile's. + """ + if adapter is None: + adapter = self.adapters.get(Platform.DISCORD) + if not adapter: + return + + text_ch_id = adapter._voice_text_channels.get(guild_id) + if not text_ch_id: + return + + # Build source — reuse the linked text channel's metadata when available + # so voice input shares the same session as the bound text conversation. + source_data = getattr(adapter, "_voice_sources", {}).get(guild_id) + if source_data: + source = SessionSource.from_dict(source_data) + source.user_id = str(user_id) + source.user_name = str(user_id) + else: + source = SessionSource( + platform=Platform.DISCORD, + chat_id=str(text_ch_id), + user_id=str(user_id), + user_name=str(user_id), + chat_type="channel", + profile=getattr(adapter, "_owner_profile", None), + ) + + # Check authorization before processing voice input + if not self._is_user_authorized(source): + logger.debug("Unauthorized voice input from user %d, ignoring", user_id) + return + + if self._is_duplicate_voice_transcript(guild_id, user_id, transcript): + logger.info( + "Suppressing duplicate voice transcript for guild=%s user=%s: %s", + guild_id, + user_id, + transcript[:100], + ) + return + + # Show transcript in text channel (after auth, with mention sanitization) + try: + channel = adapter._client.get_channel(text_ch_id) + if channel: + safe_text = transcript[:2000].replace("@everyone", "@\u200beveryone").replace("@here", "@\u200bhere") + await channel.send(f"**[Voice]** <@{user_id}>: {safe_text}") + except Exception: + pass + + # Build a synthetic MessageEvent for the normal pipeline; SimpleNamespace raw_message lets + # _get_guild_id() extract guild_id and _send_voice_reply() play audio in the voice channel. + from types import SimpleNamespace + # Resolve the bound text channel's channel_prompt so voice input gets + # the same per-channel context as typed messages (#50149). + channel_prompt: Optional[str] = None + resolver = getattr(adapter, "_resolve_channel_prompt", None) + if callable(resolver): + try: + resolved = resolver(str(text_ch_id)) + channel_prompt = resolved if isinstance(resolved, str) else None + except Exception: + channel_prompt = None + event = MessageEvent( + source=source, + text=transcript, + message_type=MessageType.VOICE, + raw_message=SimpleNamespace(guild_id=guild_id, guild=None), + channel_prompt=channel_prompt, + ) + + await adapter.handle_message(event) + + def _should_send_voice_reply( + self, + event: MessageEvent, + response: str, + agent_messages: list, + already_sent: bool = False, + ) -> bool: + """Decide whether the runner should send a TTS voice reply. + + False when voice_mode is off for this chat, the response is empty/an error, the agent + already called text_to_speech (dedup), or voice input + base adapter auto-TTS already + handled it (skip_double) — UNLESS streaming consumed the response (already_sent=True), + since then the base adapter has no text for auto-TTS and the runner must handle it. + """ + if not response or response.startswith("Error:"): + return False + + chat_id = event.source.chat_id + voice_key = self._voice_key_for_source(event.source) + voice_mode = self._voice_mode.get(voice_key) + is_voice_input = (event.message_type == MessageType.VOICE) + + adapter = self._adapter_for_source(event.source) + adapter_auto_tts = False + if adapter and hasattr(adapter, "_should_auto_tts_for_chat"): + try: + adapter_auto_tts = bool(adapter._should_auto_tts_for_chat(chat_id)) + except Exception: + adapter_auto_tts = False + + should = ( + (voice_mode == "all") + or (voice_mode == "voice_only" and is_voice_input) + # ``voice.auto_tts`` (synced into the adapter at startup) is the fallback only when the + # chat has no explicit mode; the chat-level all/voice_only/off choice takes precedence. + or (voice_mode is None and adapter_auto_tts) + ) + if not should: + logger.debug( + "Auto voice reply skipped: mode=%s adapter_auto_tts=%s chat=%s platform=%s", + voice_mode, adapter_auto_tts, chat_id, event.source.platform.value, + ) + return False + + # Dedup: agent already called TTS tool in THIS turn only + last_user_idx = None + for i, msg in enumerate(reversed(agent_messages)): + if msg.get("role") == "user": + last_user_idx = len(agent_messages) - 1 - i; break + turn_messages = agent_messages[last_user_idx:] if last_user_idx is not None else agent_messages + has_agent_tts = any( + msg.get("role") == "assistant" + and any( + (tc.get("function") or {}).get("name") == "text_to_speech" + for tc in (msg.get("tool_calls") or []) + ) + for msg in turn_messages + ) + if has_agent_tts: + return False + + # Dedup: base adapter auto-TTS already handles voice input (play_tts plays in VC when + # connected), so the runner can skip — unless streaming already delivered the text + # (already_sent): then the base adapter gets None, can't run auto-TTS, and the runner must. + return not (is_voice_input and not already_sent) + + def _should_echo_stt_transcripts(self) -> bool: + """Return whether inbound voice/STT transcripts should be echoed to chat.""" + return bool(getattr(self.config, "stt_echo_transcripts", True)) + + async def _send_voice_reply(self, event: MessageEvent, text: str) -> None: + """Generate TTS audio and send as a voice message before the text reply.""" + audio_path = None + actual_paths: List[str] = [] + try: + from tools.tts_tool import text_to_speech_tool, _strip_markdown_for_tts + + tts_text = _strip_markdown_for_tts(text) + if not tts_text: + return + + # Platforms whose native voice bubbles require Ogg/Opus (OPUS_VOICE_PLATFORMS — + # Telegram, Matrix, Feishu, WhatsApp, Signal) get an explicit .ogg path; the TTS tool's + # central container repair guarantees real Ogg/Opus bytes for every provider. + audio_path = build_auto_tts_output_path(event.source.platform) + + result_json = await asyncio.to_thread( + text_to_speech_tool, text=tts_text, output_path=audio_path + ) + try: + result = json.loads(result_json) + except (json.JSONDecodeError, TypeError): + logger.warning("Auto voice reply TTS returned invalid JSON: %s", result_json[:200] if result_json else result_json) + return + + # Delivery may be one combined file or several separately valid files (combination + # unavailable or over a platform limit); preserve legacy single-file results. + actual_paths = result.get("file_paths") or [ + result.get("file_path", audio_path) + ] + actual_paths = [ + str(path) for path in actual_paths + if path and os.path.isfile(path) + ] + if not result.get("success") or not actual_paths: + logger.warning("Auto voice reply TTS failed: %s", result.get("error")) + return + + adapter = self._adapter_for_source(event.source) + + # If connected to a voice channel, play there instead of sending a file + guild_id = self._get_guild_id(event) + play_in_voice_channel = getattr(adapter, "play_in_voice_channel", None) + is_in_voice_channel = getattr(adapter, "is_in_voice_channel", None) + send_voice = getattr(adapter, "send_voice", None) + in_voice_channel = bool( + guild_id + and callable(play_in_voice_channel) + and callable(is_in_voice_channel) + and is_in_voice_channel(guild_id) + ) + reply_anchor = self._reply_anchor_for_event(event) + thread_meta = self._thread_metadata_for_source(event.source, reply_anchor) + if not in_voice_channel and callable(send_voice): + # Mark the auto voice reply as notify-worthy (mirrors the final-text path in + # platforms/base.py) so adapters that gate push notifications (Telegram "important" + # mode) deliver it as a normal notification, not a silent message. Clone first so + # we don't mutate metadata shared with concurrent typing-indicator state. + if thread_meta is not None: + thread_meta = dict(thread_meta) + thread_meta["notify"] = True + else: + thread_meta = {"notify": True} + for actual_path in actual_paths: + if in_voice_channel: + play_voice = cast(Callable[..., Awaitable[Any]], play_in_voice_channel) + await play_voice(guild_id, actual_path) + elif callable(send_voice): + send_voice_call = cast(Callable[..., Awaitable[Any]], send_voice) + send_kwargs: Dict[str, Any] = { + "chat_id": event.source.chat_id, + "audio_path": actual_path, + "reply_to": reply_anchor, + "metadata": thread_meta, + } + await send_voice_call(**send_kwargs) + except Exception as e: + logger.warning("Auto voice reply failed: %s", e, exc_info=True) + finally: + for p in ({audio_path, *actual_paths} - {None}): + with suppress(OSError): + os.unlink(p) diff --git a/gateway/run_watchers.py b/gateway/run_watchers.py new file mode 100644 index 0000000000..d0fc1e9821 --- /dev/null +++ b/gateway/run_watchers.py @@ -0,0 +1,455 @@ +"""Session expiry / stall / catalog-refresh watcher loops for GatewayRunner. + +Split out of ``gateway/run.py``; bound onto ``GatewayRunner`` via the MRO. +``gateway.run`` internals are imported lazily inside method bodies (import cycle), +so ``patch("gateway.run.X")`` keeps intercepting them at call time. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING +import asyncio +import time +from typing import Any, Dict, Optional + +if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle) + from gateway.run import GatewayRunner, TurnRunner # noqa: F401 + +# Log-record parity with the origin module. +logger = logging.getLogger("gateway.run") + + +class GatewaySessionWatchersMixin: + """Session expiry / stall / catalog-refresh watcher loops for GatewayRunner.""" + + async def _session_expiry_watcher(self, interval: int = 300): + """Background task that finalizes expired sessions: runs ``on_session_finalize`` hooks, + cleans up the cached agent's tool resources, evicts the cache entry, and marks the session + finalized so it is not finalized again. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + await asyncio.sleep(60) # initial delay — let the gateway fully start + _finalize_failures: dict[str, int] = {} # session_id -> consecutive failure count + _MAX_FINALIZE_RETRIES = 3 + while self._running: + try: + await self.async_session_store._ensure_loaded() + # Collect expired sessions first, then log a single summary. + _expired_entries = [] + for key, entry in list(self.session_store._entries.items()): + if entry.expiry_finalized: + continue + if not await self.async_session_store._is_session_expired(entry): + continue + _expired_entries.append((key, entry)) + + if _expired_entries: + # Extract platform names from session keys for a compact summary. + # Keys look like "agent:main:telegram:dm:12345" — platform is field [2]. + _platforms: dict[str, int] = {} + for _k, _e in _expired_entries: + _parts = _k.split(":") + _plat = _parts[2] if len(_parts) > 2 else "unknown" + _platforms[_plat] = _platforms.get(_plat, 0) + 1 + _plat_summary = ", ".join( + f"{p}:{c}" for p, c in sorted(_platforms.items()) + ) + logger.info( + "Session expiry: %d sessions to finalize (%s)", + len(_expired_entries), _plat_summary, + ) + + for key, entry in _expired_entries: + try: + try: + _parts = key.split(":") + _platform = _parts[2] if len(_parts) > 2 else "" + # Off-loop + bounded: plugin finalize hooks can block arbitrarily, and + # this watcher runs on the gateway event loop. + await self._finalize_session_off_loop( + session_id=entry.session_id, + platform=_platform, + reason="session_expired", + ) + except Exception: + pass + # Close the cached agent's memory provider and tool resources. Idle agents + # live in _agent_cache (not _running_agents), so look there. + _cached_agent = None + _cache_lock = getattr(self, "_agent_cache_lock", None) + if _cache_lock is not None: + with _cache_lock: + _cached = self._agent_cache.get(key) + _cached_agent = _cached[0] if isinstance(_cached, tuple) else _cached if _cached else None + # Fall back to _running_agents in case the agent is + # still mid-turn when the expiry fires. + if _cached_agent is None: + _exp_state = self._peek_session_state(key) + _cached_agent = _exp_state.turn.agent if _exp_state else None + if _cached_agent and _cached_agent is not _AGENT_PENDING_SENTINEL: + await self._cleanup_agent_resources_off_loop( + _cached_agent, context="session expiry" + ) + # Drop the cache entry so the AIAgent (LLM clients, tool schemas, memory + # provider refs) can be GC'd; otherwise the cache grows unbounded. + self._evict_cached_agent(key) + # Permanent finalization: one funnel call drops every conversation-scoped + # dict AND boundary security state so they don't grow unbounded. Idle + # agent-cache eviction must NOT do this — that session is still alive and a + # resumed turn rebuilds from these overrides. Only finalize, /new, /reset clear. + self._clear_conversation_scope( + key, reason="expiry_finalized" + ) + # Persist finalized flag (sessions.json AND state.db, single write-path); + # also drops the /model override — finalization is a conversation boundary. + await self.async_session_store.set_expiry_finalized(entry) + logger.debug( + "Session expiry finalized for %s", + entry.session_id, + ) + _finalize_failures.pop(entry.session_id, None) + except Exception as e: + failures = _finalize_failures.get(entry.session_id, 0) + 1 + _finalize_failures[entry.session_id] = failures + if failures >= _MAX_FINALIZE_RETRIES: + logger.warning( + "Session finalize gave up after %d attempts for %s: %s. " + "Marking as finalized to prevent infinite retry loop.", + failures, entry.session_id, e, + ) + await self.async_session_store.set_expiry_finalized( + entry, clear_model_override=False + ) + _finalize_failures.pop(entry.session_id, None) + else: + logger.debug( + "Session finalize failed (%d/%d) for %s: %s", + failures, _MAX_FINALIZE_RETRIES, entry.session_id, e, + ) + + if _expired_entries: + _done = sum( + 1 for _, e in _expired_entries if e.expiry_finalized + ) + _failed = len(_expired_entries) - _done + if _failed: + logger.info( + "Session expiry done: %d finalized, %d pending retry", + _done, _failed, + ) + else: + logger.info( + "Session expiry done: %d finalized", _done, + ) + + # Sweep agents idle beyond the TTL regardless of session reset policy: sessions with + # long / "never" reset windows would otherwise pin memory for the gateway's life. + try: + _idle_evicted = self._sweep_idle_cached_agents() + if _idle_evicted: + logger.info( + "Agent cache idle sweep: evicted %d agent(s)", + _idle_evicted, + ) + except Exception as _e: + logger.debug("Idle agent sweep failed: %s", _e) + + # Neither LRU cap nor idle TTL knows what a cached transcript costs in memory, so a + # busy gateway keeps every warm session's tool output resident until the RSS limit. + try: + self._sweep_agent_cache_under_pressure() + except Exception as _e: + logger.debug("Agent cache pressure sweep failed: %s", _e) + + # Prune stale SessionStore entries; the in-memory dict (and sessions.json) would + # otherwise grow unbounded with many rotating chats / threads / users. + _last_prune_ts = getattr(self, "_last_session_store_prune_ts", 0.0) + _prune_interval = 3600.0 # once per hour + if time.time() - _last_prune_ts > _prune_interval: + try: + _max_age = int( + getattr(self.config, "session_store_max_age_days", 0) or 0 + ) + if _max_age > 0: + _pruned = await self.async_session_store.prune_old_entries(_max_age) + if _pruned: + logger.info( + "SessionStore prune: dropped %d stale entries", + _pruned, + ) + except Exception as _e: + logger.debug("SessionStore prune failed: %s", _e) + self._last_session_store_prune_ts = time.time() + except Exception as e: + logger.debug("Session expiry watcher error: %s", e) + # Sleep in small increments so we can stop quickly + for _ in range(interval): + if not self._running: + break + await asyncio.sleep(1) + + def _session_stall_timeout_seconds(self) -> float: + """Return configured stall timeout (seconds); 0 disables the watchdog.""" + from gateway.run import _float_env + return _float_env("HERMES_SESSION_STALL_TIMEOUT", 300) + + def _iter_gateway_adapters(self): + """Yield every live platform adapter (default + multiplex profiles).""" + seen: set[int] = set() + for adapter in list(getattr(self, "adapters", {}).values()): + if adapter is None: + continue + aid = id(adapter) + if aid in seen: + continue + seen.add(aid) + yield adapter + for amap in list(getattr(self, "_profile_adapters", {}).values()): + for adapter in list(amap.values()): + if adapter is None: + continue + aid = id(adapter) + if aid in seen: + continue + seen.add(aid) + yield adapter + + def _session_activity_for_stall(self, session_key: str) -> Optional[dict]: + """Return the shared activity snapshot for stall progress: the single source is + ``AIAgent.get_activity_summary()`` / ``agent.session_activity``; no turn-start or + pending-inbound clocks. + """ + from gateway.run import _AGENT_PENDING_SENTINEL + agent = (getattr(self, "_running_agents", None) or {}).get(session_key) + if agent is None or agent is _AGENT_PENDING_SENTINEL: + return None + if not hasattr(agent, "get_activity_summary"): + return None + try: + summary = agent.get_activity_summary() + except Exception: + return None + return summary if isinstance(summary, dict) else None + + async def _check_session_stalls(self, timeout_seconds: float) -> int: + """Scan pending inbound sessions and notify once per stall episode; returns the number of + notifications sent this pass (for tests). + """ + from gateway.run import _STALL_NOTIFY_SEND_TIMEOUT_SECONDS + from gateway.session_stall import ( + format_session_stall_notification, + resolve_session_idle_seconds_from_activity, + should_clear_session_stall_notification, + should_emit_session_stall_notification, + ) + + notified_map = getattr(self, "_session_stall_notified", None) + if notified_map is None: + notified_map = {} + self._session_stall_notified = notified_map + + sent = 0 + now = time.time() + candidates: Dict[str, tuple[Any, Any]] = {} + + for adapter in self._iter_gateway_adapters(): + pending_slot = getattr(adapter, "_pending_messages", None) or {} + for session_key, event in list(pending_slot.items()): + if session_key and session_key not in candidates and event is not None: + candidates[session_key] = (adapter, event) + + for session_key, overflow in list( + (getattr(self, "_queued_events", None) or {}).items() + ): + if not session_key or session_key in candidates or not overflow: + continue + event = overflow[0] + source = getattr(event, "source", None) + adapter = ( + self._adapter_for_source(source) if source is not None else None + ) + if adapter is None: + continue + candidates[session_key] = (adapter, event) + + for session_key, (adapter, pending_event) in list(candidates.items()): + has_pending = pending_event is not None + activity = ( + self._session_activity_for_stall(session_key) if has_pending else None + ) + idle_seconds = ( + resolve_session_idle_seconds_from_activity(activity, now=now) + if has_pending + else None + ) + already = bool(notified_map.get(session_key)) + if should_clear_session_stall_notification( + timeout_seconds=timeout_seconds, + idle_seconds=idle_seconds, + has_pending_inbound=has_pending, + ): + notified_map.pop(session_key, None) + already = False + if not should_emit_session_stall_notification( + timeout_seconds=timeout_seconds, + idle_seconds=idle_seconds, + has_pending_inbound=has_pending, + already_notified=already, + ): + continue + + if idle_seconds is None: + continue + mins = max(1, int(idle_seconds // 60)) + activity = activity or {} + logger.warning( + "Session stall detected: session=%s idle=%.0fs " + "(timeout=%.0fs, ~%d min); pending inbound present " + "| last_activity=%s | provenance=%s " + "(agent.session_stall_timeout)", + session_key, + idle_seconds, + timeout_seconds, + mins, + activity.get("last_activity_desc") + or activity.get("last_activity_description") + or "unknown", + activity.get("provenance") + or activity.get("last_activity_provenance") + or "unknown", + ) + source = getattr(pending_event, "source", None) + chat_id = getattr(source, "chat_id", None) if source is not None else None + if not chat_id: + logger.warning( + "Session stall notify skipped (no chat_id): session=%s", + session_key, + ) + # Cannot deliver; latch to avoid log spam every tick. + notified_map[session_key] = True + continue + # Re-read pending state + activity IMMEDIATELY before delivery: the snapshot above ages + # while earlier candidates await sends; an agent that progressed (or drained its queue) + # must not get a false stall notice. Abort, latch un-set, so the next tick re-evaluates. + still_pending = ( + (getattr(adapter, "_pending_messages", None) or {}).get( + session_key + ) + is not None + or bool( + (getattr(self, "_queued_events", None) or {}).get( + session_key + ) + ) + ) + fresh_idle = resolve_session_idle_seconds_from_activity( + self._session_activity_for_stall(session_key), + now=time.time(), + ) + if not still_pending or ( + fresh_idle is not None and fresh_idle < timeout_seconds + ): + logger.info( + "Session stall notify aborted (no longer stale): " + "session=%s pending=%s fresh_idle=%s", + session_key, + still_pending, + fresh_idle, + ) + # Re-arm: drop any stale latch so a FUTURE genuine stall + # episode notifies again. + notified_map.pop(session_key, None) + continue + try: + metadata = ( + self._thread_metadata_for_source(source) + if source is not None and hasattr(self, "_thread_metadata_for_source") + else None + ) + # Bound the send: a wedged adapter transport (network hang, dead websocket) must not + # block the watcher pass — siblings would go unevaluated and the watcher stop. + try: + result = await asyncio.wait_for( + adapter.send( + str(chat_id), + format_session_stall_notification(idle_seconds), + metadata=metadata, + ), + timeout=_STALL_NOTIFY_SEND_TIMEOUT_SECONDS, + ) + except asyncio.TimeoutError: + logger.warning( + "Session stall notify send timed out after %.0fs " + "for %s; will retry next tick", + _STALL_NOTIFY_SEND_TIMEOUT_SECONDS, + session_key, + ) + continue # do not latch; retry next tick + # Adapters often return SendResult(success=False) instead of raising. + if result is not None and getattr(result, "success", True) is False: + logger.warning( + "Session stall notify failed for %s: %s", + session_key, + getattr(result, "error", "send returned success=False"), + ) + continue # do not latch; retry next tick + sent += 1 + notified_map[session_key] = True + except Exception as exc: + logger.warning( + "Session stall notify failed for %s: %s", + session_key, + exc, + ) + # Do not latch — retry next watcher tick until delivery or episode clear. + + # Drop latches for sessions that no longer appear in any pending map. + for key in list(notified_map.keys()): + if key not in candidates: + notified_map.pop(key, None) + + return sent + + async def _model_catalog_refresh_watcher(self) -> None: + """Refresh the /model picker's remote catalogs every TTL window. The picker itself only + refreshes on a cold/stale open, so if nobody opens ``/model`` the cache never updates. + """ + from hermes_cli.model_catalog import refresh_catalogs, refresh_interval_seconds + + await asyncio.sleep(30) # let startup settle + while self._running: + try: + await asyncio.to_thread(refresh_catalogs) + except Exception as exc: + logger.debug("Model catalog refresh failed: %s", exc) + try: + interval = refresh_interval_seconds() + except Exception: + interval = 1200.0 + deadline = time.monotonic() + interval + while self._running and time.monotonic() < deadline: + await asyncio.sleep(min(30.0, max(0.0, deadline - time.monotonic()))) + + async def _session_stall_watcher(self, interval: float = 30.0): + """Periodic pending-inbound + stale-activity stall watchdog. + + Progress comes only from ``get_activity_summary()``. Pending inbound is a notify policy + gate, not a progress clock. Notify-only: does not kill the turn (contrast + ``gateway_timeout`` / ``shutdown_watchdog``). + """ + # Short initial delay so startup reconnect noise does not false-fire. + await asyncio.sleep(min(30.0, max(1.0, float(interval)))) + while self._running: + try: + timeout = self._session_stall_timeout_seconds() + if timeout > 0: + await self._check_session_stalls(timeout) + except Exception as exc: + logger.debug("Session stall watcher error: %s", exc) + # Interruptible sleep + steps = max(1, int(float(interval))) + for _ in range(steps): + if not self._running: + break + await asyncio.sleep(1) diff --git a/tests/gateway/test_10710_auto_reset_evicts_cached_agent.py b/tests/gateway/test_10710_auto_reset_evicts_cached_agent.py index 7a450ef613..e8a65cda89 100644 --- a/tests/gateway/test_10710_auto_reset_evicts_cached_agent.py +++ b/tests/gateway/test_10710_auto_reset_evicts_cached_agent.py @@ -21,6 +21,8 @@ import ast import inspect from gateway import run as gateway_run +from gateway import run_turn as gateway_run_turn +from gateway import run_turn as gateway_run_turn def _calls(node: ast.AST) -> set[str]: @@ -53,7 +55,7 @@ def test_auto_reset_cleanup_evicts_cached_agent(): conversation's cached agent (and its leaked ``context_compressor._previous_summary``) — the cache is keyed on the stable ``session_key`` (#10710).""" - tree = ast.parse(inspect.getsource(gateway_run)) + tree = ast.parse(inspect.getsource(gateway_run_turn)) # Fingerprint the cleanup branch: the `if :` block that # clears the conversation scope via the funnel (post-#64934 refactor: diff --git a/tests/gateway/test_35809_auto_reset_clean_context.py b/tests/gateway/test_35809_auto_reset_clean_context.py index 4f6d4149f9..c75b063408 100644 --- a/tests/gateway/test_35809_auto_reset_clean_context.py +++ b/tests/gateway/test_35809_auto_reset_clean_context.py @@ -37,6 +37,8 @@ import ast import inspect from gateway import run as gateway_run +from gateway import run_turn as gateway_run_turn +from gateway import run_turn as gateway_run_turn from gateway.config import GatewayConfig, Platform from gateway.session import SessionSource, SessionStore from hermes_state import SessionDB @@ -47,7 +49,7 @@ from hermes_state import SessionDB # --------------------------------------------------------------------------- def _find_compression_exhausted_reset_block() -> ast.If: """Return the ``if agent_result.get('compression_exhausted') ...`` block.""" - tree = ast.parse(inspect.getsource(gateway_run)) + tree = ast.parse(inspect.getsource(gateway_run_turn)) for node in ast.walk(tree): if not isinstance(node, ast.If): diff --git a/tests/gateway/test_48031_model_switch_after_auto_reset.py b/tests/gateway/test_48031_model_switch_after_auto_reset.py index bdbfc2b71a..be34b99a81 100644 --- a/tests/gateway/test_48031_model_switch_after_auto_reset.py +++ b/tests/gateway/test_48031_model_switch_after_auto_reset.py @@ -24,6 +24,8 @@ import ast import inspect from gateway import run as gateway_run +from gateway import run_turn as gateway_run_turn +from gateway import run_turn as gateway_run_turn from gateway import slash_commands as gateway_slash @@ -47,7 +49,7 @@ def test_run_consumes_was_auto_reset_in_cleanup_block(): `session_entry.was_auto_reset = False` so the cleanup (which pops the session model/reasoning overrides) cannot re-fire on the next message and wipe an override stored between turns (#48031).""" - tree = ast.parse(inspect.getsource(gateway_run)) + tree = ast.parse(inspect.getsource(gateway_run_turn)) # Find the cleanup branch: an `if :` block that clears the # conversation scope (post-funnel: one _clear_conversation_scope call diff --git a/tests/gateway/test_approval_prompt_redaction.py b/tests/gateway/test_approval_prompt_redaction.py index bc17ad4e6c..592a71dcbf 100644 --- a/tests/gateway/test_approval_prompt_redaction.py +++ b/tests/gateway/test_approval_prompt_redaction.py @@ -112,7 +112,7 @@ class TestApprovalCommandWiring: ) def test_chat_platform_path_redacts_before_send(self): - import gateway.run as run + import gateway.run_turn_runner as run self._assert_redacts_then_uses(run, "_approval_notify_sync", "send_exec_approval") diff --git a/tests/gateway/test_compression_deferred_soft_result.py b/tests/gateway/test_compression_deferred_soft_result.py index e5099600e3..24284f74a7 100644 --- a/tests/gateway/test_compression_deferred_soft_result.py +++ b/tests/gateway/test_compression_deferred_soft_result.py @@ -22,6 +22,8 @@ import ast import inspect from gateway import run as gateway_run +from gateway import run_turn as gateway_run_turn +from gateway import run_turn as gateway_run_turn def _calls(node: ast.AST) -> set[str]: @@ -35,7 +37,7 @@ def _calls(node: ast.AST) -> set[str]: def _find_deferred_guarded_reset_chain() -> ast.If: """Return the ``if agent_result.get('compression_deferred') ... elif agent_result.get('compression_exhausted') ... reset_session`` chain.""" - tree = ast.parse(inspect.getsource(gateway_run)) + tree = ast.parse(inspect.getsource(gateway_run_turn)) for node in ast.walk(tree): if not isinstance(node, ast.If): diff --git a/tests/gateway/test_compression_session_id_persistence.py b/tests/gateway/test_compression_session_id_persistence.py index 90690352dd..1a5f1215cf 100644 --- a/tests/gateway/test_compression_session_id_persistence.py +++ b/tests/gateway/test_compression_session_id_persistence.py @@ -25,6 +25,10 @@ import textwrap from unittest.mock import MagicMock, call from gateway import run as gateway_run +from gateway import run_turn as gateway_run_turn +from gateway import run_turn_runner as gateway_run_turn_runner +from gateway import run_turn as gateway_run_turn +from gateway import run_turn_runner as gateway_run_turn_runner from gateway.session_context import set_current_session_id, get_session_env @@ -108,8 +112,9 @@ def test_every_post_compression_session_id_assignment_persists(): would compress correctly, the gateway would update its in-memory session_id, then drop it on next gateway restart. """ - source = inspect.getsource(gateway_run) - assignments = _session_id_assignments_followed_by_save(source) + assignments = [] + for mod in (gateway_run, gateway_run_turn, gateway_run_turn_runner): + assignments += _session_id_assignments_followed_by_save(inspect.getsource(mod)) assert assignments, ( "No ``session_entry.session_id = ...`` assignments found in gateway/run.py — " "either the structure changed or the AST walker is broken." diff --git a/tests/gateway/test_fallback_chain_reload.py b/tests/gateway/test_fallback_chain_reload.py index 431b100cfa..a94b8fff78 100644 --- a/tests/gateway/test_fallback_chain_reload.py +++ b/tests/gateway/test_fallback_chain_reload.py @@ -77,9 +77,8 @@ def test_background_and_main_agent_paths_call_refresh(): """ from pathlib import Path - source = ( - Path(__file__).resolve().parent.parent.parent / "gateway" / "run.py" - ).read_text(encoding="utf-8") + _gw = Path(__file__).resolve().parent.parent.parent / "gateway" + source = "\n".join(p.read_text(encoding="utf-8") for p in sorted(_gw.glob("run*.py"))) # The agent-construction site inside TurnRunner.run_sync (extracted from # the old _run_agent_inner closure) references the runner as # ``self._runner``; the background-agent site still uses bare ``self``. diff --git a/tests/gateway/test_profile_resolution.py b/tests/gateway/test_profile_resolution.py index e79e0415ca..ee827ca262 100644 --- a/tests/gateway/test_profile_resolution.py +++ b/tests/gateway/test_profile_resolution.py @@ -21,6 +21,8 @@ def mock_runner(): # Bind the actual methods to the mock runner._profile_name_for_source = GatewayRunner._profile_name_for_source.__get__(runner) runner._resolve_profile_home_for_source = GatewayRunner._resolve_profile_home_for_source.__get__(runner) + # _handle_message's ingress gates (profile route rejection) live in this helper. + runner._hm_admit_event = GatewayRunner._hm_admit_event.__get__(runner) return runner diff --git a/tests/gateway/test_restart_drain_recovery_dedup.py b/tests/gateway/test_restart_drain_recovery_dedup.py index 15c2b54e1e..01012f3386 100644 --- a/tests/gateway/test_restart_drain_recovery_dedup.py +++ b/tests/gateway/test_restart_drain_recovery_dedup.py @@ -170,7 +170,7 @@ def test_gateway_run_agent_threads_the_event_message_id_into_the_turn(): import ast import inspect - import gateway.run as gateway_run + import gateway.run_turn_runner as gateway_run source = inspect.getsource(gateway_run) tree = ast.parse(source) diff --git a/tests/gateway/test_stream_final_adoption_gate.py b/tests/gateway/test_stream_final_adoption_gate.py index 02f07bfa5f..a44ecb3104 100644 --- a/tests/gateway/test_stream_final_adoption_gate.py +++ b/tests/gateway/test_stream_final_adoption_gate.py @@ -75,7 +75,7 @@ class TestGateWiring: completed checks — a source-level pin so the contract test above cannot drift green while the call site regresses.""" import inspect - import gateway.run as run_mod + import gateway.run_turn_runner as run_mod src = inspect.getsource(run_mod) anchor = src.index("_final_for_stream = None") diff --git a/tests/gateway/test_voice_mode_platform_isolation.py b/tests/gateway/test_voice_mode_platform_isolation.py index 799029911f..f6c5932f8b 100644 --- a/tests/gateway/test_voice_mode_platform_isolation.py +++ b/tests/gateway/test_voice_mode_platform_isolation.py @@ -77,7 +77,7 @@ class TestLegacyKeyMigration: voice_path.write_text(json.dumps(legacy_data)) with patch.object(runner, "_VOICE_MODE_PATH", voice_path): - with patch("gateway.run.logger") as mock_logger: + with patch("gateway.run_voice.logger") as mock_logger: result = runner._load_voice_modes() # Legacy keys without ':' should be skipped