diff --git a/acp_adapter/server.py b/acp_adapter/server.py index 3cf2083719..488345e15f 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -18,48 +18,15 @@ from urllib.parse import unquote, urlparse import acp from acp.schema import ( - AgentCapabilities, - AgentMessageChunk, - AuthenticateResponse, - AvailableCommand, - AvailableCommandsUpdate, - BlobResourceContents, - ClientCapabilities, - EmbeddedResourceContentBlock, - ForkSessionResponse, - ImageContentBlock, - AudioContentBlock, - Implementation, - InitializeResponse, - ListSessionsResponse, - LoadSessionResponse, - McpServerHttp, - McpServerSse, - McpServerStdio, - ModelInfo, - NewSessionResponse, - PromptCapabilities, - PromptResponse, - ResumeSessionResponse, - SetSessionConfigOptionResponse, - SetSessionModelResponse, - SetSessionModeResponse, - ResourceContentBlock, - SessionCapabilities, - SessionForkCapabilities, - SessionInfoUpdate, - SessionListCapabilities, - SessionMode, - SessionModeState, - SessionModelState, - SessionResumeCapabilities, - SessionInfo, - TextContentBlock, - TextResourceContents, - UnstructuredCommandInput, - Usage, - UsageUpdate, - UserMessageChunk, + AgentCapabilities, AgentMessageChunk, AudioContentBlock, AuthenticateResponse, AvailableCommand, + AvailableCommandsUpdate, BlobResourceContents, ClientCapabilities, EmbeddedResourceContentBlock, + ForkSessionResponse, ImageContentBlock, Implementation, InitializeResponse, ListSessionsResponse, + LoadSessionResponse, McpServerHttp, McpServerSse, McpServerStdio, ModelInfo, NewSessionResponse, + PromptCapabilities, PromptResponse, ResourceContentBlock, ResumeSessionResponse, SessionCapabilities, + SessionForkCapabilities, SessionInfo, SessionInfoUpdate, SessionListCapabilities, SessionMode, + SessionModeState, SessionModelState, SessionResumeCapabilities, SetSessionConfigOptionResponse, + SetSessionModeResponse, SetSessionModelResponse, TextContentBlock, TextResourceContents, + UnstructuredCommandInput, Usage, UsageUpdate, UserMessageChunk, ) from acp_adapter.auth import TERMINAL_SETUP_AUTH_METHOD_ID, build_auth_methods, detect_provider @@ -96,22 +63,13 @@ PromptBlock = ( def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, str]]]]: - """Return ``(slug, label, [(model_id, description), ...])`` for named endpoints. + """``(slug, label, [(model_id, description), ...])`` for named endpoints (v12 ``providers:`` + and legacy ``custom_providers:``), which canonical provider enumeration never lists. - Covers both the v12 ``providers:`` mapping and the legacy ``custom_providers:`` - list. These endpoints never appear in canonical provider enumeration, so - without this the ACP model selector hides every named endpoint the TUI - ``/model`` picker renders. - - Model lists come from the entry's declared models (``default_model`` + - ``models``), refreshed from the endpoint's live ``/models`` listing when a - credential is available and ``discover_models`` is not disabled. Declared - models survive a failed live discovery — some OpenAI-compatible endpoints - expose no ``/models`` route yet serve the declared models fine. - - Slugs use the ``custom:`` shape that ``parse_model_input`` and - ``resolve_runtime_provider`` already resolve, so encoded choice ids - (``custom::``) round-trip through ``set_session_model``. + Models = the entry's declared models, refreshed from the live ``/models`` listing when a + credential exists and ``discover_models`` isn't disabled; declared models survive a failed + discovery (some endpoints have no ``/models`` route). Slugs use the ``custom:`` shape + ``parse_model_input``/``resolve_runtime_provider`` resolve, so choice ids round-trip. """ try: from hermes_cli.config import ( @@ -138,8 +96,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st logger.debug("Could not load named custom providers", exc_info=True) return [] - # ``get_compatible_custom_providers`` drops the ``enabled`` flag during - # normalization; collect disabled keys from the raw config instead. + # ``get_compatible_custom_providers`` drops ``enabled``; read disabled keys from raw config. raw_providers = cfg.get("providers") if isinstance(cfg, dict) else None disabled_keys = { str(key).strip().lower() @@ -209,16 +166,13 @@ try: except Exception: HERMES_VERSION = "0.0.0" -# Thread pool for running AIAgent (synchronous) in parallel. +# Runs the synchronous AIAgent off the event loop. _executor = ThreadPoolExecutor(max_workers=4, thread_name_prefix="acp-agent") -# ACP ListSessionsRequest has no client-side limit; clients paginate this fixed -# page via `cursor` / `next_cursor`. +# ListSessionsRequest has no client-side limit; clients paginate via `cursor`/`next_cursor`. _LIST_SESSIONS_PAGE_SIZE = 50 -# Per-provider cap for the ACP model selector: clients (Zed, Buzz) render the -# whole `availableModels` array in one dropdown. Mirrors the MoA picker cap -# (`hermes_cli/moa_cmd.py`). Bounds each provider's row, not the total; the -# current model is always kept via the fallback insert in `_build_model_state`. +# Per-provider row cap (clients render all `availableModels` in one dropdown; mirrors the +# MoA picker cap). Not a total cap; the current model is always kept via the fallback insert. ACP_MAX_MODELS_PER_PROVIDER = 200 _MAX_ACP_RESOURCE_BYTES = 512 * 1024 _TEXT_RESOURCE_MIME_PREFIXES = ("text/",) @@ -280,11 +234,8 @@ def _image_data_url(data: bytes, mime_type: str) -> str: def _path_from_file_uri(uri: str) -> Path | None: - """Convert local file URIs/paths from ACP clients into a readable Path. - - Zed may send POSIX file URIs from Linux/WSL workspaces or Windows-ish paths - when launched through wsl.exe; the Windows drive form becomes /mnt//... - """ + """Local file URI/path from an ACP client -> readable Path (None for non-file URIs). + Windows drive forms (Zed via wsl.exe) become ``/mnt//...``.""" raw = (uri or "").strip() if not raw: return None @@ -343,12 +294,8 @@ def _image_parts(uri: str, display: str, data: bytes, mime: str) -> list[dict[st def _resource_link_to_parts(block: ResourceContentBlock) -> list[dict[str, Any]]: - """Convert an ACP resource_link block to OpenAI content parts. - - Image resources produce an image_url part with a small text header so the - model knows which attachment it is; other resources return a single text - part with the inlined file body (or a binary-omit note). - """ + """ACP resource_link -> OpenAI content parts: images become a text header + image_url, + everything else a single text part with the inlined body (or a binary-omit note).""" uri = str(getattr(block, "uri", "") or "").strip() if not uri: return [] @@ -485,8 +432,7 @@ def _content_blocks_to_openai_user_content(prompt: list[PromptBlock]) -> str | l if not parts: return _extract_text(prompt) - # Pure text prompts stay strings so slash-command handling and text-only - # providers keep the legacy path; structured content only for real media. + # Pure text stays a string (slash commands, text-only providers); structured only for media. if all(part.get("type") == "text" for part in parts): return "\n".join(text_parts) @@ -554,12 +500,8 @@ def _estimate_tokens(history: list, agent: Any, system_prompt: str | None = None def _flatten_history_text(value: Any) -> str: - """Normalize a persisted text-or-text-parts value into one stripped string. - - Content (and provider reasoning fields) may be a scalar string or a list of - ``{"text": ...}`` / ``{"type": "text", "content": ...}`` parts. Whitespace-only - input collapses to ``""`` so callers can treat that as "nothing to emit". - """ + """Persisted content/reasoning (str, or list of ``{"text"}`` / ``{"type": "text", "content"}`` + parts) -> one stripped string; whitespace-only collapses to ``""`` ("nothing to emit").""" if isinstance(value, str): return value.strip() if isinstance(value, list): @@ -578,9 +520,8 @@ def _flatten_history_text(value: Any) -> str: def _history_reasoning_text(message: dict[str, Any]) -> str: - """First non-empty of ``reasoning_content`` (DeepSeek/Moonshot, chat-completions - normalizer) and ``reasoning`` (codex projector and other transports). Both are - live keys for different transports, not old-vs-new.""" + """First non-empty of ``reasoning_content`` and ``reasoning`` — both live keys, for + different transports (not old-vs-new).""" for key in ("reasoning_content", "reasoning"): text = _flatten_history_text(message.get(key)) if text: @@ -591,19 +532,14 @@ def _history_reasoning_text(message: dict[str, Any]) -> str: def _history_summary_meta(message: dict[str, Any], text: str) -> dict[str, Any] | None: """``_meta`` for a replayed compaction summary, else None. - Summaries are persisted as ordinary history messages — standalone handoffs - under either role (whichever keeps alternation valid) or merged into the - first preserved tail message. Two distinct ``_meta.hermes`` keys so clients - cannot accidentally hide real content: ``compactionSummary`` (whole chunk is - the summary; safe to collapse) vs ``containsCompactionSummary`` (real turn - content followed by the summary; collapsing would hide preserved content). - Honors the in-process ``_compressed_summary`` flag and falls back to content - classification so DB-reloaded sessions still tag correctly. + Summaries persist as ordinary messages, standalone (either role) or merged into the first + preserved tail message. Two keys so clients can't hide real content: ``compactionSummary`` + (whole chunk; safe to collapse) vs ``containsCompactionSummary`` (real content + summary). + Uses the in-process flag, falling back to content classification for DB-reloaded sessions. """ kind = ContextCompressor.classify_summary_content(text) if kind is None and message.get(COMPRESSED_SUMMARY_METADATA_KEY): - # Flagged but unclassified (prefix drift): the flag is only ever set on - # summary-bearing messages, so treat as standalone. + # Flagged but unclassified (prefix drift): the flag only marks summaries -> standalone. kind = "standalone" if kind == "standalone": return {"hermes": {"compactionSummary": True}} @@ -660,12 +596,11 @@ def _attach_interrupted_prompt(interrupted_prompt: str, guidance: str) -> str: @dataclass class _ModelCatalog: - """Deduplicated ACP model rows collected from the inventory + named endpoints. + """Deduplicated ACP model rows from the inventory + named endpoints. - Dedupes on both the encoded choice id and a semantic ``provider:model`` id - (``ollama``/``custom:ollama`` are one provider). Also resolves the current - provider identity: a bare/``custom`` current provider whose base_url matches - an ollama inventory row is really ``custom:ollama``. + Dedupes on the encoded choice id AND a semantic ``provider:model`` id (``ollama`` == + ``custom:ollama``). A bare/``custom`` current provider whose base_url matches an ollama + inventory row is resolved to ``custom:ollama``. """ normalize_provider: Callable[[str], str] @@ -843,12 +778,8 @@ class HermesACPAgent(acp.Agent): loop.call_soon(asyncio.create_task, make_coro()) def _session_modes(self, state: SessionState) -> SessionModeState: - """ACP session modes carrying the edit-approval policy. - - Zed renders ``config_options`` in the prominent selector slot where the - model picker lives; Claude/Codex expose policy controls as ACP modes, - which coexist with the picker, so Hermes maps edit approval onto modes. - """ + """Edit-approval policy as ACP modes. Zed renders ``config_options`` in the model + picker's slot; modes (as Claude/Codex use) coexist with the picker.""" current = str(getattr(state, "mode", "") or self._MODE_DEFAULT) if current not in self._MODES: current = self._MODE_DEFAULT @@ -872,12 +803,8 @@ class HermesACPAgent(acp.Agent): return f"{raw_provider}:{raw_model}" if raw_provider else raw_model def _build_model_state(self, state: SessionState) -> SessionModelState | None: - """Authenticated providers and their models for ACP clients. - - Uses the shared Hermes inventory (also behind ``hermes model``, the TUI - and the dashboard) so the selector doesn't collapse to the current - provider's curated list. - """ + """Authenticated providers + models, from the shared Hermes inventory (same substrate + as ``hermes model``/TUI/dashboard) so the selector isn't just the current curated list.""" model = str(state.model or getattr(state.agent, "model", "") or "").strip() provider = getattr(state.agent, "provider", None) or detect_provider() or "openrouter" @@ -973,11 +900,8 @@ class HermesACPAgent(acp.Agent): def _switch_model( self, state: SessionState, raw_model: str, *, keep_endpoint: bool = False ) -> tuple[str | None, str, str]: - """Rebuild the session agent on a new model; returns (old provider, new provider, model). - - ``keep_endpoint`` carries the current base_url/api_mode over when the - provider is unchanged (ACP ``set_session_model``). - """ + """Rebuild the session agent on a new model -> (old provider, new provider, model). + ``keep_endpoint`` carries base_url/api_mode over when the provider is unchanged.""" current_provider = getattr(state.agent, "provider", None) target_provider, new_model = self._resolve_model_selection(raw_model, current_provider or "openrouter") state.model = new_model @@ -996,9 +920,8 @@ class HermesACPAgent(acp.Agent): @staticmethod def _build_usage_update(state: SessionState) -> UsageUpdate | None: - """ACP ``usage_update`` driving Zed's context indicator: ``size`` is the - model context window, ``used`` the estimated request pressure (system - prompt + history + tool schemas — the same buckets sent to providers).""" + """``usage_update`` for Zed's context indicator: ``size`` = context window, ``used`` = + estimated request pressure (system prompt + history + tool schemas).""" agent = state.agent compressor = getattr(agent, "context_compressor", None) size = int(getattr(compressor, "context_length", 0) or 0) @@ -1038,9 +961,8 @@ class HermesACPAgent(acp.Agent): self, session_id: str, *, current_hermes_session_id: Optional[str] = None, previous_hermes_session_id: Optional[str] = None, ) -> None: - """Send ACP session metadata after Hermes changes it. Pass - ``previous_hermes_session_id`` when the internal head rotated - (compression split) so the provenance meta flags the reason.""" + """Session metadata update; pass ``previous_hermes_session_id`` when the internal head + rotated (compression split) so provenance flags the reason.""" if not self._conn: return try: @@ -1052,8 +974,7 @@ class HermesACPAgent(acp.Agent): return title = row.get("title") - # `sessions` has no `updated_at` column (only started_at/ended_at); "now" - # is right because this fires precisely when the title was refreshed. + # `sessions` has no `updated_at`; "now" is right since this fires when the title changed. update = SessionInfoUpdate( session_update="session_info_update", title=title if isinstance(title, str) and title.strip() else None, @@ -1114,19 +1035,14 @@ class HermesACPAgent(acp.Agent): ) def _schedule_mcp_late_refresh(self, state: SessionState) -> None: - """Refresh the agent's tool snapshot when background MCP discovery lands late. + """Refresh the tool snapshot when background MCP discovery lands after agent build + (``_make_agent`` only joins ~1.5s). Waits up to 30s off the critical path, then rebuilds + via ``refresh_agent_mcp_tools`` (same as ``/reload-mcp``). - entry.py runs MCP discovery in a daemon thread; ``_make_agent`` joins it - only briefly (~1.5s), so a slower server lands after the agent is built - and its tools would be absent for the whole session. This waits for - discovery (bounded 30s) off the critical path, then rebuilds via the same - ``refresh_agent_mcp_tools`` that ``/reload-mcp`` uses. - - Cache safety: rebuild only while the session is pre-first-turn (nothing - cached yet). After the first message the snapshot stays frozen; later - servers are picked up cache-safely by the between-turns prologue refresh - (``agent/turn_context.py``). No-op when discovery already finished, the - join times out, the registry was unchanged, or the session was closed. + Cache safety: only pre-first-turn (nothing cached yet); afterwards the snapshot stays + frozen and late servers land via the between-turns prologue refresh + (``agent/turn_context.py``). No-op if discovery finished, join timed out, registry + unchanged, or session closed. """ try: from hermes_cli.mcp_startup import mcp_discovery_in_flight @@ -1147,17 +1063,14 @@ class HermesACPAgent(acp.Agent): if not join_mcp_discovery(timeout=30.0): return - # In-memory-only lookup on purpose: ``get_session()`` would - # restore from DB and build a whole new AIAgent just to say no-op. + # In-memory only: ``get_session()`` would restore from DB and build a new AIAgent. with self.session_manager._lock: current = self.session_manager._sessions.get(session_id) if current is None or current.agent is not agent: return - # Serialized with turn start: ``prompt()`` flips ``is_running`` - # under ``runtime_lock`` before dispatching, so holding it here - # (and bailing when a turn is running) closes the window where - # the refresh would swap ``tools=`` mid-turn and break the cache. + # ``prompt()`` flips ``is_running`` under ``runtime_lock`` before dispatching, so + # holding it here closes the window where a refresh would swap ``tools=`` mid-turn. with current.runtime_lock: if current.is_running: return @@ -1212,16 +1125,14 @@ class HermesACPAgent(acp.Agent): ) async def authenticate(self, method_id: str, **kwargs: Any) -> AuthenticateResponse | None: - # Only acknowledge the method_id advertised in initialize(); accepting - # any id would be poor hygiene if ACP ever grows multi-method auth. + # Only acknowledge the method_id advertised in initialize(). if not isinstance(method_id, str): return None normalized_method = method_id.strip().lower() provider = detect_provider() if normalized_method == TERMINAL_SETUP_AUTH_METHOD_ID: - # Terminal auth runs Hermes setup out-of-band; succeed only once it - # has produced usable runtime credentials. + # Terminal auth runs setup out-of-band; succeed only once credentials exist. return AuthenticateResponse() if provider else None if not provider or normalized_method != provider: @@ -1231,10 +1142,8 @@ class HermesACPAgent(acp.Agent): # ---- Session management ------------------------------------------------- async def _replay_session_history(self, state: SessionState) -> None: - """Replay persisted history as user/assistant chunks, thought chunks and - reconstructed tool-call start/completion notifications, so the editor - shows the transcript instead of a clean thread. Awaited inline from - ``load_session``/``resume_session`` (see there for why).""" + """Replay history as user/assistant/thought chunks plus reconstructed tool-call + start/complete events so the editor shows the transcript, not a clean thread.""" if not self._conn or not state.history: return @@ -1299,12 +1208,9 @@ class HermesACPAgent(acp.Agent): return async def _replay_history_guarded(self, state: SessionState, verb: str) -> None: - """Per ACP spec, ``session/load`` and ``session/resume`` must stream the - prior conversation via ``session/update`` BEFORE responding, so clients - get the transcript within the request's lifetime (Codex, Claude Code, - OpenCode, Zed all rely on this; deferring via ``call_soon`` broke them). - Replay is best-effort: a corrupt message shape must not turn a - successful load into a JSON-RPC error.""" + """Per ACP spec, load/resume must stream history via ``session/update`` BEFORE + responding (Codex/Claude Code/OpenCode/Zed rely on this; deferring via ``call_soon`` + broke them). Best-effort: a corrupt message must not turn the load into an error.""" try: await self._replay_session_history(state) except Exception: @@ -1316,8 +1222,8 @@ class HermesACPAgent(acp.Agent): ) def _session_response_fields(self, state: SessionState) -> dict[str, Any]: - """Common ``models``/``modes``/``field_meta`` for session responses; also - schedules the command advertisement and usage refresh.""" + """``models``/``modes``/``field_meta`` for session responses; schedules command + advertisement + usage refresh.""" self._schedule_available_commands_update(state.session_id) self._schedule_usage_update(state) return { @@ -1367,8 +1273,7 @@ class HermesACPAgent(acp.Agent): with state.runtime_lock: if state.is_running and state.current_prompt_text: state.interrupted_prompt_text = state.current_prompt_text - # Publish cancellation and hard-stop the agent before another - # prompt can acquire this lock and mistake the turn for + # Cancel + hard-stop under the lock so no other prompt mistakes this turn for # redirectable work. state.cancel_event.set() try: @@ -1397,10 +1302,8 @@ class HermesACPAgent(acp.Agent): async def list_sessions( self, cursor: str | None = None, cwd: str | None = None, **kwargs: Any ) -> ListSessionsResponse: - """``cwd`` filtering is done by ``SessionManager.list_sessions``. ``cursor`` - is a ``session_id`` previously returned as ``next_cursor``; results - resume after it (unknown cursor -> empty page, never the full list). - Pages are capped at ``_LIST_SESSIONS_PAGE_SIZE``.""" + """``cursor`` is a ``session_id`` returned as ``next_cursor``; results resume after it + (unknown cursor -> empty page, never the full list). Pages cap at the fixed size.""" infos = self.session_manager.list_sessions(cwd=cwd) if cursor: @@ -1431,17 +1334,11 @@ class HermesACPAgent(acp.Agent): def _rewrite_prompt_for_interrupt( self, state: SessionState, user_text: str, user_content: Any, text_only: bool ) -> tuple[str, Any]: - """Attach a client-cancelled prompt to the follow-up text, and run idle - ``/steer`` as a normal prompt. - - ``/steer`` on an idle session has no in-flight tool call to inject into - (matching the gateway): if a prior prompt was just cancelled, replay it - with the steer text as explicit correction so the in-flight work isn't - lost; otherwise run the steer payload as a plain prompt instead of - silently queueing it ("No active turn — queued") as if the user typed - ``/queue``. Plain text after a cancel likewise keeps the cancelled request - attached ("stop and send" clients) so deictic follow-ups have a target. - """ + """Idle ``/steer`` has nothing to inject into (gateway parity): if a prompt was just + cancelled, replay it with the steer text as explicit correction; otherwise run the steer + payload as a plain prompt rather than silently queueing it as if ``/queue`` was typed. + Plain text after a cancel likewise keeps the cancelled request attached ("stop and + send" clients) so deictic follow-ups have a target.""" if not (text_only and isinstance(user_content, str)): return user_text, user_content @@ -1477,9 +1374,8 @@ class HermesACPAgent(acp.Agent): def _claim_turn_or_queue( self, state: SessionState, session_id: str, user_text: str, user_content: Any, text_only: bool ) -> str | None: - """Mark the session running, or — if a turn is active — redirect it - (text-only, runtime supports it) or queue for the next turn. Returns the - message to send the client when the prompt was absorbed, else None.""" + """Mark the session running; if a turn is active, redirect it (text-only, supported + runtime) or queue it. Returns the client message when absorbed, else None.""" redirected = False queued_depth: int | None = None with state.runtime_lock: @@ -1511,22 +1407,19 @@ class HermesACPAgent(acp.Agent): self, *, state: SessionState, session_id: str, user_text: str, user_content: Any, conn: Any, loop: asyncio.AbstractEventLoop, approval_cb: Any, edit_approval_requester: Any, ) -> dict: - """Executor-thread body of one turn. Runs inside ``contextvars.copy_context()`` - so every ContextVar write below is isolated from concurrent sessions. + """Executor-thread body of one turn, run inside ``contextvars.copy_context()`` so + ContextVar writes are isolated from concurrent sessions. - Approval routing is thread-local, so it MUST be bound here (executor - thread), not on the event-loop thread. Interactive routing uses the - ``tools.approval`` contextvar rather than ``os.environ["HERMES_INTERACTIVE"]`` - so concurrent workers can't race a process-global flag and drop another - session onto the non-interactive auto-approve path (GHSA-96vc-wcxf-jjff). + Approval routing is thread-local, so it MUST be bound here, not on the loop thread. + Interactive routing is a ``tools.approval`` contextvar, not ``HERMES_INTERACTIVE`` in + os.environ, so concurrent workers can't race a global flag onto the non-interactive + auto-approve path (GHSA-96vc-wcxf-jjff). """ agent = state.agent - # Bind HERMES_SESSION_KEY so per-session caches (e.g. the interactive - # sudo password cache) scope to this ACP session, not the reused thread. - # ``cwd`` pins the logical working directory the system prompt reports - # (resolve_agent_cwd); without it the prompt advertises the global - # Hermes workspace while tools are rooted at the client's project, and - # edits land outside the editor's workspace. ``cron_session=""`` masks + # HERMES_SESSION_KEY scopes per-session caches (interactive sudo password) to this + # session, not the reused thread. ``cwd`` pins what the system prompt reports as the + # working directory — otherwise it advertises the Hermes workspace while tools are + # rooted at the client's project and edits land outside it. ``cron_session=""`` masks # any leaked process-global HERMES_CRON_SESSION. try: from gateway.session_context import clear_session_vars, set_session_vars @@ -1555,13 +1448,11 @@ class HermesACPAgent(acp.Agent): except Exception: logger.debug("Could not set ACP edit approval requester", exc_info=True) interactive_token = set_hermes_interactive_context(True) - # Tools tag side-effects with the originating ACP session (e.g. - # ``kanban_create``); save/restore so a reused thread never leaks it. + # Tools tag side-effects with the ACP session (``kanban_create``); save/restore it. previous_session_id = os.environ.get("HERMES_SESSION_ID") os.environ["HERMES_SESSION_ID"] = session_id - # Auto-titling fires in the turn prologue; deliver the new title now - # as a session-info update instead of waiting for the next one. + # Auto-titling fires in the turn prologue; push the title now as a session-info update. def _notify_title_update(_title: str, _source: str) -> None: if conn: loop.call_soon_threadsafe(asyncio.create_task, self._send_session_info_update(session_id)) @@ -1619,8 +1510,7 @@ class HermesACPAgent(acp.Agent): state, user_text, user_content, text_only_prompt ) - # Slash commands are text-only and handled locally without the LLM; a - # prompt with images/resources goes to the agent even if it starts with "/". + # Slash commands are text-only; a prompt with media goes to the agent even if it starts with "/". if text_only_prompt and isinstance(user_content, str) and user_text.startswith("/"): response_text = self._handle_slash_command(user_text, state) if response_text is not None: @@ -1652,12 +1542,10 @@ class HermesACPAgent(acp.Agent): ) try: - # The ACP `session_id` is the stable client handle; agent.session_id - # is the live internal head that compression may rotate. Snapshot it - # to detect a rotation after the turn. + # ACP `session_id` is the stable handle; agent.session_id is the internal head that + # compression may rotate — snapshot it to detect rotation after the turn. pre_turn_hermes_id = getattr(state.agent, "session_id", None) - # Fresh context copy so concurrent sessions on the shared executor - # don't stomp on each other's ContextVar writes. + # Fresh context copy: concurrent sessions on the shared executor must not share ContextVars. ctx = contextvars.copy_context() result = await loop.run_in_executor(_executor, ctx.run, _run_agent) except Exception: @@ -1703,9 +1591,7 @@ class HermesACPAgent(acp.Agent): agent = state.agent agent.tool_progress_callback = cbs.tool_progress_cb - # ACP thought panes get provider reasoning deltas only — never Hermes' - # local status updates, and no fake "thinking" accordion when the - # provider emits no reasoning. + # Thought panes get provider reasoning only — no local status updates, no fake accordion. agent.thinking_callback = None agent.reasoning_callback = cbs.reasoning_cb agent.step_callback = cbs.step_cb @@ -1721,8 +1607,7 @@ class HermesACPAgent(acp.Agent): state.history = result["messages"] self.session_manager.save_session(session_id) - # Internal head rotated (compression split): emit provenance so clients - # can render the boundary; the ACP session_id is unchanged. + # Head rotated (compression split): emit provenance so clients can render the boundary. post_turn_hermes_id = getattr(state.agent, "session_id", None) if ( conn @@ -1741,13 +1626,11 @@ class HermesACPAgent(acp.Agent): final_response = result.get("final_response", "") cancelled = bool(state.cancel_event and state.cancel_event.is_set()) interrupted = bool(result.get("interrupted")) or cancelled - # The local "waiting for model response" interrupt status is metadata, - # not assistant prose — clients learn cancellation from stop_reason. + # The local "waiting for model" interrupt status is metadata, not prose; stop_reason carries it. from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX suppress_interrupt_response = interrupted and final_response.startswith(INTERRUPT_WAITING_FOR_MODEL_PREFIX) - # Deliver the final response when streaming didn't already, or when a - # plugin hook transformed it after streaming (transform_llm_output). + # Send the final text unless already streamed — or if a plugin hook transformed it after. if ( final_response and conn @@ -1756,8 +1639,7 @@ class HermesACPAgent(acp.Agent): ): await conn.session_update(session_id, acp.update_agent_message_text(final_response)) - # Go idle before draining queued work so recursive prompt() calls can - # acquire the session; queued turns run as normal follow-up prompts. + # Go idle before draining so recursive prompt() calls can acquire the session. with state.runtime_lock: state.is_running = False state.current_prompt_text = "" @@ -1805,8 +1687,7 @@ class HermesACPAgent(acp.Agent): self._schedule_soon(lambda: self._send_available_commands_update(session_id)) def _handle_slash_command(self, text: str, state: SessionState) -> str | None: - """Dispatch a slash command; ``None`` for unknown commands so they fall - through to the LLM (the user may have typed ``/something`` as prose).""" + """Dispatch a slash command; ``None`` for unknown ones so they fall through to the LLM.""" parts = text.split(maxsplit=1) cmd = parts[0].lstrip("/").lower() args = parts[1].strip() if len(parts) > 1 else "" @@ -1815,11 +1696,9 @@ class HermesACPAgent(acp.Agent): return None handler = getattr(self, f"_cmd_{cmd}") - # Handlers run on the event-loop thread, OUTSIDE the per-turn context - # that pins the session cwd. ``/compress`` and ``/model`` REBUILD the - # system prompt (resolve_agent_cwd), so an unpinned handler would bake - # the Hermes install tree into the persisted cached prompt and poison - # every later turn. Pin inside a fresh context: no leak, no teardown. + # Handlers run on the loop thread, outside the per-turn cwd-pinning context. ``/compress`` + # and ``/model`` REBUILD the system prompt, so unpinned they'd bake the Hermes install tree + # into the persisted cached prompt. Pin inside a fresh context: no leak, no teardown. def _dispatch() -> str | None: try: from agent.runtime_cwd import set_session_cwd @@ -1968,22 +1847,19 @@ class HermesACPAgent(acp.Agent): return "Nothing to compress — conversation is empty." try: agent = state.agent - # No compression_enabled gate: that flag only disables *automatic* - # compaction; manual /compress must keep working (CLI/gateway parity). + # No compression_enabled gate: it only disables *automatic* compaction (CLI/gateway parity). if not hasattr(agent, "_compress_context"): return "Context compression not available for this agent." original_count = len(state.history) - # System prompt + tool schemas included so the figure reflects real - # request pressure, not a transcript-only underestimate. + # Include system prompt + tool schemas so the figure reflects real request pressure. _sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" _tools = getattr(agent, "tools", None) or None approx_tokens = _estimate_tokens(state.history, agent, _sys_prompt, _tools) original_session_db = getattr(agent, "_session_db", None) try: - # ACP sessions keep a stable session id: suppress the SQLite - # session-splitting side effect inside _compress_context. + # Stable ACP session id: suppress _compress_context's SQLite session split. agent._session_db = None compressed, _ = agent._compress_context( state.history, getattr(agent, "_cached_system_prompt", "") or "", diff --git a/tests/tui_gateway/test_compression_config_hot_reload.py b/tests/tui_gateway/test_compression_config_hot_reload.py index 37583161a3..ad9dbd15f5 100644 --- a/tests/tui_gateway/test_compression_config_hot_reload.py +++ b/tests/tui_gateway/test_compression_config_hot_reload.py @@ -133,7 +133,8 @@ def test_clearing_threshold_tokens_restores_ratio_trigger(monkeypatch): def test_prompt_submit_calls_compression_sync_after_model_sync(): - source = open(server.__file__, encoding="utf-8").read() + # Read the module that actually defines the turn (it moved out of server.py). + source = open(server._run_prompt_submit.__code__.co_filename, encoding="utf-8").read() model_idx = source.find("_sync_agent_model_with_config(sid, session)") compression_idx = source.find("_sync_agent_compression_with_config(sid, session)") assert model_idx != -1 diff --git a/tui_gateway/agent_callbacks.py b/tui_gateway/agent_callbacks.py index 7c2c993ac7..7a9d586dfb 100644 --- a/tui_gateway/agent_callbacks.py +++ b/tui_gateway/agent_callbacks.py @@ -13,25 +13,17 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# ── Child-session live mirror ──────────────────────────────────────── -# A delegated child is not a live gateway session — it runs synchronously -# inside the parent's turn, and its activity reaches the gateway only as -# relayed ``subagent.*`` events on the PARENT sid. When a UI opens the child's -# own session (session.resume on ``child_session_id``, e.g. the desktop's -# open-in-new-window), that window would otherwise sit silent until the run -# persists. Translate the relayed events into the native stream events the -# window already renders — emitted on the CHILD sid, routed to its transport -# by write_json — so the window shows a real midstream turn. +# Child-session live mirror: a delegated child's activity reaches the gateway only +# as relayed ``subagent.*`` events on the PARENT sid, so a window opened on the +# child's own session would sit silent until the run persists. Translate them into +# the native stream events emitted on the CHILD sid (write_json routes by sid). _child_mirrors: dict[str, dict] = {} _child_mirrors_lock = threading.Lock() -# Stored child session ids with a delegation run currently in flight (refreshed -# on every relayed subagent.* event, popped on subagent.complete). Lets a lazy -# watch resume report running=true so the window shows a busy indicator even -# while the child is silent inside a long tool call (no events for 25s+). +# Child session ids with a run in flight (refreshed per relayed event, popped on +# complete) so a lazy watch resume reports running=true during a silent long tool. _active_child_runs: dict[str, float] = {} -# Staleness bound for the registry: entries refresh on every relayed event, so -# anything this quiet means the completion event was lost (callback raised, -# parent crashed) — don't let a leaked entry pin "running" forever. +# Anything quiet this long lost its completion event (callback raised, parent +# crashed) — don't pin "running". _CHILD_RUN_STALE_S = 3600.0 @@ -44,41 +36,36 @@ def _mirror_subagent_to_child(event_type: str, payload: dict) -> None: child_key = str(payload.get("child_session_id") or "") if not child_key: return - # Liveness registry first — it must be accurate even when no window is - # open, so a window opened mid-run can immediately know the child is busy. + # Liveness registry first: accurate with no window open, so one opened mid-run + # immediately knows the child is busy. if event_type == "subagent.complete": _active_child_runs.pop(child_key, None) else: _active_child_runs[child_key] = time.time() - # Mirror only into a live watch session (keyed by session_key; its live sid - # differs from the stored id) that has NOT been upgraded to a full agent. - # No window / closed → nothing to mirror; an upgraded session owns a real - # native stream and mirroring on top would interleave two turns on one sid. - # Either way drop state so a reopened window starts a fresh synthetic turn. + # Mirror only into a live watch session NOT upgraded to a full agent: an + # upgraded one owns a real native stream and mirroring would interleave two + # turns on one sid. Either way drop state so a reopened window starts fresh. live = _find_live_session_by_key(child_key) if live is None or live[1].get("agent") is not None: with _child_mirrors_lock: _child_mirrors.pop(child_key, None) return csid = live[0] + text = str(payload.get("text") or "") with _child_mirrors_lock: st = _child_mirrors.setdefault(child_key, {"seq": 0, "open_tool": None, "started": False}) if not st["started"]: st["started"] = True _emit("message.start", csid) if event_type == "subagent.thinking": - if text := str(payload.get("text") or ""): + if text: _emit("reasoning.delta", csid, {"text": text}) elif event_type == "subagent.text": - # The child's streamed reply text — the actual "agent talking". - # Relayed token-by-token from the child's run_conversation - # stream_callback, so the watch window streams the reply live. - if text := str(payload.get("text") or ""): + if text: _emit("message.delta", csid, {"text": text}) elif event_type == "subagent.start": - # One-time header line (the child's goal) so a freshly opened window - # shows immediate context before the first reply token streams. - if text := str(payload.get("text") or ""): + # One-time header (the child's goal) so a fresh window has context first. + if text: _emit("message.delta", csid, {"text": f"{text}\n"}) elif event_type == "subagent.tool": if st["open_tool"]: @@ -102,124 +89,59 @@ def _mirror_subagent_to_child(event_type: str, payload: dict) -> None: def _agent_cbs(sid: str) -> dict: + def _read_block(event: str, timeout: int): + # read_terminal / read_preview (desktop GUI): blocking bridge like clarify; the + # preview read gets longer since a URL tab extracts text from a live page. + return lambda start=None, count=None: _block( + event, sid, {k: v for k, v in (("start", start), ("count", count)) if v is not None}, timeout=timeout + ) + callbacks = { - "tool_start_callback": lambda tc_id, name, args: _on_tool_start( - sid, tc_id, name, args - ), - "tool_complete_callback": lambda tc_id, name, args, result: _on_tool_complete( - sid, tc_id, name, args, result - ), + "tool_start_callback": lambda tc_id, name, args: _on_tool_start(sid, tc_id, name, args), + "tool_complete_callback": lambda tc_id, name, args, result: _on_tool_complete(sid, tc_id, name, args, result), "tool_progress_callback": lambda event_type, name=None, preview=None, args=None, **kwargs: _on_tool_progress( sid, event_type, name, preview, args, **kwargs ), - "tool_gen_callback": lambda name: _tool_progress_enabled(sid) - and _emit("tool.generating", sid, {"name": name}), + "tool_gen_callback": lambda name: _tool_progress_enabled(sid) and _emit("tool.generating", sid, {"name": name}), "thinking_callback": lambda text: _emit("thinking.delta", sid, {"text": text}), - # Affection reaction (ily / <3 / good bot) → hearts. Core-detected, so - # the TUI heart and desktop floating hearts share one signal. + # Affection reaction (ily / <3 / good bot) → hearts; core-detected so TUI and desktop share it. "reaction_callback": lambda kind: _emit("reaction", sid, {"kind": kind}), "reasoning_callback": lambda text: _emit( - "reasoning.delta", - sid, - {"text": text, **({"verbose": True} if _session_verbose(sid) else {})}, + "reasoning.delta", sid, {"text": text, **({"verbose": True} if _session_verbose(sid) else {})} ), - "status_callback": lambda kind, text=None: _status_update( - sid, str(kind), None if text is None else str(text) - ), - # Credits/notice spine (L1): an AgentNotice fired by the agent becomes a - # notification.show WS event; a recovery clear becomes notification.clear. - # Snake_case payload to match the existing gateway-event convention. + "status_callback": lambda kind, text=None: _status_update(sid, str(kind), None if text is None else str(text)), + # Credits/notice spine: AgentNotice → notification.show; recovery clear → notification.clear. "notice_callback": lambda n: _emit( "notification.show", sid, - { - "text": n.text, - "level": n.level, - "kind": n.kind, - "ttl_ms": n.ttl_ms, - "key": n.key, - "id": n.id, - }, - ), - "notice_clear_callback": lambda key: _emit( - "notification.clear", sid, {"key": key} + {"text": n.text, "level": n.level, "kind": n.kind, "ttl_ms": n.ttl_ms, "key": n.key, "id": n.id}, ), + "notice_clear_callback": lambda key: _emit("notification.clear", sid, {"key": key}), "clarify_callback": lambda q, c, multi_select=False, questions=None: ( _clarify_block(sid, q, c, multi_select=multi_select, questions=questions) ), - # read_terminal tool (desktop GUI): same blocking bridge as clarify — the - # renderer answers terminal.read.respond with the serialized buffer. - "read_terminal_callback": lambda start=None, count=None: _block( - "terminal.read.request", - sid, - {k: v for k, v in (("start", start), ("count", count)) if v is not None}, - timeout=30, - ), - # read_preview tool (desktop GUI): the renderer serializes the active - # preview tab (a Browser webview's readable text, a file's identity) - # and answers preview.read.respond. Longer timeout than the terminal - # read — a URL tab extracts text from a live page. - "read_preview_callback": lambda start=None, count=None: _block( - "preview.read.request", - sid, - {k: v for k, v in (("start", start), ("count", count)) if v is not None}, - timeout=45, - ), - # drive_preview tool (desktop GUI): the renderer injects the interaction - # engine into the preview pane's webview (or drives the pane's history) - # and answers preview.act.respond with the outcome plus a refreshed - # element inventory. Same budget as the preview read, which it ends - # with — a click on a slow page pays for the settle and the re-scan. - # annotate_preview rides this same callback: it resolves a target - # through the same engine and differs only in the verb it sends, so it - # needs a tool of its own but not a channel of its own. - "drive_preview_callback": lambda payload: _block( - "preview.act.request", - sid, - dict(payload), - timeout=45, - ), - # read_window_below tool (desktop GUI): the renderer asks its main - # process (which owns native window enumeration) which OS window sits - # directly underneath the Hermes window, and answers - # window.read.respond with the serialized metadata. - "read_window_below_callback": lambda: _block( - "window.read.request", - sid, - {}, - timeout=30, - ), - # setup_mcp tool (desktop GUI): the renderer shows an inline consent - # card and walks the user through install/enable/OAuth via the REST - # endpoints, then answers mcp.setup.respond with the JSON outcome. - # Long timeout on purpose — the flow can include typing an API key or - # a browser OAuth round-trip. Same lifecycle as clarify: on timeout - # the tool returns "unanswered" and a late answer is tolerated. + "read_terminal_callback": _read_block("terminal.read.request", 30), + "read_preview_callback": _read_block("preview.read.request", 45), + # drive_preview / annotate_preview (desktop GUI): renderer drives the preview webview and + # answers with outcome + refreshed element inventory; same budget as the preview read it ends with. + "drive_preview_callback": lambda payload: _block("preview.act.request", sid, dict(payload), timeout=45), + # read_window_below (desktop GUI): main process enumerates native windows. + "read_window_below_callback": lambda: _block("window.read.request", sid, {}, timeout=30), + # setup_mcp (desktop GUI): consent card + install/enable/OAuth. Long timeout on purpose (typing + # an API key, browser OAuth); like clarify, timeout returns "unanswered" and a late answer is tolerated. "setup_mcp_callback": lambda server, action, reason: _block( - "mcp.setup.request", - sid, - {"server": server, "action": action, "reason": reason}, - timeout=600, + "mcp.setup.request", sid, {"server": server, "action": action, "reason": reason}, timeout=600 ), - # tour tool (desktop GUI): the renderer drives driver.js — highlighting - # elements in the app's own DOM or injecting the engine into the - # preview pane's webview — and answers tour.respond with the outcome - # (did the selector match, which step is active). + # tour (desktop GUI): renderer drives driver.js and answers tour.respond. "tour_callback": lambda payload: _tour_request(sid, payload), } - # Interim assistant commentary (text alongside tool calls, or the attempted - # final answer before a verify-on-stop nudge). Gated on - # display.interim_assistant_messages (default true). Also set per-turn in - # _run_prompt_submit as defense-in-depth — the per-turn set overwrites - # this, and the finally block clears it so a stale closure can't fire. + # Interim assistant commentary (text alongside tool calls). Gated on + # display.interim_assistant_messages (default true); _run_prompt_submit overwrites + # it per turn and clears it in its finally so a stale closure can't fire. if _load_interim_assistant_messages(): - callbacks["interim_assistant_callback"] = ( - lambda text, *, already_streamed=False: _emit( - "message.interim", - sid, - {"text": str(text), "already_streamed": bool(already_streamed)}, - ) + callbacks["interim_assistant_callback"] = lambda text, *, already_streamed=False: _emit( + "message.interim", sid, {"text": str(text), "already_streamed": bool(already_streamed)} ) return callbacks @@ -227,17 +149,13 @@ def _agent_cbs(sid: str) -> dict: def _apply_project_workspace(task_id: str, path: str, _name: str = "") -> None: """Intentional workspace move from the project_* tools: re-anchor the live - session's cwd to the chosen project's folder and push session.info so the - desktop follows (refresh tree + scope into the project). This is the ONLY + session's cwd and push session.info so the desktop follows. This is the ONLY auto-cwd path — driven by an explicit tool call, never a terminal `cd`.""" if not path: return - - # The tool's task_id is the durable session_key, but _sessions is keyed by a - # short sid uuid (and the desktop routes events by that sid). Resolve it. + # task_id is the durable session_key; _sessions (and desktop event routing) key by sid. key = str(task_id or "") - sid = "" - session = None + sid, session = "", None with _sessions_lock: if key in _sessions: sid, session = key, _sessions[key] @@ -246,34 +164,25 @@ def _apply_project_workspace(task_id: str, path: str, _name: str = "") -> None: if cand.get("session_key") == key or getattr(cand.get("agent"), "session_id", None) == key: sid, session = cand_sid, cand break - if session is None: return - resolved = os.path.abspath(os.path.expanduser(str(path))) if not os.path.isdir(resolved): return - session["cwd"] = resolved session["explicit_cwd"] = True - # An explicit project switch supersedes any earlier settle-adopted cwd. - session["cwd_from_settle"] = False + session["cwd_from_settle"] = False # explicit switch supersedes a settle-adopted cwd _register_session_cwd(session) - _persist_session_cwd_and_schedule_git_meta(session, resolved) - try: agent = session.get("agent") - info = ( - _session_info(agent, session) - if agent is not None - else { - "cwd": resolved, - "branch": _git_branch_for_cwd(resolved), - "project": _project_info_for_cwd(resolved), - "lazy": True, + if agent is not None: + info = _session_info(agent, session) + else: + info = { + "cwd": resolved, "branch": _git_branch_for_cwd(resolved), + "project": _project_info_for_cwd(resolved), "lazy": True, } - ) _emit("session.info", sid, info) except Exception: logger.debug("failed to emit session.info after project workspace move", exc_info=True) @@ -293,20 +202,10 @@ def _wire_callbacks(sid: str): pl["metadata"] = metadata val = _block("secret.request", sid, pl) if not val: - return { - "success": True, - "stored_as": env_var, - "validated": False, - "skipped": True, - "message": "skipped", - } + return {"success": True, "stored_as": env_var, "validated": False, "skipped": True, "message": "skipped"} from hermes_cli.config import save_env_value_secure - return { - **save_env_value_secure(env_var, val), - "skipped": False, - "message": "ok", - } + return {**save_env_value_secure(env_var, val), "skipped": False, "message": "ok"} set_secret_capture_callback(secret_cb) @@ -322,18 +221,14 @@ def _available_personalities(cfg: dict | None = None) -> dict: """Built-ins + user overrides, via hermes_cli.personality (single owner).""" from hermes_cli.personality import available_personalities - if cfg is None: - cfg = _load_cfg() - return available_personalities(cfg) + return available_personalities(_load_cfg() if cfg is None else cfg) def _validate_personality(value: str, cfg: dict | None = None) -> tuple[str, str]: - """Resolve a requested personality against _available_personalities. + """Resolve a requested personality to (name, prompt) or raise ValueError. - Same contract as hermes_cli.personality.resolve_personality — (name, - prompt) or ValueError — but resolves through the module-level - _available_personalities so tests (and future gateway-side overrides) - keep a single patch point. + Same contract as hermes_cli.personality.resolve_personality, but goes through + the module-level _available_personalities so tests keep a single patch point. """ from hermes_cli.personality import normalize_personality_name @@ -343,17 +238,12 @@ def _validate_personality(value: str, cfg: dict | None = None) -> tuple[str, str personalities = _available_personalities(cfg) if name not in personalities: names = ", ".join(f"`{n}`" for n in sorted(personalities)) - raise ValueError( - f"Unknown personality: `{str(value).strip()}`.\n\nAvailable: `none`, {names}" - ) + raise ValueError(f"Unknown personality: `{str(value).strip()}`.\n\nAvailable: `none`, {names}") return name, _render_personality_prompt(personalities[name]) def _prompt_text(value) -> str: - """Normalize config prompt values from YAML before handing them to AIAgent. - - Delegates to hermes_cli.personality (single owner). - """ + """Normalize config prompt values from YAML for AIAgent (hermes_cli.personality owns this).""" from hermes_cli.personality import prompt_text return prompt_text(value) @@ -362,68 +252,50 @@ def _prompt_text(value) -> str: def _apply_personality_to_session( sid: str, session: dict, new_prompt: str, personality: str = "" ) -> tuple[bool, dict | None]: - """Apply a personality change to an existing session without resetting history. + """Apply a personality change to a live session without resetting history. - Updates the agent's ephemeral system prompt in-place so the new personality - takes effect on the next turn. The cached base system prompt is left intact - (ephemeral_system_prompt is appended at API-call time, not baked into the - cache), which preserves prompt-cache hits. - - Also injects a system-role marker into the conversation history so the model - knows to pivot its style from this point forward (without this, LLMs tend to - continue the tone established by earlier messages in the transcript). - - Returns (history_reset, info) — history_reset is always False since we - preserve the conversation. + Updates the ephemeral system prompt in place (appended at API-call time, so + prompt-cache hits survive) and injects a pivot marker so the model stops + pattern-matching its earlier tone. Returns (history_reset=False, info). """ if not session: return False, None session["personality"] = personality agent = session.get("agent") - if agent: - agent.ephemeral_system_prompt = new_prompt or None - # Inject a pivot marker into history so the model sees the change point. - # This prevents it from pattern-matching its prior style. - if new_prompt: - marker = ( - "[System: The user has changed the assistant's personality. " - "From this point forward, adopt the following persona and respond " - f"accordingly: {new_prompt}]" - ) - else: - marker = ( - "[System: The user has cleared the personality overlay. " - "From this point forward, respond in your normal default style.]" - ) - # Tagged like the model-switch marker (`_append_model_switch_marker`): - # the marker rides as role=user so strict OpenAI-compatible providers - # accept it mid-conversation, but `display_kind` keeps it out of the - # `truncate_before_user_ordinal` addressing space. Untagged, it counts - # as a real user turn on the gateway side while no client counts it, so - # every later rewind resolves one turn too early and `replace_messages` - # hard-deletes the difference (#82756). - with session["history_lock"]: - session["history"].append( - {"role": "user", "content": marker, "display_kind": "personality_switch"} - ) - session["history_version"] = int(session.get("history_version", 0)) + 1 - info = _session_info(agent) - _emit("session.info", sid, info) - return False, info - return False, None + if not agent: + return False, None + agent.ephemeral_system_prompt = new_prompt or None + if new_prompt: + marker = ( + "[System: The user has changed the assistant's personality. " + "From this point forward, adopt the following persona and respond " + f"accordingly: {new_prompt}]" + ) + else: + marker = ( + "[System: The user has cleared the personality overlay. " + "From this point forward, respond in your normal default style.]" + ) + # Like the model-switch marker: role=user so strict providers accept it + # mid-conversation, but `display_kind` keeps it out of the + # `truncate_before_user_ordinal` addressing space (untagged, every rewind would + # land one turn early and `replace_messages` hard-delete the difference). + with session["history_lock"]: + session["history"].append({"role": "user", "content": marker, "display_kind": "personality_switch"}) + session["history_version"] = int(session.get("history_version", 0)) + 1 + info = _session_info(agent) + _emit("session.info", sid, info) + return False, info def _cfg_max_turns(cfg: dict, default: int) -> int: from hermes_cli.config import resolve_turn_limit as _resolve_turn_limit - # Env var override (highest priority) + # Env override wins; resolve_turn_limit makes "none"/"unlimited"/0 first-class spellings. env_val = os.environ.get("HERMES_TUI_MAX_TURNS") if env_val: return _resolve_turn_limit(env_val, default=default) - # Config file value — route through resolve_turn_limit so that - # "none"/"unlimited"/0 are first-class spellings, not int() crashes. - agent_cfg = cfg.get("agent") or {} - raw = agent_cfg.get("max_turns") + raw = (cfg.get("agent") or {}).get("max_turns") if raw is None: raw = cfg.get("max_turns") if raw is not None: @@ -434,24 +306,17 @@ def _cfg_max_turns(cfg: dict, default: int) -> int: def _parse_tui_skills_env() -> list[str]: raw = os.environ.get("HERMES_TUI_SKILLS", "") skills: list[str] = [] - seen: set[str] = set() for part in raw.replace("\n", ",").split(","): item = part.strip() - if item and item not in seen: - seen.add(item) + if item and item not in skills: skills.append(item) return skills def _load_fallback_model(): - """Return the configured fallback chain for TUI-created agents. - - Delegates to the shared ``get_fallback_chain`` helper so the TUI path - stays in parity with ``HermesCLI.__init__`` and ``gateway/run.py``: - ``fallback_providers`` is the primary source of truth and keeps its - order, with legacy ``fallback_model`` entries merged in afterwards - (deduped on provider/model/base_url). - """ + """Configured fallback chain for TUI-created agents, via the shared + ``get_fallback_chain`` (parity with HermesCLI/gateway: ``fallback_providers`` + first in order, legacy ``fallback_model`` merged after, deduped).""" from hermes_cli.fallback_config import get_fallback_chain return get_fallback_chain(_load_cfg()) @@ -460,9 +325,9 @@ def _load_fallback_model(): def _agent_fallback_model(agent): """Return an agent's fallback chain without rehydrating deliberately empty chains.""" if hasattr(agent, "_fallback_chain"): - return getattr(agent, "_fallback_chain") or [] + return agent._fallback_chain or [] if hasattr(agent, "_fallback_model"): - return getattr(agent, "_fallback_model", None) + return agent._fallback_model return _load_fallback_model() @@ -478,23 +343,17 @@ def _background_agent_kwargs(agent, task_id: str) -> dict: "acp_args": getattr(agent, "acp_args", None) or None, "model": getattr(agent, "model", None) or _resolve_model(), "max_iterations": _cfg_max_turns(cfg, 25), - "enabled_toolsets": getattr(agent, "enabled_toolsets", None) - # Detached background tasks declare platform="tui" below: they have no - # UI session id, so a renderer-routed event has nowhere to land. Resolve - # their toolsets against that same platform rather than the gateway - # process's, so they never carry GUI schema they cannot use. - or _load_enabled_toolsets("tui"), + # Detached tasks declare platform="tui" (no UI sid for renderer-routed + # events), so resolve toolsets against it — never GUI schema they can't use. + "enabled_toolsets": getattr(agent, "enabled_toolsets", None) or _load_enabled_toolsets("tui"), "quiet_mode": True, "verbose_logging": False, - "ephemeral_system_prompt": getattr(agent, "ephemeral_system_prompt", None) - or None, + "ephemeral_system_prompt": getattr(agent, "ephemeral_system_prompt", None) or None, "providers_allowed": getattr(agent, "providers_allowed", None), "providers_ignored": getattr(agent, "providers_ignored", None), "providers_order": getattr(agent, "providers_order", None), "provider_sort": getattr(agent, "provider_sort", None), - "provider_require_parameters": getattr( - agent, "provider_require_parameters", False - ), + "provider_require_parameters": getattr(agent, "provider_require_parameters", False), "provider_data_collection": getattr(agent, "provider_data_collection", None), "openrouter_min_coding_score": getattr(agent, "openrouter_min_coding_score", None), "session_id": task_id, @@ -510,30 +369,15 @@ def _background_agent_kwargs(agent, task_id: str) -> dict: def _ephemeral_preview_agent_kwargs(agent, task_id: str) -> dict: kwargs = _background_agent_kwargs(agent, task_id) - kwargs.update( - { - "enabled_toolsets": ["terminal", "file"], - "session_db": None, - "skip_memory": True, - } - ) + kwargs.update({"enabled_toolsets": ["terminal", "file"], "session_db": None, "skip_memory": True}) return kwargs def _preview_restart_history(session: dict, max_messages: int = 24, max_tool_chars: int = 1200) -> list[dict]: - """Distill the parent session's recent history into a context the - ephemeral preview-restart agent can actually use. - - The restart agent has no idea what app the user was building, what - server they ran, what cwd was active, or which port belongs to which - project. Without this, it would take the bare URL + console logs and - guess — usually starting the wrong thing. - - We keep the last ``max_messages`` messages from the parent session so - the restart agent sees recent user prompts, assistant replies, and - most importantly any terminal/tool calls. Tool result payloads are - truncated so we don't blow the context window with file dumps. - """ + """Distill recent parent history for the ephemeral preview-restart agent, which + otherwise would guess app/server/cwd/port from the bare URL + console logs. + Keeps the last ``max_messages`` (always back to the last user turn); tool + results are truncated so file dumps don't blow the context window.""" try: with session["history_lock"]: history = list(session.get("history", []) or []) @@ -542,20 +386,12 @@ def _preview_restart_history(session: dict, max_messages: int = 24, max_tool_cha if not history: return [] - - # Anchor on the last user turn so we always include at least the most - # recent request and the assistant/tool work that followed it. Then - # extend backwards up to max_messages so we capture the prior context. - last_user_idx = None + start = max(0, len(history) - max_messages) for idx in range(len(history) - 1, -1, -1): if history[idx].get("role") == "user": - last_user_idx = idx + start = min(start, idx) break - start = max(0, len(history) - max_messages) - if last_user_idx is not None: - start = min(start, last_user_idx) - trimmed: list[dict] = [] for msg in history[start:]: if not isinstance(msg, dict): @@ -563,19 +399,12 @@ def _preview_restart_history(session: dict, max_messages: int = 24, max_tool_cha role = msg.get("role") if role not in ("user", "assistant", "tool", "system"): continue - copy = {k: v for k, v in msg.items() if k != "reasoning"} - # Truncate heavy tool outputs so a single 50KB file read doesn't - # crowd out the rest of the context. if role == "tool": content = copy.get("content") if isinstance(content, str) and len(content) > max_tool_chars: - copy["content"] = ( - content[:max_tool_chars] - + f"\n... (truncated, original {len(content)} chars)" - ) + copy["content"] = content[:max_tool_chars] + f"\n... (truncated, original {len(content)} chars)" trimmed.append(copy) - return trimmed @@ -584,20 +413,16 @@ def _preview_tool_result_preview(name: str, result: str) -> str: data = json.loads(result) except Exception: return "" - if not isinstance(data, dict): return "" - if name == "terminal": output = str(data.get("output") or "").strip() - exit_code = data.get("exit_code") if output: return output[-1200:] if data.get("session_id"): return f"Background process started: {data.get('session_id')}" - if exit_code is not None: - return f"terminal exited with code {exit_code}" - + if data.get("exit_code") is not None: + return f"terminal exited with code {data.get('exit_code')}" return str(data.get("error") or "").strip()[:1200] @@ -638,40 +463,31 @@ def _preview_restart_callbacks(parent: str, task_id: str) -> dict: def _reset_session_agent(sid: str, session: dict) -> dict: tokens = _set_session_context(session["session_key"]) try: - # /new is a full conversation boundary: session-scoped runtime - # overrides (/model, /reasoning, /fast) do NOT carry forward — the - # fresh agent re-derives model/provider, reasoning, and service tier - # from config.yaml (#48055, #23131). Session pins are cleared below so - # a rebuild can't resurrect them. (Global process state is still never - # touched — see the cross-session-contamination note in - # _apply_model_switch.) - session.pop("model_override", None) - session.pop("create_reasoning_override", None) - session.pop("create_service_tier_override", None) - session.pop("one_turn_model_restore", None) + # /new is a full conversation boundary: session-scoped runtime overrides + # (/model, /reasoning, /fast) do NOT carry forward — the fresh agent + # re-derives them from config.yaml, and the pins are cleared so a rebuild + # can't resurrect them. Global process state is never touched (see the + # cross-session-contamination note in _apply_model_switch). + for k in ("model_override", "create_reasoning_override", "create_service_tier_override", "one_turn_model_restore"): + session.pop(k, None) new_agent = _make_agent( sid, session["session_key"], session_id=session["session_key"], platform_override=_session_source(session), - context_cwd_is_launch_artifact=( - _context_cwd_is_launch_artifact(session) - ), + context_cwd_is_launch_artifact=_context_cwd_is_launch_artifact(session), ) finally: _clear_session_context(tokens) - session["agent"] = new_agent - session["config_model_seen"] = _config_model_target() - session["attached_images"] = [] - session["queued_prompt"] = None + session.update( + agent=new_agent, config_model_seen=_config_model_target(), attached_images=[], queued_prompt=None + ) session.pop("queued_prompts", None) session["_queued_prompt_generation"] = int(session.get("_queued_prompt_generation", 0)) + 1 - session["edit_snapshots"] = {} - session["image_counter"] = 0 - session["running"] = False - session["show_reasoning"] = _load_show_reasoning() - session["tool_progress_mode"] = _load_tool_progress_mode() - session["tool_started_at"] = {} + session.update( + edit_snapshots={}, image_counter=0, running=False, show_reasoning=_load_show_reasoning(), + tool_progress_mode=_load_tool_progress_mode(), tool_started_at={}, + ) with session["history_lock"]: session["history"] = [] session["history_version"] = int(session.get("history_version", 0)) + 1 diff --git a/tui_gateway/change_watcher.py b/tui_gateway/change_watcher.py index ddb7d8f87f..369eed7a64 100644 --- a/tui_gateway/change_watcher.py +++ b/tui_gateway/change_watcher.py @@ -21,8 +21,7 @@ def resolve_skin() -> dict: return { "name": skin.name, "colors": skin.colors, - # Paired palettes: the TUI detects the terminal's polarity and - # prefers the matching hand-tuned block over adapting `colors`. + # Paired palettes: the TUI prefers the block matching terminal polarity. "light_colors": skin.light_colors, "dark_colors": skin.dark_colors, "branding": skin.branding, @@ -35,28 +34,39 @@ def resolve_skin() -> dict: return {} -# Signature of the last skin broadcast: (name, active user-file mtime). Lets the -# per-tool reconcile fire ``skin.changed`` on any real move — a name switch OR a -# live color edit to the active skin — and nothing else. +# (name, user-file mtime) of the last skin broadcast: ``skin.changed`` fires on a +# name switch OR a live color edit of the active skin, and nothing else. _last_skin_sig: tuple[str, float | None] | None = None +def _watcher_home() -> Path: + """Active profile home for the change watcher's signature probes.""" + override = get_hermes_home_override() + return Path(override if isinstance(override, str) and override else _hermes_home) + + +def _watcher_mtime_ns(path: Path): + """``st_mtime_ns`` of ``path``, or None when it cannot be stat'ed.""" + try: + return path.stat().st_mtime_ns + except OSError: + return None + + def _skin_sig() -> tuple[str, float | None]: """(active skin name, its user-file mtime). Built-ins have no file, so only their name moves; a user skin's mtime lets an in-place color edit repaint too.""" name = str((_load_cfg().get("display") or {}).get("skin") or "default") - override = get_hermes_home_override() - home = override if isinstance(override, str) and override else _hermes_home + home = _watcher_home() try: - mtime: float | None = (Path(home) / "skins" / f"{name}.yaml").stat().st_mtime + mtime: float | None = (home / "skins" / f"{name}.yaml").stat().st_mtime except OSError: mtime = None return name, mtime def _note_skin_broadcast() -> None: - """Sync the reconcile baseline after the /skin RPC emits, so the per-tool - check doesn't re-broadcast the skin /skin just applied.""" + """Sync the baseline after the /skin RPC emits so the watcher doesn't re-broadcast it.""" global _last_skin_sig try: _last_skin_sig = _skin_sig() @@ -65,14 +75,8 @@ def _note_skin_broadcast() -> None: def _broadcast_skin_if_changed() -> None: - """Emit ``skin.changed`` when the active skin moved — the agent switched it - (``hermes config set display.skin``) OR edited the active skin's colors in - place ("I don't like that coral" → tweak the YAML). - - Routes through the SAME live path as ``/skin`` so every surface (TUI + desktop) - repaints, no slash command. The signature check is a dict lookup + one stat, - so polling it is ~free. - """ + """Emit ``skin.changed`` when the active skin moved, via the SAME live path as + ``/skin`` so every surface repaints. The check is a dict lookup + one stat.""" global _last_skin_sig try: sig = _skin_sig() @@ -87,18 +91,8 @@ def _broadcast_skin_if_changed() -> None: pass -def _watcher_home() -> Path: - """Active profile home for the change watcher's signature probes.""" - override = get_hermes_home_override() - return Path(override if isinstance(override, str) and override else _hermes_home) - - def _pet_sig() -> tuple: - """(slug, spritesheet revision, scale) of the active pet — ("off",) when none. - - Cheap by construction: config comes from the mtime-cached ``_load_cfg`` and - the sheet revision is one stat. Moves when ``/pet`` (de)activates a pet, the - hatch flow rebuilds a sheet, or the scale changes.""" + """(slug, spritesheet revision, scale) of the active pet — ("off",) when none.""" display = _load_cfg().get("display") or {} pet_cfg = display.get("pet") if isinstance(display.get("pet"), dict) else {} if not pet_cfg or not is_truthy_value(pet_cfg.get("enabled"), default=False): @@ -113,8 +107,7 @@ def _pet_sig() -> tuple: def _pet_changed_payload() -> dict: - """``pet.info.meta``-shaped payload for ``pet.changed`` — enough for the - renderer to decide whether the heavy sprite payload needs a refetch.""" + """``pet.info.meta``-shaped payload so the renderer can decide whether to refetch sprites.""" try: enabled, pet, scale = _pet_active_selection() if not enabled or pet is None or not pet.exists: @@ -131,56 +124,33 @@ def _pet_changed_payload() -> dict: def _cron_sig(): - """mtime of the profile's cron/jobs.json — moves on create/edit/pause/ - remove AND on scheduler tick bookkeeping (last_run/next_run).""" - try: - return (_watcher_home() / "cron" / "jobs.json").stat().st_mtime_ns - except OSError: - return None + """mtime of cron/jobs.json — moves on edits AND scheduler tick bookkeeping.""" + return _watcher_mtime_ns(_watcher_home() / "cron" / "jobs.json") def _sessions_sig(): - """Newest mtime across state.db and its WAL — the cross-process change - signal. Messaging-gateway turns and cron runs are written by OTHER - processes that never touch this gateway's transports; the shared SQLite - file is the one thing they all move (#58671). A backend serving several - profiles owns one store per profile, so every served sibling home is - probed too — otherwise a routed profile's Bot Chat never refreshes.""" - sig = None - for root in (_watcher_home(), *_served_profile_homes): - for name in ("state.db", "state.db-wal"): - try: - mtime = (root / name).stat().st_mtime_ns - except OSError: - continue - sig = mtime if sig is None else max(sig, mtime) - return sig + """Newest mtime across state.db + WAL: the one thing messaging-gateway turns and + cron runs (which never touch this gateway's transports) all move. Served sibling + profile homes are probed too, else a routed profile's Bot Chat never refreshes.""" + mtimes = [ + _watcher_mtime_ns(root / name) + for root in (_watcher_home(), *_served_profile_homes) + for name in ("state.db", "state.db-wal") + ] + return max((m for m in mtimes if m is not None), default=None) def _platforms_sig(): - """mtime of gateway_state.json — the messaging gateway process persists - platform connect/disconnect/health there, so its movement is the - "connection status changed" signal for the Messaging page.""" - try: - return (_watcher_home() / "gateway_state.json").stat().st_mtime_ns - except OSError: - return None + """mtime of gateway_state.json — where the messaging gateway persists platform + connect/disconnect/health, i.e. the Messaging page's status-changed signal.""" + return _watcher_mtime_ns(_watcher_home() / "gateway_state.json") def _pairing_sig(): - """Newest mtime across every profile's pairing store. - - An unknown DMer's pending code is written by the messaging gateway — a - DIFFERENT process that never touches this gateway's transports — so the - files are the only shared signal. ``platforms.changed`` cannot stand in - for this: it tracks connect/disconnect/health, and a pairing request - moves nothing in gateway_state.json. - """ + """Newest mtime across every profile's pairing store (legacy ``pairing/`` and + ``platforms/pairing/``). Pending codes are written by the gateway process, so the + files are the only shared signal; a pairing request moves nothing in gateway_state.json.""" home = _watcher_home() - sig = None - # Global store (legacy `pairing/` and consolidated `platforms/pairing/`) - # plus every named profile's own — the Messaging page can be scoped to any - # of them, and a request landing in a profile store must still tick. roots = [home / "pairing", home / "platforms" / "pairing"] try: for profile_dir in (home / "profiles").iterdir(): @@ -189,53 +159,39 @@ def _pairing_sig(): except OSError: pass + sig = None for root in roots: try: entries = list(root.iterdir()) except OSError: continue for entry in entries: - # Only the pending/approved ledgers — _rate_limits.json moves on - # every unauthorized DM, including ones that produce no new row. + # Only the ledgers: _rate_limits.json moves on every unauthorized DM. if not entry.name.endswith(("-pending.json", "-approved.json")): continue - try: - mtime = entry.stat().st_mtime_ns - except OSError: - continue - sig = mtime if sig is None else max(sig, mtime) + mtime = _watcher_mtime_ns(entry) + if mtime is not None: + sig = mtime if sig is None else max(sig, mtime) return sig -# Newest outbox-envelope mtime the watcher has EVER seen (monotone). A drain -# empties the outbox (rename → claimed/), and letting the signature fall back -# to None on empty would fire a spurious pending event right after every -# drain — so the signature only moves forward, on genuinely new envelopes. +# Newest outbox-envelope mtime EVER seen (monotone): a drain empties the outbox, +# and falling back to None would fire a spurious pending event after every drain. _bot_relay_outbox_seen = 0 def _bot_relay_outbox_sig(): - """Newest mtime across pending bot-relay outbox envelopes (monotone). - - Envelopes are written by the AGENT process (``message_agent`` → - ``tools.bot_relay.enqueue_envelope``) — a different process that never - touches this gateway's transports — so the files are the only shared - signal, exactly like the pairing store. The Desktop reacts to - ``bot_relay.outbox.pending`` with an immediate (debounced) drain instead - of waiting out its poll interval (#93091, motivated by #92760). - """ + """Newest mtime across pending bot-relay outbox envelopes (monotone). Written by + the AGENT process, so the files are the only shared signal; the Desktop reacts + to ``bot_relay.outbox.pending`` with an immediate debounced drain.""" global _bot_relay_outbox_seen home = _watcher_home() root = home.parent.parent if home.parent.name == "profiles" else home newest = 0 try: for entry in (root / "bot_relay" / "outbox").iterdir(): - if not entry.name.endswith(".json"): - continue - try: - newest = max(newest, entry.stat().st_mtime_ns) - except OSError: - continue + if entry.name.endswith(".json"): + newest = max(newest, _watcher_mtime_ns(entry) or 0) except OSError: pass if newest > _bot_relay_outbox_seen: @@ -243,25 +199,21 @@ def _bot_relay_outbox_sig(): return _bot_relay_outbox_seen or None -# Watched change signals: event → (check interval, signature fn, payload fn). -# Signatures are stat/dict-lookup cheap, same bar as the skin watcher; the -# check interval keeps the pricier probes (pet resolves the active sheet off -# disk) off the 0.5s tick. +# event → (check interval, signature fn, payload fn). Signatures are stat-cheap; +# the interval keeps pricier probes (pet resolves the sheet off disk) off the 0.5s tick. _CHANGE_WATCHES: dict[str, tuple[float, Any, Any]] = { "pet.changed": (2.0, _pet_sig, _pet_changed_payload), "cron.changed": (1.0, _cron_sig, lambda: {}), "sessions.changed": (0.5, _sessions_sig, lambda: {}), "platforms.changed": (2.0, _platforms_sig, lambda: {}), "pairing.changed": (2.0, _pairing_sig, lambda: {}), - # Cross-connection DM latency: 1s check so a queued envelope reaches the - # Desktop's push-triggered drain fast; the Desktop's poll stays backstop. + # 1s so a queued DM envelope reaches the Desktop's push-triggered drain fast. "bot_relay.outbox.pending": (1.0, _bot_relay_outbox_sig, lambda: {}), } -# state.db moves on every message append during a streaming turn, and the -# gateway rewrites gateway_state.json for in-flight-count bookkeeping; the -# floor coalesces those bursts to one broadcast per window (trailing edge -# included — a floored change keeps its old signature and re-fires next tick). +# state.db moves on every append of a streaming turn and gateway_state.json on +# in-flight bookkeeping; the floor coalesces bursts to one broadcast per window, +# trailing edge included (a floored change keeps its old signature, re-fires later). _CHANGE_BROADCAST_FLOOR_S = {"sessions.changed": 2.0, "platforms.changed": 5.0} _change_sigs: dict[str, Any] = {} @@ -270,9 +222,8 @@ _change_broadcast_at: dict[str, float] = {} def _broadcast_watched_changes(now: float | None = None) -> None: - """One pass over ``_CHANGE_WATCHES``: recompute due signatures, broadcast - the events whose signature moved. First sighting seeds silently so a - gateway boot never fires a spurious refresh storm.""" + """One pass: recompute due signatures, broadcast events whose signature moved. + First sighting seeds silently so a gateway boot never fires a refresh storm.""" now = time.monotonic() if now is None else now for event, (interval, sig_fn, payload_fn) in _CHANGE_WATCHES.items(): if now - _change_checked_at.get(event, -interval) < interval: @@ -289,9 +240,7 @@ def _broadcast_watched_changes(now: float | None = None) -> None: continue floor = _CHANGE_BROADCAST_FLOOR_S.get(event, 0.0) if floor and now - _change_broadcast_at.get(event, -floor) < floor: - # Floored: leave the old signature in place so the change re-fires - # once the window opens (the trailing edge of the burst). - continue + continue # floored: old signature stays so it re-fires when the window opens _change_sigs[event] = sig _change_broadcast_at[event] = now try: @@ -304,12 +253,9 @@ _skin_watcher_started = False def _ensure_skin_watcher() -> None: - """Watch cheap on-disk signatures and broadcast change events — so a skin - Hermes activates, a pet ``/pet`` adopts, a cron the scheduler fires, or a - messaging turn another process writes goes live on every surface within a - couple seconds, on its own, with no client-side poll in the loop. - Idempotent; started at gateway.ready. (Named for its original skin-only - duty; it is the process's one change watcher.)""" + """Start the process's one change watcher (named for its original skin-only + duty): cheap on-disk signatures → broadcast events, so skin/pet/cron/cross-process + changes go live everywhere within seconds without client polling. Idempotent.""" global _skin_watcher_started if _skin_watcher_started: return diff --git a/tui_gateway/compute_host_bridge.py b/tui_gateway/compute_host_bridge.py index 40cbd513c7..696f7b4156 100644 --- a/tui_gateway/compute_host_bridge.py +++ b/tui_gateway/compute_host_bridge.py @@ -15,10 +15,9 @@ _registry = HandlerRegistry() _compute_host_supervisor = None _compute_host_supervisor_lock = threading.Lock() -# Hard cap on how long session.compress blocks its RPC waiting for the compute -# host (#97948). Must stay below the desktop's SESSION_COMPRESS_TIMEOUT_MS -# (660s) so the client receives the `pending` answer instead of its own -# timeout error; the late-ack path covers anything slower. +# Cap on how long session.compress blocks its RPC on the compute host. Must stay +# below the desktop's SESSION_COMPRESS_TIMEOUT_MS (660s) so the client gets the +# `pending` answer, not its own timeout; the late-ack path covers anything slower. _COMPUTE_HOST_COMPRESS_WAIT_CAP_SECS = 630.0 @@ -36,9 +35,8 @@ def _turn_isolation_enabled(cfg: dict | None = None) -> bool: def _session_uses_compute_host(session: dict, cfg: dict | None = None) -> bool: if not _turn_isolation_enabled(cfg): return False - # Phase 1 routes lazy/dashboard sessions whose live AIAgent has not been - # built inside the serving process. Already-built in-process sessions keep - # the historical path unless a prior isolated turn marked host ownership. + # Routes lazy sessions whose AIAgent was never built in-process; already-built + # sessions keep the in-process path unless a prior isolated turn marked host ownership. return bool(session.get("_compute_host_active")) or ( session.get("agent") is None and session.get("agent_ready") is not None ) @@ -60,22 +58,13 @@ def _get_compute_host_supervisor(cfg: dict | None = None): def _compute_host_turn_frame( - rid: str, - sid: str, - session: dict, - text: Any, - image_paths: list[str] | None = None, - queued_prompt_generation: int | None = None, - display_kind: str | None = None, + rid: str, sid: str, session: dict, text: Any, image_paths: list[str] | None = None, + queued_prompt_generation: int | None = None, display_kind: str | None = None, ) -> dict: with session["history_lock"]: history = list(session.get("history", [])) history_version = int(session.get("history_version", 0)) - attached_images = ( - list(image_paths) - if image_paths is not None - else list(session.get("attached_images", [])) - ) + attached_images = list(image_paths if image_paths is not None else session.get("attached_images", [])) return { "type": "turn.start", "sid": sid, @@ -103,26 +92,44 @@ def _metadata_mirror(session: dict | None) -> dict: return mirror if isinstance(mirror, dict) else {} +def _compute_host_session_info(session: dict) -> dict: + # Tolerate a legacy one-arg _session_info (tests patch it that way). + try: + return _session_info(session.get("agent"), session) + except TypeError: + return _session_info(session.get("agent")) + + +def _compute_host_adopt_frame_meta(session: dict, frame: dict) -> None: + """Adopt a host frame's session_key / history_version. Caller holds history_lock.""" + if frame.get("session_key"): + session["session_key"] = str(frame.get("session_key")) + if frame.get("history_version") is not None: + try: + session["history_version"] = max( + int(session.get("history_version", 0)), + int(frame.get("history_version") or 0), + ) + except Exception: + pass + + def _relay_compute_host_rpc(message: dict) -> bool: """Relay host events while retaining the clarify snapshot needed on resume.""" params = message.get("params") if isinstance(message, dict) else None - if isinstance(params, dict) and params.get("type") == "clarify.request": - sid = str(params.get("session_id") or "") + kind = params.get("type") if isinstance(params, dict) else None + if kind in {"clarify.request", "clarify.expire"}: + session = _sessions.get(str(params.get("session_id") or "")) payload = params.get("payload") - session = _sessions.get(sid) - if session is not None and isinstance(payload, dict) and payload.get("request_id"): - with session.get("history_lock", threading.Lock()): - session["_compute_host_pending_clarify"] = dict(payload) - elif isinstance(params, dict) and params.get("type") == "clarify.expire": - sid = str(params.get("session_id") or "") - payload = params.get("payload") - session = _sessions.get(sid) request_id = payload.get("request_id") if isinstance(payload, dict) else None if session is not None and request_id: with session.get("history_lock", threading.Lock()): - pending = session.get("_compute_host_pending_clarify") - if isinstance(pending, dict) and pending.get("request_id") == request_id: - session.pop("_compute_host_pending_clarify", None) + if kind == "clarify.request": + session["_compute_host_pending_clarify"] = dict(payload) + else: + pending = session.get("_compute_host_pending_clarify") + if isinstance(pending, dict) and pending.get("request_id") == request_id: + session.pop("_compute_host_pending_clarify", None) return write_json(message) @@ -185,25 +192,13 @@ def _respond_compute_host_clarify(rid: str, params: dict) -> dict | None: def _apply_compute_host_metadata_mirror(session: dict, frame: dict | None) -> None: - """Mirror host-owned session metadata in the serving process. - - The compute host is the only writer of live agent/history state while turn - isolation is active. The serving process keeps read metadata from the last - host frame so UI reads do not construct a second in-process agent. - """ + """Mirror host-owned session metadata: while turn isolation is active the host is + the only writer of live agent/history state, and UI reads must not build a + second in-process agent.""" if not isinstance(frame, dict): return with session.get("history_lock", threading.Lock()): - if frame.get("session_key"): - session["session_key"] = str(frame.get("session_key")) - if frame.get("history_version") is not None: - try: - session["history_version"] = max( - int(session.get("history_version", 0)), - int(frame.get("history_version") or 0), - ) - except Exception: - pass + _compute_host_adopt_frame_meta(session, frame) if frame.get("message_count") is not None: try: session["_metadata_message_count"] = int(frame.get("message_count") or 0) @@ -218,60 +213,36 @@ def _apply_compute_host_metadata_mirror(session: dict, frame: dict | None) -> No def _on_compute_host_turn_done(rid: str, sid: str, session: dict, frame: dict) -> None: - is_error = frame.get("type") == "turn.error" with session["history_lock"]: - if frame.get("session_key"): - session["session_key"] = str(frame.get("session_key")) - if frame.get("history_version") is not None: - try: - session["history_version"] = max( - int(session.get("history_version", 0)), - int(frame.get("history_version") or 0), - ) - except Exception: - pass + _compute_host_adopt_frame_meta(session, frame) session["running"] = False session["last_active"] = time.time() _clear_inflight_turn(session) session.pop("_compute_host_pending_clarify", None) - if is_error: + if frame.get("type") == "turn.error": message = str(frame.get("message") or "compute host turn failed") _emit("message.complete", sid, {"text": f"Error: {message}", "status": "error"}) _apply_compute_host_metadata_mirror(session, frame) - try: - info = _session_info(session.get("agent"), session) - except TypeError: - info = _session_info(session.get("agent")) + info = _compute_host_session_info(session) if not frame.get("session_info_emitted"): _emit("session.info", sid, info) _drain_queued_prompt(rid, sid, session) def _submit_prompt_to_compute_host( - rid: str, - sid: str, - session: dict, - text: Any, - image_paths: list[str] | None = None, - queued_prompt_generation: int | None = None, - display_kind: str | None = None, + rid: str, sid: str, session: dict, text: Any, image_paths: list[str] | None = None, + queued_prompt_generation: int | None = None, display_kind: str | None = None, ) -> dict: cfg = _load_dashboard_process_isolation_config() frame = _compute_host_turn_frame( - rid, - sid, - session, - text, - image_paths=image_paths, - queued_prompt_generation=queued_prompt_generation, - display_kind=display_kind, + rid, sid, session, text, image_paths=image_paths, + queued_prompt_generation=queued_prompt_generation, display_kind=display_kind, ) def _complete(done: dict) -> None: - # submit_turn reports a synchronous pipe failure through the callback - # before re-raising. Leave the parent session untouched so prompt.submit - # can fail open to the historical in-process path without emitting a - # duplicate terminal error. + # submit_turn reports a synchronous pipe failure via the callback before + # re-raising; leave the session untouched so prompt.submit can fail open + # to the in-process path without a duplicate terminal error. if done.get("reason") == "send_failed": return _on_compute_host_turn_done(rid, sid, session, done) @@ -288,74 +259,45 @@ def _submit_prompt_to_compute_host( def _send_compute_host_control( - sid: str, - *, - route_name: str, - command: str = "", - payload: dict | None = None, - wait: bool = True, - timeout: float = 30.0, - on_late_ack=None, + sid: str, *, route_name: str, command: str = "", payload: dict | None = None, + wait: bool = True, timeout: float = 30.0, on_late_ack=None, ) -> dict: frame = dict(payload or {}) frame.setdefault("type", "control") frame.setdefault("command", command) return _get_compute_host_supervisor().control( - sid, - route_name=route_name, - payload=frame, - wait=wait, - timeout=timeout, - on_late_ack=on_late_ack, + sid, route_name=route_name, payload=frame, wait=wait, timeout=timeout, on_late_ack=on_late_ack ) def _compute_host_compress_wait_seconds(cfg: dict | None = None) -> float: - """RPC wait budget for a compute-host compress control (#97948). - - Manual compression legitimately runs up to the configured - ``compression.context_total_ceiling_seconds`` (default 600s), so a fixed - 120s waiter reported a false timeout while the host kept working. Follow - the ceiling with a little slack, but cap the blocking wait so it stays - below the desktop's own RPC timeout; anything longer is adopted through - the late-ack path instead of failing. - """ + """RPC wait budget for a compute-host compress control: the configured + ``compression.context_total_ceiling_seconds`` plus slack, capped below the + desktop's RPC timeout (a fixed waiter reported false timeouts while the host + kept working); anything slower lands via the late-ack path.""" from agent.conversation_compression import resolve_context_compression_timeouts try: compression_cfg = (cfg if cfg is not None else _load_cfg()).get("compression", {}) except Exception: compression_cfg = {} - _idle, ceiling = resolve_context_compression_timeouts( - compression_cfg if isinstance(compression_cfg, dict) else {} - ) + _idle, ceiling = resolve_context_compression_timeouts(compression_cfg if isinstance(compression_cfg, dict) else {}) return float(min(max(ceiling + 30.0, 120.0), _COMPUTE_HOST_COMPRESS_WAIT_CAP_SECS)) def _announce_compute_host_compress_done(sid: str, session: dict, ack: dict) -> None: - """Mirror a compute-host compress ack and push the client-visible edges. - - Emits the same ``session.info`` the in-process /compress path does plus - the ``compacted`` status edge, so a client whose own RPC wait already - expired still learns the transcript changed. - """ + """Mirror a compress ack and push the same ``session.info`` + ``compacted`` edges + the in-process /compress path emits, so a client whose RPC wait expired still + learns the transcript changed.""" _apply_compute_host_metadata_mirror(session, ack) - try: - info = _session_info(session.get("agent"), session) - except TypeError: - info = _session_info(session.get("agent")) - _emit("session.info", sid, info) + _emit("session.info", sid, _compute_host_session_info(session)) _status_update(sid, "compacted", "✓ Context compression complete") def _adopt_late_compute_host_compress_ack(sid: str, session: dict, ack: dict, *, route_name: str) -> None: - """Adopt a compute-host compress ack that arrived after its RPC waiter gave up. - - The RPC already answered ``status: pending``; this is the only place the - rotated session_key / history_version / session_info mirror can land, and - the only signal the client gets that the transcript changed. A late - ``control.error`` surfaces through the existing ``error`` event path. - """ + """Adopt a compress ack that arrived after its RPC waiter answered ``pending``: + the only place the rotated session_key / history_version / mirror can land and + the client's only signal. A late ``control.error`` goes out via ``error``.""" with _sessions_lock: live = _sessions.get(sid) if live is not session: diff --git a/tui_gateway/methods_browser.py b/tui_gateway/methods_browser.py index 2f0f8ed815..c7d64a49c2 100644 --- a/tui_gateway/methods_browser.py +++ b/tui_gateway/methods_browser.py @@ -12,26 +12,14 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# ── Methods: browser / plugins / cron / skills ─────────────────────── - - def _resolve_browser_cdp_url() -> str: - """Return the configured browser CDP override without network I/O. + """Configured browser CDP override without network I/O. - ``/browser status`` must be fast — calling - ``tools.browser_tool._get_cdp_override`` would invoke - ``_resolve_cdp_override``, which performs an HTTP probe to - ``.../json/version`` for discovery-style URLs. That probe has - a multi-second timeout and would block the TUI on a slow or - unreachable host even though status only needs to report whether - an override is set. - - Mirrors the env/config precedence of ``_get_cdp_override`` (env - var first, then ``browser.cdp_url`` from config.yaml) without the - websocket-resolution step, so the answer reflects user intent - even when the configured host is not currently reachable. The - actual WS normalization happens in ``browser_navigate`` on the - next tool call. + ``/browser status`` must be fast: ``tools.browser_tool._get_cdp_override`` runs an + HTTP probe with a multi-second timeout for discovery-style URLs. Mirrors its + precedence (env var, then ``browser.cdp_url``) minus the WS-resolution step, so the + answer reflects user intent even when the host is unreachable; ``browser_navigate`` + normalizes on the next tool call. """ env_url = os.environ.get("BROWSER_CDP_URL", "").strip() if env_url: @@ -49,23 +37,18 @@ def _resolve_browser_cdp_url() -> str: def _is_default_local_cdp(parsed) -> bool: - """Match the discovery-style local default; never the concrete WS form. - - A user-supplied ``ws://127.0.0.1:9222/devtools/browser/`` is a - real, connectable endpoint — collapsing it to bare ``http://...:9222`` - would strip the path and break the connect. - """ + """Match the discovery-style local default; never the concrete WS form — a + ``ws://127.0.0.1:9222/devtools/browser/`` is connectable as-is and collapsing + it to bare ``http://...:9222`` would break the connect.""" try: port = parsed.port or 80 except ValueError: return False - - discovery_path = parsed.path in {"", "/", "/json", "/json/version"} return ( parsed.scheme in {"http", "ws"} and parsed.hostname in {"127.0.0.1", "localhost"} and port == 9222 - and discovery_path + and parsed.path in {"", "/", "/json", "/json/version"} ) @@ -86,10 +69,8 @@ def _probe_urls(parsed) -> list[str]: def _normalize_cdp_url(parsed) -> str: - # Concrete ``/devtools/browser/`` endpoints (Browserbase et al.) - # are connectable as-is. Discovery-style inputs collapse to bare - # ``scheme://host:port`` so ``_resolve_cdp_override`` can append - # ``/json/version`` later without doubling the path. + # Concrete ``/devtools/browser/`` endpoints stay as-is; discovery-style inputs + # collapse to ``scheme://host:port`` so ``_resolve_cdp_override`` can append ``/json/version``. if parsed.path.startswith("/devtools/browser/"): return parsed.geturl() return parsed._replace(path="", params="", query="", fragment="").geturl() @@ -123,9 +104,7 @@ def _browser_connect(rid, params: dict) -> dict: raw_url = params.get("url") if raw_url is not None and not isinstance(raw_url, str): - return _err( - rid, 4015, f"browser url must be a string, got {type(raw_url).__name__}" - ) + return _err(rid, 4015, f"browser url must be a string, got {type(raw_url).__name__}") url = (raw_url or "").strip() or DEFAULT_BROWSER_CDP_URL sid = params.get("session_id") or "" @@ -134,9 +113,7 @@ def _browser_connect(rid, params: dict) -> dict: def announce(message: str, *, level: str = "info") -> None: messages.append(message) - # Without a session id the TUI prints `messages` from the - # response; emitting an event would double-render. Only stream - # progress when there's a real session to scope it to. + # Without a session id the TUI prints `messages` from the response; an event would double-render. if sid: _emit("browser.progress", sid, {"message": message, "level": level}) @@ -150,20 +127,16 @@ def _browser_connect(rid, params: dict) -> dict: except ValueError: return _err(rid, 4015, f"invalid port in browser url: {url}") - # Always normalize default-local to 127.0.0.1:9222 so downstream - # comparisons + messaging match what we'll actually persist. + # Normalize default-local to 127.0.0.1:9222 so comparisons + messaging match what we persist. if _is_default_local_cdp(parsed): url = DEFAULT_BROWSER_CDP_URL parsed = urlparse(url) port = parsed.port or 9222 try: - # ws[s]://.../devtools/browser/ endpoints (hosted CDP - # providers) don't serve the HTTP discovery path; just check - # TCP-level reachability and let browser_navigate handshake. - if parsed.scheme in {"ws", "wss"} and parsed.path.startswith( - "/devtools/browser/" - ): + # Hosted ws[s]://.../devtools/browser/ endpoints don't serve the HTTP discovery + # path: check TCP reachability only and let browser_navigate handshake. + if parsed.scheme in {"ws", "wss"} and parsed.path.startswith("/devtools/browser/"): import socket try: @@ -179,12 +152,9 @@ def _browser_connect(rid, params: dict) -> dict: local_port_in_use, ) - # Dual-stack discovery: when another app (an IDE debugger, - # a dev server) squats the IPv4 loopback on the debug port, - # a browser asked to bind that port comes up on [::1] only. - # An IPv4-only probe misses it AND hangs against squatters - # that accept TCP but never answer HTTP — the historic - # cause of `browser.manage` RPC timeouts. + # Dual-stack discovery: when another app squats the IPv4 loopback on the debug + # port, a browser bound there comes up on [::1] only. An IPv4-only probe misses + # it AND hangs against squatters that accept TCP but never answer HTTP. discovered = discover_local_cdp_url(port, timeout=2.0) launch_port = port @@ -192,20 +162,16 @@ def _browser_connect(rid, params: dict) -> dict: if local_port_in_use(port): launch_port = find_free_debug_port(port) announce( - f"Port {port} is occupied by another application that " - "isn't a CDP browser (an IDE debugger or dev server may " - f"be using it) — launching a debug browser on port " - f"{launch_port} instead..." + f"Port {port} is occupied by another application that isn't a CDP browser " + "(an IDE debugger or dev server may be using it) — launching a debug browser " + f"on port {launch_port} instead..." ) else: - announce( - "Chromium-family browser isn't running with remote debugging — attempting to launch..." - ) + announce("Chromium-family browser isn't running with remote debugging — attempting to launch...") launch = launch_chrome_debug(launch_port, system) if launch.launched: - # Bounded wait: the whole connect must finish well - # inside the client RPC timeout. + # Bounded wait: the whole connect must finish inside the client RPC timeout. deadline = time.monotonic() + 10.0 while time.monotonic() < deadline: discovered = discover_local_cdp_url(launch_port, timeout=1.0) @@ -214,37 +180,27 @@ def _browser_connect(rid, params: dict) -> dict: time.sleep(0.5) if discovered: - announce( - f"Chromium-family browser launched and listening on port {launch_port}" - ) + announce(f"Chromium-family browser launched and listening on port {launch_port}") else: hint = launch.hint if hint: announce(hint, level="error") for line in _failure_messages(url, launch_port, system)[1:]: announce(line, level="error") - return _ok( - rid, {"connected": False, "url": url, "messages": messages} - ) + return _ok(rid, {"connected": False, "url": url, "messages": messages}) else: announce(f"Chromium-family browser is already listening at {discovered}") - # Adopt whatever loopback/port actually answered (may be - # [::1] and/or an alternate port when 9222 was squatted). + # Adopt whatever loopback/port answered ([::1] and/or an alternate port when 9222 was squatted). url = discovered parsed = urlparse(url) - else: - probes = _probe_urls(parsed) - ok = any(_http_ok(p, timeout=2.0) for p in probes) - if not ok: - return _err(rid, 5031, f"could not reach browser CDP at {url}") + elif not any(_http_ok(p, timeout=2.0) for p in _probe_urls(parsed)): + return _err(rid, 5031, f"could not reach browser CDP at {url}") normalized = _normalize_cdp_url(parsed) - # Order matters: reap sessions BEFORE publishing the new env - # so an in-flight tool call sees the old supervisor closed, - # then again AFTER so the default task's cached supervisor - # is drained against the new URL. + # Reap BEFORE publishing the new env (an in-flight tool call sees the old supervisor + # closed) and AFTER (the default task's cached supervisor drains against the new URL). cleanup_all_browsers() os.environ["BROWSER_CDP_URL"] = normalized cleanup_all_browsers() @@ -258,8 +214,7 @@ def _browser_connect(rid, params: dict) -> dict: def _browser_disconnect(rid) -> dict: - # Reap, drop the env override, reap again — closes the same swap - # window covered by ``_browser_connect``. + # Reap, drop the override, reap again — same swap window as ``_browser_connect``. def reap() -> None: try: from tools.browser_tool import cleanup_all_browsers diff --git a/tui_gateway/methods_complete.py b/tui_gateway/methods_complete.py index d18a55e462..60f47469db 100644 --- a/tui_gateway/methods_complete.py +++ b/tui_gateway/methods_complete.py @@ -1,8 +1,7 @@ """Completion / model-key / paste JSON-RPC handlers. -Everything defined here is rebound onto server.py's globals at install time -(``method_ctx.bind_module``), so handlers and module-level helpers may -reference server globals bare (``_ok``, ``_err``, ``_sessions``, ...). +Rebound onto server.py's globals at install time (``method_ctx.bind_module``), so +bodies reference server globals bare (``_ok``, ``_err``, ``_sessions``, ...). """ @@ -55,10 +54,9 @@ def _(rid, params: dict) -> dict: def _profile_mention_items(prefix: str) -> list[dict]: - """`@` completions: agent profiles as mentionable names (multi-agent - UIs and the Bot Mode plugin route `@` text to another profile). - Bare-word matches only, never for `@kind:` directives. The primary profile is - also offered as 'hermes' when no real profile claims that name.""" + """`@` completions (multi-agent UIs route `@` text to another + profile). Bare-word matches only, never `@kind:` directives; the primary profile + is also offered as 'hermes' when no real profile claims that name.""" out: list[dict] = [] try: from hermes_cli.profiles import list_profiles @@ -83,33 +81,32 @@ def _plugin_reference_items(pfx: str, qval: str) -> list[dict] | None: """`@:` autocomplete for a plugin ContextReferenceProvider; None when no provider owns ``pfx`` or it fails.""" try: - from agent.context_references import get_context_reference_providers as _gcr + from agent.context_references import get_context_reference_providers - _prov = _gcr().get(pfx) - if _prov is None: + prov = get_context_reference_providers().get(pfx) + if prov is None: return None - import asyncio as _asyncio + import asyncio - _coro = _prov.autocomplete(qval, limit=20) + coro = prov.autocomplete(qval, limit=20) try: - _loop = _asyncio.get_running_loop() + loop = asyncio.get_running_loop() except RuntimeError: - _loop = None - if _loop and _loop.is_running(): - import concurrent.futures as _cf + loop = None + if loop and loop.is_running(): + import concurrent.futures - with _cf.ThreadPoolExecutor(max_workers=1) as _pool: - _ac = _pool.submit(_asyncio.run, _coro).result() + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: + ac = pool.submit(asyncio.run, coro).result() else: - _ac = _asyncio.run(_coro) - return [{"text": f"@{pfx}:{it.text}", "display": it.display, "meta": it.meta} for it in _ac] + ac = asyncio.run(coro) + return [{"text": f"@{pfx}:{it.text}", "display": it.display, "meta": it.meta} for it in ac] except Exception: return None def _fuzzy_basename_items(root: str, path_part: str, prefix_tag: str) -> list[dict]: - """Fuzzy basename search across the repo for a bare `@name` (what Cursor / - VS Code do for Cmd-P); path-ish queries take the directory-listing path.""" + """Cmd-P style fuzzy basename search for a bare `@name`; path-ish queries take the listing path.""" ranked: list[tuple[tuple[int, int], str, str, bool]] = [] walked_dirs: set[str] = set() seen: set[str] = set() @@ -123,9 +120,8 @@ def _fuzzy_basename_items(root: str, path_part: str, prefix_tag: str) -> list[di seen.add(rel) ranked.append((rank, rel, name, is_dir)) - # Seed with root's immediate children: `_list_repo_files` is capped at - # _FUZZY_CACHE_MAX_FILES and outside a git repo the fallback walk can burn - # the whole budget on one deep subtree before reaching a sibling. + # Seed with root's immediate children: `_list_repo_files` is capped at _FUZZY_CACHE_MAX_FILES + # and the non-git fallback walk can burn the whole budget on one deep subtree. try: for entry in os.listdir(root): if entry not in _FUZZY_FALLBACK_EXCLUDES: @@ -135,8 +131,7 @@ def _fuzzy_basename_items(root: str, path_part: str, prefix_tag: str) -> list[di for rel in _list_repo_files(root): _consider(rel, os.path.basename(rel), False) - # Directories are only implied by the file listing, so rank each ancestor - # too — otherwise a folder with no name-matching file inside is invisible. + # Rank each ancestor dir too — a folder with no name-matching file inside is otherwise invisible. parent = os.path.dirname(rel) while parent and parent not in walked_dirs: walked_dirs.add(parent) @@ -170,8 +165,7 @@ def _(rid, params: dict) -> dict: if is_context and not query: items = [_item(t, m) for t, m in _AT_DIRECTIVE_HINTS] - # Agent profiles are mentionable — `@` alone reveals them too. - items.extend(_profile_mention_items("")) + items.extend(_profile_mention_items("")) # `@` alone reveals agent profiles too try: from agent.context_references import get_context_reference_providers @@ -189,8 +183,7 @@ def _(rid, params: dict) -> dict: if plugin_items is not None: return _ok(rid, {"items": plugin_items}) - # Accept both `@folder:path` and bare `@folder` so listings appear as soon - # as the keyword is typed, without accepting the static `@folder:` hint. + # Bare `@folder` lists as soon as the keyword is typed (the static `@folder:` hint is not accepted). if is_context and query in {"file", "folder"}: prefix_tag, path_part = query, "" elif is_context and query.startswith(("file:", "folder:")): @@ -198,21 +191,15 @@ def _(rid, params: dict) -> dict: else: prefix_tag, path_part = "", query - # `@/foo` almost always means "foo, from here": take the absolute reading - # only when that prefix exists, else drop the slash and resolve relative - # to cwd — otherwise `@/Desktop` dead-ends. `@/usr/local` still resolves. - if ( - is_context - and path_part.startswith("/") - and not path_part.startswith("//") - and not _abs_completion_prefix_exists(path_part) - ): - path_part = path_part.lstrip("/") + # `@/foo` usually means "foo, from here": absolute only when that prefix exists, + # else resolve relative to cwd (`@/Desktop` must not dead-end; `@/usr/local` still resolves). + if is_context and path_part.startswith("/") and not path_part.startswith("//"): + if not _abs_completion_prefix_exists(path_part): + path_part = path_part.lstrip("/") if is_context and path_part and len(path_part.strip()) >= 2 and "/" not in path_part and prefix_tag != "folder": items = _fuzzy_basename_items(root, path_part, prefix_tag) - # Bare `@name` may equally be an agent mention: profiles rank ABOVE file hits. - if not prefix_tag: + if not prefix_tag: # bare `@name` may be an agent mention: profiles rank ABOVE file hits items = _profile_mention_items(path_part) + items return _ok(rid, {"items": items}) @@ -240,8 +227,7 @@ def _(rid, params: dict) -> dict: continue full = os.path.join(search_dir, entry) is_dir = os.path.isdir(full) - # Explicit `@folder:` / `@file:` filters: skip the opposite kind rather - # than rewriting the tag (which let `@folder:` list files). + # Explicit `@folder:` / `@file:` skip the opposite kind (never rewrite the tag). if prefix_tag and want_dir != is_dir: continue rel = os.path.relpath(full, root).replace(os.sep, "/") @@ -264,8 +250,7 @@ def _(rid, params: dict) -> dict: except Exception as e: return _err(rid, 5021, str(e)) - # Bare-word `@name` (incl. single chars, which skip the fuzzy branch) may be - # an agent mention — profiles rank above path entries. + # Bare-word `@name` (incl. single chars, which skip the fuzzy branch): profiles rank above paths. try: if is_context and not prefix_tag and path_part and "/" not in path_part: items = _profile_mention_items(path_part) + items @@ -292,15 +277,13 @@ def _(rid, params: dict) -> dict: completer = SlashCommandCompleter( skill_commands_provider=lambda: get_skill_commands(), skill_bundles_provider=lambda: get_skill_bundles() ) - # Skills/bundles are the only completions for an inline `/skill` typed - # mid-message, so the class reaches the TUI as data — derived from the - # completer's own providers, not sniffed from the ⚡/▣ display glyphs. + # `kind` reaches the TUI as data (from the providers, not sniffed from ⚡/▣ glyphs): + # skills/bundles are the only completions for an inline `/skill` typed mid-message. skill_names = {key.lstrip("/").lower() for key in (*get_skill_commands(), *get_skill_bundles())} def to_items(doc: Document) -> list[dict]: - # prompt_toolkit's display/display_meta are FormattedText; the TUI's - # CompletionItem.display contract is a plain string (the raw list - # trips Ink's row layout into 1-char truncation of the next column). + # display/display_meta are FormattedText; the TUI contract is a plain string + # (the raw list trips Ink's row layout into 1-char truncation). return [ { "text": c.text, @@ -313,16 +296,12 @@ def _(rid, params: dict) -> dict: items = to_items(Document(text, len(text))) - # Rank and bound (see _rank_slash_completions) while a `/token` is under - # the cursor — the one stage skills are offered at. An argument stage - # (`/personality `, `/details c`) keeps the order its command chose. + # Rank + bound while a `/token` is under the cursor (the one stage skills are + # offered at); an argument stage (`/personality `) keeps its command's order. if text.rsplit(" ", 1)[-1].startswith("/"): score_of = None - # Description-aware fuzzy scoring at the command-token stage: the - # completer only emits name-prefix matches, so merge in catalog entries - # whose name SUBSTRING or DESCRIPTION words match (`/summary` surfaces a - # command whose description mentions summaries). Name matches outrank - # description matches. + # Command-token stage: the completer only emits name-prefix matches, so merge in + # catalog entries whose name SUBSTRING or DESCRIPTION words match (name outranks description). if " " not in text and len(text) > 1: from tui_gateway.slash_fuzzy import fuzzy_rank_slash_items, normalize_slash_search_query @@ -357,8 +336,8 @@ def _(rid, params: dict) -> dict: session = _sessions.get(params.get("session_id", "")) agent = session.get("agent") if session else None - # Once an agent is spawned IT owns the live provider/model/base_url; empty - # agent attributes must NOT clobber disk config (with_overrides is truthy-only). + # A spawned agent owns the live provider/model/base_url; empty attributes must + # NOT clobber disk config (with_overrides is truthy-only). ctx = _model_picker_context(agent) payload = build_model_options_payload( ctx, @@ -373,8 +352,7 @@ def _(rid, params: dict) -> dict: @method("model.save_key") def _(rid, params: dict) -> dict: - """Save an API key for provider ``slug`` and return its refreshed provider row - (same shape as model.options entries, with ``authenticated``).""" + """Save an API key for ``slug``; return its refreshed provider row (model.options shape + ``authenticated``).""" try: from hermes_cli.auth import PROVIDER_REGISTRY from hermes_cli.config import is_managed @@ -384,10 +362,8 @@ def _(rid, params: dict) -> dict: api_key = (params.get("api_key") or "").strip() if not slug or not api_key: return _err(rid, 4001, "slug and api_key are required") - if is_managed(): return _err(rid, 4006, "managed install — credentials are read-only") - pconfig = PROVIDER_REGISTRY.get(slug) if not pconfig: return _err(rid, 4002, f"unknown provider: {slug}") @@ -396,32 +372,20 @@ def _(rid, params: dict) -> dict: if not pconfig.api_key_env_vars: return _err(rid, 4004, f"no env var defined for {pconfig.name}") - # Unified credential lifecycle so any stale config.yaml mirror of the old - # key (model.api_key, custom_providers[*].api_key) rotates in the same action. + # Unified lifecycle rotates stale config.yaml mirrors of the old key too. env_var = pconfig.api_key_env_vars[0] from hermes_cli.credential_lifecycle import save_provider_env_credential save_provider_env_credential(env_var, api_key) - import os - os.environ[env_var] = api_key # so the refreshed inventory sees it - # Shared inventory builder keeps this in lock-step with model.options and - # the dashboard; picker_hints=True carries `authenticated` for the TUI. + # Shared inventory builder (lock-step with model.options / dashboard); picker_hints carries `authenticated`. session = _sessions.get(params.get("session_id", "")) agent = session.get("agent") if session else None payload = build_models_payload(_model_picker_context(agent), picker_hints=True, max_models=50) provider_data = next((p for p in payload["providers"] if p["slug"] == slug), None) - if provider_data is None: - # Key saved but provider didn't appear — still success. - provider_data = { - "slug": slug, - "name": pconfig.name, - "is_current": False, - "models": [], - "total_models": 0, - "authenticated": True, - } + if provider_data is None: # key saved but provider didn't appear — still success + provider_data = {"slug": slug, "name": pconfig.name, "is_current": False, "models": [], "total_models": 0} provider_data["authenticated"] = True # synthetic fallback bypasses picker_hints return _ok(rid, {"provider": provider_data}) except Exception as e: @@ -438,13 +402,10 @@ def _(rid, params: dict) -> dict: slug = (params.get("slug") or "").strip() if not slug: return _err(rid, 4001, "slug is required") - pconfig = PROVIDER_REGISTRY.get(slug) cleared_env = False - - # Remove env vars from .env/process plus every mirror (env-seeded pool - # entries, model cache rows, value-matched config.yaml copies) — otherwise - # the provider resurrects in the picker after restart. + # Remove env vars plus every mirror (env-seeded pool entries, model cache rows, + # value-matched config.yaml copies) or the provider resurrects in the picker after restart. if pconfig and pconfig.api_key_env_vars: for ev in pconfig.api_key_env_vars: if remove_provider_env_credential(ev).get("found"): @@ -452,10 +413,8 @@ def _(rid, params: dict) -> dict: # Full disconnect: removing OAuth grants is intended here, unlike key-only deletes. cleared_auth = clear_provider_auth(slug) - if not cleared_env and not cleared_auth: return _err(rid, 4005, f"no credentials found for {slug}") - return _ok(rid, {"slug": slug, "name": pconfig.name if pconfig else slug, "disconnected": True}) except Exception as e: return _err(rid, 5035, str(e)) diff --git a/tui_gateway/methods_complete_helpers.py b/tui_gateway/methods_complete_helpers.py index f11e0827dd..1e5df8c83c 100644 --- a/tui_gateway/methods_complete_helpers.py +++ b/tui_gateway/methods_complete_helpers.py @@ -18,40 +18,18 @@ _registry = HandlerRegistry() _FUZZY_CACHE_TTL_S = 5.0 _FUZZY_CACHE_MAX_FILES = 20000 _FUZZY_FALLBACK_EXCLUDES = frozenset( - { - ".git", - ".hg", - ".svn", - ".next", - ".cache", - ".venv", - "venv", - "node_modules", - "__pycache__", - "dist", - "build", - "target", - ".mypy_cache", - ".pytest_cache", - ".ruff_cache", - } + {".git", ".hg", ".svn", ".next", ".cache", ".venv", "venv", "node_modules", "__pycache__", + "dist", "build", "target", ".mypy_cache", ".pytest_cache", ".ruff_cache"} ) _fuzzy_cache_lock = threading.Lock() _fuzzy_cache: dict[str, tuple[float, list[str]]] = {} def _list_repo_files(root: str) -> list[str]: - """Return file paths relative to ``root``. - - Uses ``git ls-files`` from the repo top (resolved via - ``rev-parse --show-toplevel``) so the listing covers tracked + untracked - files anywhere in the repo, then converts each path back to be relative - to ``root``. Files outside ``root`` (parent directories of cwd, sibling - subtrees) are excluded so the picker stays scoped to what's reachable - from the gateway's cwd. Falls back to a bounded ``os.walk(root)`` when - ``root`` isn't inside a git repo. Result cached per-root for - ``_FUZZY_CACHE_TTL_S`` so rapid keystrokes don't respawn git processes. - """ + """File paths relative to ``root`` (tracked + untracked via ``git ls-files`` from the + repo top; files outside ``root`` excluded so the picker stays Cmd-P scoped). Falls + back to a bounded ``os.walk(root)`` outside a git repo. Cached per-root for + ``_FUZZY_CACHE_TTL_S`` so rapid keystrokes don't respawn git.""" now = time.monotonic() with _fuzzy_cache_lock: cached = _fuzzy_cache.get(root) @@ -61,45 +39,20 @@ def _list_repo_files(root: str) -> list[str]: files: list[str] = [] from hermes_cli._subprocess_compat import windows_hide_flags - _creationflags = windows_hide_flags() + run_kw = dict(capture_output=True, timeout=2.0, check=False, stdin=subprocess.DEVNULL, creationflags=windows_hide_flags()) try: - top_result = subprocess.run( - ["git", "-C", root, "rev-parse", "--show-toplevel"], - capture_output=True, - timeout=2.0, - check=False, - stdin=subprocess.DEVNULL, - creationflags=_creationflags, - ) + top_result = subprocess.run(["git", "-C", root, "rev-parse", "--show-toplevel"], **run_kw) if top_result.returncode == 0: top = top_result.stdout.decode("utf-8", "replace").strip() list_result = subprocess.run( - [ - "git", - "-C", - top, - "ls-files", - "-z", - "--cached", - "--others", - "--exclude-standard", - ], - capture_output=True, - timeout=2.0, - check=False, - stdin=subprocess.DEVNULL, - creationflags=_creationflags, + ["git", "-C", top, "ls-files", "-z", "--cached", "--others", "--exclude-standard"], **run_kw ) if list_result.returncode == 0: for p in list_result.stdout.decode("utf-8", "replace").split("\0"): if not p: continue - rel = os.path.relpath(os.path.join(top, p), root).replace( - os.sep, "/" - ) - # Skip parents/siblings of cwd — keep the picker scoped - # to root-and-below, matching Cmd-P workspace semantics. - if rel.startswith("../"): + rel = os.path.relpath(os.path.join(top, p), root).replace(os.sep, "/") + if rel.startswith("../"): # parents/siblings of cwd: keep Cmd-P workspace scope continue files.append(rel) if len(files) >= _FUZZY_CACHE_MAX_FILES: @@ -108,16 +61,11 @@ def _list_repo_files(root: str) -> list[str]: pass if not files: - # Fallback walk: skip vendor/build dirs + dot-dirs so the walk stays - # tractable. Dotfiles themselves survive — the ranker decides based - # on whether the query starts with `.`. + # Fallback walk skips vendor/build dirs + dot-dirs; dotfiles survive (the ranker + # decides based on whether the query starts with `.`). try: for dirpath, dirnames, filenames in os.walk(root, followlinks=False): - dirnames[:] = [ - d - for d in dirnames - if d not in _FUZZY_FALLBACK_EXCLUDES and not d.startswith(".") - ] + dirnames[:] = [d for d in dirnames if d not in _FUZZY_FALLBACK_EXCLUDES and not d.startswith(".")] rel_dir = os.path.relpath(dirpath, root) for f in filenames: rel = f if rel_dir == "." else f"{rel_dir}/{f}" @@ -136,17 +84,9 @@ def _list_repo_files(root: str) -> list[str]: def _fuzzy_basename_rank(name: str, query: str) -> tuple[int, int] | None: - """Rank ``name`` against ``query``; lower is better. Returns None to reject. - - Tiers (kind): - 0 — exact basename - 1 — basename prefix (e.g. `app` → `appChrome.tsx`) - 2 — word-boundary / camelCase hit (e.g. `chrome` → `appChrome.tsx`) - 3 — substring anywhere in basename - 4 — subsequence match (every query char appears in order) - - Secondary key is `len(name)` so shorter names win ties. - """ + """Rank ``name`` against ``query`` as (tier, len(name)); lower wins, None rejects. + Tiers: 0 exact · 1 prefix · 2 word-boundary/camelCase hit (`chrome` → `appChrome.tsx`) + · 3 substring · 4 subsequence (query chars appear in order).""" if not query: return (3, len(name)) @@ -159,8 +99,7 @@ def _fuzzy_basename_rank(name: str, query: str) -> tuple[int, int] | None: if nl.startswith(ql): return (1, len(name)) - # Word-boundary split: `foo-bar_baz.qux` → ["foo","bar","baz","qux"]. - # camelCase split: `appChrome` → ["app","Chrome"]. Cheap approximation; + # Split on -_. and camelCase (`appChrome` → ["app","Chrome"]); cheap approximation, # falls through to substring/subsequence if it misses. parts: list[str] = [] buf = "" @@ -191,13 +130,9 @@ def _fuzzy_basename_rank(name: str, query: str) -> tuple[int, int] | None: def _abs_completion_prefix_exists(path_part: str) -> bool: - """True when ``path_part`` reads sensibly as an absolute path. - - A leading `/` is only meant literally if something is actually there: - the parent directory has to exist, and a partially-typed final segment - has to match at least one of its entries. Used to decide whether - `@/foo` is the absolute `/foo` or shorthand for `foo` under the cwd. - """ + """True when ``path_part`` reads sensibly as an absolute path: the parent dir exists + and a partially-typed final segment matches at least one entry. Decides whether + `@/foo` is the absolute `/foo` or shorthand for `foo` under the cwd.""" expanded = _normalize_completion_path(path_part) parent = os.path.dirname(expanded.rstrip("/")) or "/" tail = os.path.basename(expanded.rstrip("/")) @@ -219,13 +154,18 @@ def _details_completion_item(value: str, meta: str = "") -> dict: return {"text": value, "display": value, "meta": meta} -def _details_root_completion_item( - value: str, meta: str, needs_leading_space: bool -) -> dict: - return _details_completion_item( - f" {value}" if needs_leading_space else value, - meta, - ) +def _details_root_completion_item(value: str, meta: str, needs_leading_space: bool) -> dict: + return _details_completion_item(f" {value}" if needs_leading_space else value, meta) + + +_DETAILS_SECTIONS = ("thinking", "tools", "subagents", "activity") +_DETAILS_MODES = ("hidden", "collapsed", "expanded") + + +def _details_root_meta(candidate: str) -> str: + if candidate in _DETAILS_SECTIONS: + return "section override" + return "cycle global mode" if candidate == "cycle" else "global mode" def _details_completions(text: str) -> list[dict] | None: @@ -241,66 +181,34 @@ def _details_completions(text: str) -> list[dict] | None: body = body[1:] parts = body.split() has_trailing_space = text.endswith(" ") - sections = ("thinking", "tools", "subagents", "activity") - modes = ("hidden", "collapsed", "expanded") + sections, modes = _DETAILS_SECTIONS, _DETAILS_MODES + root_candidates = (*modes, "cycle", *sections) if not body or (len(parts) == 0 and has_trailing_space): - return [ - *[ - _details_root_completion_item( - mode, "global mode", not has_trailing_space - ) - for mode in modes - ], - _details_root_completion_item( - "cycle", "cycle global mode", not has_trailing_space - ), - *[ - _details_root_completion_item( - section, "section override", not has_trailing_space - ) - for section in sections - ], - ] + return [_details_root_completion_item(c, _details_root_meta(c), not has_trailing_space) for c in root_candidates] if len(parts) == 1 and not has_trailing_space: prefix = parts[0].lower() - candidates = [*modes, "cycle", *sections] return [ - _details_completion_item( - candidate, - ( - "section override" - if candidate in sections - else "cycle global mode" if candidate == "cycle" else "global mode" - ), - ) - for candidate in candidates - if candidate.startswith(prefix) and candidate != prefix + _details_completion_item(c, _details_root_meta(c)) + for c in root_candidates + if c.startswith(prefix) and c != prefix ] - if len(parts) == 1 and has_trailing_space and parts[0].lower() in sections: - return [ - *[ - _details_completion_item(mode, f"set {parts[0].lower()}") - for mode in modes - ], - _details_completion_item("reset", f"clear {parts[0].lower()} override"), - ] + section = parts[0].lower() if parts else "" + if section not in sections: + return [] - if len(parts) == 2 and not has_trailing_space and parts[0].lower() in sections: + def section_meta(candidate: str) -> str: + return f"clear {section} override" if candidate == "reset" else f"set {section}" + + if len(parts) == 1 and has_trailing_space: + return [_details_completion_item(c, section_meta(c)) for c in (*modes, "reset")] + + if len(parts) == 2 and not has_trailing_space: prefix = parts[1].lower() return [ - _details_completion_item( - candidate, - ( - f"clear {parts[0].lower()} override" - if candidate == "reset" - else f"set {parts[0].lower()}" - ), - ) - for candidate in (*modes, "reset") - if candidate.startswith(prefix) and candidate != prefix + _details_completion_item(c, section_meta(c)) for c in (*modes, "reset") if c.startswith(prefix) and c != prefix ] return [] @@ -313,30 +221,22 @@ def _model_picker_context(agent): ctx = load_picker_context() provider = getattr(agent, "provider", "") if agent else "" base_url = getattr(agent, "base_url", "") if agent else "" + model = getattr(agent, "model", "") if agent else "" if str(provider or "").strip().lower() == "custom": try: from hermes_cli.runtime_provider import canonical_custom_identity provider = ( canonical_custom_identity( - base_url=base_url or None, - config_provider=ctx.current_provider, - model=(getattr(agent, "model", "") if agent else "") - or None, + base_url=base_url or None, config_provider=ctx.current_provider, model=model or None ) or provider ) except Exception: - logger.debug( - "custom provider identity recovery failed (model picker)", - exc_info=True, - ) + logger.debug("custom provider identity recovery failed (model picker)", exc_info=True) return ctx.with_overrides( - current_provider=provider, - current_model=(getattr(agent, "model", "") if agent else "") - or _resolve_model(), - current_base_url=base_url, + current_provider=provider, current_model=model or _resolve_model(), current_base_url=base_url ) diff --git a/tui_gateway/methods_config_set.py b/tui_gateway/methods_config_set.py index 0cdb68c5d9..aa1b316c92 100644 --- a/tui_gateway/methods_config_set.py +++ b/tui_gateway/methods_config_set.py @@ -1,13 +1,8 @@ """``config.set`` — one JSON-RPC method, dispatched on ``key`` through a table. -Handlers are rebound onto server.py's globals at install time (see -method_ctx.bind_module), so bodies reference server.py globals bare -(``_ok``, ``_err``, ``_load_cfg``, ``_sessions``, ``_write_config_key``, ...). - -Each ``_set_*`` handler takes ``(rid, params, key, value, session)`` and returns -the JSON-RPC envelope. Table order does not matter: keys are exact matches -except ``details_mode.
`` (prefix) and ``_DISPLAY_TOGGLE_KEYS`` (set), -which are tried after the exact table, in that order. +Bodies are rebound onto server.py's globals (method_ctx.bind_module) and reference them bare. +Each ``_set_*`` handler takes ``(rid, params, key, value, session)`` and returns the JSON-RPC +envelope. Keys match exactly except ``details_mode.
`` (prefix) and ``_DISPLAY_TOGGLE_KEYS``. """ import os @@ -29,18 +24,17 @@ def _display_section(cfg: dict) -> dict: return display if isinstance(display, dict) else {} -def _write_display_sections(*, thinking=None, **display_fields) -> None: - """Persist ``display.`` values and optionally ``display.sections.thinking``. - - Write-back round-trip through the raw (uncached) config so other keys survive. - """ +def _write_display_sections(*, sections=None, drop_sections=(), **display_fields) -> None: + """Persist ``display.`` + ``display.sections`` edits via the raw (uncached) config write-back.""" cfg = _load_cfg_raw() display = _display_section(cfg) - sections = display.get("sections") if isinstance(display.get("sections"), dict) else {} + cur = display.get("sections") + cur = cur if isinstance(cur, dict) else {} display.update(display_fields) - if thinking is not None: - sections["thinking"] = thinking - display["sections"] = sections + cur.update(sections or {}) + for name in drop_sections: + cur.pop(name, None) + display["sections"] = cur cfg["display"] = display _save_cfg(cfg) @@ -72,6 +66,21 @@ def _toggle_display_bool(rid, key, value, *, cfg_key, on_words, off_words): return _ok(rid, {"key": key, "value": "on" if nv_b else "off"}) +def _cfgset_await_agent(session, rid): + """Wait for an in-progress agent build; the error envelope if it failed, else None.""" + init_err = _wait_agent(session, rid) + if init_err: + return init_err + if session.get("agent") is None: + return _err(rid, 5032, "agent initialization failed") + return None + + +def _cfgset_model_ok(rid, key, value, warning, confirm_required, confirm_message, scope, **extra): + return _ok(rid, {"key": key, "value": value, "warning": warning, "confirm_required": confirm_required, + "confirm_message": confirm_message, "scope": scope, **extra}) + + # ── per-key handlers ────────────────────────────────────────────────── @@ -80,165 +89,74 @@ def _set_model(rid, params, key, value, session): try: if not value: return _err(rid, 4002, "model value required") + confirmed = bool(params.get("confirm_expensive_model", False)) if session: from hermes_cli.model_switch import parse_model_switch_args - # A live swap can't run in-place while a turn streams: - # agent.switch_model() mutates self.model / self.provider / - # self.base_url / self.client, and the worker thread running - # agent.run_conversation reads those every iteration — a - # mid-turn swap can fire an HTTP request with the new base_url - # but old model (400/404s). So instead of rejecting the pick - # (the old 4009), stash it and apply it at the NEXT turn start - # (_apply_pending_model_switch), where nothing is in flight. - # The user gets to pick, keep typing, and send the next turn on - # the new model without waiting for the swap or interrupting. + sid = params.get("session_id", "") + # No live swap while a turn streams: agent.switch_model() mutates model/provider/ + # base_url/client that the worker thread reads every iteration. Stash the pick and + # apply it at the NEXT turn start (_apply_pending_model_switch). if session.get("running"): parsed = parse_model_switch_args(value) try: pending_model = parsed.model_input except Exception: pending_model = str(value) - pending_provider = ( - getattr(parsed, "explicit_provider", "") or "" - ).strip() - confirmed = bool(params.get("confirm_expensive_model", False)) - # Run the selection guards HERE, not only at apply time. - # This branch used to answer confirm_required=False without - # consulting them, so a client that implements the confirm - # round-trip was told no consent was needed. It stashed the - # pick, and _apply_pending_model_switch -- which calls the - # guards with the stashed (unconfirmed) flag -- dropped the - # switch at the next turn start. The model reverted with no - # confirm ever offered, because the one moment a round-trip - # was possible had already passed. + pending_provider = (getattr(parsed, "explicit_provider", "") or "").strip() + # Selection guards run HERE (the only moment a confirm round-trip is possible); + # otherwise an unconfirmed stashed pick is dropped at turn start, never confirmed. if not confirmed: - pending_warning = _pending_switch_selection_warning( - pending_model, pending_provider - ) + pending_warning = _pending_switch_selection_warning(pending_model, pending_provider) if pending_warning is not None: - # Nothing is stashed: an unconfirmed guarded pick - # leaves the session exactly as it was, and the - # client re-sends with confirm_expensive_model to - # queue it for real. - return _ok( - rid, - { - "key": key, - "value": pending_model, - # `confirm_message` is the field to read. - # `warning` carries the same text only so - # clients written before the confirm - # round-trip existed still show something; - # `_apply_pending_model_switch` already - # prefers confirm_message and falls back to - # warning. Keep them identical or drop - # `warning` -- do not let them diverge. - "warning": pending_warning, - "confirm_required": True, - "confirm_message": pending_warning, - "scope": "session", - "deferred": False, - }, + # Nothing stashed; the client re-sends with confirm_expensive_model. + # `confirm_message` is canonical, `warning` its legacy alias — identical. + return _cfgset_model_ok( + rid, key, pending_model, pending_warning, True, pending_warning, "session", deferred=False ) session["pending_model_switch"] = { "raw": value, "confirm_expensive_model": confirmed, - # The resolved model/provider the next turn will run on. - # _session_info reports these while the switch is pending - # so the end-of-turn settle keeps showing the user's pick - # instead of blipping back to the still-live old model. + # _session_info reports these while pending so the end-of-turn settle keeps + # showing the user's pick instead of the still-live old model. "display_model": pending_model, "display_provider": pending_provider, } - return _ok( - rid, - { - "key": key, - "value": pending_model, - "warning": "", - "confirm_required": False, - "confirm_message": "", - "scope": "session", - "deferred": True, - }, - ) + return _cfgset_model_ok(rid, key, pending_model, "", False, "", "session", deferred=True) parsed_flags = parse_model_switch_args(value) explicit_provider = parsed_flags.explicit_provider - failed_agent_init = ( - session.get("agent") is None - and session.get("agent_error") is not None - ) + failed_agent_init = session.get("agent") is None and session.get("agent_error") is not None failed_ready = session.get("agent_ready") if failed_agent_init else None if failed_agent_init: if failed_ready is None: - return _err( - rid, - 5032, - session.get("agent_error") - or "agent initialization failed", - ) + return _err(rid, 5032, session.get("agent_error") or "agent initialization failed") if not failed_ready.wait(timeout=30.0): return _err(rid, 5032, "agent initialization timed out") failed_agent_init = ( - failed_agent_init - and session.get("agent") is None - and session.get("agent_error") is not None - and session.get("agent_ready") is failed_ready - and failed_ready.is_set() + failed_agent_init and session.get("agent") is None and session.get("agent_error") is not None + and session.get("agent_ready") is failed_ready and failed_ready.is_set() ) - if ( - session.get("agent") is None - and not explicit_provider.strip() - and not failed_agent_init - ): - session_id = params.get("session_id", "") - _start_agent_build(session_id, session) - init_err = _wait_agent(session, rid) + if session.get("agent") is None and not explicit_provider.strip() and not failed_agent_init: + _start_agent_build(sid, session) + init_err = _cfgset_await_agent(session, rid) if init_err: return init_err - if session.get("agent") is None: - return _err(rid, 5032, "agent initialization failed") with _session_profile_runtime_scope(session): result = _apply_model_switch( - params.get("session_id", ""), - session, - value, - confirm_expensive_model=bool( - params.get("confirm_expensive_model", False) - ), - parsed_flags=parsed_flags, + sid, session, value, confirm_expensive_model=confirmed, parsed_flags=parsed_flags ) if failed_agent_init and not result.get("confirm_required"): - _restart_completed_failed_agent_build( - params.get("session_id", ""), session, failed_ready - ) - init_err = _wait_agent(session, rid) + _restart_completed_failed_agent_build(sid, session, failed_ready) + init_err = _cfgset_await_agent(session, rid) if init_err: return init_err - if session.get("agent") is None: - return _err(rid, 5032, "agent initialization failed") with _session_profile_runtime_scope(session): _persist_live_session_runtime(session) else: - result = _apply_model_switch( - "", - {"agent": None}, - value, - confirm_expensive_model=bool( - params.get("confirm_expensive_model", False) - ), - ) - return _ok( - rid, - { - "key": key, - "value": result["value"], - "warning": result["warning"], - "confirm_required": result.get("confirm_required", False), - "confirm_message": result.get("confirm_message", ""), - "scope": result.get("scope", "session"), - }, + result = _apply_model_switch("", {"agent": None}, value, confirm_expensive_model=confirmed) + return _cfgset_model_ok( + rid, key, result["value"], result["warning"], result.get("confirm_required", False), + result.get("confirm_message", ""), result.get("scope", "session"), ) except Exception as e: return _err(rid, 5001, str(e)) @@ -250,20 +168,14 @@ def _set_fast(rid, params, key, value, session): if agent is not None: current_tier = getattr(agent, "service_tier", None) elif session is not None and session.get("create_service_tier_override") is not None: - # Pre-build session with a pinned tier (desktop draft pick or an - # earlier session-scoped toggle) — report/toggle from the pin, not - # the global default. + # Pre-build session with a pinned tier: report/toggle from the pin, not the global. current_tier = session["create_service_tier_override"] or None else: current_tier = _load_service_tier() current_fast = current_tier == "priority" if raw in {"status"}: - return _ok( - rid, - {"key": key, "value": {"priority": "fast", None: "normal"}.get(current_tier, current_tier)}, - ) - + return _ok(rid, {"key": key, "value": {"priority": "fast", None: "normal"}.get(current_tier, current_tier)}) if raw in {"", "toggle"}: nv = "normal" if current_fast else "fast" elif raw in {"fast", "on"}: @@ -282,42 +194,21 @@ def _set_fast(rid, params, key, value, session): if agent is not None: target_model = getattr(agent, "model", None) else: - # A pre-build session may already have a picked model riding in - # model_override (desktop draft) — validate fast support against - # THAT model, not the global default it will never use. + # A pre-build session may carry a picked model (desktop draft) — validate against THAT. session_override = (session or {}).get("model_override") or {} - target_model = ( - session_override.get("model") - if isinstance(session_override, dict) - else None - ) or _resolve_model() + target_model = (isinstance(session_override, dict) and session_override.get("model")) or _resolve_model() if not target_model: - return _err( - rid, - 4002, - "fast mode is not available without a selected model", - ) + return _err(rid, 4002, "fast mode is not available without a selected model") overrides = resolve_fast_mode_overrides( - target_model, - provider=getattr(agent, "provider", None), - base_url=getattr(agent, "base_url", None), + target_model, provider=getattr(agent, "provider", None), base_url=getattr(agent, "base_url", None) ) if overrides is None: - return _err( - rid, - 4002, - "fast mode is not available for this model", - ) + return _err(rid, 4002, "fast mode is not available for this model") if session is not None: - # Session-scoped, like `reasoning` below (global persistence is - # `--global` / Settings → Model territory). Writing config.yaml - # here let every desktop model-menu selection (per-model fast - # preset) rewrite the user's global agent.service_tier — flipping - # fast mode for every OTHER session, profile, CLI, and gateway - # build ("switch one session, switches everywhere"). Pin the - # create override so lazily-built sessions and rebuilds (/new, - # deferred resume) keep the choice; "" pins normal explicitly. + # Session-scoped like `reasoning` (global persistence is `--global` / Settings → Model): + # writing config.yaml here flipped fast mode for every other session/profile/CLI/gateway. + # The create override keeps the choice across lazy builds and rebuilds; "" pins normal. session["create_service_tier_override"] = {"fast": "priority", "normal": ""}.get(nv, nv) else: _write_config_key("agent.service_tier", nv) @@ -330,7 +221,7 @@ def _set_fast(rid, params, key, value, session): current_overrides.update(overrides) agent.request_overrides = current_overrides _persist_live_session_runtime(session) - _emit("session.info", params.get("session_id", ""), _session_info(agent, session)) + _emit_session_info(params.get("session_id", ""), session) return _ok(rid, {"key": key, "value": nv}) @@ -346,20 +237,13 @@ def _set_busy(rid, params, key, value, session): def _set_verbose(rid, params, key, value, session): cycle = ["off", "new", "all", "verbose"] - cur = ( - session.get("tool_progress_mode", _load_tool_progress_mode()) - if session - else _load_tool_progress_mode() - ) + cur = session.get("tool_progress_mode", _load_tool_progress_mode()) if session else _load_tool_progress_mode() if value and value != "cycle": nv = str(value).strip().lower() if nv not in cycle: return _err(rid, 4002, f"unknown verbose mode: {value}") else: - try: - idx = cycle.index(cur) - except ValueError: - idx = 2 + idx = cycle.index(cur) if cur in cycle else 2 nv = cycle[(idx + 1) % len(cycle)] _write_config_key("display.tool_progress", nv) if session: @@ -371,16 +255,9 @@ def _set_verbose(rid, params, key, value, session): def _set_focus(rid, params, key, value, session): - # Focus view — display-only reduced-output mode (/focus). Composes with - # the tool_progress machinery rather than duplicating it: enabling it - # pins tool_progress to "off" (the same value /verbose off uses) after - # stashing the configured mode, and disabling it restores that mode. - # Nothing about the request payload changes. - from hermes_cli.focus_view import ( - FOCUS_TOOL_PROGRESS_MODE, - normalize_tool_progress_mode, - resolve_focus_arg, - ) + # Focus view (/focus): display-only reduced output composed with tool_progress — enabling + # stashes the configured mode and pins tool_progress "off"; disabling restores the stash. + from hermes_cli.focus_view import FOCUS_TOOL_PROGRESS_MODE, normalize_tool_progress_mode, resolve_focus_arg d_f = _display_section(_load_cfg()) cur_focus = bool(d_f.get("focus_view", False)) @@ -388,30 +265,16 @@ def _set_focus(rid, params, key, value, session): if action == "usage": return _err(rid, 4002, f"unknown focus value: {value} (use on|off|status)") if action == "status" or target is None: - return _ok( - rid, - { - "key": key, - "value": "on" if cur_focus else "off", - "tool_progress": _load_tool_progress_mode(), - }, - ) + return _ok(rid, {"key": key, "value": "on" if cur_focus else "off", "tool_progress": _load_tool_progress_mode()}) if target: - saved = normalize_tool_progress_mode( - (d_f.get("focus_saved_tool_progress") or _load_tool_progress_mode()) - if cur_focus - else _load_tool_progress_mode() - ) - _write_config_key("display.focus_saved_tool_progress", saved) + saved = (cur_focus and d_f.get("focus_saved_tool_progress")) or _load_tool_progress_mode() + _write_config_key("display.focus_saved_tool_progress", normalize_tool_progress_mode(saved)) _write_config_key("display.tool_progress", FOCUS_TOOL_PROGRESS_MODE) effective = FOCUS_TOOL_PROGRESS_MODE else: - saved = normalize_tool_progress_mode( - d_f.get("focus_saved_tool_progress") or "all" - ) - _write_config_key("display.tool_progress", saved) - effective = saved + effective = normalize_tool_progress_mode(d_f.get("focus_saved_tool_progress") or "all") + _write_config_key("display.tool_progress", effective) _write_config_key("display.focus_view", bool(target)) if session: @@ -419,144 +282,95 @@ def _set_focus(rid, params, key, value, session): session["tool_progress_mode"] = effective agent_f = session.get("agent") if agent_f is not None: - try: + with contextlib.suppress(Exception): agent_f.tool_progress_mode = effective - except Exception: - pass - return _ok( - rid, - { - "key": key, - "value": "on" if target else "off", - "tool_progress": effective, - }, - ) + return _ok(rid, {"key": key, "value": "on" if target else "off", "tool_progress": effective}) def _set_approval_mode(rid, params, key, value, session): raw = str(value or "").strip().lower() if raw not in _APPROVAL_MODES: - return _err( - rid, - 4002, - f"unknown approval mode: {value}; pick one of manual|smart|off", - ) - + return _err(rid, 4002, f"unknown approval mode: {value}; pick one of manual|smart|off") _write_config_key("approvals.mode", raw) _emit_all_session_info() return _ok(rid, {"key": "approvals.mode", "value": raw}) def _set_yolo(rid, params, key, value, session): - # Approval bypass. Two scopes: - # scope="session" (default) — same as the TUI's Shift+Tab. Toggles - # ONLY this session's _session_yolo flag; never touches global - # config, so CLI / TUI / cron behavior is unaffected. - # scope="global" (Shift+click the zap) — flips the persistent global - # approvals.mode in config.yaml between "off" (bypass on) and - # "manual" (bypass off). This DOES affect every session, the CLI, - # the TUI, and cron, and survives restarts. + # Approval bypass. scope="session" (default; TUI Shift+Tab) toggles ONLY this session's flag. + # scope="global" (Shift+click the zap) flips persistent approvals.mode between "off" (bypass + # on) and "manual" (bypass off) for every surface, surviving restarts. scope = str(params.get("scope") or "session").strip().lower() try: - from tools.approval import ( - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, - ) + from tools.approval import disable_session_yolo, enable_session_yolo, is_session_yolo_enabled raw = str(value or "").strip().lower() def _resolve_toggle(current: bool) -> bool: - if raw in {"1", "on", "true", "yes"}: - return True - if raw in {"0", "off", "false", "no"}: - return False - return not current + return _BOOL_WORDS.get(raw, not current) if scope == "global": from tools.approval import _normalize_approval_mode cfg = _load_cfg() appr = cfg.get("approvals") if isinstance(cfg, dict) else None - if not isinstance(appr, dict): - appr = {} - current = _normalize_approval_mode(appr.get("mode", "manual")) == "off" - enable = _resolve_toggle(current) - # Toggle between full bypass and the default manual gate. We do - # not try to restore a prior "smart"/custom mode — the zap is a - # binary on/off affordance; users with bespoke modes set them in - # config.yaml. + appr = appr if isinstance(appr, dict) else {} + enable = _resolve_toggle(_normalize_approval_mode(appr.get("mode", "manual")) == "off") + # Binary affordance: no restore of a prior "smart"/custom mode (those live in config.yaml). _write_config_key("approvals.mode", "off" if enable else "manual") - nv = "1" if enable else "0" - # Reflect the global flip in every live session's indicator. - _emit_all_session_info() - return _ok(rid, {"key": key, "value": nv, "scope": "global"}) + _emit_all_session_info() # reflect the flip in every live indicator + return _ok(rid, {"key": key, "value": "1" if enable else "0", "scope": "global"}) if session: - current = is_session_yolo_enabled(session["session_key"]) - enable = _resolve_toggle(current) - if enable: - enable_session_yolo(session["session_key"]) - nv = "1" - else: - disable_session_yolo(session["session_key"]) - nv = "0" + skey = session["session_key"] + enable = _resolve_toggle(is_session_yolo_enabled(skey)) + (enable_session_yolo if enable else disable_session_yolo)(skey) _emit_session_info(params.get("session_id", ""), session) else: - current = is_truthy_value(os.environ.get("HERMES_YOLO_MODE")) - enable = _resolve_toggle(current) + enable = _resolve_toggle(is_truthy_value(os.environ.get("HERMES_YOLO_MODE"))) if enable: os.environ["HERMES_YOLO_MODE"] = "1" - nv = "1" else: os.environ.pop("HERMES_YOLO_MODE", None) - nv = "0" - return _ok(rid, {"key": key, "value": nv, "scope": "session"}) + return _ok(rid, {"key": key, "value": "1" if enable else "0", "scope": "session"}) except Exception as e: return _err(rid, 5001, str(e)) +# /reasoning display words: (accepted inputs, reported value, display field, sections.thinking, +# session show_reasoning or None). full/clamp mirror the CLI's reasoning_full toggle; the TUI +# renders thinking as an expand/collapse section and display.reasoning_full is persisted too. +_REASONING_DISPLAY_WORDS = ( + ({"show", "on"}, "show", {"show_reasoning": True}, "expanded", True), + ({"hide", "off"}, "hide", {"show_reasoning": False}, "hidden", False), + ({"full", "all"}, "full", {"reasoning_full": True}, "expanded", None), + ({"clamp", "collapse", "short"}, "clamp", {"reasoning_full": False}, "collapsed", None), +) + + def _set_reasoning(rid, params, key, value, session): try: from hermes_constants import parse_reasoning_effort arg = str(value or "").strip().lower() scope = str(params.get("scope") or "").strip().lower() - global_scope = scope == "global" - if arg in {"show", "on"}: - _write_display_sections(show_reasoning=True, thinking="expanded") - if session: - session["show_reasoning"] = True - return _ok(rid, {"key": key, "value": "show"}) - if arg in {"hide", "off"}: - _write_display_sections(show_reasoning=False, thinking="hidden") - if session: - session["show_reasoning"] = False - return _ok(rid, {"key": key, "value": "hide"}) - - # /reasoning full | clamp — parity with the classic CLI's reasoning_full - # toggle. The TUI renders thinking as an expand/collapse section, so full - # maps to sections.thinking=expanded and clamp to collapsed; - # display.reasoning_full is persisted too so CLI and TUI stay consistent. - if arg in {"full", "all"}: - _write_display_sections(reasoning_full=True, thinking="expanded") - return _ok(rid, {"key": key, "value": "full"}) - if arg in {"clamp", "collapse", "short"}: - _write_display_sections(reasoning_full=False, thinking="collapsed") - return _ok(rid, {"key": key, "value": "clamp"}) + for words, reported, fields, thinking, show in _REASONING_DISPLAY_WORDS: + if arg in words: + _write_display_sections(sections={"thinking": thinking}, **fields) + if show is not None and session: + session["show_reasoning"] = show + return _ok(rid, {"key": key, "value": reported}) parsed = parse_reasoning_effort(arg) if parsed is None: return _err(rid, 4002, f"unknown reasoning value: {value}") - if global_scope or session is None: + if scope == "global" or session is None: _write_config_key("agent.reasoning_effort", arg) if session is not None: session.pop("create_reasoning_override", None) else: - # Session-scoped, like the messaging gateway's `/reasoning ` - # (global persistence is `--global` / Settings → Model territory). - # Writing config.yaml here let every desktop model-menu selection - # rewrite the user's global agent.reasoning_effort to the preset default. + # Session-scoped like the messaging gateway's `/reasoning `; otherwise every + # desktop model-menu pick rewrote the global default. session["create_reasoning_override"] = parsed if session and session.get("agent") is not None: session["agent"].reasoning_config = parsed @@ -571,54 +385,33 @@ def _set_details_mode(rid, params, key, value, session): nv = str(value or "").strip().lower() if nv not in _DETAIL_MODES: return _err(rid, 4002, f"unknown details_mode: {value}") - cfg = _load_cfg_raw() # write-back round-trip - display = _display_section(cfg) - sections = display.get("sections") if isinstance(display.get("sections"), dict) else {} - display["details_mode"] = nv - for section in _DETAIL_SECTION_NAMES: - sections[section] = nv - display["sections"] = sections - cfg["display"] = display - _save_cfg(cfg) + _write_display_sections(sections={section: nv for section in _DETAIL_SECTION_NAMES}, details_mode=nv) return _ok(rid, {"key": key, "value": nv}) def _set_details_section(rid, params, key, value, session): - # Per-section override: `details_mode.
` writes to - # `display.sections.
`. Empty value clears the explicit override and - # lets frontend resolution apply built-in section defaults before the global - # details_mode. + # `details_mode.
` -> `display.sections.
`. Empty value clears the explicit + # override so the frontend applies built-in section defaults before the global details_mode. section = key.split(".", 1)[1] if section not in _DETAIL_SECTION_NAMES: return _err(rid, 4002, f"unknown section: {section}") - - cfg = _load_cfg_raw() # write-back round-trip - display = _display_section(cfg) - sections_cfg = display.get("sections") if isinstance(display.get("sections"), dict) else {} - nv = str(value or "").strip().lower() if not nv: - sections_cfg.pop(section, None) + _write_display_sections(drop_sections=(section,)) elif nv not in _DETAIL_MODES: return _err(rid, 4002, f"unknown details_mode: {value}") else: - sections_cfg[section] = nv - display["sections"] = sections_cfg - cfg["display"] = display - _save_cfg(cfg) + _write_display_sections(sections={section: nv}) return _ok(rid, {"key": key, "value": nv}) def _set_thinking_mode(rid, params, key, value, session): nv = str(value or "").strip().lower() - allowed_tm = frozenset({"collapsed", "truncated", "full"}) - if nv not in allowed_tm: + if nv not in {"collapsed", "truncated", "full"}: return _err(rid, 4002, f"unknown thinking_mode: {value}") _write_config_key("display.thinking_mode", nv) # Backward compatibility bridge: keep details_mode aligned. - _write_config_key( - "display.details_mode", "expanded" if nv == "full" else "collapsed" - ) + _write_config_key("display.details_mode", "expanded" if nv == "full" else "collapsed") return _ok(rid, {"key": key, "value": nv}) @@ -633,8 +426,7 @@ def _set_battery(rid, params, key, value, session): def _set_theme(rid, params, key, value, session): - # TUI light/dark mode pin: 'light'/'dark' beat background - # auto-detection (xterm.js hosts misreport OSC 11); 'auto' trusts it. + # 'light'/'dark' pin beats background auto-detection (xterm.js hosts misreport OSC 11). raw = str(value or "").strip().lower() if raw not in {"auto", "light", "dark"}: return _err(rid, 4002, f"unknown theme value: {value} (use auto|light|dark)") @@ -645,7 +437,6 @@ def _set_theme(rid, params, key, value, session): def _set_statusbar(rid, params, key, value, session): raw = str(value or "").strip().lower() current = _coerce_statusbar(_display_section(_load_cfg()).get("tui_statusbar", "top")) - if raw in {"", "toggle"}: nv = "top" if current == "off" else "off" elif raw == "on": @@ -654,42 +445,30 @@ def _set_statusbar(rid, params, key, value, session): nv = raw else: return _err(rid, 4002, f"unknown statusbar value: {value}") - _write_config_key("display.tui_statusbar", nv) return _ok(rid, {"key": key, "value": nv}) def _set_mouse(rid, params, key, value, session): - # Explicit None check rather than `value or ""` so falsy non-string - # inputs (0, False) reach the alias map as themselves — both map to - # 'off' via _MOUSE_TRACKING_ALIASES — instead of being collapsed to - # '' and triggering the toggle path. The slash command always passes - # a string, but programmatic JSON-RPC callers may send booleans. + # Explicit None check (not `value or ""`) so falsy non-string inputs (0, False from + # programmatic callers) reach the alias map as themselves (-> 'off') instead of toggling. raw = ("" if value is None else str(value)).strip().lower() current = _display_mouse_tracking(_display_section(_load_cfg())) - if raw in {"", "toggle"}: nv = "all" if current == "off" else "off" elif raw in _MOUSE_TRACKING_ALIASES: nv = _MOUSE_TRACKING_ALIASES[raw] else: return _err(rid, 4002, f"unknown mouse value: {value}") - _write_config_key("display.mouse_tracking", nv) return _ok(rid, {"key": key, "value": nv}) def _set_indicator(rid, params, key, value, session): - # Use an explicit None check rather than `value or ""` so falsy - # non-string inputs (0, False, []) still surface as themselves - # in the error message instead of looking like a blank value. + # Explicit None check so falsy non-string inputs (0, False, []) surface in the error message. raw = ("" if value is None else str(value)).strip().lower() if raw not in INDICATOR_STYLES: - return _err( - rid, - 4002, - f"unknown indicator: {raw!r}; pick one of {'|'.join(INDICATOR_STYLES)}", - ) + return _err(rid, 4002, f"unknown indicator: {raw!r}; pick one of {'|'.join(INDICATOR_STYLES)}") _write_config_key("display.tui_status_indicator", raw) return _ok(rid, {"key": key, "value": raw}) @@ -703,50 +482,39 @@ def _set_cwd(rid, params, key, value, session): return _err(rid, 4002, f"working directory does not exist: {raw}") _write_config_key("terminal.cwd", cwd) os.environ["TERMINAL_CWD"] = cwd - return _ok( - rid, - {"key": "terminal.cwd", "value": cwd, "cwd": cwd, "branch": _git_branch_for_cwd(cwd)}, - ) + return _ok(rid, {"key": "terminal.cwd", "value": cwd, "cwd": cwd, "branch": _git_branch_for_cwd(cwd)}) def _set_prompt_like(rid, params, key, value, session): try: cfg = _load_cfg_raw() # write-back round-trip ("prompt" saves cfg) + resp = {"key": key, "value": value} if key == "prompt": if value == "clear": cfg.pop("custom_prompt", None) - nv = "" + resp["value"] = "" else: cfg["custom_prompt"] = value - nv = value _save_cfg(cfg) elif key == "personality": - sid_key = params.get("session_id", "") pname, new_prompt = _validate_personality(str(value or ""), cfg) - # Personality text is an in-session overlay. Persistence goes - # through hermes_cli.personality (single owner) and never - # touches the user-owned global system prompt. + # Personality text is an in-session overlay; persistence goes through + # hermes_cli.personality (single owner), never the user-owned global system prompt. from hermes_cli.personality import persist_personality persist_personality(pname) - nv = str(value or "none") - history_reset, info = _apply_personality_to_session( - sid_key, session, new_prompt, pname - ) - else: - _write_config_key(f"display.{key}", value) - nv = value - if key == "skin": - # Every connected surface repaints, not just the RPC's - # client; then sync the watcher baseline so the poll loop - # doesn't re-broadcast the skin this RPC just applied. - _broadcast_global_event("skin.changed", resolve_skin()) - _note_skin_broadcast() - resp = {"key": key, "value": nv} - if key == "personality": + resp["value"] = str(value or "none") + history_reset, info = _apply_personality_to_session(params.get("session_id", ""), session, new_prompt, pname) resp["history_reset"] = history_reset if info is not None: resp["info"] = info + else: + _write_config_key(f"display.{key}", value) + if key == "skin": + # Every connected surface repaints; then sync the watcher baseline so the poll + # loop doesn't re-broadcast the skin this RPC just applied. + _broadcast_global_event("skin.changed", resolve_skin()) + _note_skin_broadcast() return _ok(rid, resp) except Exception as e: return _err(rid, 5001, str(e)) @@ -763,29 +531,13 @@ def _set_display_toggle(rid, params, key, value, session): # ── dispatch ────────────────────────────────────────────────────────── _CONFIG_SETTERS = { - "model": _set_model, - "fast": _set_fast, - "busy": _set_busy, - "verbose": _set_verbose, - "focus": _set_focus, - "approval_mode": _set_approval_mode, - "approvals.mode": _set_approval_mode, - "yolo": _set_yolo, - "reasoning": _set_reasoning, - "details_mode": _set_details_mode, - "thinking_mode": _set_thinking_mode, - "density": _set_density, - "battery": _set_battery, - "theme": _set_theme, - "statusbar": _set_statusbar, - "mouse": _set_mouse, - "indicator": _set_indicator, - "cwd": _set_cwd, - "terminal.cwd": _set_cwd, - "workdir": _set_cwd, - "prompt": _set_prompt_like, - "personality": _set_prompt_like, - "skin": _set_prompt_like, + "model": _set_model, "fast": _set_fast, "busy": _set_busy, "verbose": _set_verbose, "focus": _set_focus, + "approval_mode": _set_approval_mode, "approvals.mode": _set_approval_mode, "yolo": _set_yolo, + "reasoning": _set_reasoning, "details_mode": _set_details_mode, "thinking_mode": _set_thinking_mode, + "density": _set_density, "battery": _set_battery, "theme": _set_theme, "statusbar": _set_statusbar, + "mouse": _set_mouse, "indicator": _set_indicator, + "cwd": _set_cwd, "terminal.cwd": _set_cwd, "workdir": _set_cwd, + "prompt": _set_prompt_like, "personality": _set_prompt_like, "skin": _set_prompt_like, } diff --git a/tui_gateway/methods_images.py b/tui_gateway/methods_images.py index 5938e0070f..5a68a10fa9 100644 --- a/tui_gateway/methods_images.py +++ b/tui_gateway/methods_images.py @@ -1,79 +1,68 @@ -"""Image-generation JSON-RPC handler (ws twin of the image_generate tool). +"""Image-generation JSON-RPC handler (ws twin of the image_generate tool), so UI +surfaces (avatar pickers, artifact panes) can generate directly. The result is +returned as a data URL: a remote desktop can't read a gateway file path and hosted +URLs are often CORS-opaque to a renderer canvas. -Desktop plugins reach the backend only through ws JSON-RPC; the image -generation capability existed solely as a model tool. ``image.generate`` -lets UI surfaces (avatar pickers, artifact panes) generate directly. - -The result image is returned as a data URL (``image_data``): a remote -desktop client cannot read a file path on the gateway host, and hosted -result URLs are often CORS-opaque to a renderer canvas. Data URLs work -identically over local and remote gateways. - -Handlers are rebound onto server.py's globals at install time (see -method_ctx.py) — helpers must stay nested inside the handler body. +Bodies are rebound onto server.py's globals at install time (see +method_ctx.bind_module), so they reference server.py globals bare. """ -from .method_ctx import HandlerRegistry +from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() method = _registry.method +def _image_gen_available() -> bool: + try: + from tools.image_generation_tool import check_image_generation_requirements + + return bool(check_image_generation_requirements()) + except Exception: + return False + + +def _image_to_data_url(ref: str, cap: int): + """Fetch a URL or read a local path into a data URL; None when missing, over *cap*, or failing.""" + import base64 + import mimetypes + import os + + try: + if ref.startswith(("http://", "https://")): + import urllib.request + + req = urllib.request.Request(ref, headers={"User-Agent": "hermes-agent"}) + with urllib.request.urlopen(req, timeout=60) as resp: + if resp.length is not None and resp.length > cap: + return None + data = resp.read(cap + 1) + mime = resp.headers.get_content_type() or "image/png" + elif os.path.isfile(ref): + if os.path.getsize(ref) > cap: + return None + with open(ref, "rb") as fh: + data = fh.read(cap + 1) + mime = mimetypes.guess_type(ref)[0] or "image/png" + else: + return None + if len(data) > cap: + return None + if not mime.startswith("image/"): + mime = "image/png" + return f"data:{mime};base64,{base64.b64encode(data).decode('ascii')}" + except Exception: + return None + + @method("image.generate") def _(rid, params: dict) -> dict: - """Generate an image with the configured backend. - - Params: ``prompt`` (required unless ``probe``), ``aspect_ratio`` - (landscape|square|portrait), ``probe`` (return availability only), - ``max_bytes`` (cap on the returned data URL payload, default 8MB). - - Result: ``{available, success, image, image_data, error}`` where - ``image`` is the backend's URL/path and ``image_data`` is a data URL - of the downloaded bytes (omitted when the download fails — callers - should fall back to ``image``). - """ - - def _availability() -> bool: - try: - from tools.image_generation_tool import check_image_generation_requirements - - return bool(check_image_generation_requirements()) - except Exception: - return False - - def _to_data_url(ref: str, cap: int): - """Fetch a URL or read a local path into a data URL, size-capped.""" - import base64 - import mimetypes - import os - - try: - if ref.startswith(("http://", "https://")): - import urllib.request - - req = urllib.request.Request(ref, headers={"User-Agent": "hermes-agent"}) - with urllib.request.urlopen(req, timeout=60) as resp: - if resp.length is not None and resp.length > cap: - return None - data = resp.read(cap + 1) - mime = resp.headers.get_content_type() or "image/png" - elif os.path.isfile(ref): - if os.path.getsize(ref) > cap: - return None - with open(ref, "rb") as fh: - data = fh.read(cap + 1) - mime = mimetypes.guess_type(ref)[0] or "image/png" - else: - return None - if len(data) > cap: - return None - if not mime.startswith("image/"): - mime = "image/png" - return f"data:{mime};base64,{base64.b64encode(data).decode('ascii')}" - except Exception: - return None - - available = _availability() + """Params: ``prompt`` (required unless ``probe``), ``aspect_ratio`` + (landscape|square|portrait), ``probe`` (availability only), ``max_bytes`` (cap + on the data URL, default 8MB, max 16MB). Result: ``{available, success, image, + image_data, error}`` — ``image_data`` is omitted when the download fails, so + callers fall back to ``image`` (the backend's URL/path).""" + available = _image_gen_available() if is_truthy_value(params.get("probe", False)): return _ok(rid, {"available": available}) if not available: @@ -99,10 +88,9 @@ def _(rid, params: dict) -> dict: try: from tools.image_generation_tool import _handle_image_generate - # Full provider dispatcher — the same path the model tool takes: - # source-image confinement, plugin-registered providers, managed - # Krea routing, then the in-tree FAL fallback. Calling the FAL leaf - # (image_generate_tool) directly here bypassed configured providers. + # Full provider dispatcher — same path as the model tool (source-image + # confinement, plugin providers, managed routing, FAL fallback); calling + # the FAL leaf directly bypassed configured providers. raw = _handle_image_generate({"prompt": prompt, "aspect_ratio": aspect}) result = json.loads(raw) except Exception as e: @@ -120,11 +108,12 @@ def _(rid, params: dict) -> dict: image_ref = str(result.get("image") or "") payload = {"available": True, "success": True, "image": image_ref} - data_url = _to_data_url(image_ref, cap) if image_ref else None + data_url = _image_to_data_url(image_ref, cap) if image_ref else None if data_url: payload["image_data"] = data_url return _ok(rid, payload) def register(server) -> None: - _registry.install(server) + """Publish this module's helpers + handlers onto ``server``, rebound to its globals.""" + bind_module(globals(), server, skip=("_",)) diff --git a/tui_gateway/methods_profiles.py b/tui_gateway/methods_profiles.py index bbfb455c3e..f1c88105c2 100644 --- a/tui_gateway/methods_profiles.py +++ b/tui_gateway/methods_profiles.py @@ -1,14 +1,9 @@ -"""Profile JSON-RPC handlers — the ws twin of the dashboard's /api/profiles. +"""Profile JSON-RPC handlers — the ws twin of the dashboard's /api/profiles (desktop plugins +only have the ws door), on the same `hermes_cli.profiles` primitives. -Desktop plugins reach the backend only through the ws JSON-RPC door, so bot rosters -and profile pickers need profile enumeration/creation/editing here, delegating to -the same `hermes_cli.profiles` primitives the REST endpoints use. - -Every function here is rebound onto server.py's globals at install time -(method_ctx.bind_module): bodies use server.py globals bare (`_ok`, `_err`, `os`, -`json`, `Path`, `is_truthy_value`, `get_hermes_home`, `*_hermes_home_override`, -`_profile_ui_meta_lock`), and module-level names are published onto server.py, so -they must not collide with its own globals. +Bodies are rebound onto server.py's globals (method_ctx.bind_module) and use them bare +(`_ok`, `_err`, `os`, `json`, `Path`, `is_truthy_value`, `get_hermes_home`, ...); module-level +names are published onto server.py, so they must not collide with its globals. """ import contextlib @@ -20,16 +15,24 @@ method = _registry.method # ext -> mime; iteration order is the on-disk lookup order for assets. _ASSET_EXTS = {"png": "image/png", "jpg": "image/jpeg", "webp": "image/webp"} -# session.list's deny-list: sub-agent and kanban dispatcher workers. -_WORKER_SOURCES = frozenset({"kanban", "tool"}) + + +def _profile_handler(name: str, code: int): + """``@method(name)`` whose body's uncaught exception becomes ``_err(rid, code, str(e))``.""" + + def deco(fn): + def handler(rid, params: dict) -> dict: + try: + return fn(rid, params) + except Exception as e: + return _err(rid, code, str(e)) + return method(name)(handler) + return deco def _lazy(module, name): - """Late-bound attribute lookup (the wrapped modules are heavy / cyclic at import time). - - Uses the ``__import__`` builtin: rebound bodies only see server.py globals, so this - module's own imports (e.g. ``importlib``) are NOT available here. - """ + """Late-bound attribute lookup (heavy / cyclic modules). ``__import__`` builtin on purpose: + rebound bodies see only server.py globals, not this module's imports.""" return getattr(__import__(module, fromlist=[name]), name) @@ -69,28 +72,21 @@ def _hermes_home_scope(path): reset_hermes_home_override(token) -def _profile_dir_or_err(rid, name): - """``(profile_dir, None)`` for an existing profile, else ``(None, 4064 error)``.""" - from hermes_cli.profiles import get_profile_dir - profile_dir = Path(get_profile_dir(name)) - if not profile_dir.is_dir(): - return None, _err(rid, 4064, f"profile '{name}' not found") - return profile_dir, None - - def _resolve_profile(rid, params): """``(name, profile_dir, err)`` — err is the 4063 (name required) / 4064 (not found) response.""" name = str(params.get("name") or "").strip() if not name: return name, None, _err(rid, 4063, "name required") - profile_dir, err = _profile_dir_or_err(rid, name) - return name, profile_dir, err + from hermes_cli.profiles import get_profile_dir + profile_dir = Path(get_profile_dir(name)) + if not profile_dir.is_dir(): + return name, None, _err(rid, 4064, f"profile '{name}' not found") + return name, profile_dir, None def _read_profile_yaml(profile_dir) -> dict: """profile.yaml as a mapping; ``{}`` when missing, unparseable, or not a mapping.""" import yaml - meta_path = profile_dir / "profile.yaml" loaded = (yaml.safe_load(meta_path.read_text(encoding="utf-8")) or {}) if meta_path.is_file() else {} return loaded if isinstance(loaded, dict) else {} @@ -102,12 +98,9 @@ def _clean_revisions(raw: dict) -> dict: def _latest_message_preview(db, session_id): - """Excerpt (≤80 chars) of the NEWEST active user/assistant message, or "". - - Messaging-app semantics for rosters (latest exchange), unlike the first-message - preview session lists use for recognition. Agent-delivery prefixes are kept. - Same query shape as ``SessionDB.latest_message_row_id`` — keep them in step. - """ + """≤80-char excerpt of the NEWEST active user/assistant message, or "" (roster semantics: + latest exchange, not the first-message preview). Same query shape as + ``SessionDB.latest_message_row_id`` — keep them in step.""" try: with db._lock: row = db._conn.execute( @@ -126,12 +119,8 @@ def _latest_message_preview(db, session_id): def _open_profile_session_db_readonly(profile_path): - """Read-only attach for roster previews, or None. - - A writable ``SessionDB()`` waits up to 20s for the write lock and runs schema init; - the roster polls every 5s while the live backend holds the writer, which stalled the - RPC past the desktop timeout. ``read_only=True`` takes neither the lock nor the DDL. - """ + """Read-only attach for roster previews, or None. A writable ``SessionDB()`` waits up to + 20s for the write lock + runs DDL; the 5s roster poll stalled past the desktop timeout.""" db_path = Path(profile_path) / "state.db" if not _try(db_path.exists, False): return None @@ -139,11 +128,8 @@ def _open_profile_session_db_readonly(profile_path): def _resurrect_recoverable_canonical(db, profile_path, session_id): - """Un-archive an accidentally archived canonical row, or False. - - Recoverability is judged on the read-only handle first; the write uses a - short-lived writable handle so the 20s lock patience is never paid on the poll. - """ + """Un-archive an accidentally archived canonical row, or False. Recoverability is judged on + the read-only handle; the write uses a short-lived writable handle.""" try: row = db.get_session(session_id) if not row or not row.get("archived"): @@ -163,14 +149,10 @@ def _resurrect_recoverable_canonical(db, profile_path, session_id): def _canonical_session_row(db, profile_path): - """Summary of the profile's canonical "Bot Chat" registry row, or None. - - Identity is the NAME (UNIQUE(title) ⇒ ≤1 row), so preview and click target agree - without a client pointer. Exact lookup: hidden rows resolve (canonical chats are - always hidden); lineages resolve via ``get_compression_tip``, NOT the resume walker - whose unmarked-child fallback can pick an ordinary child. Worker sources count as - absent. ``id`` is the durable registry row, ``resolved_id`` the live tip. - """ + """Summary of the profile's canonical "Bot Chat" row, or None. Identity is the NAME, so + preview and click target agree without a client pointer. Hidden rows resolve; lineages + via ``get_compression_tip`` (NOT the resume walker's unmarked-child fallback); worker + sources count as absent. ``id`` is the registry row, ``resolved_id`` the live tip.""" if db is None: return None try: @@ -178,10 +160,9 @@ def _canonical_session_row(db, profile_path): if not row: return None session_id = str(row.get("id") or "").strip() - if not session_id or (row.get("source") or "").strip().lower() in _WORKER_SOURCES: + if not session_id or _denied_source(row): return None - # Archived usually means the user retired it — report absent — but the - # ws-orphan reaper / older cleanup can archive by accident: resurrect those. + # Archived = retired (absent), except accidental reaper archives: resurrect those. if row.get("archived") and not _resurrect_recoverable_canonical(db, profile_path, session_id): return None tip = _try(lambda: db.get_compression_tip(session_id), None) or session_id @@ -202,22 +183,18 @@ def _canonical_session_row(db, profile_path): def _latest_profile_session_rows(db): - """(newest human-facing session, newest worker session) for a profile. - - The second is the newest DENIED row so rosters can show a profile as working - even though worker sessions never surface in conversation lists (workers - heartbeat ``last_activity_at`` every ≤60s; the client picks a liveness window). - """ + """(newest human-facing session, newest worker session). The worker row lets rosters show + a profile as working (workers heartbeat ``last_activity_at`` every ≤60s).""" if db is None: return None, None try: human = worker = None for s in db.list_sessions_rich(source=None, limit=20, order_by_last_active=True, compact_rows=True): - src = (s.get("source") or "").strip().lower() title = s.get("title") or "" last_active = s.get("last_active") or s.get("started_at") or 0 - if src in _WORKER_SOURCES: + if _denied_source(s): if worker is None: + src = (s.get("source") or "").strip().lower() worker = {"id": s["id"], "source": src, "title": title, "last_active": last_active} continue if human is not None: @@ -251,50 +228,53 @@ def _profile_session_fields(row, profile_path): _best_effort(db.close) -@method("profiles.list") +def _profile_ui_meta_fields(row: dict, profile_dir) -> None: + """Attach ``ui_meta`` / ``ui_meta_revisions`` / ``has_avatar`` from profile.yaml + assets. + + Client-agnostic UI metadata lives in profile.yaml so every client paints the + same roster. ``ui_meta_revisions`` is always present: it feature-detects + gateway-owned CAS even for a brand-new profile. + """ + row["ui_meta_revisions"] = {} + raw_meta = _try(lambda: _read_profile_yaml(profile_dir), {}) + ui_meta = raw_meta.get("ui_meta") + if isinstance(ui_meta, dict) and ui_meta: + row["ui_meta"] = ui_meta + revisions = raw_meta.get("_ui_meta_revisions") + if isinstance(revisions, dict) and revisions: + row["ui_meta_revisions"] = _try(lambda: _clean_revisions(revisions), {}) + # Cheap existence flag so rosters skip a get_asset probe per paint. + row["has_avatar"] = _try(lambda: any((profile_dir / "assets" / f"avatar.{e}").is_file() for e in _ASSET_EXTS), False) + + +@_profile_handler("profiles.list", 5061) def _(rid, params: dict) -> dict: """List Hermes profiles (name, path, model, description, skill count). ``include_sessions`` (default true) adds ``last_session`` / ``worker_session`` / ``canonical_session`` so a roster paints per-agent previews without N calls. """ - try: - from hermes_cli.profiles import list_profiles - include_sessions = is_truthy_value(params.get("include_sessions", True)) - out = [] - for p in list_profiles(): - row = { - "name": p.name, - "path": str(p.path), - "is_default": bool(p.is_default), - "model": p.model, - "provider": p.provider, - "description": getattr(p, "description", "") or "", - "display_name": getattr(p, "display_name", "") or "", - "skill_count": getattr(p, "skill_count", 0) or 0, - } - if include_sessions: - _profile_session_fields(row, p.path) - # Client-agnostic UI metadata lives in profile.yaml so every client paints - # the same roster. ``ui_meta_revisions`` is always present: it - # feature-detects gateway-owned CAS even for a brand-new profile. - profile_dir = Path(str(p.path)) - row["ui_meta_revisions"] = {} - raw_meta = _try(lambda: _read_profile_yaml(profile_dir), {}) - ui_meta = raw_meta.get("ui_meta") - if isinstance(ui_meta, dict) and ui_meta: - row["ui_meta"] = ui_meta - revisions = raw_meta.get("_ui_meta_revisions") - if isinstance(revisions, dict) and revisions: - row["ui_meta_revisions"] = _try(lambda: _clean_revisions(revisions), {}) - # Cheap existence flag so rosters skip a get_asset probe per paint. - row["has_avatar"] = _try(lambda: any((profile_dir / "assets" / f"avatar.{e}").is_file() for e in _ASSET_EXTS), False) - out.append(row) - # Capability flag: this backend injects the Bot Mode teammate-messaging - # protocol into every session, so clients must not append it to SOUL.md. - return _ok(rid, {"profiles": out, "bot_mode_protocol": True}) - except Exception as e: - return _err(rid, 5061, str(e)) + from hermes_cli.profiles import list_profiles + include_sessions = is_truthy_value(params.get("include_sessions", True)) + out = [] + for p in list_profiles(): + row = { + "name": p.name, + "path": str(p.path), + "is_default": bool(p.is_default), + "model": p.model, + "provider": p.provider, + "description": getattr(p, "description", "") or "", + "display_name": getattr(p, "display_name", "") or "", + "skill_count": getattr(p, "skill_count", 0) or 0, + } + if include_sessions: + _profile_session_fields(row, p.path) + _profile_ui_meta_fields(row, Path(str(p.path))) + out.append(row) + # Capability flag: this backend injects the Bot Mode teammate-messaging + # protocol into every session, so clients must not append it to SOUL.md. + return _ok(rid, {"profiles": out, "bot_mode_protocol": True}) def _has_real_env_content(env_path) -> bool: @@ -320,30 +300,21 @@ def _mirror_env(path, launch_home) -> bool: def _mirror_auth(path, launch_home) -> bool: - """Copy the launch auth.json when absent, dropping single-use OAuth grants. - - Skipped under ``share_auth`` so the profile reads token state via the global-root - fallback (refreshes write through): a copy forks token state and the first refresh - in either store strands the other. Static .env keys have no refresh semantics. - """ + """Copy the launch auth.json when absent (skipped under ``share_auth``: a copy forks token + state and the first refresh in either store strands the other).""" src, dst = launch_home / "auth.json", path / "auth.json" if not (src.is_file() and not dst.exists()): return False _copy_secret_file(src, dst) - # Never fork single-use OAuth grants (Anthropic / Codex / xAI): the first profile - # to refresh strands every sibling. API keys stay; OAuth rows are dropped and - # read from the root grant via the pool fallback. + # Drop single-use OAuth grants (first refresh strands every sibling); they read from + # the root grant via the pool fallback. API keys stay. _best_effort(lambda: _lazy("hermes_cli.auth", "strip_cloned_single_use_oauth_grants")(path)) return True def _mirror_voice_sections(path) -> bool: - """Copy voice config (stt/tts/voice) from the launch profile; True if written. - - Dictation/TTS resolve ``stt`` inside the TARGET profile's home, and a fresh - profile has only a ``model`` section, so voice fell back to defaults. Goes - through the canonical loaders under the home override (config-read-guard). - """ + """Copy stt/tts/voice sections from the launch profile (a fresh profile has only ``model``, + so voice fell back to defaults); True if written. Canonical loaders under the home override.""" try: from hermes_cli.config import load_config_readonly, read_user_config_raw, save_config src_cfg = load_config_readonly() or {} @@ -351,9 +322,8 @@ def _mirror_voice_sections(path) -> bool: if not sections: return False with _hermes_home_scope(path): - # Round-trip the RAW file: load_config() merges DEFAULT_CONFIG, making every - # section look present (no-op mirror) and save_config would then persist - # the whole default tree into the fresh profile. + # RAW file: load_config() merges DEFAULT_CONFIG (every section looks present + # and save_config would persist the whole default tree). dst_cfg = read_user_config_raw() or {} missing = {k: v for k, v in sections.items() if k not in dst_cfg} if missing: @@ -365,14 +335,9 @@ def _mirror_voice_sections(path) -> bool: def _inherit_launch_model(path) -> bool: - """Inherit the launch profile's model.provider/default when the new profile has none. - - Gate on the MODEL SECTION being absent, not on config.yaml existing: voice - mirroring legitimately creates the file first, and a file-existence gate - silently skipped inheritance for every non-clone bot. Clones keep theirs. - """ + """Inherit launch model.provider/default when the new profile has none. Gate on the MODEL + SECTION, not config.yaml existing: voice mirroring creates the file first.""" from hermes_cli.config import load_config_readonly, read_user_config_raw - with _hermes_home_scope(path): dst_model = (read_user_config_raw() or {}).get("model") or {} if dst_model.get("provider") and dst_model.get("default"): @@ -385,17 +350,34 @@ def _inherit_launch_model(path) -> bool: return True +def _mirror_launch_credentials(path, params: dict) -> dict: + """Copy launch .env / auth.json / voice sections into a new profile (best-effort per item). + + ``share_auth`` reports ``auth: "shared"`` and skips the auth copy; ``mirror_credentials`` + false skips everything. ``model_inherited`` is filled in by the caller. + """ + mirrored = {"env": False, "auth": False, "model_inherited": False, "voice": False} + share_auth = is_truthy_value(params.get("share_auth", False)) + if share_auth: + mirrored["auth"] = "shared" + if not is_truthy_value(params.get("mirror_credentials", True)): + return mirrored + launch_home = get_hermes_home() + mirrored["env"] = _try(lambda: _mirror_env(path, launch_home), False) + if not share_auth: + mirrored["auth"] = _try(lambda: _mirror_auth(path, launch_home), False) + mirrored["voice"] = _mirror_voice_sections(path) + return mirrored + + @method("profiles.create") def _(rid, params: dict) -> dict: - """Create a profile — the ws twin of POST /api/profiles. + """Create a profile (ws twin of POST /api/profiles). - Params: ``name`` (lowercase slug), ``description``, ``clone_from`` (omitted = - fresh profile with bundled skills), ``clone_all``, ``no_skills``, ``soul``, - ``model`` + ``provider`` (optional pin), ``share_auth``, ``mirror_credentials`` - (default true: copy the launch .env, auth.json and voice sections; inherit its - model when unpinned). Mirroring exists because ``create_profile()`` seeds a - comment-only .env and never copies auth.json, so a headlessly created profile had - NO inference provider and no interactive ``hermes setup`` to recover. + Params: ``name``, ``description``, ``clone_from`` (omitted = fresh + bundled skills), + ``clone_all``, ``no_skills``, ``soul``, ``model`` + ``provider``, ``share_auth``, + ``mirror_credentials`` (default true). Mirroring exists because ``create_profile()`` + seeds a comment-only .env and no auth.json — a headless profile had NO provider. """ name = str(params.get("name") or "").strip() if not name: @@ -405,9 +387,7 @@ def _(rid, params: dict) -> dict: clone_from = str(params.get("clone_from") or "").strip() or None clone_all = is_truthy_value(params.get("clone_all", False)) path = profiles_mod.create_profile( - name=name, - clone_from=clone_from, - clone_all=clone_all, + name=name, clone_from=clone_from, clone_all=clone_all, clone_config=bool(clone_from) and not clone_all, no_skills=is_truthy_value(params.get("no_skills", False)), description=str(params.get("description") or "").strip() or None, @@ -417,37 +397,22 @@ def _(rid, params: dict) -> dict: except Exception as e: return _err(rid, 5062, str(e)) - # Mirror the CLI/REST create flow: bundled skills for fresh profiles, then the - # alias wrapper. Both best-effort. + # CLI/REST create flow: bundled skills for fresh profiles, then the alias wrapper. if not clone_from: _best_effort(lambda: profiles_mod.seed_profile_skills(path, quiet=True)) _best_effort(lambda: profiles_mod.check_alias_collision(name) or profiles_mod.create_wrapper_script(name)) - soul = params.get("soul") soul_written = False if isinstance(soul, str) and soul.strip(): soul_written = _best_effort(lambda: (path / "SOUL.md").write_text(soul, encoding="utf-8")) - - mirrored = {"env": False, "auth": False, "model_inherited": False, "voice": False} - share_auth = is_truthy_value(params.get("share_auth", False)) - if share_auth: - mirrored["auth"] = "shared" - mirror = is_truthy_value(params.get("mirror_credentials", True)) - if mirror: - launch_home = get_hermes_home() - mirrored["env"] = _try(lambda: _mirror_env(path, launch_home), False) - if not share_auth: - mirrored["auth"] = _try(lambda: _mirror_auth(path, launch_home), False) - mirrored["voice"] = _mirror_voice_sections(path) - + mirrored = _mirror_launch_credentials(path, params) model = str(params.get("model") or "").strip() provider = str(params.get("provider") or "").strip() model_set = False if model and provider: model_set = _best_effort(lambda: _pin_profile_model(path, provider, model)) - elif mirror: + elif is_truthy_value(params.get("mirror_credentials", True)): mirrored["model_inherited"] = _try(lambda: _inherit_launch_model(path), False) - return _ok( rid, {"ok": True, "name": name, "path": str(path), "soul_written": soul_written, "model_set": model_set, "mirrored": mirrored}, @@ -455,12 +420,8 @@ def _(rid, params: dict) -> dict: def _describe_toolsets(cfg): - """``(toolsets, pinned_set)`` as the `hermes tools` checklist presents them. - - Configurable universe minus platform-restricted, enablement resolved as the runtime - does. The raw registry leaks internal platform composites and reports everything - "enabled" when the profile has no pin. - """ + """``(toolsets, pinned_set)`` as the `hermes tools` checklist presents them (the raw registry + leaks platform composites and reports everything "enabled" without a pin).""" from hermes_cli.tools_config import _get_effective_configurable_toolsets, _get_platform_tools, _toolset_allowed_for_platform from toolsets import resolve_toolset pinned = (cfg.get("tools") if isinstance(cfg.get("tools"), dict) else {}).get("enabled_toolsets") @@ -472,8 +433,7 @@ def _describe_toolsets(cfg): if not _toolset_allowed_for_platform(ts_name, "cli"): continue enabled = ts_name in pinned_set if pinned_set is not None else ts_name in platform_enabled - # Default-off integrations (a2a, spotify, ...) and the equally opt-in yuanbao - # are noise in a per-profile editor unless already enabled. + # Default-off integrations (+ opt-in yuanbao) are noise unless already enabled. if (ts_name in default_off or ts_name == "yuanbao") and not enabled: continue tool_count = _try(lambda: len(set(resolve_toolset(ts_name))), 0) @@ -503,57 +463,45 @@ def _describe_mcp_servers(cfg): ) -@method("profiles.describe") +@_profile_handler("profiles.describe", 5063) def _(rid, params: dict) -> dict: - """Full configuration snapshot of one profile, for an editor UI. - - Result: ``{name, description, soul, model: {provider, default}, skills: - [{name, enabled}], toolsets: [...], toolsets_pinned, mcp_servers}``. Skill - enablement mirrors the disabled-list model (installed = enabled unless in - ``skills.disabled``). All reads are scoped to the profile via the home override. - """ - try: - name, profile_dir, err = _resolve_profile(rid, params) - if err is not None: - return err - with _hermes_home_scope(profile_dir): - from hermes_cli.config import load_config - from hermes_cli.skills_config import get_disabled_skills - cfg = load_config() or {} - disabled = {s.lower() for s in get_disabled_skills(cfg)} - skills_root = profile_dir / "skills" - installed = [ - {"name": md.parent.name, "enabled": md.parent.name.lower() not in disabled} - for md in (sorted(skills_root.rglob("SKILL.md")) if skills_root.is_dir() else ()) - ] - toolsets_out, pinned_set = _describe_toolsets(cfg) - soul = _read_text_if_file(profile_dir / "SOUL.md") - mcp_out = _describe_mcp_servers(cfg) - model_cfg = cfg.get("model") if isinstance(cfg.get("model"), dict) else {} - meta = _try(lambda: _lazy("hermes_cli.profiles", "read_profile_meta")(profile_dir), {}) - result = { - "name": name, - "description": str(meta.get("description") or ""), - "soul": soul, - "model": {"provider": str(model_cfg.get("provider") or ""), "default": str(model_cfg.get("default") or "")}, - "skills": installed, - "toolsets": toolsets_out, - "toolsets_pinned": pinned_set is not None, - "mcp_servers": mcp_out, - } - return _ok(rid, result) - except Exception as e: - return _err(rid, 5063, str(e)) + """Editor snapshot: ``{name, description, soul, model, skills: [{name, enabled}], toolsets, + toolsets_pinned, mcp_servers}``; installed skills are enabled unless in ``skills.disabled``.""" + name, profile_dir, err = _resolve_profile(rid, params) + if err is not None: + return err + with _hermes_home_scope(profile_dir): + from hermes_cli.config import load_config + from hermes_cli.skills_config import get_disabled_skills + cfg = load_config() or {} + disabled = {s.lower() for s in get_disabled_skills(cfg)} + skills_root = profile_dir / "skills" + installed = [ + {"name": md.parent.name, "enabled": md.parent.name.lower() not in disabled} + for md in (sorted(skills_root.rglob("SKILL.md")) if skills_root.is_dir() else ()) + ] + toolsets_out, pinned_set = _describe_toolsets(cfg) + soul = _read_text_if_file(profile_dir / "SOUL.md") + mcp_out = _describe_mcp_servers(cfg) + model_cfg = cfg.get("model") if isinstance(cfg.get("model"), dict) else {} + meta = _try(lambda: _lazy("hermes_cli.profiles", "read_profile_meta")(profile_dir), {}) + result = { + "name": name, + "description": str(meta.get("description") or ""), + "soul": soul, + "model": {"provider": str(model_cfg.get("provider") or ""), "default": str(model_cfg.get("default") or "")}, + "skills": installed, + "toolsets": toolsets_out, + "toolsets_pinned": pinned_set is not None, + "mcp_servers": mcp_out, + } + return _ok(rid, result) def _configure_ui_meta(profile_dir, params, applied) -> None: - """Merge ``params["ui_meta"]`` key-wise into profile.yaml (None deletes a key). - - Size-capped (64KB) because it rides profiles.list on every roster paint. - ``ui_meta_expected_revisions`` are per-key CAS preconditions; any mismatch rejects - the whole write. Revisions survive deletion so a stale client cannot recreate a - removed key by presenting the initial revision. - """ + """Merge ``params["ui_meta"]`` key-wise into profile.yaml (None deletes). 64KB cap (rides + every roster paint). ``ui_meta_expected_revisions``: per-key CAS, any mismatch rejects the + whole write; revisions survive deletion so a stale client cannot recreate a removed key.""" try: incoming = params["ui_meta"] if len(json.dumps(incoming)) > 65536: @@ -598,13 +546,9 @@ def _configure_ui_meta(profile_dir, params, applied) -> None: def _configure_model(profile_dir, params, applied): - """Apply a ``model`` + ``provider`` pin; returns a confirm message instead of writing. - - Same handshake as ``config.set model``: without ``confirm_expensive_model`` a - guarded (data-policy / expensive) pick answers ``confirm_required`` and writes - NOTHING; the client resends with the flag once confirmed. A misbehaving guard - must never break the save (treated as "no warning"), matching ``_apply_model_switch``. - """ + """Apply a ``model`` + ``provider`` pin, or return a confirm message and write NOTHING (the + ``config.set model`` handshake: client resends with ``confirm_expensive_model``). A failing + guard counts as "no warning", matching ``_apply_model_switch``.""" model = str(params.get("model") or "").strip() provider = str(params.get("provider") or "").strip() confirm_message = None @@ -625,18 +569,12 @@ def _configure_model(profile_dir, params, applied): def _configure_cfg_sections(profile_dir, params, applied) -> None: - """Apply ``disabled_skills`` / ``enabled_toolsets`` / ``enabled_mcp_servers`` (replace semantics). - - An empty ``enabled_toolsets`` clears the pin. ``enabled_mcp_servers`` toggles the - ``disabled`` flag; enabling a server the profile doesn't define copies its - definition from the LAUNCH profile's catalog — unknown names are skipped, never - invented. Server defs are config, not secrets; credentials stay in .env/auth. - """ + """Apply ``disabled_skills`` / ``enabled_toolsets`` / ``enabled_mcp_servers`` (replace + semantics; empty toolsets clears the pin). Enabling an undefined MCP server copies its + definition from the LAUNCH catalog (unknown names skipped); credentials stay in .env/auth.""" want_mcp = isinstance(params.get("enabled_mcp_servers"), list) - # Launch profile's MCP catalog, read BEFORE the home override flips config - # resolution to the target profile. + # Launch catalog read BEFORE the home override flips config resolution. launch_mcp = _try(_launch_mcp_catalog, {}) if want_mcp else {} - with _hermes_home_scope(profile_dir): from hermes_cli.config import load_config, save_config cfg = load_config() or {} @@ -688,42 +626,34 @@ def _save_mcp_toggles(cfg, enabled, launch_mcp, save_config) -> None: save_config(cfg) -@method("profiles.configure") +@_profile_handler("profiles.configure", 5064) def _(rid, params: dict) -> dict: - """Apply configuration changes to a profile (editor Save). - - Params: ``name`` plus any of ``ui_meta`` (+ ``ui_meta_expected_revisions``), ``soul``, - ``description``, ``model`` + ``provider`` (+ ``confirm_expensive_model``), - ``disabled_skills``, ``enabled_toolsets``, ``enabled_mcp_servers``. Sections are - independent and best-effort; ``applied`` reports per-section success. - """ - try: - _name, profile_dir, err = _resolve_profile(rid, params) - if err is not None: - return err - applied = {} - if isinstance(params.get("ui_meta"), dict): - _configure_ui_meta(profile_dir, params, applied) - if isinstance(params.get("soul"), str): - applied["soul"] = _best_effort(lambda: (profile_dir / "SOUL.md").write_text(params["soul"], encoding="utf-8")) - if isinstance(params.get("description"), str): - applied["description"] = _best_effort( - lambda: _lazy("hermes_cli.profiles", "write_profile_meta")( - profile_dir, description=params["description"].strip(), description_auto=False - ) + """Editor Save: ``name`` plus any of ``ui_meta`` (+ ``ui_meta_expected_revisions``), ``soul``, + ``description``, ``model`` + ``provider`` (+ ``confirm_expensive_model``), ``disabled_skills``, + ``enabled_toolsets``, ``enabled_mcp_servers``. Sections are independent; ``applied`` reports each.""" + _name, profile_dir, err = _resolve_profile(rid, params) + if err is not None: + return err + applied = {} + if isinstance(params.get("ui_meta"), dict): + _configure_ui_meta(profile_dir, params, applied) + if isinstance(params.get("soul"), str): + applied["soul"] = _best_effort(lambda: (profile_dir / "SOUL.md").write_text(params["soul"], encoding="utf-8")) + if isinstance(params.get("description"), str): + applied["description"] = _best_effort( + lambda: _lazy("hermes_cli.profiles", "write_profile_meta")( + profile_dir, description=params["description"].strip(), description_auto=False ) - confirm_message = _configure_model(profile_dir, params, applied) - if any(isinstance(params.get(k), list) for k in ("disabled_skills", "enabled_toolsets", "enabled_mcp_servers")): - _configure_cfg_sections(profile_dir, params, applied) - - result = {"ok": all(applied.values()) if applied else True, "applied": applied} - if confirm_message is not None: - # Same shape config.set returns, so clients reuse one confirm handler. - result["confirm_required"] = True - result["confirm_message"] = confirm_message - return _ok(rid, result) - except Exception as e: - return _err(rid, 5064, str(e)) + ) + confirm_message = _configure_model(profile_dir, params, applied) + if any(isinstance(params.get(k), list) for k in ("disabled_skills", "enabled_toolsets", "enabled_mcp_servers")): + _configure_cfg_sections(profile_dir, params, applied) + result = {"ok": all(applied.values()) if applied else True, "applied": applied} + if confirm_message is not None: + # Same shape config.set returns, so clients reuse one confirm handler. + result["confirm_required"] = True + result["confirm_message"] = confirm_message + return _ok(rid, result) def _sniff_asset_ext(blob): @@ -743,73 +673,61 @@ def _unlink_asset_files(assets_dir, asset) -> int: return len(present) -@method("profiles.set_asset") +@_profile_handler("profiles.set_asset", 5065) def _(rid, params: dict) -> dict: - """Store a small binary asset (avatar image) as ``assets/.``, atomically. - - Params: ``name``, ``asset`` (only ``"avatar"``), ``data`` (data URL or raw base64; - PNG/JPEG/WebP; decoded ≤2MB), or ``clear: true``. Result: ``{ok, asset, size}``. - """ + """Store ``assets/.`` atomically. Params: ``name``, ``asset`` (``"avatar"`` only), + ``data`` (data URL or base64; PNG/JPEG/WebP ≤2MB) or ``clear: true``. Result ``{ok, asset, size}``.""" asset = str(params.get("asset") or "avatar").strip().lower() if not str(params.get("name") or "").strip(): return _err(rid, 4063, "name required") if asset != "avatar": return _err(rid, 4066, f"unknown asset '{asset}' (supported: avatar)") + import base64 + import re + _name, profile_dir, err = _resolve_profile(rid, params) + if err is not None: + return err + assets_dir = profile_dir / "assets" + if is_truthy_value(params.get("clear", False)): + removed = _unlink_asset_files(assets_dir, asset) + return _ok(rid, {"ok": True, "asset": asset, "size": 0, "removed": removed}) + data = str(params.get("data") or "") + if not data: + return _err(rid, 4067, "data required (data URL or base64)") + match = re.match(r"^data:(image/(?:png|jpeg|webp));base64,(.*)$", data, re.DOTALL) try: - import base64 - import re - _name, profile_dir, err = _resolve_profile(rid, params) - if err is not None: - return err - assets_dir = profile_dir / "assets" - if is_truthy_value(params.get("clear", False)): - removed = _unlink_asset_files(assets_dir, asset) - return _ok(rid, {"ok": True, "asset": asset, "size": 0, "removed": removed}) - data = str(params.get("data") or "") - if not data: - return _err(rid, 4067, "data required (data URL or base64)") - match = re.match(r"^data:(image/(?:png|jpeg|webp));base64,(.*)$", data, re.DOTALL) - try: - blob = base64.b64decode(match.group(2) if match else data, validate=True) - except Exception: - return _err(rid, 4068, "data is not valid base64") - if len(blob) > 2_000_000: - return _err(rid, 4069, f"asset too large ({len(blob)} bytes; max 2MB)") - ext = _sniff_asset_ext(blob) - if ext is None: - return _err(rid, 4070, "unsupported image format (PNG/JPEG/WebP only)") - assets_dir.mkdir(parents=True, exist_ok=True) - _unlink_asset_files(assets_dir, asset) # one canonical file per asset - target = assets_dir / f"{asset}.{ext}" - tmp = target.with_suffix(target.suffix + ".tmp") - tmp.write_bytes(blob) - tmp.replace(target) - return _ok(rid, {"ok": True, "asset": asset, "size": len(blob)}) - except Exception as e: - return _err(rid, 5065, str(e)) + blob = base64.b64decode(match.group(2) if match else data, validate=True) + except Exception: + return _err(rid, 4068, "data is not valid base64") + if len(blob) > 2_000_000: + return _err(rid, 4069, f"asset too large ({len(blob)} bytes; max 2MB)") + ext = _sniff_asset_ext(blob) + if ext is None: + return _err(rid, 4070, "unsupported image format (PNG/JPEG/WebP only)") + assets_dir.mkdir(parents=True, exist_ok=True) + _unlink_asset_files(assets_dir, asset) # one canonical file per asset + target = assets_dir / f"{asset}.{ext}" + tmp = target.with_suffix(target.suffix + ".tmp") + tmp.write_bytes(blob) + tmp.replace(target) + return _ok(rid, {"ok": True, "asset": asset, "size": len(blob)}) -@method("profiles.get_asset") +@_profile_handler("profiles.get_asset", 5066) def _(rid, params: dict) -> dict: - """Fetch a profile asset as a data URL: ``{found, data?, mime?, size?}``. - - ``found: false`` (not an error) when absent, so rosters can probe cheaply. - """ + """Profile asset as a data URL: ``{found, data?, mime?, size?}``; absent is ``found: false``, not an error.""" asset = str(params.get("asset") or "avatar").strip().lower() - try: - import base64 - _name, profile_dir, err = _resolve_profile(rid, params) - if err is not None: - return err - for ext, mime in _ASSET_EXTS.items(): - target = profile_dir / "assets" / f"{asset}.{ext}" - if target.is_file(): - blob = target.read_bytes() - data = f"data:{mime};base64,{base64.b64encode(blob).decode('ascii')}" - return _ok(rid, {"found": True, "mime": mime, "size": len(blob), "data": data}) - return _ok(rid, {"found": False}) - except Exception as e: - return _err(rid, 5066, str(e)) + import base64 + _name, profile_dir, err = _resolve_profile(rid, params) + if err is not None: + return err + for ext, mime in _ASSET_EXTS.items(): + target = profile_dir / "assets" / f"{asset}.{ext}" + if target.is_file(): + blob = target.read_bytes() + data = f"data:{mime};base64,{base64.b64encode(blob).decode('ascii')}" + return _ok(rid, {"found": True, "mime": mime, "size": len(blob), "data": data}) + return _ok(rid, {"found": False}) def register(server) -> None: diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index 87575549bd..0419bcaa08 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -1,27 +1,26 @@ -"""Prompt / attachment / respond JSON-RPC handlers (moved verbatim from server.py). +"""Prompt / attachment / respond JSON-RPC handlers. -Handler bodies are byte-identical to their pre-split server.py form; they -are rebound onto server.py's globals at install time — see method_ctx.py. +Bodies are rebound onto server.py's globals at install time (see +method_ctx.bind_module), so they reference server.py globals bare. """ -from .method_ctx import HandlerRegistry +import contextlib -import types +from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() method = _registry.method _profile_scoped = _registry.profile_scoped +_STALE_TARGET_MSG = "target user message is no longer in session history" + + def _history_user_indices(history: list) -> list: """Indices of canonical live-user turns, including composite carriers.""" from agent.context_compressor import user_originated_turn_view - return [ - i - for i, m in enumerate(history) - if user_originated_turn_view(m) is not None - ] + return [i for i, m in enumerate(history) if user_originated_turn_view(m) is not None] def _message_row_id(msg: dict): @@ -40,13 +39,10 @@ def _message_row_id(msg: dict): def _mem_db_pair_agrees(mem, db_msg) -> bool: """True when a live-memory entry plausibly corresponds to a durable row. - Positional trust across the live and durable lists needs evidence, not - just equal lengths/ordinals: roles must match, display-marker status must - match (a marker living only on one side shifts every later position), and - an addressable user turn must show the same text. Non-string (multimodal) - content can't be compared cheaply — role/marker agreement suffices there. - Self-contained on builtins: register() rebinds callers onto server - globals, so any helper this calls must be in that namespace too. + Positional trust needs evidence beyond equal lengths: roles must match, + display-marker status must match (a marker on one side only shifts every + later position), and an addressable user turn must show the same text. + Multimodal content can't be compared cheaply — role/marker agreement suffices. """ if not isinstance(mem, dict) or not isinstance(db_msg, dict): return False @@ -61,9 +57,7 @@ def _mem_db_pair_agrees(mem, db_msg) -> bool: if (mem_view is None) != (db_view is None): return False if mem_view is None: - return bool(mem.get("display_kind")) == bool( - db_msg.get("display_kind") - ) + return bool(mem.get("display_kind")) == bool(db_msg.get("display_kind")) mem_content = mem_view.get("content") db_content = db_view.get("content") if isinstance(mem_content, str) and isinstance(db_content, str): @@ -100,14 +94,11 @@ def _load_durable_truncation_history( if not callable(get_conv): return None history = get_conv( - session_key, - repair_alternation=repair_alternation, - include_row_ids=True, + session_key, repair_alternation=repair_alternation, include_row_ids=True, ) except Exception: logger.debug( - "prompt.submit: failed loading durable history for session %s", - session_key, + "prompt.submit: failed loading durable history for session %s", session_key, exc_info=True, ) return None @@ -117,11 +108,10 @@ def _load_durable_truncation_history( def _resolve_truncate_row_id(session: dict, history: list, target_row_id: int): """Resolve ``truncate_before_row_id`` to ``(user_ordinal, history_index)``. - Prefer in-memory ``_row_id`` / ``row_id`` stamps. When a live turn rewrote - ``session["history"]`` without stamps (provider-format messages), load the - session's durable transcript with ``include_row_ids=True`` and map the - matched user-turn ordinal onto the live list. Does **not** fall back to a - client-supplied ordinal — unknown row ids must refuse (#82959). + Prefer in-memory ``_row_id``/``row_id`` stamps. When a live turn rewrote + ``session["history"]`` without stamps, load the durable transcript with + ``include_row_ids=True`` and map the matched user-turn ordinal onto the live + list. Never falls back to a client-supplied ordinal — unknown row ids refuse. """ hit = _find_user_turn_by_row_id(history, target_row_id) if hit is not None: @@ -131,15 +121,11 @@ def _resolve_truncate_row_id(session: dict, history: list, target_row_id: int): if db_history is None: return None - # Heal missing in-memory stamps when the live list still lines up 1:1 with - # the durable transcript (common after turn-completion rewrites). Equal - # length alone is NOT proof of alignment: the durable copy above is loaded - # with repair_alternation=True (which can merge/drop rows) while the live - # list is unrepaired, and memory can carry optimistic/marker rows — so the - # two can coincide in length while position-shifted. A positional stamp on - # a misaligned pair is sticky and re-aims every later rewind at the wrong - # durable row. Stamp only when EVERY pair agrees (all-or-nothing): roles - # must match on every pair, and addressable user turns must match content. + # Heal missing stamps only when EVERY pair agrees (all-or-nothing). Equal + # length alone is not alignment: the durable copy is alternation-repaired + # (may merge/drop rows) while the live list is not and can carry + # optimistic/marker rows; a stamp on a misaligned pair is sticky and + # re-aims every later rewind at the wrong durable row. if len(db_history) == len(history) and all( _mem_db_pair_agrees(mem, db_msg) for mem, db_msg in zip(history, db_history) @@ -160,23 +146,17 @@ def _resolve_truncate_row_id(session: dict, history: list, target_row_id: int): if db_ord < 0 or db_ord >= len(mem_user_indices): return None mem_idx = mem_user_indices[db_ord] - # Same-ordinal mapping across two lists that can diverge (the repaired - # durable copy may have merged a user;user pair, shifting every later - # user ordinal). Trust the mapping only when the mapped live turn shows - # the same content as the durable target — otherwise refuse (the caller - # returns fail-closed 4018) rather than cut the wrong turn (#82959). + # Same-ordinal mapping across lists that can diverge (repair may have merged + # a user;user pair): trust it only when the mapped live turn shows the same + # content as the durable target — else refuse (caller fails closed, 4018). if not _mem_db_pair_agrees(history[mem_idx], db_history[db_idx]): return None return db_ord, mem_idx def _coerce_truncate_int(rid, value, param_name="truncate_before_user_ordinal"): - """Return ``(int_value, error_response)`` for a client-supplied integer param. - - bool is an int subclass: a JSON ``true`` would coerce via int() to - 1 and aim a confirmed rewind at the wrong turn — refuse it like any - other non-integer. - """ + """``(int_value, error_response)`` for a client integer param. bool is refused + like any non-integer: JSON ``true`` would int() to 1 and aim at the wrong turn.""" if isinstance(value, bool): return None, _err(rid, 4004, f"{param_name} must be an integer") try: @@ -191,19 +171,12 @@ def _reconcile_client_ordinal( ): """Cross-check a client ordinal against a resolved durable target. - Returns ``(ordinal, error_response)``: the target's tip-relative ordinal - when the client sent none or agreed, else the 4004/4030 refusal. A stale - ordinal alongside a *resolved* durable id is the #82756 drift class — - refuse rather than guess which address the user meant. - - Desktop/TUI ordinals count the full displayed lineage: after context - compression the client still renders the ancestor turns from - ``display_history_prefix`` while ``msg_ordinal`` is relative to the tip - segment only (#82462). A client ordinal that equals - ``msg_ordinal + prefix_user_count`` is therefore the SAME turn counted in - lineage space, not drift — accept it. The cut itself is always aimed by - the resolved durable target, never by the client ordinal, so this wider - acceptance can never re-aim a truncation. + Returns ``(ordinal, error_response)``: the target's tip-relative ordinal when + the client sent none or agreed, else the 4004/4030 refusal — a stale ordinal + beside a *resolved* durable id is drift; never guess which the user meant. + Client ordinals count the full displayed lineage, so after compression + ``msg_ordinal + prefix_user_count`` is the SAME turn, not drift. The cut is + always aimed by the durable target, so this can never re-aim a truncation. """ if client_ordinal is None: return msg_ordinal, None @@ -235,19 +208,13 @@ def _reconcile_client_ordinal( def _pending_reaction_notes(session: dict) -> str: - """Note block describing reactions the user added since the last turn, or "". - - Applied to the MODEL INPUT only (``run_message``, beside the - speech-interrupted note) — never to the text that gets persisted. Prefixing - the persisted prompt bakes scaffolding into the transcript. Each reaction is - announced once — the row is stamped ``seen`` on read. - """ + """Note block for reactions added since the last turn, or "". Applied to the + MODEL INPUT only, never the persisted prompt; each reaction is announced once + (rows are stamped ``seen`` on read). Feature-gated (display.message_reactions).""" session_key = str(session.get("session_key") or "") if not session_key: return "" - # Feature-gated (off by default, Settings → Appearance): when disabled the - # model hears nothing, even about reactions set while it was on. try: display = _load_cfg().get("display") if not (isinstance(display, dict) and bool(display.get("message_reactions", False))): @@ -277,687 +244,405 @@ def _pending_reaction_notes(session: dict) -> str: if snippet: notes.append(f'[The user reacted {emoji} to {whose} message: "{snippet}"]') else: - # A row with no plain text (attachment-only, or a tool-call-only - # assistant turn) — an empty quote reads worse than no quote. + # Attachment-only / tool-call-only rows: no quote beats an empty quote. notes.append(f"[The user reacted {emoji} to {whose} earlier message]") return "\n".join(notes) -@method("prompt.submit") -def _(rid, params: dict) -> dict: - from hermes_cli.input_sanitize import sanitize_user_prompt_text +# ── prompt.submit pieces ──────────────────────────────────────────────────── - sid = params.get("session_id", "") - raw_text = params.get("text", "") - text = sanitize_user_prompt_text(raw_text) if isinstance(raw_text, str) else raw_text - # Off-screen sends (widget intents): type the persisted user row so no - # client renders it as a bubble. Whitelisted to "hidden" — display_kind - # is a DB-only sidecar and this RPC must not mint arbitrary kinds. - display_kind = "hidden" if params.get("display_kind") == "hidden" else None - # Typed bare stop phrase while backend voice mode is active ends the - # voice chat instead of sending "stop" to the agent — the typed twin of - # the spoken stop phrase (PR #73106), applied at the ONE server-side - # choke point every TUI submit passes through. Guarded on voice mode - # being ON: typed "stop" outside a voice chat is a normal message. - # (The desktop's voice conversation is renderer-owned and never flips - # the backend flag, so it handles its own typed stop client-side.) - if isinstance(text, str) and _voice_mode_enabled(): - try: - from tools.voice_mode import is_voice_stop_phrase - typed_stop = is_voice_stop_phrase(text) - except Exception: - typed_stop = False - if typed_stop: - os.environ["HERMES_VOICE"] = "0" - os.environ["HERMES_VOICE_TTS"] = "0" - try: - from hermes_cli.voice import stop_continuous +def _typed_stop_phrase_response(rid, text): + """End the voice chat when a bare stop phrase is TYPED while backend voice mode + is on (typed twin of the spoken stop phrase, at the one server-side choke + point). Returns the RPC reply, or None when this is a normal message. The + desktop's renderer-owned voice chat never flips the backend flag and handles + its own typed stop.""" + if not (isinstance(text, str) and _voice_mode_enabled()): + return None + try: + from tools.voice_mode import is_voice_stop_phrase - stop_continuous() - except Exception: - pass - try: - _tts_stream_stop(user_barge=False) - except Exception: - pass - _voice_emit("voice.transcript", {"stop_phrase": True, "typed": True}) - logger.info("prompt.submit: typed stop phrase — voice chat ended") - return _ok(rid, {"voice_stopped": True}) + typed_stop = is_voice_stop_phrase(text) + except Exception: + typed_stop = False + if not typed_stop: + return None + _end_voice_chat(stop_loop=True, stop_tts=True) + _voice_emit("voice.transcript", {"stop_phrase": True, "typed": True}) + logger.info("prompt.submit: typed stop phrase — voice chat ended") + return _ok(rid, {"voice_stopped": True}) + + +_HOSTED_TASK_FIELDS = {"room_id", "task_id", "thread_id", "turn_id", "execution_generation"} + + +def _hosted_submit_error(rid, session, hosted_task, hosted_terminal_callback): + """Validate the hosted-room turn proof carried by an internal submit.""" + if session.get("source") != "bot_room": + return _err(rid, 4120, "hosted room turns require a bot_room session") + if not isinstance(hosted_task, dict) or not callable(hosted_terminal_callback): + return _err(rid, 4120, "invalid hosted room turn proof") + if set(hosted_task) != _HOSTED_TASK_FIELDS or not all( + isinstance(hosted_task.get(field), str) and hosted_task[field] + for field in _HOSTED_TASK_FIELDS - {"execution_generation"} + ) or not isinstance(hosted_task.get("execution_generation"), int): + return _err(rid, 4120, "invalid hosted room turn proof") + return None + + +def _legacy_group_fence_error(rid, session, params): + """Older Desktop builds know the ``Group: `` title but not the hosted + authority marker; once a gateway owns that room a direct prompt would start a + second renderer driver. Fence server-side instead of trusting the client.""" + title = str(session.get("title") or "") + if not title.startswith("Group: "): + return None + room_id = title.removeprefix("Group: ").strip() + if not room_id: + return None + try: + from gateway.hosted_rooms import ( + HostedRoomError, + RoomProbeUnavailableError, + default_db_path, + probe_hosted_room, + probe_peer_room_reservation, + ) + + hosted = probe_hosted_room(default_db_path(), room_id=room_id) + peer = False + if not hosted: + from hermes_constants import named_profile_home + + session_profile_home = named_profile_home(str(session.get("profile_home") or "")) + requested_profile = ( + ( + session_profile_home.name + if session_profile_home is not None + else "" + ) + or str(params.get("profile") or "").strip() + or str(_current_profile_name() or "default").strip() + ) + peer = probe_peer_room_reservation( + default_db_path(), room_id=room_id, target_profile=requested_profile, + ) + except RoomProbeUnavailableError: + return _err(rid, 5122, "Could not verify this group. Try again after the gateway recovers.") + except HostedRoomError: + # Legacy Desktop sessions used the display name after "Group: "; those + # names are not hosted room ids. + return None + except Exception: + return _err(rid, 5122, "Could not verify this group. Try again after the gateway recovers.") + if hosted or peer: + return _err( + rid, + 4122, + ( + "This room is managed by its gateway. " + if hosted + else "This room is managed by its home host. " + ) + + "Update Hermes Desktop to continue it.", + ) + return None + + +def _resolve_truncation_ordinal(rid, sid, session, params, history): + """Resolve the truncation target to ``(ordinal, cut_index, err)``. + + Refusal precedence: malformed params (4004) → unconfirmed (4029, checked + BEFORE target resolution so a leaked-state request never pays the durable + read or heal-stamps live dicts) → unresolvable target (4018, fail closed — + never degrade a missing row_id/message_id into an ordinal cut) → ordinal + drift (4030) → ordinal-only on a durable session (4004). + """ truncate_user_ordinal = params.get("truncate_before_user_ordinal") - if params.get("interrupted"): - # Client-side barge-in (desktop VAD / typing over playback) — latch it - # so this turn's model message carries the interruption note. - from tools.tts_streaming import mark_speech_interrupted + truncate_message_id = params.get("truncate_before_message_id") + truncate_row_id = params.get("truncate_before_row_id") - mark_speech_interrupted() - session, err = _sess_nowait(params, rid) - if err: - return err - hosted_task = params.get("_hosted_task") - hosted_terminal_callback = params.get("_hosted_terminal_callback") - internal_hosted_submit = hosted_task is not None or hosted_terminal_callback is not None - if internal_hosted_submit: - if session.get("source") != "bot_room": - return _err(rid, 4120, "hosted room turns require a bot_room session") - if not isinstance(hosted_task, dict) or not callable(hosted_terminal_callback): - return _err(rid, 4120, "invalid hosted room turn proof") - required_hosted_fields = { - "room_id", - "task_id", - "thread_id", - "turn_id", - "execution_generation", - } - if set(hosted_task) != required_hosted_fields or not all( - isinstance(hosted_task.get(field), str) and hosted_task[field] - for field in required_hosted_fields - {"execution_generation"} - ) or not isinstance(hosted_task.get("execution_generation"), int): - return _err(rid, 4120, "invalid hosted room turn proof") + target_row_id = None + if truncate_row_id is not None: + target_row_id, err = _coerce_truncate_int(rid, truncate_row_id, "truncate_before_row_id") + if err is not None: + return None, None, err + client_ordinal = None + if truncate_user_ordinal is not None: + client_ordinal, err = _coerce_truncate_int(rid, truncate_user_ordinal) + if err is not None: + return None, None, err + + # An ordinal/id alone is not consent: a leftover ordinal on an ORDINARY + # submit is field-for-field indistinguishable from a real rewind, and the + # cut is a destructive replace_messages(). Only the client knows. + if not is_truthy_value(params.get("confirm_truncate")): + logger.warning( + "prompt.submit: REFUSED unconfirmed truncation of session %s " + "(%d messages held; ordinal=%s, row_id=%s, message_id=%s). " + "The client attached truncation parameters without " + "confirm_truncate — likely stale truncation parameters on " + "an ordinary submit.", + sid, + len(history), + client_ordinal, + target_row_id, + truncate_message_id, + ) + return None, None, _err( + rid, + 4029, + "truncation parameters require confirm_truncate=true; " + "an ordinary prompt.submit must not drop session history " + "(update your Hermes client if a rewind was intended)", + ) + # Client ordinals count the full displayed lineage; after compression the + # tip segment is session["history"] and the ancestors live in + # display_history_prefix. Count the ancestor user turns once so client and + # tip-relative ordinals can translate without loading ancestors into the tip. + prefix_user_count = len(_history_user_indices(session.get("display_history_prefix") or [])) + user_indices = _history_user_indices(history) + + def _stale(resolved_ordinal=None): + # Structured recovery fields: Desktop resyncs + retries on a stale target + # and shows "compressed away" when segment_ordinal < 0 (ancestor-only). + segment = ( + client_ordinal - prefix_user_count if client_ordinal is not None else resolved_ordinal + ) + return None, None, _err(rid, 4018, _STALE_TARGET_MSG, data={ + "user_turn_count": len(user_indices), "ordinal": client_ordinal, + "segment_ordinal": segment, "prefix_user_count": prefix_user_count, + }) + + if target_row_id is not None: + found_match = _resolve_truncate_row_id(session, history, target_row_id) + if found_match is None: + logger.warning( + "prompt.submit: target row_id %d not found for session %s " + "(in-memory + durable); refusing truncation without fallback", + target_row_id, + sid, + ) + return _stale() + ordinal, err = _reconcile_client_ordinal( + rid, sid, client_ordinal, found_match[0], "truncate_before_row_id", target_row_id, + prefix_user_count=prefix_user_count, + ) + if err is not None: + return None, None, err + elif truncate_message_id is not None: + msg_id_str = str(truncate_message_id) + found_match = next( + ( + (u_ord, h_idx) + for u_ord, h_idx in enumerate(user_indices) + if history[h_idx].get("id") == msg_id_str + or history[h_idx].get("message_id") == msg_id_str + ), + None, + ) + if found_match is None: + logger.warning( + "prompt.submit: target message_id %s not found in history " + "for session %s; refusing truncation without fallback", + msg_id_str, + sid, + ) + return _stale() + ordinal, err = _reconcile_client_ordinal( + rid, sid, client_ordinal, found_match[0], "truncate_before_message_id", msg_id_str, + prefix_user_count=prefix_user_count, + ) + if err is not None: + return None, None, err else: - # Older Desktop builds know the `Group: ` session title but - # not the hosted authority marker. Once a gateway owns that room, a - # direct prompt into its member session would start a second renderer - # driver. Fence it server-side instead of trusting client awareness. - title = str(session.get("title") or "") - if title.startswith("Group: "): - room_id = title.removeprefix("Group: ").strip() - if room_id: - try: - from gateway.hosted_rooms import ( - HostedRoomError, - RoomProbeUnavailableError, - default_db_path, - probe_hosted_room, - probe_peer_room_reservation, - ) - - hosted = probe_hosted_room(default_db_path(), room_id=room_id) - peer = False - if not hosted: - from hermes_constants import named_profile_home - - session_profile_home = named_profile_home( - str(session.get("profile_home") or "") - ) - requested_profile = ( - ( - session_profile_home.name - if session_profile_home is not None - else "" - ) - or str(params.get("profile") or "").strip() - or str(_current_profile_name() or "default").strip() - ) - peer = probe_peer_room_reservation( - default_db_path(), - room_id=room_id, - target_profile=requested_profile, - ) - except RoomProbeUnavailableError: - return _err( - rid, - 5122, - "Could not verify this group. Try again after the gateway recovers.", - ) - except HostedRoomError: - # Legacy Desktop sessions used the display name after - # "Group: "; those names are not hosted room ids. - pass - except Exception: - return _err( - rid, - 5122, - "Could not verify this group. Try again after the gateway recovers.", - ) - else: - if hosted or peer: - return _err( - rid, - 4122, - ( - "This room is managed by its gateway. " - if hosted - else "This room is managed by its home host. " - ) - + "Update Hermes Desktop to continue it.", - ) - if (limit_message := _ensure_active_session_slot(sid, session)) is not None: - # The refusal reason travels as machine-readable data, not as prose. - # - # An automated client has to tell "the machine is at capacity, retry later" - # from "this session has a live owner, and your write would interleave with - # theirs". Those call for different behaviour, and a client that had to - # distinguish them by matching the message text would silently change - # behaviour the next time the wording improved. - # - # Refused HERE, before the busy-queue check, before _ensure_session_db_row - # and before _start_agent_build: no user row is persisted and no model turn - # begins, so a refusal leaves the session exactly as it was. - reason = getattr(limit_message, "reason", None) - return _err( - rid, - 4090, - str(limit_message), - {"reason": reason} if reason else None, + segment_ordinal = client_ordinal - prefix_user_count + if segment_ordinal < 0 or segment_ordinal >= len(user_indices): + return _stale() + # Durability is a state.db property, not an optional annotation on the + # live copy (resume paths historically omitted _row_id stamps). If the + # durable state cannot be read, fail closed too: absence of proof is + # not proof of an ephemeral conversation. + has_stamped_user = any( + _message_row_id(history[h_idx]) is not None for h_idx in user_indices ) - # Which desktop window this message was typed into. Rewritten on every - # submit, because one session can be driven from the app window and the HUD - # in turn: a stale "hud" would tell the model the user is still floating - # over another app when they are back in Hermes. - session["client_surface"] = "hud" if params.get("surface") == "hud" else "" - has_truncation = ( - truncate_user_ordinal is not None - or params.get("truncate_before_row_id") is not None - or params.get("truncate_before_message_id") is not None - ) - if has_truncation and isinstance(text, str): - # A rewind/regenerate replays a turn from what the transcript shows. A - # skill turn shows its invocation, so re-expand it here — otherwise - # re-running `/work fix it` sends the agent nine literal characters - # instead of the skill it originally loaded. - text = _expand_skill_invocation_for_replay( - text, str(session.get("session_key") or "") + durable_history = ( + [] if has_stamped_user else _load_durable_truncation_history(session, sid) ) - isolation_cfg = _load_dashboard_process_isolation_config() - turn_isolation = _session_uses_compute_host(session, isolation_cfg) - if internal_hosted_submit and turn_isolation: - return _err( - rid, - 4121, - "hosted room turns do not support isolated compute workers yet", - ) - # Re-bind to the current client transport for this request. This keeps - # streaming events on the active websocket even if an earlier disconnect - # or fallback moved the session transport to stdio. - if (t := current_transport()) is not None: - session["transport"] = t - while True: - busy_transport = None - with session["history_lock"]: - if session.get("running"): - if internal_hosted_submit: - return _err(rid, 4091, "hosted room member session is busy") - # Don't reject a mid-turn prompt — queue it (and, by default, - # interrupt the live turn) so it runs as the next turn. The - # provider interrupt itself must happen after this lock is - # released: a non-interruptible tool may keep it waiting. - busy_transport = t or session.get("transport") - else: - break - busy_response = _handle_busy_submit( - rid, sid, session, text, busy_transport, - queued=bool(params.get("queued")), - ) - if busy_response is not None: - return busy_response - # The old turn finished between the two lock acquisitions. Retry the - # claim so this prompt starts normally instead of being stranded in a - # queue whose drain already ran. - - # Filled when this submit performed a truncation against a durable session: - # the fresh post-rewrite row ids of the surviving user turns, for client - # rowId rebinding (see comment at the assignment site). - survivor_user_row_ids = None - survivor_row_id_map = None - raw_rebind_ids = params.get("rebind_survivor_row_ids") - requested_rebind_ids = ( - { - row_id - for row_id in raw_rebind_ids - if isinstance(row_id, int) and not isinstance(row_id, bool) - } - if isinstance(raw_rebind_ids, list) - else None - ) - with session["history_lock"]: - # A watch session's run lives in the PARENT turn, so its own running - # flag is False — without this, typing mid-run builds a second agent - # racing the in-flight child on the same stored session (interleaved - # transcript, stale fork). After the run completes, submitting is fine: - # the upgrade resumes the child's transcript as a normal conversation. - if session.get("lazy") and _child_run_active(str(session.get("session_key") or "")): - return _err(rid, 4009, "subagent still running — wait for it to finish") - truncate_message_id = params.get("truncate_before_message_id") - truncate_row_id = params.get("truncate_before_row_id") - if ( - is_truthy_value(params.get("confirm_truncate")) - and truncate_user_ordinal is None - and truncate_message_id is None - and truncate_row_id is None - ): - return _err( + if has_stamped_user or durable_history is None or durable_history: + logger.warning( + "prompt.submit: REFUSED ordinal-only truncation of durable " + "session %s (ordinal=%d); truncate_before_row_id required", + sid, + client_ordinal, + ) + return None, None, _err( rid, 4004, - "confirm_truncate requires truncate_before_user_ordinal, truncate_before_message_id, or truncate_before_row_id", - ) - if ( - truncate_user_ordinal is not None - or truncate_message_id is not None - or truncate_row_id is not None - ): - history = _history_without_ephemeral_scaffolding( - session.get("history", []) + "ordinal-only truncation is unsafe for durable session history; " + "include truncate_before_row_id", ) + ordinal = segment_ordinal - # Malformed params refuse first (4004), regardless of consent — - # the historical ordinal-path precedence. - target_row_id = None - if truncate_row_id is not None: - target_row_id, err = _coerce_truncate_int( - rid, truncate_row_id, "truncate_before_row_id" + # Reject out-of-range on BOTH ends: a negative ordinal would hit Python's + # negative indexing (user_indices[-1] → the LAST user turn) and persist the loss. + if ordinal < 0 or ordinal >= len(user_indices): + return _stale(resolved_ordinal=ordinal) + return ordinal, user_indices[ordinal], None + + +def _row_ids_of(messages) -> set: + return {row_id for message in messages if isinstance((row_id := _message_row_id(message)), int)} + + +def _persist_truncation(rid, sid, session, history, truncated, ordinal, requested_rebind_ids): + """Write the truncated transcript BEFORE touching memory (fail closed). + + If replace_messages failed after session["history"] was rewritten, the turn + would run against the short list while state.db kept the old tail; the + append-only agent flush would then stack the new exchange on the "undone" + turns — zombie history on resume. Writes through ``_session_db`` (the db that + owns this session's row), never ``_get_db()``: a profile session's transcript + lives in its own profile's state.db. + + Returns ``(err, survivor_user_row_ids, survivor_row_id_map)``. + """ + survivor_user_row_ids = None + survivor_row_id_map = None + with _session_db(session) as db: + if db is not None: + try: + # session_key can be NULL for old CLI-origin sessions; fall back to + # sid or replace_messages(None) trips an FK violation. + truncation_key = session.get("session_key") or sid + old_active_row_ids = _row_ids_of(history) + if requested_rebind_ids is not None: + # Row-id fallback can resolve a target the live list is too + # misaligned to stamp, and repair can merge a user;user pair + # keeping only the first id: read the authoritative + # un-repaired pre-write active-id set so a rewritten row is + # never mistaken for an untouched archived/ancestor row. + durable_rebind_history = _load_durable_truncation_history( + session, truncation_key, repair_alternation=False, + ) + if durable_rebind_history is None: + raise RuntimeError("could not load durable row identities for truncation") + old_active_row_ids.update(_row_ids_of(durable_rebind_history)) + old_survivor_row_ids = [_message_row_id(message) for message in truncated] + # active_only=True: in-place compaction keeps the pre-compaction + # transcript as active=0 rows under this key; a bare replace would + # DELETE that archive on every edit. archive_dropped=True: this + # write is the last step before the dropped turns are gone — + # soft-archive (active=0, still in FTS) so a mis-aimed cut is + # recoverable. + db.replace_messages( + truncation_key, truncated, active_only=True, archive_dropped=True, + reject_active_turn_lease=True, ) - if err is not None: - return err - client_ordinal = None - if truncate_user_ordinal is not None: - client_ordinal, err = _coerce_truncate_int(rid, truncate_user_ordinal) - if err is not None: - return err - - # An ordinal/id alone is not consent. A client that carries a leftover - # ordinal into an ORDINARY submit sends a request that is - # indistinguishable, field by field, from a real rewind — same - # method, same shape, an in-range target — and the cut it asks for - # is a destructive replace_messages() the user never requested - # (#80763: 296 -> 52 messages, 244 durable rows gone). Only the - # client knows whether this submit is a rewind/edit/regenerate, so - # it has to say so; refuse the cut when it doesn't. Consent is - # checked BEFORE target resolution: an unconfirmed (leaked-state) - # request must refuse with 4029 without paying the durable - # transcript read or heal-stamping live history dicts that - # row-id resolution performs. - if not is_truthy_value(params.get("confirm_truncate")): - logger.warning( - "prompt.submit: REFUSED unconfirmed truncation of session %s " - "(%d messages held; ordinal=%s, row_id=%s, message_id=%s). " - "The client attached truncation parameters without " - "confirm_truncate — likely stale truncation parameters on " - "an ordinary submit.", + except Exception as exc: + logger.error( + "prompt.submit: replace_messages failed for session %s " + "(ordinal=%d); refusing turn so memory and DB stay " + "aligned: %s", sid, - len(history), - client_ordinal, - target_row_id, - truncate_message_id, - ) - return _err( - rid, - 4029, - "truncation parameters require confirm_truncate=true; " - "an ordinary prompt.submit must not drop session history " - "(update your Hermes client if a rewind was intended)", - ) - # Desktop/TUI ordinals count the full displayed lineage. After - # compression, session["history"] holds only the tip segment while - # display_history_prefix holds the immutable ancestor display rows - # still shown in the transcript (#82462 / #69107). Count the - # ancestor user turns once so every comparison between a client - # ordinal and a tip-relative ordinal below can translate, instead - # of loading ancestors into the tip (which would duplicate - # compressed history on later resumes). - prefix_user_count = len( - _history_user_indices( - session.get("display_history_prefix") or [] - ) - ) - - user_indices = _history_user_indices(history) - - def _stale_target_data(resolved_ordinal=None): - # Structured recovery fields for clients (#82462): Desktop - # resyncs + retries on a stale target, and shows an explicit - # "compressed away" state when segment_ordinal < 0 (the target - # only exists in the immutable ancestor prefix). - segment = ( - client_ordinal - prefix_user_count - if client_ordinal is not None - else resolved_ordinal - ) - return { - "user_turn_count": len(user_indices), - "ordinal": client_ordinal, - "segment_ordinal": segment, - "prefix_user_count": prefix_user_count, - } - - ordinal = None - - if target_row_id is not None: - # Durable address first — never degrade a missing row_id into a - # client ordinal cut (#82959 / #82766 review). Unknown id refuses - # without touching data; stale ordinal with a *resolved* row_id - # is a separate 4030 mismatch below. - found_match = _resolve_truncate_row_id( - session, history, target_row_id - ) - - if found_match is None: - logger.warning( - "prompt.submit: target row_id %d not found for session %s " - "(in-memory + durable); refusing truncation without fallback", - target_row_id, - sid, - ) - return _err( - rid, - 4018, - "target user message is no longer in session history", - data=_stale_target_data(), - ) - - msg_ordinal, _ = found_match - ordinal, err = _reconcile_client_ordinal( - rid, sid, client_ordinal, msg_ordinal, - "truncate_before_row_id", target_row_id, - prefix_user_count=prefix_user_count, - ) - if err is not None: - return err - elif truncate_message_id is not None: - msg_id_str = str(truncate_message_id) - found_match = None - for u_ord, h_idx in enumerate(user_indices): - msg = history[h_idx] - if msg.get("id") == msg_id_str or msg.get("message_id") == msg_id_str: - found_match = (u_ord, h_idx) - break - - if found_match is None: - # Fail closed: a supplied message_id that does not resolve - # must not fall back to a (possibly stale) ordinal. Desktop - # clients should send truncate_before_row_id instead. - logger.warning( - "prompt.submit: target message_id %s not found in history " - "for session %s; refusing truncation without fallback", - msg_id_str, - sid, - ) - return _err( - rid, - 4018, - "target user message is no longer in session history", - data=_stale_target_data(), - ) - - msg_ordinal, _ = found_match - ordinal, err = _reconcile_client_ordinal( - rid, sid, client_ordinal, msg_ordinal, - "truncate_before_message_id", msg_id_str, - prefix_user_count=prefix_user_count, - ) - if err is not None: - return err - else: - # Client ordinals count the full displayed lineage; translate - # into the tip segment before the bounds check (#82462). An - # ancestor-only target (segment_ordinal < 0) is not editable - # from this continuation segment — same stale-target refusal, - # with the structured fields so the client can tell the - # "compressed away" case apart from plain drift. - segment_ordinal = client_ordinal - prefix_user_count - if segment_ordinal < 0 or segment_ordinal >= len(user_indices): - return _err( - rid, - 4018, - "target user message is no longer in session history", - data=_stale_target_data(), - ) - # Durability is a state.db property, not an optional annotation - # on the live copy. Resume/reload paths historically omitted - # _row_id stamps, which made an ordinal-only request look safe - # even though it could destructively replace a long transcript. - # If the durable state cannot be read, fail closed too: absence - # of proof is not proof that this is an ephemeral conversation. - has_stamped_user = any( - _message_row_id(history[h_idx]) is not None - for h_idx in user_indices - ) - durable_history = ( - [] - if has_stamped_user - else _load_durable_truncation_history(session, sid) - ) - if has_stamped_user or durable_history is None or durable_history: - logger.warning( - "prompt.submit: REFUSED ordinal-only truncation of durable " - "session %s (ordinal=%d); truncate_before_row_id required", - sid, - client_ordinal, - ) - return _err( - rid, - 4004, - "ordinal-only truncation is unsafe for durable session history; " - "include truncate_before_row_id", - ) - ordinal = segment_ordinal - - # Reject out-of-range ordinals on BOTH ends. A negative value would - # otherwise sail past the upper-bound check and hit Python's negative - # indexing below (user_indices[-1] -> the LAST user turn), silently - # truncating history to everything before it and persisting that loss - # via replace_messages — an unrecoverable overwrite of the session DB. - if ordinal < 0 or ordinal >= len(user_indices): - return _err( - rid, - 4018, - "target user message is no longer in session history", - data=_stale_target_data(resolved_ordinal=ordinal), - ) - from agent.context_compressor import history_before_user_originated_turn - - truncated, _live_view = history_before_user_originated_turn( - history, user_indices[ordinal] - ) - # Second gate, on top of confirm_truncate: ordinal 0 resolves to - # history[:0] == [] and replace_messages() DELETEs every durable - # row. A confirmed rewind that happens to erase the whole - # transcript still needs its own opt-in (legitimate restore/ - # regenerate of the first user turn). - if ( - not truncated - and history - and not is_truthy_value(params.get("confirm_empty_truncate")) - ): - logger.warning( - "prompt.submit: REFUSED empty truncation of session %s " - "(%d messages would be wiped; ordinal=%d).", - sid, - len(history), ordinal, + exc, + exc_info=True, ) - return _err( - rid, - 4028, - "truncation would erase the entire session transcript; " - "resubmit with confirm_empty_truncate=true if this is intended", - ) - # Info for routine rewind/edit cuts; warning only when the client - # explicitly opts into wiping the whole transcript. - log_fn = logger.warning if not truncated else logger.info - log_fn( - "prompt.submit: truncating session %s history %d -> %d messages " - "(ordinal=%d)", - sid, - len(history), - len(truncated), - ordinal, - ) - # Write-before-memory (mirrors gateway hygiene / manual /compress): - # persist the truncated transcript first. If replace_messages fails - # after we already rewrote session["history"], the turn still runs - # against the short list while state.db keeps the old tail. The - # agent flush is append-only for history-dict identities, so the - # new exchange is appended on top of the "undone" turns — durable - # zombie history on resume, and the edit/regenerate never sticks. - # Fail closed: refuse the turn and leave memory/DB unchanged. - # - # _session_db, not _get_db(): the truncation has to land in the db - # that owns this session's row. A profile session (app-global - # remote mode) keeps its transcript in its own profile's state.db, - # so writing through the launch handle both loses the edit — resume - # reopens the profile db and resurrects the undone turns — and - # copies the transcript into a foreign profile under this session's - # id when that profile happens to hold a row for it. Fail-closed - # only holds if the handle we check is the one that owns the row. - with _session_db(session) as db: - if db is not None: - try: - # active_only=True: replace only the live (active=1) - # rows. In-place compaction (#38763) keeps the - # pre-compaction transcript as active=0/compacted=1 - # rows under this same session key; a bare - # replace_messages() would DELETE that durable archive - # on every edit/regenerate — the same bug class #80216 - # fixed for /retry. On an uncompacted session all rows - # are active=1, so this is behaviorally identical to - # the full replace. - # archive_dropped: a rewind overwrites turns the user - # may not have meant to drop, and this write is the - # last step before they are gone — three reported - # incidents ended here with nothing to restore from - # (#70516, #80763, #82756). Soft-archiving keeps them - # on disk (active=0) and in the FTS index, so a - # mis-aimed cut is recoverable instead of terminal. - # The live transcript is unchanged. - # Fall back to session id when session_key is NULL — - # CLI-origin sessions created before the session_key - # default fix have no key, and replace_messages(None) - # triggers an FK violation. - truncation_key = session.get("session_key") or sid - old_active_row_ids = { - row_id - for message in history - if isinstance( - (row_id := _message_row_id(message)), int - ) - } - if requested_rebind_ids is not None: - # Row-id fallback can resolve a durable target even - # when the live list is too misaligned to stamp safely, - # and alternation repair can merge a physical user;user - # pair while preserving only the first row id. Read the - # authoritative un-repaired pre-write active-id set so - # a rewritten row is never mistaken for an untouched - # archived/ancestor row by the bounded client map. - durable_rebind_history = ( - _load_durable_truncation_history( - session, - truncation_key, - repair_alternation=False, - ) - ) - if durable_rebind_history is None: - raise RuntimeError( - "could not load durable row identities for truncation" - ) - old_active_row_ids.update( - row_id - for message in durable_rebind_history - if isinstance( - (row_id := _message_row_id(message)), int - ) - ) - old_survivor_row_ids = [ - _message_row_id(message) for message in truncated - ] - db.replace_messages( - truncation_key, - truncated, - active_only=True, - archive_dropped=True, - reject_active_turn_lease=True, - ) - except Exception as exc: - logger.error( - "prompt.submit: replace_messages failed for session %s " - "(ordinal=%d); refusing turn so memory and DB stay " - "aligned: %s", - sid, - ordinal, - exc, - exc_info=True, - ) - return _err( - rid, - 5008, - f"failed to persist history truncation: {exc}", - ) - # replace_messages re-inserted the surviving prefix as NEW - # rows and stamped fresh _row_id values onto these same - # dicts. Surface the surviving user-turn ids (in - # visible-user-ordinal order) so the client can rebind its - # cached rowId stamps — otherwise a second rewind targeting - # an older surviving turn sends the pre-rewind id and the - # fail-closed resolver refuses it with 4018 (#83202 review: - # consecutive-rewind staleness). Ordinal order matches the - # client's visible-user filter the same way truncate - # ordinals already do. Entries are None when a row somehow - # has no stamp — the client must drop its cached id for - # that turn rather than keep a stale one. - survivor_user_row_ids = [ - _message_row_id(truncated[i]) - for i in _history_user_indices(truncated) - ] - if requested_rebind_ids is not None: - survivor_row_id_map = { - str(old_row_id): new_row_id - for old_row_id, new_row_id in zip( - old_survivor_row_ids, - ( - _message_row_id(message) - for message in truncated - ), - ) - if isinstance(old_row_id, int) - and isinstance(new_row_id, int) - and old_row_id in requested_rebind_ids - } - for dropped_row_id in requested_rebind_ids.intersection( - old_active_row_ids - ): - survivor_row_id_map.setdefault( - str(dropped_row_id), None - ) - session["history"] = truncated - session["history_version"] = int(session.get("history_version", 0)) + 1 - session["running"] = True - session["_turn_cancel_requested"] = False - session["last_active"] = time.time() - if internal_hosted_submit: - session["_hosted_room_task"] = dict(hosted_task) - _start_inflight_turn(session, text) + return _err(rid, 5008, f"failed to persist history truncation: {exc}"), None, None + # replace_messages re-inserted the survivors as NEW rows and stamped + # fresh _row_id values onto these dicts. Surface the surviving + # user-turn ids (visible-user-ordinal order) so the client rebinds + # its cached rowIds — else a second rewind sends the pre-rewind id + # and the fail-closed resolver refuses with 4018. None entries mean + # the client must drop its cached id for that turn. + survivor_user_row_ids = [ + _message_row_id(truncated[i]) for i in _history_user_indices(truncated) + ] + if requested_rebind_ids is not None: + survivor_row_id_map = { + str(old_row_id): new_row_id + for old_row_id, new_row_id in zip( + old_survivor_row_ids, + (_message_row_id(message) for message in truncated), + ) + if isinstance(old_row_id, int) + and isinstance(new_row_id, int) + and old_row_id in requested_rebind_ids + } + for dropped_row_id in requested_rebind_ids.intersection( + old_active_row_ids + ): + survivor_row_id_map.setdefault(str(dropped_row_id), None) + return None, survivor_user_row_ids, survivor_row_id_map - if turn_isolation: - isolated_response = _submit_prompt_to_compute_host( - rid, sid, session, text, display_kind=display_kind - ) - if not isolated_response.get("error"): - if survivor_user_row_ids is not None and requested_rebind_ids is None: - # The truncation already happened inline above (memory + DB), - # before compute-host dispatch — the rebind payload applies to - # this path exactly as it does to the inline one. - isolated_response["result"][ - "survivor_user_row_ids" - ] = survivor_user_row_ids - if survivor_row_id_map is not None: - isolated_response["result"]["survivor_row_id_map"] = survivor_row_id_map - return isolated_response + +def _truncate_history_for_submit(rid, sid, session, params, requested_rebind_ids): + """Rewind/regenerate cut, under ``history_lock``. Returns + ``(err, survivor_user_row_ids, survivor_row_id_map)``; on success + ``session["history"]`` is replaced and ``history_version`` bumped.""" + history = _history_without_ephemeral_scaffolding(session.get("history", [])) + ordinal, cut_index, err = _resolve_truncation_ordinal(rid, sid, session, params, history) + if err is not None: + return err, None, None + from agent.context_compressor import history_before_user_originated_turn + + truncated, _live_view = history_before_user_originated_turn(history, cut_index) + # Second gate on top of confirm_truncate: ordinal 0 → history[:0] == [] and + # replace_messages() DELETEs every durable row. Wiping the whole transcript + # needs its own opt-in (legitimate restore/regenerate of the first turn). + if ( + not truncated + and history + and not is_truthy_value(params.get("confirm_empty_truncate")) + ): logger.warning( - "compute-host dispatch failed for session %s; falling back inline: %s", + "prompt.submit: REFUSED empty truncation of session %s " + "(%d messages would be wiped; ordinal=%d).", sid, - isolated_response["error"].get("message", "unknown error"), + len(history), + ordinal, ) + return _err( + rid, + 4028, + "truncation would erase the entire session transcript; " + "resubmit with confirm_empty_truncate=true if this is intended", + ), None, None + log_fn = logger.warning if not truncated else logger.info + log_fn( + "prompt.submit: truncating session %s history %d -> %d messages (ordinal=%d)", + sid, len(history), len(truncated), ordinal, + ) + err, survivor_user_row_ids, survivor_row_id_map = _persist_truncation( + rid, sid, session, history, truncated, ordinal, requested_rebind_ids + ) + if err is not None: + return err, None, None + session["history"] = truncated + session["history_version"] = int(session.get("history_version", 0)) + 1 + return None, survivor_user_row_ids, survivor_row_id_map - # Persist the DB row lazily, now that the user has actually sent a message. - # Disk-full must fail the RPC (not stream silently): desktop maps the error - # string to a "disk full" toast so the user knows why the send vanished. + +def _survivor_fields(survivor_user_row_ids, survivor_row_id_map, requested_rebind_ids) -> dict: + """Client rowId-rebind payload for a submit that truncated a durable session.""" + fields = {} + if survivor_user_row_ids is not None and requested_rebind_ids is None: + fields["survivor_user_row_ids"] = survivor_user_row_ids + if survivor_row_id_map is not None: + fields["survivor_row_id_map"] = survivor_row_id_map + return fields + + +def _persist_session_row_for_submit(rid, session): + """Lazily persist the DB row now that the user actually sent a message; a + branch becomes real here (parent transcript copied as its seed). Returns an + error reply — the only user-visible signal; desktop maps the string to a + toast — or None. On failure the in-flight turn is released.""" try: if _ensure_session_db_row(session) is False: - # Store unavailable: failing the RPC is the only user-visible - # signal — same principle as the disk-full path above (#98924). - # _db_error carries the SessionDB open failure for the toast. return _err( rid, 5072, @@ -965,8 +650,6 @@ def _(rid, params: dict) -> dict: f"{_db_error or 'state.db could not be opened'} — the message " "was not saved; repair state.db and try again", ) - # A branch becomes real here: copy its parent's transcript into the row so it - # resumes with full context (the agent won't persist the seed itself). _persist_branch_seed(session) except Exception as exc: from hermes_state import is_disk_full_error @@ -982,99 +665,210 @@ def _(rid, params: dict) -> dict: "disk full: session storage could not be written — free some disk space and try again", ) logger.warning("prompt.submit: session persist failed: %s", exc, exc_info=True) - return _err( - rid, - 5071, - f"session storage could not be written: {exc}", + return _err(rid, 5071, f"session storage could not be written: {exc}") + return None + + +def _run_after_agent_ready(rid, sid, session, text, display_kind, hosted_terminal_callback): + """Turn thread body: patient wait for a deferred build (the message is already + the accepted in-flight turn, so a slow build must not eat it), then run.""" + err = _wait_agent_for_prompt(session, rid, sid) + if err: + # Terminal frame + retained snapshot (not a bare "error" event): if the + # client is disconnected, the snapshot is the only way resume shows this. + _emit_terminal_turn_error( + sid, + session, + (err.get("error") or {}).get("message", "agent initialization failed"), + # Construction never reached the provider: local-runtime failure. + error_surface={"layer": "runtime", "code": "agent_init_failed", "retryable": True}, ) - # A completed FAILED build must not wedge the session: the error frame - # says retryable, so a new send (or the error card's Retry) rebuilds the - # agent with fresh provider resolution instead of replaying the cached - # failure forever. Before this, only a model switch reset the failed - # generation — a session that failed once (local server off) kept - # erroring after the server came back, while new sessions worked. Falls - # through to the normal build when there is no completed failure to - # clear. + with session["history_lock"]: + session["running"] = False + session["last_active"] = time.time() + _emit("session.info", sid, _session_info(session.get("agent"), session)) + return + with session["history_lock"]: + if session.get("_turn_cancel_requested") or not session.get("running"): + session["running"] = False + _clear_inflight_turn(session) + # Without this emit the turn vanishes silently: the client saw + # {"status": "streaming"} but never gets message.start or error. + _emit( + "error", + sid, + { + "message": "Turn cancelled before the agent was ready" + if session.get("_turn_cancel_requested") + else "Session no longer running before the agent was ready" + }, + ) + return + _run_prompt_submit( + rid, sid, session, text, display_kind=display_kind, + terminal_callback=hosted_terminal_callback, + ) + + +@method("prompt.submit") +def _(rid, params: dict) -> dict: + from hermes_cli.input_sanitize import sanitize_user_prompt_text + + sid = params.get("session_id", "") + raw_text = params.get("text", "") + text = sanitize_user_prompt_text(raw_text) if isinstance(raw_text, str) else raw_text + # Off-screen sends (widget intents) type the persisted row so no client + # renders a bubble. Whitelisted to "hidden": this RPC must not mint kinds. + display_kind = "hidden" if params.get("display_kind") == "hidden" else None + if (stopped := _typed_stop_phrase_response(rid, text)) is not None: + return stopped + if params.get("interrupted"): + # Client-side barge-in (desktop VAD / typing over playback): latch it so + # this turn's model message carries the interruption note. + from tools.tts_streaming import mark_speech_interrupted + + mark_speech_interrupted() + session, err = _sess_nowait(params, rid) + if err: + return err + hosted_task = params.get("_hosted_task") + hosted_terminal_callback = params.get("_hosted_terminal_callback") + internal_hosted_submit = hosted_task is not None or hosted_terminal_callback is not None + if internal_hosted_submit: + err = _hosted_submit_error(rid, session, hosted_task, hosted_terminal_callback) + else: + err = _legacy_group_fence_error(rid, session, params) + if err is not None: + return err + if (limit_message := _ensure_active_session_slot(sid, session)) is not None: + # Refused HERE — before the busy queue, _ensure_session_db_row and + # _start_agent_build — so a refusal leaves the session exactly as it was. + # The reason travels as machine-readable data ("at capacity, retry" vs + # "live owner, your write would interleave"), never as matched prose. + reason = getattr(limit_message, "reason", None) + return _err(rid, 4090, str(limit_message), {"reason": reason} if reason else None) + # Rewritten on every submit: one session can be driven from the app window + # and the HUD in turn, and a stale "hud" misinforms the model. + session["client_surface"] = "hud" if params.get("surface") == "hud" else "" + has_truncation = any( + params.get(k) is not None + for k in ("truncate_before_user_ordinal", "truncate_before_row_id", "truncate_before_message_id") + ) + if has_truncation and isinstance(text, str): + # A rewind replays what the transcript shows; a skill turn shows its + # invocation, so re-expand it or `/work fix it` sends nine literal chars. + text = _expand_skill_invocation_for_replay(text, str(session.get("session_key") or "")) + isolation_cfg = _load_dashboard_process_isolation_config() + turn_isolation = _session_uses_compute_host(session, isolation_cfg) + if internal_hosted_submit and turn_isolation: + return _err(rid, 4121, "hosted room turns do not support isolated compute workers yet") + # Re-bind to the current client transport so streaming stays on the active + # websocket even if a disconnect/fallback moved the session to stdio. + if (t := current_transport()) is not None: + session["transport"] = t + while True: + busy_transport = None + with session["history_lock"]: + if session.get("running"): + if internal_hosted_submit: + return _err(rid, 4091, "hosted room member session is busy") + # Queue a mid-turn prompt (and by default interrupt the live turn) + # instead of rejecting. The provider interrupt must happen after + # this lock is released: a non-interruptible tool may hold it. + busy_transport = t or session.get("transport") + else: + break + busy_response = _handle_busy_submit( + rid, sid, session, text, busy_transport, queued=bool(params.get("queued")), + ) + if busy_response is not None: + return busy_response + # The old turn finished between the two lock acquisitions: retry the + # claim rather than strand this prompt in a queue whose drain already ran. + + survivor_user_row_ids = None + survivor_row_id_map = None + raw_rebind_ids = params.get("rebind_survivor_row_ids") + requested_rebind_ids = ( + { + row_id + for row_id in raw_rebind_ids + if isinstance(row_id, int) and not isinstance(row_id, bool) + } + if isinstance(raw_rebind_ids, list) + else None + ) + with session["history_lock"]: + # A watch session's run lives in the PARENT turn, so its own running flag + # is False; typing mid-run would build a second agent racing the child + # on the same stored session. After the run completes, submitting is fine. + if session.get("lazy") and _child_run_active(str(session.get("session_key") or "")): + return _err(rid, 4009, "subagent still running — wait for it to finish") + if is_truthy_value(params.get("confirm_truncate")) and not has_truncation: + return _err( + rid, + 4004, + "confirm_truncate requires truncate_before_user_ordinal, truncate_before_message_id, or truncate_before_row_id", + ) + if has_truncation: + err, survivor_user_row_ids, survivor_row_id_map = _truncate_history_for_submit( + rid, sid, session, params, requested_rebind_ids + ) + if err is not None: + return err + session["running"] = True + session["_turn_cancel_requested"] = False + session["last_active"] = time.time() + if internal_hosted_submit: + session["_hosted_room_task"] = dict(hosted_task) + _start_inflight_turn(session, text) + + survivor_fields = _survivor_fields( + survivor_user_row_ids, survivor_row_id_map, requested_rebind_ids + ) + if turn_isolation: + isolated_response = _submit_prompt_to_compute_host( + rid, sid, session, text, display_kind=display_kind + ) + if not isolated_response.get("error"): + # The truncation already happened inline above (memory + DB). + isolated_response["result"].update(survivor_fields) + return isolated_response + logger.warning( + "compute-host dispatch failed for session %s; falling back inline: %s", sid, + isolated_response["error"].get("message", "unknown error"), + ) + + if (err := _persist_session_row_for_submit(rid, session)) is not None: + return err + # A completed FAILED build must not wedge the session: rebuild with fresh + # provider resolution instead of replaying the cached failure forever. if not _restart_completed_failed_agent_build( sid, session, session.get("agent_ready") ): _start_agent_build(sid, session) - def run_after_agent_ready() -> None: - # Patient wait (#63078): the user's message is already the accepted - # in-flight turn, so a slow deferred build must not eat it. The wait - # delivers the prompt when the still-running build completes, honors a - # cancel promptly, notices the user once past the slow threshold, and - # only errors when the build itself fails or the bounded cap expires. - err = _wait_agent_for_prompt(session, rid, sid) - if err: - # Terminal frame + retained snapshot (not a bare "error" event + - # cleared inflight): if the client is disconnected right now, the - # retained snapshot is the only way resume can show this failure. - _emit_terminal_turn_error( - sid, - session, - (err.get("error") or {}).get("message", "agent initialization failed"), - # Agent construction never reached the provider: this is a - # local-runtime failure (env/config/venv), not an API error. - error_surface={"layer": "runtime", "code": "agent_init_failed", "retryable": True}, - ) - with session["history_lock"]: - session["running"] = False - session["last_active"] = time.time() - _emit("session.info", sid, _session_info(session.get("agent"), session)) - return - with session["history_lock"]: - if session.get("_turn_cancel_requested") or not session.get("running"): - session["running"] = False - _clear_inflight_turn(session) - # Surface the cancellation to the client. Without this emit the - # turn vanishes silently — the Desktop sees `prompt.submit` - # return `{"status": "streaming"}` but never receives a - # `message.start` or `error` event, so the composer shows no - # feedback (issue #63078 server-side half). Match the - # `_wait_agent` error branch above: emit, then bail. - _emit( - "error", - sid, - { - "message": "Turn cancelled before the agent was ready" - if session.get("_turn_cancel_requested") - else "Session no longer running before the agent was ready" - }, - ) - return - _run_prompt_submit( - rid, - sid, - session, - text, - display_kind=display_kind, - terminal_callback=hosted_terminal_callback, - ) - - run_thread = threading.Thread(target=run_after_agent_ready, daemon=True) - # Keep a handle so session.interrupt can tell a live turn from a stuck - # `running` flag (a turn that died without clearing it) and recover the latter. + run_thread = threading.Thread( + target=lambda: _run_after_agent_ready( + rid, sid, session, text, display_kind, hosted_terminal_callback + ), + daemon=True, + ) + # Handle lets session.interrupt tell a live turn from a stuck `running` flag. session["_run_thread"] = run_thread run_thread.start() - return _ok( - rid, - { - "status": "streaming", - **( - {"survivor_user_row_ids": survivor_user_row_ids} - if survivor_user_row_ids is not None - and requested_rebind_ids is None - else {} - ), - **( - {"survivor_row_id_map": survivor_row_id_map} - if survivor_row_id_map is not None - else {} - ), - }, - ) + return _ok(rid, {"status": "streaming", **survivor_fields}) + + +# ── attachments ───────────────────────────────────────────────────────────── + + +def _attached_image_result(session, image_path, **extra) -> dict: + """Common ``{attached, path, count, ...meta}`` reply after queuing an image.""" + return { + "attached": True, "path": str(image_path), "count": len(session["attached_images"]), + **extra, **_image_meta(image_path), + } @method("clipboard.paste") @@ -1091,11 +885,10 @@ def _(rid, params: dict) -> dict: img_dir = _session_images_dir(session) img_dir.mkdir(parents=True, exist_ok=True) img_path = ( - img_dir - / f"clip_{datetime.now().strftime('%Y%m%d_%H%M%S')}_{session['image_counter']}.png" + img_dir / f"clip_{datetime.now().strftime('%Y%m%d_%H%M%S')}_{session['image_counter']}.png" ) - # Save-first: mirrors CLI keybinding path; more robust than has_image() precheck + # Save-first (CLI keybinding parity): more robust than a has_image() precheck. if not save_clipboard_image(img_path): session["image_counter"] = max(0, session["image_counter"] - 1) msg = ( @@ -1106,15 +899,7 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"attached": False, "message": msg}) session.setdefault("attached_images", []).append(str(img_path)) - return _ok( - rid, - { - "attached": True, - "path": str(img_path), - "count": len(session["attached_images"]), - **_image_meta(img_path), - }, - ) + return _ok(rid, _attached_image_result(session, img_path)) @method("image.attach") @@ -1145,38 +930,21 @@ def _(rid, params: dict) -> dict: if image_path.suffix.lower() not in _IMAGE_EXTENSIONS: return _err(rid, 4016, f"unsupported image: {image_path.name}") session.setdefault("attached_images", []).append(str(image_path)) - return _ok( - rid, - { - "attached": True, - "path": str(image_path), - "count": len(session["attached_images"]), - "remainder": remainder, - "text": remainder or f"[User attached image: {image_path.name}]", - **_image_meta(image_path), - }, - ) + return _ok(rid, _attached_image_result( + session, image_path, + remainder=remainder, + text=remainder or f"[User attached image: {image_path.name}]", + )) except Exception as e: return _err(rid, 5027, str(e)) @method("image.attach_bytes") def _(rid, params: dict) -> dict: - """Attach an image to the session from base64 bytes (remote-client path). - - A desktop app or web dashboard running on a DIFFERENT machine than the - gateway can't hand us a local path — that file only exists on the client's - disk. So it uploads the raw image bytes (base64) and we write them into the - gateway's own images dir. The response shape mirrors ``image.attach`` so the - client treats both identically. - - Params: - content_base64 / data (str, required): base64 image bytes. Accepts a - ``data:image/...;base64,`` prefix and embedded whitespace. ``data`` is - an accepted alias for older desktop builds. - filename / ext (str, optional): extension hint. Without it, magic bytes - identify PNG/JPEG/GIF/WebP/BMP, falling back to ``.png``. - """ + """Attach an image from base64 bytes (remote client: its file isn't on our disk). + Reply shape mirrors ``image.attach``. ``content_base64``/``data`` accept a + ``data:image/...;base64,`` prefix; ``filename``/``ext`` hint the extension, else + magic bytes decide (PNG/JPEG/GIF/WebP/BMP, fallback ``.png``).""" session, err = _sess_building(params, rid) if err: return err @@ -1185,14 +953,12 @@ def _(rid, params: dict) -> dict: if not raw_b64: return _err(rid, 4015, "content_base64 required") - img_bytes = _decode_attach_base64(raw_b64, mime_prefix="image/") - if img_bytes is None: - return _err(rid, 4017, "data is not valid base64") - if not img_bytes: - return _err(rid, 4017, "image is empty") - if len(img_bytes) > _ATTACH_BYTES_MAX_BYTES: - mb = _ATTACH_BYTES_MAX_BYTES // (1024 * 1024) - return _err(rid, 4018, f"image too large ({len(img_bytes)} bytes; cap is {mb} MB)") + img_bytes, err = _decode_attach_payload( + rid, raw_b64, mime_prefix="image/", max_bytes=_ATTACH_BYTES_MAX_BYTES, + label="image", empty_msg="image is empty", + ) + if err is not None: + return err filename = str(params.get("filename", "") or "") ext_hint = str(params.get("ext", "") or "").strip().lower() @@ -1207,32 +973,19 @@ def _(rid, params: dict) -> dict: except Exception as e: return _err(rid, 5027, f"write failed: {e}") - return _ok( - rid, - { - "attached": True, - "path": str(img_path), - "count": len(session["attached_images"]), - "remainder": "", - "text": f"[User attached image: {img_path.name}]", - "bytes": len(img_bytes), - **_image_meta(img_path), - }, - ) + return _ok(rid, _attached_image_result( + session, img_path, + remainder="", + text=f"[User attached image: {img_path.name}]", + bytes=len(img_bytes), + )) @method("pdf.attach") def _(rid, params: dict) -> dict: - """Attach a PDF by rendering each page to PNG and queuing the pages. - - Anthropic's vision pipeline accepts images, not PDFs, so this runs - ``pdftoppm`` (poppler-utils) at 150 DPI per page and queues each rendered - page as an attached image. Accepts either a host ``path`` (local mode) or - base64 ``content_base64`` (remote upload). Caps at 50 MB / 25 pages per call. - - Requires ``pdftoppm`` on $PATH (``apt install poppler-utils``); returns 5028 - if missing. - """ + """Attach a PDF by rendering each page to PNG (``pdftoppm`` @150 DPI, poppler-utils; + 5028 if missing) and queuing the pages as images. Accepts a host ``path`` or + base64 ``content_base64``. Caps: 50 MB / 25 pages per call.""" import shutil import subprocess import tempfile @@ -1252,14 +1005,12 @@ def _(rid, params: dict) -> dict: with tempfile.TemporaryDirectory(prefix="pdf_attach_") as td: td_path = Path(td) if raw_b64: - pdf_bytes = _decode_attach_base64(raw_b64, mime_prefix="application/pdf") - if pdf_bytes is None: - return _err(rid, 4017, "data is not valid base64") - if not pdf_bytes: - return _err(rid, 4017, "decoded PDF is empty") - if len(pdf_bytes) > _PDF_ATTACH_MAX_BYTES: - mb = _PDF_ATTACH_MAX_BYTES // (1024 * 1024) - return _err(rid, 4018, f"PDF too large ({len(pdf_bytes)} bytes; cap is {mb} MB)") + pdf_bytes, err = _decode_attach_payload( + rid, raw_b64, mime_prefix="application/pdf", max_bytes=_PDF_ATTACH_MAX_BYTES, + label="PDF", empty_msg="decoded PDF is empty", + ) + if err is not None: + return err if pdf_bytes[:5] != b"%PDF-": return _err(rid, 4017, "payload is not a PDF (missing %PDF- magic bytes)") pdf_path = td_path / "input.pdf" @@ -1300,8 +1051,7 @@ def _(rid, params: dict) -> dict: out_prefix = td_path / "page" argv = [ - "pdftoppm", "-png", "-r", "150", - "-f", str(first_page), "-l", str(last_page), + "pdftoppm", "-png", "-r", "150", "-f", str(first_page), "-l", str(last_page), str(pdf_path), str(out_prefix), ] from hermes_cli._subprocess_compat import windows_hide_flags @@ -1309,8 +1059,8 @@ def _(rid, params: dict) -> dict: try: res = subprocess.run( argv, capture_output=True, text=True, timeout=120, stdin=subprocess.DEVNULL, - # Force UTF-8 + lossy decode so non-UTF-8 child output can't - # crash the gateway thread on locale-mismatched Windows (#53137). + # UTF-8 + lossy decode: non-UTF-8 child output must not crash the + # gateway thread on locale-mismatched Windows. encoding="utf-8", errors="replace", creationflags=windows_hide_flags(), ) @@ -1349,23 +1099,10 @@ def _(rid, params: dict) -> dict: @method("file.attach") def _(rid, params: dict) -> dict: - """Stage a non-image file attachment into the session workspace. - - The image/PDF path renders to vision tiles; this one keeps the file as a - readable artifact and returns a workspace-relative ``@file:`` ref so the - agent's file tools (and ``agent.context_references``) can read it. Solves the - remote-gateway case where the desktop passes a path that only exists on the - CLIENT's disk: the client uploads ``data_url`` bytes and we materialize the - file on the gateway. - - Params: - session_id (str, required) - path (str): client/host path of the file (used for naming + local-mode - gateway-visible resolution). - data_url (str): ``data:;base64,`` upload of the file bytes, - required when the path isn't visible to the gateway. - name (str, optional): preferred filename. - """ + """Stage a non-image file into the session workspace and return a + workspace-relative ``@file:`` ref the agent's file tools can read. ``path`` is + the client/host path (naming + local resolution); ``data_url`` carries the bytes + when the path isn't visible to the gateway; ``name`` overrides the filename.""" session, err = _sess_building(params, rid) if err: return err @@ -1444,9 +1181,7 @@ def _(rid, params: dict) -> dict: }, ) - text = f"[User attached file: {drop_path}]" + ( - f"\n{remainder}" if remainder else "" - ) + text = f"[User attached file: {drop_path}]" + (f"\n{remainder}" if remainder else "") return _ok( rid, { @@ -1461,6 +1196,27 @@ def _(rid, params: dict) -> dict: return _err(rid, 5027, str(e)) +# ── side agents (background / btw / preview.restart) ──────────────────────── + + +@contextlib.contextmanager +def _session_profile_home_scope(session): + """Bind the session's HERMES_HOME override for an ephemeral agent thread: the + ContextVar set on the session-create thread doesn't propagate, so a turn under + a non-default profile would otherwise run against the wrong home.""" + profile_home = session.get("profile_home") + home_token = set_hermes_home_override(profile_home) if profile_home else None + try: + yield + finally: + if home_token is not None: + reset_hermes_home_override(home_token) + + +def _final_response_text(result) -> str: + return (result.get("final_response", str(result)) if isinstance(result, dict) else str(result)) + + @method("prompt.background") def _(rid, params: dict) -> dict: session, err = _sess(params, rid) @@ -1476,46 +1232,19 @@ def _(rid, params: dict) -> dict: try: from run_agent import AIAgent - # Bug #50233: ephemeral agent threads don't inherit the session's - # HERMES_HOME override (the ContextVar set on the session-create - # thread doesn't propagate here), so a background turn under a - # non-default profile would run against the wrong home. Re-bind the - # override for the duration of this turn, exactly as the normal - # prompt turn does, and restore it afterward. - _profile_home_str = session.get("profile_home") - home_token = ( - set_hermes_home_override(_profile_home_str) - if _profile_home_str - else None - ) - try: + with _session_profile_home_scope(session): result = AIAgent( **_background_agent_kwargs(session["agent"], task_id) ).run_conversation( user_message=text, task_id=task_id, ) - finally: - if home_token is not None: - reset_hermes_home_override(home_token) _emit( - "background.complete", - parent, - { - "task_id": task_id, - "text": ( - result.get("final_response", str(result)) - if isinstance(result, dict) - else str(result) - ), - }, + "background.complete", parent, + {"task_id": task_id, "text": _final_response_text(result)}, ) except Exception as e: - _emit( - "background.complete", - parent, - {"task_id": task_id, "text": f"error: {e}"}, - ) + _emit("background.complete", parent, {"task_id": task_id, "text": f"error: {e}"}) finally: _clear_session_context(session_tokens) @@ -1525,14 +1254,10 @@ def _(rid, params: dict) -> dict: @method("prompt.btw") def _(rid, params: dict) -> dict: - """Answer a side question about the session without touching its history. - - Snapshots the live conversation (in-flight ``_session_messages`` when a - turn is running, else the persisted ``session["history"]``) and runs a - one-shot auxiliary LLM call against it (``agent/side_question.py``). The - session's history, role alternation, and prompt cache are untouched; the - answer arrives as a ``btw.complete`` event. - """ + """Answer a side question without touching session history: snapshot the live + conversation (in-flight ``_session_messages`` else ``session["history"]``) and + run a one-shot auxiliary call (``agent/side_question.py``). History, role + alternation and prompt cache stay untouched; answer arrives as ``btw.complete``.""" session, err = _sess(params, rid) if err: return err @@ -1542,16 +1267,10 @@ def _(rid, params: dict) -> dict: task_id = f"btw_{uuid.uuid4().hex[:6]}" agent = session.get("agent") - snapshot = list( - getattr(agent, "_session_messages", None) - or session.get("history") - or [] - ) + snapshot = list(getattr(agent, "_session_messages", None) or session.get("history") or []) main_runtime = { - "model": getattr(agent, "model", None), - "provider": getattr(agent, "provider", None), - "base_url": getattr(agent, "base_url", None), - "api_key": getattr(agent, "api_key", None), + "model": getattr(agent, "model", None), "provider": getattr(agent, "provider", None), + "base_url": getattr(agent, "base_url", None), "api_key": getattr(agent, "api_key", None), "api_mode": getattr(agent, "api_mode", None), } @@ -1560,31 +1279,17 @@ def _(rid, params: dict) -> dict: try: from agent.side_question import answer_side_question - _profile_home_str = session.get("profile_home") - home_token = ( - set_hermes_home_override(_profile_home_str) - if _profile_home_str - else None - ) - try: + with _session_profile_home_scope(session): answer = answer_side_question( - text, - snapshot, - parent_agent=agent, - main_runtime=main_runtime, + text, snapshot, parent_agent=agent, main_runtime=main_runtime, ) - finally: - if home_token is not None: - reset_hermes_home_override(home_token) _emit( - "btw.complete", - parent, + "btw.complete", parent, {"task_id": task_id, "question": text, "text": answer or ""}, ) except Exception as e: _emit( - "btw.complete", - parent, + "btw.complete", parent, {"task_id": task_id, "question": text, "text": f"error: {e}"}, ) finally: @@ -1594,6 +1299,28 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"task_id": task_id}) +_PREVIEW_RESTART_RULES = ( + "Restart exactly the app intended for the Preview URL, not Hermes Desktop itself.", + "The Preview URL and port are the target. Preserve that target unless you conclude it is impossible.", + "If the prior conversation shows a specific command that bound this URL/port, prefer re-running THAT exact command (in the same cwd) over guessing a new one.", + "First inspect what process, if any, owns the Preview URL port. If a stale server exists, inspect its cwd and prefer that cwd over the Hermes/Desktop process cwd.", + "The Current working directory is only a hint. Do not assume it is the preview app root when the port owner or files indicate another root.", + "If the console shows a module-script MIME error for src/main.tsx or similar, a static server is serving source files. Do not restart python -m http.server or any dumb static server for that app.", + "For module-script MIME failures, inspect package.json/vite config in the candidate app root and start the real dev server/bundler (for example npm/pnpm/yarn dev) so module transforms happen.", + "Before declaring success, verify the Preview URL responds with the intended app, not Hermes Desktop. If it serves Hermes/Desktop UI or another unrelated app, stop that process and report failure.", + "Do not modify files. Do not ask the user unless blocked.", + "Prefer existing project scripts or commands when they are clear.", + "If a stale process owns the needed port, handle it safely.", + "Start long-running servers detached/in the background, then return immediately.", + "Do not run a foreground dev server command that blocks this background task.", + "Keep the final response short: what command/server was started, or why it could not be restarted.", +) + +_PREVIEW_RESTART_HISTORY_NOTE = ( + "The conversation history above is from the user's main session — including the commands you (the assistant) previously ran to start servers, edit files, or check ports. Use it to figure out exactly which server should be running at this Preview URL. The user did not start a brand new task; recover what they had working." +) + + @method("preview.restart") def _(rid, params: dict) -> dict: session, err = _sess(params, rid) @@ -1610,42 +1337,21 @@ def _(rid, params: dict) -> dict: task_id = f"preview_{uuid.uuid4().hex[:6]}" parent = params.get("session_id", "") parent_history = _preview_restart_history(session) - has_history = bool(parent_history) prompt = "\n".join( line for line in [ "The desktop preview pane cannot load a local server URL.", - "", f"Preview URL: {url}", f"Current working directory: {cwd or '(unknown)'}", - "", f"Preview console:\n{context}" if context else "", - "" if context else "", - ( - "The conversation history above is from the user's main session — including the commands you (the assistant) previously ran to start servers, edit files, or check ports. Use it to figure out exactly which server should be running at this Preview URL. The user did not start a brand new task; recover what they had working." - if has_history - else None - ), - "Restart exactly the app intended for the Preview URL, not Hermes Desktop itself.", - "The Preview URL and port are the target. Preserve that target unless you conclude it is impossible.", - "If the prior conversation shows a specific command that bound this URL/port, prefer re-running THAT exact command (in the same cwd) over guessing a new one.", - "First inspect what process, if any, owns the Preview URL port. If a stale server exists, inspect its cwd and prefer that cwd over the Hermes/Desktop process cwd.", - "The Current working directory is only a hint. Do not assume it is the preview app root when the port owner or files indicate another root.", - "If the console shows a module-script MIME error for src/main.tsx or similar, a static server is serving source files. Do not restart python -m http.server or any dumb static server for that app.", - "For module-script MIME failures, inspect package.json/vite config in the candidate app root and start the real dev server/bundler (for example npm/pnpm/yarn dev) so module transforms happen.", - "Before declaring success, verify the Preview URL responds with the intended app, not Hermes Desktop. If it serves Hermes/Desktop UI or another unrelated app, stop that process and report failure.", - "Do not modify files. Do not ask the user unless blocked.", - "Prefer existing project scripts or commands when they are clear.", - "If a stale process owns the needed port, handle it safely.", - "Start long-running servers detached/in the background, then return immediately.", - "Do not run a foreground dev server command that blocks this background task.", - "Keep the final response short: what command/server was started, or why it could not be restarted.", + _PREVIEW_RESTART_HISTORY_NOTE if parent_history else None, + *_PREVIEW_RESTART_RULES, ] if line ) - # Normalize defensively: a malformed client path (embedded NUL, etc.) must - # not blow up the whole restart — treat it as "no validated cwd". + # A malformed client path (embedded NUL, etc.) must not blow up the restart: + # treat it as "no validated cwd". try: preview_cwd = os.path.abspath(os.path.expanduser(cwd)) if cwd else "" if preview_cwd and not os.path.isdir(preview_cwd): @@ -1655,7 +1361,7 @@ def _(rid, params: dict) -> dict: def run(): # Pin the validated preview cwd, else the parent workspace — never an - # invalid client path, which would silently fall back to the launch dir. + # invalid client path (which would silently fall back to the launch dir). session_tokens = _set_session_context(task_id, cwd=(preview_cwd or _session_cwd(session))) try: from run_agent import AIAgent @@ -1670,49 +1376,26 @@ def _(rid, params: dict) -> dict: else "" ) _emit( - "preview.restart.progress", - parent, + "preview.restart.progress", parent, {"task_id": task_id, "text": f"Starting hidden restart agent{history_note}"}, ) - # Bug #50233: ephemeral preview-restart agent threads don't inherit - # the session's HERMES_HOME override (the ContextVar set on the - # session-create thread doesn't propagate here). Re-bind it for the - # duration of the turn, mirroring the normal prompt turn, then - # restore it. NOTE: we deliberately do NOT close this agent through - # task-wide process cleanup — the whole point of preview.restart is - # to leave a background server running under this task_id, and - # AIAgent.close() would kill every process for the task_id and tear - # down the very server the restart just started. - _profile_home_str = session.get("profile_home") - home_token = ( - set_hermes_home_override(_profile_home_str) - if _profile_home_str - else None - ) - try: + # Deliberately NOT closed through task-wide process cleanup: the whole + # point is to leave a background server running under this task_id, + # and AIAgent.close() would kill every process for it. + with _session_profile_home_scope(session): result = AIAgent( **_ephemeral_preview_agent_kwargs(session["agent"], task_id), **_preview_restart_callbacks(parent, task_id), ).run_conversation( - user_message=prompt, - task_id=task_id, + user_message=prompt, task_id=task_id, conversation_history=parent_history or None, ) - finally: - if home_token is not None: - reset_hermes_home_override(home_token) - text = ( - result.get("final_response", str(result)) - if isinstance(result, dict) - else str(result) - ) - _emit("preview.restart.complete", parent, {"task_id": task_id, "text": text}) - except Exception as e: _emit( - "preview.restart.complete", - parent, - {"task_id": task_id, "text": f"error: {e}"}, + "preview.restart.complete", parent, + {"task_id": task_id, "text": _final_response_text(result)}, ) + except Exception as e: + _emit("preview.restart.complete", parent, {"task_id": task_id, "text": f"error: {e}"}) finally: try: from tools.terminal_tool import clear_task_env_overrides @@ -1726,31 +1409,24 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"task_id": task_id}) +# ── late-answer RPCs for tool-driven UI cards ─────────────────────────────── +# All use allow_expired=True: each tool's bounded wait (read_terminal 30s, +# setup_mcp 10min, clarify ...) can expire — its _pending entry popped — while the +# card is still visible (e.g. a WS reconnect dropped tool.complete). A late answer +# must resolve gracefully instead of the raw 4009 "no pending answer request". + + @method("clarify.respond") def _(rid, params: dict) -> dict: - # allow_expired=True: a clarify can time out server-side (its entry is popped - # from _pending) while the card is still visible — common when a WebSocket - # reconnect during the wait drops tool.complete. A late answer must resolve - # gracefully instead of hitting the raw 4009 "no pending answer request". if proxied := _respond_compute_host_clarify(rid, params): return proxied return _respond(rid, params, "answer", allow_expired=True) -# Late-answer RPCs for tool-driven UI cards. All use allow_expired=True: each -# tool's bounded wait (read_terminal 30s, setup_mcp 10min, ...) can expire while -# a slow renderer/page/OAuth round-trip is still in flight, and a late answer -# must resolve gracefully instead of surfacing the raw 4009 "no pending answer -# request". ``text``/``result`` carry a JSON string of the card's outcome. _LATE_RESPOND_KEYS = { - "terminal.read.respond": "text", - "preview.read.respond": "text", - "preview.act.respond": "text", - "window.read.respond": "text", - "tour.respond": "text", - "mcp.setup.respond": "result", - "sudo.respond": "password", - "secret.respond": "value", + "terminal.read.respond": "text", "preview.read.respond": "text", "preview.act.respond": "text", + "window.read.respond": "text", "tour.respond": "text", "mcp.setup.respond": "result", + "sudo.respond": "password", "secret.respond": "value", } @@ -1766,17 +1442,27 @@ for _name, _key in _LATE_RESPOND_KEYS.items(): del _name, _key +# ── approvals ─────────────────────────────────────────────────────────────── + + +def _approval_reply(rid, result_key, call): + """``_ok({result_key: call(tools.approval)})``, 5004 on any failure.""" + try: + import tools.approval as approval + + return _ok(rid, {result_key: call(approval)}) + except Exception as e: + return _err(rid, 5004, str(e)) + + @method("approval.pending") def _(rid, params: dict) -> dict: session, err = _sess(params, rid) if err: return err - try: - from tools.approval import list_gateway_approvals - - return _ok(rid, {"approvals": list_gateway_approvals(session["session_key"])}) - except Exception as e: - return _err(rid, 5004, str(e)) + return _approval_reply( + rid, "approvals", lambda a: a.list_gateway_approvals(session["session_key"]) + ) @method("approval.received") @@ -1787,31 +1473,17 @@ def _(rid, params: dict) -> dict: request_id = params.get("request_id") if not isinstance(request_id, str) or not request_id: return _err(rid, 4006, "request_id required") - try: - from tools.approval import ack_gateway_approval - - return _ok( - rid, - {"acknowledged": ack_gateway_approval(session["session_key"], request_id)}, - ) - except Exception as e: - return _err(rid, 5004, str(e)) + return _approval_reply( + rid, "acknowledged", lambda a: a.ack_gateway_approval(session["session_key"], request_id), + ) def _approval_respond_session_fallback(params: dict): - """Durable-identity fallback for ``approval.respond`` (#91684). - - The desktop can answer an approval prompt with a stale live sid (its - runtime record was re-minted after a reconnect while the prompt stayed - on screen). Before failing with 4001, try resolving the target session: - - 1. by the approval ``request_id`` — unique across sessions — scanning - every live session's pending gateway approvals; - 2. by treating ``session_id`` as a STORED session id and mapping it to - the live runtime record for that stored id. - - Returns the live session record or None. - """ + """Durable-identity fallback for ``approval.respond``: the desktop can answer + with a stale live sid (runtime re-minted after a reconnect while the prompt + stayed on screen). Try (1) the approval ``request_id`` — unique across sessions + — against every live session's pending approvals, then (2) ``session_id`` as a + STORED id mapped to its live record. Returns the live session or None.""" request_id = str(params.get("request_id") or "") if request_id: try: @@ -1827,9 +1499,7 @@ def _approval_respond_session_fallback(params: dict): if str(pending.get("request_id") or "") == request_id: return session except Exception: - logger.debug( - "approval.respond request_id fallback failed", exc_info=True - ) + logger.debug("approval.respond request_id fallback failed", exc_info=True) target = str(params.get("session_id") or "") if target: try: @@ -1837,9 +1507,7 @@ def _approval_respond_session_fallback(params: dict): if live is not None: return live[1] except Exception: - logger.debug( - "approval.respond stored-id fallback failed", exc_info=True - ) + logger.debug("approval.respond stored-id fallback failed", exc_info=True) return None @@ -1847,60 +1515,24 @@ def _approval_respond_session_fallback(params: dict): def _(rid, params: dict) -> dict: session, err = _sess(params, rid) if err: - # Session-not-found (4001) only: the client may hold a stale live - # sid for a session whose runtime was re-minted after a reconnect. - # Resolve by durable identity before failing (#91684). + # Session-not-found (4001) only: resolve by durable identity before failing. code = (err.get("error") or {}).get("code") if code != 4001: return err session = _approval_respond_session_fallback(params) if session is None: return err - try: - from tools.approval import resolve_gateway_approval - - return _ok( - rid, - { - "resolved": resolve_gateway_approval( - session["session_key"], - params.get("choice", "deny"), - resolve_all=params.get("all", False), - request_id=params.get("request_id"), - ) - }, - ) - except Exception as e: - return _err(rid, 5004, str(e)) + return _approval_reply( + rid, "resolved", + lambda a: a.resolve_gateway_approval( + session["session_key"], + params.get("choice", "deny"), + resolve_all=params.get("all", False), + request_id=params.get("request_id"), + ), + ) def register(server) -> None: - """Bind this module's handlers onto ``server``'s globals and registry.""" - _registry.install(server) - # Module-level helpers aren't @method handlers, so install() doesn't see - # them. Rebind onto server globals so handler bodies (and server.py call - # sites) resolve the same free names after the split. - g = vars(server) - for helper in ( - _history_user_indices, - _message_row_id, - _mem_db_pair_agrees, - _find_user_turn_by_row_id, - _load_durable_truncation_history, - _resolve_truncate_row_id, - _coerce_truncate_int, - _reconcile_client_ordinal, - _pending_reaction_notes, - _approval_respond_session_fallback, - ): - setattr( - server, - helper.__name__, - types.FunctionType( - helper.__code__, - g, - helper.__name__, - helper.__defaults__, - helper.__closure__, - ), - ) + """Publish this module's helpers + handlers onto ``server``, rebound to its globals.""" + bind_module(globals(), server, skip=("_",)) diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 23d4a79bbd..c0b3d0944a 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -1,10 +1,9 @@ """Session / delegation / spawn-tree / billing / pet JSON-RPC handlers. -Handler bodies are rebound onto server.py's globals at install time (see -method_ctx.py), so they reference server helpers (``_sessions``, ``_ok``, -``_err``, ...) bare. Module-level helpers defined here are published onto -server.py by :func:`register` the same way, so handlers and helpers share one -namespace (and tests that monkeypatch ``server.X`` still intercept). +Bodies are rebound onto server.py's globals at install time (method_ctx.py), so +they use server helpers (``_sessions``, ``_ok``, ``_err``, ...) bare; module-level +helpers are published onto server.py the same way (tests monkeypatching ``server.X`` +still intercept). """ import contextlib @@ -20,16 +19,25 @@ _profile_scoped = _registry.profile_scoped # ── shared handler plumbing ────────────────────────────────────────── -def _with_session(fn): - """Resolve ``params.session_id`` via ``_sess_nowait`` and pass the record as a 3rd arg.""" +def _session_arg(resolve): + """Decorator factory: resolve ``params.session_id`` with ``resolve`` and pass the record as a 3rd arg. - def handler(rid, params: dict) -> dict: - session, err = _sess_nowait(params, rid) - if err: - return err - return fn(rid, params, session) + ``resolve`` is a lambda over the server helper (import-time decoration runs before + bind_module publishes ``_sess``/``_sess_nowait``; the lambda is rebound with the handler). + """ - return handler + def deco(fn): + def handler(rid, params: dict) -> dict: + session, err = resolve(params, rid) + if err: + return err + return fn(rid, params, session) + return handler + return deco + + +_with_session = _session_arg(lambda params, rid: _sess_nowait(params, rid)) # no agent-build wait +_with_live_session = _session_arg(lambda params, rid: _sess(params, rid)) # waits for the agent build def _with_session_db(code: int): @@ -44,24 +52,10 @@ def _with_session_db(code: int): if db is None: return _db_unavailable_error(rid, code=code) return fn(rid, params, session, db) - return handler - return deco -def _with_live_session(fn): - """Like :func:`_with_session` but via ``_sess`` (waits for the agent build).""" - - def handler(rid, params: dict) -> dict: - session, err = _sess(params, rid) - if err: - return err - return fn(rid, params, session) - - return handler - - def _new_runtime_ids(params: dict) -> tuple[str, str]: """Fresh runtime sid + resolved DB ``source`` for a session minted from ``params``.""" return (uuid.uuid4().hex[:8], _resolve_session_source(str(params.get("source") or "").strip() or None)) @@ -71,9 +65,8 @@ def _new_runtime_ids(params: dict) -> tuple[str, str]: def _profile_build_scope(profile_home): """Bind HERMES_HOME + the profile's secret scope while building/initializing an agent. - The home override alone only moves config/skills/memory; credentials resolve - through get_secret(), which without a scope falls through to the LAUNCH - profile's .env — so both are installed together. No-op for the launch profile. + The home override alone only moves config/skills/memory; get_secret() without a + scope falls through to the LAUNCH profile's .env. No-op for the launch profile. """ if not profile_home: yield @@ -87,6 +80,15 @@ def _profile_build_scope(profile_home): reset_secret_scope(secret_token) +def _make_agent_in_context(sid: str, key: str, **kwargs): + """``_make_agent`` with the session context bound for the build and cleared after.""" + tokens = _set_session_context(key) + try: + return _make_agent(sid, key, session_id=key, **kwargs) + finally: + _clear_session_context(tokens) + + def _branch_title(db, parent_key: str) -> str: """Next title in the parent's lineage (mirrors the TUI /branch naming).""" current = db.get_session_title(parent_key) or "branch" @@ -111,24 +113,19 @@ def _cwd_info(session: dict, cwd: str, branch=None) -> dict: def _session_row_summary(row: dict, *, tip_row: dict | None = None, resolved_id=None) -> dict: """Compact session.list row; ``tip_row``/``resolved_id`` come from the compression tip.""" tip_row = tip_row or row - out = {"id": row["id"]} - if resolved_id is not None: - out["resolved_id"] = resolved_id - out.update( - { - "title": row.get("title") or "", - "preview": tip_row.get("preview") or "", - "started_at": row.get("started_at") or 0, - "message_count": tip_row.get("message_count") or 0, - "source": row.get("source") or "", - } - ) - return out + return { + "id": row["id"], + **({} if resolved_id is None else {"resolved_id": resolved_id}), + "title": row.get("title") or "", + "preview": tip_row.get("preview") or "", + "started_at": row.get("started_at") or 0, + "message_count": tip_row.get("message_count") or 0, + "source": row.get("source") or "", + } -# Sources hidden from human-facing listings: ``tool`` sub-agent runs and -# ``kanban`` dispatcher workers. A deny-list (not an allow-list) so new -# platforms / custom HERMES_SESSION_SOURCE values surface automatically. +# Hidden from human-facing listings (sub-agent runs, kanban workers). A deny-list so +# new platforms / custom HERMES_SESSION_SOURCE values surface automatically. _LISTING_DENY_SOURCES = frozenset({"kanban", "tool"}) @@ -136,6 +133,15 @@ def _denied_source(row: dict) -> bool: return (row.get("source") or "").strip().lower() in _LISTING_DENY_SOURCES +def _snapshot_sessions(rid): + """``(list(_sessions.items()), None)`` under the lock, or ``(None, 5036 error)`` — fail CLOSED.""" + try: + with _sessions_lock: + return list(_sessions.items()), None + except Exception as e: + return None, _err(rid, 5036, f"could not enumerate active sessions: {e}") + + def _pet_display_cfg() -> dict: """``display.pet`` config block, ``{}`` when config is unreadable.""" try: @@ -148,11 +154,8 @@ def _pet_display_cfg() -> dict: def _pet_guard(name: str, *, fail_open=None): - """Wrap a pet handler so any exception is logged at debug and never breaks the surface. - - ``fail_open`` is the result payload to return (``pet.info`` style); without it - the caller gets ``_err(5031, " failed: ...")``. - """ + """Pet handlers never break the surface: exceptions log at debug and yield ``fail_open`` + (payload or ``params -> payload`` callable) or, without it, ``_err(5031, " failed: ...")``.""" def deco(fn): def handler(rid, params: dict) -> dict: @@ -163,18 +166,55 @@ def _pet_guard(name: str, *, fail_open=None): if fail_open is not None: return _ok(rid, fail_open(params) if callable(fail_open) else dict(fail_open)) return _err(rid, 5031, f"{name} failed: {exc}") - return handler - return deco -def _billing_call(rid, fn, extra: dict | None = None) -> dict: - """Run a portal call; typed BillingError → serialized envelope, anything else → generic. +def _pet_emit(event: str, payload: dict, what: str) -> None: + """Best-effort progress emit: a transport hiccup must never abort generation.""" + try: + _emit(event, "", payload) + except Exception as exc: # noqa: BLE001 + logger.debug("%s emit failed: %s", what, exc) - ``extra`` is appended to both ERROR envelopes (e.g. the idempotency key the - TUI reuses on retry); the success payload is whatever ``fn`` returns. - """ + +def _pet_gen_abort(rid, token: str, code: int, message: str) -> dict: + """Release the cancel arm for ``token`` and return ``_err``.""" + _pet_cancel_release(token) + return _err(rid, code, message) + + +def _with_slug(fn): + """Require ``params.slug`` (4004 "missing slug") and pass it as a 3rd arg.""" + + def handler(rid, params: dict) -> dict: + slug = str(params.get("slug") or "").strip() + if not slug: + return _err(rid, 4004, "missing slug") + return fn(rid, params, slug) + return handler + + +def _pet_method(name: str, *, fail_open=None, slug: bool = False, scoped: bool = True): + """``@method(name)`` + ``@_profile_scoped`` (unless ``scoped=False``) + ``_pet_guard`` (+ ``_with_slug``).""" + + def deco(fn): + handler = _pet_guard(name, fail_open=fail_open)(_with_slug(fn) if slug else fn) + return method(name)(_profile_scoped(handler) if scoped else handler) + return deco + + +def _active_pet(): + """``(pet, scale)`` when the pet display is enabled and the pet exists, else None.""" + enabled, pet, scale = _pet_active_selection() + if not enabled or pet is None or not pet.exists: + return None + return pet, scale + + +def _billing_call(rid, fn, extra: dict | None = None) -> dict: + """Run a portal call; BillingError → serialized envelope, anything else → generic. + ``extra`` rides both ERROR envelopes (e.g. the idempotency key the TUI reuses on retry).""" from hermes_cli.nous_billing import BillingError try: return _ok(rid, fn()) @@ -188,26 +228,28 @@ def _billing_invalid(rid, message: str, error: str = "invalid_request") -> dict: return _ok(rid, {"ok": False, "error": error, "message": message}) +def _billing_pick(result: dict, **fields) -> dict: + """``{"ok": True, : result[], ...}`` in ``fields`` order.""" + return {"ok": True, **{key: result.get(src) for key, src in fields.items()}} + + +def _billing_pending_change(result: dict) -> dict: + return {"ok": True, "message": result.get("message"), "payload": result} + + # ── session.create / list / most_recent / facts ────────────────────── def _create_branch_row(db, new_key: str, parent_key: str, *, source, cwd, profile_name) -> None: """Create a branch child row. - ``_branched_from`` is the stable marker that keeps the branch visible in - list_sessions_rich(): the TUI branch leaves the parent live (no - end_reason='branched'), so the legacy end_reason heuristic never matches it. - ``profile_name`` is stamped explicitly (not just parent-backfill) — NULL rows - drop out of profile-keyed sidebar matching and deep-link resolution. + ``_branched_from`` keeps the branch visible in list_sessions_rich() (the parent stays + live, so the legacy end_reason='branched' heuristic never matches). ``profile_name`` is + stamped explicitly — NULL rows drop out of profile-keyed sidebar matching / deep links. """ db.create_session( - new_key, - source=source, - model=_resolve_model(), - model_config={"_branched_from": parent_key}, - parent_session_id=parent_key, - cwd=cwd, - profile_name=profile_name, + new_key, source=source, model=_resolve_model(), model_config={"_branched_from": parent_key}, + parent_session_id=parent_key, cwd=cwd, profile_name=profile_name, ) @@ -231,11 +273,10 @@ def _copy_branch_transcript(db, new_key: str, title: str, history: list, copy_fi def _seed_branch_row(sid: str, key: str, parent_session_id: str, history: list, source: str, profile_home) -> None: """Persist a seeded desktop branch child up front (the one session.create exception to lazy rows). - A branch carries parent_session_id AND a seeded transcript — explicit intent, not - an abandoned draft. The renderer's post-create resume re-fetches the child via REST - + defer_history hydration, both of which read the DB, so an unpersisted child 404s - and the fail-latch spins forever. Best-effort: on failure the lazy first-prompt - path stays as the fallback, exactly as for plain drafts. + parent_session_id + seeded transcript is explicit intent, not a draft; the renderer's + post-create resume re-fetches the child via REST/defer_history (DB reads), so an + unpersisted child 404s and the fail-latch spins forever. Best-effort: on failure the + lazy first-prompt path is the fallback, as for plain drafts. """ try: with _session_db(_sessions[sid]) as db: @@ -243,19 +284,14 @@ def _seed_branch_row(sid: str, key: str, parent_session_id: str, history: list, return branch_title = _branch_title(db, parent_session_id) _create_branch_row( - db, - key, - parent_session_id, - source=source, - cwd=_sessions[sid]["cwd"], + db, key, parent_session_id, source=source, cwd=_sessions[sid]["cwd"], profile_name=(Path(profile_home).name if profile_home else None), ) try: _copy_branch_transcript(db, key, branch_title, history) except Exception as exc: - # Compensation: the row committed but the transcript copy / title - # write failed; a durable-but-empty row would defeat the INSERT OR - # IGNORE first-prompt seed. Roll back just this child so it can retry. + # Row committed but transcript/title failed: a durable-but-empty row would + # defeat the INSERT OR IGNORE first-prompt seed — roll back this child. from hermes_state import is_disk_full_error if is_disk_full_error(exc): raise @@ -267,19 +303,16 @@ def _seed_branch_row(sid: str, key: str, parent_session_id: str, history: list, _sessions[sid]["pending_title"] = None except Exception: logger.warning( - "seeded-branch persistence failed for %s; falling back to lazy row creation", - key, - exc_info=True, + "seeded-branch persistence failed for %s; falling back to lazy row creation", key, exc_info=True, ) def _create_overrides(params: dict) -> tuple: """(model_override, reasoning_override, service_tier_override) from the composer's UI state. - PER-SESSION overrides only — never a global config write, so picking a model - for a new chat can't mutate the profile default. provider is optional (resolved - at build). ``fast`` presence is the contract: omitted inherits the profile, true - pins priority, false pins normal ("" — _make_agent uses None for inheritance). + PER-SESSION only — never a global config write. provider resolves at build. + ``fast`` presence is the contract: omitted inherits, true pins priority, false pins + normal ("" — _make_agent uses None for inheritance). """ create_model = str(params.get("model") or "").strip() model_override = ( @@ -307,27 +340,23 @@ def _(rid, params: dict) -> dict: cols = int(params.get("cols", 80)) history = _coerce_seed_history(params.get("messages")) title = str(params.get("title") or "").strip() - # A branch: copies an existing conversation and links back so list_sessions_rich - # keeps it visible and the sidebar nests it (mirrors the TUI /branch marker). + # Branch: links back so list_sessions_rich keeps it visible and the sidebar nests it. parent_session_id = str(params.get("parent_session_id") or "").strip() or None - # Only an explicitly chosen (existing) workspace is persisted as the session's - # cwd (_ensure_session_db_row); the gateway launch dir fallback lands in "No workspace". + # Only an explicitly chosen existing workspace persists as cwd (_ensure_session_db_row); + # the launch-dir fallback lands in "No workspace". raw_cwd = str(params.get("cwd") or "").strip() - try: + explicit_cwd = False + with contextlib.suppress(Exception): explicit_cwd = bool(raw_cwd) and os.path.isdir(os.path.abspath(os.path.expanduser(raw_cwd))) - except Exception: - explicit_cwd = False resolved_cwd = _completion_cwd(params) source = _resolve_session_source(str(params.get("source") or "").strip() or None) _enable_gateway_prompts() - # ``profile`` (app-global remote mode): build + persist against THAT profile's - # home/state.db. Stored on the session so the build and every turn re-bind HERMES_HOME. + # ``profile`` (app-global remote mode): build + persist against THAT profile's home; + # stored on the session so the build and every turn re-bind HERMES_HOME. profile = (params.get("profile") or "").strip() or None profile_home = _profile_home(profile) - session_model_override, create_reasoning_override, create_service_tier_override = _create_overrides(params) - now = time.time() with _sessions_lock: _sessions[sid] = { @@ -368,18 +397,14 @@ def _(rid, params: dict) -> dict: } _register_session_cwd(_sessions[sid]) - # No DB row here: every launch/draft opens a session just to paint the - # composer, and eager rows left "Untitled" litter. The row is created lazily - # on the first prompt (_ensure_session_db_row + prompt.submit) — except for - # seeded branch children, which must exist immediately. + # No DB row here: drafts opened just to paint the composer left "Untitled" litter. The + # row is created on the first prompt — except seeded branch children (must exist now). if parent_session_id and history: _seed_branch_row(sid, key, parent_session_id, history, source, profile_home) - # Return immediately so Ink can paint; the real AIAgent builds right after - # this response is flushed (no first prompt needed to hydrate tools/skills). + # Return immediately so Ink can paint; the AIAgent builds right after the flush. _schedule_agent_build(sid) _schedule_session_cap_enforcement() # trim detached idle sessions over the cap - return _ok( rid, { @@ -388,8 +413,7 @@ def _(rid, params: dict) -> dict: "message_count": len(history), "messages": _history_to_messages(history), "info": { - # Reflect the per-session model override immediately so the client - # doesn't briefly clobber its sticky pick with the global default. + # Reflect the override now so the client doesn't clobber its sticky pick. "model": ( session_model_override.get("model") if session_model_override else _resolve_model() ), @@ -412,30 +436,25 @@ def _(rid, params: dict) -> dict: def _session_list_by_title(rid, db, title_lookup: str) -> dict: - """EXACT-title registry lookup (not a listing) for callers that treat a title as identity. + """EXACT-title registry lookup for callers that treat a title as identity. - Window-free on purpose: a busy profile's recency-windowed listing can push the - row out, so scanning ``session.list`` output is not a reliable identity lookup. - Hidden rows resolve (canonical chats are born hidden); archived rows and - deny-listed sources do not; compression lineages resolve to the live tip - (``resolved_id``), mirroring profiles.list's canonical_session resolver. + Window-free on purpose (a busy profile's windowed listing can push the row out). + Hidden rows resolve (canonical chats are born hidden); archived / deny-listed do not; + lineages resolve to the live tip (``resolved_id``), as profiles.list's canonical_session. """ row = db.get_session_by_title(title_lookup) if row and row.get("archived"): from tools.bot_mode_probe import BOT_CHAT_TITLE - # The canonical Bot Chat is identity-scoped: an archive stamped by the - # ws-orphan reaper / agent_close is an accident, and hiding it makes the - # desktop mint transient replacements forever. Resurrect recoverable - # reasons only; deliberate archives still hide. Re-fetch by ID — title - # has no DB-level UNIQUE, so a title re-query could grab a duplicate. + # A Bot Chat archived by the ws-orphan reaper / agent_close is an accident (hiding + # it makes the desktop mint replacements forever): resurrect recoverable reasons + # only. Re-fetch by ID — title has no DB UNIQUE, a title re-query could grab a dup. if title_lookup == BOT_CHAT_TITLE and db.unarchive_recoverable_session(row["id"]): row = db.get_session(row["id"]) if not row or row.get("archived") or _denied_source(row): return _ok(rid, {"sessions": []}) try: - # Only a real compression continuation: the generic resume resolver's - # legacy unmarked-child fallback could redirect the canonical Bot Chat - # to an unrelated normal child. + # Real compression continuation only: the resume resolver's unmarked-child + # fallback could redirect the canonical Bot Chat to an unrelated child. tip = db.get_compression_tip(row["id"]) or row["id"] except Exception: tip = row["id"] @@ -449,26 +468,19 @@ def _(rid, params: dict) -> dict: if db is None: return _db_unavailable_error(rid, code=5006) try: - # Older clients never send ``title``; newer clients on older gateways - # just get the windowed listing back and scan it. + # Older clients never send ``title``; newer ones on old gateways scan the listing. title_lookup = str(params.get("title") or "").strip() if title_lookup: return _session_list_by_title(rid, db, title_lookup) - limit = int(params.get("limit", 200) or 200) - # ``include_hidden``: only for surfaces that OWN hidden sessions (Bots - # pane, plugin pickers); off for the resume picker and every global caller. + # ``include_hidden``: only for surfaces that OWN hidden sessions (Bots pane, pickers). include_hidden = is_truthy_value(params.get("include_hidden", False)) - # Over-fetch so per-source filtering (and tip projection merging in - # list_sessions_rich) doesn't leave us short. + # Over-fetch: per-source filtering + tip merging must not leave us short. fetch_limit = max(limit * 2, 200) rows = [ s for s in db.list_sessions_rich( - source=None, - limit=fetch_limit, - order_by_last_active=True, - compact_rows=True, + source=None, limit=fetch_limit, order_by_last_active=True, compact_rows=True, include_hidden=include_hidden, ) if not _denied_source(s) @@ -480,18 +492,13 @@ def _(rid, params: dict) -> dict: @method("session.most_recent") def _(rid, params: dict) -> dict: - """Most recent human-facing session id, or ``None`` (same deny-list as session.list). - - ``{"session_id": null}`` means "no eligible session right now"; errors fold - into that shape (and log) so callers never special-case error envelopes. - Honors ``params.profile`` (mirrors ``session.resume``). - """ + """Most recent human-facing session id (same deny-list as session.list), honoring ``params.profile``. + Errors fold into ``{"session_id": null}`` (and log) so callers never special-case envelopes.""" with _profile_db(params) as db: if db is None: return _ok(rid, {"session_id": None}) try: - # Generous over-fetch so heavy sub-agent users (many ``tool`` rows) - # don't get a false "no eligible session". + # Generous over-fetch: many ``tool`` rows must not yield a false "none". rows = db.list_sessions_rich(source=None, limit=200, order_by_last_active=True, compact_rows=True) for row in rows: if _denied_source(row): @@ -513,11 +520,8 @@ def _(rid, params: dict) -> dict: @method("project.facts") def _(rid, params: dict) -> dict: - """Project facts for a cwd (manifests, package manager, verify commands, context files). - - Same detection the coding-context posture bakes into the system prompt, - exposed so UIs consume it instead of re-sniffing. ``{"facts": null}`` = not a code workspace. - """ + """Project facts for a cwd — the coding-context detection the system prompt uses, so UIs + don't re-sniff. ``{"facts": null}`` = not a code workspace.""" try: from agent.coding_context import project_facts_for return _ok(rid, {"facts": project_facts_for(params.get("cwd"))}) @@ -529,19 +533,15 @@ def _(rid, params: dict) -> dict: @method("verification.status") @_profile_scoped def _(rid, params: dict) -> dict: - """Best known coding verification evidence for a cwd/session. - - Read-only consumer of the core ledger: never runs checks, never upgrades - targeted evidence into a repository-wide guarantee. - """ + """Best known verification evidence for a cwd/session. Read-only: never runs checks, + never upgrades targeted evidence into a repository-wide guarantee.""" try: from agent.verification_evidence import verification_status return _ok( rid, { "verification": verification_status( - session_id=params.get("session_id") or params.get("session_key"), - cwd=params.get("cwd"), + session_id=params.get("session_id") or params.get("session_key"), cwd=params.get("cwd"), ) }, ) @@ -553,16 +553,12 @@ def _(rid, params: dict) -> dict: # ── session.resume ─────────────────────────────────────────────────── -# repr/eq off: bind_module rebinds every function on the class onto server.py's -# globals, and dataclasses' generated __repr__ wrapper reads its own module globals. +# repr/eq off: dataclass-generated methods read their own module globals, which +# bind_module cannot rebind. @dataclass(repr=False, eq=False) class _Resume: - """Per-call state for ``session.resume`` shared by the path helpers below. - - ``owns_db`` tracks the DEDICATED profile-scoped handle: it is ours to close - (the handler's ``finally``) until a path hands it to the hydration worker or - the agent (``_init_session``), which flips it False. - """ + """Per-call ``session.resume`` state. ``owns_db``: the DEDICATED profile handle is ours + to close (handler ``finally``) until handed to the hydration worker or the agent.""" rid: object params: dict @@ -583,24 +579,23 @@ class _Resume: return self.profile_resume_cwd or _default_session_cwd() def record(self, source: str, cwd: str, history: list, **extra) -> dict: - """``_deferred_session_record`` with this resume's common fields; the active-session - lease is always claimed lazily on the first turn (_ensure_active_session_slot).""" + """``_deferred_session_record`` with this resume's common fields (lease claimed lazily on turn 1).""" return _deferred_session_record( - self.target, - cols=self.cols, - cwd=cwd, - history=history, - lease=None, - source=source, + self.target, cols=self.cols, cwd=cwd, history=history, lease=None, source=source, close_on_disconnect=is_truthy_value(self.params.get("close_on_disconnect", False)), - profile_home=self.profile_home, - explicit_cwd=bool(self.profile_resume_cwd), - **extra, + profile_home=self.profile_home, explicit_cwd=bool(self.profile_resume_cwd), **extra, ) def resume_failed(self, exc) -> dict: return _err(self.rid, 5000, f"resume failed: {exc}") + def info(self, cwd: str, overrides: dict) -> dict: + model_override = overrides.get("model_override") or {} + return _lazy_resume_info( + cwd, model=model_override.get("model") or "", provider=overrides.get("provider_override") or "", + profile=self.profile, + ) + def _find_live_unpersisted(needle: str, home) -> str: """Runtime sid of a live, not-yet-persisted session matched by stored key or pending title.""" @@ -616,13 +611,11 @@ def _find_live_unpersisted(needle: str, home) -> str: def _resume_live_unpersisted(ctx: _Resume, live_sid: str, live: dict) -> dict: - """Reattach a LIVE lazy session (no state.db row yet — every fresh Bot Chat). + """Reattach a LIVE lazy session with no state.db row yet (every fresh Bot Chat). - session.create persists no row until the first prompt, so a resume by stored - key / pending title for a never-messaged session lands here; a hard 404 killed - messaging for exactly the bots that had never spoken. A WS drop may have - sentinel-parked the record, so rebind the transport and cancel the armed - orphan-reap Timer or it fires against a client that is attached right now. + A hard 404 here killed messaging for bots that had never spoken. A WS drop may have + sentinel-parked the record: rebind the transport and cancel the armed orphan-reap + Timer or it fires against the client attached right now. """ if ctx.owns_db: with contextlib.suppress(Exception): @@ -654,12 +647,10 @@ def _resume_live_unpersisted(ctx: _Resume, live_sid: str, live: dict) -> dict: def _resume_adopt_stranded(ctx: _Resume) -> None: """Adopt a lineage stranded in the DEFAULT store into this profile's db (profile-scoped only). - Before session RPCs routed by their TARGET session, a profile bot's turns ran - on the focused tile's backend, so its canonical session accumulated in the - default profile's state.db; without adoption that chat 4001s forever. - Exact-id match ONLY: title lookup has no archived filter and bot titles - collide by design, so a title-matched donor could retire an UNRELATED - conversation. Never re-adopt an already-retired donor (two "canonical" clones). + Older builds ran a profile bot's turns on the focused tile's backend, so its canonical + session landed in the default state.db; without adoption that chat 4001s forever. + Exact-id match ONLY (bot titles collide by design — a title-matched donor could retire + an UNRELATED chat). Never re-adopt a retired donor (two "canonical" clones). """ try: default_db = _get_db() @@ -693,11 +684,9 @@ def _resume_locate(ctx: _Resume) -> dict | None: ctx.target = ctx.found["id"] return None if ctx.lazy and _child_run_active(ctx.target): - # Race: a watch window opened on a freshly-spawned subagent. The child - # relays `subagent.start` BEFORE its first run_conversation() flushes the - # DB row, so the row is momentarily missing (reliably on WSL2). The child - # is provably live, so proceed lazily with empty history — the live - # mirror streams the turn and the row exists by upgrade time. + # Watch window on a fresh subagent: `subagent.start` relays BEFORE the child's first + # DB flush, so the row is momentarily missing. Proceed lazily with empty history — + # the live mirror streams the turn and the row exists by upgrade time. ctx.found = {} return None live_sid = _find_live_unpersisted(ctx.target, ctx.profile_home) @@ -714,13 +703,10 @@ def _resume_locate(ctx: _Resume) -> dict | None: def _resume_follow_tip(ctx: _Resume) -> None: """Rebind a rotated-out parent id to its compression-continuation tip. - Auto-compression ends the session and forks a child; resuming the original - id would reload the parent transcript and lose the post-compression reply. - Resolving here also re-anchors the live fast path so a rotated live session - is reused (by its new key) instead of rebuilding a duplicate on the stale - parent. Skipped for lazy watch windows (they attach to the exact child). - Bot Chat stays on a proven compression edge so an unmarked side chat cannot - steal the open; other sessions keep the legacy unmarked-child walker. + Resuming the original id would reload the parent transcript and lose the + post-compression reply; resolving here also lets the live fast path reuse the + rotated session by its new key. Skipped for lazy watch windows (exact child). + Bot Chat follows proven compression edges only; others keep the unmarked-child walker. """ if not ctx.found or ctx.lazy: return @@ -740,12 +726,9 @@ def _resume_follow_tip(ctx: _Resume) -> None: def _resume_guard(ctx: _Resume) -> dict | None: """Refuse a runaway transcript before any history read (sessions.max_resume_messages). - Only the non-deferred, non-omitted resume reads the whole lineage; the - deferred Desktop resume, omit_messages resume and lazy watch load the TIP - segment only, so they are guarded tip-only (a full-lineage count rejected - exactly the well-compressed conversations compaction produces). Metadata - fallback keeps lightweight adaptor DBs compatible. Fails OPEN on guard errors - — only a genuine over-limit blocks. + Deferred / omit_messages / lazy paths load the TIP segment only, so they are guarded + tip-only (a full-lineage count rejected exactly the well-compressed conversations). + Metadata fallback keeps lightweight adaptor DBs compatible. Fails OPEN on guard errors. """ from hermes_state import SessionResumeTooLargeError, resolved_max_resume_messages guard_tip_only = ctx.lazy or ctx.omit_messages or (ctx.defer_history and not ctx.eager_build) @@ -769,26 +752,17 @@ def _resume_guard(ctx: _Resume) -> dict | None: def _resume_reuse_live(ctx: _Resume, sid: str, session: dict) -> dict: - """Reattach an already-live session under the resume lock. - - Holding the lock across the client-gone check, transport rebind and reap - cancel makes grace expiry atomic across every reuse path (slow-path claim - races discover a winner after releasing their own lock). - """ + """Reattach an already-live session under the resume lock: holding it across the + client-gone check, transport rebind and reap cancel makes grace expiry atomic.""" with _session_resume_lock: if _sessions.get(sid) is not session: return _err(ctx.rid, 4007, "session no longer live; retry resume") if session.get("_client_gone_interrupt_requested"): return _err(ctx.rid, 4009, "session disconnect interrupt settling") - # Cancel unconditionally (the payload's rebind only cancels when a - # transport is passed) so the fast path can never race the reap Timer. + # Cancel unconditionally so the fast path can never race the reap Timer. _cancel_ws_orphan_reap(sid) payload = _live_session_payload( - sid, - session, - cols=ctx.cols, - touch=True, - transport=current_transport() or _stdio_transport, + sid, session, cols=ctx.cols, touch=True, transport=current_transport() or _stdio_transport, omit_messages=ctx.omit_messages, ) payload["resumed"] = ctx.target @@ -796,44 +770,20 @@ def _resume_reuse_live(ctx: _Resume, sid: str, session: dict) -> dict: payload["messages"] = [] payload["message_count"] = int(session.get("resume_message_count") or payload["message_count"]) payload["hydrating"] = bool(session.get("resume_hydrating")) - # A lazy watch session never owns a run loop (running always False) — - # overlay the child-run registry so a reconnecting window stays busy. + # A lazy watch session never owns a run loop — overlay the child-run registry. if session.get("agent") is None and _child_run_active(ctx.target): payload["running"] = True payload["status"] = "streaming" return _ok(ctx.rid, payload) -def _resume_info(ctx: _Resume, cwd: str, overrides: dict | None = None) -> dict: - overrides = overrides or {} - model_override = overrides.get("model_override") or {} - return _lazy_resume_info( - cwd, - model=model_override.get("model") or "", - provider=overrides.get("provider_override") or "", - profile=ctx.profile, - ) - - def _resume_response( - ctx: _Resume, - sid: str, - record: dict, - *, - info: dict, - display: list = (), - count_source: list | None = None, - messages: list | None = None, - message_count: int | None = None, - running: bool = False, - status: str = "idle", - hydrating: bool | None = None, - started_at=None, - auto_continue=None, + ctx: _Resume, sid: str, record: dict, *, info: dict, display: list = (), count_source: list | None = None, + messages: list | None = None, message_count: int | None = None, running: bool = False, + status: str = "idle", hydrating: bool | None = None, started_at=None, auto_continue=None, ) -> dict: - """Common resume payload. ``messages`` is the display projection (empty when - omit_messages); the count then falls back to ``count_source`` so the client still - learns the stored size. ``hydrating`` (deferred path) replaces ``messages_omitted``.""" + """Common resume payload. With omit_messages the count falls back to ``count_source`` + so the client still learns the stored size. ``hydrating`` replaces ``messages_omitted``.""" if messages is None: messages = [] if ctx.omit_messages else _history_to_messages(display) if message_count is None: @@ -859,10 +809,8 @@ def _resume_response( def _resume_read_history(ctx: _Resume): - """One lineage SELECT feeds both projections: model-fed copy alternation-repaired - for live replay, display copy verbatim (inspection/export shows what is stored). - The repaired copy becomes the resumed session's working conversation, so healing - a durable violation once here avoids re-firing the pre-request repair every turn.""" + """One lineage SELECT, two projections: model-fed copy alternation-repaired (healed once + here instead of every turn's pre-request repair), display copy verbatim.""" ctx.db.reopen_session(ctx.target) if ctx.omit_messages: raw = ctx.db.get_messages_as_conversation(ctx.target, repair_alternation=True, include_row_ids=True) @@ -871,19 +819,14 @@ def _resume_read_history(ctx: _Resume): def _resume_lazy(ctx: _Resume) -> dict: - """Lazy/watch resume: register the live session WITHOUT building an agent. - - Used by the desktop's subagent windows — the child runs inside the parent's - turn, so the window only needs stored history plus a transport for the - child-mirror's live events. A later prompt.submit upgrades it via - _start_agent_build (resume_session_id keeps it on the stored conversation). - """ + """Lazy/watch resume (desktop subagent windows): register the live session WITHOUT an + agent — the child runs inside the parent's turn, so the window needs stored history + plus a transport. A later prompt.submit upgrades it via _start_agent_build.""" sid, source = _new_runtime_ids(ctx.params) try: ctx.db.reopen_session(ctx.target) - # The child's OWN conversation only (include_ancestors would prepend the - # parent's transcript). repair_alternation heals a durable ``user;user`` - # once here instead of re-firing the pre-request repair every turn. + # Child's OWN conversation only (no ancestors); repair_alternation heals a + # durable ``user;user`` once here. history = ctx.db.get_messages_as_conversation( ctx.target, repair_alternation=True, include_row_ids=True ) @@ -893,12 +836,10 @@ def _resume_lazy(ctx: _Resume) -> dict: record = ctx.record(source, cwd, history, lazy=True, todo_state=_todo_state_from_history(history)) if (live := _claim_or_reuse_live(sid, ctx.target, record, None)) is not None: return _resume_reuse_live(ctx, *live) - # A delegated child mid-run emits no session events of its own — report - # liveness from the relay registry so the window shows a busy turn. + # A child mid-run emits no session events — liveness comes from the relay registry. child_running = _child_run_active(ctx.target) - # Display uses the VERBATIM projection (child-only, matching the repaired - # read) so model-invisible rows survive in the watch window as on the eager - # + REST paths; the repaired ``history`` still feeds live replay. + # Display uses the VERBATIM child-only projection so model-invisible rows survive; + # the repaired ``history`` still feeds live replay. try: display_history = ctx.db.get_messages_as_conversation( ctx.target, repair_alternation=False, include_row_ids=True @@ -907,31 +848,20 @@ def _resume_lazy(ctx: _Resume) -> dict: logger.debug("child-watch display projection read failed", exc_info=True) display_history = history return _resume_response( - ctx, - sid, - record, - info=_lazy_resume_info(cwd, profile=ctx.profile), - display=display_history, - count_source=display_history, - running=child_running, - status="streaming" if child_running else "idle", + ctx, sid, record, info=_lazy_resume_info(cwd, profile=ctx.profile), display=display_history, + count_source=display_history, running=child_running, status="streaming" if child_running else "idle", ) def _resume_deferred(ctx: _Resume) -> dict: - """Bounded acknowledgement; the transcript hydrates in the background and the - display copy pages over REST. defer_history SUPERSEDES omit_messages: the - response never carries a transcript and the ONE history read happens in the - hydration worker, so it is never loaded twice for one resume.""" + """Bounded ack; the transcript hydrates in the background and pages over REST. + defer_history SUPERSEDES omit_messages: the ONE history read happens in the worker.""" sid, source = _new_runtime_ids(ctx.params) _enable_gateway_prompts() overrides = _stored_session_runtime_overrides(ctx.found) or {} cwd = ctx.cwd() record = ctx.record( - source, - cwd, - [], - model_override=overrides.get("model_override"), + source, cwd, [], model_override=overrides.get("model_override"), resume_runtime_overrides=overrides or None, ) record["resume_history_ready"] = threading.Event() @@ -939,79 +869,51 @@ def _resume_deferred(ctx: _Resume) -> dict: record["resume_message_count"] = int(ctx.found.get("message_count") or 0) if (live := _claim_or_reuse_live(sid, ctx.target, record, None)) is not None: return _resume_reuse_live(ctx, *live) - _schedule_resume_hydration(sid, ctx.target, ctx.db, close_db=ctx.owns_db) - # The hydration worker now owns a profile-scoped handle and closes it after - # the read. The shared launch DB is process-owned. + # The hydration worker now owns (and closes) the profile-scoped handle. ctx.owns_db = False _schedule_session_cap_enforcement() return _resume_response( - ctx, - sid, - record, - info=_resume_info(ctx, cwd, overrides), - messages=[], - message_count=record["resume_message_count"], - status="resuming", - hydrating=True, + ctx, sid, record, info=ctx.info(cwd, overrides), messages=[], + message_count=record["resume_message_count"], status="resuming", hydrating=True, ) def _resume_cold(ctx: _Resume) -> dict: - """Cold resume default: register the session and read its transcript, but build - the agent OFF the response path — _make_agent can block for seconds and every - resume caller awaits this RPC before it paints. Pre-warms on a short timer - (session.create's deferred-build contract); _sess() builds on demand if the - first prompt beats it. Unlike the lazy branch this restores the full ancestor - history and persisted runtime identity, and is a real (upgradable) session.""" + """Default cold resume: read the transcript, build the agent OFF the response path + (_make_agent can block for seconds; callers await this RPC before painting). Pre-warms + on a timer; _sess() builds on demand if the first prompt beats it. Unlike lazy, restores + full ancestor history + persisted runtime identity.""" sid, source = _new_runtime_ids(ctx.params) _enable_gateway_prompts() try: raw_history, display_history = _resume_read_history(ctx) except Exception as e: return ctx.resume_failed(e) - # Display keeps the full transcript; the model-fed history drops a dangling - # tool-call tail so a session killed mid-loop does not replay it forever. + # Model-fed history drops a dangling tool-call tail (killed mid-loop) — display keeps it. prefix = [] if ctx.omit_messages else ctx.db.get_ancestor_display_prefix(ctx.target) history = sanitize_replay_history(raw_history) - # Restore model/provider/reasoning/tier so the deferred build matches the - # eager path — without them the build drops the provider. + # Restore model/provider/reasoning/tier so the deferred build matches eager. overrides = _stored_session_runtime_overrides(ctx.found) or {} cwd = ctx.cwd() record = ctx.record( - source, - cwd, - history, - display_history_prefix=prefix, - model_override=overrides.get("model_override"), - resume_runtime_overrides=overrides or None, - todo_state=_todo_state_from_history(history), + source, cwd, history, display_history_prefix=prefix, model_override=overrides.get("model_override"), + resume_runtime_overrides=overrides or None, todo_state=_todo_state_from_history(history), ) if (live := _claim_or_reuse_live(sid, ctx.target, record, None)) is not None: return _resume_reuse_live(ctx, *live) - _schedule_agent_build(sid) _schedule_session_cap_enforcement() # trim detached idle sessions over the cap auto_continue = _maybe_schedule_auto_continue(sid, record, ctx.target) - return _resume_response( - ctx, - sid, - record, - info=_resume_info(ctx, cwd, overrides), - display=display_history, - count_source=raw_history, - auto_continue=auto_continue, + ctx, sid, record, info=ctx.info(cwd, overrides), display=display_history, + count_source=raw_history, auto_continue=auto_continue, ) def _resume_eager(ctx: _Resume) -> dict: - """Synchronous build (``eager_build: true``, e.g. build-race tests). - - The agent is built OUTSIDE _session_resume_lock (it can block for seconds and - would stall session.close on the dispatch thread), then double-checked: if a - concurrent resume won meanwhile, discard our agent and reuse theirs. - """ + """Synchronous build (``eager_build: true``). Built OUTSIDE _session_resume_lock (would + stall session.close), then double-checked: a concurrent winner's agent is reused.""" sid, source = _new_runtime_ids(ctx.params) _enable_gateway_prompts() with _profile_build_scope(ctx.profile_home): @@ -1020,59 +922,38 @@ def _resume_eager(ctx: _Resume) -> dict: display_history_prefix = [] if ctx.omit_messages else ctx.db.get_ancestor_display_prefix(ctx.target) history = sanitize_replay_history(raw_history) messages = [] if ctx.omit_messages else _history_to_messages(display_history) - tokens = _set_session_context(ctx.target) - try: - # The profile's db so turns persist to the right state.db; runtime - # identity from the stored row so switching chats does not inherit - # whatever global model another chat last selected. - stored_runtime_overrides = _stored_session_runtime_overrides(ctx.found) - agent = _make_agent( - sid, - ctx.target, - session_id=ctx.target, - session_db=ctx.db, - platform_override=source, - context_cwd_is_launch_artifact=( - source in _LAUNCH_CWD_NOT_A_WORKSPACE and not ctx.profile_resume_cwd - ), - **stored_runtime_overrides, - ) - finally: - _clear_session_context(tokens) + # Profile db so turns persist to the right state.db; runtime identity from the + # stored row so switching chats does not inherit another chat's global model. + stored_runtime_overrides = _stored_session_runtime_overrides(ctx.found) + agent = _make_agent_in_context( + sid, + ctx.target, + session_db=ctx.db, + platform_override=source, + context_cwd_is_launch_artifact=( + source in _LAUNCH_CWD_NOT_A_WORKSPACE and not ctx.profile_resume_cwd + ), + **stored_runtime_overrides, + ) except Exception as e: return ctx.resume_failed(e) - with _session_resume_lock: live = _find_live_session_by_key(ctx.target, ctx.profile_home) if live is not None: - try: - if hasattr(agent, "close"): - agent.close() - except Exception: - pass + with contextlib.suppress(Exception): + agent.close() return _resume_reuse_live(ctx, *live) try: with _profile_build_scope(ctx.profile_home): _init_session( - sid, - ctx.target, - agent, - history, - cols=ctx.cols, - cwd=ctx.profile_resume_cwd, - session_db=ctx.db, - source=source, - explicit_cwd=bool(ctx.profile_resume_cwd), + sid, ctx.target, agent, history, cols=ctx.cols, cwd=ctx.profile_resume_cwd, + session_db=ctx.db, source=source, explicit_cwd=bool(ctx.profile_resume_cwd), ) - # Ownership TRANSFER: the registered agent holds this handle for - # its life and AIAgent.close() releases it at teardown - # (_init_session never closes a caller-supplied db). The drop is - # UNCONDITIONAL — past this line the session is registered against - # the handle, so the finally must not close it even if the transfer - # was refused (a leak is survivable; "Cannot operate on a closed - # database" on every later turn is not). The transfer is gated on - # owns_db: the SHARED launch handle must never move onto one - # session, or session.close tears down the process-wide database. + # Ownership TRANSFER: the agent holds the handle for life (AIAgent.close() + # releases it). The owns_db drop is UNCONDITIONAL — the session is now + # registered against the handle, so the finally must not close it even if + # the transfer was refused (a leak beats "closed database" every turn). + # Gated on owns_db: the SHARED launch handle must never move onto one session. if ctx.owns_db: _transfer_db_to_agent(agent, ctx.db) ctx.owns_db = False @@ -1080,16 +961,13 @@ def _resume_eager(ctx: _Resume) -> dict: if stored_runtime_overrides.get("model_override") is not None: _sessions[sid]["model_override"] = stored_runtime_overrides["model_override"] _sessions[sid]["display_history_prefix"] = display_history_prefix - # Each turn re-binds HERMES_HOME (mid-turn home reads — memory, - # skills — must resolve to the resumed profile too). + # Each turn re-binds HERMES_HOME (mid-turn memory/skills reads). if ctx.profile_home is not None: _sessions[sid]["profile_home"] = str(ctx.profile_home) _sessions[sid]["active_session_lease"] = None # claimed lazily on the first turn except Exception as e: - # _init_session registers _sessions[sid] BEFORE its first read through - # this handle ("database is locked" is the realistic trigger). Left in - # place, the live fast path would serve that dead session forever; - # owns_db still True means the registration is ours to undo. + # _init_session registers _sessions[sid] BEFORE its first db read ("database is + # locked"); left in place the fast path would serve that dead session forever. if ctx.owns_db: with _sessions_lock: _sessions.pop(sid, None) @@ -1097,14 +975,8 @@ def _resume_eager(ctx: _Resume) -> dict: session = _sessions.get(sid) or {} auto_continue = _maybe_schedule_auto_continue(sid, session, ctx.target) if session else None return _resume_response( - ctx, - sid, - session, - info=_session_info(agent, session), - messages=messages, - count_source=raw_history, - started_at=float(session.get("created_at") or time.time()), - auto_continue=auto_continue, + ctx, sid, session, info=_session_info(agent, session), messages=messages, count_source=raw_history, + started_at=float(session.get("created_at") or time.time()), auto_continue=auto_continue, ) @@ -1128,13 +1000,12 @@ def _(rid, params: dict) -> dict: profile_home=_profile_home(profile), lazy=is_truthy_value(params.get("lazy", False)), defer_history=is_truthy_value(params.get("defer_history", False)), - # Desktop hydrates transcripts over REST in parallel; suppress the - # duplicate WebSocket copy only when explicitly asked. + # Desktop hydrates over REST; suppress the duplicate WS copy only when asked. omit_messages=is_truthy_value(params.get("omit_messages", False)), eager_build=is_truthy_value(params.get("eager_build", False)), ) - # Profile scope opens a DEDICATED handle we own until the agent takes it; - # otherwise the shared launch db, which outlives the RPC and is never closed here. + # Profile scope: a DEDICATED handle we own until the agent takes it; else the shared + # launch db, never closed here. if ctx.profile_home is not None: from hermes_state import get_shared_session_db ctx.db = get_shared_session_db(ctx.profile_home / "state.db") @@ -1152,8 +1023,7 @@ def _(rid, params: dict) -> dict: ctx.profile_resume_cwd = str(ctx.found.get("cwd") or "").strip() or _profile_configured_cwd( ctx.profile_home ) - # Fast path: reuse a session already live IN THIS PROFILE — never another - # profile's runtime of the same stored id. + # Fast path: reuse a session live IN THIS PROFILE (never another profile's runtime). with _session_resume_lock: live = _find_live_session_by_key(ctx.target, ctx.profile_home) if live is not None: @@ -1166,10 +1036,8 @@ def _(rid, params: dict) -> dict: return _resume_cold(ctx) return _resume_eager(ctx) finally: - # Every return that does not transfer the handle abandons it. Refcounting - # alone does not release the sqlite fds: SessionDB pins ITSELF once its - # background token writer starts (atexit.register in hermes_state), which - # only close() unregisters. + # Refcounting alone does not release the sqlite fds: SessionDB pins ITSELF once its + # background token writer starts (atexit.register); only close() unregisters. if ctx.owns_db and ctx.db is not None: with contextlib.suppress(Exception): ctx.db.close() @@ -1197,13 +1065,11 @@ def _(rid, params: dict, session: dict) -> dict: @method("session.workspace.move") def _(rid, params: dict) -> dict: - """Re-home a STORED session's workspace (by ``session_key``) into another folder/project. + """Re-home a STORED session's workspace (by ``session_key``); no live agent required. - Unlike ``session.cwd.set`` no live agent is required. The git branch/root - columns are REPLACED, not enriched — a stale ``git_repo_root`` would keep the - session grouped under the project it left. A live agent bound to the row - follows too, even mid-turn (refusing made the UI claim success while state.db - kept the old cwd); in-flight tool calls keep their cwd, the NEXT one moves. + git branch/root columns are REPLACED (a stale ``git_repo_root`` would keep the session + under the project it left). A live agent follows too, even mid-turn (refusing made the + UI claim success while state.db kept the old cwd); the NEXT tool call moves. """ target = str(params.get("session_key") or "").strip() if not target: @@ -1224,14 +1090,12 @@ def _(rid, params: dict) -> dict: if sess.get("session_key") == target: live, live_sid = sess, sid break - branch = _git_branch_for_cwd(resolved) root = _git_common_repo_root_for_cwd(resolved) with _profile_db(params) as db: if db is None: return _db_unavailable_error(rid, code=5007) - # A brand-new draft has no row yet; the live re-home still applies and - # the row inherits the cwd when first written. + # A draft has no row yet; the live re-home still applies (row inherits cwd on write). row_exists = bool(db.get_session(target)) if not row_exists and live is None: return _err(rid, 4007, "session not found") @@ -1240,34 +1104,26 @@ def _(rid, params: dict) -> dict: db.update_session_cwd(target, resolved, branch, root, replace_git_meta=True) except Exception as e: return _err(rid, 5007, f"move failed: {e}") - if live is not None: try: _set_session_cwd(live, resolved) except ValueError as e: return _err(rid, 4017, str(e)) _emit("session.info", live_sid, _cwd_info(live, resolved, branch=branch)) - return _ok(rid, {"cwd": resolved, "branch": branch, "git_repo_root": root}) @method("session.active_list") def _(rid, params: dict) -> dict: - """Live TUI sessions in this process (not a DB browser): only sessions with - in-memory agents/workers the current TUI can switch to without closing siblings.""" + """Live TUI sessions in this process (not a DB browser).""" current = str(params.get("current_session_id") or "") - try: - with _sessions_lock: - snapshot = list(_sessions.items()) - except Exception as e: - return _err(rid, 5036, f"could not enumerate active sessions: {e}") + snapshot, err = _snapshot_sessions(rid) + if err: + return err - # ``_finalized`` sessions are dead (teardown begun) but may linger in - # ``_sessions`` until the reaper pops them; counting them inflated the footer - # forever. Do NOT filter on the WS-detached sentinel: a detached session is - # still attachable via reconnect until grace-reap finalizes it, and a - # standalone ``hermes --tui`` rides the real stdio transport. Keep insertion - # order — the focused session must not jump to the top. + # ``_finalized`` sessions linger until the reaper pops them (counting them inflated the + # footer). Do NOT filter on the WS-detached sentinel: detached is still attachable until + # grace-reap, and ``hermes --tui`` rides stdio. Keep insertion order (focused must not jump). rows = [ _session_live_item(sid, session, current) for sid, session in snapshot @@ -1278,8 +1134,7 @@ def _(rid, params: dict) -> dict: @method("session.activate") def _(rid, params: dict) -> dict: - """Attach the frontend to an already-live TUI session (does not close the - previously focused one — just enough state for Ink to redraw).""" + """Attach the frontend to a live TUI session without closing the previously focused one.""" sid = str(params.get("session_id") or "") session, err = _sess_nowait({"session_id": sid}, rid) if err: @@ -1287,10 +1142,7 @@ def _(rid, params: dict) -> dict: return _ok( rid, _live_session_payload( - sid, - session, - touch=True, - transport=current_transport() or _stdio_transport, + sid, session, touch=True, transport=current_transport() or _stdio_transport, omit_messages=is_truthy_value(params.get("omit_messages", False)), ), ) @@ -1298,23 +1150,15 @@ def _(rid, params: dict) -> dict: @method("session.delete") def _(rid, params: dict) -> dict: - """Delete a stored session and its transcript files (TUI resume picker ``d``). - - Refuses sessions live in this process — removing rows under a live agent - corrupts message ordering and trips FK constraints on the next flush. - Honors ``params.profile`` (mirrors ``session.resume``). - """ + """Delete a stored session + transcript files (honors ``params.profile``). Refuses sessions + live in this process — deleting under a live agent trips FK constraints on the next flush.""" target = params.get("session_id", "") if not target: return _err(rid, 4006, "session_id required") - # Snapshot via list(): _sessions is mutated by concurrent RPCs. If even the - # snapshot raises, fail CLOSED (refuse the delete). - try: - with _sessions_lock: - snapshot = list(_sessions.values()) - except Exception as e: - return _err(rid, 5036, f"could not enumerate active sessions: {e}") - active = {s.get("session_key") for s in snapshot if s.get("session_key")} + snapshot, err = _snapshot_sessions(rid) + if err: + return err + active = {s.get("session_key") for _sid, s in snapshot if s.get("session_key")} if target in active: return _err(rid, 4023, "cannot delete an active session") profile_home = _profile_home((params.get("profile") or "").strip() or None) @@ -1372,7 +1216,6 @@ def _(rid, params: dict, session: dict, db) -> dict: session["pending_title"] = value if pending else None _emit_session_info_for_session(sid, session) return _ok(rid, {"pending": pending, "title": value}) - try: if db.set_session_title(key, title): return _done(False, title) @@ -1380,17 +1223,14 @@ def _(rid, params: dict, session: dict, db) -> dict: existing_row = db.get_session(key) if existing_row: return _done(False, existing_row.get("title") or title) - # No row yet (deferred to the first prompt). An explicit /title is clear - # intent, so persist the row NOW (mirrors the messaging gateway's - # _handle_title_command) instead of queuing pending_title and hoping the - # post-turn apply block lands under this key. The min-messages sidebar - # filter keeps a titled 0-message row hidden. + # No row yet: an explicit /title is clear intent, so persist the row NOW (as the + # gateway's _handle_title_command) rather than hoping the post-turn apply lands + # under this key. The min-messages sidebar filter hides a titled 0-message row. _ensure_session_db_row(session) with _session_db(session) as scoped_db: if scoped_db is not None and scoped_db.set_session_title(key, title): return _done(False, title) - # Row creation didn't take (DB unavailable / concurrent writer) — queue - # so the post-turn apply block can still recover. + # Row creation didn't take — queue so the post-turn apply block can recover. return _done(True, title) except ValueError as e: return _err(rid, 4022, str(e)) @@ -1400,14 +1240,9 @@ def _(rid, params: dict, session: dict, db) -> dict: @method("session.set_hidden") def _(rid, params: dict) -> dict: - """Set/clear the ``hidden`` flag on a session (and its compression lineage). - - A hidden session is dropped from the default Sessions list but stays fully - resumable by the surface that owns it. Two-tier resolution: a LIVE runtime - id first (covers the not-yet-persisted draft via ``pending_hidden``), then a - durable stored id/key in the target profile's state.db — plugins reconciling - sessions they own hold stored ids for chats that aren't live right now. - """ + """Set/clear ``hidden`` on a session (and its compression lineage); hidden sessions leave + the default list but stay resumable by their owner. Resolution: LIVE runtime id first + (covers unpersisted drafts via ``pending_hidden``), then a stored id/key in the profile db.""" hidden = is_truthy_value(params.get("hidden", True)) session, err = _sess_nowait(params, rid) if session is not None: @@ -1417,8 +1252,7 @@ def _(rid, params: dict) -> dict: key = session["session_key"] try: if not db.set_session_hidden(key, hidden): - # No row yet: remember the intent so _ensure_session_db_row - # is born hidden (mirrors the pending_title deferral). + # No row yet: _ensure_session_db_row is born hidden (as pending_title). session["pending_hidden"] = hidden return _ok(rid, {"hidden": hidden, "session_key": key}) except Exception as e: @@ -1441,28 +1275,21 @@ def _(rid, params: dict) -> dict: @method("message.react") @_with_session def _(rid, params: dict, session: dict) -> dict: - """Set or clear one author's emoji reaction on a persisted message. - - iOS Tapback semantics enforced in the DB layer: one reaction per author per - message, re-sending the same emoji retracts it, ``emoji: null`` clears. - ``row_id`` is the durable ``messages.id``; a live message that hasn't - round-tripped through a resume can instead name ``newest_role``. - """ + """Set/clear one author's emoji reaction (Tapback semantics in the DB layer: one per + author, same emoji retracts, ``emoji: null`` clears). ``row_id`` is ``messages.id``; a + live message not yet round-tripped can name ``newest_role`` instead.""" newest_role = str(params.get("newest_role") or "").strip() row_id = params.get("row_id") if row_id is None and newest_role not in {"user", "assistant"}: return _err(rid, 4023, "row_id or newest_role required") - emoji = params.get("emoji") if emoji is not None: emoji = str(emoji).strip() if not emoji: return _err(rid, 4024, "emoji must be a non-empty string or null") - author = str(params.get("author") or "user").strip() if author not in {"user", "agent"}: return _err(rid, 4025, "author must be 'user' or 'agent'") - with _session_db(session) as db: if db is None: return _db_unavailable_error(rid, code=5007) @@ -1474,7 +1301,6 @@ def _(rid, params: dict, session: dict) -> dict: reactions = db.set_message_reaction(session["session_key"], int(row_id), emoji, author=author) except Exception as e: return _err(rid, 5007, str(e)) - if reactions is None: return _err(rid, 4040, "message not found in this session") return _ok(rid, {"row_id": int(row_id), "reactions": reactions}) @@ -1482,18 +1308,13 @@ def _(rid, params: dict, session: dict) -> dict: @method("llm.oneshot") def _(rid, params: dict) -> dict: - """Single stateless LLM request outside any conversation (e.g. a commit message). - - Accepts a named ``template`` + ``variables`` or ``instructions``/``input``. - A live ``session_id`` lends its agent's model; otherwise the auxiliary - ``task`` backend. Never mutates session history (prompt cache untouched). - """ + """Stateless one-shot LLM request (``template``+``variables`` or ``instructions``/``input``). + A live ``session_id`` lends its model, else the auxiliary ``task`` backend. Never touches history.""" template = (params.get("template") or "").strip() or None instructions = params.get("instructions") or "" user_input = params.get("input") or "" variables = params.get("variables") if isinstance(params.get("variables"), dict) else {} task = (params.get("task") or "title_generation").strip() or "title_generation" - try: max_tokens = int(params.get("max_tokens") or 1024) except (TypeError, ValueError): @@ -1504,23 +1325,15 @@ def _(rid, params: dict) -> dict: temperature = float(temperature) except (TypeError, ValueError): temperature = None - if not template and not str(instructions).strip() and not str(user_input).strip(): return _err(rid, 4030, "llm.oneshot requires a template or instructions/input") - session = _sessions.get(params.get("session_id") or "") main_runtime = _main_runtime_from_agent(session.get("agent")) if session else None - try: from agent.oneshot import run_oneshot text = run_oneshot( - instructions=instructions, - user_input=user_input, - template=template, - variables=variables, - task=task, - max_tokens=max_tokens, - temperature=temperature if temperature is not None else 0.3, + instructions=instructions, user_input=user_input, template=template, variables=variables, + task=task, max_tokens=max_tokens, temperature=temperature if temperature is not None else 0.3, main_runtime=main_runtime, ) except KeyError as e: @@ -1530,7 +1343,6 @@ def _(rid, params: dict) -> dict: except Exception as e: logger.warning("llm.oneshot failed: %s", e) return _err(rid, 5030, f"one-shot generation failed: {e}") - return _ok(rid, {"text": text}) @@ -1540,22 +1352,16 @@ def _(rid, params: dict) -> dict: @method("handoff.request") @_with_session def _(rid, params: dict, session: dict) -> dict: - """Queue a handoff of this session to a messaging platform (desktop parity with /handoff). - - Only writes ``handoff_state='pending'`` on the persisted row; the separate - ``hermes gateway`` process's ``_handoff_watcher`` claims it, re-binds the - session to the platform's home channel and forges a synthetic turn. The - desktop then polls ``handoff.state``. - """ + """Queue a handoff to a messaging platform (desktop /handoff). Only writes + ``handoff_state='pending'``; the gateway's ``_handoff_watcher`` claims it and re-binds + the session to the home channel. The desktop polls ``handoff.state``.""" if session.get("running"): return _err(rid, 4009, "session busy — wait for the current turn to finish, then retry the handoff") - platform_name = (params.get("platform", "") or "").strip().lower() if not platform_name: return _err(rid, 4023, "platform required") - # Validate against the live gateway config up front: an unconfigured platform - # or missing home channel would leave the handoff pending forever. + # Validate up front: an unconfigured platform / missing home channel pends forever. try: from gateway.config import Platform, load_gateway_config except Exception as e: # pragma: no cover — gateway pkg always ships @@ -1581,10 +1387,8 @@ def _(rid, params: dict, session: dict) -> dict: "/sethome on the destination chat first", ) - # The watcher transfers a persisted row, so make sure one exists even for a - # brand-new empty chat (mirrors the CLI's set_session_title stub). + # The watcher transfers a persisted row, so make sure one exists for an empty chat. _ensure_session_db_row(session) - with _session_db(session) as db: if db is None: return _db_unavailable_error(rid, code=5007) @@ -1595,7 +1399,6 @@ def _(rid, params: dict, session: dict) -> dict: ok = db.request_handoff(key, platform_name) except Exception as e: return _err(rid, 5007, str(e)) - if not ok: return _err(rid, 4027, "session is already in flight for handoff — wait for it to settle, then retry") return _ok(rid, {"queued": True, "session_key": key, "platform": platform_name, "home_name": home.name}) @@ -1618,14 +1421,9 @@ def _(rid, params: dict, session: dict, db) -> dict: @method("handoff.fail") def _(rid, params: dict) -> dict: - """Mark a not-yet-claimed handoff failed so the user can retry (desktop poll timeout). - - Only PENDING rows change (compare-and-swap in ``fail_handoff``): once the - watcher has claimed the row (``running``) it owns the terminal state — failing - it from the waiter races the dispatch and later flips failed→completed after - the user was told it failed. A ``running`` row yields - ``{"failed": False, "state": "running"}`` ("still transferring"). - """ + """Mark a not-yet-claimed handoff failed (desktop poll timeout). Only PENDING rows change + (CAS in ``fail_handoff``): a claimed ``running`` row is the watcher's to finish and yields + ``{"failed": False, "state": "running"}``.""" # Undecorated on purpose: tests rebind this handler's __code__ directly. session, err = _sess_nowait(params, rid) if err: @@ -1646,7 +1444,6 @@ def _(rid, params: dict) -> dict: if failed: return _ok(rid, {"failed": True, "state": "failed"}) record = db.get_handoff_state(key) or {} - return _ok(rid, {"failed": False, "state": record.get("state") or ""}) @@ -1660,15 +1457,11 @@ def _(rid, params: dict, session: dict) -> dict: usage: dict = _session_usage_snapshot(session) if agent is None and not usage: usage = {"calls": 0, "input": 0, "output": 0, "total": 0} - # Nous credits are agent-independent (portal fetch) so they show even with - # zero API calls. Fail-open: absent when not logged in / portal hiccup. - try: + # Nous credits are agent-independent (portal fetch); fail-open when absent. + with contextlib.suppress(Exception): from agent.account_usage import nous_credits_lines - credits = nous_credits_lines() - if credits: + if credits := nous_credits_lines(): usage["credits_lines"] = credits - except Exception: - pass return _ok(rid, usage) @@ -1704,21 +1497,15 @@ def _(rid, params: dict, session: dict) -> dict: _PET_OFF = {"enabled": False} -@method("pet.info") -@_profile_scoped -@_pet_guard("pet.info", fail_open=_PET_OFF) +@_pet_method("pet.info", fail_open=_PET_OFF) def _(rid, params: dict) -> dict: - """Active petdex pet for sprite-rendering surfaces (desktop canvas + TUI half-block). - - Carries the spritesheet (base64) plus frame geometry + state-row taxonomy so - the renderer is a thin consumer. Agent-independent; fail-open ``enabled=False``. - """ - enabled, pet, scale = _pet_active_selection() - if not enabled or pet is None or not pet.exists: + """Active pet for sprite-rendering surfaces: spritesheet (base64) + frame geometry + + state-row taxonomy so the renderer is a thin consumer.""" + if (active := _active_pet()) is None: return _ok(rid, {"enabled": False}) + pet, scale = active payload = {"enabled": True, **_pet_sprite_payload(pet, scale=scale)} - # Send-once for the multi-MB sheet: a caller holding revision R gets - # metadata only (spritesheetUnchanged) when the sheet hasn't changed. + # Send-once for the multi-MB sheet: same revision → metadata only. known_revision = str(params.get("knownRevision", "") or "") if known_revision and known_revision == payload.get("spritesheetRevision"): payload.pop("spritesheetBase64", None) @@ -1726,14 +1513,12 @@ def _(rid, params: dict) -> dict: return _ok(rid, payload) -@method("pet.info.meta") -@_profile_scoped -@_pet_guard("pet.info.meta", fail_open=_PET_OFF) +@_pet_method("pet.info.meta", fail_open=_PET_OFF) def _(rid, params: dict) -> dict: """Cheap active-pet metadata used to avoid full payload refreshes.""" - enabled, pet, scale = _pet_active_selection() - if not enabled or pet is None or not pet.exists: + if (active := _active_pet()) is None: return _ok(rid, {"enabled": False}) + pet, scale = active return _ok( rid, { @@ -1746,16 +1531,38 @@ def _(rid, params: dict) -> dict: ) -@method("pet.cells") -@_profile_scoped -@_pet_guard("pet.cells", fail_open=_PET_OFF) -def _(rid, params: dict) -> dict: - """Half-block cell frames for one pet state (TUI renderer). +def _pet_kitty_cells(pet, pet_cfg: dict, state: str, scale: float) -> dict | None: + """kitty graphics payload for a TTY that speaks it (env shared with the Ink process; the + dashboard PTY falls through). Only kitty is grid-safe in Ink — iTerm/sixel stay on half-blocks.""" + from agent.pet import constants, render + from agent.pet.render import PetRenderer + configured = str(pet_cfg.get("render_mode", "auto") or "auto").lower() + gmode = render.detect_terminal_graphics() if configured in ("", "auto") else configured + if gmode != "kitty": + return None + image_id = render.kitty_image_id(pet.slug) + # kitty sizes from scaled pixels, so unicode_cols is moot here. + payload = PetRenderer(str(pet.spritesheet), mode="kitty", scale=scale).kitty_payload(state, image_id=image_id) + if not payload: + return None + return { + "graphics": "kitty", + "imageId": image_id, + "color": render.kitty_color_hex(image_id), + "cols": payload["cols"], + "rows": payload["rows"], + "placeholder": payload["placeholder"], + "frames": payload["frames"], + "frameMs": constants.LOOP_MS / max(1, len(payload["frames"]) or 1), + "scale": scale, + } - Each cell is ``[tr,tg,tb,ta, br,bg,bb,ba]`` (top + bottom pixel). - Params: ``state`` (idle/run/review/failed/wave/jump), ``cols``, ``graphics``. - """ - from agent.pet import constants, render, store + +@_pet_method("pet.cells", fail_open=_PET_OFF) +def _(rid, params: dict) -> dict: + """Half-block cell frames for one pet state (TUI); each cell is ``[tr,tg,tb,ta, br,bg,bb,ba]``. + Params: ``state`` (idle/run/review/failed/wave/jump), ``cols``, ``graphics``.""" + from agent.pet import constants, store from agent.pet.render import PetRenderer pet_cfg = _pet_display_cfg() if not is_truthy_value(pet_cfg.get("enabled"), default=False): @@ -1763,42 +1570,13 @@ def _(rid, params: dict) -> dict: pet = store.resolve_active_pet(str(pet_cfg.get("slug", "") or "")) if pet is None or not pet.exists: return _ok(rid, {"enabled": False}) - state = str(params.get("state") or constants.PetState.IDLE.value) scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) cols = int(params.get("cols") or 0) or constants.resolve_cols(scale, pet_cfg.get("unicode_cols", 0)) base = {"enabled": True, "slug": pet.slug, "displayName": pet.display_name, "state": state} - # Graphics path: a real TTY speaking kitty gets a Unicode-placeholder image - # instead of half-blocks. Env detection is shared with the Ink process (it - # spawns us); the dashboard PTY has no such env and falls through. Only - # kitty is grid-safe in Ink — iTerm/sixel stay on the fallback. - if params.get("graphics"): - configured = str(pet_cfg.get("render_mode", "auto") or "auto").lower() - gmode = render.detect_terminal_graphics() if configured in ("", "auto") else configured - if gmode == "kitty": - image_id = render.kitty_image_id(pet.slug) - # kitty sizes from scaled pixels, so unicode_cols is moot here. - payload = PetRenderer(str(pet.spritesheet), mode="kitty", scale=scale).kitty_payload( - state, image_id=image_id - ) - if payload: - return _ok( - rid, - { - **base, - "graphics": "kitty", - "imageId": image_id, - "color": render.kitty_color_hex(image_id), - "cols": payload["cols"], - "rows": payload["rows"], - "placeholder": payload["placeholder"], - "frames": payload["frames"], - "frameMs": constants.LOOP_MS / max(1, len(payload["frames"]) or 1), - "scale": scale, - }, - ) - + if params.get("graphics") and (kitty := _pet_kitty_cells(pet, pet_cfg, state, scale)): + return _ok(rid, {**base, **kitty}) renderer = PetRenderer(str(pet.spritesheet), mode="unicode", scale=scale, unicode_cols=cols) count = renderer.frame_count(state) or 1 frames = [ @@ -1811,20 +1589,14 @@ def _(rid, params: dict) -> dict: ) -@method("pet.gallery") -@_profile_scoped -@_pet_guard("pet.gallery", fail_open={"enabled": False, "active": "", "pets": []}) +@_pet_method("pet.gallery", fail_open={"enabled": False, "active": "", "pets": []}) def _(rid, params: dict) -> dict: - """Adoptable pets for the desktop picker: petdex gallery merged with local install state. - - Fail-open to whatever is installed locally when the gallery is unreachable. - ``localOnly`` skips the remote manifest so the user's own pets render instantly. - """ + """Petdex gallery merged with local install state; falls back to installed pets offline. + ``localOnly`` skips the remote manifest so the user's own pets render instantly.""" local_only = bool(params.get("localOnly")) from agent.pet import store pet_cfg = _pet_display_cfg() installed = {p.slug: p for p in store.installed_pets()} - gallery: list[dict] = [] seen: set[str] = set() try: @@ -1832,7 +1604,6 @@ def _(rid, params: dict) -> dict: # Local-only still warms the manifest cache in the background. if local_only: prefetch() - for entry in [] if local_only else fetch_manifest(): seen.add(entry.slug) gallery.append( @@ -1841,15 +1612,13 @@ def _(rid, params: dict) -> dict: "displayName": entry.display_name, "installed": entry.slug in installed, "spritesheetUrl": entry.spritesheet_url, - # petdex has no popularity metric; "curated" (its hand-picked - # set, identified by asset path) is the closest signal. + # No popularity metric; petdex's hand-picked set (by asset path) is closest. "curated": "/curated/" in entry.spritesheet_url, "generated": entry.slug in installed and installed[entry.slug].generated, } ) except Exception as exc: # noqa: BLE001 - offline: fall back to installed logger.debug("pet.gallery manifest fetch failed: %s", exc) - for slug, pet in installed.items(): if slug not in seen: gallery.append( @@ -1861,7 +1630,6 @@ def _(rid, params: dict) -> dict: "generated": pet.generated, } ) - return _ok( rid, { @@ -1872,22 +1640,7 @@ def _(rid, params: dict) -> dict: ) -def _with_slug(fn): - """Require ``params.slug`` (4004 "missing slug") and pass it as a 3rd arg.""" - - def handler(rid, params: dict) -> dict: - slug = str(params.get("slug") or "").strip() - if not slug: - return _err(rid, 4004, "missing slug") - return fn(rid, params, slug) - - return handler - - -@method("pet.select") -@_profile_scoped -@_pet_guard("pet.select") -@_with_slug +@_pet_method("pet.select", slug=True) def _(rid, params: dict, slug: str) -> dict: """Adopt a pet: install (if needed) + activate; writes ``display.pet.*`` to config.""" from agent.pet import store @@ -1901,10 +1654,7 @@ def _(rid, params: dict, slug: str) -> dict: return _ok(rid, {"ok": True, "slug": slug, "displayName": pet.display_name}) -@method("pet.remove") -@_profile_scoped -@_pet_guard("pet.remove") -@_with_slug +@_pet_method("pet.remove", slug=True) def _(rid, params: dict, slug: str) -> dict: """Uninstall a pet (delete its directory); if it was active, turn the display off.""" from agent.pet import store @@ -1917,26 +1667,18 @@ def _(rid, params: dict, slug: str) -> dict: return _ok(rid, {"ok": removed, "slug": slug}) -@method("pet.export") -@_profile_scoped -@_pet_guard("pet.export") -@_with_slug +@_pet_method("pet.export", slug=True) def _(rid, params: dict, slug: str) -> dict: """Export an installed pet as a re-importable ``.zip`` → ``{ok, filename, zipBase64}``.""" import base64 from agent.pet import store - filename, data = store.export_pet(slug) return _ok( - rid, - {"ok": True, "filename": filename, "zipBase64": base64.standard_b64encode(data).decode("ascii")}, + rid, {"ok": True, "filename": filename, "zipBase64": base64.standard_b64encode(data).decode("ascii")}, ) -@method("pet.rename") -@_profile_scoped -@_pet_guard("pet.rename") -@_with_slug +@_pet_method("pet.rename", slug=True) def _(rid, params: dict, slug: str) -> dict: """Rename a pet's display name + realign its slug/dir; follows the active slug in config.""" name = str(params.get("name") or "").strip() @@ -1955,16 +1697,12 @@ def _(rid, params: dict, slug: str) -> dict: return _ok(rid, {"ok": True, "slug": new_slug, "displayName": name}) -@method("pet.thumb") -@_profile_scoped -@_pet_guard("pet.thumb", fail_open=lambda params: {"ok": False, "slug": str(params.get("slug") or "").strip()}) -@_with_slug +@_pet_method("pet.thumb", slug=True, fail_open=lambda params: {"ok": False, "slug": str(params.get("slug") or "").strip()}) def _(rid, params: dict, slug: str) -> dict: - """Small idle-frame PNG data URI for the picker preview (same-origin; the desktop - CSP / R2 hotlink rules break a CDN ````). ``url`` serves not-yet-installed pets.""" + """Idle-frame PNG data URI for the picker (desktop CSP / R2 hotlink rules break a CDN + ````). ``url`` serves not-yet-installed pets.""" import base64 from agent.pet import store - data = store.thumbnail_png(slug, source_url=str(params.get("url") or "")) if not data: return _ok(rid, {"ok": False, "slug": slug}) @@ -1978,9 +1716,7 @@ def _(rid, params: dict, slug: str) -> dict: ) -@method("pet.disable") -@_profile_scoped -@_pet_guard("pet.disable") +@_pet_method("pet.disable") def _(rid, params: dict) -> dict: """``display.pet.enabled=false`` from the desktop picker.""" from hermes_cli.pets import _set_enabled @@ -1988,9 +1724,7 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"ok": True}) -@method("pet.scale") -@_profile_scoped -@_pet_guard("pet.scale") +@_pet_method("pet.scale") def _(rid, params: dict) -> dict: """Persist ``display.pet.scale`` (clamped to engine bounds) from the desktop slider.""" from hermes_cli.pets import set_pet_scale @@ -2002,18 +1736,15 @@ def _(rid, params: dict) -> dict: @method("pet.cancel") def _(rid, params: dict) -> dict: - """Signal an in-flight ``pet.generate``/``pet.hatch`` (by token) to stop. - - Idempotent; stays off the worker pool so it lands while a generation occupies it. - """ + """Stop an in-flight ``pet.generate``/``pet.hatch`` by token. Idempotent; stays off the + worker pool so it lands while a generation occupies it.""" token = str(params.get("token") or "").strip() if token: _pet_cancel_request(token) return _ok(rid, {"ok": True}) -@method("pet.generate.status") -@_pet_guard("pet.generate.status", fail_open={"available": False, "providers": []}) +@_pet_method("pet.generate.status", scoped=False, fail_open={"available": False, "providers": []}) def _(rid, params: dict) -> dict: """Whether pet generation is possible: a reference-capable image backend is configured.""" from agent.pet.generate.imagegen import GenerationError, list_sprite_providers, resolve_provider @@ -2030,15 +1761,11 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"available": available, "providers": providers}) -@method("pet.generate") -@_pet_guard("pet.generate") +@_pet_method("pet.generate", scoped=False) def _(rid, params: dict) -> dict: - """Generate candidate base looks for a new pet (the draft step). Heavy: worker pool. - - Params: ``prompt`` (required unless ``referenceImage`` — a data URL every draft - is grounded on), ``count`` (≤4), ``style``, ``provider``. Returns - ``{ok, token, drafts:[{index, dataUri}]}``; the token keys a later ``pet.hatch``. - """ + """Candidate base looks for a new pet (draft step; worker pool). Params: ``prompt`` + (required unless ``referenceImage`` data URL), ``count`` (≤4), ``style``, ``provider``. + Returns ``{ok, token, drafts:[{index, dataUri}]}``; the token keys ``pet.hatch``.""" prompt = str(params.get("prompt") or "").strip() ref_raw = str(params.get("referenceImage") or "").strip() if not prompt and not ref_raw: @@ -2048,7 +1775,6 @@ def _(rid, params: dict) -> dict: except (TypeError, ValueError): count = 4 style = str(params.get("style") or "auto").strip() or "auto" - import shutil from agent.pet.generate import generate_base_drafts from agent.pet.generate.imagegen import GenerationError, resolve_provider @@ -2060,14 +1786,12 @@ def _(rid, params: dict) -> dict: _pet_cancel_arm(token) stage = root / token stage.mkdir(parents=True, exist_ok=True) - reference_images = None if ref_raw: try: reference_images = _pet_reference_images_from_data_url(ref_raw, stage) except ValueError as exc: - _pet_cancel_release(token) - return _err(rid, 4004, str(exc)) + return _pet_gen_abort(rid, token, 4004, str(exc)) # Resolve a picker-chosen provider up front so a bad pick fails fast, not mid-fan-out. provider_name = str(params.get("provider") or "").strip() @@ -2076,17 +1800,12 @@ def _(rid, params: dict) -> dict: try: sprite = resolve_provider(require_references=bool(reference_images), prefer=provider_name) except GenerationError as exc: - _pet_cancel_release(token) - return _err(rid, 5031, str(exc)) - + return _pet_gen_abort(rid, token, 5031, str(exc)) concept = prompt or "a pet based on the reference image" out: list[dict] = [] # Token-only init event so a Stop fired before the first draft can target this run. - try: - _emit("pet.generate.progress", "", {"token": token, "count": count}) - except Exception as exc: # noqa: BLE001 - streaming is best-effort - logger.debug("pet.generate init emit failed: %s", exc) + _pet_emit("pet.generate.progress", {"token": token, "count": count}, "pet.generate init") def _on_draft(index: int, src) -> None: dest = stage / f"draft-{index}.png" @@ -2097,30 +1816,18 @@ def _(rid, params: dict) -> dict: logger.debug("pet.generate draft %d failed: %s", index, exc) return out.append({"index": index, "dataUri": data_uri}) - # Stream the draft so the grid fills live; a transport hiccup must not abort generation. - try: - _emit( - "pet.generate.progress", - "", - {"token": token, "index": index, "dataUri": data_uri, "count": count}, - ) - except Exception as exc: # noqa: BLE001 - logger.debug("pet.generate progress emit failed: %s", exc) - + # Stream the draft so the grid fills live. + _pet_emit( + "pet.generate.progress", {"token": token, "index": index, "dataUri": data_uri, "count": count}, + "pet.generate progress", + ) try: generate_base_drafts( - concept, - n=count, - style=style, - reference_images=reference_images, - provider=sprite, - on_draft=_on_draft, - is_cancelled=lambda: _pet_is_cancelled(token), + concept, n=count, style=style, reference_images=reference_images, provider=sprite, + on_draft=_on_draft, is_cancelled=lambda: _pet_is_cancelled(token), ) except GenerationError as exc: - _pet_cancel_release(token) - return _err(rid, 5031, str(exc)) - + return _pet_gen_abort(rid, token, 5031, str(exc)) cancelled = _pet_is_cancelled(token) _pet_cancel_release(token) if cancelled: @@ -2131,19 +1838,13 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"ok": True, "token": token, "drafts": out}) -@method("pet.hatch") -@_pet_guard("pet.hatch") +@_pet_method("pet.hatch", scoped=False) def _(rid, params: dict) -> dict: - """Turn a chosen base draft into a full pet — installed but NOT yet active. Heavy: worker pool. - - The result is a preview the surface plays before the user commits (``pet.select`` - adopts, ``pet.remove`` discards). Params: ``token`` + ``index`` (from - ``pet.generate``), ``name`` (required), ``description``, ``prompt``, ``style``, - ``cancelToken``. Returns ``{ok, slug, displayName, warnings, pet}``. - """ + """Turn a base draft into a full pet — installed but NOT active (``pet.select`` adopts, + ``pet.remove`` discards). Params: ``token`` + ``index``, ``name`` (required), ``description``, + ``prompt``, ``style``, ``cancelToken``. Returns ``{ok, slug, displayName, warnings, pet}``.""" token = str(params.get("token") or "").strip() - # Hatch cancellation rides its own key: pet.generate may still be releasing - # `token`, which would wipe the arm set here. Falls back for old clients. + # Own cancel key: pet.generate may still be releasing `token`. Falls back for old clients. cancel_token = str(params.get("cancelToken") or "").strip() or token name = str(params.get("name") or "").strip() if not token: @@ -2154,7 +1855,6 @@ def _(rid, params: dict) -> dict: index = int(params.get("index", 0)) except (TypeError, ValueError): index = 0 - from agent.pet import store from agent.pet.generate import hatch_pet from agent.pet.generate.imagegen import GenerationError, resolve_provider @@ -2170,39 +1870,27 @@ def _(rid, params: dict) -> dict: sprite = resolve_provider(require_references=True, prefer=provider_name) except GenerationError as exc: return _err(rid, 5031, str(exc)) - _pet_cancel_arm(cancel_token) slug = store.unique_slug(name) def _on_progress(event: str, detail: str) -> None: - # Row progress is "::" so the egg screen can show - # "Drawing … (n/total)"; other phases pass through as-is. + # Row progress "::" → "Drawing … (n/total)". payload: dict = {"event": event, "detail": detail} if event == "row" and detail.count(":") == 2: state, done, total = detail.split(":") payload = {"event": "row", "state": state, "done": done, "total": total} - try: - _emit("pet.hatch.progress", "", payload) - except Exception as exc: # noqa: BLE001 - logger.debug("pet.hatch progress emit failed: %s", exc) - + _pet_emit("pet.hatch.progress", payload, "pet.hatch progress") try: result = hatch_pet( - base_image=base, - slug=slug, - display_name=name, - description=str(params.get("description") or ""), + base_image=base, slug=slug, display_name=name, description=str(params.get("description") or ""), concept=str(params.get("prompt") or name), - style=str(params.get("style") or "auto").strip() or "auto", - provider=sprite, - on_progress=_on_progress, - is_cancelled=lambda: _pet_is_cancelled(cancel_token), + style=str(params.get("style") or "auto").strip() or "auto", provider=sprite, + on_progress=_on_progress, is_cancelled=lambda: _pet_is_cancelled(cancel_token), ) except GenerationError as exc: return _err(rid, 5031, str(exc)) finally: _pet_cancel_release(cancel_token) - pet = store.load_pet(result.slug) payload = _pet_sprite_payload(pet, scale=_pet_config_scale()) if pet else {} return _ok( @@ -2218,10 +1906,9 @@ def _(rid, params: dict) -> dict: # ── billing / subscription ─────────────────────────────────────────── -# All fail-open: a logged-out / unreachable portal yields an ``ok`` envelope -# with a typed ``error`` (via _serialize_billing_error) rather than a JSON-RPC -# error, so the TUI maps it to the right copy. ``billing:manage`` routes return -# error=insufficient_scope on 403, which drives the ``billing.step_up`` device flow. +# All fail-open: a logged-out / unreachable portal yields an ``ok`` envelope with a typed +# ``error`` (not a JSON-RPC error) so the TUI maps it to copy. ``billing:manage`` routes +# return error=insufficient_scope on 403, which drives the ``billing.step_up`` device flow. @method("billing.state") @@ -2279,12 +1966,10 @@ def _(rid, params: dict) -> dict: tier_id = params.get("subscription_type_id") if not cancel and not tier_id: return _billing_invalid(rid, "subscription_type_id or cancel is required") - - def call(): - result = put_subscription_pending_change(subscription_type_id=tier_id, cancel=cancel) - return {"ok": True, "message": result.get("message"), "payload": result} - - return _billing_call(rid, call) + return _billing_call( + rid, + lambda: _billing_pending_change(put_subscription_pending_change(subscription_type_id=tier_id, cancel=cancel)), + ) @method("subscription.resume") @@ -2292,21 +1977,14 @@ def _(rid, params: dict) -> dict: """DELETE /api/billing/subscription/pending-change: clear a scheduled downgrade / cancellation. Re-enables recurring spend → billing:manage + kill-switch.""" from hermes_cli.nous_billing import delete_subscription_pending_change - def call(): - result = delete_subscription_pending_change() - return {"ok": True, "message": result.get("message"), "payload": result} - - return _billing_call(rid, call) + return _billing_call(rid, lambda: _billing_pending_change(delete_subscription_pending_change())) @method("subscription.upgrade") def _(rid, params: dict) -> dict: - """POST /api/billing/subscription/upgrade — the single money route: prorate + charge + flip plan. - - SCA / decline come back as status requires_action / payment_failed with a - recovery_url. The idempotency key is minted if absent and echoed (also on - error) so the TUI reuses it on retry of the SAME upgrade. billing:manage. - """ + """POST /api/billing/subscription/upgrade — the money route (prorate + charge + flip plan). + SCA / decline → status requires_action / payment_failed + recovery_url. Idempotency key + minted if absent and echoed (also on error) for retry of the SAME upgrade. billing:manage.""" from agent.billing_view import new_idempotency_key from hermes_cli.nous_billing import post_subscription_upgrade tier_id = params.get("subscription_type_id") @@ -2317,14 +1995,11 @@ def _(rid, params: dict) -> dict: def call(): result = post_subscription_upgrade(subscription_type_id=tier_id, idempotency_key=key) return { - "ok": True, - "status": result.get("status"), - "target_tier_name": result.get("targetTierName"), - "recovery_url": result.get("recoveryUrl"), - "reason": result.get("reason"), + **_billing_pick( + result, status="status", target_tier_name="targetTierName", recovery_url="recoveryUrl", reason="reason" + ), "idempotency_key": key, } - return _billing_call(rid, call, extra={"idempotency_key": key}) @@ -2338,12 +2013,14 @@ def _(rid, params: dict) -> dict: if amount is None: return _billing_invalid(rid, "amount_usd is required") key = params.get("idempotency_key") or new_idempotency_key() - - def call(): - result = post_charge(amount_usd=amount, idempotency_key=key) - return {"ok": True, "charge_id": result.get("chargeId"), "idempotency_key": key} - - return _billing_call(rid, call, extra={"idempotency_key": key}) + return _billing_call( + rid, + lambda: { + **_billing_pick(post_charge(amount_usd=amount, idempotency_key=key), charge_id="chargeId"), + "idempotency_key": key, + }, + extra={"idempotency_key": key}, + ) @method("billing.charge_status") @@ -2353,18 +2030,12 @@ def _(rid, params: dict) -> dict: charge_id = params.get("charge_id") if not charge_id: return _billing_invalid(rid, "charge_id is required", error="invalid_charge_id") - - def call(): - result = get_charge_status(charge_id) - return { - "ok": True, - "status": result.get("status"), - "amount_usd": result.get("amountUsd"), - "settled_at": result.get("settledAt"), - "reason": result.get("reason"), - } - - return _billing_call(rid, call) + return _billing_call( + rid, + lambda: _billing_pick( + get_charge_status(charge_id), status="status", amount_usd="amountUsd", settled_at="settledAt", reason="reason" + ), + ) @method("billing.auto_reload") @@ -2380,76 +2051,67 @@ def _(rid, params: dict) -> dict: def call(): patch_auto_top_up(enabled=enabled, threshold=threshold, top_up_amount=top_up_amount) return {"ok": True} - return _billing_call(rid, call) @method("billing.step_up") def _(rid, params: dict) -> dict: - """Lazy billing:manage step-up device flow → {ok, granted}; granted:false when the - server silently downscopes. - - Runs on the thread pool (_LONG_HANDLERS): the device flow blocks for minutes. - The verification URL/code reach the TUI via the out-of-band - ``billing.step_up.verification`` event (a print would be lost in the JSON-RPC - stdout pipe) and the browser is opened TUI-side — never via the gateway's - headless webbrowser.open (open_browser=False). - """ + """billing:manage step-up device flow → {ok, granted} (false when the server downscopes). + Runs on the pool (_LONG_HANDLERS; blocks for minutes). URL/code reach the TUI via the + ``billing.step_up.verification`` event (stdout is the RPC pipe) and the browser opens + TUI-side, never via the gateway's headless webbrowser.open.""" sid = params.get("session_id") or "" def call(): from hermes_cli.auth import step_up_nous_billing_scope def _on_verification(url: str, code: str) -> None: _emit("billing.step_up.verification", sid, {"verification_url": url, "user_code": code}) - granted = step_up_nous_billing_scope(open_browser=False, on_verification=_on_verification) return {"ok": True, "granted": bool(granted)} - return _billing_call(rid, call, extra={"granted": False}) # ── session status / history / undo / compress / save / close ──────── +def _status_row(session: dict, params: dict, key: str) -> dict: + """Stored row for ``key``: the live session's bound profile db first, else params.profile / launch.""" + if not key: + return {} + with _session_db(session) as db: + if db is not None: + return _try_get_session(db, key) + with _profile_db(params) as db2: + return _try_get_session(db2, key) if db2 else {} + + +def _try_get_session(db, key: str) -> dict: + try: + return db.get_session(key) or {} + except Exception: + return {} + + +def _status_dt(value, fallback=None): + if value: + with contextlib.suppress(Exception): + return datetime.fromtimestamp(float(value)) + return fallback or datetime.now() + + @method("session.status") @_with_session def _(rid, params: dict, session: dict) -> dict: from hermes_constants import display_hermes_home key = session.get("session_key") or params.get("session_id") or "" agent = session.get("agent") - - def _row(db) -> dict: - try: - return db.get_session(key) or {} - except Exception: - return {} - - meta = {} - # Prefer the live session's bound profile db, else params.profile / launch. - with _session_db(session) as db: - if db is not None: - if key: - meta = _row(db) - else: - with _profile_db(params) as db2: - if db2 and key: - meta = _row(db2) - - def _dt(value, fallback: datetime | None = None) -> datetime: - if value: - try: - return datetime.fromtimestamp(float(value)) - except Exception: - pass - return fallback or datetime.now() - - created = _dt(meta.get("started_at")) + meta = _status_row(session, params, key) + created = _status_dt(meta.get("started_at")) updated = created for field in ("updated_at", "last_updated_at", "last_activity_at"): if meta.get(field): - updated = _dt(meta.get(field), created) + updated = _status_dt(meta.get(field), created) break - mirror = _metadata_mirror(session) usage = _session_usage_snapshot(session) provider = getattr(agent, "provider", None) or mirror.get("provider") or "unknown" @@ -2463,8 +2125,7 @@ def _(rid, params: dict, session: dict) -> dict: lines.append(f"Title: {title}") lines.extend( [ - f"Model: {model} ({provider})", - f"Created: {created.strftime('%Y-%m-%d %H:%M')}", + f"Model: {model} ({provider})", f"Created: {created.strftime('%Y-%m-%d %H:%M')}", f"Last Activity: {updated.strftime('%Y-%m-%d %H:%M')}", f"Tokens: {int(usage.get('total') or 0):,}", f"Agent Running: {'Yes' if session.get('running') else 'No'}", @@ -2480,23 +2141,19 @@ def _(rid, params: dict, session: dict) -> dict: if session.get("session_key"): with _session_db(session) as db: if db is not None: - try: - # include_row_ids: the durable row id is how clients address a - # persisted turn (reactions, content-based truncation targets); - # _history_to_messages only forwards row_id when stamped. + # include_row_ids: the durable row id is how clients address a persisted + # turn (reactions, truncation targets); _history_to_messages forwards it. + with contextlib.suppress(Exception): history = db.get_messages_as_conversation( session["session_key"], include_ancestors=True, include_row_ids=True ) - except Exception: - pass return _ok(rid, {"count": len(history), "messages": _history_to_messages(history)}) @method("session.undo") @_with_live_session def _(rid, params: dict, session: dict) -> dict: - # Mutating history under a running turn would make prompt.submit's post-run - # write either clobber the undo or drop the agent's output — /interrupt first. + # Under a running turn the post-run write would clobber the undo — /interrupt first. busy = _err(rid, 4009, "session busy — /interrupt the current turn before /undo") if session.get("running"): return busy @@ -2505,8 +2162,7 @@ def _(rid, params: dict, session: dict) -> dict: if session.get("running"): return busy history = _history_without_ephemeral_scaffolding(session.get("history", [])) - # Truncate from the last *real* user turn: popping trailing assistant/tool - # then one user left timeline markers / compaction handoffs as the target. + # Truncate from the last *real* user turn (not a timeline marker / compaction handoff). from agent.context_compressor import user_originated_turn_view user_indices = [ index for index, message in enumerate(history) if user_originated_turn_view(message) is not None @@ -2521,6 +2177,13 @@ def _(rid, params: dict, session: dict) -> dict: return _ok(rid, {"removed": removed}) +def _compute_host_ack_error(rid, ack: dict, code: int, default: str): + """``_err`` for a ``control.error``/``error`` ack, else None.""" + if ack.get("type") in {"control.error", "error"}: + return _err(rid, code, str(ack.get("message") or default)) + return None + + def _compress_via_compute_host(rid, params: dict, session: dict) -> dict: """``session.compress`` for a turn-isolated session: forward ``/compress`` to the host.""" sid = str(params.get("session_id") or "") @@ -2529,7 +2192,6 @@ def _compress_via_compute_host(rid, params: dict, session: dict) -> dict: def _on_late_ack(late: dict, _sid=sid) -> None: _adopt_late_compute_host_compress_ack(_sid, session, late, route_name="session.compress") - try: ack = _send_compute_host_control( sid, @@ -2541,10 +2203,8 @@ def _compress_via_compute_host(rid, params: dict, session: dict) -> dict: on_late_ack=_on_late_ack, ) except queue.Empty: - # The waiter gave up but the host is still compressing; the late-ack - # handler adopts the rotated session and pushes session.info when it - # lands. Not an error (a 5019 here made clients report a timeout while - # compression later succeeded silently). + # Waiter gave up, host still compressing; the late-ack handler adopts the rotated + # session when it lands. Not an error (5019 reported timeouts that later succeeded). return _ok( rid, { @@ -2558,18 +2218,16 @@ def _compress_via_compute_host(rid, params: dict, session: dict) -> dict: ) except Exception as exc: return _err(rid, 5019, f"compute-host compress failed: {exc}") - if ack.get("type") in {"control.error", "error"}: - return _err(rid, 4009, str(ack.get("message") or "compute-host compress failed")) + if (resp := _compute_host_ack_error(rid, ack, 4009, "compute-host compress failed")) is not None: + return resp _apply_compute_host_metadata_mirror(session, ack) host_result = ack.get("result") if isinstance(host_result, dict): - # The host owns the isolated session; preserve its structured result - # verbatim (it carries `status: aborted` / `summary.aborted`). + # Host-owned result verbatim (carries `status: aborted` / `summary.aborted`). return _ok(rid, {**host_result, "turn_isolation": True}) host_info = ack.get("session_info") if isinstance(ack.get("session_info"), dict) else {} host_messages = _history_to_messages(ack.get("messages")) if isinstance(ack.get("messages"), list) else [] - # `messages` goes at top level for the transcript replacement; don't send the - # same (large) transcript a second time inside the ack. + # `messages` goes top-level for the transcript replacement; don't duplicate it in the ack. host_ack = {key: value for key, value in ack.items() if key != "messages"} return _ok( rid, @@ -2612,23 +2270,16 @@ def _(rid, params: dict) -> dict: def _tokens(msgs, sys_prompt, tools) -> int: return estimate_request_tokens_rough(msgs, system_prompt=sys_prompt, tools=tools) if msgs else 0 - before_tokens = _tokens(before_messages, _sys_prompt, _tools) - if before_count >= 4: focus_suffix = f', focus: "{focus_topic}"' if focus_topic else "" _status_update( - sid, - "compressing", + sid, "compressing", f"⠋ compressing {before_count} messages (~{before_tokens:,} tok){focus_suffix}…", ) - try: removed, usage = _compress_session_history( - session, - focus_topic, - approx_tokens=before_tokens, - before_messages=before_messages, + session, focus_topic, approx_tokens=before_tokens, before_messages=before_messages, history_version=history_version, ) with session["history_lock"]: @@ -2636,17 +2287,13 @@ def _(rid, params: dict) -> dict: after_count = len(messages) # Re-read prompt + tools: _compress_context may have rebuilt the system prompt. after_tokens = _tokens( - messages, - getattr(_agent, "_cached_system_prompt", "") or _sys_prompt, + messages, getattr(_agent, "_cached_system_prompt", "") or _sys_prompt, getattr(_agent, "tools", None) or _tools, ) agent = session["agent"] _sync_session_key_after_compress(sid, session) summary = summarize_manual_compression( - before_messages, - messages, - before_tokens, - after_tokens, + before_messages, messages, before_tokens, after_tokens, compression_state=getattr(agent, "context_compressor", None), ) info = _session_info(agent, session) @@ -2664,8 +2311,7 @@ def _(rid, params: dict) -> dict: "summary": summary, "usage": usage, "info": info, - # Same projection as session.resume / session.history: raw tool - # results belong in persisted history, not the transcript response. + # Same projection as session.resume / session.history. "messages": _history_to_messages(messages), }, ) @@ -2690,26 +2336,22 @@ def _(rid, params: dict, session: dict) -> dict: ack = _send_compute_host_control(sid, route_name="session.save", wait=True) except Exception as exc: return _err(rid, 5011, f"compute-host session save failed: {exc}") - if ack.get("type") in {"control.error", "error"}: - return _err(rid, 5011, str(ack.get("message") or "compute-host session save failed")) + if (resp := _compute_host_ack_error(rid, ack, 5011, "compute-host session save failed")) is not None: + return resp result = ack.get("result") if not isinstance(result, dict): return _err(rid, 5011, "compute-host session save returned an invalid response") return _ok(rid, result) - agent = session["agent"] - # Mirror the classic CLI /save: snapshot under the profile home (not the - # workspace cwd) and include the system prompt so it matches the dashboard save. + # Classic CLI /save: under the profile home, with the system prompt (dashboard parity). saved_dir = get_hermes_home() / "sessions" / "saved" try: saved_dir.mkdir(parents=True, exist_ok=True) except Exception as e: return _err(rid, 5011, f"failed to create save directory {saved_dir}: {e}") - path = saved_dir / f"hermes_conversation_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json" with session["history_lock"]: messages = list(session.get("history", [])) - session_id = getattr(agent, "session_id", None) or session.get("session_key") or "" # Prefer the agent's session_start (classic CLI export); else the gateway created_at. agent_start = getattr(agent, "session_start", None) @@ -2718,7 +2360,6 @@ def _(rid, params: dict, session: dict) -> dict: else: created_at = session.get("created_at") session_start = datetime.fromtimestamp(created_at).isoformat() if isinstance(created_at, (int, float)) else "" - try: with open(path, "w", encoding="utf-8") as f: json.dump( @@ -2741,8 +2382,7 @@ def _(rid, params: dict, session: dict) -> dict: @method("session.close") def _(rid, params: dict) -> dict: sid = params.get("session_id", "") - # Serialize only the ownership claim against session.resume / the reaper; - # finalization may run arbitrary plugin cleanup and must not block other resumes. + # Lock only the ownership claim; finalization (plugin cleanup) must not block resumes. with _session_resume_lock: session = _pop_session_by_id(sid) closed = _teardown_popped_session(session, end_reason="tui_close") @@ -2753,8 +2393,8 @@ def _(rid, params: dict) -> dict: def _visible_branch_history(messages) -> list: - """user/assistant rows with visible text, as FULL row copies (reasoning fields and - timeline-marker tags — display_kind/display_metadata — must survive the branch).""" + """user/assistant rows with visible text, as FULL row copies (reasoning + timeline-marker + tags must survive the branch).""" visible = [] for message in messages or []: if not isinstance(message, dict) or message.get("role") not in {"user", "assistant"}: @@ -2765,14 +2405,48 @@ def _visible_branch_history(messages) -> list: return visible +def _build_branch_agent(session: dict, new_sid: str, new_key: str, history: list, source: str): + """Build + register the branched agent bound to the parent's profile (home + secret scope, + the profile's own state.db handle). The DEDICATED handle is ours until + ``_transfer_db_to_agent`` (unconditional drop, as session.resume); released here on failure.""" + parent_home = session.get("profile_home") + branch_db = None + branch_owns_db = False + try: + if parent_home: + from hermes_state import get_shared_session_db + branch_db = get_shared_session_db(Path(parent_home) / "state.db") + branch_owns_db = True + with _profile_build_scope(parent_home): + agent = _make_agent_in_context( + new_sid, new_key, session_db=branch_db, platform_override=source, + context_cwd_is_launch_artifact=_context_cwd_is_launch_artifact(session), + ) + _init_session( + new_sid, new_key, agent, list(history), cols=session.get("cols", 80), + cwd=_session_cwd(session), session_db=branch_db, source=source, profile_home=parent_home, + explicit_cwd=bool(session.get("explicit_cwd")), + ) + _transfer_db_to_agent(agent, branch_db) + branch_owns_db = False + if new_sid in _sessions: + _sessions[new_sid]["active_session_lease"] = None # claimed lazily on the first turn + return agent + finally: + if branch_owns_db and branch_db is not None: + with contextlib.suppress(Exception): + from hermes_state import release_or_close + release_or_close(branch_db) + + _BRANCH_COPY_FIELDS = ( "reasoning", "reasoning_content", "reasoning_details", "codex_reasoning_items", "codex_message_items", - # Timeline markers ride as role=user; dropping the tag re-plants them as bare - # user turns after a restart, corrupting the truncate ordinal address space. + # Timeline markers ride as role=user; without the tag they become bare user turns + # after a restart, corrupting the truncate ordinal address space. "display_kind", "display_metadata", # Branch copies are history, not new activity: keep the parent's timestamps. @@ -2783,8 +2457,7 @@ _BRANCH_COPY_FIELDS = ( @method("session.branch") @_with_live_session def _(rid, params: dict, session: dict) -> dict: - # Branch writes into the parent's profile-scoped state.db (app-global remote - # mode); the launch handle would orphan branch rows + history. + # Write into the parent's profile-scoped state.db; the launch handle would orphan rows. with _session_db(session) as db: if db is None: return _db_unavailable_error(rid, code=5008) @@ -2796,9 +2469,8 @@ def _(rid, params: dict, session: dict) -> dict: if isinstance(msg, dict) ] - # The live history is the MODEL projection — after compaction only a - # summary + protected tail. Snapshot the persisted display projection - # instead, or the child permanently loses every turn archived before the fork. + # Live history is the MODEL projection (post-compaction: summary + tail). Snapshot + # the persisted display projection or the child loses every archived turn. history = None get_resume_conversations = getattr(db, "get_resume_conversations", None) if callable(get_resume_conversations): @@ -2833,58 +2505,10 @@ def _(rid, params: dict, session: dict) -> dict: _copy_branch_transcript(db, new_key, title, history, _BRANCH_COPY_FIELDS) except Exception as e: return _err(rid, 5008, f"branch failed: {e}") - # Bound before the try so the ownership finally can never see them unbound. - branch_db = None - branch_owns_db = False try: - # Bind the branched AGENT to the parent's profile like session.create/ - # resume: home + secret scope for the build, and the profile's own state.db - # handle so message flushes and later compression rotation persist there. - parent_home = session.get("profile_home") - if parent_home: - # DEDICATED handle, same ownership rule as session.resume: ours until - # the branched agent takes it below. - from hermes_state import get_shared_session_db - branch_db = get_shared_session_db(Path(parent_home) / "state.db") - branch_owns_db = True - with _profile_build_scope(parent_home): - tokens = _set_session_context(new_key) - try: - agent = _make_agent( - new_sid, - new_key, - session_id=new_key, - session_db=branch_db, - platform_override=source, - context_cwd_is_launch_artifact=_context_cwd_is_launch_artifact(session), - ) - finally: - _clear_session_context(tokens) - _init_session( - new_sid, - new_key, - agent, - list(history), - cols=session.get("cols", 80), - cwd=_session_cwd(session), - session_db=branch_db, - source=source, - profile_home=parent_home, - explicit_cwd=bool(session.get("explicit_cwd")), - ) - # Ownership TRANSFER (unconditional drop, as in session.resume): past - # _init_session the branched session is registered against this handle. - _transfer_db_to_agent(agent, branch_db) - branch_owns_db = False - if new_sid in _sessions: - _sessions[new_sid]["active_session_lease"] = None # claimed lazily on the first turn + agent = _build_branch_agent(session, new_sid, new_key, history, source) except Exception as e: return _err(rid, 5000, f"agent init failed on branch: {e}") - finally: - if branch_owns_db and branch_db is not None: - with contextlib.suppress(Exception): - from hermes_state import release_or_close - release_or_close(branch_db) branched_session = _sessions.get(new_sid) return _ok( rid, @@ -2931,10 +2555,9 @@ def _(rid, params: dict) -> dict: if err: return err _interrupt_session_turn(str(params.get("session_id") or ""), session) - # Retire the crash-recovery marker on a confirmed local Stop now: waiting for - # the run thread's finally leaves a window where a backend exit looks like a - # crash and session.resume auto-continues the turn the user just stopped. - # Extra key covers compression rotating session_key mid-turn. + # Retire the crash-recovery marker NOW: until the run thread's finally, a backend exit + # looks like a crash and session.resume auto-continues the turn the user just stopped. + # The extra key covers compression rotating session_key mid-turn. with session["history_lock"]: active_marker_key = str(session.pop("_active_turn_marker_key", "") or "") _retire_turn_marker(session, active_marker_key) @@ -2942,9 +2565,8 @@ def _(rid, params: dict) -> dict: def _record_accepted_correction(session: dict, text: str) -> None: - """Record a steer/redirect on the live turn so a mid-turn resume rebuilds the user - bubble, and purge server-queue self-copies of the live original so post-turn - drain cannot re-fire the pre-correction prompt.""" + """Record a steer/redirect on the live turn (mid-turn resume rebuilds the bubble) and purge + queued self-copies so post-turn drain cannot re-fire the pre-correction prompt.""" with session["history_lock"]: _record_inflight_correction(session, text) _drop_queued_duplicates_of_inflight_user(session) @@ -2953,11 +2575,8 @@ def _record_accepted_correction(session: dict, text: str) -> None: @method("session.steer") def _(rid, params: dict) -> dict: - """Inject a user message into the next tool result without interrupting. - - Mirrors AIAgent.steer(): the text lands on the last tool result of the next - tool batch. No interrupt, no new user turn, no role alternation violation. - """ + """Inject text into the next tool result without interrupting (AIAgent.steer(): no new + user turn, no role alternation violation).""" text = (params.get("text") or "").strip() if not text: return _err(rid, 4002, "text is required") @@ -2986,9 +2605,8 @@ def _(rid, params: dict) -> dict: if err: return err agent = session.get("agent") - # Turn-build window: a fresh turn flips running=True with agent still None. - # Queue the correction server-side for the next turn instead of a misleading - # 4010 the client swallows into a lost follow-up. + # Turn-build window (running=True, agent None): queue for the next turn instead of a + # misleading 4010 the client swallows into a lost follow-up. if agent is None and session.get("running"): _enqueue_prompt(session, text, current_transport() or _stdio_transport) session["last_active"] = time.time() @@ -3014,12 +2632,8 @@ def _(rid, params: dict) -> dict: @method("delegation.status") def _(rid, params: dict) -> dict: from tools.delegate_tool import ( - is_spawn_paused, - list_active_subagents, - _get_max_concurrent_children, - _get_max_spawn_depth, + is_spawn_paused, list_active_subagents, _get_max_concurrent_children, _get_max_spawn_depth, ) - return _ok( rid, { @@ -3048,13 +2662,9 @@ def _(rid, params: dict) -> dict: @method("subagent.steer") def _(rid, params: dict) -> dict: - """Queue steering text into a live delegated child without stopping it. - - Resolves the child in the delegation registry and calls AIAgent.steer(); the - in-flight tool call is never cut. "queued" is not "delivered": a child past - its final tool batch has no boundary left, and that race surfaces as - ``missed_steer`` on the parent's completion entry. - """ + """Queue steering text into a live delegated child (AIAgent.steer(); the in-flight tool call + is never cut). "queued" is not "delivered": a child past its final tool batch surfaces + the race as ``missed_steer`` on the parent's completion entry.""" from tools.delegate_tool import steer_subagent subagent_id = str(params.get("subagent_id") or "").strip() if not subagent_id: @@ -3070,10 +2680,7 @@ def _(rid, params: dict) -> dict: queued = False if invoking_transport is not None and invoking_session is not None: queued = steer_subagent( - subagent_id, - text, - owner_session_id=invoking_session_id, - owner_transport=invoking_transport, + subagent_id, text, owner_session_id=invoking_session_id, owner_transport=invoking_transport, owner_session_record=invoking_session, ) return _ok(rid, {"status": "queued" if queued else "rejected", "subagent_id": subagent_id, "text": text}) @@ -3085,7 +2692,6 @@ def _(rid, params: dict) -> dict: subagents = params.get("subagents") or [] if not isinstance(subagents, list) or not subagents: return _err(rid, 4000, "subagents list required") - started_at = params.get("started_at") finished_at = params.get("finished_at") or time.time() label = str(params.get("label") or "") @@ -3103,7 +2709,6 @@ def _(rid, params: dict) -> dict: path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") except OSError as exc: return _err(rid, 5000, f"spawn_tree.save failed: {exc}") - _append_spawn_tree_index( d, { @@ -3118,6 +2723,27 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"path": str(path), "session_id": session_id}) +def _legacy_spawn_tree_entry(p, session_dir_name: str) -> dict | None: + """Index-shaped entry for a pre-index snapshot file (None when unreadable).""" + try: + stat = p.stat() + except OSError: + return None + try: + raw = json.loads(p.read_text(encoding="utf-8")) + except Exception: + raw = {} + subagents = raw.get("subagents") or [] + return { + "path": str(p), + "session_id": raw.get("session_id") or session_dir_name, + "finished_at": raw.get("finished_at") or stat.st_mtime, + "started_at": raw.get("started_at"), + "label": raw.get("label") or "", + "count": len(subagents) if isinstance(subagents, list) else 0, + } + + @method("spawn_tree.list") def _(rid, params: dict) -> dict: session_id = str(params.get("session_id") or "").strip() @@ -3126,7 +2752,6 @@ def _(rid, params: dict) -> dict: roots = [p for p in _spawn_trees_root().iterdir() if p.is_dir()] else: roots = [_spawn_tree_session_dir(session_id or "default")] - entries: list[dict] = [] for d in roots: indexed = _read_spawn_tree_index(d) @@ -3136,28 +2761,8 @@ def _(rid, params: dict) -> dict: continue # Legacy (pre-index) sessions: full scan, once per session until the next save. for p in d.glob("*.json"): - if p.name == _SPAWN_TREE_INDEX: - continue - try: - stat = p.stat() - try: - raw = json.loads(p.read_text(encoding="utf-8")) - except Exception: - raw = {} - subagents = raw.get("subagents") or [] - entries.append( - { - "path": str(p), - "session_id": raw.get("session_id") or d.name, - "finished_at": raw.get("finished_at") or stat.st_mtime, - "started_at": raw.get("started_at"), - "label": raw.get("label") or "", - "count": len(subagents) if isinstance(subagents, list) else 0, - } - ) - except OSError: - continue - + if p.name != _SPAWN_TREE_INDEX and (entry := _legacy_spawn_tree_entry(p, d.name)) is not None: + entries.append(entry) entries.sort(key=lambda e: e.get("finished_at") or 0, reverse=True) return _ok(rid, {"entries": entries[:limit]}) @@ -3193,11 +2798,8 @@ def _(rid, params: dict, session: dict) -> dict: @method("session.events.since") def _(rid, params: dict) -> dict: - """Replay recorded events newer than the client's last-seen seq (WS reconnect contract). - - Frames older than the ring window report ``truncated`` so the client refetches - history instead of silently accepting a gap. - """ + """Replay events newer than the client's last-seen seq (WS reconnect). Frames older than + the ring window report ``truncated`` so the client refetches instead of accepting a gap.""" sid = str(params.get("session_id") or "") try: last_seen = int(params.get("last_seen", 0)) @@ -3212,8 +2814,7 @@ def _(rid, params: dict) -> dict: "latest_seq": event_replay.latest_seq(sid), "truncated": event_replay.is_truncated(sid, last_seen), "count": len(frames), - # seq counters are in-process: clients compare this against the epoch - # from gateway.ready and reset watermarks on mismatch (restart detection). + # In-process seq: clients reset watermarks when this differs from gateway.ready's. "epoch": event_replay.replay_epoch(), }, ) @@ -3227,10 +2828,5 @@ def _(rid, params: dict) -> dict: def register(server) -> None: - """Publish this module's helpers onto ``server`` and install its handlers. - - Helpers are module-level functions/classes, so install() alone would leave them - bound to THIS module's (empty) globals; ``bind_module`` rebinds them onto - server.py's namespace so they resolve the same free names as the handlers. - """ + """Publish this module's helpers onto ``server`` (rebound to its globals) and install handlers.""" bind_module(globals(), server, skip=("_",)) diff --git a/tui_gateway/methods_slash.py b/tui_gateway/methods_slash.py index 70e66be887..b8155536e1 100644 --- a/tui_gateway/methods_slash.py +++ b/tui_gateway/methods_slash.py @@ -1,4 +1,4 @@ -"""slash.exec helpers: command resolution, side-effect mirroring after a slash command ran in the worker. +"""slash.exec helpers: live-session command output and side-effect mirroring after a slash command ran in the worker. Bodies are rebound onto server.py's globals at install time (see method_ctx.bind_module), so they reference server.py globals bare. @@ -7,61 +7,44 @@ method_ctx.bind_module), so they reference server.py globals bare. from __future__ import annotations +import contextlib + from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# ── Methods: slash.exec ────────────────────────────────────────────── - +# ── Live-session slash output ──────────────────────────────────────── _LIVE_SESSION_DIRECT_COMMANDS = frozenset( - { - "clear", - "compress", - "effort", - "history", - "models", - "prompt", - "rename", - "review", - "status", - "usage", - } + {"clear", "compress", "effort", "history", "models", "prompt", "rename", "review", "status", "usage"} ) - +# Answered from the live session ONLY when the agent lives on a compute host. _ISOLATED_SESSION_READ_COMMANDS = frozenset({"context", "tools", "help"}) +_NO_AGENT_USAGE = "(._.) No active agent -- send a message first." +_NO_AGENT = "No active agent -- send a message first." + def _format_live_review_output(session: Optional[dict], arg: str) -> str: - """Dispatch /review against the live TUI/desktop session's agent. + """Dispatch /review against the live session's agent. - Spawns the reviewer subagent on the async delegation rail; the TUI - notification poller already drains async-delegation completions for the - owning session, so the finished review re-enters this chat as a normal - completion turn. The dispatch stamps the parent agent's durable - session_id as the completion's session_key (the delegate_task CLI-path - fallback), which is exactly what ``_session_owns_notification_event`` - matches against. + The reviewer subagent runs on the async delegation rail; the TUI notification + poller drains its completion back into this chat. The dispatch stamps the + parent's durable session_id as the completion's session_key, which is what + ``_session_owns_notification_event`` matches against. """ if session is None: return "Nothing to review yet — send a message first." if _session_uses_compute_host(session): - return ( - "/review runs on the local agent only for now — this session's " - "agent lives on a remote compute host." - ) + return "/review runs on the local agent only for now — this session's agent lives on a remote compute host." agent = session.get("agent") if agent is None: return "Nothing to review yet — send a message first." if session.get("running"): return "session busy — wait for the current turn to finish, then /review" - history_lock = session.get("history_lock") - if history_lock is not None: - with history_lock: - snapshot = list(session.get("history", [])) - else: + with session.get("history_lock") or contextlib.nullcontext(): snapshot = list(session.get("history", [])) if not snapshot: snapshot = list(getattr(agent, "_session_messages", None) or []) @@ -81,61 +64,60 @@ def _format_live_usage_output(session: dict) -> str: agent = session.get("agent") usage = _session_usage_snapshot(session) if agent is None and not usage: - return "(._.) No active agent -- send a message first." + return _NO_AGENT_USAGE if session.get("_metadata_message_count") is not None: message_count = int(session.get("_metadata_message_count") or 0) else: with session["history_lock"]: message_count = len(session.get("history", [])) + + def n(key: str) -> str: + return f"{int(usage.get(key) or 0):,}" + lines = [ "Session Token Usage", "────────────────────────────────────────", f"Model: {usage.get('model') or _metadata_mirror(session).get('model') or getattr(agent, 'model', '') or '(unknown)'}", - f"Input tokens: {int(usage.get('input') or 0):,}", - f"Output tokens: {int(usage.get('output') or 0):,}", + f"Input tokens: {n('input')}", + f"Output tokens: {n('output')}", + ] + if int(usage.get("reasoning") or 0): + lines.append(f"Reasoning tokens: {n('reasoning')}") + lines += [ + f"Prompt tokens: {n('prompt')}", + f"Completion tokens: {n('completion')}", + f"Total tokens: {n('total')}", + f"API calls: {n('calls')}", ] - reasoning = int(usage.get("reasoning") or 0) - if reasoning: - lines.append(f"Reasoning tokens: {reasoning:,}") - lines.extend( - [ - f"Prompt tokens: {int(usage.get('prompt') or 0):,}", - f"Completion tokens: {int(usage.get('completion') or 0):,}", - f"Total tokens: {int(usage.get('total') or 0):,}", - f"API calls: {int(usage.get('calls') or 0):,}", - ] - ) if usage.get("context_max"): lines.append( - "Current context: " - f"{int(usage.get('context_used') or 0):,} / " - f"{int(usage.get('context_max') or 0):,} " + f"Current context: {n('context_used')} / {n('context_max')} " f"({int(usage.get('context_percent') or 0)}%)" ) - lines.extend( - [ - f"Messages: {message_count:,}", - f"Compressions: {int(usage.get('compressions') or 0):,}", - ] - ) + lines += [f"Messages: {message_count:,}", f"Compressions: {n('compressions')}"] return "\n".join(lines) +def _live_session_messages(session: dict) -> Optional[list]: + """Session-scoped transcript read; None when no db/key or the read fails. Uses + ``_session_db`` (not ``_get_db()``): a profile session's rows live in its own + profile's state.db, and through the launch handle this read comes back empty.""" + with _session_db(session) as db: + if db is not None and session.get("session_key"): + try: + return db.get_messages_as_conversation( + session["session_key"], include_ancestors=True, include_row_ids=True + ) + except Exception: + pass + return None + + def _format_live_history_output(session: dict) -> str: with session["history_lock"]: history = list(session.get("history", [])) - # _session_db, not _get_db(): a profile session's transcript lives in its - # own profile's state.db, and this read is scoped by session id — through - # the launch handle it comes back empty and /history renders nothing. - with _session_db(session) as db: - if db is not None and session.get("session_key"): - try: - history = db.get_messages_as_conversation( - session["session_key"], include_ancestors=True, include_row_ids=True - ) - except Exception: - pass - messages = _history_to_messages(history) + db_history = _live_session_messages(session) + messages = _history_to_messages(history if db_history is None else db_history) if not messages: return "No conversation history yet." lines = ["Conversation History", "────────────────────────────────────────"] @@ -153,7 +135,7 @@ def _format_live_prompt_output(session: dict) -> str: agent = session.get("agent") mirror = _metadata_mirror(session) if agent is None and "system_prompt" not in mirror: - return "No active agent -- send a message first." + return _NO_AGENT prompt = ( mirror.get("system_prompt") or getattr(agent, "ephemeral_system_prompt", None) @@ -166,27 +148,16 @@ def _format_live_prompt_output(session: dict) -> str: def _format_live_context_output(session: dict) -> str: - messages = [] - # Same session-scoped read as /history — resolve it against the db that - # owns this session's rows, not the launch profile's handle. - with _session_db(session) as db: - if db is not None and session.get("session_key"): - try: - messages = _history_to_messages( - db.get_messages_as_conversation( - session["session_key"], include_ancestors=True, include_row_ids=True - ) - ) - except Exception: - messages = [] + try: + messages = _history_to_messages(_live_session_messages(session) or []) + except Exception: + messages = [] # malformed db rows fall back to the live history below if not messages: with session["history_lock"]: messages = _history_to_messages(list(session.get("history", []))) usage = _session_usage_snapshot(session) mirror = _metadata_mirror(session) - lines = [ - f"Conversation: {len(messages)} messages" if messages else "Conversation is empty (no messages yet)." - ] + lines = [f"Conversation: {len(messages)} messages" if messages else "Conversation is empty (no messages yet)."] roles: dict[str, int] = {} for msg in messages: role = str(msg.get("role") or "unknown") @@ -196,20 +167,17 @@ def _format_live_context_output(session: dict) -> str: f"tool: {roles.get('tool', 0)}, system: {roles.get('system', 0)}" ) model = mirror.get("model") or usage.get("model") or "" - provider = mirror.get("provider") or "auto" if model: lines.append(f"Model: {model}") - lines.append(f"Provider: {provider}") + lines.append(f"Provider: {mirror.get('provider') or 'auto'}") context_used = int(usage.get("context_used") or usage.get("total") or 0) context_max = int(usage.get("context_max") or 0) - if context_used: - if context_max: - usage_pct = (context_used / context_max) * 100 - lines.append( - f"Context usage: ~{context_used:,} / {context_max:,} tokens ({usage_pct:.1f}%)" - ) - else: - lines.append(f"Context usage: ~{context_used:,} tokens") + if context_used and context_max: + lines.append( + f"Context usage: ~{context_used:,} / {context_max:,} tokens ({(context_used / context_max) * 100:.1f}%)" + ) + elif context_used: + lines.append(f"Context usage: ~{context_used:,} tokens") if usage.get("compressions"): lines.append(f"Compressions: {int(usage.get('compressions') or 0):,}") return "\n".join(lines) @@ -220,16 +188,10 @@ def _format_live_tools_output(session: dict) -> str: groups = info.get("tools") if isinstance(info, dict) else {} if not isinstance(groups, dict) or not groups: return "No tools available." - names: list[str] = [] - for group_names in groups.values(): - if isinstance(group_names, list): - names.extend(str(name) for name in group_names) - names = sorted(set(names)) + names = sorted({str(n) for g in groups.values() if isinstance(g, list) for n in g}) if not names: return "No tools available." - return "Available tools ({}):\n{}".format( - len(names), "\n".join(f" {name}" for name in names) - ) + return "Available tools ({}):\n{}".format(len(names), "\n".join(f" {name}" for name in names)) def _format_live_help_output() -> str: @@ -239,8 +201,7 @@ def _format_live_help_output() -> str: lines = ["Available commands:", ""] for category, commands in COMMANDS_BY_CATEGORY.items(): lines.append(f"{category}:") - for cmd, desc in commands.items(): - lines.append(f" {cmd:<15} {desc}") + lines.extend(f" {cmd:<15} {desc}" for cmd, desc in commands.items()) return "\n".join(lines) except Exception as exc: return f"help unavailable: {exc}" @@ -252,244 +213,267 @@ def _format_live_model_output(session: dict) -> str: provider = getattr(agent, "provider", "") if agent is not None else "" if model and provider: return f"Current model: {model} ({provider})" - if model: - return f"Current model: {model}" - return "Current model: (unknown)" + return f"Current model: {model}" if model else "Current model: (unknown)" + + +def _format_live_status_output(sid: str) -> str: + response = _methods["session.status"]("status", {"session_id": sid}) + if response.get("error"): + return str(response["error"].get("message") or "status unavailable") + return str(response.get("result", {}).get("output") or "") + + +# name → (reply when there is no session, formatter(sid, session, arg)). A None +# no-session reply means the formatter handles a missing session itself. +_LIVE_SLASH_OUTPUT = { + "compress": ("no active session for /compress", lambda sid, s, a: _mirror_slash_side_effects(sid, s, f"/compress {a}".strip())), + "usage": (_NO_AGENT_USAGE, lambda sid, s, a: _format_live_usage_output(s)), + "review": (None, lambda sid, s, a: _format_live_review_output(s, a)), + "history": ("No conversation history yet.", lambda sid, s, a: _format_live_history_output(s)), + "prompt": (_NO_AGENT, lambda sid, s, a: _format_live_prompt_output(s)), + "status": (None, lambda sid, s, a: _format_live_status_output(sid)), + "context": ("Conversation is empty (no messages yet).", lambda sid, s, a: _format_live_context_output(s)), + "tools": ("No tools available.", lambda sid, s, a: _format_live_tools_output(s)), + "help": (None, lambda sid, s, a: _format_live_help_output()), + "clear": (None, lambda sid, s, a: "Screen clear is terminal-only; desktop/TUI chat left unchanged."), + "models": (None, lambda sid, s, a: "Use /model to view or switch the current model; desktop users can also open the model picker."), + "rename": (None, lambda sid, s, a: "Use /title to rename this session."), + "effort": (None, lambda sid, s, a: "Use /reasoning to change reasoning effort."), +} def _live_slash_command_output(sid: str, session: Optional[dict], name: str, arg: str) -> Optional[str]: + """Answer a slash command from the live session instead of the slash worker; None = not ours.""" name = (name or "").lstrip("/").lower() arg = arg or "" if name == "model" and not arg.strip(): return _format_live_model_output(session or {}) - if name not in _LIVE_SESSION_DIRECT_COMMANDS: - if not ( - name in _ISOLATED_SESSION_READ_COMMANDS - and session is not None - and _session_uses_compute_host(session) - ): + if name in _ISOLATED_SESSION_READ_COMMANDS: + if not (session is not None and _session_uses_compute_host(session)): return None - - if name in _ISOLATED_SESSION_READ_COMMANDS and not ( - session is not None and _session_uses_compute_host(session) - ): + elif name not in _LIVE_SESSION_DIRECT_COMMANDS: return None - if name == "compress": - if session is None: - return "no active session for /compress" - return _mirror_slash_side_effects(sid, session, f"/compress {arg}".strip()) - if name == "usage": - if session is None: - return "(._.) No active agent -- send a message first." - return _format_live_usage_output(session) - if name == "review": - return _format_live_review_output(session, arg) - if name == "history": - if session is None: - return "No conversation history yet." - return _format_live_history_output(session) - if name == "prompt": - if session is None: - return "No active agent -- send a message first." - return _format_live_prompt_output(session) - if name == "status": - response = _methods["session.status"]("status", {"session_id": sid}) - if response.get("error"): - return str(response["error"].get("message") or "status unavailable") - return str(response.get("result", {}).get("output") or "") - if name == "context": - if session is None: - return "Conversation is empty (no messages yet)." - return _format_live_context_output(session) - if name == "tools": - if session is None: - return "No tools available." - return _format_live_tools_output(session) - if name == "help": - return _format_live_help_output() - if name == "clear": - return "Screen clear is terminal-only; desktop/TUI chat left unchanged." - if name == "models": - return "Use /model to view or switch the current model; desktop users can also open the model picker." - if name == "rename": - return "Use /title to rename this session." - if name == "effort": - return "Use /reasoning to change reasoning effort." - return None + entry = _LIVE_SLASH_OUTPUT.get(name) + if entry is None: + return None + no_session_reply, fmt = entry + if session is None and no_session_reply is not None: + return no_session_reply + return fmt(sid, session, arg) +# ── Side-effect mirroring ──────────────────────────────────────────── + +# Read-then-mutate live agent/session state that a running turn is using; rejected +# while running (parity with session.compress / session.undo and the gateway's +# running-agent /model guard). +_MUTATES_WHILE_RUNNING = frozenset({"model", "personality", "prompt", "compress"}) + + +def _compress_live_with_feedback(sid: str, session: dict, agent, arg: str, *, snapshot_kwargs: bool) -> dict: + """Compress the live session; return the ``summarize_manual_compression`` dict. + + Shared by command.dispatch /compress and the slash mirror so every route shows + "compressed N → M messages / ~X → ~Y tokens". ``snapshot_kwargs`` forwards the + pre-read snapshot (approx_tokens/before_messages/history_version) to + ``_compress_session_history``; the slash mirror passes only the raw arg. The raw + arg goes through unparsed — the choke point parses ``here [N]`` / ``--keep N``. + CompressionLockHeld and other errors propagate to the caller, which finalizes + the deferred context-engine notification. + """ + from agent.conversation_compression import finalize_context_engine_compression_notification + from agent.manual_compression_feedback import summarize_manual_compression + from agent.model_metadata import estimate_request_tokens_rough + + with session["history_lock"]: + before_messages = list(session.get("history", [])) + history_version = int(session.get("history_version", 0)) + sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" + tools = getattr(agent, "tools", None) or None + before_tokens = ( + estimate_request_tokens_rough(before_messages, system_prompt=sys_prompt, tools=tools) if before_messages else 0 + ) + if snapshot_kwargs: + _compress_session_history( + session, + arg.strip() or None, + approx_tokens=before_tokens, + before_messages=before_messages, + history_version=history_version, + ) + else: + _compress_session_history(session, arg) + _sync_session_key_after_compress(sid, session) + with session["history_lock"]: + after_messages = list(session.get("history", [])) + after_tokens = ( + estimate_request_tokens_rough( + after_messages, + system_prompt=getattr(agent, "_cached_system_prompt", "") or sys_prompt, + tools=getattr(agent, "tools", None) or tools, + ) + if after_messages + else 0 + ) + _emit("session.info", sid, _session_info(agent, session)) + fb = summarize_manual_compression( + before_messages, + after_messages, + before_tokens, + after_tokens, + compression_state=getattr(agent, "context_compressor", None), + ) + finalize_context_engine_compression_notification(agent, committed=True) + return fb + + +def _mirror_model(sid, session, agent, arg) -> str: + if arg and agent: + return _apply_model_switch(sid, session, arg).get("warning", "") + return "" + + +def _mirror_approvals(sid, session, agent, arg) -> str: + # The worker already persisted approvals.mode; the bare read-only form needs no repaint. + if arg: + broadcast_session_info() + return "" + + +def _mirror_personality(sid, session, agent, arg) -> str: + if arg and agent: + pname, new_prompt = _validate_personality(arg, _load_cfg()) + # Persist through the single owner so this surface never drifts from the others. + from hermes_cli.personality import persist_personality + + persist_personality(pname) + _apply_personality_to_session(sid, session, new_prompt, pname) + return "" + + +def _mirror_prompt(sid, session, agent, arg) -> str: + if agent: + cfg = _load_cfg() + agent.ephemeral_system_prompt = _prompt_text((cfg.get("agent") or {}).get("system_prompt", "")) or None + agent._cached_system_prompt = None + return "" + + +def _mirror_compress(sid, session, agent, arg) -> str: + if not agent: + return "" + try: + fb = _compress_live_with_feedback(sid, session, agent, arg, snapshot_kwargs=False) + except CompressionLockHeld as e: + from agent.manual_compression_feedback import describe_compression_lock_skip + + return describe_compression_lock_skip(e.holder) + lines = [fb["headline"], fb["token_line"]] + if fb.get("note"): + lines.append(fb["note"]) + return "\n".join(lines) + + +def _mirror_fast(sid, session, agent, arg) -> str: + if agent: + mode = arg.lower() + if mode in {"fast", "on"}: + agent.service_tier = "priority" + elif mode in {"normal", "off"}: + agent.service_tier = None + elif mode in {"auto", "cold"}: + agent.service_tier = mode + _emit("session.info", sid, _session_info(agent, session)) + return "" + + +def _mirror_reload_mcp(sid, session, agent, arg) -> str: + if agent and hasattr(agent, "reload_mcp_tools"): + agent.reload_mcp_tools() + return "" + + +def _mirror_stop(sid, session, agent, arg) -> str: + from tools.process_registry import process_registry + + process_registry.kill_all() + return "" + + +_SLASH_MIRRORS = { + "model": _mirror_model, + "approvals": _mirror_approvals, + "personality": _mirror_personality, + "prompt": _mirror_prompt, + "compress": _mirror_compress, + "fast": _mirror_fast, + "reload-mcp": _mirror_reload_mcp, + "stop": _mirror_stop, +} + + +def _compute_host_slash(sid: str, session: dict, name: str, command: str) -> tuple[str, str]: + """Forward a mutating slash command to the session's compute host. + + Returns ``(status, text)``: ``pending`` (compress still running after the wait), + ``failed`` (transport error/timeout), ``rejected`` (host control.error), ``ok`` + (host output; metadata mirror already applied). Compress waits longer and installs + a late-ack adopter so a slow compression still lands in this session. + """ + route_name = f"slash.{name}" + is_compress = name == "compress" + _late_session = session + + def _on_late_ack(late: dict, _sid=sid) -> None: + _adopt_late_compute_host_compress_ack(_sid, _late_session, late, route_name=route_name) + + try: + ack = _send_compute_host_control( + sid, + route_name=route_name, + command=command, + wait=True, + **({"timeout": _compute_host_compress_wait_seconds(), "on_late_ack": _on_late_ack} if is_compress else {}), + ) + except queue.Empty: + if is_compress: + return "pending", "compression still running in the background; the transcript will refresh when it finishes" + return "failed", f"compute-host {route_name} failed: timed out" + except Exception as exc: + return "failed", f"compute-host {route_name} failed: {exc}" + if ack.get("type") in {"control.error", "error"}: + return "rejected", str(ack.get("message") or f"compute-host {route_name} failed") + _apply_compute_host_metadata_mirror(session, ack) + return "ok", str(ack.get("output") or "") + def _mirror_slash_side_effects(sid: str, session: dict, command: str) -> str: """Apply side effects that must also hit the gateway's live agent.""" parts = command.lstrip("/").split(None, 1) if not parts: return "" - name, arg, agent = ( - parts[0], - (parts[1].strip() if len(parts) > 1 else ""), - session.get("agent"), - ) + name, arg, agent = parts[0], (parts[1].strip() if len(parts) > 1 else ""), session.get("agent") if name == "compact": - # /compact is an alias of /compress in every host. The compute-host - # slash.compress control forwards the user's raw alias verbatim, so - # without normalizing here the child mirror silently no-ops — the - # session never compresses and the deferred context-engine - # notification wiring below is never exercised for that route. + # /compact aliases /compress everywhere; the compute-host control forwards the + # raw alias verbatim, so without this the child mirror silently no-ops. name = "compress" - # Reject agent-mutating commands during an in-flight turn. These - # all do read-then-mutate on live agent/session state that the - # worker thread running agent.run_conversation is using. Parity - # with the session.compress / session.undo guards and the gateway - # runner's running-agent /model guard. - _MUTATES_WHILE_RUNNING = {"model", "personality", "prompt", "compress"} if _session_uses_compute_host(session) and name in _MUTATES_WHILE_RUNNING: - route_name = f"slash.{name}" - is_compress = name == "compress" - _late_session = session - - def _on_late_ack(late: dict, _sid=sid) -> None: - _adopt_late_compute_host_compress_ack(_sid, _late_session, late, route_name=route_name) - - try: - ack = _send_compute_host_control( - sid, - route_name=route_name, - command=command, - wait=True, - **( - {"timeout": _compute_host_compress_wait_seconds(), "on_late_ack": _on_late_ack} - if is_compress - else {} - ), - ) - except queue.Empty: - if is_compress: - return "compression still running in the background; the transcript will refresh when it finishes" - return f"compute-host {route_name} failed: timed out" - except Exception as exc: - return f"compute-host {route_name} failed: {exc}" - if ack.get("type") in {"control.error", "error"}: - return str(ack.get("message") or f"compute-host {route_name} failed") - _apply_compute_host_metadata_mirror(session, ack) - return str(ack.get("output") or "") + return _compute_host_slash(sid, session, name, command)[1] if name in _MUTATES_WHILE_RUNNING and session.get("running"): return f"session busy — /interrupt the current turn before running /{name}" + mirror = _SLASH_MIRRORS.get(name) + if mirror is None: + return "" try: - if name == "model" and arg and agent: - result = _apply_model_switch(sid, session, arg) - return result.get("warning", "") - elif name == "approvals" and arg: - # The slash worker already persisted the new approvals.mode; the - # bare (read-only) form has no arg and needs no repaint. - broadcast_session_info() - elif name == "personality" and arg and agent: - pname, new_prompt = _validate_personality(arg, _load_cfg()) - # Persist through the single owner so this surface can never - # drift from the others (the old TUI slash path applied the - # overlay in-session but skipped persistence entirely). - from hermes_cli.personality import persist_personality - - persist_personality(pname) - _apply_personality_to_session(sid, session, new_prompt, pname) - elif name == "prompt" and agent: - cfg = _load_cfg() - new_prompt = _prompt_text((cfg.get("agent") or {}).get("system_prompt", "")) - agent.ephemeral_system_prompt = new_prompt or None - agent._cached_system_prompt = None - elif name == "compress" and agent: - # Mirror the session.compress RPC: build a before/after summary so - # the user gets feedback (#46686). The slash path previously just - # compressed + emitted session.info and returned "", so the TUI - # showed no "compressed N → M messages / ~X → ~Y tokens" stats - # while CLI and gateway both did. - from agent.manual_compression_feedback import summarize_manual_compression - from agent.model_metadata import estimate_request_tokens_rough - from agent.conversation_compression import ( - finalize_context_engine_compression_notification, - ) - - with session["history_lock"]: - _before_messages = list(session.get("history", [])) - _before_count = len(_before_messages) - _sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" - _tools = getattr(agent, "tools", None) or None - _before_tokens = ( - estimate_request_tokens_rough( - _before_messages, system_prompt=_sys_prompt, tools=_tools - ) - if _before_count - else 0 - ) - - # The raw argument goes through unparsed: _compress_session_history - # (the choke point shared by all three manual-compress routes) - # parses the boundary-aware forms (here [N], up to here, --keep N) - # and does the partial head/tail split there (#35533). - try: - _compress_session_history(session, arg) - except CompressionLockHeld as e: - from agent.manual_compression_feedback import ( - describe_compression_lock_skip, - ) - return describe_compression_lock_skip(e.holder) - _sync_session_key_after_compress(sid, session) - - with session["history_lock"]: - _after_messages = list(session.get("history", [])) - _sys_prompt_after = getattr(agent, "_cached_system_prompt", "") or _sys_prompt - _tools_after = getattr(agent, "tools", None) or _tools - _after_tokens = ( - estimate_request_tokens_rough( - _after_messages, system_prompt=_sys_prompt_after, tools=_tools_after - ) - if _after_messages - else 0 - ) - _emit("session.info", sid, _session_info(agent, session)) - _fb = summarize_manual_compression( - _before_messages, - _after_messages, - _before_tokens, - _after_tokens, - compression_state=getattr(agent, "context_compressor", None), - ) - _lines = [_fb["headline"], _fb["token_line"]] - if _fb.get("note"): - _lines.append(_fb["note"]) - finalize_context_engine_compression_notification( - agent, - committed=True, - ) - return "\n".join(_lines) - elif name == "fast" and agent: - mode = arg.lower() - if mode in {"fast", "on"}: - agent.service_tier = "priority" - elif mode in {"normal", "off"}: - agent.service_tier = None - elif mode in {"auto", "cold"}: - agent.service_tier = mode - _emit("session.info", sid, _session_info(agent, session)) - elif name == "reload-mcp" and agent and hasattr(agent, "reload_mcp_tools"): - agent.reload_mcp_tools() - elif name == "stop": - from tools.process_registry import process_registry - - process_registry.kill_all() + return mirror(sid, session, agent, arg) except Exception as e: if name == "compress" and agent: - from agent.conversation_compression import ( - finalize_context_engine_compression_notification, - ) + from agent.conversation_compression import finalize_context_engine_compression_notification - finalize_context_engine_compression_notification( - agent, - committed=False, - ) + finalize_context_engine_compression_notification(agent, committed=False) return f"live session sync failed: {e}" - return "" def register(server) -> None: diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index 3344574b0a..6002256861 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -1,10 +1,8 @@ """Tools & system / slash / insights / rollback / plugins / cron / skills / MCP JSON-RPC handlers. -Everything defined here is rebound onto server.py's globals at install time -(``method_ctx.bind_module``), so handler bodies AND module-level helpers may -reference server globals bare (``_ok``, ``_err``, ``_sessions``, ...). Names -must not collide with server.py's own; helpers here use a ``_cmd_`` / -``_slash_`` / ``_toolset_`` / ``_mcp_`` prefix. +Rebound onto server.py's globals at install time (``method_ctx.bind_module``), so +bodies reference server globals bare (``_ok``, ``_err``, ``_sessions``, ...). +Helper names must not collide with server.py's own (``_cmd_`` / ``_toolset_`` / ``_mcp_`` prefixes). """ import sys @@ -19,15 +17,15 @@ _profile_scoped = _registry.profile_scoped # ─── Shared helpers ────────────────────────────────────────────────────────── -def _profile_scoped_rpc(fail_code: int, *, required=(), catch_resolve: bool = True): +def _profile_scoped_rpc(fail_code: int, *, required=(), catch_resolve: bool = True, prefix: str = "", scoped: bool = True): """Wrap a handler body with the optional ``profile`` HERMES_HOME scope. - Order preserved from the original handlers: ``required`` params are checked - first (4063 `` required``), then the profile is resolved (4064 when the - profile dir is missing), then the body runs; any body exception becomes - ``fail_code``. ``catch_resolve`` also maps resolve-time exceptions to - ``fail_code`` (cron/skills/catalog); the mcp.servers.* handlers let them - propagate to dispatch(). The override is always reset afterwards. + Order: ``required`` params checked first (4063 `` required``), then the + profile resolved (4064 when its dir is missing), then the body; body exceptions + become ``fail_code`` (message prefixed with ``prefix``). ``catch_resolve`` also maps + resolve-time exceptions to ``fail_code`` (cron/skills/catalog); mcp.servers.* let + them propagate to dispatch(). The override is always reset afterwards. + ``scoped=False`` (see ``_guarded``) ignores ``profile`` entirely. """ def deco(body): @@ -35,7 +33,7 @@ def _profile_scoped_rpc(fail_code: int, *, required=(), catch_resolve: bool = Tr for key, present in required: if not present(params.get(key)): return _err(rid, 4063, f"{key} required") - profile = str(params.get("profile") or "").strip() + profile = str(params.get("profile") or "").strip() if scoped else "" token = None if profile: try: @@ -53,7 +51,7 @@ def _profile_scoped_rpc(fail_code: int, *, required=(), catch_resolve: bool = Tr try: return body(rid, params) except Exception as e: - return _err(rid, fail_code, str(e)) + return _err(rid, fail_code, f"{prefix}{e}") finally: _mcp_reset_profile(token) @@ -63,6 +61,11 @@ def _profile_scoped_rpc(fail_code: int, *, required=(), catch_resolve: bool = Tr return deco +def _guarded(fail_code: int, prefix: str = ""): + """Handler body exceptions → ``_err(rid, fail_code, prefix + str(e))``.""" + return _profile_scoped_rpc(fail_code, prefix=prefix, scoped=False) + + def _stripped(v) -> bool: return bool(str(v or "").strip()) @@ -80,12 +83,32 @@ def _mcp_server_scoped(body): return _profile_scoped_rpc(5024, required=_NAME, catch_resolve=False)(body) +def _mcp_named_server(rid, params): + """(name, servers, None) for a configured server, else (name, servers, 4064 error).""" + from hermes_cli.mcp_config import _get_mcp_servers + + name = str(params.get("name") or "").strip() + servers = _get_mcp_servers() + err = None if name in servers else _err(rid, 4064, f"server '{name}' not found") + return name, servers, err + + def _busy_error(rid, session, cmd: str): if session.get("running"): return _err(rid, 4009, f"session busy — /interrupt the current turn before /{cmd}") return None +def _session_key_or_err(rid, session): + """(session_key, None) or (None, 4001 error) for the /goal and /loop managers.""" + if not session: + return None, _err(rid, 4001, "no active session") + sid_key = session.get("session_key") or "" + if not sid_key: + return None, _err(rid, 4001, "no session key") + return sid_key, None + + def _user_turn_indices(session): """(history, indices of user-originated turns) minus ephemeral scaffolding. Call under history_lock.""" from agent.context_compressor import user_originated_turn_view @@ -98,6 +121,23 @@ def _clip(text: str, n: int = 120) -> str: return text[:n] + ("…" if len(text) > n else "") +def _capture_run_kwargs(timeout: int) -> dict: + """subprocess.run kwargs shared by cli.exec / shell.exec / quick commands: captured + text, UTF-8 + lossy decode (non-UTF-8 child output must not crash the gateway thread + on locale-mismatched Windows), no stdin, no console flash under the desktop parent.""" + from hermes_cli._subprocess_compat import windows_hide_flags + + return dict( + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + timeout=timeout, + stdin=subprocess.DEVNULL, + creationflags=windows_hide_flags(), + ) + + def _toolset_rows(params: dict, *, with_tools: bool) -> list[dict]: from toolsets import get_all_toolsets, get_toolset_info @@ -146,13 +186,11 @@ def _(rid, params: dict) -> dict: @method("process.stop") +@_guarded(5010) def _(rid, params: dict) -> dict: - try: - from tools.process_registry import process_registry + from tools.process_registry import process_registry - return _ok(rid, {"killed": process_registry.kill_all()}) - except Exception as e: - return _err(rid, 5010, str(e)) + return _ok(rid, {"killed": process_registry.kill_all()}) @method("process.list") @@ -187,273 +225,236 @@ def _(rid, params: dict) -> dict: return _err(rid, 5010, str(e)) +def _mcp_reload_confirm_required() -> bool: + """``approvals.mcp_reload_confirm`` from disk config; True (safe) on any failure.""" + try: + from hermes_cli.config import load_config + + cfg = load_config() + approvals = cfg.get("approvals") if isinstance(cfg, dict) else None + return bool(approvals.get("mcp_reload_confirm", True)) if isinstance(approvals, dict) else True + except Exception: + return True + + @method("reload.mcp") +@_guarded(5015) def _(rid, params: dict) -> dict: session = _sessions.get(params.get("session_id", "")) - try: - # /reload-mcp invalidates the prompt cache. Unless the caller passed - # confirm=true, honour ``approvals.mcp_reload_confirm`` (default true) by - # returning a confirm_required payload instead of reloading; Ink prints - # ``message`` and re-invokes with confirm=true (or flips the config). - if not bool(params.get("confirm", False)): - try: - from hermes_cli.config import load_config as _load_config + # /reload-mcp invalidates the prompt cache: without confirm=true, honour + # ``approvals.mcp_reload_confirm`` (default true) by returning confirm_required; + # Ink prints ``message`` and re-invokes with confirm=true (or flips the config). + if not bool(params.get("confirm", False)) and _mcp_reload_confirm_required(): + message = ( + "⚠️ /reload-mcp invalidates the prompt cache (next message re-sends full input tokens). " + "Reply `/reload-mcp now` to proceed, or `/reload-mcp always` to proceed and " + "silence this prompt permanently." + ) + return _ok(rid, {"status": "confirm_required", "message": message}) - _cfg = _load_config() - _approvals = _cfg.get("approvals") if isinstance(_cfg, dict) else None - _confirm_required = True - if isinstance(_approvals, dict): - _confirm_required = bool(_approvals.get("mcp_reload_confirm", True)) - except Exception: - _confirm_required = True - if _confirm_required: - return _ok( - rid, - { - "status": "confirm_required", - "message": ( - "⚠️ /reload-mcp invalidates the prompt cache (next " - "message re-sends full input tokens). Reply `/reload-mcp " - "now` to proceed, or `/reload-mcp always` to proceed and " - "silence this prompt permanently." - ), - }, - ) + if session and _session_uses_compute_host(session): + try: + ack = _get_compute_host_supervisor().reload_mcp( + str(params.get("session_id") or ""), request_id=f"reload-mcp-{rid}" + ) + except Exception as exc: + return _err(rid, 5019, f"compute-host reload_mcp failed: {exc}") + return _ok(rid, {"status": "reloaded", "turn_isolation": True, "host_ack": ack}) - if session and _session_uses_compute_host(session): - try: - ack = _get_compute_host_supervisor().reload_mcp( - str(params.get("session_id") or ""), request_id=f"reload-mcp-{rid}" - ) - except Exception as exc: - return _err(rid, 5019, f"compute-host reload_mcp failed: {exc}") - return _ok(rid, {"status": "reloaded", "turn_isolation": True, "host_ack": ack}) + from tools.mcp_tool import shutdown_mcp_servers, discover_mcp_tools, reprobe_tool_availability - from tools.mcp_tool import shutdown_mcp_servers, discover_mcp_tools, reprobe_tool_availability + def _refresh_session_agent() -> None: + """Rebuild THIS session's cached tool snapshot from the live registry and push + session.info (the agent never re-reads the registry itself; mirrors + gateway/run.py::_execute_mcp_reload). Runs under _mcp_reload_lock so a + concurrent reload can't tear the registry down mid-refresh.""" + if not session: + return + agent = session["agent"] + try: + from tools.mcp_tool import refresh_agent_mcp_tools - def _refresh_session_agent() -> None: - """Rebuild THIS session's cached tool snapshot from the live registry and - push session.info (the agent never re-reads the registry on its own; - mirrors gateway/run.py::_execute_mcp_reload). Runs under _mcp_reload_lock - so a concurrent reload can't tear the registry down mid-refresh.""" - if not session: - return - agent = session["agent"] - try: - from tools.mcp_tool import refresh_agent_mcp_tools + # enabled_override re-resolves toolsets so a server enabled in config this session is picked up. + refresh_agent_mcp_tools(agent, enabled_override=_load_enabled_toolsets(), quiet_mode=True) + except Exception as _exc: + logger.warning("Failed to refresh cached agent tools after /reload-mcp: %s", _exc) + _emit("session.info", params.get("session_id", ""), _session_info(agent, session)) - # Re-resolve enabled toolsets so a server enabled in config this - # session is picked up. - refresh_agent_mcp_tools(agent, enabled_override=_load_enabled_toolsets(), quiet_mode=True) - except Exception as _exc: - logger.warning("Failed to refresh cached agent tools after /reload-mcp: %s", _exc) - _emit("session.info", params.get("session_id", ""), _session_info(agent, session)) + global _mcp_reload_gen, _mcp_reload_loaded_rev + # Revision the CALLER wants loaded (the mcp_rev its poll observed); empty on + # legacy clients / manual /reload-mcp, which coalesce on generation alone. + req_rev = str(params.get("rev") or "") + + def _do_full_reload() -> None: + """shutdown+discover+refresh under the lock, then mark a completed generation. + The lock spans the refresh too, else a second reload could tear the registry + down mid-rebuild. Config can change WHILE discover connects: re-hash after + discovery and repeat until stable so the marked generation matches what loaded.""" global _mcp_reload_gen, _mcp_reload_loaded_rev - # Revision the CALLER wants loaded (the mcp_rev its poll observed). Empty - # on legacy clients / manual /reload-mcp — those coalesce on generation alone. - req_rev = str(params.get("rev") or "") + loaded = _compute_mcp_rev() + for _ in range(_MCP_RELOAD_MAX_PASSES): + shutdown_mcp_servers() + reprobe_tool_availability() + discover_mcp_tools() + after = _compute_mcp_rev() + if after == loaded: + break + loaded = after - def _do_full_reload() -> None: - """shutdown+discover+refresh under the lock, then mark a completed generation. - The lock spans the refresh too, else a second reload could tear the registry - down while this one is still rebuilding the session snapshot. Config can - change WHILE discover connects: re-hash after discovery and repeat until - stable, so the marked generation reflects the config actually loaded.""" - global _mcp_reload_gen, _mcp_reload_loaded_rev + _refresh_session_agent() + _mcp_reload_loaded_rev = loaded + _mcp_reload_gen += 1 - loaded = _compute_mcp_rev() - for _ in range(_MCP_RELOAD_MAX_PASSES): - shutdown_mcp_servers() - reprobe_tool_availability() - discover_mcp_tools() - after = _compute_mcp_rev() - if after == loaded: - break - loaded = after + # LEADER (won the non-blocking acquire) runs the full reload. FOLLOWER snapshots + # the generation, waits, then — still holding the lock — coalesces only if a + # reload COMPLETED meanwhile (generation advanced ⇒ leader didn't throw) AND it + # loaded the requested revision; otherwise it re-runs the full reload so a + # failed/stale leader never leaves a follower acking an unloaded revision. + if _mcp_reload_lock.acquire(blocking=False): + try: + _do_full_reload() + finally: + _mcp_reload_lock.release() + return _finish_reload(rid, params, coalesced=False) + + gen_before = _mcp_reload_gen + + with _mcp_reload_lock: + leader_completed = _mcp_reload_gen > gen_before + rev_satisfied = not req_rev or req_rev == _mcp_reload_loaded_rev + + if leader_completed and rev_satisfied: _refresh_session_agent() - _mcp_reload_loaded_rev = loaded - _mcp_reload_gen += 1 + coalesced = True + else: + _do_full_reload() + coalesced = False - # Serialize reloads. LEADER (won the non-blocking acquire) runs the full - # reload. FOLLOWER snapshots the generation, waits, then — still holding the - # lock — coalesces only if a reload COMPLETED meanwhile (generation advanced, - # so the leader didn't throw) AND it loaded the revision this request asked - # for; otherwise it re-runs the full reload so a failed/stale leader can - # never leave a follower acking a revision that was never loaded. - if _mcp_reload_lock.acquire(blocking=False): - try: - _do_full_reload() - finally: - _mcp_reload_lock.release() - - return _finish_reload(rid, params, coalesced=False) - - gen_before = _mcp_reload_gen - - with _mcp_reload_lock: - leader_completed = _mcp_reload_gen > gen_before - rev_satisfied = not req_rev or req_rev == _mcp_reload_loaded_rev - - if leader_completed and rev_satisfied: - _refresh_session_agent() - coalesced = True - else: - _do_full_reload() - coalesced = False - - return _finish_reload(rid, params, coalesced=coalesced) - except Exception as e: - return _err(rid, 5015, str(e)) + return _finish_reload(rid, params, coalesced=coalesced) @method("reload.env") +@_guarded(5015) def _(rid, params: dict) -> dict: - """Re-read ``~/.hermes/.env`` into the gateway (classic CLI ``/reload`` parity). + """Re-read ``~/.hermes/.env`` (classic CLI ``/reload`` parity). Already-built agents + keep their credential pool / provider routing; ``/new`` resolves fresh.""" + from hermes_cli.config import reload_env - Already-constructed agents keep their credential pool / provider routing — - same as classic CLI; ``/new`` gets a fresh credential resolution. - """ - try: - from hermes_cli.config import reload_env - - count = reload_env() - return _ok(rid, {"updated": int(count)}) - except Exception as e: - return _err(rid, 5015, str(e)) + return _ok(rid, {"updated": int(reload_env())}) # ─── Command catalog / dispatch ────────────────────────────────────────────── @method("commands.catalog") +@_guarded(5020) def _(rid, params: dict) -> dict: """Registry-backed slash metadata for the TUI — categorized, no aliases.""" + from hermes_cli.commands import COMMAND_REGISTRY, SUBCOMMANDS, _build_description, command_desktop_meta + + all_pairs: list[list[str]] = [] + canon: dict[str, str] = {} + commands: dict[str, dict[str, str | None]] = {} + cat_map: dict[str, list[list[str]]] = {} + cat_order: list[str] = [] + + def bucket(cat: str) -> list[list[str]]: + if cat not in cat_map: + cat_map[cat] = [] + cat_order.append(cat) + return cat_map[cat] + + def add(key: str, desc: str, rows: list[list[str]]) -> None: + canon[key.lower()] = key + all_pairs.append([key, desc]) + rows.append([key, desc]) + + for cmd in COMMAND_REGISTRY: + meta = command_desktop_meta(cmd) + commands[f"/{cmd.name}"] = dict(meta) + for alias in cmd.aliases: + commands[f"/{alias}"] = dict(meta) + if cmd.name in _TUI_HIDDEN or cmd.gateway_only: + continue + c = f"/{cmd.name}" + add(c, _build_description(cmd), bucket(cmd.category)) + for a in cmd.aliases: + canon[f"/{a}".lower()] = c + + for name, desc, cat in _TUI_EXTRA: + # Registry command/alias wins over a colliding TUI extra (e.g. /compact, /sessions). + if name.lower() not in canon: + add(name, desc, bucket(cat)) + + warning = "" try: - from hermes_cli.commands import COMMAND_REGISTRY, SUBCOMMANDS, _build_description, command_desktop_meta - - all_pairs: list[list[str]] = [] - canon: dict[str, str] = {} - commands: dict[str, dict[str, str | None]] = {} - cat_map: dict[str, list[list[str]]] = {} - cat_order: list[str] = [] - - def bucket(cat: str) -> list[list[str]]: - if cat not in cat_map: - cat_map[cat] = [] - cat_order.append(cat) - return cat_map[cat] - - for cmd in COMMAND_REGISTRY: - meta = command_desktop_meta(cmd) - commands[f"/{cmd.name}"] = dict(meta) - for alias in cmd.aliases: - commands[f"/{alias}"] = dict(meta) - - if cmd.name in _TUI_HIDDEN or cmd.gateway_only: - continue - - c = f"/{cmd.name}" - canon[c.lower()] = c - for a in cmd.aliases: - canon[f"/{a}".lower()] = c - - desc = _build_description(cmd) - all_pairs.append([c, desc]) - bucket(cmd.category).append([c, desc]) - - for name, desc, cat in _TUI_EXTRA: - # A TUI extra colliding with a registry command/alias (e.g. /compact, - # /sessions) is skipped: the registry entry is canonical. - if name.lower() in canon: - continue - canon[name.lower()] = name - all_pairs.append([name, desc]) - bucket(cat).append([name, desc]) - - warning = "" - try: - qcmds = _load_cfg().get("quick_commands", {}) or {} - if isinstance(qcmds, dict) and qcmds: - rows = bucket("User commands") - for qname, qc in sorted(qcmds.items()): - if not isinstance(qc, dict): - continue - key = f"/{qname}" - canon[key.lower()] = key - qtype = qc.get("type", "") - if qtype == "exec": - default_desc = f"exec: {qc.get('command', '')}" - elif qtype == "alias": - default_desc = f"alias → {qc.get('target', '')}" - else: - default_desc = qtype or "quick command" - qdesc = _clip(str(qc.get("description") or default_desc)) - all_pairs.append([key, qdesc]) - rows.append([key, qdesc]) - except Exception as e: - if not warning: - warning = f"quick_commands discovery unavailable: {e}" - - try: - from hermes_cli.plugins import get_plugin_commands - - plugin_cmds = get_plugin_commands() or {} - if plugin_cmds: - rows = bucket("Plugin commands") - for pname, info in sorted(plugin_cmds.items()): - if not isinstance(info, dict): - continue - key = f"/{pname}" - if key.lower() in canon: - continue - canon[key.lower()] = key - pdesc = _clip(str(info.get("description") or "Plugin command")) - all_pairs.append([key, pdesc]) - rows.append([key, pdesc]) - hint = str(info.get("args_hint") or "").strip() - mode = info.get("argument_mode") - if mode not in {"options", "text", "mixed"}: - mode = "text" if hint else None - commands[key] = {"argument_mode": mode, "desktop": None} - except Exception as e: - if not warning: - warning = f"plugin command discovery unavailable: {e}" - - skill_count = 0 - skills: dict[str, dict] = {} - try: - from agent.skill_commands import scan_skill_commands - - # Usage + origin ride along here (not a second RPC): every catalog - # consumer also ranks it, and both sidecars are already loaded. - usage, origin_of = _skill_usage_lookup() - - for k, info in sorted(scan_skill_commands().items()): - all_pairs.append([k, _clip(str(info.get("description", "Skill")))]) - name = str(info.get("name") or k.lstrip("/")) - skills[k] = {"usage": usage(name), "origin": origin_of(name)} - skill_count += 1 - except Exception as e: - warning = f"skill discovery unavailable: {e}" - - return _ok( - rid, - { - "pairs": all_pairs, - "sub": {k: v[:] for k, v in SUBCOMMANDS.items()}, - "canon": canon, - "commands": commands, - "categories": [{"name": cat, "pairs": cat_map[cat]} for cat in cat_order], - "skills": skills, - "skill_count": skill_count, - "warning": warning, - }, - ) + qcmds = _load_cfg().get("quick_commands", {}) or {} + if isinstance(qcmds, dict) and qcmds: + rows = bucket("User commands") + for qname, qc in sorted(qcmds.items()): + if not isinstance(qc, dict): + continue + qtype = qc.get("type", "") + default_desc = { + "exec": f"exec: {qc.get('command', '')}", + "alias": f"alias → {qc.get('target', '')}", + }.get(qtype, qtype or "quick command") + add(f"/{qname}", _clip(str(qc.get("description") or default_desc)), rows) except Exception as e: - return _err(rid, 5020, str(e)) + warning = f"quick_commands discovery unavailable: {e}" + + try: + from hermes_cli.plugins import get_plugin_commands + + plugin_cmds = get_plugin_commands() or {} + if plugin_cmds: + rows = bucket("Plugin commands") + for pname, info in sorted(plugin_cmds.items()): + if not isinstance(info, dict): + continue + key = f"/{pname}" + if key.lower() in canon: + continue + add(key, _clip(str(info.get("description") or "Plugin command")), rows) + hint = str(info.get("args_hint") or "").strip() + mode = info.get("argument_mode") + if mode not in {"options", "text", "mixed"}: + mode = "text" if hint else None + commands[key] = {"argument_mode": mode, "desktop": None} + except Exception as e: + if not warning: + warning = f"plugin command discovery unavailable: {e}" + + skill_count = 0 + skills: dict[str, dict] = {} + try: + from agent.skill_commands import scan_skill_commands + + # Usage + origin ride along (not a second RPC): every catalog consumer also ranks it. + usage, origin_of = _skill_usage_lookup() + + for k, info in sorted(scan_skill_commands().items()): + all_pairs.append([k, _clip(str(info.get("description", "Skill")))]) + name = str(info.get("name") or k.lstrip("/")) + skills[k] = {"usage": usage(name), "origin": origin_of(name)} + skill_count += 1 + except Exception as e: + warning = f"skill discovery unavailable: {e}" + + payload = { + "pairs": all_pairs, + "sub": {k: v[:] for k, v in SUBCOMMANDS.items()}, + "canon": canon, + "commands": commands, + "categories": [{"name": cat, "pairs": cat_map[cat]} for cat in cat_order], + "skills": skills, + "skill_count": skill_count, + "warning": warning, + } + return _ok(rid, payload) @method("cli.exec") @@ -466,24 +467,12 @@ def _(rid, params: dict) -> dict: if hint: return _ok(rid, {"blocked": True, "hint": hint, "code": -1, "output": ""}) try: - # CREATE_NO_WINDOW on Windows: under the windowless desktop parent this - # spawn otherwise flashes a console. - from hermes_cli._subprocess_compat import windows_hide_flags - r = subprocess.run( [sys.executable, "-m", "hermes_cli.main", *argv], - capture_output=True, - text=True, - # UTF-8 + lossy decode: non-UTF-8 child output must not crash the - # gateway thread on locale-mismatched Windows. - encoding="utf-8", - errors="replace", - timeout=min(int(params.get("timeout", 240)), 600), cwd=os.getcwd(), # Can drive the agent → needs provider credentials; tier-1 secrets still stripped. env=hermes_subprocess_env(inherit_credentials=True), - stdin=subprocess.DEVNULL, - creationflags=windows_hide_flags(), + **_capture_run_kwargs(min(int(params.get("timeout", 240)), 600)), ) parts = [r.stdout or "", r.stderr or ""] out = "\n".join(p for p in parts if p).strip() or "(no output)" @@ -495,16 +484,14 @@ def _(rid, params: dict) -> dict: @method("command.resolve") +@_guarded(5012) def _(rid, params: dict) -> dict: - try: - from hermes_cli.commands import resolve_command + from hermes_cli.commands import resolve_command - r = resolve_command(params.get("name", "")) - if r: - return _ok(rid, {"canonical": r.name, "description": r.description, "category": r.category}) - return _err(rid, 4011, f"unknown command: {params.get('name')}") - except Exception as e: - return _err(rid, 5012, str(e)) + r = resolve_command(params.get("name", "")) + if r: + return _ok(rid, {"canonical": r.name, "description": r.description, "category": r.category}) + return _err(rid, 4011, f"unknown command: {params.get('name')}") # command.dispatch stages. Each takes (rid, params, session, name, arg) and @@ -517,25 +504,11 @@ def _dispatch_quick(rid, params, session, name, arg): return None qc = qcmds[name] if qc.get("type") == "exec": - # Sanitized env: quick commands run in the TUI server process, which - # holds every API key in os.environ. + # Sanitized env: the TUI server process holds every API key in os.environ. from tools.environments.local import build_subprocess_env sanitized_env = build_subprocess_env() - from hermes_cli._subprocess_compat import windows_hide_flags - - r = subprocess.run( - qc.get("command", ""), - shell=True, - capture_output=True, - text=True, - encoding="utf-8", - errors="replace", # lossy decode: see cli.exec - timeout=30, - stdin=subprocess.DEVNULL, - env=sanitized_env, - creationflags=windows_hide_flags(), - ) + r = subprocess.run(qc.get("command", ""), shell=True, env=sanitized_env, **_capture_run_kwargs(30)) output = ((r.stdout or "") + ("\n" if r.stdout and r.stderr else "") + (r.stderr or "")).strip()[:4000] if output: from agent.redact import redact_sensitive_text @@ -549,29 +522,64 @@ def _dispatch_quick(rid, params, session, name, arg): return None -def _dispatch_plugin(rid, params, session, name, arg): +def _plugin_command_handler(name: str): try: - from hermes_cli.plugins import get_plugin_command_handler, resolve_plugin_command_result + from hermes_cli.plugins import get_plugin_command_handler + + return get_plugin_command_handler(name) + except Exception: + return None + + +def _is_profile_skill_command(session: dict, base: str) -> bool: + """True when ``/base`` is a skill command of the session's profile. HERMES_HOME is bound + to that profile so get_skill_commands() sees its skills.external_dirs: dispatch() runs on + the pool and nothing upstream binds the override. False on any failure.""" + try: + from agent.skill_commands import get_skill_commands + from hermes_constants import reset_hermes_home_override, set_hermes_home_override + + profile_home = session.get("profile_home") + token = set_hermes_home_override(profile_home) if profile_home else None + try: + return f"/{base}" in get_skill_commands() + finally: + if token is not None: + reset_hermes_home_override(token) + except Exception: + return False + + +def _dispatch_plugin(rid, params, session, name, arg): + handler = _plugin_command_handler(name) + if handler: + try: + from hermes_cli.plugins import resolve_plugin_command_result - handler = get_plugin_command_handler(name) - if handler: result = resolve_plugin_command_result(handler(arg)) return _ok(rid, {"type": "plugin", "output": str(result or "")}) - except Exception: - pass + except Exception: + pass return None -def _dispatch_bundle(rid, params, session, name, arg): +def _bundle_key_for(name: str): + """Skill-bundle key for ``name`` when it is NOT a registry command; None otherwise / on failure.""" try: - from agent.skill_bundles import build_bundle_invocation_message, get_skill_bundles, resolve_bundle_command_key + from agent.skill_bundles import resolve_bundle_command_key from hermes_cli.commands import resolve_command - bundle_key = resolve_bundle_command_key(name) if resolve_command(name) is None else None + return resolve_bundle_command_key(name) if resolve_command(name) is None else None except Exception: - bundle_key = None + return None + + +def _dispatch_bundle(rid, params, session, name, arg): + bundle_key = _bundle_key_for(name) if bundle_key is None: return None + from agent.skill_bundles import build_bundle_invocation_message, get_skill_bundles + try: bundle_result = build_bundle_invocation_message( bundle_key, @@ -589,16 +597,8 @@ def _dispatch_bundle(rid, params, session, name, arg): notice = f"⚡ Loading bundle: {bundle_name} ({len(loaded_names)} skills)" if missing: notice += f"\nSkipped missing skills: {', '.join(missing)}" - return _ok( - rid, - { - "type": "send", - "message": msg, - "notice": notice, - # UIs render `display`, never `message`: the expanded body is model-facing scaffolding. - "display": _skill_scaffold_projection(msg), - }, - ) + # UIs render `display`, never `message`: the expanded body is model-facing scaffolding. + return _ok(rid, {"type": "send", "message": msg, "notice": notice, "display": _skill_scaffold_projection(msg)}) def _dispatch_skill(rid, params, session, name, arg): @@ -610,23 +610,16 @@ def _dispatch_skill(rid, params, session, name, arg): if key in cmds: msg = build_skill_invocation_message(key, arg, task_id=session.get("session_key", "") if session else "") if msg: - return _ok( - rid, - { - "type": "skill", - "message": msg, - "name": cmds[key].get("name", name), - "display": _skill_scaffold_projection(msg), # UIs render this, never `message` - }, - ) + # UIs render `display`, never `message`. + display = _skill_scaffold_projection(msg) + return _ok(rid, {"type": "skill", "message": msg, "name": cmds[key].get("name", name), "display": display}) except Exception: pass return None -# Built-in commands that queue messages onto _pending_input in the CLI. The TUI -# slash worker has no reader for that queue, so they are handled here and return -# a structured payload. +# Built-ins that queue onto _pending_input in the CLI; the TUI slash worker has no +# reader for that queue, so they are handled here and return a structured payload. def _cmd_queue(rid, params, session, name, arg): @@ -636,32 +629,29 @@ def _cmd_queue(rid, params, session, name, arg): def _cmd_learn(rid, params, session, name, arg): - # Standards-guided prompt submitted as a normal turn; the live agent gathers - # sources with its own tools and authors the skill via skill_manage. + # Submitted as a normal turn; the live agent gathers sources and authors the skill via skill_manage. from agent.learn_prompt import build_learn_prompt return _ok(rid, {"type": "send", "message": build_learn_prompt(arg)}) def _cmd_plan(rid, params, session, name, arg): - # Plan-mode prompt as a normal turn (same pattern as /learn); the agent saves - # the plan under .hermes/plans/ via write_file. + # Normal turn (as /learn); the agent saves the plan under .hermes/plans/ via write_file. from agent.plan_prompt import build_plan_prompt return _ok(rid, {"type": "send", "message": build_plan_prompt(arg)}) def _cmd_init(rid, params, session, name, arg): - # Generate-or-update AGENTS.md as a normal turn (same pattern as /learn). + # Generate-or-update AGENTS.md as a normal turn (as /learn). from hermes_cli.init_command import build_init_prompt_for_cwd return _ok(rid, {"type": "send", "message": build_init_prompt_for_cwd(extra=arg)}) def _cmd_moa(rid, params, session, name, arg): - # One-shot sugar: run ONE prompt through the default MoA preset, then restore - # the prior model. Switching for the whole session goes through the model - # picker (MoA presets surface as a virtual "Mixture of Agents" provider). + # One prompt through the default MoA preset, then restore the prior model. Whole-session + # switching goes through the model picker (MoA presets = virtual "Mixture of Agents" provider). try: from hermes_cli.moa_config import moa_usage, normalize_moa_config @@ -671,9 +661,8 @@ def _cmd_moa(rid, params, session, name, arg): return _err(rid, 4001, "no active session") sid = params.get("session_id", "") preset = normalize_moa_config(_load_cfg().get("moa") or {})["default_preset"] - # Record the live model identity for post-turn restore, then swap the - # agent's client in place: setting session["model_override"] alone never - # switches an already-built agent. + # Record the live identity for post-turn restore, then swap the agent's client in + # place: session["model_override"] alone never switches an already-built agent. agent = session.get("agent") session["moa_one_shot_restore"] = { "override": session.get("model_override"), @@ -702,21 +691,14 @@ def _cmd_moa(rid, params, session, name, arg): "api_key": "moa-virtual-provider", "api_mode": "chat_completions", } - return _ok( - rid, - { - "type": "send", - "notice": f"MoA one-shot queued with preset {preset}; previous model will be restored after this turn.", - "message": arg, - }, - ) + notice = f"MoA one-shot queued with preset {preset}; previous model will be restored after this turn." + return _ok(rid, {"type": "send", "notice": notice, "message": arg}) except Exception as exc: return _err(rid, 5030, f"moa unavailable: {exc}") def _cmd_focus(rid, params, session, name, arg): - # Display-only. Routed through the same config.set branch the Ink slash - # command uses so both surfaces share one state machine and persistence path. + # Display-only; routed through the config.set branch Ink uses so both surfaces share one state machine. from hermes_cli.focus_view import format_focus_status, format_focus_toggle_message, resolve_focus_arg _display_focus = _load_cfg().get("display") @@ -734,10 +716,8 @@ def _cmd_focus(rid, params, session, name, arg): if "error" in _res: return _res _payload = _res.get("result") or {} - return _ok( - rid, - {"type": "exec", "output": format_focus_toggle_message(bool(_target), _payload.get("tool_progress") or "all")}, - ) + output = format_focus_toggle_message(bool(_target), _payload.get("tool_progress") or "all") + return _ok(rid, {"type": "exec", "output": output}) def _cmd_retry(rid, params, session, name, arg): @@ -779,13 +759,8 @@ def _cmd_steer(rid, params, session, name, arg): if agent and hasattr(agent, "steer"): try: if agent.steer(arg): - return _ok( - rid, - { - "type": "exec", - "output": f"⏩ Steer queued — arrives after the next tool call: {arg[:80]}{'...' if len(arg) > 80 else ''}", - }, - ) + shown = f"{arg[:80]}{'...' if len(arg) > 80 else ''}" + return _ok(rid, {"type": "exec", "output": f"⏩ Steer queued — arrives after the next tool call: {shown}"}) except Exception: pass # No active run: treat as next-turn message. @@ -799,11 +774,9 @@ def _cmd_goal(rid, params, session, name, arg): from hermes_cli.goals import GoalManager except Exception as exc: return _err(rid, 5030, f"goals unavailable: {exc}") - - sid_key = session.get("session_key") or "" - if not sid_key: - return _err(rid, 4001, "no session key") - + sid_key, err = _session_key_or_err(rid, session) + if err: + return err try: max_turns = int((_load_cfg().get("goals") or {}).get("max_turns", 20) or 20) except Exception: @@ -821,28 +794,19 @@ def _cmd_goal(rid, params, session, name, arg): state = mgr.resume() if state is None: return _ok(rid, {"type": "exec", "output": "No goal to resume."}) - # Resume must restart work, not just flip persisted state: an `exec` - # result is display-only, so return a `send` carrying the continuation - # prompt; `display` keeps the transcript free of model-facing scaffolding. + # Resume must restart work: `exec` is display-only, so return a `send` with the + # continuation prompt; `display` keeps model-facing scaffolding out of the transcript. prompt = mgr.next_continuation_prompt() if not prompt: return _ok(rid, {"type": "exec", "output": f"▶ Goal resumed: {state.goal}"}) - return _ok( - rid, - { - "type": "send", - "notice": f"▶ Goal resumed: {state.goal}\nContinuing now — taking the next step.", - "message": prompt, - "display": "/goal resume", - }, - ) + notice = f"▶ Goal resumed: {state.goal}\nContinuing now — taking the next step." + return _ok(rid, {"type": "send", "notice": notice, "message": prompt, "display": "/goal resume"}) if lower in {"clear", "stop", "done"}: had = mgr.has_goal() mgr.clear() return _ok(rid, {"type": "exec", "output": "✓ Goal cleared." if had else "No active goal."}) - # Remaining text = the new goal. The client renders `notice` as a sys line - # then submits `message`; the post-turn judge in _run_prompt_submit takes over. + # Remaining text = new goal. Client renders `notice`, submits `message`; the post-turn judge takes over. try: state = mgr.set(arg) except ValueError as exc: @@ -856,23 +820,20 @@ def _cmd_goal(rid, params, session, name, arg): def _cmd_loop(rid, params, session, name, arg): - # Recurring in-session wakeups; the notification poller fires due wakeups - # into this session while it's idle. + # Recurring in-session wakeups; the notification poller fires due ones while the session is idle. if not session: return _err(rid, 4001, "no active session") try: from hermes_cli.loops import LoopManager, dispatch_loop_command except Exception as exc: return _err(rid, 5030, f"loops unavailable: {exc}") - - sid_key = session.get("session_key") or "" - if not sid_key: - return _err(rid, 4001, "no session key") - + sid_key, err = _session_key_or_err(rid, session) + if err: + return err result = dispatch_loop_command(LoopManager(session_id=sid_key), arg) output = result.get("output") or "" if result.get("created"): - try: + with contextlib.suppress(Exception): from hermes_cli.loops import goal_blocks_loop_tick if goal_blocks_loop_tick(sid_key): @@ -880,14 +841,11 @@ def _cmd_loop(rid, params, session, name, arg): "\nNote: an active /goal is driving this session — loop " "wakeups defer until the goal finishes, pauses, or parks." ) - except Exception: - pass return _ok(rid, {"type": "exec", "output": output}) def _cmd_undo(rid, params, session, name, arg): - # /undo [N]: back up N user turns (default 1), soft-delete the truncated rows - # on disk, and prefill the composer with the backed-up user text. + # /undo [N]: back up N user turns, soft-delete truncated rows on disk, prefill the composer. if not session: return _err(rid, 4001, "no active session to undo") if busy := _busy_error(rid, session, "undo"): @@ -902,8 +860,7 @@ def _cmd_undo(rid, params, session, name, arg): n = int(arg_str.split()[0]) except (ValueError, IndexError): return _err(rid, 4004, f"undo: invalid count {arg_str!r} — use /undo or /undo N") - if n < 1: - n = 1 + n = max(n, 1) from agent.message_content import flatten_message_text with session["history_lock"]: @@ -920,30 +877,21 @@ def _cmd_undo(rid, params, session, name, arg): except Exception as exc: return _err(rid, 5008, f"undo: {exc}") target_text = flatten_message_text(live_view.get("content")) - # Notify memory providers (same hook /branch fires) with rewound=True so - # providers caching per-turn document state invalidate. + # Notify memory providers (same hook /branch fires) with rewound=True so cached per-turn state invalidates. agent = session.get("agent") if agent is not None: mm = getattr(agent, "_memory_manager", None) if mm is not None: - try: + with contextlib.suppress(Exception): mm.on_session_switch(session_key, parent_session_id="", reset=False, rewound=True) - except Exception: - pass if hasattr(agent, "_invalidate_system_prompt"): - try: + with contextlib.suppress(Exception): agent._invalidate_system_prompt() - except Exception: - pass if hasattr(agent, "_last_flushed_db_idx"): - try: + with contextlib.suppress(Exception): agent._last_flushed_db_idx = len(active) - except Exception: - pass turn_word = "turn" if turns_undone == 1 else "turns" - notice = ( - f"↶ Undid {turns_undone} {turn_word} ({rewound_count} message(s)). Edit and resubmit, or send a new message." - ) + notice = f"↶ Undid {turns_undone} {turn_word} ({rewound_count} message(s)). Edit and resubmit, or send a new message." return _ok(rid, {"type": "prefill", "message": target_text, "notice": notice}) @@ -951,17 +899,11 @@ def _cmd_snapshot(rid, params, session, name, arg): subcommand = arg.split(maxsplit=1)[0].lower() if arg else "" if subcommand not in {"restore", "rewind"}: return None - return _ok( - rid, - { - "type": "exec", - "output": ( - "/snapshot restore is blocked in the TUI because it changes " - "config/state on disk while the live agent has cached settings. " - "Run it in the classic CLI, then restart the TUI." - ), - }, + output = ( + "/snapshot restore is blocked in the TUI because it changes config/state on disk " + "while the live agent has cached settings. Run it in the classic CLI, then restart the TUI." ) + return _ok(rid, {"type": "exec", "output": output}) def _cmd_compress(rid, params, session, name, arg): @@ -973,90 +915,18 @@ def _cmd_compress(rid, params, session, name, arg): sid = params.get("session_id", "") if _session_uses_compute_host(session): - command = f"/{name}" + (f" {arg}" if arg else "") - _late_session = session - - def _on_late_ack(late: dict, _sid=sid) -> None: - _adopt_late_compute_host_compress_ack(_sid, _late_session, late, route_name="slash.compress") - - try: - ack = _send_compute_host_control( - sid, - route_name="slash.compress", - command=command, - wait=True, - timeout=_compute_host_compress_wait_seconds(), - on_late_ack=_on_late_ack, - ) - except queue.Empty: - return _ok( - rid, - { - "type": "exec", - "status": "pending", - "output": "compression still running in the background; the transcript will refresh when it finishes", - }, - ) - except Exception as exc: - return _err(rid, 5019, f"compute-host slash.compress failed: {exc}") - if ack.get("type") in {"control.error", "error"}: - return _err(rid, 4009, str(ack.get("message") or "compute-host slash.compress failed")) - _apply_compute_host_metadata_mirror(session, ack) - return _ok(rid, {"type": "exec", "output": str(ack.get("output") or "")}) + status, text = _compute_host_slash(sid, session, "compress", f"/{name}" + (f" {arg}" if arg else "")) + if status in {"failed", "rejected"}: + return _err(rid, 5019 if status == "failed" else 4009, text) + payload = {"type": "exec", "status": "pending", "output": text} if status == "pending" else {"type": "exec", "output": text} + return _ok(rid, payload) try: - from agent.manual_compression_feedback import summarize_manual_compression - from agent.model_metadata import estimate_request_tokens_rough - - with session["history_lock"]: - before_messages = list(session.get("history", [])) - history_version = int(session.get("history_version", 0)) - _agent = session["agent"] - _sys_prompt = getattr(_agent, "_cached_system_prompt", "") or "" - _tools = getattr(_agent, "tools", None) or None - before_tokens = ( - estimate_request_tokens_rough(before_messages, system_prompt=_sys_prompt, tools=_tools) - if before_messages - else 0 - ) - removed, usage = _compress_session_history( - session, - arg.strip() or None, - approx_tokens=before_tokens, - before_messages=before_messages, - history_version=history_version, - ) - with session["history_lock"]: - after_messages = list(session.get("history", [])) - after_tokens = ( - estimate_request_tokens_rough( - after_messages, - system_prompt=getattr(_agent, "_cached_system_prompt", "") or _sys_prompt, - tools=getattr(_agent, "tools", None) or _tools, - ) - if after_messages - else 0 - ) - _sync_session_key_after_compress(sid, session) - summary = summarize_manual_compression( - before_messages, - after_messages, - before_tokens, - after_tokens, - compression_state=getattr(_agent, "context_compressor", None), - ) - _emit("session.info", sid, _session_info(session.get("agent"), session)) - finalize_context_engine_compression_notification(_agent, committed=True) - return _ok( - rid, - { - "type": "exec", - "output": "\n".join(filter(None, [summary["headline"], summary["token_line"], summary.get("note")])), - }, - ) + summary = _compress_live_with_feedback(sid, session, session["agent"], arg, snapshot_kwargs=True) + output = "\n".join(filter(None, [summary["headline"], summary["token_line"], summary.get("note")])) + return _ok(rid, {"type": "exec", "output": output}) except CompressionLockHeld as e: - # Lock-skip is a clean no-op (matches the slash mirror and session.compress - # RPC), never a "compress failed" error. _compress_session_history already - # discarded the deferred context-engine notification before raising. + # Clean no-op (parity with the slash mirror / session.compress), never "compress failed"; + # _compress_session_history already discarded the deferred context-engine notification. from agent.manual_compression_feedback import describe_compression_lock_skip return _ok(rid, {"type": "exec", "output": describe_compression_lock_skip(e.holder)}) @@ -1065,26 +935,13 @@ def _cmd_compress(rid, params, session, name, arg): return _err(rid, 5009, f"compress failed: {exc}") -def _slash_builtin_table() -> dict: - """name → handler. Built per call so the entries resolve to the rebound helpers.""" - return { - "queue": _cmd_queue, - "q": _cmd_queue, - "learn": _cmd_learn, - "plan": _cmd_plan, - "init": _cmd_init, - "moa": _cmd_moa, - "focus": _cmd_focus, - "retry": _cmd_retry, - "steer": _cmd_steer, - "goal": _cmd_goal, - "loop": _cmd_loop, - "undo": _cmd_undo, - "snapshot": _cmd_snapshot, - "snap": _cmd_snapshot, - "compress": _cmd_compress, - "compact": _cmd_compress, - } +# name → built-in handler (values are rebound onto server globals by bind_module). +_SLASH_BUILTINS = { + "queue": _cmd_queue, "q": _cmd_queue, "learn": _cmd_learn, "plan": _cmd_plan, "init": _cmd_init, + "moa": _cmd_moa, "focus": _cmd_focus, "retry": _cmd_retry, "steer": _cmd_steer, "goal": _cmd_goal, + "loop": _cmd_loop, "undo": _cmd_undo, "snapshot": _cmd_snapshot, "snap": _cmd_snapshot, + "compress": _cmd_compress, "compact": _cmd_compress, +} @method("command.dispatch") @@ -1098,7 +955,7 @@ def _(rid, params: dict) -> dict: res = stage(rid, params, session, name, arg) if res is not None: return res - builtin = _slash_builtin_table().get(name) + builtin = _SLASH_BUILTINS.get(name) if builtin is not None: res = builtin(rid, params, session, name, arg) if res is not None: @@ -1116,9 +973,8 @@ def _(rid, params: dict) -> dict: if not cmd: return _err(rid, 4004, "empty command") - # Skill/bundle and _pending_input commands must NOT reach the slash worker - # (see _PENDING_INPUT_COMMANDS). Plugin commands also bypass the worker but - # still return normal slash.exec output so the TUI keeps the pager path. + # Skill/bundle and _PENDING_INPUT_COMMANDS must NOT reach the slash worker. Plugin + # commands also bypass it but return normal slash.exec output (TUI keeps the pager path). _cmd_text = cmd.lstrip("/") if cmd.startswith("/") else cmd _cmd_parts = _cmd_text.split(maxsplit=1) _cmd_base = (_cmd_parts[0] if _cmd_parts else "").lower() @@ -1130,61 +986,26 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"output": live_output or "(no output)"}) if _cmd_base in _PENDING_INPUT_COMMANDS: - # Route straight to command.dispatch rather than erroring and relying on a - # client-side retry (some clients fail the fallback → "empty command"). + # Route straight to command.dispatch: some clients fail the error-then-retry fallback ("empty command"). return _methods["command.dispatch"](rid, {"name": _cmd_base, "arg": _cmd_arg, "session_id": sid}) if _cmd_base in _WORKER_BLOCKED_COMMANDS: subcommand = _cmd_arg.split(maxsplit=1)[0].lower() if _cmd_arg else "" if subcommand in {"restore", "rewind"}: - return _err( - rid, 4018, "snapshot restore mutates live config/state; use command.dispatch for /snapshot restore" - ) + return _err(rid, 4018, "snapshot restore mutates live config/state; use command.dispatch for /snapshot restore") - try: - from agent.skill_bundles import resolve_bundle_command_key - from hermes_cli.commands import resolve_command + _bundle_key = _bundle_key_for(_cmd_base) + if _bundle_key is not None: + return _methods["command.dispatch"](rid, {"name": _bundle_key.lstrip("/"), "arg": _cmd_arg, "session_id": sid}) - _bundle_key = resolve_bundle_command_key(_cmd_base) if resolve_command(_cmd_base) is None else None - if _bundle_key is not None: - return _methods["command.dispatch"]( - rid, {"name": _bundle_key.lstrip("/"), "arg": _cmd_arg, "session_id": sid} - ) - except Exception: - pass + if _is_profile_skill_command(session, _cmd_base): + return _err(rid, 4018, f"skill command: use command.dispatch for /{_cmd_base}") - try: - from agent.skill_commands import get_skill_commands - from hermes_constants import reset_hermes_home_override, set_hermes_home_override - - # Bind HERMES_HOME to the session's profile so get_skill_commands() sees - # that profile's skills.external_dirs: dispatch() runs this on the pool - # with a copied context and nothing upstream binds the override here. - _profile_home = session.get("profile_home") - _home_token = set_hermes_home_override(_profile_home) if _profile_home else None + plugin_handler = _plugin_command_handler(_cmd_base) if _cmd_base else None + if plugin_handler: try: - _cmd_key = f"/{_cmd_base}" - if _cmd_key in get_skill_commands(): - return _err(rid, 4018, f"skill command: use command.dispatch for {_cmd_key}") - finally: - if _home_token is not None: - reset_hermes_home_override(_home_token) - except Exception: - pass + from hermes_cli.plugins import resolve_plugin_command_result - plugin_handler = None - resolve_plugin_command_result = None - if _cmd_base: - try: - from hermes_cli.plugins import get_plugin_command_handler, resolve_plugin_command_result - - plugin_handler = get_plugin_command_handler(_cmd_base) - except Exception: - plugin_handler = None - resolve_plugin_command_result = None - - if plugin_handler and resolve_plugin_command_result: - try: result = resolve_plugin_command_result(plugin_handler(_cmd_arg)) return _ok(rid, {"output": str(result or "(no output)")}) except Exception as e: @@ -1192,10 +1013,9 @@ def _(rid, params: dict) -> dict: worker = session.get("slash_worker") if not worker: - # On-demand spawn is the ONLY spawn path, and slash.exec runs on the RPC - # pool: two concurrent commands could both see slash_worker=None and each - # fork a full MCP-fleet worker (the _attach_worker loser leaks). Serialize - # first-use spawn per session. + # slash.exec runs on the RPC pool: two concurrent commands could both see + # slash_worker=None and each fork a full MCP-fleet worker (the _attach_worker + # loser leaks). Serialize first-use spawn per session. with _sessions_lock: spawn_lock = session.setdefault("_slash_spawn_lock", threading.Lock()) with spawn_lock: @@ -1219,10 +1039,8 @@ def _(rid, params: dict) -> dict: payload["warning"] = warning return _ok(rid, payload) except Exception as e: - try: + with contextlib.suppress(Exception): worker.close() - except Exception: - pass session["slash_worker"] = None return _err(rid, 5030, str(e)) @@ -1254,20 +1072,11 @@ def _(rid, params: dict) -> dict: def go(mgr, cwd): if not mgr.enabled: return _ok(rid, {"enabled": False, "checkpoints": []}) - return _ok( - rid, - { - "enabled": True, - "checkpoints": [ - { - "hash": c.get("hash", ""), - "timestamp": c.get("timestamp", ""), - "message": c.get("message", ""), - } - for c in mgr.list_checkpoints(cwd) - ], - }, - ) + rows = [ + {"hash": c.get("hash", ""), "timestamp": c.get("timestamp", ""), "message": c.get("message", "")} + for c in mgr.list_checkpoints(cwd) + ] + return _ok(rid, {"enabled": True, "checkpoints": rows}) return _with_checkpoints(session, go) except Exception as e: @@ -1283,9 +1092,8 @@ def _(rid, params: dict) -> dict: file_path = params.get("file_path", "") if not target: return _err(rid, 4014, "hash required") - # A full-history rollback mutates session history, so it is rejected during - # an in-flight turn (prompt.submit would drop the agent's output or clobber - # the rollback). A file-scoped rollback only touches disk and is allowed. + # Full-history rollback mutates session history → rejected mid-turn (prompt.submit + # would drop the agent's output or clobber it). File-scoped only touches disk. if not file_path and session.get("running"): return _err(rid, 4009, "session busy — /interrupt the current turn before full rollback.restore") try: @@ -1299,13 +1107,9 @@ def _(rid, params: dict) -> dict: _history, user_indices = _user_turn_indices(session) if user_indices: try: - _active, _live_view, removed = _rewind_active_session_history( - session, len(user_indices) - 1 - ) + _active, _live_view, removed = _rewind_active_session_history(session, len(user_indices) - 1) except Exception as exc: - raise RuntimeError( - f"checkpoint restored, but session history rewind failed: {exc}" - ) from exc + raise RuntimeError(f"checkpoint restored, but session history rewind failed: {exc}") from exc result["history_removed"] = removed return result @@ -1342,107 +1146,79 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"connected": bool(url), "url": url}) if action == "disconnect": return _browser_disconnect(rid) - if action != "connect": - return _err(rid, 4015, f"unknown action: {action}") - return _browser_connect(rid, params) + if action == "connect": + return _browser_connect(rid, params) + return _err(rid, 4015, f"unknown action: {action}") @method("plugins.list") +@_guarded(5032) def _(rid, params: dict) -> dict: - try: - from hermes_cli.plugins import get_plugin_manager + from hermes_cli.plugins import get_plugin_manager - return _ok( - rid, - { - "plugins": [ - {"name": n, "version": getattr(i, "version", "?"), "enabled": getattr(i, "enabled", True)} - for n, i in get_plugin_manager()._plugins.items() - ] - }, - ) - except Exception as e: - return _err(rid, 5032, str(e)) + rows = [ + {"name": n, "version": getattr(i, "version", "?"), "enabled": getattr(i, "enabled", True)} + for n, i in get_plugin_manager()._plugins.items() + ] + return _ok(rid, {"plugins": rows}) @method("config.show") +@_guarded(5030) def _(rid, params: dict) -> dict: - try: - cfg = _load_cfg() - model = _resolve_model() - from agent.secret_scope import get_secret + cfg = _load_cfg() + model = _resolve_model() + from agent.secret_scope import get_secret - api_key = get_secret("HERMES_API_KEY", "") or cfg.get("api_key", "") - masked = f"****{api_key[-4:]}" if len(api_key) > 4 else "(not set)" - base_url = os.environ.get("HERMES_BASE_URL", "") or cfg.get("base_url", "") - - sections = [ - {"title": "Model", "rows": [["Model", model], ["Base URL", base_url or "(default)"], ["API Key", masked]]}, - { - "title": "Agent", - "rows": [ - ["Max Turns", str(_cfg_max_turns(cfg, 500))], - ["Toolsets", ", ".join(cfg.get("enabled_toolsets", [])) or "all"], - ["Verbose", str(cfg.get("verbose", False))], - ], - }, - { - "title": "Environment", - "rows": [["Working Dir", os.getcwd()], ["Config File", str(_hermes_home / "config.yaml")]], - }, - ] - return _ok(rid, {"sections": sections}) - except Exception as e: - return _err(rid, 5030, str(e)) + api_key = get_secret("HERMES_API_KEY", "") or cfg.get("api_key", "") + masked = f"****{api_key[-4:]}" if len(api_key) > 4 else "(not set)" + base_url = os.environ.get("HERMES_BASE_URL", "") or cfg.get("base_url", "") + agent_rows = [ + ["Max Turns", str(_cfg_max_turns(cfg, 500))], + ["Toolsets", ", ".join(cfg.get("enabled_toolsets", [])) or "all"], + ["Verbose", str(cfg.get("verbose", False))], + ] + sections = [ + {"title": "Model", "rows": [["Model", model], ["Base URL", base_url or "(default)"], ["API Key", masked]]}, + {"title": "Agent", "rows": agent_rows}, + {"title": "Environment", "rows": [["Working Dir", os.getcwd()], ["Config File", str(_hermes_home / "config.yaml")]]}, + ] + return _ok(rid, {"sections": sections}) # ─── Tools / toolsets / agents ─────────────────────────────────────────────── @method("tools.list") +@_guarded(5031) def _(rid, params: dict) -> dict: - try: - return _ok(rid, {"toolsets": _toolset_rows(params, with_tools=True)}) - except Exception as e: - return _err(rid, 5031, str(e)) + return _ok(rid, {"toolsets": _toolset_rows(params, with_tools=True)}) @method("toolsets.list") +@_guarded(5032) def _(rid, params: dict) -> dict: - try: - return _ok(rid, {"toolsets": _toolset_rows(params, with_tools=False)}) - except Exception as e: - return _err(rid, 5032, str(e)) + return _ok(rid, {"toolsets": _toolset_rows(params, with_tools=False)}) @method("tools.show") +@_guarded(5034) def _(rid, params: dict) -> dict: - try: - from model_tools import get_toolset_for_tool, get_tool_definitions + from model_tools import get_toolset_for_tool, get_tool_definitions - session = _sessions.get(params.get("session_id", "")) - enabled = getattr(session["agent"], "enabled_toolsets", None) if session else _load_enabled_toolsets() - # Pre-assembly list: /tools is a discovery surface and must show tools - # deferred behind the tool_search bridge (same as the CLI). - tools = get_tool_definitions(enabled_toolsets=enabled, quiet_mode=True, skip_tool_search_assembly=True) - sections = {} - - for tool in sorted(tools, key=lambda t: t["function"]["name"]): - name = tool["function"]["name"] - desc = str(tool["function"].get("description", "") or "").split("\n")[0] - if ". " in desc: - desc = desc[: desc.index(". ") + 1] - sections.setdefault(get_toolset_for_tool(name) or "unknown", []).append({"name": name, "description": desc}) - - return _ok( - rid, - { - "sections": [{"name": name, "tools": rows} for name, rows in sorted(sections.items())], - "total": len(tools), - }, - ) - except Exception as e: - return _err(rid, 5034, str(e)) + session = _sessions.get(params.get("session_id", "")) + enabled = getattr(session["agent"], "enabled_toolsets", None) if session else _load_enabled_toolsets() + # Pre-assembly list: /tools must also show tools deferred behind the tool_search bridge (as the CLI). + tools = get_tool_definitions(enabled_toolsets=enabled, quiet_mode=True, skip_tool_search_assembly=True) + sections = {} + for tool in sorted(tools, key=lambda t: t["function"]["name"]): + name = tool["function"]["name"] + desc = str(tool["function"].get("description", "") or "").split("\n")[0] + if ". " in desc: + desc = desc[: desc.index(". ") + 1] + sections.setdefault(get_toolset_for_tool(name) or "unknown", []).append({"name": name, "description": desc}) + sections_out = [{"name": name, "tools": rows} for name, rows in sorted(sections.items())] + return _ok(rid, {"sections": sections_out, "total": len(tools)}) @method("tools.configure") @@ -1503,26 +1279,15 @@ def _(rid, params: dict) -> dict: @method("agents.list") +@_guarded(5033) def _(rid, params: dict) -> dict: - try: - from tools.process_registry import process_registry + from tools.process_registry import process_registry - return _ok( - rid, - { - "processes": [ - { - "session_id": p["session_id"], - "command": p["command"][:80], - "status": p["status"], - "uptime": p["uptime_seconds"], - } - for p in process_registry.list_sessions() - ] - }, - ) - except Exception as e: - return _err(rid, 5033, str(e)) + rows = [ + {"session_id": p["session_id"], "command": p["command"][:80], "status": p["status"], "uptime": p["uptime_seconds"]} + for p in process_registry.list_sessions() + ] + return _ok(rid, {"processes": rows}) # ─── Cron / learning / skills ──────────────────────────────────────────────── @@ -1531,102 +1296,81 @@ def _(rid, params: dict) -> dict: @method("cron.manage") @_profile_scoped_rpc(5023) def _(rid, params: dict) -> dict: - """cronjob() keys off HERMES_HOME, so the optional ``profile`` scope lets a - per-profile cron store be listed/mutated even when that profile runs its own - gateway (mirrors skills.manage / mcp.catalog).""" + """cronjob() keys off HERMES_HOME, so the optional ``profile`` scope reaches a + per-profile cron store even when that profile runs its own gateway.""" from tools.cronjob_tools import cronjob action, jid = params.get("action", "list"), params.get("name", "") if action == "list": - # Paused jobs are excluded by default, which reads as deletion in any UI - # with an enable/disable toggle — forward the flag. + # Paused jobs are excluded by default (reads as deletion in a toggle UI) — forward the flag. result = json.loads( cronjob(action="list", include_disabled=is_truthy_value(params.get("include_disabled", False))) ) - # ``scoped`` proves the gateway honored the profile scope: new clients may - # treat every job as owned by that profile; older gateways omit it and - # keep the safe [bot:] compatibility filter. + # ``scoped`` proves the profile scope was honored: new clients treat every job as that + # profile's; older gateways omit it and clients keep the safe [bot:] filter. profile = str(params.get("profile") or "").strip() if profile: result["scoped"] = profile return _ok(rid, result) if action == "add": - return _ok( - rid, - json.loads( - cronjob( - action="create", - name=jid, - schedule=params.get("schedule", ""), - prompt=params.get("prompt", ""), - # Optional repeat cap; None keeps the schedule-kind default. - repeat=int(params["repeat"]) if str(params.get("repeat", "")).strip().isdigit() else None, - # Optional continuity toggle: previous output injected into each run. - continuity=( - is_truthy_value(params.get("continuity")) if params.get("continuity") is not None else None - ), - # Optional delivery target, e.g. 'bot-chat[:name]'; empty keeps the cronjob() default. - deliver=(str(params.get("deliver") or "").strip() or None), - ) - ), + # Optional repeat / continuity / deliver ('bot-chat[:name]'): None keeps each cronjob() default. + raw = cronjob( + action="create", + name=jid, + schedule=params.get("schedule", ""), + prompt=params.get("prompt", ""), + repeat=int(params["repeat"]) if str(params.get("repeat", "")).strip().isdigit() else None, + continuity=is_truthy_value(params.get("continuity")) if params.get("continuity") is not None else None, + deliver=str(params.get("deliver") or "").strip() or None, ) + return _ok(rid, json.loads(raw)) if action in {"remove", "pause", "resume"}: return _ok(rid, json.loads(cronjob(action=action, job_id=jid))) return _err(rid, 4016, f"unknown cron action: {action}") @method("learning.frames") +@_guarded(5000, "learning.frames failed: ") def _(rid, params: dict) -> dict: - """Pre-render the learning timeline for the TUI ``/journey`` overlay: ``frames`` - (reveal 0→1) plus legend/summary/bucket metadata so Ink walks the tree locally. - Shares its renderer with ``hermes journey``.""" + """Pre-render the ``/journey`` timeline: ``frames`` (reveal 0→1) plus legend/summary/ + bucket metadata so Ink walks the tree locally. Shares its renderer with ``hermes journey``.""" try: cols = int(params.get("cols", 80) or 80) rows = int(params.get("rows", 24) or 24) frames = int(params.get("frames", 48) or 48) except (TypeError, ValueError): cols, rows, frames = 80, 24, 48 - try: - from agent.learning_graph import build_learning_graph - from agent.learning_graph_render import render_frames + from agent.learning_graph import build_learning_graph + from agent.learning_graph_render import render_frames - payload = build_learning_graph() - return _ok(rid, render_frames(payload, cols=max(20, cols), rows=max(10, rows), frames=frames)) - except Exception as exc: # noqa: BLE001 - return _err(rid, 5000, f"learning.frames failed: {exc}") + return _ok(rid, render_frames(build_learning_graph(), cols=max(20, cols), rows=max(10, rows), frames=frames)) @method("learning.detail") +@_guarded(5000, "learning.detail failed: ") def _(rid, params: dict) -> dict: """Current content of a journey node, for an edit prefill.""" - try: - from agent.learning_mutations import node_detail + from agent.learning_mutations import node_detail - return _ok(rid, node_detail(str(params.get("id", "")))) - except Exception as exc: # noqa: BLE001 - return _err(rid, 5000, f"learning.detail failed: {exc}") + return _ok(rid, node_detail(str(params.get("id", "")))) @method("learning.delete") +@_guarded(5000, "learning.delete failed: ") def _(rid, params: dict) -> dict: """Delete a journey node — skills are archived (restorable), memories removed.""" - try: - from agent.learning_mutations import delete_node + from agent.learning_mutations import delete_node - return _ok(rid, delete_node(str(params.get("id", "")))) - except Exception as exc: # noqa: BLE001 - return _err(rid, 5000, f"learning.delete failed: {exc}") + return _ok(rid, delete_node(str(params.get("id", "")))) @method("learning.edit") +@_guarded(5000, "learning.edit failed: ") def _(rid, params: dict) -> dict: """Rewrite a journey node's content (SKILL.md or memory chunk).""" - try: - from agent.learning_mutations import edit_node + from agent.learning_mutations import edit_node - return _ok(rid, edit_node(str(params.get("id", "")), str(params.get("content", "")))) - except Exception as exc: # noqa: BLE001 - return _err(rid, 5000, f"learning.edit failed: {exc}") + return _ok(rid, edit_node(str(params.get("id", "")), str(params.get("content", "")))) def _skills_list(rid, params, query): @@ -1669,8 +1413,7 @@ def _skills_inspect(rid, params, query): @method("skills.manage") @_profile_scoped_rpc(5024) def _(rid, params: dict) -> dict: - """list/install operate on the scoped profile's skills dir; search/browse/ - inspect hit the shared hub catalog (the override is harmless there).""" + """list/install use the scoped profile's skills dir; search/browse/inspect hit the shared hub.""" action, query = params.get("action", "list"), params.get("query", "") handler = { "list": _skills_list, @@ -1685,43 +1428,34 @@ def _(rid, params: dict) -> dict: @method("skills.reload") +@_guarded(5025) def _(rid, params: dict) -> dict: - try: - from agent.skill_commands import reload_skills + from agent.skill_commands import reload_skills - result = reload_skills() - added = result.get("added") or [] - removed = result.get("removed") or [] - total = int(result.get("total") or 0) - - lines = ["Reloading skills..."] - if not added and not removed: - lines.append("No new skills detected.") - if added: - lines.append("Added skills:") - lines.extend(f" - {item.get('name', '')}" for item in added) - if removed: - lines.append("Removed skills:") - lines.extend(f" - {item.get('name', '')}" for item in removed) - lines.append(f"{total} skill(s) available") - return _ok(rid, {"output": "\n".join(lines), "result": result}) - except Exception as e: - return _err(rid, 5025, str(e)) + result = reload_skills() + added = result.get("added") or [] + removed = result.get("removed") or [] + lines = ["Reloading skills..."] + if not added and not removed: + lines.append("No new skills detected.") + for label, items in (("Added skills:", added), ("Removed skills:", removed)): + if items: + lines.append(label) + lines.extend(f" - {item.get('name', '')}" for item in items) + lines.append(f"{int(result.get('total') or 0)} skill(s) available") + return _ok(rid, {"output": "\n".join(lines), "result": result}) # ─── MCP catalog + per-profile server lifecycle (mcp.servers.*) ───────────── -# -# Gateway mirrors of the dashboard REST surface (hermes_cli/web_routers/mcp.py) so -# a desktop plugin can manage MCP servers for ANY profile. Persistence reuses -# hermes_cli/mcp_config.py; summaries come from tui_gateway.mcp_rpc_helpers. +# Gateway mirrors of the dashboard REST surface (hermes_cli/web_routers/mcp.py) so a +# desktop plugin can manage MCP servers for ANY profile. Persistence: hermes_cli/mcp_config.py. @method("mcp.catalog") @_profile_scoped_rpc(5024) def _(rid, params: dict) -> dict: - """Bundled MCP catalog with per-profile install/enable state: ``{servers: - [{name, description, installed, enabled, requires: [env keys], transport}]}`` - — the same menu `hermes mcp` offers, so UIs know which entries need setup.""" + """``{servers: [{name, description, installed, enabled, requires: [env keys], transport}]}`` + — the `hermes mcp` menu with per-profile state, so UIs know which entries need setup.""" from hermes_cli import mcp_catalog out = [] @@ -1730,6 +1464,7 @@ def _(rid, params: dict) -> dict: requires = [str(k) for k in (getattr(entry, "env_keys", None) or [])] except Exception: requires = [] + transport = getattr(entry, "transport", None) # TransportSpec → its kind string out.append( { "name": entry.name, @@ -1737,10 +1472,7 @@ def _(rid, params: dict) -> dict: "installed": bool(mcp_catalog.is_installed(entry.name)), "enabled": bool(mcp_catalog.is_enabled(entry.name)), "requires": requires, - # TransportSpec object — reduce to its kind string. - "transport": str( - getattr(getattr(entry, "transport", None), "kind", "") or getattr(entry, "transport", "") or "stdio" - ), + "transport": str(getattr(transport, "kind", "") or transport or "stdio"), } ) return _ok(rid, {"servers": out}) @@ -1760,8 +1492,7 @@ def _(rid, params: dict) -> dict: @method("mcp.servers.add") @_mcp_server_scoped def _(rid, params: dict) -> dict: - """Add a server to the profile's config.yaml. ``name`` plus EITHER ``preset`` - (catalog id, via ``_apply_mcp_preset``) or ``config`` (url/command/args/env/ + """Add ``name`` with EITHER ``preset`` (catalog id) or ``config`` (url/command/args/env/ headers/auth/tools). ``bearer_token`` goes to the profile's .env; only the ``Authorization`` header template is persisted. Duplicate names → 4090.""" from hermes_cli.mcp_config import _apply_mcp_preset, _get_mcp_servers, _save_bearer_auth_token, _save_mcp_server @@ -1769,13 +1500,10 @@ def _(rid, params: dict) -> dict: name = str(params.get("name") or "").strip() if name in _get_mcp_servers(): return _err(rid, 4090, f"server '{name}' already exists") - preset = str(params.get("preset") or "").strip() raw_cfg = params.get("config") server_config: dict = dict(raw_cfg) if isinstance(raw_cfg, dict) else {} - - if preset: - # Fills url/command/args from the preset when omitted; mutates server_config in place. + if preset: # fills url/command/args when omitted; mutates server_config in place _apply_mcp_preset( name, preset_name=preset, @@ -1787,11 +1515,9 @@ def _(rid, params: dict) -> dict: if not server_config.get("url") and not server_config.get("command"): return _err(rid, 4063, "config must specify a 'url' (http) or 'command' (stdio), or a valid 'preset'") - bearer_token = params.get("bearer_token") if bearer_token: server_config["headers"] = _save_bearer_auth_token(name, str(bearer_token)) - if not _save_mcp_server(name, server_config): return _err(rid, 4001, f"server '{name}' rejected: suspicious command/args configuration") saved = _get_mcp_servers().get(name, server_config) @@ -1801,21 +1527,17 @@ def _(rid, params: dict) -> dict: @method("mcp.servers.set_api_key") @_profile_scoped_rpc(5024, required=(("name", _stripped), ("value", _nonempty)), catch_resolve=False) def _(rid, params: dict) -> dict: - """Store a credential for a server: the secret goes to the profile's .env under - ``env_var`` (default ``MCP__API_KEY``); config.yaml gets a reference — - ``Authorization: Bearer ${ENV}`` header for http, ``env: {VAR: "${ENV}"}`` for - stdio — matching ``cmd_mcp_configure`` / ``_save_bearer_auth_token``.""" + """Secret → profile .env under ``env_var`` (default ``MCP__API_KEY``); config.yaml + gets a reference: ``Authorization: Bearer ${ENV}`` header (http) or ``env: {VAR: "${ENV}"}`` + (stdio), matching ``cmd_mcp_configure`` / ``_save_bearer_auth_token``.""" from hermes_cli.config import load_config, save_config, save_env_value - from hermes_cli.mcp_config import _bearer_auth_headers, _env_key_for_server, _get_mcp_servers, _strip_bearer_prefix + from hermes_cli.mcp_config import _bearer_auth_headers, _env_key_for_server, _strip_bearer_prefix - name = str(params.get("name") or "").strip() + name, servers, err = _mcp_named_server(rid, params) + if err: + return err value = params.get("value") - servers = _get_mcp_servers() - if name not in servers: - return _err(rid, 4064, f"server '{name}' not found") - env_var = str(params.get("env_var") or "").strip() or _env_key_for_server(name) - entry = servers[name] if not isinstance(entry, dict): return _err(rid, 4001, "malformed server config") @@ -1826,10 +1548,9 @@ def _(rid, params: dict) -> dict: return _err(rid, 4063, "value is not a valid credential") save_env_value(env_var, normalized) if env_var == _env_key_for_server(name): - headers = _bearer_auth_headers(name) + entry["headers"] = _bearer_auth_headers(name) else: - headers = {"Authorization": f"Bearer ${{{env_var}}}"} - entry["headers"] = headers + entry["headers"] = {"Authorization": f"Bearer ${{{env_var}}}"} else: save_env_value(env_var, str(value)) env_block = entry.get("env") @@ -1847,58 +1568,40 @@ def _(rid, params: dict) -> dict: @method("mcp.servers.test") @_mcp_server_scoped def _(rid, params: dict) -> dict: - """Connect, list tools, disconnect (``_probe_single_server``). Success: - ``{ok, tools, prompts, resources, oauth_needed, oauth_tokens_present}``; - failure: ``{ok: false, error, tools: [], oauth_needed}``. Runs on the RPC - pool (_LONG_HANDLERS): a cold stdio `npx` spawn can block for seconds.""" - from hermes_cli.mcp_config import _get_mcp_servers, _oauth_tokens_present, _probe_single_server - - name = str(params.get("name") or "").strip() - servers = _get_mcp_servers() - if name not in servers: - return _err(rid, 4064, f"server '{name}' not found") + """Connect, list tools, disconnect. Success: ``{ok, tools, prompts, resources, oauth_needed, + oauth_tokens_present}``; failure: ``{ok: false, error, tools: [], oauth_needed, ...}``. + Runs on the RPC pool (_LONG_HANDLERS): a cold stdio `npx` spawn can block for seconds.""" + from hermes_cli.mcp_config import _oauth_tokens_present, _probe_single_server + name, servers, err = _mcp_named_server(rid, params) + if err: + return err cfg = servers[name] - # An `auth: oauth` server that serves tools/list anonymously would probe OK - # with no token — a false green. Require a token on disk for it. + # An `auth: oauth` server serving tools/list anonymously would probe OK with no + # token — a false green. Require a token on disk for it. needs_oauth_token = cfg.get("auth") == "oauth" details: dict = {} + + def failure(error: str, oauth_needed: bool, tokens_present) -> dict: + payload = {"ok": False, "error": error, "tools": [], "oauth_needed": oauth_needed} + return _ok(rid, {**payload, "oauth_tokens_present": tokens_present}) + try: tools = _probe_single_server(name, cfg, details=details) token_present = _oauth_tokens_present(name) if needs_oauth_token else True except Exception as exc: - return _ok( - rid, - { - "ok": False, - "error": str(exc), - "tools": [], - "oauth_needed": needs_oauth_token, - "oauth_tokens_present": _oauth_tokens_present(name) if needs_oauth_token else None, - }, - ) + return failure(str(exc), needs_oauth_token, _oauth_tokens_present(name) if needs_oauth_token else None) if not token_present: - return _ok( - rid, - { - "ok": False, - "error": "OAuth authentication required — no token found.", - "tools": [], - "oauth_needed": True, - "oauth_tokens_present": False, - }, - ) - return _ok( - rid, - { - "ok": True, - "tools": [{"name": t, "description": d} for t, d in tools], - "prompts": details.get("prompts", 0), - "resources": details.get("resources", 0), - "oauth_needed": needs_oauth_token, - "oauth_tokens_present": True if needs_oauth_token else None, - }, - ) + return failure("OAuth authentication required — no token found.", True, False) + payload = { + "ok": True, + "tools": [{"name": t, "description": d} for t, d in tools], + "prompts": details.get("prompts", 0), + "resources": details.get("resources", 0), + "oauth_needed": needs_oauth_token, + "oauth_tokens_present": True if needs_oauth_token else None, + } + return _ok(rid, payload) @method("mcp.servers.remove") @@ -1918,24 +1621,19 @@ def _(rid, params: dict) -> dict: def _(rid, params: dict) -> dict: """Begin a session-backed OAuth flow → ``{ok, session_id, auth_url, flow: "pkce"}``. - The client opens ``auth_url`` in the native browser and polls - ``mcp.servers.oauth.poll`` until ``status == "approved"``. A background - worker drives the same interactive machinery as ``hermes mcp login`` - (``_probe_single_server`` under ``force_interactive_oauth``) with a loopback - listener for the redirect. ``client_redirect_uri`` (remote backends): the - CLIENT hosts the loopback and relays the code via - ``mcp.servers.oauth.callback`` — the only flow that works when desktop and + The client opens ``auth_url`` and polls ``mcp.servers.oauth.poll`` until ``approved``. + A background worker drives the ``hermes mcp login`` machinery with a loopback + listener. With ``client_redirect_uri`` the CLIENT hosts the loopback and relays the + code via ``mcp.servers.oauth.callback`` — the only flow that works when desktop and gateway are on different machines. Runs on the RPC pool (_LONG_HANDLERS).""" - name = str(params.get("name") or "").strip() client_redirect_uri = str(params.get("client_redirect_uri") or "").strip() or None try: - from hermes_cli.mcp_config import _get_mcp_servers from hermes_constants import get_hermes_home from tui_gateway import mcp_oauth_sessions - servers = _get_mcp_servers() - if name not in servers: - return _err(rid, 4064, f"server '{name}' not found") + name, servers, err = _mcp_named_server(rid, params) + if err: + return err cfg = dict(servers[name]) if not cfg.get("url"): return _err(rid, 4001, "stdio servers authenticate via env keys, not OAuth") @@ -1947,17 +1645,14 @@ def _(rid, params: dict) -> dict: result = mcp_oauth_sessions.start_flow(hermes_home, name, cfg, client_redirect_uri=client_redirect_uri) except ValueError as e: return _err(rid, 4001, str(e)) - return _ok( - rid, {"ok": True, "session_id": result["session_id"], "auth_url": result["auth_url"], "flow": result["flow"]} - ) + return _ok(rid, {"ok": True, "session_id": result["session_id"], "auth_url": result["auth_url"], "flow": result["flow"]}) @method("mcp.servers.oauth.poll") @_profile_scoped_rpc(5024, required=_NAME_SESSION, catch_resolve=False) def _(rid, params: dict) -> dict: - """Poll a flow → ``{ok, status: pending|approved|error, error_message?, auth_url?, - tools?}``. On ``approved`` the tokens are persisted for that server/profile; - the profile scope applies here too so a same-profile token read resolves.""" + """Poll a flow → ``{ok, status: pending|approved|error, error_message?, auth_url?, tools?}``. + On ``approved`` tokens persist for that server/profile (profile scope applies here too).""" from tui_gateway import mcp_oauth_sessions name = str(params.get("name") or "").strip() @@ -1969,9 +1664,8 @@ def _(rid, params: dict) -> dict: @method("mcp.servers.oauth.callback") @_profile_scoped_rpc(5024, required=_NAME_SESSION, catch_resolve=False) def _(rid, params: dict) -> dict: - """Relay a client-captured OAuth redirect (``code``/``state``/``error``) into a - running flow started with ``client_redirect_uri``. ``{ok: true}`` once - accepted (state verified in the flow bridge), else ``{ok: false, error_message}``.""" + """Relay a client-captured redirect (``code``/``state``/``error``) into a flow started with + ``client_redirect_uri``. ``{ok: true}`` once accepted (state verified), else ``{ok: false, error_message}``.""" from tui_gateway import mcp_oauth_sessions name = str(params.get("name") or "").strip() @@ -2004,22 +1698,19 @@ def _plugin_rows() -> list[dict]: out = [] for name, version, desc, source, _dir, key in sorted(_discover_all_plugins()): status = _plugin_status(name, enabled, disabled, key=key) - # Bundled backends/platforms/providers run without an explicit enable; - # report the truthful default instead of "not enabled" (reads as OFF). + # Bundled backends/platforms/providers run without an explicit enable: report the + # truthful default instead of "not enabled" (reads as OFF). if status == "not enabled" and source == "bundled" and _bundled_default_on(_dir): status = "enabled" out.append( { "name": name, - # Canonical registry key (``image_gen/fal``): names collide across - # category dirs, so toggles must address the key. - "key": key, + "key": key, # canonical registry key (``image_gen/fal``): names collide across category dirs "version": str(version or ""), "description": desc or "", "source": source, "status": status, - # Agent Plugins v1 package (plugin.json) vs a native Hermes plugin. - "portable": _is_portable_plugin_dir(_dir), + "portable": _is_portable_plugin_dir(_dir), # Agent Plugins v1 package vs native Hermes plugin } ) return out @@ -2060,14 +1751,11 @@ def _plugins_install(rid, params): @method("plugins.manage") @_profile_scoped_rpc(5026, catch_resolve=False) def _(rid, params: dict) -> dict: - """TUI Plugins Hub backend, sharing discovery + enable/disable primitives with - ``hermes plugins`` and the dashboard. - - ``list`` → {plugins: [{name, key, version, description, source, status, - portable}], user_count, bundled_count} + """TUI Plugins Hub backend (shares primitives with ``hermes plugins`` / the dashboard). + - ``list`` → {plugins: [{name, key, version, description, source, status, portable}], user_count, bundled_count} - ``toggle`` → flip ``key`` (or ``name``) per ``enable``; returns the row + {ok, unchanged} - - ``install`` → git-clone ``identifier``/``repo`` into ~/.hermes/plugins/ - (``force``, ``enable`` default True); returns the dashboard dict. - Optional ``profile`` scopes to that profile's HERMES_HOME (mcp.servers.* contract).""" + - ``install`` → git-clone ``identifier``/``repo`` into ~/.hermes/plugins/ (``force``, ``enable`` default True) + Optional ``profile`` scopes HERMES_HOME (mcp.servers.* contract).""" action = params.get("action", "list") handler = {"list": _plugins_list, "toggle": _plugins_toggle, "install": _plugins_install}.get(action) if handler is None: @@ -2092,20 +1780,7 @@ def _(rid, params: dict) -> dict: except ImportError: return _err(rid, 5001, "shell.exec unavailable: approval safety module not importable") try: - from hermes_cli._subprocess_compat import windows_hide_flags - - r = subprocess.run( - cmd, - shell=True, - capture_output=True, - text=True, - timeout=30, - cwd=os.getcwd(), - encoding="utf-8", - errors="replace", # lossy decode: see cli.exec - stdin=subprocess.DEVNULL, - creationflags=windows_hide_flags(), - ) + r = subprocess.run(cmd, shell=True, cwd=os.getcwd(), **_capture_run_kwargs(30)) return _ok(rid, {"stdout": r.stdout[-4000:], "stderr": r.stderr[-2000:], "code": r.returncode}) except subprocess.TimeoutExpired: return _err(rid, 5002, "command timed out (30s)") diff --git a/tui_gateway/methods_voice.py b/tui_gateway/methods_voice.py index ccbbd64554..ef5d28ead6 100644 --- a/tui_gateway/methods_voice.py +++ b/tui_gateway/methods_voice.py @@ -14,20 +14,20 @@ _registry = HandlerRegistry() method = _registry.method -# ── Methods: voice ─────────────────────────────────────────────────── - +# ── Voice state ────────────────────────────────────────────────────────── _voice_sid_lock = threading.Lock() _voice_event_sid: str = "" _voice_wake_owner: "Optional[Transport]" = None +def _caller_transport(): + return current_transport() or _stdio_transport + + def _voice_emit(event: str, payload: dict | None = None) -> None: - """Emit a voice event toward the session that most recently turned the - mode on. Voice is process-global (one microphone), so there's only ever - one sid to target; the TUI handler treats an empty sid as "active - session". Kept separate from _emit to make the lack of per-call sid - argument explicit.""" + """Emit toward the session that most recently turned voice on (one mic → one + target sid; the TUI treats an empty sid as "active session").""" with _voice_sid_lock: sid = _voice_event_sid _emit(event, sid, payload) @@ -42,31 +42,39 @@ def _resume_voice_wake() -> None: def _voice_mode_enabled() -> bool: - """Current voice-mode flag (runtime-only, CLI parity). - - cli.py initialises ``_voice_mode = False`` at startup and only flips - it via ``/voice on``; it never reads a persisted enable bit from - config.yaml. We match that: no config lookup, env var only. This - avoids the TUI auto-starting in REC the next time the user opens it - just because they happened to enable voice in a prior session. - """ + """Runtime-only flag (CLI parity): env var only, never config.yaml, so the TUI + can't auto-start in REC because voice was on in a prior session.""" return os.environ.get("HERMES_VOICE", "").strip() == "1" def _voice_tts_enabled() -> bool: - """Whether agent replies should be spoken back via TTS (runtime only).""" + """Whether agent replies are spoken back via TTS (runtime only).""" return os.environ.get("HERMES_VOICE_TTS", "").strip() == "1" -def _tts_lease_async(lease: str, active: bool) -> None: - """Acquire/release a TTS engine lease off the RPC thread. +def _end_voice_chat(*, stop_loop: bool, stop_tts: bool) -> None: + """Flip voice + TTS mode off; optionally halt the continuous loop / cut live TTS. + Every step is best-effort.""" + os.environ["HERMES_VOICE"] = "0" + os.environ["HERMES_VOICE_TTS"] = "0" + if stop_loop: + try: + from hermes_cli.voice import stop_continuous - Speech-output toggles are the signal that TTS is about to be needed (or - no longer is). Acquiring warms the configured provider — for local - engines that is a model load, possibly a voice download — so it must not - block the toggle's RPC reply. Release is cheap but rides the same thread - for symmetry. Best-effort: a failure here never affects the toggle. - """ + stop_continuous() + except Exception: + pass + if stop_tts: + try: + _tts_stream_stop(user_barge=False) + except Exception: + pass + + +def _tts_lease_async(lease: str, active: bool) -> None: + """Acquire/release a TTS engine lease off the RPC thread: acquiring warms the + provider (local engines load a model, maybe download a voice) and must not + block the toggle's reply. Best-effort — failure never affects the toggle.""" def _run(): try: @@ -83,13 +91,8 @@ def _tts_lease_async(lease: str, active: bool) -> None: def _any_session_running() -> bool: - """True while any session's agent turn is in flight. - - Registered as the voice busy-probe (``hermes_cli.voice.set_voice_busy_probe``) - so silent capture cycles during a long agent turn don't count toward the - no-speech limit — the user is correctly quiet while the agent works. - Voice is process-global (one microphone), so any running session holds. - """ + """Voice busy-probe (``hermes_cli.voice.set_voice_busy_probe``): silent capture + cycles during a long agent turn must not count toward the no-speech limit.""" try: with _sessions_lock: return any(s.get("running") for s in _sessions.values()) @@ -98,11 +101,8 @@ def _any_session_running() -> bool: # ── Streaming TTS (one active pipeline per process — one speaker) ────────── -# Token deltas from the running turn feed a sentence-buffering consumer -# (tools.tts_tool.stream_tts_to_speaker) so speech starts on the first -# sentence instead of after the full reply. Voice is process-global, so a -# single slot suffices; starting a new turn's pipeline barges in on the -# previous one. +# Token deltas feed a sentence-buffering consumer (tools.tts_tool.stream_tts_to_speaker) +# so speech starts on the first sentence; a new turn's pipeline barges in on the previous. _tts_stream_lock = threading.Lock() _tts_stream_state: Optional[dict] = None @@ -139,12 +139,8 @@ def _tts_stream_begin() -> Optional[queue.Queue]: def _tts_stream_stop(user_barge: bool = True) -> None: - """Cut any in-flight streaming TTS (new turn, interrupt, /voice off). - - *user_barge* latches the interruption for the next turn's model note - (``mark_speech_interrupted``) — pass ``False`` for mode changes like - ``/voice off`` where the user isn't talking over the reply. - """ + """Cut any in-flight streaming TTS. *user_barge* latches the interruption for + the next turn's model note — pass ``False`` for mode changes (/voice off).""" global _tts_stream_state with _tts_stream_lock: state, _tts_stream_state = _tts_stream_state, None @@ -170,17 +166,14 @@ def _tts_stream_stop(user_barge: bool = True) -> None: # ── Full-duplex agent-turn listener (one mic, whole turn) ────────────────── -# Replaces the per-playback barge monitors: those only opened the mic once -# TTS playback started (deaf during LLM generation) and calibrated the VAD -# floor against active speaker bleed (deaf during playback too, in practice). -# This listener arms at utterance-submit, spans generation AND playback, and -# disarms when no session is running, no TTS is pending, and no audio flows. +# Arms at utterance-submit, spans generation AND playback (per-playback barge +# monitors were deaf during generation and mis-calibrated against speaker bleed), +# disarms when no session runs, no TTS is pending, and no audio flows. _fd_listener_lock = threading.Lock() _fd_listener_active = False -# (stop, done) pairs for fallback whole-reply speak paths currently active — -# the listener must cut THEIR private stop events too, and must keep -# listening while any of them is still speaking. +# (stop, done) pairs of fallback whole-reply speak paths: the listener must cut +# their private stop events too, and keep listening while any is still speaking. _fd_speak_pipelines: "set[tuple[threading.Event, threading.Event]]" = set() @@ -191,9 +184,7 @@ def _arm_full_duplex_listener() -> None: if _fd_listener_active: return _fd_listener_active = True - threading.Thread( - target=_full_duplex_listener, daemon=True, name="voice-full-duplex" - ).start() + threading.Thread(target=_full_duplex_listener, daemon=True, name="voice-full-duplex").start() def _fd_tts_pending() -> bool: @@ -210,16 +201,11 @@ def _fd_tts_pending() -> bool: def _full_duplex_listener() -> None: """Mic live from utterance-submit to turn-complete; phase-aware trip. - * generation phase (no TTS audio flowing): user speech interrupts every - running session's agent turn — the same ``agent.interrupt()`` seam - ``session.interrupt`` uses — and cuts any pending TTS pipeline so the - stale reply never plays. The captured utterance is transcribed and - emitted as ``voice.transcript`` (the TUI submits it as the next turn). - * playback phase: cuts TTS (streaming pipeline + fallback speak paths + - file player) and submits the captured interruption. - - Stop phrase is honored in both phases: mid-generation it interrupts the - turn AND ends the voice chat ("stop everything"). + Generation phase: user speech interrupts every running session's turn (the + ``agent.interrupt()`` seam ``session.interrupt`` uses) and cuts pending TTS so + the stale reply never plays. Playback phase: cuts TTS (streaming + fallback + speak paths + file player). Either way the utterance is transcribed and emitted + as ``voice.transcript``; a bare stop phrase also ends the voice chat. """ global _fd_listener_active try: @@ -253,9 +239,7 @@ def _full_duplex_listener() -> None: tripped = threading.Event() def _cut_all_tts() -> None: - # Streaming pipeline (private stop event + player). _tts_stream_stop(user_barge=True) - # Fallback whole-reply speak paths (their own stop events). with _fd_listener_lock: pipelines = list(_fd_speak_pipelines) for _stop, _done in pipelines: @@ -266,9 +250,7 @@ def _full_duplex_listener() -> None: tripped.set() mark_speech_interrupted() if phase == "playback": - logger.debug( - "TTS CUT: full-duplex listener tripped during playback" - ) + logger.debug("TTS CUT: full-duplex listener tripped during playback") _cut_all_tts() else: logger.debug( @@ -277,13 +259,9 @@ def _full_duplex_listener() -> None: ) # Cut pending TTS FIRST so the stale reply can never speak. _cut_all_tts() - # Interrupt every running session's turn — voice is - # process-global, and the same seam session.interrupt uses. try: with _sessions_lock: - running = [ - s for s in _sessions.values() if s.get("running") - ] + running = [s for s in _sessions.values() if s.get("running")] for s in running: agent = s.get("agent") if agent is not None and hasattr(agent, "interrupt"): @@ -296,11 +274,8 @@ def _full_duplex_listener() -> None: _voice_emit("voice.interrupted") wav_path = full_duplex_listen( - _should_stop, - is_playing=is_audio_output_active, - on_trigger=_on_trigger, - multiplier=_mult or None, - grace_ms=max(0, _grace_ms), + _should_stop, is_playing=is_audio_output_active, on_trigger=_on_trigger, + multiplier=_mult or None, grace_ms=max(0, _grace_ms), ) if not (wav_path and tripped.is_set()): return @@ -308,9 +283,8 @@ def _full_duplex_listener() -> None: result = transcribe_recording(wav_path) text = (result.get("transcript") or "").strip() if result.get("success") else "" if text: - # Stop-check must never break transcript delivery — if the - # helper is unavailable (stubbed voice_mode in tests, partial - # installs), treat as not-a-stop-phrase. + # Stop-check must never break transcript delivery (stubbed + # voice_mode in tests, partial installs) — treat as not-a-stop. try: from tools.voice_mode import is_voice_stop_phrase _is_stop = is_voice_stop_phrase(text) @@ -318,17 +292,8 @@ def _full_duplex_listener() -> None: _is_stop = False if _is_stop: - # Bare stop phrase — in EITHER phase the user means - # "stop everything": the turn was already interrupted / - # TTS cut at trip time; now end the voice chat. - os.environ["HERMES_VOICE"] = "0" - os.environ["HERMES_VOICE_TTS"] = "0" - try: - from hermes_cli.voice import stop_continuous - - stop_continuous() - except Exception: - pass + # Turn already interrupted / TTS cut at trip time; now end the chat. + _end_voice_chat(stop_loop=True, stop_tts=False) _voice_emit("voice.transcript", {"stop_phrase": True, "text": text}) else: _voice_emit("voice.transcript", {"text": text}) @@ -345,15 +310,9 @@ def _full_duplex_listener() -> None: def _speak_text_with_barge(text: str) -> None: - """Speak *text* via hermes_cli.voice.speak_text with spoken barge-in. - - The fallback whole-reply path (streaming couldn't start) and the - ``voice.tts`` RPC previously called ``speak_text`` bare — speech over - those paths was UNINTERRUPTIBLE by voice. The full-duplex agent-turn - listener covers this path too: the (stop, done) pair is registered in - ``_fd_speak_pipelines`` so the listener can cut the private stop event - on a playback trip and keeps listening while this speak is pending. - """ + """Speak via hermes_cli.voice.speak_text with spoken barge-in: the (stop, done) + pair is registered in ``_fd_speak_pipelines`` so the full-duplex listener can cut + it on a playback trip and keeps listening while it is pending.""" from hermes_cli.voice import speak_text stop = threading.Event() @@ -378,22 +337,20 @@ def _speak_text_with_barge(text: str) -> None: def _voice_cfg_dict() -> dict: - """Shape-safe accessor for the ``voice:`` block in config.yaml. - - ``_load_cfg()`` does not deep-merge DEFAULT_CONFIG, so both the - root AND ``voice`` may be any YAML scalar / list / None. A hand-edit - like ``voice: true`` or a malformed top-level config that parses to - a scalar would otherwise break ``.get("…")`` and take every - ``voice.*`` branch down with it (Copilot round-3..7 review on - #19835). Coerce through ``isinstance`` at every level so malformed - config falls back to an empty dict instead of crashing /voice. - """ + """Shape-safe ``voice:`` block. ``_load_cfg()`` doesn't deep-merge defaults, so + root and ``voice`` may be any YAML scalar/list/None; malformed → {}.""" cfg = _load_cfg() voice_cfg = cfg.get("voice") if isinstance(cfg, dict) else None return voice_cfg if isinstance(voice_cfg, dict) else {} +def _voice_cfg_number(value, default): + """Numeric config value, else *default*. bool is excluded explicitly (int + subclass): ``silence_threshold: true`` must not forward as ``1``.""" + return value if isinstance(value, (int, float)) and not isinstance(value, bool) else default + + def _voice_record_key() -> str: """Current ``voice.record_key`` value, documented default on error.""" record_key = _voice_cfg_dict().get("record_key") @@ -402,11 +359,11 @@ def _voice_record_key() -> str: # ── Wake word ("Hey Hermes") ────────────────────────────────────────────── -# The detector is process-global (one mic), like voice. The first eligible -# transport to call wake.start owns it until stop, disconnect, or stream failure. -# On detection we emit wake.detected; the client opens a new session and starts -# its own voice capture. The detector yields the mic to gateway voice.record -# (pause/resume below) and to the desktop's browser mic (wake.pause/resume RPCs). +# Process-global detector (one mic). The first eligible transport to call +# wake.start owns it until stop, disconnect, or stream failure. On detection we +# emit wake.detected; the client opens a session and starts its own capture. The +# detector yields the mic to voice.record (pause/resume) and to the desktop's +# browser mic (wake.pause/resume RPCs). _wake_lock = threading.Lock() _wake_owner_transport: "Optional[Transport]" = None _wake_owner_surface = "" @@ -447,15 +404,12 @@ def _wake_resume_if_owner(owner: "Transport", *, retry_seconds: float = 15.0, retry_interval: float = 1.0) -> bool: """Resume the wake detector for ``owner``; self-heal a busy microphone. - Reopening the mic right after a voice turn can fail while the capture - device is still being released (browser WebRTC tracks release async). - The CLI covers this with its idle watchdog; the gateway had nothing, so - one failed resume left the listener silently dead until the user toggled - it by hand — despite ``wake_word.enabled: true``. On an exception (mic - open failure) we retry in a background thread until it sticks, the lease - changes hands, or ``retry_seconds`` elapses. ``False`` from - ``resume_listening`` (lease gone / different owner) is final — never - retried, so this can't steal another surface's mic. + Reopening the mic right after a voice turn can fail while the device is still + being released (browser WebRTC tracks release async). On an exception we retry + in a background thread until it sticks, the lease changes hands, or + ``retry_seconds`` elapses. ``False`` from ``resume_listening`` (lease gone / + different owner) is final — never retried, so this can't steal another + surface's mic. """ from tools.wake_word import resume_listening @@ -497,12 +451,8 @@ def _wake_resume_if_owner(owner: "Transport", *, retry_seconds: float = 15.0, def _persist_wake_enabled(enabled: bool) -> bool: - """Write ``wake_word.enabled`` to config.yaml. - - Only called for explicit user gestures (the desktop ear toggle, ``/wake - on|off``) — never from passive auto-arm paths, so a mic can't become - persistently enabled without a deliberate click. - """ + """Write ``wake_word.enabled`` to config.yaml. Only for explicit user gestures + (ear toggle, /wake on|off) — never passive auto-arm paths.""" try: from cli import save_config_value @@ -512,20 +462,60 @@ def _persist_wake_enabled(enabled: bool) -> bool: return False +def _wake_prefers_client(params: dict, surface: str) -> bool: + """Desktop remote (gui) prefers client capture (Mac mic → wake.feed PCM) while + the engine runs on the backend; CLI/TUI stay local.""" + return surface in ("gui", "desktop") or bool(params.get("client_capture")) + + +def _wake_probe(cfg: dict, prefer_client: bool) -> tuple[str, dict]: + """``(capture_mode, requirements)`` with capture stamped so the probe matches + the mode that would actually arm.""" + from tools.wake_word import check_wake_word_requirements, resolve_capture_mode + + capture_mode = resolve_capture_mode(cfg, prefer_client=prefer_client) + probe_cfg = dict(cfg) + probe_cfg["capture"] = capture_mode + return capture_mode, check_wake_word_requirements(probe_cfg) + + +def _wake_detect_handler(transport, sid: str, phrase: str, new_session: bool): + """Build the on-detect callback: pause, verify ownership, emit ``wake.detected`` + on the owner's transport.""" + + def _on_detect() -> None: + from tools.wake_word import get_last_match, owns_listener, pause_listening + + if not pause_listening(owner=transport): + return + if not owns_listener(transport): + return + if _transport_is_dead(transport): + _release_wake_for_transport(transport) + return + # Multi-phrase engines report WHICH phrase fired and its profile, so one + # listener wakes any enrolled profile; single-phrase engines fall back. + matched_phrase, matched_profile = get_last_match() or (phrase, "") + logger.info("wake.detected: emitting to sid=%r (transport=%s, profile=%r)", + sid, type(transport).__name__, matched_profile) + token = bind_transport(transport) + try: + _emit("wake.detected", sid, { + "phrase": matched_phrase or phrase, "profile": matched_profile or None, + "start_new_session": new_session, + }) + finally: + reset_transport(token) + + return _on_detect + + @method("gateway.capabilities") def _(rid, params: dict) -> dict: - """What guarantees THIS BUILD enforces, for a client that must not assume. - - An automated client cannot tell a gateway that fences concurrent writers to - one session from one that silently allows them -- both accept the same calls - and both answer prompt.submit the same way. It only finds out by corrupting a - conversation. So the guarantee is advertised, and a client that does not see - it advertised is expected to withhold rather than hope. - - Sourced from the module that performs the enforcement, never from config: a - capability an operator can switch on without also having the mechanism is - worse than no capability at all, because it is believed. - """ + """Advertise what THIS BUILD enforces. A client can't tell a gateway that fences + concurrent writers from one that doesn't (both accept the same calls), so it + withholds unless the guarantee is advertised. Sourced from the enforcing module, + never config: a believed-but-absent capability is worse than none.""" from hermes_cli.active_sessions import PER_SESSION_EXCLUSIVE_SUBMIT return _ok(rid, {"per_session_exclusive_submit": bool(PER_SESSION_EXCLUSIVE_SUBMIT)}) @@ -533,16 +523,9 @@ def _(rid, params: dict) -> dict: @method("ping") def _(rid, params: dict) -> dict: - """Cheapest possible liveness probe for the desktop client. - - Answered synchronously on the WS reader thread, so it works even while - every agent is mid-turn or the GIL is contended — the round-trip only - measures socket health, not backend load. A desktop client uses it after - sleep/wake to distinguish a half-open TCP connection (no close event, so - ``connectionState`` still reads ``open`` while every RPC hangs until its - per-call timeout) from a genuinely healthy socket, and forces a reconnect - in the former case instead of letting the next ``prompt.submit`` hang. - """ + """Cheapest liveness probe, answered on the WS reader thread so it works while + every agent is mid-turn: lets the desktop tell a half-open TCP socket after + sleep/wake from a healthy one and force a reconnect.""" return _ok(rid, {"pong": True}) @@ -550,25 +533,20 @@ def _(rid, params: dict) -> dict: def _(rid, params: dict) -> dict: """Arm the wake-word listener for the calling surface ("tui" | "gui"). - Idempotent and gated: returns ``{started: False, reason}`` when the wake - word is disabled, scoped to another surface, or its deps/mic aren't ready. - - ``persist: true`` marks an explicit user gesture (toggle click, /wake on): - when the feature is disabled in config, it flips ``wake_word.enabled`` on - and saves it before arming, so the choice sticks for future sessions. - Passive auto-arm callers omit it and keep getting the config-gated refusal. + Idempotent and gated: ``{started: False, reason}`` when disabled, scoped to + another surface, or deps/mic aren't ready. ``persist: true`` marks an explicit + user gesture: when disabled in config it flips ``wake_word.enabled`` on before + arming; passive auto-arm callers omit it and keep the config-gated refusal. """ surface = str(params.get("surface") or "auto").strip().lower() persist = bool(params.get("persist")) - transport = current_transport() or _stdio_transport + transport = _caller_transport() try: from tools.wake_word import ( WakeWordInUse, - check_wake_word_requirements, detector_frame_info, load_wake_word_config, owns_listener, - resolve_capture_mode, start_listening, wake_phrase, wake_surface_enabled, @@ -577,24 +555,14 @@ def _(rid, params: dict) -> dict: return _err(rid, 5026, f"wake module unavailable: {e}") cfg = load_wake_word_config() - # Desktop remote (gui) prefers client capture: Mac mic → wake.feed PCM, - # while the engine still runs on the backend. CLI/TUI stay local. - prefer_client = surface in ("gui", "desktop") or bool(params.get("client_capture")) - capture_mode = resolve_capture_mode(cfg, prefer_client=prefer_client) + capture_mode, reqs = _wake_probe(cfg, _wake_prefers_client(params, surface)) external_audio = capture_mode == "client" - # Requirements first: a gesture on an unarmed-able setup (no STT/TTS, no - # mic, missing key) must refuse WITHOUT flipping wake_word.enabled — else - # config says on while nothing can ever arm, and auto-arm paths churn. - # Temporarily stamp capture so the probe matches the arm mode. - probe_cfg = dict(cfg) - probe_cfg["capture"] = capture_mode - reqs = check_wake_word_requirements(probe_cfg) + # Requirements first: a gesture on an un-armable setup must refuse WITHOUT + # flipping wake_word.enabled — else config says on while nothing can arm. if not reqs["available"]: logger.warning("wake.start(%s): not available — %s", surface, reqs.get("hint")) return _ok(rid, { - "started": False, - "reason": "unavailable", - "hint": reqs.get("hint") or "", + "started": False, "reason": "unavailable", "hint": reqs.get("hint") or "", "capture": capture_mode, }) enabled_persisted = False @@ -604,10 +572,8 @@ def _(rid, params: dict) -> dict: cfg = dict(cfg) cfg["enabled"] = True if not wake_surface_enabled(surface, cfg): - # Distinguish "feature off in config" (reason: disabled — a persist:true - # retry can turn it on) from "scoped to a different surface" (reason: - # disabled_for_surface — respects an explicit wake_word.surface choice, - # which persist does NOT override). + # "disabled" (a persist:true retry can turn it on) vs "disabled_for_surface" + # (explicit wake_word.surface choice, which persist does NOT override). reason = "disabled" if not cfg.get("enabled") else "disabled_for_surface" logger.info("wake.start(%s): %s (enabled=%s, surface=%s)", surface, reason, cfg.get("enabled"), cfg.get("surface")) @@ -621,55 +587,21 @@ def _(rid, params: dict) -> dict: existing_owner = None existing_surface = "" if existing_owner is not None and existing_owner is not transport: - return _ok(rid, { - "started": False, - "reason": "owned", - "owner_surface": existing_surface, - }) + return _ok(rid, {"started": False, "reason": "owned", "owner_surface": existing_surface}) sid = str(params.get("session_id") or "") - phrase = wake_phrase(cfg) - new_session = bool(cfg.get("start_new_session", True)) - - def _on_detect() -> None: - from tools.wake_word import get_last_match, owns_listener, pause_listening - - if not pause_listening(owner=transport): - return - if not owns_listener(transport): - return - if _transport_is_dead(transport): - _release_wake_for_transport(transport) - return - # Multi-phrase engines report WHICH phrase fired and the profile it - # belongs to, so one listener can wake any enrolled profile. Falls - # back to the owner's configured phrase / no profile for - # single-phrase engines. - matched_phrase, matched_profile = get_last_match() or (phrase, "") - logger.info("wake.detected: emitting to sid=%r (transport=%s, profile=%r)", - sid, type(transport).__name__, matched_profile) - token = bind_transport(transport) - try: - _emit("wake.detected", sid, { - "phrase": matched_phrase or phrase, - "profile": matched_profile or None, - "start_new_session": new_session, - }) - finally: - reset_transport(token) - try: start_listening( - _on_detect, + _wake_detect_handler( + transport, sid, wake_phrase(cfg), bool(cfg.get("start_new_session", True)) + ), owner=transport, config=cfg, external_audio=external_audio, ) except WakeWordInUse: return _ok(rid, { - "started": False, - "reason": "owned", - "owner_surface": existing_surface or None, + "started": False, "reason": "owned", "owner_surface": existing_surface or None, }) except Exception as e: logger.warning("wake.start(%s): failed to start listener: %s", surface, e) @@ -684,12 +616,8 @@ def _(rid, params: dict) -> dict: surface, reqs["phrase"], reqs["provider"], capture_mode, frame.get("frame_length"), ) return _ok(rid, { - "started": True, - "phrase": reqs["phrase"], - "provider": reqs["provider"], - "owner_surface": surface, - "enabled_persisted": enabled_persisted, - "capture": capture_mode, + "started": True, "phrase": reqs["phrase"], "provider": reqs["provider"], + "owner_surface": surface, "enabled_persisted": enabled_persisted, "capture": capture_mode, "sample_rate": frame.get("sample_rate", 16000), "frame_length": frame.get("frame_length", 1280), }) @@ -697,13 +625,9 @@ def _(rid, params: dict) -> dict: @method("wake.stop") def _(rid, params: dict) -> dict: - """Stop this surface's listener. - - ``persist: true`` (explicit user gesture) also writes - ``wake_word.enabled: false`` to config.yaml so auto-arm stays off in - future sessions — the toggle is the config, not just the live listener. - """ - transport = current_transport() or _stdio_transport + """Stop this surface's listener. ``persist: true`` also writes + ``wake_word.enabled: false`` so auto-arm stays off in future sessions.""" + transport = _caller_transport() stopped = _release_wake_for_transport(transport) disabled_persisted = False if bool(params.get("persist")): @@ -716,8 +640,7 @@ def _(rid, params: dict) -> dict: if currently_enabled: disabled_persisted = _persist_wake_enabled(False) return _ok(rid, { - "stopped": stopped, - "reason": None if stopped else "not_owner", + "stopped": stopped, "reason": None if stopped else "not_owner", "disabled_persisted": disabled_persisted, }) @@ -725,7 +648,7 @@ def _(rid, params: dict) -> dict: @method("wake.pause") def _(rid, params: dict) -> dict: """Release the mic (e.g. while the desktop's browser captures audio).""" - transport = current_transport() or _stdio_transport + transport = _caller_transport() try: from tools.wake_word import pause_listening @@ -734,22 +657,15 @@ def _(rid, params: dict) -> dict: except Exception as e: logger.debug("wake.pause failed: %s", e) paused = False - return _ok(rid, { - "paused": paused, - "reason": None if paused else "not_owner", - }) + return _ok(rid, {"paused": paused, "reason": None if paused else "not_owner"}) @method("wake.resume") def _(rid, params: dict) -> dict: """Reclaim the mic after a pause; no-op if the listener isn't armed.""" - transport = current_transport() or _stdio_transport - resumed = _wake_resume_if_owner(transport) + resumed = _wake_resume_if_owner(_caller_transport()) logger.info("wake.resume: detector resumed=%s", resumed) - return _ok(rid, { - "resumed": resumed, - "reason": None if resumed else "not_owner", - }) + return _ok(rid, {"resumed": resumed, "reason": None if resumed else "not_owner"}) @method("wake.status") @@ -757,24 +673,18 @@ def _(rid, params: dict) -> dict: try: from tools.wake_word import ( audio_is_silent, - check_wake_word_requirements, detector_frame_info, get_input_device_status, is_listening, load_wake_word_config, owns_listener, - resolve_capture_mode, silent_audio_hint, ) cfg = load_wake_word_config() - # Prefer client when the GUI asks (desktop remote re-arm / status). - prefer_client = bool(params.get("client_capture")) or str( - params.get("surface") or "" - ).strip().lower() in ("gui", "desktop") - probe_cfg = dict(cfg) - probe_cfg["capture"] = resolve_capture_mode(cfg, prefer_client=prefer_client) - reqs = check_wake_word_requirements(probe_cfg) - transport = current_transport() or _stdio_transport + probe_capture, reqs = _wake_probe( + cfg, _wake_prefers_client(params, str(params.get("surface") or "").strip().lower()) + ) + transport = _caller_transport() owner, owner_surface = _wake_owner_snapshot() owned_by_caller = owns_listener(transport) listening = owned_by_caller and is_listening() @@ -785,19 +695,16 @@ def _(rid, params: dict) -> dict: hint = f"Wake-word input device could not be resolved: {input_device['error']}" if silent and not hint: hint = silent_audio_hint(input_device) - # Effective capture: prefer the *armed* detector over config/auto. - # With capture:auto the GUI arms client mode, but a bare status probe - # would otherwise report "local" and the desktop would not reattach - # the PCM feeder after wake.detected. + # Effective capture: prefer the *armed* detector over config/auto, else + # with capture:auto a bare status probe reports "local" and the desktop + # never reattaches the PCM feeder after wake.detected. frame = detector_frame_info() if owned_by_caller and frame.get("external_audio"): capture = "client" elif owned_by_caller and listening: capture = "local" else: - capture = probe_cfg.get("capture") or reqs.get("capture") or str( - cfg.get("capture") or "auto" - ) + capture = probe_capture or reqs.get("capture") or str(cfg.get("capture") or "auto") return _ok(rid, { "listening": listening, "owned_by_caller": owned_by_caller, @@ -808,8 +715,7 @@ def _(rid, params: dict) -> dict: "input_device": input_device, "available": reqs["available"], "hint": hint, - # Config truth: clients use this to re-arm after a voice turn - # ("permanent on") without guessing from runtime listener state. + # Config truth: clients re-arm after a voice turn ("permanent on") from this. "enabled": bool(cfg.get("enabled")), # Armed but deaf despite an open stream; see platform-specific hint. "audio_silent": silent, @@ -824,18 +730,10 @@ def _(rid, params: dict) -> dict: @method("wake.feed") def _(rid, params: dict) -> dict: - """Push client-captured PCM into the armed wake detector. - - Params: - pcm: base64-encoded int16 mono little-endian samples (preferred), OR - pcm_b64: alias of pcm - Optional: - sample_rate: must be 16000 (ignored if missing; mismatched rates rejected) - - Used when ``wake.start`` returned ``capture: "client"`` so remote backends - without a microphone can still run openWakeWord on Mac/desktop audio. - """ - transport = current_transport() or _stdio_transport + """Push client-captured PCM (``pcm``/``pcm_b64``: base64 int16 mono LE, 16 kHz + only) into the armed detector — used when ``wake.start`` returned + ``capture: "client"`` so mic-less remote backends can run openWakeWord.""" + transport = _caller_transport() raw_b64 = params.get("pcm") or params.get("pcm_b64") or "" if not isinstance(raw_b64, str) or not raw_b64.strip(): return _err(rid, 4001, "wake.feed requires base64 pcm") @@ -861,168 +759,165 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"fed": bool(ok), "reason": None if ok else "not_owner"}) -@method("voice.toggle") -def _(rid, params: dict) -> dict: - """CLI parity for the ``/voice`` slash command. +def _voice_toggle_status(rid, params: dict) -> dict: + # Mirrors CLI _show_voice_status: STT/TTS availability tells the user WHY + # voice isn't working; record_key lets the TUI bind and display the shortcut. + payload: dict = { + "enabled": _voice_mode_enabled(), + "record_key": _voice_record_key(), + "tts": _voice_tts_enabled(), + } + try: + from tools.voice_mode import check_voice_requirements - Subcommands: + reqs = check_voice_requirements() + payload["available"] = bool(reqs.get("available")) + payload["audio_available"] = bool(reqs.get("audio_available")) + payload["stt_available"] = bool(reqs.get("stt_available")) + payload["details"] = reqs.get("details") or "" + except Exception as e: + # Optional transcription deps — /voice status must always answer. + logger.warning("voice.toggle status: requirements probe failed: %s", e) - * ``status`` — report mode + TTS flags (default when action is unknown). - * ``on`` / ``off`` — flip voice *mode* (the umbrella bit). Turning it - off also tears down any active continuous recording loop. Does NOT - start recording on its own; recording is driven by ``voice.record`` - (Ctrl+B) after mode is on, matching cli.py's enable/Ctrl+B split. - * ``tts`` — toggle speech-output of agent replies. Requires mode on - (mirrors CLI's _toggle_voice_tts guard). - """ - action = params.get("action", "status") + return _ok(rid, payload) - if action == "status": - # Mirror CLI's _show_voice_status: include STT/TTS provider - # availability so the user can tell at a glance *why* voice mode - # isn't working ("STT provider: MISSING ..." is the common case). - # ``record_key`` mirrors the configured ``voice.record_key`` so the - # TUI can both bind it (frontend ``isVoiceToggleKey``) and display - # it in /voice status — previously the TUI hardcoded Ctrl+B and - # ignored the config (#18994). - payload: dict = { - "enabled": _voice_mode_enabled(), + +def _voice_toggle_mode(rid, params: dict) -> dict: + enabled = params.get("action") == "on" + # Runtime-only flag (CLI parity) — never persisted, so the next TUI launch + # starts with voice OFF instead of auto-REC from a stale toggle. + os.environ["HERMES_VOICE"] = "1" if enabled else "0" + + stop_hint = "" + if enabled: + # Spoken-stop hint for the client; sourced from voice.stop_phrases, + # empty when the feature is disabled. + try: + from tools.voice_mode import voice_stop_hint + + stop_hint = voice_stop_hint() + except Exception: + stop_hint = "" + + # Speech output already on → warm the engine now, not on the first reply. + if _voice_tts_enabled(): + _tts_lease_async("tui:voice-tts", True) + + if not enabled: + # The continuous loop holds the microphone; tear it down with the mode. + try: + from hermes_cli.voice import stop_continuous + + stop_continuous() + except ImportError: + pass + except Exception as e: + logger.warning("voice: stop_continuous failed during toggle off: %s", e) + + # Clear TTS so it can be toggled independently later; silence live speech. + os.environ["HERMES_VOICE_TTS"] = "0" + _tts_stream_stop(user_barge=False) + _tts_lease_async("tui:voice-tts", False) + + return _ok( + rid, + { + "enabled": enabled, "record_key": _voice_record_key(), "tts": _voice_tts_enabled(), - } - try: - from tools.voice_mode import check_voice_requirements + "stop_hint": stop_hint, + }, + ) - reqs = check_voice_requirements() - payload["available"] = bool(reqs.get("available")) - payload["audio_available"] = bool(reqs.get("audio_available")) - payload["stt_available"] = bool(reqs.get("stt_available")) - payload["details"] = reqs.get("details") or "" - except Exception as e: - # check_voice_requirements pulls optional transcription deps — - # swallow so /voice status always returns something useful. - logger.warning("voice.toggle status: requirements probe failed: %s", e) - return _ok(rid, payload) +def _voice_toggle_tts(rid, params: dict) -> dict: + if not _voice_mode_enabled(): + return _err(rid, 4014, "enable voice mode first: /voice on") + new_value = not _voice_tts_enabled() + os.environ["HERMES_VOICE_TTS"] = "1" if new_value else "0" + if not new_value: + _tts_stream_stop(user_barge=False) + # on → pre-load the engine so the first reply starts hot; off → release the + # lease (last holder gone = resident local model freed). + _tts_lease_async("tui:voice-tts", new_value) + # record_key on every branch so a tts toggle never resets a custom binding. + return _ok(rid, {"enabled": True, "record_key": _voice_record_key(), "tts": new_value}) - if action in {"on", "off"}: - enabled = action == "on" - # Runtime-only flag (CLI parity) — no _write_config_key, so the - # next TUI launch starts with voice OFF instead of auto-REC from a - # persisted stale toggle. - os.environ["HERMES_VOICE"] = "1" if enabled else "0" - stop_hint = "" - if enabled: - # Spoken-stop hint for the client to render on voice-mode start. - # Sourced from voice.stop_phrases (custom phrases render - # correctly); empty when the feature is disabled. - try: - from tools.voice_mode import voice_stop_hint +_VOICE_TOGGLE_ACTIONS = { + "status": _voice_toggle_status, "on": _voice_toggle_mode, "off": _voice_toggle_mode, + "tts": _voice_toggle_tts, +} - stop_hint = voice_stop_hint() - except Exception: - stop_hint = "" - # Voice mode with speech output already on (voice.auto_tts / - # prior /voice tts) means replies will be spoken — warm the - # engine now rather than on the first reply. - if _voice_tts_enabled(): - _tts_lease_async("tui:voice-tts", True) +@method("voice.toggle") +def _(rid, params: dict) -> dict: + """CLI parity for ``/voice``: ``status``; ``on``/``off`` flip voice *mode* (off + also tears down the continuous loop; recording itself is driven by + ``voice.record``/Ctrl+B); ``tts`` toggles speech output (requires mode on).""" + action = params.get("action", "status") + handler = _VOICE_TOGGLE_ACTIONS.get(action) if isinstance(action, str) else None + if handler is None: + return _err(rid, 4013, f"unknown voice action: {action}") + return handler(rid, params) - if not enabled: - # Disabling the mode must tear the continuous loop down; the - # loop holds the microphone and would otherwise keep running. - try: - from hermes_cli.voice import stop_continuous - stop_continuous() - except ImportError: - pass - except Exception as e: - logger.warning("voice: stop_continuous failed during toggle off: %s", e) +# voice.record callbacks (module-level: they touch only process-global state). +def _vr_on_transcript(t): + _voice_emit("voice.transcript", {"text": t}) + _resume_voice_wake() - # Clear TTS so it can be toggled independently after voice is off, - # and silence any in-flight streaming speech. - os.environ["HERMES_VOICE_TTS"] = "0" - _tts_stream_stop(user_barge=False) - _tts_lease_async("tui:voice-tts", False) - return _ok( - rid, - { - "enabled": enabled, - "record_key": _voice_record_key(), - "tts": _voice_tts_enabled(), - "stop_hint": stop_hint, - }, - ) +def _vr_on_silent(): + _voice_emit("voice.transcript", {"no_speech_limit": True}) + _resume_voice_wake() - if action == "tts": - if not _voice_mode_enabled(): - return _err(rid, 4014, "enable voice mode first: /voice on") - new_value = not _voice_tts_enabled() - # Runtime-only flag (CLI parity) — see voice.toggle on/off above. - os.environ["HERMES_VOICE_TTS"] = "1" if new_value else "0" - if not new_value: - _tts_stream_stop(user_barge=False) - # The TTS toggle is the "speech is about to be needed" signal: on → - # pre-load the configured engine so the first reply starts hot; off → - # release the lease (last holder gone = resident local model freed). - _tts_lease_async("tui:voice-tts", new_value) - # Include ``record_key`` on every branch so a /voice tts toggle - # doesn't reset the TUI's cached shortcut to the default when a - # user has a custom binding configured (Copilot review, round 2 - # on #19835). Keeps parity with the status/on/off branches above. - return _ok( - rid, - { - "enabled": True, - "record_key": _voice_record_key(), - "tts": new_value, - }, - ) - return _err(rid, 4013, f"unknown voice action: {action}") +def _vr_on_stop_phrase(t): + # The user SAID a bare stop phrase: end the chat like /voice off and emit a + # distinct signal so clients end the conversation instead of treating it as + # a no-speech timeout. The continuous loop has already halted. + _end_voice_chat(stop_loop=False, stop_tts=True) + _voice_emit("voice.transcript", {"stop_phrase": True, "text": t}) + _resume_voice_wake() + + +def _vr_on_status(state): + _voice_emit("voice.status", {"state": state}) + if state == "idle": + _resume_voice_wake() @method("voice.record") def _(rid, params: dict) -> dict: - """VAD-bounded push-to-talk capture, CLI-parity. - - ``start`` begins one VAD-bounded capture and emits ``voice.transcript`` - after silence stops the recorder. ``stop`` forces transcription of the - active buffer, matching classic CLI push-to-talk. The voice wrapper retains - no-speech counts across single-shot starts, so three consecutive silent - captures emit ``voice.transcript`` with ``no_speech_limit=True``. - """ + """VAD-bounded push-to-talk capture, CLI-parity. ``start`` begins one capture + and emits ``voice.transcript`` when silence stops it; ``stop`` forces + transcription of the active buffer. The wrapper retains no-speech counts across + starts, so three silent captures emit ``no_speech_limit=True``.""" action = params.get("action", "start") wake_paused = False if action not in {"start", "stop"}: return _err(rid, 4019, f"unknown voice action: {action}") - transport = current_transport() or _stdio_transport + transport = _caller_transport() wake_owner, _surface = _wake_owner_snapshot() if wake_owner is not None and wake_owner is not transport: return _ok(rid, {"status": "busy", "reason": "wake_owned"}) try: + global _voice_event_sid, _voice_wake_owner if action == "start": if not _voice_mode_enabled(): return _err(rid, 4015, "voice mode is off — enable with /voice on") with _voice_sid_lock: - global _voice_event_sid, _voice_wake_owner _voice_event_sid = params.get("session_id") or _voice_event_sid from hermes_cli.voice import start_continuous - # Register the agent-busy probe so the shared voice wrapper can - # hold the no-speech counter during long agent turns (item: - # silence must not end the chat while the agent works). Safe to - # re-register on every start; older wrappers without the setter - # are tolerated. + # Busy probe holds the no-speech counter during long agent turns. + # Safe to re-register every start; older wrappers lack the setter. try: from hermes_cli.voice import set_voice_busy_probe @@ -1030,30 +925,15 @@ def _(rid, params: dict) -> dict: except Exception: pass - # Shape-safe lookups: malformed ``voice:`` YAML (bool/scalar/list) - # must not crash /voice with a 5025 — fall back to VAD defaults. - # - # Exclude ``bool`` from the numeric check since Python's bool is - # a subclass of int — a hand-edit like ``silence_threshold: true`` - # would otherwise forward as ``1`` instead of falling back to - # the documented 200 / 3.0 defaults (Copilot round-12 on #19835). + # Shape-safe: malformed voice YAML falls back to documented defaults. voice_cfg = _voice_cfg_dict() - threshold = voice_cfg.get("silence_threshold") - duration = voice_cfg.get("silence_duration") - safe_threshold = ( - threshold - if isinstance(threshold, (int, float)) - and not isinstance(threshold, bool) - else 200 - ) - safe_duration = ( - duration - if isinstance(duration, (int, float)) and not isinstance(duration, bool) - else 3.0 - ) - # Hand the mic to STT if the wake-word detector holds it; resume - # once a terminal capture event fires (one-shot transcript / silence - # limit), so wake-triggered and manual captures both coexist. + safe_threshold = _voice_cfg_number(voice_cfg.get("silence_threshold"), 200) + safe_duration = _voice_cfg_number(voice_cfg.get("silence_duration"), 3.0) + # max_recording_seconds: explicit numeric <= 0 disables the cap (0.0). + max_rec = _voice_cfg_number(voice_cfg.get("max_recording_seconds"), 120.0) + safe_max_rec = max_rec if max_rec > 0 else 0.0 + # Hand the mic to STT if the wake detector holds it; resume on a + # terminal capture event so wake-triggered and manual captures coexist. try: from tools.wake_word import pause_listening @@ -1064,55 +944,11 @@ def _(rid, params: dict) -> dict: with _voice_sid_lock: _voice_wake_owner = transport - def _on_transcript(t): - _voice_emit("voice.transcript", {"text": t}) - _resume_voice_wake() - - def _on_silent(): - _voice_emit("voice.transcript", {"no_speech_limit": True}) - _resume_voice_wake() - - def _on_stop_phrase(t): - # Explicit user intent: the user SAID a bare stop phrase - # ("stop"). End the voice chat exactly like a manual - # /voice off — flip the mode flags and silence any live - # streaming TTS — and emit a distinct signal so clients - # (TUI, desktop) end the conversation instead of treating - # it as a no-speech timeout. The continuous loop has - # already halted before this callback fires. - os.environ["HERMES_VOICE"] = "0" - os.environ["HERMES_VOICE_TTS"] = "0" - try: - _tts_stream_stop(user_barge=False) - except Exception: - pass - _voice_emit("voice.transcript", {"stop_phrase": True, "text": t}) - _resume_voice_wake() - - def _on_status(state): - _voice_emit("voice.status", {"state": state}) - if state == "idle": - _resume_voice_wake() - - # voice.max_recording_seconds — hard cap on a single recording's - # length. Same guard as the silence params: non-numeric / bool / - # missing falls back to the documented 120 default, while an - # explicit numeric value <= 0 disables the cap (0.0). - max_rec = voice_cfg.get("max_recording_seconds") - safe_max_rec = ( - (max_rec if max_rec > 0 else 0.0) - if isinstance(max_rec, (int, float)) and not isinstance(max_rec, bool) - else 120.0 - ) started = start_continuous( - on_transcript=_on_transcript, - on_status=_on_status, - on_silent_limit=_on_silent, - silence_threshold=safe_threshold, - silence_duration=safe_duration, - auto_restart=False, - max_recording_seconds=safe_max_rec, - on_stop_phrase=_on_stop_phrase, + on_transcript=_vr_on_transcript, on_status=_vr_on_status, + on_silent_limit=_vr_on_silent, silence_threshold=safe_threshold, + silence_duration=safe_duration, auto_restart=False, + max_recording_seconds=safe_max_rec, on_stop_phrase=_vr_on_stop_phrase, ) if started is False: _resume_voice_wake() @@ -1128,15 +964,11 @@ def _(rid, params: dict) -> dict: stop_continuous(force_transcribe=True) _resume_voice_wake() return _ok(rid, {"status": "stopped"}) - except ImportError: - if wake_paused or action == "stop": - _resume_voice_wake() - return _err( - rid, 5025, "voice module not available — install audio dependencies" - ) except Exception as e: if wake_paused or action == "stop": _resume_voice_wake() + if isinstance(e, ImportError): + return _err(rid, 5025, "voice module not available — install audio dependencies") return _err(rid, 5025, str(e)) @@ -1146,8 +978,8 @@ def _(rid, params: dict) -> dict: if not text: return _err(rid, 4020, "text required") try: - # Import check up front so a missing voice module still returns the - # documented 5026 instead of failing silently in the thread. + # Import check up front so a missing voice module returns 5026 instead + # of failing silently in the thread. import hermes_cli.voice # noqa: F401 threading.Thread( diff --git a/tui_gateway/model_switch.py b/tui_gateway/model_switch.py index 05ba33901c..e9ad77844e 100644 --- a/tui_gateway/model_switch.py +++ b/tui_gateway/model_switch.py @@ -14,35 +14,22 @@ _registry = HandlerRegistry() def _persist_model_switch(result) -> None: - # Use targeted, atomic key writes (comment/ordering-preserving) instead of - # rewriting the whole `model:` block. A full-block rewrite via save_config() - # destroys sibling keys the user set under `model:` — `model_slots`, - # `model_fallback`, etc. — when switching models from the TUI (#48305). + # Targeted key writes: a full `model:` block rewrite via save_config() would + # destroy sibling keys the user set there (`model_slots`, `model_fallback`, ...). from cli import save_config_value save_config_value("model.default", result.new_model) save_config_value("model.provider", result.target_provider) - if result.base_url: - save_config_value("model.base_url", result.base_url) - else: - # Clear any stale base_url when switching to a provider that doesn't use - # one (e.g. custom endpoint -> native provider). Reads coalesce null to - # absent (`model_cfg.get("base_url") or ""`), so a null is equivalent to - # removal without needing a key-delete. Leaving the old value would - # route the new model at the previous custom host (#48305). - save_config_value("model.base_url", None) + # A provider without a base_url must clear the stale one (custom endpoint -> + # native) or the new model routes at the old host; reads coalesce null to absent. + save_config_value("model.base_url", result.base_url or None) def _snapshot_agent_model_runtime(agent) -> dict: """Capture the current agent model runtime for a one-turn restore.""" - return { - "model": getattr(agent, "model", ""), - "provider": getattr(agent, "provider", ""), - "api_key": getattr(agent, "api_key", ""), - "base_url": getattr(agent, "base_url", ""), - "api_mode": getattr(agent, "api_mode", ""), - "primary_runtime": copy.deepcopy(getattr(agent, "_primary_runtime", None)), - } + snap = {k: getattr(agent, k, "") for k in ("model", "provider", "api_key", "base_url", "api_mode")} + snap["primary_runtime"] = copy.deepcopy(getattr(agent, "_primary_runtime", None)) + return snap def _restore_agent_model_runtime(agent, snapshot: dict | None) -> None: @@ -61,12 +48,9 @@ def _restore_agent_model_runtime(agent, snapshot: dict | None) -> None: logger.debug("TUI one-turn model restore via primary runtime failed", exc_info=True) if hasattr(agent, "switch_model"): agent.switch_model( - new_model=snapshot.get("model", ""), - new_provider=snapshot.get("provider", ""), - api_key=snapshot.get("api_key", ""), - base_url=snapshot.get("base_url", ""), - api_mode=snapshot.get("api_mode", ""), - capabilities=snapshot.get("capabilities"), + new_model=snapshot.get("model", ""), new_provider=snapshot.get("provider", ""), + api_key=snapshot.get("api_key", ""), base_url=snapshot.get("base_url", ""), + api_mode=snapshot.get("api_mode", ""), capabilities=snapshot.get("capabilities"), ) @@ -79,27 +63,20 @@ def _session_profile_runtime_scope(session: dict): return home_token = set_hermes_home_override(profile_home) secret_token = set_secret_scope(build_profile_secret_scope(Path(profile_home))) - # Same authoritative terminal policy the gateway binds per turn (#68559): - # a docker-configured dashboard profile must never resolve the launch - # process's pinned env. Failure → refusal scope (fail closed). - from tools.terminal_scope import ( - install_profile_terminal_scope as _install_term_scope, - ) + # Same terminal policy the gateway binds per turn: a docker-configured profile + # must never resolve the launch process's pinned env. Failure → refusal scope. + from tools.terminal_scope import install_profile_terminal_scope, reset_terminal_scope - terminal_token = _install_term_scope(Path(profile_home)) + terminal_token = install_profile_terminal_scope(Path(profile_home)) try: yield finally: - from tools.terminal_scope import reset_terminal_scope - reset_terminal_scope(terminal_token) reset_secret_scope(secret_token) reset_hermes_home_override(home_token) -def _restart_completed_failed_agent_build( - sid: str, session: dict, failed_ready: threading.Event | None -) -> bool: +def _restart_completed_failed_agent_build(sid: str, session: dict, failed_ready: threading.Event | None) -> bool: """Replace one completed failed build generation and start its retry.""" if failed_ready is None: return False @@ -141,11 +118,8 @@ def _apply_model_switch( persist_override: bool | None = None, ) -> dict: from hermes_cli.model_switch import ( - parse_model_switch_args, - resolve_persist_behavior, - switch_model, - MODEL_SWITCH_ERR_ONCE_WITH_GLOBAL, - MODEL_SWITCH_ERROR_TEXT, + parse_model_switch_args, resolve_persist_behavior, switch_model, + MODEL_SWITCH_ERR_ONCE_WITH_GLOBAL, MODEL_SWITCH_ERROR_TEXT, ) from hermes_cli.runtime_provider import resolve_runtime_provider @@ -160,20 +134,13 @@ def _apply_model_switch( else: model_input, explicit_provider, is_global_flag, _force_refresh, is_session = parsed_flags one_turn = False - # Conflict validation delegates to the shared single-owner parser; the - # TUI surfaces it as a raised ValueError (its historical behavior) - # using the canonical error copy. + # Conflict validation is the shared parser's; surface it with the canonical copy. if is_global_flag and one_turn: raise ValueError(MODEL_SWITCH_ERROR_TEXT[MODEL_SWITCH_ERR_ONCE_WITH_GLOBAL]) persist_global = ( persist_override if persist_override is not None - else resolve_persist_behavior( - is_global_flag, - is_session, - is_once=one_turn, - explicit_provider=explicit_provider, - ) + else resolve_persist_behavior(is_global_flag, is_session, is_once=one_turn, explicit_provider=explicit_provider) ) if not model_input: raise ValueError("model value required") @@ -195,22 +162,15 @@ def _apply_model_switch( runtime = resolve_runtime_provider(requested=None) current_provider = str(runtime.get("provider", "") or "") current_base_url = str(runtime.get("base_url", "") or "") - # Preserve a callable api_key (Azure Foundry Entra ID bearer - # provider) unchanged — ``str(...)`` would produce - # ``""`` and poison downstream switch_model - # validation. Match the agent-present branch's behavior at the - # top of this block. + # Keep a callable api_key (Azure Entra bearer) unchanged: ``str()`` would + # yield "" and poison switch_model validation. _runtime_key = runtime.get("api_key", "") - if callable(_runtime_key) and not isinstance(_runtime_key, str): - current_api_key = _runtime_key - else: - current_api_key = str(_runtime_key or "") + is_bearer = callable(_runtime_key) and not isinstance(_runtime_key, str) + current_api_key = _runtime_key if is_bearer else str(_runtime_key or "") - # Load user-defined providers so switch_model can resolve named custom - # endpoints (e.g. "ollama-launch") and validate against saved model lists. - user_provs = None - custom_provs = None - cfg = None + # User-defined providers let switch_model resolve named custom endpoints + # (e.g. "ollama-launch") and validate against saved model lists. + user_provs = custom_provs = cfg = None try: from hermes_cli.config import get_compatible_custom_providers, load_config @@ -221,15 +181,9 @@ def _apply_model_switch( pass result = switch_model( - raw_input=model_input, - current_provider=current_provider, - current_model=current_model, - current_base_url=current_base_url, - current_api_key=current_api_key, - is_global=persist_global, - explicit_provider=explicit_provider, - user_providers=user_provs, - custom_providers=custom_provs, + raw_input=model_input, current_provider=current_provider, current_model=current_model, + current_base_url=current_base_url, current_api_key=current_api_key, is_global=persist_global, + explicit_provider=explicit_provider, user_providers=user_provs, custom_providers=custom_provs, ) if not result.success: raise ValueError(result.error_message or "model switch failed") @@ -246,11 +200,8 @@ def _apply_model_switch( if isinstance(_mc, dict) and _mc.get("context_length") is not None: _cfg_ctx = int(_mc["context_length"]) merge_preflight_compression_warning( - result, - agent=agent, - messages=list(session.get("history", [])), - custom_providers=custom_provs, - config_context_length=_cfg_ctx, + result, agent=agent, messages=list(session.get("history", [])), + custom_providers=custom_provs, config_context_length=_cfg_ctx, ) except Exception as exc: logger.debug("preflight-compression switch warning failed: %s", exc) @@ -260,11 +211,8 @@ def _apply_model_switch( from hermes_cli.model_selection_guards import combined_selection_warning warning = combined_selection_warning( - result.new_model, - provider=result.target_provider, - base_url=result.base_url or current_base_url, - api_key=result.api_key or current_api_key, - model_info=result.model_info, + result.new_model, provider=result.target_provider, base_url=result.base_url or current_base_url, + api_key=result.api_key or current_api_key, model_info=result.model_info, ) except Exception: warning = None @@ -272,34 +220,21 @@ def _apply_model_switch( confirm_msg = warning.message if result.warning_message: confirm_msg = f"{confirm_msg}\n\n{result.warning_message}" - # Same contract as the deferred branch below: confirm_message is - # canonical, warning is the pre-confirm-era alias. Identical by - # design, not by accident. - return { - "value": result.new_model, - "warning": confirm_msg, - "confirm_required": True, - "confirm_message": confirm_msg, - } + # Same contract as _set_model's deferred branch: confirm_message is + # canonical, warning is the legacy alias — keep identical. + return {"value": result.new_model, "warning": confirm_msg, "confirm_required": True, "confirm_message": confirm_msg} if agent: try: agent.switch_model( - new_model=result.new_model, - new_provider=result.target_provider, - api_key=result.api_key, - base_url=result.base_url, - api_mode=result.api_mode, + new_model=result.new_model, new_provider=result.target_provider, api_key=result.api_key, + base_url=result.base_url, api_mode=result.api_mode, capabilities=getattr(result, "runtime_capabilities", None), ) except Exception as exc: - # The in-place swap rolled the agent back to the old working - # model/client and re-raised. Abort the commit: do NOT restart the - # slash worker, persist runtime, append the switch marker, set a - # session model_override, or persist to config — all of which would - # otherwise leave the session pinned to a broken model and kill the - # conversation on the next turn (#50163). A failed switch is a - # no-op; surface a clean error to the client. + # The in-place swap rolled the agent back and re-raised. Abort the whole + # commit (worker restart, persist, marker, override, config write) or the + # session stays pinned to a broken model. A failed switch is a no-op. logger.warning("In-place model switch failed for TUI agent: %s", exc) raise ValueError( f"Model switch to {result.new_model} failed ({exc}); " @@ -308,35 +243,21 @@ def _apply_model_switch( _restart_slash_worker(sid, session) _persist_live_session_runtime(session) _persist_live_session_system_prompt(session) - _append_model_switch_marker( - session, model=result.new_model, provider=result.target_provider - ) - _emit("session.info", sid, _session_info(agent, session)) + _append_model_switch_marker(session, model=result.new_model, provider=result.target_provider) + _emit_session_info(sid, session) if one_turn: session["one_turn_model_restore"] = restore_snapshot else: session.pop("one_turn_model_restore", None) - # Record the switch as a PER-SESSION override so a later rebuild of THIS - # session (e.g. /new via _reset_session_agent, or resume) re-derives the - # user's chosen model/provider instead of falling back to global config. - # - # We deliberately do NOT write process-global env vars (HERMES_MODEL / - # HERMES_INFERENCE_MODEL / HERMES_TUI_PROVIDER / HERMES_INFERENCE_PROVIDER) - # here. The desktop backend hosts every same-profile session in ONE process, - # so mutating os.environ on a /model switch leaked the new model/provider - # into every OTHER live session's next agent rebuild — switching the model - # in one session silently changed it in the others (the cross-session - # contamination bug). agent.switch_model() above already mutated the right - # agent in place; the override dict makes that choice survive a rebuild - # without touching shared process state. + # PER-SESSION override so a rebuild of THIS session (/new, resume) re-derives + # the chosen model. Deliberately NOT written to process-global env vars + # (HERMES_MODEL & co.): the desktop hosts every same-profile session in one + # process, so os.environ would leak the switch into every other session. if pin_session_override and isinstance(session, dict) and not one_turn: session["model_override"] = { - "model": result.new_model, - "provider": result.target_provider, - "base_url": result.base_url, - "api_key": result.api_key, - "api_mode": result.api_mode, + "model": result.new_model, "provider": result.target_provider, + "base_url": result.base_url, "api_key": result.api_key, "api_mode": result.api_mode, } if persist_global: _persist_model_switch(result) @@ -351,14 +272,10 @@ def _apply_model_switch( def _sync_bot_capabilities(sid: str, session: dict) -> None: """Rebuild a Bot Chat session's agent when its capability surface changed. - Bot Chats are eternal sessions; toolsets/MCP tool definitions are baked - into the live agent at construction, so a capability edit (Settings → - Capabilities, skill install, MCP toggle) would otherwise not apply until - /new. At turn start, hash the profile's capability surface - (tools/bot_mode_probe.capability_fingerprint) and, on change, swap in a - freshly built agent for the SAME session — history is session/DB-backed, - and the prompt-restore epoch check rebuilds the system prompt to match. - One rebuild per user-initiated change; identical state is a no-op. + Bot Chats are eternal sessions with toolsets/MCP baked in at construction, so a + capability edit would otherwise wait for /new. At turn start, fingerprint the + profile's capabilities and on change swap in a fresh agent for the SAME session + (history is DB-backed). One rebuild per change; identical state is a no-op. """ agent = session.get("agent") if agent is None: @@ -373,8 +290,7 @@ def _sync_bot_capabilities(sid: str, session: dict) -> None: return from tools.bot_mode_probe import capability_fingerprint - home = session.get("profile_home") or None - current = capability_fingerprint(home) + current = capability_fingerprint(session.get("profile_home") or None) if current == "unavailable": return seen = session.get("bot_caps_seen") @@ -384,28 +300,18 @@ def _sync_bot_capabilities(sid: str, session: dict) -> None: except Exception: return - # Capability surface changed — rebuild the agent in place. Same - # session_id/key, so the DB-backed history and (epoch-refreshed) system - # prompt carry over; only tool definitions and prompt bytes change. try: tokens = _set_session_context(sid, cwd=_session_cwd(session)) try: new_agent = _make_agent( - sid, - session["session_key"], - session_id=session["session_key"], - platform_override=_session_source(session), + sid, session["session_key"], session_id=session["session_key"], platform_override=_session_source(session) ) finally: _clear_session_context(tokens) new_agent._session_title_hint = "Bot Chat" session["agent"] = new_agent session["config_model_seen"] = _config_model_target() - _emit( - "notice", - sid, - {"message": "Capabilities updated — this bot's tools and prompt were refreshed."}, - ) + _emit("notice", sid, {"message": "Capabilities updated — this bot's tools and prompt were refreshed."}) except Exception as e: logger.warning("Bot capability sync failed for %s: %s", sid, e) @@ -427,49 +333,27 @@ def _sync_agent_model_with_config(sid: str, session: dict) -> None: if target == seen: return model, provider = target - # Already running the configured model (branched/resumed session before - # its first sync, or a config revert after a failed switch): adopt the - # baseline without a redundant switch. - if model == getattr(agent, "model", "") and ( - not provider or provider == getattr(agent, "provider", "") - ): + # Already on the configured model (resumed before first sync, or a config + # revert after a failed switch): adopt without switching. + if model == getattr(agent, "model", "") and (not provider or provider == getattr(agent, "provider", "")): return raw = f"{model} --provider {provider}" if provider else model try: + # This sync ADOPTS a config.yaml change; it must never write config back + # (that is how `hermes --tui -m` once leaked into config.yaml). _apply_model_switch( - sid, - session, - raw, - confirm_expensive_model=True, - pin_session_override=False, - # This sync ADOPTS a config.yaml change into the live session; it - # must never write config back. Without this, the flag/config - # default (persist_switch_by_default=True) re-persisted whatever - # target the sync computed — the path that leaked `hermes --tui -m` - # into config.yaml as the permanent global model. - persist_override=False, + sid, session, raw, confirm_expensive_model=True, pin_session_override=False, persist_override=False ) except Exception as e: - _emit( - "error", - sid, - {"message": f"Could not switch to configured model {model}: {e}"}, - ) + _emit("error", sid, {"message": f"Could not switch to configured model {model}: {e}"}) def _pending_switch_selection_warning(model: str, provider: str) -> str | None: """Selection-guard message for a model queued mid-turn, or ``None``. - Runs BEFORE the pick is stashed, while the client still has a live response - it can turn into a confirm prompt. Only pre-resolution inputs exist here -- - the model id the user picked and any explicit ``--provider`` -- which is - exactly what the data-policy guard keys on. Guards that can only decide - once base_url / api_key / model_info have settled still get their chance in - ``_apply_model_switch``; the cost guard returns ``None`` when pricing is - unknown, so an early call can only under-fire, never over-fire. - - A misbehaving guard must never break the pick, so exceptions are swallowed - and treated as "no warning" -- the apply-time check remains the backstop. + Runs BEFORE the pick is stashed, while the client can still turn the response + into a confirm prompt. Only pre-resolution inputs exist here, so this can only + under-fire; ``_apply_model_switch`` is the backstop. Exceptions mean "no warning". """ if not model: return None diff --git a/tui_gateway/prompt_attachments.py b/tui_gateway/prompt_attachments.py index 9f3b6cfd0a..da3821d0d0 100644 --- a/tui_gateway/prompt_attachments.py +++ b/tui_gateway/prompt_attachments.py @@ -6,6 +6,7 @@ method_ctx.bind_module), so they reference server.py globals bare. from __future__ import annotations +import re as _re from .method_ctx import HandlerRegistry, bind_module @@ -25,21 +26,20 @@ _IMAGE_MAGIC: tuple[tuple[bytes, str], ...] = ( (b"BM", ".bmp"), ) +# Context-ref values containing any of these must be quoted (desktop formatRefValue parity). +_ATTACHMENT_REF_NEEDS_QUOTING_RE = _re.compile(r"""[\s()\[\]{}<>"'`]""") +del _re # bodies are rebound onto server globals: import inside functions only + def _decode_attach_base64(raw: str, *, mime_prefix: str) -> bytes | None: - """Decode a base64 (optionally data-URL-wrapped) payload. - - Accepts ``data:...;base64,`` plus embedded whitespace. - Returns the decoded bytes, or ``None`` when the input isn't valid base64. - """ + """Decode a base64 payload, optionally ``data:...;base64,``-wrapped, + tolerating embedded whitespace. ``None`` when not valid base64.""" import base64 as _base64 import re as _re cleaned = raw.strip() m = _re.match( - rf"^data:{_re.escape(mime_prefix)}[a-zA-Z0-9.+-]*;base64,(.*)$", - cleaned, - _re.DOTALL, + rf"^data:{_re.escape(mime_prefix)}[a-zA-Z0-9.+-]*;base64,(.*)$", cleaned, _re.DOTALL, ) if m: cleaned = m.group(1) @@ -50,12 +50,24 @@ def _decode_attach_base64(raw: str, *, mime_prefix: str) -> bytes | None: return None -def _sniff_image_ext(img_bytes: bytes, filename: str = "") -> str: - """Resolve an image extension from a filename hint, else magic bytes. +def _decode_attach_payload(rid, raw_b64: str, *, mime_prefix: str, max_bytes: int, + label: str, empty_msg: str): + """``(bytes, None)`` or ``(None, error)`` for an upload: 4017 on bad/empty + base64, 4018 over *max_bytes*.""" + data = _decode_attach_base64(raw_b64, mime_prefix=mime_prefix) + if data is None: + return None, _err(rid, 4017, "data is not valid base64") + if not data: + return None, _err(rid, 4017, empty_msg) + if len(data) > max_bytes: + mb = max_bytes // (1024 * 1024) + return None, _err(rid, 4018, f"{label} too large ({len(data)} bytes; cap is {mb} MB)") + return data, None - Falls back to ``.png``. WebP needs the RIFF/WEBP container check, handled - before the generic table. - """ + +def _sniff_image_ext(img_bytes: bytes, filename: str = "") -> str: + """Extension from the filename hint, else magic bytes (WebP needs the RIFF/WEBP + container check), else ``.png``.""" if filename: suffix = Path(filename).suffix.lower() if suffix: @@ -78,33 +90,28 @@ def _allowed_image_extensions() -> frozenset[str]: return frozenset({".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}) -def _session_images_dir(session: dict) -> Path: - """Resolve the uploads ``images/`` dir against the session's effective home. +def _session_home_dir(session: dict, name: str) -> Path: + """``/``, anchored on the session's stored ``profile_home``. - Attach RPCs (``image.attach_bytes``, ``clipboard.paste``, ``pdf.attach``) - run BEFORE ``prompt.submit`` installs the session's profile HERMES_HOME - override, so ``get_hermes_home()`` here would return the gateway's launch - home. In a multi-profile / root-gateway deployment that writes the upload to - the launch home's ``images/`` while the sandbox mount and the vision host- - read allowlist both resolve the *session profile's* ``images/`` at run time - — so the file the agent tries to read is never the file we wrote (#69575). - - Anchor the write on the session's stored ``profile_home`` when present - (matching the mount/read scope), else fall back to the launch home. Keeps - per-profile isolation: a profile's uploads stay under that profile's home. + Attach RPCs run BEFORE ``prompt.submit`` installs the profile HERMES_HOME + override, so ``get_hermes_home()`` would return the gateway's launch home — + while the sandbox mounts and the vision host-read allowlist resolve the + *session profile's* dirs at run time. Writing anywhere else means the agent + can never see the file. """ profile_home = session.get("profile_home") base = Path(profile_home) if profile_home else _hermes_home - return base / "images" + return base / name + + +def _session_images_dir(session: dict) -> Path: + """Uploads ``images/`` dir for the session (see ``_session_home_dir``).""" + return _session_home_dir(session, "images") def _queue_attached_image(session: dict, img_bytes: bytes, ext: str, *, prefix: str) -> Path: - """Write image bytes into the gateway's images dir and queue them. - - Mirrors what ``image.attach`` does for a local path: appends to - ``session["attached_images"]`` so the next ``prompt.submit`` picks it up via - the existing native-image-attach pipeline. Returns the written path. - """ + """Write image bytes into the session images dir and append to + ``session["attached_images"]`` so the next ``prompt.submit`` picks them up.""" session["image_counter"] = session.get("image_counter", 0) + 1 img_dir = _session_images_dir(session) img_dir.mkdir(parents=True, exist_ok=True) @@ -119,20 +126,9 @@ def _queue_attached_image(session: dict, img_bytes: bytes, ext: str, *, prefix: return img_path -_ATTACHMENT_REF_NEEDS_QUOTING_RE = None - - def _format_ref_value(value: str) -> str: - """Quote a context-ref value when it contains whitespace or bracket chars. - - Mirrors the desktop ``formatRefValue`` so the staged ``@file:`` ref round-trips - through ``agent.context_references`` cleanly. - """ - import re as _re - - global _ATTACHMENT_REF_NEEDS_QUOTING_RE - if _ATTACHMENT_REF_NEEDS_QUOTING_RE is None: - _ATTACHMENT_REF_NEEDS_QUOTING_RE = _re.compile(r"""[\s()\[\]{}<>"'`]""") + """Quote a context-ref value containing whitespace/brackets/quotes so the staged + ``@file:`` ref round-trips through ``agent.context_references``.""" if not value or not _ATTACHMENT_REF_NEEDS_QUOTING_RE.search(value): return value if "`" not in value: @@ -155,19 +151,10 @@ def _attachment_ref_path(session: dict, target: Path) -> str: def _desktop_attachment_dir(session: dict) -> Path: - """Resolve the file-attachment staging dir against the session's effective home. - - Anchored on the session profile's ``attachments/`` dir (same rule as - ``_session_images_dir``): ``file.attach`` runs BEFORE ``prompt.submit`` - installs the session's profile HERMES_HOME override, while the docker/ssh - sandbox mounts are resolved against the *session profile's* home at run - time — so the staged file must land where the bind mount points, or the - container can never see it (#76577). ``attachments/`` is registered in - ``tools.credential_files._CACHE_DIRS`` and auto-mounted into containers. - """ - profile_home = session.get("profile_home") - base = Path(profile_home) if profile_home else _hermes_home - root = base / "attachments" + """File-attachment staging dir (``attachments/``, see ``_session_home_dir``); + registered in ``tools.credential_files._CACHE_DIRS`` and auto-mounted into + containers, so a staged file lands where the bind mount points.""" + root = _session_home_dir(session, "attachments") root.mkdir(parents=True, exist_ok=True) return root @@ -213,12 +200,8 @@ def _resolve_gateway_attachment_path(raw: str) -> Path | None: def _decode_attachment_data_url(data_url: str) -> bytes: - """Decode a ``data:;base64,`` payload to bytes. - - Unlike ``_decode_attach_base64`` (image-mime-specific), this accepts any - media type — text/csv, application/pdf, etc. — so non-image file uploads - round-trip. Also tolerates a bare base64 string with no data-URL prefix. - """ + """Decode a ``data:;base64,`` payload (any media type, unlike the + image-specific ``_decode_attach_base64``); bare base64 also accepted.""" import base64 as _base64 import binascii as _binascii import re as _re @@ -241,18 +224,13 @@ def _stage_session_file_attachment( data_url: str, name: str, ) -> tuple[Path, bool]: - """Make a desktop file attachment available to the remote gateway agent. - - Three cases: - 1. The path resolves to a file already INSIDE the session workspace — use - it as-is (no copy, ``uploaded=False``). - 2. The path resolves to a gateway-visible file OUTSIDE the workspace — copy - it into the session home's ``attachments/`` dir (bind-mounted into - container backends) so the ``@file:`` ref resolves inside the sandbox. - 3. The path doesn't exist on the gateway (the common remote case: it's a - path on the CLIENT's disk) — decode the uploaded ``data_url`` bytes and - write them into the session home's ``attachments/`` dir. + """Make a desktop file attachment available to the gateway agent. + 1. Path resolves INSIDE the session workspace → use as-is (``uploaded=False``). + 2. Gateway-visible file OUTSIDE the workspace → copy into ``attachments/`` + (bind-mounted into container backends) so ``@file:`` resolves in the sandbox. + 3. Not on the gateway (remote client disk) → decode ``data_url`` bytes into + ``attachments/``. Returns ``(stored_path, uploaded)``. """ workspace = Path(_session_cwd(session)).resolve() diff --git a/tui_gateway/prompt_turn.py b/tui_gateway/prompt_turn.py new file mode 100644 index 0000000000..009276dbff --- /dev/null +++ b/tui_gateway/prompt_turn.py @@ -0,0 +1,1083 @@ +"""The prompt turn: ``_run_prompt_submit`` and the per-phase helpers it drives. + +Bodies are rebound onto server.py's globals at install time (see +method_ctx.bind_module), so they reference server.py globals bare. + +Turn shape (all on a fresh daemon thread): + admit -> crash marker -> bind scopes -> resolve message -> run_conversation + -> commit history / message.complete -> goal & loop hooks -> release scopes + -> post-turn follow-ups (queued prompt, goal continuation, notifications). +""" + +from __future__ import annotations + +from .method_ctx import HandlerRegistry, bind_module + +_registry = HandlerRegistry() + + +def _is_successful_goal_turn(result: Any, status: str, raw: Any) -> bool: + """Return whether a turn produced a real response the goal judge can use.""" + return bool( + status == "complete" + and isinstance(raw, str) + and raw.strip() + and not (isinstance(result, dict) and result.get("failed")) + and not (isinstance(result, dict) and result.get("completed") is False) + ) + + +def _goal_max_turns() -> int: + try: + goals_cfg = _load_cfg().get("goals") or {} + return int(goals_cfg.get("max_turns", 20) or 20) + except Exception: + return 20 + + +def _plan_goal_compression_recovery( + session: dict, result: Any, *, status: str, raw: Any, +) -> tuple[str | None, str | None]: + """Plan a bounded active-goal retry after compression exhaustion. + + Compression exhaustion is a failed turn, so it must not be sent to the + goal judge or consume the goal's turn budget. One fresh continuation turn + is allowed. If that turn also exhausts compression, pause the goal rather + than spinning until an arbitrary user message happens to wake it up. + + Returns ``(continuation_prompt, status_notice)``. Sessions without an + active goal retain the existing error-only behavior. + """ + compression_exhausted = bool(isinstance(result, dict) and result.get("compression_exhausted")) + if not compression_exhausted: + if _is_successful_goal_turn(result, status, raw): + session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) + return None, None + from hermes_cli.goals import GoalManager + sid_key = str(session.get("session_key") or "") + if not sid_key: + return None, None + goal_mgr = GoalManager(session_id=sid_key, default_max_turns=_goal_max_turns()) + if not goal_mgr.is_active(): + session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) + return None, None + goal_created_at = float(getattr(goal_mgr.state, "created_at", 0.0) or 0.0) + recovery_state = session.get(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS) + attempts = 0 + if ( + isinstance(recovery_state, dict) + and recovery_state.get("goal_created_at") == goal_created_at + and recovery_state.get("goal") == getattr(goal_mgr.state, "goal", "") + ): + try: + attempts = int(recovery_state.get("attempts", 0) or 0) + except (TypeError, ValueError): + attempts = 0 + continuation_prompt = goal_mgr.next_continuation_prompt() + if attempts < _GOAL_COMPRESSION_RECOVERY_LIMIT and continuation_prompt: + session[_GOAL_COMPRESSION_RECOVERY_ATTEMPTS] = { + "goal_created_at": goal_created_at, "goal": getattr(goal_mgr.state, "goal", ""), + "attempts": attempts + 1, + } + return ( + continuation_prompt, + "Context compression was exhausted. Retrying the active goal once.", + ) + goal_mgr.pause(reason="context compression exhausted twice consecutively") + # A later explicit /goal resume gets a fresh bounded recovery cycle. + session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) + return ( + None, + "Goal paused after context compression was exhausted twice. " + "Run /compress, then /goal resume to continue.", + ) + + +# ── turn admission ─────────────────────────────────────────────────── + + +def _admit_prompt_turn( + sid: str, session: dict, text: Any, image_paths: list[str] | None, + queued_prompt_generation: int | None, +) -> tuple[list[str], Any] | None: + """Ownership + liveness gate every fresh turn source must cross. + + prompt.submit already claims the slot in its RPC handler, but crash + auto-continue, wake-ups and other synthesized turns call + ``_run_prompt_submit`` directly — the bypass that once let a second backend + run a duplicate turn. Returns ``(images, agent)`` or None when refused + (``running`` already reset). + """ + if (ownership_refusal := _ensure_active_session_slot(sid, session)) is not None: + logger.info( + "Refusing turn for session %s at _run_prompt_submit: %s", + session.get("session_key") or sid, + getattr(ownership_refusal, "reason", None) or "refused", + ) + with session["history_lock"]: + session["running"] = False + _emit("error", sid, {"message": str(ownership_refusal)}) + return None + with session["history_lock"]: + if session.get("_closing") or ( + queued_prompt_generation is not None + and int(session.get("_queued_prompt_generation", 0)) != queued_prompt_generation + ): + session["running"] = False + return None + if image_paths is None: + images = list(session.get("attached_images", [])) + session["attached_images"] = [] + else: + images = list(image_paths) + inflight = session.get("inflight_turn") + # A retained failed turn (see _fail_inflight_turn) is a stale leftover + # by the time a new turn starts — replace it, never append onto it. + if not isinstance(inflight, dict) or inflight.get("status") == "error": + _start_inflight_turn(session, text) + agent = session["agent"] + if hasattr(agent, "clear_interrupt"): + try: + agent.clear_interrupt() + except Exception: + pass + return images, agent + + +def _record_turn_marker(session: dict, text: Any) -> str: + """Write the durable crash marker; returns the session key it was written under. + + Retired the moment the outcome reaches the client (_retire_turn_marker), so + a surviving marker means the process died mid-turn and session.resume + auto-continues from it. Compression can rotate session_key mid-turn, so + the caller keeps this key. The key is published before the disk write so + an interrupt racing startup can retire it; the post-write cancel check + closes the inverse race where Stop lands first and clears no file yet. + """ + marker_home = _session_home(session) + marker_key = str(session.get("session_key") or "") + marker_attempt = int(session.pop("_auto_continue_attempt", 0) or 0) + marker_text = session.pop("_auto_continue_prompt", None) or text + if isinstance(marker_text, str) and marker_text.strip(): + with session["history_lock"]: + session["_active_turn_marker_key"] = marker_key + record_turn_start(marker_home, marker_key, marker_text, attempts=marker_attempt) + with session["history_lock"]: + marker_cancelled = bool(session.get("_turn_cancel_requested")) + if marker_cancelled: + clear_turn_marker(marker_home, marker_key) + return marker_key + + +# ── per-turn scopes ────────────────────────────────────────────────── + + +class _TurnScopes: + """Reset tokens for the thread/context scopes a turn binds (filled incrementally).""" + + __slots__ = ("approval", "session_tokens", "home", "secret", "terminal") + + def __init__(self) -> None: + self.approval = None + self.session_tokens: list = [] + self.home = None # per-turn HERMES_HOME override for a resumed remote profile + self.secret = None + self.terminal = None + + +def _bind_turn_scopes(sid: str, session: dict, scopes: _TurnScopes) -> None: + """Bind approval/session/profile/terminal scopes for this turn thread. + + Fills ``scopes`` field by field so a failure midway still leaves every + already-bound token for ``_release_turn_scopes``. The profile's COMPLETE + terminal policy is bound too (dashboard/TUI analogue of the gateway's + per-turn scope): terminal_tool otherwise reads the launch process's pinned + env, and a failed install leaves a refusal scope so terminal tools fail + closed instead of inheriting ambient policy. + """ + from tools.approval import set_current_session_key + scopes.approval = set_current_session_key(session["session_key"]) + scopes.session_tokens = _set_session_context(session["session_key"], ui_session_id=sid) + profile_home = session.get("profile_home") + if profile_home: + scopes.home = set_hermes_home_override(profile_home) + scopes.secret = set_secret_scope(build_profile_secret_scope(Path(profile_home))) + from tools.terminal_scope import install_profile_terminal_scope + scopes.terminal = install_profile_terminal_scope(Path(profile_home)) + # The sudo password callback is thread-local (tools.terminal_tool + # _callback_tls), so the build thread's wiring doesn't reach this turn + # thread — sudo prompts would fall through to /dev/tty and hang the + # headless gateway. Re-wire so they route to the sudo.request overlay + # (secret capture is a module global, so re-running is a no-op). + _wire_callbacks(sid) + + +def _release_turn_scopes(scopes: _TurnScopes) -> None: + try: + if scopes.approval is not None: + from tools.approval import reset_current_session_key + reset_current_session_key(scopes.approval) + except Exception: + pass + if scopes.home is not None: + reset_hermes_home_override(scopes.home) + if scopes.secret is not None: + reset_secret_scope(scopes.secret) + if scopes.terminal is not None: + from tools.terminal_scope import reset_terminal_scope + reset_terminal_scope(scopes.terminal) + _clear_session_context(scopes.session_tokens) + + +# ── message resolution ─────────────────────────────────────────────── + + +def _expand_context_references(agent, prompt: str, cwd: str): + """Expand ``@file`` references; returns the preprocess result (``.blocked``/``.message``).""" + from agent.context_references import preprocess_context_references + from agent.model_metadata import get_model_context_length + ctx_len = get_model_context_length( + getattr(agent, "model", "") or _resolve_model(), + base_url=getattr(agent, "base_url", "") or "", api_key=getattr(agent, "api_key", "") or "", + provider=getattr(agent, "provider", "") or "", + config_context_length=getattr(agent, "_config_context_length", None), + ) + return preprocess_context_references(prompt, cwd=cwd, allowed_root=cwd, context_length=ctx_len) + + +def _route_turn_images(agent, prompt: Any, images: list[str]) -> Any: + """Build the run message for a turn with attached images. + + "native" passes pixels to the main model as OpenAI-style content parts + (adapters translate per provider); "text" references the paths so the + agent analyzes them in-loop with vision_analyze, never blocking the submit + path on vision calls. Decision table: agent/image_routing.py. + """ + try: + from agent.image_routing import build_native_content_parts, decide_image_input_mode + from hermes_cli.config import load_config as _tui_load_config + _provider, _model = _active_image_routing_identity(agent) + mode = decide_image_input_mode( + _provider, _model, _tui_load_config(), + requested_provider=getattr(agent, "requested_provider", ""), + ) + if getattr(agent, "api_mode", "") == "codex_app_server": + mode = "text" + except Exception as _img_exc: + print( + f"[tui_gateway] image_routing decision failed, defaulting to text: {_img_exc}", + file=sys.stderr, + ) + mode = "text" + if mode != "native": + return _build_image_ref_message(prompt, images) + try: + parts, skipped = build_native_content_parts(prompt, images) + if skipped: + print( + f"[tui_gateway] native image attachment skipped {len(skipped)} unreadable path(s)", + file=sys.stderr, + ) + if any(p.get("type") == "image_url" for p in parts): + return parts + except Exception as _img_exc: + print( + f"[tui_gateway] native attach failed, falling back to text: {_img_exc}", + file=sys.stderr, + ) + return _build_image_ref_message(prompt, images) + + +def _start_turn_voice() -> tuple[Any, bool]: + """Arm voice-mode turn audio; returns ``(tts_queue, thinking_started)``. + + Streaming TTS speaks replies sentence-by-sentence as tokens arrive (CLI + parity). ``_tts_stream_begin`` goes first: it cuts any still-speaking + previous turn, and that cut IS this turn's barge-in, so it must latch + before the caller consumes the latch. The full-duplex listener lets the + user interject DURING generation (covers voice mode without working TTS). + The ambient "thinking" sound keeps long silent stretches from reading as + a dead session; per-blip gate skips while TTS audio flows or the mic is + capturing; stopped in the turn's finally. + """ + tts_queue = _tts_stream_begin() + if not _voice_mode_enabled(): + return tts_queue, False + if _voice_cfg_dict().get("barge_in", True): + _arm_full_duplex_listener() + try: + from tools.voice_mode import is_audio_output_active, start_thinking_sound + + def _thinking_should_play() -> bool: + if is_audio_output_active(): + return False + try: + from hermes_cli.voice import is_continuous_active + return not is_continuous_active() + except Exception: + return True + return tts_queue, start_thinking_sound(should_play=_thinking_should_play) + except Exception: + return tts_queue, False + + +def _stop_thinking_sound() -> None: + try: + from tools.voice_mode import stop_thinking_sound + stop_thinking_sound() + except Exception: + pass + + +def _apply_turn_notes(run_message: Any, session: dict) -> Any: + """Prepend the per-turn API-message notes (same enrichment channel as images). + + Barge mid-speech → tell the model so it can react instead of being + oblivious to its own interruption; then reactions added since the last + turn; then which window the message was typed into (HUD mode is per-turn + state, so it cannot live in the byte-stable system prompt). + """ + from tools.tts_streaming import SPEECH_INTERRUPTED_NOTE, take_speech_interrupted + if take_speech_interrupted(): + run_message = _prepend_note(run_message, SPEECH_INTERRUPTED_NOTE) + run_message = _prepend_note(run_message, _pending_reaction_notes(session)) + return _prepend_note(run_message, _hud_surface_note(session)) + + +def _build_run_kwargs( + agent, session: dict, history: list, prompt: Any, images: list[str], run_message: Any, + stream_cb, display_kind: str | None, display_metadata: dict | None, +) -> dict: + """Assemble ``run_conversation`` kwargs, feature-detecting optional parameters. + + A synthesized turn is typed at turn START so the crash persist writes its + row as a timeline event instead of a raw user bubble (forever, if the turn + never ends — the auto-continue case). The post-turn stamp is the fallback + for an older agent without the parameter; re-stamping is a no-op. + """ + run_kwargs = { + "conversation_history": list(history), + "stream_callback": stream_cb, + "persist_user_message": ( + _build_persist_user_message(prompt, images, run_message) if images else prompt + ), + } + try: + run_params = inspect.signature(agent.run_conversation).parameters + except (TypeError, ValueError): + run_params = {} + if "task_id" in run_params: + run_kwargs["task_id"] = session["session_key"] + if display_kind and "persist_user_display_kind" in run_params: + run_kwargs["persist_user_display_kind"] = display_kind + run_kwargs["persist_user_display_metadata"] = display_metadata + return run_kwargs + + +# ── post-run bookkeeping ───────────────────────────────────────────── + + +def _stamp_synthetic_display_kind( + agent, session: dict, result: Any, text: str, display_kind: str, display_metadata: dict | None +) -> None: + """Post-turn fallback stamp of a synthesized turn's display kind (DB row + result).""" + db = getattr(agent, "_session_db", None) + current_session_id = getattr(agent, "session_id", None) or session.get("session_key") + if db is not None: + try: + db.set_latest_matching_message_display_kind( + current_session_id, role="user", content=text, display_kind=display_kind, + display_metadata=display_metadata, + ) + except Exception: + logger.debug("failed to stamp synthetic display kind", exc_info=True) + if isinstance(result, dict) and isinstance(result.get("messages"), list): + for message in reversed(result["messages"]): + if message.get("role") == "user" and message.get("content") == text: + message["display_kind"] = display_kind + if display_metadata: + message["display_metadata"] = display_metadata + break + + +def _restore_moa_one_shot(sid: str, session: dict) -> None: + """Undo a /moa one-shot after its turn. + + The one-shot did a real in-place ``agent.switch_model()`` to MoA, so undoing + it must go back through the switch path — resetting ``model_override`` + alone would leave the live client pinned to MoA for the next turn. + """ + _restore = session.pop("moa_one_shot_restore", None) + if isinstance(_restore, dict): + _prev_override = _restore.get("override") + _prev_model = _restore.get("model") + _prev_provider = _restore.get("provider") + if _prev_override is None: + session.pop("model_override", None) + else: + session["model_override"] = _prev_override + if _prev_model: + _raw = f"{_prev_model} --provider {_prev_provider}" if _prev_provider else _prev_model + try: + _apply_model_switch( + sid, + session, + _raw, + confirm_expensive_model=False, + pin_session_override=bool(_prev_override), + persist_override=False, # session-internal restore, never config.yaml + ) + except Exception as _moa_restore_exc: + logger.warning("MoA one-shot model restore failed: %s", _moa_restore_exc) + elif _restore is None: + session.pop("model_override", None) + else: + session["model_override"] = _restore + + +def _commit_turn_history( + session: dict, result: dict, history: list, history_version: int +) -> str | None: + """Write the agent's messages back to session history; returns a client warning or None. + + Caller holds no lock. If history_version moved during the turn, the only + tolerated mutation is a pivot marker the gateway itself inserted mid-turn + (model switch, or /personality which lands immediately with no pending + queue); then the agent output is merged after the current history. + ``_append_model_switch_marker`` strips prior markers in place then appends, + so the delta is NOT a tail slice — compare content, not indices. Any other + desync (undo/compress/retry/rollback) is surfaced instead of silently + dropping the output. + """ + with session["history_lock"]: + current_version = int(session.get("history_version", 0)) + if current_version == history_version: + session["history"] = result["messages"] + session["history_version"] = history_version + 1 + return None + current_history = list(session["history"]) + history_no_markers = [e for e in history if not _is_pivot_marker(e)] + current_no_markers = [e for e in current_history if not _is_pivot_marker(e)] + pivot_only = current_no_markers == history_no_markers and any( + _is_pivot_marker(e) for e in current_history + ) + if pivot_only: + # Auto-compression can make result["messages"] shorter than the + # turn-start history; then the full result is the base. + if len(result["messages"]) > len(history): + new_messages = result["messages"][len(history):] + else: + new_messages = list(result["messages"]) + session["history"] = current_history + new_messages + session["history_version"] = current_version + 1 + return None + print( + f"[tui_gateway] prompt.submit: history_version mismatch " + f"(expected={history_version} current={current_version}) — " + f"agent output NOT written to session history", + file=sys.stderr, + ) + return ( + "History changed during this turn — the response above is visible " + "but was not saved to session history." + ) + + +def _turn_outcome(result: Any) -> tuple[Any, str, str | None]: + """Reduce a run_conversation result to ``(raw_text, status, last_reasoning)``.""" + if not isinstance(result, dict): + return str(result), "complete", None + raw = result.get("final_response", "") + status = ( + "interrupted" if result.get("interrupted") else "error" if result.get("error") else "complete" + ) + # No visible response AND a real error (e.g. invalid model slug → provider + # 4xx): surface the error as the text (classic CLI parity) instead of an + # empty turn. An empty successful turn still renders as empty. + if (not raw) and result.get("error") and (result.get("failed") or result.get("partial")): + raw = f"Error: {result.get('error')}" + # "Operation interrupted: waiting for model response (…)" is cancellation + # metadata, not assistant prose (gateway/run.py and ACP suppress it too). + if status == "interrupted" and isinstance(raw, str) and raw.strip().startswith( + INTERRUPT_WAITING_FOR_MODEL_PREFIX + ): + raw = "" + lr = result.get("last_reasoning") + last_reasoning = lr.strip() if isinstance(lr, str) and lr.strip() else None + return raw, status, last_reasoning + + +def _turn_error_surface(agent, result: Any) -> Any: + """Structured {layer, code, retryable} descriptor for an error result (advisory; never raises).""" + try: + from agent.error_surface import build_error_surface_from_result + return build_error_surface_from_result( + result, provider=str(getattr(agent, "provider", "") or ""), + model=str(getattr(agent, "model", "") or ""), + ) + except Exception: + return None + + +# ── post-turn hooks ────────────────────────────────────────────────── + + +def _goal_followup_after_turn(sid: str, session: dict, result: Any, status: str, raw: Any) -> str | None: + """/goal continuation (Ralph-style loop; mirrors gateway/run._post_turn_goal_continuation). + + Asks the judge whether the goal is done and, if not and still under budget, + returns the continuation prompt to chain once this thread releases + ``running``. The verdict is surfaced as a status line either way. + Compression failures are never judge input: the error text is not work + toward the goal, and evaluating it would spend a turn. + """ + goal_followup = None + compression_exhausted = bool(isinstance(result, dict) and result.get("compression_exhausted")) + try: + recovery_prompt, recovery_notice = _plan_goal_compression_recovery( + session, result, status=status, raw=raw + ) + if recovery_notice: + _emit("status.update", sid, {"kind": "goal", "text": recovery_notice}) + if recovery_prompt: + goal_followup = recovery_prompt + except Exception as _goal_recovery_exc: + print( + f"[tui_gateway] goal compression recovery failed: " + f"{type(_goal_recovery_exc).__name__}: {_goal_recovery_exc}", + file=sys.stderr, + ) + if compression_exhausted or not _is_successful_goal_turn(result, status, raw): + return goal_followup + try: + from hermes_cli.goals import GoalManager + sid_key = session.get("session_key") or "" + if sid_key: + goal_mgr = GoalManager(session_id=sid_key, default_max_turns=_goal_max_turns()) + if goal_mgr.is_active(): + try: + from hermes_cli.goals import gather_background_processes as _gather_bg + _bg_procs = _gather_bg() + except Exception: + _bg_procs = None + decision = goal_mgr.evaluate_after_turn( + raw, user_initiated=True, background_processes=_bg_procs + ) + verdict_msg = decision.get("message") or "" + if verdict_msg: + _emit("status.update", sid, {"kind": "goal", "text": verdict_msg}) + if decision.get("should_continue"): + cont_prompt = decision.get("continuation_prompt") or "" + if cont_prompt: + goal_followup = cont_prompt + except Exception as _goal_exc: + print( + f"[tui_gateway] goal continuation hook failed: " + f"{type(_goal_exc).__name__}: {_goal_exc}", + file=sys.stderr, + ) + return goal_followup + + +def _complete_loop_tick(sid: str, session: dict, raw: Any) -> None: + """If this turn was a /loop wakeup, evaluate it (LOOP_COMPLETE, --until judge, caps, next tick).""" + try: + from hermes_cli.loops import LoopManager + loop_sid_key = session.get("session_key") or "" + if loop_sid_key: + loop_mgr = LoopManager(session_id=loop_sid_key) + loop_state = loop_mgr.state + if loop_state is not None and loop_state.awaiting_response: + loop_decision = loop_mgr.complete_tick(raw if isinstance(raw, str) else "") + loop_msg = loop_decision.get("message") or "" + if loop_msg: + _emit("status.update", sid, {"kind": "loop", "text": loop_msg}) + except Exception as _loop_exc: + print( + f"[tui_gateway] loop completion hook failed: " + f"{type(_loop_exc).__name__}: {_loop_exc}", + file=sys.stderr, + ) + + +def _apply_pending_title(sid: str, session: dict) -> None: + """Apply pending_title now that the DB row exists — in the session-owned profile store.""" + _pending = session.get("pending_title") + if not _pending: + return + _session_key = session.get("session_key") or sid + try: + with _session_db(session) as _pdb: + if _pdb and _pdb.set_session_title(_session_key, _pending): + session["pending_title"] = None + except ValueError as exc: + # Invalid/duplicate title — non-retryable, drop it; auto-title takes over. + session["pending_title"] = None + logger.info("Dropping pending title for session %s: %s", _session_key, exc) + except Exception: + pass # transient DB failure — keep pending_title for retry + + +def _speak_turn_fallback(raw: str) -> None: + """Voice TTS fallback when the streaming pipeline couldn't start: speak the final text whole.""" + try: + # Barge-aware: spoken interruptions must cut this playback too. + threading.Thread(target=_speak_text_with_barge, args=(raw,), daemon=True).start() + except ImportError: + logger.warning("voice TTS skipped: hermes_cli.voice unavailable") + except Exception as e: + logger.warning("voice TTS dispatch failed: %s", e) + + +def _append_turn_crash_log(sid: str, trace: str) -> None: + try: + os.makedirs(os.path.dirname(_CRASH_LOG), exist_ok=True) + with open(_CRASH_LOG, "a", encoding="utf-8") as f: + f.write( + f"\n=== turn-dispatcher exception · " + f"{time.strftime('%Y-%m-%d %H:%M:%S')} · sid={sid} ===\n" + ) + f.write(trace) + except Exception: + pass + + +def _run_post_turn_followups(rid, sid: str, session: dict, result: Any, goal_followup: str | None) -> None: + """Chain whatever should run after ``running`` was released. + + Order: a user prompt that arrived mid-turn (interrupt + queue) wins over + every auto follow-up — drain it and skip the rest this cycle (the goal + judge / notifications re-evaluate after that turn). A leftover /steer the + agent couldn't inject (arrived during the final API call) is requeued + first so it isn't dropped; a real queued prompt still wins because + ``_enqueue_prompt`` merges both texts. Then the goal continuation, then + completion notifications that arrived mid-turn. Each nested + ``_run_prompt_submit`` checks ``running`` under the lock first, so a + racing user prompt (which sets running=True) wins. + """ + _leftover_steer = result.get("pending_steer") if isinstance(result, dict) else None + if isinstance(_leftover_steer, str) and _leftover_steer.strip(): + with session["history_lock"]: + _enqueue_prompt(session, _leftover_steer, session.get("transport")) + if _drain_queued_prompt(rid, sid, session): + return + if goal_followup: + with session["history_lock"]: + if session.get("running"): + return # user already sent something — their turn wins + session["running"] = True + try: + _emit("message.start", sid) + _run_prompt_submit(rid, sid, session, goal_followup) + except Exception as _cont_exc: + print( + f"[tui_gateway] goal continuation dispatch failed: " + f"{type(_cont_exc).__name__}: {_cont_exc}", + file=sys.stderr, + ) + with session["history_lock"]: + session["running"] = False + + # The background poller handles between-turn delivery; this is the safety + # net for events that arrived mid-turn. Ownership is positive-proof and + # compression-chain aware (same fail-closed gate as the poller): a turn + # finishing in session B must not consume session A's event, and a + # post-compression session still claims its pre-compression dispatches. + # Everything this session cannot claim is requeued for the poller. + try: + from tools.process_registry import process_registry + drained = process_registry.drain_notifications( + session_key=session.get("session_key", ""), + owns_event=lambda e: _session_owns_notification_event(sid, session, e), + skip_poll_observed=False, + ) + for index, (_evt, synth) in enumerate(drained): + with session["history_lock"]: + if session.get("running"): + for pending_evt, _pending_synth in drained[index:]: + process_registry.completion_queue.put(pending_evt) + break + session["running"] = True + from tools.async_delegation import ( + claim_event_delivery, complete_event_delivery, release_event_delivery, + ) + _claim = claim_event_delivery(_evt, "tui-post-turn") + if _claim is None: + continue + try: + _emit("message.start", sid) + _run_prompt_submit(rid, sid, session, synth) + complete_event_delivery(_evt, _claim) + except Exception as _n_exc: + release_event_delivery(_evt, _claim) + print( + f"[tui_gateway] completion notification dispatch failed: " + f"{type(_n_exc).__name__}: {_n_exc}", + file=sys.stderr, + ) + with session["history_lock"]: + session["running"] = False + except Exception as _drain_exc: + print( + f"[tui_gateway] completion queue drain failed: " + f"{type(_drain_exc).__name__}: {_drain_exc}", + file=sys.stderr, + ) + + +# ── the turn ───────────────────────────────────────────────────────── + + +def _run_prompt_submit( + rid, sid: str, session: dict, text: Any, *, display_kind: str | None = None, + display_metadata: dict | None = None, image_paths: list[str] | None = None, + queued_prompt_generation: int | None = None, + terminal_callback: Callable[[dict[str, Any]], None] | None = None, +) -> bool: + admitted = _admit_prompt_turn(sid, session, text, image_paths, queued_prompt_generation) + if admitted is None: + return False + images, agent = admitted + # The ONE INFO record proving a Desktop/TUI prompt was accepted by THIS + # process; ties together the UI session id, the gateway session_key and + # the agent's live session_id (compression rotates the last independently), + # which is what a rotation-mute trace needs. No prompt content is logged. + _turn_started_monotonic = time.monotonic() + logger.info( + "tui prompt accepted: ui_session=%s session_key=%s agent_session_id=%s " + "kind=%s chars=%s images=%d", + sid, + session.get("session_key") or "", + getattr(agent, "session_id", "") or "", + display_kind or "user", + len(text) if isinstance(text, str) else "-", + len(images), + ) + _emit("message.start", sid) + + def run(): + terminal_receipt_attempted = False + terminal_receipt_committed = terminal_callback is None + # ContextVars from the RPC dispatcher do not follow onto this thread. + # Rebind the exact transport stored on this session generation before + # any tool can commission a child; delegate_task captures it as + # non-serializable authority. + transport_token = bind_transport(session.get("transport")) + runtime_session_token = _current_runtime_session_record.set(session) + # Bound eagerly so the except/finally paths always have an agent even + # if turn setup throws; re-read after _sync_bot_capabilities, which may + # swap in a rebuilt agent for Bot Chat sessions. + agent = session["agent"] + scopes = _TurnScopes() + goal_followup = None + result = None # turn outcome; read after the finally for leftover /steer + tts_queue = None + thinking_started = False + history: list = [] + run_kwargs = None + one_turn_restore = session.pop("one_turn_model_restore", None) + # True once a failed turn's snapshot was retained for resume replay — + # tells the finally to skip the normal inflight clear. + turn_error_retained = False + # Cause for the "tui turn finished" bookend, stashed by both failure + # paths because the finally sees neither `result` nor the exception + # reliably. turn_prompt_text is what was actually submitted (post + # @-expansion, so injected file contents count), kept only so the cause + # can be checked for quoting it back (_strip_prompt_echo). + turn_error_detail = "" + turn_prompt_text = "" + marker_key = _record_turn_marker(session, text) + try: + _bind_turn_scopes(sid, session, scopes) + # Skip the config-model sync while a /model --once override is + # active: the once-model is intentionally not pinned as a session + # model_override, so the sync would clobber it back to the config + # model. A config.yaml change is adopted on the NEXT turn. A model + # picked mid-turn was queued, not applied in place — apply it on + # this thread before the first model call and before the config + # sync so the explicit pick wins over a config.yaml change. + if not one_turn_restore: + _apply_pending_model_switch(sid, session) + _sync_agent_model_with_config(sid, session) + _sync_agent_compression_with_config(sid, session) + # Bot Chat: adopt Settings→Capabilities edits into the eternal bot + # session before the turn runs. No-op for other session shapes. + _sync_bot_capabilities(sid, session) + agent = session["agent"] + # Snapshot after turn-start model sync: a deferred switch mutates + # history and its version, and that mutation belongs to this turn. + with session["history_lock"]: + history = list(session["history"]) + history_version = int(session.get("history_version", 0)) + cwd = _session_cwd(session) + _register_session_cwd(session) + cols = session.get("cols", 80) + streamer = make_stream_renderer(cols) + prompt = text + if isinstance(prompt, str) and "@" in prompt: + ctx = _expand_context_references(agent, prompt, cwd) + if ctx.blocked: + _emit( + "error", sid, + {"message": "\n".join(ctx.warnings) or "Context injection refused."}, + ) + return + prompt = ctx.message + turn_prompt_text = prompt if isinstance(prompt, str) else "" + run_message: Any = _route_turn_images(agent, prompt, images) if images else prompt + tts_queue, thinking_started = _start_turn_voice() + run_message = _apply_turn_notes(run_message, session) + + def _stream(delta): + with session["history_lock"]: + _append_inflight_delta(session, delta) + payload = {"text": delta} + if streamer and (r := streamer.feed(delta)) is not None: + payload["rendered"] = r + if tts_queue is not None and isinstance(delta, str): + tts_queue.put(delta) + _emit("message.delta", sid, payload) + + # Interim assistant text (commentary beside tool calls, or the + # attempted final answer before a verify-on-stop nudge) is sealed + # by the desktop as its own segment instead of being lost when + # message.complete replaces the streaming buffer. Gated on + # display.interim_assistant_messages (default true). + if _load_interim_assistant_messages(): + def _interim_assistant_cb(text: str, *, already_streamed: bool = False) -> None: + _emit("message.interim", sid, { + "text": text, + "already_streamed": already_streamed, + }) + agent.interim_assistant_callback = _interim_assistant_cb + else: + agent.interim_assistant_callback = None + run_kwargs = _build_run_kwargs( + agent, session, history, prompt, images, run_message, _stream, + display_kind, display_metadata, + ) + # Auto-titling fires inside the turn prologue; this live-rename + # hook repaints the sidebar the moment a title lands. + _title_key = session.get("session_key") or sid + agent._on_session_title = lambda t, _src, _k=_title_key: _emit( + "session.title", sid, {"session_id": _k, "title": t} + ) + _usage_stop, _usage_thread = _start_usage_ticker(sid, agent) + try: + result = agent.run_conversation(run_message, **run_kwargs) + finally: + # Stop AND join before anything below emits: a tick surviving + # past message.complete would roll the client's final usage + # back to a stale snapshot. The join is deliberately unbounded + # — once stop is set it only waits out one in-flight + # _get_usage/_emit, whose worst case (a stalled transport + # write) would stall the message.complete emit just the same. + _usage_stop.set() + _usage_thread.join() + if display_kind and isinstance(text, str): + _stamp_synthetic_display_kind( + agent, session, result, text, display_kind, display_metadata + ) + if "moa_one_shot_restore" in session: + _restore_moa_one_shot(sid, session) + status_note = None + if isinstance(result, dict): + if isinstance(result.get("messages"), list): + status_note = _commit_turn_history(session, result, history, history_version) + # Auto-compression inside run_conversation() may have rotated + # agent.session_id: sync session_key before title/goal/finalize + # handling uses it, keep pending_title (user intent) for the + # continuation, and restart the slash worker so worker-backed + # commands (/title etc.) target the live session. + _sync_session_key_after_compress( + sid, session, clear_pending_title=False, restart_slash_worker=True, + ) + raw, status, last_reasoning = _turn_outcome(result) + payload = {"text": raw, "usage": _get_usage(agent), "status": status} + if last_reasoning: + payload["reasoning"] = last_reasoning + if status_note: + payload["warning"] = status_note + if result.get("response_previewed"): + payload["response_previewed"] = True + # Structured billing-wall descriptor so the client renders a + # billing-specific recovery surface instead of re-parsing text. + _billing_block = result.get("billing_block") if isinstance(result, dict) else None + if _billing_block: + payload["billing"] = _billing_block + payload["failure_reason"] = result.get("failure_reason") + rendered = render_message(raw, cols) + if rendered: + payload["rendered"] = rendered + # Layer descriptor computed before the retain below so resume + # replay carries the same one (advisory; older clients ignore it). + _error_surface = _turn_error_surface(agent, result) if status == "error" else None + _result_error = result.get("error") if isinstance(result, dict) else None + with session["history_lock"]: + if status == "error": + # Retain the failed turn for resume replay: if this + # terminal frame is lost to a disconnect, resume's inflight + # payload is the only carrier of the failure. + _fail_inflight_turn( + session, _result_error if isinstance(result, dict) else raw, + error_surface=_error_surface, + ) + turn_error_retained = True + turn_error_detail = _turn_failure_detail( + (_result_error if isinstance(result, dict) else raw), + (result.get("failure_reason") if isinstance(result, dict) else None), + turn_prompt_text, + ) + else: + _clear_inflight_turn(session) + if status == "error": + payload["error"] = str((_result_error if isinstance(result, dict) else "") or raw) + payload["recoverable"] = True + if _error_surface: + payload["error_surface"] = _error_surface + if terminal_callback is not None: + terminal_receipt_attempted = True + terminal_callback( + { + "status": ( + "cancelled" + if status == "interrupted" + else "failed" if status == "error" else "settled" + ), + "text": raw if isinstance(raw, str) else str(raw), + **( + {"error": str(_result_error or raw)} + if status == "error" and isinstance(result, dict) + else {} + ), + } + ) + terminal_receipt_committed = True + if terminal_receipt_committed: + _retire_turn_marker(session, marker_key) + _emit("message.complete", sid, payload) + goal_followup = _goal_followup_after_turn(sid, session, result, status, raw) + if status == "complete": + _complete_loop_tick(sid, session, raw) + _apply_pending_title(sid, session) + # The streaming path already spoke everything via tts_queue. + if tts_queue is None and isinstance(raw, str) and raw.strip() and _voice_tts_enabled(): + _speak_turn_fallback(raw) + except Exception as e: + import traceback + _append_turn_crash_log(sid, traceback.format_exc()) + print(f"[gateway-turn] {type(e).__name__}: {e}", file=sys.stderr, flush=True) + # An exception in the agent's finalizer can leave the gateway's + # in-memory history at the turn-start snapshot; keep the partial + # turn available to the next prompt (the durable inflight record + # still carries the recoverable error state). + _restore_agent_history_after_turn_error(session, agent) + if terminal_callback is not None and not terminal_receipt_attempted: + terminal_receipt_attempted = True + try: + terminal_callback({"status": "failed", "text": "", "error": str(e)}) + terminal_receipt_committed = True + except Exception: + logger.exception("hosted room terminal receipt commit failed") + try: + # Same terminal error frame shape as the returned-error path + # (uniform client handling), retaining the turn for replay. + _emit_terminal_turn_error(sid, session, e, retire_marker=terminal_receipt_committed) + turn_error_retained = True + turn_error_detail = _turn_failure_detail(e, type(e).__name__, turn_prompt_text) + except Exception as emit_exc: + print( + f"[gateway-turn] terminal error emit failed: " + f"{type(emit_exc).__name__}: {emit_exc}", + file=sys.stderr, + flush=True, + ) + _emit("error", sid, {"message": str(e)}) + finally: + # Drop both local snapshots of the pre-turn history before asking + # glibc to return pages; session["history"] already points at the + # new/pruned result. + history.clear() + if isinstance(run_kwargs, dict): + run_kwargs.clear() + # While any profile-specific HERMES_HOME override is still active, + # so context.memory_trim resolves from the session's own config. + try: + from hermes_cli.mem_trim import trim_memory + trim_memory(reason="tui turn completion") + except Exception: + logger.debug("post-turn memory trim failed", exc_info=True) + if thinking_started: + _stop_thinking_sound() + if tts_queue is not None: + tts_queue.put(None) # end-of-text sentinel — flush + finish speaking + if one_turn_restore: + try: + _restore_agent_model_runtime(agent, one_turn_restore) + _restart_slash_worker(sid, session) + _persist_live_session_runtime(session) + _persist_live_session_system_prompt(session) + except Exception: + logger.debug("TUI one-turn model restore failed", exc_info=True) + _release_turn_scopes(scopes) + _current_runtime_session_record.reset(runtime_session_token) + reset_transport(transport_token) + # A stale interim closure must not fire during a later turn. + agent.interim_assistant_callback = None + with session["history_lock"]: + session["running"] = False + session["last_active"] = time.time() + if not turn_error_retained: + _clear_inflight_turn(session) + # Closing bookend of "tui prompt accepted" — fires on every path, + # so one accepted prompt always produces exactly one finished + # record. agent.session_id is re-read because compression may have + # rotated it mid-turn (an accepted/finished pair whose id changed IS + # a rotation trace). A missing finished record means the thread + # died before this finally. + logger.info( + "tui turn finished: ui_session=%s session_key=%s " + "agent_session_id=%s status=%s error_retained=%s duration=%.1fs" + "%s", + sid, + session.get("session_key") or "", + getattr(agent, "session_id", "") or "", + ( + result.get("interrupted") + and "interrupted" + or result.get("error") + and "error" + or "complete" + ) + if isinstance(result, dict) + else ("error" if turn_error_retained else "complete"), + turn_error_retained, + time.monotonic() - _turn_started_monotonic, + turn_error_detail, + ) + # Backstop for turns that never reached a terminal frame. + if terminal_receipt_committed: + _retire_turn_marker(session, marker_key) + with session["history_lock"]: + if session.get("_active_turn_marker_key") == marker_key: + session.pop("_active_turn_marker_key", None) + session.pop("_hosted_room_task", None) + session.pop("_auto_continue_scheduled", None) + _emit_settled_session_info(sid, session, agent) + _run_post_turn_followups(rid, sid, session, result, goal_followup) + run_thread = threading.Thread(target=run, daemon=True) + with _sessions_lock: + registered = _sessions.get(sid) + can_start = (not session.get("_closing") and (registered is None or registered is session)) + if can_start: + session["_run_thread"] = run_thread + run_thread.start() + if not can_start: + with session["history_lock"]: + session["running"] = False + return can_start + + +def register(server) -> None: + """Publish this module's helpers onto ``server``, rebound to its globals.""" + bind_module(globals(), server, skip=("_",)) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index c9db8d23bb..275f6eb6df 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -4,6 +4,7 @@ import contextlib import contextvars import copy import hashlib +import importlib import inspect import json import logging @@ -18,18 +19,10 @@ from datetime import datetime from pathlib import Path from typing import Any, Callable, NamedTuple, Optional -from agent.secret_scope import ( - build_profile_secret_scope, - reset_secret_scope, - set_secret_scope, -) +from agent.secret_scope import (build_profile_secret_scope, reset_secret_scope, set_secret_scope) from hermes_constants import ( - DEFAULT_INDICATOR_STYLE, - INDICATOR_STYLES, - get_hermes_home, - get_hermes_home_override, - reset_hermes_home_override, - set_hermes_home_override, + DEFAULT_INDICATOR_STYLE, INDICATOR_STYLES, get_hermes_home, get_hermes_home_override, + reset_hermes_home_override, set_hermes_home_override, ) from hermes_cli.env_loader import load_hermes_dotenv from utils import is_truthy_value @@ -40,52 +33,32 @@ from agent.skill_commands import describe_skill_invocation from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX from tui_gateway import git_probe from tui_gateway._env import env_float, env_int -from tui_gateway.turn_marker import ( - clear_turn_marker, - read_turn_marker, - record_turn_start, -) +from tui_gateway.turn_marker import (clear_turn_marker, read_turn_marker, record_turn_start) from tui_gateway.transport import ( - StdioTransport, - Transport, - bind_transport, - current_transport, - reset_transport, + StdioTransport, Transport, bind_transport, current_transport, reset_transport, ) logger = logging.getLogger(__name__) _hermes_home = get_hermes_home() -load_hermes_dotenv( - hermes_home=_hermes_home, project_env=Path(__file__).parent.parent / ".env" -) +load_hermes_dotenv(hermes_home=_hermes_home, project_env=Path(__file__).parent.parent / ".env") # ── Panic logger ───────────────────────────────────────────────────── -# Gateway crashes in a TUI session leave no forensics: stdout is the -# JSON-RPC pipe (TUI side parses it, doesn't log raw), the root logger -# only catches handled warnings, and the subprocess exits before stderr -# flushes through the stderr->gateway.stderr event pump. This hook -# appends every unhandled exception to ~/.hermes/logs/tui_gateway_crash.log -# AND re-emits a one-line summary to stderr so the TUI can surface it in -# Activity — exactly what was missing when the voice-mode turns started -# exiting the gateway mid-TTS. +# Crashes otherwise leave no forensics (stdout is the JSON-RPC pipe, stderr +# doesn't flush before exit): append every unhandled exception to +# logs/tui_gateway_crash.log and re-emit a one-line stderr summary for Activity. _CRASH_LOG = os.path.join(_hermes_home, "logs", "tui_gateway_crash.log") def _panic_hook(exc_type, exc_value, exc_tb): import traceback - trace = "".join(traceback.format_exception(exc_type, exc_value, exc_tb)) - try: + with contextlib.suppress(Exception): os.makedirs(os.path.dirname(_CRASH_LOG), exist_ok=True) with open(_CRASH_LOG, "a", encoding="utf-8") as f: - f.write( - f"\n=== unhandled exception · {time.strftime('%Y-%m-%d %H:%M:%S')} ===\n" - ) + f.write(f"\n=== unhandled exception · {time.strftime('%Y-%m-%d %H:%M:%S')} ===\n") f.write(trace) - except Exception: - pass # Stderr goes through to the TUI as a gateway.stderr Activity line — # the first line here is what the user will see without opening any # log files. Rest of the stack is still in the log for full context. @@ -105,11 +78,8 @@ sys.excepthook = _panic_hook def _thread_panic_hook(args): # threading.excepthook signature: SimpleNamespace(exc_type, exc_value, exc_traceback, thread) import traceback - - trace = "".join( - traceback.format_exception(args.exc_type, args.exc_value, args.exc_traceback) - ) - try: + trace = "".join(traceback.format_exception(args.exc_type, args.exc_value, args.exc_traceback)) + with contextlib.suppress(Exception): os.makedirs(os.path.dirname(_CRASH_LOG), exist_ok=True) with open(_CRASH_LOG, "a", encoding="utf-8") as f: f.write( @@ -117,8 +87,6 @@ def _thread_panic_hook(args): f"· thread={args.thread.name} ===\n" ) f.write(trace) - except Exception: - pass first_line = ( str(args.exc_value).strip().splitlines()[0] if str(args.exc_value).strip() @@ -126,19 +94,16 @@ def _thread_panic_hook(args): ) print( f"[gateway-crash] thread {args.thread.name} raised {args.exc_type.__name__}: {first_line}", - file=sys.stderr, - flush=True, + file=sys.stderr, flush=True, ) threading.excepthook = _thread_panic_hook -try: +with contextlib.suppress(Exception): from hermes_cli.banner import prefetch_update_check prefetch_update_check() -except Exception: - pass from tui_gateway.render import make_stream_renderer, render_diff, render_message @@ -167,16 +132,10 @@ _cfg_path = None _session_resume_lock = threading.Lock() _SLASH_WORKER_TIMEOUT_S = max(5.0, env_float("HERMES_TUI_SLASH_TIMEOUT_S", 45.0)) -# When a WebSocket client (the dashboard's embedded-chat tab / desktop app) -# disconnects, ``tui_gateway.ws`` detaches the transport but intentionally -# leaves the session parked so a quick reconnect can reattach it (see ws.py). -# That park is unbounded, though: a browser refresh spins up a brand-new -# ``session.create`` (new sid + a fresh _SlashWorker via _deferred_build) and -# never reattaches the OLD sid, so the old session's slash-worker subprocess -# lingers forever — one leaked python process per refresh (#38591 fallout). -# After this grace window, an orphaned WS session is interrupted if it is still -# running, then reaped once the normal turn-finalization path settles. -# Set to 0 to disable (park forever, pre-fix behaviour). +# On WS disconnect ws.py parks the session for a quick reattach, but a browser +# refresh creates a NEW sid and never reattaches the old one (leaking its slash +# worker per refresh). After this grace an orphaned WS session is interrupted +# if running, then reaped once turn finalization settles. 0 = park forever. def _resolve_ws_orphan_reap_grace() -> float: """Resolve the WS-orphan reap grace window (seconds). @@ -188,10 +147,7 @@ def _resolve_ws_orphan_reap_grace() -> float: if raw is None or not str(raw).strip(): try: from hermes_cli.config import load_config - - raw = (load_config().get("dashboard") or {}).get( - "ws_orphan_reap_grace_s" - ) + raw = (load_config().get("dashboard") or {}).get("ws_orphan_reap_grace_s") except Exception: raw = None try: @@ -222,10 +178,7 @@ def _resolve_ws_orphan_activity_stale() -> float: if raw is None or not str(raw).strip(): try: from hermes_cli.config import load_config - - raw = (load_config().get("dashboard") or {}).get( - "ws_orphan_activity_stale_s" - ) + raw = (load_config().get("dashboard") or {}).get("ws_orphan_activity_stale_s") except Exception: raw = None try: @@ -237,26 +190,19 @@ def _resolve_ws_orphan_activity_stale() -> float: _WS_ORPHAN_ACTIVITY_STALE_S = _resolve_ws_orphan_activity_stale() _WS_ORPHAN_INTERRUPT_REAP_POLL_S = 1.0 -# Total budget for the interrupt-then-reap poll chain. If an interrupted turn -# never settles (agent thread hung in a syscall, supervisor lost), each 1s poll -# would otherwise reschedule forever — trading the old leak-one-worker bug for -# leak-one-session-plus-timer-chain (review finding, PR #90373). After this -# many polls we log loudly and force-reap, mirroring the pre-existing -# stuck-`running` safety net's role of breaking the deadlock. +# Budget for the interrupt-then-reap poll chain: an interrupted turn that never +# settles (thread hung in a syscall) would reschedule the 1s poll forever. After +# this many polls, log loudly and force-reap. _WS_ORPHAN_INTERRUPT_REAP_MAX_POLLS = 60 _TURN_SETTLE_BEFORE_CLOSE_SECONDS = 5.0 _DETAIL_SECTION_NAMES = ("thinking", "tools", "subagents", "activity") _DETAIL_MODES = frozenset({"hidden", "collapsed", "expanded"}) -# ── Async RPC dispatch (#12546) ────────────────────────────────────── -# A handful of handlers block the dispatcher loop in entry.py for seconds -# to minutes (slash.exec, cli.exec, shell.exec, session.resume, -# session.branch, session.compress, skills.manage). While they're running, inbound RPCs — -# notably approval.respond and session.interrupt — sit unread in the -# stdin pipe. We route only those slow handlers onto a small thread pool; -# everything else stays on the main thread so ordering stays sane for the -# fast path. write_json is already _stdout_lock-guarded, so concurrent -# response writes are safe. +# ── Async RPC dispatch ─────────────────────────────────────────────── +# Slow handlers (seconds to minutes) would leave approval.respond and +# session.interrupt unread in the stdin pipe; only those go to a small thread +# pool, everything else stays inline so fast-path ordering stays sane. +# write_json is _stdout_lock-guarded, so concurrent response writes are safe. _LONG_HANDLERS = frozenset( { # Billing/usage reads each do a blocking portal HTTP fetch (state + usage @@ -276,27 +222,17 @@ _LONG_HANDLERS = frozenset( "billing.step_up", "browser.manage", "cli.exec", - # Completion RPCs run inline on the reader thread by default, but both - # can block it for seconds: complete.path spawns `git ls-files` and - # fuzzy-ranks the whole repo (slow on large repos / WSL2 mounts), and - # complete.slash does first-call prompt_toolkit imports + a skill-dir - # scan. While either runs inline, prompt.submit / session.interrupt sit - # unread in the stdin pipe — the TUI appears frozen until the 120s RPC - # timeout fires (#21123). Routing them to the pool keeps the fast path - # responsive; completion is read-only and write_json is lock-guarded. + # complete.path spawns `git ls-files` + fuzzy-ranks the repo; + # complete.slash does first-call prompt_toolkit imports + a skill scan. + # Inline either freezes the TUI until the 120s RPC timeout. "complete.path", "complete.slash", "llm.oneshot", - # model.options builds the full picker payload — per-provider credential - # pool checks, pricing fetch, Nous tier check, optional custom-provider - # probe — measured seconds inline. While it runs on the reader thread, - # prompt.submit / session.interrupt sit unread (same class as #21123), - # and the Desktop model pill / picker block on it every open. + # model.options: credential pool checks, pricing fetch, tier check, + # provider probe — seconds inline, blocking the picker on every open. "model.options", - # Pet RPCs hit the network (manifest fetch / spritesheet download) or do - # per-frame PNG decode/encode (pet.cells): inline they serialize on the - # reader thread, so picker previews trickle in one at a time and the - # animation poll stutters. On the pool they run concurrently. + # Pet RPCs hit the network or decode PNG frames; inline they serialize + # on the reader thread and the animation poll stutters. "pet.cells", "pet.gallery", # Generation is the heaviest pet path by far — multiple image-model @@ -308,34 +244,24 @@ _LONG_HANDLERS = frozenset( "pet.thumb", "learning.frames", "plugins.manage", - # reload.mcp shuts down and rediscovers every MCP server — with a - # flapping server (retry loops, connect timeouts up to 120s) that can - # block for minutes. Inline it froze the reader thread: config.set, - # complete.slash, prompt.submit all sat unread and the TUI appeared - # dead after a few skin switches. The handler serializes concurrent - # reloads via _mcp_reload_lock. + # reload.mcp shuts down and rediscovers every server (minutes with a + # flapping one); concurrent reloads serialize via _mcp_reload_lock. "reload.mcp", - # MCP server test/OAuth RPCs block on network: a probe spawns a stdio - # server (cold `npx` cold start = many seconds) or connects to a remote - # endpoint; oauth.start blocks up to ~30s waiting for the provider to - # publish an authorization URL. Keep them off the reader thread. + # MCP test/OAuth RPCs block on network (cold npx spawn; oauth.start + # waits up to ~30s for an authorization URL). "mcp.servers.test", "mcp.servers.oauth.start", "process.list", - # profiles.list runs list_profiles() (recursive skill-tree walk per - # profile) and opens each profile's state.db for the last-session - # preview; profiles.create copies skill bundles. Both are seconds- - # scale on cold disks — keep them off the WS reader thread. + # profiles.list walks every profile's skill tree + opens its state.db; + # profiles.create copies skill bundles — seconds on cold disks. "profiles.configure", "profiles.create", "profiles.describe", "profiles.get_asset", "profiles.list", "profiles.set_asset", - # Bot-relay RPCs: roster.sync/outbox.drain/reply are cheap file I/O, - # but bot_relay.deliver runs a FULL one-turn agent conversation - # (subprocess, up to 600s) — all four stay off the WS reader thread - # so a slow relay delivery can never block prompt.submit. + # bot_relay.deliver runs a FULL one-turn agent conversation (up to + # 600s); all four stay off the reader thread together. "bot_relay.roster.sync", "bot_relay.outbox.drain", "bot_relay.deliver", @@ -347,42 +273,22 @@ _LONG_HANDLERS = frozenset( "projects.for_cwd", "projects.tree", "projects.project_sessions", - # Setup readiness RPCs are polled by the Desktop frontend on connect - # and periodically (use-status-snapshot → evaluateRuntimeReadiness). - # setup.runtime_check calls resolve_runtime_provider() which reads - # config, checks auth state, and may probe the provider endpoint; - # setup.status calls _has_any_provider_configured() which scans - # provider config + credential files. Under GIL pressure from - # concurrent agent turns, either can take seconds inline, blocking - # the WS read loop and causing false "needs setup" (#50005 family). + # Setup readiness RPCs (polled by the Desktop) may probe the provider + # endpoint / scan credential files; under GIL pressure they block the WS + # read loop and cause false "needs setup". "setup.runtime_check", "setup.status", - # Voice RPCs can trigger check_voice_requirements() → STT provider - # auto-detect → a SYNCHRONOUS faster-whisper lazy install (uv/pip - # subprocess with a 300s timeout). Inline they stall the WS reader - # loop (handle_ws awaits dispatch before reading the next frame), so - # prompt.submit / session.list queued behind a voice.toggle sit - # unread and the desktop "send message" appears dead for minutes - # (reproduced: voice.toggle → session.list 40s+ timeout). Route them - # to the pool so a slow lazy install can't block message handling. + # Voice RPCs can trigger a SYNCHRONOUS faster-whisper lazy install + # (300s subprocess); inline that leaves prompt.submit unread for minutes. "voice.toggle", "voice.record", "voice.tts", - # wake.start calls check_wake_word_requirements() → _stt_ready() → - # _get_provider() → _try_lazy_install_stt() → ensure("stt.faster_whisper") - # (same synchronous subprocess install chain as the voice RPCs above). - # It also calls start_listening() → _build_engine() whose constructors - # call lazy_deps.ensure("wake.openwakeword" / "wake.sherpa" / …). - # wake.status calls check_wake_word_requirements() too and is polled - # by the desktop on every gateway-ready, so it can re-trigger the - # same block on a fresh launch. Same bug class as #21123 / #50005. + # wake.* hit the same synchronous STT install chain plus lazy_deps for + # the wake engine; wake.status is polled on every gateway-ready. "wake.start", "wake.status", - # Desktop also polls the in-memory live-session registry every 15s. - # The handler is normally cheap, but under heavy agent GIL pressure it - # can still stall for tens of seconds. Keep it off the WS reader thread - # so a delayed status rehydrate cannot block runtime readiness, prompt - # submission, or interrupts queued behind it on the same socket. + # Polled every 15s by the Desktop; cheap normally, but under GIL + # pressure it can stall and block interrupts queued behind it. "session.active_list", "session.branch", "session.compress", @@ -399,8 +305,7 @@ _LONG_HANDLERS = frozenset( _rpc_pool_workers = max(2, env_int("HERMES_TUI_RPC_POOL_WORKERS", 8)) _pool = concurrent.futures.ThreadPoolExecutor( - max_workers=_rpc_pool_workers, - thread_name_prefix="tui-rpc", + max_workers=_rpc_pool_workers, thread_name_prefix="tui-rpc", ) atexit.register(lambda: _pool.shutdown(wait=False, cancel_futures=True)) @@ -410,11 +315,8 @@ _current_runtime_session_record: contextvars.ContextVar[dict | None] = ( contextvars.ContextVar("hermes_gateway_runtime_session_record", default=None) ) -# JSON-RPC method being dispatched on this thread/task. Purely diagnostic: the -# 4001 "session not found" warning below is the only signal a stale-runtime -# retry loop leaves behind, and without the method name it cannot say WHICH -# client poll is looping (a 5s `process.list` poll produced 18,614 rejections -# against one id before the caller could be identified). Never used for +# JSON-RPC method being dispatched on this thread/task. Diagnostic only (names +# WHICH client poll is looping in the 4001 warning); never used for # authorization — the method string is client-supplied. _current_rpc_method: contextvars.ContextVar[str] = contextvars.ContextVar( "hermes_gateway_rpc_method", default="" @@ -456,12 +358,9 @@ def _prepend_tool_paths(env: dict[str, str]) -> dict[str, str]: Desktop/Dashboard app). Managed bin leads, matching the managed-first resolution policy for the Browser Use CLI.""" managed_bin = "" - try: + with contextlib.suppress(Exception): from hermes_constants import get_hermes_home - managed_bin = str(Path(get_hermes_home()) / "bin") - except Exception: - pass venv_bin = str(Path(sys.executable).parent) # /bin (POSIX) or /Scripts (Windows) user_bin = str(Path.home() / ".local" / "bin") existing = env.get("PATH") or "" @@ -480,27 +379,16 @@ class _SlashWorker: self._seq = 0 self.stderr_tail: list[str] = [] self.stdout_queue: queue.Queue[dict | None] = queue.Queue() - - argv = [ - sys.executable, - "-m", - "tui_gateway.slash_worker", - "--session-key", - session_key, - ] + argv = [sys.executable, "-m", "tui_gateway.slash_worker", "--session-key", session_key] if model: argv += ["--model", model] - self._closed = False from hermes_cli._subprocess_compat import windows_hide_flags - # slash_worker runs the Hermes agent → needs provider credentials. - # Tier-1 secrets (gateway/GitHub/infra) are still stripped (#29157). - # Global-remote / multi-profile sessions: the worker must resolve - # config/skills/state against the session's profile home, not the - # gateway's launch HERMES_HOME (#40677). The override goes through the - # build_subprocess_env factory's `extra` (applied last, always wins) - # instead of a hand-rolled env["HERMES_HOME"] assignment. + # The worker runs the agent → needs provider credentials; tier-1 secrets + # (gateway/GitHub/infra) are still stripped. Multi-profile sessions must + # resolve against the session's profile home, via the factory's `extra` + # (applied last, always wins). from tools.environments.local import build_subprocess_env env = build_subprocess_env( hermes_subprocess_env(inherit_credentials=True), @@ -508,30 +396,21 @@ class _SlashWorker: inherit_profile_home=False, # base already carries the HOME contract extra={"HERMES_HOME": str(profile_home)} if profile_home else None, ) - # Prepend the Hermes venv bin dir and the user-local bin dir to PATH so - # slash_worker child processes can resolve Hermes-managed CLIs - # (browser-use, uvx) even when the parent gateway was launched with a - # minimal PATH (e.g. by the Desktop/Dashboard app). See #83845. + # Hermes venv/user-local bin on PATH so worker children resolve + # Hermes-managed CLIs under the Desktop's minimal PATH. env = _prepend_tool_paths(env) - # start_new_session=True detaches the slash worker into its own - # process group / session. Without this, the worker inherits the - # gateway's pgid (= TUI parent PID). When mcp_tool's - # _kill_orphaned_mcp_children races with slash_worker spawn and sweeps - # the gateway's child set, it captures the worker PID, records the - # inherited pgid, and killpg() then kills the TUI parent itself. - # See agent/lsp/client.py for the symmetric LSP server fix and - # tools/mcp_tool.py _filter_mcp_children for defense-in-depth. + # start_new_session=True: otherwise the worker inherits the gateway's + # pgid and mcp_tool's orphan sweep, racing the spawn, killpg()s the TUI + # parent itself (see agent/lsp/client.py, mcp_tool._filter_mcp_children). self.proc = subprocess.Popen( argv, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, - # Force UTF-8 with lossy decoding so child output containing bytes - # that are invalid in the system locale (e.g. GBK on Chinese - # Windows) can't raise UnicodeDecodeError inside the drain threads - # and crash the gateway. See #53137. + # Lossy UTF-8: bytes invalid in the system locale (GBK Windows) + # must not raise UnicodeDecodeError in the drain threads. encoding="utf-8", errors="replace", bufsize=1, @@ -559,13 +438,11 @@ class _SlashWorker: def run(self, command: str) -> str: if self.proc.poll() is not None: raise RuntimeError("slash worker exited") - with self._lock: self._seq += 1 rid = self._seq self.proc.stdin.write(json.dumps({"id": rid, "command": command}) + "\n") self.proc.stdin.flush() - while True: try: msg = self.stdout_queue.get(timeout=_SLASH_WORKER_TIMEOUT_S) @@ -578,7 +455,6 @@ class _SlashWorker: if not msg.get("ok"): raise RuntimeError(msg.get("error", "slash worker failed")) return str(msg.get("output", "")).rstrip() - raise RuntimeError( f"slash worker closed pipe{': ' + chr(10).join(self.stderr_tail[-8:]) if self.stderr_tail else ''}" ) @@ -595,29 +471,26 @@ class _SlashWorker: proc.wait(timeout=1) except Exception: proc.kill() - try: + with contextlib.suppress(Exception): proc.wait(timeout=1) # reap the zombie SIGKILL leaves behind - except Exception: - pass except Exception: - try: + with contextlib.suppress(Exception): proc.kill() proc.wait(timeout=1) - except Exception: - pass finally: for stream in (proc.stdin, proc.stdout, proc.stderr): - try: + with contextlib.suppress(Exception): stream.close() - except Exception: - pass + + +def _display_cfg() -> dict: + """``display`` section of the behavioral config, or ``{}`` when absent/malformed.""" + display = _load_cfg().get("display") + return display if isinstance(display, dict) else {} def _load_busy_input_mode() -> str: - display = _load_cfg().get("display") - if not isinstance(display, dict): - display = {} - raw = str(display.get("busy_input_mode", "") or "").strip().lower() + raw = str(_display_cfg().get("busy_input_mode", "") or "").strip().lower() return raw if raw in {"queue", "steer", "interrupt"} else "interrupt" @@ -629,27 +502,17 @@ def _load_interim_assistant_messages() -> bool: interim text from tool-call turns and verify-on-stop candidates is never emitted as ``message.interim`` — mirroring the messaging gateway's gating. """ - display = _load_cfg().get("display") - if not isinstance(display, dict): - return True - return is_truthy_value(display.get("interim_assistant_messages", True)) - + return is_truthy_value(_display_cfg().get("interim_assistant_messages", True)) def _shutdown_sessions() -> None: - # Durable-first (#94724 item 2): persist every session's un-flushed - # transcript within a bounded budget BEFORE the slow per-session - # teardown below (plugin hooks, memory commit, delegation interrupts, - # agent.close). A supervisor that SIGKILLs a slow shutdown mid-way can - # then no longer lose the transcripts — the flush already landed. - try: + # Durable-first: flush every un-flushed transcript (bounded budget) BEFORE + # the slow per-session teardown, so a supervisor SIGKILL mid-shutdown + # can no longer lose them. + with contextlib.suppress(Exception): _flush_sessions_before_exit() - except Exception: - pass - try: + with contextlib.suppress(Exception): _release_gateway_wake_owner() - except Exception: - pass with _sessions_lock: sids = list(_sessions) for sid in sids: @@ -672,11 +535,8 @@ def _start_idle_reaper() -> None: def _loop(): while True: time.sleep(_REAPER_SCAN_S) - try: + with contextlib.suppress(Exception): _reap_idle_sessions() - except Exception: - pass - threading.Thread(target=_loop, daemon=True).start() @@ -691,15 +551,13 @@ def _get_db(): global _db, _db_error if _db is None: from hermes_state import get_shared_session_db - try: _db = get_shared_session_db() _db_error = None except Exception as exc: _db_error = str(exc) logger.warning( - "TUI session store unavailable — continuing without state.db features: %s", - exc, + "TUI session store unavailable — continuing without state.db features: %s", exc, ) return None return _db @@ -720,14 +578,9 @@ def _db_for_profile(profile: str | None = None): return _get_db(), False try: from hermes_state import get_shared_session_db - return get_shared_session_db(Path(profile_home) / "state.db"), True except Exception as exc: - logger.warning( - "TUI profile session store unavailable for %s: %s", - profile, - exc, - ) + logger.warning("TUI profile session store unavailable for %s: %s", profile, exc) return None, False @@ -753,12 +606,9 @@ def _transfer_db_to_agent(agent, db) -> bool: try: if getattr(agent, "_session_db", None) is not db: return False - # Defense in depth (#91610): the shared launch handle must never - # transfer. Identity alone passes for it — a launch-profile agent IS - # holding that handle — and ownership would make session.close() tear - # down the process-wide database every other session shares. Refuse it - # explicitly even if a caller invokes the transfer incorrectly; the - # caller's own `owns_db` gate is the first line of defense. + # The shared launch handle must never transfer: identity alone passes + # for it, and ownership would let session.close() tear down the + # process-wide database every other session shares. if db is _get_db(): logger.warning( "Refused transfer of the shared launch SessionDB to a session " @@ -784,14 +634,11 @@ def _open_profile_session_db(profile_home): launch handle. """ from hermes_state import get_shared_session_db - db_path = Path(profile_home) / "state.db" try: return get_shared_session_db(db_path) except Exception as exc: - raise RuntimeError( - f"profile session store unavailable: {db_path}: {exc}" - ) from exc + raise RuntimeError(f"profile session store unavailable: {db_path}: {exc}") from exc @contextlib.contextmanager @@ -831,13 +678,10 @@ def _db_unavailable_error(rid, *, code: int): # ── per-session profile scoping (global remote mode) ─────────────────────────── -# One dashboard normally serves its launch profile. But the desktop's app-global -# remote mode points every profile at this single backend, so resume/prompt must -# be able to act on ANOTHER local profile's state.db + home. The desktop passes -# ``profile`` on those calls; we open that profile's db and bind its HERMES_HOME -# (a ContextVar override) for the duration of the call so config/skills/model and -# message persistence all resolve to the right profile. Omitted/own profile → the -# launch profile (unchanged for single-profile and per-profile-remote setups). +# The desktop's app-global remote mode points every profile at this backend, so +# calls carry ``profile``: open that profile's db and bind its HERMES_HOME +# (ContextVar override) for the call so config/skills/model/persistence resolve +# to it. Omitted/own profile → the launch profile. def _profile_home(profile: str | None) -> Path | None: """Resolve a named profile's home on THIS host, or None for the launch profile.""" name = (profile or "").strip() @@ -845,7 +689,6 @@ def _profile_home(profile: str | None) -> Path | None: return None try: from hermes_cli import profiles as profiles_mod - home = Path(profiles_mod.get_profile_dir(name)) except Exception: return None @@ -884,7 +727,6 @@ def _profile_scoped(handler): return handler(rid, params) finally: reset_hermes_home_override(token) - return wrapper @@ -925,14 +767,11 @@ def _profile_configured_cwd(profile_home: Path | None) -> str | None: return None try: from hermes_cli.config import _expand_env_vars, read_user_config_raw - p = Path(profile_home) / "config.yaml" if not p.exists(): return None - # Behavioral read of a NON-launch profile's config: load_config() - # would resolve the ACTIVE profile's path, so read this profile's - # file directly, then apply the same read-side pipeline as - # _load_cfg (managed overlay + ${VAR} expansion). Fail-open. + # load_config() resolves the ACTIVE profile, so read this profile's + # file directly + the same read-side pipeline as _load_cfg. Fail-open. data = _apply_managed(read_user_config_raw(p)) expanded = _expand_env_vars(data) if isinstance(expanded, dict): @@ -993,12 +832,9 @@ def write_json(obj: dict) -> bool: sid = ((params or {}).get("session_id")) if isinstance(params, dict) else "" if sid and (t := (_sessions.get(sid) or {}).get("transport")) is not None: from tui_gateway.event_replay import _stamp_event - _stamp_event(obj) return t.write(obj) - from tui_gateway.event_replay import _stamp_event - _stamp_event(obj) return (current_transport() or _stdio_transport).write(obj) @@ -1014,10 +850,9 @@ def _emit(event: str, sid: str, payload: dict | None = None): write_json(_event_frame(event, sid, payload)) -# Live client transports, one per connected WS peer (maintained by tui_gateway.ws). -# A session-less event from a background thread has neither a session transport -# nor a contextvar binding, so write_json would drop it on stdio — this registry -# is how such events reach WS clients at all. See _broadcast_global_event. +# Live WS peer transports (maintained by tui_gateway.ws): the only route for +# session-less background events, which write_json would otherwise drop on +# stdio. See _broadcast_global_event. _live_transports: set[Transport] = set() _live_transports_lock = threading.Lock() @@ -1045,11 +880,9 @@ def _broadcast_global_event(event: str, payload: dict | None = None) -> None: """ with _live_transports_lock: targets = list(_live_transports) - if not targets: _emit(event, "", payload) return - frame = _event_frame(event, "", payload) for transport in targets: try: @@ -1076,7 +909,6 @@ def _approval_request_payload(data: dict | None) -> dict: payload["choices"] = choices if "command" in payload: from gateway.run import _redact_approval_command - payload["command"] = _redact_approval_command(payload.get("command")) return payload @@ -1118,7 +950,6 @@ def _pending_approval_request_payload(session_key: str) -> dict | None: """Read the oldest unresolved approval in a session, if there is one.""" try: from tools.approval import get_pending_gateway_approval - approval = get_pending_gateway_approval(session_key) except Exception: logger.debug("failed to read pending approval for %s", session_key, exc_info=True) @@ -1147,7 +978,6 @@ def _status_update(sid: str, kind: str, text: str | None = None): # otherwise idle/preflight compaction looks like a hung turn (#97239). if out_kind == "lifecycle": from agent.conversation_compression import is_compaction_progress_status - if is_compaction_progress_status(body): out_kind = "compacting" _emit("status.update", sid, {"kind": out_kind, "text": body}) @@ -1166,16 +996,13 @@ def _estimate_image_tokens(width: int, height: int) -> int: def _image_meta(path: Path) -> dict: meta = {"name": path.name} - try: + with contextlib.suppress(Exception): from PIL import Image - with Image.open(path) as img: width, height = img.size meta["width"] = int(width) meta["height"] = int(height) meta["token_estimate"] = _estimate_image_tokens(int(width), int(height)) - except Exception: - pass return meta @@ -1194,7 +1021,6 @@ def method(name: str): def dec(fn): _methods[name] = fn return fn - return dec @@ -1202,18 +1028,15 @@ def _normalize_request(req: Any) -> tuple[Any, str, dict] | dict: """Validate a JSON-RPC request enough for safe local dispatch.""" if not isinstance(req, dict): return _err(None, -32600, "invalid request: expected an object") - rid = req.get("id") method = req.get("method") if not isinstance(method, str) or not method: return _err(rid, -32600, "invalid request: method must be a non-empty string") - params = req.get("params", {}) if params is None: params = {} elif not isinstance(params, dict): return _err(rid, -32602, "invalid params: expected an object") - return rid, method, params @@ -1221,7 +1044,6 @@ def handle_request(req: dict) -> dict | None: normalized = _normalize_request(req) if isinstance(normalized, dict): return normalized - rid, method, params = normalized fn = _methods.get(method) if not fn: @@ -1233,9 +1055,7 @@ def handle_request(req: dict) -> dict | None: _current_rpc_method.reset(token) -def _current_session_steer_authority( - session_id: str, -) -> tuple[Transport | None, dict | None]: +def _current_session_steer_authority(session_id: str) -> tuple[Transport | None, dict | None]: """Resolve unforgeable steering authority for this exact RPC context. The public session id is only a lookup hint. Authority is the identity of @@ -1276,7 +1096,6 @@ def dispatch(req: dict, transport: Optional[Transport] = None) -> dict | None: normalized = _normalize_request(req) if isinstance(normalized, dict): return normalized - _rid, method, _params = normalized if method not in _LONG_HANDLERS: return handle_request(req) @@ -1291,9 +1110,7 @@ def dispatch(req: dict, transport: Optional[Transport] = None) -> dict | None: resp = _err(req.get("id"), -32000, f"handler error: {exc}") if resp is not None: t.write(resp) - _pool.submit(lambda: ctx.run(run)) - return None finally: reset_transport(token) @@ -1319,43 +1136,29 @@ def _agent_build_wait_cap() -> float: build before failing permanently. ``agent.build_wait_timeout`` in config.yaml overrides the 600s default (raise it for deployments with many slow/unreachable MCP servers or high-latency provider metadata).""" - try: + with contextlib.suppress(Exception): agent_cfg = _load_cfg().get("agent") or {} raw = agent_cfg.get("build_wait_timeout") if raw is not None: value = float(raw) if value > 0: return value - except Exception: - pass return 600.0 def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: """Patient variant of ``_wait_agent`` for the deferred prompt.submit path. - The flat 30s ``_wait_agent`` ceiling was a message-eating cliff (#63078): - ``prompt.submit`` has already returned ``{"status": "streaming"}``, the - user's first message IS the turn in flight, and the deferred agent build - (MCP discovery with per-server retry backoff, synchronous model-metadata - HTTP, skills scanning) routinely outlives 30 seconds on cold starts. On - timeout the old path emitted an error EVENT and returned without ever - calling ``_run_prompt_submit`` — the first message was permanently - discarded while the build finished successfully in the background, leaving - the blank first session. - - This wait instead: - - keeps the pending prompt attached to this (already off-RPC) thread and - delivers it the moment the still-running build completes; - - waits in short slices so a cancel (session.interrupt / session churn) - is honored promptly instead of after the full timeout; - - tells the client once, via a keyed notice, when the build outlives - ``_AGENT_BUILD_SLOW_NOTICE_AFTER`` — the wait is patient but never - silent; - - fails permanently only when the build itself fails: the build thread - died without signalling ready, or the bounded cap - (``agent.build_wait_timeout``, default 600s — no infinite waits) - expired on a genuinely hung build. + prompt.submit has already answered ``{"status": "streaming"}`` and the + user's first message IS the turn in flight, while a cold deferred build + (MCP discovery, model-metadata HTTP, skills scan) routinely outlives the + flat 30s ceiling — timing out here silently discarded that first message. + So: keep the prompt attached to this (off-RPC) thread and deliver it when + the build lands; wait in short slices so a cancel is honored promptly; + tell the client once (keyed notice) when the build outlives + ``_AGENT_BUILD_SLOW_NOTICE_AFTER``; fail only when the build thread died + without signalling ready or the bounded cap (``agent.build_wait_timeout``, + default 600s) expired on a genuinely hung build. Returns ``None`` on success OR when the turn was cancelled mid-wait (the caller's cancel branch owns that messaging), an ``_err`` dict otherwise. @@ -1368,9 +1171,7 @@ def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: notified_slow = False while not ready.wait(timeout=_AGENT_BUILD_WAIT_SLICE): with session["history_lock"]: - cancelled = session.get("_turn_cancel_requested") or not session.get( - "running" - ) + cancelled = session.get("_turn_cancel_requested") or not session.get("running") if cancelled: # The caller's cancel/not-running branch emits the user-visible # event for this — bail without an error of our own. @@ -1384,14 +1185,9 @@ def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: "your message was not sent; retry once the session is ready", ) build_thread = session.get("_agent_build_thread") - if ( - build_thread is not None - and not build_thread.is_alive() - and not ready.is_set() - ): - # _build's ``finally`` guarantees ready.set(); a dead thread with - # ready still unset means the build died hard (interpreter-level - # kill) — don't wait on a corpse for the rest of the cap. + if (build_thread is not None and not build_thread.is_alive() and not ready.is_set()): + # _build's finally guarantees ready.set(); dead thread + unset + # ready = the build died hard — don't wait on a corpse. return _err( rid, 5032, @@ -1399,9 +1195,7 @@ def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: or "agent initialization failed before completing", ) if not notified_slow and waited >= _AGENT_BUILD_SLOW_NOTICE_AFTER: - # One keyed, replace-in-place notice: the desktop shows it as a - # toast, the TUI in its status bar. Without this the extended wait - # would be exactly the silent hang this function exists to fix. + # One keyed, replace-in-place notice (toast / status bar). notified_slow = True _emit( "notification.show", @@ -1425,25 +1219,129 @@ def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: return _err(rid, 5032, err) if err else None +def _bind_build_profile_scopes(profile_home: str) -> "_TurnScopes": + """Bind a session profile's HERMES_HOME / secret / terminal scopes for an agent build. + + Fail-open per scope (the build must not die on a scope helper), except that + the terminal scope installer itself fails closed (malformed policy → + refusal scope) so _make_agent's terminal probing / cwd hints resolve the + routed profile, never the launch process. + """ + scopes = _TurnScopes() + scopes.home = set_hermes_home_override(profile_home) + with contextlib.suppress(Exception): + from agent.secret_scope import build_profile_secret_scope, set_secret_scope + + scopes.secret = set_secret_scope(build_profile_secret_scope(Path(profile_home))) + try: + from tools.terminal_scope import install_profile_terminal_scope + + scopes.terminal = install_profile_terminal_scope(Path(profile_home)) + except Exception: + scopes.terminal = None + return scopes + + +def _release_build_profile_scopes(scopes: "_TurnScopes") -> None: + if scopes.home is not None: + reset_hermes_home_override(scopes.home) + if scopes.secret is not None: + with contextlib.suppress(Exception): + from agent.secret_scope import reset_secret_scope + + reset_secret_scope(scopes.secret) + if scopes.terminal is not None: + with contextlib.suppress(Exception): + from tools.terminal_scope import reset_terminal_scope + + reset_terminal_scope(scopes.terminal) + + +def _deferred_build_agent_kwargs(current: dict, session_db) -> dict: + """_make_agent kwargs for a deferred (first-prompt) build. + + A lazy-resumed (watch) session carries the stored conversation id so the + upgrade continues that session instead of starting a fresh one under the + same key. A cold deferred resume restores the full persisted runtime + identity exactly as the eager resume path's _stored_session_runtime_overrides + splat did, so a deferred build can't drop the provider and fail with "No LLM + provider configured". When there is no stored runtime, or its provider no + longer resolves (renamed/removed), fall back to the model/effort/tier the + desktop picked for THIS session, else the configured default — never sink + agent init with "Unknown provider". + """ + kw = { + "session_db": session_db, + "context_cwd_is_launch_artifact": _context_cwd_is_launch_artifact(current), + } + if resume_sid := current.get("resume_session_id"): + kw["session_id"] = resume_sid + kw["platform_override"] = _session_source(current) + resume_overrides = current.get("resume_runtime_overrides") + if ( + isinstance(resume_overrides, dict) + and resume_overrides + and _overrides_have_routable_provider(resume_overrides) + ): + kw.update(resume_overrides) + else: + if override := current.get("model_override"): + kw["model_override"] = override + if (reasoning := current.get("create_reasoning_override")) is not None: + kw["reasoning_config_override"] = reasoning + if (tier := current.get("create_service_tier_override")) is not None: + kw["service_tier_override"] = tier + return kw + + +def _wire_session_agent(sid: str, key: str, agent) -> bool: + """Common post-build wiring for a session agent; returns whether notify registered. + + Approval prompts route to the client; the self-improvement review's "💾 …" + summary is emitted as review.summary so the TUI/desktop render it in the + transcript (the CLI prints it via prompt_toolkit; the TUI has no print + surface), honoring display.memory_notifications like the gateway and CLI. + """ + notify_registered = False + with contextlib.suppress(Exception): + from tools.approval import load_permanent_allowlist, register_gateway_notify + + register_gateway_notify(key, lambda data: _emit_approval_request(sid, data)) + notify_registered = True + load_permanent_allowlist() + _wire_callbacks(sid) + try: + agent.background_review_callback = lambda message, _sid=sid: _emit( + "review.summary", _sid, {"text": str(message)} + ) + agent.memory_notifications = _load_memory_notifications() + except Exception: + pass # bare agents without the attribute must not break startup + return notify_registered + + +def _start_session_services(sid: str, key: str, current: dict) -> None: + """Start the notification poller and fire the session-reset boundary hook.""" + with _sessions_lock: + if sid in _sessions: + _sessions[sid]["_notif_stop"] = _start_notification_poller(sid, _sessions[sid]) + _notify_session_boundary("on_session_reset", key, _session_source(current)) + + def _start_agent_build(sid: str, session: dict) -> None: """Start building the real AIAgent for a TUI session, once. - Classic `hermes` shows the prompt before constructing AIAgent; the TUI used - to eagerly build it during session.create, making startup feel blocked on - tool discovery/model metadata even though the composer was visible. Keep - the shell responsive by deferring this work until the first prompt (or any - command that actually needs the agent), while retaining the same ready/error - event contract for the frontend. + Deferred until the first prompt (or any command that needs the agent) so + the composer is responsive instead of blocked on tool discovery / model + metadata; the ready/error event contract for the frontend is unchanged. """ ready = session.get("agent_ready") if ready is None: return # A lazy watch session spectating an in-flight child must stay lazy so the - # subagent live-mirror keeps flowing. Incidental RPCs (session.info, model - # metadata, etc.) resolve through _sess(), which would otherwise upgrade it - # to a full agent mid-stream and silently kill the mirror (the mirror bails - # once agent is set). Once the child completes, the guard lifts and the next - # prompt/RPC builds the agent normally so the user can talk to the session. + # subagent live-mirror keeps flowing (the mirror bails once agent is set); + # incidental RPCs resolve through _sess() and would otherwise upgrade it + # mid-stream. Once the child completes the guard lifts. if session.get("lazy") and _child_run_active(str(session.get("session_key") or "")): return lock = session.setdefault("agent_build_lock", threading.Lock()) @@ -1464,11 +1362,8 @@ def _start_agent_build(sid: str, session: dict) -> None: return notify_registered = False - home_token = None - secret_token = None - build_terminal_token = None + scopes = None session_db = None - owns_db = False profile_home = current.get("profile_home") try: history_ready = current.get("resume_history_ready") @@ -1484,40 +1379,13 @@ def _start_agent_build(sid: str, session: dict) -> None: # Build against the session's profile (global-remote): bind its # HERMES_HOME so config/skills/model resolve to it, and hand the # agent that profile's db so turns persist to the right state.db. - session_db = None if profile_home: - home_token = set_hermes_home_override(profile_home) - try: - from agent.secret_scope import build_profile_secret_scope, set_secret_scope - - secret_token = set_secret_scope(build_profile_secret_scope(Path(profile_home))) - except Exception: - pass - # Bind the profile's COMPLETE terminal policy for the agent - # build (fail-closed: malformed policy → refusal scope) so - # _make_agent's terminal probing / cwd hints resolve the - # routed profile, never the launch process (#98581 class). - try: - from tools.terminal_scope import ( - install_profile_terminal_scope, - reset_terminal_scope, - ) - - build_terminal_token = install_profile_terminal_scope( - Path(profile_home) - ) - except Exception: - build_terminal_token = None - # DEDICATED handle — ours until _transfer_db_to_agent hands - # it to the built agent in the finally below. Every path - # that leaves this build without that transfer (the except - # below, and a session reaped mid-build) must close it. - # FAIL CLOSED on open failure: the raise routes to the - # ``except`` below (clear agent_error, no agent turn) instead - # of silently binding _make_agent's launch-DB default and - # bleeding this session into the wrong profile's state.db. + scopes = _bind_build_profile_scopes(profile_home) + # DEDICATED handle, ours until _transfer_db_to_agent in the + # finally; any non-transfer exit must close it. FAIL CLOSED on + # open failure (routes to the except) rather than binding the + # launch DB and bleeding rows into the wrong profile's state.db. session_db = _open_profile_session_db(profile_home) - owns_db = True try: from tui_gateway.entry import ensure_mcp_discovery_started @@ -1527,113 +1395,34 @@ def _start_agent_build(sid: str, session: dict) -> None: logger.warning("MCP discovery startup failed", exc_info=True) try: - # Lazy-resumed (watch) sessions carry the stored conversation - # id — pass it through so the upgrade continues that session - # instead of starting a fresh one under the same key. - kw = { - "session_db": session_db, - "context_cwd_is_launch_artifact": ( - _context_cwd_is_launch_artifact(current) - ), - } - if resume_sid := current.get("resume_session_id"): - kw["session_id"] = resume_sid - kw["platform_override"] = _session_source(current) - resume_overrides = current.get("resume_runtime_overrides") - if ( - isinstance(resume_overrides, dict) - and resume_overrides - and _overrides_have_routable_provider(resume_overrides) - ): - # Cold deferred resume: restore the full persisted runtime - # identity (model/provider/base_url/api_mode/reasoning/tier) - # exactly as the eager resume path's _stored_session_runtime_ - # overrides splat did, so a deferred build can't drop the - # provider and fail with "No LLM provider configured". - kw.update(resume_overrides) - else: - # No stored runtime, or the stored provider no longer - # resolves (renamed/removed since the row was written) — - # never let that sink agent init with "Unknown provider". - # Fall back to the model/effort/fast the desktop picked - # for THIS session, else the configured default. - if override := current.get("model_override"): - kw["model_override"] = override - if (reasoning := current.get("create_reasoning_override")) is not None: - kw["reasoning_config_override"] = reasoning - if (tier := current.get("create_service_tier_override")) is not None: - kw["service_tier_override"] = tier - agent = _make_agent(sid, key, **kw) + agent = _make_agent(sid, key, **_deferred_build_agent_kwargs(current, session_db)) finally: _clear_session_context(tokens) - # Bot Mode gate hint: the DB title lands post-first-turn - # (pending_title), but the system prompt builds at turn START — - # hand the agent its intended title so the "Bot Chat" protocol - # gate (agent/system_prompt.py) doesn't depend on write order. + # Bot Mode gate hint: the DB title lands post-first-turn but the + # system prompt builds at turn START, so hand the agent its title. _title_hint = str(current.get("pending_title") or "").strip() if _title_hint: agent._session_title_hint = _title_hint # Session DB row deferred to first run_conversation() call. - # pending_title applied post-first-message (see cli.exec handler). current["agent"] = agent _session_todo_state(current) - # Baseline for the per-turn config sync; the profile home - # override is still active here. + # Baseline for the per-turn config sync (profile home override still active). current["config_model_seen"] = _config_model_target() - # No eager slash-worker pre-warm: slash.exec spawns one on demand - # (its error path already relies on that respawn to recover from a - # dead worker). Each worker child runs its own MCP discovery - # (#61891), so pre-warming one per session forks the full stdio - # MCP fleet — ~20 OS processes per retained session on a config - # with a few stdio servers — even for sessions that never run a - # worker-routed command. Sessions held by a live transport are - # never reaped, so with the desktop app open for days those - # fleets accumulate until the OS refuses new process spawns. - - try: - from tools.approval import ( - register_gateway_notify, - load_permanent_allowlist, - ) - - register_gateway_notify( - key, lambda data: _emit_approval_request(sid, data) - ) - notify_registered = True - load_permanent_allowlist() - except Exception: - pass - - _wire_callbacks(sid) - # Surface the self-improvement review's "💾 …" summary as an event - # the TUI/desktop render in-transcript, honoring - # display.memory_notifications. _init_session wires this for the - # eager/branch paths; deferred-built sessions (session.create and the - # default cold resume) build through here, so without this their - # review summaries would leak to stdout instead of the chat. - try: - agent.background_review_callback = lambda message, _sid=sid: _emit( - "review.summary", _sid, {"text": str(message)} - ) - agent.memory_notifications = _load_memory_notifications() - except Exception: - pass - # Hydrate credits notices at session OPEN (not just on the first - # message), so depletion / usage-band warnings show at "ready". Runs - # off the build thread, after the notice_callback is wired. Fail-open. - try: + # No eager slash-worker pre-warm (slash.exec spawns on demand): + # each worker forks the full stdio MCP fleet (~20 processes), and + # live-transport sessions are never reaped, so fleets accumulate + # until the OS refuses new spawns. + notify_registered = _wire_session_agent(sid, key, agent) + # Credits notices at session OPEN so depletion / usage-band + # warnings show at "ready"; after notice_callback is wired. Fail-open. + with contextlib.suppress(Exception): from agent.credits_tracker import seed_credits_at_session_start seed_credits_at_session_start(agent) - except Exception: - pass - with _sessions_lock: - if sid in _sessions: - _sessions[sid]["_notif_stop"] = _start_notification_poller(sid, _sessions[sid]) - _notify_session_boundary("on_session_reset", key, _session_source(current)) + _start_session_services(sid, key, current) info = _session_info(agent, current) cfg_warn = _probe_config_health(_load_cfg()) @@ -1641,51 +1430,31 @@ def _start_agent_build(sid: str, session: dict) -> None: info["config_warning"] = cfg_warn logger.warning(cfg_warn) _emit("session.info", sid, info) - # If MCP discovery is still in flight (a server slower than the - # bounded wait_for_mcp_discovery join in _make_agent), the agent - # was built without those tools. Catch up once they land — see - # _schedule_mcp_late_refresh. Cache-safe (pre-first-turn only). + # MCP servers slower than _make_agent's bounded discovery wait are + # missing from the agent's tool list; catch up once they land + # (cache-safe: pre-first-turn only). _schedule_mcp_late_refresh(sid, agent) except Exception as e: current["agent_error"] = str(e) _emit("error", sid, {"message": f"agent init failed: {e}"}) finally: - if home_token is not None: - reset_hermes_home_override(home_token) - if secret_token is not None: - try: - from agent.secret_scope import reset_secret_scope - - reset_secret_scope(secret_token) - except Exception: - pass - if build_terminal_token is not None: - try: - from tools.terminal_scope import reset_terminal_scope - - reset_terminal_scope(build_terminal_token) - except Exception: - pass + if scopes is not None: + _release_build_profile_scopes(scopes) # _attach_worker already closed the worker if this session was # reaped mid-build; only the late notify registration can still # leak (session.close unregistered before _build registered it). with _sessions_lock: replaced = _sessions.get(sid) is not current if replaced and notify_registered: - try: + with contextlib.suppress(Exception): from tools.approval import unregister_gateway_notify unregister_gateway_notify(key) - except Exception: - pass # Dedicated profile handle: hand it to the agent that will actually - # be torn down, or close it here when no such agent exists. Both - # non-transfer cases are real: the except above (build raised, so - # nothing holds the handle) and `replaced` (the session was reaped - # mid-build, so this agent is discarded and _teardown_session will - # never reach it). Transferring to a discarded agent would leak the - # handle exactly as before. - if owns_db and session_db is not None: + # be torn down, or close it when no such agent exists — the except + # above (nothing holds it) and `replaced` (session reaped mid-build, + # this agent is discarded and _teardown_session never reaches it). + if session_db is not None: built = None if replaced else current.get("agent") if not _transfer_db_to_agent(built, session_db): with contextlib.suppress(Exception): @@ -1705,14 +1474,9 @@ def _sess_nowait(params, rid): s = _sessions.get(sid) if s: return (s, None) - # A session-scoped RPC hit a runtime id the gateway no longer holds - # (detached on WS disconnect and orphan-reaped, LRU-evicted, or torn down - # after an idle TTL). The client is expected to recover via - # session.resume on the STORED session id, but a plain stale-id send - # leaves no trace anywhere when the resume never fires — every RPC in - # this class returned a silent 4001. Log it so a "message vanished" - # report is diagnosable as "request arrived and was rejected" instead of - # "request never arrived" (see #90428). + # Stale runtime id (orphan-reaped, LRU-evicted, or idle-TTL torn down); the + # client should session.resume the STORED id. Log it so a "message vanished" + # report reads as "arrived and was rejected", not "never arrived". logger.warning( "session-scoped RPC rejected: method=%s session_id=%r not in memory " "(detached/reaped runtime; client should resume the stored session), rid=%r", @@ -1733,27 +1497,13 @@ def _sess(params, rid): def _sess_building(params, rid): """Resolve a session and warm its agent build WITHOUT waiting for it. - For handlers that need the session record but not the agent. The attach - RPCs are the whole reason this exists: ``image.attach``, - ``image.attach_bytes``, ``file.attach``, ``pdf.attach``, - ``clipboard.paste`` and ``image.detach`` only read ``cwd`` / - ``profile_home`` and mutate ``attached_images`` — every one of those - fields is populated when the session record is created, so ``_sess``'s - ``_wait_agent`` was buying nothing and charging up to 30 seconds for it. - - That charge landed in the worst possible place. Attach runs BEFORE - ``prompt.submit``, none of these methods is in ``_LONG_HANDLERS``, and a - non-pooled handler runs inline on the socket reader thread — so pasting an - image into a session whose deferred build was still running (MCP - discovery, model metadata, skills scan: routinely tens of seconds on a - cold start) stalled the send AND every RPC queued behind it on the same - socket, with no spinner to explain it. Plain text was unaffected because - ``prompt.submit`` already resolves via ``_sess_nowait`` and waits later, - off the reader thread — which is exactly why the bug reads as "text is - instant, images hang." - - The build is still kicked off (it warms the agent the following - ``prompt.submit`` needs); we simply stop blocking on it here. + For handlers that need the session record but not the agent — the attach + RPCs (image/file/pdf attach, clipboard.paste, image.detach) only read + ``cwd``/``profile_home`` and mutate ``attached_images``, all populated at + record creation. They run inline on the socket reader thread, so waiting + on a cold deferred build there stalled the paste AND every RPC queued + behind it ("text is instant, images hang"). The build is still kicked off + to warm the agent the following ``prompt.submit`` needs. """ s, err = _sess_nowait(params, rid) if err: @@ -1793,17 +1543,14 @@ def _load_dashboard_process_isolation_config(cfg: dict | None = None) -> dict[st dashboard = {} return { "turn_isolation": is_truthy_value( - dashboard.get("turn_isolation"), - default=_DASHBOARD_TURN_ISOLATION_DEFAULT, + dashboard.get("turn_isolation"), default=_DASHBOARD_TURN_ISOLATION_DEFAULT, ), "compute_host_heartbeat_secs": _coerce_int_config_value( dashboard.get("compute_host_heartbeat_secs"), - _DASHBOARD_COMPUTE_HOST_HEARTBEAT_SECS_DEFAULT, - min_value=1, + _DASHBOARD_COMPUTE_HOST_HEARTBEAT_SECS_DEFAULT, min_value=1, ), "compute_host_respawn_max": _coerce_int_config_value( - dashboard.get("compute_host_respawn_max"), - _DASHBOARD_COMPUTE_HOST_RESPAWN_MAX_DEFAULT, + dashboard.get("compute_host_respawn_max"), _DASHBOARD_COMPUTE_HOST_RESPAWN_MAX_DEFAULT, min_value=0, ), } @@ -1820,10 +1567,8 @@ def _load_cfg_raw() -> dict: """ global _cfg_cache, _cfg_mtime, _cfg_path try: - # Honor a per-session profile override (see session.resume) so a resumed - # remote profile loads ITS config (model, skills, prompt); otherwise the - # launch profile's _hermes_home. Cache is keyed on the resolved path, so - # profiles don't clobber each other. + # Per-session profile override (session.resume) → that profile's + # config; cache keyed on the resolved path so profiles don't clobber. override = get_hermes_home_override() home = override if isinstance(override, str) and override else _hermes_home p = Path(home) / "config.yaml" @@ -1837,10 +1582,8 @@ def _load_cfg_raw() -> dict: else: data = {} with _cfg_lock: - # Cache the RAW user config (no managed overlay) so _save_cfg, which - # writes _cfg_cache back to disk, never persists managed values into - # the user's file. The managed overlay is applied on every return - # path instead (read-side only). + # Cache the RAW config: _save_cfg writes _cfg_cache back to disk, + # so managed values must be overlaid read-side only. _cfg_cache = copy.deepcopy(data) _cfg_mtime = mtime _cfg_path = p @@ -1864,14 +1607,11 @@ def _load_cfg() -> dict: file. """ cfg = _apply_managed(_load_cfg_raw()) - try: + with contextlib.suppress(Exception): from hermes_cli.config import _expand_env_vars - expanded = _expand_env_vars(cfg) if isinstance(expanded, dict): cfg = expanded - except Exception: - pass return cfg @@ -1885,7 +1625,6 @@ def _apply_managed(cfg: dict) -> dict: """ try: from hermes_cli import managed_scope - return managed_scope.apply_managed_overlay(cfg if isinstance(cfg, dict) else {}) except Exception: return cfg @@ -1893,20 +1632,12 @@ def _apply_managed(cfg: dict) -> dict: def _save_cfg(cfg: dict): global _cfg_cache, _cfg_mtime, _cfg_path - from utils import atomic_roundtrip_yaml_save - override = get_hermes_home_override() - home = Path(override) if isinstance(override, str) and override else _hermes_home - path = Path(home) / "config.yaml" - # Comment-, ordering-, and Unicode-preserving full-state write. - # Replaces the previous `yaml.safe_dump(cfg, f)` (and later - # `atomic_config_write`, which is not comment-preserving) which clobbered - # the user's hand-written config every time we touched a single setting - # (top-level keys reordered alphabetically, comments dropped, kaomoji - # mangled to \\uXXXX escapes). Fails closed on an unreadable existing - # config.yaml the same way atomic_config_write does (see - # atomic_roundtrip_yaml_save's require_readable_config_before_write call). + path = Path(override if isinstance(override, str) and override else _hermes_home) / "config.yaml" + # Comment-, ordering-, and Unicode-preserving write (a plain safe_dump + # clobbered hand-written configs). Fails closed on an unreadable existing + # config.yaml like atomic_config_write does. atomic_roundtrip_yaml_save(path, cfg) with _cfg_lock: _cfg_cache = copy.deepcopy(cfg) @@ -1934,39 +1665,27 @@ def _cwd_for_session_key(session_key: str) -> str: def _set_session_context( - session_key: str, - cwd: str | None = None, - *, - ui_session_id: str = "", + session_key: str, cwd: str | None = None, *, ui_session_id: str = "", ) -> list: try: from gateway.session_context import set_session_vars - # Ephemeral task IDs (background, preview) aren't in `_sessions`, so the - # reverse-map returns "" and would clear the cwd override. Callers that - # know the parent workspace pass it explicitly so spawned agents inherit - # it instead of falling back to the gateway launch dir. + # Ephemeral task ids aren't in `_sessions` (reverse-map → "" would + # clear the cwd override); callers that know the workspace pass it. resolved = cwd if cwd is not None else _cwd_for_session_key(session_key) source = _resolve_session_platform() browser_control_principal = "" browser_control_transport_family = "" - # Derive the live conversation id so terminal/execute_code subprocesses - # can read HERMES_SESSION_ID. Without this, set_session_vars leaves the - # session-id contextvar as "" (explicitly empty), and the subprocess-env - # bridge treats that as authoritative — NOT falling back to os.environ — - # so every command in a dashboard/TUI/web session saw an empty - # HERMES_SESSION_ID even though agent_init set it via - # set_current_session_id(). Prefer the agent's durable session_id, then - # fall back to the session_key (matching the id derivation used at - # session-finalize), so an identified session is never left blank. + # Live conversation id for subprocess HERMES_SESSION_ID: an explicitly + # empty contextvar is authoritative for the subprocess-env bridge (no + # os.environ fallback), so never leave it "". Prefer the agent's durable + # session_id, then session_key (same derivation as session-finalize). session_id = session_key with _sessions_lock: for sess in list(_sessions.values()): if sess.get("session_key") == session_key: source = _session_source(sess) - session_id = ( - getattr(sess.get("agent"), "session_id", None) or session_key - ) + session_id = (getattr(sess.get("agent"), "session_id", None) or session_key) transport = sess.get("transport") identity = getattr(transport, "auth_identity", None) if _methods_browser_control._is_authenticated_identity(identity): @@ -1978,14 +1697,10 @@ def _set_session_context( ) break return set_session_vars( - session_key=session_key, - session_id=session_id, - source=source, + session_key=session_key, session_id=session_id, source=source, browser_control_principal=browser_control_principal, - browser_control_transport_family=browser_control_transport_family, - cwd=resolved, - ui_session_id=ui_session_id, - cron_session="", + browser_control_transport_family=browser_control_transport_family, cwd=resolved, + ui_session_id=ui_session_id, cron_session="", ) except Exception: return [] @@ -1994,12 +1709,9 @@ def _set_session_context( def _clear_session_context(tokens: list) -> None: if not tokens: return - try: + with contextlib.suppress(Exception): from gateway.session_context import clear_session_vars - clear_session_vars(tokens) - except Exception: - pass def _enable_gateway_prompts() -> None: @@ -2013,10 +1725,7 @@ def _enable_gateway_prompts() -> None: def _block( - event: str, - sid: str, - payload: dict, - timeout: float | None = 300, + event: str, sid: str, payload: dict, timeout: float | None = 300, batch_qids: list[str] | None = None, ) -> str: rid = uuid.uuid4().hex[:8] @@ -2049,7 +1758,6 @@ def _block( batch_state = _batch_clarify.pop(rid, None) if batch_state is not None: batch_answers = dict(batch_state["answers"]) - if batch_qids is not None: # Cancel-all (respond with no question_id) resolves via _answers with # an empty string — that stays a plain cancel, not a partial result. @@ -2061,36 +1769,18 @@ def _block( # are absences (not skips), and still fire the expire # notification so live cards tear down. result["timed_out"] = True - _emit( - f"{event.removesuffix('.request')}.expire", - sid, - {"request_id": rid}, - ) + _emit(f"{event.removesuffix('.request')}.expire", sid, {"request_id": rid}) return json.dumps(result, ensure_ascii=False) - # Emit an `.expire` notification on timeout for every blocking request type - # whose `*.respond` handler tolerates a late reply (allow_expired=True). - # All four blocking bridges — secret, sudo, clarify, terminal.read — share - # the same lifecycle: the tool gives up on timeout and returns empty, but a - # slow renderer (or a reconnect that dropped tool.complete) can still answer - # afterward. Without this the late `*.respond` would hit the generic 4009 - # "no pending request" error and clients would surface a raw JSON-RPC string. + # `.expire` on timeout for every blocking bridge whose `*.respond` tolerates + # a late reply (allow_expired=True): the tool returns empty, but a slow + # renderer can still answer and would otherwise hit a raw 4009. if not answered and not answer_present and event in { - "secret.request", - "sudo.request", - "clarify.request", - "terminal.read.request", - "preview.read.request", - "preview.act.request", - "window.read.request", - "mcp.setup.request", + "secret.request", "sudo.request", "clarify.request", "terminal.read.request", + "preview.read.request", "preview.act.request", "window.read.request", "mcp.setup.request", "tour.request", }: - _emit( - f"{event.removesuffix('.request')}.expire", - sid, - {"request_id": rid}, - ) + _emit(f"{event.removesuffix('.request')}.expire", sid, {"request_id": rid}) return answer @@ -2120,25 +1810,17 @@ def _clarify_block(sid: str, q, c, multi_select=False, questions=None) -> str: if questions: wire = [ { - "qid": entry["qid"], - "question": entry["question"], - "choices": entry["choices"], + "qid": entry["qid"], "question": entry["question"], "choices": entry["choices"], "multi_select": bool(entry["multi_select"]), } for entry in questions ] return _block( - "clarify.request", - sid, - {"questions": wire}, - timeout=_clarify_timeout_seconds(), + "clarify.request", sid, {"questions": wire}, timeout=_clarify_timeout_seconds(), batch_qids=[entry["qid"] for entry in questions], ) - # multi_select is a pass-through hint: renderers with checkbox - # support can honor it; older renderers ignore the extra field - # and stay single-select (a single answer still parses as a - # one-element list on the tool side). Only emitted when True so - # single-select payloads keep the exact pre-multi-select shape. + # multi_select is a pass-through hint older renderers ignore; emitted + # only when True so single-select payloads keep their exact shape. return _block( "clarify.request", sid, @@ -2178,21 +1860,14 @@ def _tour_request(sid: str, payload: dict) -> str: """Bridge the tour tool callback onto _block, without paying for a client that cannot answer it. - The renderer's ``tour.request`` handler ships in the desktop bundle, but - the tool is offered by this backend — and the two update on different - clocks. Against an app older than the tool the event lands in a renderer - with no branch for it, nobody ever calls ``tour.respond``, and the agent - blocks for the full deadline. The model then does what the schema tells it - to and tries the next action, so a single "give me a tour" turn stacks - those waits (the timeouts reported against #89620). - - A session's first action therefore gets the probe deadline, and an - unanswered probe marks the bridge unavailable for that session: every later - call returns immediately, telling the user what to actually fix instead of - stalling again. Once a client has answered, real actions get the full - deadline back and a single slow one no longer condemns it. The verdict - lives on the session record, so it dies with the session and a new one - re-probes. + The renderer's ``tour.request`` handler and this backend's tool update on + different clocks: against an older app nobody ever calls ``tour.respond`` + and each action blocks for the full deadline, stacking per turn. So a + session's first action gets the short probe deadline; an unanswered probe + marks the bridge unavailable for that session (later calls return at once + with what to fix). Once a client has answered, actions get the full + deadline and one slow action no longer condemns it. The verdict lives on + the session record, so a new session re-probes. """ # A detached caller has no session record; the throwaway keeps it on the # plain bridge, unprobed. @@ -2200,22 +1875,16 @@ def _tour_request(sid: str, payload: dict) -> str: if session is None: session = {} state = session.get("tour_bridge") - if state == "unanswered": return _TOUR_BRIDGE_UNAVAILABLE - answer = _block( - "tour.request", - sid, - dict(payload), + "tour.request", sid, dict(payload), timeout=_TOUR_TIMEOUT_S if state == "answered" else _TOUR_PROBE_TIMEOUT_S, ) - if answer: session["tour_bridge"] = "answered" elif state != "answered": session["tour_bridge"] = "unanswered" - return answer or _TOUR_BRIDGE_UNAVAILABLE @@ -2255,7 +1924,6 @@ def _resolve_model() -> str: # flagship the user didn't pick. try: from hermes_cli.models import get_preferred_silent_default_model - return get_preferred_silent_default_model() except Exception: return "z-ai/glm-5.2" @@ -2264,21 +1932,12 @@ def _resolve_model() -> str: def _resolve_session_platform() -> str: """Resolve the platform tag for a tui_gateway-routed session. - The desktop app's chat panel and the standalone TUI both speak to this - gateway; without a branch they all get stamped ``platform="tui"``, - which makes the agent think it's talking to a terminal user. That - mis-tag is the root cause of the desktop chat agent suggesting - TUI-only slash commands (``/reload-mcp``, …) to chat-panel users. - - Resolution: - * ``HERMES_DESKTOP=1`` and ``HERMES_DESKTOP_TERMINAL`` unset → "desktop" - (the chat-panel backend — a graphical React surface, not a terminal). - * ``HERMES_DESKTOP_TERMINAL=1`` → "tui" - (``hermes --tui`` running in the desktop's embedded terminal pane; - it IS a TUI, just embedded. The clarifier attached to the tui hint - in system_prompt.py tells the agent about the embedding.) - * neither set → "tui" - (standalone ``hermes --tui``.) + Stamping the desktop chat panel ``platform="tui"`` makes the agent suggest + TUI-only slash commands to chat-panel users. + * ``HERMES_DESKTOP=1`` with ``HERMES_DESKTOP_TERMINAL`` unset → "desktop" + * ``HERMES_DESKTOP_TERMINAL=1`` → "tui" (embedded terminal pane; the tui + hint's clarifier in system_prompt.py describes the embedding) + * neither → "tui" (standalone ``hermes --tui``) """ if is_truthy_value(os.environ.get("HERMES_DESKTOP")) and not is_truthy_value( os.environ.get("HERMES_DESKTOP_TERMINAL") @@ -2324,15 +1983,9 @@ def _config_model_target() -> tuple[str, str]: provider = "" elif isinstance(cfg_model, str): model = cfg_model.strip() - # No fallback to _resolve_model() here: that reads HERMES_MODEL / - # HERMES_INFERENCE_MODEL, which `hermes --tui -m ` sets as a - # session-scoped seed for THIS launch. When config.yaml has no - # model.default (custom-provider-only setups), falling back to the env - # seed made the per-turn sync treat the -m flag as "the configured - # model" and replay it as a /model switch — which then persisted the - # one-shot flag into config.yaml globally (#-m leak). An empty model - # simply means "config expresses no preference": the sync is a no-op - # and the agent keeps whatever it was built with. + # No _resolve_model() fallback: that reads the launch-scoped -m env seed, + # which the per-turn sync would replay as a /model switch and persist + # globally. Empty model = "config expresses no preference" → sync is a no-op. return model, provider @@ -2341,24 +1994,17 @@ def _resolve_startup_runtime() -> tuple[str, str | None]: explicit_provider = os.environ.get("HERMES_TUI_PROVIDER", "").strip() if explicit_provider: return model, explicit_provider - explicit_model = ( os.environ.get("HERMES_MODEL", "") or os.environ.get("HERMES_INFERENCE_MODEL", "") ).strip() if not explicit_model: return model, None - - try: + with contextlib.suppress(Exception): from hermes_cli.models import detect_static_provider_for_model - cfg = _load_cfg().get("model") or {} current_provider = ( - ( - str(cfg.get("provider") or "").strip().lower() - if isinstance(cfg, dict) - else "" - ) + (str(cfg.get("provider") or "").strip().lower() if isinstance(cfg, dict) else "") or os.environ.get("HERMES_INFERENCE_PROVIDER", "").strip().lower() or "auto" ) @@ -2366,20 +2012,13 @@ def _resolve_startup_runtime() -> tuple[str, str | None]: if detected: provider, detected_model = detected return detected_model, provider - except Exception: - pass return model, None # Bare billing buckets are not routable provider identities; restoring one as a -# session provider override breaks resume. (agent_init's fail-fast gate is a -# DIFFERENT set that also skips "openrouter" — there it means "default route, -# don't fail fast", not "unroutable".) -# ``openrouter`` is deliberately excluded here — it is a fully routable provider -# with its own API key and base_url. Sessions that used OpenRouter store -# ``billing_provider="openrouter"``; dropping it forces resume to the current -# global model (e.g. a custom endpoint), which is the wrong provider for the -# stored model. See #57588. +# session provider override breaks resume. ``openrouter`` is deliberately NOT in +# this set (fully routable; dropping it resumed OpenRouter sessions on the wrong +# provider) — agent_init's fail-fast gate is a different set that skips it. from hermes_state import _BARE_BILLING_PROVIDERS @@ -2394,14 +2033,11 @@ def _overrides_have_routable_provider(overrides: dict) -> bool: """ provider = str(overrides.get("provider_override") or "").strip() if not provider: - provider = str( - (overrides.get("model_override") or {}).get("provider") or "" - ).strip() + provider = str((overrides.get("model_override") or {}).get("provider") or "").strip() if not provider: return False try: from hermes_cli.runtime_provider import is_routable_provider - return is_routable_provider(provider) except Exception: return False @@ -2410,83 +2046,23 @@ def _overrides_have_routable_provider(overrides: dict) -> bool: def _stored_session_runtime_overrides(row: dict | None) -> dict: """Return runtime fields persisted with a stored session. - ``session.resume`` is a session-scoped operation: reopening an older chat - must restore the model/provider/reasoning state that chat actually used, - not whatever global model the user most recently selected in another chat. - The durable session row stores the model directly, the billing provider in - ``billing_provider``, and richer runtime knobs in JSON ``model_config``. + ``session.resume`` is session-scoped: reopening an older chat must restore + the model/provider/reasoning state that chat actually used, not the global + model most recently selected elsewhere. The row stores the model directly, + the billing provider in ``billing_provider``, and richer knobs in JSON + ``model_config``. + + Plugin-owned Bot-Mode sessions are exempt and always rebuild from the member + profile's CURRENT config — restoring a stale provider pin is what left room + bots / bot DMs failing ("out of Nous credits") after the profile switched. + Signals, in order: the explicit ``room_plumbing`` / ``follow_profile_config`` + markers persisted by session.create consumers; the legacy hidden + + "Group:" title shape (older desktop builds sent no marker); and the title + exactly "Bot Chat" (the plugin's own identity rule for the forever-DM, + UNIQUE(title) makes it exact; pre-policy rows may be visible or hidden). """ if not row: return {} - - # Bot-Mode room plumbing sessions (hidden, titled "Group: ") are - # per-member scratch conversations inside a group chat. They must always - # rebuild from the member profile's CURRENT config: restoring the stored - # model/provider pin from an old row is what left room bots stuck on - # Nous (or any earlier provider) long after the profile was switched — - # every room message then failed with "out of Nous credits" while the - # same bots worked fine in DMs. 1:1 chats keep the stored-runtime - # restore (opening an older chat must show the model it actually used); - # only the room plumbing is exempt. - # - # The primary signal is the EXPLICIT ``room_plumbing`` contract persisted - # by session.create/room consumers (desktop Bot Mode) — a deliberate - # marker, not a presentation heuristic. The hidden + "Group:" title - # shape is kept as a legacy fallback so rows created by older desktop - # builds (which never sent the marker) still behave correctly until the - # client catches up. - raw_plumbing = row.get("model_config") - if isinstance(raw_plumbing, dict): - _plumbing_marker = raw_plumbing.get("room_plumbing") - elif isinstance(raw_plumbing, str) and raw_plumbing.strip(): - try: - _plumbing_marker = json.loads(raw_plumbing).get("room_plumbing") - except Exception: - _plumbing_marker = None - else: - _plumbing_marker = None - if _plumbing_marker: - return {} - _row_title = str(row.get("title") or "").strip() - _row_hidden = row.get("hidden") - if _row_hidden and _row_title.startswith("Group:"): - return {} - - # Bot-Mode canonical chats (the ONE forever DM per bot) and room plumbing - # sessions are plugin-owned scratch conversations. They must always rebuild - # from the member profile's CURRENT config: restoring the stored - # model/provider pin from an old row is what left bot DMs stuck on a stale - # provider (e.g. "out of Nous credits" after the profile was switched to - # ollama-cloud) while the same bot worked fine in rooms. 1:1 user chats - # keep the stored-runtime restore (opening an older chat must show the - # model it actually used); only the plugin-owned bot sessions are exempt. - # - # The primary signal is the EXPLICIT ``follow_profile_config`` contract - # persisted by session.create consumers (desktop Bot Mode) — a deliberate - # marker, not a presentation heuristic. - raw_follow = row.get("model_config") - if isinstance(raw_follow, dict): - _follow_marker = raw_follow.get("follow_profile_config") - elif isinstance(raw_follow, str) and raw_follow.strip(): - try: - _follow_marker = json.loads(raw_follow).get("follow_profile_config") - except Exception: - _follow_marker = None - else: - _follow_marker = None - if _follow_marker: - return {} - # Legacy backfill: canonical Bot Chats created BEFORE the - # follow_profile_config contract existed carry no marker, yet they are - # still the plugin-owned forever-DM. The plugin's own identity rule is - # "the profile's session titled exactly 'Bot Chat'" (UNIQUE(title) makes - # that an exact registry, and pre-policy rows may be visible OR hidden), - # so mirror that rule here. Without this, every Bot Chat that already - # exists in the field stays pinned to its stale stored provider until - # the user deletes it — the exact live-report shape (#89497 / #94818). - if _row_title == "Bot Chat": - return {} - raw_config = row.get("model_config") model_config: dict = {} if isinstance(raw_config, dict): @@ -2498,14 +2074,21 @@ def _stored_session_runtime_overrides(row: dict | None) -> dict: model_config = parsed except Exception: logger.debug("failed to parse stored session model_config", exc_info=True) - + _row_title = str(row.get("title") or "").strip() + if ( + model_config.get("room_plumbing") + or (row.get("hidden") and _row_title.startswith("Group:")) + or model_config.get("follow_profile_config") + or _row_title == "Bot Chat" + ): + return {} overrides: dict = {} model = str(row.get("model") or model_config.get("model") or "").strip() - # ``billing_provider`` is only the billing bucket — for a custom endpoint it is the - # bare class ``"custom"``, which agent_init treats as non-routable, so restoring it as - # the provider override makes ``session.resume`` fail with "No LLM provider configured". - # Only restore an explicit provider; otherwise leave it unset so resume falls back to - # the configured default, matching the working CLI path. + # ``billing_provider`` is only the billing bucket — for a custom endpoint it + # is the bare class ``"custom"``, which agent_init treats as non-routable, so + # restoring it as the provider override fails resume with "No LLM provider + # configured". Only restore an explicit provider; otherwise leave it unset + # so resume falls back to the configured default (CLI parity). explicit_provider = str(model_config.get("provider") or "").strip() billing_provider = str( model_config.get("billing_provider") or row.get("billing_provider") or "" @@ -2518,20 +2101,14 @@ def _stored_session_runtime_overrides(row: dict | None) -> dict: reasoning_config = model_config.get("reasoning_config") service_tier = str(model_config.get("service_tier") or "").strip() - # Heal a stale/expired provider name persisted by an older build — not - # just the bare ``"custom"`` billing class. A renamed or removed custom - # provider (e.g. ``oldone`` -> ``newone``) stored in the session row - # would otherwise fail agent init with "Unknown provider ''". - # Recover the durable ``custom:`` menu key from the stored - # base_url, then from the entry that serves the stored model, falling - # back to the configured provider when the row has neither. When - # nothing names a real entry, drop the provider entirely so resume - # falls back to the configured default rather than the broken route. + # Heal a stale/expired provider name persisted by an older build (a renamed + # or removed custom provider would fail agent init with "Unknown provider"). + # Recover the durable ``custom:`` key from the stored base_url, then + # from the entry serving the stored model; when nothing names a real entry, + # drop the provider so resume falls back to the configured default. if provider: - routable = False try: from hermes_cli.runtime_provider import is_routable_provider - routable = is_routable_provider(provider) except Exception: routable = False @@ -2539,36 +2116,23 @@ def _stored_session_runtime_overrides(row: dict | None) -> dict: healed = None try: from hermes_cli.runtime_provider import canonical_custom_identity - - healed = canonical_custom_identity( - base_url=base_url or None, model=model or None - ) + healed = canonical_custom_identity(base_url=base_url or None, model=model or None) except Exception: - logger.debug( - "custom provider identity recovery failed", exc_info=True - ) + logger.debug("custom provider identity recovery failed", exc_info=True) if healed: - logger.info( - "healed stale session provider %r to %r", provider, healed - ) + logger.info("healed stale session provider %r to %r", provider, healed) provider = healed - # The healed identity owns a registered endpoint; drop the - # snapshot's base_url so it can't override the registry URL - # (e.g. a stale direct endpoint behind a renamed proxy). + # The healed identity owns a registered endpoint; the snapshot's + # base_url must not override the registry URL. base_url = "" else: provider = "" - if model: - # Use the same dict-shaped override that live /model switches use so a - # DB-restored session can preserve custom endpoint metadata across both - # initial resume and later rebuilds (/new). Deliberately do not persist - # or restore raw api_key here; endpoint credentials should continue to - # come from config/env/provider resolution rather than the session DB. + # Same dict-shaped override live /model switches use, so a DB-restored + # session keeps custom endpoint metadata across resume and rebuilds + # (/new). Raw api_key is deliberately never persisted or restored here. overrides["model_override"] = { - "model": model, - "provider": provider or None, - "base_url": base_url or None, + "model": model, "provider": provider or None, "base_url": base_url or None, "api_mode": api_mode or None, } if provider: @@ -2576,12 +2140,11 @@ def _stored_session_runtime_overrides(row: dict | None) -> dict: if isinstance(reasoning_config, dict): overrides["reasoning_config_override"] = reasoning_config if service_tier.lower() == "normal": - # None means "inherit the profile" at _make_agent. Empty string is a - # real override that means "do not request a priority service tier". + # None means "inherit the profile" at _make_agent; "" is a real override + # meaning "do not request a priority service tier". overrides["service_tier_override"] = "" elif service_tier: overrides["service_tier_override"] = service_tier - return overrides @@ -2604,63 +2167,32 @@ def _runtime_model_config(agent, existing: dict | None = None) -> dict: model = str(getattr(agent, "model", "") or "").strip() provider = str(getattr(agent, "provider", "") or "").strip() base_url = str(getattr(agent, "base_url", "") or "").strip() - api_mode = str(getattr(agent, "api_mode", "") or "").strip() + if provider.lower() == "custom": + # ``agent.provider`` resolves every named custom entry to the literal + # "custom", which loses the entry identity (api_key is never persisted, + # so resume couldn't re-resolve credentials). Recover the canonical + # ``custom:`` key from the endpoint URL, else from the configured + # provider (the no-base_url case that routed to OpenRouter with no key). + try: + from hermes_cli.runtime_provider import canonical_custom_identity + provider = canonical_custom_identity(base_url=base_url, model=model or None) or provider + except Exception: + logger.debug("custom provider identity lookup failed", exc_info=True) reasoning_config = getattr(agent, "reasoning_config", None) - service_tier = getattr(agent, "service_tier", None) - - if model: - config["model"] = model - else: - config.pop("model", None) - if provider: - if provider.strip().lower() == "custom": - # ``agent.provider`` is the RESOLVED provider, and for any named - # ``providers:`` / ``custom_providers:`` entry that is the literal - # string "custom" — persisting it loses the entry identity, so a - # later resume/rebuild cannot re-resolve the entry's credentials - # (the api_key is deliberately never persisted; see - # _stored_session_runtime_overrides). Recover the canonical - # ``custom:`` menu key from the endpoint URL when present, - # else from the configured provider — this second fallback is the - # fix for sessions built WITHOUT a base_url on the override (the - # recurring Desktop/TUI "No LLM provider configured" regression: - # bare "custom" with no base_url was persisted verbatim and routed - # to OpenRouter with no key on the next resume). - try: - from hermes_cli.runtime_provider import ( - canonical_custom_identity, - ) - - provider = ( - canonical_custom_identity( - base_url=base_url, model=model or None - ) - or provider - ) - except Exception: - logger.debug( - "custom provider identity lookup failed", exc_info=True - ) - config["provider"] = provider - else: - config.pop("provider", None) - if base_url: - config["base_url"] = base_url - else: - config.pop("base_url", None) - if api_mode: - config["api_mode"] = api_mode - else: - config.pop("api_mode", None) - if isinstance(reasoning_config, dict): - config["reasoning_config"] = reasoning_config - else: - config.pop("reasoning_config", None) - if service_tier: - config["service_tier"] = service_tier - else: - config.pop("service_tier", None) - + live = { + "model": model, + "provider": provider, + "base_url": base_url, + "api_mode": str(getattr(agent, "api_mode", "") or "").strip(), + # An empty dict is still a real (present) reasoning config. + "reasoning_config": reasoning_config if isinstance(reasoning_config, dict) else None, + "service_tier": getattr(agent, "service_tier", None), + } + for key, value in live.items(): + if value or isinstance(value, dict): + config[key] = value + else: + config.pop(key, None) return config @@ -2672,11 +2204,9 @@ def _persist_live_session_runtime(session: dict | None) -> None: session_key = str(session.get("session_key") or "").strip() if agent is None or not session_key: return - db = getattr(agent, "_session_db", None) or _get_db() if db is None: return - try: row = db.get_session(session_key) or {} raw_config = row.get("model_config") @@ -2711,38 +2241,25 @@ def _persist_live_session_system_prompt(session: dict | None) -> None: session_key = str(session.get("session_key") or "").strip() if agent is None or not session_key or not hasattr(agent, "_build_system_prompt"): return - db = getattr(agent, "_session_db", None) or _get_db() if db is None or not hasattr(db, "update_system_prompt"): return - # Re-bind HERMES_HOME to the session's profile so load_soul_md() and - # build_skills_system_prompt() resolve to the correct profile. Without - # this, _start_agent_build's finally block has already reset the - # override and the rebuilt prompt silently uses the root profile's - # SOUL.md and skills. See issue #50233. + # Re-bind HERMES_HOME to the session's profile: the build's finally already + # reset it, and the rebuilt prompt would use the root profile's SOUL.md/skills. profile_home = session.get("profile_home") - home_token = ( - set_hermes_home_override(profile_home) if profile_home else None - ) - # Bind the session context too. This function runs on the RPC dispatcher - # thread (model.switch, config.set model). On that thread the _SESSION_CWD - # contextvar is not set, so resolve_agent_cwd() falls back to the process - # TERMINAL_CWD, which the desktop pins to the home directory. The rebuilt - # prompt then records the wrong working directory and persists it. Later - # turns restore the stored bytes without change, because the turn - # prologue rebuilds only when _cached_system_prompt is None. - session_tokens = _set_session_context( - session_key, cwd=_session_cwd(session) - ) + home_token = (set_hermes_home_override(profile_home) if profile_home else None) + # Bind the session context too: on the RPC dispatcher thread _SESSION_CWD is + # unset, so resolve_agent_cwd() falls back to the process TERMINAL_CWD and + # the rebuilt prompt persists the wrong cwd (later turns reuse the bytes). + session_tokens = _set_session_context(session_key, cwd=_session_cwd(session)) try: prompt = agent._build_system_prompt(None) agent._cached_system_prompt = prompt db.update_system_prompt(getattr(agent, "session_id", None) or session_key, prompt) except Exception: logger.warning( - "failed to persist live session system prompt for session %s", - session_key, + "failed to persist live session system prompt for session %s", session_key, exc_info=True, ) finally: @@ -2751,10 +2268,8 @@ def _persist_live_session_system_prompt(session: dict | None) -> None: reset_hermes_home_override(home_token) -# Stable leading text of the model-switch marker, shared by the builder and the -# dedup below. Only the newest marker is meaningful (it names the *currently* -# active model); older ones are stale and would otherwise be re-sent to the -# provider on every turn (#65891). +# Stable leading text of the model-switch marker (builder + dedup). Only the +# newest marker is meaningful; stale ones would be re-sent every turn. _MODEL_SWITCH_MARKER_PREFIX = "[System: The active model for this chat has changed to " @@ -2795,17 +2310,14 @@ def _append_model_switch_marker(session: dict | None, *, model: str, provider: s session_key = str(session.get("session_key") or "").strip() if not session_key: return - provider_part = f" via provider {provider}" if provider else "" marker = ( f"{_MODEL_SWITCH_MARKER_PREFIX}" f"{model}{provider_part}. From this point forward, use this runtime " "metadata when answering questions about what model/provider is active.]" ) - # Persist as a user message, not a system message. The gateway appends - # this marker after prior conversation turns, and strict OpenAI-compatible - # providers (vLLM, Qwen) reject system messages that are not at the - # beginning of the API message list (#48338). + # A user message, not system: strict OpenAI-compatible providers (vLLM, + # Qwen) reject non-leading system messages. entry = {"role": "user", "content": marker, "display_kind": "model_switch"} def _replace_markers() -> None: @@ -2814,33 +2326,25 @@ def _append_model_switch_marker(session: dict | None, *, model: str, provider: s history[:] = [h for h in history if not _is_model_switch_marker(h)] history.append(entry) session["history_version"] = int(session.get("history_version", 0)) + 1 - lock = session.get("history_lock") if lock is not None: with lock: _replace_markers() else: _replace_markers() - try: agent = session.get("agent") db = getattr(agent, "_session_db", None) if agent is not None else None if db is not None: db.append_message( - session_id=session_key, - role="user", - content=marker, - display_kind="model_switch", + session_id=session_key, role="user", content=marker, display_kind="model_switch", ) return - _ensure_session_db_row(session) with _session_db(session) as scoped_db: if scoped_db is not None: scoped_db.append_message( - session_id=session_key, - role="user", - content=marker, + session_id=session_key, role="user", content=marker, display_kind="model_switch", ) except Exception: @@ -2864,31 +2368,15 @@ def _write_config_key(key_path: str, value): _STATUSBAR_MODES = frozenset({"off", "top", "bottom"}) _APPROVAL_MODES = frozenset({"manual", "smart", "off"}) -# Appearance switches the desktop renderer owns but the AGENT has to see: each -# one gates a tool's `check_fn`, so the toggle has to reach the config of -# whichever gateway the app is actually talking to — local, SSH, URL, or cloud. -# -# `config.set` matches an explicit key list and answers 4002 for anything else, -# so a renderer mirroring a key that is not listed here writes nothing at all. -# That is not hypothetical: the reactions toggle shipped mirroring -# `display.message_reactions`, every write was rejected into a swallowed -# `.catch()`, and `react_to_message` therefore stayed dark no matter what the -# user picked. Adding a mirrored switch to the renderer means adding it here. +# Appearance switches the renderer owns but the AGENT must see (each gates a +# tool's `check_fn`), so the toggle must reach whichever gateway the app talks +# to. `config.set` answers 4002 for unlisted keys — a mirrored switch missing +# here writes nothing and its tool stays dark. Add renderer mirrors here too. _DISPLAY_TOGGLE_KEYS = frozenset( - { - "display.message_reactions", - "display.in_app_tips", - "display.in_app_tours", - } + {"display.message_reactions", "display.in_app_tips", "display.in_app_tours"} ) _BOOL_WORDS = { - "1": True, - "on": True, - "true": True, - "yes": True, - "0": False, - "off": False, - "false": False, + "1": True, "on": True, "true": True, "yes": True, "0": False, "off": False, "false": False, "no": False, } @@ -2896,20 +2384,11 @@ _BOOL_WORDS = { def _load_approval_mode() -> str: """Resolve the effective ``approvals.mode`` for the TUI surface. - Delegates to the canonical resolver in ``tools.approval`` - (``_get_approval_mode``) so mode resolution cannot drift per surface — - the same normalization, defaults, and config precedence the approval - gate itself uses (see ``tools/approval.py``). - - Previously this re-read the config raw via ``_load_cfg`` + - ``_deep_merge(DEFAULT_CONFIG, ...)`` and normalized locally, which - could disagree with the gate's own view of the mode (e.g. the - canonical ``hermes_cli.config.load_config`` path applies managed-scope - overlays and ``${VAR}`` env expansion that the TUI's raw YAML read did - not fully mirror). + Delegates to ``tools.approval._get_approval_mode`` so the mode cannot drift + from the approval gate's own view (a local raw-config re-read missed the + managed overlay and ``${VAR}`` expansion). """ from tools.approval import _get_approval_mode - mode = _get_approval_mode() return mode if mode in _APPROVAL_MODES else "manual" @@ -2923,22 +2402,9 @@ def _coerce_statusbar(raw) -> str: _MOUSE_TRACKING_ALIASES = { - "0": "off", - "1": "all", - "all": "all", - "any": "all", - "button": "buttons", - "buttons": "buttons", - "click": "buttons", - "false": "off", - "full": "all", - "no": "off", - "off": "off", - "on": "all", - "scroll": "wheel", - "true": "all", - "wheel": "wheel", - "yes": "all", + "0": "off", "1": "all", "all": "all", "any": "all", "button": "buttons", "buttons": "buttons", + "click": "buttons", "false": "off", "full": "all", "no": "off", "off": "off", "on": "all", + "scroll": "wheel", "true": "all", "wheel": "wheel", "yes": "all", } @@ -2977,16 +2443,11 @@ def _load_reasoning_config(model: str = "") -> dict | None: Closes #21256. """ from hermes_constants import resolve_reasoning_config - return resolve_reasoning_config(_load_cfg(), model) def _load_service_tier() -> str | None: - raw = ( - str((_load_cfg().get("agent") or {}).get("service_tier", "") or "") - .strip() - .lower() - ) + raw = (str((_load_cfg().get("agent") or {}).get("service_tier", "") or "") .strip() .lower()) if not raw or raw in {"normal", "default", "standard", "off", "none"}: return None if raw in {"fast", "priority", "on"}: @@ -3013,7 +2474,7 @@ def _load_provider_routing() -> dict: def _load_show_reasoning() -> bool: # Fallback True — keep in sync with DEFAULT_CONFIG display.show_reasoning # (this loader reads the raw user YAML without the DEFAULT_CONFIG merge). - return bool((_load_cfg().get("display") or {}).get("show_reasoning", True)) + return bool(_display_cfg().get("show_reasoning", True)) def _load_memory_notifications() -> str: @@ -3026,7 +2487,7 @@ def _load_memory_notifications() -> str: who set ``off``. Accepts ``off`` / ``on`` (default) / ``verbose``; a bool is normalized for back-compat. """ - raw = (_load_cfg().get("display") or {}).get("memory_notifications") + raw = _display_cfg().get("memory_notifications") if isinstance(raw, bool): return "on" if raw else "off" return str(raw).lower() if raw else "on" @@ -3036,7 +2497,7 @@ def _load_tool_progress_mode() -> str: env = os.environ.get("HERMES_TUI_TOOL_PROGRESS", "").strip().lower() if env in {"off", "new", "all", "verbose"}: return env - raw = (_load_cfg().get("display") or {}).get("tool_progress", "all") + raw = _display_cfg().get("tool_progress", "all") if raw is False: return "off" if raw is True: @@ -3048,17 +2509,12 @@ def _load_tool_progress_mode() -> str: def _gui_surface_toolsets(platform: str) -> set[str]: """Toolsets that exist because of the CLIENT on the other end, not the host. - Both entries are deliberately off ``_HERMES_CORE_TOOLS`` — every other - platform would carry their schema for nothing — so this resolver is the one - gate that exposes them. - - ``platform`` is the SESSION's source (``session.create``'s ``source`` - field), never a process env var. The desktop app is a client: it can be - driving a local, SSH, URL, or cloud backend, and only the local/SSH spawn - paths run with ``HERMES_DESKTOP=1``. Keying GUI capability off that env var - silently stripped every pane/browser tool from URL and cloud gateways while - the same backend told the model it was "chatting inside the Hermes desktop - app". See the surface-capability rule in AGENTS.md. + Both entries are off ``_HERMES_CORE_TOOLS`` (no other platform should carry + their schema), so this resolver is the one gate that exposes them. + ``platform`` is the SESSION's source, never a process env var: the desktop + may drive a URL/cloud backend where ``HERMES_DESKTOP`` is unset, and keying + off the env var stripped every pane/browser tool there. See the + surface-capability rule in AGENTS.md. """ surfaces = {"project"} if platform == "desktop": @@ -3066,149 +2522,119 @@ def _gui_surface_toolsets(platform: str) -> set[str]: return surfaces +def _enabled_mcp_server_names() -> tuple[set[str], set[str]]: + """(enabled, disabled) MCP server names from raw config; empty on any failure.""" + try: + from hermes_cli.config import read_raw_config + from hermes_cli.tools_config import _parse_enabled_flag + raw_cfg = read_raw_config() + mcp_servers = raw_cfg.get("mcp_servers") if isinstance(raw_cfg.get("mcp_servers"), dict) else {} + enabled, disabled = set(), set() + for name, server_cfg in mcp_servers.items(): + if not isinstance(server_cfg, dict): + continue + if _parse_enabled_flag(server_cfg.get("enabled", True), default=True): + enabled.add(str(name)) + else: + disabled.add(str(name)) + return enabled, disabled + except Exception: + return set(), set() + + +def _resolve_explicit_toolsets(explicit: list[str], validate_toolset) -> list[str] | None | bool: + """Resolve a HERMES_TUI_TOOLSETS pin: list, None for "all", False when nothing was valid.""" + built_in = [name for name in explicit if validate_toolset(name)] + unresolved = [name for name in explicit if name not in built_in] + if unresolved: + try: + from hermes_cli.plugins import discover_plugins + discover_plugins() + plugin_valid = [name for name in unresolved if validate_toolset(name)] + except Exception: + plugin_valid = [] + if plugin_valid: + built_in.extend(plugin_valid) + unresolved = [name for name in unresolved if name not in plugin_valid] + if any(name in {"all", "*"} for name in built_in): + ignored = [name for name in explicit if name not in {"all", "*"}] + if ignored: + print( + "[tui] HERMES_TUI_TOOLSETS=all enables every toolset; " + f"ignoring additional entries: {', '.join(ignored)}", + file=sys.stderr, + flush=True, + ) + return None + if not unresolved: + return built_in + mcp_names, mcp_disabled = _enabled_mcp_server_names() + mcp_valid = [name for name in unresolved if name in mcp_names] + disabled = [name for name in unresolved if name in mcp_disabled] + unknown = [name for name in unresolved if name not in mcp_names and name not in mcp_disabled] + if unknown: + print( + f"[tui] ignoring unknown HERMES_TUI_TOOLSETS entries: {', '.join(unknown)}", + file=sys.stderr, flush=True, + ) + if disabled: + print( + "[tui] ignoring disabled MCP servers in HERMES_TUI_TOOLSETS " + "(set enabled: true in config.yaml to use): " + f"{', '.join(disabled)}", + file=sys.stderr, + flush=True, + ) + return (built_in + mcp_valid) or False + + def _load_enabled_toolsets(platform: str | None = None) -> list[str] | None: + """Resolve the agent's toolsets for this desktop/TUI session (None = all). + + Order: an explicit HERMES_TUI_TOOLSETS pin; else the coding posture + (collapse to the coding toolset + enabled MCP servers when sitting in a code + workspace — agent/coding_context.py, config loaded lazily there); else the + configured CLI toolsets. The client-surface (pane/project) toolsets are off + _HERMES_CORE_TOOLS so no other platform carries their schema; this resolver + runs only in the desktop/TUI gateway, so folding them in here is the gate + that exposes them on exactly the surface that can answer them. + """ session_platform = platform or _resolve_session_platform() explicit = [ item.strip() for item in os.environ.get("HERMES_TUI_TOOLSETS", "").split(",") if item.strip() ] - cfg = None fallback_notice = None - - # Coding posture (base Hermes): with no explicit pin, collapse to the - # coding toolset (+ enabled MCP servers) when sitting in a code workspace. - # The desktop app and `hermes --tui` both land here. See - # agent/coding_context.py. No config is loaded yet at this point, so we let - # coding_selection() load it lazily (cli.py passes its already-resolved - # CLI_CONFIG instead, purely to avoid a redundant read). if not explicit: - try: + with contextlib.suppress(Exception): from agent.coding_context import coding_selection - selection = coding_selection(platform=session_platform) if selection is not None: - # Fold in the client-surface toolsets here too: the focus-mode - # coding posture returns before the fallback path that normally - # adds them — without this the desktop loses its pane/project - # tools exactly when sitting in a repo (see below). return sorted({*selection, *_gui_surface_toolsets(session_platform)}) - except Exception: - pass - try: from toolsets import validate_toolset except Exception: validate_toolset = None - if explicit and validate_toolset is not None: - built_in = [name for name in explicit if validate_toolset(name)] - unresolved = [name for name in explicit if name not in built_in] - - if unresolved: - try: - from hermes_cli.plugins import discover_plugins - - discover_plugins() - plugin_valid = [name for name in unresolved if validate_toolset(name)] - except Exception: - plugin_valid = [] - - if plugin_valid: - built_in.extend(plugin_valid) - unresolved = [name for name in unresolved if name not in plugin_valid] - - if any(name in {"all", "*"} for name in built_in): - ignored = [name for name in explicit if name not in {"all", "*"}] - if ignored: - print( - "[tui] HERMES_TUI_TOOLSETS=all enables every toolset; " - f"ignoring additional entries: {', '.join(ignored)}", - file=sys.stderr, - flush=True, - ) - return None - - if not unresolved: - return built_in - - mcp_names: set[str] = set() - mcp_disabled: set[str] = set() - try: - from hermes_cli.config import read_raw_config - from hermes_cli.tools_config import _parse_enabled_flag - - raw_cfg = read_raw_config() - mcp_servers = ( - raw_cfg.get("mcp_servers") - if isinstance(raw_cfg.get("mcp_servers"), dict) - else {} - ) - for name, server_cfg in mcp_servers.items(): - if not isinstance(server_cfg, dict): - continue - if _parse_enabled_flag(server_cfg.get("enabled", True), default=True): - mcp_names.add(str(name)) - else: - mcp_disabled.add(str(name)) - except Exception: - mcp_names = set() - mcp_disabled = set() - - mcp_valid = [name for name in unresolved if name in mcp_names] - disabled = [name for name in unresolved if name in mcp_disabled] - unknown = [ - name - for name in unresolved - if name not in mcp_names and name not in mcp_disabled - ] - valid = built_in + mcp_valid - - if unknown: - print( - f"[tui] ignoring unknown HERMES_TUI_TOOLSETS entries: {', '.join(unknown)}", - file=sys.stderr, - flush=True, - ) - if disabled: - print( - "[tui] ignoring disabled MCP servers in HERMES_TUI_TOOLSETS " - "(set enabled: true in config.yaml to use): " - f"{', '.join(disabled)}", - file=sys.stderr, - flush=True, - ) - - if valid: - return valid - + resolved = _resolve_explicit_toolsets(explicit, validate_toolset) + if resolved is not False: + return resolved fallback_notice = ( "[tui] no valid HERMES_TUI_TOOLSETS entries; using configured CLI toolsets" ) - try: from hermes_cli.config import load_config from hermes_cli.tools_config import _get_platform_tools - - cfg = cfg if cfg is not None else load_config() - - # Runtime toolset resolution must include default MCP servers so the - # agent can actually call them. Passing ``False`` here is the - # config-editing variant — used when we need to persist a toolset - # list without baking in implicit MCP defaults. Using the wrong - # variant at agent creation time makes MCP tools silently missing - # from the TUI. See PR #3252 for the original design split. + cfg = load_config() + # include_default_mcp_servers=True is the runtime variant (the agent + # must be able to call default MCP servers); False is the config-editing + # variant. Using the wrong one here silently drops MCP tools from the TUI. enabled = _get_platform_tools(cfg, "cli", include_default_mcp_servers=True) if fallback_notice is not None: print(fallback_notice, file=sys.stderr, flush=True) if not enabled: return None - # The client-surface toolsets are off _HERMES_CORE_TOOLS (every other - # platform would carry their schema for nothing), so the platform - # recovery above — which keys off hermes-cli's tool universe — can't - # surface them. This resolver runs ONLY in the desktop/TUI gateway, so - # folding them in here is the gate that exposes them on exactly the - # surface that can answer them. return sorted(enabled | _gui_surface_toolsets(session_platform)) except Exception: if fallback_notice is not None: @@ -3234,38 +2660,29 @@ def _tool_progress_enabled(sid: str) -> bool: def _tool_lifecycle_required_for_ui(name: str) -> bool: """Return True for tool events that are interactive UI, not optional chrome.""" - # Desktop renders the clarify choices/question from the tool-call part, then - # wires request_id from clarify.request. If tool progress is off, suppressing - # clarify's lifecycle events leaves only the sidebar attention dot visible. - # setup_mcp is the same shape: its consent card mounts on the tool part. + # Desktop renders clarify / setup_mcp cards from the tool-call part; with + # tool progress off, suppressing them would leave only the sidebar dot. return name in ("clarify", "setup_mcp") def _restart_slash_worker(sid: str, session: dict): worker = session.get("slash_worker") - # A session that never spawned a worker has nothing stale to replace — - # the next slash.exec builds one with the current session key/model. - # Spawning here would fork the per-worker stdio MCP fleet for sessions - # that never use worker-routed commands. + # Nothing to replace for a session that never spawned a worker; spawning + # here would fork the per-worker MCP fleet for nothing. if worker is None: return - try: + with contextlib.suppress(Exception): worker.close() - except Exception: - pass try: new_worker = _SlashWorker( - session["session_key"], - getattr(session.get("agent"), "model", _resolve_model()), + session["session_key"], getattr(session.get("agent"), "model", _resolve_model()), profile_home=session.get("profile_home"), ) except Exception: session["slash_worker"] = None return - # Route through the same store-iff-still-mapped guard as the spawn sites: - # the post-turn restart runs as `running` flips false, exactly when a - # close_on_disconnect reap can pop this session — a bare store would orphan - # the fresh worker (it self-heals only on gateway exit via the watchdog). + # Store-iff-still-mapped: the post-turn restart races a close_on_disconnect + # reap, and a bare store would orphan the fresh worker. _attach_worker(sid, session, new_worker) @@ -3275,31 +2692,18 @@ def _get_usage(agent) -> dict: "model": getattr(agent, "model", "") or "", "input": g("session_input_tokens", "session_prompt_tokens"), "output": g("session_output_tokens", "session_completion_tokens"), - "reasoning": g("session_reasoning_tokens"), - "prompt": g("session_prompt_tokens"), - "completion": g("session_completion_tokens"), - "total": g("session_total_tokens"), + "reasoning": g("session_reasoning_tokens"), "prompt": g("session_prompt_tokens"), + "completion": g("session_completion_tokens"), "total": g("session_total_tokens"), "calls": g("session_api_calls"), } comp = getattr(agent, "context_compressor", None) if comp: - # context_used is the *current-window* occupancy. Do NOT fall back to - # usage["total"] (cumulative lifetime session_total_tokens): for an - # external context engine that doesn't report last_prompt_tokens that - # substitution showed lifetime totals as the live context fill, yielding - # impossible readings such as 1.9m/120k clamped to 100% (#50421). - # - # Per the issue, populate context_used/percent only from a *real* - # current-occupancy value and "leave it unknown otherwise" — so a falsy - # last_prompt_tokens (0 or missing, i.e. an engine that doesn't track - # per-window occupancy) intentionally emits no gauge rather than a - # fabricated 0% or the old cumulative reading. The built-in compressor - # always reports a real last_prompt_tokens once a turn runs, so it is - # unaffected. - # Clamp the -1 "compression just ran, awaiting real usage" sentinel - # (conversation_compression.py) to 0 so the transitional turn reads as - # unknown (no gauge) instead of leaking context_used=-1. Matches the - # CLI status-bar path (cli.py _get_status_bar_snapshot). + # context_used is *current-window* occupancy. Never fall back to + # usage["total"] (cumulative lifetime) — an external engine without + # last_prompt_tokens then showed 1.9m/120k clamped to 100%. A falsy + # last_prompt_tokens emits NO gauge rather than a fabricated one; the -1 + # "compression just ran" sentinel is clamped to 0 for the same reason + # (matches cli.py _get_status_bar_snapshot). last_prompt = getattr(comp, "last_prompt_tokens", 0) or 0 if last_prompt < 0: last_prompt = 0 @@ -3309,21 +2713,16 @@ def _get_usage(agent) -> dict: usage["context_max"] = ctx_max usage["context_percent"] = max(0, min(100, round(last_prompt / ctx_max * 100))) usage["compressions"] = getattr(comp, "compression_count", 0) or 0 - # Cache-hit ratio + rolling latency/throughput for the TUI status bar. - # Mirrors the classic CLI bar (cli.py _get_status_bar_snapshot / PR #98250): - # hit = session_cache_read_tokens / session_prompt_tokens - # (CanonicalUsage.prompt_tokens = input + cache_read + cache_write) - # latency/tps read the deque(maxlen=10) history maintained per API call in - # agent/conversation_loop.py. Values are omitted (not fabricated) when no - # data exists — e.g. Codex app-server reports no latency, and a session - # with zero cache reads shows no hit% rather than an alarming 0. - try: + # Cache-hit ratio + rolling latency/tps (CLI status-bar parity): + # hit = cache_read / prompt_tokens (prompt = input + cache_read + cache_write); + # latency/tps read the per-call deque history from conversation_loop. + # Omitted, not fabricated, when there is no data (Codex reports no latency; + # zero cache reads shows no hit% rather than an alarming 0). + with contextlib.suppress(Exception): _prompt_total = int(getattr(agent, "session_prompt_tokens", 0) or 0) _cache_read = int(getattr(agent, "session_cache_read_tokens", 0) or 0) if _prompt_total > 0 and _cache_read > 0: usage["cache_hit_pct"] = max(0, min(100, round(_cache_read / _prompt_total * 100))) - except Exception: - pass try: _lhist = list(getattr(agent, "_api_latency_history", []) or []) _ohist = list(getattr(agent, "_api_output_history", []) or []) @@ -3345,20 +2744,16 @@ def _get_usage(agent) -> dict: # Live count of background/async subagents still running (delegate_task # batches + background single delegations). Mirrors the classic CLI status # bar's ⛓ indicator; sourced from the same async_delegation registry. - try: + with contextlib.suppress(Exception): from tools.async_delegation import active_count as _async_active_count usage["active_subagents"] = _async_active_count() - except Exception: - pass # Dev-only live credits-spent readout (L0 usage-aware-credits). Gated on # HERMES_DEV_CREDITS so the payload stays clean when the flag is off. if is_truthy_value(os.environ.get("HERMES_DEV_CREDITS")): - try: + with contextlib.suppress(Exception): spent = agent.get_credits_spent_micros() if spent is not None: usage["dev_credits_spent_micros"] = int(spent) - except Exception: - pass return usage @@ -3368,13 +2763,11 @@ def _probe_credentials(agent) -> str: ``no-key-required`` is a valid sentinel for keyless custom providers; only warn when the key is genuinely missing. """ - try: + with contextlib.suppress(Exception): key = getattr(agent, "api_key", "") or "" provider = getattr(agent, "provider", "") or "" if not key: return f"No API key configured for provider '{provider}'. First message will fail." - except Exception: - pass return "" @@ -3399,17 +2792,14 @@ def _probe_config_health(cfg: dict) -> str: if isinstance(display_cfg, dict): personality = str(display_cfg.get("personality", "") or "").strip().lower() if personality and personality not in {"default", "none", "neutral"}: - try: + with contextlib.suppress(Exception): from hermes_cli.personality import available_personalities - if personality not in available_personalities(cfg): warnings.append( f"`display.personality: {personality}` does not match any " "built-in or `agent.personalities` entry; personality " "overlay will be skipped." ) - except Exception: - pass _ = agent_cfg # retained for shape parity; built-ins exist without config return " ".join(warnings).strip() @@ -3417,16 +2807,14 @@ def _probe_config_health(cfg: dict) -> str: def _current_profile_name() -> str: try: from hermes_cli.profiles import get_active_profile_name - return get_active_profile_name() or "default" except Exception: return "default" -# Monotonic GUI<->backend contract version. The desktop app refuses to drive a -# backend reporting less than its required value (or none at all — a pre-GUI -# checkout), surfacing a one-click "update to align" prompt instead of failing -# cryptically downstream. Bump whenever the desktop's backend contract changes. +# Monotonic GUI<->backend contract version: the desktop refuses a backend +# reporting less (or none) with a one-click "update to align" prompt. Bump +# whenever the desktop's backend contract changes. # v2: adds the file.attach RPC (remote-gateway non-image file upload). # v3: adds approvals.mode config RPCs and session.info reconciliation. # v4: session.create fast=false is an explicit per-session normal-tier override. @@ -3459,15 +2847,12 @@ def _project_info_for_cwd(cwd: str) -> dict | None: return None try: from hermes_cli import projects_db as pdb - with pdb.connect_closing() as conn: project = pdb.project_for_path(conn, cwd) if project is None: return None return { - "id": project.id, - "slug": project.slug, - "name": project.name, + "id": project.id, "slug": project.slug, "name": project.name, "primary_path": project.primary_path, } except Exception: @@ -3483,46 +2868,35 @@ def _session_info(agent, session: dict | None = None) -> dict: break mirror = _metadata_mirror(session) cwd = _display_session_cwd(session) - session_key = str( - (session or {}).get("session_key") or getattr(agent, "session_id", "") or "" - ) - cfg_personality = ((_load_cfg().get("display") or {}).get("personality") or "") + session_key = str((session or {}).get("session_key") or getattr(agent, "session_id", "") or "") + cfg_personality = _display_cfg().get("personality") or "" personality = (session or {}).get("personality", cfg_personality) reasoning_config = getattr(agent, "reasoning_config", None) reasoning_effort = "" if isinstance(reasoning_config, dict): if reasoning_config.get("enabled") is False: - # Disabled must be distinguishable from unset ("" = provider - # default). Reporting "" here made the desktop adopt the empty - # value after the first turn, wiping its sticky "thinking off" - # pick and re-creating every later chat at the default effort. + # Disabled must differ from unset ("" = provider default), or the + # desktop adopts "" after the first turn and loses "thinking off". reasoning_effort = "none" else: reasoning_effort = str(reasoning_config.get("effort", "") or "") service_tier = getattr(agent, "service_tier", None) or mirror.get("service_tier") or "" - # Effective approval-bypass state — the same three sources that - # check_all_command_guards() ORs together: persistent config - # (approvals.mode=off), the process-scoped --yolo env, and the - # per-session flag. Reporting only the per-session flag here would lie to - # the desktop status bar (it would show YOLO "off" while approvals.mode=off - # silently auto-approves every dangerous command). + # Effective approval bypass = the same three sources check_all_command_guards() + # ORs: approvals.mode=off, the process --yolo env, the per-session flag. + # Reporting only the session flag would show YOLO "off" while config + # silently auto-approves every dangerous command. yolo = False approval_mode = "manual" try: from tools.approval import _YOLO_MODE_FROZEN, is_session_yolo_enabled - - session_yolo = ( - bool(is_session_yolo_enabled(session_key)) if session_key else False - ) + session_yolo = (bool(is_session_yolo_enabled(session_key)) if session_key else False) approval_mode = _load_approval_mode() yolo = bool(_YOLO_MODE_FROZEN) or session_yolo or approval_mode == "off" except Exception: yolo = False - # A model switch queued mid-turn (pending_model_switch) applies at the next - # turn start, so agent.model still reads the OLD model until then. Report the - # pending pick instead — it's the model the next turn will run, and it stops - # the end-of-turn settle from blipping the UI back to the old model before - # the switch lands. Cleared once _apply_pending_model_switch consumes it. + # A switch queued mid-turn applies at the next turn start, so agent.model + # still reads the OLD model; report the pending pick so the end-of-turn + # settle doesn't blip the UI back before the switch lands. pending_switch = (session or {}).get("pending_model_switch") or {} pending_model = str(pending_switch.get("display_model") or "").strip() pending_provider = str(pending_switch.get("display_provider") or "").strip() @@ -3535,7 +2909,6 @@ def _session_info(agent, session: dict | None = None) -> dict: if isinstance(inflight, dict) and inflight.get("started_at") else None ) - info: dict = { "model": pending_model or mirror.get("model", getattr(agent, "model", "")), "provider": pending_provider @@ -3562,65 +2935,45 @@ def _session_info(agent, session: dict | None = None) -> dict: "update_behind": None, "update_command": "", "usage": _session_usage_snapshot(session), - "profile_name": _response_profile_name( - Path(session["profile_home"]).name + "profile_name": ( + _response_profile_name(Path(session["profile_home"]).name) if isinstance(session, dict) and session.get("profile_home") - else None - ) - if isinstance(session, dict) and session.get("profile_home") - else _current_profile_name(), + else _current_profile_name() + ), } - try: + with contextlib.suppress(Exception): from hermes_cli import __version__, __release_date__ - info["version"] = __version__ info["release_date"] = __release_date__ - except Exception: - pass - if agent is not None and not (session or {}).get("_compute_host_active"): - try: + live_agent = agent is not None and not (session or {}).get("_compute_host_active") + if live_agent: + with contextlib.suppress(Exception): from model_tools import get_toolset_for_tool - info["tools"] = {} for t in getattr(agent, "tools", []) or []: name = t["function"]["name"] - info["tools"].setdefault(get_toolset_for_tool(name) or "other", []).append( - name - ) - except Exception: - pass - try: + info["tools"].setdefault(get_toolset_for_tool(name) or "other", []).append(name) + with contextlib.suppress(Exception): from hermes_cli.banner import get_available_skills - info["skills"] = get_available_skills() - except Exception: - pass try: from tools.mcp_tool import get_mcp_status - info["mcp_servers"] = get_mcp_status() except Exception: info["mcp_servers"] = [] - try: + with contextlib.suppress(Exception): info["system_prompt"] = ( mirror.get("system_prompt") if "system_prompt" in mirror else getattr(agent, "_cached_system_prompt", "") or "" ) - except Exception: - pass - try: + with contextlib.suppress(Exception): from hermes_cli.banner import get_update_result from hermes_cli.config import recommended_update_command - info["update_behind"] = get_update_result(timeout=0.5) info["update_command"] = recommended_update_command() - except Exception: - pass - if agent is not None and not (session or {}).get("_compute_host_active"): - warn = _probe_credentials(agent) - if warn: - info["credential_warning"] = warn + if live_agent and (warn := _probe_credentials(agent)): + info["credential_warning"] = warn return info @@ -3637,7 +2990,6 @@ def _tool_ctx(name: str, args: dict) -> str: """ try: from agent.display import build_tool_preview - return build_tool_preview(name, args, max_len=80) or "" except Exception: return "" @@ -3647,10 +2999,8 @@ def _emit_session_info_for_session(sid: str, session: dict) -> None: agent = session.get("agent") if agent is None and not _metadata_mirror(session): return - try: + with contextlib.suppress(Exception): _emit("session.info", sid, _session_info(agent, session)) - except Exception: - pass def broadcast_session_info() -> None: @@ -3673,28 +3023,16 @@ def broadcast_session_info() -> None: def _schedule_mcp_late_refresh(sid: str, agent) -> None: """Refresh a session's tool snapshot when MCP discovery lands late. - The agent snapshots ``agent.tools`` once at build time and never re-reads - the registry (run_agent/agent_init). ``_make_agent`` briefly joins the - background MCP discovery thread (``wait_for_mcp_discovery``, bounded by the - ``mcp_discovery_timeout`` config value, default 1.5s) so - already-spawning servers land in that snapshot — but a server that takes - longer than the bound to connect (common for an HTTP MCP server on first - connect) lands *after* the agent is built. Its tools are then absent from - both the agent and the banner for the whole session, even though the - classic CLI shows them (the CLI re-derives ``get_tool_definitions`` at - banner render time, which re-waits, so it picks them up). + The agent snapshots ``agent.tools`` once at build; ``_make_agent`` only + waits a bounded ``mcp_discovery_timeout`` (default 1.5s), so a slow server + (HTTP MCP on first connect) lands after the build and its tools are missing + for the whole session. A daemon waits for discovery to finish, then does + the same rebuild ``/reload-mcp`` performs and re-emits ``session.info``. - This schedules an off-critical-path daemon that waits for discovery to - finish, then rebuilds the snapshot and re-emits ``session.info`` so both - the agent's callable tools and the banner count catch up — the same - rebuild ``/reload-mcp`` performs, but automatic. - - Cache safety: the rebuild only runs while the session is still pre-first- - turn (no API call made yet → nothing cached to invalidate). If the user - has already sent a message, we leave the snapshot frozen rather than - invalidate the prompt cache mid-conversation — those late tools then - require an explicit ``/reload-mcp`` (which gates on user consent), exactly - as today. No-op when discovery already finished before the agent build. + Cache safety: the rebuild runs only while the session is pre-first-turn + (nothing cached to invalidate). Once a message was sent the snapshot stays + frozen — late tools then need an explicit, consent-gated ``/reload-mcp``. + No-op when discovery already finished before the build. """ try: from tui_gateway.entry import mcp_discovery_in_flight, join_mcp_discovery @@ -3722,13 +3060,10 @@ def _schedule_mcp_late_refresh(sid: str, agent) -> None: return try: from tools.mcp_tool import refresh_agent_mcp_tools - added = refresh_agent_mcp_tools(agent, quiet_mode=True) except Exception as exc: logger.warning( - "Late MCP refresh: tool snapshot rebuild failed for %s: %s", - sid, - exc, + "Late MCP refresh: tool snapshot rebuild failed for %s: %s", sid, exc, ) return # No new tools landed (discovery added nothing) → don't churn the client. @@ -3738,9 +3073,7 @@ def _schedule_mcp_late_refresh(sid: str, agent) -> None: # Emit outside the lock — write_json must not block under _sessions_lock. _emit("session.info", sid, info) threading.Thread( - target=_wait_then_refresh, - name=f"tui-mcp-late-refresh-{sid}", - daemon=True, + target=_wait_then_refresh, name=f"tui-mcp-late-refresh-{sid}", daemon=True, ).start() @@ -3762,14 +3095,9 @@ def _resolve_runtime_with_fallback( """ from hermes_cli.auth import AuthError from hermes_cli.runtime_provider import resolve_runtime_provider - kwargs = resolve_kwargs or {} try: - return _RuntimeFallbackResolution( - resolve_runtime_provider(**kwargs), - None, - False, - ) + return _RuntimeFallbackResolution(resolve_runtime_provider(**kwargs), None, False) except AuthError as primary_exc: fb_chain = _load_fallback_model() or [] for entry in fb_chain: @@ -3781,11 +3109,7 @@ def _resolve_runtime_with_fallback( continue try: from hermes_cli.fallback_config import resolve_entry_api_key - - fb_kwargs: dict = { - "requested": fb_provider, - "target_model": fb_model, - } + fb_kwargs: dict = {"requested": fb_provider, "target_model": fb_model} if entry.get("base_url"): fb_kwargs["explicit_base_url"] = entry["base_url"] fb_api_key = resolve_entry_api_key(entry) @@ -3793,12 +3117,9 @@ def _resolve_runtime_with_fallback( fb_kwargs["explicit_api_key"] = fb_api_key runtime = resolve_runtime_provider(**fb_kwargs) import logging - logging.getLogger(__name__).warning( - "Primary auth failed (%s), falling back to %s model %s", - primary_exc, - fb_provider, - fb_model, + "Primary auth failed (%s), falling back to %s model %s", primary_exc, + fb_provider, fb_model, ) return _RuntimeFallbackResolution(runtime, fb_model, True) except Exception: @@ -3806,65 +3127,91 @@ def _resolve_runtime_with_fallback( raise -def _make_agent( - sid: str, - key: str, - session_id: str | None = None, - session_db=None, - model_override: dict | str | None = None, - provider_override: str | None = None, - reasoning_config_override: dict | None = None, - service_tier_override: str | None = None, - platform_override: str | None = None, - context_cwd_is_launch_artifact: bool | None = None, -): - # AC-4 test seam: dead unless explicitly armed by the isolated certify - # harness. Both inline and compute-host paths construct through _make_agent, - # leaving the process boundary as the only experimental variable. - from tui_gateway.synthetic_turn import maybe_build_synthetic_agent +def _resolve_agent_model_runtime(model_override, provider_override) -> tuple[str, dict]: + """Resolve (model, runtime) for a new agent. + A per-session override (prior in-session /model switch, or the persisted + runtime of a resumed row) wins over global config/env resolution. Rows + persisted before the custom-provider identity fix stored the resolved + provider "custom", which no named ``providers:`` entry matches — recover + the entry identity from the persisted base_url (falling back to the + configured provider) or the rebuild surfaces as "No LLM provider + configured". Persisted base_url/api_key/api_mode are honored only while + the original runtime is used; they must not leak into a fallback pair. + """ + if isinstance(model_override, dict) and model_override.get("model"): + model = str(model_override.get("model") or "") + requested_provider = model_override.get("provider") or provider_override or None + override_base_url = model_override.get("base_url") + resolve_kwargs = {} + if str(requested_provider or "").strip().lower() == "custom": + from hermes_cli.runtime_provider import canonical_custom_identity + recovered = canonical_custom_identity(base_url=override_base_url or None, model=model or None) + if recovered: + requested_provider = recovered + if override_base_url: + # Failing identity recovery, still hand the base_url to the + # direct-alias branch so pool/env credentials resolve for it. + resolve_kwargs["explicit_base_url"] = override_base_url + resolve_kwargs["requested"] = requested_provider + resolve_kwargs["target_model"] = model or None + overrides = { + "base_url": override_base_url, "api_key": model_override.get("api_key"), + "api_mode": model_override.get("api_mode"), + } + else: + model, requested_provider = _resolve_startup_runtime() + if isinstance(model_override, str) and model_override: + model = model_override + if provider_override: + requested_provider = provider_override + resolve_kwargs = {"requested": requested_provider, "target_model": model or None} + overrides = {} + resolution = _resolve_runtime_with_fallback(resolve_kwargs) + runtime = resolution.runtime + if resolution.used_fallback: + if not resolution.selected_model: + raise RuntimeError("Auth fallback resolved without a model") + return resolution.selected_model, runtime + for k, v in overrides.items(): + if v: + runtime[k] = v + return model, runtime + + +def _make_agent( + sid: str, key: str, session_id: str | None = None, session_db=None, + model_override: dict | str | None = None, provider_override: str | None = None, + reasoning_config_override: dict | None = None, service_tier_override: str | None = None, + platform_override: str | None = None, context_cwd_is_launch_artifact: bool | None = None, +): + # AC-4 test seam: dead unless armed by the isolated certify harness. + from tui_gateway.synthetic_turn import maybe_build_synthetic_agent synthetic = maybe_build_synthetic_agent(session_id or key, model_override) if synthetic is not None: return synthetic - from run_agent import AIAgent - # MCP tool discovery runs in a background daemon thread at startup so a - # dead server can't freeze the shell. The agent snapshots its tool list - # once here and never re-reads it, so briefly wait for in-flight discovery - # to land before building — bounded, so a slow/dead server still can't - # block. Dashboard /api/ws uses hermes_cli.mcp_startup; TUI stdio keeps - # its existing tui_gateway.entry-owned thread. - try: - from hermes_cli.mcp_startup import wait_for_mcp_discovery - - wait_for_mcp_discovery() - except Exception: - pass - try: - from tui_gateway.entry import wait_for_mcp_discovery - - wait_for_mcp_discovery() - except Exception: - pass - + # MCP discovery runs in a background daemon thread so a dead server can't + # freeze the shell; the agent snapshots its tool list once, so briefly + # (bounded) wait for in-flight discovery. Dashboard /api/ws uses + # hermes_cli.mcp_startup; TUI stdio keeps the tui_gateway.entry thread. + for _mod in ("hermes_cli.mcp_startup", "tui_gateway.entry"): + with contextlib.suppress(Exception): + importlib.import_module(_mod).wait_for_mcp_discovery() cfg = _load_cfg() from hermes_cli.config import resolve_ephemeral_system_prompt_from_config - system_prompt = resolve_ephemeral_system_prompt_from_config(cfg) startup_skills = _parse_tui_skills_env() if startup_skills: from agent.skill_commands import build_preloaded_skills_prompt - skills_prompt, loaded_skills, missing_skills = build_preloaded_skills_prompt( - startup_skills, - task_id=session_id or key, + startup_skills, task_id=session_id or key, ) if missing_skills: missing_display = ", ".join(missing_skills) - # Degrade gracefully when some skills loaded; only hard-fail when - # every requested skill is missing. Mirrors cli.py — a typo'd skill - # name should not crash the worker and auto-block the Kanban task. + # Hard-fail only when EVERY requested skill is missing (cli.py + # parity): a typo'd name must not auto-block the Kanban task. if loaded_skills: logger.warning( "Unknown skill(s) requested, skipping: %s. " @@ -3879,72 +3226,7 @@ def _make_agent( system_prompt = "\n\n".join( part for part in (system_prompt, skills_prompt) if part ).strip() - # Prefer a per-session model override (set by a prior in-session /model - # switch) over global config/env resolution. Resume-time stored sessions may - # also pass scalar model/provider/runtime knobs from the persisted DB row. - if isinstance(model_override, dict) and model_override.get("model"): - model = str(model_override.get("model") or "") - requested_provider = model_override.get("provider") or provider_override or None - override_base_url = model_override.get("base_url") - override_api_key = model_override.get("api_key") - override_api_mode = model_override.get("api_mode") - resolve_kwargs = {} - if str(requested_provider or "").strip().lower() == "custom": - # Session rows persisted before the custom-provider identity fix - # (see _runtime_model_config) stored the resolved provider - # "custom", which _get_named_custom_provider cannot match back to - # a named ``providers:`` / ``custom_providers:`` entry — the - # rebuild then either raised auth_unavailable, silently resolved - # placeholder credentials against the patched-back base_url, or - # (when no base_url was stored) routed to the OpenRouter default - # with no key, surfacing as "No LLM provider configured". Recover - # the entry identity from the persisted base_url, falling back to - # the configured provider when the override carries no base_url - # (the recurring Desktop/TUI regression vector). - from hermes_cli.runtime_provider import canonical_custom_identity - - recovered = canonical_custom_identity( - base_url=override_base_url or None, model=model or None - ) - if recovered: - requested_provider = recovered - if override_base_url: - # Failing identity recovery, still hand the base_url to the - # direct-alias branch so pool/env credentials resolve for it. - resolve_kwargs["explicit_base_url"] = override_base_url - resolve_kwargs["requested"] = requested_provider - resolve_kwargs["target_model"] = model or None - resolution = _resolve_runtime_with_fallback(resolve_kwargs) - runtime = resolution.runtime - if resolution.used_fallback: - if not resolution.selected_model: - raise RuntimeError("Auth fallback resolved without a model") - model = resolution.selected_model - else: - # The switch already resolved concrete credentials/endpoint; honor - # persisted overrides only while using that original runtime. They - # must not leak into a different fallback provider/model pair. - if override_base_url: - runtime["base_url"] = override_base_url - if override_api_key: - runtime["api_key"] = override_api_key - if override_api_mode: - runtime["api_mode"] = override_api_mode - else: - model, requested_provider = _resolve_startup_runtime() - if isinstance(model_override, str) and model_override: - model = model_override - if provider_override: - requested_provider = provider_override - resolution = _resolve_runtime_with_fallback({ - "requested": requested_provider, - "target_model": model or None, - }) - runtime = resolution.runtime - if resolution.used_fallback: - if not resolution.selected_model: - raise RuntimeError("Auth fallback resolved without a model") - model = resolution.selected_model + model, runtime = _resolve_agent_model_runtime(model_override, provider_override) _pr = _load_provider_routing() agent = AIAgent( model=model, @@ -3957,11 +3239,7 @@ def _make_agent( acp_args=runtime.get("args"), credential_pool=runtime.get("credential_pool"), quiet_mode=True, - # verbose_logging controls DEBUG-level agent logging; it is intentionally - # independent of tool_progress_mode (which only controls per-tool - # display detail). See cli.py PR (decoupling fix) for the matching - # change on the classic CLI side. - verbose_logging=False, + verbose_logging=False, # DEBUG agent logging; independent of tool_progress_mode reasoning_config=( reasoning_config_override if reasoning_config_override is not None @@ -3973,9 +3251,7 @@ def _make_agent( else _load_service_tier() ), enabled_toolsets=_load_enabled_toolsets(_resolve_agent_platform(platform_override)), - # OpenRouter provider-routing prefs (config.yaml `provider_routing`). - # Mirrors the messaging gateway + CLI so the desktop/TUI honors the same - # routing instead of letting OpenRouter pick providers at random. + # OpenRouter provider_routing prefs (gateway + CLI parity). providers_allowed=_pr.get("only"), providers_ignored=_pr.get("ignore"), providers_order=_pr.get("order"), @@ -3996,25 +3272,14 @@ def _make_agent( if context_cwd_is_launch_artifact is None: with _sessions_lock: context_session = _sessions.get(sid) - context_cwd_is_launch_artifact = _context_cwd_is_launch_artifact( - context_session - ) - agent._context_cwd_is_launch_artifact = bool( - context_cwd_is_launch_artifact - ) + context_cwd_is_launch_artifact = _context_cwd_is_launch_artifact(context_session) + agent._context_cwd_is_launch_artifact = bool(context_cwd_is_launch_artifact) return agent def _init_session( - sid: str, - key: str, - agent, - history: list, - cols: int = 80, - cwd: str | None = None, - session_db=None, - source: str | None = None, - profile_home: str | None = None, + sid: str, key: str, agent, history: list, cols: int = 80, cwd: str | None = None, + session_db=None, source: str | None = None, profile_home: str | None = None, explicit_cwd: bool = False, ): now = time.time() @@ -4040,16 +3305,14 @@ def _init_session( "tool_progress_mode": _load_tool_progress_mode(), "edit_snapshots": {}, "tool_started_at": {}, - # Profile-scoped HERMES_HOME for app-global remote mode; None = - # launch profile. SessionBranch copies the parent's value so the - # child stays on the same state.db. + # Profile-scoped HERMES_HOME (None = launch profile); SessionBranch + # copies the parent's so the child stays on the same state.db. "profile_home": profile_home, - # Per-session model override set by an in-session /model switch. - # Honored on rebuild (/new, resume) so a switch in THIS session + # In-session /model switch, honored on rebuild (/new, resume) so it # never leaks into siblings via process-global env vars. "model_override": None, - # Pin async event emissions to whichever transport created the - # session (stdio for Ink, JSON-RPC WS for the dashboard sidebar). + # Async events go to the transport that created the session + # (stdio for Ink, JSON-RPC WS for the dashboard sidebar). "transport": current_transport() or _stdio_transport, } _session_todo_state(_sessions[sid]) @@ -4061,11 +3324,9 @@ def _init_session( db = _open_profile_session_db(profile_home) _init_owns_db = True except Exception: - # FAIL CLOSED — same class as the deferred-build bind: a - # named-profile session must never read/write the launch - # ``state.db``. Skip the cwd hydration/persist (the row lands on - # the agent's own lazy-create once the store recovers) rather - # than writing this session's row into the wrong profile's store. + # FAIL CLOSED (same class as the deferred-build bind): a named-profile + # session must never touch the launch state.db — skip cwd hydration + # (the row lands on the agent's own lazy-create once the store recovers). logger.warning( "profile session store unavailable for %s — skipping cwd " "hydration instead of touching the launch state.db", @@ -4086,53 +3347,17 @@ def _init_session( try: _cwd = _sessions[sid]["cwd"] if hasattr(db, "update_session_cwd"): - _persist_session_cwd_and_schedule_git_meta( - _sessions[sid], _cwd, db=db - ) + _persist_session_cwd_and_schedule_git_meta(_sessions[sid], _cwd, db=db) except Exception: - logger.debug( - "failed to persist resumed session cwd", exc_info=True - ) + logger.debug("failed to persist resumed session cwd", exc_info=True) finally: if _init_owns_db and db is not None: - try: + with contextlib.suppress(Exception): db.close() - except Exception: - pass _register_session_cwd(_sessions[sid]) - # No eager slash-worker pre-warm — the session dict already carries - # slash_worker=None and slash.exec builds one on demand. See the - # deferred-build path in _start_agent_build for the full rationale - # (per-worker MCP fleets accumulating across retained sessions). - try: - from tools.approval import register_gateway_notify, load_permanent_allowlist - - register_gateway_notify(key, lambda data: _emit_approval_request(sid, data)) - load_permanent_allowlist() - except Exception: - pass - # Surface the self-improvement background review's "💾 …" summary as a - # review.summary event so Ink can render it as a persistent system line - # in the transcript. In the CLI path this message is printed via - # prompt_toolkit; the TUI has no equivalent print surface, so without - # this callback the review would write the skill/memory change silently. - try: - agent.background_review_callback = lambda message, _sid=sid: _emit( - "review.summary", _sid, {"text": str(message)} - ) - # Honor display.memory_notifications (off | on | verbose) like the - # messaging gateway and CLI do — otherwise the review always behaved as - # "on" on the TUI/desktop and a user who set "off" was ignored. - agent.memory_notifications = _load_memory_notifications() - except Exception: - # Bare AIAgents that don't expose the attribute (unlikely, but keep - # session startup resilient). - pass - _wire_callbacks(sid) - with _sessions_lock: - if sid in _sessions: - _sessions[sid]["_notif_stop"] = _start_notification_poller(sid, _sessions[sid]) - _notify_session_boundary("on_session_reset", key, _session_source(_sessions.get(sid, {}))) + # No eager slash-worker pre-warm (see _start_agent_build). + _wire_session_agent(sid, key, agent) + _start_session_services(sid, key, _sessions.get(sid, {})) _emit("session.info", sid, _session_info(agent, _sessions.get(sid, {}))) _schedule_mcp_late_refresh(sid, agent) @@ -4160,22 +3385,13 @@ def _resolve_checkpoint_hash(mgr, cwd: str, ref: str) -> str: def _lazy_resume_info( - cwd: str, - *, - model: str = "", - provider: str = "", - profile: str | None = None, + cwd: str, *, model: str = "", provider: str = "", profile: str | None = None, ) -> dict: """session.info for a not-yet-built session (the shape session.create returns). tools/skills land later when the deferred build emits session.info.""" info = { - "cwd": cwd, - "branch": _git_branch_for_cwd(cwd), - "project": _project_info_for_cwd(cwd), - "model": model or _resolve_model(), - "tools": {}, - "skills": {}, - "lazy": True, + "cwd": cwd, "branch": _git_branch_for_cwd(cwd), "project": _project_info_for_cwd(cwd), + "model": model or _resolve_model(), "tools": {}, "skills": {}, "lazy": True, "desktop_contract": DESKTOP_BACKEND_CONTRACT, "profile_name": _response_profile_name(profile), } @@ -4185,58 +3401,28 @@ def _lazy_resume_info( def _deferred_session_record( - session_key: str, - *, - cols: int, - cwd: str, - history: list, - lease, - source: str = "tui", - close_on_disconnect: bool = False, - display_history_prefix: list | None = None, - profile_home: Path | None = None, - lazy: bool = False, - model_override=None, - resume_runtime_overrides: dict | None = None, - todo_state: dict | None = None, + session_key: str, *, cols: int, cwd: str, history: list, lease, source: str = "tui", + close_on_disconnect: bool = False, display_history_prefix: list | None = None, + profile_home: Path | None = None, lazy: bool = False, model_override=None, + resume_runtime_overrides: dict | None = None, todo_state: dict | None = None, explicit_cwd: bool = False, ) -> dict: """A live-session record whose AIAgent is built later (lazy watch / cold resume) — _init_session's shape minus the agent.""" now = time.time() return { - "agent": None, - "agent_error": None, - "agent_ready": threading.Event(), - "attached_images": [], - "close_on_disconnect": close_on_disconnect, - "active_session_lease": lease, - "cols": cols, - "created_at": now, - "cwd": cwd, - "display_history_prefix": display_history_prefix or [], - "edit_snapshots": {}, - "explicit_cwd": bool(explicit_cwd), - "history": history, - "history_lock": threading.Lock(), - "history_version": 0, - "image_counter": 0, - "inflight_turn": None, - "last_active": now, - "lazy": lazy, - "model_override": model_override, + "agent": None, "agent_error": None, "agent_ready": threading.Event(), "attached_images": [], + "close_on_disconnect": close_on_disconnect, "active_session_lease": lease, "cols": cols, + "created_at": now, "cwd": cwd, "display_history_prefix": display_history_prefix or [], + "edit_snapshots": {}, "explicit_cwd": bool(explicit_cwd), "history": history, + "history_lock": threading.Lock(), "history_version": 0, "image_counter": 0, + "inflight_turn": None, "last_active": now, "lazy": lazy, "model_override": model_override, "pending_title": None, "profile_home": str(profile_home) if profile_home is not None else None, - "resume_runtime_overrides": resume_runtime_overrides, - "resume_session_id": session_key, - "running": False, - "session_key": session_key, - "show_reasoning": _load_show_reasoning(), - "slash_worker": None, - "source": source, - "tool_progress_mode": _load_tool_progress_mode(), - "tool_started_at": {}, - "todo_state": todo_state, + "resume_runtime_overrides": resume_runtime_overrides, "resume_session_id": session_key, + "running": False, "session_key": session_key, "show_reasoning": _load_show_reasoning(), + "slash_worker": None, "source": source, "tool_progress_mode": _load_tool_progress_mode(), + "tool_started_at": {}, "todo_state": todo_state, "transport": current_transport() or _stdio_transport, } @@ -4279,11 +3465,9 @@ def _claim_or_reuse_live( with _sessions_lock: _sessions[sid] = record _register_session_cwd(_sessions[sid]) - # A fresh runtime was minted for this stored session id: the new sid - # has no pending reap, but a PRIOR runtime for the same stored id may - # still be sentinel-parked with a reap Timer armed. Cancel + finalize - # those quietly so the reap doesn't later broadcast session.reclaimed - # for a session the client just re-resumed (auto-re-resume storm). + # A PRIOR runtime for this stored id may still be sentinel-parked with + # a reap Timer armed; cancel + finalize it quietly so the reap doesn't + # broadcast session.reclaimed for a just-re-resumed session (storm). _cancel_ws_orphan_reap(sid) stale = _claim_parked_runtimes(session_key, keep_sid=sid, profile_home=profile_home) # Slow finalization work stays OUTSIDE _session_resume_lock (see @@ -4337,9 +3521,7 @@ def _finalize_superseded_runtimes(stale: list[tuple[str, dict]]) -> None: try: _teardown_popped_session(popped, end_reason="superseded_by_resume") except Exception: - logger.exception( - "superseded runtime teardown failed sid=%s", old_sid - ) + logger.exception("superseded runtime teardown failed sid=%s", old_sid) def _schedule_agent_build(sid: str, delay: float = 0.05) -> None: @@ -4350,15 +3532,12 @@ def _schedule_agent_build(sid: str, delay: float = 0.05) -> None: session = _sessions.get(sid) if session is not None: _start_agent_build(sid, session) - timer = threading.Timer(delay, _run) timer.daemon = True timer.start() -def _schedule_resume_hydration( - sid: str, stored_id: str, db, *, close_db: bool = False -) -> None: +def _schedule_resume_hydration(sid: str, stored_id: str, db, *, close_db: bool = False) -> None: """Load a cold resume's transcript off the JSON-RPC response path.""" def _run() -> None: @@ -4366,22 +3545,13 @@ def _schedule_resume_hydration( try: if session is None: return - _emit( - "session.resume_progress", - sid, - {"phase": "history", "status": "loading"}, - ) + _emit("session.resume_progress", sid, {"phase": "history", "status": "loading"}) db.reopen_session(stored_id) from hermes_state import SessionResumeTooLargeError - # The deferred resume is guarded tip-only (session.resume): the - # display transcript is REST-paginated, so the ancestor prefix is - # an in-memory convenience (rewind ordinal translation, branch - # snapshots), not a requirement. Materialize the full lineage only - # while it fits sessions.max_resume_messages; past that, hydrate - # the tip alone instead of loading the runaway lineage the guard - # exists to keep out of memory (the omit_messages resume already - # runs with an empty prefix, so this is an existing shape). + # The ancestor prefix is an in-memory convenience (the transcript + # is REST-paginated): materialize the full lineage only while it + # fits sessions.max_resume_messages, else hydrate the tip alone. prefix_fits = True guard = getattr(db, "assert_resume_safe", None) if callable(guard): @@ -4406,7 +3576,6 @@ def _schedule_resume_hydration( display_history = raw_history prefix = [] history = sanitize_replay_history(raw_history) - if _sessions.get(sid) is not session: return with session["history_lock"]: @@ -4421,13 +3590,8 @@ def _schedule_resume_hydration( session["todo_state"] = todo_state session["resume_history_ready"].set() _emit( - "session.resume_progress", - sid, - { - "message_count": len(display_history), - "phase": "history", - "status": "complete", - }, + "session.resume_progress", sid, + {"message_count": len(display_history), "phase": "history", "status": "complete"}, ) _maybe_schedule_auto_continue(sid, session, stored_id) _start_agent_build(sid, session) @@ -4441,8 +3605,7 @@ def _schedule_resume_hydration( session["resume_history_ready"].set() session["agent_ready"].set() _emit( - "session.resume_progress", - sid, + "session.resume_progress", sid, {"message": message, "phase": "history", "status": "failed"}, ) _emit("error", sid, {"message": message}) @@ -4457,7 +3620,6 @@ def _schedule_resume_hydration( db.close() except Exception: logger.debug("failed to close resume db for %s", sid, exc_info=True) - threading.Thread(target=_run, daemon=True).start() @@ -4493,12 +3655,10 @@ def _message_preview(history: list) -> str: def _session_live_title(session: dict, key: str) -> str: title = str(session.get("pending_title") or "").strip() - try: + with contextlib.suppress(Exception): with _session_db(session) as db: if db is not None: title = str(db.get_session_title(key) or title or "").strip() - except Exception: - pass return title @@ -4518,36 +3678,26 @@ def _session_live_item(sid: str, session: dict, current_sid: str = "") -> dict: preview = " ".join(str(preview).split())[:160] now = time.time() return { - "current": sid == current_sid, - "id": sid, + "current": sid == current_sid, "id": sid, "last_active": float(session.get("last_active") or session.get("created_at") or now), "message_count": len(history), - "model": str(getattr(agent, "model", "") or _resolve_model()), - "preview": preview, - "session_key": key, - "started_at": float(session.get("created_at") or now), - "status": status, + "model": str(getattr(agent, "model", "") or _resolve_model()), "preview": preview, + "session_key": key, "started_at": float(session.get("created_at") or now), "status": status, "title": _session_live_title(session, key), } def _session_lookup_key(session: dict, *, fallback: str = "") -> str: agent = session.get("agent") - return str( - getattr(agent, "session_id", None) - or session.get("session_key") - or fallback - or "" - ) + return str(getattr(agent, "session_id", None) or session.get("session_key") or fallback or "") def _find_live_session_by_key( session_key: str, profile_home=_ANY_PROFILE ) -> tuple[str, dict] | None: - # Stored session ids are timestamp-based and can legitimately exist in more - # than one profile's store, so a bare-id match can hand profile B's resume - # profile A's live runtime (#100029). Profile-aware callers pass the home - # they resolved; the match must then be on (profile_home, session_key). + # Timestamp-based stored ids can exist in several profiles' stores; a + # bare-id match would hand profile B's resume profile A's runtime, so + # profile-aware callers match on (profile_home, session_key). for sid, session in list(_sessions.items()): if session.get("_finalized"): continue @@ -4562,13 +3712,9 @@ def _fallback_session_info(session: dict) -> dict: agent = session.get("agent") if agent is not None: return _session_info(agent) - # The SESSION's own workspace, not the gateway's launch directory. Reporting - # `_default_session_cwd()` here told a lazily-resumed session's client that - # its workspace was wherever the gateway process happened to start, so the - # desktop Files pane painted the wrong project even after the renderer - # rebound correctly (#71254). `branch` is always emitted ("" outside a git - # repo) so a client can clear a stale label instead of retaining it — the - # same contract `_lazy_session_info` above already follows. + # The SESSION's own workspace, not the gateway launch dir (that painted the + # wrong project in the desktop Files pane). `branch` is always emitted ("" + # outside git) so a client clears a stale label — same as _lazy_session_info. cwd = _session_cwd(session) return { "cwd": cwd, @@ -4578,18 +3724,13 @@ def _fallback_session_info(session: dict) -> dict: "model": _resolve_model(), "skills": {}, "tools": {}, - # A lazy session (agent not built yet) is still served by *this* backend, - # so it must advertise the current contract. Desktop feeds this straight - # into reportBackendContract(); a missing field is read as contract 0 and - # a current backend is falsely flagged "out of date" (#68392). The sibling - # session.create shape (_lazy_resume_info) already carries it (#36112). + # A lazy session is still served by THIS backend: a missing contract + # field reads as 0 and flags a current backend "out of date". "desktop_contract": DESKTOP_BACKEND_CONTRACT, } -def _reconcile_display_with_live( - db_display: list[dict], in_memory: list[dict] -) -> list[dict]: +def _reconcile_display_with_live(db_display: list[dict], in_memory: list[dict]) -> list[dict]: """Merge the persisted DISPLAY lineage with the in-memory live history. Two projections of the same session that each hold something the other @@ -4618,7 +3759,6 @@ def _reconcile_display_with_live( def _key(msg: dict) -> tuple: return (msg.get("role"), _coerce_message_text(msg.get("content"))) - anchor = _key(db_display[-1]) last_shared = -1 for idx, msg in enumerate(in_memory): @@ -4634,19 +3774,13 @@ def _reconcile_display_with_live( def _live_visible_history(session: dict, db, in_memory_fallback: list[dict]) -> list[dict]: """Return the user-visible DISPLAY projection for a live/warm session. - Serving the raw in-memory *model* history for the user-visible payload - dropped model-invisible rows (verification candidates persisted by #65919) - whenever a warm/live session was reused, while the eager ``session.resume`` - path (which reads the verbatim display lineage) still showed them — the two - payloads disagreed about the same session, which is the cross-session - "substantive answer vanishes on switch" class of bug. - - This reconciles the persisted display lineage (candidate-inclusive, via - ``get_messages_as_conversation(..., include_ancestors=True)`` — the same - read the eager resume and REST paths use) with the fresh in-memory tail, so - all surfaces agree while a not-yet-flushed turn is still shown. Falls back to - the in-memory history when the DB/session_key is unavailable or the DB read - fails. + The raw in-memory *model* history lacks model-invisible rows (verification + candidates) that the eager ``session.resume`` display lineage shows, so the + two payloads disagreed ("substantive answer vanishes on switch"). Reconcile + the persisted display lineage (``get_messages_as_conversation(..., + include_ancestors=True)``, same read as resume/REST) with the fresh + in-memory tail; fall back to in-memory when the DB/session_key is + unavailable or the read fails. """ key = session.get("session_key") if db is not None and key: @@ -4668,24 +3802,17 @@ def _live_visible_history(session: dict, db, in_memory_fallback: list[dict]) -> def _live_session_payload( - sid: str, - session: dict, - *, - cols: int | None = None, - touch: bool = False, - transport: Transport | None = None, - omit_messages: bool = False, + sid: str, session: dict, *, cols: int | None = None, touch: bool = False, + transport: Transport | None = None, omit_messages: bool = False, ) -> dict: with session["history_lock"]: if cols is not None: session["cols"] = cols if transport is not None: session["transport"] = transport - # Track every transport that has shown this session (multi-window: - # pop-out windows each resume the same sid). The last viewer - # becomes the transport on the disconnect path so closing a - # pop-out re-binds the session to a still-open window instead of - # stranding it on the drop sentinel (#83716). + # Every transport that has shown this session (pop-out windows + # resume the same sid); the last viewer becomes the transport on + # disconnect instead of stranding it on the drop sentinel. viewers = session.setdefault("viewers", {}) viewers[transport] = time.time() if transport is not _detached_ws_transport: @@ -4706,28 +3833,21 @@ def _live_session_payload( if isinstance(inflight_turn, dict) and inflight_turn.get("started_at") else None ) - # Prefer the persisted display lineage (candidate-inclusive) so this payload - # matches the eager session.resume + REST transcript. Use the session's - # profile-aware DB (not launch ``_get_db()``): app-global remote profile - # sessions store candidates in ``profile_home``/state.db, and reading the - # launch DB here falls back to collapsed in-memory history and drops them. - # The DB has its own lock, so read it outside the session history lock. - # ``omit_messages`` skips the DB read entirely (callers only need counts / - # status); keep that fast path from main. + # Persisted display lineage (candidate-inclusive) so this matches the eager + # resume + REST transcript; via the session's profile-aware DB, not the + # launch ``_get_db()`` (remote-profile candidates live in profile_home). + # The DB has its own lock — read outside the history lock. ``omit_messages`` + # skips the read entirely (fast path for counts/status). if omit_messages: history = in_memory_history else: with _session_db(session) as db: history = _live_visible_history(session, db, in_memory_history) payload = { - "info": _fallback_session_info(session), - "message_count": len(history), + "info": _fallback_session_info(session), "message_count": len(history), "messages": [] if omit_messages else _history_to_messages(history), - "messages_omitted": omit_messages, - "running": running, - "turn_started_at": turn_started_at, - "session_id": sid, - "session_key": _session_lookup_key(session, fallback=sid), + "messages_omitted": omit_messages, "running": running, "turn_started_at": turn_started_at, + "session_id": sid, "session_key": _session_lookup_key(session, fallback=sid), "started_at": float(session.get("created_at") or time.time()), "status": _session_live_status(sid, session), } @@ -4769,7 +3889,6 @@ def _pet_frame_counts(spritesheet) -> dict: """ try: from agent.pet import render - return render.state_frame_counts(str(spritesheet)) except Exception: # noqa: BLE001 - cosmetic, never break the surface return {} @@ -4795,11 +3914,7 @@ def _pet_payload_cache_key(pet, *, scale: float) -> tuple | None: except Exception: # noqa: BLE001 return None return ( - str(pet.spritesheet), - stat.st_mtime_ns, - stat.st_size, - pet.slug, - pet.display_name, + str(pet.spritesheet), stat.st_mtime_ns, stat.st_size, pet.slug, pet.display_name, round(scale, 4), ) @@ -4820,9 +3935,7 @@ def _pet_row_frame_counts(spritesheet) -> dict: """Real frame count per concrete spritesheet row name.""" try: from PIL import Image - from agent.pet import constants, render - with Image.open(spritesheet) as opened: image = opened.convert("RGBA") cols = max(1, image.width // constants.FRAME_W) @@ -4847,10 +3960,8 @@ def _pet_row_frame_counts(spritesheet) -> dict: def _pet_config_scale() -> float: """Configured ``display.pet.scale`` (or the engine default), never raises.""" from agent.pet import constants - try: from hermes_cli.config import load_config - cfg = load_config() display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} @@ -4866,33 +3977,24 @@ def _pet_sprite_payload(pet, *, scale: float) -> dict: preview) so both feed the desktop canvas / TUI from one shape. """ import base64 - from agent.pet import constants - cache_key = _pet_payload_cache_key(pet, scale=scale) if cache_key is not None: with _pet_payload_cache_lock: cached = _pet_payload_cache.get(cache_key) if cached is not None: return _clone_pet_payload(cached) - raw = pet.spritesheet.read_bytes() suffix = pet.spritesheet.suffix.lower() mime = "image/png" if suffix == ".png" else "image/webp" payload = { - "slug": pet.slug, - "displayName": pet.display_name, - "mime": mime, + "slug": pet.slug, "displayName": pet.display_name, "mime": mime, "spritesheetBase64": base64.standard_b64encode(raw).decode("ascii"), - "spritesheetRevision": _pet_sheet_revision(pet.spritesheet), - "frameW": constants.FRAME_W, - "frameH": constants.FRAME_H, - "framesPerState": constants.FRAMES_PER_STATE, + "spritesheetRevision": _pet_sheet_revision(pet.spritesheet), "frameW": constants.FRAME_W, + "frameH": constants.FRAME_H, "framesPerState": constants.FRAMES_PER_STATE, "framesByState": _pet_frame_counts(pet.spritesheet), - "framesByRow": _pet_row_frame_counts(pet.spritesheet), - "loopMs": constants.LOOP_MS, - "scale": scale, - "stateRows": _pet_state_rows(pet.spritesheet), + "framesByRow": _pet_row_frame_counts(pet.spritesheet), "loopMs": constants.LOOP_MS, + "scale": scale, "stateRows": _pet_state_rows(pet.spritesheet), } if cache_key is not None: with _pet_payload_cache_lock: @@ -4905,16 +4007,13 @@ def _pet_sprite_payload(pet, *, scale: float) -> dict: def _pet_active_selection(): """Resolve configured active pet + scale from config.""" from agent.pet import constants, store - try: from hermes_cli.config import load_config - cfg = load_config() display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} except Exception: pet_cfg = {} - enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) configured_slug = str(pet_cfg.get("slug", "") or "") pet = store.resolve_active_pet(configured_slug) if enabled else None @@ -4931,22 +4030,18 @@ def _pet_state_rows(spritesheet) -> list[str]: """ try: from PIL import Image - from agent.pet import constants - with Image.open(spritesheet) as image: row_count = max(1, image.height // constants.FRAME_H) return list(constants.state_rows_for_grid(row_count)) except Exception: # noqa: BLE001 - cosmetic, never break the surface from agent.pet import constants - return list(constants.STATE_ROWS) def _pet_gen_root(): """Profile-scoped staging dir for in-progress generation drafts.""" from hermes_constants import get_hermes_home - root = get_hermes_home() / "cache" / "pet-gen" root.mkdir(parents=True, exist_ok=True) return root @@ -4956,7 +4051,6 @@ def _pet_gen_sweep(root, *, max_age_s: float = 3600.0) -> None: """Drop stale draft staging dirs so cache never grows unbounded.""" import shutil import time - try: now = time.time() for child in root.iterdir(): @@ -4970,9 +4064,7 @@ def _pet_png_data_uri(path, *, max_px: int = 160) -> str: """Downscaled PNG data URI for a draft image (small preview payload).""" import base64 import io - from PIL import Image - with Image.open(path) as opened: img = opened.convert("RGBA") img.thumbnail((max_px, max_px), Image.LANCZOS) @@ -4981,23 +4073,15 @@ def _pet_png_data_uri(path, *, max_px: int = 160) -> str: return "data:image/png;base64," + base64.standard_b64encode(buf.getvalue()).decode("ascii") -# Cooperative cancellation for the heavy pet generation paths. The client's Stop -# aborts its RPC immediately, but the worker-pool generation keeps running unless -# told to stop — pet.cancel flips a token's flag, which generate_base_drafts / -# hatch_pet poll between provider calls to skip work they haven't started. +# Cooperative cancellation for pet generation: Stop aborts the RPC, but the +# pool job keeps running unless pet.cancel flips its token (polled between +# provider calls). _pet_cancel_lock = threading.Lock() _pet_cancelled: set[str] = set() -_PET_REFERENCE_MIME_EXT = { - "png": "png", - "jpeg": "jpg", - "jpg": "jpg", - "webp": "webp", - "gif": "gif", -} +_PET_REFERENCE_MIME_EXT = {"png": "png", "jpeg": "jpg", "jpg": "jpg", "webp": "webp", "gif": "gif"} try: _PET_REFERENCE_MAX_BYTES = max( - 1, - int(os.environ.get("HERMES_PET_REFERENCE_MAX_BYTES") or str(16 * 1024 * 1024)), + 1, int(os.environ.get("HERMES_PET_REFERENCE_MAX_BYTES") or str(16 * 1024 * 1024)), ) except (TypeError, ValueError): _PET_REFERENCE_MAX_BYTES = 16 * 1024 * 1024 @@ -5008,29 +4092,23 @@ def _pet_reference_images_from_data_url(ref_raw: str, stage) -> list: import base64 import binascii import re as _re - match = _re.match(r"^data:image/([a-zA-Z0-9.+-]+);base64,(.*)$", ref_raw, _re.DOTALL) if not match: raise ValueError("invalid reference image format") - mime = match.group(1).lower() ext = _PET_REFERENCE_MIME_EXT.get(mime) if ext is None: raise ValueError("unsupported reference image type") - payload = "".join(match.group(2).split()) approx = (len(payload) * 3) // 4 if approx > _PET_REFERENCE_MAX_BYTES: raise ValueError("reference image too large") - try: raw = base64.b64decode(payload, validate=True) except (binascii.Error, ValueError) as exc: raise ValueError("invalid reference image data") from exc - if len(raw) > _PET_REFERENCE_MAX_BYTES: raise ValueError("reference image too large") - ref_path = stage / f"reference.{ext}" ref_path.write_bytes(raw) return [ref_path] @@ -5057,27 +4135,17 @@ def _pet_cancel_release(token: str) -> None: _pet_cancelled.discard(token) -# =========================================================================== -# Phase 2b Remote Spending RPC methods -# =========================================================================== -# -# These return STRUCTURED success envelopes (result.ok / result.error) rather -# than JSON-RPC-level errors, so the TUI's rpc() promise always resolves and the -# Ink side can branch on the typed billing error code (insufficient_scope, -# rate_limited, no_payment_method, …) to render the right affordance instead of -# landing in a generic catch. The data-building lives in the shared core -# (agent/billing_view.py + hermes_cli/nous_billing.py) — same as /topup. +# ── Remote Spending (billing) RPC methods ──────────────────────────── +# STRUCTURED envelopes (result.ok / result.error) rather than JSON-RPC errors, +# so rpc() always resolves and the client branches on the typed billing code. +# Data-building lives in agent/billing_view.py + hermes_cli/nous_billing.py. def _serialize_billing_error(exc) -> dict: """Map a BillingError into the result.error envelope the TUI branches on.""" from hermes_cli.nous_billing import ( - BillingRemoteSpendingRevoked, - BillingScopeRequired, - BillingSessionRevoked, - BillingTransient, + BillingRemoteSpendingRevoked, BillingScopeRequired, BillingSessionRevoked, BillingTransient, ) - kind = "error" if isinstance(exc, BillingRemoteSpendingRevoked): kind = "remote_spending_revoked" @@ -5110,18 +4178,14 @@ def _serialize_billing_state(state) -> dict: def _s(value): return None if value is None else str(value) - card = None if state.card is not None: card = { "brand": state.card.brand, "last4": state.card.last4, "masked": state.card.masked, - # Post-card-resolver fields (None/False on older NAS payloads): - # display = "Visa ····4242 — the card on your subscription"; - # resolved_via = the raw resolution rung, for rung-gated surfaces - # (the /subscription confirm only shows the card when the rung - # matches what a subscription charge would use). + # None/False on older NAS payloads; resolved_via is the resolution + # rung for rung-gated surfaces (/subscription confirm). "display": state.card.display, "resolved_via": state.card.resolved_via, } @@ -5133,30 +4197,20 @@ def _serialize_billing_state(state) -> dict: # would read every Link method as a card. if pm.kind == "card": payment_method = { - "kind": "card", - "brand": pm.brand, - "last4": pm.last4, - "wallet": pm.wallet, + "kind": "card", "brand": pm.brand, "last4": pm.last4, "wallet": pm.wallet, "resolved_via": pm.resolved_via, } elif pm.kind == "link": - payment_method = { - "kind": "link", - "email": pm.email, - "resolved_via": pm.resolved_via, - } + payment_method = {"kind": "link", "email": pm.email, "resolved_via": pm.resolved_via} else: payment_method = { - "kind": "unknown", - "raw_kind": pm.raw_kind, - "resolved_via": pm.resolved_via, + "kind": "unknown", "raw_kind": pm.raw_kind, "resolved_via": pm.resolved_via, } monthly_cap = None if state.monthly_cap is not None: mc = state.monthly_cap monthly_cap = { - "limit_usd": _s(mc.limit_usd), - "limit_display": format_money(mc.limit_usd), + "limit_usd": _s(mc.limit_usd), "limit_display": format_money(mc.limit_usd), "spent_this_month_usd": _s(mc.spent_this_month_usd), "spent_display": format_money(mc.spent_this_month_usd), "is_default_ceiling": mc.is_default_ceiling, @@ -5168,20 +4222,16 @@ def _serialize_billing_state(state) -> dict: if ar.card is not None: if ar.card.kind == "distinct": card_out = { - "kind": "distinct", - "payment_method_id": ar.card.payment_method_id, - "brand": ar.card.brand, - "last4": ar.card.last4, + "kind": "distinct", "payment_method_id": ar.card.payment_method_id, + "brand": ar.card.brand, "last4": ar.card.last4, } else: card_out = {"kind": ar.card.kind} auto_reload = { - "enabled": ar.enabled, - "threshold_usd": _s(ar.threshold_usd), + "enabled": ar.enabled, "threshold_usd": _s(ar.threshold_usd), "threshold_display": format_money(ar.threshold_usd), "reload_to_usd": _s(ar.reload_to_usd), - "reload_to_display": format_money(ar.reload_to_usd), - "card": card_out, + "reload_to_display": format_money(ar.reload_to_usd), "card": card_out, } return { "ok": True, @@ -5205,10 +4255,8 @@ def _serialize_billing_state(state) -> dict: "auto_reload": auto_reload, "portal_url": state.portal_url, "error": state.error, - # Shared dollar usage model (two-bar view) embedded so /topup renders the - # same plan + top-up bars as /usage and /subscription from its single - # fetch. Built from the separate account-info path; fail-open when logged - # out or the portal is down. + # Shared two-bar dollar usage model so /topup matches /usage and + # /subscription from one fetch; fail-open. "usage": _usage_payload(state), } @@ -5223,7 +4271,6 @@ def _usage_payload(state) -> dict: return {"available": False} try: from agent.billing_usage import build_usage_model - return _serialize_usage_model(build_usage_model()) except Exception: return {"available": False} @@ -5234,14 +4281,10 @@ def _serialize_usage_bar(bar) -> Optional[dict]: if bar is None: return None from agent.billing_usage import _fmt_usd - return { - "kind": bar.kind, - "remaining_display": _fmt_usd(bar.remaining_usd), - "total_display": _fmt_usd(bar.total_usd), - "spent_display": _fmt_usd(bar.spent_usd), - "pct_used": bar.pct_used, - "fill_fraction": bar.fill_fraction, + "kind": bar.kind, "remaining_display": _fmt_usd(bar.remaining_usd), + "total_display": _fmt_usd(bar.total_usd), "spent_display": _fmt_usd(bar.spent_usd), + "pct_used": bar.pct_used, "fill_fraction": bar.fill_fraction, } @@ -5252,10 +4295,8 @@ def _serialize_usage_model(model) -> dict: ({ok, available:false} when logged out / unreachable). """ from agent.billing_usage import _fmt_usd, format_renews - if model is None or not getattr(model, "available", False): return {"ok": True, "available": False} - return { "ok": True, "available": True, @@ -5285,15 +4326,12 @@ def _serialize_subscription_state(state) -> dict: def _s(value): return None if value is None else str(value) - current = None if state.current is not None: c = state.current current = { - "tier_id": c.tier_id, - "tier_name": c.tier_name, - "monthly_credits": _s(c.monthly_credits), - "credits_remaining": _s(c.credits_remaining), + "tier_id": c.tier_id, "tier_name": c.tier_name, + "monthly_credits": _s(c.monthly_credits), "credits_remaining": _s(c.credits_remaining), "cycle_ends_at": c.cycle_ends_at, "pending_downgrade_tier_name": c.pending_downgrade_tier_name, "pending_downgrade_at": c.pending_downgrade_at, @@ -5306,12 +4344,9 @@ def _serialize_subscription_state(state) -> dict: # ($X / $X.YY) so the TUI renders it directly. tiers = [ { - "tier_id": t.tier_id, - "name": t.name, - "tier_order": t.tier_order, + "tier_id": t.tier_id, "name": t.name, "tier_order": t.tier_order, "dollars_per_month_display": format_money(t.dollars_per_month), - "monthly_credits": _s(t.monthly_credits), - "is_current": t.is_current, + "monthly_credits": _s(t.monthly_credits), "is_current": t.is_current, "is_enabled": t.is_enabled, } for t in state.tiers @@ -5329,11 +4364,8 @@ def _serialize_subscription_state(state) -> dict: "tiers": tiers, "portal_url": state.portal_url, "error": state.error, - # Shared dollar usage model (two-bar view) embedded so /subscription - # renders the same bars as /usage from its single fetch. Built from the - # separate account-info path (the only source with top-up dollars); - # fail-open → {available:false}. Computed lazily so a logged-out state - # adds no cost. + # Shared two-bar usage model (account-info is the only source with + # top-up dollars); fail-open → {available:false}; lazy when logged out. "usage": _usage_payload(state), } @@ -5363,26 +4395,20 @@ def _serialize_subscription_preview(p) -> dict: # ── Spawn-tree snapshots: TUI-written, disk-persisted ──────────────── -# The TUI is the source of truth for subagent state (it assembles payloads -# from the event stream). On turn-complete it posts the final tree here; -# /replay and /replay-diff fetch past snapshots by session_id + filename. -# -# Layout: $HERMES_HOME/spawn-trees//.json -# Each file contains { session_id, started_at, finished_at, subagents: [...] }. +# The TUI owns subagent state; on turn-complete it posts the final tree here, +# /replay fetches by session_id + filename. +# Layout: $HERMES_HOME/spawn-trees//.json def _spawn_trees_root(): from hermes_constants import get_hermes_home - root = get_hermes_home() / "spawn-trees" root.mkdir(parents=True, exist_ok=True) return root def _spawn_tree_session_dir(session_id: str): - safe = ( - "".join(c if c.isalnum() or c in "-_" else "_" for c in session_id) or "unknown" - ) + safe = ("".join(c if c.isalnum() or c in "-_" else "_" for c in session_id) or "unknown") d = _spawn_trees_root() / safe d.mkdir(parents=True, exist_ok=True) return d @@ -5431,132 +4457,30 @@ _GOAL_COMPRESSION_RECOVERY_ATTEMPTS = "_goal_compression_recovery_attempts" _GOAL_COMPRESSION_RECOVERY_LIMIT = 1 -def _is_successful_goal_turn(result: Any, status: str, raw: Any) -> bool: - """Return whether a turn produced a real response the goal judge can use.""" - return bool( - status == "complete" - and isinstance(raw, str) - and raw.strip() - and not (isinstance(result, dict) and result.get("failed")) - and not (isinstance(result, dict) and result.get("completed") is False) - ) -def _plan_goal_compression_recovery( - session: dict, - result: Any, - *, - status: str, - raw: Any, -) -> tuple[str | None, str | None]: - """Plan a bounded active-goal retry after compression exhaustion. - - Compression exhaustion is a failed turn, so it must not be sent to the - goal judge or consume the goal's turn budget. One fresh continuation turn - is allowed. If that turn also exhausts compression, pause the goal rather - than spinning until an arbitrary user message happens to wake it up. - - Returns ``(continuation_prompt, status_notice)``. Sessions without an - active goal retain the existing error-only behavior. - """ - compression_exhausted = bool( - isinstance(result, dict) and result.get("compression_exhausted") - ) - if not compression_exhausted: - if _is_successful_goal_turn(result, status, raw): - session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) - return None, None - - from hermes_cli.goals import GoalManager - - sid_key = str(session.get("session_key") or "") - if not sid_key: - return None, None - - try: - goals_cfg = _load_cfg().get("goals") or {} - goal_max_turns = int(goals_cfg.get("max_turns", 20) or 20) - except Exception: - goal_max_turns = 20 - - goal_mgr = GoalManager( - session_id=sid_key, - default_max_turns=goal_max_turns, - ) - if not goal_mgr.is_active(): - session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) - return None, None - - goal_created_at = float(getattr(goal_mgr.state, "created_at", 0.0) or 0.0) - recovery_state = session.get(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS) - attempts = 0 - if ( - isinstance(recovery_state, dict) - and recovery_state.get("goal_created_at") == goal_created_at - and recovery_state.get("goal") == getattr(goal_mgr.state, "goal", "") - ): - try: - attempts = int(recovery_state.get("attempts", 0) or 0) - except (TypeError, ValueError): - attempts = 0 - - continuation_prompt = goal_mgr.next_continuation_prompt() - if attempts < _GOAL_COMPRESSION_RECOVERY_LIMIT and continuation_prompt: - session[_GOAL_COMPRESSION_RECOVERY_ATTEMPTS] = { - "goal_created_at": goal_created_at, - "goal": getattr(goal_mgr.state, "goal", ""), - "attempts": attempts + 1, - } - return ( - continuation_prompt, - "Context compression was exhausted. Retrying the active goal once.", - ) - - goal_mgr.pause(reason="context compression exhausted twice consecutively") - # A later explicit /goal resume gets a fresh bounded recovery cycle. - session.pop(_GOAL_COMPRESSION_RECOVERY_ATTEMPTS, None) - return ( - None, - "Goal paused after context compression was exhausted twice. " - "Run /compress, then /goal resume to continue.", - ) - - -# Captured at import time. Several _run_prompt_submit tests monkeypatch -# threading.Thread with a stub that runs the target synchronously to keep the -# turn deterministic. This ticker's loop only exits once the caller sets `stop` -# *after* run_conversation returns, so running it inline would spin forever. -# It's a non-critical, fire-and-forget background poller, so it always uses a -# real daemon thread regardless of any such patch. +# Captured at import time: tests monkeypatch threading.Thread with a synchronous +# stub, and this ticker only exits once `stop` is set AFTER run_conversation +# returns — inline it would spin forever. Always a real daemon thread. _RealThread = threading.Thread def _start_usage_ticker( sid: str, agent, interval: float = 1.0 ) -> tuple[threading.Event, threading.Thread]: - """Push live usage snapshots while a turn runs. + """Push live ``session.usage`` snapshots every ``interval`` s while a turn runs. - The desktop/TUI status-bar context-window figure is otherwise refreshed - only at ``message.complete``, so it stays frozen for the whole (often - multi-minute, multi-tool) turn. On the standard chat-completions path the - agent's token counters grow after every internal API call, so this daemon - emits a lightweight ``session.usage`` event every ``interval`` seconds and - the bar tracks context growth live. (The codex app-server runtime folds - usage into the counters only at turn end — codex_runtime. - _record_codex_app_server_usage — so it gets no mid-turn ticks; its final - value still lands via ``message.complete``.) - - The caller must set the returned Event AND join the returned thread - before emitting ``message.complete``: a tick that survived past it would - roll the client's final usage back to a stale mid-turn snapshot. + Otherwise the status-bar context figure is frozen until ``message.complete``. + (The codex app-server runtime folds usage in only at turn end, so it gets no + mid-turn ticks.) The caller must set the Event AND join the thread before + emitting ``message.complete``: a late tick would roll the client's final + usage back to a stale snapshot. """ stop = threading.Event() - # Sample the dedup baseline BEFORE the thread starts: the client already - # has the turn-start values from the previous message.complete / - # session.info. Seeding here (not in the thread) guarantees the baseline - # predates the turn's first API call — a late-scheduled thread would - # otherwise absorb that first counter growth and never emit it. + # Dedup baseline sampled BEFORE the thread starts (the client already has + # the turn-start values); a late-scheduled thread would otherwise absorb the + # first counter growth and never emit it. try: baseline: dict | None = _get_usage(agent) except Exception: @@ -5565,7 +4489,7 @@ def _start_usage_ticker( def _loop() -> None: last = baseline while not stop.wait(interval): - try: + with contextlib.suppress(Exception): usage = _get_usage(agent) if usage == last: # Counters frozen (e.g. one long API call in flight) — @@ -5578,1115 +4502,11 @@ def _start_usage_ticker( # message.complete carries the authoritative usage. break _emit("session.usage", sid, {"usage": usage}) - except Exception: - pass - thread = _RealThread(target=_loop, daemon=True) thread.start() return stop, thread -def _run_prompt_submit( - rid, - sid: str, - session: dict, - text: Any, - *, - display_kind: str | None = None, - display_metadata: dict | None = None, - image_paths: list[str] | None = None, - queued_prompt_generation: int | None = None, - terminal_callback: Callable[[dict[str, Any]], None] | None = None, -) -> bool: - # Ownership admission at the ONE chokepoint every fresh turn source must - # cross. prompt.submit already claims the slot in its RPC handler (so this - # is a no-op re-check there), but crash auto-continue, wake-ups and other - # synthesized turns call _run_prompt_submit directly — the exact bypass - # that let a second backend run a duplicate turn in #94778. When the - # session already holds its lease this is a cheap dict check. - if (ownership_refusal := _ensure_active_session_slot(sid, session)) is not None: - logger.info( - "Refusing turn for session %s at _run_prompt_submit: %s", - session.get("session_key") or sid, - getattr(ownership_refusal, "reason", None) or "refused", - ) - with session["history_lock"]: - session["running"] = False - _emit("error", sid, {"message": str(ownership_refusal)}) - return False - with session["history_lock"]: - if session.get("_closing"): - session["running"] = False - return False - if ( - queued_prompt_generation is not None - and int(session.get("_queued_prompt_generation", 0)) != queued_prompt_generation - ): - session["running"] = False - return False - if image_paths is None: - images = list(session.get("attached_images", [])) - session["attached_images"] = [] - else: - images = list(image_paths) - inflight = session.get("inflight_turn") - # A retained failed turn (see _fail_inflight_turn) is a stale leftover - # by the time a new turn starts — replace it, never append onto it. - if not isinstance(inflight, dict) or inflight.get("status") == "error": - _start_inflight_turn(session, text) - agent = session["agent"] - if hasattr(agent, "clear_interrupt"): - try: - agent.clear_interrupt() - except Exception: - pass - # Desktop/TUI observability (#86647): this is the ONE INFO record proving - # a Desktop/TUI prompt was accepted by THIS process, and it ties together - # every id a rotation-mute trace needs — the UI session id, the gateway - # session_key, and the agent's live session_id (which compression rotates - # independently of the other two). Before this line a Desktop request left - # no trace in agent.log at all ("0 platform=desktop" — see #86647), so a - # muted window was structurally indistinguishable from a request that - # never arrived. No prompt content is logged. - _turn_started_monotonic = time.monotonic() - logger.info( - "tui prompt accepted: ui_session=%s session_key=%s agent_session_id=%s " - "kind=%s chars=%s images=%d", - sid, - session.get("session_key") or "", - getattr(agent, "session_id", "") or "", - display_kind or "user", - len(text) if isinstance(text, str) else "-", - len(images), - ) - _emit("message.start", sid) - - def run(): - terminal_receipt_attempted = False - terminal_receipt_committed = terminal_callback is None - # The conversation runs on a fresh thread, so ContextVars from the RPC - # dispatcher do not follow automatically. Rebind the exact transport - # stored on this session generation before any tool can commission a - # child; delegate_task then captures it as non-serializable authority. - transport_token = bind_transport(session.get("transport")) - runtime_session_token = _current_runtime_session_record.set(session) - # Bound eagerly so the except/finally paths below always have an agent - # even if turn setup throws; re-read after _sync_bot_capabilities, - # which may swap in a rebuilt agent for Bot Chat sessions. - agent = session["agent"] - approval_token = None - session_tokens = [] - home_token = None # per-turn HERMES_HOME override for a resumed remote profile - secret_token = None - goal_followup = None # set by the post-turn goal hook below - result = None # turn outcome; read after the finally for leftover /steer - tts_queue = None # streaming-TTS feed for this turn (voice mode) - thinking_started = False # ambient thinking sound armed for this turn - one_turn_restore = session.pop("one_turn_model_restore", None) - # True once a failed turn's snapshot was retained for resume replay — - # tells the finally below to skip the normal inflight clear. - turn_error_retained = False - # One-line cause for the "tui turn finished" bookend below. The record - # fires from a `finally`, where neither `result` nor the caught - # exception is reliably in scope, so both failure paths stash their - # cause here on the way past. - turn_error_detail = "" - # What this turn actually submitted, kept only so the cause can be - # checked for quoting it back (see _strip_prompt_echo). Bound here - # rather than read from the turn body because the exception path can - # fire before the prompt is resolved. - turn_prompt_text = "" - # Durable crash marker: written before the turn runs, retired the - # moment its outcome reaches the client (see _retire_turn_marker). - # Any concluded turn — success, handled error, interrupt — retires - # it, so a marker that survives means the process died mid-turn; - # session.resume auto-continues from it. Compression can rotate - # session_key mid-turn, so remember the key we wrote under. - marker_home = _session_home(session) - marker_key = str(session.get("session_key") or "") - marker_attempt = int(session.pop("_auto_continue_attempt", 0) or 0) - marker_text = session.pop("_auto_continue_prompt", None) or text - if isinstance(marker_text, str) and marker_text.strip(): - # Publish the original key before the disk write so an interrupt - # racing startup can retire it even if compression rotates the - # session key later. The post-write cancel check closes the inverse - # race where Stop lands first and therefore clears no file yet. - with session["history_lock"]: - session["_active_turn_marker_key"] = marker_key - record_turn_start(marker_home, marker_key, marker_text, attempts=marker_attempt) - with session["history_lock"]: - marker_cancelled = bool(session.get("_turn_cancel_requested")) - if marker_cancelled: - clear_turn_marker(marker_home, marker_key) - try: - from tools.approval import ( - reset_current_session_key, - set_current_session_key, - ) - - approval_token = set_current_session_key(session["session_key"]) - session_tokens = _set_session_context( - session["session_key"], - ui_session_id=sid, - ) - _profile_home_str = session.get("profile_home") - if _profile_home_str: - home_token = set_hermes_home_override(_profile_home_str) - secret_token = set_secret_scope(build_profile_secret_scope(Path(_profile_home_str))) - # Fourth profile seam: bind the session profile's COMPLETE - # terminal policy for this turn (dashboard/TUI analogue of the - # gateway's per-turn scope). #98581's unified-desktop - # reproduction ran a docker-configured profile on the host - # because terminal_tool read the launch process's pinned env. - # Failure installs a refusal scope → terminal tools raise - # (fail closed) instead of inheriting ambient policy. - from tools.terminal_scope import ( - install_profile_terminal_scope as _install_term_scope, - ) - - _terminal_scope_token = _install_term_scope(Path(_profile_home_str)) - else: - _terminal_scope_token = None - # The sudo password callback is thread-local (tools.terminal_tool - # _callback_tls), so wiring it on the build thread doesn't reach this - # turn thread — terminal sudo prompts would fall through to /dev/tty - # and hang the headless gateway. Re-wire here so the prompt routes to - # the sudo.request overlay. (secret capture is a module global, so - # re-running is a harmless no-op.) - _wire_callbacks(sid) - # Skip the config-model sync while a /model --once override is - # active: the once-model is intentionally not pinned as a session - # model_override (it must not persist), so without this guard the - # sync would see "agent model != config model" and clobber the - # once-override back to the config model before the turn runs - # (#29923 review defect). Any config.yaml change is adopted on - # the NEXT turn, after the finally-restore below. - if not one_turn_restore: - # A model picked mid-turn was queued (not applied in-place) — - # apply it now, on the turn thread before the first model call, - # so this turn runs on the model the user chose. Runs before the - # config sync so an explicit pick wins over a config.yaml change. - _apply_pending_model_switch(sid, session) - _sync_agent_model_with_config(sid, session) - _sync_agent_compression_with_config(sid, session) - # Bot Chat capability sync — adopt Settings→Capabilities edits - # (skills/toolsets/MCP/SOUL) into the eternal bot session before - # the turn runs. No-op for every other session shape. - _sync_bot_capabilities(sid, session) - agent = session["agent"] - # Snapshot after turn-start model sync. A deferred switch mutates - # history and its version; that mutation belongs to this turn. - with session["history_lock"]: - history = list(session["history"]) - history_version = int(session.get("history_version", 0)) - cwd = _session_cwd(session) - _register_session_cwd(session) - cols = session.get("cols", 80) - streamer = make_stream_renderer(cols) - prompt = text - - if isinstance(prompt, str) and "@" in prompt: - from agent.context_references import preprocess_context_references - from agent.model_metadata import get_model_context_length - - ctx_len = get_model_context_length( - getattr(agent, "model", "") or _resolve_model(), - base_url=getattr(agent, "base_url", "") or "", - api_key=getattr(agent, "api_key", "") or "", - provider=getattr(agent, "provider", "") or "", - config_context_length=getattr( - agent, "_config_context_length", None - ), - ) - ctx = preprocess_context_references( - prompt, - cwd=cwd, - allowed_root=cwd, - context_length=ctx_len, - ) - if ctx.blocked: - _emit( - "error", - sid, - { - "message": "\n".join(ctx.warnings) - or "Context injection refused." - }, - ) - return - prompt = ctx.message - - # After @-expansion on purpose: an injected file's contents are - # exactly the kind of private material a provider echo would carry - # back, and they are not in `text`. - turn_prompt_text = prompt if isinstance(prompt, str) else "" - - # Decide image routing per-turn based on active provider/model. - # "native" → pass pixels to the main model as OpenAI-style content - # parts (adapters translate for Anthropic/Gemini/Bedrock/etc.). - # "text" → reference the image paths in the message so the agent - # analyzes them in-loop with vision_analyze (never - # blocking the submit path on vision calls — #83291). - # See agent/image_routing.py for the full decision table. - run_message: Any = prompt - if images: - try: - from agent.image_routing import ( - decide_image_input_mode, - build_native_content_parts, - ) - from hermes_cli.config import load_config as _tui_load_config - - _cfg = _tui_load_config() - _provider, _model = _active_image_routing_identity(agent) - _mode = decide_image_input_mode( - _provider, - _model, - _cfg, - requested_provider=getattr( - agent, "requested_provider", "" - ), - ) - if getattr(agent, "api_mode", "") == "codex_app_server": - _mode = "text" - except Exception as _img_exc: - print( - f"[tui_gateway] image_routing decision failed, defaulting to text: {_img_exc}", - file=sys.stderr, - ) - _mode = "text" - - if _mode == "native": - try: - _parts, _skipped = build_native_content_parts( - prompt, - images, - ) - if _skipped: - print( - f"[tui_gateway] native image attachment skipped {len(_skipped)} unreadable path(s)", - file=sys.stderr, - ) - if any(p.get("type") == "image_url" for p in _parts): - run_message = _parts - else: - run_message = _build_image_ref_message(prompt, images) - except Exception as _img_exc: - print( - f"[tui_gateway] native attach failed, falling back to text: {_img_exc}", - file=sys.stderr, - ) - run_message = _build_image_ref_message(prompt, images) - else: - run_message = _build_image_ref_message(prompt, images) - - # Streaming TTS: voice-mode replies are spoken sentence-by-sentence - # as tokens arrive (CLI parity) instead of after the full turn. - # begin() first — it cuts any still-speaking previous turn, and - # that cut IS this turn's barge-in, so it must latch before we - # consume the latch below. - tts_queue = _tts_stream_begin() - - # Full-duplex agent-turn listener: armed at utterance-submit so - # the user can interject DURING generation, not just during - # playback. _tts_stream_begin arms it too when a pipeline - # starts; this covers voice mode without working TTS. - if _voice_mode_enabled() and _voice_cfg_dict().get("barge_in", True): - _arm_full_duplex_listener() - - # Ambient "thinking" sound (voice mode only): calm bubble blips - # while the agent works with no audio flowing, so long - # thinking/tool stretches don't read as a dead session. Per-blip - # gate skips while real TTS audio flows or the mic is capturing; - # stopped in the finally the instant the turn ends. - # voice.thinking_sound config-gates it; macOS TCC handled inside. - thinking_started = False - if _voice_mode_enabled(): - try: - from tools.voice_mode import ( - is_audio_output_active, - start_thinking_sound, - ) - - def _thinking_should_play() -> bool: - if is_audio_output_active(): - return False - try: - from hermes_cli.voice import is_continuous_active - - return not is_continuous_active() - except Exception: - return True - - thinking_started = start_thinking_sound( - should_play=_thinking_should_play - ) - except Exception: - thinking_started = False - - # Barged mid-speech? Tell the model (API-message note, same - # enrichment channel as attached images) so it can react - # ("rude!") instead of being oblivious to its own interruption. - from tools.tts_streaming import SPEECH_INTERRUPTED_NOTE, take_speech_interrupted - - if take_speech_interrupted(): - run_message = _prepend_note(run_message, SPEECH_INTERRUPTED_NOTE) - - # Reactions the user added since the last turn. - run_message = _prepend_note(run_message, _pending_reaction_notes(session)) - - # Which window the message was typed into. HUD mode is per-turn - # state, so it cannot live in the (byte-stable) system prompt. - run_message = _prepend_note(run_message, _hud_surface_note(session)) - - def _stream(delta): - with session["history_lock"]: - _append_inflight_delta(session, delta) - payload = {"text": delta} - if streamer and (r := streamer.feed(delta)) is not None: - payload["rendered"] = r - if tts_queue is not None and isinstance(delta, str): - tts_queue.put(delta) - _emit("message.delta", sid, payload) - - # Surface interim assistant text (commentary emitted alongside - # tool calls, or the attempted final answer before a verify-on-stop - # nudge) so the desktop can seal it as its own segment instead of - # losing it when message.complete replaces the streaming buffer. - # Gated on display.interim_assistant_messages (default true). - if _load_interim_assistant_messages(): - def _interim_assistant_cb(text: str, *, already_streamed: bool = False) -> None: - _emit("message.interim", sid, { - "text": text, - "already_streamed": already_streamed, - }) - - agent.interim_assistant_callback = _interim_assistant_cb - else: - agent.interim_assistant_callback = None - - run_kwargs = { - "conversation_history": list(history), - "stream_callback": _stream, - "persist_user_message": ( - _build_persist_user_message(prompt, images, run_message) if images else prompt - ), - } - # Type a synthesized turn at turn START so the crash persist writes - # its row as a timeline event, instead of leaving a raw user bubble - # until the turn ends — and forever if it never does, which is - # exactly the auto-continue case. The post-turn stamp below is the - # fallback for an older agent without the parameter; re-stamping - # the same value is a no-op. - try: - _run_params = inspect.signature(agent.run_conversation).parameters - except (TypeError, ValueError): - _run_params = {} - if "task_id" in _run_params: - run_kwargs["task_id"] = session["session_key"] - if display_kind and "persist_user_display_kind" in _run_params: - run_kwargs["persist_user_display_kind"] = display_kind - run_kwargs["persist_user_display_metadata"] = display_metadata - # Auto-titling now fires inside the turn prologue (shared by every - # surface). Hand the agent this session's live-rename hook so the - # sidebar repaints the moment a title lands, rather than waiting - # for the next list refresh. - _title_key = session.get("session_key") or sid - agent._on_session_title = lambda t, _src, _k=_title_key: _emit( - "session.title", sid, {"session_id": _k, "title": t} - ) - _usage_stop, _usage_thread = _start_usage_ticker(sid, agent) - try: - result = agent.run_conversation(run_message, **run_kwargs) - finally: - # Stop AND join before anything below emits: an in-flight tick - # surviving past message.complete would roll the client's final - # usage back to a stale mid-turn snapshot. The join is - # deliberately unbounded — once stop is set it only ever waits - # out one in-flight _get_usage/_emit, and the worst case there - # (a stalled transport write, up to _WS_WRITE_TIMEOUT_S) would - # stall the message.complete emit below just the same. A - # timed-out join would abandon the tick to land after - # message.complete. - _usage_stop.set() - _usage_thread.join() - if display_kind and isinstance(text, str): - db = getattr(agent, "_session_db", None) - current_session_id = getattr(agent, "session_id", None) or session.get("session_key") - if db is not None: - try: - db.set_latest_matching_message_display_kind( - current_session_id, - role="user", - content=text, - display_kind=display_kind, - display_metadata=display_metadata, - ) - except Exception: - logger.debug("failed to stamp synthetic display kind", exc_info=True) - if isinstance(result, dict) and isinstance(result.get("messages"), list): - for message in reversed(result["messages"]): - if message.get("role") == "user" and message.get("content") == text: - message["display_kind"] = display_kind - if display_metadata: - message["display_metadata"] = display_metadata - break - if "moa_one_shot_restore" in session: - _restore = session.pop("moa_one_shot_restore", None) - # Restore the model the user was on before the /moa one-shot. - # The one-shot did a real in-place agent.switch_model() to MoA - # (#53444), so undoing it must go back through the switch path — - # resetting session["model_override"] alone would leave the live - # agent's client pinned to MoA for the next turn. - if isinstance(_restore, dict): - _prev_override = _restore.get("override") - _prev_model = _restore.get("model") - _prev_provider = _restore.get("provider") - if _prev_override is None: - session.pop("model_override", None) - else: - session["model_override"] = _prev_override - if _prev_model: - _raw = ( - f"{_prev_model} --provider {_prev_provider}" - if _prev_provider - else _prev_model - ) - try: - _apply_model_switch( - sid, - session, - _raw, - confirm_expensive_model=False, - pin_session_override=bool(_prev_override), - # Session-internal restore after the /moa - # one-shot — never persist to config.yaml. - persist_override=False, - ) - except Exception as _moa_restore_exc: - logger.warning( - "MoA one-shot model restore failed: %s", - _moa_restore_exc, - ) - elif _restore is None: - session.pop("model_override", None) - else: - session["model_override"] = _restore - - last_reasoning = None - status_note = None - if isinstance(result, dict): - if isinstance(result.get("messages"), list): - with session["history_lock"]: - current_version = int(session.get("history_version", 0)) - if current_version == history_version: - session["history"] = result["messages"] - session["history_version"] = history_version + 1 - else: - # History mutated externally during the turn. - # Check if the only mutation was a pivot marker - # the gateway itself inserted mid-turn (#76870). - # If so the agent output is still valid — merge it - # into the current history that now contains the - # marker. A personality change counts here too: - # unlike a model switch it has no pending queue, so - # `/personality` during a running turn lands - # immediately and used to read as a genuine desync, - # dropping the finished turn (#82756). - # - # _append_model_switch_marker strips prior markers - # in-place then appends a new one, so the delta - # is NOT a simple tail-slice — we must compare - # content, not indices. - current_history = list(session["history"]) - history_no_markers = [ - e for e in history if not _is_pivot_marker(e) - ] - current_no_markers = [ - e for e in current_history if not _is_pivot_marker(e) - ] - pivot_only = ( - current_no_markers == history_no_markers - and any( - _is_pivot_marker(e) - for e in current_history - ) - ) - if pivot_only: - # The agent's new messages start after the - # turn-start history. Guard against - # auto-compression making result["messages"] - # shorter than history (#77274 review). - if len(result["messages"]) > len(history): - new_messages = result["messages"][len(history):] - else: - # Compression rebound the messages list — - # use the full result as the base. - new_messages = list(result["messages"]) - session["history"] = current_history + new_messages - session["history_version"] = current_version + 1 - else: - # Genuine desync (undo/compress/retry/rollback). - # Surface the desync rather than silently - # dropping the agent's output — the UI can - # show the response and warn that it was - # not persisted. - print( - f"[tui_gateway] prompt.submit: history_version mismatch " - f"(expected={history_version} current={current_version}) — " - f"agent output NOT written to session history", - file=sys.stderr, - ) - status_note = ( - "History changed during this turn — the response above is visible " - "but was not saved to session history." - ) - - # If auto-compression fired inside run_conversation(), agent.session_id - # may have rotated. Sync session_key before downstream title/goal/finalize - # handling uses it. Preserve pending_title (user intent) so it can be - # applied to the continuation. Restart slash worker so subsequent - # worker-backed commands (/title etc.) target the live session. - # Fix for #20001. - _sync_session_key_after_compress( - sid, session, clear_pending_title=False, restart_slash_worker=True, - ) - - raw = result.get("final_response", "") - status = ( - "interrupted" - if result.get("interrupted") - else "error" if result.get("error") else "complete" - ) - # When the backend produced no visible response AND reported a - # real error (e.g. invalid model slug → provider 4xx), surface - # that error as the visible text instead of shipping an empty - # turn to Ink. Mirrors classic CLI behavior at cli.py where - # (failed|partial) + no final_response → "Error: ". - # Leaves the None-with-no-error path untouched: an empty - # successful turn still renders as empty, and the existing - # "(empty)" sentinel handling stays in its own lane. - if (not raw) and result.get("error") and ( - result.get("failed") or result.get("partial") - ): - raw = f"Error: {result.get('error')}" - # "Operation interrupted: waiting for model response (…)" is - # cancellation metadata, not assistant prose. gateway/run.py - # and the ACP adapter already suppress this sentinel; without - # this the desktop paints it as the agent's reply whenever a - # stop/steer lands mid-request (#7921). - if status == "interrupted" and isinstance(raw, str) and raw.strip().startswith( - INTERRUPT_WAITING_FOR_MODEL_PREFIX - ): - raw = "" - lr = result.get("last_reasoning") - if isinstance(lr, str) and lr.strip(): - last_reasoning = lr.strip() - else: - raw = str(result) - status = "complete" - - payload = {"text": raw, "usage": _get_usage(agent), "status": status} - if last_reasoning: - payload["reasoning"] = last_reasoning - if status_note: - payload["warning"] = status_note - if result.get("response_previewed"): - payload["response_previewed"] = True - # Forward the structured billing-wall descriptor (provider, - # billing_url, is_nous, message) so the TUI/desktop render a - # billing-specific recovery surface instead of re-parsing text. - _billing_block = result.get("billing_block") if isinstance(result, dict) else None - if _billing_block: - payload["billing"] = _billing_block - payload["failure_reason"] = result.get("failure_reason") - rendered = render_message(raw, cols) - if rendered: - payload["rendered"] = rendered - # Structured layer descriptor ({layer, code, retryable}) so - # clients can name WHICH part of the stack failed (provider / - # streaming / auth / gateway / …) and offer layer-appropriate - # recovery actions instead of sniffing the message string. - # Advisory: older clients ignore it, absence falls back to - # string heuristics on newer clients. Computed before the retain - # below so resume replay carries the same descriptor. - _error_surface = None - if status == "error": - try: - from agent.error_surface import build_error_surface_from_result - - _error_surface = build_error_surface_from_result( - result, - provider=str(getattr(agent, "provider", "") or ""), - model=str(getattr(agent, "model", "") or ""), - ) - except Exception: - _error_surface = None - with session["history_lock"]: - if status == "error": - # Returned-error result (provider 4xx, budget, etc.): retain - # the failed turn for resume replay instead of clearing it. - # If this terminal frame is lost to a disconnect, resume's - # inflight payload is the only carrier of the failure. - _fail_inflight_turn( - session, - result.get("error") if isinstance(result, dict) else raw, - error_surface=_error_surface, - ) - turn_error_retained = True - turn_error_detail = _turn_failure_detail( - (result.get("error") if isinstance(result, dict) else raw), - (result.get("failure_reason") if isinstance(result, dict) else None), - turn_prompt_text, - ) - else: - _clear_inflight_turn(session) - if status == "error": - payload["error"] = str( - (result.get("error") if isinstance(result, dict) else "") or raw - ) - payload["recoverable"] = True - if _error_surface: - payload["error_surface"] = _error_surface - if terminal_callback is not None: - terminal_receipt_attempted = True - terminal_callback( - { - "status": ( - "cancelled" - if status == "interrupted" - else "failed" if status == "error" else "settled" - ), - "text": raw if isinstance(raw, str) else str(raw), - **( - {"error": str(result.get("error") or raw)} - if status == "error" and isinstance(result, dict) - else {} - ), - } - ) - terminal_receipt_committed = True - if terminal_receipt_committed: - _retire_turn_marker(session, marker_key) - _emit("message.complete", sid, payload) - - # ── /goal continuation (Ralph-style loop) ───────────────── - # After every TUI turn, if a /goal is active, ask the judge - # whether the goal is done and — if not and we're still under - # budget — queue a continuation prompt to run after this - # thread releases session["running"]. The verdict message - # ("✓ Goal achieved" / "⏸ budget exhausted") is surfaced as - # a system line so the user sees progress regardless of - # outcome. Mirrors gateway/run._post_turn_goal_continuation. - compression_exhausted = bool( - isinstance(result, dict) and result.get("compression_exhausted") - ) - try: - recovery_prompt, recovery_notice = _plan_goal_compression_recovery( - session, - result, - status=status, - raw=raw, - ) - if recovery_notice: - _emit( - "status.update", - sid, - {"kind": "goal", "text": recovery_notice}, - ) - if recovery_prompt: - goal_followup = recovery_prompt - except Exception as _goal_recovery_exc: - print( - f"[tui_gateway] goal compression recovery failed: " - f"{type(_goal_recovery_exc).__name__}: {_goal_recovery_exc}", - file=sys.stderr, - ) - - # Compression failures are never judge input: the error text is - # not work toward the goal, and evaluating it would spend a turn. - if not compression_exhausted and _is_successful_goal_turn( - result, status, raw - ): - try: - from hermes_cli.goals import GoalManager - - sid_key = session.get("session_key") or "" - if sid_key: - try: - goals_cfg = _load_cfg().get("goals") or {} - goal_max_turns = int(goals_cfg.get("max_turns", 20) or 20) - except Exception: - goal_max_turns = 20 - goal_mgr = GoalManager( - session_id=sid_key, - default_max_turns=goal_max_turns, - ) - if goal_mgr.is_active(): - try: - from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() - except Exception: - _bg_procs = None - decision = goal_mgr.evaluate_after_turn( - raw, - user_initiated=True, - background_processes=_bg_procs, - ) - verdict_msg = decision.get("message") or "" - if verdict_msg: - _emit( - "status.update", - sid, - {"kind": "goal", "text": verdict_msg}, - ) - if decision.get("should_continue"): - cont_prompt = decision.get("continuation_prompt") or "" - if cont_prompt: - goal_followup = cont_prompt - except Exception as _goal_exc: - print( - f"[tui_gateway] goal continuation hook failed: " - f"{type(_goal_exc).__name__}: {_goal_exc}", - file=sys.stderr, - ) - - # ── /loop tick completion ────────────────────────────────── - # If the turn that just finished was a /loop wakeup (fired by - # the notification poller), evaluate it: LOOP_COMPLETE marker, - # --until judge, --times / max_ticks caps, next-tick schedule. - if status == "complete": - try: - from hermes_cli.loops import LoopManager - - loop_sid_key = session.get("session_key") or "" - if loop_sid_key: - loop_mgr = LoopManager(session_id=loop_sid_key) - loop_state = loop_mgr.state - if loop_state is not None and loop_state.awaiting_response: - loop_decision = loop_mgr.complete_tick( - raw if isinstance(raw, str) else "" - ) - loop_msg = loop_decision.get("message") or "" - if loop_msg: - _emit( - "status.update", - sid, - {"kind": "loop", "text": loop_msg}, - ) - except Exception as _loop_exc: - print( - f"[tui_gateway] loop completion hook failed: " - f"{type(_loop_exc).__name__}: {_loop_exc}", - file=sys.stderr, - ) - - # Apply pending_title now that the DB row exists — in the - # session-owned profile store (not the launch profile). - _pending = session.get("pending_title") - if _pending and status == "complete": - _session_key = session.get("session_key") or sid - try: - with _session_db(session) as _pdb: - if _pdb and _pdb.set_session_title(_session_key, _pending): - session["pending_title"] = None - except ValueError as exc: - # Invalid/duplicate title — non-retryable, drop it. - # Auto-title will take over. Fix for #19029. - session["pending_title"] = None - logger.info( - "Dropping pending title for session %s: %s", - _session_key, exc, - ) - except Exception: - # Transient DB failure — keep pending_title for retry. - pass - - # Voice TTS fallback: when the streaming pipeline couldn't start - # (no provider / missing deps probed at turn start), speak the - # final text whole (cli.py:_voice_speak_response parity). The - # streaming path already spoke everything via tts_queue. - if ( - status == "complete" - and tts_queue is None - and isinstance(raw, str) - and raw.strip() - and _voice_tts_enabled() - ): - try: - spoken = raw - # Barge-aware: spoken interruptions must cut this - # fallback playback too, not just the streaming path. - threading.Thread( - target=_speak_text_with_barge, args=(spoken,), daemon=True - ).start() - except ImportError: - logger.warning("voice TTS skipped: hermes_cli.voice unavailable") - except Exception as e: - logger.warning("voice TTS dispatch failed: %s", e) - except Exception as e: - import traceback - - trace = traceback.format_exc() - try: - os.makedirs(os.path.dirname(_CRASH_LOG), exist_ok=True) - with open(_CRASH_LOG, "a", encoding="utf-8") as f: - f.write( - f"\n=== turn-dispatcher exception · " - f"{time.strftime('%Y-%m-%d %H:%M:%S')} · sid={sid} ===\n" - ) - f.write(trace) - except Exception: - pass - print( - f"[gateway-turn] {type(e).__name__}: {e}", file=sys.stderr, flush=True - ) - # The agent persists its working transcript on normal finalization, - # but an exception in that finalizer can otherwise leave the - # gateway's separate in-memory history at the turn-start snapshot. - # Keep the partial turn available to the next prompt; the durable - # inflight record still carries the recoverable error state. - _restore_agent_history_after_turn_error(session, agent) - if terminal_callback is not None and not terminal_receipt_attempted: - terminal_receipt_attempted = True - try: - terminal_callback( - {"status": "failed", "text": "", "error": str(e)} - ) - terminal_receipt_committed = True - except Exception: - logger.exception("hosted room terminal receipt commit failed") - try: - # Close the turn with the same terminal error frame shape as - # the returned-error path (uniform client handling), retaining - # the failed turn for resume replay. - _emit_terminal_turn_error( - sid, - session, - e, - retire_marker=terminal_receipt_committed, - ) - turn_error_retained = True - turn_error_detail = _turn_failure_detail( - e, type(e).__name__, turn_prompt_text - ) - except Exception as emit_exc: - print( - f"[gateway-turn] terminal error emit failed: " - f"{type(emit_exc).__name__}: {emit_exc}", - file=sys.stderr, - flush=True, - ) - _emit("error", sid, {"message": str(e)}) - finally: - # Drop both local snapshots of the pre-turn history before asking - # glibc to return pages. session["history"] already points at the - # new/pruned result; retaining either list defeats this trim. - history.clear() - local_run_kwargs = locals().get("run_kwargs") - if isinstance(local_run_kwargs, dict): - local_run_kwargs.clear() - - # Run while any profile-specific HERMES_HOME override is still active - # so context.memory_trim is resolved from the session's own config. - try: - from hermes_cli.mem_trim import trim_memory - - trim_memory(reason="tui turn completion") - except Exception: - logger.debug("post-turn memory trim failed", exc_info=True) - - if thinking_started: - # Kill the ambient thinking sound the moment the turn ends — - # error and success paths both land here. - try: - from tools.voice_mode import stop_thinking_sound - - stop_thinking_sound() - except Exception: - pass - if tts_queue is not None: - tts_queue.put(None) # end-of-text sentinel — flush + finish speaking - if one_turn_restore: - try: - _restore_agent_model_runtime(agent, one_turn_restore) - _restart_slash_worker(sid, session) - _persist_live_session_runtime(session) - _persist_live_session_system_prompt(session) - except Exception: - logger.debug("TUI one-turn model restore failed", exc_info=True) - try: - if approval_token is not None: - reset_current_session_key(approval_token) - except Exception: - pass - if home_token is not None: - reset_hermes_home_override(home_token) - if secret_token is not None: - reset_secret_scope(secret_token) - if _terminal_scope_token is not None: - from tools.terminal_scope import reset_terminal_scope - - reset_terminal_scope(_terminal_scope_token) - _clear_session_context(session_tokens) - _current_runtime_session_record.reset(runtime_session_token) - reset_transport(transport_token) - # Clear the per-turn interim callback so a stale closure from - # this turn can't fire during a later turn on the same agent. - agent.interim_assistant_callback = None - with session["history_lock"]: - session["running"] = False - session["last_active"] = time.time() - if not turn_error_retained: - _clear_inflight_turn(session) - # Closing bookend of the "tui prompt accepted" record above — - # fires on every path (success, returned error, exception, - # interrupt), so one accepted prompt always produces exactly one - # finished record. agent.session_id is re-read here because - # compression may have rotated it mid-turn: an accepted/finished - # pair whose agent_session_id changed IS a rotation trace - # (#86647). A missing finished record means the turn thread died - # without reaching this finally. - logger.info( - "tui turn finished: ui_session=%s session_key=%s " - "agent_session_id=%s status=%s error_retained=%s duration=%.1fs" - "%s", - sid, - session.get("session_key") or "", - getattr(agent, "session_id", "") or "", - ( - result.get("interrupted") - and "interrupted" - or result.get("error") - and "error" - or "complete" - ) - if isinstance(result, dict) - else ("error" if turn_error_retained else "complete"), - turn_error_retained, - time.monotonic() - _turn_started_monotonic, - turn_error_detail, - ) - # Backstop for turns that never reached a terminal frame (the - # frame paths retire the marker as they emit). - if terminal_receipt_committed: - _retire_turn_marker(session, marker_key) - with session["history_lock"]: - if session.get("_active_turn_marker_key") == marker_key: - session.pop("_active_turn_marker_key", None) - session.pop("_hosted_room_task", None) - session.pop("_auto_continue_scheduled", None) - _emit_settled_session_info(sid, session, agent) - - # A user prompt that arrived mid-turn (interrupt + queue) wins over - # every auto follow-up below — drain it first and skip them this cycle; - # the goal judge / notifications re-evaluate at the end of that turn. - # Leftover /steer: the steer arrived after the last tool batch (e.g. - # during the final API call), so the agent couldn't inject it and - # returned it in result["pending_steer"]. Requeue it as the next turn - # so it isn't silently dropped — same rule as cli.py and gateway/run.py. - # A real queued prompt still wins: the merge in _enqueue_prompt keeps - # both texts. - _leftover_steer = result.get("pending_steer") if isinstance(result, dict) else None - if isinstance(_leftover_steer, str) and _leftover_steer.strip(): - with session["history_lock"]: - _enqueue_prompt(session, _leftover_steer, session.get("transport")) - if _drain_queued_prompt(rid, sid, session): - return - - # Chain a goal-continuation turn if the judge said so. We do - # this AFTER the finally releases session["running"], so the - # nested _run_prompt_submit doesn't deadlock on the busy - # guard. A real user prompt that races us wins because - # prompt.submit sets running=True under the history_lock and - # we check that guard before re-firing. - if goal_followup: - with session["history_lock"]: - if session.get("running"): - # User already sent something — their turn wins, - # the judge will re-run on the next turn anyway. - return - session["running"] = True - try: - _emit("message.start", sid) - _run_prompt_submit(rid, sid, session, goal_followup) - except Exception as _cont_exc: - print( - f"[tui_gateway] goal continuation dispatch failed: " - f"{type(_cont_exc).__name__}: {_cont_exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False - - # Drain completion notifications that arrived during this turn. - # The background poller handles between-turn delivery; this is - # the safety net for events that arrived mid-turn. - # - # Ownership filter (#42674, #35652): a turn finishing in session B - # must not consume an event that belongs to session A. The registry - # requeues every addressed event this session cannot positively claim; - # the poller then delivers it to a live owner or drops an orphan. - try: - from tools.process_registry import process_registry - - # Positive-proof ownership (compression-chain aware) — the same - # fail-closed gate the poller uses, so the post-turn drain can't - # adopt another session's addressed notification while a - # post-compression session still claims its own pre-compression - # dispatches (#55578). - drained = process_registry.drain_notifications( - session_key=session.get("session_key", ""), - owns_event=lambda e: _session_owns_notification_event(sid, session, e), - skip_poll_observed=False, - ) - for index, (_evt, synth) in enumerate(drained): - with session["history_lock"]: - if session.get("running"): - for pending_evt, _pending_synth in drained[index:]: - process_registry.completion_queue.put(pending_evt) - break - session["running"] = True - from tools.async_delegation import ( - claim_event_delivery, complete_event_delivery, release_event_delivery, - ) - _claim = claim_event_delivery(_evt, "tui-post-turn") - if _claim is None: - continue - try: - _emit("message.start", sid) - _run_prompt_submit(rid, sid, session, synth) - complete_event_delivery(_evt, _claim) - except Exception as _n_exc: - release_event_delivery(_evt, _claim) - print( - f"[tui_gateway] completion notification dispatch failed: " - f"{type(_n_exc).__name__}: {_n_exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False - except Exception as _drain_exc: - print( - f"[tui_gateway] completion queue drain failed: " - f"{type(_drain_exc).__name__}: {_drain_exc}", - file=sys.stderr, - ) - - run_thread = threading.Thread(target=run, daemon=True) - with _sessions_lock: - registered = _sessions.get(sid) - can_start = ( - not session.get("_closing") - and (registered is None or registered is session) - ) - if can_start: - session["_run_thread"] = run_thread - run_thread.start() - if not can_start: - with session["history_lock"]: - session["running"] = False - return can_start # Byte-upload attach caps. 25 MB matches Anthropic's per-image limit; 50 MB / 25 @@ -6706,16 +4526,12 @@ def _respond(rid, params, key, *, allow_expired=False): _, ev = entry batch = _batch_clarify.get(r) if batch is not None and question_id: - # Per-question lock (multi-question clarify). Update-in-place is - # deliberate: a locked answer stays editable until the batch - # completes, and completion is exactly "every qid locked" — the - # final lock is the Confirm-and-continue click. + # Per-question lock; update-in-place so a locked answer stays + # editable until every qid is locked (the Confirm click). if question_id not in batch["qids"]: return _err(rid, 4002, f"unknown question_id {question_id!r}") batch["answers"][question_id] = params.get(key, "") - remaining = [ - qid for qid in batch["qids"] if qid not in batch["answers"] - ] + remaining = [qid for qid in batch["qids"] if qid not in batch["answers"]] if not remaining: ev.set() return _ok(rid, {"status": "ok", "remaining": remaining}) @@ -6744,7 +4560,6 @@ class _NoProject(Exception): def _projects_payload(conn) -> dict: from hermes_cli import projects_db as pdb - return { "projects": [p.to_dict() for p in pdb.list_projects(conn, include_archived=True)], "active_id": pdb.get_active_id(conn), @@ -6765,7 +4580,6 @@ def _projects_method(name: str): def handler(rid, params: dict) -> dict: try: from hermes_cli import projects_db as pdb - with pdb.connect_closing() as conn: return fn(rid, params, pdb, conn) except _NoProject: @@ -6774,9 +4588,7 @@ def _projects_method(name: str): return _err(rid, _E_PROJECT_ARG, str(e)) except Exception as e: return _err(rid, _E_PROJECTS, str(e)) - return handler - return decorator @@ -6801,14 +4613,9 @@ def _(rid, params, pdb, conn) -> dict: @_projects_method("projects.create") def _(rid, params, pdb, conn) -> dict: pid = pdb.create_project( - conn, - name=str(params.get("name") or ""), - slug=params.get("slug"), - folders=params.get("folders") or [], - primary_path=params.get("primary_path"), - description=params.get("description"), - icon=params.get("icon"), - color=params.get("color"), + conn, name=str(params.get("name") or ""), slug=params.get("slug"), + folders=params.get("folders") or [], primary_path=params.get("primary_path"), + description=params.get("description"), icon=params.get("icon"), color=params.get("color"), board_slug=params.get("board_slug"), ) if params.get("use"): @@ -6821,13 +4628,8 @@ def _(rid, params, pdb, conn) -> dict: def _(rid, params, pdb, conn) -> dict: proj = _require_project(pdb, conn, params) pdb.update_project( - conn, - proj.id, - name=params.get("name"), - description=params.get("description"), - icon=params.get("icon"), - color=params.get("color"), - board_slug=params.get("board_slug"), + conn, proj.id, name=params.get("name"), description=params.get("description"), + icon=params.get("icon"), color=params.get("color"), board_slug=params.get("board_slug"), ) return _ok(rid, {"project": pdb.get_project(conn, proj.id).to_dict()}) @@ -6836,10 +4638,7 @@ def _(rid, params, pdb, conn) -> dict: def _(rid, params, pdb, conn) -> dict: proj = _require_project(pdb, conn, params) pdb.add_folder( - conn, - proj.id, - str(params.get("path") or ""), - label=params.get("label"), + conn, proj.id, str(params.get("path") or ""), label=params.get("label"), is_primary=bool(params.get("is_primary")), ) return _ok(rid, {"project": pdb.get_project(conn, proj.id).to_dict()}) @@ -6899,7 +4698,6 @@ def _non_workspace_dirs() -> set[str]: """ home = os.path.realpath(os.path.expanduser("~")) candidates = (os.sep, home, os.path.dirname(home), "/home", "/Users") - return {os.path.normcase(os.path.realpath(path)) for path in candidates if path} @@ -6910,12 +4708,9 @@ def _is_repo_junk(root: str) -> bool: pointing there are still honored.""" if not root: return True - from hermes_constants import get_hermes_home - real = os.path.realpath(root) hermes_home = os.path.realpath(str(get_hermes_home())) - return ( os.path.normcase(real) in _non_workspace_dirs() or real == hermes_home @@ -6934,9 +4729,7 @@ def _is_session_cwd_junk(cwd: str) -> bool: """ if not cwd: return True - from hermes_constants import get_hermes_home - real = os.path.normcase(os.path.realpath(cwd)) hermes_home = os.path.normcase(os.path.realpath(str(get_hermes_home()))) return real in _non_workspace_dirs() or real == hermes_home @@ -6945,19 +4738,15 @@ def _is_session_cwd_junk(cwd: str) -> bool: def _repo_discovery_policy(raw: dict | None = None) -> dict: """Return the effective, profile-local Desktop repository scan policy.""" from hermes_cli.config import DEFAULT_CONFIG - defaults = DEFAULT_CONFIG["desktop"] source = raw if isinstance(raw, dict) else (_load_cfg().get("desktop") or {}) if not isinstance(source, dict): source = {} - enabled = source.get("enabled", source.get("repo_scan_enabled", defaults["repo_scan_enabled"])) roots = source.get("roots", source.get("repo_scan_roots", defaults["repo_scan_roots"])) excludes = source.get( - "exclude_paths", - source.get("repo_scan_exclude_paths", defaults["repo_scan_exclude_paths"]), + "exclude_paths", source.get("repo_scan_exclude_paths", defaults["repo_scan_exclude_paths"]), ) - return { "enabled": enabled if isinstance(enabled, bool) else defaults["repo_scan_enabled"], "roots": [value.strip() for value in roots if isinstance(value, str) and value.strip()] @@ -6983,10 +4772,8 @@ def _repo_discovery_policy_key(policy: dict) -> str: expanded = os.path.join(home, expanded) normalized.add(os.path.normcase(os.path.abspath(expanded))) return sorted(normalized) - canonical = { - "enabled": bool(policy["enabled"]), - "roots": _paths(policy["roots"]), + "enabled": bool(policy["enabled"]), "roots": _paths(policy["roots"]), "exclude_paths": _paths(policy["exclude_paths"]), } return json.dumps(canonical, sort_keys=True, separators=(",", ":")) @@ -6994,7 +4781,6 @@ def _repo_discovery_policy_key(policy: dict) -> str: def _repo_discovery_policy_is_default(policy: dict) -> bool: from hermes_cli.config import DEFAULT_CONFIG - return _repo_discovery_policy_key(policy) == _repo_discovery_policy_key( _repo_discovery_policy(DEFAULT_CONFIG["desktop"]) ) @@ -7020,7 +4806,6 @@ def _scan_discovered_repos_remote(conn, policy: dict) -> bool: silent, unpopulated sidebar of #81723. """ from hermes_cli import projects_db as pdb - roots = policy.get("roots") or [] excludes = policy.get("exclude_paths") or [] pairs: list[tuple[str, str | None]] = [] @@ -7029,15 +4814,11 @@ def _scan_discovered_repos_remote(conn, policy: dict) -> bool: def _is_excluded(path: str) -> bool: return any(path == ex or path.startswith(ex.rstrip("/\\") + os.sep) for ex in excludes if ex) - for root in roots: if not os.path.isdir(root): - # `os.walk` on a missing root silently yields nothing instead of - # raising, so a temporarily unavailable root (unmounted volume, - # moved path) would otherwise look like a genuinely empty scan and - # let `authoritative` stay True — letting the replace wipe every - # cached repo that lived under the missing root. A missing root - # contributes nothing and must not be treated as authoritative. + # `os.walk` on a missing root yields nothing instead of raising; an + # unmounted volume would look like an empty scan and let the + # authoritative replace wipe every cached repo under it. authoritative = False logger.debug("discover_repos scan root missing, skipping: %s", root) continue @@ -7071,7 +4852,6 @@ def _scan_discovered_repos_remote(conn, policy: dict) -> bool: # set must not be treated as the complete authoritative universe. authoritative = False break - if pairs: try: pdb.record_discovered_repos( @@ -7122,61 +4902,42 @@ def _discover_repos_payload( agg = _agg(root) agg["sessions"] += int(row.get("sessions") or 0) agg["last_active"] = max(agg["last_active"], float(row.get("last_active") or 0)) - if backfill: try: db.backfill_repo_roots(cwd_to_root) except Exception: logger.debug("failed to backfill repo roots", exc_info=True) + if include_cached: + # Filesystem-scanned roots from the cache (may have zero sessions). Reuse + # the caller's projects.db connection when given, else a short-lived one. + try: + from hermes_cli import projects_db as pdb - if not include_cached: - out = sorted(repos.values(), key=lambda repo: repo["last_active"], reverse=True) - for repo in out: - repo["label"] = ( - repo["label"] - or os.path.basename(repo["root"].rstrip("/\\")) - or repo["root"] - ) - return out - - # Filesystem-scanned roots from the cache (may have zero sessions). Reuse the - # caller's projects.db connection when given, else open a short-lived one. - try: - from hermes_cli import projects_db as pdb - - def _read(c) -> None: - for entry in pdb.list_discovered_repos(c): - root = str(entry.get("root") or "") - if not root or _is_junk(root): - continue - agg = _agg(root) - if entry.get("label"): - agg["label"] = entry["label"] - # NOTE: `last_seen` is when the disk scan last saw the directory, - # not when the user last worked in it. Folding it into - # `last_active` stamped every scanned repo with the scan time — - # i.e. "just now" — so a git checkout with zero Hermes sessions - # outranked the repos the user actually works in. Activity stays - # session-derived; a repo with no sessions has no activity. - - if conn is not None: - _read(conn) - else: - with pdb.connect_closing() as own: - _read(own) - except Exception: - logger.debug("failed to read discovered repo cache", exc_info=True) - + def _read(c) -> None: + for entry in pdb.list_discovered_repos(c): + root = str(entry.get("root") or "") + if not root or _is_junk(root): + continue + agg = _agg(root) + if entry.get("label"): + agg["label"] = entry["label"] + # `last_seen` is scan time, not user activity; folding it + # into `last_active` made every scanned repo "just now". + if conn is not None: + _read(conn) + else: + with pdb.connect_closing() as own: + _read(own) + except Exception: + logger.debug("failed to read discovered repo cache", exc_info=True) out = sorted(repos.values(), key=lambda r: r["last_active"], reverse=True) for r in out: r["label"] = r["label"] or os.path.basename(r["root"].rstrip("/\\")) or r["root"] return out -# Sources excluded from the project tree: cron runs, and kanban dispatcher -# workers, are not user conversations. Subagent/compression children are -# already dropped by list_sessions_rich(include_children=False); cron has its -# own section, and kanban runs are read on the board. +# Not user conversations (cron has its own section; kanban runs are read on +# the board). Subagent/compression children are dropped by include_children=False. _PROJECT_TREE_EXCLUDED_SOURCES = ["cron", "kanban"] @@ -7247,32 +5008,22 @@ def _project_tree_inputs( # cold-probing each cwd in sequence (matters on the drill-in path, which # skips the discovery warm-up below). git_probe.warm_roots(s["cwd"] for s in sessions if s.get("cwd")) - from hermes_cli import projects_db as pdb - policy = _repo_discovery_policy() policy_key = _repo_discovery_policy_key(policy) with pdb.connect_closing() as conn: if include_discovered: pdb.reconcile_discovered_repos_policy( - conn, - policy_key, - preserve_unversioned=_repo_discovery_policy_is_default(policy), + conn, policy_key, preserve_unversioned=_repo_discovery_policy_is_default(policy), ) projects = [p.to_dict() for p in pdb.list_projects(conn)] active_id = pdb.get_active_id(conn) # backfill stays off the hot tree path — grouping uses the live resolver. discovered = ( - _discover_repos_payload( - db, - conn=conn, - backfill=False, - include_cached=policy["enabled"], - ) + _discover_repos_payload(db, conn=conn, backfill=False, include_cached=policy["enabled"]) if include_discovered else [] ) - return sessions, projects, discovered, active_id @@ -7302,7 +5053,6 @@ def _build_project_tree( ) -> tuple[dict, str | None]: """Gather inputs and run the one authoritative builder. Returns (tree, active_id).""" from tui_gateway import project_tree - _DIR_EXISTS_CACHE.clear() sessions, projects, discovered, active_id = _project_tree_inputs( db, session_limit, include_discovered=include_discovered @@ -7315,14 +5065,8 @@ def _build_project_tree( + [str(r.get("root") or "") for r in discovered] ) tree = project_tree.build_tree( - projects, - sessions, - discovered, - _resolve_cwd_git, - preview_limit=preview_limit, - hydrate=hydrate, - is_junk_root=_is_repo_junk, - is_junk_cwd=_is_session_cwd_junk, + projects, sessions, discovered, _resolve_cwd_git, preview_limit=preview_limit, + hydrate=hydrate, is_junk_root=_is_repo_junk, is_junk_cwd=_is_session_cwd_junk, exists=_dir_exists_cached, ) return tree, active_id @@ -7334,7 +5078,6 @@ def _build_project_tree( def _session_processes(session: dict) -> list: """Background processes owned by this session (registry session_key match).""" from tools.process_registry import process_registry - key = str(session.get("session_key") or "") owned = [] for entry in process_registry.list_sessions(): @@ -7348,22 +5091,15 @@ def _session_processes(session: dict) -> list: return owned -# reload.mcp runs on the RPC pool (see _LONG_HANDLERS) so a slow/flapping MCP -# server can't freeze the reader thread. Serialize reloads: overlapping -# shutdown+discover pairs from stacked config-change polls would interleave -# and leave the registry half-built. +# Serialize reload.mcp (it runs on the pool): overlapping shutdown+discover +# pairs would leave the registry half-built. _mcp_reload_lock = threading.Lock() -# Bumped once per SUCCESSFUL shutdown+discover. A follower that waited on the -# lock only skips the redundant reload if this advanced while it waited — i.e. -# the leader actually completed. If the leader threw (flapping server), the -# follower sees no advance and re-runs the full reload itself. +# Bumped per SUCCESSFUL reload; a follower skips only if it advanced while it +# waited (a leader that threw leaves it unchanged → follower reloads itself). _mcp_reload_gen = 0 -# The mcp_rev hash that the last successful reload actually LOADED (config -# re-hashed after discovery, so it reflects what discover_mcp_tools read — -# not what the caller hoped for). A follower coalesces only when the -# revision it was asked to load matches this; otherwise the config changed -# under the leader (rev A loaded, rev B requested) and the follower must -# re-run the full reload itself instead of acking B against A's registry. +# The mcp_rev the last successful reload actually LOADED (re-hashed after +# discovery). A follower coalesces only when its requested rev matches; +# otherwise the config changed under the leader and it must reload itself. _mcp_reload_loaded_rev = "" # Bounded convergence for a config edit racing a slow reload: the leader # re-hashes after discovery and repeats until the hash is stable. @@ -7377,12 +5113,9 @@ def _compute_mcp_rev() -> str: revision-aware coalescing. Empty string = unknown (fail open).""" try: cfg = _load_cfg() - # mcp_servers holds the server DEFINITIONS the classic CLI watches - # for auto-reload (cli.py::_check_config_mcp_changes) — omitting it - # meant editing a server bumped mtime but not mcp_rev, so the TUI - # skipped reload.mcp and new servers never connected until a manual - # /reload-mcp. `mcp` (settings) and `tools` (enable/disable) round - # out the MCP-relevant surface. + # mcp_servers (definitions, what the CLI auto-reload watches) + mcp + # (settings) + tools (enable/disable); omitting mcp_servers meant an + # edited server never bumped mcp_rev and never connected. rev_src = json.dumps( {"mcp": cfg.get("mcp"), "mcp_servers": cfg.get("mcp_servers"), "tools": cfg.get("tools")}, sort_keys=True, @@ -7399,60 +5132,30 @@ def _finish_reload(rid, params: dict, *, coalesced: bool) -> dict: if bool(params.get("always", False)): try: from cli import save_config_value as _save_cfg - _save_cfg("approvals.mcp_reload_confirm", False) except Exception as _exc: logger.warning("Failed to persist mcp_reload_confirm=false: %s", _exc) - payload = {"status": "reloaded", "loaded_rev": _mcp_reload_loaded_rev} if coalesced: payload["coalesced"] = True - return _ok(rid, payload) -_TUI_HIDDEN: frozenset[str] = frozenset( - { - "sethome", - "set-home", - "commands", - "approve", - "deny", - } -) +_TUI_HIDDEN: frozenset[str] = frozenset({"sethome", "set-home", "commands", "approve", "deny"}) _TUI_EXTRA: list[tuple[str, str, str]] = [ ("/density", "Toggle compact display mode", "TUI"), ("/logs", "Show recent gateway log lines", "TUI"), - ( - "/mouse", - "Set mouse tracking preset [on|off|toggle|wheel|buttons|all]", - "TUI", - ), + ("/mouse", "Set mouse tracking preset [on|off|toggle|wheel|buttons|all]", "TUI"), ("/sessions", "Switch between live TUI sessions", "TUI"), ] -# Commands that queue messages onto _pending_input in the CLI. -# In the TUI the slash worker subprocess has no reader for that queue, -# so slash.exec routes them to command.dispatch internally (which handles -# them and returns a structured payload) instead of erroring out and -# relying on a client-side fallback. See #48848. +# Commands that queue onto _pending_input in the CLI; the slash worker has no +# reader for that queue, so slash.exec routes them to command.dispatch instead. _PENDING_INPUT_COMMANDS: frozenset[str] = frozenset( { - "retry", - "queue", - "q", - "steer", - "plan", - "goal", - "loop", - "proactive", - "moa", - "undo", - "learn", - "init", - "compress", - "compact", + "retry", "queue", "q", "steer", "plan", "goal", "loop", "proactive", "moa", "undo", "learn", + "init", "compress", "compact", } ) @@ -7471,12 +5174,8 @@ def _skill_usage_lookup(): """ try: from tools.skill_usage import ( - _read_bundled_manifest_names, - _read_hub_installed_names, - activity_count, - load_usage, + _read_bundled_manifest_names, _read_hub_installed_names, activity_count, load_usage, ) - records = load_usage() bundled = _read_bundled_manifest_names() hub = _read_hub_installed_names() @@ -7496,7 +5195,6 @@ def _skill_usage_lookup(): if name in bundled: return "bundled" return "local" - return usage, origin @@ -7504,60 +5202,38 @@ _SLASH_COMPLETION_LIMIT = 30 def _rank_slash_completions( - items: list[dict], - usage, - origin_of, - *, - browsing: bool, - score_of=None, + items: list[dict], usage, origin_of, *, browsing: bool, score_of=None, ) -> list[dict]: """Rank and bound slash completions the way the menu should read. - ``usage``/``origin_of`` are the callables :func:`_skill_usage_lookup` - returns. Registry commands keep their existing order — only the skill - block is reordered, most-used first and A-Z within a tie, so the handful - of skills someone invokes daily lead the ones that shipped with Hermes - and were never opened. + ``usage``/``origin_of`` come from :func:`_skill_usage_lookup`. Registry + commands keep their order; only the skill block is reordered: fuzzy + ``score_of`` first (a name match beats a description match), then + most-used, then A-Z. - ``score_of`` (optional) is the fuzzy-match scorer from - :func:`tui_gateway.slash_fuzzy.fuzzy_rank_slash_items` — when a typed - query produced scores, they lead the skill sort so a name match beats a - description match before usage breaks ties. Commands arrive already - score-sorted and keep their order either way. + The limit is spent PER KIND, not as one flat truncation: commands are + emitted before the first skill, so a flat cut on a large install offered + no skill at all and dropped heavily-used skills for never-opened ones. - The limit is spent PER KIND rather than on one flat truncation. A flat - cut is positional, not editorial: the completer emits every registry - command before the first skill, so on a 230-skill install a bare ``/`` - hit the cap while still inside the command block and offered no skill at - all, and ``/p`` dropped ``/proving-a-fix-works`` (471 uses) while keeping - ``/pretext`` (2). - - ``browsing`` separates the two things a slash means. A bare ``/`` is - BROWSING, so bundled skills with no recorded activity are dropped as - noise. A typed query is SEARCHING, and a search that hides a match is - broken — there nothing is pruned, the ranking only reorders. + ``browsing`` (bare ``/``) drops bundled skills with no recorded activity + as noise; a typed query is SEARCHING, and a search that hides a match is + broken — nothing is pruned there, only reordered. """ def name_of(item: dict) -> str: return str(item.get("text", "")).strip().lstrip("/").lower() - commands = [item for item in items if item.get("kind") != "skill"] skills = [item for item in items if item.get("kind") == "skill"] - if browsing: skills = [ item for item in skills if origin_of(name_of(item)) != "bundled" or usage(name_of(item)) > 0 ] - if score_of is not None: - skills.sort( - key=lambda item: (score_of(item), -usage(name_of(item)), name_of(item)) - ) + skills.sort(key=lambda item: (score_of(item), -usage(name_of(item)), name_of(item))) else: skills.sort(key=lambda item: (-usage(name_of(item)), name_of(item))) - return commands[:_SLASH_COMPLETION_LIMIT] + skills[:_SLASH_COMPLETION_LIMIT] @@ -7580,7 +5256,6 @@ def _cli_exec_blocked(argv: list[str]) -> str | None: def _resolve_name(name: str) -> str: try: from hermes_cli.commands import resolve_command - r = resolve_command(name) return r.name if r else name except Exception: @@ -7637,36 +5312,17 @@ from . import ( # noqa: E402 methods_prompt as _methods_prompt, methods_session as _methods_session, methods_tools as _methods_tools, + prompt_turn as _prompt_turn, ) for _m in ( - _session_reaper, - _session_lifecycle, - _session_workdir, - _compute_host_bridge, - _model_switch, - _session_compression, - _change_watcher, - _tool_progress, - _session_notifications, - _prompt_attachments, - _session_history, - _agent_callbacks, - _session_auto_continue, - _methods_complete_helpers, - _methods_slash, - _methods_voice, - _methods_browser, - _methods_browser_control, - _methods_session, - _methods_prompt, - _methods_config, - _methods_config_set, - _methods_complete, - _methods_tools, - _methods_profiles, - _methods_images, - _methods_bot_relay, + _session_reaper, _session_lifecycle, _session_workdir, _compute_host_bridge, _model_switch, + _session_compression, _change_watcher, _tool_progress, _session_notifications, + _prompt_attachments, _session_history, _agent_callbacks, _session_auto_continue, + _methods_complete_helpers, _methods_slash, _methods_voice, _methods_browser, + _methods_browser_control, _methods_session, _methods_prompt, _methods_config, + _methods_config_set, _methods_complete, _methods_tools, _methods_profiles, _methods_images, + _methods_bot_relay, _prompt_turn, ): _m.register(sys.modules[__name__]) del _m diff --git a/tui_gateway/session_auto_continue.py b/tui_gateway/session_auto_continue.py index 00aef44121..7c263d1557 100644 --- a/tui_gateway/session_auto_continue.py +++ b/tui_gateway/session_auto_continue.py @@ -12,17 +12,11 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# ── Auto-continue: resume a turn killed by a process/machine death ──── -# -# A turn that concludes — success, handled error, interrupt — clears its -# durable marker (see tui_gateway/turn_marker.py) in _run_prompt_submit's -# finally. Only a process death leaves the marker behind, so a marker found -# at session.resume time is positive proof the turn never finished AND the -# client never saw a terminal frame. If the interruption is fresh, re-submit -# the interrupted prompt automatically (the messaging gateway has done this -# for restart-interrupted sessions since #27856); if it's stale, clear the -# marker and let the recovered partial transcript speak for itself — the -# user can ask to continue manually. +# A concluded turn (success, handled error, interrupt) clears its durable marker +# (turn_marker.py) in _run_prompt_submit's finally; only a process death leaves it +# behind, so a marker at session.resume proves the turn never finished AND the +# client never saw a terminal frame. Fresh: re-submit automatically (as the +# messaging gateway does). Stale: clear it and let the partial transcript speak. _AUTO_CONTINUE_ENABLED_DEFAULT = True _AUTO_CONTINUE_FRESHNESS_MINUTES_DEFAULT = 15 @@ -57,12 +51,10 @@ def _session_home(session: dict) -> Path: def _retire_turn_marker(session: dict, *keys: str) -> None: """Drop the crash marker for a turn whose outcome is about to reach the client. - Called immediately before the terminal frame rather than at the end of the - turn thread: post-turn work (titles, memory sync, goal hooks) runs for a - second or more after the client has its answer, and quitting inside that - window would leave a marker that looks like a crash — re-running a finished - turn on the next launch. Extra ``keys`` cover a session_key that - compression rotated mid-turn. + Called right before the terminal frame, not at turn-thread end: post-turn work + (titles, memory sync, goal hooks) outlives the client's answer, and quitting in + that window would leave a marker that re-runs a finished turn on next launch. + Extra ``keys`` cover a session_key that compression rotated mid-turn. """ home = _session_home(session) for key in dict.fromkeys((*keys, str(session.get("session_key") or ""))): @@ -71,10 +63,8 @@ def _retire_turn_marker(session: dict, *keys: str) -> None: def _auto_continue_note(prompt: str) -> str: - # Same opening as the messaging gateway's recovery notes so transcript - # tooling recognizes both. The original prompt is embedded because a hard - # crash persists nothing of the interrupted turn to the session DB — this - # note is the only copy the model will see. + # Same opening as the gateway's recovery notes (transcript tooling recognizes + # both). The prompt is embedded: a hard crash persists nothing else of the turn. return ( f"{_AUTO_CONTINUE_NOTE_PREFIX} — the app or its backend process " "stopped before the turn could finish. Some of the work may already " @@ -84,19 +74,23 @@ def _auto_continue_note(prompt: str) -> str: ) +def _ac_release_turn(session: dict, *, unschedule: bool = False) -> None: + with session["history_lock"]: + session["running"] = False + if unschedule: + session["_auto_continue_scheduled"] = False + + def _maybe_schedule_auto_continue(sid: str, session: dict, session_key: str) -> dict | None: """Kick off a continuation turn for a crash-interrupted session. - Called from session.resume's cold paths after the live record is - registered. Returns a small descriptor for the resume payload when a - continuation was scheduled, else None. The turn itself runs on a - background thread after the (deferred) agent build finishes, through the - same _run_prompt_submit machinery as every other synthesized turn — so - the client that just resumed streams it live. + Called from session.resume's cold paths once the live record is registered. + Returns a descriptor for the resume payload when scheduled, else None. The turn + runs on a background thread after the deferred agent build via the normal + _run_prompt_submit path, so the client that just resumed streams it live. """ - # Hosted room turns are recovered by their durable task/lease state - # machine. Generic session auto-continue would bypass its execution - # generation and can duplicate work after a process restart. + # Hosted room turns are recovered by their durable task/lease state machine; + # generic auto-continue would bypass its execution generation and duplicate work. if session.get("source") == "bot_room": return None @@ -107,8 +101,7 @@ def _maybe_schedule_auto_continue(sid: str, session: dict, session_key: str) -> enabled, freshness_secs, max_attempts = _auto_continue_config() age = time.time() - marker["started_at"] if not enabled or age > freshness_secs or marker["attempts"] >= max_attempts: - # Stale, disabled, or crash-looping: stop trying. The journal/partial - # transcript still shows what happened; a manual message continues it. + # Stale, disabled, or crash-looping: stop trying; a manual message continues. clear_turn_marker(home, session_key) return None if session.get("_auto_continue_scheduled"): @@ -131,92 +124,56 @@ def _maybe_schedule_auto_continue(sid: str, session: dict, session_key: str) -> return with session["history_lock"]: if session.get("running") or session.get("_turn_cancel_requested") or session.get("_finalized"): - # A real user prompt beat us to it — their turn wins, and its - # own conclusion clears the marker. + # A real user prompt beat us; its own conclusion clears the marker. session["_auto_continue_scheduled"] = False return session["running"] = True session["last_active"] = time.time() - # Ownership admission BEFORE message.start: the interrupted-turn - # marker this continuation is recovering may have been written by a - # sibling backend that is still alive and mid-turn (#94778 — two - # backends share one HERMES_HOME; B resumes S while A runs it and - # sees A's fresh marker). Running the continuation anyway would be - # the double-writer this fence exists to prevent. Leave the marker: - # once the owner finishes or dies, a later resume retries. + # Ownership admission BEFORE message.start: a sibling backend sharing this + # HERMES_HOME may have written the marker and still be mid-turn. Leave the + # marker so a later resume retries once the owner finishes or dies. if _ensure_active_session_slot(sid, session) is not None: - logger.info( - "auto-continue for %s refused: session has another live owner", - session_key, - ) - with session["history_lock"]: - session["running"] = False - session["_auto_continue_scheduled"] = False + logger.info("auto-continue for %s refused: session has another live owner", session_key) + _ac_release_turn(session, unschedule=True) return with session["history_lock"]: - # Hand this turn its own marker inputs (read back by - # _run_prompt_submit): count the attempt so a crash during the - # continuation trips the breaker, and re-record the ORIGINAL - # prompt so a second crash doesn't nest note inside note. Set - # here, not at schedule time, so a bail above leaves nothing - # behind for a racing user turn to inherit. + # Marker inputs read back by _run_prompt_submit: count the attempt (crash + # breaker) and re-record the ORIGINAL prompt (no nested notes). Set here, + # not at schedule time, so a bail above leaves nothing for a racing user turn. session["_auto_continue_attempt"] = attempt session["_auto_continue_prompt"] = marker["prompt"] try: - _emit( - "status.update", - sid, - {"kind": "process", "text": "Resuming interrupted turn…"}, - ) + _emit("status.update", sid, {"kind": "process", "text": "Resuming interrupted turn…"}) _emit("message.start", sid) _run_prompt_submit(rid, sid, session, text, display_kind="auto_continue") except Exception as exc: - print( - f"[tui_gateway] auto-continue dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False + print(f"[tui_gateway] auto-continue dispatch failed: {type(exc).__name__}: {exc}", file=sys.stderr) + _ac_release_turn(session) threading.Thread(target=kickoff, daemon=True).start() logger.info( "auto-continue scheduled for session %s (attempt %d, interrupted %.0fs ago)", - session_key, - attempt, - age, + session_key, attempt, age, ) return {"attempt": attempt, "interrupted_at": marker["started_at"]} -def _enqueue_prompt( - session: dict, - text: Any, - transport: Any, - image_paths: list[str] | None = None, -) -> None: +def _enqueue_prompt(session: dict, text: Any, transport: Any, image_paths: list[str] | None = None) -> None: """Stash a message to run as the very next turn once the live one ends. - Used when a prompt arrives mid-turn (see ``_handle_busy_submit``). Text-only - arrivals share a slot and merge losslessly (mirroring the consecutive-user - merge in ``repair_message_sequence``). Image-bearing submissions stay as - separate envelopes, so their attachment ownership and chronology survive. - ``transport`` is pinned so the drained turn streams back to the client that - sent it even if the session transport is rebound meanwhile. + Text-only arrivals share a slot and merge losslessly (like the consecutive-user + merge in ``repair_message_sequence``); image-bearing ones stay separate envelopes + so attachment ownership/chronology survive. ``transport`` is pinned so the drained + turn streams back to its sender even if the session transport is rebound. """ image_paths = list(image_paths or []) - # #84417: scrub any live-turn self-duplicates first so the consecutive-text - # merge below cannot glue "{original}\\n\\n{later}" and re-fire original - # on drain after a later correction settles. + # Scrub live-turn self-duplicates first so the text merge below can't glue + # "{original}\n\n{later}" and re-fire the original after a correction settles. _drop_queued_duplicates_of_inflight_user(session) - # Never queue a text-only self-copy of the live inflight user prompt. The - # live turn already owns that text; draining it after settle would restart - # the same user turn as a fresh agent invocation. + # Never queue a text-only self-copy of the live prompt: draining it would restart it. if not image_paths and isinstance(text, str): turn = session.get("inflight_turn") - original = ( - str(turn.get("user") or "").strip() if isinstance(turn, dict) else "" - ) + original = (str(turn.get("user") or "").strip() if isinstance(turn, dict) else "") if original and text.strip() == original: return queued = {"text": text, "transport": transport} @@ -240,17 +197,12 @@ def _enqueue_prompt( session["queued_prompt"] = queued -def _sanitize_queued_entry_vs_inflight_user( - entry: Any, original: str -) -> dict | None: - """Drop or rewrite a queue envelope that re-carries the live user text. +def _sanitize_queued_entry_vs_inflight_user(entry: Any, original: str) -> dict | None: + """Drop (``None``) or rewrite a queue envelope that re-carries the live user text. - Returns ``None`` to drop the envelope, or a (possibly rewritten) dict to - keep. Text-only self-duplicates of ``original`` are dropped. A merged - slot ``"{original}\\n\\n{later}"`` (from ``_enqueue_prompt``'s consecutive - text merge) is rewritten to just ``later`` so a later correction is not - lost and the original is not re-fired (#84417). Image-bearing envelopes - are left alone — their chronology/ownership is load-bearing. + Text-only self-duplicates are dropped; a merged slot ``"{original}\\n\\n{later}"`` + is rewritten to ``later`` so the correction survives without re-firing the + original. Image-bearing envelopes are left alone (chronology is load-bearing). """ if not original or not isinstance(entry, dict): return entry if isinstance(entry, dict) else None @@ -260,9 +212,7 @@ def _sanitize_queued_entry_vs_inflight_user( if not isinstance(text, str): return entry stripped = text.strip() - if not stripped: - return None - if stripped == original: + if not stripped or stripped == original: return None # Lossless text-merge glued the live original onto a later follow-up. for sep in ("\n\n", "\n"): @@ -271,24 +221,16 @@ def _sanitize_queued_entry_vs_inflight_user( rest = text[len(prefix) :].strip() if not rest or rest == original: return None - cleaned = dict(entry) - cleaned["text"] = rest - return cleaned + return {**entry, "text": rest} return entry def _drop_queued_duplicates_of_inflight_user(session: dict) -> None: """Remove server-queue copies of the live turn's original user text. - A mid-turn ``prompt.submit`` of the same text can land in - ``queued_prompt`` when redirect is not yet available (model not active, - build window, tool boundary). If the user then corrects the turn with a - different prompt via redirect, that stale self-duplicate must not - ``_drain_queued_prompt`` after the redirected turn completes — otherwise - the original prompt restarts as a fresh agent turn (#84417). - - Unrelated follow-ups (different text, image-bearing envelopes) stay. - Merged ``original + later`` slots are rewritten to ``later`` only. + A mid-turn ``prompt.submit`` of the same text can be queued while redirect is + unavailable (build window, tool boundary); after a later redirect it must not + drain and restart the original as a fresh turn. Unrelated follow-ups stay. """ turn = session.get("inflight_turn") if not isinstance(turn, dict): @@ -305,26 +247,24 @@ def _drop_queued_duplicates_of_inflight_user(session: dict) -> None: if cleaned is not None: kept.append(cleaned) - if not kept: - session["queued_prompt"] = None - session.pop("queued_prompts", None) - return - session["queued_prompt"] = kept[0] - if len(kept) > 1: - session["queued_prompts"] = kept[1:] + _ac_set_queue(session, kept) + + +def _ac_set_queue(session: dict, entries: list) -> None: + """Write ``entries`` back as queued_prompt (head) + queued_prompts (rest).""" + session["queued_prompt"] = entries[0] if entries else None + if len(entries) > 1: + session["queued_prompts"] = entries[1:] else: session.pop("queued_prompts", None) def _interrupt_busy_session(sid: str, session: dict, agent: Any) -> None: - """Interrupt a busy turn without blocking the RPC reader or session lock. + """Interrupt a busy turn on a worker thread, never under ``history_lock``. - Some providers cannot apply ``interrupt()`` until a synchronous tool or - network call returns. Running that call inline used to leave - ``prompt.submit`` holding ``history_lock`` for the whole wait, which in turn - blocked ``session.resume`` and delayed the queued prompt itself. Keep at - most one interrupt worker per session so repeated steering cannot leak an - unbounded number of blocked threads. + Some providers can't apply ``interrupt()`` until a blocking tool/network call + returns; doing it inline stalled ``session.resume`` and the queued prompt. + At most one interrupt worker per session so repeated steering can't leak threads. """ use_agent = agent is not None and hasattr(agent, "interrupt") use_compute_host = not use_agent and _session_uses_compute_host(session) @@ -351,58 +291,49 @@ def _interrupt_busy_session(sid: str, session: dict, agent: Any) -> None: threading.Thread(target=interrupt, daemon=True, name=f"busy-interrupt-{sid}").start() +def _ac_record_inflight_correction(session: dict, plain_text: str) -> None: + """Record an accepted steer/redirect; scrub stale self-duplicates so the + live turn's original text is not re-fired from the queue after settle.""" + with session["history_lock"]: + _record_inflight_correction(session, plain_text) + _drop_queued_duplicates_of_inflight_user(session) + session["last_active"] = time.time() + + def _handle_busy_submit( rid, sid: str, session: dict, text: Any, transport: Any, queued: bool = False ) -> dict | None: - """Apply the ``display.busy_input_mode`` policy to a prompt that lands while - a turn is in flight, instead of rejecting it with ``session busy``. + """Apply ``display.busy_input_mode`` to a prompt that lands mid-turn instead of + rejecting it with ``session busy`` (rejection made clients busy-retry and + silently drop sends when teardown outlived their deadline). - The old rejection forced clients into a deadline-bounded busy-retry that - silently dropped the send when turn teardown outlived the deadline. The - default policy now redirects a capable core agent in place; older agents - retain the proven interrupt-and-queue path drained from ``run``'s tail. - - Modes: ``interrupt`` (default) → redirect the live turn, falling back to - hard interrupt + queue for older agents; ``queue`` → queue without - interrupting; ``steer`` → inject after the current atomic action. - - ``queued=True`` (client's queue drain, ``prompt.submit`` param) overrides - the mode entirely: the message was explicitly queued as "run after", so it - must NEVER become a live-turn correction or interrupt. Without this, a - drain that loses the settle race (client observed idle, server still - unwinding the turn) redirected the live turn with next-turn text — queue - semantics betrayed by a millisecond race the user can't see. + Modes: ``interrupt`` (default) → redirect the live turn, falling back to hard + interrupt + queue for older agents; ``queue`` → queue only; ``steer`` → inject + after the current atomic action. ``queued=True`` (client queue drain) forces + queue mode: a "run after" message must NEVER become a live-turn correction, + even when the drain loses the settle race against a still-unwinding turn. """ mode = "queue" if queued else _load_busy_input_mode() agent = session.get("agent") with session["history_lock"]: if not session.get("running"): - # The turn ended between prompt.submit's first busy check and this - # helper. Let the caller retry and claim the now-idle session. - return None - with session["history_lock"]: - if not session.get("running"): + # Turn ended since prompt.submit's busy check; caller retries on the idle session. return None image_paths = list(session.get("attached_images", [])) if image_paths: - # Claim at submission time. A later paste must not be consumed by - # this prompt after the active turn finally yields. + # Claim now so a later paste isn't consumed by this prompt when the turn yields. session["attached_images"] = [] text_only = not image_paths and _is_text_only_busy_payload(text) plain_text = _coerce_message_text(text).strip() if text_only else "" if mode == "steer" and text_only and plain_text and agent is not None and hasattr(agent, "steer"): try: if agent.steer(plain_text): - with session["history_lock"]: - _record_inflight_correction(session, plain_text) - _drop_queued_duplicates_of_inflight_user(session) - session["last_active"] = time.time() + _ac_record_inflight_correction(session, plain_text) return _ok(rid, {"status": "steered"}) except Exception: pass # fall through to queue - # Text-only corrections redirect the live turn in place when the runtime - # supports it; media/attachment payloads and older agents fall through to - # the proven interrupt + queue path below. + # Text-only corrections redirect in place when supported; media payloads and + # older agents fall through to the proven interrupt + queue path. if ( mode == "interrupt" and text_only @@ -413,18 +344,12 @@ def _handle_busy_submit( ): try: if agent.redirect(plain_text): - with session["history_lock"]: - _record_inflight_correction(session, plain_text) - # #84417: do not re-fire the live turn's original user text - # from a stale server-queue self-duplicate after settle. - _drop_queued_duplicates_of_inflight_user(session) - session["last_active"] = time.time() + _ac_record_inflight_correction(session, plain_text) return _ok(rid, {"status": "redirected"}) except Exception: pass # preserve the proven interrupt + queue fallback below - # Queue before asking the live turn to stop. In particular, never call a - # provider or compute-host method while holding history_lock: an interrupt - # can wait behind the very operation it is trying to cancel. + # Queue before asking the live turn to stop. Never call a provider/compute-host + # method under history_lock: an interrupt can wait behind the op it cancels. with session["history_lock"]: if not session.get("running"): if image_paths: @@ -433,17 +358,10 @@ def _handle_busy_submit( _enqueue_prompt(session, text, transport, image_paths=image_paths) session["last_active"] = time.time() - # Attachments need a separate model invocation. Queue them without - # cancelling the active turn so the user gets both results in order. - # - # #86134: ``steer`` mode must NEVER escalate to a hard interrupt. A burst - # of user messages while the agent is busy can land as a mix of accepted - # steers (stashed in ``AIAgent._pending_steer``) and fall-through queue - # envelopes (payload not steerable, ``steer()`` rejected/raised). A hard - # interrupt here kills the live turn AND ``AIAgent.interrupt()`` drops - # the pending steer buffer — silently destroying the earlier messages of - # the burst. Steer-mode fall-throughs keep queue semantics: preserved - # FIFO in ``queued_prompt``/``queued_prompts`` and drained on turn end. + # Attachments need their own model invocation: queue without cancelling so the + # user gets both results in order. ``steer`` must NEVER escalate to a hard + # interrupt: it would kill the live turn AND drop ``AIAgent._pending_steer``, + # destroying earlier accepted steers; steer fall-throughs stay FIFO-queued. if mode == "interrupt" and not image_paths: _interrupt_busy_session(sid, session, agent) return _ok(rid, {"status": "queued"}) @@ -452,9 +370,8 @@ def _handle_busy_submit( def _drain_queued_prompt(rid, sid: str, session: dict) -> bool: """Fire a queued next-turn prompt if one is waiting and the session is idle. - Returns True if a queued prompt was dispatched (the caller should then skip - lower-priority follow-ups this cycle — the user's message wins). Mirrors the - claim-under-lock pattern used by the goal-continuation re-fire. + True when dispatched: the caller skips lower-priority follow-ups this cycle + (the user's message wins). Claim-under-lock like the goal-continuation re-fire. """ with session["history_lock"]: if session.get("_closing"): @@ -473,39 +390,24 @@ def _drain_queued_prompt(rid, sid: str, session: dict) -> bool: use_compute_host = _session_uses_compute_host(session) with session["history_lock"]: if int(session.get("_queued_prompt_generation", 0)) != queue_generation: - # Generation cancelled the claim (Stop, compress re-anchor, …). - # Do not dispatch — but put the claimed envelope back so a - # legitimate follow-up is not silently dropped. Order: claimed - # head first, then whatever advanced into the slot while we held - # the claim (#84417 belt accuracy). + # Generation bump cancelled the claim (Stop, compress re-anchor, …): don't + # dispatch, but restore the envelope (claimed head first, then whatever + # advanced into the slot) so a legitimate follow-up isn't dropped. rest: list = [] advanced = session.get("queued_prompt") if advanced: rest.append(advanced) rest.extend(session.get("queued_prompts") or []) - session["queued_prompt"] = queued - if rest: - session["queued_prompts"] = rest - else: - session.pop("queued_prompts", None) + _ac_set_queue(session, [queued, *rest]) session["running"] = False return True + kwargs: dict = {"queued_prompt_generation": queue_generation} + if queued.get("image_paths"): + kwargs["image_paths"] = queued["image_paths"] dispatch_failed = False try: if use_compute_host: - if queued.get("image_paths"): - resp = _submit_prompt_to_compute_host( - rid, - sid, - session, - queued["text"], - image_paths=queued["image_paths"], - queued_prompt_generation=queue_generation, - ) - else: - resp = _submit_prompt_to_compute_host( - rid, sid, session, queued["text"], queued_prompt_generation=queue_generation - ) + resp = _submit_prompt_to_compute_host(rid, sid, session, queued["text"], **kwargs) if resp.get("error"): message = str(((resp.get("error") or {}).get("message")) or "queued prompt failed") with session["history_lock"]: @@ -514,37 +416,14 @@ def _drain_queued_prompt(rid, sid: str, session: dict) -> bool: _emit("error", sid, {"message": message}) dispatch_failed = True else: - if queued.get("image_paths"): - _run_prompt_submit( - rid, - sid, - session, - queued["text"], - image_paths=queued["image_paths"], - queued_prompt_generation=queue_generation, - ) - else: - _run_prompt_submit( - rid, - sid, - session, - queued["text"], - queued_prompt_generation=queue_generation, - ) + _run_prompt_submit(rid, sid, session, queued["text"], **kwargs) except Exception as exc: - print( - f"[tui_gateway] queued prompt dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False + print(f"[tui_gateway] queued prompt dispatch failed: {type(exc).__name__}: {exc}", file=sys.stderr) + _ac_release_turn(session) dispatch_failed = True if dispatch_failed: with session["history_lock"]: - drain_next = bool(session.get("queued_prompt")) and not session.get( - "_turn_cancel_requested" - ) + drain_next = bool(session.get("queued_prompt")) and not session.get("_turn_cancel_requested") if drain_next: _drain_queued_prompt(rid, sid, session) return True @@ -560,11 +439,7 @@ def _inflight_snapshot(session: dict) -> dict | None: error = str(turn.get("error") or "").strip() if not user and not assistant and not streaming and not error: return None - snapshot = { - "assistant": assistant, - "streaming": streaming, - "user": user, - } + snapshot = {"assistant": assistant, "streaming": streaming, "user": user} raw_corrections = turn.get("corrections") or [] raw_offsets = turn.get("correction_offsets") or [] correction_pairs = [ @@ -573,19 +448,15 @@ def _inflight_snapshot(session: dict) -> dict | None: if str(c).strip() ] if correction_pairs: - # Mid-turn redirects. Carried alongside the original prompt (not over - # it) so resume can rebuild every user bubble the turn produced. + # Mid-turn redirects alongside (not over) the original prompt so resume can + # rebuild every user bubble; offsets only when every correction has one so + # clients can trust the pairing. snapshot["corrections"] = [c for c, _ in correction_pairs] - # Assistant-text lengths at each correction boundary (parallel list). - # Only sent when every correction has one, so clients can trust the - # pairing; older in-memory turns without offsets omit the field and - # clients fall back to placing corrections after the assistant dump. if all(isinstance(offset, int) and offset >= 0 for _, offset in correction_pairs): snapshot["correction_offsets"] = [int(offset) for _, offset in correction_pairs] # type: ignore[arg-type] if error: - # Retained failed turn (see _fail_inflight_turn): carry the error - # semantics so a resuming client can rebuild the failed-turn bubble - # instead of rendering the partial text as a healthy reply. + # Retained failed turn (_fail_inflight_turn): a resuming client must rebuild + # the failed bubble, not render the partial text as a healthy reply. snapshot["error"] = error snapshot["status"] = str(turn.get("status") or "error") snapshot["recoverable"] = bool(turn.get("recoverable")) @@ -596,29 +467,17 @@ def _inflight_snapshot(session: dict) -> dict | None: def _emit_terminal_turn_error( - sid: str, - session: dict, - error: Any, - error_surface: Optional[dict] = None, - *, - retire_marker: bool = True, + sid: str, session: dict, error: Any, error_surface: Optional[dict] = None, *, retire_marker: bool = True ) -> None: - """Close a failed turn with a terminal ``message.complete`` frame. - - Emits the same ``status: "error"`` frame shape the returned-error path in - ``_run_prompt_submit`` already produces (so TUI/desktop handling is - uniform), and retains the failed turn via ``_fail_inflight_turn`` so a - client that missed this frame (disconnect window) can recover it from - ``session.resume``'s ``inflight`` payload. - - ``error_surface`` lets callers that already know the failing layer (e.g. - agent-init failures = local runtime) pass it explicitly; exception - callers leave it None and the classifier derives it here. + """Close a failed turn with the same ``status: "error"`` ``message.complete`` + frame as ``_run_prompt_submit``'s returned-error path, retaining the turn via + ``_fail_inflight_turn`` so a client that missed the frame recovers it from + ``session.resume``'s ``inflight``. Callers that know the failing layer pass + ``error_surface``; exception callers leave it None and it is classified here. """ agent = session.get("agent") - # Classify the failure into a {layer, code, retryable} descriptor so the - # desktop can say "Provider error" / "Gateway error" with matching - # recovery actions instead of a generic toast. Never raises (advisory). + # {layer, code, retryable} descriptor so the desktop can say "Provider error" / + # "Gateway error" with matching recovery actions. Advisory: never raises. if error_surface is None and isinstance(error, BaseException): try: from agent.error_surface import build_error_surface_from_exception @@ -660,12 +519,9 @@ def _emit_terminal_turn_error( def _restore_agent_history_after_turn_error(session: dict, agent) -> bool: - """Keep a failed turn's working transcript in the gateway session. - - ``AIAgent`` persists its working messages independently of the gateway's - history snapshot. If the turn raises after that persistence, the next - prompt must see the working transcript instead of the pre-turn snapshot. - """ + """Keep a failed turn's working transcript: ``AIAgent`` persists its messages + independently, so after a raise the next prompt must see them, not the + pre-turn snapshot.""" agent_messages = getattr(agent, "_session_messages", None) if not isinstance(agent_messages, list): return False @@ -676,13 +532,8 @@ def _restore_agent_history_after_turn_error(session: dict, agent) -> bool: def _queued_prompt_snapshot(session: dict) -> dict | None: - """Return the accepted next-turn prompt without its transport handle. - - A busy ``prompt.submit`` lives only in ``session["queued_prompt"]`` until - the current turn winds down. Desktop may reconnect or restart during that - window, so the live-session projection must carry the user-visible text; - otherwise the accepted prompt disappears until it finally drains. - """ + """The accepted next-turn prompt without its transport handle, for the + live-session projection (Desktop may reconnect while it is still queued).""" queued = session.get("queued_prompt") if not isinstance(queued, dict): return None diff --git a/tui_gateway/session_compression.py b/tui_gateway/session_compression.py index 76f72b940c..61b933f64c 100644 --- a/tui_gateway/session_compression.py +++ b/tui_gateway/session_compression.py @@ -1,33 +1,27 @@ -"""Live compression: config hot-reload onto a running agent, pending model switch apply, /compress (CompressionLockHeld when a turn holds the lock), session-key sync after compress. +"""Live compression: config hot-reload onto a running agent, pending model switch apply, /compress +(CompressionLockHeld when a turn holds the lock), session-key sync after compress. -Bodies are rebound onto server.py's globals at install time (see -method_ctx.bind_module), so they reference server.py globals bare. +Bodies are rebound onto server.py's globals (method_ctx.bind_module) and reference them bare. """ from __future__ import annotations +import contextlib + from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() def _tui_compression_config_signature(cfg: dict | None) -> tuple: - """Stable snapshot of compression/context keys that must apply next turn. - - Reuses the messaging-gateway cache-busting extract so Desktop/TUI and - messaging stay on the same key set. Adds ``idle_compact_after_seconds`` - and ``tail_mode``, which affect live TUI sessions but are not in the - gateway tuple today. - """ + """Stable snapshot of compression/context keys that must apply next turn: the messaging-gateway + cache-busting extract (same key set as messaging) plus ``idle_compact_after_seconds`` and + ``tail_mode`` (affect live TUI sessions, not in the gateway tuple).""" from gateway.run import GatewayRunner keys = GatewayRunner._extract_cache_busting_config(cfg) - picked = { - key: value - for key, value in keys.items() - if key.startswith("compression.") or key == "model.context_length" - } + picked = {k: v for k, v in keys.items() if k.startswith("compression.") or k == "model.context_length"} compression = cfg.get("compression") if isinstance(cfg, dict) and isinstance(cfg.get("compression"), dict) else {} for extra in ("idle_compact_after_seconds", "tail_mode"): picked[f"compression.{extra}"] = compression.get(extra) @@ -35,88 +29,65 @@ def _tui_compression_config_signature(cfg: dict | None) -> tuple: def _compressor_ctor_default(name: str, fallback: Any) -> Any: - """Read a normalized default from ContextCompressor's REAL signature. - - Unset restoration must go through the same derivation the construction - path uses (#94724 review finding on #95980) — pulling the default off - ``ContextCompressor.__init__`` itself instead of hardcoding copies keeps - the two from drifting. - """ + """Default read off ContextCompressor.__init__'s REAL signature, so unset-key restoration uses the + construction path's derivation instead of a hardcoded copy that could drift.""" try: import inspect from agent.context_compressor import ContextCompressor - default = inspect.signature(ContextCompressor.__init__).parameters[ - name - ].default - if default is inspect.Parameter.empty: - return fallback - return default + default = inspect.signature(ContextCompressor.__init__).parameters[name].default + return fallback if default is inspect.Parameter.empty else default except Exception: return fallback def _derived_default_threshold_percent(agent: Any, compression: dict) -> float: - """Default compaction threshold when ``compression.threshold`` is unset. - - Mirrors agent_init exactly: the ctor's global default, then the per-model - resolution (Codex gpt-5.4/5.5 + spark autoraise, Arcee Trinity, etc.) - via the SAME ``_resolve_compression_threshold`` helper — so removing the - key restores the model-derived value, not a bare constant. - """ + """Default compaction threshold when ``compression.threshold`` is unset. Mirrors agent_init: ctor + global default, then per-model resolution (Codex autoraise etc.) via the SAME + ``_resolve_compression_threshold`` — removing the key restores the model-derived value.""" try: pct = float(_compressor_ctor_default("threshold_percent", 0.50)) except (TypeError, ValueError): pct = 0.50 try: from agent.agent_init import _resolve_compression_threshold - from agent.auxiliary_client import ( - _compression_threshold_for_model, - _is_codex_gpt54_or_gpt55, - _is_codex_spark, - ) + from agent.auxiliary_client import _compression_threshold_for_model, _is_codex_gpt54_or_gpt55, _is_codex_spark model = getattr(agent, "model", "") or "" provider = getattr(agent, "provider", "") or "" - autoraise_enabled = str( - compression.get("codex_gpt55_autoraise", True) - ).lower() in {"true", "1", "yes"} + autoraise_enabled = str(compression.get("codex_gpt55_autoraise", True)).lower() in {"true", "1", "yes"} model_cthresh = _compression_threshold_for_model( - model, - provider, - allow_codex_gpt55_autoraise=autoraise_enabled, + model, provider, allow_codex_gpt55_autoraise=autoraise_enabled ) pct, _notice = _resolve_compression_threshold( - pct, - model_cthresh, - model=model, - is_codex_autoraise=( - _is_codex_gpt54_or_gpt55(model, provider) - or _is_codex_spark(model, provider) - ), + pct, model_cthresh, model=model, + is_codex_autoraise=_is_codex_gpt54_or_gpt55(model, provider) or _is_codex_spark(model, provider), ) except Exception: pass return pct +# (config key == compressor attr, ctor-default fallback, min_value) +_COMPRESSION_INT_KEYS = ( + ("proactive_prune_tokens", 0, 0), + ("proactive_prune_min_result_chars", 8000, 0), + ("proactive_prune_min_reclaim_tokens", 4096, 0), + ("protect_last_n", 20, 0), + ("min_tail_user_messages", 1, 1), +) + + def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: - """Update a live session's compressor from current config.yaml. + """Update a live session's compressor from current config.yaml, preserving the agent object, + session identity, history and callbacks. Recomputes the trigger from the ratio threshold, then + applies ``compression.threshold_tokens`` so raising/lowering/clearing the cap lands next preflight. - Preserves the agent object, session identity, history, and callbacks. - Recomputes the trigger from the ratio-based threshold and then applies - ``compression.threshold_tokens`` so raising, lowering, or clearing the - cap all take effect on the next preflight. - - Every adopted key has UNSET semantics (#94724 review finding on the - merged #95980): removing a key from config.yaml restores the normalized - default — or the model-derived value — on the next turn, through the - same derivation the construction path uses (ContextCompressor ctor - defaults read off its real signature, the Codex threshold autoraise via - ``_resolve_compression_threshold``, context-length re-inference via the - deferred ``get_model_context_length`` resolution). The old behavior - acted only on PRESENT keys, leaving stale values active forever. + Every adopted key has UNSET semantics: a removed key restores the normalized default (or the + model-derived value) through the construction path's own derivation (ctor signature defaults, + Codex autoraise, deferred context-length re-inference). Acting only on PRESENT keys would leave + stale values active forever. """ cfg = cfg if isinstance(cfg, dict) else {} compression_raw = cfg.get("compression") @@ -129,12 +100,8 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: else: agent.compression_enabled = str(enabled_raw).lower() in {"true", "1", "yes"} - agent.codex_responses_native_compaction = is_truthy_value( - compression.get("codex_responses_native", False) - ) - native_threshold_raw = compression.get( - "codex_responses_compact_threshold", 200_000 - ) + agent.codex_responses_native_compaction = is_truthy_value(compression.get("codex_responses_native", False)) + native_threshold_raw = compression.get("codex_responses_compact_threshold", 200_000) try: if isinstance(native_threshold_raw, bool): raise ValueError @@ -142,11 +109,7 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: if native_threshold <= 0: raise ValueError except (TypeError, ValueError): - logger.warning( - "Invalid compression.codex_responses_compact_threshold=%r; " - "using 200000.", - native_threshold_raw, - ) + logger.warning("Invalid compression.codex_responses_compact_threshold=%r; using 200000.", native_threshold_raw) native_threshold = 200_000 agent.codex_responses_compact_threshold = native_threshold @@ -161,52 +124,22 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: if cc is None: return - # tail_mode: ctor normalization — unknown/absent values land on "lean", - # matching agent_init's default and the compressor's own fallback. + # tail_mode: unknown/absent values land on the ctor default ("lean"), matching agent_init. default_tail = str(_compressor_ctor_default("tail_mode", "lean")) - mode = str(compression.get("tail_mode", default_tail) or default_tail) - mode = mode.strip().lower() + mode = str(compression.get("tail_mode", default_tail) or default_tail).strip().lower() cc.tail_mode = mode if mode in ("legacy", "lean") else default_tail - def _assign_int(key: str, attr: str, default: int, min_value: int = 0) -> None: + for key, fallback, min_value in _COMPRESSION_INT_KEYS: + default = int(_compressor_ctor_default(key, fallback)) raw = compression.get(key, default) try: value = default if raw is None else int(raw) except (TypeError, ValueError): - return - setattr(cc, attr, max(min_value, value)) - - _assign_int( - "proactive_prune_tokens", - "proactive_prune_tokens", - int(_compressor_ctor_default("proactive_prune_tokens", 0)), - ) - _assign_int( - "proactive_prune_min_result_chars", - "proactive_prune_min_result_chars", - int(_compressor_ctor_default("proactive_prune_min_result_chars", 8000)), - ) - _assign_int( - "proactive_prune_min_reclaim_tokens", - "proactive_prune_min_reclaim_tokens", - int(_compressor_ctor_default("proactive_prune_min_reclaim_tokens", 4096)), - ) - _assign_int( - "protect_last_n", - "protect_last_n", - int(_compressor_ctor_default("protect_last_n", 20)), - ) - _assign_int( - "min_tail_user_messages", - "min_tail_user_messages", - int(_compressor_ctor_default("min_tail_user_messages", 1)), - min_value=1, - ) + continue + setattr(cc, key, max(min_value, value)) try: - ratio_raw = compression.get( - "target_ratio", _compressor_ctor_default("summary_target_ratio", 0.20) - ) + ratio_raw = compression.get("target_ratio", _compressor_ctor_default("summary_target_ratio", 0.20)) cc.summary_target_ratio = max(0.10, min(float(ratio_raw), 0.80)) except (TypeError, ValueError): pass @@ -215,16 +148,13 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: if isinstance(raw_thresholds, dict): cc.model_thresholds = { str(k): float(v) - for k, v in raw_thresholds.items() - if isinstance(v, (int, float)) and not isinstance(v, bool) + for k, v in raw_thresholds.items() if isinstance(v, (int, float)) and not isinstance(v, bool) } else: - # Absent (or invalid shape — agent_init treats both as empty): - # stale per-model overrides must stop steering the live threshold. + # Absent or invalid shape (agent_init treats both as empty): stale overrides must stop steering. cc.model_thresholds = {} - # threshold: present value wins; absence derives the default through - # the same agent_init resolution (global default + per-model autoraise). + # threshold: present value wins; absence derives via the agent_init resolution (default + autoraise). pct: float | None = None if "threshold" in compression: try: @@ -241,17 +171,11 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: if model_thresholds: from agent.context_compressor import resolve_model_threshold - base = resolve_model_threshold( - getattr(agent, "model", "") or "", - model_thresholds, - pct, - ) + base = resolve_model_threshold(getattr(agent, "model", "") or "", model_thresholds, pct) cc._base_threshold_percent = base if hasattr(cc, "_effective_threshold_percent"): try: - cc.threshold_percent = cc._effective_threshold_percent( - cc.context_length, base - ) + cc.threshold_percent = cc._effective_threshold_percent(cc.context_length, base) except Exception: cc.threshold_percent = pct else: @@ -272,11 +196,8 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: except Exception: pass elif getattr(cc, "_config_context_length", None) is not None: - # model.context_length removed: drop the config override and force - # re-inference from model metadata on next access — the same - # deferred get_model_context_length resolution agent construction - # uses (#32221). The re-resolve also re-applies the small-context - # threshold floor for the genuinely re-inferred window. + # model.context_length removed: drop the override and force re-inference from model metadata + # on next access (construction's deferred resolution); re-applies the small-context floor too. cc._config_context_length = None cc._resolved_context_length = None @@ -292,8 +213,7 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: else: cc.threshold_tokens_cap = None - # Invalidate cached trigger so the next preflight re-derives from the - # current percent/window and then applies the (possibly new) cap. + # Invalidate the cached trigger so the next preflight re-derives from percent/window, then the cap. if hasattr(cc, "_threshold_tokens"): cc._threshold_tokens = None if hasattr(cc, "_tail_token_budget"): @@ -315,12 +235,8 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: def _sync_agent_compression_with_config(sid: str, session: dict) -> None: - """Adopt compression.* / model.context_length edits at turn start. - - Messaging gateways already rebuild a cached agent when these keys change. - Desktop/TUI only synced the model; the live compressor kept the threshold - captured at agent creation (#95151). - """ + """Adopt compression.* / model.context_length edits at turn start (messaging gateways rebuild the + agent on these keys; Desktop/TUI keeps the live compressor, so it must be updated in place).""" agent = session.get("agent") if agent is None: return @@ -333,20 +249,14 @@ def _sync_agent_compression_with_config(sid: str, session: dict) -> None: try: _apply_live_compression_config(agent, cfg) except Exception as e: - logger.warning( - "Could not apply live compression config for %s: %s", sid, e - ) + logger.warning("Could not apply live compression config for %s: %s", sid, e) def _apply_pending_model_switch(sid: str, session: dict) -> None: - """Apply a model switch queued while a turn was running. + """Apply a model switch queued (``session["pending_model_switch"]``) while a turn was running. - ``config.set model`` on a busy session doesn't mutate the live agent (the - worker thread is reading model/client mid-request); it stashes the pick in - ``session["pending_model_switch"]``. This runs on the TURN thread at turn - start — before the first model call, nothing in flight — so the in-place - swap (client rebuild, the slow part) is safe here. A failed switch keeps - the current model and never blocks the turn, matching + Runs on the TURN thread at turn start — nothing in flight — so the in-place swap (client rebuild) + is safe. A failed switch keeps the current model and never blocks the turn, matching ``_sync_agent_model_with_config``. """ pending = session.pop("pending_model_switch", None) @@ -354,85 +264,55 @@ def _apply_pending_model_switch(sid: str, session: dict) -> None: return try: result = _apply_model_switch( - sid, - session, - pending["raw"], - confirm_expensive_model=bool(pending.get("confirm_expensive_model")), + sid, session, pending["raw"], confirm_expensive_model=bool(pending.get("confirm_expensive_model")) ) - # A queued pick is a deliberate user action; honour the expensive-model - # confirm by NOT applying it silently — surface the warning and drop the - # switch rather than spend on a pricey model the user never confirmed. + # Honour the expensive-model confirm: surface the warning and drop the switch rather than + # spend on a model the user never confirmed. if result.get("confirm_required"): - _emit( - "error", - sid, - {"message": result.get("confirm_message") or result.get("warning") or ""}, - ) + _emit("error", sid, {"message": result.get("confirm_message") or result.get("warning") or ""}) except Exception as e: - _emit( - "error", - sid, - {"message": f"Could not switch model: {e}"}, - ) + _emit("error", sid, {"message": f"Could not switch model: {e}"}) class CompressionLockHeld(Exception): - """Raised by _compress_session_history when compression skipped due - to a concurrent lock on the session's compression_locks row.""" + """Raised by _compress_session_history when a concurrent compression_locks row skipped compression.""" + def __init__(self, holder: str | None = None): self.holder = holder super().__init__(f"Compression lock held: {holder or 'unknown'}") def _compress_session_history( - session: dict, - focus_topic: str | None = None, - approx_tokens: int | None = None, - before_messages: list | None = None, - history_version: int | None = None, + session: dict, focus_topic: str | None = None, approx_tokens: int | None = None, + before_messages: list | None = None, history_version: int | None = None, ) -> tuple[int, dict]: - """Compress a session's history — the single choke point shared by all - three manual-compress routes (session.compress RPC, command.dispatch - /compress|/compact, and the slash-exec mirror). + """Single choke point for all manual-compress routes (session.compress RPC, command.dispatch + /compress|/compact, slash-exec mirror). - ``focus_topic`` is the RAW argument string after ``/compress``. It is - parsed here with :func:`parse_partial_compress_args` so boundary-aware - forms (``here [N]``, ``up to here``, ``--keep N``) trigger a partial - compress — head summarized, most recent ``keep_last`` exchanges kept - verbatim — on EVERY route, mirroring cli.py's ``_manual_compress`` and - gateway/slash_commands.py (PR #35252). Parsing at the choke point (not - per-route) is what fixes #35533: previously "/compress here 3" reached - this helper unparsed and ran a FULL compress focused on the literal - text "here 3". + ``focus_topic`` is the RAW argument string after ``/compress``, parsed HERE (not per-route) with + :func:`parse_partial_compress_args` so boundary forms (``here [N]``, ``up to here``, ``--keep N``) + trigger a partial compress on EVERY route — otherwise "/compress here 3" would run a FULL compress + focused on the literal text "here 3". Mirrors cli.py ``_manual_compress`` / gateway slash_commands. """ - from agent.conversation_compression import ( - finalize_context_engine_compression_notification, - ) + from agent.conversation_compression import finalize_context_engine_compression_notification from agent.model_metadata import estimate_request_tokens_rough from hermes_cli.partial_compress import ( - parse_partial_compress_args, - rejoin_compressed_head_and_tail, - split_history_for_partial_compress, + parse_partial_compress_args, rejoin_compressed_head_and_tail, split_history_for_partial_compress, ) agent = session["agent"] - # Snapshot history under the lock so the LLM-bound compression call - # below does NOT hold history_lock for the duration of the request — - # otherwise other handlers acquiring the lock (prompt.submit etc.) - # block on the dispatcher loop while compaction runs. + # Snapshot under the lock so the LLM-bound compression call does NOT hold history_lock for the + # request — otherwise prompt.submit etc. block on the dispatcher loop while compaction runs. if before_messages is None or history_version is None: with session["history_lock"]: before_messages = list(session.get("history", [])) history_version = int(session.get("history_version", 0)) history = before_messages if len(history) < 4: - usage = _get_usage(agent) - return 0, usage + return 0, _get_usage(agent) partial, keep_last, focus_topic = parse_partial_compress_args(focus_topic or "") - # Boundary-aware split: only the head is summarized; the most recent - # `keep_last` exchanges ride along verbatim. A degenerate split (empty - # tail — everything would be kept, or no head left to compress) falls - # back to full compression so the user still gets an action. + # Only the head is summarized; the last `keep_last` exchanges ride along verbatim. A degenerate + # split (empty tail) falls back to full compression so the user still gets an action. tail: list = [] head = history if partial: @@ -441,100 +321,53 @@ def _compress_session_history( partial = False head = history if approx_tokens is None: - # Include system prompt + tool schemas so the figure reflects real - # request pressure, not a transcript-only underestimate (#6217). + # Include system prompt + tool schemas so the figure reflects real request pressure. _sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" _tools = getattr(agent, "tools", None) or None - approx_tokens = estimate_request_tokens_rough( - history, system_prompt=_sys_prompt, tools=_tools - ) - # Pass system_message=None so AIAgent._compress_context rebuilds the - # system prompt cleanly via _build_system_prompt(None). Passing the - # cached prompt (which already contains the agent identity block) - # makes the rebuild append the identity a second time. Mirrors the - # CLI's _manual_compress fix for issue #15281. - # force=True: every caller of this helper is a manual /compress path - # (session.compress RPC, slash compress/compact, slash-worker mirror) — - # auto-compaction runs inside the agent loop, not here. Manual - # compaction bypasses the summary-failure cooldown, matching the CLI - # and gateway handlers. + approx_tokens = estimate_request_tokens_rough(history, system_prompt=_sys_prompt, tools=_tools) + # system_message=None: _compress_context rebuilds the system prompt via _build_system_prompt(None); + # passing the cached prompt (already holding the identity block) appends the identity twice. + # force=True: every caller is a manual /compress path, which bypasses the summary-failure + # cooldown like the CLI and gateway handlers. Partial compress has no focus topic (exclusive modes). try: compressed, _ = agent._compress_context( - head, - None, - approx_tokens=approx_tokens, - # Partial compress has no focus topic (the modes are exclusive; - # parse_partial_compress_args returns focus_topic=None for the - # boundary-aware forms). - focus_topic=focus_topic or None, - force=True, + head, None, approx_tokens=approx_tokens, focus_topic=focus_topic or None, force=True, defer_context_engine_notification=True, ) except Exception: - finalize_context_engine_compression_notification( - agent, - committed=False, - ) + finalize_context_engine_compression_notification(agent, committed=False) raise - # If _compress_context returned unchanged because a concurrent - # compression lock is held, raise so callers can surface a clear - # message instead of the misleading "No changes from compression" text. - # Type-pinned (is True / str): real values are None/True/holder-string; - # bare truthiness is fooled by MagicMock auto-attrs on test doubles. + # Lock-skipped: raise so callers surface a clear message instead of "No changes from compression". + # Type-pinned (is True / str) because bare truthiness is fooled by MagicMock auto-attrs. _lock_skipped = getattr(agent, "_compression_skipped_due_to_lock", None) if _lock_skipped is True or isinstance(_lock_skipped, str): agent._compression_skipped_due_to_lock = None - # No boundary was committed on a lock-skip; discard any pending - # deferred context-engine notification (exactly-once, no-op safe). - finalize_context_engine_compression_notification( - agent, - committed=False, - ) - raise CompressionLockHeld( - _lock_skipped if isinstance(_lock_skipped, str) else None - ) + # No boundary committed; discard the pending deferred notification (exactly-once, no-op safe). + finalize_context_engine_compression_notification(agent, committed=False) + raise CompressionLockHeld(_lock_skipped if isinstance(_lock_skipped, str) else None) if partial and tail: compressed = rejoin_compressed_head_and_tail(compressed, tail) with session["history_lock"]: if int(session.get("history_version", 0)) != history_version: - # External mutation during compaction — drop the compressed - # result so we don't clobber concurrent edits. - finalize_context_engine_compression_notification( - agent, - committed=False, - ) - usage = _get_usage(agent) - return 0, usage + # External mutation during compaction — drop the result so we don't clobber concurrent edits. + finalize_context_engine_compression_notification(agent, committed=False) + return 0, _get_usage(agent) session["history"] = compressed session["history_version"] = history_version + 1 - usage = _get_usage(agent) - return len(history) - len(compressed), usage + return len(history) - len(compressed), _get_usage(agent) def _sync_session_key_after_compress( - sid: str, - session: dict, - *, - clear_pending_title: bool = True, - restart_slash_worker: bool = True, + sid: str, session: dict, *, clear_pending_title: bool = True, restart_slash_worker: bool = True ) -> None: - """Re-anchor session_key when AIAgent._compress_context rotates session_id. + """Re-anchor the gateway-side ``session_key`` when _compress_context rotates ``agent.session_id`` + to a SessionDB continuation; otherwise approval routing, slash worker init, DB title/history + lookups and yolo state keep targeting the ended parent. - AIAgent._compress_context ends the current SessionDB session and creates - a new continuation session, rotating ``agent.session_id``. The TUI - gateway keeps the gateway-side ``session_key`` separate (used for - approval routing, slash worker init, DB title/history lookups, yolo - state). Without this sync, those operations would target the ended - parent session while the agent writes to the new continuation session. - - Policy flags: - clear_pending_title: True for manual /compress (title belongs to old - session). False for post-turn auto-compression (preserve user - intent so pending_title can be applied to the continuation). - restart_slash_worker: True for manual /compress and post-turn - auto-compression (worker holds stale session key). False only - if the caller manages the worker lifecycle separately. + clear_pending_title: True for manual /compress (title belongs to the old session); False for + post-turn auto-compression so pending_title applies to the continuation. + restart_slash_worker: True unless the caller manages the worker (it holds the stale key). """ agent = session.get("agent") new_session_id = getattr(agent, "session_id", None) or "" @@ -542,73 +375,45 @@ def _sync_session_key_after_compress( if not new_session_id or new_session_id == old_key: return - lease_reanchored = _transfer_active_session_slot( - sid, - session, - new_session_id=new_session_id, - ) + lease_reanchored = _transfer_active_session_slot(sid, session, new_session_id=new_session_id) if not lease_reanchored: logger.warning( "Compression session lease did not re-anchor: sid=%s old_session_id=%s new_session_id=%s", - sid, - old_key, - new_session_id, + sid, old_key, new_session_id, ) try: from tools.approval import ( - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, - register_gateway_notify, + disable_session_yolo, enable_session_yolo, is_session_yolo_enabled, register_gateway_notify, unregister_gateway_notify, ) - try: + with contextlib.suppress(Exception): unregister_gateway_notify(old_key) - except Exception: - pass session["session_key"] = new_session_id try: yolo_was_on = is_session_yolo_enabled(old_key) except Exception: yolo_was_on = False if yolo_was_on: - try: + with contextlib.suppress(Exception): enable_session_yolo(new_session_id) disable_session_yolo(old_key) - except Exception: - pass - try: - register_gateway_notify( - new_session_id, - lambda data: _emit_approval_request(sid, data), - ) - except Exception: - pass + with contextlib.suppress(Exception): + register_gateway_notify(new_session_id, lambda data: _emit_approval_request(sid, data)) except Exception: - # Even if the approval module fails to import, still anchor the - # session_key on the new continuation id so downstream lookups - # don't keep targeting the ended row. + # Even if the approval module fails to import, anchor session_key on the continuation id. session["session_key"] = new_session_id - # #84417 (belt): invalidate any in-flight ``_drain_queued_prompt`` claim - # that captured generation under the pre-rotation session_key. A raced - # drain must not dispatch on the continuation with a stale claim; the - # claimed envelope is restored to the queue (see ``_drain_queued_prompt``) - # so legitimate follow-ups still survive. Complements self-duplicate - # scrubbing on redirect. - session["_queued_prompt_generation"] = int( - session.get("_queued_prompt_generation", 0) - ) + 1 + # Invalidate any in-flight ``_drain_queued_prompt`` claim taken under the pre-rotation key: a raced + # drain must not dispatch on the continuation (its envelope is restored to the queue). + session["_queued_prompt_generation"] = int(session.get("_queued_prompt_generation", 0)) + 1 if clear_pending_title: session["pending_title"] = None if restart_slash_worker: - try: + with contextlib.suppress(Exception): _restart_slash_worker(sid, session) - except Exception: - pass def register(server) -> None: diff --git a/tui_gateway/session_history.py b/tui_gateway/session_history.py index 5dbbd35c10..1a786cf190 100644 --- a/tui_gateway/session_history.py +++ b/tui_gateway/session_history.py @@ -1,7 +1,7 @@ -"""Session history/message shaping: image-ref messages, content coercion, history->wire messages, in-flight turn tracking and turn-failure detail. +"""Session history/message shaping: image-ref messages, content coercion, history->wire messages, +in-flight turn tracking and turn-failure detail. -Bodies are rebound onto server.py's globals at install time (see -method_ctx.bind_module), so they reference server.py globals bare. +Bodies are rebound onto server.py's globals (method_ctx.bind_module) and reference them bare. """ from __future__ import annotations @@ -16,40 +16,17 @@ def _active_image_routing_identity(agent: Any) -> tuple[str, str]: """Return the live provider/model, falling back before agent startup.""" from agent.auxiliary_client import _read_main_model, _read_main_provider - return ( - getattr(agent, "provider", "") or _read_main_provider(), - getattr(agent, "model", "") or _read_main_model(), - ) + return (getattr(agent, "provider", "") or _read_main_provider(), getattr(agent, "model", "") or _read_main_model()) def _build_image_ref_message(user_text: str, image_paths: list[str]) -> str: - """Reference attached images by path so the agent analyzes them in-loop. - - This used to pre-analyze every image with the auxiliary vision model - *before* the turn was dispatched (``_enrich_with_attached_images``): - serial blocking calls on the submit path — 60-90s per large photo — - with failures silently swallowed and an interrupt during the window - killing the turn with zero API calls (#83291). It also prepended the - vision description to the first user message, poisoning session - auto-titles (#82339). The CLI never gates turn dispatch on vision - like this, which is why the same message was seconds there and - minutes on desktop. - - Now the turn starts immediately. The agent examines each image itself - with ``vision_analyze`` — its own retries, visible tool progress — - exactly how the ``@folder:`` reference path already behaves, which - responds in seconds for the same images. - """ - parts: list[str] = [] - for path in image_paths: - p = Path(path) - if not p.exists(): - continue - parts.append( - f"[The user attached an image: {p.name}]\n" - f"[Examine it with the vision_analyze tool using image_url: {p}]" - ) - + """Reference attached images by path so the agent analyzes them in-loop with ``vision_analyze``. + Pre-analyzing with the auxiliary vision model blocked submit 60-90s per photo and poisoned + auto-titles with the description.""" + parts = [ + f"[The user attached an image: {p.name}]\n[Examine it with the vision_analyze tool using image_url: {p}]" + for p in map(Path, image_paths) if p.exists() + ] text = user_text or "" prefix = "\n\n".join(parts) if prefix: @@ -58,21 +35,10 @@ def _build_image_ref_message(user_text: str, image_paths: list[str]) -> str: def _build_persist_message_with_image_refs(user_text: str, image_paths: list[str]) -> str: - """Build the clean, UI-recognizable version of the user's message for - persisting to session history. Uses ``@image:`` directives — the - format the desktop client (directive-text.tsx / HERMES_DIRECTIVE_RE) - actually parses and renders as an image — unlike - ``_build_image_ref_message``, which embeds an - ``image_url:`` hint meant only for the model and must never be - persisted as-is (it silently breaks image rendering after a full - restart, and reorders image/text on live session-switch reconciliation). - - The caption leads and the directives trail: session previews are the first - 60 characters of the first user message (``list_sessions_rich``), so a - leading directive would label the session with a truncated file path in the - sidebar, switcher, and command palette. Clients lift the refs out of the - body by line, so their position does not affect how the turn renders. - """ + """Persisted form of the user's message: ``@image:`` directives (the desktop renders them + as images). ``_build_image_ref_message``'s ``image_url:`` hint is model-only, never persisted. + Caption first, directives last: session previews are the first 60 chars of the first user + message, so a leading directive would label the session with a truncated path.""" from agent.context_references import format_reference_value text = user_text or "" @@ -83,16 +49,9 @@ def _build_persist_message_with_image_refs(user_text: str, image_paths: list[str def _build_persist_user_message(user_text: str, image_paths: list[str], run_message: Any) -> Any: - """Shape the persisted user turn to match what was sent to the model. - - Native-vision turns send ``content`` as a parts list, and - ``_flush_messages_to_session_db`` deliberately ignores a plain-string - override for a list payload (a text override must not erase a turn's - image/audio summary). So mirror the shape: replace only the text part with - the ``@image:`` ref form and keep the image parts, so the model still has - the pixels for the rest of the session. Any API-only text part (the - barge-in note) is dropped along the way, which is the point of the override. - """ + """Shape the persisted user turn like the model payload: ``_flush_messages_to_session_db`` ignores + a plain-string override for a list (native-vision) payload, so swap only the text part for the + ``@image:`` form, keep image parts, and drop API-only text parts (barge-in note).""" persist_text = _build_persist_message_with_image_refs(user_text, image_paths) if not isinstance(run_message, list): return persist_text @@ -100,6 +59,36 @@ def _build_persist_user_message(user_text: str, image_paths: list[str], run_mess return [{"type": "text", "text": persist_text}, *image_parts] +_HISTORY_TEXT_KINDS = frozenset({"text", "input_text", "output_text"}) +_HISTORY_IMAGE_KINDS = frozenset({"image_url", "input_image", "image"}) +_HISTORY_AUDIO_KINDS = frozenset({"input_audio", "audio"}) + + +def _history_part_image_url(part: dict) -> str: + """The URL carried by an image part (``image_url`` dict or str), else "".""" + image_url = part.get("image_url") + if isinstance(image_url, dict): + candidate = image_url.get("url") + return candidate if isinstance(candidate, str) else "" + return image_url if isinstance(image_url, str) else "" + + +def _history_dict_text(content: dict, *, image_urls: bool) -> str: + """Placeholder/text rendering of one structured content dict.""" + kind = content.get("type") + if kind in _HISTORY_TEXT_KINDS: + return str(content.get("text") or content.get("content") or "") + if kind in _HISTORY_IMAGE_KINDS: + return (_history_part_image_url(content) if image_urls else "") or "[image]" + if kind in _HISTORY_AUDIO_KINDS: + return "[audio]" + if kind: + return f"[{kind}]" + if "text" in content: + return str(content.get("text") or "") + return "[structured content]" + + def _content_display_text(content: Any) -> str: if content is None: return "" @@ -108,47 +97,18 @@ def _content_display_text(content: Any) -> str: if isinstance(content, (int, float)): return str(content) if isinstance(content, list): - parts = [] - for part in content: - text = _content_display_text(part).strip() - if text: - parts.append(text) - return "\n".join(parts) + parts = (_content_display_text(part).strip() for part in content) + return "\n".join(text for text in parts if text) if isinstance(content, dict): - kind = content.get("type") - if kind in {"text", "input_text", "output_text"}: - return str(content.get("text") or content.get("content") or "") - if kind in {"image_url", "input_image", "image"}: - return "[image]" - if kind in {"input_audio", "audio"}: - return "[audio]" - if kind: - return f"[{kind}]" - if "text" in content: - return str(content.get("text") or "") - return "[structured content]" + return _history_dict_text(content, image_urls=False) return str(content) def _coerce_message_text(content: Any) -> str: - """Render ``message['content']`` as a plain string for transport. - - Provider-side, ``content`` may be a string (most common), a list of - multimodal parts (e.g. ``[{"type": "text", "text": "..."}, - {"type": "image_url", "image_url": {...}}]``), or a single structured - dict. Calling ``.strip()`` on a list raises ``'list' object has no - attribute 'strip'`` and breaks session resume entirely. - - Image parts (``image_url``) are preserved by appending the underlying - URL (data: or http:) into the text. The desktop renderer pulls these - back out via ``extractEmbeddedImages`` so the user sees the image - instead of the URL — and it stops the resume payload from disagreeing - with the cached message (which would otherwise cause the inline image - to flash, then disappear when the resume payload overwrites the cache). - - Other structured dict shapes (audio, unknown types) fall back to a - bracketed placeholder so resume doesn't drop the message entirely. - """ + """Render ``message['content']`` (str, parts list, or one structured dict) as a plain string. + Image parts keep their URL inline so the desktop's ``extractEmbeddedImages`` and the resume payload + agree with the cached message (else the inline image flashed, then vanished); other structured + shapes become a bracketed placeholder so resume doesn't drop the message.""" if content is None: return "" if isinstance(content, str): @@ -168,56 +128,23 @@ def _coerce_message_text(content: Any) -> str: chunks.append(text) continue kind = part.get("type") - if kind in {"text", "input_text", "output_text"}: + if kind in _HISTORY_TEXT_KINDS: t = part.get("text") or part.get("content") or "" if t: chunks.append(str(t)) - continue - if kind in {"image_url", "input_image", "image"}: - image_url = part.get("image_url") - url = "" - if isinstance(image_url, dict): - candidate = image_url.get("url") - if isinstance(candidate, str): - url = candidate - elif isinstance(image_url, str): - url = image_url - if url: - chunks.append(f"\n{url}") - else: - chunks.append("\n[image]") - continue - if kind in {"input_audio", "audio"}: + elif kind in _HISTORY_IMAGE_KINDS: + chunks.append(f"\n{_history_part_image_url(part) or '[image]'}") + elif kind in _HISTORY_AUDIO_KINDS: chunks.append("\n[audio]") - continue - if kind: + elif kind: chunks.append(f"\n[{kind}]") return "".join(chunks) if isinstance(content, dict): - kind = content.get("type") - if kind in {"text", "input_text", "output_text"}: - return str(content.get("text") or content.get("content") or "") - if kind in {"image_url", "input_image", "image"}: - image_url = content.get("image_url") - url = "" - if isinstance(image_url, dict): - candidate = image_url.get("url") - if isinstance(candidate, str): - url = candidate - elif isinstance(image_url, str): - url = image_url - return url or "[image]" - if kind in {"input_audio", "audio"}: - return "[audio]" - if kind: - return f"[{kind}]" - if "text" in content: - return str(content.get("text") or "") - return "[structured content]" + return _history_dict_text(content, image_urls=True) return str(content) -_TEXT_ONLY_BUSY_PART_KINDS = frozenset({"text", "input_text", "output_text"}) +_TEXT_ONLY_BUSY_PART_KINDS = _HISTORY_TEXT_KINDS def _is_text_only_busy_payload(content: Any) -> bool: @@ -227,109 +154,69 @@ def _is_text_only_busy_payload(content: Any) -> bool: if isinstance(content, (str, int, float)): return True if isinstance(content, list): - if not content: - return False - for part in content: - if isinstance(part, str): - continue - if not isinstance(part, dict): - return False - kind = part.get("type") - if kind in _TEXT_ONLY_BUSY_PART_KINDS: - continue - if kind is None and isinstance(part.get("text"), str): - continue - return False - return True + return bool(content) and all( + isinstance(part, str) or (isinstance(part, dict) and _history_text_only_part(part)) for part in content + ) if isinstance(content, dict): - kind = content.get("type") - if kind in _TEXT_ONLY_BUSY_PART_KINDS: - return True - return kind is None and isinstance(content.get("text"), str) + return _history_text_only_part(content) return False -def _is_display_hidden_marker(role: str | None, text: str) -> bool: - """Gateway bookkeeping notices (model-switch, personality) are persisted as - role=user ``[System: …]`` rows so strict providers accept them mid-history. - They are model-facing runtime metadata, not user turns, and must never - render as a user bubble in ANY client transcript (desktop, TUI, CLI, web). +def _history_text_only_part(part: dict) -> bool: + kind = part.get("type") + return kind in _TEXT_ONLY_BUSY_PART_KINDS or (kind is None and isinstance(part.get("text"), str)) - Filtering here — the single display projection every surface reads — hides - them everywhere while the raw marker stays in ``session["history"]`` for the - model. It also removes the stored marker from the payload the desktop - reconciles against, so it can no longer shift user-message ordinals and - duplicate the optimistic prompt (#67603).""" + +def _is_display_hidden_marker(role: str | None, text: str) -> bool: + """Gateway notices (model-switch, personality) persist as role=user ``[System: …]`` rows so strict + providers accept them mid-history; they must never render as a user bubble. Filtering in this one + projection hides them everywhere (raw marker stays in ``session["history"]``) and keeps them from + shifting the user-message ordinals the desktop reconciles against.""" return role == "user" and text.lstrip().startswith("[System:") def _skill_scaffold_projection(content_text: str) -> str: - """Return the invocation a slash-skill-expanded turn came from, else "". - - A ``/skill`` invocation expands into a model-facing message that embeds the - whole skill body. That payload belongs to the agent — every UI renders the - invocation (``/work fix the leak``) instead, so no surface can leak the - body into a chat bubble. - """ + """The invocation a slash-skill-expanded turn came from, else "" — every UI renders + ``/work fix the leak`` instead of the embedded skill body.""" return describe_skill_invocation(content_text, separator=" ") or "" def _expand_skill_invocation_for_replay(text: str, task_id: str) -> str: - """Re-expand a projected `/skill` invocation before re-running that turn. - - The inverse of :func:`_skill_scaffold_projection`. Because a skill turn is - displayed as its invocation, a rewind/regenerate hands us back - ``/work fix the leak`` rather than the body the agent originally saw — - re-running that verbatim would drop the skill. Re-expanding here keeps the - body server-side (no client ever holds it) and makes the replayed turn - identical to the original. - - Returns *text* unchanged when it isn't a resolvable skill invocation. - """ + """Inverse of :func:`_skill_scaffold_projection`: rewind/regenerate hands back the projected + invocation, and re-running it verbatim would drop the skill. Unchanged when not resolvable.""" head, _, arg = (text or "").strip().partition(" ") if not head.startswith("/"): return text - try: - from agent.skill_commands import ( - build_skill_invocation_message, - resolve_skill_command_key, - ) + from agent.skill_commands import build_skill_invocation_message, resolve_skill_command_key cmd_key = resolve_skill_command_key(head.lstrip("/")) if cmd_key is None: return text - return build_skill_invocation_message(cmd_key, arg.strip(), task_id=task_id) or text except Exception: - # A skill that no longer resolves (renamed, disabled, external dir - # gone) must not break the rewind — replay the text as typed. + # A skill that no longer resolves must not break the rewind. logger.debug("skill re-expansion failed for replay", exc_info=True) return text -# Opening of the crash-recovery note synthesized by _auto_continue_note. -# Matched (not just built) so a row persisted before the display type was -# stamped at turn start still reads as a timeline event, and to recognize the -# messaging gateway's twin note. +# Opening of the crash-recovery note synthesized by _auto_continue_note; matched (not just built) for +# rows persisted before display typing existed and for the messaging gateway's twin note. _AUTO_CONTINUE_NOTE_PREFIX = "[System note: Your previous turn was interrupted mid-run" def _legacy_display_kind(role: str, text: str) -> str | None: - """Infer the display type of a synthetic row persisted without one. - - Turn-start typing (see ``persist_user_display_kind``) covers everything - written from here on. Sessions already on disk carry untyped rows — and a - turn killed mid-run never reached the post-turn stamp at all, which is - exactly the auto-continue case — so the raw recovery note would paint as a - user bubble forever. Sniffing the one fixed synthetic prefix is the - migration for those rows; it is not how new rows get typed. - """ + """Infer the display type of a synthetic row persisted without one. New rows are typed at turn + start (``persist_user_display_kind``); this prefix sniff migrates untyped rows already on disk (a + turn killed mid-run never reached the stamp), which would otherwise paint as a user bubble.""" if role == "user" and text.lstrip().startswith(_AUTO_CONTINUE_NOTE_PREFIX): return "auto_continue" return None +_HISTORY_REASONING_KEYS = ("reasoning", "reasoning_content", "reasoning_details", "codex_reasoning_items") + + def _history_to_messages(history: list[dict]) -> list[dict]: messages = [] tool_call_args = {} @@ -343,11 +230,7 @@ def _history_to_messages(history: list[dict]) -> list[dict]: role = m.get("role") if role not in {"user", "assistant", "tool", "system"}: continue - # An explicit display_kind="hidden" row is model-facing scaffolding - # (compaction references, interrupted-turn checkpoints). The string - # sniff below only catches the "[System:" convention; honor the - # declared field too, or scaffolding reaches every surface that reads - # this projection. + # display_kind="hidden": model-facing scaffolding the "[System:" sniff does not catch. if m.get("display_kind") == "hidden": continue content_text = _coerce_message_text(m.get("content")) @@ -371,61 +254,34 @@ def _history_to_messages(history: list[dict]) -> list[dict]: name = (tc_info[0] if tc_info else None) or m.get("tool_name") or "tool" args = (tc_info[1] if tc_info else None) or {} tool_msg = {"role": "tool", "name": name, "context": _tool_ctx(name, args)} - # This is the display projection, so keep it faithful. `context` - # is an 80-char preview for collapsed row titles. A renderer that - # shows the full call (the expanded `$` transcript in the desktop) - # rebuilds it from args. When only the preview shipped, that - # truncation was permanent. + # `context` is an 80-char preview; ship args so a full-call renderer isn't truncated. if args: tool_msg["args"] = args messages.append(tool_msg) continue - # An assistant turn may carry only reasoning/thinking content with no - # visible text (extended-thinking turns, thinking-only recovery - # responses). Such a turn is persisted with its reasoning fields and is - # recallable from the transcript, but dropping it here as "empty" makes - # it vanish from the resumed/reloaded session view while the desktop's - # reasoning disclosure has nothing to render. Keep it when it carries - # reasoning so the "Thinking…" block still shows. (#44022) - reasoning_keys = ( - "reasoning", - "reasoning_content", - "reasoning_details", - "codex_reasoning_items", - ) - has_reasoning = role == "assistant" and any( - m.get(key) for key in reasoning_keys - ) + # A reasoning-only assistant turn is kept so "Thinking…" still shows after resume/reload. + has_reasoning = role == "assistant" and any(m.get(key) for key in _HISTORY_REASONING_KEYS) if not content_text.strip() and not has_reasoning: continue msg = {"role": role, "text": content_text} - # Persisted authoring time (Unix seconds) for display.timestamps - # renderers (#41531). Display-only: never fed back into model context. + # Authoring time (Unix seconds) for display.timestamps; display-only. ts = m.get("timestamp") if isinstance(ts, (int, float)) and ts > 0: msg["timestamp"] = float(ts) - # Durable row identity, stamped by _rows_to_conversation. The renderer's - # own message ids are ephemeral (timestamp+index derived, and a - # different shape for live vs rehydrated vs optimistic rows), so - # anything that addresses a specific persisted message later — message - # reactions — needs this instead. + # Durable row identity (_rows_to_conversation); reactions etc. address persisted messages by it. if m.get("_row_id") is not None: msg["row_id"] = m["_row_id"] if role == "user": invocation = _skill_scaffold_projection(content_text) if invocation: - # Show the invocation, never the expanded skill body. The raw - # payload stays server-side: a rewind/regenerate re-sends the - # turn by ordinal, so no client needs it. + # The invocation, never the expanded body (rewind re-sends by ordinal). msg["text"] = invocation msg["display_kind"] = "skill_invocation" if role == "assistant": - for key in reasoning_keys: + for key in _HISTORY_REASONING_KEYS: if key in m and m.get(key) is not None: msg[key] = m.get(key) - # Forward display-only timeline metadata so the TUI can render - # model switches and delegation completions as events instead of - # opaque user messages, and hide compaction handoffs entirely. + # Display-only timeline metadata (model switches, delegation events). display_kind = m.get("display_kind") or _legacy_display_kind(role, content_text) if display_kind: msg["display_kind"] = display_kind @@ -439,24 +295,19 @@ def _history_to_messages(history: list[dict]) -> list[dict]: def _coerce_seed_history(value: Any) -> list[dict]: if not isinstance(value, list): return [] - history = [] for item in value: if not isinstance(item, dict): continue - role = item.get("role") if role not in ("user", "assistant", "system"): continue - content = item.get("content") if content is None: content = item.get("text") if not isinstance(content, str) or not content.strip(): continue - history.append({"role": role, "content": content}) - return history @@ -467,11 +318,7 @@ def _inflight_text(value: Any) -> str: def _start_inflight_turn(session: dict, text: Any) -> None: now = time.time() session["inflight_turn"] = { - "assistant": "", - "started_at": now, - "streaming": True, - "updated_at": now, - "user": _inflight_text(text), + "assistant": "", "started_at": now, "streaming": True, "updated_at": now, "user": _inflight_text(text), } @@ -489,14 +336,8 @@ def _append_inflight_delta(session: dict, delta: Any) -> None: def _record_inflight_correction(session: dict, text: Any) -> None: - """Record an accepted mid-turn correction on the live turn. - - The correction is appended, never written over ``user``: a resuming client - must be able to rebuild BOTH bubbles. Overwriting the slot erased the - prompt that started the turn from the only snapshot resume can read, so a - reconnect (or a dev hot-reload that wipes the renderer cache) repainted the - thread with the user's original message missing. - """ + """Record an accepted mid-turn correction on the live turn — appended, never written over ``user``, + so a resuming client can rebuild BOTH bubbles.""" correction = _inflight_text(text) if not correction: return @@ -504,16 +345,10 @@ def _record_inflight_correction(session: dict, text: Any) -> None: if not isinstance(turn, dict): return turn = dict(turn) - corrections = list(turn.get("corrections") or []) - corrections.append(correction) - turn["corrections"] = corrections - # Arrival-order boundary: how much assistant text had already streamed - # when this correction was accepted. Resuming clients use it to place the - # correction bubble AFTER the output the user had already seen and BEFORE - # the output it redirected (#73793) instead of above the whole reply. - offsets = list(turn.get("correction_offsets") or []) - offsets.append(len(str(turn.get("assistant") or ""))) - turn["correction_offsets"] = offsets + turn["corrections"] = [*(turn.get("corrections") or []), correction] + # Arrival-order boundary (assistant chars already streamed) so resuming clients place the bubble + # between the output seen and the output redirected. + turn["correction_offsets"] = [*(turn.get("correction_offsets") or []), len(str(turn.get("assistant") or ""))] turn["updated_at"] = time.time() session["inflight_turn"] = turn @@ -522,22 +357,11 @@ def _clear_inflight_turn(session: dict) -> None: session["inflight_turn"] = None -def _fail_inflight_turn( - session: dict, error: Any, error_surface: Optional[dict] = None -) -> None: - """Mark the in-flight turn terminal-error but keep it replayable. - - Normal completion clears ``inflight_turn`` because the response is now in - canonical history. Failures are different: the terminal frame can be lost - on a WS disconnect, and the failed turn may never have been committed. - Retaining a compact error snapshot lets ``session.resume`` replay the - user's prompt, any partial assistant text, and the error itself instead of - leaving the client stranded on a spinner or hydrating from stale DB state. - The snapshot lives until the next turn starts (``_start_inflight_turn`` - overwrites it) or the session closes. - - Caller must hold ``session["history_lock"]``. - """ +def _fail_inflight_turn(session: dict, error: Any, error_surface: Optional[dict] = None) -> None: + """Mark the in-flight turn terminal-error but keep it replayable: a failure's terminal frame can be + lost on WS disconnect and the turn may never have been committed, so the snapshot lets + ``session.resume`` replay prompt, partial text and error instead of stranding the client on a + spinner. Lives until the next turn starts or the session closes. Caller holds history_lock.""" message = str(error) if not isinstance(error, BaseException) else (str(error) or type(error).__name__) now = time.time() turn = session.get("inflight_turn") @@ -549,9 +373,7 @@ def _fail_inflight_turn( turn["status"] = "error" turn["recoverable"] = True if error_surface: - # Structured {layer, code, retryable} descriptor — replayed to - # resuming clients via the resume snapshot so a reconnect renders the - # same layered error card the live frame carried. + # {layer, code, retryable} so a reconnect renders the same layered error card. turn["error_surface"] = dict(error_surface) else: turn.pop("error_surface", None) @@ -561,37 +383,17 @@ def _fail_inflight_turn( _TURN_FAILURE_DETAIL_LIMIT = 240 -# Shortest run of the submitted prompt that counts as the provider quoting it -# back. Long enough that shared boilerplate ("Invalid request for model ") does -# not trip it, short enough to catch a quoted sentence. +# Shortest prompt run counting as a quote-back: above shared boilerplate, below a quoted sentence. _TURN_PROMPT_ECHO_WINDOW = 24 -# Ceiling on the prompt we shingle. An @-expanded prompt can carry a whole -# file; the failure path must stay cheap. +# Ceiling on the prompt we shingle (an @-expanded prompt can carry a whole file). _TURN_PROMPT_ECHO_MAX_PROMPT = 65536 def _strip_prompt_echo(message: str, prompt: Any) -> str: - """Blank runs of the submitted prompt that ``message`` quotes back. - - Secret redaction and prompt omission are different contracts, and only the - first one is pattern-based. A provider 4xx that echoes the request carries - ordinary private prose -- a paragraph about a person, a pasted file from an - ``@`` reference -- that matches no credential pattern and would otherwise - reach the log intact. This closes that path directly: anything the message - shares with the prompt for ``_TURN_PROMPT_ECHO_WINDOW`` characters or more - becomes ````. - - Shingle-set matching, not a diff: cost is linear in both strings, which - matters because this runs on every failed turn and an ``@`` reference can - make the prompt arbitrarily long. The JSON-escaped form of the prompt is - shingled too, since a provider that hands back its own request body often - hands it back escaped. - - Verbatim echo is what this stops. A paraphrase, a re-encoding (base64, a - different unicode normalization) or a summary of the prompt would survive, - so this is a floor and not a proof; the guarantee it does give is that the - prompt cannot reach the record by being quoted. - """ + """Blank runs of the submitted prompt that ``message`` quotes back: secret redaction is pattern-based + and a provider 4xx echoing the request carries private prose matching no pattern. Any run of + ``_TURN_PROMPT_ECHO_WINDOW``+ chars shared with the prompt (or its JSON-escaped form) becomes + ````. Shingle-set matching keeps it linear. Only verbatim echo is stopped — a floor.""" if not message or not prompt: return message needle = " ".join(str(prompt).split())[:_TURN_PROMPT_ECHO_MAX_PROMPT] @@ -604,9 +406,7 @@ def _strip_prompt_echo(message: str, prompt: Any) -> str: except Exception: escaped = "" if escaped and escaped != needle: - shingles.update( - escaped[i:i + window] for i in range(len(escaped) - window + 1) - ) + shingles.update(escaped[i:i + window] for i in range(len(escaped) - window + 1)) out: list[str] = [] i = 0 n = len(message) @@ -625,34 +425,11 @@ def _strip_prompt_echo(message: str, prompt: Any) -> str: def _turn_failure_detail(error: Any, reason: Any = None, prompt: Any = None) -> str: - """Render why a turn failed, for the ``tui turn finished`` bookend. - - Returns ``""`` when there is nothing to say, otherwise a fragment that - already carries its own leading space, so the caller can append it to the - record unconditionally. - - #86865 added the bookend to trace compression rotations, so it logs - identities and a coarse ``status`` and deliberately logs no content. - #89117 is what the missing cause costs: a report consisting of two lines - reading ``status=error error_retained=True duration=0.9s`` with no way to - tell a provider 4xx from a budget wall from a crashed finalizer. The - returned-error path -- the one a 0.9 s failure almost always takes -- - emits no other log line at all; only the exception path prints to stderr, - which is why the quiet failures are the ones that get filed. - - Content discipline follows #86865's, and it takes two separate steps - because it is two separate contracts. ``redact_sensitive_text`` removes - credentials, which are pattern-shaped. It does nothing about a 4xx body - that quotes the request back, because ordinary private prose is not - pattern-shaped -- so ``_strip_prompt_echo`` removes that separately, using - the submitted ``prompt`` itself as the thing to look for. The invariant the - two of them keep is: this record may gain failure classification and - provider detail, and may not newly persist the user's own content. - - ``prompt`` is optional so the helper stays callable from a path that has no - prompt in scope, but the turn paths always pass it; without it, only the - secret contract is enforced. - """ + """Why a turn failed, for the ``tui turn finished`` bookend: ``""`` when nothing to say, else a + fragment with its own leading space (distinguishes a provider 4xx from a budget wall or crashed + finalizer). Two content contracts: ``redact_sensitive_text`` removes credentials; + ``_strip_prompt_echo`` removes a 4xx body quoting ``prompt`` back. Invariant: this record may gain + failure classification and provider detail, never the user's own content.""" reason_text = str(reason or "").strip() message = str(error or "").strip() if isinstance(error, BaseException): @@ -664,13 +441,9 @@ def _turn_failure_detail(error: Any, reason: Any = None, prompt: Any = None) -> message = redact_sensitive_text(message, force=True) except Exception: - # A redactor that cannot run must not be able to leak the raw - # message into the log by failing open. - message = "" + message = "" # never fail open message = " ".join(message.split()) - # After the collapse, so both sides are compared in the same shape, and - # before the truncation, so a quote that starts inside the kept prefix - # cannot survive by being cut mid-run. + # After the collapse (same shape both sides), before truncation (a quote must not survive the cut). message = _strip_prompt_echo(message, prompt) if len(message) > _TURN_FAILURE_DETAIL_LIMIT: message = message[:_TURN_FAILURE_DETAIL_LIMIT] + "\u2026" diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index 96e6961438..c69455a957 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -13,47 +13,26 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -def _notify_session_boundary( - event_type: str, session_id: str | None, platform: str | None = None -) -> None: +def _notify_session_boundary(event_type: str, session_id: str | None, platform: str | None = None) -> None: """Fire session lifecycle hooks with CLI parity.""" - try: + with contextlib.suppress(Exception): from hermes_cli.lifecycle import finalize_session, invoke_hook if event_type == "on_session_finalize": - finalize_session( - session_id=session_id, - platform=_resolve_agent_platform(platform), - ) + finalize_session(session_id=session_id, platform=_resolve_agent_platform(platform)) else: - invoke_hook( - event_type, - session_id=session_id, - platform=_resolve_agent_platform(platform), - ) - except Exception: - pass + invoke_hook(event_type, session_id=session_id, platform=_resolve_agent_platform(platform)) -_SESSION_OWNERSHIP_UNAVAILABLE = ( - "Hermes could not safely reserve this session. Try again." -) +_SESSION_OWNERSHIP_UNAVAILABLE = "Hermes could not safely reserve this session. Try again." _AUTOMATIC_SESSION_END_REASONS = frozenset({ - "ws_orphan_reap", - "ws_disconnect", - "idle_timeout", - "lru_evict", - "tui_shutdown", + "ws_orphan_reap", "ws_disconnect", "idle_timeout", "lru_evict", "tui_shutdown", }) def _claim_active_session_slot( - session_key: str, - *, - live_session_id: str, - surface: str = "tui", - profile_home: str | Path | None = None, + session_key: str, *, live_session_id: str, surface: str = "tui", profile_home: str | Path | None = None ) -> tuple[Any, str | None]: track_liveness = str(surface or "").strip().lower() == "desktop" try: @@ -69,27 +48,17 @@ def _claim_active_session_slot( ) except Exception as exc: logger.warning("Failed to claim active session slot: %s", exc) - # Fail CLOSED regardless of surface: per-session exclusivity is a - # correctness guarantee (see PER_SESSION_EXCLUSIVE_SUBMIT), and a - # claim that errors out has NOT proven the session is unowned. - # Proceeding without a lease here is the silent double-writer hole - # flagged in the #94595 review (blocker 2). + # Fail CLOSED regardless of surface: an errored claim has NOT proven the session + # unowned, and proceeding lease-less is a silent double-writer hole. return (None, _SESSION_OWNERSHIP_UNAVAILABLE) def _ensure_active_session_slot(sid: str, session: dict) -> str | None: """Claim this session's cap slot on its first real turn; None when ok. - session.create / session.resume deliberately do NOT claim one. Every - desktop tile paint, background reconnect-resume and abandoned draft opens a - session just to paint a composer, and a slot held by one of those is - invisible everywhere: an unprompted draft has no DB row, and the sidebar - filters it out with min_messages=1. Idle desktop tabs therefore silently - starved the messaging gateway, which shares this cap — five parked tabs on - a websocket-flappy host locked a Discord bot out of a 5-slot cap while - running no agents at all. Claiming on the first turn mirrors the lazy - contract _ensure_session_db_row already uses for the row itself, and keeps - the invariant that anything holding a slot is something the user can see. + session.create/resume deliberately do NOT claim: tile paints, reconnect-resumes and + abandoned drafts would hold invisible slots (no DB row) that starve the messaging + gateway sharing the cap. Anything holding a slot must be user-visible. """ if session.get("active_session_lease") is not None: return None @@ -118,9 +87,7 @@ def _release_active_session_lease(lease) -> bool: logger.warning("Failed to release active session slot", exc_info=True) return False time.sleep(0.05 * (attempt + 1)) - released = getattr(lease, "released", True) - enabled = getattr(lease, "enabled", True) - return bool(released or not enabled) + return bool(getattr(lease, "released", True) or not getattr(lease, "enabled", True)) def _release_active_session_slot(session: dict | None) -> bool: @@ -145,9 +112,7 @@ def _other_runtime_lease_guard(session_id: str, session: dict): ) except Exception as exc: logger.warning( - "Failed to load active session ownership guard; preserving session %s: %s", - session_id, - exc, + "Failed to load active session ownership guard; preserving session %s: %s", session_id, exc ) yield True return @@ -159,9 +124,7 @@ def _other_runtime_lease_guard(session_id: str, session: dict): if lease is not None and getattr(lease, "enabled", False): guard = release_active_session_liveness_guard(lease, session_id) else: - guard = active_session_liveness_guard( - session_id, registry_home=session.get("profile_home") - ) + guard = active_session_liveness_guard(session_id, registry_home=session.get("profile_home")) active = stack.enter_context(guard) break except Exception as exc: @@ -172,9 +135,7 @@ def _other_runtime_lease_guard(session_id: str, session: dict): time.sleep(0.05 * (attempt + 1)) else: logger.warning( - "Failed to inspect active session leases; preserving session %s: %s", - session_id, - last_error, + "Failed to inspect active session leases; preserving session %s: %s", session_id, last_error ) yield True return @@ -190,12 +151,7 @@ def _other_runtime_lease_guard(session_id: str, session: dict): session.pop("active_session_lease", None) -def _transfer_active_session_slot( - sid: str, - session: dict, - *, - new_session_id: str, -) -> bool: +def _transfer_active_session_slot(sid: str, session: dict, *, new_session_id: str) -> bool: if not new_session_id: return False lease = session.get("active_session_lease") @@ -204,11 +160,7 @@ def _transfer_active_session_slot( try: from hermes_cli.active_sessions import transfer_active_session - if transfer_active_session( - lease, - session_id=new_session_id, - metadata={"live_session_id": sid}, - ): + if transfer_active_session(lease, session_id=new_session_id, metadata={"live_session_id": sid}): return True except Exception: logger.debug("Failed to transfer active session slot", exc_info=True) @@ -216,11 +168,9 @@ def _transfer_active_session_slot( if getattr(lease, "track_liveness", False): return False - # Fallback: the in-place transfer could not move the lease (entry pruned / - # pid-check transiently failed). Reserve the new slot BEFORE releasing the - # old one, so a concurrent gateway at the session cap cannot grab the freed - # slot in a release-then-reacquire window and leave this session with no - # lease at all (#49041 review). If the reserve fails, KEEP the old lease. + # Fallback (entry pruned / pid-check transiently failed): reserve the new slot BEFORE + # releasing the old one so a gateway at the cap can't grab the freed slot and leave + # this session lease-less. On reserve failure KEEP the old lease. new_lease, limit_message = _claim_active_session_slot( new_session_id, live_session_id=sid, @@ -236,24 +186,18 @@ def _transfer_active_session_slot( logger.debug("Failed to release stale active session slot", exc_info=True) session["active_session_lease"] = new_lease return True - # Reserve failed — retain the existing lease rather than dropping it. if limit_message: logger.warning( "Compression session lease re-anchor failed (kept old lease): " "sid=%s new_session_id=%s reason=%s", - sid, - new_session_id, - limit_message, + sid, new_session_id, limit_message, ) return False -# Session sources the TUI/desktop backend must never end in state.db: the -# messaging gateway owns those sessions' lifecycle — the TUI is only a viewer -# (a resume of a Telegram/Discord/... session). Ending one creates the -# #60609 Groundhog Day routing loop (see _finalize_session). Sources the -# TUI backend itself creates ("tui", plus whatever a client passes as its -# own ``source``) and the CLI's own sessions are NOT gateway-owned. +# Sources this backend must never end in state.db: the messaging gateway owns those +# sessions and the TUI is only a viewer (ending one causes the Groundhog Day routing +# loop, see _finalize_session). Self-created and CLI sources are NOT gateway-owned. _NON_GATEWAY_SOURCES = frozenset({ "", "tui", "cli", "webui", "desktop", "cron", "kanban", "subagent", "test", "local", "acp", "webhook", "api_server", "msgraph_webhook", @@ -261,16 +205,11 @@ _NON_GATEWAY_SOURCES = frozenset({ def _is_gateway_owned_source(source: str) -> bool: - """True when ``source`` names a messaging-gateway platform whose session - lifecycle belongs to the gateway, not to this TUI backend. + """True when ``source`` is a messaging-gateway platform owning its session lifecycle. - Structural rather than a hardcoded platform list: any source that - resolves to a known gateway ``Platform`` (built-in enum member OR a - registered platform plugin, via ``Platform._missing_``) counts, so new - platforms are covered automatically. Local/self-owned sources are - excluded explicitly — ``local``/``webhook``/``api_server`` are Platform - members but their sessions are not owned by a remote chat surface that - routes by session_key, so reaping them is safe and keeps /resume clean. + Structural: any source resolving to a gateway ``Platform`` (enum member or plugin via + ``Platform._missing_``) counts, so new platforms are covered automatically. Self-owned + sources (``local``/``webhook``/``api_server`` are Platform members) are excluded explicitly. """ src = (source or "").strip().lower() if src in _NON_GATEWAY_SOURCES: @@ -284,15 +223,24 @@ def _is_gateway_owned_source(source: str) -> bool: return False -def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> None: - """Best-effort finalize hook + memory commit for a session. +def _lifecycle_own_sid(session: dict, sid_hint: str = "") -> str: + """Live UI sid for ``session``: hint, stamped ``_sid``, else registry scan.""" + own_sid = str(sid_hint or session.get("_sid") or "") + if not own_sid: + try: + with _sessions_lock: + for _cand_sid, _cand in _sessions.items(): + if _cand is session: + own_sid = _cand_sid + break + except Exception: + own_sid = "" + return own_sid - Fires ``on_session_end`` plugin hook and attempts to persist any - unflushed messages before closing the session. This mirrors the - CLI's exit-path behaviour and prevents data loss when the TUI is - force-quit (double Ctrl‑C, terminal‑close, SIGHUP) while the agent - is mid‑turn. - """ + +def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> None: + """Best-effort finalize hook + memory commit; mirrors the CLI exit path so a + force-quit mid-turn (double Ctrl-C, terminal close, SIGHUP) loses nothing.""" if not session or session.get("_finalized"): return session["_finalized"] = True @@ -300,12 +248,12 @@ def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> No if history_ready is not None and not history_ready.is_set(): session["resume_history_error"] = "session resume cancelled" history_ready.set() - _desktop_automatic_cleanup = bool( + _desktop_automatic_cleanup = ( end_reason in _AUTOMATIC_SESSION_END_REASONS and _session_source(session).strip().lower() == "desktop" ) - # Automatic Desktop cleanup removes its lease inside the lock-held lifecycle - # guard below. Explicit close and non-Desktop paths keep force/end semantics. + # Automatic Desktop cleanup releases its lease inside the lock-held lifecycle guard + # below; explicit close and non-Desktop paths keep force/end semantics. if not _desktop_automatic_cleanup: _release_active_session_slot(session) stop_event = session.get("_notif_stop") @@ -314,67 +262,45 @@ def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> No agent = session.get("agent") lock = session.get("history_lock") - if lock is not None: - with lock: - history = list(session.get("history", [])) - else: + with (lock if lock is not None else contextlib.nullcontext()): history = list(session.get("history", [])) - # ── Persist unflushed messages to SQLite ────────────────────────── - # Flush ``agent._session_messages`` via ``_persist_session``'s marker-based - # dedup (same contract as the gateway-shutdown flush, #13121). Do NOT pass - # ``conversation_history``: ``session["history"]`` and ``_session_messages`` - # alias the SAME list once a turn completes, so passing it made - # ``_flush_messages_to_session_db`` treat every message as already-durable - # and skip it — a data-loss bug when finalize is the sole persist path after - # a WS disconnect/restart (e.g. the in-turn flush hit a transient SQLite - # failure). Markers persist the genuinely-unflushed tail without duplicating - # durable rows (including a resumed-but-not-run session's already-in-DB - # transcript, which stays in ``session["history"]`` only). + # Persist via ``_persist_session``'s marker-based dedup (same contract as the + # gateway-shutdown flush). Do NOT pass ``conversation_history``: ``session["history"]`` + # and ``_session_messages`` alias the SAME list after a turn, so the flush would treat + # every message as durable and skip it — data loss when finalize is the sole persist + # path after a WS disconnect/restart. if agent is not None and hasattr(agent, "_persist_session"): snapshot = getattr(agent, "_session_messages", None) if snapshot: - try: + with contextlib.suppress(Exception): agent._persist_session(snapshot) - except Exception: - pass - # ── Plugin hook: on_session_end ──────────────────────────────────── - # Signals every plugin that the session is closing, with - # interrupted=True so crash‑recovery plugins can flush buffers, - # persist state, or close connections before the gateway exits. - # Mirrors cli.py's atexit handler that fires the same hook when - # the user Ctrl‑C's mid‑turn. + # interrupted=True so crash-recovery plugins can flush state (mirrors cli.py atexit). if agent is not None: - try: + with contextlib.suppress(Exception): from hermes_cli.lifecycle import invoke_hook invoke_hook( "on_session_end", - session_id=getattr(agent, "session_id", None) - or session.get("session_key", ""), + session_id=getattr(agent, "session_id", None) or session.get("session_key", ""), completed=False, interrupted=True, model=getattr(agent, "model", "unknown"), platform=getattr(agent, "platform", None) or "tui", ) - except Exception: - pass if agent is not None and history and hasattr(agent, "commit_memory_session"): - try: + with contextlib.suppress(Exception): agent.commit_memory_session(history) - except Exception: - pass session_key = session.get("session_key") session_id = getattr(agent, "session_id", None) or session_key _notify_session_boundary("on_session_finalize", session_id, _session_source(session)) - # Mark session ended in DB so it doesn't linger as a ghost row in /resume. - # Use session_id (from agent.session_id) not session_key — after compression, - # session_key may be stale (the ended parent) while session_id is the live - # continuation. Fix for #20001. + # End the state.db row so it doesn't linger as a ghost in /resume. Use session_id + # (agent.session_id), not session_key: after compression the key may be the stale + # ended parent while session_id is the live continuation. if _desktop_automatic_cleanup and not session_id: _release_active_session_slot(session) _lifecycle_guard = ( @@ -387,92 +313,55 @@ def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> No if _other_runtime_owns_lifecycle: logger.info( "Preserving session %s during %s: another backend owns an active lease", - session_id, - end_reason, + session_id, end_reason, ) if session_id: - try: - # End the row in the *session's* profile state.db (app-global - # remote mode), not the launch profile's shared handle. + with contextlib.suppress(Exception): + # The *session's* profile state.db (app-global remote mode), not the launch profile's. with _session_db(session) as db: if db is not None: - # Don't end gateway-originated sessions — the gateway owns - # their lifecycle. The TUI is a viewer, not the owner. - # Ending a gateway session in state.db triggers a Groundhog - # Day routing loop: the gateway's #54878 self-heal detects - # the stale entry, recovers to the parent session, context - # compression splits back to the reaped child, and the cycle - # repeats on every inbound message. (#60609) + # Never end gateway-originated sessions (Groundhog Day loop: the + # gateway's self-heal recovers to the parent, compression splits back + # to the reaped child, repeat on every message). row = db.get_session(session_id) source = (row or {}).get("source", "") if _is_gateway_owned_source(source): _tui_owns_lifecycle = False elif _tui_owns_lifecycle: db.end_session(session_id, end_reason) - except Exception: - pass - # A session's in-flight async delegations end WITH the session (#55578): - # once nobody owns the return address, a still-running background subagent - # can only burn tokens and park an orphaned completion on the shared - # queue. Always interrupt delegations commissioned by THIS live UI session - # (its sid); additionally interrupt by durable session_key, but only when - # the TUI owns the lifecycle — closing a viewer tab on a live gateway - # session must not kill the gateway's own background work. - try: + # In-flight async delegations end WITH the session (no return address left). Always + # interrupt by THIS live UI sid; by durable session_key only when the TUI owns the + # lifecycle — closing a viewer tab must not kill the gateway's own background work. + with contextlib.suppress(Exception): from tools.async_delegation import interrupt_for_session - _own_sid = str(session.get("_sid") or "") - if not _own_sid: - try: - with _sessions_lock: - for _cand_sid, _cand in _sessions.items(): - if _cand is session: - _own_sid = _cand_sid - break - except Exception: - _own_sid = "" interrupt_for_session( session_key=str(session_key or "") if _tui_owns_lifecycle else "", - origin_ui_session_id=_own_sid, + origin_ui_session_id=_lifecycle_own_sid(session), reason=end_reason, ) - except Exception: - pass - # Close the slash-worker subprocess as part of finalize itself, not just - # in the callers. Defense-in-depth: every session-end path goes through - # _finalize_session (it's the single ``_finalized``-guarded chokepoint), so - # folding worker cleanup in here means a future code path that calls - # _finalize_session directly — without the surrounding _teardown_session / - # _shutdown_sessions worker.close() — can't reintroduce the #38095 leak. - # Idempotent: _SlashWorker.close() is poll()-guarded, so the explicit - # close() still in those callers is harmless. - try: + # Close the slash-worker in this single ``_finalized``-guarded chokepoint so a direct + # _finalize_session caller can't leak it. Idempotent: close() is poll()-guarded. + with contextlib.suppress(Exception): worker = session.get("slash_worker") if worker: worker.close() - except Exception: - pass -# End reasons where the BACKEND reclaimed a session the client never asked to -# close: the idle-TTL reaper, the LRU cap, and the WS-orphan reap. A client -# holding that live session id gets no signal today — its next prompt fails -# against an id the backend has already forgotten, which reads as the session -# silently vanishing rather than being reclaimed. ``tui_close`` and friends are -# deliberately absent: the client initiated those and already knows. +# End reasons where the BACKEND reclaimed a session the client never asked to close; +# without a signal the client's next prompt fails against a forgotten id. Client-initiated +# reasons (``tui_close`` etc.) are deliberately absent. _RECLAIM_END_REASONS = frozenset({"idle_timeout", "lru_evict", "ws_orphan_reap"}) def _announce_session_reclaimed(session: dict, end_reason: str) -> None: """Tell connected clients a session was reclaimed out from under them. - Broadcast rather than session-targeted: the reap paths run on background - timer threads with no contextvar binding, and the WS-orphan case has by - definition lost its own transport — ``_emit`` would bottom out on stdio and - the peer that owns the session would never see it. Best-effort; a failed - notify must never break teardown. + Broadcast, not session-targeted: reap paths run on timer threads with no contextvar + binding and the WS-orphan case has lost its transport, so ``_emit`` would bottom out + on stdio. Best-effort; never breaks teardown. """ if end_reason not in _RECLAIM_END_REASONS: return @@ -490,42 +379,29 @@ def _announce_session_reclaimed(session: dict, end_reason: str) -> None: def _teardown_session(session: dict | None, *, end_reason: str = "tui_close") -> None: - """Fully tear down a session: finalize, unregister, close agent + worker. + """Fully tear down a session: finalize, unregister notifier, close agent. - Shared by ``session.close`` and the orphaned-WS-session reaper. The - slash-worker subprocess is closed inside ``_finalize_session`` (the single - finalize chokepoint); this still unregisters the approval notifier and - closes the in-process agent. Idempotent: the ``_finalized`` guard in - ``_finalize_session`` and the ``poll()`` guard in ``_SlashWorker.close`` - make repeat calls harmless. + Shared by ``session.close`` and the orphaned-WS reaper. The slash-worker is closed in + ``_finalize_session`` (the single chokepoint), NOT here. Idempotent via ``_finalized``. """ if not session: return _finalize_session(session, end_reason=end_reason) _announce_session_reclaimed(session, end_reason) - try: + with contextlib.suppress(Exception): from tools.approval import unregister_gateway_notify if key := session.get("session_key"): unregister_gateway_notify(key) - except Exception: - pass - try: + with contextlib.suppress(Exception): agent = session.get("agent") if agent is not None and hasattr(agent, "close"): agent.close() - except Exception: - pass - # NOTE: the slash-worker is closed inside _finalize_session (the single - # _finalized-guarded chokepoint that main folded it into), exactly once. - # We deliberately do NOT re-close it here — _teardown_session's job beyond - # finalize is unregistering the notifier and closing the in-process agent. def _attach_worker(sid: str, session: dict, worker) -> None: """Store worker on session iff sid still maps to it, else close it — a - concurrent teardown already popped the session and would orphan the - worker. Closes the create/close race at every slash-worker spawn site.""" + concurrent teardown already popped the session and would orphan the worker.""" with _sessions_lock: if _sessions.get(sid) is session: session["slash_worker"] = worker @@ -534,15 +410,10 @@ def _attach_worker(sid: str, session: dict, worker) -> None: def _pop_session_by_id(sid: str) -> dict | None: - """Atomically detach one live session from the registry. - - Detaching is the ownership claim for teardown: once the record is no - longer in ``_sessions``, a concurrent close/reaper becomes a no-op. Keep - this operation separate from ``_teardown_session`` because finalization can - flush SQLite state, invoke plugins, commit memory, interrupt delegations, - and close agents/workers. None of that slow external work belongs under - the global ``_session_resume_lock``. - """ + """Atomically detach one live session from the registry — the ownership claim for + teardown (a concurrent close/reaper then no-ops). Separate from ``_teardown_session`` + because finalization does slow external work that must not run under + ``_session_resume_lock``.""" with _sessions_lock: session = _sessions.get(sid) if session is not None: @@ -550,25 +421,17 @@ def _pop_session_by_id(sid: str) -> dict | None: _sessions.pop(sid, None) if session is None: return None - # The session is already out of _sessions here, so downstream teardown - # (e.g. _finalize_session's per-session async-delegation interrupt) can't - # recover its live id by scanning the dict — stamp it on the record. + # Out of _sessions now, so teardown can't recover the live id by scanning — stamp it. session["_sid"] = sid return session -def _teardown_popped_session( - session: dict | None, *, end_reason: str = "tui_close" -) -> bool: +def _teardown_popped_session(session: dict | None, *, end_reason: str = "tui_close") -> bool: """Finish a close after the caller has atomically detached the session.""" if session is None: return False run_thread = session.get("_run_thread") - if ( - end_reason != "tui_shutdown" - and run_thread is not None - and run_thread is not threading.current_thread() - ): + if end_reason != "tui_shutdown" and run_thread is not None and run_thread is not threading.current_thread(): try: if run_thread.is_alive(): run_thread.join(timeout=_TURN_SETTLE_BEFORE_CLOSE_SECONDS) @@ -584,23 +447,12 @@ def _teardown_popped_session( def _close_session_by_id( - sid: str, - *, - end_reason: str = "tui_close", - predicate: Callable[[dict], bool] | None = None, + sid: str, *, end_reason: str = "tui_close", predicate: Callable[[dict], bool] | None = None ) -> bool: - """Single idempotent teardown funnel for callers needing no resume race. - - Resume-sensitive callers first pop under ``_session_resume_lock`` and then - call ``_teardown_popped_session`` after releasing it. Other reapers can use - this convenience wrapper directly. The pop remains the single atomic - ownership claim, so concurrent/repeat close attempts stay harmless. - - Automatic reapers can pass ``predicate`` to revalidate under - ``_sessions_lock`` immediately before the ownership claim. This prevents a - stale scan result from closing a session that reattached or gained active - delegated work before teardown. - """ + """Idempotent teardown funnel for callers with no resume race (resume-sensitive callers + pop under ``_session_resume_lock`` and call ``_teardown_popped_session`` after releasing + it). Automatic reapers pass ``predicate`` to revalidate under ``_sessions_lock`` right + before the claim, so a stale scan can't close a session that reattached.""" if predicate is None: session = _pop_session_by_id(sid) else: @@ -615,50 +467,28 @@ def _close_session_by_id( def _ws_session_is_detached(session: dict | None) -> bool: """True if a live session is still bound to the disconnected-WS sentinel.""" return bool( - session - and not session.get("_finalized") - and session.get("transport") is _detached_ws_transport + session and not session.get("_finalized") and session.get("transport") is _detached_ws_transport ) def _ws_session_is_orphaned(session: dict | None) -> bool: - """True if a WS session has no live transport and no in-flight turn. - - After ``handle_ws`` detaches a disconnected client it points the session at - ``_detached_ws_transport``. A session left on that transport (and not - mid-turn) is genuinely orphaned and safe to reap. - """ - return bool( - _ws_session_is_detached(session) - and session is not None - and not session.get("running") - ) + """True if a WS session sits on ``_detached_ws_transport`` (where + ``handle_ws`` parks disconnected clients) with no in-flight turn.""" + return bool(_ws_session_is_detached(session) and not session.get("running")) -def _interrupt_session_turn( - sid: str, session: dict, *, request_id: str | None = None -) -> bool: - """Apply the shared ``session.interrupt`` contract to one claimed session. - - Returns whether the interrupt used the compute-host control channel. The WS - orphan reaper calls this same helper after its reconnect grace expires, so a - dead client gets the same partial-history and queued-prompt semantics as an - explicit user interrupt. - """ +def _interrupt_session_turn(sid: str, session: dict, *, request_id: str | None = None) -> bool: + """Apply the shared ``session.interrupt`` contract to one claimed session; returns + whether the compute-host control channel was used. The WS orphan reaper reuses this so + a dead client gets the same partial-history and queued-prompt semantics.""" use_compute_host = _session_uses_compute_host(session) should_interrupt = bool(session.get("running")) run_thread_alive = False if use_compute_host: - # The host owns the live turn. Parent `running` is only a mirror and - # can lag behind a blocked interactive tool (a clarify parked on its - # Event keeps the host turn alive after the parent flag went stale), - # so let the host decide whether there is work to interrupt. - # Gate on `_compute_host_active` too: `_session_uses_compute_host` - # is also true for lazy sessions that never ran a hosted turn, and - # `HostSupervisor.interrupt()` calls `start()` — forwarding - # unconditionally would spawn a compute-host child just to deliver - # an interrupt no session ever submitted work to. + # The host owns the live turn (parent `running` can lag a blocked tool), so let it + # decide. Gate on `_compute_host_active`: HostSupervisor.interrupt() calls start(), + # so forwarding blindly for a lazy session would spawn a child just to interrupt. if should_interrupt or session.get("_compute_host_active"): _get_compute_host_supervisor().interrupt(sid, request_id=request_id) else: @@ -669,9 +499,7 @@ def _interrupt_session_turn( session["_turn_cancel_requested"] = True session["queued_prompt"] = None session.pop("queued_prompts", None) - session["_queued_prompt_generation"] = int( - session.get("_queued_prompt_generation", 0) - ) + 1 + session["_queued_prompt_generation"] = int(session.get("_queued_prompt_generation", 0)) + 1 if not use_compute_host: if should_interrupt: @@ -685,48 +513,33 @@ def _interrupt_session_turn( _clear_inflight_turn(session) _clear_pending(sid) - try: + with contextlib.suppress(Exception): from tools.approval import resolve_gateway_approval resolve_gateway_approval(session["session_key"], "deny", resolve_all=True) - except Exception: - pass return use_compute_host def _session_owns_durable_lifecycle(session_id: str | None) -> bool: - """Whether this TUI/desktop session may end its durable DB row by key.""" + """Whether this TUI/desktop session may end its durable DB row by key + (never for gateway-originated sessions — the TUI is only a viewer there).""" if not session_id: return True try: db = _get_db() if db is None: return True - # Don't end gateway-originated sessions — the gateway owns their - # lifecycle. The TUI is only a viewer there (#60609). row = db.get_session(session_id) - source = (row or {}).get("source", "") - return not _is_gateway_owned_source(source) + return not _is_gateway_owned_source((row or {}).get("source", "")) except Exception: return True -def _session_async_delegation_selectors( - session: dict | None, *, sid_hint: str = "" -) -> tuple[str, str]: +def _session_async_delegation_selectors(session: dict | None, *, sid_hint: str = "") -> tuple[str, str]: """Ownership selectors for async background work tied to one UI session.""" if not session: return "", "" - own_sid = str(sid_hint or session.get("_sid") or "") - if not own_sid: - try: - with _sessions_lock: - for _cand_sid, _cand in _sessions.items(): - if _cand is session: - own_sid = _cand_sid - break - except Exception: - own_sid = "" + own_sid = _lifecycle_own_sid(session, sid_hint) agent = session.get("agent") session_key = str(session.get("session_key") or "") session_id = getattr(agent, "session_id", None) or session_key @@ -735,83 +548,48 @@ def _session_async_delegation_selectors( def _session_has_active_delegations(sid: str, session: dict | None = None) -> bool: - """True when UI session ``sid`` still owns live background work. - - Matches by the live UI sid AND — when the TUI owns the durable lifecycle - (never for gateway-viewer tabs, #60609) — by the durable session_key, so a - delegation dispatched from an earlier tab of the same resumed session still - keeps it alive. - """ + """True when UI session ``sid`` still owns live background work — by live UI sid AND, + when the TUI owns the durable lifecycle (never for gateway-viewer tabs), by durable + session_key so a delegation from an earlier tab of the same session keeps it alive.""" if session is None: with _sessions_lock: session = _sessions.get(sid) - own_sid, owned_session_key = _session_async_delegation_selectors( - session, sid_hint=sid - ) + own_sid, owned_session_key = _session_async_delegation_selectors(session, sid_hint=sid) if not own_sid and not owned_session_key: return False try: from tools.async_delegation import has_live_for_session - return has_live_for_session( - session_key=owned_session_key, - origin_ui_session_id=own_sid, - ) + return has_live_for_session(session_key=owned_session_key, origin_ui_session_id=own_sid) except Exception: - logger.debug( - "Failed to query active delegations for UI session %s", - sid, - exc_info=True, - ) - # A transient registry/import failure must not turn into destructive - # cleanup. Conservatively keep the detached session and let the next - # orphan timer retry the lookup. + logger.debug("Failed to query active delegations for UI session %s", sid, exc_info=True) + # A transient registry/import failure must not become destructive cleanup. return True -# One pending WS-orphan reap Timer per live sid. Registered by -# _schedule_ws_orphan_reap, popped when its _reap fires, and cancelled by -# _cancel_ws_orphan_reap from every resume/reuse/transport-rebind path. Without -# this cancellation the reap could fire against an already-reattached session, -# broadcast session.reclaimed, and trigger the client's auto-re-resume — a -# reap->broadcast->resume feedback storm. Guarded by _sessions_lock. +# One pending WS-orphan reap Timer per live sid; guarded by _sessions_lock. Cancelled by +# _cancel_ws_orphan_reap from every resume/reuse/transport-rebind path — otherwise a reap +# could fire on a reattached session and trigger a reap->broadcast->resume storm. _pending_ws_reaps: dict[str, threading.Timer] = {} def _cancel_ws_orphan_reap(sid: str) -> None: - """Cancel a pending WS-orphan reap for ``sid`` (client came back). - - Called from every path that re-binds a live transport onto the session: - the session.resume fast-path reuse, _claim_or_reuse_live winners, and the - _live_session_payload transport rebind. Cancelling here (rather than - relying on the reap's own orphan re-check) removes the window where a - fired-but-not-yet-run Timer races the resume, and stops dead Timers from - accumulating for sessions that reconnect frequently. - """ + """Cancel a pending WS-orphan reap for ``sid`` (client came back). Called from every + path that re-binds a live transport; closes the fired-but-not-run Timer race and stops + dead Timers accumulating on flappy clients.""" with _sessions_lock: timer = _pending_ws_reaps.pop(sid, None) if timer is not None: - try: + with contextlib.suppress(Exception): timer.cancel() - except Exception: - pass def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: - """Whether a detached RUNNING turn's activity clock is still fresh. - - Reuses the agent's existing activity summary (``_touch_activity`` is - stamped by API waits, stream tokens, and tool heartbeats — the same - clock the turn-liveness watchdog samples; see agent/turn_liveness.py). - Fresh means the WS-orphan reaper must NOT interrupt the turn yet - (#98028/#100325): deliberate client absence (closed laptop, backgrounded - mobile app, desktop update/relaunch) keeps healthy work running detached. - - Conservative fallbacks preserve the wedged-turn safety net: a disabled - threshold (<= 0), a missing/opaque agent, an unreadable summary, or a - never-stamped clock all report NOT fresh, i.e. eligible for the - interrupt-at-grace path exactly as before. - """ + """Whether a detached RUNNING turn's activity clock (``_touch_activity``, the watchdog's + clock) is still fresh — the reaper must NOT interrupt healthy detached work (closed + laptop, backgrounded app). Conservative fallbacks keep the wedged-turn safety net: + disabled threshold, missing/opaque agent, unreadable summary or never-stamped clock all + report NOT fresh, i.e. eligible for interrupt-at-grace.""" if _WS_ORPHAN_ACTIVITY_STALE_S <= 0: return False agent = session.get("agent") @@ -826,33 +604,22 @@ def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: - """After a grace window, reap session ``sid`` iff it's still orphaned. - - Called from the WS-disconnect path. The grace window lets a transient - reconnect (or a ``session.resume`` that reattaches the transport) cancel - the reap by re-binding a live transport. Disabled when the grace is 0. - """ + """After a grace window, reap session ``sid`` iff it's still orphaned. Called from the + WS-disconnect path; a reconnect or ``session.resume`` cancels the reap by re-binding a + live transport. Disabled when the grace is 0.""" if _WS_ORPHAN_REAP_GRACE_S <= 0: return def _reap() -> None: - # Serialize the orphan re-check against session.resume (which re-binds a - # live transport under _session_resume_lock and would make this session - # non-orphaned). Claim teardown by popping under both lifecycle locks, - # then release the global resume lock before the slow finalization work. - # The dict mutation still happens under _sessions_lock — consistent - # with every other _sessions mutator - # (#39591: _reap previously popped under _session_resume_lock, giving no - # mutual exclusion against _init_session / _close_session_by_id, which - # guard with _sessions_lock). _sessions_lock is an RLock and the global - # ordering is always resume_lock -> sessions_lock, so nesting is safe. + # Serialize the re-check against session.resume (rebinds under _session_resume_lock). + # Claim teardown by popping under both locks, then release the resume lock before + # slow finalization. Ordering is always resume_lock -> sessions_lock (RLock). reschedule_delay = None interrupt_session = None session = None with _session_resume_lock: - # This Timer is running: drop its registration so a concurrent - # _cancel_ws_orphan_reap doesn't cancel a dead Timer object while - # a rescheduled one (registered below) is the live owner. + # Drop this Timer's registration so a concurrent _cancel_ws_orphan_reap can't + # cancel a dead Timer while a rescheduled one (registered below) is the owner. with _sessions_lock: _pending_ws_reaps.pop(sid, None) current = _sessions.get(sid) @@ -861,40 +628,27 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: if _session_has_active_delegations(sid, current): reschedule_delay = _WS_ORPHAN_REAP_GRACE_S elif current.get("running"): - if not current.get( - "_client_gone_interrupt_requested" - ) and _ws_orphan_turn_activity_is_fresh(current): - # Client-absent but actively producing (#98028/#100325): - # the turn keeps running detached (the sentinel transport - # already buffers emits) and the reaper re-checks each - # grace interval. Only a turn whose activity clock has - # gone stale — genuinely wedged, the case the interrupt - # was added for — falls through to the interrupt below. + if not current.get("_client_gone_interrupt_requested") and ( + _ws_orphan_turn_activity_is_fresh(current) + ): + # Client-absent but producing: keep running detached (the sentinel + # buffers emits) and re-check each grace interval. logger.debug( - "client_gone sid=%s action=defer (turn activity " - "fresh; stale threshold %.0fs)", - sid, - _WS_ORPHAN_ACTIVITY_STALE_S, + "client_gone sid=%s action=defer (turn activity fresh; stale threshold %.0fs)", + sid, _WS_ORPHAN_ACTIVITY_STALE_S, ) reschedule_delay = _WS_ORPHAN_REAP_GRACE_S else: - # Mid-turn detached sessions must never drop the single - # Timer (#85578): after the reconnect grace the turn is - # interrupted once, then the reap keeps polling until the - # normal turn-finalization path settles. + # Mid-turn detached sessions must never drop the single Timer: interrupt + # once after grace, then poll until turn-finalization settles. polls = int(current.get("_client_gone_interrupt_polls") or 0) + 1 current["_client_gone_interrupt_polls"] = polls if polls > _WS_ORPHAN_INTERRUPT_REAP_MAX_POLLS: - # The interrupted turn never settled inside the budget - # — force-reap rather than parking the session + a - # timer chain forever. Loud by design: this only fires - # when a turn is genuinely stuck past interrupt. + # Never settled inside the budget — force-reap rather than park forever. logger.error( "client_gone sid=%s: turn did not settle after %d " - "interrupt polls (%.0fs) — force-reaping detached " - "session", - sid, polls - 1, - (polls - 1) * _WS_ORPHAN_INTERRUPT_REAP_POLL_S, + "interrupt polls (%.0fs) — force-reaping detached session", + sid, polls - 1, (polls - 1) * _WS_ORPHAN_INTERRUPT_REAP_POLL_S, ) session = _pop_session_by_id(sid) else: @@ -907,64 +661,37 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: if interrupt_session is not None: try: - isolated = _interrupt_session_turn( - sid, - interrupt_session, - request_id=f"client-gone-{sid}", - ) - logger.info( - "client_gone sid=%s action=interrupt turn_isolation=%s", - sid, - isolated, - ) + isolated = _interrupt_session_turn(sid, interrupt_session, request_id=f"client-gone-{sid}") + logger.info("client_gone sid=%s action=interrupt turn_isolation=%s", sid, isolated) except Exception: logger.exception("client_gone interrupt failed sid=%s", sid) with _sessions_lock: if _sessions.get(sid) is interrupt_session: - interrupt_session.pop( - "_client_gone_interrupt_requested", None - ) + interrupt_session.pop("_client_gone_interrupt_requested", None) if reschedule_delay is not None: _schedule_ws_orphan_reap(sid, delay_s=reschedule_delay) return - if session is not None and session.get( - "_client_gone_interrupt_requested" - ): + if session is not None and session.get("_client_gone_interrupt_requested"): logger.info("client_gone sid=%s action=reap", sid) _teardown_popped_session(session, end_reason="ws_orphan_reap") - timer = threading.Timer( - _WS_ORPHAN_REAP_GRACE_S if delay_s is None else max(0.0, delay_s), - _reap, - ) + timer = threading.Timer(_WS_ORPHAN_REAP_GRACE_S if delay_s is None else max(0.0, delay_s), _reap) timer.daemon = True with _sessions_lock: prior = _pending_ws_reaps.pop(sid, None) _pending_ws_reaps[sid] = timer if prior is not None: - try: + with contextlib.suppress(Exception): prior.cancel() - except Exception: - pass timer.start() -def _close_sessions_for_transport( - transport, *, end_reason: str = "ws_disconnect" -) -> tuple[int, int]: - """On transport disconnect, reap the sessions that opted into - close_on_disconnect (sidecar/dashboard) immediately and re-point the rest - at the detached transport so later emits don't hit a dead socket. - - Non-flagged detached sessions are handed to the grace-windowed WS-orphan - reaper (``_schedule_ws_orphan_reap``): a quick reconnect / session.resume - that re-binds a live transport cancels the reap, otherwise the orphan is - torn down through the same idempotent ``_teardown_session`` path. This is - the single WS-disconnect teardown entry point — there is no second - independent reap loop in ``handle_ws``. - - Returns ``(reaped, detached)`` counts for disconnect-path observability.""" +def _close_sessions_for_transport(transport, *, end_reason: str = "ws_disconnect") -> tuple[int, int]: + """Single WS-disconnect teardown entry point: reap close_on_disconnect sessions + (sidecar/dashboard) immediately and re-point the rest at the detached transport so + later emits don't hit a dead socket; those go to the grace-windowed WS-orphan reaper. + Returns ``(reaped, detached)`` counts.""" with _sessions_lock: owned = [(sid, s) for sid, s in _sessions.items() if s.get("transport") is transport] reaped = 0 @@ -972,43 +699,29 @@ def _close_sessions_for_transport( for sid, session in owned: claimed_for_teardown = None should_schedule_reap = False - # A session.resume fast-path rebinds its live session while holding - # _session_resume_lock. Take that lock before re-checking the snapshot - # so a reconnect cannot move the transport between this check and the - # close/detach ownership claim. Keep the slow teardown below both locks. + # session.resume fast-path rebinds under _session_resume_lock: take it before + # re-checking so a reconnect can't move the transport between check and claim. with _session_resume_lock: with _sessions_lock: current = _sessions.get(sid) if current is not session: continue if current.get("transport") is not transport: - # The reconnect owns this session now. Drop only the old - # viewer registration; it must not affect the new owner. - viewers = current.get("viewers") - if viewers: - viewers.pop(transport, None) + # The reconnect owns this session now; drop only the old viewer registration. + (current.get("viewers") or {}).pop(transport, None) continue if current.get("close_on_disconnect"): claimed_for_teardown = _pop_session_by_id(sid) else: - # Point detached sessions at the drop sentinel (NOT real - # stdio) so _ws_session_is_orphaned recognizes them and - # the grace-reap can actually fire; a standalone - # `hermes --tui` keeps real _stdio. UNLESS another window - # still shows the session: multi-window pop-outs all - # register as viewers, so on disconnect re-bind the - # session to the most recent surviving viewer instead of - # stranding the original window on the sentinel (#83716). - viewers = current.get("viewers") - if viewers: - viewers.pop(transport, None) + # Point at the drop sentinel (NOT real stdio) so _ws_session_is_orphaned + # recognizes it; standalone `hermes --tui` keeps real _stdio. UNLESS + # another window (multi-window pop-out viewer) still shows the session: + # re-bind to the most recent surviving viewer instead of stranding it. + viewers = current.get("viewers") or {} + viewers.pop(transport, None) remaining = [ - (ts, viewer_transport) - for viewer_transport, ts in (viewers or {}).items() - if ( - viewer_transport is not transport - and not _transport_is_dead(viewer_transport) - ) + (ts, vt) for vt, ts in viewers.items() + if vt is not transport and not _transport_is_dead(vt) ] if remaining: remaining.sort(key=lambda item: item[0]) @@ -1022,10 +735,8 @@ def _close_sessions_for_transport( reaped += 1 elif should_schedule_reap: detached += 1 - try: + with contextlib.suppress(Exception): _schedule_ws_orphan_reap(sid) - except Exception: - pass return reaped, detached diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index f300e9389e..2cb14a14c3 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -7,111 +7,92 @@ method_ctx.bind_module), so they reference server.py globals bare. from __future__ import annotations +import contextlib + from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() +def _notif_locked_sessions(fn, default): + """Run ``fn(_sessions)`` under ``_sessions_lock``; ``default`` if that fails (the poller + thread must never crash on a lock/enumeration failure).""" + try: + with _sessions_lock: + return fn(_sessions) + except Exception: + return default + + +def _notif_current_keys(sid: str, session: dict) -> set: + return {str(session.get("session_key") or ""), _session_lookup_key(session, fallback=sid)} + + +def _notif_session_matches(s: dict, keys) -> bool: + return str(s.get("session_key") or "") in keys or _session_lookup_key(s, fallback="") in keys + + +def _notif_resolve_event_key(evt_key: str) -> str: + """Resolve a compression-rotated session key to its continuation tip (or itself).""" + try: + db = _get_db() + return (db.resolve_resume_session_id(evt_key) if db is not None else evt_key) or evt_key + except Exception: + return evt_key + + def _notification_event_belongs_elsewhere(sid: str, session: dict, evt: dict) -> bool: """True if ``evt`` is owned by a *different* live session. - Background completions carry the ``session_key`` of the session that started - the work. Async delegation completions from the desktop also carry - ``origin_ui_session_id``: the live TUI tab/window that commissioned them. - Since all desktop sessions share one process-wide completion queue, each - poller must skip events it doesn't own so a detached result surfaces in the - launching session, not whichever poller happened to dequeue first. - """ + Background completions carry the ``session_key`` of the session that started the work; async + delegation completions also carry ``origin_ui_session_id`` (the live TUI tab that commissioned + them). All desktop sessions share one process-wide completion queue, so each poller must skip + events it doesn't own or a detached result surfaces in whichever poller dequeued first.""" evt_ui_sid = str(evt.get("origin_ui_session_id") or "") if evt_ui_sid: if evt_ui_sid == str(sid or "") and not session.get("_finalized"): return False - try: - with _sessions_lock: - owner_live = evt_ui_sid in _sessions and not _sessions[evt_ui_sid].get("_finalized") - except Exception: - owner_live = False - if owner_live: + if _notif_locked_sessions(lambda ss: evt_ui_sid in ss and not ss[evt_ui_sid].get("_finalized"), False): return True - # If the exact UI tab is gone, fall through to durable session_key - # routing. That avoids wrong-session delivery while still allowing a - # resumed continuation with the same durable key/lineage to claim it. + # Exact UI tab gone: fall through to durable session_key routing so a + # resumed continuation with the same key/lineage can still claim it. evt_key = str(evt.get("session_key") or "") if not evt_key: return False - current_keys = { - str(session.get("session_key") or ""), - _session_lookup_key(session, fallback=sid), - } + current_keys = _notif_current_keys(sid, session) - # Compression can rotate AIAgent.session_id while the detached child is - # still running. Resolve the event's original key to its continuation tip so - # an event captured before or after compression still maps to the same live - # desktop session instead of becoming an orphan that any poller may consume. - resolved_key = evt_key - try: - db = _get_db() - if db is not None: - resolved_key = db.resolve_resume_session_id(evt_key) or evt_key - except Exception: - resolved_key = evt_key - - # If the key has a live continuation, prefer that continuation over the - # compressed parent. Otherwise a stale parent tab could consume the event - # before the real current conversation sees it. + # Compression can rotate AIAgent.session_id while the detached child is still running: map + # the event's original key to its continuation tip so it reaches the live session instead of + # becoming an orphan any poller may consume. A live continuation wins over the compressed + # parent, else a stale parent tab could consume the event before the current conversation. + resolved_key = _notif_resolve_event_key(evt_key) if resolved_key != evt_key: if resolved_key in current_keys: return False - try: - with _sessions_lock: - continuation_live = any( - not s.get("_finalized") - and ( - str(s.get("session_key") or "") == resolved_key - or _session_lookup_key(s, fallback="") == resolved_key - ) - for s in _sessions.values() - ) - except Exception: - continuation_live = False - if continuation_live: + if _notif_locked_sessions( + lambda ss: any(not s.get("_finalized") and _notif_session_matches(s, {resolved_key}) for s in ss.values()), + False, + ): return True if evt_key in current_keys: return False - try: - with _sessions_lock: - snapshot = list(_sessions.values()) - except Exception: - # If we can't safely enumerate live sessions, fail open so we don't - # crash the poller thread or drop the event. - return False + snapshot = _notif_locked_sessions(lambda ss: list(ss.values()), None) + if snapshot is None: + return False # can't enumerate: fail open rather than drop the event - return any( - s is not session - and not s.get("_finalized") - and ( - str(s.get("session_key") or "") in {evt_key, resolved_key} - or _session_lookup_key(s, fallback="") in {evt_key, resolved_key} - ) - for s in snapshot - ) + keys = {evt_key, resolved_key} + return any(s is not session and not s.get("_finalized") and _notif_session_matches(s, keys) for s in snapshot) def _session_owns_notification_event(sid: str, session: dict, evt: dict) -> bool: - """True iff *this* session PROVABLY owns ``evt``. - - Positive ownership — the mirror of ``_notification_event_belongs_elsewhere`` - minus its orphan-adoption fallback. An event owns-matches when its - ``origin_ui_session_id`` is this live session, or its ``session_key`` - (raw or resolved through the compression chain) matches this session's - key/lineage. Used as the fail-closed gate for every addressed notification: - "not provably elsewhere" is NOT good enough to inject a payload into this - chat (#55578). - """ + """True iff *this* session PROVABLY owns ``evt``: positive mirror of + ``_notification_event_belongs_elsewhere`` minus its orphan-adoption fallback (UI origin is + this live session, or ``session_key`` raw/compression-resolved matches this session). Fail- + closed gate for every addressed notification — "not provably elsewhere" is NOT enough.""" if session.get("_finalized"): return False if str(evt.get("origin_ui_session_id") or "") == str(sid or ""): @@ -119,63 +100,33 @@ def _session_owns_notification_event(sid: str, session: dict, evt: dict) -> bool evt_key = str(evt.get("session_key") or "") if not evt_key: return False - current_keys = { - str(session.get("session_key") or ""), - _session_lookup_key(session, fallback=sid), - } + current_keys = _notif_current_keys(sid, session) if evt_key in current_keys: return True - try: - db = _get_db() - resolved_key = ( - db.resolve_resume_session_id(evt_key) if db is not None else evt_key - ) or evt_key - except Exception: - resolved_key = evt_key - return resolved_key in current_keys + return _notif_resolve_event_key(evt_key) in current_keys def _notification_event_requires_owner(evt: dict) -> bool: """Whether ``evt`` must be positively claimed before TUI delivery.""" return evt.get("type") == "async_delegation" or bool( - str(evt.get("origin_ui_session_id") or "") - or str(evt.get("session_key") or "") + str(evt.get("origin_ui_session_id") or "") or str(evt.get("session_key") or "") ) def _notification_event_dedup_key(evt: dict) -> tuple: - """Return the UI-emission identity for a process notification event. - - Completion events are terminal notifications for a background process, so - they remain one-shot per process session. Watch-match events are not - terminal: a single background process can legitimately match the same or - different patterns many times, so include event-specific content to avoid - suppressing later distinct matches from the same process. - """ + """UI-emission identity for a process notification event. Completions are terminal (one-shot + per process session); watch events are not — one process can match patterns many times, so + include event content to avoid suppressing later distinct matches.""" evt_type = evt.get("type", "completion") evt_sid = evt.get("session_id", "") if evt_type == "watch_match": - return ( - evt_sid, - evt_type, - evt.get("command", ""), - evt.get("pattern", ""), - evt.get("output", ""), - evt.get("suppressed", 0), - evt.get("message_id", ""), - ) + return (evt_sid, evt_type, evt.get("command", ""), evt.get("pattern", ""), evt.get("output", ""), + evt.get("suppressed", 0), evt.get("message_id", "")) if evt_type.startswith("watch_overflow_") or evt_type == "watch_disabled": - return ( - evt_sid, - evt_type, - evt.get("command", ""), - evt.get("message", ""), - evt.get("suppressed", 0), - ) + return (evt_sid, evt_type, evt.get("command", ""), evt.get("message", ""), evt.get("suppressed", 0)) if evt_type == "async_delegation": - # Async-delegation completions have no process session_id; without - # this the fallthrough keys every one as ("", "async_delegation") - # and the second completion's status update is suppressed forever. + # No process session_id: without this every completion keys as + # ("", "async_delegation") and the second one is suppressed forever. return (evt.get("delegation_id", ""), evt_type) return (evt_sid, evt_type) @@ -183,23 +134,25 @@ def _notification_event_dedup_key(evt: dict) -> tuple: # Mirror gateway/kanban_watchers.py TERMINAL_KINDS: claim silent kinds too so # the cursor advances past them and they can't wedge a later completed/blocked # event behind an unclaimed row. -_KANBAN_NOTIFY_KINDS = ( - "completed", "blocked", "gave_up", "crashed", "timed_out", - "status", "archived", "unblocked", -) +_KANBAN_NOTIFY_KINDS = ("completed", "blocked", "gave_up", "crashed", "timed_out", "status", "archived", "unblocked") _KANBAN_SILENT_KINDS = frozenset({"archived", "unblocked"}) _KANBAN_POLL_SECONDS = 5.0 _LOOP_POLL_SECONDS = 5.0 -def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: - """Fire a due /loop wakeup for an idle TUI/Desktop/dashboard session. +def _notif_release_turn(session: dict) -> None: + with session["history_lock"]: + session["running"] = False - Called from the per-session notification poller thread on a coarse - cadence. Claims the session under history_lock (running=True) before - dispatching so a racing user prompt wins cleanly. The post-turn hook - in the turn dispatcher completes the tick. - """ + +def _notif_log_failure(what: str, exc: BaseException) -> None: + print(f"[tui_gateway] {what}: {type(exc).__name__}: {exc}", file=sys.stderr) + + +def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: + """Fire a due /loop wakeup for an idle TUI/Desktop/dashboard session (per-session poller, + coarse cadence). Claims the session under history_lock (running=True) before dispatching so + a racing user prompt wins cleanly; the turn dispatcher's post-turn hook completes the tick.""" try: from hermes_cli.loops import LoopManager, goal_blocks_loop_tick except Exception: @@ -209,9 +162,7 @@ def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: if not sid_key: return mgr = LoopManager(session_id=sid_key) - if not mgr.is_due(): - return - if goal_blocks_loop_tick(sid_key): + if not mgr.is_due() or goal_blocks_loop_tick(sid_key): return with session["history_lock"]: @@ -221,41 +172,30 @@ def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: wakeup = mgr.fire_tick() if not wakeup: - with session["history_lock"]: - session["running"] = False + _notif_release_turn(session) return tick_no = mgr.state.ticks_fired if mgr.state else "?" rid = f"__loop__{int(time.time() * 1000)}" try: - _emit( - "status.update", - sid, - {"kind": "loop", "text": f"↻ /loop wakeup #{tick_no} firing…"}, - ) + _emit("status.update", sid, {"kind": "loop", "text": f"↻ /loop wakeup #{tick_no} firing…"}) if wakeup.lstrip().startswith("/"): - # Slash-command loop: route through the slash pipeline instead of - # the model. No model reply to evaluate — complete immediately. - with session["history_lock"]: - session["running"] = False + # Slash-command loop: route through the slash pipeline, not the + # model. No model reply to evaluate — complete immediately. + _notif_release_turn(session) try: parts = wakeup.lstrip()[1:].split(None, 1) resp = _methods["command.dispatch"]( rid, - { - "name": parts[0] if parts else "", - "arg": parts[1] if len(parts) > 1 else "", - "session_id": sid, - }, + {"name": parts[0] if parts else "", "arg": parts[1] if len(parts) > 1 else "", "session_id": sid}, ) payload = (resp or {}).get("result") or {} out = str(payload.get("output") or "").strip() if out: _emit("status.update", sid, {"kind": "loop", "text": out}) if payload.get("type") == "send" and payload.get("message"): - # The command resolves to a prompt (skill command etc.) — - # run it as a normal turn; the post-turn hook completes - # the tick. + # Command resolves to a prompt (skill command etc.) — run it + # as a normal turn; the post-turn hook completes the tick. with session["history_lock"]: if session.get("running"): mgr.abandon_tick() @@ -273,26 +213,15 @@ def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: _emit("message.start", sid) _run_prompt_submit(rid, sid, session, wakeup) except Exception as exc: - print( - f"[tui_gateway] loop wakeup dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False - try: + _notif_log_failure("loop wakeup dispatch failed", exc) + _notif_release_turn(session) + with contextlib.suppress(Exception): mgr.abandon_tick() - except Exception: - pass def _format_kanban_event_text(sub: dict, task, ev, board_slug: str) -> Optional[str]: - """Single-line notification text for one kanban event. - - Wording mirrors the gateway notifier (gateway/kanban_watchers.py) so a - task completion reads the same in the TUI as it does on Telegram. - Returns None for kinds that are claimed but intentionally silent. - """ + """Single-line notification text for one kanban event; wording mirrors + gateway/kanban_watchers.py so it reads the same as on Telegram. None for silent kinds.""" kind = getattr(ev, "kind", "") if not kind or kind in _KANBAN_SILENT_KINDS: return None @@ -321,11 +250,10 @@ def _format_kanban_event_text(sub: dict, task, ev, board_slug: str) -> Optional[ if kind == "crashed": return f"✖ {board_tag}{tag}Kanban {task_id} worker crashed (pid gone); dispatcher will retry" if kind == "timed_out": - limit = 0 try: limit = int(payload.get("limit_seconds") or 0) except (TypeError, ValueError): - pass + limit = 0 return f"⏱ {board_tag}{tag}Kanban {task_id} timed out (max_runtime={limit}s); will retry" if kind == "status": return f"🔄 {board_tag}{tag}Kanban {task_id} → {payload.get('status') or ''}" @@ -335,17 +263,11 @@ def _format_kanban_event_text(sub: dict, task, ev, board_slug: str) -> Optional[ def _collect_kanban_notifications(session: dict) -> list: """Claim unseen terminal kanban events for this TUI session's subscriptions. - ``kanban_create`` auto-subscribes TUI/desktop sessions with - ``platform="tui"`` and ``chat_id=HERMES_SESSION_KEY`` (see - tools/kanban_tools.py ``_maybe_auto_subscribe``). The gateway notifier - can't deliver those — there is no "tui" messaging adapter — so this - poller is the delivery path for them (issue #59890). Uses the same - atomic cursor-claim (``claim_unseen_events_for_sub``) as the gateway - notifier, so a subscription is delivered exactly once even if a gateway - and a TUI poll the same board DB. - - Returns the list of formatted notification texts (may be empty). - """ + ``kanban_create`` auto-subscribes TUI/desktop sessions with ``platform="tui"`` and + ``chat_id=HERMES_SESSION_KEY``; there is no "tui" messaging adapter, so this poller is the + delivery path. Same atomic cursor-claim (``claim_unseen_events_for_sub``) as the gateway + notifier, so a sub is delivered exactly once even if a gateway and a TUI poll the same board + DB. Returns formatted notification texts (may be empty).""" session_key = str(session.get("session_key") or "") if not session_key or session.get("_finalized"): return [] @@ -362,35 +284,25 @@ def _collect_kanban_notifications(session: dict) -> list: except Exception: return [] # Poll each resolved DB path once — multiple slugs can point at the same - # DB when HERMES_KANBAN_DB pins the board path (same guard as the gateway - # notifier). + # DB when HERMES_KANBAN_DB pins the board path (same guard as the gateway). seen_db_paths: set = set() for board_meta in boards: slug = (board_meta or {}).get("slug") or _kb.DEFAULT_BOARD db_path = (board_meta or {}).get("db_path") try: - resolved = ( - str(Path(db_path).expanduser().resolve()) - if db_path else str(_kb.kanban_db_path(slug).resolve()) - ) + resolved = str(Path(db_path).expanduser().resolve() if db_path else _kb.kanban_db_path(slug).resolve()) except Exception: resolved = f"slug:{slug}" if resolved in seen_db_paths: continue seen_db_paths.add(resolved) - # A poller runs per live TUI/Desktop session. Avoid opening this board - # writable unless it has a subscription owned by this exact session; - # subscriptions for gateways or other sessions are not actionable here. + # One poller per live session: don't open this board writable unless it + # has a subscription owned by this exact session. If the read-only probe + # fails (locked/corrupt DB), preserve delivery and fall through. try: - if _kb.count_notify_subs( - board=slug, - platform="tui", - chat_id=session_key, - ) == 0: + if _kb.count_notify_subs(board=slug, platform="tui", chat_id=session_key) == 0: continue except Exception: - # Preserve delivery if the read-only probe cannot inspect a - # locked, corrupt, or otherwise unusual database. pass try: conn = _kb.connect(board=slug) @@ -402,18 +314,13 @@ def _collect_kanban_notifications(session: dict) -> list: except Exception: continue for sub in subs: - if (sub.get("platform") or "").lower() != "tui": + if (sub.get("platform") or "").lower() != "tui" or sub.get("chat_id") != session_key: continue - if sub.get("chat_id") != session_key: - continue - _old, _new, events = _kb.claim_unseen_events_for_sub( - conn, - task_id=sub["task_id"], - platform=sub["platform"], - chat_id=sub["chat_id"], + sub_ident = dict( + task_id=sub["task_id"], platform=sub["platform"], chat_id=sub["chat_id"], thread_id=sub.get("thread_id") or "", - kinds=_KANBAN_NOTIFY_KINDS, ) + _old, _new, events = _kb.claim_unseen_events_for_sub(conn, kinds=_KANBAN_NOTIFY_KINDS, **sub_ident) if not events: continue task = _kb.get_task(conn, sub["task_id"]) @@ -421,157 +328,142 @@ def _collect_kanban_notifications(session: dict) -> list: text = _format_kanban_event_text(sub, task, ev, slug) if text: texts.append(text) - # Unsubscribe only on archive. ``done`` is reversible in - # review/controller flows, so retaining the subscription lets - # a later reopen notify the same originating TUI/Desktop - # session. The claimed cursor prevents historical replay. + # Unsubscribe only on archive: ``done`` is reversible in review/controller flows, + # so keeping the sub lets a later reopen notify the same session. The claimed + # cursor prevents replay. if task and getattr(task, "status", "") == "archived": - try: - _kb.remove_notify_sub( - conn, - task_id=sub["task_id"], - platform=sub["platform"], - chat_id=sub["chat_id"], - thread_id=sub.get("thread_id") or "", - ) - except Exception: - pass + with contextlib.suppress(Exception): + _kb.remove_notify_sub(conn, **sub_ident) finally: conn.close() return texts -def _notification_poller_loop( - stop_event: threading.Event, sid: str, session: dict -) -> None: - """Poll completion_queue and dispatch notifications autonomously. +def _notif_poll_kanban(sid: str, session: dict) -> None: + """One kanban poll: emit new texts, buffer them, and run the buffered batch as a turn if idle. + Events are cursor-claimed (never re-queued), so they are buffered until the session is idle + instead of dropping the agent turn.""" + try: + _kanban_texts = _collect_kanban_notifications(session) + except Exception as _kb_exc: + _notif_log_failure("kanban notification poll failed", _kb_exc) + _kanban_texts = [] + if _kanban_texts: + for _kb_text in _kanban_texts: + _emit("status.update", sid, {"kind": "process", "text": _kb_text}) + session.setdefault("_kanban_pending", []).extend(_kanban_texts) + _pending = session.get("_kanban_pending") or [] + if not _pending: + return + _batch: list = [] + with session["history_lock"]: + if not session.get("running"): + session["running"] = True + _batch = list(_pending) + session["_kanban_pending"] = [] + if _batch: + rid = f"__notif__{int(time.time() * 1000)}" + try: + _emit("message.start", sid) + _run_prompt_submit(rid, sid, session, "\n".join(_batch)) + except Exception as exc: + _notif_log_failure("kanban notification dispatch failed", exc) + _notif_release_turn(session) - Runs in a daemon thread started by _init_session(). Emits a - status.update (kind=process) for user visibility, then chains an - agent turn via _run_prompt_submit if the session is idle. - The completion_queue is process-global. In multi-session Desktop each - poller requeues events owned by another live session and drops addressed - events whose owner is gone; ownerless legacy notifications remain global. +def _notif_render_event(sid: str, evt: dict, emitted: set, process_registry, format_process_notification) -> Optional[str]: + """Format ``evt`` and emit its status.update once; None means skip the event (consumed + completion or unformattable). Re-queued completions would otherwise re-emit every 0.5s while + the session is busy, while distinct watch_match events from one process must stay visible.""" + if evt.get("type") == "completion" and process_registry.is_completion_consumed(evt.get("session_id", "")): + return None + text = format_process_notification(evt) + if not text: + return None + dedup_key = _notification_event_dedup_key(evt) + if dedup_key not in emitted: + _emit("status.update", sid, {"kind": "process", "text": text}) + emitted.add(dedup_key) + return text - Also polls ``kanban_notify_subs`` every ``_KANBAN_POLL_SECONDS`` for this - session's TUI kanban subscriptions and delivers terminal task events the - same way (status.update + agent turn) — the delivery path - tools/kanban_tools.py documents for platform="tui" rows (issue #59890). - """ + +def _notif_dispatch_event(sid: str, session: dict, evt: dict, text: str) -> None: + """Run the claimed (running=True) agent turn for one notification event.""" + from tools.async_delegation import claim_event_delivery, complete_event_delivery, release_event_delivery + + rid = f"__notif__{int(time.time() * 1000)}" + claim = claim_event_delivery(evt, "tui-poller") + if claim is None: + return + try: + _emit("message.start", sid) + if evt.get("type") == "async_delegation": + _run_prompt_submit( + rid, sid, session, text, display_kind="async_delegation_complete", + display_metadata=_async_delegation_display_metadata(evt), + ) + else: + _run_prompt_submit(rid, sid, session, text) + complete_event_delivery(evt, claim) + except Exception as exc: + release_event_delivery(evt, claim) + _notif_log_failure("notification poller dispatch failed", exc) + _notif_release_turn(session) + + +def _notification_poller_loop(stop_event: threading.Event, sid: str, session: dict) -> None: + """Poll completion_queue and dispatch notifications autonomously (daemon thread started by + _init_session()): emit a status.update (kind=process), then chain an agent turn via + _run_prompt_submit if the session is idle. The queue is process-global: each poller requeues + events owned by another live session and drops addressed events whose owner is gone; + ownerless legacy notifications remain global. Also polls ``kanban_notify_subs`` every + ``_KANBAN_POLL_SECONDS`` — the delivery path for platform="tui" rows.""" from tools.process_registry import process_registry, format_process_notification - _emitted = set() # dedup re-queued events so same completion isn't emitted 50 times while session is busy + _emitted = set() # dedup re-queued events so one completion isn't emitted 50 times while busy _last_kanban_poll = 0.0 _last_loop_poll = 0.0 while not stop_event.is_set() and not session.get("_finalized"): _now = time.monotonic() - # ── /loop wakeup driver ────────────────────────────────────── - # Fire a due /loop tick for THIS session while it's idle. Same - # claim-under-lock pattern as the kanban dispatch below. Active - # non-parked /goal owns the idle boundary and defers the tick. + # /loop wakeup driver: fire a due tick for THIS session while idle (same claim-under-lock + # as kanban dispatch). An active non-parked /goal owns the idle boundary and defers it. if _now - _last_loop_poll >= _LOOP_POLL_SECONDS: _last_loop_poll = _now try: _maybe_fire_tui_loop_tick(sid, session) except Exception as _loop_exc: - print( - f"[tui_gateway] loop wakeup poll failed: " - f"{type(_loop_exc).__name__}: {_loop_exc}", - file=sys.stderr, - ) + _notif_log_failure("loop wakeup poll failed", _loop_exc) if _now - _last_kanban_poll >= _KANBAN_POLL_SECONDS: _last_kanban_poll = _now - try: - _kanban_texts = _collect_kanban_notifications(session) - except Exception as _kb_exc: - print( - f"[tui_gateway] kanban notification poll failed: " - f"{type(_kb_exc).__name__}: {_kb_exc}", - file=sys.stderr, - ) - _kanban_texts = [] - if _kanban_texts: - for _kb_text in _kanban_texts: - _emit("status.update", sid, {"kind": "process", "text": _kb_text}) - # Events are cursor-claimed (never re-queued), so buffer them - # until the session is idle instead of dropping the agent turn. - session.setdefault("_kanban_pending", []).extend(_kanban_texts) - _pending = session.get("_kanban_pending") or [] - if _pending: - _batch: list = [] - with session["history_lock"]: - if not session.get("running"): - session["running"] = True - _batch = list(_pending) - session["_kanban_pending"] = [] - if _batch: - rid = f"__notif__{int(time.time() * 1000)}" - try: - _emit("message.start", sid) - _run_prompt_submit(rid, sid, session, "\n".join(_batch)) - except Exception as exc: - print( - f"[tui_gateway] kanban notification dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False + _notif_poll_kanban(sid, session) try: evt = process_registry.completion_queue.get(timeout=0.5) except Exception: continue - # Multiple desktop sessions share this one process-wide queue. Only - # consume events that belong to *this* session — otherwise a background - # process started in session A would surface its completion in whichever - # session's poller happened to wake first (Ben's "reported in a - # different session" bug). Leave foreign events for their owner. + # Leave foreign events for their owner — otherwise a process started in + # session A surfaces its completion in whichever poller wakes first. if _notification_event_belongs_elsewhere(sid, session, evt): process_registry.completion_queue.put(evt) time.sleep(0.1) continue - # What reaches here is not owned by another LIVE session. Addressed - # events still require positive proof before injection: exact UI origin, - # direct durable key, or compression lineage. If none proves ownership, - # the event is orphaned and must not be adopted by this chat. Truly - # ownerless ordinary notifications retain legacy global delivery. - requires_owner = _notification_event_requires_owner(evt) - if requires_owner and not _session_owns_notification_event(sid, session, evt): - log = ( - logger.warning - if evt.get("type") == "async_delegation" - else logger.debug - ) + # Not owned by another LIVE session, but addressed events still need positive proof + # (exact UI origin, direct durable key, or compression lineage) — an orphan must not be + # adopted by this chat. Truly ownerless ordinary notifications keep legacy global delivery. + if _notification_event_requires_owner(evt) and not _session_owns_notification_event(sid, session, evt): + log = logger.warning if evt.get("type") == "async_delegation" else logger.debug log( - "Dropping unowned %s notification (origin=%r key=%r) instead " - "of delivering to session %s", - evt.get("type", "completion"), - str(evt.get("origin_ui_session_id") or ""), - str(evt.get("session_key") or ""), - sid, + "Dropping unowned %s notification (origin=%r key=%r) instead of delivering to session %s", + evt.get("type", "completion"), str(evt.get("origin_ui_session_id") or ""), + str(evt.get("session_key") or ""), sid, ) continue - _evt_sid = evt.get("session_id", "") - if evt.get("type") == "completion" and process_registry.is_completion_consumed(_evt_sid): - continue - - text = format_process_notification(evt) + text = _notif_render_event(sid, evt, _emitted, process_registry, format_process_notification) if not text: continue - # Only emit the same notification identity to TUI once — re-queued - # completions get re-emitted every 0.5s otherwise when session is busy, - # while distinct watch_match events from the same process must remain - # visible independently. - _dedup_key = _notification_event_dedup_key(evt) - if _dedup_key not in _emitted: - _emit("status.update", sid, {"kind": "process", "text": text}) - _emitted.add(_dedup_key) - _requeued = False with session["history_lock"]: if session.get("running"): @@ -580,47 +472,16 @@ def _notification_poller_loop( else: session["running"] = True if _requeued: - # Back off before re-polling: the re-queued event keeps the queue - # non-empty, so without a sleep this loop spins at full speed - # (100% CPU, GIL churn) for as long as the session stays busy. + # Back off: the re-queued event keeps the queue non-empty, so + # without a sleep this loop spins at 100% CPU while busy. time.sleep(0.25) continue - rid = f"__notif__{int(time.time() * 1000)}" - from tools.async_delegation import ( - claim_event_delivery, complete_event_delivery, release_event_delivery, - ) - _claim = claim_event_delivery(evt, "tui-poller") - if _claim is None: - continue - try: - _emit("message.start", sid) - if evt.get("type") == "async_delegation": - _run_prompt_submit( - rid, - sid, - session, - text, - display_kind="async_delegation_complete", - display_metadata=_async_delegation_display_metadata(evt), - ) - else: - _run_prompt_submit(rid, sid, session, text) - complete_event_delivery(evt, _claim) - except Exception as exc: - release_event_delivery(evt, _claim) - print( - f"[tui_gateway] notification poller dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False + _notif_dispatch_event(sid, session, evt, text) - # Drain any remaining events after stop signal (process all pending - # before exiting so nothing is lost on shutdown). Events owned by other - # live sessions are set aside and re-queued so their poller still sees them. - # Orphaned events (owner gone) are dropped — same guard as the main loop. + # Drain remaining events after the stop signal so nothing is lost on shutdown. Other live + # sessions' events are set aside and re-queued; orphaned events (owner gone) are dropped — + # same guard as the main loop, except orphaned delegation payloads are deferred for a resume. deferred: list = [] while not process_registry.completion_queue.empty(): try: @@ -630,70 +491,27 @@ def _notification_poller_loop( if _notification_event_belongs_elsewhere(sid, session, evt): deferred.append(evt) continue - # Same positive-proof rule as the live loop. Preserve the existing - # shutdown behavior for orphaned delegation payloads by deferring them - # for a later resume; ordinary addressed orphans are dropped. - requires_owner = _notification_event_requires_owner(evt) - if requires_owner and not _session_owns_notification_event(sid, session, evt): + if _notification_event_requires_owner(evt) and not _session_owns_notification_event(sid, session, evt): if evt.get("type") == "async_delegation": deferred.append(evt) else: logger.debug( - "Dropping unowned %s notification during shutdown drain " - "(origin=%r key=%r)", - evt.get("type", "completion"), - str(evt.get("origin_ui_session_id") or ""), + "Dropping unowned %s notification during shutdown drain (origin=%r key=%r)", + evt.get("type", "completion"), str(evt.get("origin_ui_session_id") or ""), str(evt.get("session_key") or ""), ) continue - _evt_sid = evt.get("session_id", "") - if evt.get("type") == "completion" and process_registry.is_completion_consumed(_evt_sid): - continue - text = format_process_notification(evt) + text = _notif_render_event(sid, evt, _emitted, process_registry, format_process_notification) if not text: continue - _dedup_key = _notification_event_dedup_key(evt) - if _dedup_key not in _emitted: - _emit("status.update", sid, {"kind": "process", "text": text}) - _emitted.add(_dedup_key) - with session["history_lock"]: if session.get("running"): process_registry.completion_queue.put(evt) break session["running"] = True - rid = f"__notif__{int(time.time() * 1000)}" - from tools.async_delegation import ( - claim_event_delivery, complete_event_delivery, release_event_delivery, - ) - _claim = claim_event_delivery(evt, "tui-poller") - if _claim is None: - continue - try: - _emit("message.start", sid) - if evt.get("type") == "async_delegation": - _run_prompt_submit( - rid, - sid, - session, - text, - display_kind="async_delegation_complete", - display_metadata=_async_delegation_display_metadata(evt), - ) - else: - _run_prompt_submit(rid, sid, session, text) - complete_event_delivery(evt, _claim) - except Exception as exc: - release_event_delivery(evt, _claim) - print( - f"[tui_gateway] notification poller dispatch failed: " - f"{type(exc).__name__}: {exc}", - file=sys.stderr, - ) - with session["history_lock"]: - session["running"] = False + _notif_dispatch_event(sid, session, evt, text) # Hand any other sessions' events back to the shared queue. for evt in deferred: @@ -703,23 +521,13 @@ def _notification_poller_loop( def _async_delegation_display_metadata(evt: dict) -> dict: """Build display-only metadata before the completion event is formatted.""" raw_results = evt.get("results") - results: list[dict] = [ - result for result in raw_results if isinstance(result, dict) - ] if isinstance(raw_results, list) else [] + results: list[dict] = [r for r in raw_results if isinstance(r, dict)] if isinstance(raw_results, list) else [] task_count = len(results) or 1 - completed_count = sum( - 1 for result in results - if result.get("status") in {"completed", "success"} - ) - failed_count = sum( - 1 for result in results - if result.get("status") in {"failed", "error"} - ) + completed_count = sum(1 for r in results if r.get("status") in {"completed", "success"}) + failed_count = sum(1 for r in results if r.get("status") in {"failed", "error"}) metadata = { - "delegation_id": str(evt.get("delegation_id") or ""), - "task_count": task_count, - "completed_count": completed_count or task_count - failed_count, - "failed_count": failed_count, + "delegation_id": str(evt.get("delegation_id") or ""), "task_count": task_count, + "completed_count": completed_count or task_count - failed_count, "failed_count": failed_count, } duration = evt.get("total_duration_seconds") or evt.get("duration_seconds") if isinstance(duration, (int, float)): @@ -728,14 +536,10 @@ def _async_delegation_display_metadata(evt: dict) -> dict: def _wire_agent_terminal_output() -> None: - """Idempotently route background-process output (and tab-close requests) to - the desktop, keyed by process id. Read-only agent terminal tabs stream - `agent.terminal.output` chunks live instead of polling the output tail, and - `process_registry.request_close_terminal` emits `terminal.close` so the agent - can drop a tab without killing the process. Events are routed to the window - that owns the process (its gateway session); `_emit`/`write_json` is - `_stdout_lock`-guarded, so calling it from the registry's reader threads is - safe.""" + """Idempotently route background-process output (`agent.terminal.output` chunks) and + `process_registry.request_close_terminal` (`terminal.close`, drops a tab without killing the + process) to the window owning the process (its gateway session), keyed by process id. + `_emit` is `_stdout_lock`-guarded, so the registry's reader threads may call it.""" from tools.process_registry import process_registry has_output_sink = getattr(process_registry, "on_output", None) is not None @@ -754,11 +558,7 @@ def _wire_agent_terminal_output() -> None: return "" def _emit_agent_terminal_output(session, chunk): - _emit( - "agent.terminal.output", - _owner_sid_for_process(session), - {"process_id": session.id, "chunk": chunk}, - ) + _emit("agent.terminal.output", _owner_sid_for_process(session), {"process_id": session.id, "chunk": chunk}) def _emit_agent_terminal_close(session, process_id): # session may be None (process already finished/pruned) — the tab can @@ -777,11 +577,8 @@ _desktop_ui_wired = False def _wire_desktop_ui() -> None: """Bridge desktop-only tools (open_preview, close_preview, focus_pane) to renderer events. - - Idempotent. The tool hands back the turn's ``HERMES_UI_SESSION_ID`` as - ``sid`` so the event routes to the window that asked (``_emit`` / - ``write_json`` is ``_stdout_lock``-guarded, so calling it from the tool's - thread is safe).""" + Idempotent. The tool hands back the turn's ``HERMES_UI_SESSION_ID`` as ``sid`` so the event + routes to the window that asked (``_emit`` is ``_stdout_lock``-guarded; tool thread may call it).""" global _desktop_ui_wired if _desktop_ui_wired: return @@ -794,9 +591,9 @@ def _wire_desktop_ui() -> None: _desktop_ui_wired = True -# (stop_event, thread) for every poller ever started in this process. -# Pruned of dead threads on each spawn; consumed by test teardowns to reap -# leaked pollers (see _start_notification_poller). +# (stop_event, thread) for every poller started in this process, pruned of dead threads on each +# spawn. Test teardowns reap leaked pollers through it: an unjoined poller steals events off the +# process-global completion_queue mid-assertion in a LATER test. _notification_pollers: list = [] @@ -805,21 +602,11 @@ def _start_notification_poller(sid: str, session: dict) -> threading.Event: _wire_agent_terminal_output() _wire_desktop_ui() stop = threading.Event() + # Thread name is greppable for debuggers/test teardowns. t = threading.Thread( - target=_notification_poller_loop, - args=(stop, sid, session), - daemon=True, - # Stable, greppable name for debuggers and test teardowns. - name=f"tui-notif-poller-{sid}", + target=_notification_poller_loop, args=(stop, sid, session), daemon=True, name=f"tui-notif-poller-{sid}" ) - # Registry of (stop, thread) pairs so test teardowns can reap pollers - # leaked by session.init/create tests — an unjoined poller steals - # events off the process-global completion_queue mid-assertion in a - # LATER test (flaky test_run_prompt_submit_requeues_all_unstarted_...). - # Bounded: entries for dead threads are pruned on each spawn. - _notification_pollers[:] = [ - (s, th) for (s, th) in _notification_pollers if th.is_alive() - ] + _notification_pollers[:] = [(s, th) for (s, th) in _notification_pollers if th.is_alive()] _notification_pollers.append((stop, t)) t.start() return stop @@ -835,14 +622,10 @@ def _hud_surface_note(session: dict) -> str: def _prepend_note(run_message: Any, note: str) -> Any: - """Prefix a per-turn note onto the MODEL INPUT, leaving the prompt alone. - - Everything the model needs to know about the turn but the user did not - type — an interrupted reply, reactions, the surface they typed into — - arrives this way. persist_user_message keeps the clean prompt, so no - scaffolding reaches the transcript, and annotating the NEW turn never - rewrites an already-sent message, so the cached prefix survives. - """ + """Prefix a per-turn note onto the MODEL INPUT, leaving the prompt alone. Everything the model + must know about the turn that the user did not type (interrupted reply, reactions, surface) + arrives this way; persist_user_message keeps the clean prompt, so no scaffolding reaches the + transcript, and annotating the NEW turn never rewrites a sent message — cached prefix survives.""" if not note: return run_message if isinstance(run_message, str): diff --git a/tui_gateway/session_reaper.py b/tui_gateway/session_reaper.py index 20037d8501..a2ebb036fc 100644 --- a/tui_gateway/session_reaper.py +++ b/tui_gateway/session_reaper.py @@ -6,6 +6,7 @@ method_ctx.bind_module), so they reference server.py globals bare. from __future__ import annotations +import contextlib import threading from tui_gateway._env import env_float @@ -15,31 +16,18 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# Knobs (_SESSION_TTL_S, _REAPER_SCAN_S, _EXIT_FLUSH_BUDGET_S, -# _INCREMENTAL_FLUSH_INTERVAL_S) live in server.py next to _start_idle_reaper. +# Knobs (_SESSION_TTL_S, _REAPER_SCAN_S, _EXIT_FLUSH_BUDGET_S, _INCREMENTAL_FLUSH_INTERVAL_S) live in server.py. -# ── Flush-on-kill + periodic incremental flush (#94724 item 2) ─────────── -# A `hermes serve` killed mid-update used to lose every un-flushed in-memory -# session: the next RPC failed with "session-scoped RPC rejected: not in -# memory (detached/reaped runtime)" and NO store held the transcript. #95576 -# made serves survive *future* updates; this closes the kill path itself: -# (a) SIGTERM/SIGINT run a bounded, best-effort flush of in-memory session -# transcripts to state.db BEFORE the normal shutdown path, chained to -# whatever handler was installed before (uvicorn's included); -# (b) the idle-reaper scan piggybacks a periodic incremental flush so even -# a SIGKILL loses at most one flush interval. +# ── Flush-on-kill + periodic incremental flush ─────────────────────────── +# (a) SIGTERM/SIGINT run a bounded flush to state.db BEFORE normal shutdown, chained to the prior +# handler; (b) the idle-reaper scan piggybacks an incremental flush so a SIGKILL loses at most one interval. def _flush_session_messages(session: dict | None) -> bool: - """Best-effort durable flush of one session's in-memory transcript. - - Rides ``agent._persist_session`` — the same marker-deduped persist - contract ``_finalize_session`` uses (#13121) — so repeated calls only - write genuinely-unflushed messages and never duplicate durable rows. - """ - if not session: - return False - agent = session.get("agent") + """Best-effort durable flush of one session's transcript via ``agent._persist_session`` + (same marker-deduped contract as ``_finalize_session``): repeated calls only write + genuinely-unflushed messages, never duplicate rows.""" + agent = session.get("agent") if session else None if agent is None or not hasattr(agent, "_persist_session"): return False snapshot = getattr(agent, "_session_messages", None) @@ -53,25 +41,22 @@ def _flush_session_messages(session: dict | None) -> bool: return False -def _flush_dirty_sessions(now: float | None = None) -> int: - """Periodic incremental flush, driven by the idle-reaper scan. +def _reaper_session_snapshot() -> list: + with _sessions_lock: + return list(_sessions.values()) - Skips ``running`` sessions: the turn thread owns mid-turn persistence - (it already flushes at every persist point) and - ``_drop_trailing_empty_response_scaffolding`` mutates the live message - list, so racing an in-flight turn from the reaper thread is never safe. - Idle/detached sessions — precisely the ones a kill strands — are flushed - at most once per ``_INCREMENTAL_FLUSH_INTERVAL_S``. ``now`` is injectable - for tests (monotonic clock). - """ + +def _flush_dirty_sessions(now: float | None = None) -> int: + """Periodic incremental flush, driven by the idle-reaper scan. Skips ``running`` + sessions: the turn thread owns mid-turn persistence and mutates the live message list, + so racing it from the reaper thread is never safe. Idle sessions flush at most once per + ``_INCREMENTAL_FLUSH_INTERVAL_S``; ``now`` (monotonic) is injectable for tests.""" if _INCREMENTAL_FLUSH_INTERVAL_S <= 0: return 0 if now is None: now = time.monotonic() - with _sessions_lock: - sessions = list(_sessions.values()) flushed = 0 - for session in sessions: + for session in _reaper_session_snapshot(): if not isinstance(session, dict) or session.get("running"): continue last = float(session.get("_last_incremental_flush") or 0.0) @@ -84,13 +69,10 @@ def _flush_dirty_sessions(now: float | None = None) -> int: def _flush_sessions_before_exit(budget_s: float | None = None) -> int: - """Bounded flush of ALL in-memory sessions on the way out. - - Runs on a daemon worker joined with the budget so a hung SQLite write - can never block exit longer than ``HERMES_TUI_EXIT_FLUSH_BUDGET_S`` - (default 5s). Running sessions are included — the process is dying, so - a best-effort partial transcript beats guaranteed loss. - """ + """Bounded flush of ALL in-memory sessions on the way out. Runs on a daemon worker + joined with the budget so a hung SQLite write can't block exit past + ``HERMES_TUI_EXIT_FLUSH_BUDGET_S`` (default 5s). Running sessions are included — the + process is dying, so a partial transcript beats guaranteed loss.""" budget = _EXIT_FLUSH_BUDGET_S if budget_s is None else max(0.0, budget_s) if budget <= 0: return 0 @@ -98,9 +80,7 @@ def _flush_sessions_before_exit(budget_s: float | None = None) -> int: def _run() -> None: deadline = time.monotonic() + budget - with _sessions_lock: - sessions = list(_sessions.values()) - for session in sessions: + for session in _reaper_session_snapshot(): if time.monotonic() >= deadline: break if _flush_session_messages(session): @@ -117,16 +97,11 @@ _exit_flush_handlers_installed = False def _handle_exit_flush_signal(signum, frame) -> None: - """Flush in-memory sessions, then hand off to the prior handler. - - Chaining preserves the pre-existing signal story (uvicorn's graceful - shutdown, a supervisor's handler, or the default terminate disposition) - — this handler only *prepends* a bounded durable flush. - """ - try: + """Flush in-memory sessions, then hand off to the prior handler (uvicorn's graceful + shutdown, a supervisor's handler, or the default disposition) — this only *prepends* + a bounded durable flush.""" + with contextlib.suppress(Exception): _flush_sessions_before_exit() - except Exception: - pass import signal as _signal prev = _exit_flush_prev_handlers.get(signum) @@ -135,8 +110,8 @@ def _handle_exit_flush_signal(signum, frame) -> None: return if prev is _signal.SIG_IGN: return - # Default disposition: restore it and re-raise so the process still dies - # with the correct signal (exit status visible to supervisors). + # Default disposition: restore it and re-raise so the process dies with the correct + # signal (exit status visible to supervisors). try: _signal.signal(signum, _signal.SIG_DFL) os.kill(os.getpid(), signum) @@ -145,14 +120,10 @@ def _handle_exit_flush_signal(signum, frame) -> None: def install_exit_flush_signal_handlers() -> bool: - """Install chaining SIGTERM/SIGINT flush handlers (main thread only). - - Called by ``hermes serve`` / dashboard startup before uvicorn takes over - signals: uvicorn's ``capture_signals()`` saves these as the "original" - handlers and restores + re-raises into them after its graceful shutdown, - so the flush also covers terminations outside uvicorn's serve window. - Idempotent; returns False off the main thread or when installation fails. - """ + """Install chaining SIGTERM/SIGINT flush handlers (main thread only). Called before + uvicorn takes over signals: its ``capture_signals()`` saves these as the "original" + handlers and re-raises into them after graceful shutdown, so the flush also covers + terminations outside uvicorn's serve window. Idempotent; False off-main-thread/on failure.""" global _exit_flush_handlers_installed if _exit_flush_handlers_installed: return True @@ -174,26 +145,25 @@ def install_exit_flush_signal_handlers() -> bool: def _transport_is_dead(transport) -> bool: - # _detached_ws_transport is the post-WS-disconnect drop sentinel; a session - # parked on it has no live client. _stdio_transport is the REAL transport - # for a standalone `hermes --tui`, so it must NOT count as dead here (doing - # so let the idle reaper evict healthy standalone TUI sessions). - if transport is _detached_ws_transport: - return True - return getattr(transport, "_closed", None) is True + # _detached_ws_transport is the post-disconnect drop sentinel. _stdio_transport is the + # REAL transport for standalone `hermes --tui` and must NOT count as dead. + return transport is _detached_ws_transport or getattr(transport, "_closed", None) is True + + +def _reaper_session_is_detached_idle(sid: str, session: dict) -> bool: + """Shared hard exemptions for both reapers: never evict a session mid-turn, awaiting + input, still building, owning live delegated work, or on a live transport. Lazy watch + sessions never start a build, so their unset agent_ready must not make them immortal.""" + if session.get("running") or _session_pending_kind(sid) or _session_has_active_delegations(sid, session): + return False + ready = session.get("agent_ready") + if ready is not None and not ready.is_set() and not session.get("lazy"): + return False + return _transport_is_dead(session.get("transport")) def _session_is_evictable(sid: str, session: dict, now: float) -> bool: - if session.get("running") or _session_pending_kind(sid): - return False - if _session_has_active_delegations(sid, session): - return False - ready = session.get("agent_ready") - # Lazy watch sessions (subagent spectator windows) never start a build, - # so their forever-unset agent_ready must not make them immortal. - if ready is not None and not ready.is_set() and not session.get("lazy"): - return False - if not _transport_is_dead(session.get("transport")): + if not _reaper_session_is_detached_idle(sid, session): return False last_active = float(session.get("last_active") or 0.0) created_at = float(session.get("created_at") or 0.0) @@ -202,10 +172,7 @@ def _session_is_evictable(sid: str, session: dict, now: float) -> bool: def _reap_idle_sessions() -> None: now = time.time() - # Piggyback the periodic incremental flush on the existing reaper tick - # (#94724 item 2) — no new timer subsystem. Even a SIGKILL then loses at - # most one flush interval of un-persisted messages. - try: + try: # piggyback the incremental flush on the reaper tick — no new timer subsystem _flush_dirty_sessions() except Exception: logger.debug("periodic incremental session flush failed", exc_info=True) @@ -215,27 +182,19 @@ def _reap_idle_sessions() -> None: _close_session_by_id( sid, end_reason="idle_timeout", - predicate=lambda session, victim_sid=sid: _session_is_evictable( - victim_sid, session, time.time() - ), + predicate=lambda session, vs=sid: _session_is_evictable(vs, session, time.time()), ) _enforce_session_cap() _reclaim_orphaned_leases() - # Periodic heap release for long-lived gateway processes. Even when no - # session is reaped, Python's generational GC rarely runs gen2 collection - # under steady-state allocation, and glibc retains freed pages as RSS. - # Calling trim_memory here ensures every reaper scan (default every 5 min) - # returns releasable pages, preventing unbounded RSS growth over days/weeks. + # Long-lived processes: gen2 GC rarely runs at steady state and glibc retains freed + # pages as RSS, so trim every scan to prevent unbounded RSS growth over days/weeks. try: from hermes_cli.mem_trim import trim_memory trim_memory(reason="idle reaper periodic trim") except Exception as exc: - # debug, not warning — persistent failure would repeat every reaper - # scan (300s) forever; sibling failure branches log at debug. - logger.debug( - "idle reaper memory trim failed: %s: %s", type(exc).__name__, exc - ) + # debug, not warning — a persistent failure would repeat every scan. + logger.debug("idle reaper memory trim failed: %s: %s", type(exc).__name__, exc) def _reclaim_orphaned_leases() -> None: @@ -255,14 +214,10 @@ def _reclaim_orphaned_leases() -> None: logger.debug("orphaned lease reclaim failed", exc_info=True) -# Soft LRU cap on in-memory sessions. The 6h TTL reaper above only frees -# sessions that have been idle for hours; a heavy user who reconnects often -# accumulates detached sessions (the report's ``detached_sessions=5``) whose -# agents sit resident for the full TTL. The cap evicts the least-recently-active -# DETACHED sessions sooner so live agents don't pile up under memory pressure. -# Default-on but provably safe: it only touches sessions with no live client -# (reopening re-resumes them from the DB) and never a running / pending / -# mid-build / live-transport one. 0/null disables. +# Soft LRU cap on in-memory sessions: the TTL reaper only frees sessions idle for hours, +# so a heavy reconnecting user accumulates resident detached agents. The cap evicts the +# least-recently-active DETACHED sessions sooner — never a running / pending / mid-build / +# live-transport one (reopening re-resumes from the DB). 0/null disables. def _max_live_sessions() -> int: try: from hermes_cli.active_sessions import coerce_max_concurrent_sessions @@ -280,18 +235,8 @@ def _max_live_sessions() -> int: def _session_is_lru_evictable(sid: str, session: dict) -> bool: - # Same hard exemptions as the TTL reaper (never evict a session mid-turn, - # awaiting input, still building, or owning active delegated work), but - # WITHOUT the hours-scale age gate: a detached session is eligible the - # moment it loses its client. - if session.get("running") or _session_pending_kind(sid): - return False - if _session_has_active_delegations(sid, session): - return False - ready = session.get("agent_ready") - if ready is not None and not ready.is_set() and not session.get("lazy"): - return False - return _transport_is_dead(session.get("transport")) + # TTL-reaper exemptions WITHOUT the age gate: eligible the moment it loses its client. + return _reaper_session_is_detached_idle(sid, session) def _enforce_session_cap() -> None: @@ -299,14 +244,10 @@ def _enforce_session_cap() -> None: if cap <= 0: return with _sessions_lock: - total = len(_sessions) - if total <= cap: + if len(_sessions) <= cap: return - evictable = [ - (sid, s) for sid, s in _sessions.items() if _session_is_lru_evictable(sid, s) - ] - # Oldest-touched first; only evict down to the cap (live/focused sessions on - # a live transport are never eligible, so we may stop short of the cap). + evictable = [(sid, s) for sid, s in _sessions.items() if _session_is_lru_evictable(sid, s)] + # Oldest-touched first; evict only down to the cap (may stop short: live sessions are never eligible). evictable.sort(key=lambda kv: float(kv[1].get("last_active") or 0.0)) for sid, _s in evictable: with _sessions_lock: @@ -315,12 +256,16 @@ def _enforce_session_cap() -> None: _close_session_by_id( sid, end_reason="lru_evict", - predicate=lambda session, victim_sid=sid: _session_is_lru_evictable( - victim_sid, session - ), + predicate=lambda session, vs=sid: _session_is_lru_evictable(vs, session), ) +def _reaper_daemon_timer(delay: float, fn) -> None: + timer = threading.Timer(delay, fn) + timer.daemon = True + timer.start() + + def _schedule_session_cap_enforcement() -> None: """Run the LRU sweep off the response path (eviction can call agent.close).""" @@ -330,38 +275,26 @@ def _schedule_session_cap_enforcement() -> None: except Exception: logger.debug("session cap enforcement failed", exc_info=True) - timer = threading.Timer(0.1, _run) - timer.daemon = True - timer.start() + _reaper_daemon_timer(0.1, _run) - - -# ── Startup sweep for orphaned session rows (#65194) ───────────────────── -# The WS-orphan reaper above is an in-process threading.Timer: a gateway -# restart (update, crash, systemd) kills it before it fires, leaving the -# session row `ended_at IS NULL` forever. This is the startup complement -# every other resource type already has (docker_orphan_reaper, compression -# orphans). Scheduled once per process from both gateway entry points -# (stdio `entry.main` and the WS sidecar's `handle_ws`) — desktop/dashboard -# never run `entry.main()`. state.db is shared by sibling processes on the -# same profile, so eligibility is conservative. Disable via -# `dashboard.startup_orphan_sweep: false` (default on). +# ── Startup sweep for orphaned session rows ────────────────────────────── +# The WS-orphan reaper is an in-process Timer: a gateway restart kills it before it fires, leaving +# the row `ended_at IS NULL` forever. Scheduled once per process from both gateway entry points +# (stdio `entry.main`, WS sidecar `handle_ws`). state.db is shared by sibling processes on the same +# profile, so eligibility is conservative. Disable via `dashboard.startup_orphan_sweep: false`. _ORPHAN_SWEEP_SOURCES = ("tui", "desktop", "subagent") _startup_orphan_sweep_ran = False _startup_orphan_sweep_lock = threading.Lock() def _session_orphan_reaper_enabled() -> bool: - """``dashboard.startup_orphan_sweep`` (default on). Fail-open on errors.""" + """``dashboard.startup_orphan_sweep`` (default on). Fail-open on errors and + on a missing key (raw yaml, no DEFAULT_CONFIG merge on this loader).""" try: dashboard_cfg = (_load_cfg() or {}).get("dashboard") or {} if isinstance(dashboard_cfg, dict) and "startup_orphan_sweep" in dashboard_cfg: - return is_truthy_value( - dashboard_cfg.get("startup_orphan_sweep"), default=True - ) - # Fail-open: a missing key (raw yaml, no DEFAULT_CONFIG merge on - # this loader) must keep the sweep on. + return is_truthy_value(dashboard_cfg.get("startup_orphan_sweep"), default=True) return True except Exception: return True @@ -374,11 +307,9 @@ def _live_session_ids() -> list[str]: for sid, session in _sessions.items(): if sid: ids.add(str(sid)) - agent = session.get("agent") if isinstance(session, dict) else None - for candidate in ( - getattr(agent, "session_id", None), - session.get("session_key") if isinstance(session, dict) else None, - ): + if not isinstance(session, dict): + continue + for candidate in (getattr(session.get("agent"), "session_id", None), session.get("session_key")): if candidate: ids.add(str(candidate)) return sorted(ids) @@ -387,56 +318,33 @@ def _live_session_ids() -> list[str]: def _sweep_orphaned_session_rows() -> list[str]: """End orphaned tui/desktop/subagent rows left by a dead process. - "Provably orphaned" is inferred conservatively from inactivity — the - row must have been created AND last messaged at least the session TTL - ago (``HERMES_TUI_SESSION_TTL_S``). A freshly created row that copied - an old transcript is protected by its own ``started_at``. Rows this - process still holds in memory (e.g. a ``session.resume`` during the - startup grace window) are excluded so the sweep never races a - mid-reconnect client. - - Cross-backend liveness (#94895): when one ``state.db`` is shared by - N serve / gateway processes, each registered a heartbeat row in - ``gateway_heartbeats``. The sweep refuses to close a row that any - live backend (heartbeat refreshed within ``2 * TTL``) could - plausibly own — see ``SessionDB.sweep_orphaned_sessions`` for the - exact predicate. + "Provably orphaned" is inferred conservatively: the row must have been created AND last + messaged at least the session TTL ago (a fresh row that copied an old transcript is + protected by its own ``started_at``). Rows held in memory (e.g. a ``session.resume`` in + the startup grace window) are excluded. Cross-backend: the sweep refuses to close a row + any live backend (heartbeat within ``2 * TTL``) could own — see + ``SessionDB.sweep_orphaned_sessions``. """ db = _get_db() - if db is None: - return [] ttl = _SESSION_TTL_S - if ttl <= 0: + if db is None or ttl <= 0: return [] swept = db.sweep_orphaned_sessions( - max_idle_seconds=ttl, - sources=_ORPHAN_SWEEP_SOURCES, - exclude_ids=tuple(_live_session_ids()), + max_idle_seconds=ttl, sources=_ORPHAN_SWEEP_SOURCES, exclude_ids=tuple(_live_session_ids()) ) if swept: logger.info( - "Closed %d orphaned session row(s) from a previous gateway " - "process (startup_orphan_reap): %s", - len(swept), - ", ".join(swept), + "Closed %d orphaned session row(s) from a previous gateway process (startup_orphan_reap): %s", + len(swept), ", ".join(swept), ) return swept -# ── Cross-backend heartbeat (#94895) ─────────────────────────────────── -# Each serve / gateway process registers a heartbeat row in -# ``gateway_heartbeats`` so the startup orphan sweep can tell "row owned -# by a live but idle backend" from "row truly orphaned". Without this, -# the first process to restart in a multi-backend topology reaped every -# inactive row — including those held by the other N−1 still-running -# processes (the #94895 reporter saw 473 sessions disappear in one shot). -# -# Refresh cadence: every HEARTBEAT_REFRESH_S (default 60s — much shorter -# than the default 6h session TTL so a refresh always lands inside the -# staleness window). The heartbeat is removed at process exit so a -# graceful shutdown doesn't leave a stale row behind. A crashed process -# leaves its row until the heartbeat ages out of the staleness window, -# at which point the sweep treats it as dead. +# ── Cross-backend heartbeat ────────────────────────────────────────────── +# Each serve / gateway process registers a heartbeat row in ``gateway_heartbeats`` so the startup +# sweep can tell "owned by a live but idle backend" from "truly orphaned" (else the first process to +# restart reaped every inactive row of the other N−1). Refresh 60s default — far shorter than the +# 6h TTL so a refresh always lands inside the staleness window. Removed at exit; a crashed row ages out. _HEARTBEAT_REFRESH_S = max(0.0, env_float("HERMES_GATEWAY_HEARTBEAT_REFRESH_S", 60.0)) @@ -444,13 +352,13 @@ _heartbeat_refresher_started = False _heartbeat_refresher_lock = threading.Lock() -def _backend_id_for_this_process() -> str: - """Stable identity for this process's heartbeat row (#94895). +def _reaper_hostname() -> str: + return os.uname().nodename if hasattr(os, "uname") else "host" - Includes pid AND a startup-time nonce so a PID-reuse respawn cannot - inherit the dead predecessor's heartbeat and protect truly orphaned - sessions. The pid is kept for human readability in diagnostics. - """ + +def _backend_id_for_this_process() -> str: + """Stable identity for this process's heartbeat row: pid (readability) AND a startup + nonce so a PID-reuse respawn cannot inherit the dead predecessor's heartbeat.""" nonce = getattr(_backend_id_for_this_process, "_nonce", None) if nonce is None: import secrets as _secrets @@ -460,11 +368,11 @@ def _backend_id_for_this_process() -> str: setattr(_backend_id_for_this_process, "_nonce", nonce) except AttributeError: # pragma: no cover - defensive pass - return f"{_current_profile_name()}@{os.uname().nodename if hasattr(os, 'uname') else 'host'}:{os.getpid()}:{nonce}" + return f"{_current_profile_name()}@{_reaper_hostname()}:{os.getpid()}:{nonce}" def _refresh_backend_heartbeat() -> None: - """Refresh this backend's heartbeat row (#94895). No-op when DB unavailable.""" + """Refresh this backend's heartbeat row. No-op when DB unavailable.""" db = _get_db() if db is None: return @@ -474,17 +382,15 @@ def _refresh_backend_heartbeat() -> None: pid=os.getpid(), started_at=_gateway_started_at(), profile=_current_profile_name(), - host=(os.uname().nodename if hasattr(os, "uname") else "host"), + host=_reaper_hostname(), ) except Exception: logger.debug("backend heartbeat refresh failed", exc_info=True) def _gateway_started_at() -> float: - """Wall-clock time when this process started. Module-import time is - a good-enough proxy: the heartbeat refresher runs after the gateway - is fully wired up. - """ + """Wall-clock time this process started (first-call time is a good-enough + proxy: the heartbeat refresher runs after the gateway is fully wired up).""" started = getattr(_gateway_started_at, "_t", None) if started is None: started = time.time() @@ -506,25 +412,14 @@ def _heartbeat_refresher_loop(stop_event: threading.Event) -> None: def _start_backend_heartbeat_refresher() -> None: - """Register this backend and start the refresher thread (#94895). - - Called once per process from both gateway entry points. The first - refresh writes the row immediately so even a very fast crash leaves - a fresh-enough row that other backends can see. Repeat calls are - no-ops. The refresher thread is only spawned when - ``_HEARTBEAT_REFRESH_S > 0`` — a refresh interval of zero means - "register the row once, never refresh" (the row ages out naturally - after the heartbeat staleness window). - """ + """Register this backend and start the refresher thread (once per process). The first + refresh writes the row synchronously so this process's own sweep sees itself in the + heartbeat table. ``_HEARTBEAT_REFRESH_S <= 0`` means "register once, never refresh".""" global _heartbeat_refresher_started with _heartbeat_refresher_lock: if _heartbeat_refresher_started: return _heartbeat_refresher_started = True - # Write a row synchronously so the sweep run later in this same - # process can see ourselves in the heartbeat table too. Without - # this, exclude_ids would have to cover every local session — a - # regression in the strict-ownership case the heartbeat exists to fix. try: _refresh_backend_heartbeat() except Exception: @@ -535,37 +430,24 @@ def _start_backend_heartbeat_refresher() -> None: def _atexit_clear(): stop_event.set() - try: + with contextlib.suppress(Exception): db = _get_db() if db is not None: db.clear_backend_heartbeat(_backend_id_for_this_process()) - except Exception: - pass atexit.register(_atexit_clear) - thread = threading.Thread( - target=_heartbeat_refresher_loop, - args=(stop_event,), - name="hermes-gateway-heartbeat", - daemon=True, - ) - thread.start() + threading.Thread( + target=_heartbeat_refresher_loop, args=(stop_event,), name="hermes-gateway-heartbeat", daemon=True + ).start() def _schedule_startup_orphan_sweep() -> None: - """Schedule the once-per-process startup orphan sweep (#65194). - - Called from both gateway entry points. Repeat calls are no-ops. The - sweep is delayed by the WS-orphan grace window so a client reconnecting - right after a restart can ``session.resume`` its row before the sweep - reads the DB. ``HERMES_TUI_WS_ORPHAN_REAP_GRACE_S=0`` (park forever) - and ``HERMES_TUI_SESSION_TTL_S=0`` both suppress the sweep; so does - ``dashboard.startup_orphan_sweep: false``. - """ + """Schedule the once-per-process startup orphan sweep, delayed by the WS-orphan grace + window so a client reconnecting right after a restart can ``session.resume`` its row + first. Grace 0 (park forever), TTL 0 and ``dashboard.startup_orphan_sweep: false`` + all suppress the sweep.""" global _startup_orphan_sweep_ran - if _WS_ORPHAN_REAP_GRACE_S <= 0 or _SESSION_TTL_S <= 0: - return - if not _session_orphan_reaper_enabled(): + if _WS_ORPHAN_REAP_GRACE_S <= 0 or _SESSION_TTL_S <= 0 or not _session_orphan_reaper_enabled(): return if _startup_orphan_sweep_ran: return @@ -580,9 +462,7 @@ def _schedule_startup_orphan_sweep() -> None: except Exception: logger.warning("startup orphan session sweep failed", exc_info=True) - timer = threading.Timer(_WS_ORPHAN_REAP_GRACE_S, _run) - timer.daemon = True - timer.start() + _reaper_daemon_timer(_WS_ORPHAN_REAP_GRACE_S, _run) def register(server) -> None: diff --git a/tui_gateway/session_workdir.py b/tui_gateway/session_workdir.py index 00431afc94..4bf66d4651 100644 --- a/tui_gateway/session_workdir.py +++ b/tui_gateway/session_workdir.py @@ -18,12 +18,7 @@ def _normalize_completion_path(path_part: str) -> str: expanded = os.path.expanduser(path_part) if os.name != "nt": normalized = expanded.replace("\\", "/") - if ( - len(normalized) >= 3 - and normalized[1] == ":" - and normalized[2] == "/" - and normalized[0].isalpha() - ): + if len(normalized) >= 3 and normalized[1] == ":" and normalized[2] == "/" and normalized[0].isalpha(): return f"/mnt/{normalized[0].lower()}/{normalized[3:]}" return expanded @@ -36,10 +31,8 @@ def _completion_cwd(params: dict | None = None) -> str: # A session bound to another profile resolves its workspace from THAT # profile's config before falling back to the launch profile's env var. or _profile_configured_cwd(_profile_home(params.get("profile"))) - # The launch profile's dashboard /chat attaches to the dashboard's - # in-memory gateway, which does NOT inherit the PTY child's bridged - # TERMINAL_CWD. Read the launch profile's config.yaml directly so a - # configured terminal.cwd wins over a stale process env / launch dir. + # The dashboard's in-memory gateway does NOT inherit the PTY child's + # bridged TERMINAL_CWD, so a configured terminal.cwd is read directly. or _launch_configured_cwd() or os.environ.get("TERMINAL_CWD") or os.getcwd() @@ -53,63 +46,35 @@ def _completion_cwd(params: dict | None = None) -> str: return os.getcwd() +def _workdir_terminal_cfg(key: str) -> str: + """Stripped ``terminal.`` from config, or "" when unset/unreadable.""" + try: + terminal_cfg = _load_cfg().get("terminal", {}) + if isinstance(terminal_cfg, dict): + return str(terminal_cfg.get(key) or "").strip() + except Exception: + pass + return "" + + def _terminal_task_cwd(session: dict | None) -> str: - """Return the cwd that terminal_tool should use for this TUI session. - - ``_completion_cwd`` validates paths on the host so file completion does not - point at nonsense. Non-local terminal backends are different: their cwd is - inside the target environment, so an SSH path like /home/user/workspace may - not exist on the local macOS host but is still the correct execution cwd. - - When ``TERMINAL_ENV`` is unset (dashboard/TUI process) the config's - ``terminal.backend`` is consulted as a fallback so the non-local cwd - resolution path is taken even when the dashboard entrypoint did not call - ``apply_terminal_config_to_env`` on its own ``os.environ``. - """ + """The cwd terminal_tool should use for this TUI session. Unlike + ``_completion_cwd`` it is NOT validated on the host: a non-local backend's + cwd lives inside the target environment.""" return _terminal_task_cwd_with_source(session)[0] def _terminal_task_cwd_with_source(session: dict | None) -> tuple[str, str]: - """Like :func:`_terminal_task_cwd` but also names the value's ORIGIN. - - Returns ``(cwd, source)`` where source is: - - * ``"session"`` — the workspace the user attached to THIS session - (``explicit_cwd``), or this session's own tracked directory. - * ``"process"`` — the process-global ``TERMINAL_CWD`` env var / config - ``terminal.cwd`` fallback. On a shared-container backend this is the - normal seed; under per-session docker isolation it is a launch - artifact from a PREVIOUS session (the workspace picker persists it - process-wide) and must never become a fresh session's bind mount — - terminal_tool refuses ``cwd_source: "process"`` as a mount source. - """ - backend = (os.environ.get("TERMINAL_ENV") or "").strip().lower() - if not backend or backend == "local": - # Fall back to config when TERMINAL_ENV is unset (dashboard/TUI process - # never calls apply_terminal_config_to_env on os.environ). - try: - terminal_cfg = _load_cfg().get("terminal", {}) - if isinstance(terminal_cfg, dict): - cfg_backend = str(terminal_cfg.get("backend") or "").strip().lower() - if cfg_backend and cfg_backend != "local": - backend = cfg_backend - except Exception: - pass - - if backend and backend != "local": - # A workspace the user explicitly attached to THIS session wins over - # the process-global env var — the env var is whatever the LAST - # session's picker wrote, not this session's choice. + """Like :func:`_terminal_task_cwd` but returns ``(cwd, source)``: ``"session"`` for THIS + session's workspace (``explicit_cwd``/tracked dir), ``"process"`` for the process-global + ``TERMINAL_CWD`` / ``terminal.cwd`` fallback — under per-session docker isolation that is a + launch artifact of a PREVIOUS session, so terminal_tool refuses it as a bind-mount source.""" + backend = _effective_terminal_backend() + if backend != "local": + # THIS session's explicit workspace beats the LAST session's env var. if session and session.get("explicit_cwd") and session.get("cwd"): return str(session["cwd"]), "session" - raw = os.environ.get("TERMINAL_CWD", "").strip() - if not raw: - try: - terminal_cfg = _load_cfg().get("terminal", {}) - if isinstance(terminal_cfg, dict): - raw = str(terminal_cfg.get("cwd") or "").strip() - except Exception: - raw = "" + raw = os.environ.get("TERMINAL_CWD", "").strip() or _workdir_terminal_cfg("cwd") if raw and raw not in {".", "auto", "cwd"}: return raw, "process" if backend == "ssh": @@ -120,9 +85,7 @@ def _terminal_task_cwd_with_source(session: dict | None) -> tuple[str, str]: return _completion_cwd(), "process" -# Git working-tree probing (run git, resolve roots, fold worktrees) lives in a -# focused, single-flight-cached module; these stay as the in-server names every -# call site already uses. +# Git probing lives in git_probe; these keep the in-server names call sites use. _git = git_probe.run_git _git_branch_for_cwd = git_probe.branch _git_repo_root_for_cwd = git_probe.repo_root @@ -137,8 +100,7 @@ def _session_cwd(session: dict | None) -> str: # Sources whose launch directory is an artifact of how the app was started, not -# a workspace the user picked. Everything else is terminal-started: the process -# runs in a directory the user deliberately cd'd into. +# a workspace the user picked (everything else is a directory the user cd'd into). _LAUNCH_CWD_NOT_A_WORKSPACE = {"desktop"} @@ -152,41 +114,26 @@ def _context_cwd_is_launch_artifact(session: dict | None) -> bool: def _persisted_session_cwd(session: dict) -> str | None: - """The cwd to stamp on the session's DB row, or None to leave it unset. - - See :func:`_ensure_session_db_row` for why the launch directory counts as a - workspace for terminal sessions but not for the desktop. - """ + """The cwd to stamp on the session's DB row, or None to leave it unset (see + :func:`_ensure_session_db_row` for the desktop vs terminal launch-dir rule).""" if session.get("explicit_cwd"): return _session_cwd(session) if _session_source(session) in _LAUNCH_CWD_NOT_A_WORKSPACE: return None - # Only the session's OWN directory. `_session_cwd` falls back to the - # gateway-wide completion cwd, which belongs to no session in particular — - # stamping that would invent a workspace for a session that never had one. + # Only the session's OWN directory, never `_session_cwd`'s gateway-wide fallback. return str(session.get("cwd") or "") or None def _heal_dead_cwd(cwd: str) -> str: - """Resolve a session cwd that points at a now-deleted directory. - - A session anchored to a linked worktree (``/.worktrees/``) keeps - that path after the worktree is removed (branch merged, `git worktree - remove`, etc). The literal dir is gone, so a probe of it returns nothing and - the composer shows no branch — while the sidebar still folds the path up to - the repo's main lane. Heal the mismatch: walk up to the first existing - ancestor, then resolve its common git root, so a dead-worktree cwd collapses - to the live repo root (and its real current branch). - - Only meaningful for local backends; a remote/SSH cwd may legitimately not - exist on the host, so callers must skip healing there. - """ + """Resolve a session cwd inside a now-deleted directory (e.g. a removed linked worktree, + which probes to no branch while the sidebar folds it to the main lane): walk up to the first + existing ancestor and take its common git root. Local backends only — a remote/SSH cwd may + legitimately not exist on the host, so callers skip healing there.""" raw = (cwd or "").strip() if not raw or os.path.isdir(raw): return raw probe = raw - # Climb to the first ancestor that still exists on disk. for _ in range(64): parent = os.path.dirname(probe) if not parent or parent == probe: @@ -212,33 +159,21 @@ def _is_local_terminal_backend() -> bool: def _effective_terminal_backend() -> str: - """Active terminal backend name (``local``, ``docker``, ``ssh``, ...). - - ``TERMINAL_ENV`` is authoritative when set (launchers bridge - ``terminal.backend`` into env at startup). Desktop/TUI in-process gateways - skip that bridge, so fall back to the ``terminal.backend`` config key — - the same rule ``_terminal_task_cwd`` uses. - """ + """Active terminal backend name (``local``, ``docker``, ``ssh``, ...): + ``TERMINAL_ENV`` when set (launchers bridge ``terminal.backend`` into env), + else the ``terminal.backend`` config key (desktop/TUI in-process gateways + skip that bridge).""" backend = (os.environ.get("TERMINAL_ENV") or "").strip().lower() if not backend or backend == "local": - try: - terminal_cfg = _load_cfg().get("terminal", {}) - if isinstance(terminal_cfg, dict): - cfg_backend = str(terminal_cfg.get("backend") or "").strip().lower() - if cfg_backend and cfg_backend != "local": - backend = cfg_backend - except Exception: - pass + cfg_backend = _workdir_terminal_cfg("backend").lower() + if cfg_backend and cfg_backend != "local": + backend = cfg_backend return backend or "local" def _display_session_cwd(session: dict | None) -> str: - """Session cwd for display/probe surfaces, healed past deleted worktrees. - - Persists the healed value back to the session row (best-effort, local only) - so the next load is already coherent and the sidebar lane stops showing a - session pinned to a vanished path. - """ + """Session cwd for display/probe surfaces, healed past deleted worktrees; + the healed value is persisted back (best-effort, local only).""" cwd = _session_cwd(session) if not _is_local_terminal_backend(): return cwd @@ -252,33 +187,18 @@ def _display_session_cwd(session: dict | None) -> str: def _reconcile_session_cwd_from_terminal(session: dict | None) -> bool: - """Re-anchor a session that SETTLED in another git checkout. Returns moved. + """Re-anchor a session that SETTLED in another worktree of the SAME repo. Returns moved. - An agent told to work in a fresh worktree does exactly that — `git worktree - add`, `cd` into it, and every later command runs there — but the session - stayed pinned to wherever it started, so the desktop kept labelling the chat - with the primary checkout's branch while all the work landed elsewhere. - - A plain `cd` is deliberately NOT a workspace move (see - ``_apply_project_workspace``): browsing to /tmp to read a log must not - re-home the chat. What we adopt here is narrower — the session's recorded - cwd is in a DIFFERENT working tree of the SAME repository (the shape - ``git worktree add`` produces). Everything else — a non-git workspace - stepping into a repo, or a git workspace visiting an unrelated repo — is - a browsing visit, and a user's explicitly chosen workspace is never - overridden at all. - - Local backends only: a remote/SSH cwd names a path on the host, which this - gateway can neither stat nor probe with git. - """ + An agent told to work in a fresh worktree `git worktree add`s and `cd`s in while the session + stays pinned (labelled with the primary checkout's branch). A plain `cd` is deliberately NOT + a workspace move (see ``_apply_project_workspace``): a non-git workspace stepping into a repo + or a visit to an unrelated repo is browsing, and an explicitly chosen workspace is never + overridden. Local backends only (a remote cwd cannot be stat'ed or git-probed here).""" if not session or not _is_local_terminal_backend(): return False - # A workspace the user (or GUI) explicitly chose is never overridden by - # where the agent's terminal happened to settle — only another explicit - # action (`_set_session_cwd`, a project switch) moves it. A cwd this very - # function adopted is marked `cwd_from_settle` so a session can keep - # following the agent through successive worktrees. + # An explicit choice only moves by another explicit action; a cwd adopted + # HERE is marked `cwd_from_settle` so successive settles keep following. if session.get("explicit_cwd") and not session.get("cwd_from_settle"): return False @@ -297,34 +217,20 @@ def _reconcile_session_cwd_from_terminal(session: dict | None) -> bool: if resolved == current or not os.path.isdir(resolved): return False - # The worktree ROOT, not the common repo root: folding worktrees together - # here is exactly what hides the move we're looking for. + # Worktree ROOTS (folding to the common root would hide the move), both in a git + # tree, different from each other, sharing the SAME common .git dir. landed = _git_repo_root_for_cwd(resolved) current_root = _git_repo_root_for_cwd(current) - # A relocation is a move between two DIFFERENT git working trees. When the - # session's own workspace is not in a git repo, the agent stepping into one - # to read a file or run a command is a browsing visit, not a re-home: - # adopting it would hijack a non-git workspace onto whatever repo a tool - # call touched first (e.g. a home-directory session pinned to the checkout - # it read a file from). if not landed or not current_root or landed == current_root: return False - - # And only between checkouts of the SAME repository — the shape a real - # `git worktree add` produces (linked worktrees share the common .git - # dir). Settling in an UNRELATED repo (`cd ~/other-project && git log`) - # is likewise a visit: adopting it would re-home the chat onto whatever - # foreign repo the terminal last touched. landed_common = _git_common_repo_root_for_cwd(resolved) current_common = _git_common_repo_root_for_cwd(current) if not landed_common or landed_common != current_common: return False session["cwd"] = resolved - # The session works here now, so this is its workspace — a desktop chat - # whose cwd was an unpersisted launch artifact earns a real row. The - # settle marker keeps this adoption overridable by the NEXT settle while - # still yielding to a user's explicit choice (see the guard above). + # This is the session's workspace now (a desktop launch-artifact cwd earns + # a real row); the settle marker keeps it overridable by the NEXT settle. session["explicit_cwd"] = True session["cwd_from_settle"] = True _register_session_cwd(session) @@ -334,14 +240,8 @@ def _reconcile_session_cwd_from_terminal(session: dict | None) -> bool: def _emit_settled_session_info(sid: str, session: dict, agent) -> None: - """Emit end-of-turn ``session.info``, reconciling a settled cwd first. - - The turn is over, so the agent has stopped moving: this is the one moment - where its recorded cwd is a stable answer to "where does this session - work". Reconciling before building the payload means the same event that - already tells the desktop the turn ended also carries the new cwd/branch — - the client follows it with no new event type and no extra round trip. - """ + """Emit end-of-turn ``session.info``, reconciling a settled cwd first: the agent has stopped + moving, and riding the reconcile on the turn-end event needs no new event type/round trip.""" try: _reconcile_session_cwd_from_terminal(session) except Exception: @@ -364,96 +264,28 @@ def _register_session_cwd(session: dict | None) -> None: from tools.terminal_tool import register_task_env_overrides cwd, cwd_source = _terminal_task_cwd_with_source(session) - register_task_env_overrides( - session["session_key"], {"cwd": cwd, "cwd_source": cwd_source} - ) + register_task_env_overrides(session["session_key"], {"cwd": cwd, "cwd_source": cwd_source}) except Exception: pass -def _ensure_session_db_row(session: dict) -> bool: - """Idempotently persist the session's DB row on first real activity. +def _workdir_row_model_config(session: dict) -> tuple[str, dict]: + """``(model, model_config)`` for a fresh session row. - Called from prompt.submit so a row only exists once the user actually sends - a message — abandoned drafts never leave an empty "Untitled" session behind. - Uses INSERT OR IGNORE under the hood, so re-calls (and the AIAgent's own - lazy create) are no-ops. - Returns False only when the store is unavailable (no openable state.db); - prompt.submit turns that into an RPC error so a send fails loudly with a - toast instead of streaming into a store that will never save it (#98924). - Every other outcome — no key, best-effort attempt, success — is True. - - A cwd the user *chose* is always persisted. When they made no explicit - choice the launch directory stands in, and whether that is meaningful - depends on how the session was started: - - * The desktop launches from wherever the app bundle was opened (often ``/`` - or the user's home), so stamping that would file every unpicked chat under - a folder the user never chose. Those stay null and group under "No - workspace", which is the desired default. - * A terminal session (``hermes`` / ``hermes --tui`` / CLI) is started from a - directory the user deliberately ``cd``'d into — that IS the workspace, and - it is also where the agent's terminal actually runs. Dropping it stranded - the session with no cwd AND no git_repo_root, so the sidebar could never - place it under its project. - """ - key = session.get("session_key") - if not key: - return - # Persist into the session's own profile db (global remote mode), not the - # launch profile's — otherwise the row lands in the wrong state.db, the - # unified list mis-tags it, and resume 404s ("session not found"). - profile_home = session.get("profile_home") - if profile_home: - from hermes_state import SessionDB - - try: - from hermes_state import get_shared_session_db - db = get_shared_session_db(Path(profile_home) / "state.db") - except Exception: - logger.debug("failed to open profile db for session row", exc_info=True) - return False - close_db = True - else: - db = _get_db() - close_db = False - if db is None: - # Fail loud ONLY when the store actually failed to open (#98924): - # _db_error records the SessionDB open exception. A None db with no - # recorded error means "no store in this context" (degraded harness, - # store deliberately absent) — that keeps the pinned best-effort - # contract and stays True. - return _db_error is None - # The session's own model/effort/fast pick — the composer override shipped on - # session.create, or a restored /model switch — must own the row's model + - # model_config. The agent isn't built yet at first prompt.submit, so derive - # the row from the live override dict; fall back to the global resolved model - # only when this chat made no explicit pick. Writing the global default here - # used to win the INSERT-OR-IGNORE race against the agent's own correct - # lazy-create, so a reconnect/resume rebuilt from the global model and - # silently reverted the chat (e.g. picked gpt-5.5, reconnect snapped back to - # the profile default). model_config carries provider/reasoning/service_tier - # so resume restores effort + fast too, not just the model name. + The session's own model/effort/fast pick (composer override or restored /model switch) must + own the row: the agent isn't built yet at first prompt.submit, and writing the global default + here wins the INSERT-OR-IGNORE race, so a reconnect silently reverts to the profile default. + model_config carries provider/reasoning/service_tier so resume restores effort + fast too.""" override = session.get("model_override") override = override if isinstance(override, dict) else {} row_model = str(override.get("model") or "").strip() or _resolve_model() model_config: dict = {} - for src_key, cfg_key in ( - ("model", "model"), - ("provider", "provider"), - ("base_url", "base_url"), - ("api_mode", "api_mode"), - ): - if val := override.get(src_key): + for cfg_key in ("model", "provider", "base_url", "api_mode"): + if val := override.get(cfg_key): model_config[cfg_key] = str(val) - # The composer override may carry the RESOLVED provider "custom" for a named - # ``providers:`` / ``custom_providers:`` entry. Persisting bare "custom" here - # (the very first DB write for a fresh desktop session, before the agent is - # built) is the origin of the recurring "No LLM provider configured" rows: - # on the next resume bare "custom" routes to OpenRouter with no key. Recover - # the durable ``custom:`` identity from the override's base_url, else - # the configured provider, so a routable identity is persisted from the - # start (matches _runtime_model_config's normalization). + # A RESOLVED provider "custom" (named ``providers:``/``custom_providers:`` entry) persisted + # bare here is the origin of "No LLM provider configured" rows (resume routes to OpenRouter + # with no key). Recover the durable ``custom:`` identity (matches _runtime_model_config). if str(model_config.get("provider") or "").strip().lower() == "custom": try: from hermes_cli.runtime_provider import canonical_custom_identity @@ -465,96 +297,100 @@ def _ensure_session_db_row(session: dict) -> bool: if healed: model_config["provider"] = healed except Exception: - logger.debug( - "custom provider identity recovery failed (db row)", exc_info=True - ) + logger.debug("custom provider identity recovery failed (db row)", exc_info=True) if (reasoning := session.get("create_reasoning_override")) is not None: model_config["reasoning_config"] = reasoning create_service_tier_override = session.get("create_service_tier_override") if create_service_tier_override is not None: - # Empty string is the in-memory sentinel for an explicit normal tier: - # it bypasses _make_agent's profile fallback without sending a bogus - # service_tier value to the provider. Persist a durable marker so resume - # can distinguish that choice from an omitted/inherited tier. + # "" is the in-memory sentinel for an explicit normal tier (bypasses _make_agent's profile + # fallback); persist a durable marker so resume can tell it from an inherited tier. model_config["service_tier"] = create_service_tier_override or "normal" - # Branch lineage: stamp the same ``_branched_from`` marker the TUI /branch - # uses so list_sessions_rich keeps the branch listed and the desktop sidebar - # can nest it under its parent. - parent_session_id = session.get("parent_session_id") or None - if parent_session_id: + # Same ``_branched_from`` marker the TUI /branch uses (list_sessions_rich + sidebar nesting). + if parent_session_id := session.get("parent_session_id"): model_config["_branched_from"] = parent_session_id - # Bot-Mode room plumbing sessions are per-member scratch conversations - # inside a group chat: their runtime must ALWAYS follow the member profile's - # CURRENT config, never the provider that was pinned when the row was first - # written. Persist that contract explicitly so resume can distinguish room - # plumbing from a normal user chat (whose stored model/provider must be - # restored verbatim). See _stored_session_runtime_overrides. - if session.get("room_plumbing"): - model_config["room_plumbing"] = True - # Bot-Mode canonical chats (the ONE forever DM per bot) and room plumbing - # sessions are plugin-owned scratch conversations: their runtime must ALWAYS - # follow the member profile's CURRENT config, never the model/provider that - # was pinned when the row was first written. Persist that contract explicitly - # so resume can distinguish them from a normal user chat (whose stored - # model/provider must be restored verbatim). See - # _stored_session_runtime_overrides. - if session.get("follow_profile_config"): - model_config["follow_profile_config"] = True - try: - db.create_session( - key, - source=_session_source(session), - model=row_model, - model_config=model_config or None, - parent_session_id=parent_session_id, - cwd=_persisted_session_cwd(session), - # Self-describing rows: aggregators that merge multiple profile DBs - # into one list can't rely on which file a row came from alone. - # Stamp the launch profile explicitly instead of leaving NULL — - # NULL is exactly what the #94724 legacy-owner backfill exists to - # repair, and rows minted AFTER that one-shot backfill ran stayed - # NULL forever: profile-keyed matching then drops them from the - # sidebar and deep links can't resolve them (#99222). - profile_name=( - Path(profile_home).name if profile_home else _current_profile_name() - ), - ) - # A session can be born hidden (session.create hidden=true, or a - # session.set_hidden that arrived before the row existed): apply the - # deferred intent now that the row exists, mirroring pending_title. - if session.get("pending_hidden"): - try: - db.set_session_hidden(key, True) - except Exception: - logger.debug("failed to apply pending hidden flag", exc_info=True) - except Exception as exc: - # Disk-full is not a soft failure: if we swallow it here, prompt.submit - # returns {"status":"streaming"} and the user's message vanishes with - # no toast. Re-raise so the submit handler can return a real RPC error. - from hermes_state import is_disk_full_error + # Bot-Mode canonical chats / room plumbing are plugin-owned scratch conversations whose runtime + # must ALWAYS follow the member profile's CURRENT config, never the provider pinned at first + # write; persist that contract for resume (see _stored_session_runtime_overrides). + for flag in ("room_plumbing", "follow_profile_config"): + if session.get(flag): + model_config[flag] = True + return row_model, model_config - if is_disk_full_error(exc): - raise - logger.debug("failed to persist desktop session row", exc_info=True) - finally: - if close_db: - try: - from hermes_state import release_or_close - release_or_close(db) - except Exception: - pass + +def _ensure_session_db_row(session: dict) -> bool: + """Idempotently persist the session's DB row on first real activity (prompt.submit), so + abandoned drafts never leave an empty "Untitled" session. INSERT OR IGNORE: re-calls and the + AIAgent's lazy create are no-ops. Returns False only when the store is unavailable (no + openable state.db) — prompt.submit fails the send loudly instead of streaming into a store + that will never save it; no key / best-effort / success are all True. + + A cwd the user *chose* is always persisted. Otherwise the launch directory stands in only for + terminal sessions (the user deliberately ``cd``'d there; dropping it left the sidebar with no + cwd AND no git_repo_root); desktop launch dirs (``/``, home) stay null -> "No workspace".""" + key = session.get("session_key") + if not key: + return + # Persist into the session's own profile db (global remote mode), not the launch profile's — + # otherwise the unified list mis-tags the row and resume 404s ("session not found"). + profile_home = session.get("profile_home") + with _workdir_owner_db(session, "failed to open profile db for session row") as db: + if db is _WORKDIR_DB_OPEN_FAILED: + return False + if db is None: + # Fail loud ONLY when the store failed to open (_db_error records the SessionDB open + # exception); None with no recorded error means "no store in this context" -> True. + return _db_error is None + row_model, model_config = _workdir_row_model_config(session) + try: + db.create_session( + key, + source=_session_source(session), + model=row_model, + model_config=model_config or None, + parent_session_id=session.get("parent_session_id") or None, + cwd=_persisted_session_cwd(session), + # Self-describing rows: aggregators merging several profile DBs can't rely on + # which file a row came from; a NULL is only repaired by the one-shot backfill. + profile_name=Path(profile_home).name if profile_home else _current_profile_name(), + ) + # Born hidden (session.create hidden=true, or set_hidden before the + # row existed): apply the deferred intent now, like pending_title. + if session.get("pending_hidden"): + try: + db.set_session_hidden(key, True) + except Exception: + logger.debug("failed to apply pending hidden flag", exc_info=True) + except Exception as exc: + # Disk-full is not a soft failure: swallowed here, prompt.submit + # returns {"status":"streaming"} and the message vanishes silently. + _workdir_reraise_disk_full(exc, "failed to persist desktop session row") return True -def _persist_branch_seed(session: dict) -> None: - """First-turn persist of a branch's copied transcript. +def _workdir_reraise_disk_full(exc: BaseException, log_msg: str) -> None: + """Re-raise a disk-full write error (the caller must surface it); debug-log the rest.""" + from hermes_state import is_disk_full_error - A branch is a draft until its first submit: the parent's messages live only - in ``session["history"]`` (they ride into the agent as ``conversation_history``, - which ``_flush_messages_to_session_db`` skips by identity). Without this the - branch row would resume missing its pre-branch context. Runs once; the row + - parent link are written by ``_ensure_session_db_row`` just before this. - """ + if is_disk_full_error(exc): + raise exc + logger.debug(log_msg, exc_info=True) + + +# Seed row fields copied from the parent transcript. display_kind/metadata: timeline markers +# ride as role=user, dropping the tag re-plants them as bare user turns after a restart and +# corrupts the truncate ordinal address space. timestamp: parent's original, not "now". +_WORKDIR_SEED_FIELDS = ( + "content", "reasoning", "reasoning_content", "reasoning_details", "codex_reasoning_items", + "codex_message_items", "display_kind", "display_metadata", "timestamp", +) + + +def _persist_branch_seed(session: dict) -> None: + """First-turn persist of a branch's copied transcript. A branch is a draft until its first + submit: the parent's messages live only in ``session["history"]`` (ridden into the agent as + ``conversation_history``, which ``_flush_messages_to_session_db`` skips by identity), so the + row would otherwise resume missing its pre-branch context. Runs once, after + ``_ensure_session_db_row`` wrote the row + parent link.""" if not session.get("parent_session_id") or session.get("_branch_seed_persisted"): return key = session.get("session_key") @@ -568,66 +404,39 @@ def _persist_branch_seed(session: dict) -> None: if db is None: return try: - # Bounded-chunk transactions (see #23254): a branch seed can be - # hundreds of rows; chunking keeps each BEGIN IMMEDIATE short so - # concurrent writers aren't starved. Recovery semantics match the - # old per-row loop (mid-copy failure leaves a partial seed with - # _branch_seed_persisted unset). + # Chunked so each BEGIN IMMEDIATE stays short (a seed can be hundreds of rows); a + # mid-copy failure leaves a partial seed with _branch_seed_persisted unset. db.append_messages_batch( key, [ - { - "role": msg.get("role", "user"), - "content": msg.get("content"), - "reasoning": msg.get("reasoning"), - "reasoning_content": msg.get("reasoning_content"), - "reasoning_details": msg.get("reasoning_details"), - "codex_reasoning_items": msg.get("codex_reasoning_items"), - "codex_message_items": msg.get("codex_message_items"), - # Timeline markers (model_switch, personality_switch, - # auto_continue, …) ride as role=user; dropping the tag - # here re-planted them as bare user turns after a - # restart, corrupting the truncate ordinal address - # space the same way #82756 did. - "display_kind": msg.get("display_kind"), - "display_metadata": msg.get("display_metadata"), - # Preserve the parent's original message timestamps — - # append_message would otherwise stamp time.time() and the - # branch's copied history would all appear authored "now". - "timestamp": msg.get("timestamp"), - } + {"role": msg.get("role", "user"), **{f: msg.get(f) for f in _WORKDIR_SEED_FIELDS}} for msg in seed ], chunk_rows=500, ) session["_branch_seed_persisted"] = True except Exception as exc: - from hermes_state import is_disk_full_error + _workdir_reraise_disk_full(exc, "branch seed persist failed") - if is_disk_full_error(exc): - raise - logger.debug("branch seed persist failed", exc_info=True) + +# Yielded by _workdir_owner_db when the profile db failed to OPEN (as opposed +# to "no store in this context"); _ensure_session_db_row fails loud on it. +_WORKDIR_DB_OPEN_FAILED = object() @contextlib.contextmanager -def _session_db(session: dict): - """Yield the SessionDB that owns this session's row (profile-aware). - - Mirrors :func:`_ensure_session_db_row`: a remote/profile session persists - into its own profile's ``state.db`` (a fresh handle we close on exit); - everything else borrows the shared ``_get_db()`` handle (left open). Yields - None when the db is unavailable. - """ +def _workdir_owner_db(session: dict, fail_log: str): + """Body of :func:`_session_db`; also used directly by ``_ensure_session_db_row`` + so a test-patched ``_session_db`` does not change row creation.""" db, close_db = None, False profile_home = session.get("profile_home") if profile_home: - from hermes_state import SessionDB - try: from hermes_state import get_shared_session_db db, close_db = get_shared_session_db(Path(profile_home) / "state.db"), True except Exception: - logger.debug("failed to open profile db for session", exc_info=True) + logger.debug(fail_log, exc_info=True) + db = _WORKDIR_DB_OPEN_FAILED else: db = _get_db() try: @@ -639,20 +448,24 @@ def _session_db(session: dict): release_or_close(db) +@contextlib.contextmanager +def _session_db(session: dict): + """Yield the SessionDB that owns this session's row (profile-aware): a remote/profile session + persists into its own profile's ``state.db`` (fresh handle, closed on exit); everything else + borrows the shared ``_get_db()`` handle (left open). Yields None when unavailable.""" + with _workdir_owner_db(session, "failed to open profile db for session") as db: + yield None if db is _WORKDIR_DB_OPEN_FAILED else db + + def _rewind_active_session_history( - session: dict, - user_ordinal: int, - *, - require_retryable: bool = False, + session: dict, user_ordinal: int, *, require_retryable: bool = False ) -> tuple[list[dict], dict, int]: """Rewind one canonical user turn while retaining carrier scaffolding. - The caller holds ``history_lock``. Persistent sessions archive the target - and tail; a composite carrier's own hidden handoff is inserted in that same - transaction. Memory is installed only after the durable commit and is - built from the already-validated prefix plus the returned scaffold row id, - so there is no fallible post-commit reload. - """ + Caller holds ``history_lock``. Persistent sessions archive the target and tail, inserting a + composite carrier's hidden handoff in the same transaction; memory is installed only after + the durable commit, from the already-validated prefix plus the returned scaffold row id + (no fallible post-commit reload).""" from agent.context_compressor import ( history_before_user_originated_turn, retryable_user_text, @@ -660,10 +473,7 @@ def _rewind_active_session_history( user_originated_turn_view, ) from agent.memory_manager import sanitize_context - from agent.tool_dispatch_helpers import ( - _is_multimodal_tool_result, - _multimodal_text_summary, - ) + from agent.tool_dispatch_helpers import _is_multimodal_tool_result, _multimodal_text_summary def _comparison_content(message: dict) -> Any: content = message.get("content") @@ -672,24 +482,22 @@ def _rewind_active_session_history( elif isinstance(content, list): text_parts = [] for part in content: - if isinstance(part, dict) and part.get("type") == "text": + if not isinstance(part, dict): + continue + if part.get("type") == "text": text_parts.append(str(part.get("text", ""))) - elif ( - isinstance(part, dict) - and part.get("type") in {"image", "image_url", "input_image"} - ): + elif part.get("type") in {"image", "image_url", "input_image"}: text_parts.append("[screenshot]") content = "\n".join(text_parts) if text_parts else None if message.get("role") in {"user", "assistant"} and isinstance(content, str): return sanitize_context(content).strip() return content + def _user_indices(messages: list[dict]) -> list[int]: + return [i for i, m in enumerate(messages) if user_originated_turn_view(m) is not None] + history = _history_without_ephemeral_scaffolding(session.get("history", [])) - user_indices = [ - index - for index, message in enumerate(history) - if user_originated_turn_view(message) is not None - ] + user_indices = _user_indices(history) if user_ordinal < 0 or user_ordinal >= len(user_indices): raise ValueError("target user message is no longer in session history") target_index = user_indices[user_ordinal] @@ -703,28 +511,17 @@ def _rewind_active_session_history( if db is None: raise RuntimeError("session database is unavailable") expected_active_ids = db.get_active_message_ids(session_key) - durable = db.get_messages_as_conversation( - session_key, - include_row_ids=True, - ) - durable_user_indices = [ - index - for index, message in enumerate(durable) - if user_originated_turn_view(message) is not None - ] + durable = db.get_messages_as_conversation(session_key, include_row_ids=True) + durable_user_indices = _user_indices(durable) if len(durable_user_indices) != len(user_indices): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) + raise RuntimeError("session history changed before the rewind could be persisted") durable_target_index = durable_user_indices[user_ordinal] durable_target = durable[durable_target_index] durable_prefix, durable_live_view = history_before_user_originated_turn( durable, durable_target_index ) if _comparison_content(durable_live_view) != _comparison_content(live_view): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) + raise RuntimeError("session history changed before the rewind could be persisted") target_row_id = durable_target.get("_row_id") if not isinstance(target_row_id, int): raise RuntimeError("rewind target has no durable row identity") @@ -741,21 +538,16 @@ def _rewind_active_session_history( if scaffold is not None: replacement_id = result.get("replacement_message_id") if not isinstance(replacement_id, int): - raise RuntimeError( - "rewind commit did not return the replacement scaffold id" - ) + raise RuntimeError("rewind commit did not return the replacement scaffold id") durable_prefix[-1]["_row_id"] = replacement_id durable_prefix[-1]["_db_persisted"] = True installed[-1] = durable_prefix[-1] - # Current clients address destructive follow-ups by durable row id. - # Preserve the richer warm content (for example image parts), but - # copy row identities when the retained warm/durable shapes align. + # Clients address follow-ups by durable row id: keep the richer warm + # content but copy row identities when the shapes align. if len(installed) == len(durable_prefix) and all( warm.get("role") == durable_message.get("role") - and bool(warm.get("display_kind")) - == bool(durable_message.get("display_kind")) - and _comparison_content(warm) - == _comparison_content(durable_message) + and bool(warm.get("display_kind")) == bool(durable_message.get("display_kind")) + and _comparison_content(warm) == _comparison_content(durable_message) for warm, durable_message in zip(installed, durable_prefix) ): for warm, durable_message in zip(installed, durable_prefix): @@ -785,37 +577,23 @@ def _history_without_ephemeral_scaffolding(history: list[dict]) -> list[dict]: """Return the durable transcript shape without transient recovery rows.""" from run_agent import _is_ephemeral_scaffolding - return [ - message.copy() - for message in history - if not _is_ephemeral_scaffolding(message) - ] + return [message.copy() for message in history if not _is_ephemeral_scaffolding(message)] + + +def _workdir_valid_generation(generation) -> bool: + """A claimed DB probe generation: a positive int (bool excluded).""" + return not isinstance(generation, bool) and isinstance(generation, int) and generation >= 1 def _persist_session_git_meta(session: dict, cwd: str, generation: int) -> None: - """Resolve + persist a session's git branch / repo root WITHOUT blocking. - - Branch and root come from ``git`` subprocess probes; running them inline on - the session-init / cwd-set path would stall startup whenever ``cwd`` is slow - or on an unreachable mount. Run them on a short-lived daemon thread instead - and persist via the same profile-aware db the caller writes ``cwd`` to. - - Best-effort: ``cwd`` itself is persisted synchronously by the caller, so a - probe failure just leaves these enrichment columns unset (the project tree - falls back to its live resolver / lazy backfill). Daemon, so a mid-flight - probe never delays gateway shutdown. - """ + """Resolve + persist a session's git branch / repo root on a daemon thread: inline ``git`` + probes on the session-init / cwd-set path would stall startup on a slow or unreachable + ``cwd``. Persists via the same profile-aware db the caller wrote ``cwd`` to. Best-effort: a + probe failure leaves the enrichment columns unset (project tree uses its live resolver).""" session_key = session.get("session_key", "") - if ( - not session_key - or not cwd - or isinstance(generation, bool) - or not isinstance(generation, int) - or generation < 1 - ): + if not session_key or not cwd or not _workdir_valid_generation(generation): return - # Snapshot the routing fields now; the live session dict may be gone by the - # time the thread runs. `_session_db` reopens the profile-correct db inside. + # Snapshot routing fields; the live session dict may be gone when the thread runs. db_session = {"session_key": session_key, "profile_home": session.get("profile_home")} def _run() -> None: @@ -826,47 +604,26 @@ def _persist_session_git_meta(session: dict, cwd: str, generation: int) -> None: return with _session_db(db_session) as db: if db is not None: - db.publish_session_git_metadata( - session_key, - cwd, - generation, - branch, - root, - ) + db.publish_session_git_metadata(session_key, cwd, generation, branch, root) except Exception: logger.debug("failed to persist session git metadata", exc_info=True) threading.Thread(target=_run, name="git-meta", daemon=True).start() -def _persist_session_cwd_and_schedule_git_meta( - session: dict, - cwd: str, - *, - db=None, -) -> int | None: +def _persist_session_cwd_and_schedule_git_meta(session: dict, cwd: str, *, db=None) -> int | None: """Claim a DB-backed probe generation, then start Git enrichment.""" try: - if db is not None: - generation = db.update_session_cwd( - session.get("session_key", ""), cwd - ) - else: - with _session_db(session) as owner_db: - if owner_db is None: - return None - generation = owner_db.update_session_cwd( - session.get("session_key", ""), cwd - ) + owner = contextlib.nullcontext(db) if db is not None else _session_db(session) + with owner as owner_db: + if owner_db is None: + return None + generation = owner_db.update_session_cwd(session.get("session_key", ""), cwd) except Exception: logger.debug("failed to persist session cwd", exc_info=True) return None - if ( - isinstance(generation, bool) - or not isinstance(generation, int) - or generation < 1 - ): + if not _workdir_valid_generation(generation): return None _persist_session_git_meta(session, cwd, generation) return generation @@ -880,15 +637,13 @@ def _set_session_cwd(session: dict, cwd: str) -> str: if not os.path.isdir(resolved): raise ValueError(f"working directory does not exist: {cwd}") session["cwd"] = resolved - # An explicit user choice — persist it as the workspace (and let a later - # lazy row creation persist it too, not the launch-dir fallback). + # An explicit user choice: persisted as the workspace (not the launch-dir + # fallback) and superseding any settle-adopted cwd. session["explicit_cwd"] = True - # A user's choice supersedes any earlier settle-adopted cwd: from here on - # the terminal wandering must not move the workspace again. session["cwd_from_settle"] = False _register_session_cwd(session) - # The synchronous DB write claims ordering authority; Git subprocesses stay - # off the hot path and may publish only for that exact generation. + # The synchronous DB write claims ordering authority; git probes may + # publish only for that exact generation. _persist_session_cwd_and_schedule_git_meta(session, resolved) try: from tools.terminal_tool import cleanup_vm diff --git a/tui_gateway/tool_progress.py b/tui_gateway/tool_progress.py index 217a1c8657..25a51158ff 100644 --- a/tui_gateway/tool_progress.py +++ b/tui_gateway/tool_progress.py @@ -12,13 +12,15 @@ from .method_ctx import HandlerRegistry, bind_module _registry = HandlerRegistry() -# waste AND feeds the Ink render-tree blowup that silently OOM-killed the TUI -# parent (#34095). Cap here to match the render budget (a hair more, so the -# "[omitted …]" label is still informative when output is genuinely large). -# Full output stays in the agent context and the SQLite session, untouched. +# Verbose tool text is capped to the Ink render budget (a hair more, so the +# "[omitted …]" label stays informative): unbounded output fed a render-tree +# blowup that OOM-killed the TUI parent. Full output stays in the agent context +# and the SQLite session, untouched. _TUI_VERBOSE_TEXT_MAX_CHARS = 1_000 _TUI_VERBOSE_TEXT_MAX_LINES = 16 +_TODO_TOOL_NAMES = ("todo_list", "todo") # legacy alias: pre-rename replays + def _cap_tui_verbose_text(text: str) -> str: if ( @@ -48,8 +50,7 @@ def _cap_tui_verbose_text(text: str) -> str: omitted_lines = text[:start].count("\n") if omitted_lines: label = ( - "[showing verbose tail; omitted " - f"{omitted_lines} lines / {omitted_chars} chars]\n" + "[showing verbose tail; omitted " f"{omitted_lines} lines / {omitted_chars} chars]\n" ) else: label = f"[showing verbose tail; omitted {omitted_chars} chars]\n" @@ -141,14 +142,23 @@ def _normalize_todo_state(value: object) -> dict | None: except (TypeError, ValueError): return None todos = list(value["todos"]) - # Unused TodoStore snapshot() is {todos: [], revision: 0}. Attaching - # that on resume stamps a client watermark and blocks unversioned - # tool.start merges. An empty list at revision >= 1 is a real clear. + # Unused TodoStore snapshot() is {todos: [], revision: 0}: attaching it on + # resume stamps a client watermark and blocks unversioned tool.start merges. + # An empty list at revision >= 1 is a real clear. if not todos and revision == 0: return None return {"todos": todos, "revision": revision} +def _cache_todo_state(session: dict, state: dict | None) -> None: + """Keep the newest snapshot on the session (revision-monotonic).""" + if state is None: + return + cached = _normalize_todo_state(session.get("todo_state")) + if cached is None or state["revision"] >= cached["revision"]: + session["todo_state"] = state + + def _session_todo_state(session: dict) -> dict | None: """Return the newest live/cached todo snapshot for a runtime session.""" cached = _normalize_todo_state(session.get("todo_state")) @@ -180,15 +190,9 @@ def _attach_todo_state(payload: dict, session: dict) -> dict: def _todo_state_from_history(history) -> dict | None: - """Derive the latest todo snapshot from an already-loaded transcript. - - Used by resume paths that answer before an AIAgent (and its live - TodoStore) exists. The canonical todo tool results already persist in - conversation history as ordinary tool messages, so the latest one paired - with an assistant ``todo`` tool call IS the durable snapshot — no side - table and no extra transcript read (each resume path passes the history - it already loaded). - """ + """Latest todo snapshot from an already-loaded transcript, for resume paths + that answer before an AIAgent (and its live TodoStore) exists: the newest + tool result paired with an assistant ``todo`` call IS the durable snapshot.""" if not isinstance(history, list) or not history: return None try: @@ -199,7 +203,7 @@ def _todo_state_from_history(history) -> dict | None: if not isinstance(msg, dict): continue for call in msg.get("tool_calls") or []: - if (call.get("function") or {}).get("name") in ("todo_list", "todo"): + if (call.get("function") or {}).get("name") in _TODO_TOOL_NAMES: cid = call.get("id") if cid: todo_call_ids.add(cid) @@ -245,19 +249,16 @@ def _on_tool_start(sid: str, tool_call_id: str, name: str, args: dict): "name": name, "context": _tool_ctx(name, args), } - # The desktop renders the expanded tool row (the `$` transcript) from - # the args of the part, and `context` is an 80-char display preview. - # tool.complete already ships full args to every client. When - # tool.start ships them too, the expanded row is complete while the - # tool runs, at the cost of one duplicate transient payload per call. + # Full args here (not just the 80-char `context` preview) so the desktop's + # expanded tool row is complete while the tool runs; tool.complete ships + # them again. args.todos may be a partial merge — tool.complete is the + # source of truth for todos. if args: payload["args"] = args if _session_verbose(sid): args_text = _tool_args_text(args) if args_text: payload["args_text"] = args_text - # tool.complete is the source of truth for todos (full list from the - # tool result). args.todos here may be a partial merge update. _emit("tool.start", sid, payload) @@ -284,14 +285,12 @@ def _on_tool_complete(sid: str, tool_call_id: str, name: str, args: dict, result if result_text: payload["result_text"] = result_text todo_state = None - if name in ("todo_list", "todo"): # legacy alias: pre-rename replays + if name in _TODO_TOOL_NAMES: todo_state = _normalize_todo_state(payload.get("result")) if todo_state is not None: payload.update(todo_state) if session is not None: - cached = _normalize_todo_state(session.get("todo_state")) - if cached is None or todo_state["revision"] >= cached["revision"]: - session["todo_state"] = todo_state + _cache_todo_state(session, todo_state) try: from agent.display import render_edit_diff_with_delta @@ -310,16 +309,154 @@ def _on_tool_complete(sid: str, tool_call_id: str, name: str, args: dict, result _tool_progress_enabled(sid) or payload.get("inline_diff") or _tool_lifecycle_required_for_ui(name) - or name in ("todo_list", "todo") + or name in _TODO_TOOL_NAMES ): _emit("tool.complete", sid, payload) - # Task state is application data, not optional tool-progress chrome. A - # dedicated full-snapshot event lets every client reconcile immediately - # without interpreting provider text or partial merge arguments. + # Task state is application data, not tool-progress chrome: a dedicated + # full-snapshot event lets every client reconcile without parsing tool args. if todo_state is not None: _emit("todo.updated", sid, todo_state) +# ── _on_tool_progress dispatch ───────────────────────────────────────────── +# Each handler takes (sid, name, preview, kw). `tool.started` is dropped on +# purpose: _on_tool_start already emits the authoritative tool.start with the +# stable id and args; an id-less duplicate row makes the desktop live view +# diverge from hydrated history. + + +def _progress_output_risk(sid, name, preview, kw): + metadata = kw.get("risk_metadata") + if not isinstance(metadata, dict): + return + _emit("tool.output_risk", sid, { + "tool_id": str(kw.get("tool_call_id") or ""), "name": str(name), + "risk": str(metadata.get("risk") or "low"), + "findings": [str(item) for item in metadata.get("findings", [])], + "redacted": bool(metadata.get("redacted", False)), + }) + + +def _progress_reasoning(sid, name, preview, kw): + payload: dict[str, object] = {"text": str(preview)} + if _session_verbose(sid): + payload["verbose"] = True + _emit("reasoning.available", sid, payload) + + +def _progress_moa_reference(sid, name, preview, kw): + # MoA reference-model output, rendered as a labelled block before the + # aggregator's response. `name` is the slot label, `preview` the text. + ref_payload: dict[str, object] = { + "label": str(name), + "text": str(preview or ""), + } + if kw.get("moa_index") is not None: + ref_payload["index"] = kw.get("moa_index") + if kw.get("moa_count") is not None: + ref_payload["count"] = kw.get("moa_count") + _emit("moa.reference", sid, ref_payload) + + +def _progress_moa_aggregating(sid, name, preview, kw): + _emit("moa.aggregating", sid, {"aggregator": str(name or "")}) + + +def _progress_moa_progress(sid, name, preview, kw): + # Drives the status-bar `MOA: 2/3 refs done`; both counters required so the + # client renders deterministically. + refs_done = kw.get("moa_refs_done") + refs_total = kw.get("moa_refs_total") + if refs_done is None or refs_total is None: + return + _emit("moa.progress", sid, { + "label": str(name or ""), "refs_done": int(refs_done), "refs_total": int(refs_total), + }) + + +def _progress_moa_phase(sid, name, preview, kw): + # Currently only phase="aggregator" fires, once fan-out completes. + phase = kw.get("moa_phase") + if not phase: + return + phase_payload: dict[str, object] = {"phase": str(phase)} + refs_done = kw.get("moa_refs_done") + refs_total = kw.get("moa_refs_total") + if refs_done is not None: + phase_payload["refs_done"] = int(refs_done) + if refs_total is not None: + phase_payload["refs_total"] = int(refs_total) + if name: + phase_payload["aggregator"] = str(name) + _emit("moa.phase", sid, phase_payload) + + +# Per-branch rollups emitted on subagent.complete. +_SUBAGENT_INT_FIELDS = ("input_tokens", "output_tokens", "reasoning_tokens", "api_calls") + + +def _progress_subagent(sid, name, preview, kw, event_type): + # Identity fields are all optional: older emitters omit them and the TUI + # spawn tree falls back to flat rendering. + payload = { + "goal": str(kw.get("goal") or ""), "task_count": int(kw.get("task_count") or 1), + "task_index": int(kw.get("task_index") or 0), + } + for key in ("subagent_id", "parent_id", "child_session_id", "delegation_id"): + if kw.get(key): + payload[key] = str(kw[key]) + if kw.get("depth") is not None: + payload["depth"] = int(kw["depth"]) + if kw.get("model"): + payload["model"] = str(kw["model"]) + if kw.get("tool_count") is not None: + payload["tool_count"] = int(kw["tool_count"]) + if kw.get("toolsets"): + payload["toolsets"] = [str(t) for t in kw["toolsets"]] + for int_key in _SUBAGENT_INT_FIELDS: + val = kw.get(int_key) + if val is not None: + try: + payload[int_key] = int(val) + except (TypeError, ValueError): + pass + for key in ("files_read", "files_written"): + if kw.get(key): + payload[key] = [str(p) for p in kw[key]] + if kw.get("output_tail"): + payload["output_tail"] = list(kw["output_tail"]) # list of dicts + if name: + payload["tool_name"] = str(name) + if preview: + payload["text"] = str(preview) + if kw.get("status"): + payload["status"] = str(kw["status"]) + if kw.get("summary"): + payload["summary"] = str(kw["summary"]) + if kw.get("duration_seconds") is not None: + payload["duration_seconds"] = float(kw["duration_seconds"]) + if preview and event_type == "subagent.tool": + payload["tool_preview"] = str(preview) + payload["text"] = str(preview) + # subagent.text is the child's per-token reply, relayed solely to feed a + # watch window's live mirror (keyed off the child sid); on the parent it's + # hundreds of ignored frames, so skip that emit. + if event_type != "subagent.text": + _emit(event_type, sid, payload) + _mirror_subagent_to_child(event_type, payload) + + +# event_type -> (handler, requires) where `requires` names the arg that must be +# truthy for the row to be emitted at all ("name" / "preview" / None). +_PROGRESS_HANDLERS = { + "tool.output_risk": (_progress_output_risk, "name"), + "reasoning.available": (_progress_reasoning, "preview"), + "moa.reference": (_progress_moa_reference, "name"), + "moa.aggregating": (_progress_moa_aggregating, None), + "moa.progress": (_progress_moa_progress, None), "moa.phase": (_progress_moa_phase, None), +} + + def _on_tool_progress( sid: str, event_type: str, @@ -331,149 +468,15 @@ def _on_tool_progress( if not _tool_progress_enabled(sid): return if event_type == "tool.started" and name: - # `_on_tool_start` already emits the authoritative `tool.start` with - # the stable tool id and args. Emitting another id-less progress row - # here makes the desktop live view diverge from hydrated history. return - if event_type == "tool.output_risk" and name: - metadata = _kwargs.get("risk_metadata") - if not isinstance(metadata, dict): - return - payload: dict[str, object] = { - "tool_id": str(_kwargs.get("tool_call_id") or ""), - "name": str(name), - "risk": str(metadata.get("risk") or "low"), - "findings": [str(item) for item in metadata.get("findings", [])], - "redacted": bool(metadata.get("redacted", False)), - } - _emit("tool.output_risk", sid, payload) - return - if event_type == "reasoning.available" and preview: - payload: dict[str, object] = {"text": str(preview)} - if _session_verbose(sid): - payload["verbose"] = True - _emit("reasoning.available", sid, payload) - return - if event_type == "moa.reference" and name: - # MoA reference-model output — relay as a labelled block the Ink/desktop - # client renders before the aggregator's response (like a thinking - # block, tagged with the source model). `name` is the slot label, - # `preview` is the reference text. - ref_payload: dict[str, object] = { - "label": str(name), - "text": str(preview or ""), - } - if _kwargs.get("moa_index") is not None: - ref_payload["index"] = _kwargs.get("moa_index") - if _kwargs.get("moa_count") is not None: - ref_payload["count"] = _kwargs.get("moa_count") - _emit("moa.reference", sid, ref_payload) - return - if event_type == "moa.aggregating": - _emit("moa.aggregating", sid, {"aggregator": str(name or "")}) - return - if event_type == "moa.progress": - # Per-reference completion — drives the status-bar progress indicator - # (`MOA: 2/3 refs done`) requested in issue #59546. Only emitted when - # both counters are present so the client can render deterministically. - refs_done = _kwargs.get("moa_refs_done") - refs_total = _kwargs.get("moa_refs_total") - if refs_done is None or refs_total is None: - return - _emit( - "moa.progress", - sid, - { - "label": str(name or ""), - "refs_done": int(refs_done), - "refs_total": int(refs_total), - }, - ) - return - if event_type == "moa.phase": - # Phase transition — currently only ``phase="aggregator"`` fires once - # the fan-out completes and the aggregator is about to act. Tells the - # client which phase of the MoA pipeline is currently running so it - # can swap status-bar copy accordingly. - phase = _kwargs.get("moa_phase") - if not phase: - return - phase_payload: dict[str, object] = {"phase": str(phase)} - refs_done = _kwargs.get("moa_refs_done") - refs_total = _kwargs.get("moa_refs_total") - if refs_done is not None: - phase_payload["refs_done"] = int(refs_done) - if refs_total is not None: - phase_payload["refs_total"] = int(refs_total) - if name: - phase_payload["aggregator"] = str(name) - _emit("moa.phase", sid, phase_payload) + entry = _PROGRESS_HANDLERS.get(event_type) + if entry is not None: + handler, requires = entry + if requires is None or {"name": name, "preview": preview}[requires]: + handler(sid, name, preview, _kwargs) return if event_type.startswith("subagent."): - payload = { - "goal": str(_kwargs.get("goal") or ""), - "task_count": int(_kwargs.get("task_count") or 1), - "task_index": int(_kwargs.get("task_index") or 0), - } - # Identity fields for the TUI spawn tree. All optional — older - # emitters that omit them fall back to flat rendering client-side. - if _kwargs.get("subagent_id"): - payload["subagent_id"] = str(_kwargs["subagent_id"]) - if _kwargs.get("parent_id"): - payload["parent_id"] = str(_kwargs["parent_id"]) - if _kwargs.get("child_session_id"): - payload["child_session_id"] = str(_kwargs["child_session_id"]) - if _kwargs.get("delegation_id"): - payload["delegation_id"] = str(_kwargs["delegation_id"]) - if _kwargs.get("depth") is not None: - payload["depth"] = int(_kwargs["depth"]) - if _kwargs.get("model"): - payload["model"] = str(_kwargs["model"]) - if _kwargs.get("tool_count") is not None: - payload["tool_count"] = int(_kwargs["tool_count"]) - if _kwargs.get("toolsets"): - payload["toolsets"] = [str(t) for t in _kwargs["toolsets"]] - # Per-branch rollups emitted on subagent.complete (features 1+2+4). - for int_key in ( - "input_tokens", - "output_tokens", - "reasoning_tokens", - "api_calls", - ): - val = _kwargs.get(int_key) - if val is not None: - try: - payload[int_key] = int(val) - except (TypeError, ValueError): - pass - if _kwargs.get("files_read"): - payload["files_read"] = [str(p) for p in _kwargs["files_read"]] - if _kwargs.get("files_written"): - payload["files_written"] = [str(p) for p in _kwargs["files_written"]] - if _kwargs.get("output_tail"): - payload["output_tail"] = list(_kwargs["output_tail"]) # list of dicts - if name: - payload["tool_name"] = str(name) - if preview: - payload["text"] = str(preview) - if _kwargs.get("status"): - payload["status"] = str(_kwargs["status"]) - if _kwargs.get("summary"): - payload["summary"] = str(_kwargs["summary"]) - if _kwargs.get("duration_seconds") is not None: - payload["duration_seconds"] = float(_kwargs["duration_seconds"]) - if preview and event_type == "subagent.tool": - payload["tool_preview"] = str(preview) - payload["text"] = str(preview) - # subagent.text is the child's per-token reply, relayed solely to feed a - # watch window's live mirror. It is meaningless on the parent session - # (which shows the child via the spawn tree, not its reply body), so - # skip the parent emit — sending hundreds of ignored token frames there - # is wasted traffic and a trap for any future parent-side subagent - # catch-all. The mirror keys off the child sid and is unaffected. - if event_type != "subagent.text": - _emit(event_type, sid, payload) - _mirror_subagent_to_child(event_type, payload) + _progress_subagent(sid, name, preview, _kwargs, event_type) def register(server) -> None: