From eb74a00c71243bc67b4c7200f4e8e50862ee7df6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:42:24 -0700 Subject: [PATCH] refactor(cli): split HermesCLI into 10 cohesive mixins (cli.py 22,284 -> 9,150) 326 methods lifted by AST (bodies identical; ast.dump-verified) into hermes_cli/cli_{tui,status_bar,voice,model_switch,session,stream,modal, terminal,info,loops}_mixin.py. cli.py-internal symbols resolve via lazy 'from cli import ...' inside each method (no import cycle; patch('cli.X') keeps working). The three 'global' writers (_skill_commands, _cli_wake_owner) now write the cli module attribute explicitly so the origin's readers still see them. Dropped imports left unused in cli.py; kept display_hermes_home / build_welcome_banner as re-exports (mixins + tests resolve them via cli). Repointed two AST change-detector tests to cli_tui_mixin.py; one test fixture now keeps 'cli' in sys.modules across its patch.dict scope. --- cli.py | 13162 +---------------- hermes_cli/cli_info_mixin.py | 1333 ++ hermes_cli/cli_loops_mixin.py | 725 + hermes_cli/cli_modal_mixin.py | 1228 ++ hermes_cli/cli_model_switch_mixin.py | 1151 ++ hermes_cli/cli_session_mixin.py | 1831 +++ hermes_cli/cli_status_bar_mixin.py | 1597 ++ hermes_cli/cli_stream_mixin.py | 986 ++ hermes_cli/cli_terminal_mixin.py | 624 + hermes_cli/cli_tui_mixin.py | 3080 ++++ hermes_cli/cli_voice_mixin.py | 1022 ++ tests/cli/test_compress_type_ahead.py | 4 +- tests/cli/test_steer_inline_repaint_34569.py | 4 +- tests/cli/test_tool_progress_scrollback.py | 11 +- 14 files changed, 13604 insertions(+), 13154 deletions(-) create mode 100644 hermes_cli/cli_info_mixin.py create mode 100644 hermes_cli/cli_loops_mixin.py create mode 100644 hermes_cli/cli_modal_mixin.py create mode 100644 hermes_cli/cli_model_switch_mixin.py create mode 100644 hermes_cli/cli_session_mixin.py create mode 100644 hermes_cli/cli_status_bar_mixin.py create mode 100644 hermes_cli/cli_stream_mixin.py create mode 100644 hermes_cli/cli_terminal_mixin.py create mode 100644 hermes_cli/cli_tui_mixin.py create mode 100644 hermes_cli/cli_voice_mixin.py diff --git a/cli.py b/cli.py index ea6b178781..26ac227b7a 100644 --- a/cli.py +++ b/cli.py @@ -24,7 +24,6 @@ except ModuleNotFoundError: pass import logging -import copy import os import functools import shutil @@ -32,10 +31,8 @@ import sys import json import re import concurrent.futures -import base64 import atexit import errno -import tempfile import time import uuid import textwrap @@ -55,21 +52,22 @@ from hermes_cli.fallback_config import get_fallback_chain from hermes_cli.cli_agent_setup_mixin import CLIAgentSetupMixin from hermes_cli.cli_commands_mixin import CLICommandsMixin from hermes_cli.cli_billing_mixin import CLIBillingMixin +from hermes_cli.cli_loops_mixin import CLILoopsMixin +from hermes_cli.cli_info_mixin import CLIInfoMixin +from hermes_cli.cli_terminal_mixin import CLITerminalMixin +from hermes_cli.cli_modal_mixin import CLIModalMixin +from hermes_cli.cli_stream_mixin import CLIStreamMixin +from hermes_cli.cli_session_mixin import CLISessionMixin +from hermes_cli.cli_model_switch_mixin import CLIModelSwitchMixin +from hermes_cli.cli_voice_mixin import CLIVoiceMixin +from hermes_cli.cli_status_bar_mixin import CLIStatusBarMixin +from hermes_cli.cli_tui_mixin import CLITuiMixin from agent.interrupt_compat import request_hard_interrupt from agent.pet import render as pet_render # prompt_toolkit for fixed input area TUI -from prompt_toolkit.history import FileHistory -from prompt_toolkit.styles import Style as PTStyle from prompt_toolkit.patch_stdout import patch_stdout from prompt_toolkit.application import Application -from prompt_toolkit.layout import Layout, HSplit, Window, FormattedTextControl, ConditionalContainer, WindowAlign -from prompt_toolkit.layout.processors import Processor, Transformation, PasswordProcessor, ConditionalProcessor -from prompt_toolkit.filters import Condition -from prompt_toolkit.layout.dimension import Dimension -from prompt_toolkit.layout.menus import CompletionsMenu -from prompt_toolkit.widgets import TextArea -from prompt_toolkit.key_binding import KeyBindings from prompt_toolkit import print_formatted_text as _pt_print from prompt_toolkit.formatted_text import ANSI as _PT_ANSI try: @@ -215,14 +213,14 @@ def realign_markdown_tables(*args, **kwargs): # NOTE: `from agent.account_usage import ...` is deliberately NOT at module # top — it transitively pulls the OpenAI SDK chain (~230 ms cold) and is only # needed when the user runs `/limits`. Lazy-imported inside the handler below. -from hermes_cli.banner import _format_context_length, format_banner_version_label +from hermes_cli.banner import format_banner_version_label _COMMAND_SPINNER_FRAMES = ("⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏") # Load .env from ~/.hermes/.env first, then project root as dev fallback. # User-managed env files should override stale shell exports on restart. -from hermes_constants import get_hermes_home, display_hermes_home +from hermes_constants import get_hermes_home, display_hermes_home # noqa: F401 (mixins import via cli) from hermes_cli.env_loader import load_hermes_dotenv from utils import base_url_host_matches, base_url_hostname, fast_safe_load @@ -904,9 +902,7 @@ def get_toolset_for_tool(*args, **kwargs): return _get_toolset_for_tool(*args, **kwargs) -# Extracted CLI modules (Phase 3) -from hermes_cli.banner import build_welcome_banner -from hermes_cli.commands import SlashCommandCompleter, SlashCommandAutoSuggest +from hermes_cli.banner import build_welcome_banner # noqa: F401 (CLIInfoMixin imports via cli) def get_all_toolsets(*args, **kwargs): @@ -939,8 +935,6 @@ def get_job(*args, **kwargs): return _get_job(*args, **kwargs) -# Resource cleanup imports for safe shutdown (terminal VMs, browser sessions) -from hermes_cli.callbacks import prompt_for_secret def _cleanup_all_terminals(*args, **kwargs): @@ -3938,8 +3932,6 @@ _IMAGE_EXTENSIONS = frozenset({ }) -from hermes_constants import is_termux as _is_termux_environment - def _termux_example_image_path(filename: str = "cat.png") -> str: """Return a realistic example media path for the current Termux setup.""" @@ -5213,7 +5205,7 @@ def _append_blank_panel_line(lines, border_style: str, box_width: int) -> None: lines.append((border_style, "│" + (" " * box_width) + "│\n")) -class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): +class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMixin, CLIStatusBarMixin, CLIVoiceMixin, CLIModelSwitchMixin, CLISessionMixin, CLIStreamMixin, CLIModalMixin, CLITerminalMixin, CLIInfoMixin, CLILoopsMixin): """ Interactive CLI for the Hermes Agent. @@ -5929,1125 +5921,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): finally: self._active_session_lease = None - def _mark_terminal_io_broken(self, reason: str = "") -> None: - """Stop UI paints after the PTY/stdout becomes unusable (#81521).""" - if getattr(self, "_terminal_io_broken", False): - return - self._terminal_io_broken = True - try: - self._pet_stop_anim() - except Exception: - pass - logger.warning( - "Terminal I/O broken%s — freezing UI paints to avoid redraw storm (#81521)", - f" ({reason})" if reason else "", - ) - - def _invalidate(self, min_interval: float = 0.25) -> None: - """Throttled UI repaint for high-frequency background updates. - - Use this for spinner frames, streaming token flushes, and other - repaints that can fire many times per second — the throttle prevents - terminal blinking on slow/SSH connections, and the resize-recovery - guard avoids stamping footer/status-bar chrome into scrollback while a - SIGWINCH reflow is in flight. - - Do NOT use this for user-blocking modal prompts (approval / clarify / - sudo). Those are rare, one-shot, user-blocking events that must paint - immediately; route them through ``self._app.invalidate()`` directly, the - same way the modal key-binding handlers already do. Sending a modal's - entry paint through this throttle lets an unrelated background repaint - within the 250ms window — or an in-flight resize — silently drop it, so - the prompt never renders and times out unseen (#41098). - """ - if getattr(self, "_terminal_io_broken", False): - return - if getattr(self, "_resize_recovery_pending", False): - return - now = time.monotonic() - if hasattr(self, "_app") and self._app and (now - getattr(self, "_last_invalidate", 0.0)) >= min_interval: - self._last_invalidate = now - try: - self._app.invalidate() - except OSError as exc: - if getattr(exc, "errno", None) == errno.EIO: - self._mark_terminal_io_broken("invalidate") - return - raise - - def _paint_now(self) -> None: - """Immediate, unthrottled repaint for user-blocking modal prompts. - - Background-thread callbacks (approval / clarify / sudo) set their modal - state then call this to make the panel visible at once. It deliberately - bypasses the ``_invalidate`` throttle and resize-recovery guard — a - modal the user is actively waiting on must never be dropped — mirroring - the direct ``event.app.invalidate()`` the modal key-binding handlers - already use. See ``_invalidate`` for why the throttle must not gate - these paints (#41098). - """ - if getattr(self, "_terminal_io_broken", False): - return - app = getattr(self, "_app", None) - if app is not None: - try: - app.invalidate() - except OSError as exc: - if getattr(exc, "errno", None) == errno.EIO: - self._mark_terminal_io_broken("paint_now") - return - raise - except Exception: - pass - - def _force_full_redraw(self) -> None: - """Force a clean full-screen repaint of the prompt_toolkit UI. - - Used to recover from terminal buffer drift caused by external - redraws we can't detect — e.g. macOS cmux / tmux tab switches, - ``clear`` issued from a subshell, or SSH window restores. These - wipe or repaint the terminal without firing SIGWINCH, so - prompt_toolkit's tracked ``_cursor_pos`` no longer matches reality - and the next incremental redraw stacks on top of stale content - (ghost status bars, duplicated prompts). - - Bound to Ctrl+L and exposed as the ``/redraw`` slash command, - matching the standard terminal-UX convention (bash, zsh, fish, - vim, htop). - """ - if getattr(self, "_terminal_io_broken", False): - return - app = getattr(self, "_app", None) - if not app: - return - self._clear_prompt_toolkit_screen( - app, - rebuild_scrollback=self._redraw_rebuilds_scrollback(), - ) - if getattr(self, "_terminal_io_broken", False): - return - _replay_output_history() - self._pet_queue_kitty_frame() - try: - app.invalidate() - except OSError as exc: - if getattr(exc, "errno", None) == errno.EIO: - self._mark_terminal_io_broken("force_full_redraw") - return - raise - except Exception: - pass - - def _schedule_focus_regain_redraw(self, min_interval: float = 1.0) -> None: - """Repaint after a terminal focus-in report (``CSI I``), rate-limited. - - Terminals with focus tracking active (Ghostty, iTerm2, xterm builds, - multiplexers that toggle DECSET 1004 upstream) emit ``\\x1b[I`` when - the Hermes tab/window becomes visible again. Emulators can coalesce - or drop hidden-tab output and repaint the surface while we're - invisible, so on regain prompt_toolkit's incremental diff stacks on - stale content — a second copy of the composer/prompt chrome next to - the ghost of the old one (#60920 focus-regain variant, #25337). - - The stock handling maps ``CSI I``/``CSI O`` to ``Keys.Ignore`` so the - bytes never pollute the input buffer; this hook additionally routes - focus-in through the same recovery as Ctrl+L / ``/redraw``. It is - self-gating: terminals that never enable focus tracking never emit - the sequence, so nothing changes for them. Rate-limited so a burst of - focus reports (rapid Alt+Tab, mux pane hops) repaints at most once - per ``min_interval`` seconds. - """ - now = time.monotonic() - last = getattr(self, "_last_focus_regain_redraw", 0.0) - if now - last < min_interval: - return - self._last_focus_regain_redraw = now - self._force_full_redraw() - - @staticmethod - def _redraw_rebuilds_scrollback() -> bool: - """Return whether CLI redraw/resize recovery should clear scrollback. - - Some terminal/tmux stacks move prompt_toolkit's non-fullscreen bottom - chrome into scrollback when the window is maximized/restored. A normal - CSI 2J viewport clear cannot remove those stale prompt/input-rule rows, - so users who hit that class of bug need CSI 3J as well, followed by the - existing bounded output-history replay. - """ - display_config = CLI_CONFIG.get("display") if isinstance(CLI_CONFIG, dict) else {} - if not isinstance(display_config, dict): - display_config = {} - raw = display_config.get("cli_rebuild_scrollback_on_redraw", False) - if isinstance(raw, str): - return raw.strip().lower() in {"1", "true", "yes", "on", "always"} - return bool(raw) - - def _recover_terminal_after_interrupt(self) -> None: - """Recover the terminal after an interrupted agent turn (#33271). - - When the user interrupts a running turn by typing a new message, - prompt_toolkit may have an in-flight ``CSI 6n`` cursor-position query - whose reply (``ESC[;R``) arrives on stdin after the input - parser has torn down. The reply then leaks as literal text - (``^[[19;1R``) and the VT100 parser can stall in a partial-escape - state, accepting no further keystrokes — the terminal appears frozen. - - Two steps recover a sane state: - 1. ``flush_stdin()`` drains stray escape bytes from the OS input - buffer (``termios.tcflush(TCIFLUSH)``; no-op on non-TTY). - 2. ``_force_full_redraw()`` drops prompt_toolkit's cached - screen/cursor state and forces a clean repaint. - - Both steps are independently safe and self-guard, so a failure of one - never prevents the other. If the PTY is already dead (EIO), skip the - redraw entirely — painting a broken fd is the #81521 redraw storm. - """ - if getattr(self, "_terminal_io_broken", False): - return - try: - from hermes_cli.curses_ui import flush_stdin - flush_stdin() - except Exception: - pass - # #60920: The interruption marker is now printed with - # _suspend_output_history in chat(), so _OUTPUT_HISTORY only - # contains the normal response text (no marker text). Do NOT - # clear history here — _force_full_redraw → _replay_output_history - # replays the response correctly without duplicating the marker. - # The /redraw + Ctrl+L paths also preserve replay for scrollback - # recovery as intended. - self._force_full_redraw() - - def _clear_prompt_toolkit_screen(self, app, *, rebuild_scrollback: bool = False) -> None: - """Clear the terminal and reset prompt_toolkit renderer state.""" - if getattr(self, "_terminal_io_broken", False): - return - try: - renderer = app.renderer - out = renderer.output - out.reset_attributes() - out.erase_screen() - if rebuild_scrollback: - try: - out.write_raw("\x1b[3J") - except Exception: - pass - out.cursor_goto(0, 0) - out.flush() - # Drop prompt_toolkit's cached screen + cursor state so the - # next _redraw() starts from a known (0, 0) origin and - # re-renders every cell rather than diffing against stale. - renderer.reset(leave_alternate_screen=False) - except OSError as exc: - if getattr(exc, "errno", None) == errno.EIO: - self._mark_terminal_io_broken("clear_screen") - return - pass - except Exception: - pass - - def _recover_after_resize(self, app, original_on_resize) -> None: - """Recover a resized classic CLI without desynchronizing cursor state. - - Unlike _force_full_redraw, we do NOT clear the physical screen or - scrollback here. The startup banner and tool summary are printed - before prompt_toolkit owns the live chrome, so they live in normal - terminal scrollback. Erasing the screen on SIGWINCH removes that - startup UI and ``_replay_output_history`` cannot reconstruct it - (the banner was never added to ``_OUTPUT_HISTORY``). - - Let prompt_toolkit's own resize path run with its renderer cursor - cache intact. Its Application._on_resize() starts with - renderer.erase(leave_alternate_screen=False), which needs the cached - cursor position to move back to the live prompt origin before - erase_down(). Resetting the renderer before that erase loses the - origin and can leave stale prompt glyphs after a narrow resize. - - We also flag ``_status_bar_suppressed_after_resize`` so the dynamic - status bar and input separator rules stay hidden while the terminal - reflow settles. On column shrink the terminal reflows already-rendered - status bar rows into scrollback before prompt_toolkit can erase them; - drawing a fresh full-width bar immediately makes the old and new - versions look duplicated (#19280, #22976). - - Suppression alone is not enough on a WIDTH change. prompt_toolkit's - ``renderer.erase()`` does ``cursor_up(_cursor_pos.y)`` + ``erase_down()`` - using the ``_cursor_pos.y`` cached from the LAST render at the OLD - width (renderer.py). When the column count shrinks, the terminal - reflows each already-painted full-width chrome row into 2+ physical - rows, so the cached ``y`` undershoots: ``cursor_up`` does not climb - past the reflowed rows and ``erase_down`` leaves the stale bar stranded - ABOVE the live origin. The next paint then stacks a fresh bar below it - — the duplicated-status-bar report (two bars, two elapsed readings). - Suppression hides the *new* bar but never erases the already-reflowed - *old* one, so the ghost survives the whole suppression window. - - Fix: on a width change, wipe the visible viewport with ``erase_screen`` - (CSI 2J) BEFORE delegating to prompt_toolkit's resize, then let its - repaint redraw from a clean origin. This is banner-safe: 2J clears - only the visible screen, NOT scrollback history (that is CSI 3J, which - we do not send here — ``rebuild_scrollback=False``), so the startup - banner that scrolled into history is preserved and - ``_replay_output_history`` is not needed. Row-count-only changes skip - the clear (no reflow, so no ghost) to avoid an unnecessary repaint. - - The suppression is transient: a short follow-up timer clears it and - repaints once the reflow has settled, so the bar returns on its own - during idle. Previously the flag was only cleared on the next - *submitted* user input, so a resize/reflow (tmux pane change, SSH - window restore, font zoom) followed by idle left the status bar hidden - indefinitely even while the refresh clock kept ticking (the dynamic - chrome rendered at height 0 on every repaint). The next-submit clear - at the input loop remains as a fast path. - """ - self._status_bar_suppressed_after_resize = True - # On a WIDTH change the terminal has already reflowed the old full-width - # chrome into extra physical rows that prompt_toolkit's stale-cursor - # erase (cursor_up(_cursor_pos.y) cached at the OLD width) will not - # reach, leaving a duplicated status bar stranded above the live origin. - # Ctrl+L / /redraw clears it cleanly, so route the resize path through - # the SAME recovery: wipe the visible viewport (banner-safe — CSI 2J - # by default; CSI 3J only when display.cli_rebuild_scrollback_on_redraw - # is enabled) and replay the transcript so nothing is lost. - # Same-width SIGWINCH (tmux attach, benign focus/tab signals) is left - # untouched — no clear, no replay — because a 2J without replay erases - # the visible transcript and a replay against preserved scrollback - # duplicates it (#65293). The stale-previous_screen crash tmux attach - # used to trigger is handled by _hermes_call_output_screen_diff's - # retry-with-first-paint instead (#83874). - try: - new_width = self._get_tui_terminal_width() - except Exception: - new_width = None - prev_width = getattr(self, "_last_resize_width", None) - # Replay only on an OBSERVED width change. The first signal of a - # session must not count as one (#65293): GNOME Terminal and friends - # deliver benign SIGWINCHes (tab bar appearing, monitor-scale change, - # focus events), and a 2J+replay against preserved scrollback - # duplicates everything ``_OUTPUT_HISTORY`` holds — after a resume - # that is the entire "Previous Conversation" recap plus the first - # live exchange. ``_install_resize_recovery`` seeds the baseline at - # startup, so an initial maximize/restore still differs from it and - # is still recovered; with no baseline (width probe failed) this - # signal just records one for the next comparison. - width_changed = ( - new_width is not None - and prev_width is not None - and new_width != prev_width - ) - if width_changed: - try: - self._clear_prompt_toolkit_screen( - app, - rebuild_scrollback=self._redraw_rebuilds_scrollback(), - ) - _replay_output_history() - except Exception: - pass - if new_width is not None: - self._last_resize_width = new_width - if width_changed: - self._pet_queue_kitty_frame() - original_on_resize() - self._schedule_status_bar_unsuppress(app) - - def _schedule_status_bar_unsuppress(self, app, delay: float = 0.35) -> None: - """Clear the post-resize status-bar suppression after the reflow settles. - - Debounced: a fresh resize cancels the pending unsuppress and restarts - the timer, so a resize storm only repaints the bar once it stops. - """ - try: - old_timer = getattr(self, "_status_bar_unsuppress_timer", None) - if old_timer is not None: - try: - old_timer.cancel() - except Exception: - pass - - def _clear(): - self._status_bar_suppressed_after_resize = False - try: - app.invalidate() - except Exception: - pass - - def _fire(): - try: - loop = getattr(app, "loop", None) - except Exception: - loop = None - if loop is not None: - try: - loop.call_soon_threadsafe(_clear) - return - except Exception: - pass - _clear() - - timer = threading.Timer(delay, _fire) - timer.daemon = True - self._status_bar_unsuppress_timer = timer - timer.start() - except Exception: - # Fail open: never leave the bar stuck hidden. - self._status_bar_suppressed_after_resize = False - - def _schedule_resize_recovery(self, app, original_on_resize, delay: float = 0.12) -> None: - """Debounce resize redraws so footer chrome is not stamped into scrollback.""" - try: - old_timer = getattr(self, "_resize_recovery_timer", None) - lock = getattr(self, "_resize_recovery_lock", None) - if lock is None: - lock = threading.Lock() - self._resize_recovery_lock = lock - - def _timer_fired(timer_ref): - def _run_recovery(): - with lock: - if getattr(self, "_resize_recovery_timer", None) is not timer_ref: - return - self._resize_recovery_timer = None - self._resize_recovery_pending = False - self._recover_after_resize(app, original_on_resize) - - try: - loop = app.loop # type: ignore[attr-defined] - except Exception: - loop = None - if loop is not None: - try: - loop.call_soon_threadsafe(_run_recovery) - return - except Exception: - pass - _run_recovery() - - with lock: - if old_timer is not None: - try: - old_timer.cancel() - except Exception: - pass - self._resize_recovery_pending = True - timer = threading.Timer(delay, lambda: _timer_fired(timer)) - timer.daemon = True - self._resize_recovery_timer = timer - timer.start() - except Exception: - self._resize_recovery_pending = False - self._recover_after_resize(app, original_on_resize) - - def _install_resize_recovery(self, app) -> None: - """Route prompt_toolkit's ``_on_resize`` through the debounced - ghost-clearing recovery (#5474/#49120) and record the current terminal - width as the baseline for width-change detection. - - Seeding the baseline here is what keeps the session's FIRST SIGWINCH - honest (#65293): ``_recover_after_resize`` replays the transcript only - on an observed width change, and without a startup baseline it could - not tell a benign signal (GNOME Terminal tab bar, monitor-scale - change) from a real one. An initial maximize/restore still differs - from the seeded width, so it is still recovered. - - The probe reads ``app.output`` directly — NOT - ``_get_tui_terminal_width`` — because this runs before ``app.run()``, - when ``get_app()`` still returns prompt_toolkit's DummyApplication - whose DummyOutput reports a hardcoded 80 columns; seeding that fake - width would make the first real signal look like a width change and - resurrect the duplicate-replay bug this exists to fix. - ``app.output`` is the same object the running app's resize handler - measures, so install-time and signal-time widths are comparable. - """ - width = None - try: - width = app.output.get_size().columns - except Exception: - width = None - if not width or width <= 0: - try: - width = shutil.get_terminal_size((80, 24)).columns - except Exception: - width = None - self._last_resize_width = width - original_on_resize = app._on_resize - - def _resize_clear_ghosts(): - self._schedule_resize_recovery(app, original_on_resize) - - app._on_resize = _resize_clear_ghosts - - def _status_bar_context_style(self, percent_used: Optional[int]) -> str: - if percent_used is None: - return "class:status-bar-dim" - if percent_used >= 95: - return "class:status-bar-critical" - if percent_used > 80: - return "class:status-bar-bad" - if percent_used >= 50: - return "class:status-bar-warn" - return "class:status-bar-good" - - def _cache_hit_rate(self, snapshot: dict, precision: int = 1) -> "tuple[float, str] | None": - """Return (cache_pct, formatted_label) or None if no cache data. - - Centralises the cache-hit-rate computation so both the plain-text - status bar and the prompt-toolkit fragment path share one formula. - Prefers the baseline-delta percentage computed in - ``_get_status_bar_snapshot`` (resets on model switch / compression, - so it reflects the *current* cache regime); falls back to the - session-lifetime ratio when no delta is available. - """ - delta_pct = snapshot.get("cache_hit_pct") - if delta_pct is not None: - return float(delta_pct), f"◎ {float(delta_pct):.{precision}f}%" - cache_read = snapshot.get("session_cache_read_tokens", 0) - prompt_total = snapshot.get("session_prompt_tokens", 0) - if cache_read > 0 and prompt_total > 0: - cache_pct = cache_read / prompt_total * 100 - return cache_pct, f"◎ {cache_pct:.{precision}f}%" - return None - - def _cache_hit_rate_style(self, cache_pct: float) -> str: - """Style for cache hit rate — higher is better (opposite of context %).""" - if cache_pct >= 70: - return "class:status-bar-good" - if cache_pct >= 40: - return "class:status-bar-warn" - return "class:status-bar-bad" - - - @staticmethod - def _battery_status_style(category: str) -> str: - """Map a battery colour category to a status-bar style class.""" - return { - "good": "class:status-bar-good", - "warn": "class:status-bar-warn", - "bad": "class:status-bar-bad", - "critical": "class:status-bar-critical", - }.get(category, "class:status-bar-dim") - - def _handle_battery_command(self, cmd_original: str) -> None: - """Toggle the status-bar battery read-out. - - ``/battery`` toggles, ``/battery on|off`` sets explicitly, and - ``/battery status`` reports the current setting plus a live reading. - The choice is persisted to ``display.battery`` so it survives restarts. - """ - parts = (cmd_original or "").split() - arg = parts[1].strip().lower() if len(parts) > 1 else "" - - try: - from agent.battery import format_battery, read_battery - reading = read_battery(use_cache=False) - except Exception: - reading = None - - if arg in ("status", "show"): - state = "on" if self._battery_visible else "off" - if reading is not None and reading.available: - self._console_print( - f" Battery indicator {state} — currently {format_battery(reading)}" - ) - elif reading is not None: - self._console_print( - f" Battery indicator {state} — no battery detected on this machine" - ) - else: - self._console_print(f" Battery indicator {state}") - return - - if arg in ("on", "true", "yes"): - target = True - elif arg in ("off", "false", "no"): - target = False - elif arg in ("", "toggle"): - target = not self._battery_visible - else: - self._console_print(" Usage: /battery [on|off|status]") - return - - self._battery_visible = target - save_config_value("display.battery", target) - - if target: - if reading is not None and not reading.available: - self._console_print( - " Battery indicator on — no battery detected, so nothing will show here" - ) - elif reading is not None and reading.available: - self._console_print( - f" Battery indicator on — {format_battery(reading)}" - ) - else: - self._console_print(" Battery indicator on") - else: - self._console_print(" Battery indicator off") - - @staticmethod - def _compression_count_style(count: int) -> str: - """Return a style class reflecting context compression pressure.""" - if count >= 10: - return "class:status-bar-bad" - if count >= 5: - return "class:status-bar-warn" - return "class:status-bar-dim" - - def _build_context_bar(self, percent_used: Optional[int], width: int = 10) -> str: - safe_percent = max(0, min(100, percent_used or 0)) - filled = round((safe_percent / 100) * width) - return f"[{('█' * filled) + ('░' * max(0, width - filled))}]" - - @staticmethod - def _format_prompt_elapsed(prompt_start_time: Optional[float], prompt_duration: float, live: bool = False) -> str: - """Format per-prompt elapsed time for the status bar. - - Always returns a string — shows 0s on fresh start before first turn. - Keeps seconds visible at all scales so it increments smoothly: - 59s → 1m → 1m 1s → ... → 1m 59s → 2m → 2m 1s → ... - 59m 59s → 1h → 1h 0m 1s → ... - 23h 59m 59s → 1d → 1d 0h 1m → ... - - Emoji prefix: ⏱ when turn is live, ⏲ when frozen or fresh start. - Uses width-1 (no variation selector) glyphs so the status bar stays - aligned in monospace terminals. - """ - if prompt_start_time is None and prompt_duration == 0.0: - return "⏲ 0s" - elapsed = time.time() - prompt_start_time if prompt_start_time is not None else prompt_duration - elapsed = max(0.0, elapsed) - - days = int(elapsed // 86400) - remaining = elapsed % 86400 - hours = int(remaining // 3600) - remaining = remaining % 3600 - minutes = int(remaining // 60) - seconds = int(remaining % 60) - - if days > 0: - time_str = f"{days}d {hours}h {minutes}m" - elif hours > 0: - time_str = f"{hours}h {minutes}m {seconds}s" if seconds else f"{hours}h {minutes}m" - elif minutes > 0: - time_str = f"{minutes}m {seconds}s" if seconds else f"{minutes}m" - else: - time_str = f"{int(elapsed)}s" - - emoji = "⏱" if live else "⏲" - return f"{emoji} {time_str}" - - @staticmethod - def _format_idle_since(last_finished_at: Optional[float], turn_live: bool) -> str: - """Format time since the last final agent response for the status bar. - - Returns an empty string while a turn is live (the per-prompt elapsed - timer covers that case) or before the first turn has completed. - Compact read-out: ``✓ 42s`` / ``✓ 3m`` / ``✓ 1h 12m``. - """ - if turn_live or last_finished_at is None: - return "" - idle = max(0.0, time.time() - last_finished_at) - return f"✓ {format_duration_compact(idle)}" - - def _get_status_bar_snapshot(self) -> Dict[str, Any]: - # Prefer the agent's model name — it updates on fallback. - # self.model reflects the originally configured model and never - # changes mid-session, so the TUI would show a stale name after - # _try_activate_fallback() switches provider/model. - agent = getattr(self, "agent", None) - model_name = (getattr(agent, "model", None) or self.model or "unknown") - # Friendly display: prefer reverse-alias from config.yaml ``model_aliases:`` - # before slash/length truncation. This turns long Palantir RIDs like - # ``ri.language-model-service..language-model.anthropic-claude-4-7-opus`` - # into the user's chosen short name (e.g. ``opus-4.7``) in the status bar. - model_short = _reverse_alias_for_display(model_name) - if model_short == model_name: - model_short = model_name.split("/")[-1] if "/" in model_name else model_name - # Strip Palantir RID prefixes via the shared display formatter so - # this site and ``ModelSwitchResult`` confirmation can't drift. - from hermes_cli.model_switch import format_model_for_display - model_short = format_model_for_display(model_short) - if model_short.endswith(".gguf"): - model_short = model_short[:-5] - if len(model_short) > 26: - model_short = f"{model_short[:23]}..." - - elapsed_seconds = max(0.0, (datetime.now() - self.session_start).total_seconds()) - snapshot = { - "model_name": model_name, - "model_short": model_short, - "duration": format_duration_compact(elapsed_seconds), - "session_title": self._get_status_bar_session_title(), - "prompt_elapsed": self._format_prompt_elapsed( - getattr(self, "_prompt_start_time", None), - getattr(self, "_prompt_duration", 0.0), - live=getattr(self, "_prompt_start_time", None) is not None, - ), - "idle_since": self._format_idle_since( - getattr(self, "_last_turn_finished_at", None), - turn_live=getattr(self, "_prompt_start_time", None) is not None, - ), - "context_tokens": 0, - "context_length": None, - "context_percent": None, - "session_input_tokens": 0, - "session_output_tokens": 0, - "session_cache_read_tokens": 0, - "session_cache_write_tokens": 0, - "session_prompt_tokens": 0, - "session_completion_tokens": 0, - "session_total_tokens": 0, - "session_api_calls": 0, - "compressions": 0, - "active_background_tasks": 0, - "active_background_processes": 0, - "active_background_subagents": 0, - "battery_label": "", - "battery_category": "dim", - # Focus view badge (/focus). Persistent indicator so the reduced - # output mode is never invisible. Display-only. - "focus_label": "", - } - - try: - from hermes_cli.focus_view import focus_statusbar_segment - - snapshot["focus_label"] = focus_statusbar_segment( - bool(getattr(self, "_focus_view_enabled", False)) - ) - except Exception: - pass - - # Battery read-out (first status-bar element when enabled). Reads are - # memoised for a few seconds inside agent.battery, so polling it on - # every status-bar repaint is cheap. - if getattr(self, "_battery_visible", False): - try: - from agent.battery import ( - battery_category, - format_battery, - read_battery, - ) - - _batt = read_battery() - snapshot["battery_label"] = format_battery(_batt) - snapshot["battery_category"] = battery_category(_batt) - except Exception: - pass - - # Count live /bg tasks. The dict entry is removed in the - # task thread's finally block, so len() reflects truly-running tasks. - # len() on a CPython dict is atomic; safe to read without a lock. - try: - bg_tasks = getattr(self, "_background_tasks", None) - if bg_tasks: - snapshot["active_background_tasks"] = len(bg_tasks) - except Exception: - pass - - # Count live background terminal processes (terminal tool background - # sessions tracked by tools.process_registry). Cheap O(1) read. - try: - from tools.process_registry import process_registry - snapshot["active_background_processes"] = process_registry.count_running() - except Exception: - pass - - # Count live background/async subagents (delegate_task batches and - # background single delegations tracked by tools.async_delegation). - # active_count() iterates an in-memory records dict under a lock — - # cheap and only counts records still in the "running" state. - try: - from tools.async_delegation import active_count as _async_active_count - snapshot["active_background_subagents"] = _async_active_count() - except Exception: - pass - - # Standing /goal state (Ralph loop). GoalManager is cached on self and - # keeps its state in memory, so this is a cheap attribute read — no DB - # hit per repaint. Only an *active* goal earns a segment; paused/done - # goals stay out of the bar (matching the desktop's active-first row). - snapshot["goal_active"] = False - snapshot["goal_turns_used"] = 0 - snapshot["goal_max_turns"] = 0 - try: - goal_mgr = self._get_goal_manager() - if goal_mgr is not None and goal_mgr.is_active(): - goal_state = goal_mgr.state - snapshot["goal_active"] = True - snapshot["goal_turns_used"] = int(getattr(goal_state, "turns_used", 0) or 0) - snapshot["goal_max_turns"] = int(getattr(goal_state, "max_turns", 0) or 0) - except Exception: - pass - - - if not agent: - return snapshot - - snapshot["session_input_tokens"] = getattr(agent, "session_input_tokens", 0) or 0 - snapshot["session_output_tokens"] = getattr(agent, "session_output_tokens", 0) or 0 - snapshot["session_cache_read_tokens"] = getattr(agent, "session_cache_read_tokens", 0) or 0 - snapshot["session_cache_write_tokens"] = getattr(agent, "session_cache_write_tokens", 0) or 0 - snapshot["session_prompt_tokens"] = getattr(agent, "session_prompt_tokens", 0) or 0 - snapshot["session_completion_tokens"] = getattr(agent, "session_completion_tokens", 0) or 0 - snapshot["session_total_tokens"] = getattr(agent, "session_total_tokens", 0) or 0 - snapshot["session_api_calls"] = getattr(agent, "session_api_calls", 0) or 0 - - compressor = getattr(agent, "context_compressor", None) - if compressor: - # last_prompt_tokens is parked at the -1 sentinel right after a - # compression, until the next real API call reports a prompt count - # (awaiting_real_usage_after_compression). The status bar must not - # render that sentinel verbatim — it produced "-1/200K" / "-1%". - # Clamp it to 0 so the one transitional turn reads as empty context. - context_tokens = getattr(compressor, "last_prompt_tokens", 0) or 0 - if context_tokens < 0: - context_tokens = 0 - # Durable-transcript view: on reasoning models a long tool loop - # replays the current turn's thinking + scaffolding on every - # request, so the LAST request's prompt_tokens can exceed the - # durable transcript by hundreds of K — all of which evaporates - # at the turn boundary. Rendering that raw figure makes the bar - # sawtooth (e.g. 850K mid-turn -> 600K next turn) and reads as a - # broken compaction. Anchor the display on the turn's FIRST - # response (minimal replay) plus a delta estimate of messages - # appended since, excluding stale thinking. Display-only: the - # compression trigger keeps using real last-request usage. - try: - from agent.model_metadata import anchored_context_tokens - - _msgs = getattr(agent, "_session_messages", None) - _anchored = anchored_context_tokens( - _msgs if isinstance(_msgs, list) else [], - getattr(agent, "_turn_base_usage_anchor", None), - charge_stale_thinking=False, - ) - if _anchored is not None and _anchored > 0: - context_tokens = _anchored - except Exception: - pass - context_length = getattr(compressor, "context_length", 0) or 0 - if context_length < 0: - context_length = 0 - snapshot["context_tokens"] = context_tokens - snapshot["context_length"] = context_length or None - snapshot["compressions"] = getattr(compressor, "compression_count", 0) or 0 - if context_length: - snapshot["context_percent"] = max(0, min(100, round((context_tokens / context_length) * 100))) - - # -- Cache-hit ratio (delta since last reset) -- - # Reset baseline on model switch and on compression — both invalidate - # the prompt cache. Formula verified against live logs: - # hit = cache_read / prompt_tokens (prompt = input+cache_read+cache_write) - # see agent/conversation_loop.py:4314 cache=read/prompt (87%) - # and CanonicalUsage.prompt_tokens = input+read+write - try: - base_model = getattr(self, "_cache_hit_baseline_model", None) - base_prompt = int(getattr(self, "_cache_hit_baseline_prompt", 0) or 0) - base_read = int(getattr(self, "_cache_hit_baseline_read", 0) or 0) - base_comps = int(getattr(self, "_cache_hit_baseline_compressions", 0) or 0) - cur_model = snapshot.get("model_name") or model_name - cur_comps = int(snapshot.get("compressions", 0) or 0) - cur_prompt = int(snapshot.get("session_prompt_tokens", 0) or 0) - cur_read = int(snapshot.get("session_cache_read_tokens", 0) or 0) - if base_model is None: - self._cache_hit_baseline_model = cur_model - self._cache_hit_baseline_compressions = cur_comps - base_model = cur_model - base_comps = cur_comps - if cur_model != base_model: - self._cache_hit_baseline_model = cur_model - self._cache_hit_baseline_prompt = cur_prompt - self._cache_hit_baseline_read = cur_read - self._cache_hit_baseline_compressions = cur_comps - base_prompt = cur_prompt - base_read = cur_read - base_comps = cur_comps - if cur_comps != base_comps: - self._cache_hit_baseline_compressions = cur_comps - self._cache_hit_baseline_prompt = cur_prompt - self._cache_hit_baseline_read = cur_read - base_prompt = cur_prompt - base_read = cur_read - delta_prompt = cur_prompt - base_prompt - delta_read = cur_read - base_read - # A zero-read regime hides the segment entirely (no cache data - # is not the same as a 0% hit worth alarming about), and the pct - # stays a float so renderers control their own precision. - if delta_prompt > 0 and delta_read > 0: - pct = max(0.0, min(100.0, (delta_read / delta_prompt) * 100)) - snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct:.0f}%" - elif cur_prompt > 0 and cur_read > 0 and base_prompt == 0 and base_read == 0: - pct = max(0.0, min(100.0, (cur_read / cur_prompt) * 100)) - snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct:.0f}%" - else: - snapshot["cache_hit_pct"] = None - snapshot["cache_hit_label"] = "" - except Exception: - snapshot["cache_hit_pct"] = None - snapshot["cache_hit_label"] = "" - - # -- Rolling avg latency / velocity (last 10 calls) -- - # Reads the deque maintained in agent/conversation_loop.py (and - # agent_init). Codex app-server has no latency, so it stays hidden there. - try: - agent_obj = getattr(self, "agent", None) - lhist = list(getattr(agent_obj, "_api_latency_history", []) or []) if agent_obj else [] - ohist = list(getattr(agent_obj, "_api_output_history", []) or []) if agent_obj else [] - # Keep the two histories aligned (they are appended together). - n = min(len(lhist), len(ohist)) - if n: - lhist = lhist[-n:] - ohist = ohist[-n:] - # Simple mean for latency; sum/sum for velocity (true throughput, not mean of ratios). - avg_lat = sum(lhist) / len(lhist) if lhist else None - total_out = sum(ohist) - total_lat = sum(lhist) - avg_vel = (total_out / total_lat) if total_lat > 0 else None - # Guard against NaN / inf from weird provider timings (e.g. -0.8s in logs). - if avg_lat is not None and (avg_lat != avg_lat or avg_lat < 0 or avg_lat > 1e6): - avg_lat = None - if avg_vel is not None and (avg_vel != avg_vel or avg_vel < 0 or avg_vel > 1e6): - avg_vel = None - snapshot["avg_latency"] = float(avg_lat) if avg_lat is not None else None - snapshot["avg_latency_label"] = f"{avg_lat:.1f}s" if avg_lat is not None else "" - snapshot["avg_velocity"] = float(avg_vel) if avg_vel is not None else None - snapshot["avg_velocity_label"] = f"{avg_vel:.0f} t/s" if avg_vel is not None else "" - else: - snapshot["avg_latency"] = None - snapshot["avg_latency_label"] = "" - snapshot["avg_velocity"] = None - snapshot["avg_velocity_label"] = "" - except Exception: - snapshot["avg_latency"] = None - snapshot["avg_latency_label"] = "" - snapshot["avg_velocity"] = None - snapshot["avg_velocity_label"] = "" - - return snapshot - - def _get_status_bar_session_title(self) -> str: - """Return the current title without polling state.db on every repaint.""" - pending = str(getattr(self, "_pending_title", None) or "").strip() - session_id = str(getattr(self, "session_id", "") or "") - if pending: - self._status_bar_title_session_id = session_id - self._status_bar_title_cache = pending - self._status_bar_title_checked_at = time.monotonic() - return pending - - now = time.monotonic() - cached_session_id = getattr(self, "_status_bar_title_session_id", None) - checked_at = float(getattr(self, "_status_bar_title_checked_at", 0.0) or 0.0) - if cached_session_id == session_id and now - checked_at < 1.5: - return str(getattr(self, "_status_bar_title_cache", "") or "") - - title = "" - db = getattr(self, "_session_db", None) - if db is not None and session_id: - try: - title = str(db.get_session_title(session_id) or "").strip() - except Exception: - title = "" - self._status_bar_title_session_id = session_id - self._status_bar_title_cache = title - self._status_bar_title_checked_at = now - return title - - @staticmethod - def _status_bar_display_width(text: str) -> int: - """Return terminal cell width for status-bar text. - - len() is not enough for prompt_toolkit layout decisions because some - glyphs can render wider than one Python codepoint. Keeping the status - bar within the real display width prevents it from wrapping onto a - second line and leaving behind duplicate rows. - """ - try: - from prompt_toolkit.utils import get_cwidth - return get_cwidth(text or "") - except Exception: - return len(text or "") - - @classmethod - def _trim_status_bar_text(cls, text: str, max_width: int) -> str: - """Trim status-bar text to a single terminal row.""" - if max_width <= 0: - return "" - try: - from prompt_toolkit.utils import get_cwidth - except Exception: - get_cwidth = None - - if cls._status_bar_display_width(text) <= max_width: - return text - - ellipsis = "..." - ellipsis_width = cls._status_bar_display_width(ellipsis) - if max_width <= ellipsis_width: - return ellipsis[:max_width] - - out = [] - width = 0 - for ch in text: - ch_width = get_cwidth(ch) if get_cwidth else len(ch) - if width + ch_width + ellipsis_width > max_width: - break - out.append(ch) - width += ch_width - return "".join(out).rstrip() + ellipsis - - @classmethod - def _right_align_status_title(cls, text: str, title: str, width: int) -> str: - """Pin a bounded session-title badge to the far-right status-bar edge.""" - title = str(title or "").strip() - if not title or width < 24: - return cls._trim_status_bar_text(text, width) - - title_width = max(6, min(30, width // 3)) - badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " - suffix = f" ─{badge}" - left_width = max(0, width - cls._status_bar_display_width(suffix)) - left = cls._trim_status_bar_text(text.rstrip(), left_width) - padding = " " * max(0, left_width - cls._status_bar_display_width(left)) - return f"{left}{padding}{suffix}" - - @classmethod - def _right_align_status_title_fragments(cls, frags, title: str, width: int): - """Styled counterpart to :meth:`_right_align_status_title`.""" - title = str(title or "").strip() - if not title or width < 24: - return frags - - title_width = max(6, min(30, width // 3)) - badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " - suffix_width = cls._status_bar_display_width(" ─") + cls._status_bar_display_width(badge) - left_width = max(0, width - suffix_width) - trimmed = [] - used = 0 - for style, value in frags: - remaining = left_width - used - if remaining <= 0: - break - value_width = cls._status_bar_display_width(value) - if value_width <= remaining: - trimmed.append((style, value)) - used += value_width - continue - clipped = cls._trim_status_bar_text(value, remaining) - if clipped: - trimmed.append((style, clipped)) - used += cls._status_bar_display_width(clipped) - break - - if used < left_width: - trimmed.append(("class:status-bar-dim", " " * (left_width - used))) - trimmed.extend([ - ("class:status-bar-dim", " ─"), - ("class:status-bar-session-title", badge), - ]) - return trimmed - - @staticmethod - def _get_tui_terminal_width(default: tuple[int, int] = (80, 24)) -> int: - """Return the live prompt_toolkit width, falling back to ``shutil``. - - The TUI layout can be narrower than ``shutil.get_terminal_size()`` reports, - especially on Termux/mobile shells, so prefer prompt_toolkit's width whenever - an app is active. - """ - try: - from prompt_toolkit.application import get_app - return get_app().output.get_size().columns - except Exception: - return shutil.get_terminal_size(default).columns - - def _use_minimal_tui_chrome(self, width: Optional[int] = None) -> bool: - """Hide low-value chrome on narrow/mobile terminals to preserve rows.""" - if width is None: - width = self._get_tui_terminal_width() - return width < 64 - - @staticmethod - def _scrollback_box_width(width: Optional[int] = None) -> int: - """Return the full viewport width for printed scrollback box rules. - - Previously this clamped to ``max(32, min(width, 56))`` as a defense - against terminal-emulator reflow on column-shrink (#25975, salvaging - #24403). That clamp made response/reasoning borders look stubby on - any modern wide terminal. We now trust the prompt_toolkit - ``_output_screen_diff`` monkey-patch landed in #26137 (salvaging - #25981) to keep chrome out of scrollback in the first place, and - accept that an aggressive column-shrink may visually reflow already - printed Panel borders — that's a cosmetic artifact of stamped - scrollback history, not a live-render bug. - - A small floor (32 cols) is kept so the box still renders on tiny - terminals without negative ``'─' * (w - 2)`` math. - """ - if width is None: - try: - width = shutil.get_terminal_size((80, 24)).columns - except Exception: - width = 80 - return max(32, int(width or 80)) - - def _tui_input_rule_height(self, position: str, width: Optional[int] = None) -> int: - """Return the visible height for the top/bottom input separator rules.""" - if position not in {"top", "bottom"}: - raise ValueError(f"Unknown input rule position: {position}") - if getattr(self, "_status_bar_suppressed_after_resize", False): - return 0 - if position == "top": - return 1 - return 0 if self._use_minimal_tui_chrome(width=width) else 1 - - def _agent_spacer_height(self, width: Optional[int] = None) -> int: - """Return the spacer height shown above the status bar while the agent runs.""" - if not getattr(self, "_agent_running", False): - return 0 - return 0 if self._use_minimal_tui_chrome(width=width) else 1 - - def _spinner_widget_height(self, width: Optional[int] = None) -> int: - """Return the visible height for the spinner/status text line above the status bar.""" - spinner_line = self._render_spinner_text() - if not spinner_line: - return 0 - if self._use_minimal_tui_chrome(width=width): - return 0 - width = width or self._get_tui_terminal_width() - if width and width > 10: - import math - text_width = self._status_bar_display_width(spinner_line) - return max(1, math.ceil(text_width / width)) - return 1 - - def _render_spinner_text(self) -> str: - """Return the live spinner/status text exactly as rendered in the TUI.""" - txt = getattr(self, "_spinner_text", "") - if not txt: - return "" - flow = self._spinner_token_flow() - t0 = getattr(self, "_tool_start_time", 0) or 0 - if t0 > 0: - elapsed = time.monotonic() - t0 - if elapsed >= 60: - _m, _s = int(elapsed // 60), int(elapsed % 60) - # Fixed-width timer to avoid status-line wrap jitter while - # scrolling/repainting (e.g. 01m05s, 12m09s). - elapsed_str = f"{_m:02d}m{_s:02d}s" - else: - # Keep width stable before the 60s rollover as well. - elapsed_str = f"{elapsed:5.1f}s" - if flow: - return f" {txt} ({elapsed_str} · {flow})" - return f" {txt} ({elapsed_str})" - if flow: - return f" {txt} ({flow})" - return f" {txt}" - # ── Per-turn accounting (display.turn_summary / spinner_token_flow) ── # # Both features are CLI-only chrome. The tally is observed from the @@ -7056,83 +5929,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # agent's cumulative session counters (bumped per API call in # agent/conversation_loop.py) and subtracts a per-turn baseline. - def _spinner_token_flow(self) -> str: - """Cumulative output tokens for the running turn, for the spinner.""" - if not getattr(self, "_spinner_token_flow_enabled", False): - return "" - if not getattr(self, "_agent_running", False): - return "" - agent = getattr(self, "agent", None) - if agent is None: - return "" - try: - from agent.turn_summary import format_token_flow - - produced = (getattr(agent, "session_output_tokens", 0) or 0) - ( - getattr(self, "_turn_token_baseline", 0) or 0 - ) - return format_token_flow(produced) - except Exception: - return "" - - def _turn_summary_is_active(self) -> bool: - """Whether the per-turn summary line should render for this surface. - - Gated off for: the config key, quiet/tool-progress-off mode, and any - non-interactive path (single query, ``-Q``, gateway/messaging) — those - surfaces either want machine-readable output or carry their own footer. - """ - if not getattr(self, "_turn_summary_enabled", False): - return False - if getattr(self, "tool_progress_mode", "all") == "off": - return False - agent = getattr(self, "agent", None) - if agent is not None and getattr(agent, "quiet_mode", False): - return False - return bool(getattr(self, "_interactive_turn", False)) - - def _turn_summary_begin(self) -> None: - """Start per-turn accounting for the turn that is about to run.""" - try: - from agent.turn_summary import TurnSummaryCollector - - collector = getattr(self, "_turn_summary_collector", None) - if collector is None: - collector = TurnSummaryCollector() - self._turn_summary_collector = collector - collector.begin() - self._turn_summary_start = time.monotonic() - agent = getattr(self, "agent", None) - self._turn_token_baseline = ( - getattr(agent, "session_output_tokens", 0) or 0 - ) if agent is not None else 0 - except Exception: - self._turn_summary_collector = None - - def _turn_summary_record(self, function_name, result, is_error: bool) -> None: - """Feed one completed tool call into the active tally.""" - collector = getattr(self, "_turn_summary_collector", None) - if collector is None: - return - try: - collector.record_tool(function_name, result=result, is_error=bool(is_error)) - except Exception: - pass - - def _turn_summary_emit(self) -> None: - """Print the post-turn accounting line, when enabled for this surface.""" - collector = getattr(self, "_turn_summary_collector", None) - if collector is None or not self._turn_summary_is_active(): - return - try: - started = getattr(self, "_turn_summary_start", 0.0) or 0.0 - elapsed = max(0.0, time.monotonic() - started) if started else 0.0 - line = collector.render(elapsed) - if line: - _cprint(f" {_DIM}{line}{_RST}") - except Exception: - logger.debug("Turn summary render failed", exc_info=True) - # ── Petdex mascot (base-CLI pet pane) ─────────────────────────────── # # Parity with the TUI: a sprite in a prompt_toolkit window above the @@ -7144,1738 +5940,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): _PET_FRAME_INTERVAL = 0.16 _PET_CFG_INTERVAL = 2.5 - def _pet_clear_runtime(self) -> None: - """Drop renderer + queued Kitty state. Caller holds ``_pet_lock``.""" - self._pet_enabled = False - self._pet_renderer = None - self._pet_frames_cache.clear() - self._pet_kitty_cache.clear() - self._pet_kitty_pending = "" - self._pet_kitty_image_id = 0 - - def _pet_resolve_config(self) -> None: - """(Re)resolve the active pet from config — picks up live enable/disable/ - - switch made via ``/pet`` or ``hermes pets`` without a restart, mirroring - the TUI's steady poll. Cheap and fail-open: any problem disables the pet. - """ - try: - from agent.pet import constants, store - from hermes_cli.config import load_config - - cfg = load_config() - display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} - pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} - - from utils import is_truthy_value - - enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) - slug = str(pet_cfg.get("slug", "") or "") - scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) - cols = constants.resolve_cols(scale, pet_cfg.get("unicode_cols", 0)) - configured_mode = str(pet_cfg.get("render_mode", "auto") or "auto").lower() - # Placeholders only on kitty/Ghostty. WezTerm speaks kitty APC but - # not U+10EEEE — detect_terminal_graphics() still returns kitty - # there, which is why this gate is narrower. - use_kitty = configured_mode in ("", "auto", "kitty") and pet_render.supports_kitty_placeholders() - renderer_mode = "kitty" if use_kitty else "unicode" - - if not enabled or configured_mode == "off": - with self._pet_lock: - self._pet_clear_runtime() - return - - pet = store.resolve_active_pet(slug) - if pet is None or not pet.exists: - with self._pet_lock: - self._pet_clear_runtime() - return - - with self._pet_lock: - # Rebuild only when the resolved pet, mode, or geometry changes. - if ( - self._pet_renderer is None - or self._pet_slug != pet.slug - or self._pet_cols != cols - or self._pet_scale != scale - or self._pet_renderer.mode != renderer_mode - ): - self._pet_renderer = pet_render.PetRenderer( - str(pet.spritesheet), mode=renderer_mode, scale=scale, unicode_cols=cols - ) - self._pet_slug = pet.slug - self._pet_cols = cols - self._pet_scale = scale - self._pet_frames_cache.clear() - self._pet_kitty_cache.clear() - self._pet_kitty_pending = "" - self._pet_kitty_image_id = pet_render.kitty_image_id(pet.slug) - self._pet_frame_idx = 0 - self._pet_enabled = True - except Exception: - with self._pet_lock: - self._pet_clear_runtime() - - def _pet_flash(self, state: str, secs: float = 1.6) -> None: - """Briefly force a transient reaction (wave/jump/failed) before resting.""" - self._pet_event = state - self._pet_event_until = time.monotonic() + secs - - def _on_reaction(self, kind: str) -> None: - """User affection (ily / <3 / good bot), core-detected — the pet's share - of the vibe signal that plays hearts on the TUI/desktop. Flash a celebrate.""" - if kind == "vibe": - self._pet_flash("jump") - - def _pet_react_turn_end(self) -> None: - """Flash the end-of-turn beat: failed on error, jump on a finished plan, else wave.""" - if not self._pet_enabled: - return - from agent.pet.state import todos_all_done - - if self._pet_turn_error: - self._pet_flash("failed") - return - try: - store = getattr(self.agent, "_todo_store", None) - done = todos_all_done(store.read()) if store else False - except Exception: - done = False - self._pet_flash("jump" if done else "wave") - - def _derive_pet_state(self) -> str: - """Map current CLI activity to a pet animation state. - - A transient reaction beat (wave/jump/failed) wins while it's live; - otherwise the steady state comes from the shared - :func:`agent.pet.state.derive_pet_state` so the CLI can't drift from the - TUI/desktop priority order. - """ - if self._pet_event and time.monotonic() < self._pet_event_until: - return self._pet_event - self._pet_event = "" - from agent.pet.state import derive_pet_state - - # A live blocking modal (approval / clarify / sudo / secret / slash - # confirm) means the agent is paused on the user → the `waiting` pose, - # which outranks the in-flight signals in derive_pet_state. - awaiting_input = bool( - self._approval_state - or self._clarify_state - or self._sudo_state - or self._secret_state - or getattr(self, "_slash_confirm_state", None) - ) - - return derive_pet_state( - awaiting_input=awaiting_input, - busy=getattr(self, "_agent_running", False), - reasoning=self._pet_reasoning, - ).value - - def _pet_frames_for(self, state: str) -> list: - """Return (and cache) the half-block grids for one state.""" - cached = self._pet_frames_cache.get(state) - if cached is not None: - return cached - renderer = self._pet_renderer - if renderer is None: - return [] - try: - count = renderer.frame_count(state) or 1 - grids = [renderer.cells(state, i, cols=self._pet_cols) for i in range(count)] - except Exception: - grids = [] - self._pet_frames_cache[state] = grids - return grids - - def _pet_kitty_payload_for(self, state: str) -> dict | None: - """Return and cache a Kitty virtual-placeholder payload for *state*.""" - with self._pet_lock: - cached = self._pet_kitty_cache.get(state) - if cached is not None: - return cached - renderer = self._pet_renderer - image_id = self._pet_kitty_image_id - if renderer is None or renderer.mode != "kitty": - return None - try: - # PNG encoding is outside _pet_lock: first visit of a state must - # not stall the prompt under the lock. - payload = renderer.kitty_payload(state, image_id=image_id) - except Exception: - payload = None - if payload is not None: - payload = {**payload, "image_id": image_id} - with self._pet_lock: - if self._pet_renderer is renderer and self._pet_kitty_image_id == image_id: - self._pet_kitty_cache[state] = payload - return payload - - def _pet_queue_kitty_frame(self, state: str | None = None) -> None: - """Queue one virtual Kitty frame for the next prompt_toolkit render. - - No-op when the pet pane was never initialized (``__new__`` fixtures - and ``_force_full_redraw`` / resize recovery on a pet-less CLI). - """ - if not getattr(self, "_pet_enabled", False): - return - if state is None: - state = self._derive_pet_state() - payload = self._pet_kitty_payload_for(state) - if not payload or not payload.get("frames"): - return - with self._pet_lock: - if self._pet_renderer is not None and self._pet_renderer.mode == "kitty": - self._pet_kitty_pending = payload["frames"][self._pet_frame_idx % len(payload["frames"])] - - def _pet_flush_kitty_frame(self, app) -> None: - """Write a queued APC after prompt_toolkit has finished its screen diff.""" - with self._pet_lock: - frame = self._pet_kitty_pending - self._pet_kitty_pending = "" - if not frame: - return - try: - # U=1/q=2 leaves the cursor and input stream untouched. - app.output.write_raw(frame) - app.output.flush() - except (OSError, ValueError): - pass - - def _pet_fragments(self): - """Return prompt_toolkit FormattedText for the current pet frame, or [].""" - with self._pet_lock: - if not self._pet_enabled or self._pet_renderer is None: - return [] - state = self._derive_pet_state() - kitty = self._pet_renderer.mode == "kitty" - if kitty: - payload = self._pet_kitty_payload_for(state) - if not payload: - return [] - color = pet_render.kitty_color_hex(payload["image_id"]) - frags = [] - for y, row in enumerate(payload["placeholder"]): - if y: - frags.append(("", "\n")) - frags.append((f"fg:{color}", row)) - return frags - with self._pet_lock: - grids = self._pet_frames_for(state) - if not grids: - return [] - grid = grids[self._pet_frame_idx % len(grids)] - - frags = [] - for y, row in enumerate(grid): - if y: - frags.append(("", "\n")) - for top, bottom in row: - tr, tg, tb, ta = top - br, bg, bb, ba = bottom - top_op = ta >= 32 - bot_op = ba >= 32 - if not top_op and not bot_op: - frags.append(("", " ")) - elif top_op and bot_op: - frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x} bg:#{br:02x}{bg:02x}{bb:02x}", "▀")) - elif top_op: - # Upper half only — leave the lower half the terminal's bg - # instead of painting it black (cleaner on light themes). - frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x}", "▀")) - else: - frags.append((f"fg:#{br:02x}{bg:02x}{bb:02x}", "▄")) - return frags - - def _pet_widget_height(self) -> int: - """Visible rows for the pet window — 0 collapses it when no pet shows.""" - with self._pet_lock: - if not self._pet_enabled or self._pet_renderer is None: - return 0 - state = self._derive_pet_state() - kitty = self._pet_renderer.mode == "kitty" - if kitty: - payload = self._pet_kitty_payload_for(state) - return int(payload.get("rows", 0)) if payload else 0 - with self._pet_lock: - grids = self._pet_frames_for(state) - if not grids or not grids[0]: - return 0 - return len(grids[0]) - - def _pet_anim_loop(self) -> None: - """Advance the frame + invalidate on a timer while a pet is enabled.""" - while self._pet_anim_running: - time.sleep(self._PET_FRAME_INTERVAL) - if getattr(self, "_terminal_io_broken", False): - self._pet_anim_running = False - break - now = time.monotonic() - if now - self._pet_cfg_checked >= self._PET_CFG_INTERVAL: - self._pet_cfg_checked = now - self._pet_resolve_config() - if not self._pet_enabled: - continue - with self._pet_lock: - self._pet_frame_idx += 1 - kitty = self._pet_renderer is not None and self._pet_renderer.mode == "kitty" - if kitty: - self._pet_queue_kitty_frame() - app = getattr(self, "_app", None) - if app is not None: - try: - app.invalidate() - except OSError as exc: - if getattr(exc, "errno", None) == errno.EIO: - self._mark_terminal_io_broken("pet_anim") - break - except Exception: - pass - - def _pet_start_anim(self) -> None: - if self._pet_anim_running: - return - self._pet_resolve_config() - with self._pet_lock: - kitty = self._pet_enabled and self._pet_renderer is not None and self._pet_renderer.mode == "kitty" - if kitty: - self._pet_queue_kitty_frame() - self._pet_anim_running = True - self._pet_anim_thread = threading.Thread(target=self._pet_anim_loop, daemon=True) - self._pet_anim_thread.start() - - def _pet_stop_anim(self) -> None: - self._pet_anim_running = False - thread = self._pet_anim_thread - if thread is not None: - thread.join(timeout=0.3) - self._pet_anim_thread = None - - def _voice_record_key_label(self) -> str: - """Return the configured voice push-to-talk key formatted for UI. - - Shared helper so every voice-facing status line / placeholder / - recording hint advertises the SAME label as the registered - prompt_toolkit binding. - - Cached at startup (see ``set_voice_record_key_cache``) rather - than re-read per render. Two reasons (Copilot round-13 on - #19835): - - * The prompt_toolkit binding is registered once at session - start via ``@kb.add(_voice_key)``; re-reading config per - render meant the status bar could advertise a new shortcut - after a config edit while the actual binding was still the - startup chord — exactly the display/binding drift this PR - is trying to eliminate. - * The label is on the hot render path (status bar + composer - placeholder invalidated every 150ms during recording), so - reading config on every call added avoidable UI overhead. - """ - return getattr(self, "_voice_record_key_display_cache", None) or "Ctrl+B" - - def set_voice_record_key_cache(self, raw_key: object) -> None: - """Populate the voice label cache from a raw ``voice.record_key``. - - Called at CLI startup after the prompt_toolkit binding is - registered so the cached label always matches the live binding. - """ - try: - from hermes_cli.voice import format_voice_record_key_for_status - self._voice_record_key_display_cache = format_voice_record_key_for_status(raw_key) - except Exception: - self._voice_record_key_display_cache = "Ctrl+B" - - def _get_voice_status_fragments(self, width: Optional[int] = None): - """Return the voice status bar fragments for the interactive TUI.""" - width = width or self._get_tui_terminal_width() - compact = self._use_minimal_tui_chrome(width=width) - label = self._voice_record_key_label() - if self._voice_recording: - if compact: - return [("class:voice-status-recording", " ● REC ")] - return [("class:voice-status-recording", f" ● REC {label} to stop ")] - if self._voice_processing: - if compact: - return [("class:voice-status", " ◉ STT ")] - return [("class:voice-status", " ◉ Transcribing... ")] - if compact: - return [("class:voice-status", f" 🎤 {label} ")] - tts = " | TTS on" if self._voice_tts else "" - cont = " | Continuous" if self._voice_continuous else "" - return [("class:voice-status", f" 🎤 Voice mode{tts}{cont} — {label} to record ")] - - @staticmethod - def _status_bar_goal_segment(snapshot: Dict[str, Any]) -> str: - """Return the ``⊙ goal 3/20`` segment, or ``""`` when no goal is active. - - Active-goal-only by design: paused/done goals don't occupy status-bar - real estate (they already print their own glyph lines in the thread). - """ - if not snapshot.get("goal_active"): - return "" - used = snapshot.get("goal_turns_used") or 0 - max_turns = snapshot.get("goal_max_turns") or 0 - if max_turns: - return f"⊙ goal {used}/{max_turns}" - return "⊙ goal" - - def _get_status_bar_field_set(self) -> Optional[frozenset]: - """Return the set of visible status-bar fields from config. - - Reads ``display.status_bar.fields`` from the module-level - ``CLI_CONFIG`` (no per-render YAML parse — the status bar repaints - every frame). Returns ``None`` when the user has not customized the - bar (use built-in defaults, i.e. show everything), or a - ``frozenset`` of field names when the list is non-empty. - - Available fields: model, context_detail, context_pct, cache_hit, - latency, tps, compressions, bg_tasks, bg_processes, bg_subagents, - goal, duration, prompt_elapsed, idle_since, focus, yolo, stash, - battery, title, total_tokens. - ``total_tokens`` is opt-in only (never shown by default). - The field order is fixed; the config controls visibility only. - """ - if hasattr(self, "_status_bar_field_set_cache"): - return self._status_bar_field_set_cache - result = None - try: - display = CLI_CONFIG.get("display") if isinstance(CLI_CONFIG, dict) else None - status_bar = (display or {}).get("status_bar") if isinstance(display, dict) else None - fields = status_bar.get("fields") if isinstance(status_bar, dict) else None - if isinstance(fields, list) and fields: - result = frozenset(str(f) for f in fields) - except Exception: - result = None - self._status_bar_field_set_cache = result - return result - - def _build_status_bar_text(self, width: Optional[int] = None) -> str: - """Return a compact one-line session status string for the TUI footer.""" - try: - snapshot = self._get_status_bar_snapshot() - if width is None: - width = self._get_tui_terminal_width() - percent = snapshot["context_percent"] - percent_label = f"{percent}%" if percent is not None else "--" - duration_label = snapshot["duration"] - battery_label = snapshot.get("battery_label") or "" - battery_prefix = f"{battery_label} │ " if battery_label else "" - focus_label = snapshot.get("focus_label") or "" - session_title = snapshot.get("session_title") or "" - - yolo_active = self._is_session_yolo_active() - goal_segment = self._status_bar_goal_segment(snapshot) - field_set = self._get_status_bar_field_set() - - def _ok(name: str) -> bool: - return field_set is None or name in field_set - - if not _ok("title"): - session_title = "" - - if not _ok("goal"): - goal_segment = "" - if not _ok("focus"): - focus_label = "" - if width < 52: - segs = [] - if _ok("model"): - segs.append(f"⚕ {snapshot['model_short']}") - if _ok("duration"): - segs.append(duration_label) - if goal_segment: - segs.append(goal_segment) - if focus_label: - segs.append(focus_label) - if yolo_active and _ok("yolo"): - segs.append("⚠ YOLO") - text = battery_prefix + " · ".join(segs) if segs else f"{battery_prefix}⚕ {snapshot['model_short']}" - return self._right_align_status_title(text, session_title, width) - if width < 76: - parts = [] - if _ok("model"): - parts.append(f"⚕ {snapshot['model_short']}") - if _ok("context_pct"): - parts.append(percent_label) - cache = self._cache_hit_rate(snapshot, precision=0) - if cache and _ok("cache_hit"): - parts.append(cache[1]) - if battery_label: - parts.insert(0, battery_label) - compressions = snapshot.get("compressions", 0) - if compressions and _ok("compressions"): - parts.append(f"🗜️ {compressions}") - bg_count = snapshot.get("active_background_tasks", 0) - if bg_count and _ok("bg_tasks"): - parts.append(f"▶ {bg_count}") - bg_proc_count = snapshot.get("active_background_processes", 0) - if bg_proc_count and _ok("bg_processes"): - parts.append(f"⚙ {bg_proc_count}") - bg_subagent_count = snapshot.get("active_background_subagents", 0) - if bg_subagent_count and _ok("bg_subagents"): - parts.append(f"⛓ {bg_subagent_count}") - if goal_segment: - parts.append(goal_segment) - if _ok("duration"): - parts.append(duration_label) - if focus_label: - parts.append(focus_label) - if yolo_active and _ok("yolo"): - parts.append("⚠ YOLO") - if not parts: - parts = [f"⚕ {snapshot['model_short']}"] - return self._right_align_status_title(" · ".join(parts), session_title, width) - - parts = [] - if _ok("model"): - parts.append(f"⚕ {snapshot['model_short']}") - if _ok("context_detail"): - if snapshot["context_length"]: - ctx_total = _format_context_length(snapshot["context_length"]) - ctx_used = format_token_count_compact(snapshot["context_tokens"]) - context_label = f"{ctx_used}/{ctx_total}" - else: - context_label = "ctx --" - parts.append(context_label) - if _ok("context_pct"): - parts.append(percent_label) - if battery_label: - parts.insert(0, battery_label) - compressions = snapshot.get("compressions", 0) - cache = self._cache_hit_rate(snapshot) - if cache and _ok("cache_hit"): - parts.append(cache[1]) - _avg_lat = snapshot.get("avg_latency_label") or "" - if _avg_lat and _ok("latency"): - parts.append(f"◷ {_avg_lat}") - _avg_vel = snapshot.get("avg_velocity_label") or "" - if _avg_vel and _ok("tps"): - parts.append(f"↑ {_avg_vel}") - if compressions and _ok("compressions"): - parts.append(f"🗜️ {compressions}") - bg_count = snapshot.get("active_background_tasks", 0) - if bg_count and _ok("bg_tasks"): - parts.append(f"▶ {bg_count}") - bg_proc_count = snapshot.get("active_background_processes", 0) - if bg_proc_count and _ok("bg_processes"): - parts.append(f"⚙ {bg_proc_count}") - bg_subagent_count = snapshot.get("active_background_subagents", 0) - if bg_subagent_count and _ok("bg_subagents"): - parts.append(f"⛓ {bg_subagent_count}") - if goal_segment: - parts.append(goal_segment) - if _ok("duration"): - parts.append(duration_label) - prompt_elapsed = snapshot.get("prompt_elapsed") - if prompt_elapsed and _ok("prompt_elapsed"): - parts.append(prompt_elapsed) - idle_since = snapshot.get("idle_since") - if idle_since and _ok("idle_since"): - parts.append(idle_since) - if focus_label: - parts.append(focus_label) - if yolo_active and _ok("yolo"): - parts.append("⚠ YOLO") - # Session token total (Σ) — opt-in only via an explicit fields - # list, so default bars never widen. - total_tokens = snapshot.get("session_total_tokens", 0) - if total_tokens and field_set is not None and "total_tokens" in field_set: - parts.append(f"Σ{format_token_count_compact(total_tokens)}") - if not parts: - parts = [f"⚕ {snapshot['model_short']}"] - return self._right_align_status_title(" │ ".join(parts), session_title, width) - except Exception: - return f"⚕ {self.model if getattr(self, 'model', None) else 'Hermes'}" - - def _get_status_bar_fragments(self): - if not self._status_bar_visible or getattr(self, '_model_picker_state', None) or getattr(self, '_command_palette_state', None): - return [] - try: - snapshot = self._get_status_bar_snapshot() - # Use prompt_toolkit's own terminal width when running inside the - # TUI — shutil.get_terminal_size() can return stale or fallback - # values (especially on SSH) that differ from what prompt_toolkit - # actually renders, causing the fragments to overflow to a second - # line and produce duplicated status bar rows over long sessions. - width = self._get_tui_terminal_width() - duration_label = snapshot["duration"] - yolo_active = self._is_session_yolo_active() - goal_segment = self._status_bar_goal_segment(snapshot) - battery_label = snapshot.get("battery_label") or "" - battery_style = self._battery_status_style(snapshot.get("battery_category", "dim")) - focus_label = snapshot.get("focus_label") or "" - session_title = snapshot.get("session_title") or "" - field_set = self._get_status_bar_field_set() - - def _ok(name: str) -> bool: - return field_set is None or name in field_set - - if not _ok("title"): - session_title = "" - - if not _ok("goal"): - goal_segment = "" - if not _ok("focus"): - focus_label = "" - - def _append(frag_list, sep, *pieces): - if frag_list: - frag_list.append(("class:status-bar-dim", sep)) - frag_list.extend(pieces) - - if width < 52: - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("duration"): - _append(frags, " · ", ("class:status-bar-dim", duration_label)) - if goal_segment: - _append(frags, " · ", ("class:status-bar-strong", goal_segment)) - if focus_label: - _append(frags, " · ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - else: - percent = snapshot["context_percent"] - percent_label = f"{percent}%" if percent is not None else "--" - if width < 76: - compressions = snapshot.get("compressions", 0) - bg_count = snapshot.get("active_background_tasks", 0) - bg_proc_count = snapshot.get("active_background_processes", 0) - bg_subagent_count = snapshot.get("active_background_subagents", 0) - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("context_pct"): - _append(frags, " · ", (self._status_bar_context_style(percent), percent_label)) - cache = self._cache_hit_rate(snapshot, precision=0) - if cache and _ok("cache_hit"): - _append(frags, " · ", (self._cache_hit_rate_style(cache[0]), cache[1])) - if compressions and _ok("compressions"): - _append(frags, " · ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) - if bg_count and _ok("bg_tasks"): - _append(frags, " · ", ("class:status-bar-strong", f"▶ {bg_count}")) - if bg_proc_count and _ok("bg_processes"): - _append(frags, " · ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) - if bg_subagent_count and _ok("bg_subagents"): - _append(frags, " · ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) - if goal_segment: - _append(frags, " · ", ("class:status-bar-strong", goal_segment)) - if _ok("duration"): - _append(frags, " · ", ("class:status-bar-dim", duration_label)) - if focus_label: - _append(frags, " · ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - else: - bar_style = self._status_bar_context_style(percent) - compressions = snapshot.get("compressions", 0) - bg_count = snapshot.get("active_background_tasks", 0) - bg_proc_count = snapshot.get("active_background_processes", 0) - bg_subagent_count = snapshot.get("active_background_subagents", 0) - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("context_detail"): - if snapshot["context_length"]: - ctx_total = _format_context_length(snapshot["context_length"]) - ctx_used = format_token_count_compact(snapshot["context_tokens"]) - context_label = f"{ctx_used}/{ctx_total}" - else: - context_label = "ctx --" - _append(frags, " │ ", ("class:status-bar-dim", context_label)) - if _ok("context_pct"): - _append( - frags, - " │ ", - (bar_style, self._build_context_bar(percent)), - ("class:status-bar-dim", " "), - (bar_style, percent_label), - ) - cache = self._cache_hit_rate(snapshot) - if cache and _ok("cache_hit"): - _append(frags, " │ ", (self._cache_hit_rate_style(cache[0]), cache[1])) - _avg_lat = snapshot.get("avg_latency_label") or "" - if _avg_lat and _ok("latency"): - _append(frags, " │ ", ("class:status-bar-dim", f"◷ {_avg_lat}")) - _avg_vel = snapshot.get("avg_velocity_label") or "" - if _avg_vel and _ok("tps"): - _append(frags, " │ ", ("class:status-bar-dim", f"↑ {_avg_vel}")) - if compressions and _ok("compressions"): - _append(frags, " │ ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) - if bg_count and _ok("bg_tasks"): - _append(frags, " │ ", ("class:status-bar-strong", f"▶ {bg_count}")) - if bg_proc_count and _ok("bg_processes"): - _append(frags, " │ ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) - if bg_subagent_count and _ok("bg_subagents"): - _append(frags, " │ ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) - if goal_segment: - _append(frags, " │ ", ("class:status-bar-strong", goal_segment)) - if _ok("duration"): - _append(frags, " │ ", ("class:status-bar-dim", duration_label)) - # Position 7: per-prompt elapsed timer (live or frozen) - prompt_elapsed = snapshot.get("prompt_elapsed") - if prompt_elapsed and _ok("prompt_elapsed"): - _append(frags, " │ ", ("class:status-bar-dim", prompt_elapsed)) - # Position 8: idle time since the last final agent response - idle_since = snapshot.get("idle_since") - if idle_since and _ok("idle_since"): - _append(frags, " │ ", ("class:status-bar-dim", idle_since)) - # Persistent focus-view badge — so the reduced-output mode - # is never invisible (mirrors the YOLO badge convention). - if focus_label: - _append(frags, " │ ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " │ ", ("class:status-bar-yolo", "⚠ YOLO")) - # Session token total (Σ) — opt-in only via an explicit - # fields list, so default bars never widen. - total_tokens = snapshot.get("session_total_tokens", 0) - if total_tokens and field_set is not None and "total_tokens" in field_set: - _append(frags, " │ ", ("class:status-bar-dim", f"Σ{format_token_count_compact(total_tokens)}")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - - # Stash indicator (📌 N) — appended after all width tiers so the - # user always knows a parked draft exists, even on narrow - # terminals. Placed before the battery prepend so it stays at the - # right edge, and it is the first thing the width trim below drops - # if the bar genuinely cannot fit. - try: - stash_indicator = self._prompt_stash.indicator() - except Exception: - stash_indicator = "" - if stash_indicator and _ok("stash"): - # Insert before the trailing pad fragment so the bar keeps its - # one-cell right margin. - if frags and frags[-1] == ("class:status-bar", " "): - frags[-1:-1] = [ - ("class:status-bar-dim", " · "), - ("class:status-bar-strong", stash_indicator), - ] - else: - frags.append(("class:status-bar-dim", " · ")) - frags.append(("class:status-bar-strong", stash_indicator)) - - # Battery is the first status-bar element when enabled: prepend it - # ahead of the leading ⚕ marker in whichever width tier ran above. - if battery_label and _ok("battery"): - frags[0:0] = [ - ("class:status-bar", " "), - (battery_style, battery_label), - ("class:status-bar-dim", " │"), - ] - - frags = self._right_align_status_title_fragments(frags, session_title, width) - - total_width = sum(self._status_bar_display_width(text) for _, text in frags) - if total_width > width: - plain_text = "".join(text for _, text in frags) - trimmed = self._trim_status_bar_text(plain_text, width) - return [("class:status-bar", trimmed)] - return frags - except Exception: - return [("class:status-bar", f" {self._build_status_bar_text()} ")] - - @staticmethod - def _fmt_stash_age(stashed_at: float) -> str: - """Return human-readable age string for a stash entry.""" - import time as _t - secs = int(_t.monotonic() - stashed_at) - if secs < 10: - return "just now" - if secs < 90: - return f"{secs}s ago" - mins = secs // 60 - if mins < 60: - return f"{mins} min ago" - return f"{mins // 60}h ago" - - def _render_stash_panel(self, stash_list: list, cursor: int, width: int) -> list: - """Return prompt_toolkit formatted_text fragments for the stash panel box. - - Every horizontal measurement goes through ``_status_bar_display_width`` - (prompt_toolkit's ``get_cwidth``) rather than ``len()``. The header - contains 📌, which is one Python codepoint but two terminal cells; the - original PR chased that off-by-one through three successive - "subtract 1 from len()" commits. Measuring in display cells fixes it - for real and keeps CJK previews from bleeding past the right border. - """ - cw = self._status_bar_display_width - W = max(12, min(width - 4, 80)) - - n = len(stash_list) - hdr_prefix_str = f"╭─ 📌 Stash ({n} item{'s' if n != 1 else ''}) " - HDR_SUFFIX = " Ctrl+S ─╮" - FTR_PREFIX = "╰" - FTR_SUFFIX = " ↑↓ Enter=restore D=delete Esc ─╯" - - # On narrow terminals the full hint text is wider than the box itself. - # Drop to compact affordances rather than letting the frame bleed past - # the right edge (which is what made the panel look broken). - if cw(hdr_prefix_str) + cw(HDR_SUFFIX) > W: - hdr_prefix_str = f"╭─ 📌 {n} " - HDR_SUFFIX = "─╮" - if cw(FTR_PREFIX) + cw(FTR_SUFFIX) > W: - FTR_SUFFIX = " ↑↓ ⏎ D Esc ─╯" - if cw(FTR_PREFIX) + cw(FTR_SUFFIX) > W: - FTR_SUFFIX = "─╯" - - hdr_dashes = max(0, W - cw(hdr_prefix_str) - cw(HDR_SUFFIX)) - ftr_dashes = max(0, W - cw(FTR_PREFIX) - cw(FTR_SUFFIX)) - - # Row inner width: W minus the two '│' border cells. - INNER = W - 2 - - frags: list = [] - - def line(text: str, style: str = "") -> None: - # Final guard: never emit a line wider than the box, whatever the - # label lengths worked out to. - frags.append((style, self._trim_status_bar_text(text, W) + "\n")) - - line(f"{hdr_prefix_str}{'─' * hdr_dashes}{HDR_SUFFIX}", "class:subagent-border") - - for i, item in enumerate(stash_list): - age = self._fmt_stash_age(item["stashed_at"]) - # Row: " ► [N] {age:<10} {preview} " - prefix = f" {'►' if i == cursor else ' '} [{i + 1}] {age:<10} " - if cw(prefix) > INNER - 2: - prefix = f" {'►' if i == cursor else ' '} [{i + 1}] " - avail = max(0, INNER - cw(prefix) - 1) - preview = self._trim_status_bar_text(item.get("preview") or "", avail) - preview = preview + " " * max(0, avail - cw(preview)) - row = self._trim_status_bar_text(f"│{prefix}{preview} │", W) - if i == cursor: - frags.append(("class:subagent-selected", row + "\n")) - else: - frags.append(("class:subagent-border", "│")) - frags.append(("class:subagent-sub", f"{prefix}{preview} ")) - frags.append(("class:subagent-border", "│\n")) - - line(f"{FTR_PREFIX}{'─' * ftr_dashes}{FTR_SUFFIX}", "class:subagent-border") - return frags - - def _normalize_model_for_provider(self, resolved_provider: str) -> bool: - """Normalize provider-specific model IDs and routing.""" - current_model = str(self.model or "").strip() - if isinstance(self.model, dict): - _m, _ = _split_model_config_default(self.model) - current_model = _m - changed = False - - try: - from hermes_cli.model_normalize import ( - _AGGREGATOR_PROVIDERS, - normalize_model_for_provider, - ) - - if resolved_provider not in _AGGREGATOR_PROVIDERS: - normalized_model = normalize_model_for_provider(current_model, resolved_provider) - if normalized_model and normalized_model != current_model: - if not self._model_is_default: - self._console_print( - f"[yellow]⚠️ Normalized model '{current_model}' to '{normalized_model}' for {resolved_provider}.[/]" - ) - self.model = normalized_model - current_model = normalized_model - changed = True - except Exception: - pass - - if resolved_provider == "copilot": - try: - from hermes_cli.models import copilot_model_api_mode, normalize_copilot_model_id - - canonical = normalize_copilot_model_id(current_model, api_key=self.api_key) - if canonical and canonical != current_model: - if not self._model_is_default: - self._console_print( - f"[yellow]⚠️ Normalized Copilot model '{current_model}' to '{canonical}'.[/]" - ) - self.model = canonical - current_model = canonical - changed = True - - resolved_mode = copilot_model_api_mode(current_model, api_key=self.api_key) - if resolved_mode != self.api_mode: - self.api_mode = resolved_mode - changed = True - except Exception: - pass - return changed - - from hermes_cli.models import opencode_provider_family - - if opencode_provider_family(resolved_provider) is not None: - try: - from hermes_cli.models import normalize_opencode_model_id, opencode_model_api_mode - - canonical = normalize_opencode_model_id(resolved_provider, current_model) - if canonical and canonical != current_model: - if not self._model_is_default: - self._console_print( - f"[yellow]⚠️ Stripped provider prefix from '{current_model}'; using '{canonical}' for {resolved_provider}.[/]" - ) - self.model = canonical - current_model = canonical - changed = True - - resolved_mode = opencode_model_api_mode(resolved_provider, current_model) - if resolved_mode != self.api_mode: - self.api_mode = resolved_mode - changed = True - except Exception: - pass - return changed - - if resolved_provider != "openai-codex": - return changed - - # 1. Strip provider prefix ("openai/gpt-5.4" → "gpt-5.4") - if "/" in current_model: - slug = current_model.split("/", 1)[1] - if not self._model_is_default: - self._console_print( - f"[yellow]⚠️ Stripped provider prefix from '{current_model}'; " - f"using '{slug}' for OpenAI Codex.[/]" - ) - self.model = slug - current_model = slug - changed = True - - # 2. Replace untouched default with a Codex model - if self._model_is_default: - fallback_model = "gpt-5.3-codex" - try: - from hermes_cli.codex_models import get_codex_model_ids - - available = get_codex_model_ids( - access_token=self.api_key if self.api_key else None, - ) - if available: - fallback_model = available[0] - except Exception: - pass - - if current_model != fallback_model: - self.model = fallback_model - changed = True - - return changed - - def _on_thinking(self, text: str) -> None: - """Called by agent when thinking starts/stops. Updates TUI spinner.""" - if not text: - self._flush_reasoning_preview(force=True) - self._spinner_text = text or "" - self._tool_start_time = 0.0 # clear tool timer when switching to thinking - self._invalidate() - - def _on_notice(self, notice) -> None: - """Queue an out-of-band AgentNotice for rendering at the next clean boundary. - - Notices fire from inside the agent turn (cold-start seed during _init_agent, - per-turn _capture_credits after the API call) — printing immediately races the - streaming response and the line gets buried behind the prompt (see _cprint's - bg-thread caveat). So we QUEUE here and flush in _flush_credit_notices(), called - right after run_conversation returns. Fail-soft: never break the turn. - """ - try: - text = getattr(notice, "text", "") or "" - if not text: - return - level = getattr(notice, "level", "info") or "info" - if not hasattr(self, "_pending_credit_notices"): - self._pending_credit_notices = [] - self._pending_credit_notices.append((level, text)) - except Exception: - pass - - def _flush_credit_notices(self) -> None: - """Print any queued credit notices as level-colored lines. Called at turn end - (after run_conversation) where _cprint paints cleanly above the prompt.""" - try: - pending = getattr(self, "_pending_credit_notices", None) - if not pending: - return - self._pending_credit_notices = [] - for level, text in pending: - color = { - "error": "\033[31m", - "warn": "\033[33m", - "success": "\033[32m", - "info": _DIM, - }.get(level, _DIM) - _cprint(f" {color}{text}{_RST}") - except Exception: - pass - - def _on_notice_clear(self, key: str) -> None: - """Notice cleared. The REPL prints lines (no persistent slot to wipe), so - this drops any still-queued notice with that key is not tracked by key here; - it's a no-op for rendering — kept so the agent's clear callback is bound - symmetrically with the show callback (and so future REPL UIs can hook it).""" - return - # ── Streaming display ──────────────────────────────────────────────── - def _current_reasoning_callback(self): - """Return the active reasoning display callback for the current mode.""" - if self.show_reasoning and self.streaming_enabled: - return self._stream_reasoning_delta - if self.verbose and not self.show_reasoning: - return self._on_reasoning - return None - - def _emit_reasoning_preview(self, reasoning_text: str) -> None: - """Render a buffered reasoning preview as a single [thinking] block.""" - preview_text = reasoning_text.strip() - if not preview_text: - return - - try: - term_width = shutil.get_terminal_size().columns - except Exception: - term_width = 80 - prefix = " [thinking] " - wrap_width = max(30, term_width - len(prefix) - 2) - - paragraphs = [] - raw_paragraphs = re.split(r"\n\s*\n+", preview_text.replace("\r\n", "\n")) - for paragraph in raw_paragraphs: - compact = " ".join(line.strip() for line in paragraph.splitlines() if line.strip()) - if compact: - paragraphs.append(textwrap.fill(compact, width=wrap_width)) - preview_text = "\n".join(paragraphs) - if not preview_text: - return - - if self.verbose: - _cprint(f" {_DIM}[thinking] {preview_text}{_RST}") - return - - lines = preview_text.splitlines() - if len(lines) > 5: - preview = "\n".join(lines[:5]) - preview += f"\n ... ({len(lines) - 5} more lines)" - else: - preview = preview_text - _cprint(f" {_DIM}[thinking] {preview}{_RST}") - - def _flush_reasoning_preview(self, *, force: bool = False) -> None: - """Flush buffered reasoning text at natural boundaries. - - Some providers stream reasoning in tiny word or punctuation chunks. - Buffer them here so the preview path does not print one `[thinking]` - line per token. - """ - buf = getattr(self, "_reasoning_preview_buf", "") - if not buf: - return - - try: - term_width = shutil.get_terminal_size().columns - except Exception: - term_width = 80 - target_width = max(40, term_width - len(" [thinking] ") - 4) - - flush_text = "" - - if force: - flush_text = buf - buf = "" - else: - line_break = buf.rfind("\n") - min_newline_flush = max(16, target_width // 3) - if line_break != -1 and ( - line_break >= min_newline_flush - or buf.endswith("\n\n") - or buf.endswith(".\n") - or buf.endswith("!\n") - or buf.endswith("?\n") - or buf.endswith(":\n") - ): - flush_text = buf[: line_break + 1] - buf = buf[line_break + 1 :] - elif len(buf) >= target_width: - search_start = max(20, target_width // 2) - search_end = min(len(buf), max(target_width + (target_width // 3), target_width + 8)) - cut = -1 - for boundary in (" ", "\t", ".", "!", "?", ",", ";", ":"): - cut = max(cut, buf.rfind(boundary, search_start, search_end)) - if cut != -1: - flush_text = buf[: cut + 1] - buf = buf[cut + 1 :] - - self._reasoning_preview_buf = buf.lstrip() if flush_text else buf - if flush_text: - self._emit_reasoning_preview(flush_text) - - def _format_submitted_user_message_preview(self, user_input: str) -> str: - """Format the submitted user-message scrollback preview.""" - ts_suffix = ( - f" [dim]{datetime.now().strftime(getattr(self, 'timestamp_format', '%H:%M'))}[/]" - if getattr(self, "show_timestamps", False) else "" - ) - lines = user_input.split("\n") - if len(lines) <= 1: - return f"[bold {_accent_hex()}]●[/] [bold]{_escape(user_input)}[/]{ts_suffix}" - - first_lines = int(getattr(self, "user_message_preview_first_lines", 2)) - last_lines = int(getattr(self, "user_message_preview_last_lines", 2)) - first_lines = max(1, first_lines) - last_lines = max(0, last_lines) - head = lines[:first_lines] - remaining_after_head = max(0, len(lines) - len(head)) - tail_count = min(last_lines, remaining_after_head) - tail = lines[-tail_count:] if tail_count else [] - - hidden_middle_count = len(lines) - len(head) - len(tail) - if hidden_middle_count < 0: - hidden_middle_count = 0 - tail = [] - - preview_lines = [ - f"[bold {_accent_hex()}]●[/] [bold]{_escape(head[0])}[/]{ts_suffix}" - ] - preview_lines.extend(f"[bold]{_escape(line)}[/]" for line in head[1:]) - - if hidden_middle_count > 0: - noun = "line" if hidden_middle_count == 1 else "lines" - preview_lines.append(f"[dim]... (+{hidden_middle_count} more {noun})[/]") - - preview_lines.extend(f"[bold]{_escape(line)}[/]" for line in tail) - return "\n".join(preview_lines) - - def _expand_paste_references(self, text: str | None) -> str: - """Expand [Pasted text #N -> file] placeholders into file contents.""" - if not isinstance(text, str) or "[Pasted text #" not in text: - return text or "" - paste_ref_re = re.compile(r'\[Pasted text #\d+: \d+ lines \u2192 (.+?)\]') - - def _expand_ref(match): - path = Path(match.group(1)) - # Use try/except instead of path.exists() to avoid TOCTOU race: - # the paste file may be deleted between check and read, causing - # the input to be silently dropped (#17666). - try: - return path.read_text(encoding="utf-8") - except (OSError, IOError): - logger.warning("Paste file gone or unreadable, returning placeholder: %s", path) - return match.group(0) - - return paste_ref_re.sub(_expand_ref, text) - - def _print_user_message_preview(self, user_input: str) -> None: - """Render a user message using the normal chat scrollback style.""" - ChatConsole().print(f"[{_accent_hex()}]{'─' * 40}[/]") - text = str(user_input or "") - if "\n" in text: - ChatConsole().print(self._format_submitted_user_message_preview(text)) - else: - ChatConsole().print(f"[bold {_accent_hex()}]●[/] [bold]{_escape(text)}[/]") - - def _stream_reasoning_delta(self, text: str) -> None: - """Stream reasoning/thinking tokens into a dim box above the response. - - Opens a dim reasoning box on first token, streams line-by-line. - The box is closed automatically when content tokens start arriving - (via _stream_delta → _emit_stream_text). - - Once the response box is open, suppress any further reasoning - rendering — a late thinking block (e.g. after an interrupt) would - otherwise draw a reasoning box inside the response box. - """ - if not text: - return - self._reasoning_shown_this_turn = True - if getattr(self, "_stream_box_opened", False): - return - - # Open reasoning box on first reasoning token - if not getattr(self, "_reasoning_box_opened", False): - self._reasoning_box_opened = True - w = self._scrollback_box_width() - r_label = " Reasoning " - r_fill = w - 2 - len(r_label) - _cprint(f"\n{_DIM}┌─{r_label}{'─' * max(r_fill - 1, 0)}┐{_RST}") - - self._reasoning_buf = getattr(self, "_reasoning_buf", "") + text - - # Emit complete lines, and force-flush long partial lines so - # reasoning is visible in real-time even without newlines. - while "\n" in self._reasoning_buf: - line, self._reasoning_buf = self._reasoning_buf.split("\n", 1) - _cprint(f"{_DIM}{line}{_RST}") - if len(self._reasoning_buf) > 80: - _cprint(f"{_DIM}{self._reasoning_buf}{_RST}") - self._reasoning_buf = "" - - def _close_reasoning_box(self) -> None: - """Close the live reasoning box if it's open.""" - if getattr(self, "_reasoning_box_opened", False): - # Flush remaining reasoning buffer - buf = getattr(self, "_reasoning_buf", "") - if buf: - _cprint(f"{_DIM}{buf}{_RST}") - self._reasoning_buf = "" - w = self._scrollback_box_width() - _cprint(f"{_DIM}└{'─' * (w - 2)}┘{_RST}") - self._reasoning_box_opened = False - - # Flush any content that was deferred while reasoning was rendering. - deferred = getattr(self, "_deferred_content", "") - if deferred: - self._deferred_content = "" - self._emit_stream_text(deferred) - - def _stream_delta(self, text) -> None: - """Line-buffered streaming callback for real-time token rendering. - - Receives text deltas from the agent as tokens arrive. Buffers - partial lines and emits complete lines via _cprint to work - reliably with prompt_toolkit's patch_stdout. - - Reasoning/thinking blocks (, , etc.) - are suppressed during streaming since they'd display raw XML tags. - The agent strips them from the final response anyway. - - A ``None`` value signals an intermediate turn boundary (tools are - about to execute). Flushes any open boxes and resets state so - tool feed lines render cleanly between turns. - """ - if text is None: - self._flush_stream() - self._reset_stream_state() - return - if not text: - return - - self._stream_started = True - - # ── Tag-based reasoning suppression ── - # Track whether we're inside a reasoning/thinking block. - # These tags are model-generated (system prompt tells the model - # to use them) and get stripped from final_response. We must - # suppress them during streaming too — unless show_reasoning is - # enabled, in which case we route the inner content to the - # reasoning display box instead of discarding it. - _OPEN_TAGS = ("", "", "", "", "", "") - _CLOSE_TAGS = ("", "", "", "", "", "") - - # Append to a pre-filter buffer first - self._stream_prefilt = getattr(self, "_stream_prefilt", "") + text - - # Check if we're entering a reasoning block. - # Only match tags that appear at a "block boundary": start of the - # stream, after a newline (with optional whitespace), or when nothing - # but whitespace has been emitted on the current line. - # This prevents false positives when models *mention* tags in prose - # like "(/think not producing tags)". - # - # _stream_last_was_newline tracks whether the last character emitted - # (or the start of the stream) is a line boundary. It's True at - # stream start and set True whenever emitted text ends with '\n'. - if not hasattr(self, "_stream_last_was_newline"): - self._stream_last_was_newline = True # start of stream = boundary - - if not getattr(self, "_in_reasoning_block", False): - # Case-insensitive matching against a lowercased view so - # mixed-case tag variants (, , …) are caught. - prefilt_lower = self._stream_prefilt.lower() - for tag in _OPEN_TAGS: - tag_lower = tag.lower() - search_start = 0 - while True: - idx = prefilt_lower.find(tag_lower, search_start) - if idx == -1: - break - # Check if this is a block boundary position - preceding = self._stream_prefilt[:idx] - if idx == 0: - # At buffer start — only a boundary if we're at - # a line start (stream start or last emit ended - # with newline) - is_block_boundary = getattr(self, "_stream_last_was_newline", True) - else: - # Find last newline in the buffer before the tag - last_nl = preceding.rfind("\n") - if last_nl == -1: - # No newline in buffer — boundary only if - # last emit was a newline AND only whitespace - # has accumulated before the tag - is_block_boundary = ( - getattr(self, "_stream_last_was_newline", True) - and preceding.strip() == "" - ) - else: - # Text between last newline and tag must be - # whitespace-only - is_block_boundary = preceding[last_nl + 1:].strip() == "" - if is_block_boundary: - # Emit everything before the tag - if preceding: - self._emit_stream_text(preceding) - self._stream_last_was_newline = preceding.endswith("\n") - self._in_reasoning_block = True - self._stream_prefilt = self._stream_prefilt[idx + len(tag):] - break - # Not a block boundary — keep searching after this occurrence - search_start = idx + 1 - if getattr(self, "_in_reasoning_block", False): - break - - # Could also be a partial open tag at the end — hold it back - if not getattr(self, "_in_reasoning_block", False): - # Check for partial tag match at the end (case-insensitive) - safe = self._stream_prefilt - for tag in _OPEN_TAGS: - tag_lower = tag.lower() - for i in range(1, len(tag)): - if prefilt_lower.endswith(tag_lower[:i]): - safe = self._stream_prefilt[:-i] - break - if safe: - self._emit_stream_text(safe) - self._stream_last_was_newline = safe.endswith("\n") - self._stream_prefilt = self._stream_prefilt[len(safe):] - return - - # Inside a reasoning block — look for close tag. - # Keep accumulating _stream_prefilt because close tags can arrive - # split across multiple tokens (e.g. "..."). - if getattr(self, "_in_reasoning_block", False): - prefilt_lower = self._stream_prefilt.lower() - for tag in _CLOSE_TAGS: - idx = prefilt_lower.find(tag.lower()) - if idx != -1: - self._in_reasoning_block = False - # When show_reasoning is on, route inner content to - # the reasoning display box instead of discarding. - if self.show_reasoning: - inner = self._stream_prefilt[:idx] - if inner: - self._stream_reasoning_delta(inner) - after = self._stream_prefilt[idx + len(tag):] - self._stream_prefilt = "" - # Process remaining text after close tag through full - # filtering (it could contain another open tag) - if after: - self._stream_delta(after) - return - # When show_reasoning is on, stream reasoning content live - # instead of silently accumulating. Keep only the tail that - # could be a partial close tag prefix. - max_tag_len = max(len(t) for t in _CLOSE_TAGS) - if len(self._stream_prefilt) > max_tag_len: - if self.show_reasoning: - # Route the safe prefix to reasoning display - safe_reasoning = self._stream_prefilt[:-max_tag_len] - self._stream_reasoning_delta(safe_reasoning) - self._stream_prefilt = self._stream_prefilt[-max_tag_len:] - return - - def _emit_stream_text(self, text: str) -> None: - """Emit filtered text to the streaming display.""" - if not text: - return - - # When show_reasoning is on and reasoning is still rendering, - # defer content until the reasoning box closes. This ensures the - # reasoning block always appears BEFORE the response in the terminal. - if self.show_reasoning and getattr(self, "_reasoning_box_opened", False): - self._deferred_content = getattr(self, "_deferred_content", "") + text - return - - # Close the live reasoning box before opening the response box - self._close_reasoning_box() - - # Open the response box header on the very first visible text - if not self._stream_box_opened: - # Strip leading whitespace/newlines before first visible content - text = text.lstrip("\n") - if not text: - return - self._stream_box_opened = True - try: - from hermes_cli.skin_engine import get_active_skin - _skin = get_active_skin() - label = _skin.get_branding("response_label", "⚕ Hermes") - _text_hex = _skin.get_color("banner_text", "#FFF8DC") - except Exception: - label = "⚕ Hermes" - _text_hex = "#FFF8DC" - # Build a true-color ANSI escape for the response text color - # so streamed content matches the Rich Panel appearance. - try: - _r = int(_text_hex[1:3], 16) - _g = int(_text_hex[3:5], 16) - _b = int(_text_hex[5:7], 16) - self._stream_text_ansi = f"\033[38;2;{_r};{_g};{_b}m" - except (ValueError, IndexError): - self._stream_text_ansi = "" - if self.show_timestamps: - label = f"{label} {datetime.now().strftime(getattr(self, 'timestamp_format', '%H:%M'))}" - w = self._scrollback_box_width() - fill = w - 2 - HermesCLI._status_bar_display_width(label) - _cprint(f"\n{_ACCENT}╭─{label}{'─' * max(fill - 1, 0)}╮{_RST}") - - self._stream_buf += text - - # Emit complete lines, keep partial remainder in buffer - _tc = getattr(self, "_stream_text_ansi", "") - - def _emit_one(printed_line: str) -> None: - _cprint(f"{_STREAM_PAD}{_tc}{printed_line}{_RST}" if _tc else f"{_STREAM_PAD}{printed_line}") - - def _flush_table_buf() -> None: - buf = self._stream_table_buf - self._stream_table_buf = [] - self._in_stream_table = False - if not buf: - return - # Strip cell-level markdown (`code`, **bold**, ~~strike~~) FIRST - # so the realigner pads to the final visible cell width, not - # the marker-decorated source width. Otherwise a body row - # like `` | Bold | `**bold**` | `` lands narrower than its - # header column once the markers are removed. - joined = "\n".join(buf) - if self.final_response_markdown == "strip": - joined = _strip_markdown_syntax(joined) - block = realign_markdown_tables(joined, _terminal_width_for_streaming()) - for ln in block.split("\n"): - _emit_one(ln) - - while "\n" in self._stream_buf: - line, self._stream_buf = self._stream_buf.split("\n", 1) - - # Hold table-shaped lines in a side-buffer so we can re-pad - # the whole block once it ends. Streaming line-by-line, we - # cannot re-align mid-table without reflowing already-printed - # rows; the cost is that the user sees the table appear in a - # single batch when the block closes instead of row-by-row. - if self._in_stream_table: - if looks_like_table_row(line) or is_table_divider(line): - self._stream_table_buf.append(line) - continue - # Block ended — flush the realigned table, then fall - # through to print the current (non-table) line. - _flush_table_buf() - elif looks_like_table_row(line): - self._stream_table_buf.append(line) - self._in_stream_table = True - continue - - if self.final_response_markdown == "strip": - line = _strip_markdown_syntax(line) - _emit_one(line) - - # Long partial lines are emitted ONLY at real newlines — we no - # longer hard-wrap paragraphs at terminal width ourselves. Each - # logical line lands in scrollback as one line; the TERMINAL - # soft-wraps it visually, and emulators (iTerm2/kitty/VTE/ - # xterm.js/Windows Terminal) rejoin soft-wrapped rows on copy, - # so highlight-copy yields the original unwrapped text — same - # outcome as the TUI's selection copy. (The pre-July-2026 chunk - # emitter baked real '\n's into every long paragraph, which is - # exactly what polluted copy/paste.) - # - # TTFT perception: while a long opening paragraph accumulates - # without a newline, mirror its tail into the status-bar spinner - # line so the user sees tokens arriving instead of a blank box. - if ( - self._stream_buf - and not self._in_stream_table - and not self._stream_buf.lstrip().startswith("|") - and len(self._stream_buf) >= 80 - ): - preview = self._stream_buf[-int(_STREAM_PARTIAL_PREVIEW_LEN):] - cut = preview.find(" ") - if 0 < cut < len(preview) - 1: - preview = preview[cut + 1:] - try: - self._spinner_text = f"… {preview}" - self._invalidate() - except Exception: - pass - - def _flush_stream(self) -> None: - """Emit any remaining partial line from the stream buffer and close the box.""" - # If we're still inside a "reasoning block" at end-of-stream, it was - # a false positive — the model mentioned a tag like in prose - # but never closed it. Recover the buffered content as regular text. - if getattr(self, "_in_reasoning_block", False) and getattr(self, "_stream_prefilt", ""): - self._in_reasoning_block = False - self._emit_stream_text(self._stream_prefilt) - self._stream_prefilt = "" - - # Close reasoning box if still open (in case no content tokens arrived) - self._close_reasoning_box() - - _tc = getattr(self, "_stream_text_ansi", "") - - # If the stream buffer has a trailing partial line that looks like - # a table row, fold it into the table buffer so the whole block - # gets re-aligned together. Otherwise the final row prints raw - # (with the model's original under-padded spacing) while the rows - # above it are aligned. - if ( - self._stream_buf - and getattr(self, "_in_stream_table", False) - and (looks_like_table_row(self._stream_buf) or is_table_divider(self._stream_buf)) - ): - self._stream_table_buf.append(self._stream_buf) - self._stream_buf = "" - - # Flush any buffered table rows first so their padding is - # finalised before the stream remainder lands. - if getattr(self, "_stream_table_buf", None): - joined = "\n".join(self._stream_table_buf) - self._stream_table_buf = [] - self._in_stream_table = False - if self.final_response_markdown == "strip": - joined = _strip_markdown_syntax(joined) - block = realign_markdown_tables(joined, _terminal_width_for_streaming()) - for ln in block.split("\n"): - _cprint(f"{_STREAM_PAD}{_tc}{ln}{_RST}" if _tc else f"{_STREAM_PAD}{ln}") - - if self._stream_buf: - line = _strip_markdown_syntax(self._stream_buf) if self.final_response_markdown == "strip" else self._stream_buf - _cprint(f"{_STREAM_PAD}{_tc}{line}{_RST}" if _tc else f"{_STREAM_PAD}{line}") - self._stream_buf = "" - - # Close the response box - if self._stream_box_opened: - w = self._scrollback_box_width() - _cprint(f"{_ACCENT}╰{'─' * (w - 2)}╯{_RST}") - - def _reset_stream_state(self) -> None: - """Reset streaming state before each agent invocation.""" - self._stream_buf = "" - self._stream_started = False - self._stream_box_opened = False - self._stream_text_ansi = "" - self._stream_prefilt = "" - self._in_reasoning_block = False - self._stream_last_was_newline = True - self._reasoning_box_opened = False - self._reasoning_buf = "" - self._reasoning_preview_buf = "" - self._deferred_content = "" - self._stream_table_buf = [] - self._in_stream_table = False - - def _slow_command_status(self, command: str) -> str: - """Return a user-facing status message for slower slash commands.""" - cmd_lower = command.lower().strip() - if cmd_lower.startswith("/skills search"): - return "Searching skills..." - if cmd_lower.startswith("/skills browse"): - return "Loading skills..." - if cmd_lower.startswith("/skills inspect"): - return "Inspecting skill..." - if cmd_lower.startswith("/skills install"): - return "Installing skill..." - if cmd_lower.startswith("/skills"): - return "Processing skills command..." - if cmd_lower == "/reload-mcp": - return "Reloading MCP servers..." - if cmd_lower == "/reload-skills" or cmd_lower == "/reload_skills": - return "Reloading skills..." - if cmd_lower.startswith("/browser"): - return "Configuring browser..." - return "Processing command..." - - def _command_spinner_frame(self) -> str: - """Return the current spinner frame for slow slash commands.""" - frame_idx = int(time.monotonic() * 10) % len(_COMMAND_SPINNER_FRAMES) - return _COMMAND_SPINNER_FRAMES[frame_idx] - - @contextmanager - def _busy_command(self, status: str, *, blocks_input: bool = True): - """Expose a temporary busy state in the TUI while a slash command runs. - - Most synchronous slash commands must reserve the composer because their - completion changes the active session state. Manual compression is safe - to draft through: the queued input is processed against the compacted - history after the command completes. - """ - previous_blocks_input = getattr(self, "_command_blocks_input", False) - self._command_running = True - self._command_blocks_input = blocks_input - self._command_status = status - self._invalidate(min_interval=0.0) - try: - print(f"⏳ {status}") - yield - finally: - self._command_running = False - self._command_blocks_input = previous_blocks_input - self._command_status = "" - self._invalidate(min_interval=0.0) - - def _open_external_editor(self, buffer=None) -> bool: - """Open the active input buffer in an external editor.""" - app = getattr(self, "_app", None) - if not app: - _cprint(f"{_DIM}External editor is only available inside the interactive CLI.{_RST}") - return False - if self._command_running: - _cprint(f"{_DIM}Wait for the current command to finish before opening the editor.{_RST}") - return False - if self._sudo_state or self._secret_state or self._approval_state or getattr(self, "_slash_confirm_state", None) or self._clarify_state: - _cprint(f"{_DIM}Finish the active prompt before opening the editor.{_RST}") - return False - target_buffer = buffer or getattr(app, "current_buffer", None) - if target_buffer is None: - _cprint(f"{_DIM}No active input buffer is available for the external editor.{_RST}") - return False - try: - # Inline pastes so the editor (and the draft it submits) sees real - # content; skip flag unconditionally so the editor-close text-change - # doesn't re-collapse it, even when there was nothing to inline. - self._inline_pastes(target_buffer) - self._skip_paste_collapse = True - # Open the editor, then submit the saved draft on a clean exit — - # matching the TUI's Ctrl+G (openEditor), which sends the buffer - # instead of requiring a second Enter. Submission in this CLI is - # driven by the custom `enter` keybinding, NOT the buffer's - # accept_handler, so validate_and_handle can't route through it; - # chain a done-callback on the returned Task that re-uses the - # real submit pipeline via _submit_editor_buffer(). - task = target_buffer.open_in_editor(validate_and_handle=False) - if task is not None and hasattr(task, "add_done_callback"): - task.add_done_callback( - lambda _t, b=target_buffer: self._submit_editor_buffer(b) - ) - return True - except Exception as exc: - _cprint(f"{_DIM}Failed to open external editor: {exc}{_RST}") - return False - - def _submit_editor_buffer(self, buffer) -> None: - """Submit the draft an external editor left in ``buffer``. - - Invoked from the Ctrl+G done-callback so saving the editor sends the - prompt (TUI parity) instead of leaving it sitting in the input area. - Mirrors the idle/queue branches of the `enter` keybinding handler: - an empty save is ignored (never submits a blank turn), a slash command - is dispatched, otherwise the text is routed through the same input - queues the normal Enter path uses. Runs on the prompt_toolkit event - loop via the Task callback, so it must be cheap and non-blocking. - """ - try: - text = (getattr(buffer, "text", "") or "").strip() - except Exception: - return - if not text: - # Editor saved empty / was cleared — match the TUI, which drops - # an empty draft instead of submitting a blank turn. - return - - app = getattr(self, "_app", None) - - # `!` shell mode, checked before slash dispatch — matches the - # Enter path in the input loop so an editor-saved bang command runs - # locally instead of being sent to the agent. - try: - if self.handle_bang_shell(text): - self._reset_input_buffer(buffer) - if app is not None: - app.invalidate() - return - except Exception as exc: - _cprint(f" {_DIM}Shell command failed: {exc}{_RST}") - self._reset_input_buffer(buffer) - if app is not None: - app.invalidate() - return - - # Slash commands: dispatch directly, same as the Enter handler's - # _looks_like_slash_command branch. - if _looks_like_slash_command(text): - try: - if not self.process_command(text): - self._should_exit = True - if app is not None and app.is_running: - app.exit() - except Exception as exc: - _cprint(f" {_DIM}Command failed: {exc}{_RST}") - finally: - self._reset_input_buffer(buffer) - if app is not None: - app.invalidate() - return - - # Regular prompt: route through the same queues the Enter handler uses. - if self._agent_running: - # Agent busy → honour the configured busy-input behaviour by - # queueing for the next turn (the safe default; interrupt/steer - # remain reachable via the normal Enter path). - self._interrupt_queue.put(text) if self.busy_input_mode == "interrupt" else self._pending_input.put(text) - preview = text[:80] + ("..." if len(text) > 80 else "") - _cprint(f" Queued for the next turn: {preview}") - else: - self._pending_input.put(text) - - self._reset_input_buffer(buffer) - if app is not None: - app.invalidate() - - def _inline_pastes(self, buffer) -> None: - """Replace collapsed-paste placeholders in ``buffer`` with real content. - - A big paste shows as a compact ``[Pasted text #N -> file]`` placeholder, - but history recall and the external editor need the actual text — a bare - reference is useless once the file is gone or on another machine. Inlining - before ``reset(append_to_history=True)`` also lets prompt_toolkit persist - the content through its normal path. Sets ``_skip_paste_collapse`` so the - ensuing text-change doesn't re-collapse it. - """ - try: - existing = getattr(buffer, "text", "") - expanded = self._expand_paste_references(existing) - if expanded != existing and hasattr(buffer, "text"): - self._skip_paste_collapse = True - buffer.text = expanded - if hasattr(buffer, "cursor_position"): - buffer.cursor_position = len(expanded) - except Exception: - logger.debug("Failed to inline paste placeholders", exc_info=True) - - def _reset_input_buffer(self, buffer) -> None: - """Clear an input buffer after a programmatic submit (best-effort).""" - try: - buffer.reset(append_to_history=True) - except Exception: - try: - buffer.text = "" - except Exception: - pass - - - def _install_tool_callbacks(self) -> None: """Install tool callbacks that need the live prompt UI.""" if getattr(self, "_tool_callbacks_installed", False): @@ -9003,721 +6069,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): ).strip() self.preloaded_skills = loaded_skills - def show_banner(self): - """Display the welcome banner in Claude Code style.""" - self.console.clear() - ctx_len = None - if hasattr(self, 'agent') and self.agent and hasattr(self.agent, 'context_compressor'): - ctx_len = self.agent.context_compressor.context_length - - # Auto-compact for narrow terminals — the full banner with caduceus - # + tool list needs ~80 columns minimum to render without wrapping. - term_width = shutil.get_terminal_size().columns - use_compact = self.compact or term_width < 80 - - if use_compact: - self._console_print(_build_compact_banner()) - self._show_status() - else: - # Warm-launch fast path: replay last launch's tool panel when the - # snapshot fingerprint (config.yaml + .env + checkout rev + - # toolsets) is unchanged, skipping the ~0.5-0.9s cold - # get_tool_definitions walk. The agent's REAL tool list is still - # computed fresh at first message; a background refresh below - # re-verifies the snapshot so any drift self-heals next launch. - from hermes_cli.banner import ( - compute_toolset_availability, - load_banner_snapshot, - save_banner_snapshot, - ) - - snapshot = None - try: - snapshot = load_banner_snapshot(self.enabled_toolsets) - except Exception: - snapshot = None - - # Get terminal working directory (where commands will execute) - cwd = os.getenv("TERMINAL_CWD", os.getcwd()) - - if snapshot is not None: - self._defer_tool_warnings = True - toolset_map = snapshot["toolset_map"] - build_welcome_banner( - console=self.console, - model=self.model, - cwd=cwd, - tools=snapshot["tools"], - enabled_toolsets=self.enabled_toolsets, - session_id=self.session_id, - get_toolset_for_tool=lambda name: toolset_map.get(name), - context_length=ctx_len, - provider=self.provider, - availability=snapshot["availability"], - skills_by_category=snapshot.get("skills_by_category"), - ) - - def _refresh_banner_snapshot() -> None: - try: - from model_tools import get_toolset_for_tool - tools = get_tool_definitions( - enabled_toolsets=self.enabled_toolsets, quiet_mode=True - ) - availability = compute_toolset_availability(self.enabled_toolsets) - tmap = { - t["function"]["name"]: get_toolset_for_tool(t["function"]["name"]) - for t in tools - } - for item in availability.get("unavailable_toolsets", []): - for name in item.get("tools", []): - tmap.setdefault( - name, item.get("id", item.get("name", "")) - ) - save_banner_snapshot( - tools, self.enabled_toolsets, availability, tmap - ) - except Exception: - logger.debug("banner snapshot refresh failed", exc_info=True) - - threading.Thread( - target=_refresh_banner_snapshot, - name="banner-snapshot-refresh", - daemon=True, - ).start() - else: - # Cold path: compute everything live, then persist the snapshot - # so the next launch replays it. - from model_tools import get_toolset_for_tool - tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) - availability = compute_toolset_availability(self.enabled_toolsets) - - build_welcome_banner( - console=self.console, - model=self.model, - cwd=cwd, - tools=tools, - enabled_toolsets=self.enabled_toolsets, - session_id=self.session_id, - context_length=ctx_len, - provider=self.provider, - availability=availability, - ) - try: - tmap = { - t["function"]["name"]: get_toolset_for_tool(t["function"]["name"]) - for t in tools - } - for item in availability.get("unavailable_toolsets", []): - for name in item.get("tools", []): - tmap.setdefault(name, item.get("id", item.get("name", ""))) - save_banner_snapshot(tools, self.enabled_toolsets, availability, tmap) - except Exception: - logger.debug("banner snapshot save failed", exc_info=True) - - # Tool discovery is intentionally deferred on the Termux bare prompt - # path; availability warnings are shown once tools are initialized. - # On the snapshot fast path (warm launch), the check walks every - # check_fn (~180ms) — run it in the background refresh thread instead - # and let its output land above the prompt (patch_stdout-safe). - if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": - if getattr(self, "_defer_tool_warnings", False): - threading.Thread( - target=self._show_tool_availability_warnings, - name="tool-availability-warnings", - daemon=True, - ).start() - else: - self._show_tool_availability_warnings() - - # Warn about low context lengths (common with local servers). Keep - # this tied to the runtime guard so guidance cannot drift again. - from agent.model_metadata import MINIMUM_CONTEXT_LENGTH - if ctx_len and ctx_len < MINIMUM_CONTEXT_LENGTH: - self._console_print() - self._console_print( - f"[yellow]⚠️ Context length is only {ctx_len:,} tokens — " - f"this is likely too low for agent use with tools.[/]" - ) - self._console_print( - f"[dim] Hermes needs at least {MINIMUM_CONTEXT_LENGTH:,} tokens. Tool schemas + system prompt use a large fixed prefix.[/]" - ) - base_url = getattr(self, "base_url", "") or "" - from urllib.parse import urlparse as _urlparse - try: - _parsed = _urlparse(base_url if "://" in base_url else f"//{base_url}") - _port = _parsed.port - except ValueError: - _port = None - _host = base_url_hostname(base_url) - if _port == 11434 or "ollama" in _host: - self._console_print( - f"[dim] Ollama fix: OLLAMA_CONTEXT_LENGTH={MINIMUM_CONTEXT_LENGTH} ollama serve[/]" - ) - elif _port == 1234: - self._console_print( - "[dim] LM Studio fix: Set context length in model settings → reload model[/]" - ) - else: - self._console_print( - "[dim] Fix: Set model.context_length in config.yaml, or increase your server's context setting[/]" - ) - - # Warn if the configured model is a Nous Hermes LLM (not agentic) - from hermes_cli.model_switch import is_nous_hermes_non_agentic - - model_name = getattr(self, "model", "") or "" - if is_nous_hermes_non_agentic(model_name): - self._console_print() - self._console_print( - "[bold yellow]⚠ Nous Research Hermes 3 & 4 models are NOT agentic and are not " - "designed for use with Hermes Agent.[/]" - ) - self._console_print( - "[dim] They lack tool-calling capabilities required for agent workflows. " - "Consider using an agentic model (Claude, GPT, Gemini, DeepSeek, etc.).[/]" - ) - self._console_print( - "[dim] Switch with: /model sonnet or /model gpt5[/]" - ) - - # Project-local skills: one-line status. Trusted → show count; - # untrusted-with-skills → point at `hermes skills trust`. Never raises. - try: - from agent.skill_utils import ( - get_project_skills_dirs, - get_untrusted_project_skills_root, - iter_skill_index_files, - ) - _proj_dirs = get_project_skills_dirs() - if _proj_dirs: - _n = sum( - sum(1 for _ in iter_skill_index_files(d, "SKILL.md")) - for d in _proj_dirs - ) - if _n: - self._console_print( - f"[dim]◆ {_n} project skill(s) loaded from this repo[/]" - ) - else: - _untrusted = get_untrusted_project_skills_root() - if _untrusted is not None: - _root, _n = _untrusted - self._console_print( - f"[yellow]◆ {_n} project skill(s) found in {_root} but not " - f"loaded — run `hermes skills trust` to enable them.[/]" - ) - except Exception: - logger.debug("project skills banner notice failed", exc_info=True) - - self._console_print() - - def _restore_session_cwd(self, session_meta: dict, *, quiet: bool = False) -> None: - """Relaunch a resumed session in the directory it was started from. - - Idempotent and safe to call from every resume path. When the stored - ``cwd`` differs from the current process directory, we both - ``os.chdir()`` (so the process and any ``os.getcwd()`` fallback agree) - and retarget ``TERMINAL_CWD`` (so the terminal tool, code-exec tool, - and relative-path resolution all land in the same place — the local - terminal backend snapshots cwd on first use, which happens after this). - - No-ops when: the session recorded no cwd (gateway/remote/older - sessions), the directory no longer exists, or we're already there. - A missing directory degrades to a single dim warning rather than a - crash — repos get moved and deleted. - """ - recorded = (session_meta or {}).get("cwd") - if not recorded: - return - recorded = os.path.expanduser(str(recorded)) - try: - current = os.getcwd() - except OSError: - current = None - if current and os.path.realpath(recorded) == os.path.realpath(current): - return # Already where the session lived — nothing to announce. - - if not os.path.isdir(recorded): - msg = f"⚠ Session's working directory is gone: {recorded} — staying in {current or '.'}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") - return - - try: - os.chdir(recorded) - except OSError as e: - msg = f"⚠ Could not enter session's working directory {recorded}: {e}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") - return - - # Retarget the terminal/code-exec tools to match the process cwd. - os.environ["TERMINAL_CWD"] = recorded - - msg = f"↻ Working directory: {recorded}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") - - def _restore_session_yolo(self, session_meta: dict, *, quiet: bool = False) -> None: - """Re-enable YOLO bypass on resume when the session had it on. - - Companion to ``_restore_session_cwd`` — called from every resume path - (startup ``--resume``/``-c`` and mid-chat ``/resume``). The persisted - flag lives in the session row's ``model_config.yolo_mode`` (written by - ``/yolo`` toggles and ``--yolo`` launches); without this restore the - in-memory ``tools.approval._session_yolo`` set starts empty in a fresh - process and the user's bypass silently reverts. - - No-op when the flag is absent/false, when YOLO is already active for - this session (idempotent across repeated resume paths), or when the - process was itself launched with ``--yolo`` (frozen bypass already - covers everything). - """ - try: - from hermes_state import SessionDB - from tools.approval import ( - _YOLO_MODE_FROZEN, - enable_session_yolo, - is_session_yolo_enabled, - ) - except Exception: - return - if _YOLO_MODE_FROZEN: - return - if not SessionDB.session_yolo_enabled(session_meta): - return - session_key = self.session_id or "default" - if is_session_yolo_enabled(session_key): - return - enable_session_yolo(session_key) - msg = "⚡ YOLO mode restored from session — all commands auto-approved. /yolo to turn off." - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") - - def _persist_model_switch_to_session(self, result) -> None: - """Persist a session-scoped /model switch to the session DB row. - - Writes the model column plus the runtime route so ``--resume`` - (CLI, reads ``gateway_runtime``) and ``session.resume`` (TUI/desktop, - reads top-level ``model_config`` keys via - ``_stored_session_runtime_overrides``) both restore the switched - provider instead of recombining the model with the ambient default - (#79536). Mirrors the gateway's ``update_session_model()`` call. - getattr: tests drive the switch paths with ``object.__new__`` stubs. - """ - db = getattr(self, "_session_db", None) - sid = getattr(self, "session_id", None) - if not db or not sid: - return - provider = result.target_provider - # Bare "custom" is the resolved billing class, not a routable - # identity — persisting it verbatim makes a later resume hard-fail - # when the config default has moved off the custom endpoint - # (resolve_runtime_provider only trusts config base_url for bare - # custom while the config provider is still custom-ish). Heal to - # the durable custom: menu key, else drop the provider — - # same recovery the TUI gateway applies on its read path. - if str(provider or "").strip().lower() == "custom": - try: - from hermes_cli.runtime_provider import canonical_custom_identity - provider = canonical_custom_identity( - base_url=result.base_url or None, - model=result.new_model or None, - ) or None - except Exception: - provider = None - # Both shapes use the same or-None discipline so stale keys from a - # previous switch are deleted (not merely omitted) in BOTH the - # nested gateway_runtime dict (CLI reader) and the top-level keys - # (TUI gateway reader). _merge_model_config_json only deletes on - # explicit None, so falsy values must be converted, not filtered. - # Deriving the top-level from **route guarantees the two shapes - # can never diverge — the asymmetry that caused the original - # stale-key bug (#85261 simplify-code review). - route = { - "provider": provider or None, - "base_url": result.base_url or None, - "api_mode": result.api_mode or None, - } - try: - db.update_session_model(sid, result.new_model) - db.patch_session_model_config(sid, { - "gateway_runtime": route, - **route, - }) - except Exception: - logger.debug( - "Failed to persist model switch to session DB", exc_info=True - ) - - def _restore_session_model(self, session_meta: dict, *, quiet: bool = False) -> None: - """Restore model/provider from the session DB row on resume. - - Companion to ``_restore_session_cwd`` / ``_restore_session_yolo`` — - called from every resume path (startup ``--resume``/``-c`` and - mid-chat ``/resume``). The persisted model lives in the session row's - ``model`` column (written at creation time and updated on ``/model`` - switches via ``update_session_model``); the provider/endpoint live in - ``model_config.gateway_runtime`` (written by the gateway's - ``_sync_session_model_from_agent`` and the CLI ``/model`` persist). - Without this restore a resumed session silently falls back to the - config default model, losing the user's last ``/model`` choice. - - When the stored provider differs from the ambient one, credentials - are re-resolved for the stored provider (mirroring the gateway's - ``_rehydrate_session_model_override``) — the ambient ``self.api_key`` - belongs to the config-default provider and must not be sent to the - session's endpoint. On resolution failure the ambient credentials are - kept so the session still opens (the first turn surfaces the auth - error instead of the resume dying). - - Skips when the session has no model recorded or when the CLI was - launched with an explicit ``-m`` override (user intent wins). - """ - stored_model = (session_meta or {}).get("model") - if not stored_model: - return - # An explicit -m / --model on the command line overrides resume. - if getattr(self, "_explicit_model_override", False): - return - # Stored provider/endpoint via the canonical row-level reader - # (prefers model_config.gateway_runtime, falls back to the TUI - # gateway's top-level keys). - from hermes_state import SessionDB as _SessionDB - _stored_runtime = _SessionDB.session_gateway_runtime(session_meta) - stored_provider = _stored_runtime.get("provider") or None - stored_base_url = _stored_runtime.get("base_url") or None - stored_api_mode = _stored_runtime.get("api_mode") or None - # Heal bare "custom" persisted by older builds / gateway turns: it's - # the resolved billing class, not a routable identity. Recover the - # durable custom: menu key from the endpoint, else drop the - # provider so resume keeps the ambient default. (Stricter than the - # TUI gateway's recovery, which keeps bare "custom" when a base_url - # exists — the CLI's resolve path would hard-fail on it, #14676.) - if str(stored_provider or "").strip().lower() == "custom": - try: - from hermes_cli.runtime_provider import canonical_custom_identity - stored_provider = canonical_custom_identity( - base_url=stored_base_url or None, - model=stored_model or None, - ) or None - except Exception: - stored_provider = None - model_changed = stored_model != self.model - provider_changed = bool(stored_provider) and stored_provider != self.provider - if not model_changed and not provider_changed: - return - self.model = stored_model - if stored_provider: - self.provider = stored_provider - self.requested_provider = stored_provider - if stored_base_url: - self.base_url = stored_base_url - if stored_api_mode: - self.api_mode = stored_api_mode - if provider_changed: - # Stale launch-time explicit overrides belong to the AMBIENT - # provider; carrying them into the restored provider's - # resolution poisons _ensure_runtime_credentials on startup - # resume (same leak _apply_model_switch_result guards against - # by overwriting _explicit_* on every switch). - self._explicit_api_key = None - self._explicit_base_url = stored_base_url - # Re-resolve credentials for the restored provider. api_key is - # never persisted to the session DB (by design) — the normal - # runtime provider resolution owns credentials. - try: - from hermes_cli.runtime_provider import resolve_runtime_provider - resolved = resolve_runtime_provider(requested=stored_provider) - if resolved.get("api_key"): - self.api_key = resolved["api_key"] - self._credential_pool = resolved.get("credential_pool") - if not stored_base_url and resolved.get("base_url"): - self.base_url = resolved["base_url"] - if not stored_api_mode and resolved.get("api_mode"): - self.api_mode = resolved["api_mode"] - except Exception: - logger.debug( - "Credential re-resolution for resumed session provider " - "%s failed; keeping ambient credentials", - stored_provider, exc_info=True, - ) - # If the agent is already running (mid-chat /resume), swap it - # in-place so the next turn uses the restored model. On startup - # --resume the agent isn't built yet — _init_agent will pick up - # self.model / self.provider when constructing AIAgent. - if self.agent is not None: - try: - self.agent.switch_model( - new_model=self.model, - new_provider=self.provider, - api_key=self.api_key or "", - base_url=self.base_url or "", - api_mode=self.api_mode or "", - ) - except Exception: - logger.debug( - "In-place agent model swap on resume failed", exc_info=True - ) - msg = f"Model restored from session: {stored_model}" - if stored_provider: - msg += f" ({stored_provider})" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") - - - - def _render_resume_history_panel_lines(self, panel) -> list[str]: - """Render the resume panel at the current terminal width for resize replay.""" - from io import StringIO - - buf = StringIO() - width = shutil.get_terminal_size((80, 24)).columns - console = Console( - file=buf, - force_terminal=True, - color_system="truecolor", - highlight=False, - width=width, - ) - with _suspend_output_history(): - console.print(panel) - return buf.getvalue().rstrip("\n").splitlines() - - def _try_attach_clipboard_image(self) -> bool: - """Check clipboard for an image and attach it if found. - - Saves the image to ~/.hermes/images/ and appends the path to - ``_attached_images``. Returns True if an image was attached. - """ - from hermes_cli.clipboard import save_clipboard_image - - img_dir = get_hermes_home() / "images" - self._image_counter += 1 - ts = datetime.now().strftime("%Y%m%d_%H%M%S") - img_path = img_dir / f"clip_{ts}_{self._image_counter}.png" - - if save_clipboard_image(img_path): - self._attached_images.append(img_path) - return True - self._image_counter -= 1 - return False - - - def _resolve_checkpoint_ref(self, ref: str, checkpoints: list) -> str | None: - """Resolve a checkpoint number or hash to a full commit hash.""" - try: - idx = int(ref) - 1 # 1-indexed for user - if 0 <= idx < len(checkpoints): - return checkpoints[idx]["hash"] - else: - print(f" Invalid checkpoint number. Use 1-{len(checkpoints)}.") - return None - except ValueError: - # Treat as a git hash - return ref - - - - - - def _write_osc52_clipboard(self, text: str) -> None: - """Copy *text* to terminal clipboard via OSC 52. - - Wrapped for tmux/screen passthrough (mirrors the TUI's - wrapForMultiplexer in ui-tui/src/lib/osc52.ts) — without the DCS - wrapper the multiplexer consumes the sequence and the copy is - silently lost. - """ - payload = base64.b64encode(text.encode("utf-8")).decode("ascii") - seq = f"\x1b]52;c;{payload}\x07" - if os.environ.get("TMUX"): - seq = "\x1bPtmux;" + seq.replace("\x1b", "\x1b\x1b") + "\x1b\\" - elif os.environ.get("STY"): - seq = "\x1bP" + seq + "\x1b\\" - out = getattr(self, "_app", None) - output = getattr(out, "output", None) if out else None - if output and hasattr(output, "write_raw"): - output.write_raw(seq) - output.flush() - return - if output and hasattr(output, "write"): - output.write(seq) - output.flush() - return - sys.stdout.write(seq) - sys.stdout.flush() - - def _recover_terminal_input_modes(self, *, reason: str) -> None: - """Best-effort reset when leaked mouse reports indicate mode drift.""" - now = time.monotonic() - # Rate-limit to avoid thrashing if a terminal floods reports. - if now - self._last_input_mode_recovery < 0.5: - return - self._last_input_mode_recovery = now - - out = getattr(self, "_app", None) - output = getattr(out, "output", None) if out else None - try: - if output and hasattr(output, "write_raw"): - output.write_raw(_TERMINAL_INPUT_MODE_RESET_SEQ) - output.flush() - elif output and hasattr(output, "write"): - output.write(_TERMINAL_INPUT_MODE_RESET_SEQ) - output.flush() - else: - sys.stdout.write(_TERMINAL_INPUT_MODE_RESET_SEQ) - sys.stdout.flush() - except Exception: - return - - # The reset sequence above pops kitty keyboard mode and resets - # modifyOtherKeys too — re-request extended keys so Shift+Enter / - # modified-key reporting isn't silently dead for the rest of the - # session after a recovery (sibling of the startup push). - try: - if _cli_multiline_shortcuts_enabled(self.config or CLI_CONFIG): - _enable_extended_enter_keys(output) - except Exception: - pass - - logger.warning("Recovered terminal input modes after leak: %s", reason) - if not self._input_mode_recovery_notice_shown: - self._input_mode_recovery_notice_shown = True - _cprint( - f" {_DIM}Recovered terminal input modes after leaked mouse reports. " - f"If this repeats, run /new or restart this tab.{_RST}" - ) - - def _check_termios_drift(self) -> None: - """Watchdog: heal the tty if it drifted back to cooked mode. - - See ``_heal_cooked_mode_drift`` for the failure class (a lost - ``run_in_terminal`` cooked→raw restore leaves the terminal - line-buffering keystrokes while the prompt_toolkit app believes it - owns raw mode — the CLI looks dead but the process is healthy). - - Called from ``process_loop``'s idle branch, so a drifted terminal - self-heals within ~a second of the agent going idle instead of - requiring an external ``stty`` rescue. Skipped while a - ``run_in_terminal`` window is legitimately holding cooked mode - (``app._running_in_terminal``), while the agent is running (approval - prompts and sudo prompts legitimately manipulate the tty), and on - Windows (no termios). - """ - if os.name == "nt": - return - app = getattr(self, "_app", None) - if app is None or not getattr(app, "_is_running", False): - return - # A run_in_terminal window is *supposed* to be cooked — don't fight it. - if getattr(app, "_running_in_terminal", False): - return - now = time.monotonic() - if now - self._last_termios_drift_check < 1.0: - return - self._last_termios_drift_check = now - try: - if not sys.stdin.isatty(): - return - fd = sys.stdin.fileno() - except Exception: - return - if _heal_cooked_mode_drift(fd): - logger.warning( - "Healed cooked-mode termios drift on stdin — a " - "run_in_terminal cooked→raw restore was lost." - ) - # Redraw so the prompt is visibly alive again. - try: - self._invalidate() - except Exception: - pass - if not self._termios_drift_notice_shown: - self._termios_drift_notice_shown = True - _cprint( - f" {_DIM}Recovered terminal from cooked-mode drift " - f"(input should respond normally again).{_RST}" - ) - - - - def _preprocess_images_with_vision(self, text: str, images: list, *, announce: bool = True) -> str: - """Analyze attached images via the vision tool and return enriched text. - - Instead of embedding raw base64 ``image_url`` content parts in the - conversation (which only works with vision-capable models), this - pre-processes each image through the auxiliary vision model (Gemini - Flash) and prepends the descriptions to the user's message — the - same approach the messaging gateway uses. - - The local file path is included so the agent can re-examine the - image later with ``vision_analyze`` if needed. - """ - import asyncio as _asyncio - from tools.vision_tools import vision_analyze_tool - - analysis_prompt = ( - "Describe everything visible in this image in thorough detail. " - "Include any text, code, data, objects, people, layout, colors, " - "and any other notable visual information." - ) - - enriched_parts = [] - for img_path in images: - if not img_path.exists(): - continue - size_kb = img_path.stat().st_size // 1024 - if announce: - _cprint(f" {_DIM}👁️ analyzing {img_path.name} ({size_kb}KB)...{_RST}") - try: - result_json = _asyncio.run( - vision_analyze_tool(image_url=str(img_path), user_prompt=analysis_prompt) - ) - result = json.loads(result_json) - if result.get("success"): - description = result.get("analysis", "") - enriched_parts.append( - f"[The user attached an image. Here's what it contains:\n{description}]\n" - f"[If you need a closer look, use vision_analyze with " - f"image_url: {img_path}]" - ) - if announce: - _cprint(f" {_DIM}✓ image analyzed{_RST}") - else: - enriched_parts.append( - f"[The user attached an image but it couldn't be analyzed. " - f"You can try examining it with vision_analyze using " - f"image_url: {img_path}]" - ) - if announce: - _cprint(f" {_DIM}⚠ vision analysis failed — path included for retry{_RST}") - except Exception as e: - enriched_parts.append( - f"[The user attached an image but analysis failed ({e}). " - f"You can try examining it with vision_analyze using " - f"image_url: {img_path}]" - ) - if announce: - _cprint(f" {_DIM}⚠ vision analysis error — path included for retry{_RST}") - - # Combine: vision descriptions first, then the user's original text - user_text = text if isinstance(text, str) and text else "" - if enriched_parts: - prefix = "\n\n".join(enriched_parts) - return f"{prefix}\n\n{user_text}" if user_text else prefix - return user_text or "What do you see in this image?" - def _show_tool_availability_warnings(self): """Show warnings about disabled tools due to missing API keys.""" try: @@ -9740,398 +6091,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): except Exception: pass # Don't crash on import errors - def _show_status(self): - """Show compact startup status line.""" - # Avoid pulling the full tool registry into the bare Termux prompt path. - if os.environ.get("HERMES_DEFER_AGENT_STARTUP") == "1": - tool_status = "tools deferred" - else: - tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) - tool_count = len(tools) if tools else 0 - tool_status = f"{tool_count} tools" - - # Format model name (shorten if needed) - model_short = self.model.split("/")[-1] if "/" in self.model else self.model - if len(model_short) > 30: - model_short = model_short[:27] + "..." - - # Get API status indicator - api_indicator = "[green bold]●[/]" if self.api_key else "[red bold]●[/]" - - # Build status line with proper markup — skin-aware colors - try: - from hermes_cli.skin_engine import get_active_skin - skin = get_active_skin() - separator_color = skin.get_color("banner_dim", "#B8860B") - accent_color = skin.get_color("ui_accent", "#FFBF00") - label_color = skin.get_color("ui_label", "#DAA520") - except Exception: - separator_color, accent_color, label_color = "#B8860B", "#FFBF00", "cyan" - toolsets_info = "" - if self.enabled_toolsets and "all" not in self.enabled_toolsets: - toolsets_info = f" [dim {separator_color}]·[/] [{label_color}]toolsets: {', '.join(self.enabled_toolsets)}[/]" - - provider_info = f" [dim {separator_color}]·[/] [dim]provider: {self.provider}[/]" - if self._provider_source: - provider_info += f" [dim {separator_color}]·[/] [dim]auth: {self._provider_source}[/]" - - self._console_print( - f" {api_indicator} [{accent_color}]{model_short}[/] " - f"[dim {separator_color}]·[/] [bold {label_color}]{tool_status}[/]" - f"{toolsets_info}{provider_info}" - ) - - def _show_session_status(self): - """Show gateway-style status for the current CLI session.""" - session_meta = {} - if self._session_db: - try: - session_meta = self._session_db.get_session(self.session_id) or {} - except Exception: - session_meta = {} - - title = (session_meta.get("title") or "").strip() - - created_at = self.session_start - started_at = session_meta.get("started_at") - if started_at: - try: - created_at = datetime.fromtimestamp(float(started_at)) - except Exception: - created_at = self.session_start - - updated_at = created_at - for field in ("updated_at", "last_updated_at", "last_activity_at"): - value = session_meta.get(field) - if not value: - continue - try: - updated_at = datetime.fromtimestamp(float(value)) - break - except Exception: - pass - - agent = getattr(self, "agent", None) - total_tokens = getattr(agent, "session_total_tokens", 0) or 0 - provider = getattr(self, "provider", None) or "unknown" - model = getattr(self, "model", None) or "(unknown)" - is_running = bool(getattr(self, "_agent_running", False)) - - # Reasoning level (C-02): resolve the effective effort for display. - reasoning_label = None - try: - rc = getattr(agent, "reasoning_config", None) or getattr(self, "reasoning_config", None) - if isinstance(rc, dict): - if rc.get("enabled") is False: - reasoning_label = "off" - elif rc.get("effort"): - reasoning_label = str(rc.get("effort")) - show_r = getattr(self, "show_reasoning", None) - if reasoning_label: - reasoning_label += f" (display: {'on' if show_r else 'off'})" if show_r is not None else "" - except Exception: - reasoning_label = None - - # Approval mode (C-02). - approval_label = None - try: - from tools.approval import _get_approval_mode, is_approval_bypass_active_for_session - approval_label = _get_approval_mode() - try: - if is_approval_bypass_active_for_session(getattr(self, "session_key", "") or ""): - approval_label += " (YOLO bypass active)" - except Exception: - pass - except Exception: - approval_label = None - - # Context window usage (C-02): reuse the status-bar snapshot which - # already computes tokens / max / percent. - ctx_label = None - try: - snap = self._get_status_bar_snapshot() - ctx_tokens = snap.get("context_tokens") or 0 - ctx_max = snap.get("context_length") - ctx_pct = snap.get("context_percent") - if ctx_max: - left = "" - if isinstance(ctx_pct, (int, float)): - left = f"{max(0, 100 - int(ctx_pct))}% left · " - ctx_label = f"{left}{ctx_tokens:,} / {ctx_max:,} tokens used" - except Exception: - ctx_label = None - - lines = [ - "Hermes CLI Status", - "", - f"Session ID: {self.session_id}", - f"Path: {display_hermes_home()}", - ] - if title: - lines.append(f"Title: {title}") - lines.append(f"Model: {model} ({provider})") - if reasoning_label: - lines.append(f"Reasoning: {reasoning_label}") - if approval_label: - lines.append(f"Approvals: {approval_label}") - if ctx_label: - lines.append(f"Context: {ctx_label}") - lines.extend([ - f"Created: {created_at.strftime('%Y-%m-%d %H:%M')}", - f"Last Activity: {updated_at.strftime('%Y-%m-%d %H:%M')}", - f"Tokens: {total_tokens:,}", - f"Agent Running: {'Yes' if is_running else 'No'}", - ]) - self._console_print("\n".join(lines), highlight=False, markup=False) - - def _fast_command_available(self) -> bool: - try: - from hermes_cli.models import model_supports_fast_mode - except Exception: - return False - agent = getattr(self, "agent", None) - model = getattr(agent, "model", None) or getattr(self, "model", None) - return model_supports_fast_mode(model) - - def _command_available(self, slash_command: str) -> bool: - if slash_command == "/fast": - return self._fast_command_available() - return True - - def show_help(self, arg: str = ""): - """Display help. Bare /help shows categorized core commands with the - skill list collapsed to one line; /help skills lists all skill - commands; /help filters commands by substring. - """ - from hermes_cli.commands import COMMANDS_BY_CATEGORY, HELP_SESSION_SUBGROUPS - - arg = (arg or "").strip() - skill_commands = _ensure_skill_commands() - - # /help skills — the full skill-command list (kept out of the default - # view so core commands don't scroll off screen). - if arg.lower() in ("skills", "skill"): - if not skill_commands: - _cprint("\n No skill commands installed.\n") - return - _cprint(f"\n ⚡ {_BOLD}Skill Commands{_RST} ({len(skill_commands)} installed):") - for cmd, info in sorted(skill_commands.items()): - ChatConsole().print( - f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] {_escape(info['description'])}" - ) - _cprint("") - return - - query = arg.lower() if arg else "" - - try: - from hermes_cli.skin_engine import get_active_help_header - header = get_active_help_header("(^_^)? Available Commands") - except Exception: - header = "(^_^)? Available Commands" - header = (header or "").strip() or "(^_^)? Available Commands" - inner_width = 55 - if len(header) > inner_width: - header = header[:inner_width] - _cprint(f"\n{_BOLD}+{'-' * inner_width}+{_RST}") - _cprint(f"{_BOLD}|{header:^{inner_width}}|{_RST}") - _cprint(f"{_BOLD}+{'-' * inner_width}+{_RST}") - - def _emit(cmd: str, desc: str) -> bool: - if not self._command_available(cmd): - return False - if query and query not in cmd.lower() and query not in desc.lower(): - return False - ChatConsole().print( - f" [bold {_accent_hex()}]{cmd:<15}[/] [dim]-[/] {_escape(desc)}" - ) - return True - - for category, commands in COMMANDS_BY_CATEGORY.items(): - if category == "Session": - # Split the oversized Session category into readable sub-groups - # (Session / Context / Background & Automation) in the renderer. - sub_of: dict[str, str] = {} - for _sub, _names in HELP_SESSION_SUBGROUPS.items(): - for _n in _names: - sub_of[f"/{_n}"] = _sub - buckets: dict[str, list[tuple[str, str]]] = {"Session": []} - for _sub in HELP_SESSION_SUBGROUPS: - buckets[_sub] = [] - for cmd, desc in commands.items(): - buckets[sub_of.get(cmd, "Session")].append((cmd, desc)) - for _sub in ("Session", *HELP_SESSION_SUBGROUPS.keys()): - rows = buckets.get(_sub) or [] - printed_header = False - for cmd, desc in rows: - if not self._command_available(cmd): - continue - if query and query not in cmd.lower() and query not in desc.lower(): - continue - if not printed_header: - _cprint(f"\n {_BOLD}── {_sub} ──{_RST}") - printed_header = True - _emit(cmd, desc) - continue - - printed_header = False - for cmd, desc in commands.items(): - if not self._command_available(cmd): - continue - if query and query not in cmd.lower() and query not in desc.lower(): - continue - if not printed_header: - _cprint(f"\n {_BOLD}── {category} ──{_RST}") - printed_header = True - _emit(cmd, desc) - - # Skill commands: collapsed to a one-line pointer by default so the - # 60+ skill entries don't bury the core command reference (C-04). - if query: - # In filter mode, DO include matching skill commands inline. - matched_skills = [ - (cmd, info) for cmd, info in sorted(skill_commands.items()) - if query in cmd.lower() or query in (info.get("description", "").lower()) - ] - if matched_skills: - _cprint(f"\n ⚡ {_BOLD}Skill Commands{_RST} (matching '{arg}'):") - for cmd, info in matched_skills: - ChatConsole().print( - f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] {_escape(info['description'])}" - ) - elif skill_commands: - _cprint( - f"\n ⚡ {_BOLD}Skill Commands{_RST}: {len(skill_commands)} installed " - f"— {_DIM}/help skills{_RST} to list them" - ) - - _bundles_now = get_skill_bundles() - if _bundles_now and not query: - _cprint(f"\n ▣ {_BOLD}Skill Bundles{_RST} ({len(_bundles_now)} installed):") - for cmd, info in sorted(_bundles_now.items()): - skill_count = len(info.get("skills", [])) - desc = info.get("description") or f"Load {skill_count} skills" - ChatConsole().print( - f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] " - f"{_escape(desc)} [dim]({skill_count} skills)[/]" - ) - - quick_commands = self.config.get("quick_commands", {}) - if quick_commands and not query: - _cprint(f"\n ⚡ {_BOLD}Quick Commands{_RST} ({len(quick_commands)} configured):") - for name, qcmd in sorted(quick_commands.items()): - desc = qcmd.get("description", qcmd.get("type", "")) - ChatConsole().print( - f" [bold {_accent_hex()}]{('/' + name):<22}[/] [dim]-[/] {_escape(desc)}" - ) - - if query: - _cprint(f"\n {_DIM}Filtered by '{arg}' — run /help for the full list.{_RST}\n") - return - - _cprint(f"\n {_DIM}Tip: /help skills lists skill commands · /help filters · Ctrl+P opens the command palette{_RST}") - _cprint(f" {_DIM}Multi-line: Ctrl+J, Alt+Enter, or \\\\+Enter for a new line{_RST}") - _cprint(f" {_DIM}Draft editor: Ctrl+G (Alt+G in VSCode/Cursor){_RST}") - if _is_termux_environment(): - _cprint(f" {_DIM}Attach image: /image {_termux_example_image_path()} or start your prompt with a local image path{_RST}\n") - else: - _cprint(f" {_DIM}Paste image: Alt+V (or /paste){_RST}\n") - - def show_tools(self): - """Display available tools with kawaii ASCII art.""" - # Pre-assembly list: /tools is a discovery/inspection surface, so it - # must show the full catalog including tools deferred behind the - # tool_search bridge (users check this to verify an MCP installed). - tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True, - skip_tool_search_assembly=True) - - if not tools: - print("(;_;) No tools available") - return - - # Header - print() - title = "(^_^)/ Available Tools" - width = 78 - pad = width - len(title) - print("+" + "-" * width + "+") - print("|" + " " * (pad // 2) + title + " " * (pad - pad // 2) + "|") - print("+" + "-" * width + "+") - print() - - # Group tools by toolset - toolsets = {} - for tool in sorted(tools, key=lambda t: t["function"]["name"]): - name = tool["function"]["name"] - toolset = get_toolset_for_tool(name) or "unknown" - if toolset not in toolsets: - toolsets[toolset] = [] - desc = tool["function"].get("description", "") - # First sentence: split on ". " (period+space) to avoid breaking on "e.g." or "v2.0" - desc = desc.split("\n")[0] - if ". " in desc: - desc = desc[:desc.index(". ") + 1] - toolsets[toolset].append((name, desc)) - - # Display by toolset - for toolset in sorted(toolsets.keys()): - print(f" [{toolset}]") - for name, desc in toolsets[toolset]: - print(f" * {name:<20} - {desc}") - print() - - print(f" Total: {len(tools)} tools ヽ(^o^)ノ") - print() - - - def show_toolsets(self): - """Display available toolsets with kawaii ASCII art.""" - all_toolsets = get_all_toolsets() - - # Header - print() - title = "(^_^)b Available Toolsets" - width = 58 - pad = width - len(title) - print("+" + "-" * width + "+") - print("|" + " " * (pad // 2) + title + " " * (pad - pad // 2) + "|") - print("+" + "-" * width + "+") - print() - - for name in sorted(all_toolsets.keys()): - info = get_toolset_info(name) - if info: - tool_count = info["tool_count"] - desc = info["description"] - - # Mark if currently enabled - marker = "(*)" if self.enabled_toolsets and name in self.enabled_toolsets else " " - print(f" {marker} {name:<18} [{tool_count:>2} tools] - {desc}") - - print() - print(" (*) = currently enabled") - print() - print(" Tip: Use 'all' or '*' to enable all toolsets") - print(" Example: python cli.py --toolsets web,terminal") - print() - - - def _handle_whoami_command(self): - """Display slash-command access for the local CLI surface.""" - import getpass - - try: - user_name = getpass.getuser() or "?" - except Exception: - user_name = "?" - - print() - print(" You: cli (local terminal)") - print(f" User: {user_name}") - print(" Tier: unrestricted") - print(" Slash commands: all available") - print() - def show_config(self): """Display current configuration with kawaii ASCII art.""" # Get terminal config from environment (which was set from cli-config.yaml) @@ -10202,2256 +6161,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): print(f" Config File: {config_path} {config_status}") print() - def _list_recent_sessions(self, limit: int = 10) -> list[dict[str, Any]]: - """Return recent CLI sessions for in-chat browsing/resume affordances.""" - if not self._session_db: - return [] - try: - from hermes_cli.session_listing import query_session_listing - - return query_session_listing( - self._session_db, - source="cli", - current_session_id=self.session_id, - include_all_sources=False, - include_unnamed=True, - limit=limit, - exclude_sources=["kanban", "tool"], - ) - except Exception: - return [] - - def _show_recent_sessions(self, *, reason: str = "history", limit: int = 10) -> bool: - """Render recent sessions inline from the active chat TUI. - - Returns True when something was shown, False if no session list was available. - """ - sessions = self._list_recent_sessions(limit=limit) - if not sessions: - return False - - from hermes_cli.main import _relative_time - - _cli_visible_print() - if reason == "history": - _cli_visible_print("(._.) No messages in the current chat yet — here are recent sessions you can resume:") - else: - _cli_visible_print(" Recent sessions:") - _cli_visible_print() - _cli_visible_print(f" {'#':<3} {'Title':<32} {'Preview':<40} {'Last Active':<13} {'ID'}") - _cli_visible_print(f" {'─' * 3} {'─' * 32} {'─' * 40} {'─' * 13} {'─' * 24}") - for idx, session in enumerate(sessions, start=1): - title = session.get("title") or "—" - preview = (session.get("preview") or "")[:38] - last_active = _relative_time(session.get("last_active")) - _cli_visible_print(f" {idx:<3} {title:<32} {preview:<40} {last_active:<13} {session['id']}") - _cli_visible_print() - _cli_visible_print(" Use /resume , /resume , or /resume to continue.") - _cli_visible_print(" Example: /resume 2") - _cli_visible_print() - return True - - def show_history(self): - """Display conversation history.""" - if not self.conversation_history: - if not self._show_recent_sessions(reason="history"): - _cli_visible_print("(._.) No conversation history yet.") - return - - preview_limit = 400 - visible_index = 0 - hidden_tool_messages = 0 - show_ts = bool(getattr(self, "show_timestamps", False)) - - def _ts_suffix(message: dict) -> str: - # Messages restored from SessionDB carry a unix `timestamp`; live - # unsaved turns may not. Only annotate when both the toggle is on - # and the turn actually has a stored time — never fabricate one. - if not show_ts: - return "" - ts = message.get("timestamp") - if not ts: - return "" - try: - from datetime import datetime - return f" [{datetime.fromtimestamp(float(ts)).strftime(getattr(self, 'timestamp_format', '%H:%M'))}]" - except (ValueError, OSError, TypeError): - return "" - - def flush_tool_summary(): - nonlocal hidden_tool_messages - if not hidden_tool_messages: - return - - noun = "message" if hidden_tool_messages == 1 else "messages" - _cli_visible_print("\n [Tools]") - _cli_visible_print(f" ({hidden_tool_messages} tool {noun} hidden)") - hidden_tool_messages = 0 - - _cli_visible_print() - _cli_visible_print("+" + "-" * 50 + "+") - _cli_visible_print("|" + " " * 12 + "(^_^) Conversation History" + " " * 11 + "|") - _cli_visible_print("+" + "-" * 50 + "+") - - for msg in self.conversation_history: - role = msg.get("role", "unknown") - - if role == "tool": - hidden_tool_messages += 1 - continue - - if role not in {"user", "assistant"}: - continue - - flush_tool_summary() - visible_index += 1 - - content = msg.get("content") - content_text = "" if content is None else str(content) - - if role == "user": - _cli_visible_print(f"\n [You #{visible_index}]{_ts_suffix(msg)}") - _cli_visible_print( - f" {content_text[:preview_limit]}{'...' if len(content_text) > preview_limit else ''}" - ) - continue - - _cli_visible_print(f"\n [Hermes #{visible_index}]{_ts_suffix(msg)}") - tool_calls = msg.get("tool_calls") or [] - if content_text: - preview = content_text[:preview_limit] - suffix = "..." if len(content_text) > preview_limit else "" - elif tool_calls: - tool_count = len(tool_calls) - noun = "call" if tool_count == 1 else "calls" - preview = f"(requested {tool_count} tool {noun})" - suffix = "" - else: - preview = "(no text response)" - suffix = "" - _cli_visible_print(f" {preview}{suffix}") - - flush_tool_summary() - _cli_visible_print() - - def _notify_session_boundary(self, event_type: str) -> None: - """Fire a session-boundary plugin hook (on_session_finalize or on_session_reset). - - Non-blocking — errors are caught and logged. Safe to call from any - lifecycle point (shutdown, /new, /reset). - """ - try: - from hermes_cli.lifecycle import finalize_session, invoke_hook - - context = { - "session_id": self.agent.session_id if self.agent else None, - "platform": getattr(self, "platform", None) or "cli", - "reason": ( - "new_session" - if event_type == "on_session_reset" - else "session_boundary" - ), - } - if event_type == "on_session_finalize": - finalize_session(**context) - else: - invoke_hook(event_type, **context) - except Exception: - pass - - def _discard_session_if_empty(self, session_id: Optional[str]) -> bool: - """Drop a just-ended session row when it never gained content. - - Starting the CLI and immediately quitting (or rotating with /new, - /clear) used to leave an empty untitled row behind that clutters - ``/resume`` and ``hermes sessions list``. Delegates the - check-and-delete to ``SessionDB.delete_session_if_empty``, which - only removes rows with no messages, no title, and no child - sessions. Ported from google-gemini/gemini-cli#27770. - """ - if not self._session_db or not session_id: - return False - # In-memory transcript is authoritative: if this CLI object holds - # conversation messages (flushed to the DB or not), the session is - # not empty. Protects against pruning a real conversation whose DB - # flush failed or hasn't happened yet. - if getattr(self, "conversation_history", None): - return False - try: - from hermes_constants import get_hermes_home as _ghh - return self._session_db.delete_session_if_empty( - session_id, sessions_dir=_ghh() / "sessions" - ) - except Exception: - logger.debug( - "Could not prune empty session %s", session_id, exc_info=True - ) - return False - - def _launch_session_boundary_memory_flush( - self, - history_snapshot: list, - *, - session_id: Optional[str] = None, - ) -> Optional[list]: - """Stage old-session memory extraction so /new stays responsive. - - The context-engine ``on_session_end`` boundary is delivered - synchronously here: it is cheap (local state clear, no LLM call) and - ordering-sensitive — it must land before ``reset_session_state()`` - rebinds the engine to the new session. - - The memory-provider half (LLM-bound extraction, seconds) is NOT run - here. The returned snapshot is handed by ``new_session()`` to - ``MemoryManager.commit_session_boundary_async`` as a single - end→switch task on the manager's serialized background worker, so - extraction can never race the provider rebinding (providers key off - internal ``_session_id`` state — a late ``on_session_end`` after - ``on_session_switch`` would misattribute the old transcript to the - new session). - - Returns the history snapshot to queue, or ``None`` when there is - nothing to extract (no agent / empty history / no memory manager). - """ - agent = getattr(self, "agent", None) - if not agent or not history_snapshot: - return None - - engine = getattr(agent, "context_compressor", None) - if engine is not None and hasattr(engine, "on_session_end"): - try: - engine.on_session_end(session_id or "", history_snapshot) - except Exception: - logger.debug( - "Context engine on_session_end failed at /new boundary", - exc_info=True, - ) - - # No provider extraction to queue when no memory manager is - # configured — new_session() falls back to the inline switch path. - if getattr(agent, "_memory_manager", None) is None: - return None - return history_snapshot - - def new_session(self, silent=False, title=None): - """Start a fresh session with a new session ID and cleared agent state.""" - old_session_id = self.session_id - _boundary_snapshot = None - if self.agent and self.conversation_history: - # Deliver the context-engine boundary synchronously and get back - # the history snapshot for the deferred provider extraction — - # queued below (after rotation) so /new never blocks on the - # LLM-bound extraction call. - _boundary_snapshot = self._launch_session_boundary_memory_flush( - list(self.conversation_history), - session_id=old_session_id, - ) - self._notify_session_boundary("on_session_finalize") - elif self.agent: - # First session or empty history — still finalize the old session - self._notify_session_boundary("on_session_finalize") - - if self._session_db and old_session_id: - # Flush any un-persisted messages from the current turn to the - # old session *before* rotating. /new can be called mid-turn - # when _flush_messages_to_session_db() has not yet run — without - # this, messages generated during the current turn are silently - # lost on session rotation (#47202). - if self.agent: - try: - self.agent._flush_messages_to_session_db( - self.conversation_history, - conversation_history=self.conversation_history, - ) - except Exception: - pass # best-effort - try: - self._session_db.end_session(old_session_id, "new_session") - except Exception: - pass - # Don't let immediately-rotated empty sessions pile up in - # /resume and `hermes sessions list` (gemini-cli#27770 port). - self._discard_session_if_empty(old_session_id) - - self.session_start = datetime.now() - timestamp_str = self.session_start.strftime("%Y%m%d_%H%M%S") - short_uuid = uuid.uuid4().hex[:6] - self.session_id = f"{timestamp_str}_{short_uuid}" - getattr(self, "_write_terminal_breadcrumb", lambda: None)() - self.conversation_history = [] - self._pending_title = None - self._resumed = False - # /new clears the -m / --model override flag: an explicit CLI model - # was for the previous session only, not for every session spawned - # afterwards. - self._explicit_model_override = False - self.reasoning_config = _parse_reasoning_config( - CLI_CONFIG["agent"].get("reasoning_effort", "") - ) - # /new is a full conversation boundary: session-scoped runtime - # overrides (/model --session, /fast, one-turn restores) do not carry - # forward. Re-derive model/provider and service tier from config.yaml - # so a session-only switch never leaks into the next session (#48055, - # #23131). - self._pending_one_turn_model_restore = None - self.service_tier = _parse_service_tier_config( - CLI_CONFIG["agent"].get("service_tier", "") - ) - _model_config = CLI_CONFIG.get("model", {}) - _raw_default2 = (_model_config.get("default") or _model_config.get("model") or "") if isinstance(_model_config, dict) else (_model_config or "") - _config_model, _ = _split_model_config_default(_raw_default2) - if _config_model and _config_model != getattr(self, "model", None): - _config_provider = ( - _model_config.get("provider", "") - if isinstance(_model_config, dict) - else "" - ) - try: - from hermes_cli.model_switch import switch_model as _switch_model - - _reset_result = _switch_model( - raw_input=_config_model, - current_provider=self.provider or "", - current_model=self.model or "", - current_base_url=self.base_url or "", - current_api_key=self.api_key or "", - is_global=False, - explicit_provider=_config_provider or "", - ) - if _reset_result.success: - if self.agent: - self.agent.switch_model( - new_model=_reset_result.new_model, - new_provider=_reset_result.target_provider, - api_key=_reset_result.api_key, - base_url=_reset_result.base_url, - api_mode=_reset_result.api_mode, - capabilities=getattr( - _reset_result, "runtime_capabilities", None - ), - ) - self.model = _reset_result.new_model - self.provider = _reset_result.target_provider - self.requested_provider = _reset_result.target_provider - self._explicit_api_key = _reset_result.api_key - self._explicit_base_url = _reset_result.base_url - if _reset_result.api_key: - self.api_key = _reset_result.api_key - if _reset_result.base_url: - self.base_url = _reset_result.base_url - if _reset_result.api_mode: - self.api_mode = _reset_result.api_mode - if not silent: - _cprint( - f" (model reset to config default: " - f"{_reset_result.new_model})" - ) - except Exception: - # Best-effort: an unreachable config default must never block - # /new. The session keeps the current working model. - logger.debug("/new model reset to config default failed", exc_info=True) - _sync_process_session_id(self.session_id) - - if self.agent: - self.agent.session_id = self.session_id - self.agent.session_start = self.session_start - self.agent.reasoning_config = self.reasoning_config - self.agent.reset_session_state() - if hasattr(self.agent, "_last_flushed_db_idx"): - self.agent._last_flushed_db_idx = 0 - if hasattr(self.agent, "_todo_store"): - try: - from tools.todo_tool import TodoStore - self.agent._todo_store = TodoStore() - except Exception: - pass - if hasattr(self.agent, "_invalidate_system_prompt"): - self.agent._invalidate_system_prompt() - - if self._session_db: - try: - self.agent._session_db_created = False - self._session_db.create_session( - session_id=self.session_id, - source=os.environ.get("HERMES_SESSION_SOURCE", "cli"), - model=self.model, - model_config={ - "max_iterations": self.max_turns, - "reasoning_config": self.reasoning_config, - }, - ) - self.agent._session_db_created = True - except Exception: - pass - if title and self._session_db: - from hermes_state import SessionDB - try: - sanitized = SessionDB.sanitize_title(title) - except ValueError as e: - _cprint(f" Title rejected: {e}") - sanitized = None - title = None - if sanitized: - try: - self._session_db.set_session_title(self.session_id, sanitized) - self._pending_title = None - self._status_bar_title_checked_at = 0.0 - title = sanitized - except ValueError as e: - _cprint(f" {e} — session started untitled.") - title = None - except Exception: - title = None - elif title is not None: - # sanitize_title returned empty (whitespace-only / unprintable) - _cprint(" Title is empty after cleanup — session started untitled.") - title = None - # Notify memory providers that session_id rotated to a fresh - # conversation. reset=True signals providers to flush accumulated - # per-session state (_session_turns, _turn_counter, _document_id). - # Fires BEFORE the plugin on_session_reset hook (shell hooks only - # see the new id; Python providers see the transition). See #6672. - # - # When the old session has history, end-of-session extraction - # (LLM-bound, seconds) and this switch are queued as ONE task on - # the memory manager's serialized worker — end strictly before - # switch, without blocking /new (#16454). With no history there - # is nothing to extract; switch inline as before. - try: - _mm = getattr(self.agent, "_memory_manager", None) - if _mm is not None: - if _boundary_snapshot: - _mm.commit_session_boundary_async( - _boundary_snapshot, - new_session_id=self.session_id, - parent_session_id=old_session_id or "", - reason="new_session", - ) - else: - _mm.on_session_switch( - self.session_id, - parent_session_id=old_session_id or "", - reset=True, - reason="new_session", - ) - except Exception: - pass - self._notify_session_boundary("on_session_reset") - - if not silent: - if title: - print(f"(^_^)v New session started: {title}") - else: - print("(^_^)v New session started!") - - - def _consume_pending_resume_selection(self, text: str) -> bool: - """Resolve a bare numeric reply that follows a bare ``/resume`` prompt. - - After ``/resume`` (no args) prints the recent-sessions list it arms - ``self._pending_resume_sessions``. The next submitted input is given - one chance to be a bare session number (``3``); if so we resume that - session here. Anything else (another command, free text, blank) simply - disarms the prompt and is handled normally by the caller. - - Returns True if the input was consumed as a resume selection (caller - must not treat it as chat); False otherwise. The pending state is - always one-shot: it is cleared on the first submitted input regardless - of outcome. See #34584. - """ - pending = self._pending_resume_sessions - if not pending: - return False - # One-shot: disarm now so a non-matching input can't leave the prompt - # armed and hijack a later number the user meant as chat. - self._pending_resume_sessions = None - - if not isinstance(text, str): - return False - stripped = text.strip() - # Only a pure number selects; let "/resume 3", titles, or any other - # text fall through to normal handling. - if not stripped.isdigit(): - return False - - index = int(stripped) - if index < 1 or index > len(pending): - _cprint(f" Resume index {index} is out of range.") - _cprint(" Use /resume with no arguments to see available sessions.") - return True - - self._handle_resume_command(f"/resume {index}") - return True - - - def save_conversation(self, cmd: str = "/save"): - """Handle /save — export the current session to json, md, or html. - - Usage: ``/save [json|md|html] [filename] [redact]`` - - The snapshot is a convenience export for sharing or off-line - inspection; every message is already persisted incrementally to the - SQLite session DB, so the live session remains resumable via - ``hermes --resume `` regardless of whether the user ever runs - ``/save``. ``redact`` runs the export through the force-mode secret - redaction pass before writing. - """ - from hermes_cli.session_export import ( - SAVE_USAGE, - normalize_save_format, - render_session_for_save, - ) - - parts = cmd.split()[1:] - if not parts: - print(SAVE_USAGE) - return - redact = False - if parts[-1].lower() in ("redact", "--redact"): - redact = True - parts = parts[:-1] - if not parts: - print(SAVE_USAGE) - return - - try: - fmt = normalize_save_format(parts[0]) - except ValueError as e: - print(f"(._.) {e}") - print(SAVE_USAGE) - return - filename = parts[1] if len(parts) > 1 else None - - # Prefer the durable DB row (has metadata + tool calls); fall back to - # the in-memory history for sessions that never touched the DB. - # getattr: test doubles (SimpleNamespace / object.__new__) may not - # carry _session_db or session_id. - session_data = None - _db = getattr(self, "_session_db", None) - _sid = getattr(self, "session_id", None) - if _db and _sid: - try: - session_data = _db.export_session(_sid) - except Exception: - session_data = None - if not session_data: - if not self.conversation_history: - print("(;_;) No conversation to save.") - return - session_data = { - "id": self.session_id, - "model": self.model, - "started_at": self.session_start.timestamp(), - "messages": self.conversation_history, - } - - if redact: - from hermes_cli.session_export_md import redact_session_data - - session_data = redact_session_data(session_data) - - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") - saved_dir = get_hermes_home() / "sessions" / "saved" - try: - saved_dir.mkdir(parents=True, exist_ok=True) - except Exception as e: - print(f"(x_x) Failed to create save directory {saved_dir}: {e}") - return - if filename: - path = Path(filename).expanduser() - if not path.is_absolute(): - path = Path.cwd() / path - else: - path = saved_dir / f"hermes_conversation_{timestamp}.{fmt}" - - try: - content = render_session_for_save(session_data, fmt) - with open(path, "w", encoding="utf-8") as f: - f.write(content) - label = {"json": "JSON", "md": "Markdown", "html": "HTML"}[fmt] - print(f"(^_^)v Conversation saved to: {path} ({label})") - if self.session_id: - print(f" Resume the live session with: hermes --resume {self.session_id}") - except Exception as e: - print(f"(x_x) Failed to save: {e}") - - def _rewind_persisted_user_turn( - self, - *, - warm_history: List[Dict[str, Any]], - user_ordinal: int, - warm_live_view: Dict[str, Any], - ) -> tuple[List[Dict[str, Any]], Dict[str, Any], Dict[str, Any]]: - """Bind one warm user ordinal to a durable row and rewind it atomically.""" - if self._session_db is None or not self.session_id: - raise RuntimeError("session database is unavailable") - - from agent.context_compressor import ( - history_before_user_originated_turn, - split_user_originated_turn, - user_originated_turn_view, - ) - from agent.memory_manager import sanitize_context - from agent.tool_dispatch_helpers import ( - _is_multimodal_tool_result, - _multimodal_text_summary, - ) - from run_agent import _is_ephemeral_scaffolding - - def _persistence_content(content: Any) -> Any: - """Project warm content exactly as the session DB flush does.""" - if _is_multimodal_tool_result(content): - return _multimodal_text_summary(content) - if isinstance(content, list): - text_parts = [] - for part in content: - if isinstance(part, dict) and part.get("type") == "text": - text_parts.append(str(part.get("text", ""))) - elif isinstance(part, dict) and part.get("type") in { - "image", - "image_url", - "input_image", - }: - text_parts.append("[screenshot]") - return "\n".join(text_parts) if text_parts else None - return content - - def _comparison_content(message: Dict[str, Any]) -> Any: - content = _persistence_content(message.get("content")) - if message.get("role") in {"user", "assistant"} and isinstance( - content, str - ): - return sanitize_context(content).strip() - return content - - expected_active_ids = self._session_db.get_active_message_ids( - self.session_id - ) - durable = self._session_db.get_messages_as_conversation( - self.session_id, - include_row_ids=True, - ) - warm_persistence_history = [ - message - for message in warm_history - if not _is_ephemeral_scaffolding(message) - ] - warm_user_indices = [ - index - for index, message in enumerate(warm_persistence_history) - if user_originated_turn_view(message) is not None - ] - durable_user_indices = [ - index - for index, message in enumerate(durable) - if user_originated_turn_view(message) is not None - ] - if len(durable_user_indices) != len(warm_user_indices): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) - if user_ordinal < 0 or user_ordinal >= len(durable_user_indices): - raise RuntimeError("persisted rewind target is no longer available") - - warm_prefix, _ = history_before_user_originated_turn( - warm_persistence_history, warm_user_indices[user_ordinal] - ) - durable_target_index = durable_user_indices[user_ordinal] - durable_target = durable[durable_target_index] - durable_prefix, durable_live_view = history_before_user_originated_turn( - durable, durable_target_index - ) - if _comparison_content(durable_live_view) != _comparison_content( - warm_live_view - ): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) - target_row_id = durable_target.get("_row_id") - if not isinstance(target_row_id, int): - raise RuntimeError("persisted rewind target has no row identity") - scaffold, _ = split_user_originated_turn(durable_target) - result = self._session_db.rewind_to_message( - self.session_id, - target_row_id, - preserve_compaction_handoff=scaffold is not None, - expected_active_ids=expected_active_ids, - expected_target_content=durable_live_view.get("content"), - ) - if scaffold is not None: - replacement_id = result.get("replacement_message_id") - if not isinstance(replacement_id, int) or not durable_prefix: - raise RuntimeError("rewind did not retain its compaction handoff") - durable_prefix[-1]["_row_id"] = replacement_id - durable_prefix[-1]["_db_persisted"] = True - warm_prefix[-1] = durable_prefix[-1] - return warm_prefix, durable_live_view, result - - def retry_last(self): - """Retry the last user message by removing the last exchange and re-sending. - - Removes the last assistant response (and any tool-call messages) and - the last user message, then re-sends that user message to the agent. - Returns the message to re-send, or None if there's nothing to retry. - """ - if not self.conversation_history: - print("(._.) No messages to retry.") - return None - - # Walk backwards to the last *real* user message. Timeline bookkeeping - # rows (display_kind set) are role=user but are not user turns — match - # CLI resume counting and user_originated_turn_view. Compaction - # handoffs are excluded too (durable role=user, sometimes without - # display_kind on legacy sessions; #80622). - from agent.context_compressor import ( - history_before_user_originated_turn, - retryable_user_text, - user_originated_turn_view, - ) - from agent.memory_manager import sanitize_context - from run_agent import _is_ephemeral_scaffolding - - warm_history = list(self.conversation_history) - - user_indices = [ - index - for index, message in enumerate(warm_history) - if not _is_ephemeral_scaffolding(message) - and user_originated_turn_view(message) is not None - ] - - if not user_indices: - print("(._.) No user message found to retry.") - return None - last_user_idx = user_indices[-1] - - # Resolve a lossless live payload before touching either persistence or - # memory. A force-user-leading compaction row is one physical carrier: - # its historical handoff remains in the prefix while only the embedded - # human ask is retried. Media cannot be replayed by /retry, so fail - # closed before archiving anything. - try: - truncated, live_view = history_before_user_originated_turn( - warm_history, last_user_idx - ) - live_content = live_view.get("content") - if isinstance(live_content, str): - live_content = sanitize_context(live_content).strip() - last_message = retryable_user_text(live_content) - except ValueError as exc: - print(f"(._.) Cannot retry that message safely: {exc}") - return None - - # Persist the rewind before publishing the shorter in-memory view. - # The DB owns the physical carrier split so the archived original and - # retained scaffold are committed atomically. A plain user row keeps - # the legacy rewind shape (no replacement scaffold). - if self._session_db is not None and self.session_id: - try: - truncated, _, _ = self._rewind_persisted_user_turn( - warm_history=warm_history, - user_ordinal=len(user_indices) - 1, - warm_live_view=live_view, - ) - except Exception as exc: - print(f"(x_x) Retry rewind failed; history was not changed: {exc}") - return None - - self.conversation_history = truncated - if self.agent is not None: - if hasattr(self.agent, "_session_messages"): - self.agent._session_messages = self.conversation_history - if hasattr(self.agent, "_last_flushed_db_idx"): - self.agent._last_flushed_db_idx = len(self.conversation_history) - if hasattr(self.agent, "_db_flush_scan_prefix"): - self.agent._db_flush_scan_prefix = self.conversation_history[:] - - print(f"(^_^)b Retrying: \"{last_message[:60]}{'...' if len(last_message) > 60 else ''}\"") - return last_message - - def undo_last(self, n: int = 1, prefill: bool = True): - """Back up N user turns: truncate history, soft-delete on disk, prefill. - - Walks backwards N user messages and discards everything from the - Nth-from-last user message onward (its assistant response, tool - calls, etc.). ``n`` defaults to 1 (the last exchange); ``/undo 3`` - backs up three user turns. If ``n`` exceeds the number of user - turns, it backs up to the oldest one. - - Beyond the in-memory ``conversation_history`` slice, this also: - • soft-deletes the truncated rows in SessionDB (``active=0``) so - they're hidden from re-prompts and search but kept for audit; - • notifies memory providers via ``on_session_switch(rewound=True)``; - • mirrors /branch's agent surgery (system-prompt invalidation + - flush-index reset); - • when ``prefill`` is set and an input buffer is available, - pre-fills the composer with the backed-up message text so it - can be edited and resubmitted. - - ``prefill=False`` is used by callers that drive the undo - programmatically (e.g. checkpoint rollback) and don't want to - touch the user's input buffer. - """ - if not self.conversation_history: - print("(._.) No messages to undo.") - return - - if n < 1: - n = 1 - - # Walk backwards collecting the indices of the last N *real* user - # messages (exclude display_kind timeline rows and compaction - # handoffs — same predicate as user_originated_turn_view, resume - # turn counting, and /retry; #80622). - from agent.context_compressor import ( - history_before_user_originated_turn, - user_originated_turn_view, - ) - from run_agent import _is_ephemeral_scaffolding - - warm_history = list(self.conversation_history) - - user_indices = [ - index - for index, message in enumerate(warm_history) - if not _is_ephemeral_scaffolding(message) - and user_originated_turn_view(message) is not None - ] - - if not user_indices: - print("(._.) No user message found to undo.") - return - - turns_undone = min(n, len(user_indices)) - target_ordinal = len(user_indices) - turns_undone - cut_idx = user_indices[target_ordinal] - - removed_count = len(warm_history) - cut_idx - truncated, live_view = history_before_user_originated_turn( - warm_history, cut_idx - ) - removed_text = self._undo_content_to_text(live_view.get("content")) - - # Soft-delete the truncated rows on disk so re-prompts and search - # see the clean transcript while the rows survive for audit. - rewound_rows = 0 - if self._session_db is not None and self.session_id: - try: - truncated, durable_live_view, result = ( - self._rewind_persisted_user_turn( - warm_history=warm_history, - user_ordinal=target_ordinal, - warm_live_view=live_view, - ) - ) - # Canonicalize the editable prefill before mutation. The raw - # physical carrier contains the reference summary wrapper. - durable_text = self._undo_content_to_text( - durable_live_view.get("content") - ) - if durable_text: - removed_text = durable_text - rewound_rows = result.get("rewound_count", 0) - except Exception as e: - logger.debug("undo: durable rewind failed: %s", e) - print(f"(x_x) Undo failed; history was not changed: {e}") - return - - # Publish only after the durable rewind succeeds (or no store exists). - self.conversation_history = truncated - - # Agent surgery: invalidate the system-prompt cache and reset the - # flush index so the next turn re-flushes from the truncated head. - if self.agent is not None: - if hasattr(self.agent, "_invalidate_system_prompt"): - try: - self.agent._invalidate_system_prompt() - except Exception: - pass - if hasattr(self.agent, "_last_flushed_db_idx"): - try: - self.agent._last_flushed_db_idx = len(self.conversation_history) - except Exception: - pass - if hasattr(self.agent, "_session_messages"): - self.agent._session_messages = self.conversation_history - if hasattr(self.agent, "_db_flush_scan_prefix"): - self.agent._db_flush_scan_prefix = self.conversation_history[:] - # Notify memory providers — same hook /branch fires, with the - # rewound flag so per-turn document caches invalidate (#6672, #21910). - try: - _mm = getattr(self.agent, "_memory_manager", None) - if _mm is not None and self.session_id: - _mm.on_session_switch( - self.session_id, - parent_session_id="", - reset=False, - rewound=True, - ) - except Exception: - pass - - turn_word = "turn" if turns_undone == 1 else "turns" - msg_count = rewound_rows or removed_count - print( - f"(^_^)b Undid {turns_undone} {turn_word} ({msg_count} message(s)). " - f"Backed up to: \"{removed_text[:60]}{'...' if len(removed_text) > 60 else ''}\"" - ) - remaining = len(self.conversation_history) - print(f" {remaining} message(s) remaining in history.") - - # Pre-fill the composer with the backed-up message so the user can - # edit and resubmit (Claude-Code-style). Editable, not auto-sent. - if prefill and removed_text: - self._prefill_input_buffer(removed_text) - - @staticmethod - def _undo_content_to_text(content) -> str: - """Flatten message content (str or content-part list) to plain text.""" - if isinstance(content, str): - return content - if isinstance(content, list): - parts = [ - p.get("text", "") - for p in content - if isinstance(p, dict) and p.get("type") == "text" - ] - return "\n".join(t for t in parts if t) - return "" - - def _prefill_input_buffer(self, text: str) -> None: - """Place ``text`` in the active prompt_toolkit buffer, editable.""" - app = getattr(self, "_app", None) - if app is None: - return - try: - buf = app.current_buffer - buf.text = text - if hasattr(buf, "cursor_position"): - buf.cursor_position = len(text) - app.invalidate() - except Exception as e: - logger.debug("undo: prefill buffer failed: %s", e) - - def _prompt_text_input(self, prompt_text: str) -> str | None: - """Prompt for free-text input safely inside or outside prompt_toolkit. - - ``run_in_terminal`` returns a coroutine that must be awaited by the prompt_toolkit event loop, - which only exists on the main thread. Slash commands are dispatched from - the ``process_loop`` daemon thread (see issue #23185), so calling - ``run_in_terminal`` from there orphans the coroutine — ``_ask`` never runs, - and user keystrokes leak into the composer instead. Fall back to a direct - ``input()`` when we're off the main thread. - """ - import threading - result = [None] - - def _ask(): - try: - result[0] = input(prompt_text).strip() or None - except (KeyboardInterrupt, EOFError): - pass - - in_main_thread = threading.current_thread() is threading.main_thread() - - # Slash-worker guard (#23185 / billing auto-reload hang): when a - # prompt_toolkit app is running but we're on a non-main thread (the - # process_loop / TUI slash-worker daemon thread), stdin is owned by the - # event loop / JSON-RPC pipe. A bare input() there blocks forever until - # the worker's 45s timeout fires. We cannot safely prompt off the main - # thread, so cancel cleanly (None) instead of hanging — mirrors the - # _stdin_fallback discipline in _prompt_text_input_modal. - if self._app and not in_main_thread: - self._invalidate() - return None - - if self._app and in_main_thread: - from prompt_toolkit.application import run_in_terminal - was_visible = self._status_bar_visible - self._status_bar_visible = False - self._app.invalidate() - try: - run_in_terminal(_ask) - except Exception: - # WSL / Warp / certain terminal emulators silently drop the - # scheduled coroutine. Fall back to a direct input() so the - # user's keystrokes don't leak into the agent buffer. - try: - _ask() - except Exception: - pass - finally: - self._status_bar_visible = was_visible - self._app.invalidate() - else: - _ask() - return result[0] - - def _prompt_text_input_modal( - self, - *, - title: str, - detail: str, - choices: list[tuple[str, str, str]], - timeout: float = 120, - ) -> str | None: - """Prompt through the prompt_toolkit composer instead of raw input(). - - This is for CLI slash-command confirmations. The old raw input() path - fought prompt_toolkit's active stdin ownership: in some terminals the - prompt appeared above the TUI, choices were redrawn later, and Enter - could be interpreted as EOF/exit. A first-class modal state keeps the - choices visible and lets the normal Enter key binding submit the typed - or highlighted choice. - - **Platform note (Windows — issue #33961):** - Earlier code bypassed the modal on ``sys.platform == "win32"`` and fell - back to a raw ``input()`` prompt. When the confirm was triggered from the - ``process_loop`` daemon thread (the normal case) that ``input()`` ran off - the main thread and deadlocked against prompt_toolkit's stdin ownership — - the user saw a frozen cursor and Ctrl-C was swallowed (bare ``/reset`` - froze; ``/reset now`` worked only because it skips the prompt entirely). - - Native Windows now uses the same path as Linux/macOS: the modal is set up - on ``self._app.loop`` via ``call_soon_threadsafe`` and answered by the - normal prompt_toolkit key bindings (the same input channel that already - handles ordinary typing on Windows). The raw ``input()`` fallback is kept - only for the genuinely safe cases: no running app (unit tests / - non-interactive), no resolvable event loop, or a scheduling failure. - """ - import threading - import time as _time - - if not choices: - return None - - # If prompt_toolkit is not running (unit tests / non-interactive calls), - # keep the simple stdin fallback. - if not getattr(self, "_app", None): - return self._prompt_text_input("Choice [1/2/3]: ") - - try: - app_loop = self._app.loop - except Exception: - app_loop = None - - in_main_thread = threading.current_thread() is threading.main_thread() - - def _stdin_fallback() -> str | None: - # On native Windows a raw input() from a non-main thread deadlocks - # against prompt_toolkit's stdin ownership (#33961). With an app - # running we cannot safely prompt off the main thread, so cancel - # cleanly (None) rather than hang the terminal. - if sys.platform == "win32" and not in_main_thread: - self._invalidate() - return None - return self._prompt_text_input("Choice [1/2/3]: ") - - if not in_main_thread and app_loop is None: - return _stdin_fallback() - - response_queue = queue.Queue() - - def _setup_modal() -> None: - self._capture_modal_input_snapshot() - self._slash_confirm_state = { - "title": title, - "detail": detail, - "choices": choices, - "selected": 0, - "response_queue": response_queue, - } - self._slash_confirm_deadline = _time.monotonic() + timeout - self._invalidate() - - def _teardown_modal() -> None: - self._slash_confirm_state = None - self._slash_confirm_deadline = 0 - self._restore_modal_input_snapshot() - self._invalidate() - - def _run_on_app_loop(fn) -> bool: - if in_main_thread or app_loop is None: - fn() - return True - ready = threading.Event() - - def _wrapped() -> None: - try: - fn() - finally: - ready.set() - - try: - app_loop.call_soon_threadsafe(_wrapped) - except Exception: - return False - return ready.wait(timeout=5) - - if not _run_on_app_loop(_setup_modal): - return _stdin_fallback() - - _last_countdown_refresh = _time.monotonic() - try: - while True: - try: - result = response_queue.get(timeout=1) - _run_on_app_loop(_teardown_modal) - return result - except queue.Empty: - remaining = self._slash_confirm_deadline - _time.monotonic() - if remaining <= 0: - break - now = _time.monotonic() - if now - _last_countdown_refresh >= 5.0: - _last_countdown_refresh = now - self._invalidate() - finally: - if self._slash_confirm_state is not None: - _run_on_app_loop(_teardown_modal) - return None - - def _submit_slash_confirm_response(self, value: str | None) -> None: - state = self._slash_confirm_state - if not state: - return - state["response_queue"].put(value) - self._slash_confirm_state = None - self._slash_confirm_deadline = 0 - self._invalidate() - - def _normalize_slash_confirm_choice( - self, - raw: str | None, - choices: list[tuple[str, str, str]], - ) -> str | None: - if raw is None: - return None - choice_raw = raw.strip().lower() - if not choice_raw: - return None - aliases = { - "1": "once", - "once": "once", - "approve": "once", - "yes": "once", - "y": "once", - "ok": "once", - "2": "always", - "always": "always", - "remember": "always", - "3": "cancel", - "cancel": "cancel", - "nevermind": "cancel", - "no": "cancel", - "n": "cancel", - } - allowed = {choice[0] for choice in choices} - normalized = aliases.get(choice_raw) - if normalized in allowed: - return normalized - if choice_raw in allowed: - return choice_raw - return None - - def _get_slash_confirm_display_fragments(self): - """Render the /new-/clear-style confirmation panel.""" - state = self._slash_confirm_state - if not state: - return [] - - title = state.get("title") or "Confirm action" - detail = state.get("detail") or "" - choices = state.get("choices") or [] - selected = state.get("selected", 0) - - _wrap_panel_text = _wrap_panel_text_keep_ws - - preview_lines = [] - for line in detail.splitlines(): - preview_lines.extend(_wrap_panel_text(line, 72)) - for idx, (_value, label, desc) in enumerate(choices): - marker = "❯" if idx == selected else " " - preview_lines.extend(_wrap_panel_text(f"{marker} [{idx + 1}] {label} — {desc}", 72, subsequent_indent=" ")) - preview_lines.append("Type 1/2/3 or use ↑/↓ then Enter. ESC/Ctrl+C cancels.") - - box_width = _panel_box_width(title, preview_lines, min_width=56, max_width=86) - inner_text_width = max(8, box_width - 2) - detail_wrapped = [] - for line in detail.splitlines(): - detail_wrapped.extend(_wrap_panel_text(line, inner_text_width)) - choice_wrapped: list[tuple[int, str]] = [] - for idx, (_value, label, desc) in enumerate(choices): - marker = "❯" if idx == selected else " " - for wrapped in _wrap_panel_text(f"{marker} [{idx + 1}] {label} — {desc}", inner_text_width, subsequent_indent=" "): - choice_wrapped.append((idx, wrapped)) - - term_rows = shutil.get_terminal_size((100, 24)).lines - reserved_below = 6 - chrome_full = 6 - available = max(0, term_rows - reserved_below) - max_detail_rows = max(1, available - chrome_full - len(choice_wrapped)) - max_detail_rows = min(max_detail_rows, 8) - if len(detail_wrapped) > max_detail_rows: - keep = max(1, max_detail_rows - 1) - detail_wrapped = detail_wrapped[:keep] + ["… (detail truncated)"] - - lines = [] - lines.append(('class:approval-border', '╭' + ('─' * box_width) + '╮\n')) - _append_panel_line(lines, 'class:approval-border', 'class:approval-title', title, box_width) - _append_blank_panel_line(lines, 'class:approval-border', box_width) - for wrapped in detail_wrapped: - _append_panel_line(lines, 'class:approval-border', 'class:approval-desc', wrapped, box_width) - _append_blank_panel_line(lines, 'class:approval-border', box_width) - for idx, wrapped in choice_wrapped: - style = 'class:approval-selected' if idx == selected else 'class:approval-choice' - _append_panel_line(lines, 'class:approval-border', style, wrapped, box_width) - _append_blank_panel_line(lines, 'class:approval-border', box_width) - _append_panel_line(lines, 'class:approval-border', 'class:approval-cmd', 'Type 1/2/3 or use ↑/↓ then Enter. ESC/Ctrl+C cancels.', box_width) - lines.append(('class:approval-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _build_command_palette_entries(self) -> list: - """Flat list of (command, description) for the Ctrl+P palette. - - Sourced from the same COMMAND_REGISTRY that backs /help, filtered to - commands available on this surface, plus installed skill commands. - Selecting an entry inserts the exact command string — never a fuzzy - resolution. - """ - from hermes_cli.commands import COMMANDS_BY_CATEGORY - - entries: list[tuple[str, str, str]] = [] # (command, category, desc) - for category, commands in COMMANDS_BY_CATEGORY.items(): - for cmd, desc in commands.items(): - if not self._command_available(cmd): - continue - entries.append((cmd, category, desc)) - try: - for cmd, info in sorted(_ensure_skill_commands().items()): - entries.append((cmd, "Skill", info.get("description", ""))) - except Exception: - pass - return entries - - def _open_command_palette(self) -> None: - """Open the Ctrl+P fuzzy command palette modal.""" - if getattr(self, "_command_palette_state", None): - return - # Don't stack over other modals. - if (self._model_picker_state or self._clarify_state or self._approval_state - or self._slash_confirm_state or self._sudo_state or self._secret_state): - return - self._capture_modal_input_snapshot() - self._command_palette_state = { - "entries": self._build_command_palette_entries(), - "filter": "", - "selected": 0, - "_scroll_offset": 0, - } - self._invalidate(min_interval=0.0) - - def _close_command_palette(self) -> None: - self._command_palette_state = None - self._restore_modal_input_snapshot() - self._invalidate(min_interval=0.0) - - def _command_palette_visible_entries(self) -> list: - """Return (command, category, desc) rows matching the active filter. - - Ranked, command-name-focused matching (a bare subsequence over the - whole "cmd category desc" string is uselessly permissive — "steer" - would match 130+ rows via description text). Priority: - 0 exact command match - 1 command startswith query - 2 query substring in command - 3 query subsequence in command - 4 query substring in description - Rows that match nowhere are dropped. Ties keep registry order. - """ - state = self._command_palette_state or {} - entries = state.get("entries") or [] - q = (state.get("filter", "") or "").strip().lower() - if not q: - return list(entries) - - def _subseq(needle: str, hay: str) -> bool: - it = iter(hay) - return all(ch in it for ch in needle) - - ranked = [] - for order, row in enumerate(entries): - cmd, _cat, desc = row - name = cmd.lower().lstrip("/") - qn = q.lstrip("/") - desc_l = (desc or "").lower() - if name == qn: - rank = 0 - elif name.startswith(qn): - rank = 1 - elif qn in name: - rank = 2 - elif _subseq(qn, name): - rank = 3 - elif q in desc_l: - rank = 4 - else: - continue - ranked.append((rank, order, row)) - ranked.sort(key=lambda t: (t[0], t[1])) - return [row for (_r, _o, row) in ranked] - - def _handle_command_palette_selection(self) -> None: - """Insert the selected command into the composer (does not auto-run).""" - state = self._command_palette_state - if not state: - return - rows = self._command_palette_visible_entries() - selected = state.get("selected", 0) - if not (0 <= selected < len(rows)): - self._close_command_palette() - return - cmd = rows[selected][0] # exact command string, e.g. "/model" - self._close_command_palette() - # Prefill the composer so the user can add args / confirm — never - # auto-execute (a palette pick should be explicit, and many commands - # take arguments). - try: - app = getattr(self, "_app", None) - if app is not None: - buf = app.current_buffer - buf.text = cmd + " " - buf.cursor_position = len(buf.text) - self._invalidate(min_interval=0.0) - except Exception: - logger.debug("command palette prefill failed", exc_info=True) - - def _open_model_picker(self, providers: list, current_model: str, current_provider: str, user_provs=None, custom_provs=None) -> None: - """Open prompt_toolkit-native /model picker modal.""" - self._capture_modal_input_snapshot() - default_idx = next((i for i, p in enumerate(providers) if p.get("is_current")), 0) - self._model_picker_state = { - "stage": "provider", - "providers": providers, - "selected": default_idx, - "current_model": current_model, - "current_provider": current_provider, - "user_provs": user_provs, - "custom_provs": custom_provs, - "filter": "", - } - self._invalidate(min_interval=0.0) - - def _confirm_expensive_model_switch(self, result) -> bool: - """Ask for explicit confirmation before applying costly model switches.""" - if not getattr(result, "success", False): - return True - try: - from hermes_cli.model_selection_guards import combined_selection_warning - - warning = combined_selection_warning( - result.new_model, - provider=result.target_provider, - base_url=result.base_url or self.base_url or "", - api_key=result.api_key or self.api_key or "", - model_info=result.model_info, - ) - except Exception: - warning = None - if warning is None: - return True - - choices = [ - ("once", "Switch anyway", "Use this model for the current Hermes session."), - ("cancel", "Cancel", "Keep the current model."), - ] - raw = self._prompt_text_input_modal( - title=f"!!! {warning.title} !!!", - detail=warning.message, - choices=choices, - timeout=120, - ) - choice = self._normalize_slash_confirm_choice(raw, choices) - return choice == "once" - - def _confirm_and_apply_model_switch_result( - self, result, persist_global: bool, custom_providers=None - ) -> None: - try: - if result.success and not self._confirm_expensive_model_switch(result): - _cprint(" Model switch cancelled.") - return - self._apply_model_switch_result( - result, persist_global, custom_providers=custom_providers - ) - except Exception as exc: - _cprint(f" ✗ Model selection failed: {exc}") - - def _close_model_picker(self) -> None: - self._model_picker_state = None - self._restore_modal_input_snapshot() - self._invalidate(min_interval=0.0) - - def _snapshot_model_runtime(self) -> dict: - """Capture current CLI and agent model runtime for one-turn restore.""" - agent = getattr(self, "agent", None) - return { - "model": self.model, - "provider": self.provider, - "requested_provider": self.requested_provider, - "_explicit_api_key": getattr(self, "_explicit_api_key", None), - "_explicit_base_url": getattr(self, "_explicit_base_url", None), - "api_key": self.api_key, - "base_url": self.base_url, - "api_mode": self.api_mode, - "agent_primary_runtime": copy.deepcopy( - getattr(agent, "_primary_runtime", None) - ) if agent is not None else None, - } - - def _restore_model_runtime_snapshot(self, snapshot: dict | None) -> None: - """Restore a model runtime captured before a one-turn override.""" - if not snapshot: - return - for key in ( - "model", - "provider", - "requested_provider", - "_explicit_api_key", - "_explicit_base_url", - "api_key", - "base_url", - "api_mode", - ): - if key in snapshot: - setattr(self, key, snapshot.get(key)) - - agent = getattr(self, "agent", None) - if agent is None: - return - - primary = snapshot.get("agent_primary_runtime") - if primary and hasattr(agent, "_restore_primary_runtime"): - try: - agent._primary_runtime = copy.deepcopy(primary) - agent._fallback_activated = True - agent._rate_limited_until = 0 - if agent._restore_primary_runtime(): - return - except Exception: - logger.debug("CLI one-turn model restore via primary runtime failed", exc_info=True) - - if hasattr(agent, "switch_model"): - try: - agent.switch_model( - new_model=snapshot.get("model", ""), - new_provider=snapshot.get("provider", ""), - api_key=snapshot.get("api_key", ""), - base_url=snapshot.get("base_url", ""), - api_mode=snapshot.get("api_mode", ""), - capabilities=snapshot.get("capabilities"), - ) - except Exception as exc: - logger.warning("CLI one-turn model restore failed: %s", exc) - - @staticmethod - def _filter_model_picker_entries(entries: list, query: str) -> list: - """Return (original_index, label) pairs for entries matching ``query``. - - Subsequence ("fuzzy") match, case-insensitive: the query characters - must appear in order in the label. An empty query matches everything. - Crucially the returned pairs carry the ORIGINAL index into ``entries``, - so a selection in the filtered view still resolves to exactly one - concrete model — filtering only narrows the list, it never introduces - an ambiguous or fuzzy *resolution* (the anti-"claude→old-model" rule). - """ - pairs = list(enumerate(entries)) - q = (query or "").strip().lower() - if not q: - return pairs - - def _subseq(needle: str, hay: str) -> bool: - it = iter(hay) - return all(ch in it for ch in needle) - - out = [(i, e) for (i, e) in pairs if _subseq(q, str(e).lower())] - return out - - @staticmethod - def _compute_model_picker_viewport( - selected: int, - scroll_offset: int, - n: int, - term_rows: int, - reserved_below: int = 6, - panel_chrome: int = 6, - min_visible: int = 3, - ) -> tuple[int, int]: - """Resolve (scroll_offset, visible) for the /model picker viewport. - - ``reserved_below`` matches the approval / clarify panels — input area, - status bar, and separators below the panel. ``panel_chrome`` covers - this panel's own borders + blanks + hint row. The remaining rows hold - the scrollable list, with the offset slid to keep ``selected`` on screen. - """ - max_visible = max(min_visible, term_rows - reserved_below - panel_chrome) - if n <= max_visible: - return 0, n - visible = max_visible - if selected < scroll_offset: - scroll_offset = selected - elif selected >= scroll_offset + visible: - scroll_offset = selected - visible + 1 - scroll_offset = max(0, min(scroll_offset, n - visible)) - return scroll_offset, visible - - def _clear_persisted_context_for_model_switch(self, result) -> None: - """Drop a global context pin when its configured owner changes.""" - try: - from hermes_cli.config import load_config_readonly - from hermes_cli.route_identity import should_clear_context_pin - - config = load_config_readonly() - model_cfg = config.get("model", {}) if isinstance(config, dict) else {} - if not isinstance(model_cfg, dict) or "context_length" not in model_cfg: - return - if should_clear_context_pin( - model_cfg.get("default") or model_cfg.get("model"), - result.new_model, - model_cfg.get("base_url"), - result.base_url, - model_cfg.get("provider"), - result.target_provider, - ): - save_config_value("model.context_length", None) - except Exception: - save_config_value("model.context_length", None) - - def _apply_model_switch_result( - self, result, persist_global: bool, custom_providers=None - ) -> None: - if not result.success: - _cprint(f" ✗ {result.error_message}") - return - - if self.agent is not None: - try: - from hermes_cli.context_switch_guard import merge_preflight_compression_warning - - # Prefer the fresh inventory list (same source as switch_model / - # TUI); fall back to the agent-init snapshot. - _cp = ( - custom_providers - if custom_providers is not None - else getattr(self.agent, "_custom_providers", None) - ) - merge_preflight_compression_warning( - result, - agent=self.agent, - messages=list(self.conversation_history or []), - custom_providers=_cp, - config_context_length=getattr(self.agent, "_config_context_length", None), - ) - except Exception as exc: - logger.debug("preflight-compression switch warning failed: %s", exc) - - old_model = self.model - # Snapshot the CLI-level credential/runtime fields BEFORE mutating them - # so a failed in-place agent swap can roll the whole CLI back to the old - # working model. Otherwise the broken credentials staged below leak into - # the next turn's resolution even though the agent itself rolled back - # (#50163). - _cli_snapshot = { - "model": self.model, - "provider": self.provider, - "requested_provider": self.requested_provider, - "_explicit_api_key": getattr(self, "_explicit_api_key", None), - "_explicit_base_url": getattr(self, "_explicit_base_url", None), - "api_key": self.api_key, - "base_url": self.base_url, - "api_mode": self.api_mode, - } - self.model = result.new_model - self.provider = result.target_provider - self.requested_provider = result.target_provider - # Always overwrite explicit overrides so stale credentials from the - # previous provider (e.g. Ollama api_key/base_url) don't leak into - # the new provider's credential resolution on the next turn. - self._explicit_api_key = result.api_key - self._explicit_base_url = result.base_url - if result.api_key: - self.api_key = result.api_key - if result.base_url: - self.base_url = result.base_url - if result.api_mode: - self.api_mode = result.api_mode - - if self.agent is not None: - try: - self.agent.switch_model( - new_model=result.new_model, - new_provider=result.target_provider, - api_key=result.api_key, - base_url=result.base_url, - api_mode=result.api_mode, - capabilities=getattr(result, "runtime_capabilities", None), - ) - except Exception as exc: - # The agent rolled itself back to the old working model/client. - # Roll the CLI's own staged fields back too and abort the rest - # of the commit (note + success print) so a failed switch is a - # no-op rather than a dead session (#50163). - for _k, _v in _cli_snapshot.items(): - setattr(self, _k, _v) - _cprint( - f" ⚠ Model switch to {result.new_model} failed ({exc}); " - f"staying on {old_model}." - ) - return - - from hermes_cli.model_switch import format_model_for_display - _display_old = format_model_for_display(old_model) - _display_new = format_model_for_display(result.new_model) - - self._pending_model_switch_note = ( - f"[Note: model was just switched from {_display_old} to {_display_new} " - f"via {result.provider_label or result.target_provider}. " - f"Adjust your self-identification accordingly.]" - ) - - provider_label = result.provider_label or result.target_provider - _cprint(f" ✓ Model switched: {_display_new}") - _cprint(f" Provider: {provider_label}") - - # Context: always resolve via the provider-aware chain so Codex OAuth, - # Copilot, and Nous-enforced caps win over the raw models.dev entry - # (e.g. gpt-5.5 is 1.05M on openai but 272K on Codex OAuth). - mi = result.model_info - try: - from hermes_cli.model_switch import resolve_display_context_length - ctx = resolve_display_context_length( - result.new_model, - result.target_provider, - base_url=result.base_url or self.base_url or "", - api_key=result.api_key or self.api_key or "", - model_info=mi, - config_context_length=getattr(self.agent, "_config_context_length", None) if self.agent else None, - custom_providers=getattr(self.agent, "_custom_providers", None) if self.agent else None, - ) - if ctx: - _cprint(f" Context: {ctx:,} tokens") - except Exception: - pass - if mi: - if mi.max_output: - _cprint(f" Max output: {mi.max_output:,} tokens") - _cprint(f" Capabilities: {mi.format_capabilities()}") - - cache_enabled = ( - (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) - or result.api_mode == "anthropic_messages" - ) - if cache_enabled: - _cprint(" Prompt caching: enabled") - if result.warning_message: - _cprint(f" ⚠ {result.warning_message}") - if persist_global: - HermesCLI._clear_persisted_context_for_model_switch(self, result) - save_config_value("model.default", result.new_model) - save_config_value("model.provider", result.target_provider) - # base_url/api_mode were previously never persisted here, so a - # global switch left the OLD provider's endpoint/wire-protocol in - # config.yaml. result.base_url/api_mode are always freshly - # resolved for the target provider (see model_switch.py), so sync - # them every time; None clears a value the new provider doesn't - # need (#25106). - save_config_value("model.base_url", result.base_url or None) - save_config_value("model.api_mode", result.api_mode or None) - _cprint(" Saved to config.yaml (--global)") - else: - _cprint(" (session only — add --global to persist)") - - # Persist the switch to this session's row so --resume / - # session.resume restore it. --global also updates config.yaml - # (future sessions), but the row still records what THIS session - # actually runs — otherwise a later resume would restore the stale - # creation-time model over the user's new global choice. - HermesCLI._persist_model_switch_to_session(self, result) - - def _handle_model_picker_selection(self, persist_global: bool = False) -> None: - state = self._model_picker_state - if not state: - return - selected = state.get("selected", 0) - stage = state.get("stage") - if stage == "provider": - providers = state.get("providers") or [] - if selected >= len(providers): - self._close_model_picker() - return - provider_data = providers[selected] - # Use the curated model list from list_authenticated_providers() - # (same lists as `hermes model` and gateway pickers). - # Only fall back to the live provider catalog when the curated - # list is empty (e.g. user-defined endpoints with no curated list). - model_list = provider_data.get("models", []) - if not model_list: - try: - from hermes_cli.models import provider_model_ids - live = provider_model_ids(provider_data["slug"]) - if live: - model_list = live - except Exception: - pass - state["stage"] = "model" - state["provider_data"] = provider_data - state["model_list"] = model_list - state["selected"] = 0 - state["filter"] = "" - state["_filtered_pairs"] = None - self._invalidate(min_interval=0.0) - return - if stage == "model": - provider_data = state.get("provider_data") or {} - model_list = state.get("model_list") or [] - # Map the selected row through the active fuzzy filter so the - # index lines up with what the picker is currently showing. The - # filtered pair carries the ORIGINAL index into model_list, so the - # resolved model is always one concrete, unambiguous entry. - filtered_pairs = state.get("_filtered_pairs") - if filtered_pairs is None: - filtered_pairs = list(enumerate(model_list)) - visible_labels = [e for (_i, e) in filtered_pairs] - back_idx = len(visible_labels) - cancel_idx = len(visible_labels) + 1 - if selected == back_idx: - state["stage"] = "provider" - state["filter"] = "" - state["_filtered_pairs"] = None - state["selected"] = next((i for i, p in enumerate(state.get("providers") or []) if p.get("slug") == provider_data.get("slug")), 0) - self._invalidate(min_interval=0.0) - return - if selected >= cancel_idx: - self._close_model_picker() - return - if 0 <= selected < len(visible_labels): - from hermes_cli.model_switch import switch_model - chosen_model = visible_labels[selected] - result = switch_model( - raw_input=chosen_model, - current_provider=self.provider or "", - current_model=self.model or "", - current_base_url=self.base_url or "", - current_api_key=self.api_key or "", - is_global=persist_global, - explicit_provider=provider_data.get("slug"), - user_providers=state.get("user_provs"), - custom_providers=state.get("custom_provs"), - ) - # Capture before close — picker state is cleared on close. - _picker_custom_provs = state.get("custom_provs") - self._close_model_picker() - if getattr(self, "_app", None): - threading.Thread( - target=self._confirm_and_apply_model_switch_result, - args=(result, persist_global, _picker_custom_provs), - daemon=True, - ).start() - else: - self._confirm_and_apply_model_switch_result( - result, persist_global, custom_providers=_picker_custom_provs - ) - return - self._close_model_picker() - - def _handle_model_switch(self, cmd_original: str): - """Handle /model command — switch model. - - Supports: - /model — show current model + usage hints - /model — switch model (this session only) - /model --once — switch for the next turn only - /model --session — switch for this session only (explicit) - /model --global — switch and persist to config.yaml - /model --provider — switch provider + model - /model --provider — switch to provider, auto-detect model - - Persistence defaults to off (``model.persist_switch_by_default`` in - config.yaml, default False — switches are session-scoped). Use - ``--global`` to persist, or ``--once`` for the next turn only. - """ - from hermes_cli.model_switch import ( - switch_model, - parse_model_switch_args, - resolve_persist_behavior, - ) - from hermes_cli.providers import get_label - - # Parse args from the original command - parts = cmd_original.split(None, 1) # split off '/model' - raw_args = parts[1].strip() if len(parts) > 1 else "" - - # Parse --provider, --global, --session, --once, and --refresh flags - # via the shared single-owner parser (hermes_cli.model_switch). - request = parse_model_switch_args(raw_args) - model_input = request.target - explicit_provider = request.explicit_provider - is_global_flag = request.is_global - force_refresh = request.force_refresh - is_session = request.is_session - one_turn = request.is_once - if request.errors: - # CLI decoration: " ✗ " prefix over the canonical error copy. - _cprint(f" ✗ {request.error_messages()[0]}") - return - # Resolve the effective persistence once: --global forces persist, - # --session/--once force session-scope, otherwise defer to - # model.persist_switch_by_default (defaults to False so /model is - # session-scoped unless the user opts in). - persist_global = resolve_persist_behavior( - is_global_flag, is_session, is_once=one_turn, - explicit_provider=explicit_provider, - ) - - # --refresh: wipe the on-disk picker cache before building the - # provider list. Forces a live re-fetch of every authed provider's - # /v1/models endpoint on this open. - if force_refresh: - try: - from hermes_cli.models import clear_provider_models_cache - clear_provider_models_cache() - _cprint(" Cleared model picker cache. Refreshing...") - except Exception: - pass - - # Single inventory context — replaces the inline config-slice the - # dashboard / TUI used to duplicate. Overlay live session state - # via with_overrides (truthy-only) so empty self.* attrs don't - # clobber disk config. - from hermes_cli.inventory import build_models_payload, load_picker_context - - try: - ctx = load_picker_context().with_overrides( - current_provider=self.provider or "", - current_model=self.model or "", - current_base_url=self.base_url or "", - ) - except Exception: - ctx = None - - # switch_model() + _open_model_picker still need the raw provider - # dicts; ConfigContext is the canonical source for both. - user_provs = ctx.user_providers if ctx is not None else None - custom_provs = ctx.custom_providers if ctx is not None else None - - # No args at all: open prompt_toolkit-native picker modal - if not model_input and not explicit_provider: - model_display = self.model or "unknown" - provider_display = get_label(self.provider) if self.provider else "unknown" - - try: - if ctx is None: - raise RuntimeError("inventory context unavailable") - providers = build_models_payload( - ctx, - probe_custom_providers=force_refresh, - probe_current_custom_provider=not force_refresh, - )["providers"] - except Exception: - providers = [] - - if not providers: - _cprint(" No authenticated providers found.") - _cprint("") - _cprint(" /model switch model (this session)") - _cprint(" /model --global switch model and persist as default") - _cprint(" /model --once switch for the next turn only") - _cprint(" /model --session switch for this session only") - _cprint(" /model --provider switch provider") - _cprint(" /model --refresh re-fetch live model lists") - return - - self._open_model_picker( - providers, - model_display, - provider_display, - user_provs=user_provs, - custom_provs=custom_provs, - ) - return - - # Perform the switch - result = switch_model( - raw_input=model_input, - current_provider=self.provider or "", - current_model=self.model or "", - current_base_url=self.base_url or "", - current_api_key=self.api_key or "", - is_global=persist_global, - explicit_provider=explicit_provider, - user_providers=user_provs, - custom_providers=custom_provs, - ) - - if not result.success: - _cprint(f" ✗ {result.error_message}") - return - - if self.agent is not None: - try: - from hermes_cli.context_switch_guard import merge_preflight_compression_warning - - merge_preflight_compression_warning( - result, - agent=self.agent, - messages=list(self.conversation_history or []), - # Same fresh inventory list passed to switch_model above. - custom_providers=custom_provs - if custom_provs is not None - else getattr(self.agent, "_custom_providers", None), - config_context_length=getattr(self.agent, "_config_context_length", None), - ) - except Exception as exc: - logger.debug("preflight-compression switch warning failed: %s", exc) - - # Run the confirm + apply sequence off the main thread. The - # expensive-model confirmation modal blocks the calling thread on a - # response queue (see _prompt_text_input_modal); running it on the - # prompt_toolkit main thread freezes TUI rendering, so the modal never - # appears and the switch silently cancels after the 120s timeout. - # Mirror the picker path (_handle_model_picker_selection), which - # already dispatches confirm+apply on a worker thread. - if getattr(self, "_app", None): - threading.Thread( - target=self._confirm_and_apply_cli_model_switch, - args=(result, persist_global, one_turn, custom_provs), - daemon=True, - ).start() - return - self._confirm_and_apply_cli_model_switch( - result, persist_global, one_turn, custom_provs - ) - return - - def _confirm_and_apply_cli_model_switch( - self, result, persist_global: bool, one_turn: bool, custom_provs=None - ) -> None: - """Confirm an expensive model switch and apply it to CLI state. - - Runs on a worker thread when the TUI is active (see - _handle_model_switch) so the confirmation modal can render. - """ - if not self._confirm_expensive_model_switch(result): - _cprint(" Model switch cancelled.") - return - - # Apply to CLI state. - # Update requested_provider so _ensure_runtime_credentials() doesn't - # overwrite the switch on the next turn (it re-resolves from this). - old_model = self.model - _one_turn_restore_snapshot = self._snapshot_model_runtime() if one_turn else None - # Snapshot CLI-level fields before mutation so a failed in-place swap - # rolls the whole CLI back to the old working model (#50163). - _cli_snapshot = { - "model": self.model, - "provider": self.provider, - "requested_provider": self.requested_provider, - "_explicit_api_key": getattr(self, "_explicit_api_key", None), - "_explicit_base_url": getattr(self, "_explicit_base_url", None), - "api_key": self.api_key, - "base_url": self.base_url, - "api_mode": self.api_mode, - } - self.model = result.new_model - self.provider = result.target_provider - self.requested_provider = result.target_provider - # Always overwrite explicit overrides so stale credentials from the - # previous provider (e.g. Ollama api_key/base_url) don't leak into - # the new provider's credential resolution on the next turn. - self._explicit_api_key = result.api_key - self._explicit_base_url = result.base_url - if result.api_key: - self.api_key = result.api_key - if result.base_url: - self.base_url = result.base_url - if result.api_mode: - self.api_mode = result.api_mode - - # Apply to running agent (in-place swap) - if self.agent is not None: - try: - self.agent.switch_model( - new_model=result.new_model, - new_provider=result.target_provider, - api_key=result.api_key, - base_url=result.base_url, - api_mode=result.api_mode, - capabilities=getattr(result, "runtime_capabilities", None), - ) - except Exception as exc: - # Agent rolled itself back; roll the CLI back too and abort so a - # failed switch is a no-op rather than a dead session (#50163). - for _k, _v in _cli_snapshot.items(): - setattr(self, _k, _v) - _cprint( - f" ⚠ Model switch to {result.new_model} failed ({exc}); " - f"staying on {old_model}." - ) - return - - # Store a note to prepend to the next user message so the model - # knows a switch occurred (avoids injecting system messages mid-history - # which breaks providers and prompt caching). - from hermes_cli.model_switch import format_model_for_display - _display_old = format_model_for_display(old_model) - _display_new = format_model_for_display(result.new_model) - - self._pending_model_switch_note = ( - f"[Note: model was just switched from {_display_old} to {_display_new} " - f"via {result.provider_label or result.target_provider}. " - f"{'This override applies to the next turn only. ' if one_turn else ''}" - f"Adjust your self-identification accordingly.]" - ) - if one_turn: - self._pending_one_turn_model_restore = _one_turn_restore_snapshot - else: - self._pending_one_turn_model_restore = None - - # Display confirmation with full metadata - provider_label = result.provider_label or result.target_provider - _cprint(f" ✓ Model switched: {_display_new}") - _cprint(f" Provider: {provider_label}") - - # Context: always resolve via the provider-aware chain so Codex OAuth, - # Copilot, and Nous-enforced caps win over the raw models.dev entry - # (e.g. gpt-5.5 is 1.05M on openai but 272K on Codex OAuth). - mi = result.model_info - from hermes_cli.model_switch import resolve_display_context_length - ctx = resolve_display_context_length( - result.new_model, - result.target_provider, - base_url=result.base_url or self.base_url or "", - api_key=result.api_key or self.api_key or "", - model_info=mi, - config_context_length=getattr(self.agent, "_config_context_length", None) if self.agent else None, - custom_providers=getattr(self.agent, "_custom_providers", None) if self.agent else None, - ) - if ctx: - _cprint(f" Context: {ctx:,} tokens") - if mi: - if mi.max_output: - _cprint(f" Max output: {mi.max_output:,} tokens") - _cprint(f" Capabilities: {mi.format_capabilities()}") - - # Cache notice - cache_enabled = ( - (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) - or result.api_mode == "anthropic_messages" - ) - if cache_enabled: - _cprint(" Prompt caching: enabled") - - # Warning from validation - if result.warning_message: - _cprint(f" ⚠ {result.warning_message}") - - # Persistence - if persist_global: - HermesCLI._clear_persisted_context_for_model_switch(self, result) - save_config_value("model.default", result.new_model) - save_config_value("model.provider", result.target_provider) - # See _apply_model_switch_result above for why base_url/api_mode - # must be synced on every global switch (#25106). - save_config_value("model.base_url", result.base_url or None) - save_config_value("model.api_mode", result.api_mode or None) - _cprint(" Saved to config.yaml") - elif one_turn: - _cprint(" (next turn only — restores after one response)") - else: - _cprint(" (session only — add --global to persist)") - - # Persist the switch to this session's row so --resume / - # session.resume restore it (--global also updates config.yaml but - # the row still records what THIS session runs; --once is ephemeral - # and restored after one turn, so it must not touch the row). - if not one_turn: - HermesCLI._persist_model_switch_to_session(self, result) - - def _handle_codex_runtime(self, cmd_original: str) -> None: - """Handle /codex-runtime — toggle the codex app-server runtime opt-in. - - Usage: - /codex-runtime — show current state - /codex-runtime auto — Hermes default (chat_completions) - /codex-runtime codex_app_server — hand turns to codex subprocess - /codex-runtime on / off — synonyms for the above - """ - from hermes_cli import codex_runtime_switch as crs - - parts = cmd_original.split(None, 1) - raw_args = parts[1].strip() if len(parts) > 1 else "" - new_value, errors = crs.parse_args(raw_args) - if errors: - for err in errors: - _cprint(f"❌ {err}") - return - - # Load + persist via the existing config helpers - try: - from hermes_cli.config import load_config, save_config - except Exception as exc: - _cprint(f"❌ could not load config: {exc}") - return - cfg = load_config() - - result = crs.apply( - cfg, - new_value, - persist_callback=(save_config if new_value is not None else None), - ) - - prefix = "✓" if result.success else "✗" - for line in result.message.splitlines(): - _cprint(f" {prefix} {line}" if line.startswith("openai_runtime") - else f" {line}") - if result.success and result.requires_new_session: - _cprint(" Tip: `/reset` starts a new session immediately.") - - def _should_handle_model_command_inline(self, text: str, has_images: bool = False) -> bool: - """Return True when /model should be handled immediately on the UI thread.""" - if not text or has_images or not _looks_like_slash_command(text): - return False - try: - from hermes_cli.commands import resolve_command - base = text.split(None, 1)[0].lower().lstrip('/') - cmd = resolve_command(base) - return bool(cmd and cmd.name == "model") - except Exception: - return False - - def _should_handle_steer_command_inline(self, text: str, has_images: bool = False) -> bool: - """Return True when /steer should be dispatched immediately while the agent is running. - - /steer MUST bypass the normal _pending_input → process_loop path when - the agent is active, because process_loop is blocked inside - self.chat() for the duration of the run. By the time the queued - command is pulled from _pending_input, _agent_running has already - flipped back to False, and process_command() takes the idle - fallback — delivering the steer as a next-turn message instead of - injecting it mid-run. Dispatching inline on the UI thread calls - agent.steer() directly, which is thread-safe (uses _pending_steer_lock). - """ - if not text or has_images or not _looks_like_slash_command(text): - return False - if not getattr(self, "_agent_running", False): - return False - try: - from hermes_cli.commands import resolve_command - base = text.split(None, 1)[0].lower().lstrip('/') - cmd = resolve_command(base) - return bool(cmd and cmd.name == "steer") - except Exception: - return False - - def _should_handle_background_command_inline( - self, text: str, has_images: bool = False - ) -> bool: - """Return True when /bg or /btw should be dispatched while the agent runs. - - Same queue problem /steer had. ``/bg`` exists to start independent - work *without* waiting for the current turn, and ``/btw`` exists to - answer a side question about the in-flight conversation, but a slash - command typed while the agent is busy goes into ``_pending_input``, - and ``process_loop`` is blocked inside ``self.chat()`` for the whole - run. The side task would therefore only start once the foreground - turn has finished, which is the one moment it was not needed. - - Both commands' ``CommandDef`` entries already declare - ``busy_policy="dispatch"``; the gateway honours that, the classic CLI - never consulted it. Dispatching inline on the UI thread starts the - side session immediately and leaves the foreground turn running - untouched: no interrupt, no steer. - """ - if not text or has_images or not _looks_like_slash_command(text): - return False - if not getattr(self, "_agent_running", False): - return False - try: - from hermes_cli.commands import resolve_command - base = text.split(None, 1)[0].lower().lstrip('/') - cmd = resolve_command(base) - return bool(cmd and cmd.name in ("bg", "btw")) - except Exception: - return False - - def _output_console(self): - """Use prompt_toolkit-safe Rich rendering once the TUI is live.""" - if getattr(self, "_app", None): - return ChatConsole() - return self.console - - def _console_print(self, *args, **kwargs): - """Print through the active command-safe console.""" - self._output_console().print(*args, **kwargs) - - def handle_bang_shell(self, text: str) -> bool: - """Run a ``!`` submission. Returns True when it was handled. - - Dispatched from the input loop BEFORE slash-command routing and before - anything is queued for the agent, so a bang command never becomes a - turn: no user message, no assistant message, no tool result touches - ``self.conversation_history``. That is what makes ``!`` free — zero - tokens, and role alternation / prompt caching are untouched by - construction. The invariant is covered by - tests/cli/test_bang_shell_mode.py. - - Returns False when the text is not a bang command or when bang mode is - disabled for this context (gateway/cron), letting the caller fall - through to normal routing. - """ - from hermes_cli.bang_shell import ( - USAGE_HINT, - bang_shell_enabled, - check_bang_approval, - is_bang_command, - parse_bang_command, - resolve_bang_cwd, - run_bang_command, - ) - - if not is_bang_command(text): - return False - if not bang_shell_enabled(): - # Gateway / cron / API contexts: no composer, no human at a - # keyboard, and those users already have their own shells. Let the - # text route normally rather than becoming remote execution. - return False - - command = parse_bang_command(text) - if not command: - # Bare `!` — show what the feature does instead of running an - # empty shell or sending "!" to the model. - self._console_print(f"[dim]{USAGE_HINT}[/]") - return True - - approval = check_bang_approval(command) - if not approval.get("approved"): - message = approval.get("message") or ( - f"Command denied: {approval.get('description', 'flagged as dangerous')}" - ) - self._console_print(f"[bold red]{_escape(str(message))}[/]") - return True - - cwd = resolve_bang_cwd(getattr(self, "session_id", None)) - exit_code = run_bang_command( - command, - cwd=cwd, - writer=lambda line: self._console_print(_rich_text_from_ansi(line)), - ) - if exit_code: - self._console_print(f"[dim]! exited {exit_code}[/]") - return True - @staticmethod def _resolve_personality_prompt(value) -> str: """Accept string or dict personality value; return system prompt string. @@ -12467,63 +6176,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): - def _show_gateway_status(self): - """Show status of the gateway and connected messaging platforms.""" - from gateway.config import load_gateway_config, Platform - - print() - print("+" + "-" * 60 + "+") - print("|" + " " * 15 + "(✿◠‿◠) Gateway Status" + " " * 17 + "|") - print("+" + "-" * 60 + "+") - print() - - try: - config = load_gateway_config() - - print(" Messaging Platform Configuration:") - print(" " + "-" * 55) - - platform_status = { - Platform.TELEGRAM: ("Telegram", "TELEGRAM_BOT_TOKEN"), - Platform.DISCORD: ("Discord", "DISCORD_BOT_TOKEN"), - Platform.SLACK: ("Slack", "SLACK_BOT_TOKEN"), - Platform.WHATSAPP: ("WhatsApp", "WHATSAPP_ENABLED"), - } - - for platform, (name, env_var) in platform_status.items(): - pconfig = config.platforms.get(platform) - if pconfig and pconfig.enabled: - home = config.get_home_channel(platform) - home_str = f" → {home.name}" if home else "" - print(f" ✓ {name:<12} Enabled{home_str}") - else: - print(f" ○ {name:<12} Not configured ({env_var})") - - print() - print(" Session Reset Policy:") - print(" " + "-" * 55) - policy = config.default_reset_policy - print(f" Mode: {policy.mode}") - print(f" Daily reset at: {policy.at_hour}:00") - print(f" Idle timeout: {policy.idle_minutes} minutes") - - print() - print(" To start the gateway:") - print(" python cli.py --gateway") - print() - print(f" Configuration file: {display_hermes_home()}/config.yaml") - print() - - except Exception as e: - print(f" Error loading gateway config: {e}") - print() - print(" To configure the gateway:") - print(" 1. Set environment variables:") - print(" TELEGRAM_BOT_TOKEN=your_token") - print(" DISCORD_BOT_TOKEN=your_token") - print(f" 2. Or configure settings in {display_hermes_home()}/config.yaml") - print() - # Slash dispatch: canonical command -> (method name, pass cmd_original?). # Commands absent here resolve by convention to ``_handle__command(cmd)`` # (dashes -> underscores). Resolved via getattr at dispatch time so @@ -12823,620 +6475,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): _cprint(f"{_DIM}{_ACCENT}Type /help for available commands{_RST}") return True - def _cmd_exit(self, cmd_original: str): - # /exit --delete also removes the session's transcripts + SQLite history. - _args = _slash_args(cmd_original).lower() - if _args in {"--delete", "-d"}: - self._delete_session_on_exit = True - elif _args: - _cprint(f" {_DIM}✗ Unknown argument: {_escape(_args)}. Use /exit --delete to also remove session history.{_RST}") - return True - return False - - def _cmd_help(self, cmd_original: str): - self.show_help(_slash_args(cmd_original)) - - def _cmd_redraw(self, cmd_original: str): - # Manual recovery for terminal buffer drift from multiplexer - # tab switches, subshell ``clear``, SSH window restores, etc. - # See issue #8688 (cmux). Ctrl+L is bound to the same helper. - self._force_full_redraw() - _cprint(f" {_DIM}✓ UI redrawn{_RST}") - - def _print_random_tip(self) -> None: - """Best-effort discovery tip (startup + /clear); never raises.""" - try: - from hermes_cli.tips import get_random_tip - _tip = get_random_tip() - try: - from hermes_cli.skin_engine import get_active_skin - _tip_color = get_active_skin().get_color("banner_dim", "#B8860B") - except Exception: - _tip_color = "#B8860B" - self._console_print(f"[dim {_tip_color}]✦ Tip: {_tip}[/]") - except Exception: - pass - - def _cmd_clear(self, cmd_original: str): - if self._confirm_destructive_slash( - "clear", - "This clears the screen and starts a new session.\n" - "The current conversation history will be discarded.", - cmd_original=cmd_original, - ) is None: - return True # confirmation cancelled — command handled, keep REPL alive - self.new_session(silent=True) - _clear_output_history() - # Clear terminal screen. Inside the TUI, Rich's console.clear() - # goes through patch_stdout's StdoutProxy which swallows the - # screen-clear escape sequences. Use prompt_toolkit's output - # object directly to actually clear the terminal. - if self._app: - out = self._app.output - out.erase_screen() - out.cursor_goto(0, 0) - out.flush() - else: - self.console.clear() - # Show fresh banner. Inside the TUI we must route Rich output - # through ChatConsole (which uses prompt_toolkit's native ANSI - # renderer) instead of self.console (which writes raw to stdout - # and gets mangled by patch_stdout). - if self._app: - cc = ChatConsole() - term_w = shutil.get_terminal_size().columns - if self.compact or term_w < 80: - cc.print(_build_compact_banner()) - else: - tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) - cwd = os.getenv("TERMINAL_CWD", os.getcwd()) - ctx_len = None - if hasattr(self, 'agent') and self.agent and hasattr(self.agent, 'context_compressor'): - ctx_len = self.agent.context_compressor.context_length - build_welcome_banner( - console=cc, - model=self.model, - cwd=cwd, - tools=tools, - enabled_toolsets=self.enabled_toolsets, - session_id=self.session_id, - context_length=ctx_len, - provider=self.provider, - ) - _cprint(" ✨ (◕‿◕)✨ Fresh start! Screen cleared and conversation reset.\n") - self._print_random_tip() - else: - self.show_banner() - print(" ✨ (◕‿◕)✨ Fresh start! Screen cleared and conversation reset.\n") - self._print_random_tip() - - def _cmd_title(self, cmd_original: str): - parts = cmd_original.split(maxsplit=1) - if len(parts) > 1: - raw_title = parts[1].strip() - if raw_title: - if self._session_db: - # Sanitize the title early so feedback matches what gets stored - try: - from hermes_state import SessionDB - new_title = SessionDB.sanitize_title(raw_title) - except ValueError as e: - # sanitize_title rejected the input (e.g. too long). - # Print that one reason and stop — don't fall - # through to the "empty after cleanup" branch and - # print a second, contradictory error (SC-05). - _cprint(f" {e}") - return True - if not new_title: - _cprint(" Title is empty after cleanup. Please use printable characters.") - elif self._session_db.get_session(self.session_id): - # Session exists in DB — set title directly - try: - if self._session_db.set_session_title(self.session_id, new_title): - self._status_bar_title_checked_at = 0.0 - _cprint(f" Session title set: {new_title}") - else: - _cprint(" Session not found in database.") - except ValueError as e: - _cprint(f" {e}") - else: - # Session not created yet — defer the title - # Check uniqueness proactively with the sanitized title - existing = self._session_db.get_session_by_title(new_title) - if existing: - _cprint(f" Title '{new_title}' is already in use by session {existing['id']}") - else: - self._pending_title = new_title - _cprint(f" Session title queued: {new_title} (will be saved on first message)") - else: - from hermes_state import format_session_db_unavailable - _cprint(f" {format_session_db_unavailable()}") - else: - _cprint(" Usage: /title ") - # Show current title and session ID if no argument given - elif self._session_db: - _cprint(f" Session ID: {self.session_id}") - session = self._session_db.get_session(self.session_id) - if session and session.get("title"): - _cprint(f" Title: {session['title']}") - elif self._pending_title: - _cprint(f" Title (pending): {self._pending_title}") - else: - _cprint(" No title set. Usage: /title ") - else: - from hermes_state import format_session_db_unavailable - _cprint(f" {format_session_db_unavailable()}") - - def _cmd_new(self, cmd_original: str): - # Strip inline-skip tokens (now/--yes/-y) before deriving the title - # so "/new now My Session" yields title="My Session" instead of - # title="now My Session". See _split_destructive_skip. - _new_args, _ = self._split_destructive_skip(cmd_original) - title = _new_args.strip() or None - if self._confirm_destructive_slash( - "new", - "This starts a fresh session.\n" - "The current conversation history will be discarded.", - cmd_original=cmd_original, - ) is None: - return True # confirmation cancelled — command handled, keep REPL alive - self.new_session(title=title) - - def _cmd_retry(self, cmd_original: str): - retry_msg = self.retry_last() - if retry_msg and hasattr(self, '_pending_input'): - # Re-queue the message so process_loop sends it to the agent - self._pending_input.put(retry_msg) - - def _cmd_undo(self, cmd_original: str): - # Parse optional turn count: "/undo" → 1, "/undo 3" → 3. - _undo_n = 1 - _undo_parts = cmd_original.split() - if len(_undo_parts) > 1: - try: - _undo_n = int(_undo_parts[1]) - except ValueError: - print(f"(._.) Invalid count {_undo_parts[1]!r} — use /undo or /undo N.") - return True # bad arg — command handled, keep the REPL alive - if _undo_n < 1: - _undo_n = 1 - # Nothing to undo → say so immediately; don't pop a destructive - # confirmation dialog for a guaranteed no-op (SC-06). - if not self.conversation_history: - print("(._.) No messages to undo.") - return True - _undo_desc = ( - "This removes the last user/assistant exchange from history." - if _undo_n == 1 - else f"This removes the last {_undo_n} user turns from history." - ) - if self._confirm_destructive_slash( - "undo", - _undo_desc, - cmd_original=cmd_original, - ) is None: - return True # confirmation cancelled — command handled, keep REPL alive - self.undo_last(_undo_n) - - def _cmd_skills(self, cmd_original: str): - with self._busy_command(self._slow_command_status(cmd_original)): - self._handle_skills_command(cmd_original) - - def _cmd_egress(self, cmd_original: str): - from hermes_cli.slash_exec import CommandContext, execute_command - - self._console_print( - execute_command("egress", CommandContext(surface="cli")).text, - highlight=False, markup=False, - ) - - def _cmd_statusbar(self, cmd_original: str): - self._status_bar_visible = not self._status_bar_visible - state = "visible" if self._status_bar_visible else "hidden" - self._console_print(f" Status bar {state}") - - def _cmd_update(self, cmd_original: str) -> bool: - # A truthy result means the process is relaunching — leave the REPL. - return not self._handle_update_command() - - def _cmd_version(self, cmd_original: str): - from hermes_cli.main import _print_version_info - - _print_version_info(check_updates=True) - - def _cmd_reload(self, cmd_original: str): - from hermes_cli.config import reload_env - count = reload_env() - print(f" Reloaded .env ({count} var(s) updated)") - - def _cmd_reload_skills(self, cmd_original: str): - with self._busy_command(self._slow_command_status(cmd_original)): - self._reload_skills() - - def _cmd_plugins(self, cmd_original: str): - try: - # Discover from disk (bundled + user), matching `hermes plugins - # list` — so installed-but-not-enabled plugins are visible here - # too. The plugin manager only knows about *loaded* plugins, so - # using it alone made freshly-installed, not-yet-enabled plugins - # look like "nothing installed". - from hermes_cli.plugins_cmd import ( - _discover_all_plugins, - _get_disabled_set, - _get_enabled_set, - _plugin_status, - ) - - entries = _discover_all_plugins() - enabled = _get_enabled_set() - disabled = _get_disabled_set() - - # `/plugins` is a quick glance — default to user-installed - # plugins (what the user actually added). Bundled provider/ - # platform plugins are summarized on one line; the full - # catalog lives behind `hermes plugins list`. - user_entries = [e for e in entries if e[3] != "bundled"] - bundled_count = len(entries) - len(user_entries) - - if not user_entries: - print("No user plugins installed.") - print(" Install one: hermes plugins install owner/repo") - print(f" Or drop a plugin directory into {display_hermes_home()}/plugins/") - if bundled_count: - print(f" ({bundled_count} bundled plugins available — see: hermes plugins list)") - else: - # Loaded-plugin details (tools/hooks/commands counts, errors) - # keyed by name, when available. - loaded: dict = {} - try: - from hermes_cli.plugins import get_plugin_manager - for p in get_plugin_manager().list_plugins(): - loaded[p["name"]] = p - except Exception: - loaded = {} - - print(f"User plugins ({len(user_entries)}):") - for name, version, _desc, source, _dir, key in sorted(user_entries): - state = _plugin_status(name, enabled, disabled, key=key) - glyph = {"enabled": "✓", "disabled": "✗"}.get(state, "○") - ver = f" v{version}" if version else "" - info = loaded.get(name) or {} - bits = [] - if info.get("tools"): - bits.append(f"{info['tools']} tools") - if info.get("hooks"): - bits.append(f"{info['hooks']} hooks") - if info.get("commands"): - bits.append(f"{info['commands']} commands") - detail = f" ({', '.join(bits)})" if bits else "" - label = "" if state == "enabled" else f" [{state}]" - error = f" — {info['error']}" if info.get("error") else "" - print(f" {glyph} {name}{ver}{label}{detail}{error}") - if bundled_count: - print(f" (+{bundled_count} bundled — see: hermes plugins list)") - print(" Enable/disable: hermes plugins enable/disable ") - except Exception as e: - print(f"Plugin system error: {e}") - - def _cmd_queue(self, cmd_original: str): - payload = self._expand_paste_references(_slash_args(cmd_original)) - if not payload: - _cprint(" Usage: /queue ") - else: - self._pending_input.put(payload) - if self._agent_running: - _cprint(f" Queued for the next turn: {payload[:80]}{'...' if len(payload) > 80 else ''}") - else: - _cprint(f" Queued: {payload[:80]}{'...' if len(payload) > 80 else ''}") - - def _cmd_steer(self, cmd_original: str): - # Inject a message after the next tool call without interrupting. - # If the agent is actively running, push the text into the agent's - # pending_steer slot — the drain hook in _execute_tool_calls_* - # will append it to the next tool result's content. If no agent - # is running, fall back to queue semantics (same as /queue). - payload = _slash_args(cmd_original) - if not payload: - _cprint(" Usage: /steer ") - elif self._agent_running and self.agent is not None and hasattr(self.agent, "steer"): - try: - accepted = self.agent.steer(payload) - except Exception as exc: - _cprint(f" Steer failed: {exc}") - else: - if accepted: - _cprint(f" ⏩ Steer queued — arrives after the next tool call: {payload[:80]}{'...' if len(payload) > 80 else ''}") - else: - _cprint(" Steer rejected (empty payload).") - else: - # No active run — treat as a normal next-turn message. - self._pending_input.put(payload) - _cprint(f" No agent running; queued as next turn: {payload[:80]}{'...' if len(payload) > 80 else ''}") - - def _cmd_moa(self, cmd_original: str): - # /moa is one-shot sugar only: run a single prompt through the - # default MoA preset, then restore the prior model. To *switch* to a - # MoA preset for the session, pick it from the model picker (MoA - # presets surface as a virtual "Mixture of Agents" provider). - from hermes_cli.moa_config import ( - moa_usage, - normalize_moa_config, - ) - - payload = _slash_args(cmd_original) - if not payload: - _cprint(f" {moa_usage()}") - return True - moa_cfg = self.config.get("moa") if isinstance(self.config, dict) else {} - normalized = normalize_moa_config(moa_cfg) - preset = normalized["default_preset"] - self._pending_moa_restore_model = { - "requested_provider": getattr(self, "requested_provider", None), - "provider": getattr(self, "provider", None), - "model": getattr(self, "model", None), - "api_key": getattr(self, "api_key", None), - "base_url": getattr(self, "base_url", None), - "api_mode": getattr(self, "api_mode", None), - } - self.requested_provider = "moa" - self.provider = "moa" - self.model = preset - self.api_key = "moa-virtual-provider" - self.base_url = "moa://local" - self.api_mode = "chat_completions" - self.agent = None - self._pending_moa_disable_after_turn = True - self._pending_agent_seed = payload - _cprint(f" MoA one-shot queued with preset {preset}; previous model will be restored after this turn.") - - - # ──────────────────────────────────────────────────────────────── - # /goal — persistent cross-turn goals (Ralph-style loop) - # ──────────────────────────────────────────────────────────────── - def _get_goal_manager(self): - """Return the GoalManager bound to the current session_id. - - Cached on ``self._goal_manager`` and rebound lazily when - ``session_id`` changes (e.g. after /new or a compression-driven - session split). - """ - try: - from hermes_cli.goals import GoalManager - from hermes_cli.config import load_config - except Exception as exc: - logging.debug("goal manager unavailable: %s", exc) - return None - - sid = getattr(self, "session_id", None) or "" - if not sid: - return None - - existing = getattr(self, "_goal_manager", None) - if existing is not None and getattr(existing, "session_id", None) == sid: - return existing - - try: - cfg = load_config() or {} - goals_cfg = cfg.get("goals") or {} - max_turns = int(goals_cfg.get("max_turns", 20) or 20) - except Exception: - max_turns = 20 - - mgr = GoalManager(session_id=sid, default_max_turns=max_turns) - self._goal_manager = mgr - return mgr - - def _get_heartbeat_manager(self): - """Return the HeartbeatManager bound to the current session_id. - - Cached on ``self._heartbeat_manager`` and rebound lazily when - ``session_id`` changes (mirrors ``_get_goal_manager``). - """ - try: - from hermes_cli.heartbeat import HeartbeatManager - except Exception as exc: - logging.debug("heartbeat manager unavailable: %s", exc) - return None - - sid = getattr(self, "session_id", None) or "" - if not sid: - return None - - existing = getattr(self, "_heartbeat_manager", None) - if existing is not None and getattr(existing, "session_id", None) == sid: - return existing - - mgr = HeartbeatManager(session_id=sid) - self._heartbeat_manager = mgr - return mgr - - def _start_heartbeat_watchdog(self): - """Start the idle-poll thread that fires due heartbeats. - - Same pattern as the wake-word watchdog: a daemon thread polls a few - times a minute; when the session is idle (no agent running, empty - input queue) and the heartbeat is due, its prompt is injected into - ``_pending_input`` as a normal user turn. Missed ticks coalesce — - the anchor resets on fire, so a busy hour yields ONE heartbeat turn, - not a backlog. Idempotent; safe to call on every /heartbeat set. - """ - if getattr(self, "_heartbeat_watchdog_started", False): - return - self._heartbeat_watchdog_started = True - - from hermes_cli.heartbeat import POLL_SECONDS - - def _loop(): - try: - while not getattr(self, "_should_exit", False): - time.sleep(POLL_SECONDS) - try: - mgr = self._get_heartbeat_manager() - if mgr is None or not mgr.is_active(): - continue - busy = ( - self._agent_running - or getattr(self, "_voice_recording", False) - or getattr(self, "_voice_processing", False) - or not self._pending_input.empty() - ) - if busy: - continue - prompt = mgr.due_prompt() - if prompt: - self._pending_input.put(prompt) - except Exception as exc: - logging.debug("heartbeat watchdog tick failed: %s", exc) - finally: - self._heartbeat_watchdog_started = False - - threading.Thread(target=_loop, daemon=True, name="heartbeat-watchdog").start() - - # ──────────────────────────────────────────────────────────────── - # /loop — recurring in-session wakeups (Claude Code /loop parity) - # ──────────────────────────────────────────────────────────────── - def _get_loop_manager(self): - """Return the LoopManager bound to the current session_id. - - Cached on ``self._loop_manager`` and rebound lazily when - ``session_id`` changes (mirrors ``_get_goal_manager``). - """ - try: - from hermes_cli.loops import LoopManager - except Exception as exc: - logging.debug("loop manager unavailable: %s", exc) - return None - - sid = getattr(self, "session_id", None) or "" - if not sid: - return None - - existing = getattr(self, "_loop_manager", None) - if existing is not None and getattr(existing, "session_id", None) == sid: - return existing - - mgr = LoopManager(session_id=sid) - self._loop_manager = mgr - return mgr - - def _maybe_fire_loop_tick(self) -> None: - """Idle hook run from process_loop: fire a due /loop wakeup. - - Only runs while the agent is idle and nothing is queued — a real - user message always wins the idle boundary. An active (non-parked) - /goal also wins: its judge-driven continuations own the idle - boundary, so the loop defers to the next poll. - """ - mgr = self._get_loop_manager() - if mgr is None or not mgr.is_due(): - return - # The idle poll runs at ~10 Hz; once a tick is due but deferred - # (queued input / active goal), every poll would otherwise hit the - # DB via goal_blocks_loop_tick. Throttle the deferred re-check. - now = time.time() - if now - getattr(self, "_last_loop_tick_check", 0.0) < 2.0: - return - self._last_loop_tick_check = now - # Real user input (or anything else queued) takes priority; the - # loop stays due and fires at the next idle poll. - try: - if not self._pending_input.empty(): - return - except Exception: - return - try: - from hermes_cli.loops import goal_blocks_loop_tick - - if goal_blocks_loop_tick(mgr.session_id): - return - except Exception: - pass - - wakeup = mgr.fire_tick() - if not wakeup: - return - try: - state = mgr.state - tick_no = state.ticks_fired if state else "?" - _cprint(f" {_DIM}↻ /loop wakeup #{tick_no} firing…{_RST}") - self._pending_input.put(wakeup) - except Exception as exc: - logging.debug("loop tick injection failed: %s", exc) - try: - mgr.abandon_tick() - except Exception: - pass - return - # A slash-command loop (e.g. `/loop 10m /recap`) is dispatched via - # process_command, which never reaches the post-turn chat() finally - # block — so the tick would never complete and the loop would wedge - # on awaiting_response. Slash ticks have no model reply to evaluate; - # complete them immediately (caps and scheduling still apply). - if wakeup.lstrip().startswith("/"): - try: - decision = mgr.complete_tick("") - msg = decision.get("message") or "" - if msg: - _cprint(f" {msg}") - except Exception: - pass - - def _maybe_complete_loop_tick_after_turn(self) -> None: - """Post-turn hook: evaluate a finished /loop wakeup turn. - - No-op unless the turn that just ended was a loop wakeup - (``awaiting_response`` set by ``fire_tick``). Detects the - LOOP_COMPLETE marker, judges --until, applies caps, and schedules - the next tick. Mirrors _maybe_continue_goal_after_turn's shape. - """ - mgr = self._get_loop_manager() - if mgr is None: - return - state = mgr.state - if state is None or not state.awaiting_response: - return - - # A user-interrupted wakeup turn pauses the loop (recoverable via - # /loop resume) — same contract as the goal loop's Ctrl+C handling. - if getattr(self, "_last_turn_interrupted", False): - try: - mgr.pause(reason="user-interrupted (Ctrl+C)") - except Exception: - pass - _cprint( - f" {_DIM}⏸ Loop paused — wakeup turn was interrupted. " - f"Use /loop resume to continue, or /loop stop to end it.{_RST}" - ) - return - - last_response = "" - try: - hist = self.conversation_history or [] - for msg in reversed(hist): - if msg.get("role") == "assistant": - content = msg.get("content", "") - if isinstance(content, list): - parts = [ - p.get("text", "") - for p in content - if isinstance(p, dict) and p.get("type") in {"text", "output_text"} - ] - last_response = "\n".join(t for t in parts if t) - else: - last_response = str(content or "") - break - except Exception: - last_response = "" - - decision = mgr.complete_tick(last_response) - msg = decision.get("message") or "" - if msg: - _cprint(f" {msg}") - elif decision.get("status") == "active" and mgr.state is not None: - _cprint(f" {_DIM}↻ Loop: {mgr.state.remaining_label()}.{_RST}") - - - def _owns_process_notification(self, event: dict) -> bool: """Return whether this CLI session provably owns a delegation event. @@ -13511,346 +6549,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): except Exception: pass # Non-fatal — never break the main loop - def _maybe_continue_goal_after_turn(self) -> None: - """Hook run after every CLI turn. Judges + maybe re-queues. - - Safe to call when no goal is set — returns quickly. - - Preemption is automatic: if a real user message is already in - ``_pending_input`` we skip judging (the user's new input takes - priority and we'll re-judge after that turn). If judge says done, - mark it done and tell the user. If judge says continue and we're - under budget, push the continuation prompt onto the queue. - - Interrupt handling: if the turn was user-cancelled (Ctrl+C), we - AUTO-PAUSE the goal instead of judging + re-queuing. Otherwise - Ctrl+C feels like it did nothing — the judge runs on whatever - partial output landed, almost always says "continue", and the - loop keeps going. Auto-pause keeps the goal recoverable via - ``/goal resume`` once the user has sorted out what they want. - The empty-response skip mirrors the gateway guard at - ``_handle_message`` in ``gateway/run.py``. - """ - mgr = self._get_goal_manager() - if mgr is None or not mgr.is_active(): - return - - # If a real user message is already queued, don't inject a - # continuation prompt on top — let the user's turn go first. - # Slash commands don't count as "real user messages" for this - # check: they're inspection/mutation (e.g. /subgoal added mid- - # run) and the process_loop dispatches them via process_command, - # not via chat(). If we treat a queued /subgoal as preempting, - # the goal loop silently stalls — we'd return here, then the - # slash command consumes its queue slot via process_command() - # which never re-fires the goal hook. Peek at all queued entries - # and only defer when there's a non-slash payload. - try: - pending = getattr(self, "_pending_input", None) - if pending is not None and not pending.empty(): - has_real_message = False - try: - # Queue.queue is the underlying deque — direct peek - # without disturbing FIFO order. - for entry in list(pending.queue): - # Bundled payloads are (text, images) tuples; - # unpack for inspection. - if isinstance(entry, tuple) and entry: - entry = entry[0] - if isinstance(entry, str) and _looks_like_slash_command(entry): - continue - has_real_message = True - break - except Exception: - # Fallback: if we can't introspect the queue, behave - # like the old check and defer to be safe. - has_real_message = True - if has_real_message: - return - except Exception: - pass - - # If the turn was user-interrupted (Ctrl+C), auto-pause the goal - # and bail. The judge call would almost always return "continue" - # on the partial output and immediately re-queue another turn, - # which is exactly what the user cancelled. Pausing (rather than - # silently skipping) is the observable, recoverable behavior. - if getattr(self, "_last_turn_interrupted", False): - try: - mgr.pause(reason="user-interrupted (Ctrl+C)") - except Exception as exc: - logging.debug("goal pause-on-interrupt failed: %s", exc) - _cprint( - f" {_DIM}⏸ Goal paused — turn was interrupted. " - f"Use /goal resume to continue, or /goal clear to stop.{_RST}" - ) - return - - # Extract the agent's final response for this turn. - last_response = "" - try: - hist = self.conversation_history or [] - for msg in reversed(hist): - if msg.get("role") == "assistant": - content = msg.get("content", "") - if isinstance(content, list): - # Multimodal content — flatten text parts. - parts = [ - p.get("text", "") - for p in content - if isinstance(p, dict) and p.get("type") in {"text", "output_text"} - ] - last_response = "\n".join(t for t in parts if t) - else: - last_response = str(content or "") - break - except Exception: - last_response = "" - - # Skip judging on empty/whitespace-only responses. These are almost - # always transient failures (API error, empty stream) where the - # judge would say "continue" and trip the consecutive-parse-failures - # backstop unnecessarily. Mirrors the gateway guard. - if not last_response.strip(): - return - - try: - from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() - except Exception: - _bg_procs = None - - decision = mgr.evaluate_after_turn( - last_response, - user_initiated=True, - background_processes=_bg_procs, - ) - msg = decision.get("message") or "" - if msg: - _cprint(f" {msg}") - - if decision.get("should_continue"): - prompt = decision.get("continuation_prompt") - if prompt: - try: - self._pending_input.put(prompt) - except Exception as exc: - logging.debug("goal continuation enqueue failed: %s", exc) - - - - def _toggle_verbose(self): - """Cycle tool progress mode: off → new → all → verbose → off. - - Tool-progress display (full args / results / think blocks at the - ``verbose`` step) is INDEPENDENT of global DEBUG logging. Cycling - through here does not change ``self.verbose`` or the agent's - ``verbose_logging`` / ``quiet_mode`` — those remain under the - explicit ``-v``/``--verbose`` flag and the ``/verbose-logging`` - toggle. See PR #6a1aa420e for the history that decoupled them. - """ - cycle = ["off", "new", "all", "verbose"] - try: - idx = cycle.index(self.tool_progress_mode) - except ValueError: - idx = 2 # default to "all" - self.tool_progress_mode = cycle[(idx + 1) % len(cycle)] - - # /verbose is the explicit tool-progress control, so cycling it takes - # ownership of the mode back from focus view. Leaving _focus_view_enabled - # set would show a "focus" status-bar badge and hidden-line counts while - # tool lines were visibly printing. Display-only state change. - if getattr(self, "_focus_view_enabled", False): - self._focus_view_enabled = False - self._focus_saved_tool_progress = None - self._focus_hidden_lines = 0 - self._focus_last_counted_tool = None - try: - from hermes_cli.focus_view import FOCUS_CONFIG_KEY - - save_config_value(FOCUS_CONFIG_KEY, False) - except Exception: - pass - - if self.agent: - self.agent.reasoning_callback = self._current_reasoning_callback() - # Keep the live agent's tool_progress_mode in sync so the - # tool_executor rendering path reflects the new mode this turn, - # without waiting for an agent rebuild. - self.agent.tool_progress_mode = self.tool_progress_mode - - # Use raw ANSI codes via _cprint so the output is routed through - # prompt_toolkit's renderer. self.console.print() with Rich markup - # writes directly to stdout which patch_stdout's StdoutProxy mangles - # into garbled sequences like '?[33mTool progress: NEW?[0m' (#2262). - from hermes_cli.colors import Colors as _Colors - labels = { - "off": f"{_Colors.DIM}Tool progress: OFF{_Colors.RESET} — silent mode, just the final response.", - "new": f"{_Colors.YELLOW}Tool progress: NEW{_Colors.RESET} — show each new tool (skip repeats).", - "all": f"{_Colors.GREEN}Tool progress: ALL{_Colors.RESET} — show every tool call.", - "verbose": f"{_Colors.BOLD}{_Colors.GREEN}Tool progress: VERBOSE{_Colors.RESET} — full args, results, and think blocks.", - } - _cprint(labels.get(self.tool_progress_mode, "")) - - def _write_terminal_breadcrumb(self) -> None: - """Record this terminal's live session for bare ``hermes -c``. - - Called at session start and whenever ``self.session_id`` is - reassigned mid-run (/new, /branch, auto-compression rotation) so a - later bare ``-c`` in THIS terminal resumes THIS conversation's live - tip. Best-effort — never raises, no-op without a terminal identity - or when session.terminal_continue is false. - """ - try: - from hermes_cli.terminal_breadcrumbs import write_breadcrumb - - write_breadcrumb(self.session_id) - except Exception: - pass - - def _transfer_session_yolo(self, old_session_id: str, new_session_id: str) -> None: - """Move YOLO bypass state from an old session key to a new one. - - Called whenever ``self.session_id`` is reassigned mid-run — ``/branch`` - forks into a new session, and auto-compression rotates the agent's - session id into a fresh continuation session. Without this transfer - the user's ``/yolo ON`` toggle would silently revert on the very next - turn (the same UX failure mode that motivated this entire fix), since - ``_session_yolo`` is keyed by session id. - - Mirrors ``tui_gateway/server.py`` (~line 1297-1305) which performs the - same transfer for the TUI's session-rename path. No-op when YOLO - wasn't enabled or when the ids match. - """ - if not old_session_id or not new_session_id or old_session_id == new_session_id: - return - try: - from tools.approval import ( - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, - ) - except Exception: - return - if is_session_yolo_enabled(old_session_id): - enable_session_yolo(new_session_id) - disable_session_yolo(old_session_id) - # Carry the persisted flag onto the continuation row so a later - # `hermes --resume ` restores the bypass too. getattr - # guard: tests call this unbound against a minimal stand-in. - _persist = getattr(self, "_persist_session_yolo", None) - if _persist: - _persist(new_session_id, True) - - def _is_session_yolo_active(self) -> bool: - """Whether YOLO bypass is currently enabled for this CLI session. - - Reads from ``tools.approval._session_yolo`` (the same set that - ``enable_session_yolo`` / ``disable_session_yolo`` write to) so the - status bar reflects the actual bypass state instead of a stale env - var. Also honors the process-start ``--yolo`` flag, which freezes - ``HERMES_YOLO_MODE`` into ``_YOLO_MODE_FROZEN`` before tool imports - happen. - """ - try: - from tools.approval import ( - _YOLO_MODE_FROZEN, - is_session_yolo_enabled, - ) - except Exception: - return False - if _YOLO_MODE_FROZEN: - return True - # Use ``getattr`` so test fixtures that build a CLI via ``__new__`` - # (skipping ``__init__``) don't trip an AttributeError here; the - # status-bar builders swallow exceptions silently but lose every - # field after the failure. - session_key = getattr(self, "session_id", None) or "default" - return is_session_yolo_enabled(session_key) - - def _toggle_yolo(self): - """Toggle YOLO mode — skip all dangerous command approval prompts. - - Per-session toggle that mirrors the gateway and TUI ``/yolo`` handlers - (see ``gateway/run.py:_handle_yolo_command`` and - ``tui_gateway/server.py`` key=="yolo"). We deliberately do NOT mutate - ``HERMES_YOLO_MODE`` here — that env var is read once at module import - time into ``tools.approval._YOLO_MODE_FROZEN`` to keep prompt-injected - skills from flipping the bypass mid-session, so setting it after CLI - startup is a silent no-op. Routing through ``enable_session_yolo`` / - ``disable_session_yolo`` gives the same auditable, per-session bypass - the other surfaces have. ``run_conversation`` binds - ``self.session_id`` as the active approval session key via - ``set_current_session_key`` so the bypass takes effect on the very - next dangerous command in this run. - """ - from hermes_cli.colors import Colors as _Colors - from tools.approval import ( - _YOLO_MODE_FROZEN, - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, - ) - - # Process-level YOLO (--yolo flag / HERMES_YOLO_MODE at startup) is - # frozen into tools.approval at import time and cannot be disabled by - # the session toggle. Before this guard, /yolo printed "YOLO mode OFF — - # dangerous commands will require approval" while every command kept - # auto-approving (the frozen flag short-circuits the approval gate - # ahead of the session check) — a false safety claim. Say the truth - # instead of toggling a bypass that has no effect. - if _YOLO_MODE_FROZEN: - _cprint( - f" ⚡ YOLO is {_Colors.BOLD}{_Colors.RED}locked ON{_Colors.RESET}" - " for this process (started with --yolo / HERMES_YOLO_MODE)." - " /yolo cannot disable it — restart without the flag to" - " re-enable approvals." - ) - return - - session_key = self.session_id or "default" - # ``getattr`` guard: tests exercise this method unbound against a - # minimal stand-in object (see tests/cli/test_cli_yolo_toggle.py); - # persistence is best-effort either way. - _persist = getattr(self, "_persist_session_yolo", None) - if is_session_yolo_enabled(session_key): - disable_session_yolo(session_key) - if _persist: - _persist(session_key, False) - _cprint( - f" ⚠ YOLO mode {_Colors.BOLD}{_Colors.RED}OFF{_Colors.RESET}" - " — dangerous commands will require approval." - ) - else: - enable_session_yolo(session_key) - if _persist: - _persist(session_key, True) - _cprint( - f" ⚡ YOLO mode {_Colors.BOLD}{_Colors.GREEN}ON{_Colors.RESET}" - " — all commands auto-approved. Use with caution." - ) - - def _persist_session_yolo(self, session_key: str, enabled: bool) -> None: - """Persist the YOLO flag to the session row so --resume restores it. - - Best-effort: the in-memory toggle is authoritative for this process; - persistence only affects a future ``hermes --resume``. Skipped when the - session store is unavailable or the row doesn't exist yet (the row is - created lazily on the first turn — ``_toggle_yolo`` before any chat - writes nothing, and the launch-time ``--yolo`` flag is carried into the - creation-time model_config instead). - """ - db = getattr(self, "_session_db", None) - if db is None or not session_key or session_key == "default": - return - try: - db.set_session_yolo(session_key, enabled) - except Exception: - pass - - - - def _on_reasoning(self, reasoning_text: str): """Callback for intermediate reasoning display during tool-call loops.""" if not reasoning_text: @@ -13858,439 +6556,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._reasoning_preview_buf = getattr(self, "_reasoning_preview_buf", "") + reasoning_text self._flush_reasoning_preview(force=False) - def _manual_compress(self, cmd_original: str = ""): - """Manually trigger context compression on the current conversation. - - Two modes: - - * ``/compress []`` — compress the *whole* history. An - optional focus topic guides the summariser to preserve - information related to *focus* while being more aggressive - about discarding everything else. Inspired by Claude Code's - ``/compact `` feature. - * ``/compress here [N]`` — boundary-aware compression. Summarize - everything *except* the most recent ``N`` exchanges (default - 2), which are preserved verbatim. Inspired by Claude Code's - Rewind "Summarize up to here" action (v2.1.139, May 2026, - https://code.claude.com/docs/en/whats-new/2026-w20). Lets the - user pick the compression boundary instead of leaving it to - the automatic token-budget heuristic. - """ - if not self.conversation_history or len(self.conversation_history) < 4: - print("(._.) Not enough conversation to compress (need at least 4 messages).") - return - - if not self.agent: - print("(._.) No active agent -- send a message first.") - return - - # No compression_enabled gate here: the config flag disables - # *automatic* compaction only. Manual /compress is an explicit user - # action — the context-overflow error path (conversation_loop.py) - # directs users here when auto-compaction is off, and the gateway's - # /compress handler has never gated on the flag. - - from hermes_cli.partial_compress import ( - extract_compress_flags, - parse_partial_compress_args, - rejoin_compressed_head_and_tail, - split_history_for_partial_compress, - summarize_compress_preview, - ) - from agent.conversation_compression import ( - finalize_context_engine_compression_notification, - ) - - # Args after the command word (e.g. "/compress here 3" -> "here 3"). - raw_args = "" - if cmd_original: - _parts = cmd_original.strip().split(None, 1) - if len(_parts) > 1: - raw_args = _parts[1].strip() - - # Strip --preview/--dry-run/--aggressive before positional parsing - # so the flags coexist with 'here [N]' / focus-topic forms. - raw_args, preview, aggressive = extract_compress_flags(raw_args) - partial, keep_last, focus_topic = parse_partial_compress_args(raw_args) - focus_topic = focus_topic or "" - - if aggressive: - # LLM-free hard truncation is not supported: it would need its - # own transcript-persistence path outside the guarded - # _compress_context rotation machinery. Surface that instead of - # silently mis-parsing the flag as a focus topic. - print("(._.) --aggressive is not supported; use '/compress here [N]' " - "to keep only recent exchanges, or /undo to drop turns.") - if not preview: - return - - if preview: - from agent.model_metadata import estimate_request_tokens_rough - _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" - _tools = getattr(self.agent, "tools", None) or None - approx_tokens = estimate_request_tokens_rough( - self.conversation_history, - system_prompt=_sys_prompt, - tools=_tools, - ) - report = summarize_compress_preview( - self.conversation_history, - partial, - keep_last, - focus_topic or None, - approx_tokens, - ) - for line in report["lines"]: - print(f"🗜️ {line}") - return - - original_count = len(self.conversation_history) - with self._busy_command("Compressing context...", blocks_input=False): - try: - from agent.model_metadata import estimate_request_tokens_rough - from agent.manual_compression_feedback import summarize_manual_compression - original_history = list(self.conversation_history) - - # Boundary-aware split: only the head is summarized; the - # most recent `keep_last` exchanges ride along verbatim. - tail: list = [] - head = original_history - if partial: - head, tail = split_history_for_partial_compress( - original_history, keep_last - ) - if not tail: - # Split degenerated (everything would be kept, or - # no head left to compress). Fall back to full - # compression so the user still gets an action. - partial = False - head = original_history - - # Include system prompt + tool schemas in the estimate — - # a transcript-only number understates real request pressure - # and can even appear to grow after compression because a - # dense handoff summary replaces many short turns (#6217). - _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" - _tools = getattr(self.agent, "tools", None) or None - approx_tokens = estimate_request_tokens_rough( - original_history, - system_prompt=_sys_prompt, - tools=_tools, - ) - if partial: - print(f"🗜️ Summarizing up to here: compressing {len(head)} of " - f"{original_count} messages (~{approx_tokens:,} tokens), " - f"keeping last {keep_last} exchange(s) verbatim...") - elif focus_topic: - print(f"🗜️ Compressing {original_count} messages (~{approx_tokens:,} tokens), " - f"focus: \"{focus_topic}\"...") - else: - print(f"🗜️ Compressing {original_count} messages (~{approx_tokens:,} tokens)...") - - # Pass None as system_message so _compress_context rebuilds - # the system prompt from scratch via _build_system_prompt(None). - # Passing _cached_system_prompt caused duplication because - # _build_system_prompt appends system_message to prompt_parts - # which already contain the agent identity — resulting in the - # identity block appearing twice (issue #15281). - compressed, _ = self.agent._compress_context( - head, - None, - approx_tokens=approx_tokens, - focus_topic=focus_topic or None, - force=True, - defer_context_engine_notification=True, - ) - - # If _compress_context returned unchanged because a - # concurrent compression lock is held, tell the user - # clearly instead of showing the misleading - # "No changes from compression" no-op text. The wording - # distinguishes a confirmed holder from an unconfirmed - # acquisition failure (describe_compression_lock_skip). - # Type-pinned check (is True / str): the flag's only real - # values are None/True/holder-string, and a bare getattr - # truthiness test is fooled by MagicMock auto-attributes on - # test-double agents (skill pitfall: MagicMock vs hasattr). - _lock_skip_signal = getattr( - self.agent, "_compression_skipped_due_to_lock", None - ) - if _lock_skip_signal is True or isinstance(_lock_skip_signal, str): - from agent.manual_compression_feedback import ( - describe_compression_lock_skip, - ) - print( - " " - + describe_compression_lock_skip( - self.agent._compression_skipped_due_to_lock - ) - ) - self.agent._compression_skipped_due_to_lock = None - # No boundary was committed on a lock-skip; discard the - # deferred context-engine notification (exactly-once). - finalize_context_engine_compression_notification( - self.agent, - committed=False, - ) - return - - if partial and tail: - compressed = rejoin_compressed_head_and_tail(compressed, tail) - self.conversation_history = compressed - # _compress_context ends the old session and creates a new child - # session on the agent (run_agent.py::_compress_context). Sync the - # CLI's session_id so /status, /resume, exit summary, and title - # generation all point at the live continuation session, not the - # ended parent. Without this, subsequent end_session() calls target - # the already-closed parent and the child is orphaned. - if ( - getattr(self.agent, "session_id", None) - and self.agent.session_id != self.session_id - ): - self.session_id = self.agent.session_id - getattr(self, "_write_terminal_breadcrumb", lambda: None)() - self._pending_title = None - # Manual /compress replaces conversation_history with a new - # compressed handoff for the child session. Persist it from - # offset 0 so resume can recover the continuation after exit. - self.agent._flush_messages_to_session_db(self.conversation_history, None) - finalize_context_engine_compression_notification( - self.agent, - committed=True, - ) - new_tokens = estimate_request_tokens_rough( - self.conversation_history, - system_prompt=_sys_prompt, - tools=_tools, - ) - summary = summarize_manual_compression( - original_history, - self.conversation_history, - approx_tokens, - new_tokens, - compression_state=getattr( - self.agent, "context_compressor", None - ), - ) - if ( - summary.get("aborted") - or summary.get("fallback_used") - or summary.get("refused_would_grow") - ): - icon = "⚠️" - else: - icon = "🗜️" if summary["noop"] else "✅" - print(f" {icon} {summary['headline']}") - print(f" {summary['token_line']}") - if summary["note"]: - print(f" {summary['note']}") - - except Exception as e: - finalize_context_engine_compression_notification( - self.agent, - committed=False, - ) - print(f" ❌ Compression failed: {e}") - - - - def _handle_usage_command(self, cmd_original: str): - """Dispatch `/usage [reset [--force]]`. - - Bare `/usage` keeps the classic display. `/usage reset` redeems one - banked Codex rate-limit reset credit (guarded: refuses when limits - aren't exhausted unless --force). - """ - parts = cmd_original.split() - args = [p.lower() for p in parts[1:]] - if args and args[0] == "reset": - self._usage_reset(force="--force" in args[1:]) - return - if args: - print(f" Unknown /usage subcommand: {' '.join(parts[1:])}. Try /usage or /usage reset [--force].") - return - self._show_usage() - - def _usage_reset(self, force: bool = False): - """`/usage reset [--force]` — redeem one banked Codex reset credit.""" - provider = ( - (getattr(self.agent, "provider", None) if self.agent else None) - or getattr(self, "provider", None) - ) - normalized = str(provider or "").strip().lower() - if normalized != "openai-codex": - print(" Banked usage resets are only available on the openai-codex provider.") - print(" Switch with `/model` or `hermes auth` first.") - return - base_url = (getattr(self.agent, "base_url", None) if self.agent else None) or getattr(self, "base_url", None) - api_key = (getattr(self.agent, "api_key", None) if self.agent else None) or getattr(self, "api_key", None) - - from agent.account_usage import redeem_codex_reset_credit - - print(" ⏳ Checking banked reset credits...") - with concurrent.futures.ThreadPoolExecutor(max_workers=1) as _pool: - try: - result = _pool.submit( - redeem_codex_reset_credit, - base_url=base_url, - api_key=api_key, - force=force, - ).result(timeout=45.0) - except concurrent.futures.TimeoutError: - print(" ❌ Timed out talking to the Codex backend — try again shortly.") - return - print(f" {result.message}") - - def _show_context_breakdown(self, cmd_original: str = ""): - """`/context [all]` — visual context-window usage breakdown. - - Renders a 5×20 glyph block grid (each cell ≈ 1% of the model context - window) plus an estimated per-category table: system prompt, tool - definitions, rules, skills index, MCP, subagents, memory, and the - conversation itself — versus free space. `/context all` appends the - expanded per-skill and per-toolset cost listings. - - Read-only: same chars/4 estimation engine as the desktop context - popover (agent.context_breakdown) — no provider calls, no prompt-cache - impact. - """ - if not self.agent: - print(" (._.) No active agent -- send a message first.") - return - - args = cmd_original.split(maxsplit=1)[1].strip().lower() if " " in cmd_original else "" - expanded = args in {"all", "full", "details"} - - from agent.context_breakdown import ( - compute_context_details, - compute_session_context_breakdown, - render_context_breakdown_lines, - ) - - try: - payload = compute_session_context_breakdown( - self.agent, self.conversation_history - ) - except Exception as e: - print(f" (._.) Could not compute context breakdown: {e}") - return - - details = None - if expanded: - try: - details = compute_context_details(self.agent) - except Exception: - details = {"skills": [], "toolsets": []} - - model = payload.get("model") or self.model - print() - print(f" 🧠 Context Usage — {model}") - print() - for line in render_context_breakdown_lines(payload, details=details, grid=True): - print(f" {line}") - print() - - def _show_usage(self): - """Rate limits + session token usage (when a live agent exists) + Nous credits. - - The Nous credits block is agent-independent (a portal fetch), so it runs even - with no live agent — important for the TUI, where /usage runs in a slash-worker - subprocess that resumes the session WITHOUT building an agent (self.agent is None), - which would otherwise early-return before any credits showed. - """ - if not self.agent: - if self._print_nous_credits_block(): - self._print_usage_cta() - else: - print("(._.) No active agent -- send a message first.") - return - - agent = self.agent - calls = agent.session_api_calls - - if calls == 0: - if self._print_nous_credits_block(): - self._print_usage_cta() - else: - print("(._.) No API calls made yet in this session.") - return - - # ── Rate limits (shown first when available) ──────────────── - rl_state = agent.get_rate_limit_state() - if rl_state and rl_state.has_data: - from agent.rate_limit_tracker import format_rate_limit_display - print() - print(format_rate_limit_display(rl_state)) - print() - - # ── Session token usage ───────────────────────────────────── - input_tokens = getattr(agent, "session_input_tokens", 0) or 0 - output_tokens = getattr(agent, "session_output_tokens", 0) or 0 - reasoning_tokens = getattr(agent, "session_reasoning_tokens", 0) or 0 - prompt = agent.session_prompt_tokens - completion = agent.session_completion_tokens - total = agent.session_total_tokens - - compressor = agent.context_compressor - last_prompt = compressor.last_prompt_tokens if compressor.last_prompt_tokens > 0 else 0 - ctx_len = compressor.context_length - pct = min(100, (last_prompt / ctx_len * 100)) if ctx_len else 0 - compressions = compressor.compression_count - - msg_count = len(self.conversation_history) - elapsed = format_duration_compact((datetime.now() - self.session_start).total_seconds()) - - print(" 📊 Session Token Usage") - print(f" {'─' * 40}") - print(f" Model: {agent.model}") - print(f" Input tokens: {input_tokens:>10,}") - print(f" Output tokens: {output_tokens:>10,}") - if reasoning_tokens: - print(f" ↳ Reasoning (subset): {reasoning_tokens:>10,}") - print(f" Prompt tokens (total): {prompt:>10,}") - print(f" Completion tokens: {completion:>10,}") - print(f" Total tokens: {total:>10,}") - print(f" API calls: {calls:>10,}") - print(f" Session duration: {elapsed:>10}") - print(f" {'─' * 40}") - print(f" Current context: {last_prompt:,} / {ctx_len:,} ({pct:.0f}%)") - print(f" Messages: {msg_count}") - print(f" Compressions: {compressions}") - - # Account limits -- fetched off-thread with a hard timeout so slow - # provider APIs don't hang the prompt. - provider = getattr(agent, "provider", None) or getattr(self, "provider", None) - base_url = getattr(agent, "base_url", None) or getattr(self, "base_url", None) - api_key = getattr(agent, "api_key", None) or getattr(self, "api_key", None) - # Lazy import — pulls the OpenAI SDK chain, only needed here. - from agent.account_usage import fetch_account_usage, render_account_usage_lines - account_snapshot = None - if provider: - with concurrent.futures.ThreadPoolExecutor(max_workers=1) as _pool: - try: - account_snapshot = _pool.submit( - fetch_account_usage, provider, - base_url=base_url, api_key=api_key, - ).result(timeout=10.0) - except (concurrent.futures.TimeoutError, Exception): - account_snapshot = None - account_lines = [f" {line}" for line in render_account_usage_lines(account_snapshot)] - if account_lines: - print() - for line in account_lines: - print(line) - - # Nous credits magnitudes + monthly-grant gauge (agent-independent — also - # runs at the no-agent / no-calls early-returns above). See the helper. - if self._print_nous_credits_block(): - self._print_usage_cta() - - if self.verbose: - logging.getLogger().setLevel(logging.DEBUG) - for noisy in ('openai', 'openai._base_client', 'httpx', 'httpcore', 'asyncio', 'hpack', 'grpc', 'modal'): - logging.getLogger(noisy).setLevel(logging.WARNING) - else: - logging.getLogger().setLevel(logging.INFO) # NOTE: We deliberately do NOT raise per-logger levels for # tools/run_agent/etc. in quiet mode. Setting logger.setLevel # above the file handler level filters records before they @@ -14299,149 +6564,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # Console quietness is enforced by hermes_logging not # installing a console StreamHandler in non-verbose mode. - def _show_insights(self, command: str = "/insights"): - """Show usage insights and analytics from session history.""" - # Parse optional --days flag - parts = command.split() - days = 30 - source = None - i = 1 - while i < len(parts): - if parts[i] == "--days" and i + 1 < len(parts): - try: - days = int(parts[i + 1]) - except ValueError: - print(f" Invalid --days value: {parts[i + 1]}") - return - i += 2 - elif parts[i] == "--source" and i + 1 < len(parts): - source = parts[i + 1] - i += 2 - elif parts[i].isdigit(): - days = int(parts[i]) - i += 1 - else: - i += 1 - - try: - from hermes_state import SessionDB - from agent.insights import InsightsEngine - - db = SessionDB() - try: - engine = InsightsEngine(db) - report = engine.generate(days=days, source=source) - print(engine.format_terminal(report)) - finally: - db.close() - except Exception as e: - print(f" Error generating insights: {e}") - - def _check_config_mcp_changes(self) -> None: - """Detect mcp_servers changes in config.yaml and react. - - Called from process_loop every CONFIG_WATCH_INTERVAL seconds. - Compares config.yaml mtime + mcp_servers section against the last - known state. When a change is detected: - - * By default (``mcp.auto_reload_on_config_change: true``) it - auto-triggers ``_reload_mcp()`` and informs the user — legacy - behaviour from #1474. - * When opted out (``mcp.auto_reload_on_config_change: false``) it - does NOT reload. Instead it notifies the user that the config - changed and that they can apply it with ``/reload-mcp`` — while - warning that ``/reload-mcp`` rebuilds the tool surface and - **invalidates the provider prompt cache** (the next message - re-sends the full input prefix, expensive on long-context / - high-reasoning models). This stops silent cache-breaking reloads - when config.yaml is rewritten frequently by external tooling or - other Hermes instances. - """ - - import yaml as _yaml - - CONFIG_WATCH_INTERVAL = 5.0 # seconds between config.yaml stat() calls - - now = time.monotonic() - if now - self._last_config_check < CONFIG_WATCH_INTERVAL: - return - self._last_config_check = now - - from hermes_cli.config import get_config_path as _get_config_path - cfg_path = _get_config_path() - if not cfg_path.exists(): - return - - try: - mtime = cfg_path.stat().st_mtime - except OSError: - return - - if mtime == self._config_mtime: - return # File unchanged — fast path - - # File changed — check whether mcp_servers section changed - self._config_mtime = mtime - try: - with open(cfg_path, encoding="utf-8") as f: - new_cfg = _yaml.safe_load(f) or {} - except Exception: - return - - new_mcp = new_cfg.get("mcp_servers") or {} - # Expand ${VAR} templates so the comparison is consistent with the - # init snapshot (self._config_mcp_servers), which was populated from - # the deep-merged + expanded config. Without this, any - # save_config_value() that rewrites config.yaml (even for unrelated - # keys) triggers a false-positive MCP reload because the raw yaml - # still has "${POWERMEM_API_KEY}" while the snapshot has the - # expanded value. - from hermes_cli.config import _expand_env_vars - new_mcp = _expand_env_vars(new_mcp) - if new_mcp == self._config_mcp_servers: - return # mcp_servers unchanged (some other section was edited) - - # Detected a change in the mcp_servers section. By default we - # auto-reload (legacy behaviour), but if the user has opted out we - # notify instead of reloading — because every reload rebuilds the - # agent tool surface and INVALIDATES the provider prompt cache (the - # next message re-sends the full input prefix, which is expensive on - # long-context / high-reasoning models). - # - # The toggle is the top-level ``mcp.auto_reload_on_config_change`` - # key (see DEFAULT_CONFIG). Read it from the config we just parsed - # so the user can flip it in the same edit that changes mcp_servers; - # missing key means default-on. - _mcp_cfg = new_cfg.get("mcp") - _auto = ( - _mcp_cfg.get("auto_reload_on_config_change", True) - if isinstance(_mcp_cfg, dict) - else True - ) - - self._config_mcp_servers = new_mcp - - if not _auto: - # Notify the user that the config changed but do NOT auto-reload. - # They can apply the new settings on their own terms with - # /reload-mcp — which we explicitly warn may invalidate the cache. - print() - print("🔄 MCP server config changed — reload skipped (auto-reload disabled).") - print(" New settings are NOT applied yet. To apply them now, run:") - print(" /reload-mcp") - print(" ⚠️ Note: /reload-mcp rebuilds the tool set and invalidates the") - print(" provider prompt cache (next message re-sends full input tokens).") - return - - # Notify user and reload. Run in a separate thread with a hard - # timeout so a hung MCP server cannot block the process_loop - # indefinitely (which would freeze the entire TUI). - print() - print("🔄 MCP server config changed — reloading connections...") - _reload_thread = threading.Thread( - target=self._reload_mcp, daemon=True - ) - _reload_thread.start() # Do NOT join here — process_loop calls this from its idle branch, so a # blocking join would freeze input consumption for up to 30s (and a hung # MCP server could block far longer). The reload runs purely in the @@ -14455,1302 +6577,18 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # in config. (Native Windows now drives the modal normally — see #33961.) _DESTRUCTIVE_SKIP_TOKENS = frozenset({"now", "--yes", "-y"}) - @classmethod - def _split_destructive_skip(cls, cmd_text: Optional[str]) -> tuple[str, bool]: - """Split inline-skip tokens out of a destructive slash command. - - Returns ``(remainder, skip)`` where ``remainder`` is the original - text with the command word and any recognized skip tokens removed, - and ``skip`` is True iff at least one skip token was found. - - Examples: - "/reset now" -> ("", True) - "/reset --yes My title" -> ("My title", True) - "/new My title" -> ("My title", False) - "/clear" -> ("", False) - """ - if not cmd_text: - return "", False - tokens = cmd_text.strip().split() - if not tokens: - return "", False - # Drop leading "/cmd" word — callers pass the full command text. - if tokens[0].startswith("/"): - tokens = tokens[1:] - skip = False - kept: list[str] = [] - for tok in tokens: - if tok.lower() in cls._DESTRUCTIVE_SKIP_TOKENS: - skip = True - continue - kept.append(tok) - return " ".join(kept), skip - - def _confirm_destructive_slash( - self, - command: str, - detail: str, - cmd_original: Optional[str] = None, - ) -> Optional[str]: - """Prompt the user to confirm a destructive session slash command. - - Used by ``/clear``, ``/new``/``/reset``, and ``/undo`` before they - discard conversation state. Three-option prompt: - - 1. Approve Once — proceed this time only - 2. Always Approve — proceed and persist - ``approvals.destructive_slash_confirm: false`` so future - destructive commands run without confirmation - 3. Cancel — abort - - Gated by ``approvals.destructive_slash_confirm`` (default on). If the - gate is off the function returns ``"once"`` immediately without - prompting. - - Inline-skip: if ``cmd_original`` contains ``now``, ``--yes``, or - ``-y`` as an argument (e.g. ``/reset now``, ``/new --yes My title``), - the modal is bypassed and ``"once"`` is returned immediately. This is - an escape hatch for non-interactive use and for the degraded path where - the modal can't be marshaled onto the app loop (native Windows itself now - drives the modal normally — see #33961). Callers are responsible - for stripping the skip tokens from any remaining argument parsing - (see :meth:`_split_destructive_skip`). - - Returns ``"once"``, ``"always"``, or ``None`` (cancelled). Callers - proceed with the destructive action when the result is non-None. - """ - # Inline-skip escape hatch — works regardless of platform/modal state. - # See class-level _DESTRUCTIVE_SKIP_TOKENS for the accepted tokens. - if cmd_original: - _, _skip = self._split_destructive_skip(cmd_original) - if _skip: - return "once" - - # Gate check — respects prior "Always Approve" clicks. - try: - cfg = load_cli_config() - approvals = cfg.get("approvals") if isinstance(cfg, dict) else None - confirm_required = True - if isinstance(approvals, dict): - confirm_required = bool(approvals.get("destructive_slash_confirm", True)) - except Exception: - confirm_required = True - - if not confirm_required: - return "once" - - # Render a prompt_toolkit-native confirmation panel. This keeps option - # labels visible above the composer and avoids raw input()/EOF races with - # the running TUI. - choices = [ - ("once", "Approve Once", "proceed this time only"), - ("always", "Always Approve", "proceed and silence this prompt permanently"), - ("cancel", "Cancel", "keep current conversation"), - ] - raw = self._prompt_text_input_modal( - title=f"⚠️ /{command} — destroys conversation state", - detail=detail, - choices=choices, - ) - if raw is None: - print(f"🟡 /{command} cancelled (no input).") - return None - choice = self._normalize_slash_confirm_choice(raw, choices) - if choice is None: - print(f"🟡 Unrecognized choice '{raw}'. /{command} cancelled.") - return None - - if choice == "cancel": - print(f"🟡 /{command} cancelled. Conversation unchanged.") - return None - - if choice == "always": - if save_config_value("approvals.destructive_slash_confirm", False): - print("🔒 Future /clear, /new, /reset, and /undo will run without confirmation.") - print(" Re-enable via `approvals.destructive_slash_confirm: true` in config.yaml.") - else: - print("⚠️ Couldn't persist opt-out — proceeding once.") - - return choice - - def _confirm_and_reload_mcp(self, cmd_original: str = "") -> None: - """Interactive /reload-mcp — confirm with the user, then reload. - - The auto-reload path (config file watcher) calls ``_reload_mcp`` - directly and never goes through this confirmation. - - Reloading MCP tools invalidates the provider prompt cache for the - active session (tool schemas are baked into the system prompt). - The next message re-sends full input tokens — can be expensive on - long-context or high-reasoning models. - - Three options: Approve Once, Always Approve (persists - ``approvals.mcp_reload_confirm: false`` so future reloads run - without this prompt), Cancel. Gated by - ``approvals.mcp_reload_confirm`` — default on. - """ - # Gate check — respects prior "Always Approve" clicks. - try: - cfg = load_cli_config() - approvals = cfg.get("approvals") if isinstance(cfg, dict) else None - confirm_required = True - if isinstance(approvals, dict): - confirm_required = bool(approvals.get("mcp_reload_confirm", True)) - except Exception: - confirm_required = True - - if not confirm_required: - with self._busy_command(self._slow_command_status(cmd_original)): - self._reload_mcp() - return - - # Render warning + prompt. Use the same prompt_toolkit-native composer - # modal as destructive slash confirmations so choices stay visible. - choices = [ - ("once", "Approve Once", "reload now"), - ("always", "Always Approve", "reload now and silence this prompt permanently"), - ("cancel", "Cancel", "leave MCP tools unchanged"), - ] - raw = self._prompt_text_input_modal( - title="⚠️ /reload-mcp — Prompt cache invalidation warning", - detail=( - "Reloading MCP servers rebuilds the tool set for this session and\n" - "invalidates the provider prompt cache. The next message will\n" - "re-send full input tokens (can be expensive on long-context or\n" - "high-reasoning models)." - ), - choices=choices, - ) - if raw is None: - print("🟡 /reload-mcp cancelled (no input).") - return - choice = self._normalize_slash_confirm_choice(raw, choices) - if choice is None: - print(f"🟡 Unrecognized choice '{raw}'. /reload-mcp cancelled.") - return - - if choice == "cancel": - print("🟡 /reload-mcp cancelled. MCP tools unchanged.") - return - - if choice == "always": - if save_config_value("approvals.mcp_reload_confirm", False): - print("🔒 Future /reload-mcp calls will run without confirmation.") - print(" Re-enable via `approvals.mcp_reload_confirm: true` in config.yaml.") - else: - print("⚠️ Couldn't persist opt-out — reloading once.") - - with self._busy_command(self._slow_command_status(cmd_original)): - self._reload_mcp() - - def _reload_mcp(self): - """Reload MCP servers: disconnect all, re-read config.yaml, reconnect. - - After reconnecting, refreshes the agent's tool list so the model - sees the updated tools on the next turn. - """ - try: - from tools.mcp_tool import ( - shutdown_mcp_servers, discover_mcp_tools, reprobe_tool_availability, _servers, _lock, - ) - - # Capture old server names - with _lock: - old_servers = set(_servers.keys()) - - if not self._command_running: - print("🔄 Reloading MCP servers...") - - # Shutdown existing connections - shutdown_mcp_servers() - - # Explicit reload also re-probes tool availability (check_fn). - reprobe_tool_availability() - # Reconnect (reads config.yaml fresh) - new_tools = discover_mcp_tools() - - # Compute what changed - with _lock: - connected_servers = set(_servers.keys()) - - added = connected_servers - old_servers - removed = old_servers - connected_servers - reconnected = connected_servers & old_servers - - if reconnected: - print(f" ♻️ Reconnected: {', '.join(sorted(reconnected))}") - if added: - print(f" ➕ Added: {', '.join(sorted(added))}") - if removed: - print(f" ➖ Removed: {', '.join(sorted(removed))}") - if not connected_servers: - print(" No MCP servers connected.") - else: - print(f" 🔧 {len(new_tools)} tool(s) available from {len(connected_servers)} server(s)") - - # Refresh the agent's tool list so the model can call new tools. - # Route through the shared helper so this CLI /reload-mcp path stays - # in lockstep with the TUI RPC / gateway reload / late-binding paths - # (name-diff, thread-safe, and — critically — additive-preserving so - # memory-provider and context-engine tools survive the rebuild). - if self.agent is not None: - from tools.mcp_tool import refresh_agent_mcp_tools - # Explicit reload: pick up MCP servers the user ENABLED in config - # this session. self.enabled_toolsets was resolved once at - # startup; merge in any now-connected server names (unless the - # user pinned `all`/`*`, which already includes everything) so a - # freshly-added server isn't filtered out. Mirrors startup, where - # MCP server names are part of enabled_toolsets (see __init__). - enabled_override = None - et = self.enabled_toolsets - if et and "all" not in et and "*" not in et: - merged = list(et) - for _name in sorted(connected_servers): - if _name not in merged: - merged.append(_name) - enabled_override = merged - refresh_agent_mcp_tools( - self.agent, - enabled_override=enabled_override, - quiet_mode=True, - ) - # Keep the CLI's own list in sync with what the agent now uses. - if enabled_override is not None: - self.enabled_toolsets = enabled_override - - # Inject a message at the END of conversation history so the - # model knows tools changed. Appended after all existing - # messages to preserve prompt-cache for the prefix. - change_parts = [] - if added: - change_parts.append(f"Added servers: {', '.join(sorted(added))}") - if removed: - change_parts.append(f"Removed servers: {', '.join(sorted(removed))}") - if reconnected: - change_parts.append(f"Reconnected servers: {', '.join(sorted(reconnected))}") - tool_summary = f"{len(new_tools)} MCP tool(s) now available" if new_tools else "No MCP tools available" - change_detail = ". ".join(change_parts) + ". " if change_parts else "" - self.conversation_history.append({ - "role": "user", - "content": f"[IMPORTANT: MCP servers have been reloaded. {change_detail}{tool_summary}. The tool list for this conversation has been updated accordingly.]", - }) - - # Persist session immediately so the session log reflects the - # updated tools list (self.agent.tools was refreshed above). - if self.agent is not None: - try: - self.agent._persist_session( - self.conversation_history, - self.conversation_history, - ) - except Exception: - pass # Best-effort - - print(f" ✅ Agent updated — {len(self.agent.tools if self.agent else [])} tool(s) available") - - except Exception as e: - print(f" ❌ MCP reload failed: {e}") - - def _reload_skills(self) -> None: - """Reload skills: rescan ~/.hermes/skills/ and queue a note for the - next user turn. - - Skills don't need to live in the system prompt for the model to use - them (they're invoked via ``/skill-name``, ``skills_list``, or - ``skill_view`` at runtime), so this does NOT clear the prompt cache. - It rescans the slash-command map, prints the diff for the user, and - — if any skills were added or removed — queues a one-shot note that - gets prepended to the next user message. This preserves message - alternation (no phantom user turn injected out of band) and keeps - prompt caching intact. - """ - try: - from agent.skill_commands import reload_skills, get_skill_commands - - if not self._command_running: - print("🔄 Reloading skills...") - - result = reload_skills() - - # Sync cli.py's module-level _skill_commands so all consumers - # (help display, command dispatch, Tab-completion lambda) see the - # updated dict without needing to restart the session. - global _skill_commands - _skill_commands = get_skill_commands() - added = result.get("added", []) # [{"name", "description"}, ...] - removed = result.get("removed", []) # [{"name", "description"}, ...] - total = result.get("total", 0) - - if not added and not removed: - print(" No new skills detected.") - print(f" 📚 {total} skill(s) available") - return - - def _fmt_line(item: dict) -> str: - nm = item.get("name", "") - desc = item.get("description", "") - return f" - {nm}: {desc}" if desc else f" - {nm}" - - if added: - print(" ➕ Added Skills:") - for item in added: - print(f" {_fmt_line(item)}") - if removed: - print(" ➖ Removed Skills:") - for item in removed: - print(f" {_fmt_line(item)}") - print(f" 📚 {total} skill(s) available") - - # Queue a one-shot note for the NEXT user turn. The CLI's agent - # loop prepends ``_pending_skills_reload_note`` (if set) to the - # API-call-local message at ~L8770, then clears it — same - # pattern as ``_pending_model_switch_note``. Nothing is written - # to conversation_history here, so message alternation stays - # intact and no out-of-band user turn is persisted. - # - # Format matches how the system prompt renders pre-existing - # skills (`` - name: description``) so the model reads the - # diff in the same shape as its original skill catalog. - sections = ["[USER INITIATED SKILLS RELOAD:"] - if added: - sections.append("") - sections.append("Added Skills:") - for item in added: - sections.append(_fmt_line(item)) - if removed: - sections.append("") - sections.append("Removed Skills:") - for item in removed: - sections.append(_fmt_line(item)) - sections.append("") - sections.append("Use skills_list to see the updated catalog.]") - self._pending_skills_reload_note = "\n".join(sections) - - except Exception as e: - print(f" ❌ Skills reload failed: {e}") - # ==================================================================== # Tool-call generation indicator (shown during streaming) # ==================================================================== - def _on_tool_gen_start(self, tool_name: str) -> None: - """Called when the model begins generating tool-call arguments. - - Closes any open streaming boxes (reasoning / response) exactly once, - then prints a short status line so the user sees activity instead of - a frozen screen while a large payload (e.g. 45 KB write_file) streams. - """ - if getattr(self, "_stream_box_opened", False): - self._flush_stream() - self._stream_box_opened = False - self._close_reasoning_box() - - from agent.display import get_tool_emoji - emoji = get_tool_emoji(tool_name, default="⚡") - _cprint(f" ┊ {emoji} preparing {tool_name}…") - # ==================================================================== # Tool progress callback (audio cues for voice mode) # ==================================================================== - def _on_tool_progress(self, event_type: str, function_name: str = None, preview: str = None, function_args: dict = None, **kwargs): - """Called on tool lifecycle events (tool.started, tool.completed, reasoning.available, etc.). - - Updates the TUI spinner widget so the user can see what the agent - is doing during tool execution (fills the gap between thinking - spinner and next response). - - On tool.started, records a monotonic timestamp so get_spinner_text() - can show a live elapsed timer (the TUI poll loop already invalidates - every ~0.15s, so the counter updates automatically). - - When tool_progress_mode is "all" or "new", also prints a persistent - stacked line to scrollback on tool.completed so users can see the - full history of tool calls (not just the current one in the spinner). - """ - # MoA reference-model outputs: render each reference's answer as a - # labelled thinking-style block BEFORE the aggregator acts, so the user - # sees the mixture-of-agents process instead of a silent pause. These - # are display-only events emitted by the MoA facade (agent_init relay); - # they never enter message history. - if event_type == "moa.reference": - label = function_name or "reference" - text = preview or "" - idx = kwargs.get("moa_index") - count = kwargs.get("moa_count") - header = f"Reference {idx}/{count} — {label}" if idx and count else f"Reference — {label}" - try: - self._flush_reasoning_preview(force=True) - except Exception: - pass - _cprint(f" {_DIM}┊ ◇ {header}{_RST}") - try: - self._emit_reasoning_preview(text) - except Exception: - # Fallback: print the raw text dimmed if the preview helper fails. - if text.strip(): - _cprint(f" {_DIM}{text.strip()}{_RST}") - self._invalidate() - return - if event_type == "moa.aggregating": - agg = function_name or "" - self._spinner_text = f"◆ aggregating ({agg})" if agg else "◆ aggregating" - self._invalidate() - return - - # Feed the pet: tools mean "running" (not reasoning); a failed tool - # latches the turn so it ends on a sulk. - if event_type == "tool.started": - self._pet_reasoning = False - elif event_type == "tool.completed" and kwargs.get("is_error"): - self._pet_turn_error = True - elif event_type and event_type.startswith("reasoning"): - self._pet_reasoning = True - - if event_type == "tool.completed": - self._tool_start_time = 0.0 - # Per-turn accounting: this feed already sees every tool call with - # its result, so the summary line needs no agent-loop state. - self._turn_summary_record( - function_name, kwargs.get("result"), kwargs.get("is_error", False) - ) - # Focus view: count the scrollback line we are NOT printing, so the - # post-turn recovery line can report how much was hidden. Counted - # against the pre-focus tool-progress mode, so a user who already - # had /verbose off is never told focus hid something it didn't. - if getattr(self, "_focus_view_enabled", False): - try: - self._note_focus_hidden_line(function_name or "") - except Exception: - pass - # Print stacked scrollback line for "new" / "all" / "verbose" modes. - # "verbose" was previously omitted here, so non-streaming model - # calls (MoA aggregator, copilot-acp) rendered each tool only into - # the transient spinner line — which overwrites itself, so no - # scrollable tool history accumulated. Streaming models hid the bug - # because _on_tool_gen_start commits a "preparing" line per tool; - # non-streaming calls never emit that, leaving verbose mode with no - # committed line at all. "verbose" is strictly more than "all", so - # it must commit at least the same line. - if function_name and self.tool_progress_mode in {"new", "all", "verbose"}: - duration = kwargs.get("duration", 0.0) - # Pop stored args from tool.started for this function - stored = self._pending_tool_info.get(function_name) - stored_args = stored.pop(0) if stored else {} - if stored is not None and not stored: - del self._pending_tool_info[function_name] - # "new" mode: skip consecutive repeats of the same tool - if self.tool_progress_mode == "new" and function_name == self._last_scrollback_tool: - self._invalidate() - return - self._last_scrollback_tool = function_name - try: - from agent.display import get_cute_tool_message - line = get_cute_tool_message(function_name, stored_args, duration, result=kwargs.get("result")) - _cprint(f" {line}") - except Exception: - pass - # First-touch onboarding: on the first tool in this process - # that takes longer than the threshold while we're in the - # noisiest progress mode, print a one-time hint about - # /verbose. Latched on self so it fires at most once per - # process; persisted to config.yaml so it never fires again - # across processes either. - try: - if ( - not getattr(self, "_long_tool_hint_fired", False) - and self.tool_progress_mode == "all" - and duration >= 30.0 - ): - from agent.onboarding import ( - TOOL_PROGRESS_FLAG, - is_seen, - mark_seen, - tool_progress_hint_cli, - ) - if not is_seen(CLI_CONFIG, TOOL_PROGRESS_FLAG): - self._long_tool_hint_fired = True - _cprint(f" {_DIM}{tool_progress_hint_cli()}{_RST}") - mark_seen(_hermes_home / "config.yaml", TOOL_PROGRESS_FLAG) - CLI_CONFIG.setdefault("onboarding", {}).setdefault("seen", {})[TOOL_PROGRESS_FLAG] = True - except Exception: - pass - self._invalidate() - return - if event_type != "tool.started": - return - if function_name and not function_name.startswith("_"): - from agent.display import get_tool_emoji - emoji = get_tool_emoji(function_name) - label = preview or function_name - from agent.display import get_tool_preview_max_len - _pl = get_tool_preview_max_len() - if _pl > 0 and len(label) > _pl: - label = label[:_pl - 3] + "..." - self._spinner_text = f"{emoji} {label}" - self._tool_start_time = time.monotonic() - # Store args for stacked scrollback line on completion - self._pending_tool_info.setdefault(function_name, []).append( - function_args if function_args is not None else {} - ) - self._invalidate() - - def _on_tool_start(self, tool_call_id: str, function_name: str, function_args: dict): - """Capture local before-state for write-capable tools.""" - try: - from agent.display import capture_local_edit_snapshot - - snapshot = capture_local_edit_snapshot(function_name, function_args) - if snapshot is not None: - self._pending_edit_snapshots[tool_call_id] = snapshot - except Exception: - logger.debug("Edit snapshot capture failed for %s", function_name, exc_info=True) - - def _on_tool_complete(self, tool_call_id: str, function_name: str, function_args: dict, function_result: str): - """Render file edits with inline diff after write-capable tools complete.""" - # A top-level delegate_task dispatches in the background and re-enters as - # a fresh turn when done. Say so once — no spinner, nothing to poll — so - # the idle prompt doesn't read as "nothing happened" (⛓ tracks the work). - if function_name == "delegate_task": - try: - parsed = json.loads(function_result) if isinstance(function_result, str) else (function_result or {}) - except Exception: - parsed = {} - if isinstance(parsed, dict) and parsed.get("status") == "dispatched" and parsed.get("mode") == "background": - n = parsed.get("count") or 1 - noun, tail = ("task", "it finishes") if n == 1 else (f"{n} tasks", "they finish") - try: - _cprint(f"\033[2m\u21a9 Background {noun} running — I'll resume when {tail}. Keep chatting.\033[0m") - except Exception: - pass - snapshot = self._pending_edit_snapshots.pop(tool_call_id, None) - try: - from agent.display import render_edit_diff_with_delta - - render_edit_diff_with_delta( - function_name, - function_result, - function_args=function_args, - snapshot=snapshot, - print_fn=_cprint, - ) - except Exception: - logger.debug("Edit diff preview failed for %s", function_name, exc_info=True) - # ==================================================================== # Voice mode methods # ==================================================================== - def _voice_start_recording(self): - """Start capturing audio from the microphone.""" - if getattr(self, '_should_exit', False): - return - from tools.voice_mode import create_audio_recorder, check_voice_requirements - - reqs = check_voice_requirements() - if not reqs["audio_available"]: - if _is_termux_environment(): - details = reqs.get("details", "") - if "Termux:API Android app is not installed" in details: - raise RuntimeError( - "Termux:API command package detected, but the Android app is missing.\n" - "Install/update the Termux:API Android app, then retry /voice on.\n" - "Fallback: pkg install python-numpy portaudio && python -m pip install sounddevice" - ) - raise RuntimeError( - "Voice mode requires either Termux:API microphone access or Python audio libraries.\n" - "Option 1: pkg install termux-api and install the Termux:API Android app\n" - "Option 2: pkg install python-numpy portaudio && python -m pip install sounddevice" - ) - raise RuntimeError( - "Voice mode requires sounddevice and numpy.\n" - f"Install with: {sys.executable} -m pip install sounddevice numpy" - ) - if not reqs.get("stt_available", reqs.get("stt_key_set")): - raise RuntimeError( - "Voice mode requires an STT provider for transcription.\n" - "Option 1: uv pip install faster-whisper " - "(free, local; `pip install faster-whisper` also works if pip is on PATH)\n" - "Option 2: Set GROQ_API_KEY (free tier)\n" - "Option 3: Set VOICE_TOOLS_OPENAI_KEY (paid)" - ) - - # Prevent double-start from concurrent threads (atomic check-and-set) - with self._voice_lock: - if self._voice_recording: - return - self._voice_recording = True - - # Load silence detection params from config. Shape-safe: a - # hand-edited ``voice: true`` / ``voice: cmd+b`` leaves - # ``load_config()['voice']`` as a non-dict; coerce to {} so - # continuous recording falls back to the documented defaults - # instead of crashing on ``.get()``. - voice_cfg: dict = {} - try: - from hermes_cli.config import load_config - _cfg = load_config().get("voice") - voice_cfg = _cfg if isinstance(_cfg, dict) else {} - except Exception: - pass - - # Recorder creation can fail (no input device, PortAudio init error). - # Reset the flag on failure or _voice_recording stays True forever and - # every future voice start is silently skipped by the guard above. - if self._voice_recorder is None: - try: - self._voice_recorder = create_audio_recorder() - except Exception: - with self._voice_lock: - self._voice_recording = False - raise - - # Apply config-driven silence params (numeric-guarded so YAML - # scalar corruption doesn't break recording start-up). - # - # ``bool`` is explicitly excluded from the numeric check — in - # Python bool is a subclass of int, so a hand-edited - # ``silence_threshold: true`` would otherwise be forwarded as - # ``1`` instead of falling back to the 200 default (Copilot - # round-12 on #19835). - _threshold = voice_cfg.get("silence_threshold") - _duration = voice_cfg.get("silence_duration") - self._voice_recorder._silence_threshold = ( - _threshold if isinstance(_threshold, (int, float)) and not isinstance(_threshold, bool) else 200 - ) - self._voice_recorder._silence_duration = ( - _duration if isinstance(_duration, (int, float)) and not isinstance(_duration, bool) else 3.0 - ) - # voice.max_recording_seconds — hard cap on a single recording's length. - # Same numeric guard as the silence params (bool excluded: a hand-edited - # ``max_recording_seconds: true`` must not become ``1`` — it falls back - # to the documented 120 default, mirroring the silence-param handling). - # An explicit numeric value <= 0 disables the cap. Previously this - # documented key was never read (dead config); wiring it here makes it - # take effect. - _max_rec = voice_cfg.get("max_recording_seconds") - self._voice_recorder._max_recording_seconds = ( - (_max_rec if _max_rec > 0 else 0.0) - if isinstance(_max_rec, (int, float)) and not isinstance(_max_rec, bool) - else 120.0 - ) - - def _on_silence(): - """Called by AudioRecorder when silence is detected after speech.""" - with self._voice_lock: - if not self._voice_recording: - return - _cprint(f"\n{_DIM}Silence detected, auto-stopping...{_RST}") - if hasattr(self, '_app') and self._app: - self._app.invalidate() - self._voice_stop_and_transcribe() - - # Audio cue: single beep BEFORE starting stream (avoid CoreAudio conflict) - if self._voice_beeps_enabled(): - try: - from tools.voice_mode import play_beep - play_beep(frequency=880, count=1) - except Exception: - pass - - try: - self._voice_recorder.start(on_silence_stop=_on_silence) - except Exception: - with self._voice_lock: - self._voice_recording = False - raise - _label = self._voice_record_key_label() - if getattr(self._voice_recorder, "supports_silence_autostop", True): - _recording_hint = f"auto-stops on silence | {_label} to stop & exit continuous" - elif _is_termux_environment(): - _recording_hint = f"Termux:API capture | {_label} to stop" - else: - _recording_hint = f"{_label} to stop" - _cprint(f"\n{_ACCENT}● Recording...{_RST} {_DIM}({_recording_hint}){_RST}") - - # Periodically refresh prompt to update audio level indicator - def _refresh_level(): - while True: - with self._voice_lock: - still_recording = self._voice_recording - if not still_recording: - break - if hasattr(self, '_app') and self._app: - self._app.invalidate() - time.sleep(0.15) - threading.Thread(target=_refresh_level, daemon=True).start() - - def _voice_stt_model(self) -> Optional[str]: - """STT model override from config, or None for the provider default. - - For the local provider, prefer stt.local.model (default ``base``) so the - CLI passes a real model name into the local STT backend. - """ - try: - from hermes_cli.config import load_config - stt_config = load_config().get("stt", {}) - if not isinstance(stt_config, dict): - return None - provider = str(stt_config.get("provider") or "").strip().lower() - if provider == "local": - local_config = stt_config.get("local") or {} - if not isinstance(local_config, dict): - local_config = {} - return local_config.get("model") or "base" - return stt_config.get("model") - except Exception: - return None - - def _voice_stt_provider(self) -> str: - """Configured STT provider name (lowercased), or empty string.""" - try: - from hermes_cli.config import load_config - stt_config = load_config().get("stt", {}) - if not isinstance(stt_config, dict): - return "" - return str(stt_config.get("provider") or "").strip().lower() - except Exception: - return "" - - def _voice_restart_recording_async(self) -> None: - """Restart continuous-mode recording off-thread (start() can block).""" - def _restart_recording(): - try: - self._voice_start_recording() - if hasattr(self, '_app') and self._app: - self._app.invalidate() - except Exception as e: - _cprint(f"{_DIM}Voice auto-restart failed: {e}{_RST}") - threading.Thread(target=_restart_recording, daemon=True).start() - - def _voice_stop_and_transcribe(self): - """Stop recording, transcribe via STT, and queue the transcript as input.""" - # Atomic guard: only one thread can enter stop-and-transcribe. - # Set _voice_processing immediately so concurrent Ctrl+B presses - # don't race into the START path while recorder.stop() holds its lock. - with self._voice_lock: - if not self._voice_recording: - return - self._voice_recording = False - self._voice_processing = True - - submitted = False - transcription_failed = False - wav_path = None - try: - if self._voice_recorder is None: - return - - wav_path = self._voice_recorder.stop() - - # Audio cue: double beep after stream stopped (no CoreAudio conflict) - if self._voice_beeps_enabled(): - try: - from tools.voice_mode import play_beep - play_beep(frequency=660, count=2) - except Exception: - pass - - if wav_path is None: - _cprint(f"{_DIM}No speech detected.{_RST}") - return - - # _voice_processing is already True (set atomically above) - if hasattr(self, '_app') and self._app: - self._app.invalidate() - - stt_model = self._voice_stt_model() - if self._voice_stt_provider() == "local": - _cprint( - f"{_DIM}Preparing local STT model '{stt_model}' " - f"(first use may download it from Hugging Face)...{_RST}" - ) - else: - _cprint(f"{_DIM}Transcribing...{_RST}") - - from tools.voice_mode import transcribe_recording - result = transcribe_recording(wav_path, model=stt_model) - - if result.get("success") and result.get("transcript", "").strip(): - transcript = result["transcript"].strip() - from tools.voice_mode import is_voice_stop_phrase - if is_voice_stop_phrase(transcript): - # Bare "stop" (or configured phrase) ends the voice chat - # instead of being sent to the agent. - _cprint(f"{_DIM}Stop phrase detected — ending voice chat.{_RST}") - self._disable_voice_mode() - return - self._attached_images.clear() - if hasattr(self, '_app') and self._app: - self._app.invalidate() - self._pending_input.put(_VoiceInputMessage(transcript)) - submitted = True - elif result.get("success"): - _cprint(f"{_DIM}No speech detected.{_RST}") - else: - error = result.get("error", "Unknown error") - _cprint(f"\n{_DIM}Transcription failed: {error}{_RST}") - transcription_failed = True - - except Exception as e: - _cprint(f"\n{_DIM}Voice processing error: {e}{_RST}") - transcription_failed = wav_path is not None - finally: - with self._voice_lock: - self._voice_processing = False - if hasattr(self, '_app') and self._app: - self._app.invalidate() - # Clean up temp file unless transcription failed. On failure, keep - # the source recording so long dictation is not lost. - try: - if wav_path and os.path.isfile(wav_path): - if transcription_failed: - _cprint(f"{_DIM}Recording preserved at: {wav_path}{_RST}") - else: - os.unlink(wav_path) - except Exception: - pass - - # Track consecutive no-speech cycles to avoid infinite restart loops. - # While the agent is mid-turn or TTS is speaking, the user is - # CORRECTLY silent (waiting/listening) — those cycles must not - # count, or a multi-minute tool run ends the voice chat under - # the user. The stop phrase and barge-in still work during the - # hold (they run on their own paths above). - stop_continuous_restart = False - _tts_done = getattr(self, "_voice_tts_done", None) - _activity_hold = bool( - getattr(self, "_agent_running", False) - or (_tts_done is not None and not _tts_done.is_set()) - ) - if not submitted: - if _activity_hold: - pass # held: keep listening without counting the cycle - else: - self._no_speech_count = getattr(self, '_no_speech_count', 0) + 1 - if self._no_speech_count >= 3: - self._voice_continuous = False - self._no_speech_count = 0 - _cprint(f"{_DIM}No speech detected 3 times, continuous mode stopped.{_RST}") - stop_continuous_restart = True - else: - self._no_speech_count = 0 - - # If no transcript was submitted but continuous mode is active, - # restart recording so the user can keep talking. - # (When transcript IS submitted, process_loop handles restart - # after chat() completes.) - if ( - self._voice_continuous - and not submitted - and not self._voice_recording - and not stop_continuous_restart - ): - self._voice_restart_recording_async() - - def _voice_speak_response_async(self, text: str) -> None: - """Schedule TTS and mark it pending before continuous recording can restart.""" - if not self._voice_tts or not text: - return - self._voice_tts_done.clear() - threading.Thread( - target=self._voice_speak_response, - args=(text,), - daemon=True, - ).start() - # Spoken barge-in must work on the whole-file fallback path too. The - # full-duplex agent-turn listener normally already covers playback - # (armed at turn start in chat()); this arm is an idempotent safety - # net for speak calls outside a chat turn — the listener refuses to - # double-arm via _voice_fd_active. - if self._voice_continuous: - threading.Thread( - target=self._voice_full_duplex_listener, - daemon=True, - ).start() - - def _voice_speak_response(self, text: str): - """Speak the agent's response aloud using TTS (runs in background thread).""" - if not self._voice_tts: - return - self._voice_tts_done.clear() - try: - from tools.tts_tool import text_to_speech_tool - from tools.voice_mode import play_audio_file - - # Strip markdown and non-speech content for cleaner TTS via the - # shared cleaner (tools/tts_text_normalize): markdown, emoji, - # ⋗ blocks, verifier footer, units, newline flattening. - # The TTS tool owns provider request limits and long-form chunking. - try: - from tools.tts_text_normalize import prepare_spoken_text - tts_text = prepare_spoken_text(text, max_chars=None) - except Exception: - # Legacy fallback pipeline — keep voice replies best-effort. - tts_text = re.sub(r'```[\s\S]*?```', ' ', text) # fenced code blocks - tts_text = re.sub(r'\[([^\]]+)\]\([^)]+\)', r'\1', tts_text) # [text](url) -> text - tts_text = re.sub(r'https?://\S+', '', tts_text) # URLs - tts_text = re.sub(r'\*\*(.+?)\*\*', r'\1', tts_text) # bold - tts_text = re.sub(r'\*(.+?)\*', r'\1', tts_text) # italic - tts_text = re.sub(r'`(.+?)`', r'\1', tts_text) # inline code - tts_text = re.sub(r'^#+\s*', '', tts_text, flags=re.MULTILINE) # headers - tts_text = re.sub(r'^\s*[-*]\s+', '', tts_text, flags=re.MULTILINE) # list items - tts_text = re.sub(r'---+', '', tts_text) # horizontal rules - tts_text = re.sub(r'\n{3,}', '\n\n', tts_text) # excessive newlines - tts_text = tts_text.strip() - if not tts_text: - return - self._voice_last_tts_text = tts_text - - # Use MP3 output for CLI playback (afplay doesn't handle OGG well). - # The TTS tool may auto-convert MP3->OGG, but the original MP3 remains. - os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True) - mp3_path = os.path.join( - tempfile.gettempdir(), "hermes_voice", - f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3", - ) - - raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path) - try: - tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {} - except Exception: - tts_result = {} - - # The tool result is authoritative — it may return multiple files - # for long-form chunked output. Play each in order. - play_paths = tts_result.get("file_paths") or [ - tts_result.get("file_path") or mp3_path - ] - for play_path in play_paths if tts_result.get("success") else []: - if os.path.isfile(play_path) and os.path.getsize(play_path) > 0: - play_audio_file(play_path) - # Clean up all generated files (play_paths + mp3_path + ogg variants) - cleanup_paths = set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]) - for path in cleanup_paths: - if os.path.isfile(path): - try: - os.unlink(path) - except OSError: - pass - except Exception as e: - logger.warning("Voice TTS playback failed: %s", e) - _cprint(f"{_DIM}TTS playback failed: {e}{_RST}") - finally: - self._voice_tts_done.set() - - - def _voice_full_duplex_listener(self) -> None: - """Full-duplex agent-turn listener: mic live for the WHOLE turn. - - Armed at utterance-submit (chat() start in continuous voice mode) and - disarmed when the turn is fully done (agent finished + TTS played). - Replaces the old per-playback ``_voice_barge_in_monitor``, which only - listened while TTS audio was playing — during LLM generation the mic - was dead, so the user could not interject by voice at all (and the - playback monitor calibrated against its own speaker bleed, making - the trigger unreachable; see tools.voice_mode.full_duplex_listen). - - Phase behaviour: - - * generation (no TTS audio yet): speech interrupts the in-flight - agent turn via ``self.agent.interrupt()`` — the same seam the - typed/Ctrl+C interrupt uses — and the captured utterance is - submitted as the next message. - * playback: speech cuts TTS (pipeline stop event + stop_playback) - and the interruption is captured with pre-roll and submitted. - - The stop phrase ends the voice chat in BOTH phases (a stop during - generation means "stop everything": the turn is already interrupted - at trip time, then ``_voice_submit_barge_utterance`` disables voice - mode). - """ - fd_active = getattr(self, "_voice_fd_active", None) - if fd_active is None: - fd_active = threading.Event() - self._voice_fd_active = fd_active - if fd_active.is_set(): - return # one listener owns the mic for this turn - fd_active.set() - try: - from hermes_cli.config import load_config - voice_cfg = load_config().get("voice") or {} - if not (isinstance(voice_cfg, dict) and voice_cfg.get("barge_in", True)): - return - from tools.voice_mode import ( - full_duplex_listen, - is_audio_output_active, - stop_playback, - ) - - try: - _mult = float(voice_cfg.get("barge_in_threshold_multiplier", 0) or 0) - except (TypeError, ValueError): - _mult = 0.0 - try: - _grace_ms = int(float(voice_cfg.get("barge_in_grace_seconds", 0.5)) * 1000) - except (TypeError, ValueError): - _grace_ms = 500 - - tts_done = getattr(self, "_voice_tts_done", None) - - def _should_stop() -> bool: - if not (getattr(self, "_voice_mode", False) and getattr(self, "_voice_continuous", False)): - return True - if getattr(self, "_agent_running", False): - return False - # Agent finished — keep listening until TTS fully played. - if tts_done is not None and not tts_done.is_set(): - return False - return not is_audio_output_active() - - def _on_trigger(phase: str) -> None: - # Latch BEFORE cutting anything: suppresses process_loop's - # auto-restart until the capture is submitted. - self._voice_barge_capture.set() - self._voice_barge_phase = phase - if phase == "playback": - logger.debug( - "TTS CUT: full-duplex listener tripped during playback" - ) - from tools.tts_streaming import mark_speech_interrupted - mark_speech_interrupted() - _pipe_stop = getattr(self, "_voice_tts_stop", None) - if _pipe_stop is not None: - _pipe_stop.set() - stop_playback() - else: - # Generation phase: no audio to cut — interrupt the - # in-flight agent turn (same seam as typed interrupt). - logger.debug( - "full-duplex listener tripped during generation — " - "interrupting agent turn" - ) - _pipe_stop = getattr(self, "_voice_tts_stop", None) - if _pipe_stop is not None: - _pipe_stop.set() # never let the stale reply speak - try: - if self.agent is not None and getattr(self, "_agent_running", False): - _cprint(f"\n{_DIM}🎤 Voice interjection — interrupting…{_RST}") - self.agent.interrupt() - except Exception as e: - logger.debug("voice interjection interrupt failed: %s", e) - - wav_path = full_duplex_listen( - _should_stop, - is_playing=is_audio_output_active, - on_trigger=_on_trigger, - multiplier=_mult or None, - grace_ms=max(0, _grace_ms), - ) - if wav_path and self._voice_barge_capture.is_set(): - self._voice_submit_barge_utterance(wav_path) - else: - self._voice_barge_capture.clear() - except Exception as e: - self._voice_barge_capture.clear() - logger.debug("Voice full-duplex listener failed: %s", e) - finally: - fd_active.clear() - - def _voice_submit_barge_utterance(self, wav_path: str) -> None: - """Transcribe a barge-captured interruption and queue it as the next turn.""" - submitted = False - try: - from tools.voice_mode import transcribe_recording - result = transcribe_recording(wav_path, model=self._voice_stt_model()) - transcript = (result.get("transcript") or "").strip() if result.get("success") else "" - if transcript: - from tools.voice_mode import is_voice_stop_phrase - if is_voice_stop_phrase(transcript): - _cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}") - self._disable_voice_mode() - return - # Fail-closed echo guard (#75780): a playback-phase capture - # has no acoustic echo cancellation, so speaker bleed alone - # can trip the barge trigger. If the transcript is a close - # match for what Hermes just spoke, treat it as self-capture - # instead of queuing it as a user turn. - if getattr(self, "_voice_barge_phase", None) == "playback": - from tools.voice_mode import is_tts_echo - if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")): - logger.debug( - "Dropping playback-phase barge transcript as TTS echo: %r", - transcript, - ) - _cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}") - return - self._pending_input.put(_VoiceInputMessage(transcript)) - submitted = True - elif not result.get("success"): - _cprint(f"\n{_DIM}Transcription failed: {result.get('error', 'Unknown error')}{_RST}") - except Exception as e: - _cprint(f"\n{_DIM}Voice processing error: {e}{_RST}") - finally: - try: - if os.path.isfile(wav_path): - os.unlink(wav_path) - except OSError: - pass - self._voice_barge_capture.clear() - self._voice_barge_phase = None - # No usable transcript: hand the mic back to the normal loop. - if not submitted and self._voice_mode and self._voice_continuous and not self._voice_recording: - self._voice_restart_recording_async() - - def _voice_beeps_enabled(self) -> bool: - """Return whether CLI voice mode should play record start/stop beeps.""" - try: - from hermes_cli.config import load_config - from utils import is_truthy_value - voice_cfg = load_config().get("voice", {}) - if isinstance(voice_cfg, dict): - # is_truthy_value handles quoted YAML strings like "false" - # which bool() would misread as True (#49883). - return is_truthy_value(voice_cfg.get("beep_enabled", True), default=True) - except Exception: - pass - return True - - def _enable_voice_mode(self): - """Enable voice mode after checking requirements.""" - if self._voice_mode: - _cprint(f"{_DIM}Voice mode is already enabled.{_RST}") - return - - from tools.voice_mode import check_voice_requirements, detect_audio_environment - - # Environment detection -- warn and block in incompatible environments - env_check = detect_audio_environment() - if not env_check["available"]: - _cprint(f"\n{_ACCENT}Voice mode unavailable in this environment:{_RST}") - for warning in env_check["warnings"]: - _cprint(f" {_DIM}{warning}{_RST}") - return - - reqs = check_voice_requirements() - if not reqs["available"]: - _cprint(f"\n{_ACCENT}Voice mode requirements not met:{_RST}") - for line in reqs["details"].split("\n"): - _cprint(f" {_DIM}{line}{_RST}") - if reqs["missing_packages"]: - if _is_termux_environment(): - _cprint(f"\n {_BOLD}Option 1: pkg install termux-api{_RST}") - _cprint(f" {_DIM}Then install/update the Termux:API Android app for microphone capture{_RST}") - _cprint(f" {_BOLD}Option 2: pkg install python-numpy portaudio && python -m pip install sounddevice{_RST}") - else: - _cprint(f"\n {_BOLD}Install: {sys.executable} -m pip install {' '.join(reqs['missing_packages'])}{_RST}") - return - - with self._voice_lock: - self._voice_mode = True - - # Check config for auto_tts (shape-safe — malformed ``voice:`` YAML - # leaves ``voice_config`` as a non-dict, so guard before .get()). - try: - from hermes_cli.config import load_config - _raw_voice = load_config().get("voice") - voice_config = _raw_voice if isinstance(_raw_voice, dict) else {} - if voice_config.get("auto_tts", False): - with self._voice_lock: - self._voice_tts = True - except Exception: - pass - - # Voice mode instruction is injected as a user message prefix (not a - # system prompt change) to avoid invalidating the prompt cache. See - # _voice_message_prefix property and its usage in _process_message(). - - tts_status = " (TTS enabled)" if self._voice_tts else "" - if self._voice_tts: - # Speech output is on from the start — warm the engine now so the - # first spoken reply doesn't pay the model load as dead air. - self._tts_lease_async(True) - # Use the startup-pinned cache so the advertised shortcut always - # matches the live prompt_toolkit binding — reading live config - # here would drift after a mid-session config edit (Copilot - # round-14 on #19835, same class as round-13). - _ptt_display = self._voice_record_key_label() - _cprint(f"\n{_ACCENT}Voice mode enabled{tts_status}{_RST}") - _cprint(f" {_DIM}{_ptt_display} to start/stop recording{_RST}") - # Spoken-stop hint sourced from voice.stop_phrases (first entry); the - # helper returns "" when stop phrases are disabled — show no hint then. - try: - from tools.voice_mode import voice_stop_hint - _stop_hint = voice_stop_hint() - except Exception: - _stop_hint = "" - if _stop_hint: - _cprint(f" {_DIM}{_stop_hint}{_RST}") - _cprint(f" {_DIM}/voice tts to toggle speech output{_RST}") - _cprint(f" {_DIM}/voice off to disable voice mode{_RST}") - - def _typed_voice_stop(self, user_input) -> bool: - """Typed bare stop phrase during an active voice chat ends the chat. - - Saying "stop" ends the voice chat (PR #73106); TYPING the same bare - stop phrase while voice mode is on must behave identically instead of - sending "stop" to the agent as a turn. Guarded on voice mode being ON - — typed "stop" outside voice chat passes through to the agent exactly - as before. Reuses ``is_voice_stop_phrase`` (same config - ``voice.stop_phrases``, same exact-match semantics), so longer typed - messages containing "stop" are never swallowed. - """ - if not isinstance(user_input, str): - return False - with self._voice_lock: - voice_on = self._voice_mode or self._voice_continuous - if not voice_on: - return False - try: - from tools.voice_mode import is_voice_stop_phrase - if not is_voice_stop_phrase(user_input): - return False - except Exception: - return False - _cprint(f"\n{_DIM}Stop phrase typed — ending voice chat.{_RST}") - self._disable_voice_mode() - return True - - def _disable_voice_mode(self): - """Disable voice mode, cancel any active recording, and stop TTS.""" - recorder = None - with self._voice_lock: - if self._voice_recording and self._voice_recorder: - self._voice_recorder.cancel() - self._voice_recording = False - recorder = self._voice_recorder - self._voice_mode = False - self._voice_tts = False - self._voice_continuous = False - - # Speech output is off with the mode — release the TTS engine lease so - # a resident local model (piper/kittentts) is freed once nothing else - # in this process still needs it. - self._tts_lease_async(False) - - # Shut down the persistent audio stream in background - if recorder is not None: - def _bg_shutdown(rec=recorder): - try: - rec.shutdown() - except Exception: - pass - threading.Thread(target=_bg_shutdown, daemon=True).start() - self._voice_recorder = None - - # Stop any active TTS playback (file player + streaming pipeline) - try: - if self._voice_tts_stop is not None: - logger.info("TTS CUT: _disable_voice_mode setting stop event") - self._voice_tts_stop.set() - from tools.voice_mode import stop_playback - stop_playback() - except Exception: - pass - self._voice_tts_done.set() - - _cprint(f"\n{_DIM}Voice mode disabled.{_RST}") - # ── Wake word ("Hey Hermes") ───────────────────────────────────────── # # An always-on hotword listener (tools/wake_word.py) that, on detecting @@ -15764,1038 +6602,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # path (transcript submitted, no speech, or transcription error) without # threading resume logic through the voice machinery. - def _maybe_start_wake_word(self): - """Start the wake-word listener at CLI startup if this surface is eligible.""" - try: - from tools.wake_word import wake_surface_enabled - if not wake_surface_enabled("cli"): - return - except Exception: - return - self._start_wake_word_listener(announce=True) - - def _start_wake_word_listener(self, announce: bool = False) -> bool: - """Build + start the hotword detector. Returns True on success.""" - try: - from tools.wake_word import ( - check_wake_word_requirements, - load_wake_word_config, - owns_listener, - start_listening, - ) - except Exception as e: - if announce: - _cprint(f"{_DIM}Wake word unavailable: {e}{_RST}") - return False - - if getattr(self, "_wake_word_active", False) and owns_listener(self): - if announce: - _cprint(f"{_DIM}Wake word is already listening.{_RST}") - return True - self._wake_word_active = False - - cfg = load_wake_word_config() - reqs = check_wake_word_requirements(cfg) - if not reqs["available"]: - if announce: - _cprint(f"\n{_ACCENT}Wake word requirements not met:{_RST}") - if reqs.get("hint"): - _cprint(f" {_DIM}{reqs['hint']}{_RST}") - return False - - if announce and not reqs.get("deps_available", True): - # Fresh install: the engine constructor lazy-installs its deps - # (onnxruntime is a large wheel) — tell the user why this is slow. - _cprint(f"{_DIM}Installing wake word engine (first use — this may take a minute)...{_RST}") - - self._wake_start_new_session = bool(cfg.get("start_new_session", True)) - try: - start_listening(self._on_wake_word, owner=self, config=cfg) - except Exception as e: - if announce: - _cprint(f"\n{_DIM}Failed to start wake word: {e}{_RST}") - return False - - self._wake_word_active = True - self._wake_suspended = False - global _cli_wake_owner - _cli_wake_owner = self - self._start_wake_watchdog() - if announce: - _cprint(f"\n{_ACCENT}Wake word listening{_RST} " - f"{_DIM}(say \"{reqs['phrase']}\" — /wake off to stop){_RST}") - return True - - def _stop_wake_word_listener(self, announce: bool = False): - """Stop and tear down the hotword detector.""" - global _cli_wake_owner - was_active = getattr(self, "_wake_word_active", False) - self._wake_word_active = False - self._wake_suspended = False - try: - from tools.wake_word import stop_listening - stop_listening(owner=self) - except Exception: - pass - if _cli_wake_owner is self: - _cli_wake_owner = None - if announce: - if was_active: - _cprint(f"{_DIM}Wake word stopped.{_RST}") - else: - _cprint(f"{_DIM}Wake word is not running.{_RST}") - - def _on_wake_word(self): - """Fired after the detector hears the wake phrase.""" - if getattr(self, "_should_exit", False): - return - # Ignore wake while a turn is in flight or the mic is already in use. - if self._agent_running or self._voice_recording or getattr(self, "_voice_processing", False): - return - - # Release the mic so STT can capture the command utterance. - try: - from tools.wake_word import pause_listening - if not pause_listening(owner=self): - self._wake_word_active = False - return - except Exception as e: - logger.debug("wake word pause failed: %s", e) - return - self._wake_suspended = True - - # Multi-profile routing: the CLI is a single-profile process, so a - # phrase enrolled by ANOTHER profile can't be routed here — print the - # switch command and re-arm rather than answering as the wrong profile. - try: - from tools.wake_word import get_last_match - _match = get_last_match() - except Exception: - _match = None - if _match and _match[1]: - from tools.wake_word import _active_profile_name - if _match[1] != _active_profile_name(): - _cprint(f"\n{_DIM}Wake phrase for profile '{_match[1]}' — " - f"run: hermes -p {_match[1]}{_RST}") - self._wake_suspended = True # watchdog resumes the listener - return - - _cprint(f"\n{_ACCENT}✦ Wake word detected — listening...{_RST}") - if getattr(self, "_app", None): - try: - self._app.invalidate() - except Exception: - pass - - if getattr(self, "_wake_start_new_session", True): - try: - self.new_session(silent=True) - except Exception as e: - logger.debug("wake word new_session failed: %s", e) - - # Single-utterance capture (not continuous) via the voice pipeline; - # VAD auto-stop transcribes and queues the transcript for process_loop. - with self._voice_lock: - self._voice_mode = True - self._voice_continuous = False - try: - self._voice_start_recording() - except Exception as e: - _cprint(f"{_DIM}Wake capture failed: {e}{_RST}") # Leave _wake_suspended set; the watchdog resumes once idle. - def _start_wake_watchdog(self): - """Resume the paused detector when the CLI returns to a stable idle.""" - if getattr(self, "_wake_watchdog_started", False): - return - self._wake_watchdog_started = True - - def _loop(): - idle_polls = 0 - try: - while getattr(self, "_wake_word_active", False) and not getattr(self, "_should_exit", False): - time.sleep(0.25) - if not getattr(self, "_wake_suspended", False): - idle_polls = 0 - continue - busy = ( - self._agent_running - or self._voice_recording - or getattr(self, "_voice_processing", False) - or not self._pending_input.empty() - ) - if busy: - idle_polls = 0 - continue - # Require a few consecutive idle polls (~0.75s) so we don't - # resume in the gap between VAD stop and the agent starting. - idle_polls += 1 - if idle_polls >= 3: - idle_polls = 0 - try: - from tools.wake_word import resume_listening - if resume_listening(owner=self): - self._wake_suspended = False - else: - self._wake_word_active = False - except Exception as e: - logger.debug("wake word resume failed: %s", e) - finally: - self._wake_watchdog_started = False - - threading.Thread(target=_loop, daemon=True, name="wake-watchdog").start() - - def _show_wake_word_status(self): - """Show current wake-word listener status.""" - from tools.wake_word import ( - audio_is_silent, - check_wake_word_requirements, - is_listening, - load_wake_word_config, - owns_listener, - ) - - cfg = load_wake_word_config() - reqs = check_wake_word_requirements(cfg) - owned = owns_listener(self) - state = "LISTENING" if owned and is_listening() else "PAUSED" if owned else "OFF" - - _cprint(f"\n{_BOLD}Wake Word Status{_RST}") - _cprint(f" State: {state}") - _cprint(f" Phrase: \"{reqs['phrase']}\"") - _cprint(f" Provider: {reqs['provider']}") - _cprint(f" Surface: {cfg.get('surface', 'auto')}") - _cprint(f" New session: {'yes' if cfg.get('start_new_session', True) else 'no'}") - if state == "LISTENING" and audio_is_silent(): - _cprint(f" {_ACCENT}⚠ Microphone delivers only silence — the listener can't hear anything.{_RST}") - _cprint(f" {_DIM}On macOS: System Settings > Privacy & Security > Microphone — allow your" - f" terminal/Hermes, then /wake off + /wake on.{_RST}") - if not reqs["available"] and reqs.get("hint"): - _cprint(f" {_DIM}{reqs['hint']}{_RST}") - if not owned: - _cprint(f" {_DIM}Enable with /wake on{_RST}") - - def _tts_lease_async(self, active: bool) -> None: - """Acquire/release this CLI's TTS engine lease in the background. - - The /voice tts toggle (and voice-mode on/off with speech output set) - is the "TTS is about to be needed / no longer needed" signal: - acquiring pre-loads the configured provider so the first reply starts - hot; releasing lets the last-holder path unload resident local models. - Never blocks the toggle and never fails it. - """ - - def _run(): - try: - from tools.tts_tool import acquire_tts_lease, release_tts_lease - - if active: - acquire_tts_lease("cli:voice-tts") - else: - release_tts_lease("cli:voice-tts") - except Exception as e: - logger.debug("voice: tts lease active=%s failed: %s", active, e) - - threading.Thread(target=_run, name="tts-lease-cli", daemon=True).start() - - def _toggle_voice_tts(self): - """Toggle TTS output for voice mode.""" - if not self._voice_mode: - _cprint(f"{_DIM}Enable voice mode first: /voice on{_RST}") - return - - with self._voice_lock: - self._voice_tts = not self._voice_tts - status = "enabled" if self._voice_tts else "disabled" - - if self._voice_tts: - from tools.tts_tool import check_tts_requirements - if not check_tts_requirements(): - _cprint(f"{_DIM}Warning: No TTS provider available. Install edge-tts or set API keys.{_RST}") - - # Toggle = warm-up / release signal for the TTS engine (see - # tools.tts_tool.acquire_tts_lease). - self._tts_lease_async(self._voice_tts) - - _cprint(f"{_ACCENT}Voice TTS {status}.{_RST}") - - def _show_voice_status(self): - """Show current voice mode status.""" - from tools.voice_mode import check_voice_requirements - - reqs = check_voice_requirements() - - _cprint(f"\n{_BOLD}Voice Mode Status{_RST}") - _cprint(f" Mode: {'ON' if self._voice_mode else 'OFF'}") - _cprint(f" TTS: {'ON' if self._voice_tts else 'OFF'}") - _cprint(f" Recording: {'YES' if self._voice_recording else 'no'}") - # Display the startup-pinned label so /voice status always - # matches the live prompt_toolkit binding (Copilot round-14 on - # #19835, same class as round-13). Reading live config here - # would drift after a mid-session config edit. - _cprint(f" Record key: {self._voice_record_key_label()}") - _cprint(f"\n {_BOLD}Requirements:{_RST}") - for line in reqs["details"].split("\n"): - _cprint(f" {line}") - - def _persist_prompt_summary(self, icon: str, label: str, detail: str, outcome: str) -> None: - """Print a one-line scrollback summary of a resolved modal prompt. - - Modal panels (approval / clarify) live in the prompt_toolkit layout and - vanish on the next repaint, so the question and the decision leave no - trace in the terminal scrollback. When display.persist_prompts is on - (default), emit a dim single line after the prompt resolves so the - decision survives in chat history. - """ - if not CLI_CONFIG.get("display", {}).get("persist_prompts", True): - return - detail = " ".join(detail.split()) - if len(detail) > 120: - detail = detail[:119] + "…" - outcome = " ".join(outcome.split()) - if len(outcome) > 120: - outcome = outcome[:119] + "…" - _cprint(f"\n{_DIM}{icon} {label}: {detail} → {outcome}{_RST}") - - def _ring_bell(self, prompt: bool = False, context: str = "", detail: str = "") -> None: - """Write a terminal bell (\\a) if the matching display.bell_* flag is on. - - ``prompt=True`` is the blocking-modal variant (clarify / approval / - sudo / secret capture) gated by ``display.bell_on_prompt``; the default - is the end-of-turn bell gated by ``display.bell_on_complete``. Works - over SSH — the BEL propagates to the user's terminal. - - The same flag also emits an OSC 9 desktop notification (Ghostty, - iTerm2, Kitty, WezTerm) and, inside a supporting Warp build, a - ``warp://cli-agent`` OSC 777 event — see ``hermes_cli.terminal_notify``. - ``context`` is the short notification body (e.g. "approval"). - """ - flag = "bell_on_prompt" if prompt else "bell_on_complete" - if not getattr(self, flag, False): - return - try: - sys.stdout.write("\a") - sys.stdout.flush() - except Exception: - pass - try: - from hermes_cli.terminal_notify import notify as _terminal_notify - - _terminal_notify( - context or ("input needed" if prompt else "turn complete"), - prompt=prompt, - session_id=getattr(self, "session_id", "") or "", - detail=detail, - ) - except Exception: - pass - - def _clarify_callback(self, question, choices, multi_select=False, questions=None): - """ - Platform callback for the clarify tool. Called from the agent thread. - - Sets up the interactive selection UI (or freetext prompt for open-ended - questions), then blocks until the user responds via the prompt_toolkit - key bindings. If no response arrives within the configured timeout the - question is dismissed and the agent is told to decide on its own. - - When ``multi_select`` is True, shows checkboxes and the user can - select multiple options with Space, confirming with Enter. - - When ``questions`` is a non-empty list (batch clarify, issue #18450), - the panel switches to the A-compact multi-question layout and the - return value is a dict ``{"answers": {qid: raw_answer}}`` (plus - ``"timed_out": True`` when the deadline expired with only partial - answers). The single-question path below is unchanged. - """ - import time as _time - - from tools.clarify_gateway import resolve_clarify_timeout - - if questions: - return self._clarify_callback_batch(questions) - - # Canonical clarify timeout, shared with the gateway/TUI path. `<= 0` - # means unlimited (never auto-skip mid-think) → a null deadline. - timeout = resolve_clarify_timeout(CLI_CONFIG) - response_queue = queue.Queue() - is_open_ended = not choices - # multi-select support: only active when multi_select is True and choices exist - effective_multi = multi_select and not is_open_ended - - self._clarify_state = { - "question": question, - "choices": choices if not is_open_ended else [], - "selected": 0, - # multi-select support - "multi_select": effective_multi, - "selected_indices": set() if effective_multi else None, - "response_queue": response_queue, - } - self._clarify_deadline = None if timeout <= 0 else _time.monotonic() + timeout - # Open-ended questions skip straight to freetext input - self._clarify_freetext = is_open_ended - self._clarify_multi_base = None - - self._ring_bell(prompt=True, context="clarify") - # Trigger an immediate prompt_toolkit repaint from this (non-main) - # thread. Modal prompts must paint at once and must not be gated by the - # _invalidate throttle / resize guard — see _paint_now / _invalidate (#41098). - self._paint_now() - - # Poll for the user's response. The countdown in the hint line updates - # on each repaint; refresh it once a second so the timer stays visible - # while we wait. Selection changes (↑/↓) trigger instant repaints via - # the key bindings. - _last_countdown_refresh = _time.monotonic() - while True: - try: - result = response_queue.get(timeout=1) - self._clarify_deadline = None - self._persist_prompt_summary("?", "Clarify", question, str(result)) - return result - except queue.Empty: - # None deadline = unlimited: never auto-skip, just keep polling. - if self._clarify_deadline is not None: - remaining = self._clarify_deadline - _time.monotonic() - if remaining <= 0: - break - now = _time.monotonic() - if now - _last_countdown_refresh >= 1.0: - _last_countdown_refresh = now - self._paint_now() - - # Timed out — tear down the UI and let the agent decide - self._clarify_state = None - self._clarify_freetext = False - self._clarify_deadline = None - self._clarify_multi_base = None - self._paint_now() - _cprint(f"\n{_DIM}(clarify timed out after {timeout}s — agent will decide){_RST}") - return ( - "The user did not provide a response within the time limit. " - "Use your best judgement to make the choice and proceed." - ) - # --- Batch clarify (multi-question, issue #18450) ----------------------- - def _clarify_batch_set_active(self, state, index) -> None: - """Point the batch clarify panel at question ``index``. - - Mirrors the active question's data into the flat keys the existing - single-question keybindings and renderer read (``question``, - ``choices``, ``selected``, ``multi_select``, ``selected_indices``), - so ↑/↓/Space/number keys operate on the active question unchanged. - Open-ended questions drop straight into freetext, matching the - single-question path. Re-visiting an answered question restores the - cursor to the earlier selection (choice answers highlight their row, - an "Other" answer highlights the Other row) so the user can see and - edit what they picked. - """ - questions_list = state["questions"] - index = max(0, min(index, len(questions_list) - 1)) - entry = questions_list[index] - state["active"] = index - state["question"] = entry["question"] - state["choices"] = entry["choices"] or [] - state["selected"] = 0 - state["multi_select"] = bool(entry["multi_select"]) - state["selected_indices"] = set() if entry["multi_select"] else None - self._clarify_freetext = not entry["choices"] - self._clarify_multi_base = None - # Restore the earlier answer's cursor/checkbox position on re-visit. - meta = (state.get("answer_meta") or {}).get(entry["qid"]) - choices = entry["choices"] or [] - if meta is None: - return - if meta.get("kind") == "choice": - answer = state["answers"].get(entry["qid"]) - if answer in choices: - state["selected"] = choices.index(answer) - elif meta.get("kind") == "other": - state["selected"] = len(choices) - elif meta.get("kind") == "multi": - checked = set() - for label in meta.get("choices") or []: - if label in choices: - checked.add(choices.index(label)) - if meta.get("other_text"): - checked.add(len(choices)) - state["selected_indices"] = checked - - def _clarify_batch_lock(self, state, answer, meta=None) -> None: - """Lock ``answer`` for the active batch question and advance. - - Overwrites any earlier answer for the same question (locked answers - stay editable until the batch completes). ``meta`` records how the - answer was produced ({"kind": "choice"|"other"|"multi", ...}) so a - re-visit can restore the cursor and prefill an "Other" edit. Advances - ``active`` to the next unanswered question; when every question has - an answer, puts the answers dict on the response queue and tears down - the panel. - """ - entry = state["questions"][state["active"]] - state["answers"][entry["qid"]] = answer - state.setdefault("answer_meta", {})[entry["qid"]] = meta or {"kind": "choice"} - self._persist_prompt_summary("?", "Clarify", entry["question"], str(answer)) - total = len(state["questions"]) - for offset in range(1, total + 1): - candidate = (state["active"] + offset) % total - if state["questions"][candidate]["qid"] not in state["answers"]: - self._clarify_batch_set_active(state, candidate) - return - # Every question answered — resolve the batch. - try: - state["response_queue"].put(dict(state["answers"])) - except Exception: - pass - self._clarify_state = None - self._clarify_freetext = False - self._clarify_multi_base = None - - def _clarify_batch_enter(self, state) -> None: - """Enter in batch choice mode: lock the active question's selection. - - Multi-select questions lock a JSON array string of the checked - labels (the tool core parses it via ``_parse_multi_select_response``). - Selecting "Other" switches to freetext; the freetext submit path - locks the typed answer. Entering "Other" on a question whose earlier - answer was typed prefills the composer with that text for editing. - """ - choices = state.get("choices") or [] - selected = state.get("selected", 0) - entry = state["questions"][state["active"]] - meta = (state.get("answer_meta") or {}).get(entry["qid"]) or {} - if state.get("multi_select"): - indices = state.get("selected_indices") or set() - sorted_idx = sorted(indices) - selected_choices = [choices[i] for i in sorted_idx if i < len(choices)] - other_checked = len(choices) in sorted_idx - if other_checked: - # Stash the checked real choices (possibly none) so the - # freetext submit appends the typed answer to the array. - self._clarify_multi_base = selected_choices - self._clarify_freetext = True - self._clarify_prefill = meta.get("other_text") or "" - return - self._clarify_batch_lock( - state, - json.dumps(selected_choices, ensure_ascii=False), - meta={"kind": "multi", "choices": selected_choices, "other_text": ""}, - ) - return - if selected < len(choices): - self._clarify_batch_lock( - state, choices[selected], meta={"kind": "choice"} - ) - return - # "Other" highlighted → switch to freetext; prefill an earlier typed - # answer so Enter on an answered Other edits instead of retyping. - self._clarify_freetext = True - self._clarify_prefill = ( - meta.get("other_text") or "" if meta.get("kind") == "other" else "" - ) - - def _clarify_callback_batch(self, questions): - """Batch clarify panel (A-compact): all questions, one active. - - Blocks on the response queue like the single-question path. Returns - ``{"answers": {qid: raw_answer}}`` when every question is locked, the - same dict plus ``"timed_out": True`` when the deadline expires with - partial (or zero) answers, and passes a cancel string through - unchanged so the tool core resolves the batch empty. - """ - import time as _time - - from tools.clarify_gateway import resolve_clarify_timeout - - timeout = resolve_clarify_timeout(CLI_CONFIG) - response_queue = queue.Queue() - - state = { - "questions": list(questions), - "answers": {}, - "answer_meta": {}, - "active": 0, - "response_queue": response_queue, - # Flat keys mirroring the active question — filled by - # _clarify_batch_set_active below. - "question": "", - "choices": [], - "selected": 0, - "multi_select": False, - "selected_indices": None, - } - self._clarify_state = state - self._clarify_batch_set_active(state, 0) - self._clarify_deadline = None if timeout <= 0 else _time.monotonic() + timeout - self._ring_bell(prompt=True, context="clarify") - self._paint_now() - - _last_countdown_refresh = _time.monotonic() - while True: - try: - result = response_queue.get(timeout=1) - self._clarify_deadline = None - if isinstance(result, dict): - return {"answers": result} - # Cancel path (Ctrl+C teardown) posts a plain string — pass - # it through so the tool core resolves the batch empty. - return result - except queue.Empty: - if self._clarify_deadline is not None: - remaining = self._clarify_deadline - _time.monotonic() - if remaining <= 0: - break - now = _time.monotonic() - if now - _last_countdown_refresh >= 1.0: - _last_countdown_refresh = now - self._paint_now() - - # Timed out — keep the answers locked so far and flag the timeout. - partial = dict(state["answers"]) - self._clarify_state = None - self._clarify_freetext = False - self._clarify_deadline = None - self._clarify_multi_base = None - self._paint_now() - _cprint(f"\n{_DIM}(clarify timed out after {timeout}s — locked answers returned){_RST}") - return {"answers": partial, "timed_out": True} - - def _sudo_password_callback(self) -> str: - """ - Prompt for sudo password through the prompt_toolkit UI. - - Called from the agent thread when a sudo command is encountered. - Uses the same clarify-style mechanism: sets UI state, waits on a - queue for the user's response via the Enter key binding. - """ - import time as _time - - timeout = 45 - response_queue = queue.Queue() - - self._capture_modal_input_snapshot() - self._sudo_state = { - "response_queue": response_queue, - } - self._sudo_deadline = _time.monotonic() + timeout - self._ring_bell(prompt=True, context="sudo password") - - # Modal prompt — paint immediately, bypassing the throttle/resize guard - # so the prompt can't be dropped and time out unseen (#41098). - self._paint_now() - - while True: - try: - result = response_queue.get(timeout=1) - self._sudo_state = None - self._sudo_deadline = 0 - self._restore_modal_input_snapshot() - self._paint_now() - if result: - _cprint(f"\n{_DIM} ✓ Password received (cached for session){_RST}") - else: - _cprint(f"\n{_DIM} ⏭ Skipped{_RST}") - return result - except queue.Empty: - remaining = self._sudo_deadline - _time.monotonic() - if remaining <= 0: - break - self._paint_now() - - self._sudo_state = None - self._sudo_deadline = 0 - self._restore_modal_input_snapshot() - self._paint_now() - _cprint(f"\n{_DIM} ⏱ Timeout — continuing without sudo{_RST}") - return "" - - def _approval_callback(self, command: str, description: str, - *, allow_permanent: bool = True, - allow_session: bool = True, - smart_denied: bool = False) -> str: - """ - Prompt for dangerous command approval through the prompt_toolkit UI. - - Called from the agent thread. Shows a selection UI similar to clarify - with choices: once / session / always / deny. Smart DENY owner - overrides show only once / deny, as do gates that re-ask every time - (allow_session=False). When allow_permanent is False for another - reason (for example tirith), only 'always' is hidden. - Long commands also get a 'view' option so the full command can be - expanded before deciding. - - Uses _approval_lock to serialize concurrent requests (e.g. from - parallel delegation subtasks) so each prompt gets its own turn - and the shared _approval_state / _approval_deadline aren't clobbered. - """ - import time as _time - - with self._approval_lock: - timeout = int(CLI_CONFIG.get("approvals", {}).get("timeout", 300)) - response_queue = queue.Queue() - - self._approval_state = { - "command": command, - "description": description, - "choices": self._approval_choices( - command, - allow_permanent=allow_permanent, - allow_session=allow_session, - smart_denied=smart_denied, - ), - "selected": 0, - "response_queue": response_queue, - } - self._approval_deadline = _time.monotonic() + timeout - - self._ring_bell(prompt=True, context="approval", detail=command) - # Modal prompt — paint immediately, bypassing the throttle/resize - # guard. A throttled paint here can be silently dropped (250ms - # window collision or in-flight resize), leaving the panel unseen so - # the command is denied on timeout without the user ever seeing it - # (#41098). The countdown refreshes below paint the same way. - self._paint_now() - - _last_countdown_refresh = _time.monotonic() - while True: - try: - result = response_queue.get(timeout=1) - self._approval_state = None - self._approval_deadline = 0 - self._paint_now() - _outcome_labels = { - "once": "allowed once", - "session": "allowed for session", - "always": "added to allowlist", - "deny": "denied", - } - self._persist_prompt_summary( - "⚠", "Approval", command, - _outcome_labels.get(result, str(result)), - ) - return result - except queue.Empty: - remaining = self._approval_deadline - _time.monotonic() - if remaining <= 0: - break - now = _time.monotonic() - if now - _last_countdown_refresh >= 1.0: - _last_countdown_refresh = now - self._paint_now() - - self._approval_state = None - self._approval_deadline = 0 - self._paint_now() - _cprint(f"\n{_DIM} ⏱ Timeout — denying command{_RST}") - self._persist_prompt_summary( - "⚠", "Approval", command, "timed out (no response)", - ) - return "timeout" - - def _approval_choices(self, command: str, *, allow_permanent: bool = True, - allow_session: bool = True, - smart_denied: bool = False) -> list[str]: - """Return approval choices for a dangerous command prompt.""" - if smart_denied or not allow_session: - choices = ["once", "deny"] - else: - choices = ["once", "session", "always", "deny"] if allow_permanent else ["once", "session", "deny"] - if len(command) > 70: - choices.append("view") - return choices - - def _computer_use_approval_callback(self, action: str, args: dict, summary: str) -> str: - """Adapt the generic approval UI for the computer_use tool. - - The computer_use handler expects verdicts of the form - `approve_once` | `approve_session` | `always_approve` | `deny`. - The CLI's built-in approval UI returns `once` | `session` | `always` - | `deny`. Translate between the two. - """ - # Build a command-ish string so the existing UI renders something - # meaningful. `summary` is already a one-line human description. - verdict = self._approval_callback( - command=f"computer_use: {summary}", - description=f"Allow computer_use to perform `{action}`?", - ) - return { - "once": "approve_once", - "session": "approve_session", - "always": "always_approve", - "deny": "deny", - "timeout": "timeout", - }.get(verdict, "deny") - - def _handle_approval_selection(self) -> None: - """Process the currently selected dangerous-command approval choice.""" - state = self._approval_state - if not state: - return - - selected = state.get("selected", 0) - choices = state.get("choices") - if not isinstance(choices, list): - choices = [] - if not (0 <= selected < len(choices)): - return - - chosen = choices[selected] - if chosen == "view": - state["show_full"] = True - state["choices"] = [choice for choice in choices if choice != "view"] - if state["selected"] >= len(state["choices"]): - state["selected"] = max(0, len(state["choices"]) - 1) - self._invalidate() - return - - state["response_queue"].put(chosen) - self._approval_state = None - self._invalidate() - - def _get_approval_display_fragments(self): - """Render the dangerous-command approval panel for the prompt_toolkit UI. - - Layout priority: title + command + choices must always render, even if - the terminal is short or the description is long. Description is placed - at the bottom of the panel and gets truncated to fit the remaining row - budget. This prevents HSplit from clipping approve/deny off-screen when - tirith findings produce multi-paragraph descriptions or when the user - runs in a compact terminal pane. - """ - state = self._approval_state - if not state: - return [] - - _wrap_panel_text = _wrap_panel_text_keep_ws - - command = state["command"] - description = state["description"] - choices = state["choices"] - selected = state.get("selected", 0) - show_full = state.get("show_full", False) - - title = "⚠️ Dangerous Command" - cmd_display = command - choice_labels = { - "once": "Allow once", - "session": "Allow for this session", - "always": "Add to permanent allowlist", - "deny": "Deny", - "view": "Show full command", - } - - preview_lines = _wrap_panel_text(description, 60) - preview_lines.extend(_wrap_panel_text(cmd_display, 60)) - for i, choice in enumerate(choices): - prefix = '❯ ' if i == selected else ' ' - preview_lines.extend(_wrap_panel_text( - f"{prefix}{choice_labels.get(choice, choice)}", - 60, - subsequent_indent=" ", - )) - - box_width = _panel_box_width(title, preview_lines) - inner_text_width = max(8, box_width - 2) - - # Pre-wrap the mandatory content — command + choices must always render. - cmd_wrapped = _wrap_panel_text(cmd_display, inner_text_width) - if not show_full and "view" in choices and len(cmd_wrapped) > 4: - cmd_wrapped = cmd_wrapped[:3] + _wrap_panel_text( - "… (choose Show full command)", - inner_text_width, - ) - - # (choice_index, wrapped_line) so we can re-apply selected styling below - choice_wrapped: list[tuple[int, str]] = [] - for i, choice in enumerate(choices): - label = choice_labels.get(choice, choice) - # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) - if i < 9: - num_prefix = str(i + 1) - elif i == 9: - num_prefix = '0' - else: - num_prefix = ' ' # No number for items beyond 10th - prefix = f'❯ {num_prefix}. ' if i == selected else f' {num_prefix}. ' - for wrapped in _wrap_panel_text(f"{prefix}{label}", inner_text_width, subsequent_indent=" "): - choice_wrapped.append((i, wrapped)) - - # Budget vertical space so HSplit never clips the command or choices. - # Panel chrome (full layout with separators): - # top border + title + blank_after_title - # + blank_between_cmd_choices + bottom border = 5 rows. - # In tight terminals we collapse to: - # top border + title + bottom border = 3 rows (no blanks). - # - # reserved_below: rows consumed below the approval panel by the - # spinner/tool-progress line, status bar, input area, separators, and - # prompt symbol. Measured at ~6 rows during live PTY approval prompts; - # budget 6 so we don't overestimate the panel's room. - term_rows = shutil.get_terminal_size((100, 24)).lines - chrome_full = 5 - chrome_tight = 3 - reserved_below = 6 - - available = max(0, term_rows - reserved_below) - mandatory_full = chrome_full + len(cmd_wrapped) + len(choice_wrapped) - - # If the full-chrome panel doesn't fit, drop the separator blanks. - # This keeps the command and every choice on-screen in compact terminals. - use_compact_chrome = mandatory_full > available - chrome_rows = chrome_tight if use_compact_chrome else chrome_full - - # If the command itself is too long to leave room for choices (e.g. user - # hit "view" on a multi-hundred-character command), truncate it so the - # approve/deny buttons still render. Keep at least 1 row of command. - max_cmd_rows = max(1, available - chrome_rows - len(choice_wrapped)) - if len(cmd_wrapped) > max_cmd_rows: - keep = max(1, max_cmd_rows - 1) if max_cmd_rows > 1 else 1 - cmd_wrapped = cmd_wrapped[:keep] + _wrap_panel_text( - "… (command truncated — use /logs or /debug for full text)", - inner_text_width, - ) - - # Allocate any remaining rows to description. The extra -1 in full mode - # accounts for the blank separator between choices and description. - mandatory_no_desc = chrome_rows + len(cmd_wrapped) + len(choice_wrapped) - desc_sep_cost = 0 if use_compact_chrome else 1 - available_for_desc = available - mandatory_no_desc - desc_sep_cost - # Even on huge terminals, cap description height so the panel stays compact. - available_for_desc = max(0, min(available_for_desc, 10)) - - desc_wrapped = _wrap_panel_text(description, inner_text_width) if description else [] - if available_for_desc < 1 or not desc_wrapped: - desc_wrapped = [] - elif len(desc_wrapped) > available_for_desc: - keep = max(1, available_for_desc - 1) - desc_wrapped = desc_wrapped[:keep] + ["… (description truncated)"] - - # Render: title → command → choices → description (description last so - # any remaining overflow clips from the bottom of the least-critical - # content, never from the command or choices). Use compact chrome (no - # blank separators) when the terminal is tight. - lines = [] - lines.append(('class:approval-border', '╭' + ('─' * box_width) + '╮\n')) - _append_panel_line(lines, 'class:approval-border', 'class:approval-title', title, box_width) - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:approval-border', box_width) - - for wrapped in cmd_wrapped: - _append_panel_line(lines, 'class:approval-border', 'class:approval-cmd', wrapped, box_width) - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:approval-border', box_width) - - for i, wrapped in choice_wrapped: - style = 'class:approval-selected' if i == selected else 'class:approval-choice' - _append_panel_line(lines, 'class:approval-border', style, wrapped, box_width) - - if desc_wrapped: - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:approval-border', box_width) - for wrapped in desc_wrapped: - _append_panel_line(lines, 'class:approval-border', 'class:approval-desc', wrapped, box_width) - - lines.append(('class:approval-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _secret_capture_callback(self, var_name: str, prompt: str, metadata=None) -> dict: - return prompt_for_secret(self, var_name, prompt, metadata) - - def _capture_modal_input_snapshot(self) -> None: - """Temporarily clear the input buffer and save the user's in-progress draft.""" - if self._modal_input_snapshot is not None or not getattr(self, "_app", None): - return - try: - buf = self._app.current_buffer - self._modal_input_snapshot = { - "text": buf.text, - "cursor_position": buf.cursor_position, - } - buf.reset() - except Exception: - self._modal_input_snapshot = None - - def _restore_modal_input_snapshot(self) -> None: - """Restore any draft text that was present before a modal prompt opened.""" - snapshot = self._modal_input_snapshot - self._modal_input_snapshot = None - if not snapshot or not getattr(self, "_app", None): - return - try: - buf = self._app.current_buffer - buf.text = snapshot.get("text", "") - buf.cursor_position = min(snapshot.get("cursor_position", 0), len(buf.text)) - except Exception: - pass - - def _clear_active_overlays_for_interrupt(self) -> None: - """Drain and clear every input-blocking overlay left by an interrupted agent. - - approval/clarify/sudo/secret prompts each block a worker thread on a - ``response_queue.get()``. When the agent is interrupted the worker - thread is torn down, but the overlay's state dict stays set — leaving - the CLI input gated (``read_only`` condition + keypress filter) with no - thread servicing the prompt. The result is a frozen terminal until the - prompt's own timeout expires. Push a terminal value onto each queue so - any still-blocked thread unblocks cleanly, then nil the state out and - restore the user's pre-modal draft (#14026). - - Safe default per prompt: approval -> "deny", clarify/sudo/secret -> - cancel (None / empty). Each step is wrapped so a dead queue can't - prevent clearing the others. - """ - if self._approval_state: - try: - self._approval_state["response_queue"].put("deny") - except Exception: - pass - self._approval_state = None - if self._clarify_state: - try: - self._clarify_state["response_queue"].put( - "The user cancelled. Use your best judgement to proceed." - ) - except Exception: - pass - self._clarify_state = None - self._clarify_freetext = False - self._clarify_multi_base = None - if self._sudo_state: - try: - self._sudo_state["response_queue"].put("") - except Exception: - pass - self._sudo_state = None - self._sudo_deadline = 0 - self._restore_modal_input_snapshot() - if self._secret_state: - try: - self._cancel_secret_capture() - except Exception: - self._secret_state = None - - def _submit_secret_response(self, value: str) -> None: - if not self._secret_state: - return - self._secret_state["response_queue"].put(value) - self._secret_state = None - self._secret_deadline = 0 - # Modal teardown — paint directly so the secret panel clears at once and - # isn't held by the _invalidate throttle/resize guard (#41098). - self._paint_now() - - def _cancel_secret_capture(self) -> None: - self._submit_secret_response("") - - def _clear_secret_input_buffer(self) -> None: - if getattr(self, "_app", None): - try: - self._app.current_buffer.reset() - except Exception: - pass - def chat(self, message, images: list = None, voice_input: bool = False) -> Optional[str]: """ Send a message to the agent and get a response. @@ -17652,426 +7462,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): if tts_thread is not None and tts_thread.is_alive(): tts_thread.join(timeout=5) - def _clear_terminal_on_exit(self): - """Clear screen + scrollback so nothing is stranded above the exit summary. - - Called from ``_print_exit_summary`` after ``app.run()`` has returned and - prompt_toolkit has torn down its renderer + restored terminal modes — - so a direct write to the real stdout fd is safe (the StdoutProxy / - patch_stdout layer is gone by now). - - Sequence: ``ESC[3J`` (erase scrollback) + ``ESC[2J`` (erase visible - screen) + ``ESC[H`` (cursor home). Modern terminals on Linux, macOS and - Windows (Terminal / conhost with VT processing, which prompt_toolkit - already enables) all honor these. Best-effort: skip silently when - stdout isn't a real console, and fall back to the platform ``clear`` / - ``cls`` command if the escape write fails. - """ - try: - stream = sys.stdout - if stream is None or not stream.isatty(): - return - except Exception: - return - try: - stream.write("\033[3J\033[2J\033[H") - stream.flush() - return - except Exception: - pass - # Fallback: shell clear command (rarely needed — escapes work on every - # VT-capable terminal, but this covers exotic stdout wrappers). - try: - os.system("cls" if os.name == "nt" else "clear") - except Exception: - pass - - def _persist_active_session_before_close(self): - """Best-effort SQLite/JSON flush before the CLI marks a session closed. - - ``run_conversation()`` normally persists at turn boundaries, but a - terminal close/SIGHUP/SIGTERM can unwind the prompt_toolkit app while - the agent thread still holds the current turn only in memory. Flush the - agent's live ``_session_messages`` before ``end_session()`` so resume, - session_search, and state.db do not lose the interrupted turn. - """ - agent = getattr(self, "agent", None) - if not agent or not hasattr(agent, "_persist_session"): - return - - persist_lock = getattr(agent, "_session_persist_lock", None) - - def _snapshot_and_persist() -> None: - # This snapshot must share the staging lock with ``chat()``. Without - # it, close can retain a mutable history baseline just before chat - # appends its pending dict; the later flush then mistakes that dict - # for durable history and stamps it without writing a row (#63766). - messages = getattr(agent, "_session_messages", None) - pending_cli_message = getattr(agent, "_pending_cli_user_message", None) - if not isinstance(messages, list): - messages = getattr(self, "conversation_history", None) - if not isinstance(messages, list): - return - if isinstance(pending_cli_message, dict) and not any( - message is pending_cli_message for message in messages - ): - # The UI has accepted a new input but the worker still exposes its - # prior snapshot. Include only that staged dict; the baseline below - # keeps any durable resumed prefix from being re-appended. - messages = [*messages, pending_cli_message] - if not messages: - return - - # A normal turn builds a new list that reuses the resumed-history dicts. - # Keep that CLI history as the baseline so a signal between assigning - # ``_session_messages`` and the turn's DB flush cannot append its durable - # prefix a second time. Once the CLI takes the turn result, however, both - # names can point at the same live list; passing that alias would mark an - # unflushed tail durable without writing it. Marker-only persistence is - # correct only in that alias case. - conversation_history = getattr(self, "conversation_history", None) - pending_cli_message = getattr(agent, "_pending_cli_user_message", None) - if ( - isinstance(conversation_history, list) - and conversation_history - and conversation_history[-1] is pending_cli_message - ): - # The UI accepted this user message before the agent finished its - # early persistence. Its dict can already be in ``messages`` but is - # not durable yet, so exclude it from the resumed-history baseline. - conversation_history = conversation_history[:-1] - elif not isinstance(conversation_history, list) or conversation_history is messages: - conversation_history = None - - # A first-turn close can arrive before the worker builds its cached - # prompt. Build or restore it before the DB row is created so the - # durable transcript never leaves a NULL system_prompt cache entry. - if getattr(agent, "_cached_system_prompt", None) is None: - try: - from agent.conversation_loop import _restore_or_build_system_prompt - - _restore_or_build_system_prompt(agent, None, conversation_history) - except Exception: - logger.debug("Could not build system prompt during CLI close", exc_info=True) - return - if getattr(agent, "_cached_system_prompt", None) is None: - return - - agent._ensure_db_session() - agent._persist_session(messages, conversation_history) - if getattr(agent, "session_id", None): - self.session_id = agent.session_id - getattr(self, "_write_terminal_breadcrumb", lambda: None)() - - try: - if persist_lock is None: - _snapshot_and_persist() - else: - with persist_lock: - _snapshot_and_persist() - except (Exception, KeyboardInterrupt) as e: - logger.debug("Could not persist active CLI session before close: %s", e) - - def _print_exit_summary(self, clear_screen: bool = True): - """Print session resume info on exit, similar to Claude Code. - - Args: - clear_screen: When True (default), clear the terminal screen and - scrollback before printing the summary. This is appropriate for - interactive TUI teardown (#38252). Single-query (-q) mode should - pass False to preserve the printed answer (#53009). - """ - if clear_screen: - # Clear the screen + scrollback before printing the summary so the - # live bottom chrome (status bar, input box, separator rules) and the - # rest of the session transcript don't get stranded above the exit - # summary (#38252). By this point app.run() has returned and - # prompt_toolkit has restored terminal modes, so writing raw escapes - # to stdout is safe. ESC[3J clears scrollback, ESC[2J clears the - # visible screen, ESC[H homes the cursor — so the summary prints at a - # clean top-left. Falls back to the platform clear command if stdout - # isn't a TTY-capable stream. Honors NO_COLOR/dumb terminals by - # skipping silently when there's no real console. - self._clear_terminal_on_exit() - print() - msg_count = len(self.conversation_history) - if msg_count > 0: - user_msgs = len([m for m in self.conversation_history if m.get("role") == "user"]) - tool_calls = len([m for m in self.conversation_history if m.get("role") == "tool" or m.get("tool_calls")]) - elapsed = datetime.now() - self.session_start - hours, remainder = divmod(int(elapsed.total_seconds()), 3600) - minutes, seconds = divmod(remainder, 60) - if hours > 0: - duration_str = f"{hours}h {minutes}m {seconds}s" - elif minutes > 0: - duration_str = f"{minutes}m {seconds}s" - else: - duration_str = f"{seconds}s" - - # Look up session title for resume-by-name hint - session_title = None - if self._session_db: - try: - session_title = self._session_db.get_session_title(self.session_id) - except Exception: - pass - - print("Resume this session with:") - # Session IDs are profile-constrained, so the resume hint must - # include `-p ` for non-default profiles. Without this, - # copying the hint from a non-default profile fails to find the - # session on the next invocation. The "default" and "custom" - # profile names use the standard HERMES_HOME, so no -p needed. - try: - from hermes_cli.profiles import get_active_profile_name - _active_profile = get_active_profile_name() - except Exception: - _active_profile = "default" - profile_flag = ( - "" if _active_profile in ("default", "custom") else f" -p {_active_profile}" - ) - print(f" hermes --resume {self.session_id}{profile_flag}") - if session_title: - print(f" hermes -c \"{session_title}\"{profile_flag}") - print() - print(f"Session: {self.session_id}") - if session_title: - print(f"Title: {session_title}") - print(f"Duration: {duration_str}") - print(f"Messages: {msg_count} ({user_msgs} user, {tool_calls} tool calls)") - else: - try: - from hermes_cli.skin_engine import get_active_goodbye - goodbye = get_active_goodbye("Goodbye! ⚕") - except Exception: - goodbye = "Goodbye! ⚕" - print(goodbye) - - def _get_tui_prompt_symbols(self) -> tuple[str, str]: - """Return ``(normal_prompt, state_suffix)`` for the active skin. - - ``normal_prompt`` is the full ``branding.prompt_symbol``. - ``state_suffix`` is what special states (sudo/secret/approval/agent) - should render after their leading icon. - - When a profile is active (not "default"), the profile name is - prepended to the prompt symbol: ``coder ❯`` instead of ``❯``. - """ - try: - from hermes_cli.skin_engine import get_active_prompt_symbol - symbol = get_active_prompt_symbol("❯ ") - except Exception: - symbol = "❯ " - - symbol = (symbol or "❯ ").rstrip() + " " - - # Prepend profile name when not default - try: - from hermes_cli.profiles import get_active_profile_name - profile = get_active_profile_name() - if profile not in {"default", "custom"}: - symbol = f"{profile} {symbol}" - except Exception: - pass - stripped = symbol.rstrip() - if not stripped: - return "❯ ", "❯ " - - parts = stripped.split() - candidate = parts[-1] if parts else "" - arrow_chars = ("❯", ">", "$", "#", "›", "»", "→") - if any(ch in candidate for ch in arrow_chars): - return symbol, candidate.rstrip() + " " - - # Icon-only custom prompts should still remain visible in special states. - return symbol, symbol - - def _audio_level_bar(self) -> str: - """Return a visual audio level indicator based on current RMS.""" - _LEVEL_BARS = " ▁▂▃▄▅▆▇" - rec = getattr(self, "_voice_recorder", None) - if rec is None: - return "" - rms = rec.current_rms - # Normalize RMS (0-32767) to 0-7 index, with log-ish scaling - # Typical speech RMS is 500-5000, we cap display at ~8000 - level = min(rms, 8000) * 7 // 8000 - return _LEVEL_BARS[level] - - def _get_tui_prompt_fragments(self): - """Return the prompt_toolkit fragments for the current interactive state.""" - symbol, state_suffix = self._get_tui_prompt_symbols() - compact = self._use_minimal_tui_chrome(width=self._get_tui_terminal_width()) - - def _state_fragment(style: str, icon: str, extra: str = ""): - if compact: - text = icon - if extra: - text = f"{text} {extra.strip()}".rstrip() - return [(style, text + " ")] - if extra: - return [(style, f"{icon} {extra} {state_suffix}")] - return [(style, f"{icon} {state_suffix}")] - - if self._voice_recording: - bar = self._audio_level_bar() - return _state_fragment("class:voice-recording", "●", bar) - if self._voice_processing: - return _state_fragment("class:voice-processing", "◉") - if self._sudo_state: - return _state_fragment("class:sudo-prompt", "🔐") - if self._secret_state: - return _state_fragment("class:sudo-prompt", "🔑") - if self._approval_state: - return _state_fragment("class:prompt-working", "⚠") - if getattr(self, "_slash_confirm_state", None): - return _state_fragment("class:prompt-working", "⚠") - if self._clarify_freetext: - return _state_fragment("class:clarify-selected", "✎") - if self._clarify_state: - return _state_fragment("class:prompt-working", "?") - if self._command_running: - return _state_fragment("class:prompt-working", self._command_spinner_frame()) - if self._agent_running: - return _state_fragment("class:prompt-working", "⚕") - if self._voice_mode: - return _state_fragment("class:voice-prompt", "🎤") - return [("class:prompt", symbol)] - - def _get_tui_prompt_text(self) -> str: - """Return the visible prompt text for width calculations.""" - return "".join(text for _, text in self._get_tui_prompt_fragments()) - - def _build_tui_style_dict(self) -> dict[str, str]: - """Layer the active skin's prompt_toolkit colors over the base TUI style. - - Also rewrites any hex-color tokens in the resulting style strings - to their light-mode equivalents (via _LIGHT_MODE_REMAP) when the - terminal is detected as light. This makes the chrome readable - on cream Terminal.app backgrounds without per-skin overrides. - """ - style_dict = dict(getattr(self, "_tui_style_base", {}) or {}) - try: - from hermes_cli.skin_engine import get_prompt_toolkit_style_overrides - style_dict.update(get_prompt_toolkit_style_overrides()) - except Exception: - pass - # Light-mode remap on the style strings. Each value is a pt - # style string like "bg:#1a1a2e #C0C0C0 bold" — split on space, - # rewrite any "#XXX" tokens (including "bg:#XXX") through the - # light-mode remap, rejoin. - # - # CRITICAL: skip the remap entirely when a style string already - # specifies its own bg (e.g. status-bar / completion-menu styles - # with `bg:#1a1a2e ...`). Those colors were tuned for that - # specific dark bg and remapping the FG to a dark equivalent - # would produce dark-on-dark (invisible). The terminal's BG - # mode is irrelevant — what matters is the bg the style itself - # paints. - try: - if _detect_light_mode(): - def _remap_value(v: str) -> str: - if not v: - return v - tokens = v.split() - has_explicit_bg = any(t.startswith("bg:") for t in tokens) - if has_explicit_bg: - # The style paints its own bg — leave its fg alone. - return v - return " ".join( - _maybe_remap_for_light_mode(t) if t.startswith("#") else t - for t in tokens - ) - style_dict = {k: _remap_value(v or "") for k, v in style_dict.items()} - except Exception: - pass - return style_dict - - def _apply_tui_skin_style(self) -> bool: - """Refresh prompt_toolkit styling for a running interactive TUI.""" - if not getattr(self, "_app", None) or not getattr(self, "_tui_style_base", None): - return False - self._app.style = PTStyle.from_dict(self._build_tui_style_dict()) - self._invalidate(min_interval=0.0) - return True - # --- Protected TUI extension hooks for wrapper CLIs --- - def _get_extra_tui_widgets(self) -> list: - """Return extra prompt_toolkit widgets to insert into the TUI layout. - - Wrapper CLIs can override this to inject widgets (e.g. a mini-player, - overlay menu) into the layout without overriding ``run()``. Widgets - are inserted between the spacer and the status bar. - """ - return [] - - def _register_extra_tui_keybindings(self, kb, *, input_area) -> None: - """Register extra keybindings on the TUI ``KeyBindings`` object. - - Wrapper CLIs can override this to add keybindings (e.g. transport - controls, modal shortcuts) without overriding ``run()``. - - Parameters - ---------- - kb : KeyBindings - The active keybinding registry for the prompt_toolkit application. - input_area : TextArea - The main input widget, for wrappers that need to inspect or - manipulate user input from a keybinding handler. - """ - - def _build_tui_layout_children( - self, - *, - sudo_widget, - secret_widget, - approval_widget, - slash_confirm_widget=None, - clarify_widget, - model_picker_widget=None, - command_palette_widget=None, - spinner_widget=None, - spacer, - status_bar, - input_rule_top, - image_bar, - input_area, - input_rule_bot, - voice_status_bar, - completions_menu, - ) -> list: - """Assemble the ordered list of children for the root ``HSplit``. - - Wrapper CLIs typically override ``_get_extra_tui_widgets`` instead of - this method. Override this only when you need full control over widget - ordering. - """ - return [ - item for item in [ - Window(height=0), - sudo_widget, - secret_widget, - approval_widget, - slash_confirm_widget, - clarify_widget, - model_picker_widget, - command_palette_widget, - spinner_widget, - spacer, - *self._get_extra_tui_widgets(), - getattr(self, "_pet_widget", None), - getattr(self, "_stash_panel_widget", None), - status_bar, - input_rule_top, - image_bar, - input_area, - input_rule_bot, - voice_status_bar, - completions_menu, - ] if item is not None - ] - def _tui_process_loop(self): while not self._should_exit: try: @@ -18430,1846 +7822,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): pass raise KeyboardInterrupt() # fallback for non-prompt_toolkit contexts - def _tui_spinner_loop(self): - while not self._should_exit: - if not self._app: - time.sleep(0.1) - continue - if self._command_running: - self._invalidate(min_interval=0.1) - time.sleep(0.1) - else: - # Do not repaint the idle prompt every second. In non-full-screen - # prompt_toolkit mode, background redraws can fight tmux/Ghostty/cmux - # viewport restoration after focus changes and visually move the - # command input area. Keep idle stable; input/agent events still - # invalidate explicitly when the UI actually changes. - time.sleep(0.2) - - def _get_clarify_batch_display_fragments(self, state): - """Build styled text for the batch (multi-question) clarify panel. - - A-compact layout mirroring the TUI: a "N questions" header, one - status line per question (✓ answered → answer / ▸ active / - · pending), and the active question's numbered choices (+ Other) - expanded directly beneath its status line. - """ - questions_list = state.get("questions") or [] - answers = state.get("answers") or {} - active = state.get("active", 0) - choices = state.get("choices") or [] - selected = state.get("selected", 0) - multi_select = state.get("multi_select", False) - selected_indices = state.get("selected_indices", set()) if multi_select else set() - - title = "Hermes needs your input" - header = f"{len(questions_list)} questions" - - def _status_rows(width): - """(style, text) rows for the status list + expanded active question.""" - rows = [] - answer_meta = state.get("answer_meta") or {} - for idx, entry in enumerate(questions_list): - answered = entry["qid"] in answers - if answered: - marker = "✓" - elif idx == active: - marker = "▸" - else: - marker = "·" - label = f"{marker} {entry['question']}" - row_style = 'class:clarify-selected' if idx == active else 'class:clarify-choice' - for wrapped in _wrap_panel_text(label, width, subsequent_indent=" "): - rows.append((row_style, wrapped)) - if answered: - # The locked answer on its own line, in its own color, - # so the current answer stays readable while walking - # the list with Tab/Shift-Tab. - for wrapped in _wrap_panel_text( - f" {answers[entry['qid']]}", width, subsequent_indent=" " - ): - rows.append(('class:clarify-answer', wrapped)) - if idx != active: - continue - # Expanded active question: numbered choices + Other. - for i, choice in enumerate(choices): - num_prefix = str(i + 1) if i < 9 else ('0' if i == 9 else ' ') - if multi_select: - cb = "[x]" if i in selected_indices else "[ ]" - cursor = "❯" if i == selected and not self._clarify_freetext else " " - prefix = f" {cursor} {cb} {num_prefix}. " - else: - cursor = "❯" if i == selected and not self._clarify_freetext else " " - prefix = f" {cursor} {num_prefix}. " - style = 'class:clarify-selected' if i == selected and not self._clarify_freetext else 'class:clarify-choice' - for wrapped in _wrap_panel_text(f"{prefix}{choice}", width, subsequent_indent=" "): - rows.append((style, wrapped)) - if choices: - other_idx = len(choices) - other_num = other_idx + 1 - other_num_prefix = str(other_num) if other_num < 10 else ('0' if other_num == 10 else ' ') - if multi_select: - cb = "[x]" if other_idx in selected_indices else "[ ]" - mid = f"{cb} {other_num_prefix}" - else: - mid = other_num_prefix - # An earlier typed answer stays visible next to Other; - # Enter on it edits (the composer is prefilled). - meta = answer_meta.get(entry["qid"]) or {} - other_text = meta.get("other_text") or "" - other_suffix = f"Other: {other_text}" if other_text else None - if self._clarify_freetext: - other_label = f" ❯ {mid}. " + (other_suffix or "Other (type below)") - other_style = 'class:clarify-active-other' - elif selected == other_idx: - other_label = f" ❯ {mid}. " + (other_suffix or "Other (type your answer)") - other_style = 'class:clarify-selected' - else: - other_label = f" {mid}. " + (other_suffix or "Other (type your answer)") - other_style = 'class:clarify-choice' - for wrapped in _wrap_panel_text(other_label, width, subsequent_indent=" "): - rows.append((other_style, wrapped)) - elif self._clarify_freetext: - for wrapped in _wrap_panel_text( - " Type your answer in the prompt below, then press Enter.", width - ): - rows.append(('class:clarify-active-other', wrapped)) - return rows - - preview_rows = _status_rows(60) - box_width = _panel_box_width(title, [header] + [text for _, text in preview_rows]) - inner_text_width = max(8, box_width - 2) - rows = _status_rows(inner_text_width) - - lines = [] - lines.append(('class:clarify-border', '╭─ ')) - lines.append(('class:clarify-title', title)) - lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) - _append_panel_line(lines, 'class:clarify-border', 'class:clarify-question', header, box_width) - for style, text in rows: - _append_panel_line(lines, 'class:clarify-border', style, text, box_width) - lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _get_clarify_display_fragments(self): - """Build styled text for the clarify question/choices panel. - - Layout priority: choices + Other option must always render even if - the question is very long. The question is budgeted to leave enough - rows for the choices and trailing chrome; anything over the budget - is truncated with a marker. - """ - state = self._clarify_state - if not state: - return [] - if state.get("questions"): - return self._get_clarify_batch_display_fragments(state) - - question = state["question"] - choices = state.get("choices") or [] - selected = state.get("selected", 0) - # multi-select support - multi_select = state.get("multi_select", False) - selected_indices = state.get("selected_indices", set()) if multi_select else set() - preview_lines = _wrap_panel_text(question, 60) - for i, choice in enumerate(choices): - # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) - if i < 9: - num_prefix = str(i + 1) - elif i == 9: - num_prefix = '0' - else: - num_prefix = ' ' - if multi_select: - cb = "[x]" if i in selected_indices else "[ ]" - if i == selected and not self._clarify_freetext: - prefix = f"❯ {cb} {num_prefix}. " - else: - prefix = f" {cb} {num_prefix}. " - elif i == selected and not self._clarify_freetext: - prefix = f"❯ {num_prefix}. " - else: - prefix = f" {num_prefix}. " - preview_lines.extend(_wrap_panel_text(f"{prefix}{choice}", 60, subsequent_indent=" ")) - # "Other" option in preview - other_num = len(choices) + 1 - if other_num < 10: - other_num_prefix = str(other_num) - elif other_num == 10: - other_num_prefix = '0' - else: - other_num_prefix = ' ' - other_idx_val = len(choices) - if multi_select: - cb = "[x]" if other_idx_val in selected_indices else "[ ]" - other_label = ( - f"❯ {cb} {other_num_prefix}. Other (type below)" if self._clarify_freetext - else f"❯ {cb} {other_num_prefix}. Other (type your answer)" if selected == other_idx_val - else f" {cb} {other_num_prefix}. Other (type your answer)" - ) - else: - other_label = ( - f"❯ {other_num_prefix}. Other (type below)" if self._clarify_freetext - else f"❯ {other_num_prefix}. Other (type your answer)" if selected == len(choices) - else f" {other_num_prefix}. Other (type your answer)" - ) - preview_lines.extend(_wrap_panel_text(other_label, 60, subsequent_indent=" ")) - box_width = _panel_box_width("Hermes needs your input", preview_lines) - inner_text_width = max(8, box_width - 2) - - # Pre-wrap choices + Other option — these are mandatory. - choice_wrapped: list[tuple[int, str]] = [] - if choices: - for i, choice in enumerate(choices): - # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) - if i < 9: - num_prefix = str(i + 1) - elif i == 9: - num_prefix = '0' - else: - num_prefix = ' ' - # multi-select support: add checkbox after cursor indicator - if multi_select: - cb = "[x]" if i in selected_indices else "[ ]" - if i == selected and not self._clarify_freetext: - prefix = f'❯ {cb} {num_prefix}. ' - else: - prefix = f' {cb} {num_prefix}. ' - elif i == selected and not self._clarify_freetext: - prefix = f'❯ {num_prefix}. ' - else: - prefix = f' {num_prefix}. ' - for wrapped in _wrap_panel_text(f"{prefix}{choice}", inner_text_width, subsequent_indent=" "): - choice_wrapped.append((i, wrapped)) - # Trailing Other row(s) - other_idx = len(choices) - other_num = other_idx + 1 - if other_num < 10: - other_num_prefix = str(other_num) - elif other_num == 10: - other_num_prefix = '0' - else: - other_num_prefix = ' ' - # multi-select support: add checkbox to Other option - if multi_select: - cb = "[x]" if other_idx in selected_indices else "[ ]" - if selected == other_idx and not self._clarify_freetext: - other_label_mand = f'❯ {cb} {other_num_prefix}. Other (type your answer)' - elif self._clarify_freetext: - other_label_mand = f'❯ {cb} {other_num_prefix}. Other (type below)' - else: - other_label_mand = f' {cb} {other_num_prefix}. Other (type your answer)' - else: - if selected == other_idx and not self._clarify_freetext: - other_label_mand = f'❯ {other_num_prefix}. Other (type your answer)' - elif self._clarify_freetext: - other_label_mand = f'❯ {other_num_prefix}. Other (type below)' - else: - other_label_mand = f' {other_num_prefix}. Other (type your answer)' - other_wrapped = _wrap_panel_text(other_label_mand, inner_text_width, subsequent_indent=" ") - elif self._clarify_freetext: - # Freetext-only mode: the guidance line takes the place of choices. - other_wrapped = _wrap_panel_text( - "Type your answer in the prompt below, then press Enter.", - inner_text_width, - ) - else: - other_wrapped = [] - - # Budget the question so mandatory rows always render. - # Chrome layouts: - # full : top border + blank_after_title + blank_after_question - # + blank_before_bottom + bottom border = 5 rows - # tight: top border + bottom border = 2 rows (drop all blanks) - # - # reserved_below matches the approval-panel budget (~6 rows for - # spinner/tool-progress + status + input + separators + prompt). - term_rows = shutil.get_terminal_size((100, 24)).lines - chrome_full = 5 - chrome_tight = 2 - reserved_below = 6 - - available = max(0, term_rows - reserved_below) - # The compact decision must reserve room for at least one question - # row on top of the choices, otherwise full chrome (3 blank - # separators) gets kept when there is no room for it and the panel - # overflows the viewport — HSplit then clips the panel's tail, - # silently dropping the choices (the reported bug). - mandatory_full = chrome_full + 1 + len(choice_wrapped) + len(other_wrapped) - - use_compact_chrome = mandatory_full > available - chrome_rows = chrome_tight if use_compact_chrome else chrome_full - - max_question_rows = max(1, available - chrome_rows - len(choice_wrapped) - len(other_wrapped)) - max_question_rows = min(max_question_rows, 12) # soft cap on huge terminals - - # When the choices alone (plus compact chrome) already exceed the - # viewport, drop the question entirely — the choices are the only - # thing the user must see to make a selection. Without this the - # question would still claim its 1-row floor above and push the - # tail of the choices off-screen (HSplit clips the overflow). - choices_overflow = chrome_rows + len(choice_wrapped) + len(other_wrapped) >= available - if choices_overflow: - max_question_rows = 0 - - question_wrapped = _wrap_panel_text(question, inner_text_width) - if max_question_rows <= 0: - question_wrapped = [] - elif len(question_wrapped) > max_question_rows: - # The truncation marker is itself a row, so it must count - # against the budget. With a 1-row budget there is no room for - # both a question line and the marker — show the marker alone - # so the rendered question never exceeds max_question_rows. - keep = max(0, max_question_rows - 1) - question_wrapped = question_wrapped[:keep] + ["… (question truncated)"] - - lines = [] - # Box top border - lines.append(('class:clarify-border', '╭─ ')) - lines.append(('class:clarify-title', 'Hermes needs your input')) - lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len("Hermes needs your input") - 3)) + '╮\n')) - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - - # Question text (bounded) - for wrapped in question_wrapped: - _append_panel_line(lines, 'class:clarify-border', 'class:clarify-question', wrapped, box_width) - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - - if self._clarify_freetext and not choices: - for wrapped in other_wrapped: - _append_panel_line(lines, 'class:clarify-border', 'class:clarify-choice', wrapped, box_width) - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - - if choices: - # Multiple-choice mode: show selectable options - for i, wrapped in choice_wrapped: - style = 'class:clarify-selected' if i == selected and not self._clarify_freetext else 'class:clarify-choice' - _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) - - # "Other" option (trailing row(s), only shown when choices exist) - other_idx = len(choices) - # Calculate number prefix for "Other" option - other_num = other_idx + 1 - if other_num < 10: - other_num_prefix = str(other_num) - elif other_num == 10: - other_num_prefix = '0' - else: - other_num_prefix = ' ' - - if selected == other_idx and not self._clarify_freetext: - other_style = 'class:clarify-selected' - elif self._clarify_freetext: - other_style = 'class:clarify-active-other' - else: - other_style = 'class:clarify-choice' - for wrapped in other_wrapped: - _append_panel_line(lines, 'class:clarify-border', other_style, wrapped, box_width) - - if not use_compact_chrome: - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _get_model_picker_display_fragments(self): - state = self._model_picker_state - if not state: - return [] - stage = state.get("stage", "provider") - if stage == "provider": - title = "⚙ Model Picker — Select Provider" - choices = [] - _providers = state.get("providers") - for p in _providers if isinstance(_providers, list) else []: - count = p.get("total_models", len(p.get("models", []))) - label = f"{p['name']} ({count} model{'s' if count != 1 else ''})" - if p.get("is_current"): - label += " ← current" - choices.append(label) - choices.append("Cancel") - hint = f"Current: {state.get('current_model', 'unknown')} on {state.get('current_provider', 'unknown')}" - else: - provider_data = state.get("provider_data") or {} - model_list = state.get("model_list") or [] - title = f"⚙ Model Picker — {provider_data.get('name', provider_data.get('slug', 'Provider'))}" - # Fuzzy filter: narrow the concrete model list by the typed - # query. Selection still resolves to a real entry (see the - # filtered_pairs index mapping in the selection handler), so - # this never introduces an ambiguous model resolution. - _query = state.get("filter", "") or "" - filtered_pairs = self._filter_model_picker_entries(model_list, _query) - state["_filtered_pairs"] = filtered_pairs - model_labels = [e for (_i, e) in filtered_pairs] - choices = list(model_labels) + ["← Back", "Cancel"] - if _query: - hint = ( - f"Filter: {_query}▏ ({len(model_labels)}/{len(model_list)} match " - "— type to narrow, Backspace to clear)" - ) - elif model_list: - hint = f"Select a model ({len(model_list)} available) — type to filter" - else: - hint = "No models listed for this provider. Use Back or Cancel." - - box_width = _panel_box_width(title, [hint] + choices, min_width=46, max_width=84) - inner_text_width = max(8, box_width - 6) - selected = state.get("selected", 0) - - # Scrolling viewport: the panel renders into a Window with no max - # height, so without limiting visible items the bottom border and - # any items past the available terminal rows get clipped on long - # provider catalogs (e.g. Ollama Cloud's 36+ models). - try: - from prompt_toolkit.application import get_app - term_rows = get_app().output.get_size().rows - except Exception: - term_rows = shutil.get_terminal_size((100, 24)).lines - scroll_offset, visible = HermesCLI._compute_model_picker_viewport( - selected, state.get("_scroll_offset", 0), len(choices), term_rows, - ) - state["_scroll_offset"] = scroll_offset - - lines = [] - lines.append(('class:clarify-border', '╭─ ')) - lines.append(('class:clarify-title', title)) - lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - _append_panel_line(lines, 'class:clarify-border', 'class:clarify-hint', hint, box_width) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - for idx in range(scroll_offset, scroll_offset + visible): - choice = choices[idx] - style = 'class:clarify-selected' if idx == selected else 'class:clarify-choice' - prefix = '❯ ' if idx == selected else ' ' - for wrapped in _wrap_panel_text(prefix + choice, inner_text_width, subsequent_indent=' '): - _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _get_command_palette_display_fragments(self): - state = self._command_palette_state - if not state: - return [] - rows = self._command_palette_visible_entries() - state["_visible_count"] = len(rows) - _query = state.get("filter", "") or "" - total = len(state.get("entries") or []) - title = "⚙ Command Palette" - if _query: - hint = f"Filter: {_query}▏ ({len(rows)}/{total} match — Enter inserts, Esc cancels)" - else: - hint = f"Type to filter {total} commands — ↑/↓ then Enter inserts, Esc cancels" - - labels = [f"{c} — {d}" if d else c for (c, _cat, d) in rows] - if not labels: - labels = ["(no matching commands)"] - box_width = _panel_box_width(title, [hint] + labels, min_width=50, max_width=90) - inner_text_width = max(8, box_width - 6) - selected = state.get("selected", 0) - try: - from prompt_toolkit.application import get_app - term_rows = get_app().output.get_size().rows - except Exception: - term_rows = shutil.get_terminal_size((100, 24)).lines - scroll_offset, visible = HermesCLI._compute_model_picker_viewport( - selected, state.get("_scroll_offset", 0), len(labels), term_rows, - ) - state["_scroll_offset"] = scroll_offset - - lines = [] - lines.append(('class:clarify-border', '╭─ ')) - lines.append(('class:clarify-title', title)) - lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - _append_panel_line(lines, 'class:clarify-border', 'class:clarify-hint', hint, box_width) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - for idx in range(scroll_offset, min(scroll_offset + visible, len(labels))): - label = labels[idx] - style = 'class:clarify-selected' if idx == selected else 'class:clarify-choice' - prefix = '❯ ' if idx == selected else ' ' - for wrapped in _wrap_panel_text(prefix + label, inner_text_width, subsequent_indent=' '): - _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) - _append_blank_panel_line(lines, 'class:clarify-border', box_width) - lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _get_sudo_display_fragments(self): - state = self._sudo_state - if not state: - return [] - title = '🔐 Sudo Password Required' - body = 'Enter password below (hidden), or press Enter to skip' - box_width = _panel_box_width(title, [body]) - lines = [] - lines.append(('class:sudo-border', '╭─ ')) - lines.append(('class:sudo-title', title)) - lines.append(('class:sudo-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) - _append_blank_panel_line(lines, 'class:sudo-border', box_width) - _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', body, box_width) - _append_blank_panel_line(lines, 'class:sudo-border', box_width) - lines.append(('class:sudo-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _get_secret_display_fragments(self): - state = self._secret_state - if not state: - return [] - - title = '🔑 Skill Setup Required' - prompt = state.get("prompt") or f"Enter value for {state.get('var_name', 'secret')}" - metadata = state.get("metadata") or {} - help_text = metadata.get("help") - body = 'Enter secret below (hidden), ESC or Ctrl+C to skip' - content_lines = [prompt, body] - if help_text: - content_lines.insert(1, str(help_text)) - box_width = _panel_box_width(title, content_lines) - lines = [] - lines.append(('class:sudo-border', '╭─ ')) - lines.append(('class:sudo-title', title)) - lines.append(('class:sudo-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) - _append_blank_panel_line(lines, 'class:sudo-border', box_width) - _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', prompt, box_width) - if help_text: - _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', str(help_text), box_width) - _append_blank_panel_line(lines, 'class:sudo-border', box_width) - _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', body, box_width) - _append_blank_panel_line(lines, 'class:sudo-border', box_width) - lines.append(('class:sudo-border', '╰' + ('─' * box_width) + '╯\n')) - return lines - - def _tui_hint_text(self): - if self._sudo_state: - remaining = max(0, int(self._sudo_deadline - time.monotonic())) - return [ - ('class:hint', ' password hidden · Enter to skip'), - ('class:clarify-countdown', f' ({remaining}s)'), - ] - - if self._secret_state: - remaining = max(0, int(self._secret_deadline - time.monotonic())) - return [ - ('class:hint', ' secret hidden · Enter to skip'), - ('class:clarify-countdown', f' ({remaining}s)'), - ] - - if self._approval_state: - remaining = max(0, int(self._approval_deadline - time.monotonic())) - return [ - ('class:hint', ' ↑/↓ to select, Enter to confirm'), - ('class:clarify-countdown', f' ({remaining}s)'), - ] - - if self._slash_confirm_state: - remaining = max(0, int(self._slash_confirm_deadline - time.monotonic())) - return [ - ('class:hint', ' type 1/2/3, or ↑/↓ to select, Enter to confirm'), - ('class:clarify-countdown', f' ({remaining}s)'), - ] - - if self._clarify_state: - # None deadline = unlimited wait → hide the countdown entirely. - if self._clarify_deadline is None: - countdown = '' - else: - remaining = max(0, int(self._clarify_deadline - time.monotonic())) - countdown = f' ({remaining}s)' - if self._clarify_freetext: - return [ - ('class:hint', ' type your answer and press Enter'), - ('class:clarify-countdown', countdown), - ] - if self._clarify_state.get("questions"): - return [ - ('class:hint', ' ↑/↓ to select, Enter to lock, Tab next question'), - ('class:clarify-countdown', countdown), - ] - return [ - ('class:hint', ' ↑/↓ to select, Enter to confirm'), - ('class:clarify-countdown', countdown), - ] - - if self._command_running: - frame = self._command_spinner_frame() - detail = "input temporarily disabled" if self._command_blocks_input else "input stays active; Enter queues" - return [ - ('class:hint', f' {frame} command in progress · {detail}'), - ] - - return [] - - def _tui_placeholder_text(self): - if self._voice_recording: - _label = self._voice_record_key_label() - return f"recording... {_label} to stop, Ctrl+C to cancel" - if self._voice_processing: - return "transcribing..." - if self._sudo_state: - return "type password (hidden), Enter to submit · ESC to skip" - if self._secret_state: - return "type secret (hidden), Enter to submit · ESC to skip" - if self._approval_state: - return "" - if self._slash_confirm_state: - return "type 1/2/3, or use ↑/↓ then Enter" - if self._clarify_freetext: - return "type your answer here and press Enter" - if self._clarify_state: - return "" - if self._command_running: - frame = self._command_spinner_frame() - status = self._command_status or "Processing command..." - return f"{frame} {status}" - if self._agent_running: - return "msg=interrupt · /queue · /bg · /steer · Ctrl+C cancel" - if self._voice_mode: - _label = self._voice_record_key_label() - return f"type or {_label} to record" - # Advertise a parked draft so the stash can never be silently - # forgotten — the composer itself tells you how to get it back. - _stash_hint = "" - try: - _stash_hint = self._prompt_stash.placeholder_hint() - except Exception: - _stash_hint = "" - if _stash_hint: - return _stash_hint - # Idle + empty composer: show a rotating task-oriented example to - # nudge the user toward a high-value first action (C-09). Chosen - # once per session (self._composer_placeholder) so it stays stable - # while being read, not flickering every render. - return getattr(self, "_composer_placeholder", "") or "" - - def _get_stash_panel_display_fragments(self): - try: - _stash = self._prompt_stash - return self._render_stash_panel( - _stash.panel_rows(), - _stash.panel_cursor, - self._get_tui_terminal_width(), - ) - except Exception: - return [] - - def _tui_handle_voice_record(self, event): - """Toggle voice recording when voice mode is active. - - IMPORTANT: This handler runs in prompt_toolkit's event-loop thread. - Any blocking call here (locks, sd.wait, disk I/O) freezes the - entire UI. All heavy work is dispatched to daemon threads. - """ - if not self._voice_mode: - return - # Always allow STOPPING a recording (even when agent is running) - if self._voice_recording: - # Manual stop via push-to-talk key: stop continuous mode - with self._voice_lock: - self._voice_continuous = False - # Flag clearing is handled atomically inside _voice_stop_and_transcribe - event.app.invalidate() - threading.Thread( - target=self._voice_stop_and_transcribe, - daemon=True, - ).start() - else: - # Allow disarming continuous mode even when the agent is - # running or transcribing — otherwise the user is stuck in - # an auto-restart loop until /voice off (#67545). - if self._agent_running or self._voice_processing: - with self._voice_lock: - self._voice_continuous = False - event.app.invalidate() - return - # Guard: don't START recording during interactive prompts - if self._clarify_state or self._sudo_state or self._approval_state or self._slash_confirm_state: - return - - # Interrupt TTS if playing, so user can start talking. - # stop_playback() is fast (just terminates a subprocess); - # the stop event drains the streaming pipeline if one is live. - if not self._voice_tts_done.is_set(): - try: - logger.info("TTS CUT: record key handler cutting TTS") - from tools.tts_streaming import mark_speech_interrupted - mark_speech_interrupted() - if self._voice_tts_stop is not None: - self._voice_tts_stop.set() - from tools.voice_mode import stop_playback - stop_playback() - self._voice_tts_done.set() - except Exception: - pass - - with self._voice_lock: - self._voice_continuous = True - - # Dispatch to a daemon thread so play_beep(sd.wait), - # AudioRecorder.start(lock acquire), and config I/O - # never block the prompt_toolkit event loop. - def _start_recording(): - try: - self._voice_start_recording() - if hasattr(self, '_app') and self._app: - self._app.invalidate() - except Exception as e: - _cprint(f"\n{_DIM}Voice recording failed: {e}{_RST}") - - threading.Thread(target=_start_recording, daemon=True).start() - event.app.invalidate() - - def _tui_handle_ctrl_c(self, event): - """Handle Ctrl+C - cancel interactive prompts, interrupt agent, or exit. - - Priority: - 0. Cancel active voice recording - 1. Cancel active sudo/approval/clarify prompt - 2. Interrupt the running agent (first press) - 3. Force exit (second press within 2s, or when idle) - """ - now = time.time() - - # Cancel active voice recording. - # Run cancel() in a background thread to prevent blocking the - # event loop if AudioRecorder._lock or CoreAudio takes time. - _should_cancel_voice = False - _recorder_ref = None - with self._voice_lock: - if self._voice_recording and self._voice_recorder: - _recorder_ref = self._voice_recorder - self._voice_recording = False - self._voice_continuous = False - _should_cancel_voice = True - if _should_cancel_voice: - _cprint(f"\n{_DIM}Recording cancelled.{_RST}") - threading.Thread( - target=_recorder_ref.cancel, daemon=True - ).start() - event.app.invalidate() - return - - # Cancel slash confirmation prompt (foreground UI, not an - # agent-blocking overlay — cancel and stop here). - if self._slash_confirm_state: - self._submit_slash_confirm_response("cancel") - event.app.current_buffer.reset() - event.app.invalidate() - return - - # Cancel /model picker (foreground UI — cancel and stop here). - if self._model_picker_state: - self._close_model_picker() - event.app.current_buffer.reset() - event.app.invalidate() - return - - # Cancel command palette (foreground UI — cancel and stop here). - if self._command_palette_state: - self._close_command_palette() - event.app.current_buffer.reset() - event.app.invalidate() - return - - # Clear all agent-blocking overlays (approval/clarify/sudo/secret) - # in one shot. We do NOT return after clearing — we fall through so - # that if the agent is also running we fire the interrupt on the same - # Ctrl+C press. This fixes the case where a stale/orphaned overlay - # (left behind by a previous interrupt) consumes the press without - # ever reaching the agent-interrupt branch, leaving the chat frozen - # (#14026). - _overlay_cleared = bool( - self._sudo_state - or self._secret_state - or self._approval_state - or self._clarify_state - ) - if _overlay_cleared: - self._clear_active_overlays_for_interrupt() - event.app.current_buffer.reset() - event.app.invalidate() - - # If we only cleared overlays and the agent is NOT running, stop here - # (don't fall through to the interrupt/exit path). - if _overlay_cleared and not (self._agent_running and self.agent): - return - - if self._agent_running and self.agent: - if now - self._last_ctrl_c_time < 2.0: - print("\n⚡ Force exiting...") - self._should_exit = True - event.app.exit() - return - - self._last_ctrl_c_time = now - print("\n⚡ Interrupting agent... (press Ctrl+C again to force exit)") - request_hard_interrupt(self.agent) - # If there's text or images, clear them (like bash). - # If everything is already empty, exit. - elif event.app.current_buffer.text or self._attached_images: - event.app.current_buffer.reset() - self._attached_images.clear() - event.app.invalidate() - else: - self._should_exit = True - event.app.exit() - - def _tui_handle_ctrl_q(self, event): - """Alternative interrupt/exit shortcut (Ctrl+Q). - - Behaves like Ctrl+C: cancels active prompts, interrupts the - running agent, or clears the input buffer. Does not support - the double-press 'force exit' feature of Ctrl+C. - """ - # Cancel active voice recording. - _should_cancel_voice = False - _recorder_ref = None - with self._voice_lock: - if self._voice_recording and self._voice_recorder: - _recorder_ref = self._voice_recorder - self._voice_recording = False - self._voice_continuous = False - _should_cancel_voice = True - if _should_cancel_voice: - _cprint(f"\n{_DIM}Recording cancelled.{_RST}") - threading.Thread( - target=_recorder_ref.cancel, daemon=True - ).start() - event.app.invalidate() - return - - # Cancel slash confirmation prompt (foreground UI — cancel and stop). - if self._slash_confirm_state: - self._submit_slash_confirm_response("cancel") - event.app.current_buffer.reset() - event.app.invalidate() - return - - # Cancel /model picker (foreground UI — cancel and stop). - if self._model_picker_state: - self._close_model_picker() - event.app.current_buffer.reset() - event.app.invalidate() - return - - # Clear all agent-blocking overlays in one shot, then fall through to - # the agent-interrupt branch so a single Ctrl+Q both clears a stale - # overlay and interrupts a still-running agent (#14026). - _overlay_cleared = bool( - self._sudo_state - or self._secret_state - or self._approval_state - or self._clarify_state - ) - if _overlay_cleared: - self._clear_active_overlays_for_interrupt() - event.app.current_buffer.reset() - event.app.invalidate() - - if _overlay_cleared and not (self._agent_running and self.agent): - return - - if self._agent_running and self.agent: - print("\n⚡ Interrupting agent...") - request_hard_interrupt(self.agent) - elif event.app.current_buffer.text or self._attached_images: - event.app.current_buffer.reset() - self._attached_images.clear() - event.app.invalidate() - else: - self._should_exit = True - event.app.exit() - - def _tui_make_clarify_number_handler(self, idx): - def handler(event): - if self._clarify_state and not self._clarify_freetext: - choices = self._clarify_state.get("choices") or [] - # multi-select support: number keys toggle checkboxes instead of submitting - if self._clarify_state.get("multi_select"): - if idx < len(choices): - indices = self._clarify_state.get("selected_indices", set()) - if idx in indices: - indices.discard(idx) - else: - indices.add(idx) - event.app.invalidate() - elif idx == len(choices): - # Toggle "Other" in multi-select mode - indices = self._clarify_state.get("selected_indices", set()) - if idx in indices: - indices.discard(idx) - else: - indices.add(idx) - event.app.invalidate() - return - # Original single-select: number keys submit directly - # Map index to choice (treating "Other" as the last option) - if idx < len(choices): - # Batch mode: lock the numbered choice for the active - # question instead of resolving the whole prompt. - if self._clarify_state.get("questions"): - self._clarify_batch_lock(self._clarify_state, choices[idx]) - event.app.invalidate() - return - # Select a numbered choice - self._clarify_state["response_queue"].put(choices[idx]) - self._clarify_state = None - self._clarify_freetext = False - event.app.invalidate() - elif idx == len(choices): - # Select "Other" option - self._clarify_freetext = True - event.app.invalidate() - return handler - - def _tui_restore_stash_payload(self, event, payload) -> None: - """Put a popped (text, images) payload back into the composer.""" - if not payload: - return - text, images = payload - buf = event.app.current_buffer - buf.text = text - buf.cursor_position = len(text) - if images: - # Restore attachments the draft was carrying. Extend rather - # than replace: the user may have attached something new since - # the stash was taken and silently dropping it would be data - # loss. - for img in images: - if img not in self._attached_images: - self._attached_images.append(img) - - def _tui_handle_stash_panel_up(self, event): - self._prompt_stash.move_cursor(-1) - event.app.invalidate() - - def _tui_handle_stash_panel_down(self, event): - self._prompt_stash.move_cursor(1) - event.app.invalidate() - - def _tui_handle_stash_panel_delete(self, event): - """D in the browse panel discards the highlighted draft.""" - self._prompt_stash.delete_at_cursor() - event.app.invalidate() - - def _tui_handle_stash_panel_close(self, event): - self._prompt_stash.close_panel() - event.app.invalidate() - - def _tui_handle_tab(self, event): - """Tab: accept completion, auto-suggestion, or start completions. - - Priority: - 1. Completion menu open → accept selected completion - 2. Ghost text suggestion available → accept auto-suggestion - 3. Otherwise → start completion menu - - After accepting a provider like 'anthropic:', the completion menu - closes and complete_while_typing doesn't fire (no keystroke). - This binding re-triggers completions so stage-2 models appear - immediately. - """ - buf = event.current_buffer - if buf.complete_state: - # Completion menu is open — accept the selection - completion = buf.complete_state.current_completion - if completion is None: - # Menu open but nothing selected — select first then grab it - buf.go_to_completion(0) - completion = buf.complete_state and buf.complete_state.current_completion - if completion is None: - return - # Accept the selected completion - buf.apply_completion(completion) - elif buf.suggestion and buf.suggestion.text: - # No completion menu, but there's a ghost text auto-suggestion — accept it - buf.insert_text(buf.suggestion.text) - else: - # No menu and no suggestion — start completions from scratch - buf.start_completion() - - def _tui_handle_double_escape(self, event): - """Double ESC: discard the current draft and any attached images. - - Matches Claude Code / Gemini CLI, where double-Esc is the - clear-the-composer gesture. It works while the agent is - streaming, which is the gap Ctrl+C leaves: Ctrl+C interrupts a - running turn and only clears the draft when idle, so mid-stream - there was no way to discard a half-typed prompt. - - The draft is appended to history first, so Up recalls it — the - same undo affordance Claude Code provides, and the reason this - is safe to bind to a key pressed by reflex. - - Single ESC is the prefix for Alt sequences (escape+enter, - escape+g, escape+v), so prompt_toolkit's escape-timeout keeps - those distinct from the double press. Modal prompts bind ESC - eagerly and are excluded here so cancel still wins. - """ - buf = event.app.current_buffer - if not (buf.text or self._attached_images): - return - buf.reset(append_to_history=bool(buf.text)) - self._attached_images.clear() - event.app.invalidate() - - def _tui_handle_ignored_terminal_sequence(self, event): - """Consume parser-level ignored terminal sequences before self-insert. - - install_ignored_terminal_sequences() in hermes_cli.pt_input_extras - registers focus reports (CSI I / CSI O) as Keys.Ignore at the - VT100 parser level. Without this no-op binding the default - self-insert path would still fire and the bytes would land in - the buffer. - - Focus-in (CSI I) additionally schedules a rate-limited full - repaint: while the tab/window was hidden the emulator may have - coalesced output or repainted the surface, so prompt_toolkit's - incremental diff would stack a fresh copy of the prompt chrome - on top of the stale one (#60920 focus-regain variant, #25337). - """ - try: - for press in getattr(event, "key_sequence", None) or (): - if getattr(press, "data", None) == "\x1b[I": - self._schedule_focus_regain_redraw() - break - except Exception: - pass - return None - - def _tui_handle_escape_modal(self, event): - """ESC cancels active secret/sudo prompts.""" - if self._secret_state: - self._cancel_secret_capture() - event.app.current_buffer.reset() - event.app.invalidate() - return - if self._sudo_state: - self._sudo_state["response_queue"].put("") - self._sudo_state = None - event.app.invalidate() - return - if self._slash_confirm_state: - self._submit_slash_confirm_response("cancel") - event.app.current_buffer.reset() - event.app.invalidate() - return - - def _tui_handle_ctrl_z(self, event): - """Handle Ctrl+Z - suspend process to background (Unix only).""" - if sys.platform == 'win32': - _cprint(f"\n{_DIM}Suspend (Ctrl+Z) is not supported on Windows.{_RST}") - event.app.invalidate() - return - import signal as _sig - from prompt_toolkit.application import run_in_terminal - from hermes_cli.skin_engine import get_active_skin - agent_name = get_active_skin().get_branding("agent_name", "Hermes Agent") - msg = f"\n{agent_name} has been suspended. Run `fg` to bring {agent_name} back." - def _suspend(): - os.write(1, msg.encode()) - os.kill(0, _sig.SIGTSTP) - run_in_terminal(_suspend) - - def _tui_handle_ctrl_d(self, event): - """Ctrl+D: delete char under cursor (standard readline behaviour). - Only exit when the input is empty — same as bash/zsh. Pending - attached images count as input and block the EOF-exit so the - user doesn't lose them silently. - """ - buf = event.app.current_buffer - if buf.text: - buf.delete() - elif self._attached_images: - # Empty text but pending attachments — no-op, don't exit. - return - else: - self._should_exit = True - event.app.exit() - - def _tui_recall_without_recollapse(self, buf, move): - """Run a history-navigation move, suppressing paste-collapse. - - Recalled history can hold the full text of a paste that was - collapsed to a placeholder at submit time. Loading it back into the - buffer looks exactly like a fresh large paste to ``_on_text_changed`` - and would be re-collapsed. Set the skip flag around the move; if the - move didn't change the text (plain cursor movement), clear the flag - so a later real paste still collapses. - """ - before = buf.text - self._skip_paste_collapse = True - move() - if buf.text == before: - self._skip_paste_collapse = False - - def _tui_handle_alt_v(self, event): - """Alt+V — paste image from clipboard. - - Alt key combos pass through all terminal emulators (sent as - ESC + key), unlike Ctrl+V which terminals intercept for text - paste. This is the reliable way to attach clipboard images - on WSL2, VSCode, and any terminal over SSH where Ctrl+V - can't reach the application for image-only clipboard. - """ - if self._try_attach_clipboard_image(): - event.app.invalidate() - else: - # No image found — show a hint - pass # silent when no image (avoid noise on accidental press) - - def _tui_handle_ctrl_v(self, event): - """Fallback image paste for terminals without bracketed paste. - - On Linux terminals (GNOME Terminal, Konsole, etc.), Ctrl+V - sends raw byte 0x16 instead of triggering a paste. This - binding catches that and checks the clipboard for images. - On terminals that DO intercept Ctrl+V for paste (macOS - Terminal, iTerm2, VSCode, Windows Terminal), the bracketed - paste handler fires instead and this binding never triggers. - """ - if self._try_attach_clipboard_image(): - event.app.invalidate() - - def _tui_handle_ctrl_l(self, event): - """Ctrl+L: force a clean full-screen repaint. - - Recovers the UI after external terminal buffer drift — tmux / - cmux tab switches, ``clear`` from a subshell, SSH window - restores, etc. — that prompt_toolkit can't detect on its own. - Matches the universal bash/zsh/fish/vim/htop convention. - """ - self._force_full_redraw() - - def _tui_insert_newline(self, event): - """Insert a newline for multi-line input (Alt+Enter, and Ctrl+J/Ctrl+Enter - when multiline shortcuts are on). - - Alt+Enter works on mac/Linux/WSL. On Windows Terminal that keystroke is - intercepted at the terminal layer (toggles fullscreen) and never reaches - here — Windows users get newline via Ctrl+Enter, which WT delivers as c-j. - """ - event.current_buffer.insert_text('\n') - - def _tui_handle_open_in_editor(self, event): - """Ctrl+G (or Alt+G in VSCode/Cursor) opens the current draft in an external editor.""" - self._open_external_editor(event.current_buffer) - - def _tui_model_picker_down(self, event): - state = self._model_picker_state - if not state: - return - if state.get("stage") == "provider": - max_idx = len(state.get("providers") or []) - else: - # +1 for "← Back" and Cancel over the filtered visible rows. - _fp = state.get("_filtered_pairs") - _visible = len(_fp) if _fp is not None else len(state.get("model_list") or []) - max_idx = _visible + 1 - state["selected"] = min(max_idx, state.get("selected", 0) + 1) - event.app.invalidate() - - def _tui_model_picker_up(self, event): - if self._model_picker_state: - self._model_picker_state["selected"] = max(0, self._model_picker_state.get("selected", 0) - 1) - event.app.invalidate() - - def _tui_model_picker_escape(self, event): - """ESC clears an active filter first, else closes the picker.""" - st = self._model_picker_state - if st and st.get("stage") == "model" and (st.get("filter") or ""): - st["filter"] = "" - st["selected"] = 0 - st["_scroll_offset"] = 0 - event.app.invalidate() - return - self._close_model_picker() - event.app.current_buffer.reset() - event.app.invalidate() - - def _tui_model_picker_filter_backspace(self, event): - st = self._model_picker_state - if not st: - return - cur = st.get("filter", "") or "" - st["filter"] = cur[:-1] - st["selected"] = 0 - st["_scroll_offset"] = 0 - event.app.invalidate() - - def _tui_make_model_filter_char_handler(self, ch: str): - def handler(event): - st = self._model_picker_state - if not st or st.get("stage") != "model": - return - st["filter"] = (st.get("filter", "") or "") + ch - st["selected"] = 0 - st["_scroll_offset"] = 0 - event.app.invalidate() - return handler - - def _tui_make_palette_char_handler(self, ch: str): - def handler(event): - st = self._command_palette_state - if not st: - return - st["filter"] = (st.get("filter", "") or "") + ch - st["selected"] = 0 - st["_scroll_offset"] = 0 - event.app.invalidate() - return handler - - def _tui_make_approval_number_handler(self, idx): - def handler(event): - if self._approval_state and idx < len(self._approval_state["choices"]): - self._approval_state["selected"] = idx - self._handle_approval_selection() - event.app.invalidate() - return handler - - def _tui_make_slash_confirm_number_handler(self, idx): - def handler(event): - if self._slash_confirm_state and idx < len(self._slash_confirm_state.get("choices") or []): - choice = self._slash_confirm_state["choices"][idx][0] - self._submit_slash_confirm_response(choice) - event.app.current_buffer.reset() - event.app.invalidate() - return handler - - def _tui_clarify_toggle(self, event): - if self._clarify_state: - selected = self._clarify_state["selected"] - indices = self._clarify_state.get("selected_indices", set()) - if selected in indices: - indices.discard(selected) - else: - indices.add(selected) - event.app.invalidate() - - def _tui_clarify_down(self, event): - """Move selection down in clarify choices.""" - if self._clarify_state: - choices = self._clarify_state.get("choices") or [] - max_idx = len(choices) # last index is the "Other" option - self._clarify_state["selected"] = min(max_idx, self._clarify_state["selected"] + 1) - event.app.invalidate() - - def _tui_clarify_up(self, event): - """Move selection up in clarify choices.""" - if self._clarify_state: - self._clarify_state["selected"] = max(0, self._clarify_state["selected"] - 1) - event.app.invalidate() - - def _tui_clarify_batch_tab(self, event): - state = self._clarify_state - if state and state.get("questions"): - self._clarify_batch_set_active( - state, (state["active"] + 1) % len(state["questions"]) - ) - event.app.invalidate() - - def _tui_clarify_batch_backtab(self, event): - state = self._clarify_state - if state and state.get("questions"): - self._clarify_batch_set_active( - state, (state["active"] - 1) % len(state["questions"]) - ) - event.app.invalidate() - - def _tui_command_palette_backspace(self, event): - st = self._command_palette_state - if st: - st["filter"] = (st.get("filter", "") or "")[:-1] - st["selected"] = 0 - st["_scroll_offset"] = 0 - event.app.invalidate() - - def _tui_command_palette_down(self, event): - st = self._command_palette_state - if st: - n = st.get("_visible_count", len(self._command_palette_visible_entries())) - st["selected"] = min(max(0, n - 1), st.get("selected", 0) + 1) - event.app.invalidate() - - def _tui_command_palette_up(self, event): - st = self._command_palette_state - if st: - st["selected"] = max(0, st.get("selected", 0) - 1) - event.app.invalidate() - - def _tui_command_palette_enter(self, event): - self._handle_command_palette_selection() - event.app.invalidate() - - def _tui_command_palette_escape(self, event): - self._close_command_palette() - event.app.invalidate() - - def _tui_open_command_palette(self, event): - self._open_command_palette() - event.app.invalidate() - - def _tui_slash_confirm_down(self, event): - if self._slash_confirm_state: - max_idx = len(self._slash_confirm_state.get("choices") or []) - 1 - self._slash_confirm_state["selected"] = min(max_idx, self._slash_confirm_state.get("selected", 0) + 1) - event.app.invalidate() - - def _tui_slash_confirm_up(self, event): - if self._slash_confirm_state: - self._slash_confirm_state["selected"] = max(0, self._slash_confirm_state.get("selected", 0) - 1) - event.app.invalidate() - - def _tui_approval_down(self, event): - if self._approval_state: - max_idx = len(self._approval_state["choices"]) - 1 - self._approval_state["selected"] = min(max_idx, self._approval_state["selected"] + 1) - event.app.invalidate() - - def _tui_approval_up(self, event): - if self._approval_state: - self._approval_state["selected"] = max(0, self._approval_state["selected"] - 1) - event.app.invalidate() - - def _tui_wake_startup(self): - try: - self._maybe_start_wake_word() - except Exception as e: - logger.debug("wake-word startup skipped: %s", e) - - def _tui_suppress_closed_loop_errors(self, loop, context): - exc = context.get("exception") - if isinstance(exc, RuntimeError) and "Event loop is closed" in str(exc): - return # silently suppress - if isinstance(exc, KeyError) and "is not registered" in str(exc): - return # suppress selector registration failures (#6393) - if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.EIO: - return # suppress I/O errors from broken stdout on interrupt (#13710) - # Fall back to default handler for everything else - loop.default_exception_handler(context) - - def _tui_handle_enter(self, event): - """Handle Enter key - submit input. - - Routes to the correct queue based on active UI state: - - Sudo password prompt: password goes to sudo response queue - - Approval selection: selected choice goes to approval response queue - - Clarify freetext mode: answer goes to the clarify response queue - - Clarify choice mode: selected choice goes to the clarify response queue - - Agent running: goes to _interrupt_queue (chat() monitors this) - - Agent idle: goes to _pending_input (process_loop monitors this) - Commands (starting with /) always go to _pending_input so they're - handled as commands, not sent as interrupt text to the agent. - """ - # --- Sudo password prompt: submit the typed password --- - if self._sudo_state: - text = event.app.current_buffer.text - self._sudo_state["response_queue"].put(text) - self._sudo_state = None - event.app.invalidate() - return - - # --- Secret prompt: submit the typed secret --- - if self._secret_state: - text = event.app.current_buffer.text - self._submit_secret_response(text) - event.app.current_buffer.reset() - event.app.invalidate() - return - - # --- Approval selection: confirm the highlighted choice --- - if self._approval_state: - self._handle_approval_selection() - event.app.invalidate() - return - - # --- Slash-command confirmation: submit typed or highlighted choice --- - if self._slash_confirm_state: - text = event.app.current_buffer.text.strip() - choices = self._slash_confirm_state.get("choices") or [] - choice = self._normalize_slash_confirm_choice(text, choices) if text else None - if choice is None: - selected = self._slash_confirm_state.get("selected", 0) - if 0 <= selected < len(choices): - choice = choices[selected][0] - self._submit_slash_confirm_response(choice or "cancel") - event.app.current_buffer.reset() - event.app.invalidate() - return - - # --- /model picker modal --- - if self._model_picker_state: - try: - # Picker selections follow the same session-scoped default - # as /model ; honour model.persist_switch_by_default. - from hermes_cli.model_switch import resolve_persist_behavior - - self._handle_model_picker_selection( - persist_global=resolve_persist_behavior(False, False) - ) - except Exception as _exc: - _cprint(f" ✗ Model selection failed: {_exc}") - self._close_model_picker() - event.app.current_buffer.reset() - event.app.invalidate() - return - - # --- Clarify freetext mode: user typed their own answer --- - if self._clarify_freetext and self._clarify_state: - text = event.app.current_buffer.text.strip() - if text: - state = self._clarify_state - # Batch mode: lock the typed answer for the active question - if state.get("questions"): - base = getattr(self, '_clarify_multi_base', None) - if base is not None: - # Multi-select "Other": append the typed answer to - # the checked labels as a JSON array string. - answer = json.dumps(base + [text], ensure_ascii=False) - meta = {"kind": "multi", "choices": list(base), "other_text": text} - self._clarify_multi_base = None - else: - answer = text - meta = {"kind": "other", "other_text": text} - self._clarify_freetext = False - self._clarify_prefill = "" - self._clarify_batch_lock(state, answer, meta=meta) - event.app.current_buffer.reset() - event.app.invalidate() - return - # multi-select: prepend previously checked real choices - base = getattr(self, '_clarify_multi_base', None) - if base: - text = ", ".join(base) + ", " + text - self._clarify_multi_base = None - self._clarify_state["response_queue"].put(text) - self._clarify_state = None - self._clarify_freetext = False - event.app.current_buffer.reset() - event.app.invalidate() - return - - # --- Clarify choice mode: confirm the highlighted selection --- - if self._clarify_state and not self._clarify_freetext: - state = self._clarify_state - # Batch mode: Enter locks the active question's answer and - # advances to the next unanswered question. - if state.get("questions"): - self._clarify_batch_enter(state) - # Editing an earlier "Other" answer: prefill the composer - # with the previously typed text. - if self._clarify_freetext and self._clarify_prefill: - event.app.current_buffer.text = self._clarify_prefill - event.app.current_buffer.cursor_position = len(self._clarify_prefill) - self._clarify_prefill = "" - event.app.invalidate() - return - selected = state["selected"] - choices = state.get("choices") or [] - # multi-select support: submit comma-joined list of checked choices - if state.get("multi_select"): - indices = state.get("selected_indices") - if not indices: - # Nothing checked → submit empty string (parses to []) - state["response_queue"].put("") - self._clarify_state = None - event.app.invalidate() - return - sorted_idx = sorted(indices) - selected_choices = [choices[i] for i in sorted_idx if i < len(choices)] - other_checked = len(choices) in sorted_idx - if other_checked and selected_choices: - # "Other" + real choices: store base choices, switch to freetext - # so the user can type a custom answer that gets appended - self._clarify_multi_base = selected_choices - self._clarify_freetext = True - event.app.invalidate() - return - if selected_choices: - state["response_queue"].put(", ".join(selected_choices)) - self._clarify_state = None - event.app.invalidate() - return - # Only "Other" was checked → switch to freetext - self._clarify_freetext = True - event.app.invalidate() - return - # Original single-select behavior: submit the highlighted choice - if selected < len(choices): - state["response_queue"].put(choices[selected]) - self._clarify_state = None - event.app.invalidate() - else: - # "Other" selected → switch to freetext - self._clarify_freetext = True - event.app.invalidate() - return - - # --- Normal input routing --- - raw_text = event.app.current_buffer.text - if ( - self._tui_multiline_shortcuts - and event.app.current_buffer.cursor_position == len(raw_text) - and _is_backslash_line_continuation(raw_text) - ): - continued = _apply_backslash_line_continuation(raw_text) - event.app.current_buffer.text = continued - event.app.current_buffer.cursor_position = len(continued) - event.app.invalidate() - return - text = raw_text.strip() - has_images = bool(self._attached_images) - if text or has_images: - # Handle /model directly on the UI thread so interactive pickers - # can safely use prompt_toolkit terminal handoff helpers. - if self._should_handle_model_command_inline(text, has_images=has_images): - if not self.process_command(text): - self._should_exit = True - if event.app.is_running: - event.app.exit() - event.app.current_buffer.reset(append_to_history=True) - # Force a repaint: process_command() prints through - # patch_stdout (scrolls output above the prompt) and never - # invalidates the app, so the just-cleared input area can - # keep showing the submitted text until some unrelated - # redraw fires. Every other early-return branch in this - # handler invalidates after reset — match them. - event.app.invalidate() - return - - # Handle /steer while the agent is running immediately on the - # UI thread. Queuing through _pending_input would deadlock the - # steer until after the agent loop finishes (process_loop is - # blocked inside self.chat()), which turns /steer into a - # post-run next-turn message — defeating mid-run injection. - # agent.steer() is thread-safe (holds _pending_steer_lock). - if self._should_handle_steer_command_inline(text, has_images=has_images): - self.process_command(text) - event.app.current_buffer.reset(append_to_history=True) - # Force a repaint after clearing the buffer. /steer is - # dispatched mid-run while the agent streams output through - # patch_stdout; process_command() never invalidates the - # app, so without this the submitted "/steer " can - # linger in the input area (looking unsent) and invite an - # accidental re-submit. See issue #34569. - event.app.invalidate() - return - - # Same treatment for /bg and /btw while the agent is - # running. Queuing them defeats the entire point of the - # commands: process_loop is blocked inside self.chat(), so the - # side task would only start once the foreground turn it was - # meant to run alongside has already finished (#75221). The - # foreground turn is left alone: no interrupt, no steer. - if self._should_handle_background_command_inline( - text, has_images=has_images - ): - self.process_command(text) - event.app.current_buffer.reset(append_to_history=True) - # Repaint for the same reason as the /steer branch above: - # process_command() prints through patch_stdout and never - # invalidates the app, so the submitted text can linger in - # the input area looking unsent. - event.app.invalidate() - return - - # Snapshot and clear attached images - images = list(self._attached_images) - self._attached_images.clear() - event.app.invalidate() - # Bundle text + images as a tuple when images are present - payload = (text, images) if images else text - # A bang command is treated like a slash command while the - # agent is busy: it must never be routed into steer/redirect - # (which would inject `!git status` into the model's context as - # a prompt). It queues and runs locally once the loop drains. - _is_local_dispatch = bool(text) and ( - _looks_like_slash_command(text) or text.strip().startswith("!") - ) - if self._agent_running and not _is_local_dispatch: - _effective_mode = self.busy_input_mode - redirected = False - if _effective_mode == "steer": - # Route Enter through /steer — inject mid-run after the - # next tool call. Images can't ride along (steer only - # appends text), so fall back to queue when images are - # attached. If the agent lacks steer() or rejects the - # payload, also fall back to queue so nothing is lost. - if images or not text: - _effective_mode = "queue" - else: - accepted = False - try: - if self.agent is not None and hasattr(self.agent, "steer"): - accepted = bool(self.agent.steer(text)) - except Exception as exc: - _cprint(f" {_DIM}Steer failed ({exc}) — queued for next turn.{_RST}") - accepted = False - if accepted: - preview = text[:80] + ("..." if len(text) > 80 else "") - _cprint(f" {_ACCENT}⏩ Steered: '{preview}'{_RST}") - else: - _effective_mode = "queue" - if _effective_mode == "queue": - # Queue for the next turn instead of interrupting - self._pending_input.put(payload) - preview = text if text else f"[{len(images)} image{'s' if len(images) != 1 else ''} attached]" - _cprint(f" Queued for the next turn: {preview[:80]}{'...' if len(preview) > 80 else ''}") - elif _effective_mode == "interrupt": - if not images and text: - try: - if ( - self.agent is not None - and getattr( - self.agent, - "_supports_active_turn_redirect", - False, - ) - is True - and hasattr(self.agent, "redirect") - ): - redirected = bool(self.agent.redirect(text)) - except Exception: - redirected = False - if redirected: - preview = text[:80] + ("..." if len(text) > 80 else "") - _cprint(f" {_ACCENT}↪ Redirected current turn: '{preview}'{_RST}") - else: - # Compatibility path for older agents, multimodal - # follow-ups, or a turn that finished in the race. - self._interrupt_queue.put(payload) - try: - _dbg = _hermes_home / "interrupt_debug.log" - with open(_dbg, "a", encoding="utf-8") as _f: - _f.write(f"{time.strftime('%H:%M:%S')} ENTER: queued interrupt msg={str(payload)[:60]!r}, " - f"agent_running={self._agent_running}\n") - except Exception: - pass - # First-touch onboarding: on the very first busy-while-running - # event for this install, print a one-line tip explaining the - # /busy knob. Flag persists to config.yaml and never fires - # again. Guarded for exceptions so onboarding can't break - # the input loop. - try: - from agent.onboarding import ( - BUSY_INPUT_FLAG, - busy_input_hint_cli, - is_seen, - mark_seen, - ) - if not is_seen(CLI_CONFIG, BUSY_INPUT_FLAG): - _hint_mode = "redirect" if redirected else _effective_mode - _cprint(f" {_DIM}{busy_input_hint_cli(_hint_mode)}{_RST}") - mark_seen(_hermes_home / "config.yaml", BUSY_INPUT_FLAG) - CLI_CONFIG.setdefault("onboarding", {}).setdefault("seen", {})[BUSY_INPUT_FLAG] = True - except Exception: - pass - else: - self._pending_input.put(payload) - # History stores real pasted content, not the placeholder, so - # up-arrow recall restores the actual text. - self._inline_pastes(event.app.current_buffer) - event.app.current_buffer.reset(append_to_history=True) - - def _tui_handle_paste(self, event): - """Handle terminal paste — detect clipboard images. - - When the terminal supports bracketed paste, Ctrl+V / Cmd+V - triggers this with the pasted text. We only auto-attach a - clipboard image for image-only/empty paste gestures so text - pastes and dictation do not accidentally attach stale images. - - Large pastes (5+ lines) are collapsed to a file reference - placeholder while preserving any existing user text in the - buffer. - """ - # Diagnostic canary: measure how long the paste handler blocks - # the prompt_toolkit event loop. If this exceeds ~500ms we log - # it so recurring "CLI freezes on paste" reports (issue #16263, - # macOS Tahoe 26 + iTerm2/Ghostty) arrive with data attached. - _paste_handler_start = time.perf_counter() - _paste_raw_size = len(event.data or "") - pasted_text = event.data or "" - # Normalise line endings — Windows \r\n and old Mac \r both become \n - # so the 5-line collapse threshold and display are consistent. - pasted_text = pasted_text.replace('\r\n', '\n').replace('\r', '\n') - pasted_text = _strip_leaked_bracketed_paste_wrappers(pasted_text) - pasted_text, _had_mouse_reports = _strip_leaked_terminal_responses_with_meta(pasted_text) - if _had_mouse_reports: - self._recover_terminal_input_modes(reason="mouse reports leaked into bracketed paste payload") - if _should_auto_attach_clipboard_image_on_paste(pasted_text) and self._try_attach_clipboard_image(): - event.app.invalidate() - if pasted_text: - # Sanitize surrogate characters (e.g. from Word/Google Docs paste) before writing - from run_agent import _sanitize_surrogates - pasted_text = _sanitize_surrogates(pasted_text) - line_count = pasted_text.count('\n') - buf = event.current_buffer - threshold = self.config.get("paste_collapse_threshold", 5) - char_threshold = self.config.get("paste_collapse_char_threshold", 2000) - lines_hit = threshold > 0 and line_count >= threshold - chars_hit = char_threshold > 0 and len(pasted_text) >= char_threshold - if (lines_hit or chars_hit) and not buf.text.strip().startswith('/'): - self._tui_paste_counter[0] += 1 - paste_dir = _hermes_home / "pastes" - paste_dir.mkdir(parents=True, exist_ok=True) - paste_file = paste_dir / f"paste_{self._tui_paste_counter[0]}_{datetime.now().strftime('%H%M%S')}.txt" - paste_file.write_text(pasted_text, encoding="utf-8") - logger.info("Collapsed paste #%d: %d lines, %d chars -> %s", self._tui_paste_counter[0], line_count + 1, len(pasted_text), paste_file) - placeholder = f"[Pasted text #{self._tui_paste_counter[0]}: {line_count + 1} lines \u2192 {paste_file}]" - prefix = "" - if buf.cursor_position > 0 and buf.text[buf.cursor_position - 1] != '\n': - prefix = "\n" - self._tui_paste_just_collapsed[0] = True - buf.insert_text(prefix + placeholder) - else: - buf.insert_text(pasted_text) - _paste_handler_elapsed_ms = (time.perf_counter() - _paste_handler_start) * 1000.0 - if _paste_handler_elapsed_ms > 500.0: - logger.warning( - "Slow bracketed-paste handler: %.1fms to process %d bytes " - "(%d lines) on %s. If the input becomes unresponsive after " - "this, attach this log line to the bug report.", - _paste_handler_elapsed_ms, - _paste_raw_size, - pasted_text.count('\n') + 1 if pasted_text else 0, - sys.platform, - ) - - def _tui_on_text_changed(self, buf): - """Detect large pastes and collapse them to a file reference. - - When bracketed paste is available, handle_paste collapses - large pastes directly. This handler is a fallback for - terminals without bracketed paste support. - - Two heuristics (either triggers collapse): - 1. Many characters added at once (chars_added > 1) — works - when the terminal delivers the paste in one event-loop tick. - 2. Newline count jumped by 4+ in a single text-change event — - catches terminals that feed characters individually but - still batch newlines. Alt+Enter only adds 1 newline per - event so it never triggers this. - """ - text = _strip_leaked_bracketed_paste_wrappers(buf.text) - text, _had_mouse_reports = _strip_leaked_terminal_responses_with_meta(text) - if _had_mouse_reports: - self._recover_terminal_input_modes(reason="mouse reports leaked into prompt buffer") - if text != buf.text: - cursor = min(buf.cursor_position, len(text)) - self._tui_paste_just_collapsed[0] = True - buf.text = text - buf.cursor_position = cursor - self._tui_prev_text_len[0] = len(text) - self._tui_prev_newline_count[0] = text.count('\n') - return - chars_added = len(text) - self._tui_prev_text_len[0] - self._tui_prev_text_len[0] = len(text) - if self._tui_paste_just_collapsed[0] or self._skip_paste_collapse: - self._tui_paste_just_collapsed[0] = False - self._skip_paste_collapse = False - self._tui_prev_newline_count[0] = text.count('\n') - return - line_count = text.count('\n') - newlines_added = line_count - self._tui_prev_newline_count[0] - self._tui_prev_newline_count[0] = line_count - is_paste = chars_added > 1 or newlines_added >= 4 - threshold = self.config.get("paste_collapse_threshold_fallback", 5) - char_threshold = self.config.get("paste_collapse_char_threshold", 2000) - lines_hit = threshold > 0 and line_count >= threshold - chars_hit = char_threshold > 0 and len(text) >= char_threshold - if (lines_hit or chars_hit) and is_paste and not text.startswith('/'): - self._tui_paste_counter[0] += 1 - paste_dir = _hermes_home / "pastes" - paste_dir.mkdir(parents=True, exist_ok=True) - paste_file = paste_dir / f"paste_{self._tui_paste_counter[0]}_{datetime.now().strftime('%H%M%S')}.txt" - paste_file.write_text(text, encoding="utf-8") - logger.info("Collapsed paste #%d: %d lines, %d chars -> %s (fallback)", self._tui_paste_counter[0], line_count + 1, len(text), paste_file) - self._tui_paste_just_collapsed[0] = True - buf.text = f"[Pasted text #{self._tui_paste_counter[0]}: {line_count + 1} lines \u2192 {paste_file}]" - buf.cursor_position = len(buf.text) - - def _tui_handle_prompt_stash(self, event): - """Ctrl+S: stash the current draft, or restore/browse a stashed one. - - - Composer has content → push it onto the stash and clear the input. - - Composer empty, one stashed draft → pop it straight back. - - Composer empty, several stashed → open the browse panel. - - Browse panel open → close it. - - Pushing onto a stack (rather than a single slot) is what makes - repeated Ctrl+S safe: a second stash never silently overwrites the - first, both stay reachable in the panel. - """ - from hermes_cli.prompt_stash import ( - ACTION_OPEN_PANEL, - ACTION_RESTORED, - ACTION_STASHED, - resolve_ctrl_s, - ) - - buf = event.app.current_buffer - action, payload = resolve_ctrl_s( - self._prompt_stash, buf.text, self._attached_images - ) - - if action == ACTION_STASHED: - # reset() (not `text = ""`) so completion state, selection, and - # the undo stack are cleared along with the text. - buf.reset() - self._attached_images.clear() - elif action == ACTION_RESTORED: - self._tui_restore_stash_payload(event, payload) - elif action == ACTION_OPEN_PANEL: - pass # resolve_ctrl_s already flipped panel_open - - event.app.invalidate() - - def _tui_handle_stash_panel_restore(self, event): - """Enter in the browse panel restores the highlighted draft.""" - payload = self._prompt_stash.restore_at_cursor() - self._tui_restore_stash_payload(event, payload) - event.app.invalidate() - - def _tui_history_up(self, event): - """Up arrow: browse history when on first line, else move cursor up.""" - buf = event.app.current_buffer - self._tui_recall_without_recollapse(buf, lambda: buf.auto_up(count=event.arg)) - - def _tui_history_down(self, event): - """Down arrow: browse history when on last line, else move cursor down.""" - buf = event.app.current_buffer - self._tui_recall_without_recollapse(buf, lambda: buf.auto_down(count=event.arg)) - - def _tui_image_bar_fragments(self): - if not self._attached_images: - return [] - badges = _format_image_attachment_badges( - self._attached_images, - self._image_counter, - ) - return [("class:image-badge", f" {badges} ")] - - def _tui_voice_status_fragments(self): - return self._get_voice_status_fragments() - - def _tui_spinner_text(self): - spinner_line = self._render_spinner_text() - if not spinner_line: - return [] - return [('class:hint', spinner_line)] - - def _tui_spinner_height(self): - return self._spinner_widget_height() - - def _tui_hint_height(self): - if self._sudo_state or self._secret_state or self._approval_state or self._slash_confirm_state or self._clarify_state or self._command_running: - return 1 - # Keep a spacer while the agent runs on roomy terminals, but reclaim - # the row on narrow/mobile screens where every line matters. - return self._agent_spacer_height() - def _tui_print_startup(self): """Startup output: light-mode probe, banner, advisories, resume/welcome lines, tips.""" # Detect light/dark terminal mode now (before pt grabs the tty). @@ -20448,692 +8000,6 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._startup_skills_line_shown = True self._console_print() - def _tui_init_run_state(self): - """Reset the per-run REPL state (queues, modal states, voice state, config watcher).""" - # State for async operation - self._agent_running = False - self._pending_input = queue.Queue() # For normal input (commands + new queries) - self._interrupt_queue = queue.Queue() # For messages typed while agent is running - # Seeded -q handoff: main() can't put directly into _pending_input - # (this reinit would discard it), so the seeded first message rides - # in on an attribute and is enqueued into the fresh queue here. - _seed_msg = getattr(self, "_seeded_first_message", None) - if _seed_msg is not None: - self._seeded_first_message = None - self._pending_input.put(_seed_msg) - # See constructor note. Mirrored here for the run() path that skips - # the earlier __init__ branch. - self._last_turn_interrupted = False - self._should_exit = False - self._last_ctrl_c_time = 0 # Track double Ctrl+C for force exit - - # Give plugin manager a CLI reference so plugins can inject messages - from hermes_cli.plugins import get_plugin_manager - get_plugin_manager()._cli_ref = self - - # Config file watcher — detect mcp_servers changes and auto-reload - from hermes_cli.config import get_config_path as _get_config_path - _cfg_path = _get_config_path() - self._config_mtime: float = _cfg_path.stat().st_mtime if _cfg_path.exists() else 0.0 - self._config_mcp_servers: dict = self.config.get("mcp_servers") or {} - self._last_config_check: float = 0.0 # monotonic time of last check - - # Clarify tool state: interactive question/answer with the user. - # When the agent calls the clarify tool, _clarify_state is set and - # the prompt_toolkit UI switches to a selection mode. - self._clarify_state = None # dict with question, choices, selected, response_queue - self._clarify_freetext = False # True when user chose "Other" and is typing - self._clarify_deadline = 0 # monotonic timestamp when the clarify times out - - # Sudo password prompt state (similar mechanism to clarify) - self._sudo_state = None # dict with response_queue when active - self._sudo_deadline = 0 - self._modal_input_snapshot = None - - # Dangerous command approval state (similar mechanism to clarify) - self._approval_state = None # dict with command, description, choices, selected, response_queue - self._approval_deadline = 0 - self._approval_lock = threading.Lock() # serialize concurrent approval prompts (delegation race fix) - - # Destructive slash-command confirmation state (/new, /clear, /undo). - # These prompts are answered through the prompt_toolkit composer, not - # raw input(), so the option labels stay visible and Enter does not EOF - # the whole app. - self._slash_confirm_state = None - self._slash_confirm_deadline = 0 - - # Slash command loading state - self._command_running = False - self._command_blocks_input = False - self._command_status = "" - - # Secure secret capture state for skill setup - self._secret_state = None # dict with var_name, prompt, metadata, response_queue - self._secret_deadline = 0 - - # Clipboard image attachments (paste images into the CLI) - self._attached_images: list[Path] = [] - self._image_counter = 0 - - # Voice mode state (protected by _voice_lock for cross-thread access) - self._voice_lock = threading.Lock() - self._voice_mode = False # Whether voice mode is enabled - self._voice_tts = False # Whether TTS output is enabled - self._voice_recorder = None # AudioRecorder instance (lazy init) - self._voice_recording = False # Whether currently recording - self._voice_processing = False # Whether STT is in progress - self._voice_continuous = False # Whether to auto-restart after agent responds - self._voice_tts_done = threading.Event() # Signals TTS playback finished - self._voice_tts_done.set() # Initially "done" (no TTS pending) - self._voice_tts_stop = None # active streaming pipeline's stop event - self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption - self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780) - self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip - - if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": - self._install_tool_callbacks() - - if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": - self._ensure_tirith_security() - - def _tui_build_key_bindings(self): - """Build the prompt_toolkit KeyBindings for the REPL input area.""" - # Key bindings for the input area - kb = KeyBindings() - - _multiline_shortcuts_enabled = _cli_multiline_shortcuts_enabled(self.config or CLI_CONFIG) - self._tui_multiline_shortcuts = _multiline_shortcuts_enabled - - from prompt_toolkit.keys import Keys as _IgnoreKeys - - kb.add(_IgnoreKeys.Ignore, eager=True)(self._tui_handle_ignored_terminal_sequence) - - _bind_prompt_submit_keys( - kb, - self._tui_handle_enter, - multiline_shortcuts_enabled=_multiline_shortcuts_enabled, - ) - - kb.add('escape', 'enter')(self._tui_insert_newline) - - # Ctrl+J inserts a newline (matches Claude Code / Codex / OpenCode). - # Windows Terminal delivers Ctrl+Enter as the same c-j code, so this - # covers Ctrl+Enter there. display.cli_multiline_shortcuts: false - # restores legacy c-j submit on unusual POSIX PTYs where Enter is LF. - if _multiline_shortcuts_enabled or _preserve_ctrl_enter_newline(): - kb.add('c-j')(self._tui_insert_newline) - - # VSCode/Cursor bind Ctrl+G to "Find Next" at the editor level, so - # the keystroke never reaches the embedded terminal. Alt+G is unbound - # in those IDEs and arrives here as ('escape', 'g') — register it as - # a fallback so the editor handoff works inside Cursor/VSCode too. - _editor_filter = Condition( - lambda: not self._clarify_state and not self._approval_state and not self._sudo_state and not self._secret_state - ) - - kb.add('c-g', filter=_editor_filter)(kb.add('escape', 'g', filter=_editor_filter)(self._tui_handle_open_in_editor)) - - # --- Ctrl+S prompt stash ------------------------------------------- - # Park a half-written draft, send something else, then bring the draft - # back. Suppressed while a modal prompt owns the composer (sudo / - # secret / approval / clarify) so Ctrl+S can't stash a password. - _stash_filter = Condition( - lambda: not self._clarify_state - and not self._approval_state - and not self._sudo_state - and not self._secret_state - and not self._slash_confirm_state - and not self._model_picker_state - ) - _stash_panel_filter = Condition( - lambda: self._prompt_stash.panel_open and bool(len(self._prompt_stash)) - ) - - kb.add('c-s', filter=_stash_filter)(self._tui_handle_prompt_stash) - - kb.add('up', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_up) - - kb.add('down', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_down) - - kb.add('enter', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_restore) - - kb.add('d', filter=_stash_panel_filter, eager=True)(kb.add('D', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_delete)) - - kb.add('escape', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_close) - - kb.add('tab', eager=True)(self._tui_handle_tab) - - # --- Clarify tool: arrow-key navigation for multiple-choice questions --- - - kb.add('up', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_clarify_up) - - kb.add('down', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_clarify_down) - - # multi-select support: Space toggles the checkbox at the current cursor position - kb.add('space', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext and self._clarify_state.get("multi_select")))(self._tui_clarify_toggle) - - # Batch clarify: Tab cycles the active question (any-order answering; - # moving onto an answered question lets the user re-answer it before - # the batch completes). Registered after the generic tab handler so - # this filtered binding wins while the batch panel is open. - kb.add('tab', filter=Condition(lambda: bool(self._clarify_state) and bool(self._clarify_state.get("questions")) and not self._clarify_freetext), eager=True)(self._tui_clarify_batch_tab) - - # Shift-Tab walks backwards through the questions. - kb.add('s-tab', filter=Condition(lambda: bool(self._clarify_state) and bool(self._clarify_state.get("questions")) and not self._clarify_freetext), eager=True)(self._tui_clarify_batch_backtab) - - # Number keys for quick clarify selection (1-9, 0 for 10th item) - - for _num in range(10): - # 1-9 select items 0-8, 0 selects item 9 (10thitem) - _idx = 9 if _num == 0 else _num - 1 - kb.add(str(_num), filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_make_clarify_number_handler(_idx)) - - # --- Dangerous command approval: arrow-key navigation --- - - kb.add('up', filter=Condition(lambda: bool(self._approval_state)))(self._tui_approval_up) - - kb.add('down', filter=Condition(lambda: bool(self._approval_state)))(self._tui_approval_down) - - # --- Slash-command confirmation: arrow-key navigation --- - kb.add('up', filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_slash_confirm_up) - - kb.add('down', filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_slash_confirm_down) - - # --- /model picker: arrow-key navigation --- - kb.add('up', filter=Condition(lambda: bool(self._model_picker_state)))(self._tui_model_picker_up) - - kb.add('down', filter=Condition(lambda: bool(self._model_picker_state)))(self._tui_model_picker_down) - - def _model_picker_typing_active() -> bool: - # Type-to-filter is only live on the model stage (concrete list). - st = self._model_picker_state - return bool(st) and st.get("stage") == "model" - - # Printable ASCII (space through ~) narrows the model list as you type. - import string as _string - for _ch in _string.digits + _string.ascii_letters + "-_.:/ ": - kb.add(_ch, filter=Condition(_model_picker_typing_active))( - self._tui_make_model_filter_char_handler(_ch) - ) - - kb.add('backspace', filter=Condition(_model_picker_typing_active))(self._tui_model_picker_filter_backspace) - - kb.add('escape', filter=Condition(lambda: bool(self._model_picker_state)), eager=True)(self._tui_model_picker_escape) - - # --- Ctrl+P command palette keybindings --- - def _palette_active() -> bool: - return bool(self._command_palette_state) - - kb.add('c-p', filter=Condition(lambda: not self._command_palette_state and not self._model_picker_state and not self._clarify_state and not self._approval_state and not self._slash_confirm_state and not self._sudo_state and not self._secret_state))(self._tui_open_command_palette) - - kb.add('up', filter=Condition(_palette_active))(self._tui_command_palette_up) - - kb.add('down', filter=Condition(_palette_active))(self._tui_command_palette_down) - - kb.add('enter', filter=Condition(_palette_active))(self._tui_command_palette_enter) - - kb.add('backspace', filter=Condition(_palette_active))(self._tui_command_palette_backspace) - - kb.add('escape', filter=Condition(_palette_active), eager=True)(self._tui_command_palette_escape) - - import string as _pstring - for _pch in _pstring.digits + _pstring.ascii_letters + "-_.:/ ": - kb.add(_pch, filter=Condition(_palette_active))(self._tui_make_palette_char_handler(_pch)) - - # Number keys for quick approval selection (1-9, 0 for 10th item) - - for _num in range(10): - # 1-9 select items 0-8, 0 selects item 9 (10th item) - _idx = 9 if _num == 0 else _num - 1 - kb.add(str(_num), filter=Condition(lambda: bool(self._approval_state)))(self._tui_make_approval_number_handler(_idx)) - - # Number keys for quick slash-confirm selection (1-9, 0 for 10th item) - - for _num in range(10): - _idx = 9 if _num == 0 else _num - 1 - kb.add(str(_num), filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_make_slash_confirm_number_handler(_idx)) - - # --- History navigation: up/down browse history in normal input mode --- - # The TextArea is multiline, so by default up/down only move the cursor. - # Buffer.auto_up/auto_down handle both: cursor movement when multi-line, - # history browsing when on the first/last line (or single-line input). - _normal_input = Condition( - lambda: not self._clarify_state and not self._approval_state and not self._slash_confirm_state and not self._sudo_state and not self._secret_state and not self._model_picker_state and not self._command_palette_state - ) - - kb.add('up', filter=_normal_input)(self._tui_history_up) - - kb.add('down', filter=_normal_input)(self._tui_history_down) - - kb.add('c-l')(self._tui_handle_ctrl_l) - - kb.add('c-c')(self._tui_handle_ctrl_c) - - # Ctrl+Shift+C: no binding needed. Terminal emulators (GNOME Terminal, - # iTerm2, kitty, Windows Terminal, etc.) intercept Ctrl+Shift+C before - # the keystroke reaches the application's stdin — prompt_toolkit never - # sees it, and prompt_toolkit's key spec parser doesn't even recognise - # 'c-S-c' anyway (the Shift modifier is meaningless on control-sequence - # keys). #19884 added a handler for this; #19895 patched the resulting - # startup crash with try/except. Both were based on a misreading of how - # terminal key events propagate. Deleting the dead handler outright. - - kb.add('c-q')(self._tui_handle_ctrl_q) - - kb.add('c-d')(self._tui_handle_ctrl_d) - - _modal_prompt_active = Condition( - lambda: bool(self._secret_state or self._sudo_state or self._slash_confirm_state) - ) - - kb.add('escape', filter=_modal_prompt_active, eager=True)(self._tui_handle_escape_modal) - - kb.add('escape', 'escape', filter=~_modal_prompt_active)(self._tui_handle_double_escape) - - kb.add('c-z')(self._tui_handle_ctrl_z) - - # Voice push-to-talk key: configurable via config.yaml (voice.record_key) - # Default: Ctrl+B (avoids conflict with Ctrl+R readline reverse-search). - # Config spellings (ctrl/control/alt/option/opt) are normalized to - # prompt_toolkit's c-x / a-x format via ``normalize_voice_record_key_for_prompt_toolkit`` - # so the same config value binds identically in the TUI and CLI - # (Copilot round-9 review on #19835). ``super``/``win``/``windows`` - # configs silently fall back to the default here since prompt_toolkit - # has no super modifier — log a warning so users notice the - # TUI/CLI split instead of a silent mismatch (round-11). - _raw_key: object = "ctrl+b" - try: - from hermes_cli.config import load_config - from hermes_cli.voice import ( - normalize_voice_record_key_for_prompt_toolkit, - pt_key_to_sequence, - voice_record_key_from_config, - ) - _raw_key = voice_record_key_from_config(load_config()) - _voice_key = normalize_voice_record_key_for_prompt_toolkit(_raw_key) - if ( - isinstance(_raw_key, str) - and _raw_key.strip().lower().split("+", 1)[0].strip() in {"super", "win", "windows"} - and _voice_key == "c-b" - ): - logger.warning( - "voice.record_key %r uses a TUI-only modifier (super/win); " - "CLI fell back to Ctrl+B. Use ctrl+ or alt+ for " - "cross-runtime parity.", - _raw_key, - ) - except Exception: - _voice_key = "c-b" - - # Cache the UI label here — same ``_raw_key`` that drives the - # prompt_toolkit binding below. Every status / placeholder / - # recording-hint render reads this cached value so display can - # never drift from the live keybinding even if the user edits - # voice.record_key mid-session (Copilot round-13 on #19835). - self.set_voice_record_key_cache(_raw_key) - - kb.add(*pt_key_to_sequence(_voice_key))(self._tui_handle_voice_record) - from prompt_toolkit.keys import Keys - - kb.add(Keys.BracketedPaste, eager=True)(self._tui_handle_paste) - - kb.add('c-v')(self._tui_handle_ctrl_v) - - kb.add('escape', 'v')(self._tui_handle_alt_v) - return kb - - def _tui_build_layout(self, kb): - """Build the TUI widgets, Layout and Style; registers wrapper keybindings on ``kb``.""" - # Dynamic prompt: shows Hermes symbol when agent is working, - # or answer prompt when clarify freetext mode is active. - cli_ref = self - - def get_prompt(): - return cli_ref._get_tui_prompt_fragments() - - # Create the input area with multiline (Alt+Enter), autocomplete, and paste handling - from prompt_toolkit.auto_suggest import AutoSuggestFromHistory - from prompt_toolkit.completion import ThreadedCompleter - - - _completer = SlashCommandCompleter( - skill_commands_provider=lambda: get_skill_commands(), - command_filter=cli_ref._command_available, - skill_bundles_provider=lambda: get_skill_bundles(), - ) - input_area = TextArea( - height=Dimension(min=1, max=8, preferred=1), - prompt=get_prompt, - style='class:input-area', - multiline=True, - wrap_lines=True, - read_only=Condition(lambda: bool(cli_ref._command_blocks_input)), - history=FileHistory(str(self._history_file)), - # complete_while_typing fires the completer on every keystroke. The - # completer does blocking work — fuzzy @-file indexing shells out to - # rg/fd (up to a 2s timeout) and path completion hits os.listdir/stat - # — so running it inline would stall the render loop on each key (very - # noticeable on WSL2/slow filesystems). ThreadedCompleter moves it off - # the UI event loop, keeping typing responsive. - completer=ThreadedCompleter(_completer), - complete_while_typing=True, - auto_suggest=SlashCommandAutoSuggest( - history_suggest=AutoSuggestFromHistory(), - completer=_completer, - ), - ) - # Keep prompt_toolkit on its simple tempfile path. Setting - # buffer.tempfile = "prompt.md" triggers its complex-tempfile branch, - # which tries to mkdir() the mkdtemp() directory again and raises - # EEXIST. The suffix keeps markdown highlighting without that bug. - input_area.buffer.tempfile_suffix = '.md' - - # Dynamic height: accounts for both explicit newlines AND visual - # wrapping of long lines so the input area always fits its content. - def _input_height(): - try: - from prompt_toolkit.application import get_app - - doc = input_area.buffer.document - try: - terminal_columns = get_app().output.get_size().columns - except Exception: - terminal_columns = shutil.get_terminal_size((80, 24)).columns - return _estimate_tui_input_height( - doc.lines, - self._get_tui_prompt_text(), - terminal_columns, - ) - except Exception: - return 1 - - input_area.window.height = _input_height - - # Paste collapsing: detect large pastes and save to temp file - self._tui_paste_counter = [0] - self._tui_prev_text_len = [0] - self._tui_prev_newline_count = [0] - self._tui_paste_just_collapsed = [False] - self._skip_paste_collapse = False - - input_area.buffer.on_text_changed += self._tui_on_text_changed - - # --- Input processors for password masking and inline placeholder --- - - # Mask input with '*' when the sudo password prompt is active - input_area.control.input_processors.append( - ConditionalProcessor( - PasswordProcessor(), - filter=Condition( - lambda: bool(cli_ref._sudo_state) or bool(cli_ref._secret_state) - ), - ) - ) - - class _PlaceholderProcessor(Processor): - """Render grayed-out placeholder text inside the input when empty.""" - def __init__(self, get_text): - self._get_text = get_text - - def apply_transformation(self, ti): - if not ti.document.text and ti.lineno == 0: - text = self._get_text() - if text: - # Append after existing fragments (preserves the ❯ prompt) - return Transformation(fragments=ti.fragments + [('class:placeholder', text)]) - return Transformation(fragments=ti.fragments) - - input_area.control.input_processors.append(_PlaceholderProcessor(self._tui_placeholder_text)) - - # Hint line above input: shown only for interactive prompts that need - # extra instructions (sudo countdown, approval navigation, clarify). - # The agent-running interrupt hint is now an inline placeholder above. - - spinner_widget = Window( - content=FormattedTextControl(self._tui_spinner_text), - height=self._tui_spinner_height, - wrap_lines=True, - ) - - # Petdex mascot — right-aligned Kitty placeholder or half-block sprite - # above the prompt. Collapses to height 0 when no pet is enabled. - # The animation thread queues virtual Kitty frames; after_render - # writes them out-of-band while prompt_toolkit owns the placeholder grid. - self._pet_widget = Window( - content=FormattedTextControl(self._pet_fragments), - height=self._pet_widget_height, - align=WindowAlign.RIGHT, - ) - - spacer = Window( - content=FormattedTextControl(self._tui_hint_text), - height=self._tui_hint_height, - ) - - # --- Clarify tool: dynamic display widget for questions + choices --- - - clarify_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_clarify_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._clarify_state is not None), - ) - - # --- Sudo password: display widget --- - - sudo_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_sudo_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._sudo_state is not None), - ) - - secret_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_secret_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._secret_state is not None), - ) - - # --- Dangerous command approval: display widget --- - - approval_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_approval_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._approval_state is not None), - ) - - slash_confirm_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_slash_confirm_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._slash_confirm_state is not None), - ) - - # --- /model picker: display widget --- - - model_picker_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_model_picker_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._model_picker_state is not None), - ) - - # --- Ctrl+P command palette: display widget --- - - command_palette_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_command_palette_display_fragments), - wrap_lines=True, - ), - filter=Condition(lambda: cli_ref._command_palette_state is not None), - ) - - # Horizontal rules above and below the input. - # On narrow/mobile terminals we keep the top separator for structure but - # hide the bottom one to recover a full row for conversation content. - input_rule_top = Window( - char='─', - height=lambda: cli_ref._tui_input_rule_height("top"), - style='class:input-rule', - ) - input_rule_bot = Window( - char='─', - height=lambda: cli_ref._tui_input_rule_height("bottom"), - style='class:input-rule', - ) - - # Image attachment indicator — shows badges like [📎 Image #1] above input - cli_ref = self - - image_bar = Window( - content=FormattedTextControl(self._tui_image_bar_fragments), - height=Condition(lambda: bool(cli_ref._attached_images)), - ) - - # Persistent voice mode status bar (visible only when voice mode is on) - - voice_status_bar = ConditionalContainer( - Window( - FormattedTextControl(self._tui_voice_status_fragments), - height=1, - ), - filter=Condition(lambda: cli_ref._voice_mode), - ) - - status_bar = ConditionalContainer( - Window( - content=FormattedTextControl(lambda: cli_ref._get_status_bar_fragments()), - height=1, - # Prevent fragments that overflow the terminal width from - # wrapping onto a second line, which causes the status bar to - # appear duplicated (one full + one partial row) during long - # sessions, especially on SSH where shutil.get_terminal_size - # may return stale values. _get_status_bar_fragments now reads - # width from prompt_toolkit's own output object, so fragments - # will always fit; wrap_lines=False is the belt-and-suspenders - # guard against any future width mismatch. - wrap_lines=False, - ), - filter=Condition( - lambda: cli_ref._status_bar_visible - and not getattr(cli_ref, "_status_bar_suppressed_after_resize", False) - ), - ) - - # Stash browse panel — appears just above the status bar when the user - # presses Ctrl+S on an empty composer with 2+ stashed drafts. - - self._stash_panel_widget = ConditionalContainer( - Window( - FormattedTextControl(self._get_stash_panel_display_fragments), - wrap_lines=False, - ), - filter=Condition( - lambda: cli_ref._prompt_stash.panel_open - and bool(len(cli_ref._prompt_stash)) - ), - ) - - # Allow wrapper CLIs to register extra keybindings. - self._register_extra_tui_keybindings(kb, input_area=input_area) - - # Layout: interactive prompt widgets + ruled input at bottom. - # The sudo, approval, and clarify widgets appear above the input when - # the corresponding interactive prompt is active. - completions_menu = CompletionsMenu(max_height=12, scroll_offset=1) - - layout = Layout( - HSplit( - self._build_tui_layout_children( - sudo_widget=sudo_widget, - secret_widget=secret_widget, - approval_widget=approval_widget, - slash_confirm_widget=slash_confirm_widget, - clarify_widget=clarify_widget, - model_picker_widget=model_picker_widget, - command_palette_widget=command_palette_widget, - spinner_widget=spinner_widget, - spacer=spacer, - status_bar=status_bar, - input_rule_top=input_rule_top, - image_bar=image_bar, - input_area=input_area, - input_rule_bot=input_rule_bot, - voice_status_bar=voice_status_bar, - completions_menu=completions_menu, - ) - ) - ) - - # Style for the application - self._tui_style_base = { - # Input area / prompt: empty style strings inherit the - # terminal's default foreground/background, so the typed - # text is readable in both light and dark Terminal.app - # color schemes. (Hardcoding a near-white #FFF8DC made - # input invisible on light backgrounds.) - 'input-area': '', - 'placeholder': '#888888 italic', - 'prompt': '', - 'prompt-working': '#888888 italic', - 'hint': '#888888 italic', - 'status-bar': 'bg:#1a1a2e #C0C0C0', - 'status-bar-strong': 'bg:#1a1a2e #FFD700 bold', - 'status-bar-dim': 'bg:#1a1a2e #8B8682', - 'status-bar-good': 'bg:#1a1a2e #8FBC8F bold', - 'status-bar-warn': 'bg:#1a1a2e #FFD700 bold', - 'status-bar-bad': 'bg:#1a1a2e #FF8C00 bold', - 'status-bar-critical': 'bg:#1a1a2e #FF6B6B bold', - 'status-bar-yolo': 'bg:#1a1a2e #FF4444 bold', - 'status-bar-session-title': 'bg:#FFD700 #1a1a2e bold', - # Bronze horizontal rules around the input area - 'input-rule': '#CD7F32', - # Clipboard image attachment badges - 'image-badge': '#87CEEB bold', - 'completion-menu': 'bg:#1a1a2e #FFF8DC', - 'completion-menu.completion': 'bg:#1a1a2e #FFF8DC', - 'completion-menu.completion.current': 'bg:#333355 #FFD700', - 'completion-menu.meta.completion': 'bg:#1a1a2e #888888', - 'completion-menu.meta.completion.current': 'bg:#333355 #FFBF00', - # Clarify question panel - 'clarify-border': '#CD7F32', - 'clarify-title': '#FFD700 bold', - 'clarify-question': '#FFF8DC bold', - 'clarify-choice': '#AAAAAA', - 'clarify-selected': '#FFD700 bold', - 'clarify-active-other': '#FFD700 italic', - 'clarify-answer': '#98FB98', - 'clarify-countdown': '#CD7F32', - # Sudo password panel - 'sudo-prompt': '#FF6B6B bold', - 'sudo-border': '#CD7F32', - 'sudo-title': '#FF6B6B bold', - 'sudo-text': '#FFF8DC', - # Dangerous command approval panel - 'approval-border': '#CD7F32', - 'approval-title': '#FF8C00 bold', - 'approval-desc': '#FFF8DC bold', - 'approval-cmd': '#AAAAAA italic', - 'approval-choice': '#AAAAAA', - 'approval-selected': '#FFD700 bold', - # Voice mode - 'voice-prompt': '#87CEEB', - 'voice-recording': '#FF4444 bold', - 'voice-processing': '#FFA500 italic', - 'voice-status': 'bg:#1a1a2e #87CEEB', - 'voice-status-recording': 'bg:#1a1a2e #FF4444 bold', - } - style = PTStyle.from_dict(self._build_tui_style_dict()) - return (layout, style) - def run(self): """Run the interactive CLI loop with persistent input at bottom.""" if not self._claim_active_session("cli"): diff --git a/hermes_cli/cli_info_mixin.py b/hermes_cli/cli_info_mixin.py new file mode 100644 index 0000000000..87c3286c5e --- /dev/null +++ b/hermes_cli/cli_info_mixin.py @@ -0,0 +1,1333 @@ +"""Informational views and reload flows for the interactive CLI: banner, help, tools, usage, insights, MCP/skills reload, bang shell + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import concurrent.futures +import logging +import os +import shutil +import threading +import time + +from hermes_constants import is_termux as _is_termux_environment +from rich.markup import escape as _escape +from utils import base_url_hostname + + +class CLIInfoMixin: + """Informational views and reload flows for the interactive CLI: banner, help, tools, usage, insights, MCP/skills reload, bang shell""" + + def show_banner(self): + """Display the welcome banner in Claude Code style.""" + from cli import _build_compact_banner, build_welcome_banner, get_tool_definitions, logger + self.console.clear() + ctx_len = None + if hasattr(self, 'agent') and self.agent and hasattr(self.agent, 'context_compressor'): + ctx_len = self.agent.context_compressor.context_length + + # Auto-compact for narrow terminals — the full banner with caduceus + # + tool list needs ~80 columns minimum to render without wrapping. + term_width = shutil.get_terminal_size().columns + use_compact = self.compact or term_width < 80 + + if use_compact: + self._console_print(_build_compact_banner()) + self._show_status() + else: + # Warm-launch fast path: replay last launch's tool panel when the + # snapshot fingerprint (config.yaml + .env + checkout rev + + # toolsets) is unchanged, skipping the ~0.5-0.9s cold + # get_tool_definitions walk. The agent's REAL tool list is still + # computed fresh at first message; a background refresh below + # re-verifies the snapshot so any drift self-heals next launch. + from hermes_cli.banner import ( + compute_toolset_availability, + load_banner_snapshot, + save_banner_snapshot, + ) + + snapshot = None + try: + snapshot = load_banner_snapshot(self.enabled_toolsets) + except Exception: + snapshot = None + + # Get terminal working directory (where commands will execute) + cwd = os.getenv("TERMINAL_CWD", os.getcwd()) + + if snapshot is not None: + self._defer_tool_warnings = True + toolset_map = snapshot["toolset_map"] + build_welcome_banner( + console=self.console, + model=self.model, + cwd=cwd, + tools=snapshot["tools"], + enabled_toolsets=self.enabled_toolsets, + session_id=self.session_id, + get_toolset_for_tool=lambda name: toolset_map.get(name), + context_length=ctx_len, + provider=self.provider, + availability=snapshot["availability"], + skills_by_category=snapshot.get("skills_by_category"), + ) + + def _refresh_banner_snapshot() -> None: + try: + from model_tools import get_toolset_for_tool + tools = get_tool_definitions( + enabled_toolsets=self.enabled_toolsets, quiet_mode=True + ) + availability = compute_toolset_availability(self.enabled_toolsets) + tmap = { + t["function"]["name"]: get_toolset_for_tool(t["function"]["name"]) + for t in tools + } + for item in availability.get("unavailable_toolsets", []): + for name in item.get("tools", []): + tmap.setdefault( + name, item.get("id", item.get("name", "")) + ) + save_banner_snapshot( + tools, self.enabled_toolsets, availability, tmap + ) + except Exception: + logger.debug("banner snapshot refresh failed", exc_info=True) + + threading.Thread( + target=_refresh_banner_snapshot, + name="banner-snapshot-refresh", + daemon=True, + ).start() + else: + # Cold path: compute everything live, then persist the snapshot + # so the next launch replays it. + from model_tools import get_toolset_for_tool + tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) + availability = compute_toolset_availability(self.enabled_toolsets) + + build_welcome_banner( + console=self.console, + model=self.model, + cwd=cwd, + tools=tools, + enabled_toolsets=self.enabled_toolsets, + session_id=self.session_id, + context_length=ctx_len, + provider=self.provider, + availability=availability, + ) + try: + tmap = { + t["function"]["name"]: get_toolset_for_tool(t["function"]["name"]) + for t in tools + } + for item in availability.get("unavailable_toolsets", []): + for name in item.get("tools", []): + tmap.setdefault(name, item.get("id", item.get("name", ""))) + save_banner_snapshot(tools, self.enabled_toolsets, availability, tmap) + except Exception: + logger.debug("banner snapshot save failed", exc_info=True) + + # Tool discovery is intentionally deferred on the Termux bare prompt + # path; availability warnings are shown once tools are initialized. + # On the snapshot fast path (warm launch), the check walks every + # check_fn (~180ms) — run it in the background refresh thread instead + # and let its output land above the prompt (patch_stdout-safe). + if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": + if getattr(self, "_defer_tool_warnings", False): + threading.Thread( + target=self._show_tool_availability_warnings, + name="tool-availability-warnings", + daemon=True, + ).start() + else: + self._show_tool_availability_warnings() + + # Warn about low context lengths (common with local servers). Keep + # this tied to the runtime guard so guidance cannot drift again. + from agent.model_metadata import MINIMUM_CONTEXT_LENGTH + if ctx_len and ctx_len < MINIMUM_CONTEXT_LENGTH: + self._console_print() + self._console_print( + f"[yellow]⚠️ Context length is only {ctx_len:,} tokens — " + f"this is likely too low for agent use with tools.[/]" + ) + self._console_print( + f"[dim] Hermes needs at least {MINIMUM_CONTEXT_LENGTH:,} tokens. Tool schemas + system prompt use a large fixed prefix.[/]" + ) + base_url = getattr(self, "base_url", "") or "" + from urllib.parse import urlparse as _urlparse + try: + _parsed = _urlparse(base_url if "://" in base_url else f"//{base_url}") + _port = _parsed.port + except ValueError: + _port = None + _host = base_url_hostname(base_url) + if _port == 11434 or "ollama" in _host: + self._console_print( + f"[dim] Ollama fix: OLLAMA_CONTEXT_LENGTH={MINIMUM_CONTEXT_LENGTH} ollama serve[/]" + ) + elif _port == 1234: + self._console_print( + "[dim] LM Studio fix: Set context length in model settings → reload model[/]" + ) + else: + self._console_print( + "[dim] Fix: Set model.context_length in config.yaml, or increase your server's context setting[/]" + ) + + # Warn if the configured model is a Nous Hermes LLM (not agentic) + from hermes_cli.model_switch import is_nous_hermes_non_agentic + + model_name = getattr(self, "model", "") or "" + if is_nous_hermes_non_agentic(model_name): + self._console_print() + self._console_print( + "[bold yellow]⚠ Nous Research Hermes 3 & 4 models are NOT agentic and are not " + "designed for use with Hermes Agent.[/]" + ) + self._console_print( + "[dim] They lack tool-calling capabilities required for agent workflows. " + "Consider using an agentic model (Claude, GPT, Gemini, DeepSeek, etc.).[/]" + ) + self._console_print( + "[dim] Switch with: /model sonnet or /model gpt5[/]" + ) + + # Project-local skills: one-line status. Trusted → show count; + # untrusted-with-skills → point at `hermes skills trust`. Never raises. + try: + from agent.skill_utils import ( + get_project_skills_dirs, + get_untrusted_project_skills_root, + iter_skill_index_files, + ) + _proj_dirs = get_project_skills_dirs() + if _proj_dirs: + _n = sum( + sum(1 for _ in iter_skill_index_files(d, "SKILL.md")) + for d in _proj_dirs + ) + if _n: + self._console_print( + f"[dim]◆ {_n} project skill(s) loaded from this repo[/]" + ) + else: + _untrusted = get_untrusted_project_skills_root() + if _untrusted is not None: + _root, _n = _untrusted + self._console_print( + f"[yellow]◆ {_n} project skill(s) found in {_root} but not " + f"loaded — run `hermes skills trust` to enable them.[/]" + ) + except Exception: + logger.debug("project skills banner notice failed", exc_info=True) + + self._console_print() + + def _fast_command_available(self) -> bool: + try: + from hermes_cli.models import model_supports_fast_mode + except Exception: + return False + agent = getattr(self, "agent", None) + model = getattr(agent, "model", None) or getattr(self, "model", None) + return model_supports_fast_mode(model) + + def _command_available(self, slash_command: str) -> bool: + if slash_command == "/fast": + return self._fast_command_available() + return True + + def show_help(self, arg: str = ""): + """Display help. Bare /help shows categorized core commands with the + skill list collapsed to one line; /help skills lists all skill + commands; /help filters commands by substring. + """ + from cli import ( + ChatConsole, + _BOLD, + _DIM, + _RST, + _accent_hex, + _cprint, + _ensure_skill_commands, + _termux_example_image_path, + get_skill_bundles, + ) + from hermes_cli.commands import COMMANDS_BY_CATEGORY, HELP_SESSION_SUBGROUPS + + arg = (arg or "").strip() + skill_commands = _ensure_skill_commands() + + # /help skills — the full skill-command list (kept out of the default + # view so core commands don't scroll off screen). + if arg.lower() in ("skills", "skill"): + if not skill_commands: + _cprint("\n No skill commands installed.\n") + return + _cprint(f"\n ⚡ {_BOLD}Skill Commands{_RST} ({len(skill_commands)} installed):") + for cmd, info in sorted(skill_commands.items()): + ChatConsole().print( + f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] {_escape(info['description'])}" + ) + _cprint("") + return + + query = arg.lower() if arg else "" + + try: + from hermes_cli.skin_engine import get_active_help_header + header = get_active_help_header("(^_^)? Available Commands") + except Exception: + header = "(^_^)? Available Commands" + header = (header or "").strip() or "(^_^)? Available Commands" + inner_width = 55 + if len(header) > inner_width: + header = header[:inner_width] + _cprint(f"\n{_BOLD}+{'-' * inner_width}+{_RST}") + _cprint(f"{_BOLD}|{header:^{inner_width}}|{_RST}") + _cprint(f"{_BOLD}+{'-' * inner_width}+{_RST}") + + def _emit(cmd: str, desc: str) -> bool: + if not self._command_available(cmd): + return False + if query and query not in cmd.lower() and query not in desc.lower(): + return False + ChatConsole().print( + f" [bold {_accent_hex()}]{cmd:<15}[/] [dim]-[/] {_escape(desc)}" + ) + return True + + for category, commands in COMMANDS_BY_CATEGORY.items(): + if category == "Session": + # Split the oversized Session category into readable sub-groups + # (Session / Context / Background & Automation) in the renderer. + sub_of: dict[str, str] = {} + for _sub, _names in HELP_SESSION_SUBGROUPS.items(): + for _n in _names: + sub_of[f"/{_n}"] = _sub + buckets: dict[str, list[tuple[str, str]]] = {"Session": []} + for _sub in HELP_SESSION_SUBGROUPS: + buckets[_sub] = [] + for cmd, desc in commands.items(): + buckets[sub_of.get(cmd, "Session")].append((cmd, desc)) + for _sub in ("Session", *HELP_SESSION_SUBGROUPS.keys()): + rows = buckets.get(_sub) or [] + printed_header = False + for cmd, desc in rows: + if not self._command_available(cmd): + continue + if query and query not in cmd.lower() and query not in desc.lower(): + continue + if not printed_header: + _cprint(f"\n {_BOLD}── {_sub} ──{_RST}") + printed_header = True + _emit(cmd, desc) + continue + + printed_header = False + for cmd, desc in commands.items(): + if not self._command_available(cmd): + continue + if query and query not in cmd.lower() and query not in desc.lower(): + continue + if not printed_header: + _cprint(f"\n {_BOLD}── {category} ──{_RST}") + printed_header = True + _emit(cmd, desc) + + # Skill commands: collapsed to a one-line pointer by default so the + # 60+ skill entries don't bury the core command reference (C-04). + if query: + # In filter mode, DO include matching skill commands inline. + matched_skills = [ + (cmd, info) for cmd, info in sorted(skill_commands.items()) + if query in cmd.lower() or query in (info.get("description", "").lower()) + ] + if matched_skills: + _cprint(f"\n ⚡ {_BOLD}Skill Commands{_RST} (matching '{arg}'):") + for cmd, info in matched_skills: + ChatConsole().print( + f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] {_escape(info['description'])}" + ) + elif skill_commands: + _cprint( + f"\n ⚡ {_BOLD}Skill Commands{_RST}: {len(skill_commands)} installed " + f"— {_DIM}/help skills{_RST} to list them" + ) + + _bundles_now = get_skill_bundles() + if _bundles_now and not query: + _cprint(f"\n ▣ {_BOLD}Skill Bundles{_RST} ({len(_bundles_now)} installed):") + for cmd, info in sorted(_bundles_now.items()): + skill_count = len(info.get("skills", [])) + desc = info.get("description") or f"Load {skill_count} skills" + ChatConsole().print( + f" [bold {_accent_hex()}]{cmd:<22}[/] [dim]-[/] " + f"{_escape(desc)} [dim]({skill_count} skills)[/]" + ) + + quick_commands = self.config.get("quick_commands", {}) + if quick_commands and not query: + _cprint(f"\n ⚡ {_BOLD}Quick Commands{_RST} ({len(quick_commands)} configured):") + for name, qcmd in sorted(quick_commands.items()): + desc = qcmd.get("description", qcmd.get("type", "")) + ChatConsole().print( + f" [bold {_accent_hex()}]{('/' + name):<22}[/] [dim]-[/] {_escape(desc)}" + ) + + if query: + _cprint(f"\n {_DIM}Filtered by '{arg}' — run /help for the full list.{_RST}\n") + return + + _cprint(f"\n {_DIM}Tip: /help skills lists skill commands · /help filters · Ctrl+P opens the command palette{_RST}") + _cprint(f" {_DIM}Multi-line: Ctrl+J, Alt+Enter, or \\\\+Enter for a new line{_RST}") + _cprint(f" {_DIM}Draft editor: Ctrl+G (Alt+G in VSCode/Cursor){_RST}") + if _is_termux_environment(): + _cprint(f" {_DIM}Attach image: /image {_termux_example_image_path()} or start your prompt with a local image path{_RST}\n") + else: + _cprint(f" {_DIM}Paste image: Alt+V (or /paste){_RST}\n") + + def show_tools(self): + """Display available tools with kawaii ASCII art.""" + from cli import get_tool_definitions, get_toolset_for_tool + # Pre-assembly list: /tools is a discovery/inspection surface, so it + # must show the full catalog including tools deferred behind the + # tool_search bridge (users check this to verify an MCP installed). + tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True, + skip_tool_search_assembly=True) + + if not tools: + print("(;_;) No tools available") + return + + # Header + print() + title = "(^_^)/ Available Tools" + width = 78 + pad = width - len(title) + print("+" + "-" * width + "+") + print("|" + " " * (pad // 2) + title + " " * (pad - pad // 2) + "|") + print("+" + "-" * width + "+") + print() + + # Group tools by toolset + toolsets = {} + for tool in sorted(tools, key=lambda t: t["function"]["name"]): + name = tool["function"]["name"] + toolset = get_toolset_for_tool(name) or "unknown" + if toolset not in toolsets: + toolsets[toolset] = [] + desc = tool["function"].get("description", "") + # First sentence: split on ". " (period+space) to avoid breaking on "e.g." or "v2.0" + desc = desc.split("\n")[0] + if ". " in desc: + desc = desc[:desc.index(". ") + 1] + toolsets[toolset].append((name, desc)) + + # Display by toolset + for toolset in sorted(toolsets.keys()): + print(f" [{toolset}]") + for name, desc in toolsets[toolset]: + print(f" * {name:<20} - {desc}") + print() + + print(f" Total: {len(tools)} tools ヽ(^o^)ノ") + print() + + def show_toolsets(self): + """Display available toolsets with kawaii ASCII art.""" + from cli import get_all_toolsets, get_toolset_info + all_toolsets = get_all_toolsets() + + # Header + print() + title = "(^_^)b Available Toolsets" + width = 58 + pad = width - len(title) + print("+" + "-" * width + "+") + print("|" + " " * (pad // 2) + title + " " * (pad - pad // 2) + "|") + print("+" + "-" * width + "+") + print() + + for name in sorted(all_toolsets.keys()): + info = get_toolset_info(name) + if info: + tool_count = info["tool_count"] + desc = info["description"] + + # Mark if currently enabled + marker = "(*)" if self.enabled_toolsets and name in self.enabled_toolsets else " " + print(f" {marker} {name:<18} [{tool_count:>2} tools] - {desc}") + + print() + print(" (*) = currently enabled") + print() + print(" Tip: Use 'all' or '*' to enable all toolsets") + print(" Example: python cli.py --toolsets web,terminal") + print() + + def _handle_whoami_command(self): + """Display slash-command access for the local CLI surface.""" + import getpass + + try: + user_name = getpass.getuser() or "?" + except Exception: + user_name = "?" + + print() + print(" You: cli (local terminal)") + print(f" User: {user_name}") + print(" Tier: unrestricted") + print(" Slash commands: all available") + print() + + def _should_handle_steer_command_inline(self, text: str, has_images: bool = False) -> bool: + """Return True when /steer should be dispatched immediately while the agent is running. + + /steer MUST bypass the normal _pending_input → process_loop path when + the agent is active, because process_loop is blocked inside + self.chat() for the duration of the run. By the time the queued + command is pulled from _pending_input, _agent_running has already + flipped back to False, and process_command() takes the idle + fallback — delivering the steer as a next-turn message instead of + injecting it mid-run. Dispatching inline on the UI thread calls + agent.steer() directly, which is thread-safe (uses _pending_steer_lock). + """ + from cli import _looks_like_slash_command + if not text or has_images or not _looks_like_slash_command(text): + return False + if not getattr(self, "_agent_running", False): + return False + try: + from hermes_cli.commands import resolve_command + base = text.split(None, 1)[0].lower().lstrip('/') + cmd = resolve_command(base) + return bool(cmd and cmd.name == "steer") + except Exception: + return False + + def _should_handle_background_command_inline( + self, text: str, has_images: bool = False + ) -> bool: + """Return True when /bg or /btw should be dispatched while the agent runs. + + Same queue problem /steer had. ``/bg`` exists to start independent + work *without* waiting for the current turn, and ``/btw`` exists to + answer a side question about the in-flight conversation, but a slash + command typed while the agent is busy goes into ``_pending_input``, + and ``process_loop`` is blocked inside ``self.chat()`` for the whole + run. The side task would therefore only start once the foreground + turn has finished, which is the one moment it was not needed. + + Both commands' ``CommandDef`` entries already declare + ``busy_policy="dispatch"``; the gateway honours that, the classic CLI + never consulted it. Dispatching inline on the UI thread starts the + side session immediately and leaves the foreground turn running + untouched: no interrupt, no steer. + """ + from cli import _looks_like_slash_command + if not text or has_images or not _looks_like_slash_command(text): + return False + if not getattr(self, "_agent_running", False): + return False + try: + from hermes_cli.commands import resolve_command + base = text.split(None, 1)[0].lower().lstrip('/') + cmd = resolve_command(base) + return bool(cmd and cmd.name in ("bg", "btw")) + except Exception: + return False + + def handle_bang_shell(self, text: str) -> bool: + """Run a ``!`` submission. Returns True when it was handled. + + Dispatched from the input loop BEFORE slash-command routing and before + anything is queued for the agent, so a bang command never becomes a + turn: no user message, no assistant message, no tool result touches + ``self.conversation_history``. That is what makes ``!`` free — zero + tokens, and role alternation / prompt caching are untouched by + construction. The invariant is covered by + tests/cli/test_bang_shell_mode.py. + + Returns False when the text is not a bang command or when bang mode is + disabled for this context (gateway/cron), letting the caller fall + through to normal routing. + """ + from cli import _rich_text_from_ansi + from hermes_cli.bang_shell import ( + USAGE_HINT, + bang_shell_enabled, + check_bang_approval, + is_bang_command, + parse_bang_command, + resolve_bang_cwd, + run_bang_command, + ) + + if not is_bang_command(text): + return False + if not bang_shell_enabled(): + # Gateway / cron / API contexts: no composer, no human at a + # keyboard, and those users already have their own shells. Let the + # text route normally rather than becoming remote execution. + return False + + command = parse_bang_command(text) + if not command: + # Bare `!` — show what the feature does instead of running an + # empty shell or sending "!" to the model. + self._console_print(f"[dim]{USAGE_HINT}[/]") + return True + + approval = check_bang_approval(command) + if not approval.get("approved"): + message = approval.get("message") or ( + f"Command denied: {approval.get('description', 'flagged as dangerous')}" + ) + self._console_print(f"[bold red]{_escape(str(message))}[/]") + return True + + cwd = resolve_bang_cwd(getattr(self, "session_id", None)) + exit_code = run_bang_command( + command, + cwd=cwd, + writer=lambda line: self._console_print(_rich_text_from_ansi(line)), + ) + if exit_code: + self._console_print(f"[dim]! exited {exit_code}[/]") + return True + + def _show_gateway_status(self): + """Show status of the gateway and connected messaging platforms.""" + from cli import display_hermes_home + from gateway.config import load_gateway_config, Platform + + print() + print("+" + "-" * 60 + "+") + print("|" + " " * 15 + "(✿◠‿◠) Gateway Status" + " " * 17 + "|") + print("+" + "-" * 60 + "+") + print() + + try: + config = load_gateway_config() + + print(" Messaging Platform Configuration:") + print(" " + "-" * 55) + + platform_status = { + Platform.TELEGRAM: ("Telegram", "TELEGRAM_BOT_TOKEN"), + Platform.DISCORD: ("Discord", "DISCORD_BOT_TOKEN"), + Platform.SLACK: ("Slack", "SLACK_BOT_TOKEN"), + Platform.WHATSAPP: ("WhatsApp", "WHATSAPP_ENABLED"), + } + + for platform, (name, env_var) in platform_status.items(): + pconfig = config.platforms.get(platform) + if pconfig and pconfig.enabled: + home = config.get_home_channel(platform) + home_str = f" → {home.name}" if home else "" + print(f" ✓ {name:<12} Enabled{home_str}") + else: + print(f" ○ {name:<12} Not configured ({env_var})") + + print() + print(" Session Reset Policy:") + print(" " + "-" * 55) + policy = config.default_reset_policy + print(f" Mode: {policy.mode}") + print(f" Daily reset at: {policy.at_hour}:00") + print(f" Idle timeout: {policy.idle_minutes} minutes") + + print() + print(" To start the gateway:") + print(" python cli.py --gateway") + print() + print(f" Configuration file: {display_hermes_home()}/config.yaml") + print() + + except Exception as e: + print(f" Error loading gateway config: {e}") + print() + print(" To configure the gateway:") + print(" 1. Set environment variables:") + print(" TELEGRAM_BOT_TOKEN=your_token") + print(" DISCORD_BOT_TOKEN=your_token") + print(f" 2. Or configure settings in {display_hermes_home()}/config.yaml") + print() + + def _print_random_tip(self) -> None: + """Best-effort discovery tip (startup + /clear); never raises.""" + try: + from hermes_cli.tips import get_random_tip + _tip = get_random_tip() + try: + from hermes_cli.skin_engine import get_active_skin + _tip_color = get_active_skin().get_color("banner_dim", "#B8860B") + except Exception: + _tip_color = "#B8860B" + self._console_print(f"[dim {_tip_color}]✦ Tip: {_tip}[/]") + except Exception: + pass + + def _toggle_verbose(self): + """Cycle tool progress mode: off → new → all → verbose → off. + + Tool-progress display (full args / results / think blocks at the + ``verbose`` step) is INDEPENDENT of global DEBUG logging. Cycling + through here does not change ``self.verbose`` or the agent's + ``verbose_logging`` / ``quiet_mode`` — those remain under the + explicit ``-v``/``--verbose`` flag and the ``/verbose-logging`` + toggle. See PR #6a1aa420e for the history that decoupled them. + """ + from cli import _cprint, save_config_value + cycle = ["off", "new", "all", "verbose"] + try: + idx = cycle.index(self.tool_progress_mode) + except ValueError: + idx = 2 # default to "all" + self.tool_progress_mode = cycle[(idx + 1) % len(cycle)] + + # /verbose is the explicit tool-progress control, so cycling it takes + # ownership of the mode back from focus view. Leaving _focus_view_enabled + # set would show a "focus" status-bar badge and hidden-line counts while + # tool lines were visibly printing. Display-only state change. + if getattr(self, "_focus_view_enabled", False): + self._focus_view_enabled = False + self._focus_saved_tool_progress = None + self._focus_hidden_lines = 0 + self._focus_last_counted_tool = None + try: + from hermes_cli.focus_view import FOCUS_CONFIG_KEY + + save_config_value(FOCUS_CONFIG_KEY, False) + except Exception: + pass + + if self.agent: + self.agent.reasoning_callback = self._current_reasoning_callback() + # Keep the live agent's tool_progress_mode in sync so the + # tool_executor rendering path reflects the new mode this turn, + # without waiting for an agent rebuild. + self.agent.tool_progress_mode = self.tool_progress_mode + + # Use raw ANSI codes via _cprint so the output is routed through + # prompt_toolkit's renderer. self.console.print() with Rich markup + # writes directly to stdout which patch_stdout's StdoutProxy mangles + # into garbled sequences like '?[33mTool progress: NEW?[0m' (#2262). + from hermes_cli.colors import Colors as _Colors + labels = { + "off": f"{_Colors.DIM}Tool progress: OFF{_Colors.RESET} — silent mode, just the final response.", + "new": f"{_Colors.YELLOW}Tool progress: NEW{_Colors.RESET} — show each new tool (skip repeats).", + "all": f"{_Colors.GREEN}Tool progress: ALL{_Colors.RESET} — show every tool call.", + "verbose": f"{_Colors.BOLD}{_Colors.GREEN}Tool progress: VERBOSE{_Colors.RESET} — full args, results, and think blocks.", + } + _cprint(labels.get(self.tool_progress_mode, "")) + + def _handle_usage_command(self, cmd_original: str): + """Dispatch `/usage [reset [--force]]`. + + Bare `/usage` keeps the classic display. `/usage reset` redeems one + banked Codex rate-limit reset credit (guarded: refuses when limits + aren't exhausted unless --force). + """ + parts = cmd_original.split() + args = [p.lower() for p in parts[1:]] + if args and args[0] == "reset": + self._usage_reset(force="--force" in args[1:]) + return + if args: + print(f" Unknown /usage subcommand: {' '.join(parts[1:])}. Try /usage or /usage reset [--force].") + return + self._show_usage() + + def _usage_reset(self, force: bool = False): + """`/usage reset [--force]` — redeem one banked Codex reset credit.""" + provider = ( + (getattr(self.agent, "provider", None) if self.agent else None) + or getattr(self, "provider", None) + ) + normalized = str(provider or "").strip().lower() + if normalized != "openai-codex": + print(" Banked usage resets are only available on the openai-codex provider.") + print(" Switch with `/model` or `hermes auth` first.") + return + base_url = (getattr(self.agent, "base_url", None) if self.agent else None) or getattr(self, "base_url", None) + api_key = (getattr(self.agent, "api_key", None) if self.agent else None) or getattr(self, "api_key", None) + + from agent.account_usage import redeem_codex_reset_credit + + print(" ⏳ Checking banked reset credits...") + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as _pool: + try: + result = _pool.submit( + redeem_codex_reset_credit, + base_url=base_url, + api_key=api_key, + force=force, + ).result(timeout=45.0) + except concurrent.futures.TimeoutError: + print(" ❌ Timed out talking to the Codex backend — try again shortly.") + return + print(f" {result.message}") + + def _show_context_breakdown(self, cmd_original: str = ""): + """`/context [all]` — visual context-window usage breakdown. + + Renders a 5×20 glyph block grid (each cell ≈ 1% of the model context + window) plus an estimated per-category table: system prompt, tool + definitions, rules, skills index, MCP, subagents, memory, and the + conversation itself — versus free space. `/context all` appends the + expanded per-skill and per-toolset cost listings. + + Read-only: same chars/4 estimation engine as the desktop context + popover (agent.context_breakdown) — no provider calls, no prompt-cache + impact. + """ + if not self.agent: + print(" (._.) No active agent -- send a message first.") + return + + args = cmd_original.split(maxsplit=1)[1].strip().lower() if " " in cmd_original else "" + expanded = args in {"all", "full", "details"} + + from agent.context_breakdown import ( + compute_context_details, + compute_session_context_breakdown, + render_context_breakdown_lines, + ) + + try: + payload = compute_session_context_breakdown( + self.agent, self.conversation_history + ) + except Exception as e: + print(f" (._.) Could not compute context breakdown: {e}") + return + + details = None + if expanded: + try: + details = compute_context_details(self.agent) + except Exception: + details = {"skills": [], "toolsets": []} + + model = payload.get("model") or self.model + print() + print(f" 🧠 Context Usage — {model}") + print() + for line in render_context_breakdown_lines(payload, details=details, grid=True): + print(f" {line}") + print() + + def _show_usage(self): + """Rate limits + session token usage (when a live agent exists) + Nous credits. + + The Nous credits block is agent-independent (a portal fetch), so it runs even + with no live agent — important for the TUI, where /usage runs in a slash-worker + subprocess that resumes the session WITHOUT building an agent (self.agent is None), + which would otherwise early-return before any credits showed. + """ + from cli import datetime, format_duration_compact + if not self.agent: + if self._print_nous_credits_block(): + self._print_usage_cta() + else: + print("(._.) No active agent -- send a message first.") + return + + agent = self.agent + calls = agent.session_api_calls + + if calls == 0: + if self._print_nous_credits_block(): + self._print_usage_cta() + else: + print("(._.) No API calls made yet in this session.") + return + + # ── Rate limits (shown first when available) ──────────────── + rl_state = agent.get_rate_limit_state() + if rl_state and rl_state.has_data: + from agent.rate_limit_tracker import format_rate_limit_display + print() + print(format_rate_limit_display(rl_state)) + print() + + # ── Session token usage ───────────────────────────────────── + input_tokens = getattr(agent, "session_input_tokens", 0) or 0 + output_tokens = getattr(agent, "session_output_tokens", 0) or 0 + reasoning_tokens = getattr(agent, "session_reasoning_tokens", 0) or 0 + prompt = agent.session_prompt_tokens + completion = agent.session_completion_tokens + total = agent.session_total_tokens + + compressor = agent.context_compressor + last_prompt = compressor.last_prompt_tokens if compressor.last_prompt_tokens > 0 else 0 + ctx_len = compressor.context_length + pct = min(100, (last_prompt / ctx_len * 100)) if ctx_len else 0 + compressions = compressor.compression_count + + msg_count = len(self.conversation_history) + elapsed = format_duration_compact((datetime.now() - self.session_start).total_seconds()) + + print(" 📊 Session Token Usage") + print(f" {'─' * 40}") + print(f" Model: {agent.model}") + print(f" Input tokens: {input_tokens:>10,}") + print(f" Output tokens: {output_tokens:>10,}") + if reasoning_tokens: + print(f" ↳ Reasoning (subset): {reasoning_tokens:>10,}") + print(f" Prompt tokens (total): {prompt:>10,}") + print(f" Completion tokens: {completion:>10,}") + print(f" Total tokens: {total:>10,}") + print(f" API calls: {calls:>10,}") + print(f" Session duration: {elapsed:>10}") + print(f" {'─' * 40}") + print(f" Current context: {last_prompt:,} / {ctx_len:,} ({pct:.0f}%)") + print(f" Messages: {msg_count}") + print(f" Compressions: {compressions}") + + # Account limits -- fetched off-thread with a hard timeout so slow + # provider APIs don't hang the prompt. + provider = getattr(agent, "provider", None) or getattr(self, "provider", None) + base_url = getattr(agent, "base_url", None) or getattr(self, "base_url", None) + api_key = getattr(agent, "api_key", None) or getattr(self, "api_key", None) + # Lazy import — pulls the OpenAI SDK chain, only needed here. + from agent.account_usage import fetch_account_usage, render_account_usage_lines + account_snapshot = None + if provider: + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as _pool: + try: + account_snapshot = _pool.submit( + fetch_account_usage, provider, + base_url=base_url, api_key=api_key, + ).result(timeout=10.0) + except (concurrent.futures.TimeoutError, Exception): + account_snapshot = None + account_lines = [f" {line}" for line in render_account_usage_lines(account_snapshot)] + if account_lines: + print() + for line in account_lines: + print(line) + + # Nous credits magnitudes + monthly-grant gauge (agent-independent — also + # runs at the no-agent / no-calls early-returns above). See the helper. + if self._print_nous_credits_block(): + self._print_usage_cta() + + if self.verbose: + logging.getLogger().setLevel(logging.DEBUG) + for noisy in ('openai', 'openai._base_client', 'httpx', 'httpcore', 'asyncio', 'hpack', 'grpc', 'modal'): + logging.getLogger(noisy).setLevel(logging.WARNING) + else: + logging.getLogger().setLevel(logging.INFO) + + def _show_insights(self, command: str = "/insights"): + """Show usage insights and analytics from session history.""" + # Parse optional --days flag + parts = command.split() + days = 30 + source = None + i = 1 + while i < len(parts): + if parts[i] == "--days" and i + 1 < len(parts): + try: + days = int(parts[i + 1]) + except ValueError: + print(f" Invalid --days value: {parts[i + 1]}") + return + i += 2 + elif parts[i] == "--source" and i + 1 < len(parts): + source = parts[i + 1] + i += 2 + elif parts[i].isdigit(): + days = int(parts[i]) + i += 1 + else: + i += 1 + + try: + from hermes_state import SessionDB + from agent.insights import InsightsEngine + + db = SessionDB() + try: + engine = InsightsEngine(db) + report = engine.generate(days=days, source=source) + print(engine.format_terminal(report)) + finally: + db.close() + except Exception as e: + print(f" Error generating insights: {e}") + + def _check_config_mcp_changes(self) -> None: + """Detect mcp_servers changes in config.yaml and react. + + Called from process_loop every CONFIG_WATCH_INTERVAL seconds. + Compares config.yaml mtime + mcp_servers section against the last + known state. When a change is detected: + + * By default (``mcp.auto_reload_on_config_change: true``) it + auto-triggers ``_reload_mcp()`` and informs the user — legacy + behaviour from #1474. + * When opted out (``mcp.auto_reload_on_config_change: false``) it + does NOT reload. Instead it notifies the user that the config + changed and that they can apply it with ``/reload-mcp`` — while + warning that ``/reload-mcp`` rebuilds the tool surface and + **invalidates the provider prompt cache** (the next message + re-sends the full input prefix, expensive on long-context / + high-reasoning models). This stops silent cache-breaking reloads + when config.yaml is rewritten frequently by external tooling or + other Hermes instances. + """ + + import yaml as _yaml + + CONFIG_WATCH_INTERVAL = 5.0 # seconds between config.yaml stat() calls + + now = time.monotonic() + if now - self._last_config_check < CONFIG_WATCH_INTERVAL: + return + self._last_config_check = now + + from hermes_cli.config import get_config_path as _get_config_path + cfg_path = _get_config_path() + if not cfg_path.exists(): + return + + try: + mtime = cfg_path.stat().st_mtime + except OSError: + return + + if mtime == self._config_mtime: + return # File unchanged — fast path + + # File changed — check whether mcp_servers section changed + self._config_mtime = mtime + try: + with open(cfg_path, encoding="utf-8") as f: + new_cfg = _yaml.safe_load(f) or {} + except Exception: + return + + new_mcp = new_cfg.get("mcp_servers") or {} + # Expand ${VAR} templates so the comparison is consistent with the + # init snapshot (self._config_mcp_servers), which was populated from + # the deep-merged + expanded config. Without this, any + # save_config_value() that rewrites config.yaml (even for unrelated + # keys) triggers a false-positive MCP reload because the raw yaml + # still has "${POWERMEM_API_KEY}" while the snapshot has the + # expanded value. + from hermes_cli.config import _expand_env_vars + new_mcp = _expand_env_vars(new_mcp) + if new_mcp == self._config_mcp_servers: + return # mcp_servers unchanged (some other section was edited) + + # Detected a change in the mcp_servers section. By default we + # auto-reload (legacy behaviour), but if the user has opted out we + # notify instead of reloading — because every reload rebuilds the + # agent tool surface and INVALIDATES the provider prompt cache (the + # next message re-sends the full input prefix, which is expensive on + # long-context / high-reasoning models). + # + # The toggle is the top-level ``mcp.auto_reload_on_config_change`` + # key (see DEFAULT_CONFIG). Read it from the config we just parsed + # so the user can flip it in the same edit that changes mcp_servers; + # missing key means default-on. + _mcp_cfg = new_cfg.get("mcp") + _auto = ( + _mcp_cfg.get("auto_reload_on_config_change", True) + if isinstance(_mcp_cfg, dict) + else True + ) + + self._config_mcp_servers = new_mcp + + if not _auto: + # Notify the user that the config changed but do NOT auto-reload. + # They can apply the new settings on their own terms with + # /reload-mcp — which we explicitly warn may invalidate the cache. + print() + print("🔄 MCP server config changed — reload skipped (auto-reload disabled).") + print(" New settings are NOT applied yet. To apply them now, run:") + print(" /reload-mcp") + print(" ⚠️ Note: /reload-mcp rebuilds the tool set and invalidates the") + print(" provider prompt cache (next message re-sends full input tokens).") + return + + # Notify user and reload. Run in a separate thread with a hard + # timeout so a hung MCP server cannot block the process_loop + # indefinitely (which would freeze the entire TUI). + print() + print("🔄 MCP server config changed — reloading connections...") + _reload_thread = threading.Thread( + target=self._reload_mcp, daemon=True + ) + _reload_thread.start() + + def _confirm_and_reload_mcp(self, cmd_original: str = "") -> None: + """Interactive /reload-mcp — confirm with the user, then reload. + + The auto-reload path (config file watcher) calls ``_reload_mcp`` + directly and never goes through this confirmation. + + Reloading MCP tools invalidates the provider prompt cache for the + active session (tool schemas are baked into the system prompt). + The next message re-sends full input tokens — can be expensive on + long-context or high-reasoning models. + + Three options: Approve Once, Always Approve (persists + ``approvals.mcp_reload_confirm: false`` so future reloads run + without this prompt), Cancel. Gated by + ``approvals.mcp_reload_confirm`` — default on. + """ + from cli import load_cli_config, save_config_value + # Gate check — respects prior "Always Approve" clicks. + try: + cfg = load_cli_config() + approvals = cfg.get("approvals") if isinstance(cfg, dict) else None + confirm_required = True + if isinstance(approvals, dict): + confirm_required = bool(approvals.get("mcp_reload_confirm", True)) + except Exception: + confirm_required = True + + if not confirm_required: + with self._busy_command(self._slow_command_status(cmd_original)): + self._reload_mcp() + return + + # Render warning + prompt. Use the same prompt_toolkit-native composer + # modal as destructive slash confirmations so choices stay visible. + choices = [ + ("once", "Approve Once", "reload now"), + ("always", "Always Approve", "reload now and silence this prompt permanently"), + ("cancel", "Cancel", "leave MCP tools unchanged"), + ] + raw = self._prompt_text_input_modal( + title="⚠️ /reload-mcp — Prompt cache invalidation warning", + detail=( + "Reloading MCP servers rebuilds the tool set for this session and\n" + "invalidates the provider prompt cache. The next message will\n" + "re-send full input tokens (can be expensive on long-context or\n" + "high-reasoning models)." + ), + choices=choices, + ) + if raw is None: + print("🟡 /reload-mcp cancelled (no input).") + return + choice = self._normalize_slash_confirm_choice(raw, choices) + if choice is None: + print(f"🟡 Unrecognized choice '{raw}'. /reload-mcp cancelled.") + return + + if choice == "cancel": + print("🟡 /reload-mcp cancelled. MCP tools unchanged.") + return + + if choice == "always": + if save_config_value("approvals.mcp_reload_confirm", False): + print("🔒 Future /reload-mcp calls will run without confirmation.") + print(" Re-enable via `approvals.mcp_reload_confirm: true` in config.yaml.") + else: + print("⚠️ Couldn't persist opt-out — reloading once.") + + with self._busy_command(self._slow_command_status(cmd_original)): + self._reload_mcp() + + def _reload_mcp(self): + """Reload MCP servers: disconnect all, re-read config.yaml, reconnect. + + After reconnecting, refreshes the agent's tool list so the model + sees the updated tools on the next turn. + """ + try: + from tools.mcp_tool import ( + shutdown_mcp_servers, discover_mcp_tools, reprobe_tool_availability, _servers, _lock, + ) + + # Capture old server names + with _lock: + old_servers = set(_servers.keys()) + + if not self._command_running: + print("🔄 Reloading MCP servers...") + + # Shutdown existing connections + shutdown_mcp_servers() + + # Explicit reload also re-probes tool availability (check_fn). + reprobe_tool_availability() + # Reconnect (reads config.yaml fresh) + new_tools = discover_mcp_tools() + + # Compute what changed + with _lock: + connected_servers = set(_servers.keys()) + + added = connected_servers - old_servers + removed = old_servers - connected_servers + reconnected = connected_servers & old_servers + + if reconnected: + print(f" ♻️ Reconnected: {', '.join(sorted(reconnected))}") + if added: + print(f" ➕ Added: {', '.join(sorted(added))}") + if removed: + print(f" ➖ Removed: {', '.join(sorted(removed))}") + if not connected_servers: + print(" No MCP servers connected.") + else: + print(f" 🔧 {len(new_tools)} tool(s) available from {len(connected_servers)} server(s)") + + # Refresh the agent's tool list so the model can call new tools. + # Route through the shared helper so this CLI /reload-mcp path stays + # in lockstep with the TUI RPC / gateway reload / late-binding paths + # (name-diff, thread-safe, and — critically — additive-preserving so + # memory-provider and context-engine tools survive the rebuild). + if self.agent is not None: + from tools.mcp_tool import refresh_agent_mcp_tools + # Explicit reload: pick up MCP servers the user ENABLED in config + # this session. self.enabled_toolsets was resolved once at + # startup; merge in any now-connected server names (unless the + # user pinned `all`/`*`, which already includes everything) so a + # freshly-added server isn't filtered out. Mirrors startup, where + # MCP server names are part of enabled_toolsets (see __init__). + enabled_override = None + et = self.enabled_toolsets + if et and "all" not in et and "*" not in et: + merged = list(et) + for _name in sorted(connected_servers): + if _name not in merged: + merged.append(_name) + enabled_override = merged + refresh_agent_mcp_tools( + self.agent, + enabled_override=enabled_override, + quiet_mode=True, + ) + # Keep the CLI's own list in sync with what the agent now uses. + if enabled_override is not None: + self.enabled_toolsets = enabled_override + + # Inject a message at the END of conversation history so the + # model knows tools changed. Appended after all existing + # messages to preserve prompt-cache for the prefix. + change_parts = [] + if added: + change_parts.append(f"Added servers: {', '.join(sorted(added))}") + if removed: + change_parts.append(f"Removed servers: {', '.join(sorted(removed))}") + if reconnected: + change_parts.append(f"Reconnected servers: {', '.join(sorted(reconnected))}") + tool_summary = f"{len(new_tools)} MCP tool(s) now available" if new_tools else "No MCP tools available" + change_detail = ". ".join(change_parts) + ". " if change_parts else "" + self.conversation_history.append({ + "role": "user", + "content": f"[IMPORTANT: MCP servers have been reloaded. {change_detail}{tool_summary}. The tool list for this conversation has been updated accordingly.]", + }) + + # Persist session immediately so the session log reflects the + # updated tools list (self.agent.tools was refreshed above). + if self.agent is not None: + try: + self.agent._persist_session( + self.conversation_history, + self.conversation_history, + ) + except Exception: + pass # Best-effort + + print(f" ✅ Agent updated — {len(self.agent.tools if self.agent else [])} tool(s) available") + + except Exception as e: + print(f" ❌ MCP reload failed: {e}") + + def _reload_skills(self) -> None: + """Reload skills: rescan ~/.hermes/skills/ and queue a note for the + next user turn. + + Skills don't need to live in the system prompt for the model to use + them (they're invoked via ``/skill-name``, ``skills_list``, or + ``skill_view`` at runtime), so this does NOT clear the prompt cache. + It rescans the slash-command map, prints the diff for the user, and + — if any skills were added or removed — queues a one-shot note that + gets prepended to the next user message. This preserves message + alternation (no phantom user turn injected out of band) and keeps + prompt caching intact. + """ + try: + from agent.skill_commands import reload_skills, get_skill_commands + + if not self._command_running: + print("🔄 Reloading skills...") + + result = reload_skills() + + # Sync cli.py's module-level _skill_commands so all consumers + # (help display, command dispatch, Tab-completion lambda) see the + # updated dict without needing to restart the session. + import cli as _cli + _cli._skill_commands = get_skill_commands() + added = result.get("added", []) # [{"name", "description"}, ...] + removed = result.get("removed", []) # [{"name", "description"}, ...] + total = result.get("total", 0) + + if not added and not removed: + print(" No new skills detected.") + print(f" 📚 {total} skill(s) available") + return + + def _fmt_line(item: dict) -> str: + nm = item.get("name", "") + desc = item.get("description", "") + return f" - {nm}: {desc}" if desc else f" - {nm}" + + if added: + print(" ➕ Added Skills:") + for item in added: + print(f" {_fmt_line(item)}") + if removed: + print(" ➖ Removed Skills:") + for item in removed: + print(f" {_fmt_line(item)}") + print(f" 📚 {total} skill(s) available") + + # Queue a one-shot note for the NEXT user turn. The CLI's agent + # loop prepends ``_pending_skills_reload_note`` (if set) to the + # API-call-local message at ~L8770, then clears it — same + # pattern as ``_pending_model_switch_note``. Nothing is written + # to conversation_history here, so message alternation stays + # intact and no out-of-band user turn is persisted. + # + # Format matches how the system prompt renders pre-existing + # skills (`` - name: description``) so the model reads the + # diff in the same shape as its original skill catalog. + sections = ["[USER INITIATED SKILLS RELOAD:"] + if added: + sections.append("") + sections.append("Added Skills:") + for item in added: + sections.append(_fmt_line(item)) + if removed: + sections.append("") + sections.append("Removed Skills:") + for item in removed: + sections.append(_fmt_line(item)) + sections.append("") + sections.append("Use skills_list to see the updated catalog.]") + self._pending_skills_reload_note = "\n".join(sections) + + except Exception as e: + print(f" ❌ Skills reload failed: {e}") diff --git a/hermes_cli/cli_loops_mixin.py b/hermes_cli/cli_loops_mixin.py new file mode 100644 index 0000000000..91fc398eec --- /dev/null +++ b/hermes_cli/cli_loops_mixin.py @@ -0,0 +1,725 @@ +"""Simple slash-command wrappers plus goal/heartbeat/loop manager hooks for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import logging +import os +import shutil +import threading +import time + +from rich.markup import escape as _escape + + +class CLILoopsMixin: + """Simple slash-command wrappers plus goal/heartbeat/loop manager hooks for the interactive CLI""" + + def _cmd_exit(self, cmd_original: str): + # /exit --delete also removes the session's transcripts + SQLite history. + from cli import _DIM, _RST, _cprint, _slash_args + _args = _slash_args(cmd_original).lower() + if _args in {"--delete", "-d"}: + self._delete_session_on_exit = True + elif _args: + _cprint(f" {_DIM}✗ Unknown argument: {_escape(_args)}. Use /exit --delete to also remove session history.{_RST}") + return True + return False + + def _cmd_help(self, cmd_original: str): + from cli import _slash_args + self.show_help(_slash_args(cmd_original)) + + def _cmd_redraw(self, cmd_original: str): + # Manual recovery for terminal buffer drift from multiplexer + # tab switches, subshell ``clear``, SSH window restores, etc. + # See issue #8688 (cmux). Ctrl+L is bound to the same helper. + from cli import _DIM, _RST, _cprint + self._force_full_redraw() + _cprint(f" {_DIM}✓ UI redrawn{_RST}") + + def _cmd_clear(self, cmd_original: str): + from cli import ( + ChatConsole, + _build_compact_banner, + _clear_output_history, + _cprint, + build_welcome_banner, + get_tool_definitions, + ) + if self._confirm_destructive_slash( + "clear", + "This clears the screen and starts a new session.\n" + "The current conversation history will be discarded.", + cmd_original=cmd_original, + ) is None: + return True # confirmation cancelled — command handled, keep REPL alive + self.new_session(silent=True) + _clear_output_history() + # Clear terminal screen. Inside the TUI, Rich's console.clear() + # goes through patch_stdout's StdoutProxy which swallows the + # screen-clear escape sequences. Use prompt_toolkit's output + # object directly to actually clear the terminal. + if self._app: + out = self._app.output + out.erase_screen() + out.cursor_goto(0, 0) + out.flush() + else: + self.console.clear() + # Show fresh banner. Inside the TUI we must route Rich output + # through ChatConsole (which uses prompt_toolkit's native ANSI + # renderer) instead of self.console (which writes raw to stdout + # and gets mangled by patch_stdout). + if self._app: + cc = ChatConsole() + term_w = shutil.get_terminal_size().columns + if self.compact or term_w < 80: + cc.print(_build_compact_banner()) + else: + tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) + cwd = os.getenv("TERMINAL_CWD", os.getcwd()) + ctx_len = None + if hasattr(self, 'agent') and self.agent and hasattr(self.agent, 'context_compressor'): + ctx_len = self.agent.context_compressor.context_length + build_welcome_banner( + console=cc, + model=self.model, + cwd=cwd, + tools=tools, + enabled_toolsets=self.enabled_toolsets, + session_id=self.session_id, + context_length=ctx_len, + provider=self.provider, + ) + _cprint(" ✨ (◕‿◕)✨ Fresh start! Screen cleared and conversation reset.\n") + self._print_random_tip() + else: + self.show_banner() + print(" ✨ (◕‿◕)✨ Fresh start! Screen cleared and conversation reset.\n") + self._print_random_tip() + + def _cmd_title(self, cmd_original: str): + from cli import _cprint + parts = cmd_original.split(maxsplit=1) + if len(parts) > 1: + raw_title = parts[1].strip() + if raw_title: + if self._session_db: + # Sanitize the title early so feedback matches what gets stored + try: + from hermes_state import SessionDB + new_title = SessionDB.sanitize_title(raw_title) + except ValueError as e: + # sanitize_title rejected the input (e.g. too long). + # Print that one reason and stop — don't fall + # through to the "empty after cleanup" branch and + # print a second, contradictory error (SC-05). + _cprint(f" {e}") + return True + if not new_title: + _cprint(" Title is empty after cleanup. Please use printable characters.") + elif self._session_db.get_session(self.session_id): + # Session exists in DB — set title directly + try: + if self._session_db.set_session_title(self.session_id, new_title): + self._status_bar_title_checked_at = 0.0 + _cprint(f" Session title set: {new_title}") + else: + _cprint(" Session not found in database.") + except ValueError as e: + _cprint(f" {e}") + else: + # Session not created yet — defer the title + # Check uniqueness proactively with the sanitized title + existing = self._session_db.get_session_by_title(new_title) + if existing: + _cprint(f" Title '{new_title}' is already in use by session {existing['id']}") + else: + self._pending_title = new_title + _cprint(f" Session title queued: {new_title} (will be saved on first message)") + else: + from hermes_state import format_session_db_unavailable + _cprint(f" {format_session_db_unavailable()}") + else: + _cprint(" Usage: /title ") + # Show current title and session ID if no argument given + elif self._session_db: + _cprint(f" Session ID: {self.session_id}") + session = self._session_db.get_session(self.session_id) + if session and session.get("title"): + _cprint(f" Title: {session['title']}") + elif self._pending_title: + _cprint(f" Title (pending): {self._pending_title}") + else: + _cprint(" No title set. Usage: /title ") + else: + from hermes_state import format_session_db_unavailable + _cprint(f" {format_session_db_unavailable()}") + + def _cmd_new(self, cmd_original: str): + # Strip inline-skip tokens (now/--yes/-y) before deriving the title + # so "/new now My Session" yields title="My Session" instead of + # title="now My Session". See _split_destructive_skip. + _new_args, _ = self._split_destructive_skip(cmd_original) + title = _new_args.strip() or None + if self._confirm_destructive_slash( + "new", + "This starts a fresh session.\n" + "The current conversation history will be discarded.", + cmd_original=cmd_original, + ) is None: + return True # confirmation cancelled — command handled, keep REPL alive + self.new_session(title=title) + + def _cmd_retry(self, cmd_original: str): + retry_msg = self.retry_last() + if retry_msg and hasattr(self, '_pending_input'): + # Re-queue the message so process_loop sends it to the agent + self._pending_input.put(retry_msg) + + def _cmd_undo(self, cmd_original: str): + # Parse optional turn count: "/undo" → 1, "/undo 3" → 3. + _undo_n = 1 + _undo_parts = cmd_original.split() + if len(_undo_parts) > 1: + try: + _undo_n = int(_undo_parts[1]) + except ValueError: + print(f"(._.) Invalid count {_undo_parts[1]!r} — use /undo or /undo N.") + return True # bad arg — command handled, keep the REPL alive + if _undo_n < 1: + _undo_n = 1 + # Nothing to undo → say so immediately; don't pop a destructive + # confirmation dialog for a guaranteed no-op (SC-06). + if not self.conversation_history: + print("(._.) No messages to undo.") + return True + _undo_desc = ( + "This removes the last user/assistant exchange from history." + if _undo_n == 1 + else f"This removes the last {_undo_n} user turns from history." + ) + if self._confirm_destructive_slash( + "undo", + _undo_desc, + cmd_original=cmd_original, + ) is None: + return True # confirmation cancelled — command handled, keep REPL alive + self.undo_last(_undo_n) + + def _cmd_skills(self, cmd_original: str): + with self._busy_command(self._slow_command_status(cmd_original)): + self._handle_skills_command(cmd_original) + + def _cmd_egress(self, cmd_original: str): + from hermes_cli.slash_exec import CommandContext, execute_command + + self._console_print( + execute_command("egress", CommandContext(surface="cli")).text, + highlight=False, markup=False, + ) + + def _cmd_statusbar(self, cmd_original: str): + self._status_bar_visible = not self._status_bar_visible + state = "visible" if self._status_bar_visible else "hidden" + self._console_print(f" Status bar {state}") + + def _cmd_update(self, cmd_original: str) -> bool: + # A truthy result means the process is relaunching — leave the REPL. + return not self._handle_update_command() + + def _cmd_version(self, cmd_original: str): + from hermes_cli.main import _print_version_info + + _print_version_info(check_updates=True) + + def _cmd_reload(self, cmd_original: str): + from hermes_cli.config import reload_env + count = reload_env() + print(f" Reloaded .env ({count} var(s) updated)") + + def _cmd_reload_skills(self, cmd_original: str): + with self._busy_command(self._slow_command_status(cmd_original)): + self._reload_skills() + + def _cmd_plugins(self, cmd_original: str): + from cli import display_hermes_home + try: + # Discover from disk (bundled + user), matching `hermes plugins + # list` — so installed-but-not-enabled plugins are visible here + # too. The plugin manager only knows about *loaded* plugins, so + # using it alone made freshly-installed, not-yet-enabled plugins + # look like "nothing installed". + from hermes_cli.plugins_cmd import ( + _discover_all_plugins, + _get_disabled_set, + _get_enabled_set, + _plugin_status, + ) + + entries = _discover_all_plugins() + enabled = _get_enabled_set() + disabled = _get_disabled_set() + + # `/plugins` is a quick glance — default to user-installed + # plugins (what the user actually added). Bundled provider/ + # platform plugins are summarized on one line; the full + # catalog lives behind `hermes plugins list`. + user_entries = [e for e in entries if e[3] != "bundled"] + bundled_count = len(entries) - len(user_entries) + + if not user_entries: + print("No user plugins installed.") + print(" Install one: hermes plugins install owner/repo") + print(f" Or drop a plugin directory into {display_hermes_home()}/plugins/") + if bundled_count: + print(f" ({bundled_count} bundled plugins available — see: hermes plugins list)") + else: + # Loaded-plugin details (tools/hooks/commands counts, errors) + # keyed by name, when available. + loaded: dict = {} + try: + from hermes_cli.plugins import get_plugin_manager + for p in get_plugin_manager().list_plugins(): + loaded[p["name"]] = p + except Exception: + loaded = {} + + print(f"User plugins ({len(user_entries)}):") + for name, version, _desc, source, _dir, key in sorted(user_entries): + state = _plugin_status(name, enabled, disabled, key=key) + glyph = {"enabled": "✓", "disabled": "✗"}.get(state, "○") + ver = f" v{version}" if version else "" + info = loaded.get(name) or {} + bits = [] + if info.get("tools"): + bits.append(f"{info['tools']} tools") + if info.get("hooks"): + bits.append(f"{info['hooks']} hooks") + if info.get("commands"): + bits.append(f"{info['commands']} commands") + detail = f" ({', '.join(bits)})" if bits else "" + label = "" if state == "enabled" else f" [{state}]" + error = f" — {info['error']}" if info.get("error") else "" + print(f" {glyph} {name}{ver}{label}{detail}{error}") + if bundled_count: + print(f" (+{bundled_count} bundled — see: hermes plugins list)") + print(" Enable/disable: hermes plugins enable/disable ") + except Exception as e: + print(f"Plugin system error: {e}") + + def _cmd_queue(self, cmd_original: str): + from cli import _cprint, _slash_args + payload = self._expand_paste_references(_slash_args(cmd_original)) + if not payload: + _cprint(" Usage: /queue ") + else: + self._pending_input.put(payload) + if self._agent_running: + _cprint(f" Queued for the next turn: {payload[:80]}{'...' if len(payload) > 80 else ''}") + else: + _cprint(f" Queued: {payload[:80]}{'...' if len(payload) > 80 else ''}") + + def _cmd_steer(self, cmd_original: str): + # Inject a message after the next tool call without interrupting. + # If the agent is actively running, push the text into the agent's + # pending_steer slot — the drain hook in _execute_tool_calls_* + # will append it to the next tool result's content. If no agent + # is running, fall back to queue semantics (same as /queue). + from cli import _cprint, _slash_args + payload = _slash_args(cmd_original) + if not payload: + _cprint(" Usage: /steer ") + elif self._agent_running and self.agent is not None and hasattr(self.agent, "steer"): + try: + accepted = self.agent.steer(payload) + except Exception as exc: + _cprint(f" Steer failed: {exc}") + else: + if accepted: + _cprint(f" ⏩ Steer queued — arrives after the next tool call: {payload[:80]}{'...' if len(payload) > 80 else ''}") + else: + _cprint(" Steer rejected (empty payload).") + else: + # No active run — treat as a normal next-turn message. + self._pending_input.put(payload) + _cprint(f" No agent running; queued as next turn: {payload[:80]}{'...' if len(payload) > 80 else ''}") + + # ──────────────────────────────────────────────────────────────── + # /goal — persistent cross-turn goals (Ralph-style loop) + # ──────────────────────────────────────────────────────────────── + def _get_goal_manager(self): + """Return the GoalManager bound to the current session_id. + + Cached on ``self._goal_manager`` and rebound lazily when + ``session_id`` changes (e.g. after /new or a compression-driven + session split). + """ + try: + from hermes_cli.goals import GoalManager + from hermes_cli.config import load_config + except Exception as exc: + logging.debug("goal manager unavailable: %s", exc) + return None + + sid = getattr(self, "session_id", None) or "" + if not sid: + return None + + existing = getattr(self, "_goal_manager", None) + if existing is not None and getattr(existing, "session_id", None) == sid: + return existing + + try: + cfg = load_config() or {} + goals_cfg = cfg.get("goals") or {} + max_turns = int(goals_cfg.get("max_turns", 20) or 20) + except Exception: + max_turns = 20 + + mgr = GoalManager(session_id=sid, default_max_turns=max_turns) + self._goal_manager = mgr + return mgr + + def _get_heartbeat_manager(self): + """Return the HeartbeatManager bound to the current session_id. + + Cached on ``self._heartbeat_manager`` and rebound lazily when + ``session_id`` changes (mirrors ``_get_goal_manager``). + """ + try: + from hermes_cli.heartbeat import HeartbeatManager + except Exception as exc: + logging.debug("heartbeat manager unavailable: %s", exc) + return None + + sid = getattr(self, "session_id", None) or "" + if not sid: + return None + + existing = getattr(self, "_heartbeat_manager", None) + if existing is not None and getattr(existing, "session_id", None) == sid: + return existing + + mgr = HeartbeatManager(session_id=sid) + self._heartbeat_manager = mgr + return mgr + + def _start_heartbeat_watchdog(self): + """Start the idle-poll thread that fires due heartbeats. + + Same pattern as the wake-word watchdog: a daemon thread polls a few + times a minute; when the session is idle (no agent running, empty + input queue) and the heartbeat is due, its prompt is injected into + ``_pending_input`` as a normal user turn. Missed ticks coalesce — + the anchor resets on fire, so a busy hour yields ONE heartbeat turn, + not a backlog. Idempotent; safe to call on every /heartbeat set. + """ + if getattr(self, "_heartbeat_watchdog_started", False): + return + self._heartbeat_watchdog_started = True + + from hermes_cli.heartbeat import POLL_SECONDS + + def _loop(): + try: + while not getattr(self, "_should_exit", False): + time.sleep(POLL_SECONDS) + try: + mgr = self._get_heartbeat_manager() + if mgr is None or not mgr.is_active(): + continue + busy = ( + self._agent_running + or getattr(self, "_voice_recording", False) + or getattr(self, "_voice_processing", False) + or not self._pending_input.empty() + ) + if busy: + continue + prompt = mgr.due_prompt() + if prompt: + self._pending_input.put(prompt) + except Exception as exc: + logging.debug("heartbeat watchdog tick failed: %s", exc) + finally: + self._heartbeat_watchdog_started = False + + threading.Thread(target=_loop, daemon=True, name="heartbeat-watchdog").start() + + # ──────────────────────────────────────────────────────────────── + # /loop — recurring in-session wakeups (Claude Code /loop parity) + # ──────────────────────────────────────────────────────────────── + def _get_loop_manager(self): + """Return the LoopManager bound to the current session_id. + + Cached on ``self._loop_manager`` and rebound lazily when + ``session_id`` changes (mirrors ``_get_goal_manager``). + """ + try: + from hermes_cli.loops import LoopManager + except Exception as exc: + logging.debug("loop manager unavailable: %s", exc) + return None + + sid = getattr(self, "session_id", None) or "" + if not sid: + return None + + existing = getattr(self, "_loop_manager", None) + if existing is not None and getattr(existing, "session_id", None) == sid: + return existing + + mgr = LoopManager(session_id=sid) + self._loop_manager = mgr + return mgr + + def _maybe_fire_loop_tick(self) -> None: + """Idle hook run from process_loop: fire a due /loop wakeup. + + Only runs while the agent is idle and nothing is queued — a real + user message always wins the idle boundary. An active (non-parked) + /goal also wins: its judge-driven continuations own the idle + boundary, so the loop defers to the next poll. + """ + from cli import _DIM, _RST, _cprint + mgr = self._get_loop_manager() + if mgr is None or not mgr.is_due(): + return + # The idle poll runs at ~10 Hz; once a tick is due but deferred + # (queued input / active goal), every poll would otherwise hit the + # DB via goal_blocks_loop_tick. Throttle the deferred re-check. + now = time.time() + if now - getattr(self, "_last_loop_tick_check", 0.0) < 2.0: + return + self._last_loop_tick_check = now + # Real user input (or anything else queued) takes priority; the + # loop stays due and fires at the next idle poll. + try: + if not self._pending_input.empty(): + return + except Exception: + return + try: + from hermes_cli.loops import goal_blocks_loop_tick + + if goal_blocks_loop_tick(mgr.session_id): + return + except Exception: + pass + + wakeup = mgr.fire_tick() + if not wakeup: + return + try: + state = mgr.state + tick_no = state.ticks_fired if state else "?" + _cprint(f" {_DIM}↻ /loop wakeup #{tick_no} firing…{_RST}") + self._pending_input.put(wakeup) + except Exception as exc: + logging.debug("loop tick injection failed: %s", exc) + try: + mgr.abandon_tick() + except Exception: + pass + return + # A slash-command loop (e.g. `/loop 10m /recap`) is dispatched via + # process_command, which never reaches the post-turn chat() finally + # block — so the tick would never complete and the loop would wedge + # on awaiting_response. Slash ticks have no model reply to evaluate; + # complete them immediately (caps and scheduling still apply). + if wakeup.lstrip().startswith("/"): + try: + decision = mgr.complete_tick("") + msg = decision.get("message") or "" + if msg: + _cprint(f" {msg}") + except Exception: + pass + + def _maybe_complete_loop_tick_after_turn(self) -> None: + """Post-turn hook: evaluate a finished /loop wakeup turn. + + No-op unless the turn that just ended was a loop wakeup + (``awaiting_response`` set by ``fire_tick``). Detects the + LOOP_COMPLETE marker, judges --until, applies caps, and schedules + the next tick. Mirrors _maybe_continue_goal_after_turn's shape. + """ + from cli import _DIM, _RST, _cprint + mgr = self._get_loop_manager() + if mgr is None: + return + state = mgr.state + if state is None or not state.awaiting_response: + return + + # A user-interrupted wakeup turn pauses the loop (recoverable via + # /loop resume) — same contract as the goal loop's Ctrl+C handling. + if getattr(self, "_last_turn_interrupted", False): + try: + mgr.pause(reason="user-interrupted (Ctrl+C)") + except Exception: + pass + _cprint( + f" {_DIM}⏸ Loop paused — wakeup turn was interrupted. " + f"Use /loop resume to continue, or /loop stop to end it.{_RST}" + ) + return + + last_response = "" + try: + hist = self.conversation_history or [] + for msg in reversed(hist): + if msg.get("role") == "assistant": + content = msg.get("content", "") + if isinstance(content, list): + parts = [ + p.get("text", "") + for p in content + if isinstance(p, dict) and p.get("type") in {"text", "output_text"} + ] + last_response = "\n".join(t for t in parts if t) + else: + last_response = str(content or "") + break + except Exception: + last_response = "" + + decision = mgr.complete_tick(last_response) + msg = decision.get("message") or "" + if msg: + _cprint(f" {msg}") + elif decision.get("status") == "active" and mgr.state is not None: + _cprint(f" {_DIM}↻ Loop: {mgr.state.remaining_label()}.{_RST}") + + def _maybe_continue_goal_after_turn(self) -> None: + """Hook run after every CLI turn. Judges + maybe re-queues. + + Safe to call when no goal is set — returns quickly. + + Preemption is automatic: if a real user message is already in + ``_pending_input`` we skip judging (the user's new input takes + priority and we'll re-judge after that turn). If judge says done, + mark it done and tell the user. If judge says continue and we're + under budget, push the continuation prompt onto the queue. + + Interrupt handling: if the turn was user-cancelled (Ctrl+C), we + AUTO-PAUSE the goal instead of judging + re-queuing. Otherwise + Ctrl+C feels like it did nothing — the judge runs on whatever + partial output landed, almost always says "continue", and the + loop keeps going. Auto-pause keeps the goal recoverable via + ``/goal resume`` once the user has sorted out what they want. + The empty-response skip mirrors the gateway guard at + ``_handle_message`` in ``gateway/run.py``. + """ + from cli import _DIM, _RST, _cprint, _looks_like_slash_command + mgr = self._get_goal_manager() + if mgr is None or not mgr.is_active(): + return + + # If a real user message is already queued, don't inject a + # continuation prompt on top — let the user's turn go first. + # Slash commands don't count as "real user messages" for this + # check: they're inspection/mutation (e.g. /subgoal added mid- + # run) and the process_loop dispatches them via process_command, + # not via chat(). If we treat a queued /subgoal as preempting, + # the goal loop silently stalls — we'd return here, then the + # slash command consumes its queue slot via process_command() + # which never re-fires the goal hook. Peek at all queued entries + # and only defer when there's a non-slash payload. + try: + pending = getattr(self, "_pending_input", None) + if pending is not None and not pending.empty(): + has_real_message = False + try: + # Queue.queue is the underlying deque — direct peek + # without disturbing FIFO order. + for entry in list(pending.queue): + # Bundled payloads are (text, images) tuples; + # unpack for inspection. + if isinstance(entry, tuple) and entry: + entry = entry[0] + if isinstance(entry, str) and _looks_like_slash_command(entry): + continue + has_real_message = True + break + except Exception: + # Fallback: if we can't introspect the queue, behave + # like the old check and defer to be safe. + has_real_message = True + if has_real_message: + return + except Exception: + pass + + # If the turn was user-interrupted (Ctrl+C), auto-pause the goal + # and bail. The judge call would almost always return "continue" + # on the partial output and immediately re-queue another turn, + # which is exactly what the user cancelled. Pausing (rather than + # silently skipping) is the observable, recoverable behavior. + if getattr(self, "_last_turn_interrupted", False): + try: + mgr.pause(reason="user-interrupted (Ctrl+C)") + except Exception as exc: + logging.debug("goal pause-on-interrupt failed: %s", exc) + _cprint( + f" {_DIM}⏸ Goal paused — turn was interrupted. " + f"Use /goal resume to continue, or /goal clear to stop.{_RST}" + ) + return + + # Extract the agent's final response for this turn. + last_response = "" + try: + hist = self.conversation_history or [] + for msg in reversed(hist): + if msg.get("role") == "assistant": + content = msg.get("content", "") + if isinstance(content, list): + # Multimodal content — flatten text parts. + parts = [ + p.get("text", "") + for p in content + if isinstance(p, dict) and p.get("type") in {"text", "output_text"} + ] + last_response = "\n".join(t for t in parts if t) + else: + last_response = str(content or "") + break + except Exception: + last_response = "" + + # Skip judging on empty/whitespace-only responses. These are almost + # always transient failures (API error, empty stream) where the + # judge would say "continue" and trip the consecutive-parse-failures + # backstop unnecessarily. Mirrors the gateway guard. + if not last_response.strip(): + return + + try: + from hermes_cli.goals import gather_background_processes as _gather_bg + _bg_procs = _gather_bg() + except Exception: + _bg_procs = None + + decision = mgr.evaluate_after_turn( + last_response, + user_initiated=True, + background_processes=_bg_procs, + ) + msg = decision.get("message") or "" + if msg: + _cprint(f" {msg}") + + if decision.get("should_continue"): + prompt = decision.get("continuation_prompt") + if prompt: + try: + self._pending_input.put(prompt) + except Exception as exc: + logging.debug("goal continuation enqueue failed: %s", exc) diff --git a/hermes_cli/cli_modal_mixin.py b/hermes_cli/cli_modal_mixin.py new file mode 100644 index 0000000000..9b8c6abb90 --- /dev/null +++ b/hermes_cli/cli_modal_mixin.py @@ -0,0 +1,1228 @@ +"""Modal overlays for the interactive CLI: clarify, approval, sudo/secret capture, command palette, slash-confirm, external editor + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import json +import queue +import sys + +from hermes_cli.callbacks import prompt_for_secret +from typing import Optional + + +class CLIModalMixin: + """Modal overlays for the interactive CLI: clarify, approval, sudo/secret capture, command palette, slash-confirm, external editor""" + + def _open_external_editor(self, buffer=None) -> bool: + """Open the active input buffer in an external editor.""" + from cli import _DIM, _RST, _cprint + app = getattr(self, "_app", None) + if not app: + _cprint(f"{_DIM}External editor is only available inside the interactive CLI.{_RST}") + return False + if self._command_running: + _cprint(f"{_DIM}Wait for the current command to finish before opening the editor.{_RST}") + return False + if self._sudo_state or self._secret_state or self._approval_state or getattr(self, "_slash_confirm_state", None) or self._clarify_state: + _cprint(f"{_DIM}Finish the active prompt before opening the editor.{_RST}") + return False + target_buffer = buffer or getattr(app, "current_buffer", None) + if target_buffer is None: + _cprint(f"{_DIM}No active input buffer is available for the external editor.{_RST}") + return False + try: + # Inline pastes so the editor (and the draft it submits) sees real + # content; skip flag unconditionally so the editor-close text-change + # doesn't re-collapse it, even when there was nothing to inline. + self._inline_pastes(target_buffer) + self._skip_paste_collapse = True + # Open the editor, then submit the saved draft on a clean exit — + # matching the TUI's Ctrl+G (openEditor), which sends the buffer + # instead of requiring a second Enter. Submission in this CLI is + # driven by the custom `enter` keybinding, NOT the buffer's + # accept_handler, so validate_and_handle can't route through it; + # chain a done-callback on the returned Task that re-uses the + # real submit pipeline via _submit_editor_buffer(). + task = target_buffer.open_in_editor(validate_and_handle=False) + if task is not None and hasattr(task, "add_done_callback"): + task.add_done_callback( + lambda _t, b=target_buffer: self._submit_editor_buffer(b) + ) + return True + except Exception as exc: + _cprint(f"{_DIM}Failed to open external editor: {exc}{_RST}") + return False + + def _submit_editor_buffer(self, buffer) -> None: + """Submit the draft an external editor left in ``buffer``. + + Invoked from the Ctrl+G done-callback so saving the editor sends the + prompt (TUI parity) instead of leaving it sitting in the input area. + Mirrors the idle/queue branches of the `enter` keybinding handler: + an empty save is ignored (never submits a blank turn), a slash command + is dispatched, otherwise the text is routed through the same input + queues the normal Enter path uses. Runs on the prompt_toolkit event + loop via the Task callback, so it must be cheap and non-blocking. + """ + from cli import _DIM, _RST, _cprint, _looks_like_slash_command + try: + text = (getattr(buffer, "text", "") or "").strip() + except Exception: + return + if not text: + # Editor saved empty / was cleared — match the TUI, which drops + # an empty draft instead of submitting a blank turn. + return + + app = getattr(self, "_app", None) + + # `!` shell mode, checked before slash dispatch — matches the + # Enter path in the input loop so an editor-saved bang command runs + # locally instead of being sent to the agent. + try: + if self.handle_bang_shell(text): + self._reset_input_buffer(buffer) + if app is not None: + app.invalidate() + return + except Exception as exc: + _cprint(f" {_DIM}Shell command failed: {exc}{_RST}") + self._reset_input_buffer(buffer) + if app is not None: + app.invalidate() + return + + # Slash commands: dispatch directly, same as the Enter handler's + # _looks_like_slash_command branch. + if _looks_like_slash_command(text): + try: + if not self.process_command(text): + self._should_exit = True + if app is not None and app.is_running: + app.exit() + except Exception as exc: + _cprint(f" {_DIM}Command failed: {exc}{_RST}") + finally: + self._reset_input_buffer(buffer) + if app is not None: + app.invalidate() + return + + # Regular prompt: route through the same queues the Enter handler uses. + if self._agent_running: + # Agent busy → honour the configured busy-input behaviour by + # queueing for the next turn (the safe default; interrupt/steer + # remain reachable via the normal Enter path). + self._interrupt_queue.put(text) if self.busy_input_mode == "interrupt" else self._pending_input.put(text) + preview = text[:80] + ("..." if len(text) > 80 else "") + _cprint(f" Queued for the next turn: {preview}") + else: + self._pending_input.put(text) + + self._reset_input_buffer(buffer) + if app is not None: + app.invalidate() + + def _inline_pastes(self, buffer) -> None: + """Replace collapsed-paste placeholders in ``buffer`` with real content. + + A big paste shows as a compact ``[Pasted text #N -> file]`` placeholder, + but history recall and the external editor need the actual text — a bare + reference is useless once the file is gone or on another machine. Inlining + before ``reset(append_to_history=True)`` also lets prompt_toolkit persist + the content through its normal path. Sets ``_skip_paste_collapse`` so the + ensuing text-change doesn't re-collapse it. + """ + from cli import logger + try: + existing = getattr(buffer, "text", "") + expanded = self._expand_paste_references(existing) + if expanded != existing and hasattr(buffer, "text"): + self._skip_paste_collapse = True + buffer.text = expanded + if hasattr(buffer, "cursor_position"): + buffer.cursor_position = len(expanded) + except Exception: + logger.debug("Failed to inline paste placeholders", exc_info=True) + + def _reset_input_buffer(self, buffer) -> None: + """Clear an input buffer after a programmatic submit (best-effort).""" + try: + buffer.reset(append_to_history=True) + except Exception: + try: + buffer.text = "" + except Exception: + pass + + def _prefill_input_buffer(self, text: str) -> None: + """Place ``text`` in the active prompt_toolkit buffer, editable.""" + from cli import logger + app = getattr(self, "_app", None) + if app is None: + return + try: + buf = app.current_buffer + buf.text = text + if hasattr(buf, "cursor_position"): + buf.cursor_position = len(text) + app.invalidate() + except Exception as e: + logger.debug("undo: prefill buffer failed: %s", e) + + def _prompt_text_input(self, prompt_text: str) -> str | None: + """Prompt for free-text input safely inside or outside prompt_toolkit. + + ``run_in_terminal`` returns a coroutine that must be awaited by the prompt_toolkit event loop, + which only exists on the main thread. Slash commands are dispatched from + the ``process_loop`` daemon thread (see issue #23185), so calling + ``run_in_terminal`` from there orphans the coroutine — ``_ask`` never runs, + and user keystrokes leak into the composer instead. Fall back to a direct + ``input()`` when we're off the main thread. + """ + import threading + result = [None] + + def _ask(): + try: + result[0] = input(prompt_text).strip() or None + except (KeyboardInterrupt, EOFError): + pass + + in_main_thread = threading.current_thread() is threading.main_thread() + + # Slash-worker guard (#23185 / billing auto-reload hang): when a + # prompt_toolkit app is running but we're on a non-main thread (the + # process_loop / TUI slash-worker daemon thread), stdin is owned by the + # event loop / JSON-RPC pipe. A bare input() there blocks forever until + # the worker's 45s timeout fires. We cannot safely prompt off the main + # thread, so cancel cleanly (None) instead of hanging — mirrors the + # _stdin_fallback discipline in _prompt_text_input_modal. + if self._app and not in_main_thread: + self._invalidate() + return None + + if self._app and in_main_thread: + from prompt_toolkit.application import run_in_terminal + was_visible = self._status_bar_visible + self._status_bar_visible = False + self._app.invalidate() + try: + run_in_terminal(_ask) + except Exception: + # WSL / Warp / certain terminal emulators silently drop the + # scheduled coroutine. Fall back to a direct input() so the + # user's keystrokes don't leak into the agent buffer. + try: + _ask() + except Exception: + pass + finally: + self._status_bar_visible = was_visible + self._app.invalidate() + else: + _ask() + return result[0] + + def _prompt_text_input_modal( + self, + *, + title: str, + detail: str, + choices: list[tuple[str, str, str]], + timeout: float = 120, + ) -> str | None: + """Prompt through the prompt_toolkit composer instead of raw input(). + + This is for CLI slash-command confirmations. The old raw input() path + fought prompt_toolkit's active stdin ownership: in some terminals the + prompt appeared above the TUI, choices were redrawn later, and Enter + could be interpreted as EOF/exit. A first-class modal state keeps the + choices visible and lets the normal Enter key binding submit the typed + or highlighted choice. + + **Platform note (Windows — issue #33961):** + Earlier code bypassed the modal on ``sys.platform == "win32"`` and fell + back to a raw ``input()`` prompt. When the confirm was triggered from the + ``process_loop`` daemon thread (the normal case) that ``input()`` ran off + the main thread and deadlocked against prompt_toolkit's stdin ownership — + the user saw a frozen cursor and Ctrl-C was swallowed (bare ``/reset`` + froze; ``/reset now`` worked only because it skips the prompt entirely). + + Native Windows now uses the same path as Linux/macOS: the modal is set up + on ``self._app.loop`` via ``call_soon_threadsafe`` and answered by the + normal prompt_toolkit key bindings (the same input channel that already + handles ordinary typing on Windows). The raw ``input()`` fallback is kept + only for the genuinely safe cases: no running app (unit tests / + non-interactive), no resolvable event loop, or a scheduling failure. + """ + import threading + import time as _time + + if not choices: + return None + + # If prompt_toolkit is not running (unit tests / non-interactive calls), + # keep the simple stdin fallback. + if not getattr(self, "_app", None): + return self._prompt_text_input("Choice [1/2/3]: ") + + try: + app_loop = self._app.loop + except Exception: + app_loop = None + + in_main_thread = threading.current_thread() is threading.main_thread() + + def _stdin_fallback() -> str | None: + # On native Windows a raw input() from a non-main thread deadlocks + # against prompt_toolkit's stdin ownership (#33961). With an app + # running we cannot safely prompt off the main thread, so cancel + # cleanly (None) rather than hang the terminal. + if sys.platform == "win32" and not in_main_thread: + self._invalidate() + return None + return self._prompt_text_input("Choice [1/2/3]: ") + + if not in_main_thread and app_loop is None: + return _stdin_fallback() + + response_queue = queue.Queue() + + def _setup_modal() -> None: + self._capture_modal_input_snapshot() + self._slash_confirm_state = { + "title": title, + "detail": detail, + "choices": choices, + "selected": 0, + "response_queue": response_queue, + } + self._slash_confirm_deadline = _time.monotonic() + timeout + self._invalidate() + + def _teardown_modal() -> None: + self._slash_confirm_state = None + self._slash_confirm_deadline = 0 + self._restore_modal_input_snapshot() + self._invalidate() + + def _run_on_app_loop(fn) -> bool: + if in_main_thread or app_loop is None: + fn() + return True + ready = threading.Event() + + def _wrapped() -> None: + try: + fn() + finally: + ready.set() + + try: + app_loop.call_soon_threadsafe(_wrapped) + except Exception: + return False + return ready.wait(timeout=5) + + if not _run_on_app_loop(_setup_modal): + return _stdin_fallback() + + _last_countdown_refresh = _time.monotonic() + try: + while True: + try: + result = response_queue.get(timeout=1) + _run_on_app_loop(_teardown_modal) + return result + except queue.Empty: + remaining = self._slash_confirm_deadline - _time.monotonic() + if remaining <= 0: + break + now = _time.monotonic() + if now - _last_countdown_refresh >= 5.0: + _last_countdown_refresh = now + self._invalidate() + finally: + if self._slash_confirm_state is not None: + _run_on_app_loop(_teardown_modal) + return None + + def _submit_slash_confirm_response(self, value: str | None) -> None: + state = self._slash_confirm_state + if not state: + return + state["response_queue"].put(value) + self._slash_confirm_state = None + self._slash_confirm_deadline = 0 + self._invalidate() + + def _normalize_slash_confirm_choice( + self, + raw: str | None, + choices: list[tuple[str, str, str]], + ) -> str | None: + if raw is None: + return None + choice_raw = raw.strip().lower() + if not choice_raw: + return None + aliases = { + "1": "once", + "once": "once", + "approve": "once", + "yes": "once", + "y": "once", + "ok": "once", + "2": "always", + "always": "always", + "remember": "always", + "3": "cancel", + "cancel": "cancel", + "nevermind": "cancel", + "no": "cancel", + "n": "cancel", + } + allowed = {choice[0] for choice in choices} + normalized = aliases.get(choice_raw) + if normalized in allowed: + return normalized + if choice_raw in allowed: + return choice_raw + return None + + def _build_command_palette_entries(self) -> list: + """Flat list of (command, description) for the Ctrl+P palette. + + Sourced from the same COMMAND_REGISTRY that backs /help, filtered to + commands available on this surface, plus installed skill commands. + Selecting an entry inserts the exact command string — never a fuzzy + resolution. + """ + from cli import _ensure_skill_commands + from hermes_cli.commands import COMMANDS_BY_CATEGORY + + entries: list[tuple[str, str, str]] = [] # (command, category, desc) + for category, commands in COMMANDS_BY_CATEGORY.items(): + for cmd, desc in commands.items(): + if not self._command_available(cmd): + continue + entries.append((cmd, category, desc)) + try: + for cmd, info in sorted(_ensure_skill_commands().items()): + entries.append((cmd, "Skill", info.get("description", ""))) + except Exception: + pass + return entries + + def _open_command_palette(self) -> None: + """Open the Ctrl+P fuzzy command palette modal.""" + if getattr(self, "_command_palette_state", None): + return + # Don't stack over other modals. + if (self._model_picker_state or self._clarify_state or self._approval_state + or self._slash_confirm_state or self._sudo_state or self._secret_state): + return + self._capture_modal_input_snapshot() + self._command_palette_state = { + "entries": self._build_command_palette_entries(), + "filter": "", + "selected": 0, + "_scroll_offset": 0, + } + self._invalidate(min_interval=0.0) + + def _close_command_palette(self) -> None: + self._command_palette_state = None + self._restore_modal_input_snapshot() + self._invalidate(min_interval=0.0) + + def _command_palette_visible_entries(self) -> list: + """Return (command, category, desc) rows matching the active filter. + + Ranked, command-name-focused matching (a bare subsequence over the + whole "cmd category desc" string is uselessly permissive — "steer" + would match 130+ rows via description text). Priority: + 0 exact command match + 1 command startswith query + 2 query substring in command + 3 query subsequence in command + 4 query substring in description + Rows that match nowhere are dropped. Ties keep registry order. + """ + state = self._command_palette_state or {} + entries = state.get("entries") or [] + q = (state.get("filter", "") or "").strip().lower() + if not q: + return list(entries) + + def _subseq(needle: str, hay: str) -> bool: + it = iter(hay) + return all(ch in it for ch in needle) + + ranked = [] + for order, row in enumerate(entries): + cmd, _cat, desc = row + name = cmd.lower().lstrip("/") + qn = q.lstrip("/") + desc_l = (desc or "").lower() + if name == qn: + rank = 0 + elif name.startswith(qn): + rank = 1 + elif qn in name: + rank = 2 + elif _subseq(qn, name): + rank = 3 + elif q in desc_l: + rank = 4 + else: + continue + ranked.append((rank, order, row)) + ranked.sort(key=lambda t: (t[0], t[1])) + return [row for (_r, _o, row) in ranked] + + def _handle_command_palette_selection(self) -> None: + """Insert the selected command into the composer (does not auto-run).""" + from cli import logger + state = self._command_palette_state + if not state: + return + rows = self._command_palette_visible_entries() + selected = state.get("selected", 0) + if not (0 <= selected < len(rows)): + self._close_command_palette() + return + cmd = rows[selected][0] # exact command string, e.g. "/model" + self._close_command_palette() + # Prefill the composer so the user can add args / confirm — never + # auto-execute (a palette pick should be explicit, and many commands + # take arguments). + try: + app = getattr(self, "_app", None) + if app is not None: + buf = app.current_buffer + buf.text = cmd + " " + buf.cursor_position = len(buf.text) + self._invalidate(min_interval=0.0) + except Exception: + logger.debug("command palette prefill failed", exc_info=True) + + @classmethod + def _split_destructive_skip(cls, cmd_text: Optional[str]) -> tuple[str, bool]: + """Split inline-skip tokens out of a destructive slash command. + + Returns ``(remainder, skip)`` where ``remainder`` is the original + text with the command word and any recognized skip tokens removed, + and ``skip`` is True iff at least one skip token was found. + + Examples: + "/reset now" -> ("", True) + "/reset --yes My title" -> ("My title", True) + "/new My title" -> ("My title", False) + "/clear" -> ("", False) + """ + if not cmd_text: + return "", False + tokens = cmd_text.strip().split() + if not tokens: + return "", False + # Drop leading "/cmd" word — callers pass the full command text. + if tokens[0].startswith("/"): + tokens = tokens[1:] + skip = False + kept: list[str] = [] + for tok in tokens: + if tok.lower() in cls._DESTRUCTIVE_SKIP_TOKENS: + skip = True + continue + kept.append(tok) + return " ".join(kept), skip + + def _confirm_destructive_slash( + self, + command: str, + detail: str, + cmd_original: Optional[str] = None, + ) -> Optional[str]: + """Prompt the user to confirm a destructive session slash command. + + Used by ``/clear``, ``/new``/``/reset``, and ``/undo`` before they + discard conversation state. Three-option prompt: + + 1. Approve Once — proceed this time only + 2. Always Approve — proceed and persist + ``approvals.destructive_slash_confirm: false`` so future + destructive commands run without confirmation + 3. Cancel — abort + + Gated by ``approvals.destructive_slash_confirm`` (default on). If the + gate is off the function returns ``"once"`` immediately without + prompting. + + Inline-skip: if ``cmd_original`` contains ``now``, ``--yes``, or + ``-y`` as an argument (e.g. ``/reset now``, ``/new --yes My title``), + the modal is bypassed and ``"once"`` is returned immediately. This is + an escape hatch for non-interactive use and for the degraded path where + the modal can't be marshaled onto the app loop (native Windows itself now + drives the modal normally — see #33961). Callers are responsible + for stripping the skip tokens from any remaining argument parsing + (see :meth:`_split_destructive_skip`). + + Returns ``"once"``, ``"always"``, or ``None`` (cancelled). Callers + proceed with the destructive action when the result is non-None. + """ + from cli import load_cli_config, save_config_value + # Inline-skip escape hatch — works regardless of platform/modal state. + # See class-level _DESTRUCTIVE_SKIP_TOKENS for the accepted tokens. + if cmd_original: + _, _skip = self._split_destructive_skip(cmd_original) + if _skip: + return "once" + + # Gate check — respects prior "Always Approve" clicks. + try: + cfg = load_cli_config() + approvals = cfg.get("approvals") if isinstance(cfg, dict) else None + confirm_required = True + if isinstance(approvals, dict): + confirm_required = bool(approvals.get("destructive_slash_confirm", True)) + except Exception: + confirm_required = True + + if not confirm_required: + return "once" + + # Render a prompt_toolkit-native confirmation panel. This keeps option + # labels visible above the composer and avoids raw input()/EOF races with + # the running TUI. + choices = [ + ("once", "Approve Once", "proceed this time only"), + ("always", "Always Approve", "proceed and silence this prompt permanently"), + ("cancel", "Cancel", "keep current conversation"), + ] + raw = self._prompt_text_input_modal( + title=f"⚠️ /{command} — destroys conversation state", + detail=detail, + choices=choices, + ) + if raw is None: + print(f"🟡 /{command} cancelled (no input).") + return None + choice = self._normalize_slash_confirm_choice(raw, choices) + if choice is None: + print(f"🟡 Unrecognized choice '{raw}'. /{command} cancelled.") + return None + + if choice == "cancel": + print(f"🟡 /{command} cancelled. Conversation unchanged.") + return None + + if choice == "always": + if save_config_value("approvals.destructive_slash_confirm", False): + print("🔒 Future /clear, /new, /reset, and /undo will run without confirmation.") + print(" Re-enable via `approvals.destructive_slash_confirm: true` in config.yaml.") + else: + print("⚠️ Couldn't persist opt-out — proceeding once.") + + return choice + + def _ring_bell(self, prompt: bool = False, context: str = "", detail: str = "") -> None: + """Write a terminal bell (\\a) if the matching display.bell_* flag is on. + + ``prompt=True`` is the blocking-modal variant (clarify / approval / + sudo / secret capture) gated by ``display.bell_on_prompt``; the default + is the end-of-turn bell gated by ``display.bell_on_complete``. Works + over SSH — the BEL propagates to the user's terminal. + + The same flag also emits an OSC 9 desktop notification (Ghostty, + iTerm2, Kitty, WezTerm) and, inside a supporting Warp build, a + ``warp://cli-agent`` OSC 777 event — see ``hermes_cli.terminal_notify``. + ``context`` is the short notification body (e.g. "approval"). + """ + flag = "bell_on_prompt" if prompt else "bell_on_complete" + if not getattr(self, flag, False): + return + try: + sys.stdout.write("\a") + sys.stdout.flush() + except Exception: + pass + try: + from hermes_cli.terminal_notify import notify as _terminal_notify + + _terminal_notify( + context or ("input needed" if prompt else "turn complete"), + prompt=prompt, + session_id=getattr(self, "session_id", "") or "", + detail=detail, + ) + except Exception: + pass + + def _clarify_callback(self, question, choices, multi_select=False, questions=None): + """ + Platform callback for the clarify tool. Called from the agent thread. + + Sets up the interactive selection UI (or freetext prompt for open-ended + questions), then blocks until the user responds via the prompt_toolkit + key bindings. If no response arrives within the configured timeout the + question is dismissed and the agent is told to decide on its own. + + When ``multi_select`` is True, shows checkboxes and the user can + select multiple options with Space, confirming with Enter. + + When ``questions`` is a non-empty list (batch clarify, issue #18450), + the panel switches to the A-compact multi-question layout and the + return value is a dict ``{"answers": {qid: raw_answer}}`` (plus + ``"timed_out": True`` when the deadline expired with only partial + answers). The single-question path below is unchanged. + """ + from cli import CLI_CONFIG, _DIM, _RST, _cprint + import time as _time + + from tools.clarify_gateway import resolve_clarify_timeout + + if questions: + return self._clarify_callback_batch(questions) + + # Canonical clarify timeout, shared with the gateway/TUI path. `<= 0` + # means unlimited (never auto-skip mid-think) → a null deadline. + timeout = resolve_clarify_timeout(CLI_CONFIG) + response_queue = queue.Queue() + is_open_ended = not choices + # multi-select support: only active when multi_select is True and choices exist + effective_multi = multi_select and not is_open_ended + + self._clarify_state = { + "question": question, + "choices": choices if not is_open_ended else [], + "selected": 0, + # multi-select support + "multi_select": effective_multi, + "selected_indices": set() if effective_multi else None, + "response_queue": response_queue, + } + self._clarify_deadline = None if timeout <= 0 else _time.monotonic() + timeout + # Open-ended questions skip straight to freetext input + self._clarify_freetext = is_open_ended + self._clarify_multi_base = None + + self._ring_bell(prompt=True, context="clarify") + # Trigger an immediate prompt_toolkit repaint from this (non-main) + # thread. Modal prompts must paint at once and must not be gated by the + # _invalidate throttle / resize guard — see _paint_now / _invalidate (#41098). + self._paint_now() + + # Poll for the user's response. The countdown in the hint line updates + # on each repaint; refresh it once a second so the timer stays visible + # while we wait. Selection changes (↑/↓) trigger instant repaints via + # the key bindings. + _last_countdown_refresh = _time.monotonic() + while True: + try: + result = response_queue.get(timeout=1) + self._clarify_deadline = None + self._persist_prompt_summary("?", "Clarify", question, str(result)) + return result + except queue.Empty: + # None deadline = unlimited: never auto-skip, just keep polling. + if self._clarify_deadline is not None: + remaining = self._clarify_deadline - _time.monotonic() + if remaining <= 0: + break + now = _time.monotonic() + if now - _last_countdown_refresh >= 1.0: + _last_countdown_refresh = now + self._paint_now() + + # Timed out — tear down the UI and let the agent decide + self._clarify_state = None + self._clarify_freetext = False + self._clarify_deadline = None + self._clarify_multi_base = None + self._paint_now() + _cprint(f"\n{_DIM}(clarify timed out after {timeout}s — agent will decide){_RST}") + return ( + "The user did not provide a response within the time limit. " + "Use your best judgement to make the choice and proceed." + ) + + def _clarify_batch_set_active(self, state, index) -> None: + """Point the batch clarify panel at question ``index``. + + Mirrors the active question's data into the flat keys the existing + single-question keybindings and renderer read (``question``, + ``choices``, ``selected``, ``multi_select``, ``selected_indices``), + so ↑/↓/Space/number keys operate on the active question unchanged. + Open-ended questions drop straight into freetext, matching the + single-question path. Re-visiting an answered question restores the + cursor to the earlier selection (choice answers highlight their row, + an "Other" answer highlights the Other row) so the user can see and + edit what they picked. + """ + questions_list = state["questions"] + index = max(0, min(index, len(questions_list) - 1)) + entry = questions_list[index] + state["active"] = index + state["question"] = entry["question"] + state["choices"] = entry["choices"] or [] + state["selected"] = 0 + state["multi_select"] = bool(entry["multi_select"]) + state["selected_indices"] = set() if entry["multi_select"] else None + self._clarify_freetext = not entry["choices"] + self._clarify_multi_base = None + # Restore the earlier answer's cursor/checkbox position on re-visit. + meta = (state.get("answer_meta") or {}).get(entry["qid"]) + choices = entry["choices"] or [] + if meta is None: + return + if meta.get("kind") == "choice": + answer = state["answers"].get(entry["qid"]) + if answer in choices: + state["selected"] = choices.index(answer) + elif meta.get("kind") == "other": + state["selected"] = len(choices) + elif meta.get("kind") == "multi": + checked = set() + for label in meta.get("choices") or []: + if label in choices: + checked.add(choices.index(label)) + if meta.get("other_text"): + checked.add(len(choices)) + state["selected_indices"] = checked + + def _clarify_batch_lock(self, state, answer, meta=None) -> None: + """Lock ``answer`` for the active batch question and advance. + + Overwrites any earlier answer for the same question (locked answers + stay editable until the batch completes). ``meta`` records how the + answer was produced ({"kind": "choice"|"other"|"multi", ...}) so a + re-visit can restore the cursor and prefill an "Other" edit. Advances + ``active`` to the next unanswered question; when every question has + an answer, puts the answers dict on the response queue and tears down + the panel. + """ + entry = state["questions"][state["active"]] + state["answers"][entry["qid"]] = answer + state.setdefault("answer_meta", {})[entry["qid"]] = meta or {"kind": "choice"} + self._persist_prompt_summary("?", "Clarify", entry["question"], str(answer)) + total = len(state["questions"]) + for offset in range(1, total + 1): + candidate = (state["active"] + offset) % total + if state["questions"][candidate]["qid"] not in state["answers"]: + self._clarify_batch_set_active(state, candidate) + return + # Every question answered — resolve the batch. + try: + state["response_queue"].put(dict(state["answers"])) + except Exception: + pass + self._clarify_state = None + self._clarify_freetext = False + self._clarify_multi_base = None + + def _clarify_batch_enter(self, state) -> None: + """Enter in batch choice mode: lock the active question's selection. + + Multi-select questions lock a JSON array string of the checked + labels (the tool core parses it via ``_parse_multi_select_response``). + Selecting "Other" switches to freetext; the freetext submit path + locks the typed answer. Entering "Other" on a question whose earlier + answer was typed prefills the composer with that text for editing. + """ + choices = state.get("choices") or [] + selected = state.get("selected", 0) + entry = state["questions"][state["active"]] + meta = (state.get("answer_meta") or {}).get(entry["qid"]) or {} + if state.get("multi_select"): + indices = state.get("selected_indices") or set() + sorted_idx = sorted(indices) + selected_choices = [choices[i] for i in sorted_idx if i < len(choices)] + other_checked = len(choices) in sorted_idx + if other_checked: + # Stash the checked real choices (possibly none) so the + # freetext submit appends the typed answer to the array. + self._clarify_multi_base = selected_choices + self._clarify_freetext = True + self._clarify_prefill = meta.get("other_text") or "" + return + self._clarify_batch_lock( + state, + json.dumps(selected_choices, ensure_ascii=False), + meta={"kind": "multi", "choices": selected_choices, "other_text": ""}, + ) + return + if selected < len(choices): + self._clarify_batch_lock( + state, choices[selected], meta={"kind": "choice"} + ) + return + # "Other" highlighted → switch to freetext; prefill an earlier typed + # answer so Enter on an answered Other edits instead of retyping. + self._clarify_freetext = True + self._clarify_prefill = ( + meta.get("other_text") or "" if meta.get("kind") == "other" else "" + ) + + def _clarify_callback_batch(self, questions): + """Batch clarify panel (A-compact): all questions, one active. + + Blocks on the response queue like the single-question path. Returns + ``{"answers": {qid: raw_answer}}`` when every question is locked, the + same dict plus ``"timed_out": True`` when the deadline expires with + partial (or zero) answers, and passes a cancel string through + unchanged so the tool core resolves the batch empty. + """ + from cli import CLI_CONFIG, _DIM, _RST, _cprint + import time as _time + + from tools.clarify_gateway import resolve_clarify_timeout + + timeout = resolve_clarify_timeout(CLI_CONFIG) + response_queue = queue.Queue() + + state = { + "questions": list(questions), + "answers": {}, + "answer_meta": {}, + "active": 0, + "response_queue": response_queue, + # Flat keys mirroring the active question — filled by + # _clarify_batch_set_active below. + "question": "", + "choices": [], + "selected": 0, + "multi_select": False, + "selected_indices": None, + } + self._clarify_state = state + self._clarify_batch_set_active(state, 0) + self._clarify_deadline = None if timeout <= 0 else _time.monotonic() + timeout + self._ring_bell(prompt=True, context="clarify") + self._paint_now() + + _last_countdown_refresh = _time.monotonic() + while True: + try: + result = response_queue.get(timeout=1) + self._clarify_deadline = None + if isinstance(result, dict): + return {"answers": result} + # Cancel path (Ctrl+C teardown) posts a plain string — pass + # it through so the tool core resolves the batch empty. + return result + except queue.Empty: + if self._clarify_deadline is not None: + remaining = self._clarify_deadline - _time.monotonic() + if remaining <= 0: + break + now = _time.monotonic() + if now - _last_countdown_refresh >= 1.0: + _last_countdown_refresh = now + self._paint_now() + + # Timed out — keep the answers locked so far and flag the timeout. + partial = dict(state["answers"]) + self._clarify_state = None + self._clarify_freetext = False + self._clarify_deadline = None + self._clarify_multi_base = None + self._paint_now() + _cprint(f"\n{_DIM}(clarify timed out after {timeout}s — locked answers returned){_RST}") + return {"answers": partial, "timed_out": True} + + def _sudo_password_callback(self) -> str: + """ + Prompt for sudo password through the prompt_toolkit UI. + + Called from the agent thread when a sudo command is encountered. + Uses the same clarify-style mechanism: sets UI state, waits on a + queue for the user's response via the Enter key binding. + """ + from cli import _DIM, _RST, _cprint + import time as _time + + timeout = 45 + response_queue = queue.Queue() + + self._capture_modal_input_snapshot() + self._sudo_state = { + "response_queue": response_queue, + } + self._sudo_deadline = _time.monotonic() + timeout + self._ring_bell(prompt=True, context="sudo password") + + # Modal prompt — paint immediately, bypassing the throttle/resize guard + # so the prompt can't be dropped and time out unseen (#41098). + self._paint_now() + + while True: + try: + result = response_queue.get(timeout=1) + self._sudo_state = None + self._sudo_deadline = 0 + self._restore_modal_input_snapshot() + self._paint_now() + if result: + _cprint(f"\n{_DIM} ✓ Password received (cached for session){_RST}") + else: + _cprint(f"\n{_DIM} ⏭ Skipped{_RST}") + return result + except queue.Empty: + remaining = self._sudo_deadline - _time.monotonic() + if remaining <= 0: + break + self._paint_now() + + self._sudo_state = None + self._sudo_deadline = 0 + self._restore_modal_input_snapshot() + self._paint_now() + _cprint(f"\n{_DIM} ⏱ Timeout — continuing without sudo{_RST}") + return "" + + def _approval_callback(self, command: str, description: str, + *, allow_permanent: bool = True, + allow_session: bool = True, + smart_denied: bool = False) -> str: + """ + Prompt for dangerous command approval through the prompt_toolkit UI. + + Called from the agent thread. Shows a selection UI similar to clarify + with choices: once / session / always / deny. Smart DENY owner + overrides show only once / deny, as do gates that re-ask every time + (allow_session=False). When allow_permanent is False for another + reason (for example tirith), only 'always' is hidden. + Long commands also get a 'view' option so the full command can be + expanded before deciding. + + Uses _approval_lock to serialize concurrent requests (e.g. from + parallel delegation subtasks) so each prompt gets its own turn + and the shared _approval_state / _approval_deadline aren't clobbered. + """ + from cli import CLI_CONFIG, _DIM, _RST, _cprint + import time as _time + + with self._approval_lock: + timeout = int(CLI_CONFIG.get("approvals", {}).get("timeout", 300)) + response_queue = queue.Queue() + + self._approval_state = { + "command": command, + "description": description, + "choices": self._approval_choices( + command, + allow_permanent=allow_permanent, + allow_session=allow_session, + smart_denied=smart_denied, + ), + "selected": 0, + "response_queue": response_queue, + } + self._approval_deadline = _time.monotonic() + timeout + + self._ring_bell(prompt=True, context="approval", detail=command) + # Modal prompt — paint immediately, bypassing the throttle/resize + # guard. A throttled paint here can be silently dropped (250ms + # window collision or in-flight resize), leaving the panel unseen so + # the command is denied on timeout without the user ever seeing it + # (#41098). The countdown refreshes below paint the same way. + self._paint_now() + + _last_countdown_refresh = _time.monotonic() + while True: + try: + result = response_queue.get(timeout=1) + self._approval_state = None + self._approval_deadline = 0 + self._paint_now() + _outcome_labels = { + "once": "allowed once", + "session": "allowed for session", + "always": "added to allowlist", + "deny": "denied", + } + self._persist_prompt_summary( + "⚠", "Approval", command, + _outcome_labels.get(result, str(result)), + ) + return result + except queue.Empty: + remaining = self._approval_deadline - _time.monotonic() + if remaining <= 0: + break + now = _time.monotonic() + if now - _last_countdown_refresh >= 1.0: + _last_countdown_refresh = now + self._paint_now() + + self._approval_state = None + self._approval_deadline = 0 + self._paint_now() + _cprint(f"\n{_DIM} ⏱ Timeout — denying command{_RST}") + self._persist_prompt_summary( + "⚠", "Approval", command, "timed out (no response)", + ) + return "timeout" + + def _approval_choices(self, command: str, *, allow_permanent: bool = True, + allow_session: bool = True, + smart_denied: bool = False) -> list[str]: + """Return approval choices for a dangerous command prompt.""" + if smart_denied or not allow_session: + choices = ["once", "deny"] + else: + choices = ["once", "session", "always", "deny"] if allow_permanent else ["once", "session", "deny"] + if len(command) > 70: + choices.append("view") + return choices + + def _computer_use_approval_callback(self, action: str, args: dict, summary: str) -> str: + """Adapt the generic approval UI for the computer_use tool. + + The computer_use handler expects verdicts of the form + `approve_once` | `approve_session` | `always_approve` | `deny`. + The CLI's built-in approval UI returns `once` | `session` | `always` + | `deny`. Translate between the two. + """ + # Build a command-ish string so the existing UI renders something + # meaningful. `summary` is already a one-line human description. + verdict = self._approval_callback( + command=f"computer_use: {summary}", + description=f"Allow computer_use to perform `{action}`?", + ) + return { + "once": "approve_once", + "session": "approve_session", + "always": "always_approve", + "deny": "deny", + "timeout": "timeout", + }.get(verdict, "deny") + + def _handle_approval_selection(self) -> None: + """Process the currently selected dangerous-command approval choice.""" + state = self._approval_state + if not state: + return + + selected = state.get("selected", 0) + choices = state.get("choices") + if not isinstance(choices, list): + choices = [] + if not (0 <= selected < len(choices)): + return + + chosen = choices[selected] + if chosen == "view": + state["show_full"] = True + state["choices"] = [choice for choice in choices if choice != "view"] + if state["selected"] >= len(state["choices"]): + state["selected"] = max(0, len(state["choices"]) - 1) + self._invalidate() + return + + state["response_queue"].put(chosen) + self._approval_state = None + self._invalidate() + + def _secret_capture_callback(self, var_name: str, prompt: str, metadata=None) -> dict: + return prompt_for_secret(self, var_name, prompt, metadata) + + def _capture_modal_input_snapshot(self) -> None: + """Temporarily clear the input buffer and save the user's in-progress draft.""" + if self._modal_input_snapshot is not None or not getattr(self, "_app", None): + return + try: + buf = self._app.current_buffer + self._modal_input_snapshot = { + "text": buf.text, + "cursor_position": buf.cursor_position, + } + buf.reset() + except Exception: + self._modal_input_snapshot = None + + def _restore_modal_input_snapshot(self) -> None: + """Restore any draft text that was present before a modal prompt opened.""" + snapshot = self._modal_input_snapshot + self._modal_input_snapshot = None + if not snapshot or not getattr(self, "_app", None): + return + try: + buf = self._app.current_buffer + buf.text = snapshot.get("text", "") + buf.cursor_position = min(snapshot.get("cursor_position", 0), len(buf.text)) + except Exception: + pass + + def _clear_active_overlays_for_interrupt(self) -> None: + """Drain and clear every input-blocking overlay left by an interrupted agent. + + approval/clarify/sudo/secret prompts each block a worker thread on a + ``response_queue.get()``. When the agent is interrupted the worker + thread is torn down, but the overlay's state dict stays set — leaving + the CLI input gated (``read_only`` condition + keypress filter) with no + thread servicing the prompt. The result is a frozen terminal until the + prompt's own timeout expires. Push a terminal value onto each queue so + any still-blocked thread unblocks cleanly, then nil the state out and + restore the user's pre-modal draft (#14026). + + Safe default per prompt: approval -> "deny", clarify/sudo/secret -> + cancel (None / empty). Each step is wrapped so a dead queue can't + prevent clearing the others. + """ + if self._approval_state: + try: + self._approval_state["response_queue"].put("deny") + except Exception: + pass + self._approval_state = None + if self._clarify_state: + try: + self._clarify_state["response_queue"].put( + "The user cancelled. Use your best judgement to proceed." + ) + except Exception: + pass + self._clarify_state = None + self._clarify_freetext = False + self._clarify_multi_base = None + if self._sudo_state: + try: + self._sudo_state["response_queue"].put("") + except Exception: + pass + self._sudo_state = None + self._sudo_deadline = 0 + self._restore_modal_input_snapshot() + if self._secret_state: + try: + self._cancel_secret_capture() + except Exception: + self._secret_state = None + + def _submit_secret_response(self, value: str) -> None: + if not self._secret_state: + return + self._secret_state["response_queue"].put(value) + self._secret_state = None + self._secret_deadline = 0 + # Modal teardown — paint directly so the secret panel clears at once and + # isn't held by the _invalidate throttle/resize guard (#41098). + self._paint_now() + + def _cancel_secret_capture(self) -> None: + self._submit_secret_response("") + + def _clear_secret_input_buffer(self) -> None: + if getattr(self, "_app", None): + try: + self._app.current_buffer.reset() + except Exception: + pass diff --git a/hermes_cli/cli_model_switch_mixin.py b/hermes_cli/cli_model_switch_mixin.py new file mode 100644 index 0000000000..10d8eed26c --- /dev/null +++ b/hermes_cli/cli_model_switch_mixin.py @@ -0,0 +1,1151 @@ +"""Model picker, /model switch application, runtime snapshot/restore, and codex runtime handling for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import copy +import sys +import threading + +from rich.markup import escape as _escape +from utils import base_url_host_matches + + +class CLIModelSwitchMixin: + """Model picker, /model switch application, runtime snapshot/restore, and codex runtime handling for the interactive CLI""" + + def _normalize_model_for_provider(self, resolved_provider: str) -> bool: + """Normalize provider-specific model IDs and routing.""" + from cli import _split_model_config_default + current_model = str(self.model or "").strip() + if isinstance(self.model, dict): + _m, _ = _split_model_config_default(self.model) + current_model = _m + changed = False + + try: + from hermes_cli.model_normalize import ( + _AGGREGATOR_PROVIDERS, + normalize_model_for_provider, + ) + + if resolved_provider not in _AGGREGATOR_PROVIDERS: + normalized_model = normalize_model_for_provider(current_model, resolved_provider) + if normalized_model and normalized_model != current_model: + if not self._model_is_default: + self._console_print( + f"[yellow]⚠️ Normalized model '{current_model}' to '{normalized_model}' for {resolved_provider}.[/]" + ) + self.model = normalized_model + current_model = normalized_model + changed = True + except Exception: + pass + + if resolved_provider == "copilot": + try: + from hermes_cli.models import copilot_model_api_mode, normalize_copilot_model_id + + canonical = normalize_copilot_model_id(current_model, api_key=self.api_key) + if canonical and canonical != current_model: + if not self._model_is_default: + self._console_print( + f"[yellow]⚠️ Normalized Copilot model '{current_model}' to '{canonical}'.[/]" + ) + self.model = canonical + current_model = canonical + changed = True + + resolved_mode = copilot_model_api_mode(current_model, api_key=self.api_key) + if resolved_mode != self.api_mode: + self.api_mode = resolved_mode + changed = True + except Exception: + pass + return changed + + from hermes_cli.models import opencode_provider_family + + if opencode_provider_family(resolved_provider) is not None: + try: + from hermes_cli.models import normalize_opencode_model_id, opencode_model_api_mode + + canonical = normalize_opencode_model_id(resolved_provider, current_model) + if canonical and canonical != current_model: + if not self._model_is_default: + self._console_print( + f"[yellow]⚠️ Stripped provider prefix from '{current_model}'; using '{canonical}' for {resolved_provider}.[/]" + ) + self.model = canonical + current_model = canonical + changed = True + + resolved_mode = opencode_model_api_mode(resolved_provider, current_model) + if resolved_mode != self.api_mode: + self.api_mode = resolved_mode + changed = True + except Exception: + pass + return changed + + if resolved_provider != "openai-codex": + return changed + + # 1. Strip provider prefix ("openai/gpt-5.4" → "gpt-5.4") + if "/" in current_model: + slug = current_model.split("/", 1)[1] + if not self._model_is_default: + self._console_print( + f"[yellow]⚠️ Stripped provider prefix from '{current_model}'; " + f"using '{slug}' for OpenAI Codex.[/]" + ) + self.model = slug + current_model = slug + changed = True + + # 2. Replace untouched default with a Codex model + if self._model_is_default: + fallback_model = "gpt-5.3-codex" + try: + from hermes_cli.codex_models import get_codex_model_ids + + available = get_codex_model_ids( + access_token=self.api_key if self.api_key else None, + ) + if available: + fallback_model = available[0] + except Exception: + pass + + if current_model != fallback_model: + self.model = fallback_model + changed = True + + return changed + + def _persist_model_switch_to_session(self, result) -> None: + """Persist a session-scoped /model switch to the session DB row. + + Writes the model column plus the runtime route so ``--resume`` + (CLI, reads ``gateway_runtime``) and ``session.resume`` (TUI/desktop, + reads top-level ``model_config`` keys via + ``_stored_session_runtime_overrides``) both restore the switched + provider instead of recombining the model with the ambient default + (#79536). Mirrors the gateway's ``update_session_model()`` call. + getattr: tests drive the switch paths with ``object.__new__`` stubs. + """ + from cli import logger + db = getattr(self, "_session_db", None) + sid = getattr(self, "session_id", None) + if not db or not sid: + return + provider = result.target_provider + # Bare "custom" is the resolved billing class, not a routable + # identity — persisting it verbatim makes a later resume hard-fail + # when the config default has moved off the custom endpoint + # (resolve_runtime_provider only trusts config base_url for bare + # custom while the config provider is still custom-ish). Heal to + # the durable custom: menu key, else drop the provider — + # same recovery the TUI gateway applies on its read path. + if str(provider or "").strip().lower() == "custom": + try: + from hermes_cli.runtime_provider import canonical_custom_identity + provider = canonical_custom_identity( + base_url=result.base_url or None, + model=result.new_model or None, + ) or None + except Exception: + provider = None + # Both shapes use the same or-None discipline so stale keys from a + # previous switch are deleted (not merely omitted) in BOTH the + # nested gateway_runtime dict (CLI reader) and the top-level keys + # (TUI gateway reader). _merge_model_config_json only deletes on + # explicit None, so falsy values must be converted, not filtered. + # Deriving the top-level from **route guarantees the two shapes + # can never diverge — the asymmetry that caused the original + # stale-key bug (#85261 simplify-code review). + route = { + "provider": provider or None, + "base_url": result.base_url or None, + "api_mode": result.api_mode or None, + } + try: + db.update_session_model(sid, result.new_model) + db.patch_session_model_config(sid, { + "gateway_runtime": route, + **route, + }) + except Exception: + logger.debug( + "Failed to persist model switch to session DB", exc_info=True + ) + + def _restore_session_model(self, session_meta: dict, *, quiet: bool = False) -> None: + """Restore model/provider from the session DB row on resume. + + Companion to ``_restore_session_cwd`` / ``_restore_session_yolo`` — + called from every resume path (startup ``--resume``/``-c`` and + mid-chat ``/resume``). The persisted model lives in the session row's + ``model`` column (written at creation time and updated on ``/model`` + switches via ``update_session_model``); the provider/endpoint live in + ``model_config.gateway_runtime`` (written by the gateway's + ``_sync_session_model_from_agent`` and the CLI ``/model`` persist). + Without this restore a resumed session silently falls back to the + config default model, losing the user's last ``/model`` choice. + + When the stored provider differs from the ambient one, credentials + are re-resolved for the stored provider (mirroring the gateway's + ``_rehydrate_session_model_override``) — the ambient ``self.api_key`` + belongs to the config-default provider and must not be sent to the + session's endpoint. On resolution failure the ambient credentials are + kept so the session still opens (the first turn surfaces the auth + error instead of the resume dying). + + Skips when the session has no model recorded or when the CLI was + launched with an explicit ``-m`` override (user intent wins). + """ + from cli import logger + stored_model = (session_meta or {}).get("model") + if not stored_model: + return + # An explicit -m / --model on the command line overrides resume. + if getattr(self, "_explicit_model_override", False): + return + # Stored provider/endpoint via the canonical row-level reader + # (prefers model_config.gateway_runtime, falls back to the TUI + # gateway's top-level keys). + from hermes_state import SessionDB as _SessionDB + _stored_runtime = _SessionDB.session_gateway_runtime(session_meta) + stored_provider = _stored_runtime.get("provider") or None + stored_base_url = _stored_runtime.get("base_url") or None + stored_api_mode = _stored_runtime.get("api_mode") or None + # Heal bare "custom" persisted by older builds / gateway turns: it's + # the resolved billing class, not a routable identity. Recover the + # durable custom: menu key from the endpoint, else drop the + # provider so resume keeps the ambient default. (Stricter than the + # TUI gateway's recovery, which keeps bare "custom" when a base_url + # exists — the CLI's resolve path would hard-fail on it, #14676.) + if str(stored_provider or "").strip().lower() == "custom": + try: + from hermes_cli.runtime_provider import canonical_custom_identity + stored_provider = canonical_custom_identity( + base_url=stored_base_url or None, + model=stored_model or None, + ) or None + except Exception: + stored_provider = None + model_changed = stored_model != self.model + provider_changed = bool(stored_provider) and stored_provider != self.provider + if not model_changed and not provider_changed: + return + self.model = stored_model + if stored_provider: + self.provider = stored_provider + self.requested_provider = stored_provider + if stored_base_url: + self.base_url = stored_base_url + if stored_api_mode: + self.api_mode = stored_api_mode + if provider_changed: + # Stale launch-time explicit overrides belong to the AMBIENT + # provider; carrying them into the restored provider's + # resolution poisons _ensure_runtime_credentials on startup + # resume (same leak _apply_model_switch_result guards against + # by overwriting _explicit_* on every switch). + self._explicit_api_key = None + self._explicit_base_url = stored_base_url + # Re-resolve credentials for the restored provider. api_key is + # never persisted to the session DB (by design) — the normal + # runtime provider resolution owns credentials. + try: + from hermes_cli.runtime_provider import resolve_runtime_provider + resolved = resolve_runtime_provider(requested=stored_provider) + if resolved.get("api_key"): + self.api_key = resolved["api_key"] + self._credential_pool = resolved.get("credential_pool") + if not stored_base_url and resolved.get("base_url"): + self.base_url = resolved["base_url"] + if not stored_api_mode and resolved.get("api_mode"): + self.api_mode = resolved["api_mode"] + except Exception: + logger.debug( + "Credential re-resolution for resumed session provider " + "%s failed; keeping ambient credentials", + stored_provider, exc_info=True, + ) + # If the agent is already running (mid-chat /resume), swap it + # in-place so the next turn uses the restored model. On startup + # --resume the agent isn't built yet — _init_agent will pick up + # self.model / self.provider when constructing AIAgent. + if self.agent is not None: + try: + self.agent.switch_model( + new_model=self.model, + new_provider=self.provider, + api_key=self.api_key or "", + base_url=self.base_url or "", + api_mode=self.api_mode or "", + ) + except Exception: + logger.debug( + "In-place agent model swap on resume failed", exc_info=True + ) + msg = f"Model restored from session: {stored_model}" + if stored_provider: + msg += f" ({stored_provider})" + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + + def _open_model_picker(self, providers: list, current_model: str, current_provider: str, user_provs=None, custom_provs=None) -> None: + """Open prompt_toolkit-native /model picker modal.""" + self._capture_modal_input_snapshot() + default_idx = next((i for i, p in enumerate(providers) if p.get("is_current")), 0) + self._model_picker_state = { + "stage": "provider", + "providers": providers, + "selected": default_idx, + "current_model": current_model, + "current_provider": current_provider, + "user_provs": user_provs, + "custom_provs": custom_provs, + "filter": "", + } + self._invalidate(min_interval=0.0) + + def _confirm_expensive_model_switch(self, result) -> bool: + """Ask for explicit confirmation before applying costly model switches.""" + if not getattr(result, "success", False): + return True + try: + from hermes_cli.model_selection_guards import combined_selection_warning + + warning = combined_selection_warning( + result.new_model, + provider=result.target_provider, + base_url=result.base_url or self.base_url or "", + api_key=result.api_key or self.api_key or "", + model_info=result.model_info, + ) + except Exception: + warning = None + if warning is None: + return True + + choices = [ + ("once", "Switch anyway", "Use this model for the current Hermes session."), + ("cancel", "Cancel", "Keep the current model."), + ] + raw = self._prompt_text_input_modal( + title=f"!!! {warning.title} !!!", + detail=warning.message, + choices=choices, + timeout=120, + ) + choice = self._normalize_slash_confirm_choice(raw, choices) + return choice == "once" + + def _confirm_and_apply_model_switch_result( + self, result, persist_global: bool, custom_providers=None + ) -> None: + from cli import _cprint + try: + if result.success and not self._confirm_expensive_model_switch(result): + _cprint(" Model switch cancelled.") + return + self._apply_model_switch_result( + result, persist_global, custom_providers=custom_providers + ) + except Exception as exc: + _cprint(f" ✗ Model selection failed: {exc}") + + def _close_model_picker(self) -> None: + self._model_picker_state = None + self._restore_modal_input_snapshot() + self._invalidate(min_interval=0.0) + + def _snapshot_model_runtime(self) -> dict: + """Capture current CLI and agent model runtime for one-turn restore.""" + agent = getattr(self, "agent", None) + return { + "model": self.model, + "provider": self.provider, + "requested_provider": self.requested_provider, + "_explicit_api_key": getattr(self, "_explicit_api_key", None), + "_explicit_base_url": getattr(self, "_explicit_base_url", None), + "api_key": self.api_key, + "base_url": self.base_url, + "api_mode": self.api_mode, + "agent_primary_runtime": copy.deepcopy( + getattr(agent, "_primary_runtime", None) + ) if agent is not None else None, + } + + def _restore_model_runtime_snapshot(self, snapshot: dict | None) -> None: + """Restore a model runtime captured before a one-turn override.""" + from cli import logger + if not snapshot: + return + for key in ( + "model", + "provider", + "requested_provider", + "_explicit_api_key", + "_explicit_base_url", + "api_key", + "base_url", + "api_mode", + ): + if key in snapshot: + setattr(self, key, snapshot.get(key)) + + agent = getattr(self, "agent", None) + if agent is None: + return + + primary = snapshot.get("agent_primary_runtime") + if primary and hasattr(agent, "_restore_primary_runtime"): + try: + agent._primary_runtime = copy.deepcopy(primary) + agent._fallback_activated = True + agent._rate_limited_until = 0 + if agent._restore_primary_runtime(): + return + except Exception: + logger.debug("CLI one-turn model restore via primary runtime failed", exc_info=True) + + if hasattr(agent, "switch_model"): + try: + agent.switch_model( + new_model=snapshot.get("model", ""), + new_provider=snapshot.get("provider", ""), + api_key=snapshot.get("api_key", ""), + base_url=snapshot.get("base_url", ""), + api_mode=snapshot.get("api_mode", ""), + capabilities=snapshot.get("capabilities"), + ) + except Exception as exc: + logger.warning("CLI one-turn model restore failed: %s", exc) + + @staticmethod + def _filter_model_picker_entries(entries: list, query: str) -> list: + """Return (original_index, label) pairs for entries matching ``query``. + + Subsequence ("fuzzy") match, case-insensitive: the query characters + must appear in order in the label. An empty query matches everything. + Crucially the returned pairs carry the ORIGINAL index into ``entries``, + so a selection in the filtered view still resolves to exactly one + concrete model — filtering only narrows the list, it never introduces + an ambiguous or fuzzy *resolution* (the anti-"claude→old-model" rule). + """ + pairs = list(enumerate(entries)) + q = (query or "").strip().lower() + if not q: + return pairs + + def _subseq(needle: str, hay: str) -> bool: + it = iter(hay) + return all(ch in it for ch in needle) + + out = [(i, e) for (i, e) in pairs if _subseq(q, str(e).lower())] + return out + + @staticmethod + def _compute_model_picker_viewport( + selected: int, + scroll_offset: int, + n: int, + term_rows: int, + reserved_below: int = 6, + panel_chrome: int = 6, + min_visible: int = 3, + ) -> tuple[int, int]: + """Resolve (scroll_offset, visible) for the /model picker viewport. + + ``reserved_below`` matches the approval / clarify panels — input area, + status bar, and separators below the panel. ``panel_chrome`` covers + this panel's own borders + blanks + hint row. The remaining rows hold + the scrollable list, with the offset slid to keep ``selected`` on screen. + """ + max_visible = max(min_visible, term_rows - reserved_below - panel_chrome) + if n <= max_visible: + return 0, n + visible = max_visible + if selected < scroll_offset: + scroll_offset = selected + elif selected >= scroll_offset + visible: + scroll_offset = selected - visible + 1 + scroll_offset = max(0, min(scroll_offset, n - visible)) + return scroll_offset, visible + + def _clear_persisted_context_for_model_switch(self, result) -> None: + """Drop a global context pin when its configured owner changes.""" + from cli import save_config_value + try: + from hermes_cli.config import load_config_readonly + from hermes_cli.route_identity import should_clear_context_pin + + config = load_config_readonly() + model_cfg = config.get("model", {}) if isinstance(config, dict) else {} + if not isinstance(model_cfg, dict) or "context_length" not in model_cfg: + return + if should_clear_context_pin( + model_cfg.get("default") or model_cfg.get("model"), + result.new_model, + model_cfg.get("base_url"), + result.base_url, + model_cfg.get("provider"), + result.target_provider, + ): + save_config_value("model.context_length", None) + except Exception: + save_config_value("model.context_length", None) + + def _apply_model_switch_result( + self, result, persist_global: bool, custom_providers=None + ) -> None: + from cli import HermesCLI, _cprint, logger, save_config_value + if not result.success: + _cprint(f" ✗ {result.error_message}") + return + + if self.agent is not None: + try: + from hermes_cli.context_switch_guard import merge_preflight_compression_warning + + # Prefer the fresh inventory list (same source as switch_model / + # TUI); fall back to the agent-init snapshot. + _cp = ( + custom_providers + if custom_providers is not None + else getattr(self.agent, "_custom_providers", None) + ) + merge_preflight_compression_warning( + result, + agent=self.agent, + messages=list(self.conversation_history or []), + custom_providers=_cp, + config_context_length=getattr(self.agent, "_config_context_length", None), + ) + except Exception as exc: + logger.debug("preflight-compression switch warning failed: %s", exc) + + old_model = self.model + # Snapshot the CLI-level credential/runtime fields BEFORE mutating them + # so a failed in-place agent swap can roll the whole CLI back to the old + # working model. Otherwise the broken credentials staged below leak into + # the next turn's resolution even though the agent itself rolled back + # (#50163). + _cli_snapshot = { + "model": self.model, + "provider": self.provider, + "requested_provider": self.requested_provider, + "_explicit_api_key": getattr(self, "_explicit_api_key", None), + "_explicit_base_url": getattr(self, "_explicit_base_url", None), + "api_key": self.api_key, + "base_url": self.base_url, + "api_mode": self.api_mode, + } + self.model = result.new_model + self.provider = result.target_provider + self.requested_provider = result.target_provider + # Always overwrite explicit overrides so stale credentials from the + # previous provider (e.g. Ollama api_key/base_url) don't leak into + # the new provider's credential resolution on the next turn. + self._explicit_api_key = result.api_key + self._explicit_base_url = result.base_url + if result.api_key: + self.api_key = result.api_key + if result.base_url: + self.base_url = result.base_url + if result.api_mode: + self.api_mode = result.api_mode + + if self.agent is not None: + try: + self.agent.switch_model( + new_model=result.new_model, + new_provider=result.target_provider, + api_key=result.api_key, + base_url=result.base_url, + api_mode=result.api_mode, + capabilities=getattr(result, "runtime_capabilities", None), + ) + except Exception as exc: + # The agent rolled itself back to the old working model/client. + # Roll the CLI's own staged fields back too and abort the rest + # of the commit (note + success print) so a failed switch is a + # no-op rather than a dead session (#50163). + for _k, _v in _cli_snapshot.items(): + setattr(self, _k, _v) + _cprint( + f" ⚠ Model switch to {result.new_model} failed ({exc}); " + f"staying on {old_model}." + ) + return + + from hermes_cli.model_switch import format_model_for_display + _display_old = format_model_for_display(old_model) + _display_new = format_model_for_display(result.new_model) + + self._pending_model_switch_note = ( + f"[Note: model was just switched from {_display_old} to {_display_new} " + f"via {result.provider_label or result.target_provider}. " + f"Adjust your self-identification accordingly.]" + ) + + provider_label = result.provider_label or result.target_provider + _cprint(f" ✓ Model switched: {_display_new}") + _cprint(f" Provider: {provider_label}") + + # Context: always resolve via the provider-aware chain so Codex OAuth, + # Copilot, and Nous-enforced caps win over the raw models.dev entry + # (e.g. gpt-5.5 is 1.05M on openai but 272K on Codex OAuth). + mi = result.model_info + try: + from hermes_cli.model_switch import resolve_display_context_length + ctx = resolve_display_context_length( + result.new_model, + result.target_provider, + base_url=result.base_url or self.base_url or "", + api_key=result.api_key or self.api_key or "", + model_info=mi, + config_context_length=getattr(self.agent, "_config_context_length", None) if self.agent else None, + custom_providers=getattr(self.agent, "_custom_providers", None) if self.agent else None, + ) + if ctx: + _cprint(f" Context: {ctx:,} tokens") + except Exception: + pass + if mi: + if mi.max_output: + _cprint(f" Max output: {mi.max_output:,} tokens") + _cprint(f" Capabilities: {mi.format_capabilities()}") + + cache_enabled = ( + (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) + or result.api_mode == "anthropic_messages" + ) + if cache_enabled: + _cprint(" Prompt caching: enabled") + if result.warning_message: + _cprint(f" ⚠ {result.warning_message}") + if persist_global: + HermesCLI._clear_persisted_context_for_model_switch(self, result) + save_config_value("model.default", result.new_model) + save_config_value("model.provider", result.target_provider) + # base_url/api_mode were previously never persisted here, so a + # global switch left the OLD provider's endpoint/wire-protocol in + # config.yaml. result.base_url/api_mode are always freshly + # resolved for the target provider (see model_switch.py), so sync + # them every time; None clears a value the new provider doesn't + # need (#25106). + save_config_value("model.base_url", result.base_url or None) + save_config_value("model.api_mode", result.api_mode or None) + _cprint(" Saved to config.yaml (--global)") + else: + _cprint(" (session only — add --global to persist)") + + # Persist the switch to this session's row so --resume / + # session.resume restore it. --global also updates config.yaml + # (future sessions), but the row still records what THIS session + # actually runs — otherwise a later resume would restore the stale + # creation-time model over the user's new global choice. + HermesCLI._persist_model_switch_to_session(self, result) + + def _handle_model_picker_selection(self, persist_global: bool = False) -> None: + state = self._model_picker_state + if not state: + return + selected = state.get("selected", 0) + stage = state.get("stage") + if stage == "provider": + providers = state.get("providers") or [] + if selected >= len(providers): + self._close_model_picker() + return + provider_data = providers[selected] + # Use the curated model list from list_authenticated_providers() + # (same lists as `hermes model` and gateway pickers). + # Only fall back to the live provider catalog when the curated + # list is empty (e.g. user-defined endpoints with no curated list). + model_list = provider_data.get("models", []) + if not model_list: + try: + from hermes_cli.models import provider_model_ids + live = provider_model_ids(provider_data["slug"]) + if live: + model_list = live + except Exception: + pass + state["stage"] = "model" + state["provider_data"] = provider_data + state["model_list"] = model_list + state["selected"] = 0 + state["filter"] = "" + state["_filtered_pairs"] = None + self._invalidate(min_interval=0.0) + return + if stage == "model": + provider_data = state.get("provider_data") or {} + model_list = state.get("model_list") or [] + # Map the selected row through the active fuzzy filter so the + # index lines up with what the picker is currently showing. The + # filtered pair carries the ORIGINAL index into model_list, so the + # resolved model is always one concrete, unambiguous entry. + filtered_pairs = state.get("_filtered_pairs") + if filtered_pairs is None: + filtered_pairs = list(enumerate(model_list)) + visible_labels = [e for (_i, e) in filtered_pairs] + back_idx = len(visible_labels) + cancel_idx = len(visible_labels) + 1 + if selected == back_idx: + state["stage"] = "provider" + state["filter"] = "" + state["_filtered_pairs"] = None + state["selected"] = next((i for i, p in enumerate(state.get("providers") or []) if p.get("slug") == provider_data.get("slug")), 0) + self._invalidate(min_interval=0.0) + return + if selected >= cancel_idx: + self._close_model_picker() + return + if 0 <= selected < len(visible_labels): + from hermes_cli.model_switch import switch_model + chosen_model = visible_labels[selected] + result = switch_model( + raw_input=chosen_model, + current_provider=self.provider or "", + current_model=self.model or "", + current_base_url=self.base_url or "", + current_api_key=self.api_key or "", + is_global=persist_global, + explicit_provider=provider_data.get("slug"), + user_providers=state.get("user_provs"), + custom_providers=state.get("custom_provs"), + ) + # Capture before close — picker state is cleared on close. + _picker_custom_provs = state.get("custom_provs") + self._close_model_picker() + if getattr(self, "_app", None): + threading.Thread( + target=self._confirm_and_apply_model_switch_result, + args=(result, persist_global, _picker_custom_provs), + daemon=True, + ).start() + else: + self._confirm_and_apply_model_switch_result( + result, persist_global, custom_providers=_picker_custom_provs + ) + return + self._close_model_picker() + + def _handle_model_switch(self, cmd_original: str): + """Handle /model command — switch model. + + Supports: + /model — show current model + usage hints + /model — switch model (this session only) + /model --once — switch for the next turn only + /model --session — switch for this session only (explicit) + /model --global — switch and persist to config.yaml + /model --provider — switch provider + model + /model --provider — switch to provider, auto-detect model + + Persistence defaults to off (``model.persist_switch_by_default`` in + config.yaml, default False — switches are session-scoped). Use + ``--global`` to persist, or ``--once`` for the next turn only. + """ + from cli import _cprint, logger + from hermes_cli.model_switch import ( + switch_model, + parse_model_switch_args, + resolve_persist_behavior, + ) + from hermes_cli.providers import get_label + + # Parse args from the original command + parts = cmd_original.split(None, 1) # split off '/model' + raw_args = parts[1].strip() if len(parts) > 1 else "" + + # Parse --provider, --global, --session, --once, and --refresh flags + # via the shared single-owner parser (hermes_cli.model_switch). + request = parse_model_switch_args(raw_args) + model_input = request.target + explicit_provider = request.explicit_provider + is_global_flag = request.is_global + force_refresh = request.force_refresh + is_session = request.is_session + one_turn = request.is_once + if request.errors: + # CLI decoration: " ✗ " prefix over the canonical error copy. + _cprint(f" ✗ {request.error_messages()[0]}") + return + # Resolve the effective persistence once: --global forces persist, + # --session/--once force session-scope, otherwise defer to + # model.persist_switch_by_default (defaults to False so /model is + # session-scoped unless the user opts in). + persist_global = resolve_persist_behavior( + is_global_flag, is_session, is_once=one_turn, + explicit_provider=explicit_provider, + ) + + # --refresh: wipe the on-disk picker cache before building the + # provider list. Forces a live re-fetch of every authed provider's + # /v1/models endpoint on this open. + if force_refresh: + try: + from hermes_cli.models import clear_provider_models_cache + clear_provider_models_cache() + _cprint(" Cleared model picker cache. Refreshing...") + except Exception: + pass + + # Single inventory context — replaces the inline config-slice the + # dashboard / TUI used to duplicate. Overlay live session state + # via with_overrides (truthy-only) so empty self.* attrs don't + # clobber disk config. + from hermes_cli.inventory import build_models_payload, load_picker_context + + try: + ctx = load_picker_context().with_overrides( + current_provider=self.provider or "", + current_model=self.model or "", + current_base_url=self.base_url or "", + ) + except Exception: + ctx = None + + # switch_model() + _open_model_picker still need the raw provider + # dicts; ConfigContext is the canonical source for both. + user_provs = ctx.user_providers if ctx is not None else None + custom_provs = ctx.custom_providers if ctx is not None else None + + # No args at all: open prompt_toolkit-native picker modal + if not model_input and not explicit_provider: + model_display = self.model or "unknown" + provider_display = get_label(self.provider) if self.provider else "unknown" + + try: + if ctx is None: + raise RuntimeError("inventory context unavailable") + providers = build_models_payload( + ctx, + probe_custom_providers=force_refresh, + probe_current_custom_provider=not force_refresh, + )["providers"] + except Exception: + providers = [] + + if not providers: + _cprint(" No authenticated providers found.") + _cprint("") + _cprint(" /model switch model (this session)") + _cprint(" /model --global switch model and persist as default") + _cprint(" /model --once switch for the next turn only") + _cprint(" /model --session switch for this session only") + _cprint(" /model --provider switch provider") + _cprint(" /model --refresh re-fetch live model lists") + return + + self._open_model_picker( + providers, + model_display, + provider_display, + user_provs=user_provs, + custom_provs=custom_provs, + ) + return + + # Perform the switch + result = switch_model( + raw_input=model_input, + current_provider=self.provider or "", + current_model=self.model or "", + current_base_url=self.base_url or "", + current_api_key=self.api_key or "", + is_global=persist_global, + explicit_provider=explicit_provider, + user_providers=user_provs, + custom_providers=custom_provs, + ) + + if not result.success: + _cprint(f" ✗ {result.error_message}") + return + + if self.agent is not None: + try: + from hermes_cli.context_switch_guard import merge_preflight_compression_warning + + merge_preflight_compression_warning( + result, + agent=self.agent, + messages=list(self.conversation_history or []), + # Same fresh inventory list passed to switch_model above. + custom_providers=custom_provs + if custom_provs is not None + else getattr(self.agent, "_custom_providers", None), + config_context_length=getattr(self.agent, "_config_context_length", None), + ) + except Exception as exc: + logger.debug("preflight-compression switch warning failed: %s", exc) + + # Run the confirm + apply sequence off the main thread. The + # expensive-model confirmation modal blocks the calling thread on a + # response queue (see _prompt_text_input_modal); running it on the + # prompt_toolkit main thread freezes TUI rendering, so the modal never + # appears and the switch silently cancels after the 120s timeout. + # Mirror the picker path (_handle_model_picker_selection), which + # already dispatches confirm+apply on a worker thread. + if getattr(self, "_app", None): + threading.Thread( + target=self._confirm_and_apply_cli_model_switch, + args=(result, persist_global, one_turn, custom_provs), + daemon=True, + ).start() + return + self._confirm_and_apply_cli_model_switch( + result, persist_global, one_turn, custom_provs + ) + return + + def _confirm_and_apply_cli_model_switch( + self, result, persist_global: bool, one_turn: bool, custom_provs=None + ) -> None: + """Confirm an expensive model switch and apply it to CLI state. + + Runs on a worker thread when the TUI is active (see + _handle_model_switch) so the confirmation modal can render. + """ + from cli import HermesCLI, _cprint, save_config_value + if not self._confirm_expensive_model_switch(result): + _cprint(" Model switch cancelled.") + return + + # Apply to CLI state. + # Update requested_provider so _ensure_runtime_credentials() doesn't + # overwrite the switch on the next turn (it re-resolves from this). + old_model = self.model + _one_turn_restore_snapshot = self._snapshot_model_runtime() if one_turn else None + # Snapshot CLI-level fields before mutation so a failed in-place swap + # rolls the whole CLI back to the old working model (#50163). + _cli_snapshot = { + "model": self.model, + "provider": self.provider, + "requested_provider": self.requested_provider, + "_explicit_api_key": getattr(self, "_explicit_api_key", None), + "_explicit_base_url": getattr(self, "_explicit_base_url", None), + "api_key": self.api_key, + "base_url": self.base_url, + "api_mode": self.api_mode, + } + self.model = result.new_model + self.provider = result.target_provider + self.requested_provider = result.target_provider + # Always overwrite explicit overrides so stale credentials from the + # previous provider (e.g. Ollama api_key/base_url) don't leak into + # the new provider's credential resolution on the next turn. + self._explicit_api_key = result.api_key + self._explicit_base_url = result.base_url + if result.api_key: + self.api_key = result.api_key + if result.base_url: + self.base_url = result.base_url + if result.api_mode: + self.api_mode = result.api_mode + + # Apply to running agent (in-place swap) + if self.agent is not None: + try: + self.agent.switch_model( + new_model=result.new_model, + new_provider=result.target_provider, + api_key=result.api_key, + base_url=result.base_url, + api_mode=result.api_mode, + capabilities=getattr(result, "runtime_capabilities", None), + ) + except Exception as exc: + # Agent rolled itself back; roll the CLI back too and abort so a + # failed switch is a no-op rather than a dead session (#50163). + for _k, _v in _cli_snapshot.items(): + setattr(self, _k, _v) + _cprint( + f" ⚠ Model switch to {result.new_model} failed ({exc}); " + f"staying on {old_model}." + ) + return + + # Store a note to prepend to the next user message so the model + # knows a switch occurred (avoids injecting system messages mid-history + # which breaks providers and prompt caching). + from hermes_cli.model_switch import format_model_for_display + _display_old = format_model_for_display(old_model) + _display_new = format_model_for_display(result.new_model) + + self._pending_model_switch_note = ( + f"[Note: model was just switched from {_display_old} to {_display_new} " + f"via {result.provider_label or result.target_provider}. " + f"{'This override applies to the next turn only. ' if one_turn else ''}" + f"Adjust your self-identification accordingly.]" + ) + if one_turn: + self._pending_one_turn_model_restore = _one_turn_restore_snapshot + else: + self._pending_one_turn_model_restore = None + + # Display confirmation with full metadata + provider_label = result.provider_label or result.target_provider + _cprint(f" ✓ Model switched: {_display_new}") + _cprint(f" Provider: {provider_label}") + + # Context: always resolve via the provider-aware chain so Codex OAuth, + # Copilot, and Nous-enforced caps win over the raw models.dev entry + # (e.g. gpt-5.5 is 1.05M on openai but 272K on Codex OAuth). + mi = result.model_info + from hermes_cli.model_switch import resolve_display_context_length + ctx = resolve_display_context_length( + result.new_model, + result.target_provider, + base_url=result.base_url or self.base_url or "", + api_key=result.api_key or self.api_key or "", + model_info=mi, + config_context_length=getattr(self.agent, "_config_context_length", None) if self.agent else None, + custom_providers=getattr(self.agent, "_custom_providers", None) if self.agent else None, + ) + if ctx: + _cprint(f" Context: {ctx:,} tokens") + if mi: + if mi.max_output: + _cprint(f" Max output: {mi.max_output:,} tokens") + _cprint(f" Capabilities: {mi.format_capabilities()}") + + # Cache notice + cache_enabled = ( + (base_url_host_matches(result.base_url or "", "openrouter.ai") and "claude" in result.new_model.lower()) + or result.api_mode == "anthropic_messages" + ) + if cache_enabled: + _cprint(" Prompt caching: enabled") + + # Warning from validation + if result.warning_message: + _cprint(f" ⚠ {result.warning_message}") + + # Persistence + if persist_global: + HermesCLI._clear_persisted_context_for_model_switch(self, result) + save_config_value("model.default", result.new_model) + save_config_value("model.provider", result.target_provider) + # See _apply_model_switch_result above for why base_url/api_mode + # must be synced on every global switch (#25106). + save_config_value("model.base_url", result.base_url or None) + save_config_value("model.api_mode", result.api_mode or None) + _cprint(" Saved to config.yaml") + elif one_turn: + _cprint(" (next turn only — restores after one response)") + else: + _cprint(" (session only — add --global to persist)") + + # Persist the switch to this session's row so --resume / + # session.resume restore it (--global also updates config.yaml but + # the row still records what THIS session runs; --once is ephemeral + # and restored after one turn, so it must not touch the row). + if not one_turn: + HermesCLI._persist_model_switch_to_session(self, result) + + def _handle_codex_runtime(self, cmd_original: str) -> None: + """Handle /codex-runtime — toggle the codex app-server runtime opt-in. + + Usage: + /codex-runtime — show current state + /codex-runtime auto — Hermes default (chat_completions) + /codex-runtime codex_app_server — hand turns to codex subprocess + /codex-runtime on / off — synonyms for the above + """ + from cli import _cprint + from hermes_cli import codex_runtime_switch as crs + + parts = cmd_original.split(None, 1) + raw_args = parts[1].strip() if len(parts) > 1 else "" + new_value, errors = crs.parse_args(raw_args) + if errors: + for err in errors: + _cprint(f"❌ {err}") + return + + # Load + persist via the existing config helpers + try: + from hermes_cli.config import load_config, save_config + except Exception as exc: + _cprint(f"❌ could not load config: {exc}") + return + cfg = load_config() + + result = crs.apply( + cfg, + new_value, + persist_callback=(save_config if new_value is not None else None), + ) + + prefix = "✓" if result.success else "✗" + for line in result.message.splitlines(): + _cprint(f" {prefix} {line}" if line.startswith("openai_runtime") + else f" {line}") + if result.success and result.requires_new_session: + _cprint(" Tip: `/reset` starts a new session immediately.") + + def _should_handle_model_command_inline(self, text: str, has_images: bool = False) -> bool: + """Return True when /model should be handled immediately on the UI thread.""" + from cli import _looks_like_slash_command + if not text or has_images or not _looks_like_slash_command(text): + return False + try: + from hermes_cli.commands import resolve_command + base = text.split(None, 1)[0].lower().lstrip('/') + cmd = resolve_command(base) + return bool(cmd and cmd.name == "model") + except Exception: + return False + + def _cmd_moa(self, cmd_original: str): + # /moa is one-shot sugar only: run a single prompt through the + # default MoA preset, then restore the prior model. To *switch* to a + # MoA preset for the session, pick it from the model picker (MoA + # presets surface as a virtual "Mixture of Agents" provider). + from cli import _cprint, _slash_args + from hermes_cli.moa_config import ( + moa_usage, + normalize_moa_config, + ) + + payload = _slash_args(cmd_original) + if not payload: + _cprint(f" {moa_usage()}") + return True + moa_cfg = self.config.get("moa") if isinstance(self.config, dict) else {} + normalized = normalize_moa_config(moa_cfg) + preset = normalized["default_preset"] + self._pending_moa_restore_model = { + "requested_provider": getattr(self, "requested_provider", None), + "provider": getattr(self, "provider", None), + "model": getattr(self, "model", None), + "api_key": getattr(self, "api_key", None), + "base_url": getattr(self, "base_url", None), + "api_mode": getattr(self, "api_mode", None), + } + self.requested_provider = "moa" + self.provider = "moa" + self.model = preset + self.api_key = "moa-virtual-provider" + self.base_url = "moa://local" + self.api_mode = "chat_completions" + self.agent = None + self._pending_moa_disable_after_turn = True + self._pending_agent_seed = payload + _cprint(f" MoA one-shot queued with preset {preset}; previous model will be restored after this turn.") diff --git a/hermes_cli/cli_session_mixin.py b/hermes_cli/cli_session_mixin.py new file mode 100644 index 0000000000..4fd1acf21b --- /dev/null +++ b/hermes_cli/cli_session_mixin.py @@ -0,0 +1,1831 @@ +"""Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo persistence, manual compression, and exit summary + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import os +import shutil +import sys +import uuid + +from hermes_constants import get_hermes_home +from pathlib import Path +from rich.console import Console +from rich.markup import escape as _escape +from typing import Any, Dict, List, Optional + + +class CLISessionMixin: + """Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo persistence, manual compression, and exit summary""" + + def _restore_session_cwd(self, session_meta: dict, *, quiet: bool = False) -> None: + """Relaunch a resumed session in the directory it was started from. + + Idempotent and safe to call from every resume path. When the stored + ``cwd`` differs from the current process directory, we both + ``os.chdir()`` (so the process and any ``os.getcwd()`` fallback agree) + and retarget ``TERMINAL_CWD`` (so the terminal tool, code-exec tool, + and relative-path resolution all land in the same place — the local + terminal backend snapshots cwd on first use, which happens after this). + + No-ops when: the session recorded no cwd (gateway/remote/older + sessions), the directory no longer exists, or we're already there. + A missing directory degrades to a single dim warning rather than a + crash — repos get moved and deleted. + """ + recorded = (session_meta or {}).get("cwd") + if not recorded: + return + recorded = os.path.expanduser(str(recorded)) + try: + current = os.getcwd() + except OSError: + current = None + if current and os.path.realpath(recorded) == os.path.realpath(current): + return # Already where the session lived — nothing to announce. + + if not os.path.isdir(recorded): + msg = f"⚠ Session's working directory is gone: {recorded} — staying in {current or '.'}" + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + return + + try: + os.chdir(recorded) + except OSError as e: + msg = f"⚠ Could not enter session's working directory {recorded}: {e}" + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + return + + # Retarget the terminal/code-exec tools to match the process cwd. + os.environ["TERMINAL_CWD"] = recorded + + msg = f"↻ Working directory: {recorded}" + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + + def _restore_session_yolo(self, session_meta: dict, *, quiet: bool = False) -> None: + """Re-enable YOLO bypass on resume when the session had it on. + + Companion to ``_restore_session_cwd`` — called from every resume path + (startup ``--resume``/``-c`` and mid-chat ``/resume``). The persisted + flag lives in the session row's ``model_config.yolo_mode`` (written by + ``/yolo`` toggles and ``--yolo`` launches); without this restore the + in-memory ``tools.approval._session_yolo`` set starts empty in a fresh + process and the user's bypass silently reverts. + + No-op when the flag is absent/false, when YOLO is already active for + this session (idempotent across repeated resume paths), or when the + process was itself launched with ``--yolo`` (frozen bypass already + covers everything). + """ + try: + from hermes_state import SessionDB + from tools.approval import ( + _YOLO_MODE_FROZEN, + enable_session_yolo, + is_session_yolo_enabled, + ) + except Exception: + return + if _YOLO_MODE_FROZEN: + return + if not SessionDB.session_yolo_enabled(session_meta): + return + session_key = self.session_id or "default" + if is_session_yolo_enabled(session_key): + return + enable_session_yolo(session_key) + msg = "⚡ YOLO mode restored from session — all commands auto-approved. /yolo to turn off." + if quiet: + print(msg, file=sys.stderr) + else: + self._console_print(f"[dim]{_escape(msg)}[/dim]") + + def _render_resume_history_panel_lines(self, panel) -> list[str]: + """Render the resume panel at the current terminal width for resize replay.""" + from cli import _suspend_output_history + from io import StringIO + + buf = StringIO() + width = shutil.get_terminal_size((80, 24)).columns + console = Console( + file=buf, + force_terminal=True, + color_system="truecolor", + highlight=False, + width=width, + ) + with _suspend_output_history(): + console.print(panel) + return buf.getvalue().rstrip("\n").splitlines() + + def _resolve_checkpoint_ref(self, ref: str, checkpoints: list) -> str | None: + """Resolve a checkpoint number or hash to a full commit hash.""" + try: + idx = int(ref) - 1 # 1-indexed for user + if 0 <= idx < len(checkpoints): + return checkpoints[idx]["hash"] + else: + print(f" Invalid checkpoint number. Use 1-{len(checkpoints)}.") + return None + except ValueError: + # Treat as a git hash + return ref + + def _show_status(self): + """Show compact startup status line.""" + from cli import get_tool_definitions + # Avoid pulling the full tool registry into the bare Termux prompt path. + if os.environ.get("HERMES_DEFER_AGENT_STARTUP") == "1": + tool_status = "tools deferred" + else: + tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) + tool_count = len(tools) if tools else 0 + tool_status = f"{tool_count} tools" + + # Format model name (shorten if needed) + model_short = self.model.split("/")[-1] if "/" in self.model else self.model + if len(model_short) > 30: + model_short = model_short[:27] + "..." + + # Get API status indicator + api_indicator = "[green bold]●[/]" if self.api_key else "[red bold]●[/]" + + # Build status line with proper markup — skin-aware colors + try: + from hermes_cli.skin_engine import get_active_skin + skin = get_active_skin() + separator_color = skin.get_color("banner_dim", "#B8860B") + accent_color = skin.get_color("ui_accent", "#FFBF00") + label_color = skin.get_color("ui_label", "#DAA520") + except Exception: + separator_color, accent_color, label_color = "#B8860B", "#FFBF00", "cyan" + toolsets_info = "" + if self.enabled_toolsets and "all" not in self.enabled_toolsets: + toolsets_info = f" [dim {separator_color}]·[/] [{label_color}]toolsets: {', '.join(self.enabled_toolsets)}[/]" + + provider_info = f" [dim {separator_color}]·[/] [dim]provider: {self.provider}[/]" + if self._provider_source: + provider_info += f" [dim {separator_color}]·[/] [dim]auth: {self._provider_source}[/]" + + self._console_print( + f" {api_indicator} [{accent_color}]{model_short}[/] " + f"[dim {separator_color}]·[/] [bold {label_color}]{tool_status}[/]" + f"{toolsets_info}{provider_info}" + ) + + def _show_session_status(self): + """Show gateway-style status for the current CLI session.""" + from cli import datetime, display_hermes_home + session_meta = {} + if self._session_db: + try: + session_meta = self._session_db.get_session(self.session_id) or {} + except Exception: + session_meta = {} + + title = (session_meta.get("title") or "").strip() + + created_at = self.session_start + started_at = session_meta.get("started_at") + if started_at: + try: + created_at = datetime.fromtimestamp(float(started_at)) + except Exception: + created_at = self.session_start + + updated_at = created_at + for field in ("updated_at", "last_updated_at", "last_activity_at"): + value = session_meta.get(field) + if not value: + continue + try: + updated_at = datetime.fromtimestamp(float(value)) + break + except Exception: + pass + + agent = getattr(self, "agent", None) + total_tokens = getattr(agent, "session_total_tokens", 0) or 0 + provider = getattr(self, "provider", None) or "unknown" + model = getattr(self, "model", None) or "(unknown)" + is_running = bool(getattr(self, "_agent_running", False)) + + # Reasoning level (C-02): resolve the effective effort for display. + reasoning_label = None + try: + rc = getattr(agent, "reasoning_config", None) or getattr(self, "reasoning_config", None) + if isinstance(rc, dict): + if rc.get("enabled") is False: + reasoning_label = "off" + elif rc.get("effort"): + reasoning_label = str(rc.get("effort")) + show_r = getattr(self, "show_reasoning", None) + if reasoning_label: + reasoning_label += f" (display: {'on' if show_r else 'off'})" if show_r is not None else "" + except Exception: + reasoning_label = None + + # Approval mode (C-02). + approval_label = None + try: + from tools.approval import _get_approval_mode, is_approval_bypass_active_for_session + approval_label = _get_approval_mode() + try: + if is_approval_bypass_active_for_session(getattr(self, "session_key", "") or ""): + approval_label += " (YOLO bypass active)" + except Exception: + pass + except Exception: + approval_label = None + + # Context window usage (C-02): reuse the status-bar snapshot which + # already computes tokens / max / percent. + ctx_label = None + try: + snap = self._get_status_bar_snapshot() + ctx_tokens = snap.get("context_tokens") or 0 + ctx_max = snap.get("context_length") + ctx_pct = snap.get("context_percent") + if ctx_max: + left = "" + if isinstance(ctx_pct, (int, float)): + left = f"{max(0, 100 - int(ctx_pct))}% left · " + ctx_label = f"{left}{ctx_tokens:,} / {ctx_max:,} tokens used" + except Exception: + ctx_label = None + + lines = [ + "Hermes CLI Status", + "", + f"Session ID: {self.session_id}", + f"Path: {display_hermes_home()}", + ] + if title: + lines.append(f"Title: {title}") + lines.append(f"Model: {model} ({provider})") + if reasoning_label: + lines.append(f"Reasoning: {reasoning_label}") + if approval_label: + lines.append(f"Approvals: {approval_label}") + if ctx_label: + lines.append(f"Context: {ctx_label}") + lines.extend([ + f"Created: {created_at.strftime('%Y-%m-%d %H:%M')}", + f"Last Activity: {updated_at.strftime('%Y-%m-%d %H:%M')}", + f"Tokens: {total_tokens:,}", + f"Agent Running: {'Yes' if is_running else 'No'}", + ]) + self._console_print("\n".join(lines), highlight=False, markup=False) + + def _list_recent_sessions(self, limit: int = 10) -> list[dict[str, Any]]: + """Return recent CLI sessions for in-chat browsing/resume affordances.""" + if not self._session_db: + return [] + try: + from hermes_cli.session_listing import query_session_listing + + return query_session_listing( + self._session_db, + source="cli", + current_session_id=self.session_id, + include_all_sources=False, + include_unnamed=True, + limit=limit, + exclude_sources=["kanban", "tool"], + ) + except Exception: + return [] + + def _show_recent_sessions(self, *, reason: str = "history", limit: int = 10) -> bool: + """Render recent sessions inline from the active chat TUI. + + Returns True when something was shown, False if no session list was available. + """ + from cli import _cli_visible_print + sessions = self._list_recent_sessions(limit=limit) + if not sessions: + return False + + from hermes_cli.main import _relative_time + + _cli_visible_print() + if reason == "history": + _cli_visible_print("(._.) No messages in the current chat yet — here are recent sessions you can resume:") + else: + _cli_visible_print(" Recent sessions:") + _cli_visible_print() + _cli_visible_print(f" {'#':<3} {'Title':<32} {'Preview':<40} {'Last Active':<13} {'ID'}") + _cli_visible_print(f" {'─' * 3} {'─' * 32} {'─' * 40} {'─' * 13} {'─' * 24}") + for idx, session in enumerate(sessions, start=1): + title = session.get("title") or "—" + preview = (session.get("preview") or "")[:38] + last_active = _relative_time(session.get("last_active")) + _cli_visible_print(f" {idx:<3} {title:<32} {preview:<40} {last_active:<13} {session['id']}") + _cli_visible_print() + _cli_visible_print(" Use /resume , /resume , or /resume to continue.") + _cli_visible_print(" Example: /resume 2") + _cli_visible_print() + return True + + def show_history(self): + """Display conversation history.""" + from cli import _cli_visible_print + if not self.conversation_history: + if not self._show_recent_sessions(reason="history"): + _cli_visible_print("(._.) No conversation history yet.") + return + + preview_limit = 400 + visible_index = 0 + hidden_tool_messages = 0 + show_ts = bool(getattr(self, "show_timestamps", False)) + + def _ts_suffix(message: dict) -> str: + # Messages restored from SessionDB carry a unix `timestamp`; live + # unsaved turns may not. Only annotate when both the toggle is on + # and the turn actually has a stored time — never fabricate one. + if not show_ts: + return "" + ts = message.get("timestamp") + if not ts: + return "" + try: + from datetime import datetime + return f" [{datetime.fromtimestamp(float(ts)).strftime(getattr(self, 'timestamp_format', '%H:%M'))}]" + except (ValueError, OSError, TypeError): + return "" + + def flush_tool_summary(): + nonlocal hidden_tool_messages + if not hidden_tool_messages: + return + + noun = "message" if hidden_tool_messages == 1 else "messages" + _cli_visible_print("\n [Tools]") + _cli_visible_print(f" ({hidden_tool_messages} tool {noun} hidden)") + hidden_tool_messages = 0 + + _cli_visible_print() + _cli_visible_print("+" + "-" * 50 + "+") + _cli_visible_print("|" + " " * 12 + "(^_^) Conversation History" + " " * 11 + "|") + _cli_visible_print("+" + "-" * 50 + "+") + + for msg in self.conversation_history: + role = msg.get("role", "unknown") + + if role == "tool": + hidden_tool_messages += 1 + continue + + if role not in {"user", "assistant"}: + continue + + flush_tool_summary() + visible_index += 1 + + content = msg.get("content") + content_text = "" if content is None else str(content) + + if role == "user": + _cli_visible_print(f"\n [You #{visible_index}]{_ts_suffix(msg)}") + _cli_visible_print( + f" {content_text[:preview_limit]}{'...' if len(content_text) > preview_limit else ''}" + ) + continue + + _cli_visible_print(f"\n [Hermes #{visible_index}]{_ts_suffix(msg)}") + tool_calls = msg.get("tool_calls") or [] + if content_text: + preview = content_text[:preview_limit] + suffix = "..." if len(content_text) > preview_limit else "" + elif tool_calls: + tool_count = len(tool_calls) + noun = "call" if tool_count == 1 else "calls" + preview = f"(requested {tool_count} tool {noun})" + suffix = "" + else: + preview = "(no text response)" + suffix = "" + _cli_visible_print(f" {preview}{suffix}") + + flush_tool_summary() + _cli_visible_print() + + def _notify_session_boundary(self, event_type: str) -> None: + """Fire a session-boundary plugin hook (on_session_finalize or on_session_reset). + + Non-blocking — errors are caught and logged. Safe to call from any + lifecycle point (shutdown, /new, /reset). + """ + try: + from hermes_cli.lifecycle import finalize_session, invoke_hook + + context = { + "session_id": self.agent.session_id if self.agent else None, + "platform": getattr(self, "platform", None) or "cli", + "reason": ( + "new_session" + if event_type == "on_session_reset" + else "session_boundary" + ), + } + if event_type == "on_session_finalize": + finalize_session(**context) + else: + invoke_hook(event_type, **context) + except Exception: + pass + + def _discard_session_if_empty(self, session_id: Optional[str]) -> bool: + """Drop a just-ended session row when it never gained content. + + Starting the CLI and immediately quitting (or rotating with /new, + /clear) used to leave an empty untitled row behind that clutters + ``/resume`` and ``hermes sessions list``. Delegates the + check-and-delete to ``SessionDB.delete_session_if_empty``, which + only removes rows with no messages, no title, and no child + sessions. Ported from google-gemini/gemini-cli#27770. + """ + from cli import logger + if not self._session_db or not session_id: + return False + # In-memory transcript is authoritative: if this CLI object holds + # conversation messages (flushed to the DB or not), the session is + # not empty. Protects against pruning a real conversation whose DB + # flush failed or hasn't happened yet. + if getattr(self, "conversation_history", None): + return False + try: + from hermes_constants import get_hermes_home as _ghh + return self._session_db.delete_session_if_empty( + session_id, sessions_dir=_ghh() / "sessions" + ) + except Exception: + logger.debug( + "Could not prune empty session %s", session_id, exc_info=True + ) + return False + + def _launch_session_boundary_memory_flush( + self, + history_snapshot: list, + *, + session_id: Optional[str] = None, + ) -> Optional[list]: + """Stage old-session memory extraction so /new stays responsive. + + The context-engine ``on_session_end`` boundary is delivered + synchronously here: it is cheap (local state clear, no LLM call) and + ordering-sensitive — it must land before ``reset_session_state()`` + rebinds the engine to the new session. + + The memory-provider half (LLM-bound extraction, seconds) is NOT run + here. The returned snapshot is handed by ``new_session()`` to + ``MemoryManager.commit_session_boundary_async`` as a single + end→switch task on the manager's serialized background worker, so + extraction can never race the provider rebinding (providers key off + internal ``_session_id`` state — a late ``on_session_end`` after + ``on_session_switch`` would misattribute the old transcript to the + new session). + + Returns the history snapshot to queue, or ``None`` when there is + nothing to extract (no agent / empty history / no memory manager). + """ + from cli import logger + agent = getattr(self, "agent", None) + if not agent or not history_snapshot: + return None + + engine = getattr(agent, "context_compressor", None) + if engine is not None and hasattr(engine, "on_session_end"): + try: + engine.on_session_end(session_id or "", history_snapshot) + except Exception: + logger.debug( + "Context engine on_session_end failed at /new boundary", + exc_info=True, + ) + + # No provider extraction to queue when no memory manager is + # configured — new_session() falls back to the inline switch path. + if getattr(agent, "_memory_manager", None) is None: + return None + return history_snapshot + + def new_session(self, silent=False, title=None): + """Start a fresh session with a new session ID and cleared agent state.""" + from cli import ( + CLI_CONFIG, + _cprint, + _parse_reasoning_config, + _parse_service_tier_config, + _split_model_config_default, + _sync_process_session_id, + datetime, + logger, + ) + old_session_id = self.session_id + _boundary_snapshot = None + if self.agent and self.conversation_history: + # Deliver the context-engine boundary synchronously and get back + # the history snapshot for the deferred provider extraction — + # queued below (after rotation) so /new never blocks on the + # LLM-bound extraction call. + _boundary_snapshot = self._launch_session_boundary_memory_flush( + list(self.conversation_history), + session_id=old_session_id, + ) + self._notify_session_boundary("on_session_finalize") + elif self.agent: + # First session or empty history — still finalize the old session + self._notify_session_boundary("on_session_finalize") + + if self._session_db and old_session_id: + # Flush any un-persisted messages from the current turn to the + # old session *before* rotating. /new can be called mid-turn + # when _flush_messages_to_session_db() has not yet run — without + # this, messages generated during the current turn are silently + # lost on session rotation (#47202). + if self.agent: + try: + self.agent._flush_messages_to_session_db( + self.conversation_history, + conversation_history=self.conversation_history, + ) + except Exception: + pass # best-effort + try: + self._session_db.end_session(old_session_id, "new_session") + except Exception: + pass + # Don't let immediately-rotated empty sessions pile up in + # /resume and `hermes sessions list` (gemini-cli#27770 port). + self._discard_session_if_empty(old_session_id) + + self.session_start = datetime.now() + timestamp_str = self.session_start.strftime("%Y%m%d_%H%M%S") + short_uuid = uuid.uuid4().hex[:6] + self.session_id = f"{timestamp_str}_{short_uuid}" + getattr(self, "_write_terminal_breadcrumb", lambda: None)() + self.conversation_history = [] + self._pending_title = None + self._resumed = False + # /new clears the -m / --model override flag: an explicit CLI model + # was for the previous session only, not for every session spawned + # afterwards. + self._explicit_model_override = False + self.reasoning_config = _parse_reasoning_config( + CLI_CONFIG["agent"].get("reasoning_effort", "") + ) + # /new is a full conversation boundary: session-scoped runtime + # overrides (/model --session, /fast, one-turn restores) do not carry + # forward. Re-derive model/provider and service tier from config.yaml + # so a session-only switch never leaks into the next session (#48055, + # #23131). + self._pending_one_turn_model_restore = None + self.service_tier = _parse_service_tier_config( + CLI_CONFIG["agent"].get("service_tier", "") + ) + _model_config = CLI_CONFIG.get("model", {}) + _raw_default2 = (_model_config.get("default") or _model_config.get("model") or "") if isinstance(_model_config, dict) else (_model_config or "") + _config_model, _ = _split_model_config_default(_raw_default2) + if _config_model and _config_model != getattr(self, "model", None): + _config_provider = ( + _model_config.get("provider", "") + if isinstance(_model_config, dict) + else "" + ) + try: + from hermes_cli.model_switch import switch_model as _switch_model + + _reset_result = _switch_model( + raw_input=_config_model, + current_provider=self.provider or "", + current_model=self.model or "", + current_base_url=self.base_url or "", + current_api_key=self.api_key or "", + is_global=False, + explicit_provider=_config_provider or "", + ) + if _reset_result.success: + if self.agent: + self.agent.switch_model( + new_model=_reset_result.new_model, + new_provider=_reset_result.target_provider, + api_key=_reset_result.api_key, + base_url=_reset_result.base_url, + api_mode=_reset_result.api_mode, + capabilities=getattr( + _reset_result, "runtime_capabilities", None + ), + ) + self.model = _reset_result.new_model + self.provider = _reset_result.target_provider + self.requested_provider = _reset_result.target_provider + self._explicit_api_key = _reset_result.api_key + self._explicit_base_url = _reset_result.base_url + if _reset_result.api_key: + self.api_key = _reset_result.api_key + if _reset_result.base_url: + self.base_url = _reset_result.base_url + if _reset_result.api_mode: + self.api_mode = _reset_result.api_mode + if not silent: + _cprint( + f" (model reset to config default: " + f"{_reset_result.new_model})" + ) + except Exception: + # Best-effort: an unreachable config default must never block + # /new. The session keeps the current working model. + logger.debug("/new model reset to config default failed", exc_info=True) + _sync_process_session_id(self.session_id) + + if self.agent: + self.agent.session_id = self.session_id + self.agent.session_start = self.session_start + self.agent.reasoning_config = self.reasoning_config + self.agent.reset_session_state() + if hasattr(self.agent, "_last_flushed_db_idx"): + self.agent._last_flushed_db_idx = 0 + if hasattr(self.agent, "_todo_store"): + try: + from tools.todo_tool import TodoStore + self.agent._todo_store = TodoStore() + except Exception: + pass + if hasattr(self.agent, "_invalidate_system_prompt"): + self.agent._invalidate_system_prompt() + + if self._session_db: + try: + self.agent._session_db_created = False + self._session_db.create_session( + session_id=self.session_id, + source=os.environ.get("HERMES_SESSION_SOURCE", "cli"), + model=self.model, + model_config={ + "max_iterations": self.max_turns, + "reasoning_config": self.reasoning_config, + }, + ) + self.agent._session_db_created = True + except Exception: + pass + if title and self._session_db: + from hermes_state import SessionDB + try: + sanitized = SessionDB.sanitize_title(title) + except ValueError as e: + _cprint(f" Title rejected: {e}") + sanitized = None + title = None + if sanitized: + try: + self._session_db.set_session_title(self.session_id, sanitized) + self._pending_title = None + self._status_bar_title_checked_at = 0.0 + title = sanitized + except ValueError as e: + _cprint(f" {e} — session started untitled.") + title = None + except Exception: + title = None + elif title is not None: + # sanitize_title returned empty (whitespace-only / unprintable) + _cprint(" Title is empty after cleanup — session started untitled.") + title = None + # Notify memory providers that session_id rotated to a fresh + # conversation. reset=True signals providers to flush accumulated + # per-session state (_session_turns, _turn_counter, _document_id). + # Fires BEFORE the plugin on_session_reset hook (shell hooks only + # see the new id; Python providers see the transition). See #6672. + # + # When the old session has history, end-of-session extraction + # (LLM-bound, seconds) and this switch are queued as ONE task on + # the memory manager's serialized worker — end strictly before + # switch, without blocking /new (#16454). With no history there + # is nothing to extract; switch inline as before. + try: + _mm = getattr(self.agent, "_memory_manager", None) + if _mm is not None: + if _boundary_snapshot: + _mm.commit_session_boundary_async( + _boundary_snapshot, + new_session_id=self.session_id, + parent_session_id=old_session_id or "", + reason="new_session", + ) + else: + _mm.on_session_switch( + self.session_id, + parent_session_id=old_session_id or "", + reset=True, + reason="new_session", + ) + except Exception: + pass + self._notify_session_boundary("on_session_reset") + + if not silent: + if title: + print(f"(^_^)v New session started: {title}") + else: + print("(^_^)v New session started!") + + def _consume_pending_resume_selection(self, text: str) -> bool: + """Resolve a bare numeric reply that follows a bare ``/resume`` prompt. + + After ``/resume`` (no args) prints the recent-sessions list it arms + ``self._pending_resume_sessions``. The next submitted input is given + one chance to be a bare session number (``3``); if so we resume that + session here. Anything else (another command, free text, blank) simply + disarms the prompt and is handled normally by the caller. + + Returns True if the input was consumed as a resume selection (caller + must not treat it as chat); False otherwise. The pending state is + always one-shot: it is cleared on the first submitted input regardless + of outcome. See #34584. + """ + from cli import _cprint + pending = self._pending_resume_sessions + if not pending: + return False + # One-shot: disarm now so a non-matching input can't leave the prompt + # armed and hijack a later number the user meant as chat. + self._pending_resume_sessions = None + + if not isinstance(text, str): + return False + stripped = text.strip() + # Only a pure number selects; let "/resume 3", titles, or any other + # text fall through to normal handling. + if not stripped.isdigit(): + return False + + index = int(stripped) + if index < 1 or index > len(pending): + _cprint(f" Resume index {index} is out of range.") + _cprint(" Use /resume with no arguments to see available sessions.") + return True + + self._handle_resume_command(f"/resume {index}") + return True + + def save_conversation(self, cmd: str = "/save"): + """Handle /save — export the current session to json, md, or html. + + Usage: ``/save [json|md|html] [filename] [redact]`` + + The snapshot is a convenience export for sharing or off-line + inspection; every message is already persisted incrementally to the + SQLite session DB, so the live session remains resumable via + ``hermes --resume `` regardless of whether the user ever runs + ``/save``. ``redact`` runs the export through the force-mode secret + redaction pass before writing. + """ + from cli import datetime + from hermes_cli.session_export import ( + SAVE_USAGE, + normalize_save_format, + render_session_for_save, + ) + + parts = cmd.split()[1:] + if not parts: + print(SAVE_USAGE) + return + redact = False + if parts[-1].lower() in ("redact", "--redact"): + redact = True + parts = parts[:-1] + if not parts: + print(SAVE_USAGE) + return + + try: + fmt = normalize_save_format(parts[0]) + except ValueError as e: + print(f"(._.) {e}") + print(SAVE_USAGE) + return + filename = parts[1] if len(parts) > 1 else None + + # Prefer the durable DB row (has metadata + tool calls); fall back to + # the in-memory history for sessions that never touched the DB. + # getattr: test doubles (SimpleNamespace / object.__new__) may not + # carry _session_db or session_id. + session_data = None + _db = getattr(self, "_session_db", None) + _sid = getattr(self, "session_id", None) + if _db and _sid: + try: + session_data = _db.export_session(_sid) + except Exception: + session_data = None + if not session_data: + if not self.conversation_history: + print("(;_;) No conversation to save.") + return + session_data = { + "id": self.session_id, + "model": self.model, + "started_at": self.session_start.timestamp(), + "messages": self.conversation_history, + } + + if redact: + from hermes_cli.session_export_md import redact_session_data + + session_data = redact_session_data(session_data) + + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + saved_dir = get_hermes_home() / "sessions" / "saved" + try: + saved_dir.mkdir(parents=True, exist_ok=True) + except Exception as e: + print(f"(x_x) Failed to create save directory {saved_dir}: {e}") + return + if filename: + path = Path(filename).expanduser() + if not path.is_absolute(): + path = Path.cwd() / path + else: + path = saved_dir / f"hermes_conversation_{timestamp}.{fmt}" + + try: + content = render_session_for_save(session_data, fmt) + with open(path, "w", encoding="utf-8") as f: + f.write(content) + label = {"json": "JSON", "md": "Markdown", "html": "HTML"}[fmt] + print(f"(^_^)v Conversation saved to: {path} ({label})") + if self.session_id: + print(f" Resume the live session with: hermes --resume {self.session_id}") + except Exception as e: + print(f"(x_x) Failed to save: {e}") + + def _rewind_persisted_user_turn( + self, + *, + warm_history: List[Dict[str, Any]], + user_ordinal: int, + warm_live_view: Dict[str, Any], + ) -> tuple[List[Dict[str, Any]], Dict[str, Any], Dict[str, Any]]: + """Bind one warm user ordinal to a durable row and rewind it atomically.""" + if self._session_db is None or not self.session_id: + raise RuntimeError("session database is unavailable") + + from agent.context_compressor import ( + history_before_user_originated_turn, + split_user_originated_turn, + user_originated_turn_view, + ) + from agent.memory_manager import sanitize_context + from agent.tool_dispatch_helpers import ( + _is_multimodal_tool_result, + _multimodal_text_summary, + ) + from run_agent import _is_ephemeral_scaffolding + + def _persistence_content(content: Any) -> Any: + """Project warm content exactly as the session DB flush does.""" + if _is_multimodal_tool_result(content): + return _multimodal_text_summary(content) + if isinstance(content, list): + text_parts = [] + for part in content: + if isinstance(part, dict) and part.get("type") == "text": + text_parts.append(str(part.get("text", ""))) + elif isinstance(part, dict) and part.get("type") in { + "image", + "image_url", + "input_image", + }: + text_parts.append("[screenshot]") + return "\n".join(text_parts) if text_parts else None + return content + + def _comparison_content(message: Dict[str, Any]) -> Any: + content = _persistence_content(message.get("content")) + if message.get("role") in {"user", "assistant"} and isinstance( + content, str + ): + return sanitize_context(content).strip() + return content + + expected_active_ids = self._session_db.get_active_message_ids( + self.session_id + ) + durable = self._session_db.get_messages_as_conversation( + self.session_id, + include_row_ids=True, + ) + warm_persistence_history = [ + message + for message in warm_history + if not _is_ephemeral_scaffolding(message) + ] + warm_user_indices = [ + index + for index, message in enumerate(warm_persistence_history) + if user_originated_turn_view(message) is not None + ] + durable_user_indices = [ + index + for index, message in enumerate(durable) + if user_originated_turn_view(message) is not None + ] + if len(durable_user_indices) != len(warm_user_indices): + raise RuntimeError( + "session history changed before the rewind could be persisted" + ) + if user_ordinal < 0 or user_ordinal >= len(durable_user_indices): + raise RuntimeError("persisted rewind target is no longer available") + + warm_prefix, _ = history_before_user_originated_turn( + warm_persistence_history, warm_user_indices[user_ordinal] + ) + durable_target_index = durable_user_indices[user_ordinal] + durable_target = durable[durable_target_index] + durable_prefix, durable_live_view = history_before_user_originated_turn( + durable, durable_target_index + ) + if _comparison_content(durable_live_view) != _comparison_content( + warm_live_view + ): + raise RuntimeError( + "session history changed before the rewind could be persisted" + ) + target_row_id = durable_target.get("_row_id") + if not isinstance(target_row_id, int): + raise RuntimeError("persisted rewind target has no row identity") + scaffold, _ = split_user_originated_turn(durable_target) + result = self._session_db.rewind_to_message( + self.session_id, + target_row_id, + preserve_compaction_handoff=scaffold is not None, + expected_active_ids=expected_active_ids, + expected_target_content=durable_live_view.get("content"), + ) + if scaffold is not None: + replacement_id = result.get("replacement_message_id") + if not isinstance(replacement_id, int) or not durable_prefix: + raise RuntimeError("rewind did not retain its compaction handoff") + durable_prefix[-1]["_row_id"] = replacement_id + durable_prefix[-1]["_db_persisted"] = True + warm_prefix[-1] = durable_prefix[-1] + return warm_prefix, durable_live_view, result + + def retry_last(self): + """Retry the last user message by removing the last exchange and re-sending. + + Removes the last assistant response (and any tool-call messages) and + the last user message, then re-sends that user message to the agent. + Returns the message to re-send, or None if there's nothing to retry. + """ + if not self.conversation_history: + print("(._.) No messages to retry.") + return None + + # Walk backwards to the last *real* user message. Timeline bookkeeping + # rows (display_kind set) are role=user but are not user turns — match + # CLI resume counting and user_originated_turn_view. Compaction + # handoffs are excluded too (durable role=user, sometimes without + # display_kind on legacy sessions; #80622). + from agent.context_compressor import ( + history_before_user_originated_turn, + retryable_user_text, + user_originated_turn_view, + ) + from agent.memory_manager import sanitize_context + from run_agent import _is_ephemeral_scaffolding + + warm_history = list(self.conversation_history) + + user_indices = [ + index + for index, message in enumerate(warm_history) + if not _is_ephemeral_scaffolding(message) + and user_originated_turn_view(message) is not None + ] + + if not user_indices: + print("(._.) No user message found to retry.") + return None + last_user_idx = user_indices[-1] + + # Resolve a lossless live payload before touching either persistence or + # memory. A force-user-leading compaction row is one physical carrier: + # its historical handoff remains in the prefix while only the embedded + # human ask is retried. Media cannot be replayed by /retry, so fail + # closed before archiving anything. + try: + truncated, live_view = history_before_user_originated_turn( + warm_history, last_user_idx + ) + live_content = live_view.get("content") + if isinstance(live_content, str): + live_content = sanitize_context(live_content).strip() + last_message = retryable_user_text(live_content) + except ValueError as exc: + print(f"(._.) Cannot retry that message safely: {exc}") + return None + + # Persist the rewind before publishing the shorter in-memory view. + # The DB owns the physical carrier split so the archived original and + # retained scaffold are committed atomically. A plain user row keeps + # the legacy rewind shape (no replacement scaffold). + if self._session_db is not None and self.session_id: + try: + truncated, _, _ = self._rewind_persisted_user_turn( + warm_history=warm_history, + user_ordinal=len(user_indices) - 1, + warm_live_view=live_view, + ) + except Exception as exc: + print(f"(x_x) Retry rewind failed; history was not changed: {exc}") + return None + + self.conversation_history = truncated + if self.agent is not None: + if hasattr(self.agent, "_session_messages"): + self.agent._session_messages = self.conversation_history + if hasattr(self.agent, "_last_flushed_db_idx"): + self.agent._last_flushed_db_idx = len(self.conversation_history) + if hasattr(self.agent, "_db_flush_scan_prefix"): + self.agent._db_flush_scan_prefix = self.conversation_history[:] + + print(f"(^_^)b Retrying: \"{last_message[:60]}{'...' if len(last_message) > 60 else ''}\"") + return last_message + + def undo_last(self, n: int = 1, prefill: bool = True): + """Back up N user turns: truncate history, soft-delete on disk, prefill. + + Walks backwards N user messages and discards everything from the + Nth-from-last user message onward (its assistant response, tool + calls, etc.). ``n`` defaults to 1 (the last exchange); ``/undo 3`` + backs up three user turns. If ``n`` exceeds the number of user + turns, it backs up to the oldest one. + + Beyond the in-memory ``conversation_history`` slice, this also: + • soft-deletes the truncated rows in SessionDB (``active=0``) so + they're hidden from re-prompts and search but kept for audit; + • notifies memory providers via ``on_session_switch(rewound=True)``; + • mirrors /branch's agent surgery (system-prompt invalidation + + flush-index reset); + • when ``prefill`` is set and an input buffer is available, + pre-fills the composer with the backed-up message text so it + can be edited and resubmitted. + + ``prefill=False`` is used by callers that drive the undo + programmatically (e.g. checkpoint rollback) and don't want to + touch the user's input buffer. + """ + from cli import logger + if not self.conversation_history: + print("(._.) No messages to undo.") + return + + if n < 1: + n = 1 + + # Walk backwards collecting the indices of the last N *real* user + # messages (exclude display_kind timeline rows and compaction + # handoffs — same predicate as user_originated_turn_view, resume + # turn counting, and /retry; #80622). + from agent.context_compressor import ( + history_before_user_originated_turn, + user_originated_turn_view, + ) + from run_agent import _is_ephemeral_scaffolding + + warm_history = list(self.conversation_history) + + user_indices = [ + index + for index, message in enumerate(warm_history) + if not _is_ephemeral_scaffolding(message) + and user_originated_turn_view(message) is not None + ] + + if not user_indices: + print("(._.) No user message found to undo.") + return + + turns_undone = min(n, len(user_indices)) + target_ordinal = len(user_indices) - turns_undone + cut_idx = user_indices[target_ordinal] + + removed_count = len(warm_history) - cut_idx + truncated, live_view = history_before_user_originated_turn( + warm_history, cut_idx + ) + removed_text = self._undo_content_to_text(live_view.get("content")) + + # Soft-delete the truncated rows on disk so re-prompts and search + # see the clean transcript while the rows survive for audit. + rewound_rows = 0 + if self._session_db is not None and self.session_id: + try: + truncated, durable_live_view, result = ( + self._rewind_persisted_user_turn( + warm_history=warm_history, + user_ordinal=target_ordinal, + warm_live_view=live_view, + ) + ) + # Canonicalize the editable prefill before mutation. The raw + # physical carrier contains the reference summary wrapper. + durable_text = self._undo_content_to_text( + durable_live_view.get("content") + ) + if durable_text: + removed_text = durable_text + rewound_rows = result.get("rewound_count", 0) + except Exception as e: + logger.debug("undo: durable rewind failed: %s", e) + print(f"(x_x) Undo failed; history was not changed: {e}") + return + + # Publish only after the durable rewind succeeds (or no store exists). + self.conversation_history = truncated + + # Agent surgery: invalidate the system-prompt cache and reset the + # flush index so the next turn re-flushes from the truncated head. + if self.agent is not None: + if hasattr(self.agent, "_invalidate_system_prompt"): + try: + self.agent._invalidate_system_prompt() + except Exception: + pass + if hasattr(self.agent, "_last_flushed_db_idx"): + try: + self.agent._last_flushed_db_idx = len(self.conversation_history) + except Exception: + pass + if hasattr(self.agent, "_session_messages"): + self.agent._session_messages = self.conversation_history + if hasattr(self.agent, "_db_flush_scan_prefix"): + self.agent._db_flush_scan_prefix = self.conversation_history[:] + # Notify memory providers — same hook /branch fires, with the + # rewound flag so per-turn document caches invalidate (#6672, #21910). + try: + _mm = getattr(self.agent, "_memory_manager", None) + if _mm is not None and self.session_id: + _mm.on_session_switch( + self.session_id, + parent_session_id="", + reset=False, + rewound=True, + ) + except Exception: + pass + + turn_word = "turn" if turns_undone == 1 else "turns" + msg_count = rewound_rows or removed_count + print( + f"(^_^)b Undid {turns_undone} {turn_word} ({msg_count} message(s)). " + f"Backed up to: \"{removed_text[:60]}{'...' if len(removed_text) > 60 else ''}\"" + ) + remaining = len(self.conversation_history) + print(f" {remaining} message(s) remaining in history.") + + # Pre-fill the composer with the backed-up message so the user can + # edit and resubmit (Claude-Code-style). Editable, not auto-sent. + if prefill and removed_text: + self._prefill_input_buffer(removed_text) + + @staticmethod + def _undo_content_to_text(content) -> str: + """Flatten message content (str or content-part list) to plain text.""" + if isinstance(content, str): + return content + if isinstance(content, list): + parts = [ + p.get("text", "") + for p in content + if isinstance(p, dict) and p.get("type") == "text" + ] + return "\n".join(t for t in parts if t) + return "" + + def _write_terminal_breadcrumb(self) -> None: + """Record this terminal's live session for bare ``hermes -c``. + + Called at session start and whenever ``self.session_id`` is + reassigned mid-run (/new, /branch, auto-compression rotation) so a + later bare ``-c`` in THIS terminal resumes THIS conversation's live + tip. Best-effort — never raises, no-op without a terminal identity + or when session.terminal_continue is false. + """ + try: + from hermes_cli.terminal_breadcrumbs import write_breadcrumb + + write_breadcrumb(self.session_id) + except Exception: + pass + + def _transfer_session_yolo(self, old_session_id: str, new_session_id: str) -> None: + """Move YOLO bypass state from an old session key to a new one. + + Called whenever ``self.session_id`` is reassigned mid-run — ``/branch`` + forks into a new session, and auto-compression rotates the agent's + session id into a fresh continuation session. Without this transfer + the user's ``/yolo ON`` toggle would silently revert on the very next + turn (the same UX failure mode that motivated this entire fix), since + ``_session_yolo`` is keyed by session id. + + Mirrors ``tui_gateway/server.py`` (~line 1297-1305) which performs the + same transfer for the TUI's session-rename path. No-op when YOLO + wasn't enabled or when the ids match. + """ + if not old_session_id or not new_session_id or old_session_id == new_session_id: + return + try: + from tools.approval import ( + disable_session_yolo, + enable_session_yolo, + is_session_yolo_enabled, + ) + except Exception: + return + if is_session_yolo_enabled(old_session_id): + enable_session_yolo(new_session_id) + disable_session_yolo(old_session_id) + # Carry the persisted flag onto the continuation row so a later + # `hermes --resume ` restores the bypass too. getattr + # guard: tests call this unbound against a minimal stand-in. + _persist = getattr(self, "_persist_session_yolo", None) + if _persist: + _persist(new_session_id, True) + + def _is_session_yolo_active(self) -> bool: + """Whether YOLO bypass is currently enabled for this CLI session. + + Reads from ``tools.approval._session_yolo`` (the same set that + ``enable_session_yolo`` / ``disable_session_yolo`` write to) so the + status bar reflects the actual bypass state instead of a stale env + var. Also honors the process-start ``--yolo`` flag, which freezes + ``HERMES_YOLO_MODE`` into ``_YOLO_MODE_FROZEN`` before tool imports + happen. + """ + try: + from tools.approval import ( + _YOLO_MODE_FROZEN, + is_session_yolo_enabled, + ) + except Exception: + return False + if _YOLO_MODE_FROZEN: + return True + # Use ``getattr`` so test fixtures that build a CLI via ``__new__`` + # (skipping ``__init__``) don't trip an AttributeError here; the + # status-bar builders swallow exceptions silently but lose every + # field after the failure. + session_key = getattr(self, "session_id", None) or "default" + return is_session_yolo_enabled(session_key) + + def _toggle_yolo(self): + """Toggle YOLO mode — skip all dangerous command approval prompts. + + Per-session toggle that mirrors the gateway and TUI ``/yolo`` handlers + (see ``gateway/run.py:_handle_yolo_command`` and + ``tui_gateway/server.py`` key=="yolo"). We deliberately do NOT mutate + ``HERMES_YOLO_MODE`` here — that env var is read once at module import + time into ``tools.approval._YOLO_MODE_FROZEN`` to keep prompt-injected + skills from flipping the bypass mid-session, so setting it after CLI + startup is a silent no-op. Routing through ``enable_session_yolo`` / + ``disable_session_yolo`` gives the same auditable, per-session bypass + the other surfaces have. ``run_conversation`` binds + ``self.session_id`` as the active approval session key via + ``set_current_session_key`` so the bypass takes effect on the very + next dangerous command in this run. + """ + from cli import _cprint + from hermes_cli.colors import Colors as _Colors + from tools.approval import ( + _YOLO_MODE_FROZEN, + disable_session_yolo, + enable_session_yolo, + is_session_yolo_enabled, + ) + + # Process-level YOLO (--yolo flag / HERMES_YOLO_MODE at startup) is + # frozen into tools.approval at import time and cannot be disabled by + # the session toggle. Before this guard, /yolo printed "YOLO mode OFF — + # dangerous commands will require approval" while every command kept + # auto-approving (the frozen flag short-circuits the approval gate + # ahead of the session check) — a false safety claim. Say the truth + # instead of toggling a bypass that has no effect. + if _YOLO_MODE_FROZEN: + _cprint( + f" ⚡ YOLO is {_Colors.BOLD}{_Colors.RED}locked ON{_Colors.RESET}" + " for this process (started with --yolo / HERMES_YOLO_MODE)." + " /yolo cannot disable it — restart without the flag to" + " re-enable approvals." + ) + return + + session_key = self.session_id or "default" + # ``getattr`` guard: tests exercise this method unbound against a + # minimal stand-in object (see tests/cli/test_cli_yolo_toggle.py); + # persistence is best-effort either way. + _persist = getattr(self, "_persist_session_yolo", None) + if is_session_yolo_enabled(session_key): + disable_session_yolo(session_key) + if _persist: + _persist(session_key, False) + _cprint( + f" ⚠ YOLO mode {_Colors.BOLD}{_Colors.RED}OFF{_Colors.RESET}" + " — dangerous commands will require approval." + ) + else: + enable_session_yolo(session_key) + if _persist: + _persist(session_key, True) + _cprint( + f" ⚡ YOLO mode {_Colors.BOLD}{_Colors.GREEN}ON{_Colors.RESET}" + " — all commands auto-approved. Use with caution." + ) + + def _persist_session_yolo(self, session_key: str, enabled: bool) -> None: + """Persist the YOLO flag to the session row so --resume restores it. + + Best-effort: the in-memory toggle is authoritative for this process; + persistence only affects a future ``hermes --resume``. Skipped when the + session store is unavailable or the row doesn't exist yet (the row is + created lazily on the first turn — ``_toggle_yolo`` before any chat + writes nothing, and the launch-time ``--yolo`` flag is carried into the + creation-time model_config instead). + """ + db = getattr(self, "_session_db", None) + if db is None or not session_key or session_key == "default": + return + try: + db.set_session_yolo(session_key, enabled) + except Exception: + pass + + def _manual_compress(self, cmd_original: str = ""): + """Manually trigger context compression on the current conversation. + + Two modes: + + * ``/compress []`` — compress the *whole* history. An + optional focus topic guides the summariser to preserve + information related to *focus* while being more aggressive + about discarding everything else. Inspired by Claude Code's + ``/compact `` feature. + * ``/compress here [N]`` — boundary-aware compression. Summarize + everything *except* the most recent ``N`` exchanges (default + 2), which are preserved verbatim. Inspired by Claude Code's + Rewind "Summarize up to here" action (v2.1.139, May 2026, + https://code.claude.com/docs/en/whats-new/2026-w20). Lets the + user pick the compression boundary instead of leaving it to + the automatic token-budget heuristic. + """ + if not self.conversation_history or len(self.conversation_history) < 4: + print("(._.) Not enough conversation to compress (need at least 4 messages).") + return + + if not self.agent: + print("(._.) No active agent -- send a message first.") + return + + # No compression_enabled gate here: the config flag disables + # *automatic* compaction only. Manual /compress is an explicit user + # action — the context-overflow error path (conversation_loop.py) + # directs users here when auto-compaction is off, and the gateway's + # /compress handler has never gated on the flag. + + from hermes_cli.partial_compress import ( + extract_compress_flags, + parse_partial_compress_args, + rejoin_compressed_head_and_tail, + split_history_for_partial_compress, + summarize_compress_preview, + ) + from agent.conversation_compression import ( + finalize_context_engine_compression_notification, + ) + + # Args after the command word (e.g. "/compress here 3" -> "here 3"). + raw_args = "" + if cmd_original: + _parts = cmd_original.strip().split(None, 1) + if len(_parts) > 1: + raw_args = _parts[1].strip() + + # Strip --preview/--dry-run/--aggressive before positional parsing + # so the flags coexist with 'here [N]' / focus-topic forms. + raw_args, preview, aggressive = extract_compress_flags(raw_args) + partial, keep_last, focus_topic = parse_partial_compress_args(raw_args) + focus_topic = focus_topic or "" + + if aggressive: + # LLM-free hard truncation is not supported: it would need its + # own transcript-persistence path outside the guarded + # _compress_context rotation machinery. Surface that instead of + # silently mis-parsing the flag as a focus topic. + print("(._.) --aggressive is not supported; use '/compress here [N]' " + "to keep only recent exchanges, or /undo to drop turns.") + if not preview: + return + + if preview: + from agent.model_metadata import estimate_request_tokens_rough + _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" + _tools = getattr(self.agent, "tools", None) or None + approx_tokens = estimate_request_tokens_rough( + self.conversation_history, + system_prompt=_sys_prompt, + tools=_tools, + ) + report = summarize_compress_preview( + self.conversation_history, + partial, + keep_last, + focus_topic or None, + approx_tokens, + ) + for line in report["lines"]: + print(f"🗜️ {line}") + return + + original_count = len(self.conversation_history) + with self._busy_command("Compressing context...", blocks_input=False): + try: + from agent.model_metadata import estimate_request_tokens_rough + from agent.manual_compression_feedback import summarize_manual_compression + original_history = list(self.conversation_history) + + # Boundary-aware split: only the head is summarized; the + # most recent `keep_last` exchanges ride along verbatim. + tail: list = [] + head = original_history + if partial: + head, tail = split_history_for_partial_compress( + original_history, keep_last + ) + if not tail: + # Split degenerated (everything would be kept, or + # no head left to compress). Fall back to full + # compression so the user still gets an action. + partial = False + head = original_history + + # Include system prompt + tool schemas in the estimate — + # a transcript-only number understates real request pressure + # and can even appear to grow after compression because a + # dense handoff summary replaces many short turns (#6217). + _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" + _tools = getattr(self.agent, "tools", None) or None + approx_tokens = estimate_request_tokens_rough( + original_history, + system_prompt=_sys_prompt, + tools=_tools, + ) + if partial: + print(f"🗜️ Summarizing up to here: compressing {len(head)} of " + f"{original_count} messages (~{approx_tokens:,} tokens), " + f"keeping last {keep_last} exchange(s) verbatim...") + elif focus_topic: + print(f"🗜️ Compressing {original_count} messages (~{approx_tokens:,} tokens), " + f"focus: \"{focus_topic}\"...") + else: + print(f"🗜️ Compressing {original_count} messages (~{approx_tokens:,} tokens)...") + + # Pass None as system_message so _compress_context rebuilds + # the system prompt from scratch via _build_system_prompt(None). + # Passing _cached_system_prompt caused duplication because + # _build_system_prompt appends system_message to prompt_parts + # which already contain the agent identity — resulting in the + # identity block appearing twice (issue #15281). + compressed, _ = self.agent._compress_context( + head, + None, + approx_tokens=approx_tokens, + focus_topic=focus_topic or None, + force=True, + defer_context_engine_notification=True, + ) + + # If _compress_context returned unchanged because a + # concurrent compression lock is held, tell the user + # clearly instead of showing the misleading + # "No changes from compression" no-op text. The wording + # distinguishes a confirmed holder from an unconfirmed + # acquisition failure (describe_compression_lock_skip). + # Type-pinned check (is True / str): the flag's only real + # values are None/True/holder-string, and a bare getattr + # truthiness test is fooled by MagicMock auto-attributes on + # test-double agents (skill pitfall: MagicMock vs hasattr). + _lock_skip_signal = getattr( + self.agent, "_compression_skipped_due_to_lock", None + ) + if _lock_skip_signal is True or isinstance(_lock_skip_signal, str): + from agent.manual_compression_feedback import ( + describe_compression_lock_skip, + ) + print( + " " + + describe_compression_lock_skip( + self.agent._compression_skipped_due_to_lock + ) + ) + self.agent._compression_skipped_due_to_lock = None + # No boundary was committed on a lock-skip; discard the + # deferred context-engine notification (exactly-once). + finalize_context_engine_compression_notification( + self.agent, + committed=False, + ) + return + + if partial and tail: + compressed = rejoin_compressed_head_and_tail(compressed, tail) + self.conversation_history = compressed + # _compress_context ends the old session and creates a new child + # session on the agent (run_agent.py::_compress_context). Sync the + # CLI's session_id so /status, /resume, exit summary, and title + # generation all point at the live continuation session, not the + # ended parent. Without this, subsequent end_session() calls target + # the already-closed parent and the child is orphaned. + if ( + getattr(self.agent, "session_id", None) + and self.agent.session_id != self.session_id + ): + self.session_id = self.agent.session_id + getattr(self, "_write_terminal_breadcrumb", lambda: None)() + self._pending_title = None + # Manual /compress replaces conversation_history with a new + # compressed handoff for the child session. Persist it from + # offset 0 so resume can recover the continuation after exit. + self.agent._flush_messages_to_session_db(self.conversation_history, None) + finalize_context_engine_compression_notification( + self.agent, + committed=True, + ) + new_tokens = estimate_request_tokens_rough( + self.conversation_history, + system_prompt=_sys_prompt, + tools=_tools, + ) + summary = summarize_manual_compression( + original_history, + self.conversation_history, + approx_tokens, + new_tokens, + compression_state=getattr( + self.agent, "context_compressor", None + ), + ) + if ( + summary.get("aborted") + or summary.get("fallback_used") + or summary.get("refused_would_grow") + ): + icon = "⚠️" + else: + icon = "🗜️" if summary["noop"] else "✅" + print(f" {icon} {summary['headline']}") + print(f" {summary['token_line']}") + if summary["note"]: + print(f" {summary['note']}") + + except Exception as e: + finalize_context_engine_compression_notification( + self.agent, + committed=False, + ) + print(f" ❌ Compression failed: {e}") + + def _persist_prompt_summary(self, icon: str, label: str, detail: str, outcome: str) -> None: + """Print a one-line scrollback summary of a resolved modal prompt. + + Modal panels (approval / clarify) live in the prompt_toolkit layout and + vanish on the next repaint, so the question and the decision leave no + trace in the terminal scrollback. When display.persist_prompts is on + (default), emit a dim single line after the prompt resolves so the + decision survives in chat history. + """ + from cli import CLI_CONFIG, _DIM, _RST, _cprint + if not CLI_CONFIG.get("display", {}).get("persist_prompts", True): + return + detail = " ".join(detail.split()) + if len(detail) > 120: + detail = detail[:119] + "…" + outcome = " ".join(outcome.split()) + if len(outcome) > 120: + outcome = outcome[:119] + "…" + _cprint(f"\n{_DIM}{icon} {label}: {detail} → {outcome}{_RST}") + + def _clear_terminal_on_exit(self): + """Clear screen + scrollback so nothing is stranded above the exit summary. + + Called from ``_print_exit_summary`` after ``app.run()`` has returned and + prompt_toolkit has torn down its renderer + restored terminal modes — + so a direct write to the real stdout fd is safe (the StdoutProxy / + patch_stdout layer is gone by now). + + Sequence: ``ESC[3J`` (erase scrollback) + ``ESC[2J`` (erase visible + screen) + ``ESC[H`` (cursor home). Modern terminals on Linux, macOS and + Windows (Terminal / conhost with VT processing, which prompt_toolkit + already enables) all honor these. Best-effort: skip silently when + stdout isn't a real console, and fall back to the platform ``clear`` / + ``cls`` command if the escape write fails. + """ + try: + stream = sys.stdout + if stream is None or not stream.isatty(): + return + except Exception: + return + try: + stream.write("\033[3J\033[2J\033[H") + stream.flush() + return + except Exception: + pass + # Fallback: shell clear command (rarely needed — escapes work on every + # VT-capable terminal, but this covers exotic stdout wrappers). + try: + os.system("cls" if os.name == "nt" else "clear") + except Exception: + pass + + def _persist_active_session_before_close(self): + """Best-effort SQLite/JSON flush before the CLI marks a session closed. + + ``run_conversation()`` normally persists at turn boundaries, but a + terminal close/SIGHUP/SIGTERM can unwind the prompt_toolkit app while + the agent thread still holds the current turn only in memory. Flush the + agent's live ``_session_messages`` before ``end_session()`` so resume, + session_search, and state.db do not lose the interrupted turn. + """ + from cli import logger + agent = getattr(self, "agent", None) + if not agent or not hasattr(agent, "_persist_session"): + return + + persist_lock = getattr(agent, "_session_persist_lock", None) + + def _snapshot_and_persist() -> None: + # This snapshot must share the staging lock with ``chat()``. Without + # it, close can retain a mutable history baseline just before chat + # appends its pending dict; the later flush then mistakes that dict + # for durable history and stamps it without writing a row (#63766). + messages = getattr(agent, "_session_messages", None) + pending_cli_message = getattr(agent, "_pending_cli_user_message", None) + if not isinstance(messages, list): + messages = getattr(self, "conversation_history", None) + if not isinstance(messages, list): + return + if isinstance(pending_cli_message, dict) and not any( + message is pending_cli_message for message in messages + ): + # The UI has accepted a new input but the worker still exposes its + # prior snapshot. Include only that staged dict; the baseline below + # keeps any durable resumed prefix from being re-appended. + messages = [*messages, pending_cli_message] + if not messages: + return + + # A normal turn builds a new list that reuses the resumed-history dicts. + # Keep that CLI history as the baseline so a signal between assigning + # ``_session_messages`` and the turn's DB flush cannot append its durable + # prefix a second time. Once the CLI takes the turn result, however, both + # names can point at the same live list; passing that alias would mark an + # unflushed tail durable without writing it. Marker-only persistence is + # correct only in that alias case. + conversation_history = getattr(self, "conversation_history", None) + pending_cli_message = getattr(agent, "_pending_cli_user_message", None) + if ( + isinstance(conversation_history, list) + and conversation_history + and conversation_history[-1] is pending_cli_message + ): + # The UI accepted this user message before the agent finished its + # early persistence. Its dict can already be in ``messages`` but is + # not durable yet, so exclude it from the resumed-history baseline. + conversation_history = conversation_history[:-1] + elif not isinstance(conversation_history, list) or conversation_history is messages: + conversation_history = None + + # A first-turn close can arrive before the worker builds its cached + # prompt. Build or restore it before the DB row is created so the + # durable transcript never leaves a NULL system_prompt cache entry. + if getattr(agent, "_cached_system_prompt", None) is None: + try: + from agent.conversation_loop import _restore_or_build_system_prompt + + _restore_or_build_system_prompt(agent, None, conversation_history) + except Exception: + logger.debug("Could not build system prompt during CLI close", exc_info=True) + return + if getattr(agent, "_cached_system_prompt", None) is None: + return + + agent._ensure_db_session() + agent._persist_session(messages, conversation_history) + if getattr(agent, "session_id", None): + self.session_id = agent.session_id + getattr(self, "_write_terminal_breadcrumb", lambda: None)() + + try: + if persist_lock is None: + _snapshot_and_persist() + else: + with persist_lock: + _snapshot_and_persist() + except (Exception, KeyboardInterrupt) as e: + logger.debug("Could not persist active CLI session before close: %s", e) + + def _print_exit_summary(self, clear_screen: bool = True): + """Print session resume info on exit, similar to Claude Code. + + Args: + clear_screen: When True (default), clear the terminal screen and + scrollback before printing the summary. This is appropriate for + interactive TUI teardown (#38252). Single-query (-q) mode should + pass False to preserve the printed answer (#53009). + """ + from cli import datetime + if clear_screen: + # Clear the screen + scrollback before printing the summary so the + # live bottom chrome (status bar, input box, separator rules) and the + # rest of the session transcript don't get stranded above the exit + # summary (#38252). By this point app.run() has returned and + # prompt_toolkit has restored terminal modes, so writing raw escapes + # to stdout is safe. ESC[3J clears scrollback, ESC[2J clears the + # visible screen, ESC[H homes the cursor — so the summary prints at a + # clean top-left. Falls back to the platform clear command if stdout + # isn't a TTY-capable stream. Honors NO_COLOR/dumb terminals by + # skipping silently when there's no real console. + self._clear_terminal_on_exit() + print() + msg_count = len(self.conversation_history) + if msg_count > 0: + user_msgs = len([m for m in self.conversation_history if m.get("role") == "user"]) + tool_calls = len([m for m in self.conversation_history if m.get("role") == "tool" or m.get("tool_calls")]) + elapsed = datetime.now() - self.session_start + hours, remainder = divmod(int(elapsed.total_seconds()), 3600) + minutes, seconds = divmod(remainder, 60) + if hours > 0: + duration_str = f"{hours}h {minutes}m {seconds}s" + elif minutes > 0: + duration_str = f"{minutes}m {seconds}s" + else: + duration_str = f"{seconds}s" + + # Look up session title for resume-by-name hint + session_title = None + if self._session_db: + try: + session_title = self._session_db.get_session_title(self.session_id) + except Exception: + pass + + print("Resume this session with:") + # Session IDs are profile-constrained, so the resume hint must + # include `-p ` for non-default profiles. Without this, + # copying the hint from a non-default profile fails to find the + # session on the next invocation. The "default" and "custom" + # profile names use the standard HERMES_HOME, so no -p needed. + try: + from hermes_cli.profiles import get_active_profile_name + _active_profile = get_active_profile_name() + except Exception: + _active_profile = "default" + profile_flag = ( + "" if _active_profile in ("default", "custom") else f" -p {_active_profile}" + ) + print(f" hermes --resume {self.session_id}{profile_flag}") + if session_title: + print(f" hermes -c \"{session_title}\"{profile_flag}") + print() + print(f"Session: {self.session_id}") + if session_title: + print(f"Title: {session_title}") + print(f"Duration: {duration_str}") + print(f"Messages: {msg_count} ({user_msgs} user, {tool_calls} tool calls)") + else: + try: + from hermes_cli.skin_engine import get_active_goodbye + goodbye = get_active_goodbye("Goodbye! ⚕") + except Exception: + goodbye = "Goodbye! ⚕" + print(goodbye) diff --git a/hermes_cli/cli_status_bar_mixin.py b/hermes_cli/cli_status_bar_mixin.py new file mode 100644 index 0000000000..f14124fe6a --- /dev/null +++ b/hermes_cli/cli_status_bar_mixin.py @@ -0,0 +1,1597 @@ +"""Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import errno +import shutil +import threading +import time + +from agent.pet import render as pet_render +from hermes_cli.banner import _format_context_length +from typing import Any, Dict, Optional + + +class CLIStatusBarMixin: + """Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the interactive CLI""" + + def _status_bar_context_style(self, percent_used: Optional[int]) -> str: + if percent_used is None: + return "class:status-bar-dim" + if percent_used >= 95: + return "class:status-bar-critical" + if percent_used > 80: + return "class:status-bar-bad" + if percent_used >= 50: + return "class:status-bar-warn" + return "class:status-bar-good" + + def _cache_hit_rate(self, snapshot: dict, precision: int = 1) -> "tuple[float, str] | None": + """Return (cache_pct, formatted_label) or None if no cache data. + + Centralises the cache-hit-rate computation so both the plain-text + status bar and the prompt-toolkit fragment path share one formula. + Prefers the baseline-delta percentage computed in + ``_get_status_bar_snapshot`` (resets on model switch / compression, + so it reflects the *current* cache regime); falls back to the + session-lifetime ratio when no delta is available. + """ + delta_pct = snapshot.get("cache_hit_pct") + if delta_pct is not None: + return float(delta_pct), f"◎ {float(delta_pct):.{precision}f}%" + cache_read = snapshot.get("session_cache_read_tokens", 0) + prompt_total = snapshot.get("session_prompt_tokens", 0) + if cache_read > 0 and prompt_total > 0: + cache_pct = cache_read / prompt_total * 100 + return cache_pct, f"◎ {cache_pct:.{precision}f}%" + return None + + def _cache_hit_rate_style(self, cache_pct: float) -> str: + """Style for cache hit rate — higher is better (opposite of context %).""" + if cache_pct >= 70: + return "class:status-bar-good" + if cache_pct >= 40: + return "class:status-bar-warn" + return "class:status-bar-bad" + + @staticmethod + def _battery_status_style(category: str) -> str: + """Map a battery colour category to a status-bar style class.""" + return { + "good": "class:status-bar-good", + "warn": "class:status-bar-warn", + "bad": "class:status-bar-bad", + "critical": "class:status-bar-critical", + }.get(category, "class:status-bar-dim") + + def _handle_battery_command(self, cmd_original: str) -> None: + """Toggle the status-bar battery read-out. + + ``/battery`` toggles, ``/battery on|off`` sets explicitly, and + ``/battery status`` reports the current setting plus a live reading. + The choice is persisted to ``display.battery`` so it survives restarts. + """ + from cli import save_config_value + parts = (cmd_original or "").split() + arg = parts[1].strip().lower() if len(parts) > 1 else "" + + try: + from agent.battery import format_battery, read_battery + reading = read_battery(use_cache=False) + except Exception: + reading = None + + if arg in ("status", "show"): + state = "on" if self._battery_visible else "off" + if reading is not None and reading.available: + self._console_print( + f" Battery indicator {state} — currently {format_battery(reading)}" + ) + elif reading is not None: + self._console_print( + f" Battery indicator {state} — no battery detected on this machine" + ) + else: + self._console_print(f" Battery indicator {state}") + return + + if arg in ("on", "true", "yes"): + target = True + elif arg in ("off", "false", "no"): + target = False + elif arg in ("", "toggle"): + target = not self._battery_visible + else: + self._console_print(" Usage: /battery [on|off|status]") + return + + self._battery_visible = target + save_config_value("display.battery", target) + + if target: + if reading is not None and not reading.available: + self._console_print( + " Battery indicator on — no battery detected, so nothing will show here" + ) + elif reading is not None and reading.available: + self._console_print( + f" Battery indicator on — {format_battery(reading)}" + ) + else: + self._console_print(" Battery indicator on") + else: + self._console_print(" Battery indicator off") + + @staticmethod + def _compression_count_style(count: int) -> str: + """Return a style class reflecting context compression pressure.""" + if count >= 10: + return "class:status-bar-bad" + if count >= 5: + return "class:status-bar-warn" + return "class:status-bar-dim" + + def _build_context_bar(self, percent_used: Optional[int], width: int = 10) -> str: + safe_percent = max(0, min(100, percent_used or 0)) + filled = round((safe_percent / 100) * width) + return f"[{('█' * filled) + ('░' * max(0, width - filled))}]" + + @staticmethod + def _format_prompt_elapsed(prompt_start_time: Optional[float], prompt_duration: float, live: bool = False) -> str: + """Format per-prompt elapsed time for the status bar. + + Always returns a string — shows 0s on fresh start before first turn. + Keeps seconds visible at all scales so it increments smoothly: + 59s → 1m → 1m 1s → ... → 1m 59s → 2m → 2m 1s → ... + 59m 59s → 1h → 1h 0m 1s → ... + 23h 59m 59s → 1d → 1d 0h 1m → ... + + Emoji prefix: ⏱ when turn is live, ⏲ when frozen or fresh start. + Uses width-1 (no variation selector) glyphs so the status bar stays + aligned in monospace terminals. + """ + if prompt_start_time is None and prompt_duration == 0.0: + return "⏲ 0s" + elapsed = time.time() - prompt_start_time if prompt_start_time is not None else prompt_duration + elapsed = max(0.0, elapsed) + + days = int(elapsed // 86400) + remaining = elapsed % 86400 + hours = int(remaining // 3600) + remaining = remaining % 3600 + minutes = int(remaining // 60) + seconds = int(remaining % 60) + + if days > 0: + time_str = f"{days}d {hours}h {minutes}m" + elif hours > 0: + time_str = f"{hours}h {minutes}m {seconds}s" if seconds else f"{hours}h {minutes}m" + elif minutes > 0: + time_str = f"{minutes}m {seconds}s" if seconds else f"{minutes}m" + else: + time_str = f"{int(elapsed)}s" + + emoji = "⏱" if live else "⏲" + return f"{emoji} {time_str}" + + @staticmethod + def _format_idle_since(last_finished_at: Optional[float], turn_live: bool) -> str: + """Format time since the last final agent response for the status bar. + + Returns an empty string while a turn is live (the per-prompt elapsed + timer covers that case) or before the first turn has completed. + Compact read-out: ``✓ 42s`` / ``✓ 3m`` / ``✓ 1h 12m``. + """ + from cli import format_duration_compact + if turn_live or last_finished_at is None: + return "" + idle = max(0.0, time.time() - last_finished_at) + return f"✓ {format_duration_compact(idle)}" + + def _get_status_bar_snapshot(self) -> Dict[str, Any]: + # Prefer the agent's model name — it updates on fallback. + # self.model reflects the originally configured model and never + # changes mid-session, so the TUI would show a stale name after + # _try_activate_fallback() switches provider/model. + from cli import _reverse_alias_for_display, datetime, format_duration_compact + agent = getattr(self, "agent", None) + model_name = (getattr(agent, "model", None) or self.model or "unknown") + # Friendly display: prefer reverse-alias from config.yaml ``model_aliases:`` + # before slash/length truncation. This turns long Palantir RIDs like + # ``ri.language-model-service..language-model.anthropic-claude-4-7-opus`` + # into the user's chosen short name (e.g. ``opus-4.7``) in the status bar. + model_short = _reverse_alias_for_display(model_name) + if model_short == model_name: + model_short = model_name.split("/")[-1] if "/" in model_name else model_name + # Strip Palantir RID prefixes via the shared display formatter so + # this site and ``ModelSwitchResult`` confirmation can't drift. + from hermes_cli.model_switch import format_model_for_display + model_short = format_model_for_display(model_short) + if model_short.endswith(".gguf"): + model_short = model_short[:-5] + if len(model_short) > 26: + model_short = f"{model_short[:23]}..." + + elapsed_seconds = max(0.0, (datetime.now() - self.session_start).total_seconds()) + snapshot = { + "model_name": model_name, + "model_short": model_short, + "duration": format_duration_compact(elapsed_seconds), + "session_title": self._get_status_bar_session_title(), + "prompt_elapsed": self._format_prompt_elapsed( + getattr(self, "_prompt_start_time", None), + getattr(self, "_prompt_duration", 0.0), + live=getattr(self, "_prompt_start_time", None) is not None, + ), + "idle_since": self._format_idle_since( + getattr(self, "_last_turn_finished_at", None), + turn_live=getattr(self, "_prompt_start_time", None) is not None, + ), + "context_tokens": 0, + "context_length": None, + "context_percent": None, + "session_input_tokens": 0, + "session_output_tokens": 0, + "session_cache_read_tokens": 0, + "session_cache_write_tokens": 0, + "session_prompt_tokens": 0, + "session_completion_tokens": 0, + "session_total_tokens": 0, + "session_api_calls": 0, + "compressions": 0, + "active_background_tasks": 0, + "active_background_processes": 0, + "active_background_subagents": 0, + "battery_label": "", + "battery_category": "dim", + # Focus view badge (/focus). Persistent indicator so the reduced + # output mode is never invisible. Display-only. + "focus_label": "", + } + + try: + from hermes_cli.focus_view import focus_statusbar_segment + + snapshot["focus_label"] = focus_statusbar_segment( + bool(getattr(self, "_focus_view_enabled", False)) + ) + except Exception: + pass + + # Battery read-out (first status-bar element when enabled). Reads are + # memoised for a few seconds inside agent.battery, so polling it on + # every status-bar repaint is cheap. + if getattr(self, "_battery_visible", False): + try: + from agent.battery import ( + battery_category, + format_battery, + read_battery, + ) + + _batt = read_battery() + snapshot["battery_label"] = format_battery(_batt) + snapshot["battery_category"] = battery_category(_batt) + except Exception: + pass + + # Count live /bg tasks. The dict entry is removed in the + # task thread's finally block, so len() reflects truly-running tasks. + # len() on a CPython dict is atomic; safe to read without a lock. + try: + bg_tasks = getattr(self, "_background_tasks", None) + if bg_tasks: + snapshot["active_background_tasks"] = len(bg_tasks) + except Exception: + pass + + # Count live background terminal processes (terminal tool background + # sessions tracked by tools.process_registry). Cheap O(1) read. + try: + from tools.process_registry import process_registry + snapshot["active_background_processes"] = process_registry.count_running() + except Exception: + pass + + # Count live background/async subagents (delegate_task batches and + # background single delegations tracked by tools.async_delegation). + # active_count() iterates an in-memory records dict under a lock — + # cheap and only counts records still in the "running" state. + try: + from tools.async_delegation import active_count as _async_active_count + snapshot["active_background_subagents"] = _async_active_count() + except Exception: + pass + + # Standing /goal state (Ralph loop). GoalManager is cached on self and + # keeps its state in memory, so this is a cheap attribute read — no DB + # hit per repaint. Only an *active* goal earns a segment; paused/done + # goals stay out of the bar (matching the desktop's active-first row). + snapshot["goal_active"] = False + snapshot["goal_turns_used"] = 0 + snapshot["goal_max_turns"] = 0 + try: + goal_mgr = self._get_goal_manager() + if goal_mgr is not None and goal_mgr.is_active(): + goal_state = goal_mgr.state + snapshot["goal_active"] = True + snapshot["goal_turns_used"] = int(getattr(goal_state, "turns_used", 0) or 0) + snapshot["goal_max_turns"] = int(getattr(goal_state, "max_turns", 0) or 0) + except Exception: + pass + + + if not agent: + return snapshot + + snapshot["session_input_tokens"] = getattr(agent, "session_input_tokens", 0) or 0 + snapshot["session_output_tokens"] = getattr(agent, "session_output_tokens", 0) or 0 + snapshot["session_cache_read_tokens"] = getattr(agent, "session_cache_read_tokens", 0) or 0 + snapshot["session_cache_write_tokens"] = getattr(agent, "session_cache_write_tokens", 0) or 0 + snapshot["session_prompt_tokens"] = getattr(agent, "session_prompt_tokens", 0) or 0 + snapshot["session_completion_tokens"] = getattr(agent, "session_completion_tokens", 0) or 0 + snapshot["session_total_tokens"] = getattr(agent, "session_total_tokens", 0) or 0 + snapshot["session_api_calls"] = getattr(agent, "session_api_calls", 0) or 0 + + compressor = getattr(agent, "context_compressor", None) + if compressor: + # last_prompt_tokens is parked at the -1 sentinel right after a + # compression, until the next real API call reports a prompt count + # (awaiting_real_usage_after_compression). The status bar must not + # render that sentinel verbatim — it produced "-1/200K" / "-1%". + # Clamp it to 0 so the one transitional turn reads as empty context. + context_tokens = getattr(compressor, "last_prompt_tokens", 0) or 0 + if context_tokens < 0: + context_tokens = 0 + # Durable-transcript view: on reasoning models a long tool loop + # replays the current turn's thinking + scaffolding on every + # request, so the LAST request's prompt_tokens can exceed the + # durable transcript by hundreds of K — all of which evaporates + # at the turn boundary. Rendering that raw figure makes the bar + # sawtooth (e.g. 850K mid-turn -> 600K next turn) and reads as a + # broken compaction. Anchor the display on the turn's FIRST + # response (minimal replay) plus a delta estimate of messages + # appended since, excluding stale thinking. Display-only: the + # compression trigger keeps using real last-request usage. + try: + from agent.model_metadata import anchored_context_tokens + + _msgs = getattr(agent, "_session_messages", None) + _anchored = anchored_context_tokens( + _msgs if isinstance(_msgs, list) else [], + getattr(agent, "_turn_base_usage_anchor", None), + charge_stale_thinking=False, + ) + if _anchored is not None and _anchored > 0: + context_tokens = _anchored + except Exception: + pass + context_length = getattr(compressor, "context_length", 0) or 0 + if context_length < 0: + context_length = 0 + snapshot["context_tokens"] = context_tokens + snapshot["context_length"] = context_length or None + snapshot["compressions"] = getattr(compressor, "compression_count", 0) or 0 + if context_length: + snapshot["context_percent"] = max(0, min(100, round((context_tokens / context_length) * 100))) + + # -- Cache-hit ratio (delta since last reset) -- + # Reset baseline on model switch and on compression — both invalidate + # the prompt cache. Formula verified against live logs: + # hit = cache_read / prompt_tokens (prompt = input+cache_read+cache_write) + # see agent/conversation_loop.py:4314 cache=read/prompt (87%) + # and CanonicalUsage.prompt_tokens = input+read+write + try: + base_model = getattr(self, "_cache_hit_baseline_model", None) + base_prompt = int(getattr(self, "_cache_hit_baseline_prompt", 0) or 0) + base_read = int(getattr(self, "_cache_hit_baseline_read", 0) or 0) + base_comps = int(getattr(self, "_cache_hit_baseline_compressions", 0) or 0) + cur_model = snapshot.get("model_name") or model_name + cur_comps = int(snapshot.get("compressions", 0) or 0) + cur_prompt = int(snapshot.get("session_prompt_tokens", 0) or 0) + cur_read = int(snapshot.get("session_cache_read_tokens", 0) or 0) + if base_model is None: + self._cache_hit_baseline_model = cur_model + self._cache_hit_baseline_compressions = cur_comps + base_model = cur_model + base_comps = cur_comps + if cur_model != base_model: + self._cache_hit_baseline_model = cur_model + self._cache_hit_baseline_prompt = cur_prompt + self._cache_hit_baseline_read = cur_read + self._cache_hit_baseline_compressions = cur_comps + base_prompt = cur_prompt + base_read = cur_read + base_comps = cur_comps + if cur_comps != base_comps: + self._cache_hit_baseline_compressions = cur_comps + self._cache_hit_baseline_prompt = cur_prompt + self._cache_hit_baseline_read = cur_read + base_prompt = cur_prompt + base_read = cur_read + delta_prompt = cur_prompt - base_prompt + delta_read = cur_read - base_read + # A zero-read regime hides the segment entirely (no cache data + # is not the same as a 0% hit worth alarming about), and the pct + # stays a float so renderers control their own precision. + if delta_prompt > 0 and delta_read > 0: + pct = max(0.0, min(100.0, (delta_read / delta_prompt) * 100)) + snapshot["cache_hit_pct"] = pct + snapshot["cache_hit_label"] = f"{pct:.0f}%" + elif cur_prompt > 0 and cur_read > 0 and base_prompt == 0 and base_read == 0: + pct = max(0.0, min(100.0, (cur_read / cur_prompt) * 100)) + snapshot["cache_hit_pct"] = pct + snapshot["cache_hit_label"] = f"{pct:.0f}%" + else: + snapshot["cache_hit_pct"] = None + snapshot["cache_hit_label"] = "" + except Exception: + snapshot["cache_hit_pct"] = None + snapshot["cache_hit_label"] = "" + + # -- Rolling avg latency / velocity (last 10 calls) -- + # Reads the deque maintained in agent/conversation_loop.py (and + # agent_init). Codex app-server has no latency, so it stays hidden there. + try: + agent_obj = getattr(self, "agent", None) + lhist = list(getattr(agent_obj, "_api_latency_history", []) or []) if agent_obj else [] + ohist = list(getattr(agent_obj, "_api_output_history", []) or []) if agent_obj else [] + # Keep the two histories aligned (they are appended together). + n = min(len(lhist), len(ohist)) + if n: + lhist = lhist[-n:] + ohist = ohist[-n:] + # Simple mean for latency; sum/sum for velocity (true throughput, not mean of ratios). + avg_lat = sum(lhist) / len(lhist) if lhist else None + total_out = sum(ohist) + total_lat = sum(lhist) + avg_vel = (total_out / total_lat) if total_lat > 0 else None + # Guard against NaN / inf from weird provider timings (e.g. -0.8s in logs). + if avg_lat is not None and (avg_lat != avg_lat or avg_lat < 0 or avg_lat > 1e6): + avg_lat = None + if avg_vel is not None and (avg_vel != avg_vel or avg_vel < 0 or avg_vel > 1e6): + avg_vel = None + snapshot["avg_latency"] = float(avg_lat) if avg_lat is not None else None + snapshot["avg_latency_label"] = f"{avg_lat:.1f}s" if avg_lat is not None else "" + snapshot["avg_velocity"] = float(avg_vel) if avg_vel is not None else None + snapshot["avg_velocity_label"] = f"{avg_vel:.0f} t/s" if avg_vel is not None else "" + else: + snapshot["avg_latency"] = None + snapshot["avg_latency_label"] = "" + snapshot["avg_velocity"] = None + snapshot["avg_velocity_label"] = "" + except Exception: + snapshot["avg_latency"] = None + snapshot["avg_latency_label"] = "" + snapshot["avg_velocity"] = None + snapshot["avg_velocity_label"] = "" + + return snapshot + + def _get_status_bar_session_title(self) -> str: + """Return the current title without polling state.db on every repaint.""" + pending = str(getattr(self, "_pending_title", None) or "").strip() + session_id = str(getattr(self, "session_id", "") or "") + if pending: + self._status_bar_title_session_id = session_id + self._status_bar_title_cache = pending + self._status_bar_title_checked_at = time.monotonic() + return pending + + now = time.monotonic() + cached_session_id = getattr(self, "_status_bar_title_session_id", None) + checked_at = float(getattr(self, "_status_bar_title_checked_at", 0.0) or 0.0) + if cached_session_id == session_id and now - checked_at < 1.5: + return str(getattr(self, "_status_bar_title_cache", "") or "") + + title = "" + db = getattr(self, "_session_db", None) + if db is not None and session_id: + try: + title = str(db.get_session_title(session_id) or "").strip() + except Exception: + title = "" + self._status_bar_title_session_id = session_id + self._status_bar_title_cache = title + self._status_bar_title_checked_at = now + return title + + @staticmethod + def _status_bar_display_width(text: str) -> int: + """Return terminal cell width for status-bar text. + + len() is not enough for prompt_toolkit layout decisions because some + glyphs can render wider than one Python codepoint. Keeping the status + bar within the real display width prevents it from wrapping onto a + second line and leaving behind duplicate rows. + """ + try: + from prompt_toolkit.utils import get_cwidth + return get_cwidth(text or "") + except Exception: + return len(text or "") + + @classmethod + def _trim_status_bar_text(cls, text: str, max_width: int) -> str: + """Trim status-bar text to a single terminal row.""" + if max_width <= 0: + return "" + try: + from prompt_toolkit.utils import get_cwidth + except Exception: + get_cwidth = None + + if cls._status_bar_display_width(text) <= max_width: + return text + + ellipsis = "..." + ellipsis_width = cls._status_bar_display_width(ellipsis) + if max_width <= ellipsis_width: + return ellipsis[:max_width] + + out = [] + width = 0 + for ch in text: + ch_width = get_cwidth(ch) if get_cwidth else len(ch) + if width + ch_width + ellipsis_width > max_width: + break + out.append(ch) + width += ch_width + return "".join(out).rstrip() + ellipsis + + @classmethod + def _right_align_status_title(cls, text: str, title: str, width: int) -> str: + """Pin a bounded session-title badge to the far-right status-bar edge.""" + title = str(title or "").strip() + if not title or width < 24: + return cls._trim_status_bar_text(text, width) + + title_width = max(6, min(30, width // 3)) + badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " + suffix = f" ─{badge}" + left_width = max(0, width - cls._status_bar_display_width(suffix)) + left = cls._trim_status_bar_text(text.rstrip(), left_width) + padding = " " * max(0, left_width - cls._status_bar_display_width(left)) + return f"{left}{padding}{suffix}" + + @classmethod + def _right_align_status_title_fragments(cls, frags, title: str, width: int): + """Styled counterpart to :meth:`_right_align_status_title`.""" + title = str(title or "").strip() + if not title or width < 24: + return frags + + title_width = max(6, min(30, width // 3)) + badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " + suffix_width = cls._status_bar_display_width(" ─") + cls._status_bar_display_width(badge) + left_width = max(0, width - suffix_width) + trimmed = [] + used = 0 + for style, value in frags: + remaining = left_width - used + if remaining <= 0: + break + value_width = cls._status_bar_display_width(value) + if value_width <= remaining: + trimmed.append((style, value)) + used += value_width + continue + clipped = cls._trim_status_bar_text(value, remaining) + if clipped: + trimmed.append((style, clipped)) + used += cls._status_bar_display_width(clipped) + break + + if used < left_width: + trimmed.append(("class:status-bar-dim", " " * (left_width - used))) + trimmed.extend([ + ("class:status-bar-dim", " ─"), + ("class:status-bar-session-title", badge), + ]) + return trimmed + + @staticmethod + def _get_tui_terminal_width(default: tuple[int, int] = (80, 24)) -> int: + """Return the live prompt_toolkit width, falling back to ``shutil``. + + The TUI layout can be narrower than ``shutil.get_terminal_size()`` reports, + especially on Termux/mobile shells, so prefer prompt_toolkit's width whenever + an app is active. + """ + try: + from prompt_toolkit.application import get_app + return get_app().output.get_size().columns + except Exception: + return shutil.get_terminal_size(default).columns + + def _use_minimal_tui_chrome(self, width: Optional[int] = None) -> bool: + """Hide low-value chrome on narrow/mobile terminals to preserve rows.""" + if width is None: + width = self._get_tui_terminal_width() + return width < 64 + + @staticmethod + def _scrollback_box_width(width: Optional[int] = None) -> int: + """Return the full viewport width for printed scrollback box rules. + + Previously this clamped to ``max(32, min(width, 56))`` as a defense + against terminal-emulator reflow on column-shrink (#25975, salvaging + #24403). That clamp made response/reasoning borders look stubby on + any modern wide terminal. We now trust the prompt_toolkit + ``_output_screen_diff`` monkey-patch landed in #26137 (salvaging + #25981) to keep chrome out of scrollback in the first place, and + accept that an aggressive column-shrink may visually reflow already + printed Panel borders — that's a cosmetic artifact of stamped + scrollback history, not a live-render bug. + + A small floor (32 cols) is kept so the box still renders on tiny + terminals without negative ``'─' * (w - 2)`` math. + """ + if width is None: + try: + width = shutil.get_terminal_size((80, 24)).columns + except Exception: + width = 80 + return max(32, int(width or 80)) + + def _agent_spacer_height(self, width: Optional[int] = None) -> int: + """Return the spacer height shown above the status bar while the agent runs.""" + if not getattr(self, "_agent_running", False): + return 0 + return 0 if self._use_minimal_tui_chrome(width=width) else 1 + + def _spinner_widget_height(self, width: Optional[int] = None) -> int: + """Return the visible height for the spinner/status text line above the status bar.""" + spinner_line = self._render_spinner_text() + if not spinner_line: + return 0 + if self._use_minimal_tui_chrome(width=width): + return 0 + width = width or self._get_tui_terminal_width() + if width and width > 10: + import math + text_width = self._status_bar_display_width(spinner_line) + return max(1, math.ceil(text_width / width)) + return 1 + + def _render_spinner_text(self) -> str: + """Return the live spinner/status text exactly as rendered in the TUI.""" + txt = getattr(self, "_spinner_text", "") + if not txt: + return "" + flow = self._spinner_token_flow() + t0 = getattr(self, "_tool_start_time", 0) or 0 + if t0 > 0: + elapsed = time.monotonic() - t0 + if elapsed >= 60: + _m, _s = int(elapsed // 60), int(elapsed % 60) + # Fixed-width timer to avoid status-line wrap jitter while + # scrolling/repainting (e.g. 01m05s, 12m09s). + elapsed_str = f"{_m:02d}m{_s:02d}s" + else: + # Keep width stable before the 60s rollover as well. + elapsed_str = f"{elapsed:5.1f}s" + if flow: + return f" {txt} ({elapsed_str} · {flow})" + return f" {txt} ({elapsed_str})" + if flow: + return f" {txt} ({flow})" + return f" {txt}" + + def _spinner_token_flow(self) -> str: + """Cumulative output tokens for the running turn, for the spinner.""" + if not getattr(self, "_spinner_token_flow_enabled", False): + return "" + if not getattr(self, "_agent_running", False): + return "" + agent = getattr(self, "agent", None) + if agent is None: + return "" + try: + from agent.turn_summary import format_token_flow + + produced = (getattr(agent, "session_output_tokens", 0) or 0) - ( + getattr(self, "_turn_token_baseline", 0) or 0 + ) + return format_token_flow(produced) + except Exception: + return "" + + def _turn_summary_is_active(self) -> bool: + """Whether the per-turn summary line should render for this surface. + + Gated off for: the config key, quiet/tool-progress-off mode, and any + non-interactive path (single query, ``-Q``, gateway/messaging) — those + surfaces either want machine-readable output or carry their own footer. + """ + if not getattr(self, "_turn_summary_enabled", False): + return False + if getattr(self, "tool_progress_mode", "all") == "off": + return False + agent = getattr(self, "agent", None) + if agent is not None and getattr(agent, "quiet_mode", False): + return False + return bool(getattr(self, "_interactive_turn", False)) + + def _turn_summary_begin(self) -> None: + """Start per-turn accounting for the turn that is about to run.""" + try: + from agent.turn_summary import TurnSummaryCollector + + collector = getattr(self, "_turn_summary_collector", None) + if collector is None: + collector = TurnSummaryCollector() + self._turn_summary_collector = collector + collector.begin() + self._turn_summary_start = time.monotonic() + agent = getattr(self, "agent", None) + self._turn_token_baseline = ( + getattr(agent, "session_output_tokens", 0) or 0 + ) if agent is not None else 0 + except Exception: + self._turn_summary_collector = None + + def _turn_summary_record(self, function_name, result, is_error: bool) -> None: + """Feed one completed tool call into the active tally.""" + collector = getattr(self, "_turn_summary_collector", None) + if collector is None: + return + try: + collector.record_tool(function_name, result=result, is_error=bool(is_error)) + except Exception: + pass + + def _turn_summary_emit(self) -> None: + """Print the post-turn accounting line, when enabled for this surface.""" + from cli import _DIM, _RST, _cprint, logger + collector = getattr(self, "_turn_summary_collector", None) + if collector is None or not self._turn_summary_is_active(): + return + try: + started = getattr(self, "_turn_summary_start", 0.0) or 0.0 + elapsed = max(0.0, time.monotonic() - started) if started else 0.0 + line = collector.render(elapsed) + if line: + _cprint(f" {_DIM}{line}{_RST}") + except Exception: + logger.debug("Turn summary render failed", exc_info=True) + + def _pet_clear_runtime(self) -> None: + """Drop renderer + queued Kitty state. Caller holds ``_pet_lock``.""" + self._pet_enabled = False + self._pet_renderer = None + self._pet_frames_cache.clear() + self._pet_kitty_cache.clear() + self._pet_kitty_pending = "" + self._pet_kitty_image_id = 0 + + def _pet_resolve_config(self) -> None: + """(Re)resolve the active pet from config — picks up live enable/disable/ + + switch made via ``/pet`` or ``hermes pets`` without a restart, mirroring + the TUI's steady poll. Cheap and fail-open: any problem disables the pet. + """ + try: + from agent.pet import constants, store + from hermes_cli.config import load_config + + cfg = load_config() + display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} + pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} + + from utils import is_truthy_value + + enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) + slug = str(pet_cfg.get("slug", "") or "") + scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) + cols = constants.resolve_cols(scale, pet_cfg.get("unicode_cols", 0)) + configured_mode = str(pet_cfg.get("render_mode", "auto") or "auto").lower() + # Placeholders only on kitty/Ghostty. WezTerm speaks kitty APC but + # not U+10EEEE — detect_terminal_graphics() still returns kitty + # there, which is why this gate is narrower. + use_kitty = configured_mode in ("", "auto", "kitty") and pet_render.supports_kitty_placeholders() + renderer_mode = "kitty" if use_kitty else "unicode" + + if not enabled or configured_mode == "off": + with self._pet_lock: + self._pet_clear_runtime() + return + + pet = store.resolve_active_pet(slug) + if pet is None or not pet.exists: + with self._pet_lock: + self._pet_clear_runtime() + return + + with self._pet_lock: + # Rebuild only when the resolved pet, mode, or geometry changes. + if ( + self._pet_renderer is None + or self._pet_slug != pet.slug + or self._pet_cols != cols + or self._pet_scale != scale + or self._pet_renderer.mode != renderer_mode + ): + self._pet_renderer = pet_render.PetRenderer( + str(pet.spritesheet), mode=renderer_mode, scale=scale, unicode_cols=cols + ) + self._pet_slug = pet.slug + self._pet_cols = cols + self._pet_scale = scale + self._pet_frames_cache.clear() + self._pet_kitty_cache.clear() + self._pet_kitty_pending = "" + self._pet_kitty_image_id = pet_render.kitty_image_id(pet.slug) + self._pet_frame_idx = 0 + self._pet_enabled = True + except Exception: + with self._pet_lock: + self._pet_clear_runtime() + + def _pet_flash(self, state: str, secs: float = 1.6) -> None: + """Briefly force a transient reaction (wave/jump/failed) before resting.""" + self._pet_event = state + self._pet_event_until = time.monotonic() + secs + + def _on_reaction(self, kind: str) -> None: + """User affection (ily / <3 / good bot), core-detected — the pet's share + of the vibe signal that plays hearts on the TUI/desktop. Flash a celebrate.""" + if kind == "vibe": + self._pet_flash("jump") + + def _pet_react_turn_end(self) -> None: + """Flash the end-of-turn beat: failed on error, jump on a finished plan, else wave.""" + if not self._pet_enabled: + return + from agent.pet.state import todos_all_done + + if self._pet_turn_error: + self._pet_flash("failed") + return + try: + store = getattr(self.agent, "_todo_store", None) + done = todos_all_done(store.read()) if store else False + except Exception: + done = False + self._pet_flash("jump" if done else "wave") + + def _derive_pet_state(self) -> str: + """Map current CLI activity to a pet animation state. + + A transient reaction beat (wave/jump/failed) wins while it's live; + otherwise the steady state comes from the shared + :func:`agent.pet.state.derive_pet_state` so the CLI can't drift from the + TUI/desktop priority order. + """ + if self._pet_event and time.monotonic() < self._pet_event_until: + return self._pet_event + self._pet_event = "" + from agent.pet.state import derive_pet_state + + # A live blocking modal (approval / clarify / sudo / secret / slash + # confirm) means the agent is paused on the user → the `waiting` pose, + # which outranks the in-flight signals in derive_pet_state. + awaiting_input = bool( + self._approval_state + or self._clarify_state + or self._sudo_state + or self._secret_state + or getattr(self, "_slash_confirm_state", None) + ) + + return derive_pet_state( + awaiting_input=awaiting_input, + busy=getattr(self, "_agent_running", False), + reasoning=self._pet_reasoning, + ).value + + def _pet_frames_for(self, state: str) -> list: + """Return (and cache) the half-block grids for one state.""" + cached = self._pet_frames_cache.get(state) + if cached is not None: + return cached + renderer = self._pet_renderer + if renderer is None: + return [] + try: + count = renderer.frame_count(state) or 1 + grids = [renderer.cells(state, i, cols=self._pet_cols) for i in range(count)] + except Exception: + grids = [] + self._pet_frames_cache[state] = grids + return grids + + def _pet_kitty_payload_for(self, state: str) -> dict | None: + """Return and cache a Kitty virtual-placeholder payload for *state*.""" + with self._pet_lock: + cached = self._pet_kitty_cache.get(state) + if cached is not None: + return cached + renderer = self._pet_renderer + image_id = self._pet_kitty_image_id + if renderer is None or renderer.mode != "kitty": + return None + try: + # PNG encoding is outside _pet_lock: first visit of a state must + # not stall the prompt under the lock. + payload = renderer.kitty_payload(state, image_id=image_id) + except Exception: + payload = None + if payload is not None: + payload = {**payload, "image_id": image_id} + with self._pet_lock: + if self._pet_renderer is renderer and self._pet_kitty_image_id == image_id: + self._pet_kitty_cache[state] = payload + return payload + + def _pet_queue_kitty_frame(self, state: str | None = None) -> None: + """Queue one virtual Kitty frame for the next prompt_toolkit render. + + No-op when the pet pane was never initialized (``__new__`` fixtures + and ``_force_full_redraw`` / resize recovery on a pet-less CLI). + """ + if not getattr(self, "_pet_enabled", False): + return + if state is None: + state = self._derive_pet_state() + payload = self._pet_kitty_payload_for(state) + if not payload or not payload.get("frames"): + return + with self._pet_lock: + if self._pet_renderer is not None and self._pet_renderer.mode == "kitty": + self._pet_kitty_pending = payload["frames"][self._pet_frame_idx % len(payload["frames"])] + + def _pet_flush_kitty_frame(self, app) -> None: + """Write a queued APC after prompt_toolkit has finished its screen diff.""" + with self._pet_lock: + frame = self._pet_kitty_pending + self._pet_kitty_pending = "" + if not frame: + return + try: + # U=1/q=2 leaves the cursor and input stream untouched. + app.output.write_raw(frame) + app.output.flush() + except (OSError, ValueError): + pass + + def _pet_fragments(self): + """Return prompt_toolkit FormattedText for the current pet frame, or [].""" + with self._pet_lock: + if not self._pet_enabled or self._pet_renderer is None: + return [] + state = self._derive_pet_state() + kitty = self._pet_renderer.mode == "kitty" + if kitty: + payload = self._pet_kitty_payload_for(state) + if not payload: + return [] + color = pet_render.kitty_color_hex(payload["image_id"]) + frags = [] + for y, row in enumerate(payload["placeholder"]): + if y: + frags.append(("", "\n")) + frags.append((f"fg:{color}", row)) + return frags + with self._pet_lock: + grids = self._pet_frames_for(state) + if not grids: + return [] + grid = grids[self._pet_frame_idx % len(grids)] + + frags = [] + for y, row in enumerate(grid): + if y: + frags.append(("", "\n")) + for top, bottom in row: + tr, tg, tb, ta = top + br, bg, bb, ba = bottom + top_op = ta >= 32 + bot_op = ba >= 32 + if not top_op and not bot_op: + frags.append(("", " ")) + elif top_op and bot_op: + frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x} bg:#{br:02x}{bg:02x}{bb:02x}", "▀")) + elif top_op: + # Upper half only — leave the lower half the terminal's bg + # instead of painting it black (cleaner on light themes). + frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x}", "▀")) + else: + frags.append((f"fg:#{br:02x}{bg:02x}{bb:02x}", "▄")) + return frags + + def _pet_widget_height(self) -> int: + """Visible rows for the pet window — 0 collapses it when no pet shows.""" + with self._pet_lock: + if not self._pet_enabled or self._pet_renderer is None: + return 0 + state = self._derive_pet_state() + kitty = self._pet_renderer.mode == "kitty" + if kitty: + payload = self._pet_kitty_payload_for(state) + return int(payload.get("rows", 0)) if payload else 0 + with self._pet_lock: + grids = self._pet_frames_for(state) + if not grids or not grids[0]: + return 0 + return len(grids[0]) + + def _pet_anim_loop(self) -> None: + """Advance the frame + invalidate on a timer while a pet is enabled.""" + while self._pet_anim_running: + time.sleep(self._PET_FRAME_INTERVAL) + if getattr(self, "_terminal_io_broken", False): + self._pet_anim_running = False + break + now = time.monotonic() + if now - self._pet_cfg_checked >= self._PET_CFG_INTERVAL: + self._pet_cfg_checked = now + self._pet_resolve_config() + if not self._pet_enabled: + continue + with self._pet_lock: + self._pet_frame_idx += 1 + kitty = self._pet_renderer is not None and self._pet_renderer.mode == "kitty" + if kitty: + self._pet_queue_kitty_frame() + app = getattr(self, "_app", None) + if app is not None: + try: + app.invalidate() + except OSError as exc: + if getattr(exc, "errno", None) == errno.EIO: + self._mark_terminal_io_broken("pet_anim") + break + except Exception: + pass + + def _pet_start_anim(self) -> None: + if self._pet_anim_running: + return + self._pet_resolve_config() + with self._pet_lock: + kitty = self._pet_enabled and self._pet_renderer is not None and self._pet_renderer.mode == "kitty" + if kitty: + self._pet_queue_kitty_frame() + self._pet_anim_running = True + self._pet_anim_thread = threading.Thread(target=self._pet_anim_loop, daemon=True) + self._pet_anim_thread.start() + + def _pet_stop_anim(self) -> None: + self._pet_anim_running = False + thread = self._pet_anim_thread + if thread is not None: + thread.join(timeout=0.3) + self._pet_anim_thread = None + + def _voice_record_key_label(self) -> str: + """Return the configured voice push-to-talk key formatted for UI. + + Shared helper so every voice-facing status line / placeholder / + recording hint advertises the SAME label as the registered + prompt_toolkit binding. + + Cached at startup (see ``set_voice_record_key_cache``) rather + than re-read per render. Two reasons (Copilot round-13 on + #19835): + + * The prompt_toolkit binding is registered once at session + start via ``@kb.add(_voice_key)``; re-reading config per + render meant the status bar could advertise a new shortcut + after a config edit while the actual binding was still the + startup chord — exactly the display/binding drift this PR + is trying to eliminate. + * The label is on the hot render path (status bar + composer + placeholder invalidated every 150ms during recording), so + reading config on every call added avoidable UI overhead. + """ + return getattr(self, "_voice_record_key_display_cache", None) or "Ctrl+B" + + def set_voice_record_key_cache(self, raw_key: object) -> None: + """Populate the voice label cache from a raw ``voice.record_key``. + + Called at CLI startup after the prompt_toolkit binding is + registered so the cached label always matches the live binding. + """ + try: + from hermes_cli.voice import format_voice_record_key_for_status + self._voice_record_key_display_cache = format_voice_record_key_for_status(raw_key) + except Exception: + self._voice_record_key_display_cache = "Ctrl+B" + + def _get_voice_status_fragments(self, width: Optional[int] = None): + """Return the voice status bar fragments for the interactive TUI.""" + width = width or self._get_tui_terminal_width() + compact = self._use_minimal_tui_chrome(width=width) + label = self._voice_record_key_label() + if self._voice_recording: + if compact: + return [("class:voice-status-recording", " ● REC ")] + return [("class:voice-status-recording", f" ● REC {label} to stop ")] + if self._voice_processing: + if compact: + return [("class:voice-status", " ◉ STT ")] + return [("class:voice-status", " ◉ Transcribing... ")] + if compact: + return [("class:voice-status", f" 🎤 {label} ")] + tts = " | TTS on" if self._voice_tts else "" + cont = " | Continuous" if self._voice_continuous else "" + return [("class:voice-status", f" 🎤 Voice mode{tts}{cont} — {label} to record ")] + + @staticmethod + def _status_bar_goal_segment(snapshot: Dict[str, Any]) -> str: + """Return the ``⊙ goal 3/20`` segment, or ``""`` when no goal is active. + + Active-goal-only by design: paused/done goals don't occupy status-bar + real estate (they already print their own glyph lines in the thread). + """ + if not snapshot.get("goal_active"): + return "" + used = snapshot.get("goal_turns_used") or 0 + max_turns = snapshot.get("goal_max_turns") or 0 + if max_turns: + return f"⊙ goal {used}/{max_turns}" + return "⊙ goal" + + def _get_status_bar_field_set(self) -> Optional[frozenset]: + """Return the set of visible status-bar fields from config. + + Reads ``display.status_bar.fields`` from the module-level + ``CLI_CONFIG`` (no per-render YAML parse — the status bar repaints + every frame). Returns ``None`` when the user has not customized the + bar (use built-in defaults, i.e. show everything), or a + ``frozenset`` of field names when the list is non-empty. + + Available fields: model, context_detail, context_pct, cache_hit, + latency, tps, compressions, bg_tasks, bg_processes, bg_subagents, + goal, duration, prompt_elapsed, idle_since, focus, yolo, stash, + battery, title, total_tokens. + ``total_tokens`` is opt-in only (never shown by default). + The field order is fixed; the config controls visibility only. + """ + from cli import CLI_CONFIG + if hasattr(self, "_status_bar_field_set_cache"): + return self._status_bar_field_set_cache + result = None + try: + display = CLI_CONFIG.get("display") if isinstance(CLI_CONFIG, dict) else None + status_bar = (display or {}).get("status_bar") if isinstance(display, dict) else None + fields = status_bar.get("fields") if isinstance(status_bar, dict) else None + if isinstance(fields, list) and fields: + result = frozenset(str(f) for f in fields) + except Exception: + result = None + self._status_bar_field_set_cache = result + return result + + def _build_status_bar_text(self, width: Optional[int] = None) -> str: + """Return a compact one-line session status string for the TUI footer.""" + from cli import format_token_count_compact + try: + snapshot = self._get_status_bar_snapshot() + if width is None: + width = self._get_tui_terminal_width() + percent = snapshot["context_percent"] + percent_label = f"{percent}%" if percent is not None else "--" + duration_label = snapshot["duration"] + battery_label = snapshot.get("battery_label") or "" + battery_prefix = f"{battery_label} │ " if battery_label else "" + focus_label = snapshot.get("focus_label") or "" + session_title = snapshot.get("session_title") or "" + + yolo_active = self._is_session_yolo_active() + goal_segment = self._status_bar_goal_segment(snapshot) + field_set = self._get_status_bar_field_set() + + def _ok(name: str) -> bool: + return field_set is None or name in field_set + + if not _ok("title"): + session_title = "" + + if not _ok("goal"): + goal_segment = "" + if not _ok("focus"): + focus_label = "" + if width < 52: + segs = [] + if _ok("model"): + segs.append(f"⚕ {snapshot['model_short']}") + if _ok("duration"): + segs.append(duration_label) + if goal_segment: + segs.append(goal_segment) + if focus_label: + segs.append(focus_label) + if yolo_active and _ok("yolo"): + segs.append("⚠ YOLO") + text = battery_prefix + " · ".join(segs) if segs else f"{battery_prefix}⚕ {snapshot['model_short']}" + return self._right_align_status_title(text, session_title, width) + if width < 76: + parts = [] + if _ok("model"): + parts.append(f"⚕ {snapshot['model_short']}") + if _ok("context_pct"): + parts.append(percent_label) + cache = self._cache_hit_rate(snapshot, precision=0) + if cache and _ok("cache_hit"): + parts.append(cache[1]) + if battery_label: + parts.insert(0, battery_label) + compressions = snapshot.get("compressions", 0) + if compressions and _ok("compressions"): + parts.append(f"🗜️ {compressions}") + bg_count = snapshot.get("active_background_tasks", 0) + if bg_count and _ok("bg_tasks"): + parts.append(f"▶ {bg_count}") + bg_proc_count = snapshot.get("active_background_processes", 0) + if bg_proc_count and _ok("bg_processes"): + parts.append(f"⚙ {bg_proc_count}") + bg_subagent_count = snapshot.get("active_background_subagents", 0) + if bg_subagent_count and _ok("bg_subagents"): + parts.append(f"⛓ {bg_subagent_count}") + if goal_segment: + parts.append(goal_segment) + if _ok("duration"): + parts.append(duration_label) + if focus_label: + parts.append(focus_label) + if yolo_active and _ok("yolo"): + parts.append("⚠ YOLO") + if not parts: + parts = [f"⚕ {snapshot['model_short']}"] + return self._right_align_status_title(" · ".join(parts), session_title, width) + + parts = [] + if _ok("model"): + parts.append(f"⚕ {snapshot['model_short']}") + if _ok("context_detail"): + if snapshot["context_length"]: + ctx_total = _format_context_length(snapshot["context_length"]) + ctx_used = format_token_count_compact(snapshot["context_tokens"]) + context_label = f"{ctx_used}/{ctx_total}" + else: + context_label = "ctx --" + parts.append(context_label) + if _ok("context_pct"): + parts.append(percent_label) + if battery_label: + parts.insert(0, battery_label) + compressions = snapshot.get("compressions", 0) + cache = self._cache_hit_rate(snapshot) + if cache and _ok("cache_hit"): + parts.append(cache[1]) + _avg_lat = snapshot.get("avg_latency_label") or "" + if _avg_lat and _ok("latency"): + parts.append(f"◷ {_avg_lat}") + _avg_vel = snapshot.get("avg_velocity_label") or "" + if _avg_vel and _ok("tps"): + parts.append(f"↑ {_avg_vel}") + if compressions and _ok("compressions"): + parts.append(f"🗜️ {compressions}") + bg_count = snapshot.get("active_background_tasks", 0) + if bg_count and _ok("bg_tasks"): + parts.append(f"▶ {bg_count}") + bg_proc_count = snapshot.get("active_background_processes", 0) + if bg_proc_count and _ok("bg_processes"): + parts.append(f"⚙ {bg_proc_count}") + bg_subagent_count = snapshot.get("active_background_subagents", 0) + if bg_subagent_count and _ok("bg_subagents"): + parts.append(f"⛓ {bg_subagent_count}") + if goal_segment: + parts.append(goal_segment) + if _ok("duration"): + parts.append(duration_label) + prompt_elapsed = snapshot.get("prompt_elapsed") + if prompt_elapsed and _ok("prompt_elapsed"): + parts.append(prompt_elapsed) + idle_since = snapshot.get("idle_since") + if idle_since and _ok("idle_since"): + parts.append(idle_since) + if focus_label: + parts.append(focus_label) + if yolo_active and _ok("yolo"): + parts.append("⚠ YOLO") + # Session token total (Σ) — opt-in only via an explicit fields + # list, so default bars never widen. + total_tokens = snapshot.get("session_total_tokens", 0) + if total_tokens and field_set is not None and "total_tokens" in field_set: + parts.append(f"Σ{format_token_count_compact(total_tokens)}") + if not parts: + parts = [f"⚕ {snapshot['model_short']}"] + return self._right_align_status_title(" │ ".join(parts), session_title, width) + except Exception: + return f"⚕ {self.model if getattr(self, 'model', None) else 'Hermes'}" + + def _get_status_bar_fragments(self): + from cli import format_token_count_compact + if not self._status_bar_visible or getattr(self, '_model_picker_state', None) or getattr(self, '_command_palette_state', None): + return [] + try: + snapshot = self._get_status_bar_snapshot() + # Use prompt_toolkit's own terminal width when running inside the + # TUI — shutil.get_terminal_size() can return stale or fallback + # values (especially on SSH) that differ from what prompt_toolkit + # actually renders, causing the fragments to overflow to a second + # line and produce duplicated status bar rows over long sessions. + width = self._get_tui_terminal_width() + duration_label = snapshot["duration"] + yolo_active = self._is_session_yolo_active() + goal_segment = self._status_bar_goal_segment(snapshot) + battery_label = snapshot.get("battery_label") or "" + battery_style = self._battery_status_style(snapshot.get("battery_category", "dim")) + focus_label = snapshot.get("focus_label") or "" + session_title = snapshot.get("session_title") or "" + field_set = self._get_status_bar_field_set() + + def _ok(name: str) -> bool: + return field_set is None or name in field_set + + if not _ok("title"): + session_title = "" + + if not _ok("goal"): + goal_segment = "" + if not _ok("focus"): + focus_label = "" + + def _append(frag_list, sep, *pieces): + if frag_list: + frag_list.append(("class:status-bar-dim", sep)) + frag_list.extend(pieces) + + if width < 52: + frags = [] + if _ok("model"): + frags.append(("class:status-bar", " ⚕ ")) + frags.append(("class:status-bar-strong", snapshot["model_short"])) + if _ok("duration"): + _append(frags, " · ", ("class:status-bar-dim", duration_label)) + if goal_segment: + _append(frags, " · ", ("class:status-bar-strong", goal_segment)) + if focus_label: + _append(frags, " · ", ("class:status-bar-strong", focus_label)) + if yolo_active and _ok("yolo"): + _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) + if not frags: + frags = [ + ("class:status-bar", " ⚕ "), + ("class:status-bar-strong", snapshot["model_short"]), + ] + frags.append(("class:status-bar", " ")) + else: + percent = snapshot["context_percent"] + percent_label = f"{percent}%" if percent is not None else "--" + if width < 76: + compressions = snapshot.get("compressions", 0) + bg_count = snapshot.get("active_background_tasks", 0) + bg_proc_count = snapshot.get("active_background_processes", 0) + bg_subagent_count = snapshot.get("active_background_subagents", 0) + frags = [] + if _ok("model"): + frags.append(("class:status-bar", " ⚕ ")) + frags.append(("class:status-bar-strong", snapshot["model_short"])) + if _ok("context_pct"): + _append(frags, " · ", (self._status_bar_context_style(percent), percent_label)) + cache = self._cache_hit_rate(snapshot, precision=0) + if cache and _ok("cache_hit"): + _append(frags, " · ", (self._cache_hit_rate_style(cache[0]), cache[1])) + if compressions and _ok("compressions"): + _append(frags, " · ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) + if bg_count and _ok("bg_tasks"): + _append(frags, " · ", ("class:status-bar-strong", f"▶ {bg_count}")) + if bg_proc_count and _ok("bg_processes"): + _append(frags, " · ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) + if bg_subagent_count and _ok("bg_subagents"): + _append(frags, " · ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) + if goal_segment: + _append(frags, " · ", ("class:status-bar-strong", goal_segment)) + if _ok("duration"): + _append(frags, " · ", ("class:status-bar-dim", duration_label)) + if focus_label: + _append(frags, " · ", ("class:status-bar-strong", focus_label)) + if yolo_active and _ok("yolo"): + _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) + if not frags: + frags = [ + ("class:status-bar", " ⚕ "), + ("class:status-bar-strong", snapshot["model_short"]), + ] + frags.append(("class:status-bar", " ")) + else: + bar_style = self._status_bar_context_style(percent) + compressions = snapshot.get("compressions", 0) + bg_count = snapshot.get("active_background_tasks", 0) + bg_proc_count = snapshot.get("active_background_processes", 0) + bg_subagent_count = snapshot.get("active_background_subagents", 0) + frags = [] + if _ok("model"): + frags.append(("class:status-bar", " ⚕ ")) + frags.append(("class:status-bar-strong", snapshot["model_short"])) + if _ok("context_detail"): + if snapshot["context_length"]: + ctx_total = _format_context_length(snapshot["context_length"]) + ctx_used = format_token_count_compact(snapshot["context_tokens"]) + context_label = f"{ctx_used}/{ctx_total}" + else: + context_label = "ctx --" + _append(frags, " │ ", ("class:status-bar-dim", context_label)) + if _ok("context_pct"): + _append( + frags, + " │ ", + (bar_style, self._build_context_bar(percent)), + ("class:status-bar-dim", " "), + (bar_style, percent_label), + ) + cache = self._cache_hit_rate(snapshot) + if cache and _ok("cache_hit"): + _append(frags, " │ ", (self._cache_hit_rate_style(cache[0]), cache[1])) + _avg_lat = snapshot.get("avg_latency_label") or "" + if _avg_lat and _ok("latency"): + _append(frags, " │ ", ("class:status-bar-dim", f"◷ {_avg_lat}")) + _avg_vel = snapshot.get("avg_velocity_label") or "" + if _avg_vel and _ok("tps"): + _append(frags, " │ ", ("class:status-bar-dim", f"↑ {_avg_vel}")) + if compressions and _ok("compressions"): + _append(frags, " │ ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) + if bg_count and _ok("bg_tasks"): + _append(frags, " │ ", ("class:status-bar-strong", f"▶ {bg_count}")) + if bg_proc_count and _ok("bg_processes"): + _append(frags, " │ ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) + if bg_subagent_count and _ok("bg_subagents"): + _append(frags, " │ ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) + if goal_segment: + _append(frags, " │ ", ("class:status-bar-strong", goal_segment)) + if _ok("duration"): + _append(frags, " │ ", ("class:status-bar-dim", duration_label)) + # Position 7: per-prompt elapsed timer (live or frozen) + prompt_elapsed = snapshot.get("prompt_elapsed") + if prompt_elapsed and _ok("prompt_elapsed"): + _append(frags, " │ ", ("class:status-bar-dim", prompt_elapsed)) + # Position 8: idle time since the last final agent response + idle_since = snapshot.get("idle_since") + if idle_since and _ok("idle_since"): + _append(frags, " │ ", ("class:status-bar-dim", idle_since)) + # Persistent focus-view badge — so the reduced-output mode + # is never invisible (mirrors the YOLO badge convention). + if focus_label: + _append(frags, " │ ", ("class:status-bar-strong", focus_label)) + if yolo_active and _ok("yolo"): + _append(frags, " │ ", ("class:status-bar-yolo", "⚠ YOLO")) + # Session token total (Σ) — opt-in only via an explicit + # fields list, so default bars never widen. + total_tokens = snapshot.get("session_total_tokens", 0) + if total_tokens and field_set is not None and "total_tokens" in field_set: + _append(frags, " │ ", ("class:status-bar-dim", f"Σ{format_token_count_compact(total_tokens)}")) + if not frags: + frags = [ + ("class:status-bar", " ⚕ "), + ("class:status-bar-strong", snapshot["model_short"]), + ] + frags.append(("class:status-bar", " ")) + + # Stash indicator (📌 N) — appended after all width tiers so the + # user always knows a parked draft exists, even on narrow + # terminals. Placed before the battery prepend so it stays at the + # right edge, and it is the first thing the width trim below drops + # if the bar genuinely cannot fit. + try: + stash_indicator = self._prompt_stash.indicator() + except Exception: + stash_indicator = "" + if stash_indicator and _ok("stash"): + # Insert before the trailing pad fragment so the bar keeps its + # one-cell right margin. + if frags and frags[-1] == ("class:status-bar", " "): + frags[-1:-1] = [ + ("class:status-bar-dim", " · "), + ("class:status-bar-strong", stash_indicator), + ] + else: + frags.append(("class:status-bar-dim", " · ")) + frags.append(("class:status-bar-strong", stash_indicator)) + + # Battery is the first status-bar element when enabled: prepend it + # ahead of the leading ⚕ marker in whichever width tier ran above. + if battery_label and _ok("battery"): + frags[0:0] = [ + ("class:status-bar", " "), + (battery_style, battery_label), + ("class:status-bar-dim", " │"), + ] + + frags = self._right_align_status_title_fragments(frags, session_title, width) + + total_width = sum(self._status_bar_display_width(text) for _, text in frags) + if total_width > width: + plain_text = "".join(text for _, text in frags) + trimmed = self._trim_status_bar_text(plain_text, width) + return [("class:status-bar", trimmed)] + return frags + except Exception: + return [("class:status-bar", f" {self._build_status_bar_text()} ")] + + @staticmethod + def _fmt_stash_age(stashed_at: float) -> str: + """Return human-readable age string for a stash entry.""" + import time as _t + secs = int(_t.monotonic() - stashed_at) + if secs < 10: + return "just now" + if secs < 90: + return f"{secs}s ago" + mins = secs // 60 + if mins < 60: + return f"{mins} min ago" + return f"{mins // 60}h ago" + + def _render_stash_panel(self, stash_list: list, cursor: int, width: int) -> list: + """Return prompt_toolkit formatted_text fragments for the stash panel box. + + Every horizontal measurement goes through ``_status_bar_display_width`` + (prompt_toolkit's ``get_cwidth``) rather than ``len()``. The header + contains 📌, which is one Python codepoint but two terminal cells; the + original PR chased that off-by-one through three successive + "subtract 1 from len()" commits. Measuring in display cells fixes it + for real and keeps CJK previews from bleeding past the right border. + """ + cw = self._status_bar_display_width + W = max(12, min(width - 4, 80)) + + n = len(stash_list) + hdr_prefix_str = f"╭─ 📌 Stash ({n} item{'s' if n != 1 else ''}) " + HDR_SUFFIX = " Ctrl+S ─╮" + FTR_PREFIX = "╰" + FTR_SUFFIX = " ↑↓ Enter=restore D=delete Esc ─╯" + + # On narrow terminals the full hint text is wider than the box itself. + # Drop to compact affordances rather than letting the frame bleed past + # the right edge (which is what made the panel look broken). + if cw(hdr_prefix_str) + cw(HDR_SUFFIX) > W: + hdr_prefix_str = f"╭─ 📌 {n} " + HDR_SUFFIX = "─╮" + if cw(FTR_PREFIX) + cw(FTR_SUFFIX) > W: + FTR_SUFFIX = " ↑↓ ⏎ D Esc ─╯" + if cw(FTR_PREFIX) + cw(FTR_SUFFIX) > W: + FTR_SUFFIX = "─╯" + + hdr_dashes = max(0, W - cw(hdr_prefix_str) - cw(HDR_SUFFIX)) + ftr_dashes = max(0, W - cw(FTR_PREFIX) - cw(FTR_SUFFIX)) + + # Row inner width: W minus the two '│' border cells. + INNER = W - 2 + + frags: list = [] + + def line(text: str, style: str = "") -> None: + # Final guard: never emit a line wider than the box, whatever the + # label lengths worked out to. + frags.append((style, self._trim_status_bar_text(text, W) + "\n")) + + line(f"{hdr_prefix_str}{'─' * hdr_dashes}{HDR_SUFFIX}", "class:subagent-border") + + for i, item in enumerate(stash_list): + age = self._fmt_stash_age(item["stashed_at"]) + # Row: " ► [N] {age:<10} {preview} " + prefix = f" {'►' if i == cursor else ' '} [{i + 1}] {age:<10} " + if cw(prefix) > INNER - 2: + prefix = f" {'►' if i == cursor else ' '} [{i + 1}] " + avail = max(0, INNER - cw(prefix) - 1) + preview = self._trim_status_bar_text(item.get("preview") or "", avail) + preview = preview + " " * max(0, avail - cw(preview)) + row = self._trim_status_bar_text(f"│{prefix}{preview} │", W) + if i == cursor: + frags.append(("class:subagent-selected", row + "\n")) + else: + frags.append(("class:subagent-border", "│")) + frags.append(("class:subagent-sub", f"{prefix}{preview} ")) + frags.append(("class:subagent-border", "│\n")) + + line(f"{FTR_PREFIX}{'─' * ftr_dashes}{FTR_SUFFIX}", "class:subagent-border") + return frags diff --git a/hermes_cli/cli_stream_mixin.py b/hermes_cli/cli_stream_mixin.py new file mode 100644 index 0000000000..5714a1dcda --- /dev/null +++ b/hermes_cli/cli_stream_mixin.py @@ -0,0 +1,986 @@ +"""Streaming output, reasoning preview, tool progress callbacks, and busy-command spinner for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import json +import re +import shutil +import textwrap +import time + +from contextlib import contextmanager +from pathlib import Path +from rich.markup import escape as _escape + + +class CLIStreamMixin: + """Streaming output, reasoning preview, tool progress callbacks, and busy-command spinner for the interactive CLI""" + + def _on_thinking(self, text: str) -> None: + """Called by agent when thinking starts/stops. Updates TUI spinner.""" + if not text: + self._flush_reasoning_preview(force=True) + self._spinner_text = text or "" + self._tool_start_time = 0.0 # clear tool timer when switching to thinking + self._invalidate() + + def _on_notice(self, notice) -> None: + """Queue an out-of-band AgentNotice for rendering at the next clean boundary. + + Notices fire from inside the agent turn (cold-start seed during _init_agent, + per-turn _capture_credits after the API call) — printing immediately races the + streaming response and the line gets buried behind the prompt (see _cprint's + bg-thread caveat). So we QUEUE here and flush in _flush_credit_notices(), called + right after run_conversation returns. Fail-soft: never break the turn. + """ + try: + text = getattr(notice, "text", "") or "" + if not text: + return + level = getattr(notice, "level", "info") or "info" + if not hasattr(self, "_pending_credit_notices"): + self._pending_credit_notices = [] + self._pending_credit_notices.append((level, text)) + except Exception: + pass + + def _flush_credit_notices(self) -> None: + """Print any queued credit notices as level-colored lines. Called at turn end + (after run_conversation) where _cprint paints cleanly above the prompt.""" + from cli import _DIM, _RST, _cprint + try: + pending = getattr(self, "_pending_credit_notices", None) + if not pending: + return + self._pending_credit_notices = [] + for level, text in pending: + color = { + "error": "\033[31m", + "warn": "\033[33m", + "success": "\033[32m", + "info": _DIM, + }.get(level, _DIM) + _cprint(f" {color}{text}{_RST}") + except Exception: + pass + + def _on_notice_clear(self, key: str) -> None: + """Notice cleared. The REPL prints lines (no persistent slot to wipe), so + this drops any still-queued notice with that key is not tracked by key here; + it's a no-op for rendering — kept so the agent's clear callback is bound + symmetrically with the show callback (and so future REPL UIs can hook it).""" + return + + def _current_reasoning_callback(self): + """Return the active reasoning display callback for the current mode.""" + if self.show_reasoning and self.streaming_enabled: + return self._stream_reasoning_delta + if self.verbose and not self.show_reasoning: + return self._on_reasoning + return None + + def _emit_reasoning_preview(self, reasoning_text: str) -> None: + """Render a buffered reasoning preview as a single [thinking] block.""" + from cli import _DIM, _RST, _cprint + preview_text = reasoning_text.strip() + if not preview_text: + return + + try: + term_width = shutil.get_terminal_size().columns + except Exception: + term_width = 80 + prefix = " [thinking] " + wrap_width = max(30, term_width - len(prefix) - 2) + + paragraphs = [] + raw_paragraphs = re.split(r"\n\s*\n+", preview_text.replace("\r\n", "\n")) + for paragraph in raw_paragraphs: + compact = " ".join(line.strip() for line in paragraph.splitlines() if line.strip()) + if compact: + paragraphs.append(textwrap.fill(compact, width=wrap_width)) + preview_text = "\n".join(paragraphs) + if not preview_text: + return + + if self.verbose: + _cprint(f" {_DIM}[thinking] {preview_text}{_RST}") + return + + lines = preview_text.splitlines() + if len(lines) > 5: + preview = "\n".join(lines[:5]) + preview += f"\n ... ({len(lines) - 5} more lines)" + else: + preview = preview_text + _cprint(f" {_DIM}[thinking] {preview}{_RST}") + + def _flush_reasoning_preview(self, *, force: bool = False) -> None: + """Flush buffered reasoning text at natural boundaries. + + Some providers stream reasoning in tiny word or punctuation chunks. + Buffer them here so the preview path does not print one `[thinking]` + line per token. + """ + buf = getattr(self, "_reasoning_preview_buf", "") + if not buf: + return + + try: + term_width = shutil.get_terminal_size().columns + except Exception: + term_width = 80 + target_width = max(40, term_width - len(" [thinking] ") - 4) + + flush_text = "" + + if force: + flush_text = buf + buf = "" + else: + line_break = buf.rfind("\n") + min_newline_flush = max(16, target_width // 3) + if line_break != -1 and ( + line_break >= min_newline_flush + or buf.endswith("\n\n") + or buf.endswith(".\n") + or buf.endswith("!\n") + or buf.endswith("?\n") + or buf.endswith(":\n") + ): + flush_text = buf[: line_break + 1] + buf = buf[line_break + 1 :] + elif len(buf) >= target_width: + search_start = max(20, target_width // 2) + search_end = min(len(buf), max(target_width + (target_width // 3), target_width + 8)) + cut = -1 + for boundary in (" ", "\t", ".", "!", "?", ",", ";", ":"): + cut = max(cut, buf.rfind(boundary, search_start, search_end)) + if cut != -1: + flush_text = buf[: cut + 1] + buf = buf[cut + 1 :] + + self._reasoning_preview_buf = buf.lstrip() if flush_text else buf + if flush_text: + self._emit_reasoning_preview(flush_text) + + def _format_submitted_user_message_preview(self, user_input: str) -> str: + """Format the submitted user-message scrollback preview.""" + from cli import _accent_hex, datetime + ts_suffix = ( + f" [dim]{datetime.now().strftime(getattr(self, 'timestamp_format', '%H:%M'))}[/]" + if getattr(self, "show_timestamps", False) else "" + ) + lines = user_input.split("\n") + if len(lines) <= 1: + return f"[bold {_accent_hex()}]●[/] [bold]{_escape(user_input)}[/]{ts_suffix}" + + first_lines = int(getattr(self, "user_message_preview_first_lines", 2)) + last_lines = int(getattr(self, "user_message_preview_last_lines", 2)) + first_lines = max(1, first_lines) + last_lines = max(0, last_lines) + head = lines[:first_lines] + remaining_after_head = max(0, len(lines) - len(head)) + tail_count = min(last_lines, remaining_after_head) + tail = lines[-tail_count:] if tail_count else [] + + hidden_middle_count = len(lines) - len(head) - len(tail) + if hidden_middle_count < 0: + hidden_middle_count = 0 + tail = [] + + preview_lines = [ + f"[bold {_accent_hex()}]●[/] [bold]{_escape(head[0])}[/]{ts_suffix}" + ] + preview_lines.extend(f"[bold]{_escape(line)}[/]" for line in head[1:]) + + if hidden_middle_count > 0: + noun = "line" if hidden_middle_count == 1 else "lines" + preview_lines.append(f"[dim]... (+{hidden_middle_count} more {noun})[/]") + + preview_lines.extend(f"[bold]{_escape(line)}[/]" for line in tail) + return "\n".join(preview_lines) + + def _expand_paste_references(self, text: str | None) -> str: + """Expand [Pasted text #N -> file] placeholders into file contents.""" + from cli import logger + if not isinstance(text, str) or "[Pasted text #" not in text: + return text or "" + paste_ref_re = re.compile(r'\[Pasted text #\d+: \d+ lines \u2192 (.+?)\]') + + def _expand_ref(match): + path = Path(match.group(1)) + # Use try/except instead of path.exists() to avoid TOCTOU race: + # the paste file may be deleted between check and read, causing + # the input to be silently dropped (#17666). + try: + return path.read_text(encoding="utf-8") + except (OSError, IOError): + logger.warning("Paste file gone or unreadable, returning placeholder: %s", path) + return match.group(0) + + return paste_ref_re.sub(_expand_ref, text) + + def _print_user_message_preview(self, user_input: str) -> None: + """Render a user message using the normal chat scrollback style.""" + from cli import ChatConsole, _accent_hex + ChatConsole().print(f"[{_accent_hex()}]{'─' * 40}[/]") + text = str(user_input or "") + if "\n" in text: + ChatConsole().print(self._format_submitted_user_message_preview(text)) + else: + ChatConsole().print(f"[bold {_accent_hex()}]●[/] [bold]{_escape(text)}[/]") + + def _stream_reasoning_delta(self, text: str) -> None: + """Stream reasoning/thinking tokens into a dim box above the response. + + Opens a dim reasoning box on first token, streams line-by-line. + The box is closed automatically when content tokens start arriving + (via _stream_delta → _emit_stream_text). + + Once the response box is open, suppress any further reasoning + rendering — a late thinking block (e.g. after an interrupt) would + otherwise draw a reasoning box inside the response box. + """ + from cli import _DIM, _RST, _cprint + if not text: + return + self._reasoning_shown_this_turn = True + if getattr(self, "_stream_box_opened", False): + return + + # Open reasoning box on first reasoning token + if not getattr(self, "_reasoning_box_opened", False): + self._reasoning_box_opened = True + w = self._scrollback_box_width() + r_label = " Reasoning " + r_fill = w - 2 - len(r_label) + _cprint(f"\n{_DIM}┌─{r_label}{'─' * max(r_fill - 1, 0)}┐{_RST}") + + self._reasoning_buf = getattr(self, "_reasoning_buf", "") + text + + # Emit complete lines, and force-flush long partial lines so + # reasoning is visible in real-time even without newlines. + while "\n" in self._reasoning_buf: + line, self._reasoning_buf = self._reasoning_buf.split("\n", 1) + _cprint(f"{_DIM}{line}{_RST}") + if len(self._reasoning_buf) > 80: + _cprint(f"{_DIM}{self._reasoning_buf}{_RST}") + self._reasoning_buf = "" + + def _close_reasoning_box(self) -> None: + """Close the live reasoning box if it's open.""" + from cli import _DIM, _RST, _cprint + if getattr(self, "_reasoning_box_opened", False): + # Flush remaining reasoning buffer + buf = getattr(self, "_reasoning_buf", "") + if buf: + _cprint(f"{_DIM}{buf}{_RST}") + self._reasoning_buf = "" + w = self._scrollback_box_width() + _cprint(f"{_DIM}└{'─' * (w - 2)}┘{_RST}") + self._reasoning_box_opened = False + + # Flush any content that was deferred while reasoning was rendering. + deferred = getattr(self, "_deferred_content", "") + if deferred: + self._deferred_content = "" + self._emit_stream_text(deferred) + + def _stream_delta(self, text) -> None: + """Line-buffered streaming callback for real-time token rendering. + + Receives text deltas from the agent as tokens arrive. Buffers + partial lines and emits complete lines via _cprint to work + reliably with prompt_toolkit's patch_stdout. + + Reasoning/thinking blocks (, , etc.) + are suppressed during streaming since they'd display raw XML tags. + The agent strips them from the final response anyway. + + A ``None`` value signals an intermediate turn boundary (tools are + about to execute). Flushes any open boxes and resets state so + tool feed lines render cleanly between turns. + """ + if text is None: + self._flush_stream() + self._reset_stream_state() + return + if not text: + return + + self._stream_started = True + + # ── Tag-based reasoning suppression ── + # Track whether we're inside a reasoning/thinking block. + # These tags are model-generated (system prompt tells the model + # to use them) and get stripped from final_response. We must + # suppress them during streaming too — unless show_reasoning is + # enabled, in which case we route the inner content to the + # reasoning display box instead of discarding it. + _OPEN_TAGS = ("", "", "", "", "", "") + _CLOSE_TAGS = ("", "", "", "", "", "") + + # Append to a pre-filter buffer first + self._stream_prefilt = getattr(self, "_stream_prefilt", "") + text + + # Check if we're entering a reasoning block. + # Only match tags that appear at a "block boundary": start of the + # stream, after a newline (with optional whitespace), or when nothing + # but whitespace has been emitted on the current line. + # This prevents false positives when models *mention* tags in prose + # like "(/think not producing tags)". + # + # _stream_last_was_newline tracks whether the last character emitted + # (or the start of the stream) is a line boundary. It's True at + # stream start and set True whenever emitted text ends with '\n'. + if not hasattr(self, "_stream_last_was_newline"): + self._stream_last_was_newline = True # start of stream = boundary + + if not getattr(self, "_in_reasoning_block", False): + # Case-insensitive matching against a lowercased view so + # mixed-case tag variants (, , …) are caught. + prefilt_lower = self._stream_prefilt.lower() + for tag in _OPEN_TAGS: + tag_lower = tag.lower() + search_start = 0 + while True: + idx = prefilt_lower.find(tag_lower, search_start) + if idx == -1: + break + # Check if this is a block boundary position + preceding = self._stream_prefilt[:idx] + if idx == 0: + # At buffer start — only a boundary if we're at + # a line start (stream start or last emit ended + # with newline) + is_block_boundary = getattr(self, "_stream_last_was_newline", True) + else: + # Find last newline in the buffer before the tag + last_nl = preceding.rfind("\n") + if last_nl == -1: + # No newline in buffer — boundary only if + # last emit was a newline AND only whitespace + # has accumulated before the tag + is_block_boundary = ( + getattr(self, "_stream_last_was_newline", True) + and preceding.strip() == "" + ) + else: + # Text between last newline and tag must be + # whitespace-only + is_block_boundary = preceding[last_nl + 1:].strip() == "" + if is_block_boundary: + # Emit everything before the tag + if preceding: + self._emit_stream_text(preceding) + self._stream_last_was_newline = preceding.endswith("\n") + self._in_reasoning_block = True + self._stream_prefilt = self._stream_prefilt[idx + len(tag):] + break + # Not a block boundary — keep searching after this occurrence + search_start = idx + 1 + if getattr(self, "_in_reasoning_block", False): + break + + # Could also be a partial open tag at the end — hold it back + if not getattr(self, "_in_reasoning_block", False): + # Check for partial tag match at the end (case-insensitive) + safe = self._stream_prefilt + for tag in _OPEN_TAGS: + tag_lower = tag.lower() + for i in range(1, len(tag)): + if prefilt_lower.endswith(tag_lower[:i]): + safe = self._stream_prefilt[:-i] + break + if safe: + self._emit_stream_text(safe) + self._stream_last_was_newline = safe.endswith("\n") + self._stream_prefilt = self._stream_prefilt[len(safe):] + return + + # Inside a reasoning block — look for close tag. + # Keep accumulating _stream_prefilt because close tags can arrive + # split across multiple tokens (e.g. "..."). + if getattr(self, "_in_reasoning_block", False): + prefilt_lower = self._stream_prefilt.lower() + for tag in _CLOSE_TAGS: + idx = prefilt_lower.find(tag.lower()) + if idx != -1: + self._in_reasoning_block = False + # When show_reasoning is on, route inner content to + # the reasoning display box instead of discarding. + if self.show_reasoning: + inner = self._stream_prefilt[:idx] + if inner: + self._stream_reasoning_delta(inner) + after = self._stream_prefilt[idx + len(tag):] + self._stream_prefilt = "" + # Process remaining text after close tag through full + # filtering (it could contain another open tag) + if after: + self._stream_delta(after) + return + # When show_reasoning is on, stream reasoning content live + # instead of silently accumulating. Keep only the tail that + # could be a partial close tag prefix. + max_tag_len = max(len(t) for t in _CLOSE_TAGS) + if len(self._stream_prefilt) > max_tag_len: + if self.show_reasoning: + # Route the safe prefix to reasoning display + safe_reasoning = self._stream_prefilt[:-max_tag_len] + self._stream_reasoning_delta(safe_reasoning) + self._stream_prefilt = self._stream_prefilt[-max_tag_len:] + return + + def _emit_stream_text(self, text: str) -> None: + """Emit filtered text to the streaming display.""" + from cli import ( + HermesCLI, + _ACCENT, + _RST, + _STREAM_PAD, + _STREAM_PARTIAL_PREVIEW_LEN, + _cprint, + _strip_markdown_syntax, + _terminal_width_for_streaming, + datetime, + is_table_divider, + looks_like_table_row, + realign_markdown_tables, + ) + if not text: + return + + # When show_reasoning is on and reasoning is still rendering, + # defer content until the reasoning box closes. This ensures the + # reasoning block always appears BEFORE the response in the terminal. + if self.show_reasoning and getattr(self, "_reasoning_box_opened", False): + self._deferred_content = getattr(self, "_deferred_content", "") + text + return + + # Close the live reasoning box before opening the response box + self._close_reasoning_box() + + # Open the response box header on the very first visible text + if not self._stream_box_opened: + # Strip leading whitespace/newlines before first visible content + text = text.lstrip("\n") + if not text: + return + self._stream_box_opened = True + try: + from hermes_cli.skin_engine import get_active_skin + _skin = get_active_skin() + label = _skin.get_branding("response_label", "⚕ Hermes") + _text_hex = _skin.get_color("banner_text", "#FFF8DC") + except Exception: + label = "⚕ Hermes" + _text_hex = "#FFF8DC" + # Build a true-color ANSI escape for the response text color + # so streamed content matches the Rich Panel appearance. + try: + _r = int(_text_hex[1:3], 16) + _g = int(_text_hex[3:5], 16) + _b = int(_text_hex[5:7], 16) + self._stream_text_ansi = f"\033[38;2;{_r};{_g};{_b}m" + except (ValueError, IndexError): + self._stream_text_ansi = "" + if self.show_timestamps: + label = f"{label} {datetime.now().strftime(getattr(self, 'timestamp_format', '%H:%M'))}" + w = self._scrollback_box_width() + fill = w - 2 - HermesCLI._status_bar_display_width(label) + _cprint(f"\n{_ACCENT}╭─{label}{'─' * max(fill - 1, 0)}╮{_RST}") + + self._stream_buf += text + + # Emit complete lines, keep partial remainder in buffer + _tc = getattr(self, "_stream_text_ansi", "") + + def _emit_one(printed_line: str) -> None: + _cprint(f"{_STREAM_PAD}{_tc}{printed_line}{_RST}" if _tc else f"{_STREAM_PAD}{printed_line}") + + def _flush_table_buf() -> None: + buf = self._stream_table_buf + self._stream_table_buf = [] + self._in_stream_table = False + if not buf: + return + # Strip cell-level markdown (`code`, **bold**, ~~strike~~) FIRST + # so the realigner pads to the final visible cell width, not + # the marker-decorated source width. Otherwise a body row + # like `` | Bold | `**bold**` | `` lands narrower than its + # header column once the markers are removed. + joined = "\n".join(buf) + if self.final_response_markdown == "strip": + joined = _strip_markdown_syntax(joined) + block = realign_markdown_tables(joined, _terminal_width_for_streaming()) + for ln in block.split("\n"): + _emit_one(ln) + + while "\n" in self._stream_buf: + line, self._stream_buf = self._stream_buf.split("\n", 1) + + # Hold table-shaped lines in a side-buffer so we can re-pad + # the whole block once it ends. Streaming line-by-line, we + # cannot re-align mid-table without reflowing already-printed + # rows; the cost is that the user sees the table appear in a + # single batch when the block closes instead of row-by-row. + if self._in_stream_table: + if looks_like_table_row(line) or is_table_divider(line): + self._stream_table_buf.append(line) + continue + # Block ended — flush the realigned table, then fall + # through to print the current (non-table) line. + _flush_table_buf() + elif looks_like_table_row(line): + self._stream_table_buf.append(line) + self._in_stream_table = True + continue + + if self.final_response_markdown == "strip": + line = _strip_markdown_syntax(line) + _emit_one(line) + + # Long partial lines are emitted ONLY at real newlines — we no + # longer hard-wrap paragraphs at terminal width ourselves. Each + # logical line lands in scrollback as one line; the TERMINAL + # soft-wraps it visually, and emulators (iTerm2/kitty/VTE/ + # xterm.js/Windows Terminal) rejoin soft-wrapped rows on copy, + # so highlight-copy yields the original unwrapped text — same + # outcome as the TUI's selection copy. (The pre-July-2026 chunk + # emitter baked real '\n's into every long paragraph, which is + # exactly what polluted copy/paste.) + # + # TTFT perception: while a long opening paragraph accumulates + # without a newline, mirror its tail into the status-bar spinner + # line so the user sees tokens arriving instead of a blank box. + if ( + self._stream_buf + and not self._in_stream_table + and not self._stream_buf.lstrip().startswith("|") + and len(self._stream_buf) >= 80 + ): + preview = self._stream_buf[-int(_STREAM_PARTIAL_PREVIEW_LEN):] + cut = preview.find(" ") + if 0 < cut < len(preview) - 1: + preview = preview[cut + 1:] + try: + self._spinner_text = f"… {preview}" + self._invalidate() + except Exception: + pass + + def _flush_stream(self) -> None: + """Emit any remaining partial line from the stream buffer and close the box.""" + from cli import ( + _ACCENT, + _RST, + _STREAM_PAD, + _cprint, + _strip_markdown_syntax, + _terminal_width_for_streaming, + is_table_divider, + looks_like_table_row, + realign_markdown_tables, + ) + # If we're still inside a "reasoning block" at end-of-stream, it was + # a false positive — the model mentioned a tag like in prose + # but never closed it. Recover the buffered content as regular text. + if getattr(self, "_in_reasoning_block", False) and getattr(self, "_stream_prefilt", ""): + self._in_reasoning_block = False + self._emit_stream_text(self._stream_prefilt) + self._stream_prefilt = "" + + # Close reasoning box if still open (in case no content tokens arrived) + self._close_reasoning_box() + + _tc = getattr(self, "_stream_text_ansi", "") + + # If the stream buffer has a trailing partial line that looks like + # a table row, fold it into the table buffer so the whole block + # gets re-aligned together. Otherwise the final row prints raw + # (with the model's original under-padded spacing) while the rows + # above it are aligned. + if ( + self._stream_buf + and getattr(self, "_in_stream_table", False) + and (looks_like_table_row(self._stream_buf) or is_table_divider(self._stream_buf)) + ): + self._stream_table_buf.append(self._stream_buf) + self._stream_buf = "" + + # Flush any buffered table rows first so their padding is + # finalised before the stream remainder lands. + if getattr(self, "_stream_table_buf", None): + joined = "\n".join(self._stream_table_buf) + self._stream_table_buf = [] + self._in_stream_table = False + if self.final_response_markdown == "strip": + joined = _strip_markdown_syntax(joined) + block = realign_markdown_tables(joined, _terminal_width_for_streaming()) + for ln in block.split("\n"): + _cprint(f"{_STREAM_PAD}{_tc}{ln}{_RST}" if _tc else f"{_STREAM_PAD}{ln}") + + if self._stream_buf: + line = _strip_markdown_syntax(self._stream_buf) if self.final_response_markdown == "strip" else self._stream_buf + _cprint(f"{_STREAM_PAD}{_tc}{line}{_RST}" if _tc else f"{_STREAM_PAD}{line}") + self._stream_buf = "" + + # Close the response box + if self._stream_box_opened: + w = self._scrollback_box_width() + _cprint(f"{_ACCENT}╰{'─' * (w - 2)}╯{_RST}") + + def _reset_stream_state(self) -> None: + """Reset streaming state before each agent invocation.""" + self._stream_buf = "" + self._stream_started = False + self._stream_box_opened = False + self._stream_text_ansi = "" + self._stream_prefilt = "" + self._in_reasoning_block = False + self._stream_last_was_newline = True + self._reasoning_box_opened = False + self._reasoning_buf = "" + self._reasoning_preview_buf = "" + self._deferred_content = "" + self._stream_table_buf = [] + self._in_stream_table = False + + def _slow_command_status(self, command: str) -> str: + """Return a user-facing status message for slower slash commands.""" + cmd_lower = command.lower().strip() + if cmd_lower.startswith("/skills search"): + return "Searching skills..." + if cmd_lower.startswith("/skills browse"): + return "Loading skills..." + if cmd_lower.startswith("/skills inspect"): + return "Inspecting skill..." + if cmd_lower.startswith("/skills install"): + return "Installing skill..." + if cmd_lower.startswith("/skills"): + return "Processing skills command..." + if cmd_lower == "/reload-mcp": + return "Reloading MCP servers..." + if cmd_lower == "/reload-skills" or cmd_lower == "/reload_skills": + return "Reloading skills..." + if cmd_lower.startswith("/browser"): + return "Configuring browser..." + return "Processing command..." + + def _command_spinner_frame(self) -> str: + """Return the current spinner frame for slow slash commands.""" + from cli import _COMMAND_SPINNER_FRAMES + frame_idx = int(time.monotonic() * 10) % len(_COMMAND_SPINNER_FRAMES) + return _COMMAND_SPINNER_FRAMES[frame_idx] + + @contextmanager + def _busy_command(self, status: str, *, blocks_input: bool = True): + """Expose a temporary busy state in the TUI while a slash command runs. + + Most synchronous slash commands must reserve the composer because their + completion changes the active session state. Manual compression is safe + to draft through: the queued input is processed against the compacted + history after the command completes. + """ + previous_blocks_input = getattr(self, "_command_blocks_input", False) + self._command_running = True + self._command_blocks_input = blocks_input + self._command_status = status + self._invalidate(min_interval=0.0) + try: + print(f"⏳ {status}") + yield + finally: + self._command_running = False + self._command_blocks_input = previous_blocks_input + self._command_status = "" + self._invalidate(min_interval=0.0) + + def _preprocess_images_with_vision(self, text: str, images: list, *, announce: bool = True) -> str: + """Analyze attached images via the vision tool and return enriched text. + + Instead of embedding raw base64 ``image_url`` content parts in the + conversation (which only works with vision-capable models), this + pre-processes each image through the auxiliary vision model (Gemini + Flash) and prepends the descriptions to the user's message — the + same approach the messaging gateway uses. + + The local file path is included so the agent can re-examine the + image later with ``vision_analyze`` if needed. + """ + from cli import _DIM, _RST, _cprint + import asyncio as _asyncio + from tools.vision_tools import vision_analyze_tool + + analysis_prompt = ( + "Describe everything visible in this image in thorough detail. " + "Include any text, code, data, objects, people, layout, colors, " + "and any other notable visual information." + ) + + enriched_parts = [] + for img_path in images: + if not img_path.exists(): + continue + size_kb = img_path.stat().st_size // 1024 + if announce: + _cprint(f" {_DIM}👁️ analyzing {img_path.name} ({size_kb}KB)...{_RST}") + try: + result_json = _asyncio.run( + vision_analyze_tool(image_url=str(img_path), user_prompt=analysis_prompt) + ) + result = json.loads(result_json) + if result.get("success"): + description = result.get("analysis", "") + enriched_parts.append( + f"[The user attached an image. Here's what it contains:\n{description}]\n" + f"[If you need a closer look, use vision_analyze with " + f"image_url: {img_path}]" + ) + if announce: + _cprint(f" {_DIM}✓ image analyzed{_RST}") + else: + enriched_parts.append( + f"[The user attached an image but it couldn't be analyzed. " + f"You can try examining it with vision_analyze using " + f"image_url: {img_path}]" + ) + if announce: + _cprint(f" {_DIM}⚠ vision analysis failed — path included for retry{_RST}") + except Exception as e: + enriched_parts.append( + f"[The user attached an image but analysis failed ({e}). " + f"You can try examining it with vision_analyze using " + f"image_url: {img_path}]" + ) + if announce: + _cprint(f" {_DIM}⚠ vision analysis error — path included for retry{_RST}") + + # Combine: vision descriptions first, then the user's original text + user_text = text if isinstance(text, str) and text else "" + if enriched_parts: + prefix = "\n\n".join(enriched_parts) + return f"{prefix}\n\n{user_text}" if user_text else prefix + return user_text or "What do you see in this image?" + + def _output_console(self): + """Use prompt_toolkit-safe Rich rendering once the TUI is live.""" + from cli import ChatConsole + if getattr(self, "_app", None): + return ChatConsole() + return self.console + + def _console_print(self, *args, **kwargs): + """Print through the active command-safe console.""" + self._output_console().print(*args, **kwargs) + + def _on_tool_gen_start(self, tool_name: str) -> None: + """Called when the model begins generating tool-call arguments. + + Closes any open streaming boxes (reasoning / response) exactly once, + then prints a short status line so the user sees activity instead of + a frozen screen while a large payload (e.g. 45 KB write_file) streams. + """ + from cli import _cprint + if getattr(self, "_stream_box_opened", False): + self._flush_stream() + self._stream_box_opened = False + self._close_reasoning_box() + + from agent.display import get_tool_emoji + emoji = get_tool_emoji(tool_name, default="⚡") + _cprint(f" ┊ {emoji} preparing {tool_name}…") + + def _on_tool_progress(self, event_type: str, function_name: str = None, preview: str = None, function_args: dict = None, **kwargs): + """Called on tool lifecycle events (tool.started, tool.completed, reasoning.available, etc.). + + Updates the TUI spinner widget so the user can see what the agent + is doing during tool execution (fills the gap between thinking + spinner and next response). + + On tool.started, records a monotonic timestamp so get_spinner_text() + can show a live elapsed timer (the TUI poll loop already invalidates + every ~0.15s, so the counter updates automatically). + + When tool_progress_mode is "all" or "new", also prints a persistent + stacked line to scrollback on tool.completed so users can see the + full history of tool calls (not just the current one in the spinner). + """ + from cli import CLI_CONFIG, _DIM, _RST, _cprint, _hermes_home + # MoA reference-model outputs: render each reference's answer as a + # labelled thinking-style block BEFORE the aggregator acts, so the user + # sees the mixture-of-agents process instead of a silent pause. These + # are display-only events emitted by the MoA facade (agent_init relay); + # they never enter message history. + if event_type == "moa.reference": + label = function_name or "reference" + text = preview or "" + idx = kwargs.get("moa_index") + count = kwargs.get("moa_count") + header = f"Reference {idx}/{count} — {label}" if idx and count else f"Reference — {label}" + try: + self._flush_reasoning_preview(force=True) + except Exception: + pass + _cprint(f" {_DIM}┊ ◇ {header}{_RST}") + try: + self._emit_reasoning_preview(text) + except Exception: + # Fallback: print the raw text dimmed if the preview helper fails. + if text.strip(): + _cprint(f" {_DIM}{text.strip()}{_RST}") + self._invalidate() + return + if event_type == "moa.aggregating": + agg = function_name or "" + self._spinner_text = f"◆ aggregating ({agg})" if agg else "◆ aggregating" + self._invalidate() + return + + # Feed the pet: tools mean "running" (not reasoning); a failed tool + # latches the turn so it ends on a sulk. + if event_type == "tool.started": + self._pet_reasoning = False + elif event_type == "tool.completed" and kwargs.get("is_error"): + self._pet_turn_error = True + elif event_type and event_type.startswith("reasoning"): + self._pet_reasoning = True + + if event_type == "tool.completed": + self._tool_start_time = 0.0 + # Per-turn accounting: this feed already sees every tool call with + # its result, so the summary line needs no agent-loop state. + self._turn_summary_record( + function_name, kwargs.get("result"), kwargs.get("is_error", False) + ) + # Focus view: count the scrollback line we are NOT printing, so the + # post-turn recovery line can report how much was hidden. Counted + # against the pre-focus tool-progress mode, so a user who already + # had /verbose off is never told focus hid something it didn't. + if getattr(self, "_focus_view_enabled", False): + try: + self._note_focus_hidden_line(function_name or "") + except Exception: + pass + # Print stacked scrollback line for "new" / "all" / "verbose" modes. + # "verbose" was previously omitted here, so non-streaming model + # calls (MoA aggregator, copilot-acp) rendered each tool only into + # the transient spinner line — which overwrites itself, so no + # scrollable tool history accumulated. Streaming models hid the bug + # because _on_tool_gen_start commits a "preparing" line per tool; + # non-streaming calls never emit that, leaving verbose mode with no + # committed line at all. "verbose" is strictly more than "all", so + # it must commit at least the same line. + if function_name and self.tool_progress_mode in {"new", "all", "verbose"}: + duration = kwargs.get("duration", 0.0) + # Pop stored args from tool.started for this function + stored = self._pending_tool_info.get(function_name) + stored_args = stored.pop(0) if stored else {} + if stored is not None and not stored: + del self._pending_tool_info[function_name] + # "new" mode: skip consecutive repeats of the same tool + if self.tool_progress_mode == "new" and function_name == self._last_scrollback_tool: + self._invalidate() + return + self._last_scrollback_tool = function_name + try: + from agent.display import get_cute_tool_message + line = get_cute_tool_message(function_name, stored_args, duration, result=kwargs.get("result")) + _cprint(f" {line}") + except Exception: + pass + # First-touch onboarding: on the first tool in this process + # that takes longer than the threshold while we're in the + # noisiest progress mode, print a one-time hint about + # /verbose. Latched on self so it fires at most once per + # process; persisted to config.yaml so it never fires again + # across processes either. + try: + if ( + not getattr(self, "_long_tool_hint_fired", False) + and self.tool_progress_mode == "all" + and duration >= 30.0 + ): + from agent.onboarding import ( + TOOL_PROGRESS_FLAG, + is_seen, + mark_seen, + tool_progress_hint_cli, + ) + if not is_seen(CLI_CONFIG, TOOL_PROGRESS_FLAG): + self._long_tool_hint_fired = True + _cprint(f" {_DIM}{tool_progress_hint_cli()}{_RST}") + mark_seen(_hermes_home / "config.yaml", TOOL_PROGRESS_FLAG) + CLI_CONFIG.setdefault("onboarding", {}).setdefault("seen", {})[TOOL_PROGRESS_FLAG] = True + except Exception: + pass + self._invalidate() + return + if event_type != "tool.started": + return + if function_name and not function_name.startswith("_"): + from agent.display import get_tool_emoji + emoji = get_tool_emoji(function_name) + label = preview or function_name + from agent.display import get_tool_preview_max_len + _pl = get_tool_preview_max_len() + if _pl > 0 and len(label) > _pl: + label = label[:_pl - 3] + "..." + self._spinner_text = f"{emoji} {label}" + self._tool_start_time = time.monotonic() + # Store args for stacked scrollback line on completion + self._pending_tool_info.setdefault(function_name, []).append( + function_args if function_args is not None else {} + ) + self._invalidate() + + def _on_tool_start(self, tool_call_id: str, function_name: str, function_args: dict): + """Capture local before-state for write-capable tools.""" + from cli import logger + try: + from agent.display import capture_local_edit_snapshot + + snapshot = capture_local_edit_snapshot(function_name, function_args) + if snapshot is not None: + self._pending_edit_snapshots[tool_call_id] = snapshot + except Exception: + logger.debug("Edit snapshot capture failed for %s", function_name, exc_info=True) + + def _on_tool_complete(self, tool_call_id: str, function_name: str, function_args: dict, function_result: str): + """Render file edits with inline diff after write-capable tools complete.""" + from cli import _cprint, logger + # A top-level delegate_task dispatches in the background and re-enters as + # a fresh turn when done. Say so once — no spinner, nothing to poll — so + # the idle prompt doesn't read as "nothing happened" (⛓ tracks the work). + if function_name == "delegate_task": + try: + parsed = json.loads(function_result) if isinstance(function_result, str) else (function_result or {}) + except Exception: + parsed = {} + if isinstance(parsed, dict) and parsed.get("status") == "dispatched" and parsed.get("mode") == "background": + n = parsed.get("count") or 1 + noun, tail = ("task", "it finishes") if n == 1 else (f"{n} tasks", "they finish") + try: + _cprint(f"\033[2m\u21a9 Background {noun} running — I'll resume when {tail}. Keep chatting.\033[0m") + except Exception: + pass + snapshot = self._pending_edit_snapshots.pop(tool_call_id, None) + try: + from agent.display import render_edit_diff_with_delta + + render_edit_diff_with_delta( + function_name, + function_result, + function_args=function_args, + snapshot=snapshot, + print_fn=_cprint, + ) + except Exception: + logger.debug("Edit diff preview failed for %s", function_name, exc_info=True) diff --git a/hermes_cli/cli_terminal_mixin.py b/hermes_cli/cli_terminal_mixin.py new file mode 100644 index 0000000000..489cf77b9e --- /dev/null +++ b/hermes_cli/cli_terminal_mixin.py @@ -0,0 +1,624 @@ +"""Terminal repaint/resize recovery, input-mode healing, and clipboard helpers for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import base64 +import errno +import os +import shutil +import sys +import threading +import time + +from hermes_constants import get_hermes_home + + +class CLITerminalMixin: + """Terminal repaint/resize recovery, input-mode healing, and clipboard helpers for the interactive CLI""" + + def _mark_terminal_io_broken(self, reason: str = "") -> None: + """Stop UI paints after the PTY/stdout becomes unusable (#81521).""" + from cli import logger + if getattr(self, "_terminal_io_broken", False): + return + self._terminal_io_broken = True + try: + self._pet_stop_anim() + except Exception: + pass + logger.warning( + "Terminal I/O broken%s — freezing UI paints to avoid redraw storm (#81521)", + f" ({reason})" if reason else "", + ) + + def _invalidate(self, min_interval: float = 0.25) -> None: + """Throttled UI repaint for high-frequency background updates. + + Use this for spinner frames, streaming token flushes, and other + repaints that can fire many times per second — the throttle prevents + terminal blinking on slow/SSH connections, and the resize-recovery + guard avoids stamping footer/status-bar chrome into scrollback while a + SIGWINCH reflow is in flight. + + Do NOT use this for user-blocking modal prompts (approval / clarify / + sudo). Those are rare, one-shot, user-blocking events that must paint + immediately; route them through ``self._app.invalidate()`` directly, the + same way the modal key-binding handlers already do. Sending a modal's + entry paint through this throttle lets an unrelated background repaint + within the 250ms window — or an in-flight resize — silently drop it, so + the prompt never renders and times out unseen (#41098). + """ + if getattr(self, "_terminal_io_broken", False): + return + if getattr(self, "_resize_recovery_pending", False): + return + now = time.monotonic() + if hasattr(self, "_app") and self._app and (now - getattr(self, "_last_invalidate", 0.0)) >= min_interval: + self._last_invalidate = now + try: + self._app.invalidate() + except OSError as exc: + if getattr(exc, "errno", None) == errno.EIO: + self._mark_terminal_io_broken("invalidate") + return + raise + + def _paint_now(self) -> None: + """Immediate, unthrottled repaint for user-blocking modal prompts. + + Background-thread callbacks (approval / clarify / sudo) set their modal + state then call this to make the panel visible at once. It deliberately + bypasses the ``_invalidate`` throttle and resize-recovery guard — a + modal the user is actively waiting on must never be dropped — mirroring + the direct ``event.app.invalidate()`` the modal key-binding handlers + already use. See ``_invalidate`` for why the throttle must not gate + these paints (#41098). + """ + if getattr(self, "_terminal_io_broken", False): + return + app = getattr(self, "_app", None) + if app is not None: + try: + app.invalidate() + except OSError as exc: + if getattr(exc, "errno", None) == errno.EIO: + self._mark_terminal_io_broken("paint_now") + return + raise + except Exception: + pass + + def _force_full_redraw(self) -> None: + """Force a clean full-screen repaint of the prompt_toolkit UI. + + Used to recover from terminal buffer drift caused by external + redraws we can't detect — e.g. macOS cmux / tmux tab switches, + ``clear`` issued from a subshell, or SSH window restores. These + wipe or repaint the terminal without firing SIGWINCH, so + prompt_toolkit's tracked ``_cursor_pos`` no longer matches reality + and the next incremental redraw stacks on top of stale content + (ghost status bars, duplicated prompts). + + Bound to Ctrl+L and exposed as the ``/redraw`` slash command, + matching the standard terminal-UX convention (bash, zsh, fish, + vim, htop). + """ + from cli import _replay_output_history + if getattr(self, "_terminal_io_broken", False): + return + app = getattr(self, "_app", None) + if not app: + return + self._clear_prompt_toolkit_screen( + app, + rebuild_scrollback=self._redraw_rebuilds_scrollback(), + ) + if getattr(self, "_terminal_io_broken", False): + return + _replay_output_history() + self._pet_queue_kitty_frame() + try: + app.invalidate() + except OSError as exc: + if getattr(exc, "errno", None) == errno.EIO: + self._mark_terminal_io_broken("force_full_redraw") + return + raise + except Exception: + pass + + def _schedule_focus_regain_redraw(self, min_interval: float = 1.0) -> None: + """Repaint after a terminal focus-in report (``CSI I``), rate-limited. + + Terminals with focus tracking active (Ghostty, iTerm2, xterm builds, + multiplexers that toggle DECSET 1004 upstream) emit ``\\x1b[I`` when + the Hermes tab/window becomes visible again. Emulators can coalesce + or drop hidden-tab output and repaint the surface while we're + invisible, so on regain prompt_toolkit's incremental diff stacks on + stale content — a second copy of the composer/prompt chrome next to + the ghost of the old one (#60920 focus-regain variant, #25337). + + The stock handling maps ``CSI I``/``CSI O`` to ``Keys.Ignore`` so the + bytes never pollute the input buffer; this hook additionally routes + focus-in through the same recovery as Ctrl+L / ``/redraw``. It is + self-gating: terminals that never enable focus tracking never emit + the sequence, so nothing changes for them. Rate-limited so a burst of + focus reports (rapid Alt+Tab, mux pane hops) repaints at most once + per ``min_interval`` seconds. + """ + now = time.monotonic() + last = getattr(self, "_last_focus_regain_redraw", 0.0) + if now - last < min_interval: + return + self._last_focus_regain_redraw = now + self._force_full_redraw() + + @staticmethod + def _redraw_rebuilds_scrollback() -> bool: + """Return whether CLI redraw/resize recovery should clear scrollback. + + Some terminal/tmux stacks move prompt_toolkit's non-fullscreen bottom + chrome into scrollback when the window is maximized/restored. A normal + CSI 2J viewport clear cannot remove those stale prompt/input-rule rows, + so users who hit that class of bug need CSI 3J as well, followed by the + existing bounded output-history replay. + """ + from cli import CLI_CONFIG + display_config = CLI_CONFIG.get("display") if isinstance(CLI_CONFIG, dict) else {} + if not isinstance(display_config, dict): + display_config = {} + raw = display_config.get("cli_rebuild_scrollback_on_redraw", False) + if isinstance(raw, str): + return raw.strip().lower() in {"1", "true", "yes", "on", "always"} + return bool(raw) + + def _recover_terminal_after_interrupt(self) -> None: + """Recover the terminal after an interrupted agent turn (#33271). + + When the user interrupts a running turn by typing a new message, + prompt_toolkit may have an in-flight ``CSI 6n`` cursor-position query + whose reply (``ESC[;R``) arrives on stdin after the input + parser has torn down. The reply then leaks as literal text + (``^[[19;1R``) and the VT100 parser can stall in a partial-escape + state, accepting no further keystrokes — the terminal appears frozen. + + Two steps recover a sane state: + 1. ``flush_stdin()`` drains stray escape bytes from the OS input + buffer (``termios.tcflush(TCIFLUSH)``; no-op on non-TTY). + 2. ``_force_full_redraw()`` drops prompt_toolkit's cached + screen/cursor state and forces a clean repaint. + + Both steps are independently safe and self-guard, so a failure of one + never prevents the other. If the PTY is already dead (EIO), skip the + redraw entirely — painting a broken fd is the #81521 redraw storm. + """ + if getattr(self, "_terminal_io_broken", False): + return + try: + from hermes_cli.curses_ui import flush_stdin + flush_stdin() + except Exception: + pass + # #60920: The interruption marker is now printed with + # _suspend_output_history in chat(), so _OUTPUT_HISTORY only + # contains the normal response text (no marker text). Do NOT + # clear history here — _force_full_redraw → _replay_output_history + # replays the response correctly without duplicating the marker. + # The /redraw + Ctrl+L paths also preserve replay for scrollback + # recovery as intended. + self._force_full_redraw() + + def _clear_prompt_toolkit_screen(self, app, *, rebuild_scrollback: bool = False) -> None: + """Clear the terminal and reset prompt_toolkit renderer state.""" + if getattr(self, "_terminal_io_broken", False): + return + try: + renderer = app.renderer + out = renderer.output + out.reset_attributes() + out.erase_screen() + if rebuild_scrollback: + try: + out.write_raw("\x1b[3J") + except Exception: + pass + out.cursor_goto(0, 0) + out.flush() + # Drop prompt_toolkit's cached screen + cursor state so the + # next _redraw() starts from a known (0, 0) origin and + # re-renders every cell rather than diffing against stale. + renderer.reset(leave_alternate_screen=False) + except OSError as exc: + if getattr(exc, "errno", None) == errno.EIO: + self._mark_terminal_io_broken("clear_screen") + return + pass + except Exception: + pass + + def _recover_after_resize(self, app, original_on_resize) -> None: + """Recover a resized classic CLI without desynchronizing cursor state. + + Unlike _force_full_redraw, we do NOT clear the physical screen or + scrollback here. The startup banner and tool summary are printed + before prompt_toolkit owns the live chrome, so they live in normal + terminal scrollback. Erasing the screen on SIGWINCH removes that + startup UI and ``_replay_output_history`` cannot reconstruct it + (the banner was never added to ``_OUTPUT_HISTORY``). + + Let prompt_toolkit's own resize path run with its renderer cursor + cache intact. Its Application._on_resize() starts with + renderer.erase(leave_alternate_screen=False), which needs the cached + cursor position to move back to the live prompt origin before + erase_down(). Resetting the renderer before that erase loses the + origin and can leave stale prompt glyphs after a narrow resize. + + We also flag ``_status_bar_suppressed_after_resize`` so the dynamic + status bar and input separator rules stay hidden while the terminal + reflow settles. On column shrink the terminal reflows already-rendered + status bar rows into scrollback before prompt_toolkit can erase them; + drawing a fresh full-width bar immediately makes the old and new + versions look duplicated (#19280, #22976). + + Suppression alone is not enough on a WIDTH change. prompt_toolkit's + ``renderer.erase()`` does ``cursor_up(_cursor_pos.y)`` + ``erase_down()`` + using the ``_cursor_pos.y`` cached from the LAST render at the OLD + width (renderer.py). When the column count shrinks, the terminal + reflows each already-painted full-width chrome row into 2+ physical + rows, so the cached ``y`` undershoots: ``cursor_up`` does not climb + past the reflowed rows and ``erase_down`` leaves the stale bar stranded + ABOVE the live origin. The next paint then stacks a fresh bar below it + — the duplicated-status-bar report (two bars, two elapsed readings). + Suppression hides the *new* bar but never erases the already-reflowed + *old* one, so the ghost survives the whole suppression window. + + Fix: on a width change, wipe the visible viewport with ``erase_screen`` + (CSI 2J) BEFORE delegating to prompt_toolkit's resize, then let its + repaint redraw from a clean origin. This is banner-safe: 2J clears + only the visible screen, NOT scrollback history (that is CSI 3J, which + we do not send here — ``rebuild_scrollback=False``), so the startup + banner that scrolled into history is preserved and + ``_replay_output_history`` is not needed. Row-count-only changes skip + the clear (no reflow, so no ghost) to avoid an unnecessary repaint. + + The suppression is transient: a short follow-up timer clears it and + repaints once the reflow has settled, so the bar returns on its own + during idle. Previously the flag was only cleared on the next + *submitted* user input, so a resize/reflow (tmux pane change, SSH + window restore, font zoom) followed by idle left the status bar hidden + indefinitely even while the refresh clock kept ticking (the dynamic + chrome rendered at height 0 on every repaint). The next-submit clear + at the input loop remains as a fast path. + """ + from cli import _replay_output_history + self._status_bar_suppressed_after_resize = True + # On a WIDTH change the terminal has already reflowed the old full-width + # chrome into extra physical rows that prompt_toolkit's stale-cursor + # erase (cursor_up(_cursor_pos.y) cached at the OLD width) will not + # reach, leaving a duplicated status bar stranded above the live origin. + # Ctrl+L / /redraw clears it cleanly, so route the resize path through + # the SAME recovery: wipe the visible viewport (banner-safe — CSI 2J + # by default; CSI 3J only when display.cli_rebuild_scrollback_on_redraw + # is enabled) and replay the transcript so nothing is lost. + # Same-width SIGWINCH (tmux attach, benign focus/tab signals) is left + # untouched — no clear, no replay — because a 2J without replay erases + # the visible transcript and a replay against preserved scrollback + # duplicates it (#65293). The stale-previous_screen crash tmux attach + # used to trigger is handled by _hermes_call_output_screen_diff's + # retry-with-first-paint instead (#83874). + try: + new_width = self._get_tui_terminal_width() + except Exception: + new_width = None + prev_width = getattr(self, "_last_resize_width", None) + # Replay only on an OBSERVED width change. The first signal of a + # session must not count as one (#65293): GNOME Terminal and friends + # deliver benign SIGWINCHes (tab bar appearing, monitor-scale change, + # focus events), and a 2J+replay against preserved scrollback + # duplicates everything ``_OUTPUT_HISTORY`` holds — after a resume + # that is the entire "Previous Conversation" recap plus the first + # live exchange. ``_install_resize_recovery`` seeds the baseline at + # startup, so an initial maximize/restore still differs from it and + # is still recovered; with no baseline (width probe failed) this + # signal just records one for the next comparison. + width_changed = ( + new_width is not None + and prev_width is not None + and new_width != prev_width + ) + if width_changed: + try: + self._clear_prompt_toolkit_screen( + app, + rebuild_scrollback=self._redraw_rebuilds_scrollback(), + ) + _replay_output_history() + except Exception: + pass + if new_width is not None: + self._last_resize_width = new_width + if width_changed: + self._pet_queue_kitty_frame() + original_on_resize() + self._schedule_status_bar_unsuppress(app) + + def _schedule_status_bar_unsuppress(self, app, delay: float = 0.35) -> None: + """Clear the post-resize status-bar suppression after the reflow settles. + + Debounced: a fresh resize cancels the pending unsuppress and restarts + the timer, so a resize storm only repaints the bar once it stops. + """ + try: + old_timer = getattr(self, "_status_bar_unsuppress_timer", None) + if old_timer is not None: + try: + old_timer.cancel() + except Exception: + pass + + def _clear(): + self._status_bar_suppressed_after_resize = False + try: + app.invalidate() + except Exception: + pass + + def _fire(): + try: + loop = getattr(app, "loop", None) + except Exception: + loop = None + if loop is not None: + try: + loop.call_soon_threadsafe(_clear) + return + except Exception: + pass + _clear() + + timer = threading.Timer(delay, _fire) + timer.daemon = True + self._status_bar_unsuppress_timer = timer + timer.start() + except Exception: + # Fail open: never leave the bar stuck hidden. + self._status_bar_suppressed_after_resize = False + + def _schedule_resize_recovery(self, app, original_on_resize, delay: float = 0.12) -> None: + """Debounce resize redraws so footer chrome is not stamped into scrollback.""" + try: + old_timer = getattr(self, "_resize_recovery_timer", None) + lock = getattr(self, "_resize_recovery_lock", None) + if lock is None: + lock = threading.Lock() + self._resize_recovery_lock = lock + + def _timer_fired(timer_ref): + def _run_recovery(): + with lock: + if getattr(self, "_resize_recovery_timer", None) is not timer_ref: + return + self._resize_recovery_timer = None + self._resize_recovery_pending = False + self._recover_after_resize(app, original_on_resize) + + try: + loop = app.loop # type: ignore[attr-defined] + except Exception: + loop = None + if loop is not None: + try: + loop.call_soon_threadsafe(_run_recovery) + return + except Exception: + pass + _run_recovery() + + with lock: + if old_timer is not None: + try: + old_timer.cancel() + except Exception: + pass + self._resize_recovery_pending = True + timer = threading.Timer(delay, lambda: _timer_fired(timer)) + timer.daemon = True + self._resize_recovery_timer = timer + timer.start() + except Exception: + self._resize_recovery_pending = False + self._recover_after_resize(app, original_on_resize) + + def _install_resize_recovery(self, app) -> None: + """Route prompt_toolkit's ``_on_resize`` through the debounced + ghost-clearing recovery (#5474/#49120) and record the current terminal + width as the baseline for width-change detection. + + Seeding the baseline here is what keeps the session's FIRST SIGWINCH + honest (#65293): ``_recover_after_resize`` replays the transcript only + on an observed width change, and without a startup baseline it could + not tell a benign signal (GNOME Terminal tab bar, monitor-scale + change) from a real one. An initial maximize/restore still differs + from the seeded width, so it is still recovered. + + The probe reads ``app.output`` directly — NOT + ``_get_tui_terminal_width`` — because this runs before ``app.run()``, + when ``get_app()`` still returns prompt_toolkit's DummyApplication + whose DummyOutput reports a hardcoded 80 columns; seeding that fake + width would make the first real signal look like a width change and + resurrect the duplicate-replay bug this exists to fix. + ``app.output`` is the same object the running app's resize handler + measures, so install-time and signal-time widths are comparable. + """ + width = None + try: + width = app.output.get_size().columns + except Exception: + width = None + if not width or width <= 0: + try: + width = shutil.get_terminal_size((80, 24)).columns + except Exception: + width = None + self._last_resize_width = width + original_on_resize = app._on_resize + + def _resize_clear_ghosts(): + self._schedule_resize_recovery(app, original_on_resize) + + app._on_resize = _resize_clear_ghosts + + def _try_attach_clipboard_image(self) -> bool: + """Check clipboard for an image and attach it if found. + + Saves the image to ~/.hermes/images/ and appends the path to + ``_attached_images``. Returns True if an image was attached. + """ + from cli import datetime + from hermes_cli.clipboard import save_clipboard_image + + img_dir = get_hermes_home() / "images" + self._image_counter += 1 + ts = datetime.now().strftime("%Y%m%d_%H%M%S") + img_path = img_dir / f"clip_{ts}_{self._image_counter}.png" + + if save_clipboard_image(img_path): + self._attached_images.append(img_path) + return True + self._image_counter -= 1 + return False + + def _write_osc52_clipboard(self, text: str) -> None: + """Copy *text* to terminal clipboard via OSC 52. + + Wrapped for tmux/screen passthrough (mirrors the TUI's + wrapForMultiplexer in ui-tui/src/lib/osc52.ts) — without the DCS + wrapper the multiplexer consumes the sequence and the copy is + silently lost. + """ + payload = base64.b64encode(text.encode("utf-8")).decode("ascii") + seq = f"\x1b]52;c;{payload}\x07" + if os.environ.get("TMUX"): + seq = "\x1bPtmux;" + seq.replace("\x1b", "\x1b\x1b") + "\x1b\\" + elif os.environ.get("STY"): + seq = "\x1bP" + seq + "\x1b\\" + out = getattr(self, "_app", None) + output = getattr(out, "output", None) if out else None + if output and hasattr(output, "write_raw"): + output.write_raw(seq) + output.flush() + return + if output and hasattr(output, "write"): + output.write(seq) + output.flush() + return + sys.stdout.write(seq) + sys.stdout.flush() + + def _recover_terminal_input_modes(self, *, reason: str) -> None: + """Best-effort reset when leaked mouse reports indicate mode drift.""" + from cli import ( + CLI_CONFIG, + _DIM, + _RST, + _TERMINAL_INPUT_MODE_RESET_SEQ, + _cli_multiline_shortcuts_enabled, + _cprint, + _enable_extended_enter_keys, + logger, + ) + now = time.monotonic() + # Rate-limit to avoid thrashing if a terminal floods reports. + if now - self._last_input_mode_recovery < 0.5: + return + self._last_input_mode_recovery = now + + out = getattr(self, "_app", None) + output = getattr(out, "output", None) if out else None + try: + if output and hasattr(output, "write_raw"): + output.write_raw(_TERMINAL_INPUT_MODE_RESET_SEQ) + output.flush() + elif output and hasattr(output, "write"): + output.write(_TERMINAL_INPUT_MODE_RESET_SEQ) + output.flush() + else: + sys.stdout.write(_TERMINAL_INPUT_MODE_RESET_SEQ) + sys.stdout.flush() + except Exception: + return + + # The reset sequence above pops kitty keyboard mode and resets + # modifyOtherKeys too — re-request extended keys so Shift+Enter / + # modified-key reporting isn't silently dead for the rest of the + # session after a recovery (sibling of the startup push). + try: + if _cli_multiline_shortcuts_enabled(self.config or CLI_CONFIG): + _enable_extended_enter_keys(output) + except Exception: + pass + + logger.warning("Recovered terminal input modes after leak: %s", reason) + if not self._input_mode_recovery_notice_shown: + self._input_mode_recovery_notice_shown = True + _cprint( + f" {_DIM}Recovered terminal input modes after leaked mouse reports. " + f"If this repeats, run /new or restart this tab.{_RST}" + ) + + def _check_termios_drift(self) -> None: + """Watchdog: heal the tty if it drifted back to cooked mode. + + See ``_heal_cooked_mode_drift`` for the failure class (a lost + ``run_in_terminal`` cooked→raw restore leaves the terminal + line-buffering keystrokes while the prompt_toolkit app believes it + owns raw mode — the CLI looks dead but the process is healthy). + + Called from ``process_loop``'s idle branch, so a drifted terminal + self-heals within ~a second of the agent going idle instead of + requiring an external ``stty`` rescue. Skipped while a + ``run_in_terminal`` window is legitimately holding cooked mode + (``app._running_in_terminal``), while the agent is running (approval + prompts and sudo prompts legitimately manipulate the tty), and on + Windows (no termios). + """ + from cli import _DIM, _RST, _cprint, _heal_cooked_mode_drift, logger + if os.name == "nt": + return + app = getattr(self, "_app", None) + if app is None or not getattr(app, "_is_running", False): + return + # A run_in_terminal window is *supposed* to be cooked — don't fight it. + if getattr(app, "_running_in_terminal", False): + return + now = time.monotonic() + if now - self._last_termios_drift_check < 1.0: + return + self._last_termios_drift_check = now + try: + if not sys.stdin.isatty(): + return + fd = sys.stdin.fileno() + except Exception: + return + if _heal_cooked_mode_drift(fd): + logger.warning( + "Healed cooked-mode termios drift on stdin — a " + "run_in_terminal cooked→raw restore was lost." + ) + # Redraw so the prompt is visibly alive again. + try: + self._invalidate() + except Exception: + pass + if not self._termios_drift_notice_shown: + self._termios_drift_notice_shown = True + _cprint( + f" {_DIM}Recovered terminal from cooked-mode drift " + f"(input should respond normally again).{_RST}" + ) diff --git a/hermes_cli/cli_tui_mixin.py b/hermes_cli/cli_tui_mixin.py new file mode 100644 index 0000000000..7bbcfea2f5 --- /dev/null +++ b/hermes_cli/cli_tui_mixin.py @@ -0,0 +1,3080 @@ +"""prompt_toolkit TUI construction, key-binding handlers, and overlay display fragments for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import errno +import json +import os +import queue +import shutil +import sys +import threading +import time + +from agent.interrupt_compat import request_hard_interrupt +from hermes_cli.commands import SlashCommandAutoSuggest, SlashCommandCompleter +from pathlib import Path +from prompt_toolkit.filters import Condition +from prompt_toolkit.history import FileHistory +from prompt_toolkit.key_binding import KeyBindings +from prompt_toolkit.layout import ( + ConditionalContainer, + FormattedTextControl, + HSplit, + Layout, + Window, + WindowAlign, +) +from prompt_toolkit.layout.dimension import Dimension +from prompt_toolkit.layout.menus import CompletionsMenu +from prompt_toolkit.layout.processors import ( + ConditionalProcessor, + PasswordProcessor, + Processor, + Transformation, +) +from prompt_toolkit.styles import Style as PTStyle +from prompt_toolkit.widgets import TextArea +from typing import Optional + + +class CLITuiMixin: + """prompt_toolkit TUI construction, key-binding handlers, and overlay display fragments for the interactive CLI""" + + def _tui_input_rule_height(self, position: str, width: Optional[int] = None) -> int: + """Return the visible height for the top/bottom input separator rules.""" + if position not in {"top", "bottom"}: + raise ValueError(f"Unknown input rule position: {position}") + if getattr(self, "_status_bar_suppressed_after_resize", False): + return 0 + if position == "top": + return 1 + return 0 if self._use_minimal_tui_chrome(width=width) else 1 + + def _get_slash_confirm_display_fragments(self): + """Render the /new-/clear-style confirmation panel.""" + from cli import ( + _append_blank_panel_line, + _append_panel_line, + _panel_box_width, + _wrap_panel_text_keep_ws, + ) + state = self._slash_confirm_state + if not state: + return [] + + title = state.get("title") or "Confirm action" + detail = state.get("detail") or "" + choices = state.get("choices") or [] + selected = state.get("selected", 0) + + _wrap_panel_text = _wrap_panel_text_keep_ws + + preview_lines = [] + for line in detail.splitlines(): + preview_lines.extend(_wrap_panel_text(line, 72)) + for idx, (_value, label, desc) in enumerate(choices): + marker = "❯" if idx == selected else " " + preview_lines.extend(_wrap_panel_text(f"{marker} [{idx + 1}] {label} — {desc}", 72, subsequent_indent=" ")) + preview_lines.append("Type 1/2/3 or use ↑/↓ then Enter. ESC/Ctrl+C cancels.") + + box_width = _panel_box_width(title, preview_lines, min_width=56, max_width=86) + inner_text_width = max(8, box_width - 2) + detail_wrapped = [] + for line in detail.splitlines(): + detail_wrapped.extend(_wrap_panel_text(line, inner_text_width)) + choice_wrapped: list[tuple[int, str]] = [] + for idx, (_value, label, desc) in enumerate(choices): + marker = "❯" if idx == selected else " " + for wrapped in _wrap_panel_text(f"{marker} [{idx + 1}] {label} — {desc}", inner_text_width, subsequent_indent=" "): + choice_wrapped.append((idx, wrapped)) + + term_rows = shutil.get_terminal_size((100, 24)).lines + reserved_below = 6 + chrome_full = 6 + available = max(0, term_rows - reserved_below) + max_detail_rows = max(1, available - chrome_full - len(choice_wrapped)) + max_detail_rows = min(max_detail_rows, 8) + if len(detail_wrapped) > max_detail_rows: + keep = max(1, max_detail_rows - 1) + detail_wrapped = detail_wrapped[:keep] + ["… (detail truncated)"] + + lines = [] + lines.append(('class:approval-border', '╭' + ('─' * box_width) + '╮\n')) + _append_panel_line(lines, 'class:approval-border', 'class:approval-title', title, box_width) + _append_blank_panel_line(lines, 'class:approval-border', box_width) + for wrapped in detail_wrapped: + _append_panel_line(lines, 'class:approval-border', 'class:approval-desc', wrapped, box_width) + _append_blank_panel_line(lines, 'class:approval-border', box_width) + for idx, wrapped in choice_wrapped: + style = 'class:approval-selected' if idx == selected else 'class:approval-choice' + _append_panel_line(lines, 'class:approval-border', style, wrapped, box_width) + _append_blank_panel_line(lines, 'class:approval-border', box_width) + _append_panel_line(lines, 'class:approval-border', 'class:approval-cmd', 'Type 1/2/3 or use ↑/↓ then Enter. ESC/Ctrl+C cancels.', box_width) + lines.append(('class:approval-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_approval_display_fragments(self): + """Render the dangerous-command approval panel for the prompt_toolkit UI. + + Layout priority: title + command + choices must always render, even if + the terminal is short or the description is long. Description is placed + at the bottom of the panel and gets truncated to fit the remaining row + budget. This prevents HSplit from clipping approve/deny off-screen when + tirith findings produce multi-paragraph descriptions or when the user + runs in a compact terminal pane. + """ + from cli import ( + _append_blank_panel_line, + _append_panel_line, + _panel_box_width, + _wrap_panel_text_keep_ws, + ) + state = self._approval_state + if not state: + return [] + + _wrap_panel_text = _wrap_panel_text_keep_ws + + command = state["command"] + description = state["description"] + choices = state["choices"] + selected = state.get("selected", 0) + show_full = state.get("show_full", False) + + title = "⚠️ Dangerous Command" + cmd_display = command + choice_labels = { + "once": "Allow once", + "session": "Allow for this session", + "always": "Add to permanent allowlist", + "deny": "Deny", + "view": "Show full command", + } + + preview_lines = _wrap_panel_text(description, 60) + preview_lines.extend(_wrap_panel_text(cmd_display, 60)) + for i, choice in enumerate(choices): + prefix = '❯ ' if i == selected else ' ' + preview_lines.extend(_wrap_panel_text( + f"{prefix}{choice_labels.get(choice, choice)}", + 60, + subsequent_indent=" ", + )) + + box_width = _panel_box_width(title, preview_lines) + inner_text_width = max(8, box_width - 2) + + # Pre-wrap the mandatory content — command + choices must always render. + cmd_wrapped = _wrap_panel_text(cmd_display, inner_text_width) + if not show_full and "view" in choices and len(cmd_wrapped) > 4: + cmd_wrapped = cmd_wrapped[:3] + _wrap_panel_text( + "… (choose Show full command)", + inner_text_width, + ) + + # (choice_index, wrapped_line) so we can re-apply selected styling below + choice_wrapped: list[tuple[int, str]] = [] + for i, choice in enumerate(choices): + label = choice_labels.get(choice, choice) + # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) + if i < 9: + num_prefix = str(i + 1) + elif i == 9: + num_prefix = '0' + else: + num_prefix = ' ' # No number for items beyond 10th + prefix = f'❯ {num_prefix}. ' if i == selected else f' {num_prefix}. ' + for wrapped in _wrap_panel_text(f"{prefix}{label}", inner_text_width, subsequent_indent=" "): + choice_wrapped.append((i, wrapped)) + + # Budget vertical space so HSplit never clips the command or choices. + # Panel chrome (full layout with separators): + # top border + title + blank_after_title + # + blank_between_cmd_choices + bottom border = 5 rows. + # In tight terminals we collapse to: + # top border + title + bottom border = 3 rows (no blanks). + # + # reserved_below: rows consumed below the approval panel by the + # spinner/tool-progress line, status bar, input area, separators, and + # prompt symbol. Measured at ~6 rows during live PTY approval prompts; + # budget 6 so we don't overestimate the panel's room. + term_rows = shutil.get_terminal_size((100, 24)).lines + chrome_full = 5 + chrome_tight = 3 + reserved_below = 6 + + available = max(0, term_rows - reserved_below) + mandatory_full = chrome_full + len(cmd_wrapped) + len(choice_wrapped) + + # If the full-chrome panel doesn't fit, drop the separator blanks. + # This keeps the command and every choice on-screen in compact terminals. + use_compact_chrome = mandatory_full > available + chrome_rows = chrome_tight if use_compact_chrome else chrome_full + + # If the command itself is too long to leave room for choices (e.g. user + # hit "view" on a multi-hundred-character command), truncate it so the + # approve/deny buttons still render. Keep at least 1 row of command. + max_cmd_rows = max(1, available - chrome_rows - len(choice_wrapped)) + if len(cmd_wrapped) > max_cmd_rows: + keep = max(1, max_cmd_rows - 1) if max_cmd_rows > 1 else 1 + cmd_wrapped = cmd_wrapped[:keep] + _wrap_panel_text( + "… (command truncated — use /logs or /debug for full text)", + inner_text_width, + ) + + # Allocate any remaining rows to description. The extra -1 in full mode + # accounts for the blank separator between choices and description. + mandatory_no_desc = chrome_rows + len(cmd_wrapped) + len(choice_wrapped) + desc_sep_cost = 0 if use_compact_chrome else 1 + available_for_desc = available - mandatory_no_desc - desc_sep_cost + # Even on huge terminals, cap description height so the panel stays compact. + available_for_desc = max(0, min(available_for_desc, 10)) + + desc_wrapped = _wrap_panel_text(description, inner_text_width) if description else [] + if available_for_desc < 1 or not desc_wrapped: + desc_wrapped = [] + elif len(desc_wrapped) > available_for_desc: + keep = max(1, available_for_desc - 1) + desc_wrapped = desc_wrapped[:keep] + ["… (description truncated)"] + + # Render: title → command → choices → description (description last so + # any remaining overflow clips from the bottom of the least-critical + # content, never from the command or choices). Use compact chrome (no + # blank separators) when the terminal is tight. + lines = [] + lines.append(('class:approval-border', '╭' + ('─' * box_width) + '╮\n')) + _append_panel_line(lines, 'class:approval-border', 'class:approval-title', title, box_width) + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:approval-border', box_width) + + for wrapped in cmd_wrapped: + _append_panel_line(lines, 'class:approval-border', 'class:approval-cmd', wrapped, box_width) + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:approval-border', box_width) + + for i, wrapped in choice_wrapped: + style = 'class:approval-selected' if i == selected else 'class:approval-choice' + _append_panel_line(lines, 'class:approval-border', style, wrapped, box_width) + + if desc_wrapped: + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:approval-border', box_width) + for wrapped in desc_wrapped: + _append_panel_line(lines, 'class:approval-border', 'class:approval-desc', wrapped, box_width) + + lines.append(('class:approval-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_tui_prompt_symbols(self) -> tuple[str, str]: + """Return ``(normal_prompt, state_suffix)`` for the active skin. + + ``normal_prompt`` is the full ``branding.prompt_symbol``. + ``state_suffix`` is what special states (sudo/secret/approval/agent) + should render after their leading icon. + + When a profile is active (not "default"), the profile name is + prepended to the prompt symbol: ``coder ❯`` instead of ``❯``. + """ + try: + from hermes_cli.skin_engine import get_active_prompt_symbol + symbol = get_active_prompt_symbol("❯ ") + except Exception: + symbol = "❯ " + + symbol = (symbol or "❯ ").rstrip() + " " + + # Prepend profile name when not default + try: + from hermes_cli.profiles import get_active_profile_name + profile = get_active_profile_name() + if profile not in {"default", "custom"}: + symbol = f"{profile} {symbol}" + except Exception: + pass + stripped = symbol.rstrip() + if not stripped: + return "❯ ", "❯ " + + parts = stripped.split() + candidate = parts[-1] if parts else "" + arrow_chars = ("❯", ">", "$", "#", "›", "»", "→") + if any(ch in candidate for ch in arrow_chars): + return symbol, candidate.rstrip() + " " + + # Icon-only custom prompts should still remain visible in special states. + return symbol, symbol + + def _audio_level_bar(self) -> str: + """Return a visual audio level indicator based on current RMS.""" + _LEVEL_BARS = " ▁▂▃▄▅▆▇" + rec = getattr(self, "_voice_recorder", None) + if rec is None: + return "" + rms = rec.current_rms + # Normalize RMS (0-32767) to 0-7 index, with log-ish scaling + # Typical speech RMS is 500-5000, we cap display at ~8000 + level = min(rms, 8000) * 7 // 8000 + return _LEVEL_BARS[level] + + def _get_tui_prompt_fragments(self): + """Return the prompt_toolkit fragments for the current interactive state.""" + symbol, state_suffix = self._get_tui_prompt_symbols() + compact = self._use_minimal_tui_chrome(width=self._get_tui_terminal_width()) + + def _state_fragment(style: str, icon: str, extra: str = ""): + if compact: + text = icon + if extra: + text = f"{text} {extra.strip()}".rstrip() + return [(style, text + " ")] + if extra: + return [(style, f"{icon} {extra} {state_suffix}")] + return [(style, f"{icon} {state_suffix}")] + + if self._voice_recording: + bar = self._audio_level_bar() + return _state_fragment("class:voice-recording", "●", bar) + if self._voice_processing: + return _state_fragment("class:voice-processing", "◉") + if self._sudo_state: + return _state_fragment("class:sudo-prompt", "🔐") + if self._secret_state: + return _state_fragment("class:sudo-prompt", "🔑") + if self._approval_state: + return _state_fragment("class:prompt-working", "⚠") + if getattr(self, "_slash_confirm_state", None): + return _state_fragment("class:prompt-working", "⚠") + if self._clarify_freetext: + return _state_fragment("class:clarify-selected", "✎") + if self._clarify_state: + return _state_fragment("class:prompt-working", "?") + if self._command_running: + return _state_fragment("class:prompt-working", self._command_spinner_frame()) + if self._agent_running: + return _state_fragment("class:prompt-working", "⚕") + if self._voice_mode: + return _state_fragment("class:voice-prompt", "🎤") + return [("class:prompt", symbol)] + + def _get_tui_prompt_text(self) -> str: + """Return the visible prompt text for width calculations.""" + return "".join(text for _, text in self._get_tui_prompt_fragments()) + + def _build_tui_style_dict(self) -> dict[str, str]: + """Layer the active skin's prompt_toolkit colors over the base TUI style. + + Also rewrites any hex-color tokens in the resulting style strings + to their light-mode equivalents (via _LIGHT_MODE_REMAP) when the + terminal is detected as light. This makes the chrome readable + on cream Terminal.app backgrounds without per-skin overrides. + """ + from cli import _detect_light_mode, _maybe_remap_for_light_mode + style_dict = dict(getattr(self, "_tui_style_base", {}) or {}) + try: + from hermes_cli.skin_engine import get_prompt_toolkit_style_overrides + style_dict.update(get_prompt_toolkit_style_overrides()) + except Exception: + pass + # Light-mode remap on the style strings. Each value is a pt + # style string like "bg:#1a1a2e #C0C0C0 bold" — split on space, + # rewrite any "#XXX" tokens (including "bg:#XXX") through the + # light-mode remap, rejoin. + # + # CRITICAL: skip the remap entirely when a style string already + # specifies its own bg (e.g. status-bar / completion-menu styles + # with `bg:#1a1a2e ...`). Those colors were tuned for that + # specific dark bg and remapping the FG to a dark equivalent + # would produce dark-on-dark (invisible). The terminal's BG + # mode is irrelevant — what matters is the bg the style itself + # paints. + try: + if _detect_light_mode(): + def _remap_value(v: str) -> str: + if not v: + return v + tokens = v.split() + has_explicit_bg = any(t.startswith("bg:") for t in tokens) + if has_explicit_bg: + # The style paints its own bg — leave its fg alone. + return v + return " ".join( + _maybe_remap_for_light_mode(t) if t.startswith("#") else t + for t in tokens + ) + style_dict = {k: _remap_value(v or "") for k, v in style_dict.items()} + except Exception: + pass + return style_dict + + def _apply_tui_skin_style(self) -> bool: + """Refresh prompt_toolkit styling for a running interactive TUI.""" + if not getattr(self, "_app", None) or not getattr(self, "_tui_style_base", None): + return False + self._app.style = PTStyle.from_dict(self._build_tui_style_dict()) + self._invalidate(min_interval=0.0) + return True + + def _get_extra_tui_widgets(self) -> list: + """Return extra prompt_toolkit widgets to insert into the TUI layout. + + Wrapper CLIs can override this to inject widgets (e.g. a mini-player, + overlay menu) into the layout without overriding ``run()``. Widgets + are inserted between the spacer and the status bar. + """ + return [] + + def _register_extra_tui_keybindings(self, kb, *, input_area) -> None: + """Register extra keybindings on the TUI ``KeyBindings`` object. + + Wrapper CLIs can override this to add keybindings (e.g. transport + controls, modal shortcuts) without overriding ``run()``. + + Parameters + ---------- + kb : KeyBindings + The active keybinding registry for the prompt_toolkit application. + input_area : TextArea + The main input widget, for wrappers that need to inspect or + manipulate user input from a keybinding handler. + """ + + def _build_tui_layout_children( + self, + *, + sudo_widget, + secret_widget, + approval_widget, + slash_confirm_widget=None, + clarify_widget, + model_picker_widget=None, + command_palette_widget=None, + spinner_widget=None, + spacer, + status_bar, + input_rule_top, + image_bar, + input_area, + input_rule_bot, + voice_status_bar, + completions_menu, + ) -> list: + """Assemble the ordered list of children for the root ``HSplit``. + + Wrapper CLIs typically override ``_get_extra_tui_widgets`` instead of + this method. Override this only when you need full control over widget + ordering. + """ + return [ + item for item in [ + Window(height=0), + sudo_widget, + secret_widget, + approval_widget, + slash_confirm_widget, + clarify_widget, + model_picker_widget, + command_palette_widget, + spinner_widget, + spacer, + *self._get_extra_tui_widgets(), + getattr(self, "_pet_widget", None), + getattr(self, "_stash_panel_widget", None), + status_bar, + input_rule_top, + image_bar, + input_area, + input_rule_bot, + voice_status_bar, + completions_menu, + ] if item is not None + ] + + def _tui_spinner_loop(self): + while not self._should_exit: + if not self._app: + time.sleep(0.1) + continue + if self._command_running: + self._invalidate(min_interval=0.1) + time.sleep(0.1) + else: + # Do not repaint the idle prompt every second. In non-full-screen + # prompt_toolkit mode, background redraws can fight tmux/Ghostty/cmux + # viewport restoration after focus changes and visually move the + # command input area. Keep idle stable; input/agent events still + # invalidate explicitly when the UI actually changes. + time.sleep(0.2) + + def _get_clarify_batch_display_fragments(self, state): + """Build styled text for the batch (multi-question) clarify panel. + + A-compact layout mirroring the TUI: a "N questions" header, one + status line per question (✓ answered → answer / ▸ active / + · pending), and the active question's numbered choices (+ Other) + expanded directly beneath its status line. + """ + from cli import _append_panel_line, _panel_box_width, _wrap_panel_text + questions_list = state.get("questions") or [] + answers = state.get("answers") or {} + active = state.get("active", 0) + choices = state.get("choices") or [] + selected = state.get("selected", 0) + multi_select = state.get("multi_select", False) + selected_indices = state.get("selected_indices", set()) if multi_select else set() + + title = "Hermes needs your input" + header = f"{len(questions_list)} questions" + + def _status_rows(width): + """(style, text) rows for the status list + expanded active question.""" + rows = [] + answer_meta = state.get("answer_meta") or {} + for idx, entry in enumerate(questions_list): + answered = entry["qid"] in answers + if answered: + marker = "✓" + elif idx == active: + marker = "▸" + else: + marker = "·" + label = f"{marker} {entry['question']}" + row_style = 'class:clarify-selected' if idx == active else 'class:clarify-choice' + for wrapped in _wrap_panel_text(label, width, subsequent_indent=" "): + rows.append((row_style, wrapped)) + if answered: + # The locked answer on its own line, in its own color, + # so the current answer stays readable while walking + # the list with Tab/Shift-Tab. + for wrapped in _wrap_panel_text( + f" {answers[entry['qid']]}", width, subsequent_indent=" " + ): + rows.append(('class:clarify-answer', wrapped)) + if idx != active: + continue + # Expanded active question: numbered choices + Other. + for i, choice in enumerate(choices): + num_prefix = str(i + 1) if i < 9 else ('0' if i == 9 else ' ') + if multi_select: + cb = "[x]" if i in selected_indices else "[ ]" + cursor = "❯" if i == selected and not self._clarify_freetext else " " + prefix = f" {cursor} {cb} {num_prefix}. " + else: + cursor = "❯" if i == selected and not self._clarify_freetext else " " + prefix = f" {cursor} {num_prefix}. " + style = 'class:clarify-selected' if i == selected and not self._clarify_freetext else 'class:clarify-choice' + for wrapped in _wrap_panel_text(f"{prefix}{choice}", width, subsequent_indent=" "): + rows.append((style, wrapped)) + if choices: + other_idx = len(choices) + other_num = other_idx + 1 + other_num_prefix = str(other_num) if other_num < 10 else ('0' if other_num == 10 else ' ') + if multi_select: + cb = "[x]" if other_idx in selected_indices else "[ ]" + mid = f"{cb} {other_num_prefix}" + else: + mid = other_num_prefix + # An earlier typed answer stays visible next to Other; + # Enter on it edits (the composer is prefilled). + meta = answer_meta.get(entry["qid"]) or {} + other_text = meta.get("other_text") or "" + other_suffix = f"Other: {other_text}" if other_text else None + if self._clarify_freetext: + other_label = f" ❯ {mid}. " + (other_suffix or "Other (type below)") + other_style = 'class:clarify-active-other' + elif selected == other_idx: + other_label = f" ❯ {mid}. " + (other_suffix or "Other (type your answer)") + other_style = 'class:clarify-selected' + else: + other_label = f" {mid}. " + (other_suffix or "Other (type your answer)") + other_style = 'class:clarify-choice' + for wrapped in _wrap_panel_text(other_label, width, subsequent_indent=" "): + rows.append((other_style, wrapped)) + elif self._clarify_freetext: + for wrapped in _wrap_panel_text( + " Type your answer in the prompt below, then press Enter.", width + ): + rows.append(('class:clarify-active-other', wrapped)) + return rows + + preview_rows = _status_rows(60) + box_width = _panel_box_width(title, [header] + [text for _, text in preview_rows]) + inner_text_width = max(8, box_width - 2) + rows = _status_rows(inner_text_width) + + lines = [] + lines.append(('class:clarify-border', '╭─ ')) + lines.append(('class:clarify-title', title)) + lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) + _append_panel_line(lines, 'class:clarify-border', 'class:clarify-question', header, box_width) + for style, text in rows: + _append_panel_line(lines, 'class:clarify-border', style, text, box_width) + lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_clarify_display_fragments(self): + """Build styled text for the clarify question/choices panel. + + Layout priority: choices + Other option must always render even if + the question is very long. The question is budgeted to leave enough + rows for the choices and trailing chrome; anything over the budget + is truncated with a marker. + """ + from cli import _append_blank_panel_line, _append_panel_line, _panel_box_width, _wrap_panel_text + state = self._clarify_state + if not state: + return [] + if state.get("questions"): + return self._get_clarify_batch_display_fragments(state) + + question = state["question"] + choices = state.get("choices") or [] + selected = state.get("selected", 0) + # multi-select support + multi_select = state.get("multi_select", False) + selected_indices = state.get("selected_indices", set()) if multi_select else set() + preview_lines = _wrap_panel_text(question, 60) + for i, choice in enumerate(choices): + # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) + if i < 9: + num_prefix = str(i + 1) + elif i == 9: + num_prefix = '0' + else: + num_prefix = ' ' + if multi_select: + cb = "[x]" if i in selected_indices else "[ ]" + if i == selected and not self._clarify_freetext: + prefix = f"❯ {cb} {num_prefix}. " + else: + prefix = f" {cb} {num_prefix}. " + elif i == selected and not self._clarify_freetext: + prefix = f"❯ {num_prefix}. " + else: + prefix = f" {num_prefix}. " + preview_lines.extend(_wrap_panel_text(f"{prefix}{choice}", 60, subsequent_indent=" ")) + # "Other" option in preview + other_num = len(choices) + 1 + if other_num < 10: + other_num_prefix = str(other_num) + elif other_num == 10: + other_num_prefix = '0' + else: + other_num_prefix = ' ' + other_idx_val = len(choices) + if multi_select: + cb = "[x]" if other_idx_val in selected_indices else "[ ]" + other_label = ( + f"❯ {cb} {other_num_prefix}. Other (type below)" if self._clarify_freetext + else f"❯ {cb} {other_num_prefix}. Other (type your answer)" if selected == other_idx_val + else f" {cb} {other_num_prefix}. Other (type your answer)" + ) + else: + other_label = ( + f"❯ {other_num_prefix}. Other (type below)" if self._clarify_freetext + else f"❯ {other_num_prefix}. Other (type your answer)" if selected == len(choices) + else f" {other_num_prefix}. Other (type your answer)" + ) + preview_lines.extend(_wrap_panel_text(other_label, 60, subsequent_indent=" ")) + box_width = _panel_box_width("Hermes needs your input", preview_lines) + inner_text_width = max(8, box_width - 2) + + # Pre-wrap choices + Other option — these are mandatory. + choice_wrapped: list[tuple[int, str]] = [] + if choices: + for i, choice in enumerate(choices): + # Show number prefix for quick selection (1-9 for items 1-9, 0 for 10th item) + if i < 9: + num_prefix = str(i + 1) + elif i == 9: + num_prefix = '0' + else: + num_prefix = ' ' + # multi-select support: add checkbox after cursor indicator + if multi_select: + cb = "[x]" if i in selected_indices else "[ ]" + if i == selected and not self._clarify_freetext: + prefix = f'❯ {cb} {num_prefix}. ' + else: + prefix = f' {cb} {num_prefix}. ' + elif i == selected and not self._clarify_freetext: + prefix = f'❯ {num_prefix}. ' + else: + prefix = f' {num_prefix}. ' + for wrapped in _wrap_panel_text(f"{prefix}{choice}", inner_text_width, subsequent_indent=" "): + choice_wrapped.append((i, wrapped)) + # Trailing Other row(s) + other_idx = len(choices) + other_num = other_idx + 1 + if other_num < 10: + other_num_prefix = str(other_num) + elif other_num == 10: + other_num_prefix = '0' + else: + other_num_prefix = ' ' + # multi-select support: add checkbox to Other option + if multi_select: + cb = "[x]" if other_idx in selected_indices else "[ ]" + if selected == other_idx and not self._clarify_freetext: + other_label_mand = f'❯ {cb} {other_num_prefix}. Other (type your answer)' + elif self._clarify_freetext: + other_label_mand = f'❯ {cb} {other_num_prefix}. Other (type below)' + else: + other_label_mand = f' {cb} {other_num_prefix}. Other (type your answer)' + else: + if selected == other_idx and not self._clarify_freetext: + other_label_mand = f'❯ {other_num_prefix}. Other (type your answer)' + elif self._clarify_freetext: + other_label_mand = f'❯ {other_num_prefix}. Other (type below)' + else: + other_label_mand = f' {other_num_prefix}. Other (type your answer)' + other_wrapped = _wrap_panel_text(other_label_mand, inner_text_width, subsequent_indent=" ") + elif self._clarify_freetext: + # Freetext-only mode: the guidance line takes the place of choices. + other_wrapped = _wrap_panel_text( + "Type your answer in the prompt below, then press Enter.", + inner_text_width, + ) + else: + other_wrapped = [] + + # Budget the question so mandatory rows always render. + # Chrome layouts: + # full : top border + blank_after_title + blank_after_question + # + blank_before_bottom + bottom border = 5 rows + # tight: top border + bottom border = 2 rows (drop all blanks) + # + # reserved_below matches the approval-panel budget (~6 rows for + # spinner/tool-progress + status + input + separators + prompt). + term_rows = shutil.get_terminal_size((100, 24)).lines + chrome_full = 5 + chrome_tight = 2 + reserved_below = 6 + + available = max(0, term_rows - reserved_below) + # The compact decision must reserve room for at least one question + # row on top of the choices, otherwise full chrome (3 blank + # separators) gets kept when there is no room for it and the panel + # overflows the viewport — HSplit then clips the panel's tail, + # silently dropping the choices (the reported bug). + mandatory_full = chrome_full + 1 + len(choice_wrapped) + len(other_wrapped) + + use_compact_chrome = mandatory_full > available + chrome_rows = chrome_tight if use_compact_chrome else chrome_full + + max_question_rows = max(1, available - chrome_rows - len(choice_wrapped) - len(other_wrapped)) + max_question_rows = min(max_question_rows, 12) # soft cap on huge terminals + + # When the choices alone (plus compact chrome) already exceed the + # viewport, drop the question entirely — the choices are the only + # thing the user must see to make a selection. Without this the + # question would still claim its 1-row floor above and push the + # tail of the choices off-screen (HSplit clips the overflow). + choices_overflow = chrome_rows + len(choice_wrapped) + len(other_wrapped) >= available + if choices_overflow: + max_question_rows = 0 + + question_wrapped = _wrap_panel_text(question, inner_text_width) + if max_question_rows <= 0: + question_wrapped = [] + elif len(question_wrapped) > max_question_rows: + # The truncation marker is itself a row, so it must count + # against the budget. With a 1-row budget there is no room for + # both a question line and the marker — show the marker alone + # so the rendered question never exceeds max_question_rows. + keep = max(0, max_question_rows - 1) + question_wrapped = question_wrapped[:keep] + ["… (question truncated)"] + + lines = [] + # Box top border + lines.append(('class:clarify-border', '╭─ ')) + lines.append(('class:clarify-title', 'Hermes needs your input')) + lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len("Hermes needs your input") - 3)) + '╮\n')) + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + + # Question text (bounded) + for wrapped in question_wrapped: + _append_panel_line(lines, 'class:clarify-border', 'class:clarify-question', wrapped, box_width) + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + + if self._clarify_freetext and not choices: + for wrapped in other_wrapped: + _append_panel_line(lines, 'class:clarify-border', 'class:clarify-choice', wrapped, box_width) + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + + if choices: + # Multiple-choice mode: show selectable options + for i, wrapped in choice_wrapped: + style = 'class:clarify-selected' if i == selected and not self._clarify_freetext else 'class:clarify-choice' + _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) + + # "Other" option (trailing row(s), only shown when choices exist) + other_idx = len(choices) + # Calculate number prefix for "Other" option + other_num = other_idx + 1 + if other_num < 10: + other_num_prefix = str(other_num) + elif other_num == 10: + other_num_prefix = '0' + else: + other_num_prefix = ' ' + + if selected == other_idx and not self._clarify_freetext: + other_style = 'class:clarify-selected' + elif self._clarify_freetext: + other_style = 'class:clarify-active-other' + else: + other_style = 'class:clarify-choice' + for wrapped in other_wrapped: + _append_panel_line(lines, 'class:clarify-border', other_style, wrapped, box_width) + + if not use_compact_chrome: + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_model_picker_display_fragments(self): + from cli import ( + HermesCLI, + _append_blank_panel_line, + _append_panel_line, + _panel_box_width, + _wrap_panel_text, + ) + state = self._model_picker_state + if not state: + return [] + stage = state.get("stage", "provider") + if stage == "provider": + title = "⚙ Model Picker — Select Provider" + choices = [] + _providers = state.get("providers") + for p in _providers if isinstance(_providers, list) else []: + count = p.get("total_models", len(p.get("models", []))) + label = f"{p['name']} ({count} model{'s' if count != 1 else ''})" + if p.get("is_current"): + label += " ← current" + choices.append(label) + choices.append("Cancel") + hint = f"Current: {state.get('current_model', 'unknown')} on {state.get('current_provider', 'unknown')}" + else: + provider_data = state.get("provider_data") or {} + model_list = state.get("model_list") or [] + title = f"⚙ Model Picker — {provider_data.get('name', provider_data.get('slug', 'Provider'))}" + # Fuzzy filter: narrow the concrete model list by the typed + # query. Selection still resolves to a real entry (see the + # filtered_pairs index mapping in the selection handler), so + # this never introduces an ambiguous model resolution. + _query = state.get("filter", "") or "" + filtered_pairs = self._filter_model_picker_entries(model_list, _query) + state["_filtered_pairs"] = filtered_pairs + model_labels = [e for (_i, e) in filtered_pairs] + choices = list(model_labels) + ["← Back", "Cancel"] + if _query: + hint = ( + f"Filter: {_query}▏ ({len(model_labels)}/{len(model_list)} match " + "— type to narrow, Backspace to clear)" + ) + elif model_list: + hint = f"Select a model ({len(model_list)} available) — type to filter" + else: + hint = "No models listed for this provider. Use Back or Cancel." + + box_width = _panel_box_width(title, [hint] + choices, min_width=46, max_width=84) + inner_text_width = max(8, box_width - 6) + selected = state.get("selected", 0) + + # Scrolling viewport: the panel renders into a Window with no max + # height, so without limiting visible items the bottom border and + # any items past the available terminal rows get clipped on long + # provider catalogs (e.g. Ollama Cloud's 36+ models). + try: + from prompt_toolkit.application import get_app + term_rows = get_app().output.get_size().rows + except Exception: + term_rows = shutil.get_terminal_size((100, 24)).lines + scroll_offset, visible = HermesCLI._compute_model_picker_viewport( + selected, state.get("_scroll_offset", 0), len(choices), term_rows, + ) + state["_scroll_offset"] = scroll_offset + + lines = [] + lines.append(('class:clarify-border', '╭─ ')) + lines.append(('class:clarify-title', title)) + lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + _append_panel_line(lines, 'class:clarify-border', 'class:clarify-hint', hint, box_width) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + for idx in range(scroll_offset, scroll_offset + visible): + choice = choices[idx] + style = 'class:clarify-selected' if idx == selected else 'class:clarify-choice' + prefix = '❯ ' if idx == selected else ' ' + for wrapped in _wrap_panel_text(prefix + choice, inner_text_width, subsequent_indent=' '): + _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_command_palette_display_fragments(self): + from cli import ( + HermesCLI, + _append_blank_panel_line, + _append_panel_line, + _panel_box_width, + _wrap_panel_text, + ) + state = self._command_palette_state + if not state: + return [] + rows = self._command_palette_visible_entries() + state["_visible_count"] = len(rows) + _query = state.get("filter", "") or "" + total = len(state.get("entries") or []) + title = "⚙ Command Palette" + if _query: + hint = f"Filter: {_query}▏ ({len(rows)}/{total} match — Enter inserts, Esc cancels)" + else: + hint = f"Type to filter {total} commands — ↑/↓ then Enter inserts, Esc cancels" + + labels = [f"{c} — {d}" if d else c for (c, _cat, d) in rows] + if not labels: + labels = ["(no matching commands)"] + box_width = _panel_box_width(title, [hint] + labels, min_width=50, max_width=90) + inner_text_width = max(8, box_width - 6) + selected = state.get("selected", 0) + try: + from prompt_toolkit.application import get_app + term_rows = get_app().output.get_size().rows + except Exception: + term_rows = shutil.get_terminal_size((100, 24)).lines + scroll_offset, visible = HermesCLI._compute_model_picker_viewport( + selected, state.get("_scroll_offset", 0), len(labels), term_rows, + ) + state["_scroll_offset"] = scroll_offset + + lines = [] + lines.append(('class:clarify-border', '╭─ ')) + lines.append(('class:clarify-title', title)) + lines.append(('class:clarify-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + _append_panel_line(lines, 'class:clarify-border', 'class:clarify-hint', hint, box_width) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + for idx in range(scroll_offset, min(scroll_offset + visible, len(labels))): + label = labels[idx] + style = 'class:clarify-selected' if idx == selected else 'class:clarify-choice' + prefix = '❯ ' if idx == selected else ' ' + for wrapped in _wrap_panel_text(prefix + label, inner_text_width, subsequent_indent=' '): + _append_panel_line(lines, 'class:clarify-border', style, wrapped, box_width) + _append_blank_panel_line(lines, 'class:clarify-border', box_width) + lines.append(('class:clarify-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_sudo_display_fragments(self): + from cli import _append_blank_panel_line, _append_panel_line, _panel_box_width + state = self._sudo_state + if not state: + return [] + title = '🔐 Sudo Password Required' + body = 'Enter password below (hidden), or press Enter to skip' + box_width = _panel_box_width(title, [body]) + lines = [] + lines.append(('class:sudo-border', '╭─ ')) + lines.append(('class:sudo-title', title)) + lines.append(('class:sudo-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) + _append_blank_panel_line(lines, 'class:sudo-border', box_width) + _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', body, box_width) + _append_blank_panel_line(lines, 'class:sudo-border', box_width) + lines.append(('class:sudo-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _get_secret_display_fragments(self): + from cli import _append_blank_panel_line, _append_panel_line, _panel_box_width + state = self._secret_state + if not state: + return [] + + title = '🔑 Skill Setup Required' + prompt = state.get("prompt") or f"Enter value for {state.get('var_name', 'secret')}" + metadata = state.get("metadata") or {} + help_text = metadata.get("help") + body = 'Enter secret below (hidden), ESC or Ctrl+C to skip' + content_lines = [prompt, body] + if help_text: + content_lines.insert(1, str(help_text)) + box_width = _panel_box_width(title, content_lines) + lines = [] + lines.append(('class:sudo-border', '╭─ ')) + lines.append(('class:sudo-title', title)) + lines.append(('class:sudo-border', ' ' + ('─' * max(0, box_width - len(title) - 3)) + '╮\n')) + _append_blank_panel_line(lines, 'class:sudo-border', box_width) + _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', prompt, box_width) + if help_text: + _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', str(help_text), box_width) + _append_blank_panel_line(lines, 'class:sudo-border', box_width) + _append_panel_line(lines, 'class:sudo-border', 'class:sudo-text', body, box_width) + _append_blank_panel_line(lines, 'class:sudo-border', box_width) + lines.append(('class:sudo-border', '╰' + ('─' * box_width) + '╯\n')) + return lines + + def _tui_hint_text(self): + if self._sudo_state: + remaining = max(0, int(self._sudo_deadline - time.monotonic())) + return [ + ('class:hint', ' password hidden · Enter to skip'), + ('class:clarify-countdown', f' ({remaining}s)'), + ] + + if self._secret_state: + remaining = max(0, int(self._secret_deadline - time.monotonic())) + return [ + ('class:hint', ' secret hidden · Enter to skip'), + ('class:clarify-countdown', f' ({remaining}s)'), + ] + + if self._approval_state: + remaining = max(0, int(self._approval_deadline - time.monotonic())) + return [ + ('class:hint', ' ↑/↓ to select, Enter to confirm'), + ('class:clarify-countdown', f' ({remaining}s)'), + ] + + if self._slash_confirm_state: + remaining = max(0, int(self._slash_confirm_deadline - time.monotonic())) + return [ + ('class:hint', ' type 1/2/3, or ↑/↓ to select, Enter to confirm'), + ('class:clarify-countdown', f' ({remaining}s)'), + ] + + if self._clarify_state: + # None deadline = unlimited wait → hide the countdown entirely. + if self._clarify_deadline is None: + countdown = '' + else: + remaining = max(0, int(self._clarify_deadline - time.monotonic())) + countdown = f' ({remaining}s)' + if self._clarify_freetext: + return [ + ('class:hint', ' type your answer and press Enter'), + ('class:clarify-countdown', countdown), + ] + if self._clarify_state.get("questions"): + return [ + ('class:hint', ' ↑/↓ to select, Enter to lock, Tab next question'), + ('class:clarify-countdown', countdown), + ] + return [ + ('class:hint', ' ↑/↓ to select, Enter to confirm'), + ('class:clarify-countdown', countdown), + ] + + if self._command_running: + frame = self._command_spinner_frame() + detail = "input temporarily disabled" if self._command_blocks_input else "input stays active; Enter queues" + return [ + ('class:hint', f' {frame} command in progress · {detail}'), + ] + + return [] + + def _tui_placeholder_text(self): + if self._voice_recording: + _label = self._voice_record_key_label() + return f"recording... {_label} to stop, Ctrl+C to cancel" + if self._voice_processing: + return "transcribing..." + if self._sudo_state: + return "type password (hidden), Enter to submit · ESC to skip" + if self._secret_state: + return "type secret (hidden), Enter to submit · ESC to skip" + if self._approval_state: + return "" + if self._slash_confirm_state: + return "type 1/2/3, or use ↑/↓ then Enter" + if self._clarify_freetext: + return "type your answer here and press Enter" + if self._clarify_state: + return "" + if self._command_running: + frame = self._command_spinner_frame() + status = self._command_status or "Processing command..." + return f"{frame} {status}" + if self._agent_running: + return "msg=interrupt · /queue · /bg · /steer · Ctrl+C cancel" + if self._voice_mode: + _label = self._voice_record_key_label() + return f"type or {_label} to record" + # Advertise a parked draft so the stash can never be silently + # forgotten — the composer itself tells you how to get it back. + _stash_hint = "" + try: + _stash_hint = self._prompt_stash.placeholder_hint() + except Exception: + _stash_hint = "" + if _stash_hint: + return _stash_hint + # Idle + empty composer: show a rotating task-oriented example to + # nudge the user toward a high-value first action (C-09). Chosen + # once per session (self._composer_placeholder) so it stays stable + # while being read, not flickering every render. + return getattr(self, "_composer_placeholder", "") or "" + + def _get_stash_panel_display_fragments(self): + try: + _stash = self._prompt_stash + return self._render_stash_panel( + _stash.panel_rows(), + _stash.panel_cursor, + self._get_tui_terminal_width(), + ) + except Exception: + return [] + + def _tui_handle_voice_record(self, event): + """Toggle voice recording when voice mode is active. + + IMPORTANT: This handler runs in prompt_toolkit's event-loop thread. + Any blocking call here (locks, sd.wait, disk I/O) freezes the + entire UI. All heavy work is dispatched to daemon threads. + """ + from cli import _DIM, _RST, _cprint, logger + if not self._voice_mode: + return + # Always allow STOPPING a recording (even when agent is running) + if self._voice_recording: + # Manual stop via push-to-talk key: stop continuous mode + with self._voice_lock: + self._voice_continuous = False + # Flag clearing is handled atomically inside _voice_stop_and_transcribe + event.app.invalidate() + threading.Thread( + target=self._voice_stop_and_transcribe, + daemon=True, + ).start() + else: + # Allow disarming continuous mode even when the agent is + # running or transcribing — otherwise the user is stuck in + # an auto-restart loop until /voice off (#67545). + if self._agent_running or self._voice_processing: + with self._voice_lock: + self._voice_continuous = False + event.app.invalidate() + return + # Guard: don't START recording during interactive prompts + if self._clarify_state or self._sudo_state or self._approval_state or self._slash_confirm_state: + return + + # Interrupt TTS if playing, so user can start talking. + # stop_playback() is fast (just terminates a subprocess); + # the stop event drains the streaming pipeline if one is live. + if not self._voice_tts_done.is_set(): + try: + logger.info("TTS CUT: record key handler cutting TTS") + from tools.tts_streaming import mark_speech_interrupted + mark_speech_interrupted() + if self._voice_tts_stop is not None: + self._voice_tts_stop.set() + from tools.voice_mode import stop_playback + stop_playback() + self._voice_tts_done.set() + except Exception: + pass + + with self._voice_lock: + self._voice_continuous = True + + # Dispatch to a daemon thread so play_beep(sd.wait), + # AudioRecorder.start(lock acquire), and config I/O + # never block the prompt_toolkit event loop. + def _start_recording(): + try: + self._voice_start_recording() + if hasattr(self, '_app') and self._app: + self._app.invalidate() + except Exception as e: + _cprint(f"\n{_DIM}Voice recording failed: {e}{_RST}") + + threading.Thread(target=_start_recording, daemon=True).start() + event.app.invalidate() + + def _tui_handle_ctrl_c(self, event): + """Handle Ctrl+C - cancel interactive prompts, interrupt agent, or exit. + + Priority: + 0. Cancel active voice recording + 1. Cancel active sudo/approval/clarify prompt + 2. Interrupt the running agent (first press) + 3. Force exit (second press within 2s, or when idle) + """ + from cli import _DIM, _RST, _cprint + now = time.time() + + # Cancel active voice recording. + # Run cancel() in a background thread to prevent blocking the + # event loop if AudioRecorder._lock or CoreAudio takes time. + _should_cancel_voice = False + _recorder_ref = None + with self._voice_lock: + if self._voice_recording and self._voice_recorder: + _recorder_ref = self._voice_recorder + self._voice_recording = False + self._voice_continuous = False + _should_cancel_voice = True + if _should_cancel_voice: + _cprint(f"\n{_DIM}Recording cancelled.{_RST}") + threading.Thread( + target=_recorder_ref.cancel, daemon=True + ).start() + event.app.invalidate() + return + + # Cancel slash confirmation prompt (foreground UI, not an + # agent-blocking overlay — cancel and stop here). + if self._slash_confirm_state: + self._submit_slash_confirm_response("cancel") + event.app.current_buffer.reset() + event.app.invalidate() + return + + # Cancel /model picker (foreground UI — cancel and stop here). + if self._model_picker_state: + self._close_model_picker() + event.app.current_buffer.reset() + event.app.invalidate() + return + + # Cancel command palette (foreground UI — cancel and stop here). + if self._command_palette_state: + self._close_command_palette() + event.app.current_buffer.reset() + event.app.invalidate() + return + + # Clear all agent-blocking overlays (approval/clarify/sudo/secret) + # in one shot. We do NOT return after clearing — we fall through so + # that if the agent is also running we fire the interrupt on the same + # Ctrl+C press. This fixes the case where a stale/orphaned overlay + # (left behind by a previous interrupt) consumes the press without + # ever reaching the agent-interrupt branch, leaving the chat frozen + # (#14026). + _overlay_cleared = bool( + self._sudo_state + or self._secret_state + or self._approval_state + or self._clarify_state + ) + if _overlay_cleared: + self._clear_active_overlays_for_interrupt() + event.app.current_buffer.reset() + event.app.invalidate() + + # If we only cleared overlays and the agent is NOT running, stop here + # (don't fall through to the interrupt/exit path). + if _overlay_cleared and not (self._agent_running and self.agent): + return + + if self._agent_running and self.agent: + if now - self._last_ctrl_c_time < 2.0: + print("\n⚡ Force exiting...") + self._should_exit = True + event.app.exit() + return + + self._last_ctrl_c_time = now + print("\n⚡ Interrupting agent... (press Ctrl+C again to force exit)") + request_hard_interrupt(self.agent) + # If there's text or images, clear them (like bash). + # If everything is already empty, exit. + elif event.app.current_buffer.text or self._attached_images: + event.app.current_buffer.reset() + self._attached_images.clear() + event.app.invalidate() + else: + self._should_exit = True + event.app.exit() + + def _tui_handle_ctrl_q(self, event): + """Alternative interrupt/exit shortcut (Ctrl+Q). + + Behaves like Ctrl+C: cancels active prompts, interrupts the + running agent, or clears the input buffer. Does not support + the double-press 'force exit' feature of Ctrl+C. + """ + from cli import _DIM, _RST, _cprint + # Cancel active voice recording. + _should_cancel_voice = False + _recorder_ref = None + with self._voice_lock: + if self._voice_recording and self._voice_recorder: + _recorder_ref = self._voice_recorder + self._voice_recording = False + self._voice_continuous = False + _should_cancel_voice = True + if _should_cancel_voice: + _cprint(f"\n{_DIM}Recording cancelled.{_RST}") + threading.Thread( + target=_recorder_ref.cancel, daemon=True + ).start() + event.app.invalidate() + return + + # Cancel slash confirmation prompt (foreground UI — cancel and stop). + if self._slash_confirm_state: + self._submit_slash_confirm_response("cancel") + event.app.current_buffer.reset() + event.app.invalidate() + return + + # Cancel /model picker (foreground UI — cancel and stop). + if self._model_picker_state: + self._close_model_picker() + event.app.current_buffer.reset() + event.app.invalidate() + return + + # Clear all agent-blocking overlays in one shot, then fall through to + # the agent-interrupt branch so a single Ctrl+Q both clears a stale + # overlay and interrupts a still-running agent (#14026). + _overlay_cleared = bool( + self._sudo_state + or self._secret_state + or self._approval_state + or self._clarify_state + ) + if _overlay_cleared: + self._clear_active_overlays_for_interrupt() + event.app.current_buffer.reset() + event.app.invalidate() + + if _overlay_cleared and not (self._agent_running and self.agent): + return + + if self._agent_running and self.agent: + print("\n⚡ Interrupting agent...") + request_hard_interrupt(self.agent) + elif event.app.current_buffer.text or self._attached_images: + event.app.current_buffer.reset() + self._attached_images.clear() + event.app.invalidate() + else: + self._should_exit = True + event.app.exit() + + def _tui_make_clarify_number_handler(self, idx): + def handler(event): + if self._clarify_state and not self._clarify_freetext: + choices = self._clarify_state.get("choices") or [] + # multi-select support: number keys toggle checkboxes instead of submitting + if self._clarify_state.get("multi_select"): + if idx < len(choices): + indices = self._clarify_state.get("selected_indices", set()) + if idx in indices: + indices.discard(idx) + else: + indices.add(idx) + event.app.invalidate() + elif idx == len(choices): + # Toggle "Other" in multi-select mode + indices = self._clarify_state.get("selected_indices", set()) + if idx in indices: + indices.discard(idx) + else: + indices.add(idx) + event.app.invalidate() + return + # Original single-select: number keys submit directly + # Map index to choice (treating "Other" as the last option) + if idx < len(choices): + # Batch mode: lock the numbered choice for the active + # question instead of resolving the whole prompt. + if self._clarify_state.get("questions"): + self._clarify_batch_lock(self._clarify_state, choices[idx]) + event.app.invalidate() + return + # Select a numbered choice + self._clarify_state["response_queue"].put(choices[idx]) + self._clarify_state = None + self._clarify_freetext = False + event.app.invalidate() + elif idx == len(choices): + # Select "Other" option + self._clarify_freetext = True + event.app.invalidate() + return handler + + def _tui_restore_stash_payload(self, event, payload) -> None: + """Put a popped (text, images) payload back into the composer.""" + if not payload: + return + text, images = payload + buf = event.app.current_buffer + buf.text = text + buf.cursor_position = len(text) + if images: + # Restore attachments the draft was carrying. Extend rather + # than replace: the user may have attached something new since + # the stash was taken and silently dropping it would be data + # loss. + for img in images: + if img not in self._attached_images: + self._attached_images.append(img) + + def _tui_handle_stash_panel_up(self, event): + self._prompt_stash.move_cursor(-1) + event.app.invalidate() + + def _tui_handle_stash_panel_down(self, event): + self._prompt_stash.move_cursor(1) + event.app.invalidate() + + def _tui_handle_stash_panel_delete(self, event): + """D in the browse panel discards the highlighted draft.""" + self._prompt_stash.delete_at_cursor() + event.app.invalidate() + + def _tui_handle_stash_panel_close(self, event): + self._prompt_stash.close_panel() + event.app.invalidate() + + def _tui_handle_tab(self, event): + """Tab: accept completion, auto-suggestion, or start completions. + + Priority: + 1. Completion menu open → accept selected completion + 2. Ghost text suggestion available → accept auto-suggestion + 3. Otherwise → start completion menu + + After accepting a provider like 'anthropic:', the completion menu + closes and complete_while_typing doesn't fire (no keystroke). + This binding re-triggers completions so stage-2 models appear + immediately. + """ + buf = event.current_buffer + if buf.complete_state: + # Completion menu is open — accept the selection + completion = buf.complete_state.current_completion + if completion is None: + # Menu open but nothing selected — select first then grab it + buf.go_to_completion(0) + completion = buf.complete_state and buf.complete_state.current_completion + if completion is None: + return + # Accept the selected completion + buf.apply_completion(completion) + elif buf.suggestion and buf.suggestion.text: + # No completion menu, but there's a ghost text auto-suggestion — accept it + buf.insert_text(buf.suggestion.text) + else: + # No menu and no suggestion — start completions from scratch + buf.start_completion() + + def _tui_handle_double_escape(self, event): + """Double ESC: discard the current draft and any attached images. + + Matches Claude Code / Gemini CLI, where double-Esc is the + clear-the-composer gesture. It works while the agent is + streaming, which is the gap Ctrl+C leaves: Ctrl+C interrupts a + running turn and only clears the draft when idle, so mid-stream + there was no way to discard a half-typed prompt. + + The draft is appended to history first, so Up recalls it — the + same undo affordance Claude Code provides, and the reason this + is safe to bind to a key pressed by reflex. + + Single ESC is the prefix for Alt sequences (escape+enter, + escape+g, escape+v), so prompt_toolkit's escape-timeout keeps + those distinct from the double press. Modal prompts bind ESC + eagerly and are excluded here so cancel still wins. + """ + buf = event.app.current_buffer + if not (buf.text or self._attached_images): + return + buf.reset(append_to_history=bool(buf.text)) + self._attached_images.clear() + event.app.invalidate() + + def _tui_handle_ignored_terminal_sequence(self, event): + """Consume parser-level ignored terminal sequences before self-insert. + + install_ignored_terminal_sequences() in hermes_cli.pt_input_extras + registers focus reports (CSI I / CSI O) as Keys.Ignore at the + VT100 parser level. Without this no-op binding the default + self-insert path would still fire and the bytes would land in + the buffer. + + Focus-in (CSI I) additionally schedules a rate-limited full + repaint: while the tab/window was hidden the emulator may have + coalesced output or repainted the surface, so prompt_toolkit's + incremental diff would stack a fresh copy of the prompt chrome + on top of the stale one (#60920 focus-regain variant, #25337). + """ + try: + for press in getattr(event, "key_sequence", None) or (): + if getattr(press, "data", None) == "\x1b[I": + self._schedule_focus_regain_redraw() + break + except Exception: + pass + return None + + def _tui_handle_escape_modal(self, event): + """ESC cancels active secret/sudo prompts.""" + if self._secret_state: + self._cancel_secret_capture() + event.app.current_buffer.reset() + event.app.invalidate() + return + if self._sudo_state: + self._sudo_state["response_queue"].put("") + self._sudo_state = None + event.app.invalidate() + return + if self._slash_confirm_state: + self._submit_slash_confirm_response("cancel") + event.app.current_buffer.reset() + event.app.invalidate() + return + + def _tui_handle_ctrl_z(self, event): + """Handle Ctrl+Z - suspend process to background (Unix only).""" + from cli import _DIM, _RST, _cprint + if sys.platform == 'win32': + _cprint(f"\n{_DIM}Suspend (Ctrl+Z) is not supported on Windows.{_RST}") + event.app.invalidate() + return + import signal as _sig + from prompt_toolkit.application import run_in_terminal + from hermes_cli.skin_engine import get_active_skin + agent_name = get_active_skin().get_branding("agent_name", "Hermes Agent") + msg = f"\n{agent_name} has been suspended. Run `fg` to bring {agent_name} back." + def _suspend(): + os.write(1, msg.encode()) + os.kill(0, _sig.SIGTSTP) + run_in_terminal(_suspend) + + def _tui_handle_ctrl_d(self, event): + """Ctrl+D: delete char under cursor (standard readline behaviour). + Only exit when the input is empty — same as bash/zsh. Pending + attached images count as input and block the EOF-exit so the + user doesn't lose them silently. + """ + buf = event.app.current_buffer + if buf.text: + buf.delete() + elif self._attached_images: + # Empty text but pending attachments — no-op, don't exit. + return + else: + self._should_exit = True + event.app.exit() + + def _tui_recall_without_recollapse(self, buf, move): + """Run a history-navigation move, suppressing paste-collapse. + + Recalled history can hold the full text of a paste that was + collapsed to a placeholder at submit time. Loading it back into the + buffer looks exactly like a fresh large paste to ``_on_text_changed`` + and would be re-collapsed. Set the skip flag around the move; if the + move didn't change the text (plain cursor movement), clear the flag + so a later real paste still collapses. + """ + before = buf.text + self._skip_paste_collapse = True + move() + if buf.text == before: + self._skip_paste_collapse = False + + def _tui_handle_alt_v(self, event): + """Alt+V — paste image from clipboard. + + Alt key combos pass through all terminal emulators (sent as + ESC + key), unlike Ctrl+V which terminals intercept for text + paste. This is the reliable way to attach clipboard images + on WSL2, VSCode, and any terminal over SSH where Ctrl+V + can't reach the application for image-only clipboard. + """ + if self._try_attach_clipboard_image(): + event.app.invalidate() + else: + # No image found — show a hint + pass # silent when no image (avoid noise on accidental press) + + def _tui_handle_ctrl_v(self, event): + """Fallback image paste for terminals without bracketed paste. + + On Linux terminals (GNOME Terminal, Konsole, etc.), Ctrl+V + sends raw byte 0x16 instead of triggering a paste. This + binding catches that and checks the clipboard for images. + On terminals that DO intercept Ctrl+V for paste (macOS + Terminal, iTerm2, VSCode, Windows Terminal), the bracketed + paste handler fires instead and this binding never triggers. + """ + if self._try_attach_clipboard_image(): + event.app.invalidate() + + def _tui_handle_ctrl_l(self, event): + """Ctrl+L: force a clean full-screen repaint. + + Recovers the UI after external terminal buffer drift — tmux / + cmux tab switches, ``clear`` from a subshell, SSH window + restores, etc. — that prompt_toolkit can't detect on its own. + Matches the universal bash/zsh/fish/vim/htop convention. + """ + self._force_full_redraw() + + def _tui_insert_newline(self, event): + """Insert a newline for multi-line input (Alt+Enter, and Ctrl+J/Ctrl+Enter + when multiline shortcuts are on). + + Alt+Enter works on mac/Linux/WSL. On Windows Terminal that keystroke is + intercepted at the terminal layer (toggles fullscreen) and never reaches + here — Windows users get newline via Ctrl+Enter, which WT delivers as c-j. + """ + event.current_buffer.insert_text('\n') + + def _tui_handle_open_in_editor(self, event): + """Ctrl+G (or Alt+G in VSCode/Cursor) opens the current draft in an external editor.""" + self._open_external_editor(event.current_buffer) + + def _tui_model_picker_down(self, event): + state = self._model_picker_state + if not state: + return + if state.get("stage") == "provider": + max_idx = len(state.get("providers") or []) + else: + # +1 for "← Back" and Cancel over the filtered visible rows. + _fp = state.get("_filtered_pairs") + _visible = len(_fp) if _fp is not None else len(state.get("model_list") or []) + max_idx = _visible + 1 + state["selected"] = min(max_idx, state.get("selected", 0) + 1) + event.app.invalidate() + + def _tui_model_picker_up(self, event): + if self._model_picker_state: + self._model_picker_state["selected"] = max(0, self._model_picker_state.get("selected", 0) - 1) + event.app.invalidate() + + def _tui_model_picker_escape(self, event): + """ESC clears an active filter first, else closes the picker.""" + st = self._model_picker_state + if st and st.get("stage") == "model" and (st.get("filter") or ""): + st["filter"] = "" + st["selected"] = 0 + st["_scroll_offset"] = 0 + event.app.invalidate() + return + self._close_model_picker() + event.app.current_buffer.reset() + event.app.invalidate() + + def _tui_model_picker_filter_backspace(self, event): + st = self._model_picker_state + if not st: + return + cur = st.get("filter", "") or "" + st["filter"] = cur[:-1] + st["selected"] = 0 + st["_scroll_offset"] = 0 + event.app.invalidate() + + def _tui_make_model_filter_char_handler(self, ch: str): + def handler(event): + st = self._model_picker_state + if not st or st.get("stage") != "model": + return + st["filter"] = (st.get("filter", "") or "") + ch + st["selected"] = 0 + st["_scroll_offset"] = 0 + event.app.invalidate() + return handler + + def _tui_make_palette_char_handler(self, ch: str): + def handler(event): + st = self._command_palette_state + if not st: + return + st["filter"] = (st.get("filter", "") or "") + ch + st["selected"] = 0 + st["_scroll_offset"] = 0 + event.app.invalidate() + return handler + + def _tui_make_approval_number_handler(self, idx): + def handler(event): + if self._approval_state and idx < len(self._approval_state["choices"]): + self._approval_state["selected"] = idx + self._handle_approval_selection() + event.app.invalidate() + return handler + + def _tui_make_slash_confirm_number_handler(self, idx): + def handler(event): + if self._slash_confirm_state and idx < len(self._slash_confirm_state.get("choices") or []): + choice = self._slash_confirm_state["choices"][idx][0] + self._submit_slash_confirm_response(choice) + event.app.current_buffer.reset() + event.app.invalidate() + return handler + + def _tui_clarify_toggle(self, event): + if self._clarify_state: + selected = self._clarify_state["selected"] + indices = self._clarify_state.get("selected_indices", set()) + if selected in indices: + indices.discard(selected) + else: + indices.add(selected) + event.app.invalidate() + + def _tui_clarify_down(self, event): + """Move selection down in clarify choices.""" + if self._clarify_state: + choices = self._clarify_state.get("choices") or [] + max_idx = len(choices) # last index is the "Other" option + self._clarify_state["selected"] = min(max_idx, self._clarify_state["selected"] + 1) + event.app.invalidate() + + def _tui_clarify_up(self, event): + """Move selection up in clarify choices.""" + if self._clarify_state: + self._clarify_state["selected"] = max(0, self._clarify_state["selected"] - 1) + event.app.invalidate() + + def _tui_clarify_batch_tab(self, event): + state = self._clarify_state + if state and state.get("questions"): + self._clarify_batch_set_active( + state, (state["active"] + 1) % len(state["questions"]) + ) + event.app.invalidate() + + def _tui_clarify_batch_backtab(self, event): + state = self._clarify_state + if state and state.get("questions"): + self._clarify_batch_set_active( + state, (state["active"] - 1) % len(state["questions"]) + ) + event.app.invalidate() + + def _tui_command_palette_backspace(self, event): + st = self._command_palette_state + if st: + st["filter"] = (st.get("filter", "") or "")[:-1] + st["selected"] = 0 + st["_scroll_offset"] = 0 + event.app.invalidate() + + def _tui_command_palette_down(self, event): + st = self._command_palette_state + if st: + n = st.get("_visible_count", len(self._command_palette_visible_entries())) + st["selected"] = min(max(0, n - 1), st.get("selected", 0) + 1) + event.app.invalidate() + + def _tui_command_palette_up(self, event): + st = self._command_palette_state + if st: + st["selected"] = max(0, st.get("selected", 0) - 1) + event.app.invalidate() + + def _tui_command_palette_enter(self, event): + self._handle_command_palette_selection() + event.app.invalidate() + + def _tui_command_palette_escape(self, event): + self._close_command_palette() + event.app.invalidate() + + def _tui_open_command_palette(self, event): + self._open_command_palette() + event.app.invalidate() + + def _tui_slash_confirm_down(self, event): + if self._slash_confirm_state: + max_idx = len(self._slash_confirm_state.get("choices") or []) - 1 + self._slash_confirm_state["selected"] = min(max_idx, self._slash_confirm_state.get("selected", 0) + 1) + event.app.invalidate() + + def _tui_slash_confirm_up(self, event): + if self._slash_confirm_state: + self._slash_confirm_state["selected"] = max(0, self._slash_confirm_state.get("selected", 0) - 1) + event.app.invalidate() + + def _tui_approval_down(self, event): + if self._approval_state: + max_idx = len(self._approval_state["choices"]) - 1 + self._approval_state["selected"] = min(max_idx, self._approval_state["selected"] + 1) + event.app.invalidate() + + def _tui_approval_up(self, event): + if self._approval_state: + self._approval_state["selected"] = max(0, self._approval_state["selected"] - 1) + event.app.invalidate() + + def _tui_wake_startup(self): + from cli import logger + try: + self._maybe_start_wake_word() + except Exception as e: + logger.debug("wake-word startup skipped: %s", e) + + def _tui_suppress_closed_loop_errors(self, loop, context): + exc = context.get("exception") + if isinstance(exc, RuntimeError) and "Event loop is closed" in str(exc): + return # silently suppress + if isinstance(exc, KeyError) and "is not registered" in str(exc): + return # suppress selector registration failures (#6393) + if isinstance(exc, OSError) and getattr(exc, "errno", None) == errno.EIO: + return # suppress I/O errors from broken stdout on interrupt (#13710) + # Fall back to default handler for everything else + loop.default_exception_handler(context) + + def _tui_handle_enter(self, event): + """Handle Enter key - submit input. + + Routes to the correct queue based on active UI state: + - Sudo password prompt: password goes to sudo response queue + - Approval selection: selected choice goes to approval response queue + - Clarify freetext mode: answer goes to the clarify response queue + - Clarify choice mode: selected choice goes to the clarify response queue + - Agent running: goes to _interrupt_queue (chat() monitors this) + - Agent idle: goes to _pending_input (process_loop monitors this) + Commands (starting with /) always go to _pending_input so they're + handled as commands, not sent as interrupt text to the agent. + """ + from cli import ( + CLI_CONFIG, + _ACCENT, + _DIM, + _RST, + _apply_backslash_line_continuation, + _cprint, + _hermes_home, + _is_backslash_line_continuation, + _looks_like_slash_command, + ) + # --- Sudo password prompt: submit the typed password --- + if self._sudo_state: + text = event.app.current_buffer.text + self._sudo_state["response_queue"].put(text) + self._sudo_state = None + event.app.invalidate() + return + + # --- Secret prompt: submit the typed secret --- + if self._secret_state: + text = event.app.current_buffer.text + self._submit_secret_response(text) + event.app.current_buffer.reset() + event.app.invalidate() + return + + # --- Approval selection: confirm the highlighted choice --- + if self._approval_state: + self._handle_approval_selection() + event.app.invalidate() + return + + # --- Slash-command confirmation: submit typed or highlighted choice --- + if self._slash_confirm_state: + text = event.app.current_buffer.text.strip() + choices = self._slash_confirm_state.get("choices") or [] + choice = self._normalize_slash_confirm_choice(text, choices) if text else None + if choice is None: + selected = self._slash_confirm_state.get("selected", 0) + if 0 <= selected < len(choices): + choice = choices[selected][0] + self._submit_slash_confirm_response(choice or "cancel") + event.app.current_buffer.reset() + event.app.invalidate() + return + + # --- /model picker modal --- + if self._model_picker_state: + try: + # Picker selections follow the same session-scoped default + # as /model ; honour model.persist_switch_by_default. + from hermes_cli.model_switch import resolve_persist_behavior + + self._handle_model_picker_selection( + persist_global=resolve_persist_behavior(False, False) + ) + except Exception as _exc: + _cprint(f" ✗ Model selection failed: {_exc}") + self._close_model_picker() + event.app.current_buffer.reset() + event.app.invalidate() + return + + # --- Clarify freetext mode: user typed their own answer --- + if self._clarify_freetext and self._clarify_state: + text = event.app.current_buffer.text.strip() + if text: + state = self._clarify_state + # Batch mode: lock the typed answer for the active question + if state.get("questions"): + base = getattr(self, '_clarify_multi_base', None) + if base is not None: + # Multi-select "Other": append the typed answer to + # the checked labels as a JSON array string. + answer = json.dumps(base + [text], ensure_ascii=False) + meta = {"kind": "multi", "choices": list(base), "other_text": text} + self._clarify_multi_base = None + else: + answer = text + meta = {"kind": "other", "other_text": text} + self._clarify_freetext = False + self._clarify_prefill = "" + self._clarify_batch_lock(state, answer, meta=meta) + event.app.current_buffer.reset() + event.app.invalidate() + return + # multi-select: prepend previously checked real choices + base = getattr(self, '_clarify_multi_base', None) + if base: + text = ", ".join(base) + ", " + text + self._clarify_multi_base = None + self._clarify_state["response_queue"].put(text) + self._clarify_state = None + self._clarify_freetext = False + event.app.current_buffer.reset() + event.app.invalidate() + return + + # --- Clarify choice mode: confirm the highlighted selection --- + if self._clarify_state and not self._clarify_freetext: + state = self._clarify_state + # Batch mode: Enter locks the active question's answer and + # advances to the next unanswered question. + if state.get("questions"): + self._clarify_batch_enter(state) + # Editing an earlier "Other" answer: prefill the composer + # with the previously typed text. + if self._clarify_freetext and self._clarify_prefill: + event.app.current_buffer.text = self._clarify_prefill + event.app.current_buffer.cursor_position = len(self._clarify_prefill) + self._clarify_prefill = "" + event.app.invalidate() + return + selected = state["selected"] + choices = state.get("choices") or [] + # multi-select support: submit comma-joined list of checked choices + if state.get("multi_select"): + indices = state.get("selected_indices") + if not indices: + # Nothing checked → submit empty string (parses to []) + state["response_queue"].put("") + self._clarify_state = None + event.app.invalidate() + return + sorted_idx = sorted(indices) + selected_choices = [choices[i] for i in sorted_idx if i < len(choices)] + other_checked = len(choices) in sorted_idx + if other_checked and selected_choices: + # "Other" + real choices: store base choices, switch to freetext + # so the user can type a custom answer that gets appended + self._clarify_multi_base = selected_choices + self._clarify_freetext = True + event.app.invalidate() + return + if selected_choices: + state["response_queue"].put(", ".join(selected_choices)) + self._clarify_state = None + event.app.invalidate() + return + # Only "Other" was checked → switch to freetext + self._clarify_freetext = True + event.app.invalidate() + return + # Original single-select behavior: submit the highlighted choice + if selected < len(choices): + state["response_queue"].put(choices[selected]) + self._clarify_state = None + event.app.invalidate() + else: + # "Other" selected → switch to freetext + self._clarify_freetext = True + event.app.invalidate() + return + + # --- Normal input routing --- + raw_text = event.app.current_buffer.text + if ( + self._tui_multiline_shortcuts + and event.app.current_buffer.cursor_position == len(raw_text) + and _is_backslash_line_continuation(raw_text) + ): + continued = _apply_backslash_line_continuation(raw_text) + event.app.current_buffer.text = continued + event.app.current_buffer.cursor_position = len(continued) + event.app.invalidate() + return + text = raw_text.strip() + has_images = bool(self._attached_images) + if text or has_images: + # Handle /model directly on the UI thread so interactive pickers + # can safely use prompt_toolkit terminal handoff helpers. + if self._should_handle_model_command_inline(text, has_images=has_images): + if not self.process_command(text): + self._should_exit = True + if event.app.is_running: + event.app.exit() + event.app.current_buffer.reset(append_to_history=True) + # Force a repaint: process_command() prints through + # patch_stdout (scrolls output above the prompt) and never + # invalidates the app, so the just-cleared input area can + # keep showing the submitted text until some unrelated + # redraw fires. Every other early-return branch in this + # handler invalidates after reset — match them. + event.app.invalidate() + return + + # Handle /steer while the agent is running immediately on the + # UI thread. Queuing through _pending_input would deadlock the + # steer until after the agent loop finishes (process_loop is + # blocked inside self.chat()), which turns /steer into a + # post-run next-turn message — defeating mid-run injection. + # agent.steer() is thread-safe (holds _pending_steer_lock). + if self._should_handle_steer_command_inline(text, has_images=has_images): + self.process_command(text) + event.app.current_buffer.reset(append_to_history=True) + # Force a repaint after clearing the buffer. /steer is + # dispatched mid-run while the agent streams output through + # patch_stdout; process_command() never invalidates the + # app, so without this the submitted "/steer " can + # linger in the input area (looking unsent) and invite an + # accidental re-submit. See issue #34569. + event.app.invalidate() + return + + # Same treatment for /bg and /btw while the agent is + # running. Queuing them defeats the entire point of the + # commands: process_loop is blocked inside self.chat(), so the + # side task would only start once the foreground turn it was + # meant to run alongside has already finished (#75221). The + # foreground turn is left alone: no interrupt, no steer. + if self._should_handle_background_command_inline( + text, has_images=has_images + ): + self.process_command(text) + event.app.current_buffer.reset(append_to_history=True) + # Repaint for the same reason as the /steer branch above: + # process_command() prints through patch_stdout and never + # invalidates the app, so the submitted text can linger in + # the input area looking unsent. + event.app.invalidate() + return + + # Snapshot and clear attached images + images = list(self._attached_images) + self._attached_images.clear() + event.app.invalidate() + # Bundle text + images as a tuple when images are present + payload = (text, images) if images else text + # A bang command is treated like a slash command while the + # agent is busy: it must never be routed into steer/redirect + # (which would inject `!git status` into the model's context as + # a prompt). It queues and runs locally once the loop drains. + _is_local_dispatch = bool(text) and ( + _looks_like_slash_command(text) or text.strip().startswith("!") + ) + if self._agent_running and not _is_local_dispatch: + _effective_mode = self.busy_input_mode + redirected = False + if _effective_mode == "steer": + # Route Enter through /steer — inject mid-run after the + # next tool call. Images can't ride along (steer only + # appends text), so fall back to queue when images are + # attached. If the agent lacks steer() or rejects the + # payload, also fall back to queue so nothing is lost. + if images or not text: + _effective_mode = "queue" + else: + accepted = False + try: + if self.agent is not None and hasattr(self.agent, "steer"): + accepted = bool(self.agent.steer(text)) + except Exception as exc: + _cprint(f" {_DIM}Steer failed ({exc}) — queued for next turn.{_RST}") + accepted = False + if accepted: + preview = text[:80] + ("..." if len(text) > 80 else "") + _cprint(f" {_ACCENT}⏩ Steered: '{preview}'{_RST}") + else: + _effective_mode = "queue" + if _effective_mode == "queue": + # Queue for the next turn instead of interrupting + self._pending_input.put(payload) + preview = text if text else f"[{len(images)} image{'s' if len(images) != 1 else ''} attached]" + _cprint(f" Queued for the next turn: {preview[:80]}{'...' if len(preview) > 80 else ''}") + elif _effective_mode == "interrupt": + if not images and text: + try: + if ( + self.agent is not None + and getattr( + self.agent, + "_supports_active_turn_redirect", + False, + ) + is True + and hasattr(self.agent, "redirect") + ): + redirected = bool(self.agent.redirect(text)) + except Exception: + redirected = False + if redirected: + preview = text[:80] + ("..." if len(text) > 80 else "") + _cprint(f" {_ACCENT}↪ Redirected current turn: '{preview}'{_RST}") + else: + # Compatibility path for older agents, multimodal + # follow-ups, or a turn that finished in the race. + self._interrupt_queue.put(payload) + try: + _dbg = _hermes_home / "interrupt_debug.log" + with open(_dbg, "a", encoding="utf-8") as _f: + _f.write(f"{time.strftime('%H:%M:%S')} ENTER: queued interrupt msg={str(payload)[:60]!r}, " + f"agent_running={self._agent_running}\n") + except Exception: + pass + # First-touch onboarding: on the very first busy-while-running + # event for this install, print a one-line tip explaining the + # /busy knob. Flag persists to config.yaml and never fires + # again. Guarded for exceptions so onboarding can't break + # the input loop. + try: + from agent.onboarding import ( + BUSY_INPUT_FLAG, + busy_input_hint_cli, + is_seen, + mark_seen, + ) + if not is_seen(CLI_CONFIG, BUSY_INPUT_FLAG): + _hint_mode = "redirect" if redirected else _effective_mode + _cprint(f" {_DIM}{busy_input_hint_cli(_hint_mode)}{_RST}") + mark_seen(_hermes_home / "config.yaml", BUSY_INPUT_FLAG) + CLI_CONFIG.setdefault("onboarding", {}).setdefault("seen", {})[BUSY_INPUT_FLAG] = True + except Exception: + pass + else: + self._pending_input.put(payload) + # History stores real pasted content, not the placeholder, so + # up-arrow recall restores the actual text. + self._inline_pastes(event.app.current_buffer) + event.app.current_buffer.reset(append_to_history=True) + + def _tui_handle_paste(self, event): + """Handle terminal paste — detect clipboard images. + + When the terminal supports bracketed paste, Ctrl+V / Cmd+V + triggers this with the pasted text. We only auto-attach a + clipboard image for image-only/empty paste gestures so text + pastes and dictation do not accidentally attach stale images. + + Large pastes (5+ lines) are collapsed to a file reference + placeholder while preserving any existing user text in the + buffer. + """ + from cli import ( + _hermes_home, + _should_auto_attach_clipboard_image_on_paste, + _strip_leaked_bracketed_paste_wrappers, + _strip_leaked_terminal_responses_with_meta, + datetime, + logger, + ) + # Diagnostic canary: measure how long the paste handler blocks + # the prompt_toolkit event loop. If this exceeds ~500ms we log + # it so recurring "CLI freezes on paste" reports (issue #16263, + # macOS Tahoe 26 + iTerm2/Ghostty) arrive with data attached. + _paste_handler_start = time.perf_counter() + _paste_raw_size = len(event.data or "") + pasted_text = event.data or "" + # Normalise line endings — Windows \r\n and old Mac \r both become \n + # so the 5-line collapse threshold and display are consistent. + pasted_text = pasted_text.replace('\r\n', '\n').replace('\r', '\n') + pasted_text = _strip_leaked_bracketed_paste_wrappers(pasted_text) + pasted_text, _had_mouse_reports = _strip_leaked_terminal_responses_with_meta(pasted_text) + if _had_mouse_reports: + self._recover_terminal_input_modes(reason="mouse reports leaked into bracketed paste payload") + if _should_auto_attach_clipboard_image_on_paste(pasted_text) and self._try_attach_clipboard_image(): + event.app.invalidate() + if pasted_text: + # Sanitize surrogate characters (e.g. from Word/Google Docs paste) before writing + from run_agent import _sanitize_surrogates + pasted_text = _sanitize_surrogates(pasted_text) + line_count = pasted_text.count('\n') + buf = event.current_buffer + threshold = self.config.get("paste_collapse_threshold", 5) + char_threshold = self.config.get("paste_collapse_char_threshold", 2000) + lines_hit = threshold > 0 and line_count >= threshold + chars_hit = char_threshold > 0 and len(pasted_text) >= char_threshold + if (lines_hit or chars_hit) and not buf.text.strip().startswith('/'): + self._tui_paste_counter[0] += 1 + paste_dir = _hermes_home / "pastes" + paste_dir.mkdir(parents=True, exist_ok=True) + paste_file = paste_dir / f"paste_{self._tui_paste_counter[0]}_{datetime.now().strftime('%H%M%S')}.txt" + paste_file.write_text(pasted_text, encoding="utf-8") + logger.info("Collapsed paste #%d: %d lines, %d chars -> %s", self._tui_paste_counter[0], line_count + 1, len(pasted_text), paste_file) + placeholder = f"[Pasted text #{self._tui_paste_counter[0]}: {line_count + 1} lines \u2192 {paste_file}]" + prefix = "" + if buf.cursor_position > 0 and buf.text[buf.cursor_position - 1] != '\n': + prefix = "\n" + self._tui_paste_just_collapsed[0] = True + buf.insert_text(prefix + placeholder) + else: + buf.insert_text(pasted_text) + _paste_handler_elapsed_ms = (time.perf_counter() - _paste_handler_start) * 1000.0 + if _paste_handler_elapsed_ms > 500.0: + logger.warning( + "Slow bracketed-paste handler: %.1fms to process %d bytes " + "(%d lines) on %s. If the input becomes unresponsive after " + "this, attach this log line to the bug report.", + _paste_handler_elapsed_ms, + _paste_raw_size, + pasted_text.count('\n') + 1 if pasted_text else 0, + sys.platform, + ) + + def _tui_on_text_changed(self, buf): + """Detect large pastes and collapse them to a file reference. + + When bracketed paste is available, handle_paste collapses + large pastes directly. This handler is a fallback for + terminals without bracketed paste support. + + Two heuristics (either triggers collapse): + 1. Many characters added at once (chars_added > 1) — works + when the terminal delivers the paste in one event-loop tick. + 2. Newline count jumped by 4+ in a single text-change event — + catches terminals that feed characters individually but + still batch newlines. Alt+Enter only adds 1 newline per + event so it never triggers this. + """ + from cli import ( + _hermes_home, + _strip_leaked_bracketed_paste_wrappers, + _strip_leaked_terminal_responses_with_meta, + datetime, + logger, + ) + text = _strip_leaked_bracketed_paste_wrappers(buf.text) + text, _had_mouse_reports = _strip_leaked_terminal_responses_with_meta(text) + if _had_mouse_reports: + self._recover_terminal_input_modes(reason="mouse reports leaked into prompt buffer") + if text != buf.text: + cursor = min(buf.cursor_position, len(text)) + self._tui_paste_just_collapsed[0] = True + buf.text = text + buf.cursor_position = cursor + self._tui_prev_text_len[0] = len(text) + self._tui_prev_newline_count[0] = text.count('\n') + return + chars_added = len(text) - self._tui_prev_text_len[0] + self._tui_prev_text_len[0] = len(text) + if self._tui_paste_just_collapsed[0] or self._skip_paste_collapse: + self._tui_paste_just_collapsed[0] = False + self._skip_paste_collapse = False + self._tui_prev_newline_count[0] = text.count('\n') + return + line_count = text.count('\n') + newlines_added = line_count - self._tui_prev_newline_count[0] + self._tui_prev_newline_count[0] = line_count + is_paste = chars_added > 1 or newlines_added >= 4 + threshold = self.config.get("paste_collapse_threshold_fallback", 5) + char_threshold = self.config.get("paste_collapse_char_threshold", 2000) + lines_hit = threshold > 0 and line_count >= threshold + chars_hit = char_threshold > 0 and len(text) >= char_threshold + if (lines_hit or chars_hit) and is_paste and not text.startswith('/'): + self._tui_paste_counter[0] += 1 + paste_dir = _hermes_home / "pastes" + paste_dir.mkdir(parents=True, exist_ok=True) + paste_file = paste_dir / f"paste_{self._tui_paste_counter[0]}_{datetime.now().strftime('%H%M%S')}.txt" + paste_file.write_text(text, encoding="utf-8") + logger.info("Collapsed paste #%d: %d lines, %d chars -> %s (fallback)", self._tui_paste_counter[0], line_count + 1, len(text), paste_file) + self._tui_paste_just_collapsed[0] = True + buf.text = f"[Pasted text #{self._tui_paste_counter[0]}: {line_count + 1} lines \u2192 {paste_file}]" + buf.cursor_position = len(buf.text) + + def _tui_handle_prompt_stash(self, event): + """Ctrl+S: stash the current draft, or restore/browse a stashed one. + + - Composer has content → push it onto the stash and clear the input. + - Composer empty, one stashed draft → pop it straight back. + - Composer empty, several stashed → open the browse panel. + - Browse panel open → close it. + + Pushing onto a stack (rather than a single slot) is what makes + repeated Ctrl+S safe: a second stash never silently overwrites the + first, both stay reachable in the panel. + """ + from hermes_cli.prompt_stash import ( + ACTION_OPEN_PANEL, + ACTION_RESTORED, + ACTION_STASHED, + resolve_ctrl_s, + ) + + buf = event.app.current_buffer + action, payload = resolve_ctrl_s( + self._prompt_stash, buf.text, self._attached_images + ) + + if action == ACTION_STASHED: + # reset() (not `text = ""`) so completion state, selection, and + # the undo stack are cleared along with the text. + buf.reset() + self._attached_images.clear() + elif action == ACTION_RESTORED: + self._tui_restore_stash_payload(event, payload) + elif action == ACTION_OPEN_PANEL: + pass # resolve_ctrl_s already flipped panel_open + + event.app.invalidate() + + def _tui_handle_stash_panel_restore(self, event): + """Enter in the browse panel restores the highlighted draft.""" + payload = self._prompt_stash.restore_at_cursor() + self._tui_restore_stash_payload(event, payload) + event.app.invalidate() + + def _tui_history_up(self, event): + """Up arrow: browse history when on first line, else move cursor up.""" + buf = event.app.current_buffer + self._tui_recall_without_recollapse(buf, lambda: buf.auto_up(count=event.arg)) + + def _tui_history_down(self, event): + """Down arrow: browse history when on last line, else move cursor down.""" + buf = event.app.current_buffer + self._tui_recall_without_recollapse(buf, lambda: buf.auto_down(count=event.arg)) + + def _tui_image_bar_fragments(self): + from cli import _format_image_attachment_badges + if not self._attached_images: + return [] + badges = _format_image_attachment_badges( + self._attached_images, + self._image_counter, + ) + return [("class:image-badge", f" {badges} ")] + + def _tui_voice_status_fragments(self): + return self._get_voice_status_fragments() + + def _tui_spinner_text(self): + spinner_line = self._render_spinner_text() + if not spinner_line: + return [] + return [('class:hint', spinner_line)] + + def _tui_spinner_height(self): + return self._spinner_widget_height() + + def _tui_hint_height(self): + if self._sudo_state or self._secret_state or self._approval_state or self._slash_confirm_state or self._clarify_state or self._command_running: + return 1 + # Keep a spacer while the agent runs on roomy terminals, but reclaim + # the row on narrow/mobile screens where every line matters. + return self._agent_spacer_height() + + def _tui_init_run_state(self): + """Reset the per-run REPL state (queues, modal states, voice state, config watcher).""" + # State for async operation + self._agent_running = False + self._pending_input = queue.Queue() # For normal input (commands + new queries) + self._interrupt_queue = queue.Queue() # For messages typed while agent is running + # Seeded -q handoff: main() can't put directly into _pending_input + # (this reinit would discard it), so the seeded first message rides + # in on an attribute and is enqueued into the fresh queue here. + _seed_msg = getattr(self, "_seeded_first_message", None) + if _seed_msg is not None: + self._seeded_first_message = None + self._pending_input.put(_seed_msg) + # See constructor note. Mirrored here for the run() path that skips + # the earlier __init__ branch. + self._last_turn_interrupted = False + self._should_exit = False + self._last_ctrl_c_time = 0 # Track double Ctrl+C for force exit + + # Give plugin manager a CLI reference so plugins can inject messages + from hermes_cli.plugins import get_plugin_manager + get_plugin_manager()._cli_ref = self + + # Config file watcher — detect mcp_servers changes and auto-reload + from hermes_cli.config import get_config_path as _get_config_path + _cfg_path = _get_config_path() + self._config_mtime: float = _cfg_path.stat().st_mtime if _cfg_path.exists() else 0.0 + self._config_mcp_servers: dict = self.config.get("mcp_servers") or {} + self._last_config_check: float = 0.0 # monotonic time of last check + + # Clarify tool state: interactive question/answer with the user. + # When the agent calls the clarify tool, _clarify_state is set and + # the prompt_toolkit UI switches to a selection mode. + self._clarify_state = None # dict with question, choices, selected, response_queue + self._clarify_freetext = False # True when user chose "Other" and is typing + self._clarify_deadline = 0 # monotonic timestamp when the clarify times out + + # Sudo password prompt state (similar mechanism to clarify) + self._sudo_state = None # dict with response_queue when active + self._sudo_deadline = 0 + self._modal_input_snapshot = None + + # Dangerous command approval state (similar mechanism to clarify) + self._approval_state = None # dict with command, description, choices, selected, response_queue + self._approval_deadline = 0 + self._approval_lock = threading.Lock() # serialize concurrent approval prompts (delegation race fix) + + # Destructive slash-command confirmation state (/new, /clear, /undo). + # These prompts are answered through the prompt_toolkit composer, not + # raw input(), so the option labels stay visible and Enter does not EOF + # the whole app. + self._slash_confirm_state = None + self._slash_confirm_deadline = 0 + + # Slash command loading state + self._command_running = False + self._command_blocks_input = False + self._command_status = "" + + # Secure secret capture state for skill setup + self._secret_state = None # dict with var_name, prompt, metadata, response_queue + self._secret_deadline = 0 + + # Clipboard image attachments (paste images into the CLI) + self._attached_images: list[Path] = [] + self._image_counter = 0 + + # Voice mode state (protected by _voice_lock for cross-thread access) + self._voice_lock = threading.Lock() + self._voice_mode = False # Whether voice mode is enabled + self._voice_tts = False # Whether TTS output is enabled + self._voice_recorder = None # AudioRecorder instance (lazy init) + self._voice_recording = False # Whether currently recording + self._voice_processing = False # Whether STT is in progress + self._voice_continuous = False # Whether to auto-restart after agent responds + self._voice_tts_done = threading.Event() # Signals TTS playback finished + self._voice_tts_done.set() # Initially "done" (no TTS pending) + self._voice_tts_stop = None # active streaming pipeline's stop event + self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption + self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780) + self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip + + if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": + self._install_tool_callbacks() + + if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": + self._ensure_tirith_security() + + def _tui_build_key_bindings(self): + """Build the prompt_toolkit KeyBindings for the REPL input area.""" + from cli import ( + CLI_CONFIG, + _bind_prompt_submit_keys, + _cli_multiline_shortcuts_enabled, + _preserve_ctrl_enter_newline, + logger, + ) + # Key bindings for the input area + kb = KeyBindings() + + _multiline_shortcuts_enabled = _cli_multiline_shortcuts_enabled(self.config or CLI_CONFIG) + self._tui_multiline_shortcuts = _multiline_shortcuts_enabled + + from prompt_toolkit.keys import Keys as _IgnoreKeys + + kb.add(_IgnoreKeys.Ignore, eager=True)(self._tui_handle_ignored_terminal_sequence) + + _bind_prompt_submit_keys( + kb, + self._tui_handle_enter, + multiline_shortcuts_enabled=_multiline_shortcuts_enabled, + ) + + kb.add('escape', 'enter')(self._tui_insert_newline) + + # Ctrl+J inserts a newline (matches Claude Code / Codex / OpenCode). + # Windows Terminal delivers Ctrl+Enter as the same c-j code, so this + # covers Ctrl+Enter there. display.cli_multiline_shortcuts: false + # restores legacy c-j submit on unusual POSIX PTYs where Enter is LF. + if _multiline_shortcuts_enabled or _preserve_ctrl_enter_newline(): + kb.add('c-j')(self._tui_insert_newline) + + # VSCode/Cursor bind Ctrl+G to "Find Next" at the editor level, so + # the keystroke never reaches the embedded terminal. Alt+G is unbound + # in those IDEs and arrives here as ('escape', 'g') — register it as + # a fallback so the editor handoff works inside Cursor/VSCode too. + _editor_filter = Condition( + lambda: not self._clarify_state and not self._approval_state and not self._sudo_state and not self._secret_state + ) + + kb.add('c-g', filter=_editor_filter)(kb.add('escape', 'g', filter=_editor_filter)(self._tui_handle_open_in_editor)) + + # --- Ctrl+S prompt stash ------------------------------------------- + # Park a half-written draft, send something else, then bring the draft + # back. Suppressed while a modal prompt owns the composer (sudo / + # secret / approval / clarify) so Ctrl+S can't stash a password. + _stash_filter = Condition( + lambda: not self._clarify_state + and not self._approval_state + and not self._sudo_state + and not self._secret_state + and not self._slash_confirm_state + and not self._model_picker_state + ) + _stash_panel_filter = Condition( + lambda: self._prompt_stash.panel_open and bool(len(self._prompt_stash)) + ) + + kb.add('c-s', filter=_stash_filter)(self._tui_handle_prompt_stash) + + kb.add('up', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_up) + + kb.add('down', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_down) + + kb.add('enter', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_restore) + + kb.add('d', filter=_stash_panel_filter, eager=True)(kb.add('D', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_delete)) + + kb.add('escape', filter=_stash_panel_filter, eager=True)(self._tui_handle_stash_panel_close) + + kb.add('tab', eager=True)(self._tui_handle_tab) + + # --- Clarify tool: arrow-key navigation for multiple-choice questions --- + + kb.add('up', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_clarify_up) + + kb.add('down', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_clarify_down) + + # multi-select support: Space toggles the checkbox at the current cursor position + kb.add('space', filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext and self._clarify_state.get("multi_select")))(self._tui_clarify_toggle) + + # Batch clarify: Tab cycles the active question (any-order answering; + # moving onto an answered question lets the user re-answer it before + # the batch completes). Registered after the generic tab handler so + # this filtered binding wins while the batch panel is open. + kb.add('tab', filter=Condition(lambda: bool(self._clarify_state) and bool(self._clarify_state.get("questions")) and not self._clarify_freetext), eager=True)(self._tui_clarify_batch_tab) + + # Shift-Tab walks backwards through the questions. + kb.add('s-tab', filter=Condition(lambda: bool(self._clarify_state) and bool(self._clarify_state.get("questions")) and not self._clarify_freetext), eager=True)(self._tui_clarify_batch_backtab) + + # Number keys for quick clarify selection (1-9, 0 for 10th item) + + for _num in range(10): + # 1-9 select items 0-8, 0 selects item 9 (10thitem) + _idx = 9 if _num == 0 else _num - 1 + kb.add(str(_num), filter=Condition(lambda: bool(self._clarify_state) and not self._clarify_freetext))(self._tui_make_clarify_number_handler(_idx)) + + # --- Dangerous command approval: arrow-key navigation --- + + kb.add('up', filter=Condition(lambda: bool(self._approval_state)))(self._tui_approval_up) + + kb.add('down', filter=Condition(lambda: bool(self._approval_state)))(self._tui_approval_down) + + # --- Slash-command confirmation: arrow-key navigation --- + kb.add('up', filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_slash_confirm_up) + + kb.add('down', filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_slash_confirm_down) + + # --- /model picker: arrow-key navigation --- + kb.add('up', filter=Condition(lambda: bool(self._model_picker_state)))(self._tui_model_picker_up) + + kb.add('down', filter=Condition(lambda: bool(self._model_picker_state)))(self._tui_model_picker_down) + + def _model_picker_typing_active() -> bool: + # Type-to-filter is only live on the model stage (concrete list). + st = self._model_picker_state + return bool(st) and st.get("stage") == "model" + + # Printable ASCII (space through ~) narrows the model list as you type. + import string as _string + for _ch in _string.digits + _string.ascii_letters + "-_.:/ ": + kb.add(_ch, filter=Condition(_model_picker_typing_active))( + self._tui_make_model_filter_char_handler(_ch) + ) + + kb.add('backspace', filter=Condition(_model_picker_typing_active))(self._tui_model_picker_filter_backspace) + + kb.add('escape', filter=Condition(lambda: bool(self._model_picker_state)), eager=True)(self._tui_model_picker_escape) + + # --- Ctrl+P command palette keybindings --- + def _palette_active() -> bool: + return bool(self._command_palette_state) + + kb.add('c-p', filter=Condition(lambda: not self._command_palette_state and not self._model_picker_state and not self._clarify_state and not self._approval_state and not self._slash_confirm_state and not self._sudo_state and not self._secret_state))(self._tui_open_command_palette) + + kb.add('up', filter=Condition(_palette_active))(self._tui_command_palette_up) + + kb.add('down', filter=Condition(_palette_active))(self._tui_command_palette_down) + + kb.add('enter', filter=Condition(_palette_active))(self._tui_command_palette_enter) + + kb.add('backspace', filter=Condition(_palette_active))(self._tui_command_palette_backspace) + + kb.add('escape', filter=Condition(_palette_active), eager=True)(self._tui_command_palette_escape) + + import string as _pstring + for _pch in _pstring.digits + _pstring.ascii_letters + "-_.:/ ": + kb.add(_pch, filter=Condition(_palette_active))(self._tui_make_palette_char_handler(_pch)) + + # Number keys for quick approval selection (1-9, 0 for 10th item) + + for _num in range(10): + # 1-9 select items 0-8, 0 selects item 9 (10th item) + _idx = 9 if _num == 0 else _num - 1 + kb.add(str(_num), filter=Condition(lambda: bool(self._approval_state)))(self._tui_make_approval_number_handler(_idx)) + + # Number keys for quick slash-confirm selection (1-9, 0 for 10th item) + + for _num in range(10): + _idx = 9 if _num == 0 else _num - 1 + kb.add(str(_num), filter=Condition(lambda: bool(self._slash_confirm_state)))(self._tui_make_slash_confirm_number_handler(_idx)) + + # --- History navigation: up/down browse history in normal input mode --- + # The TextArea is multiline, so by default up/down only move the cursor. + # Buffer.auto_up/auto_down handle both: cursor movement when multi-line, + # history browsing when on the first/last line (or single-line input). + _normal_input = Condition( + lambda: not self._clarify_state and not self._approval_state and not self._slash_confirm_state and not self._sudo_state and not self._secret_state and not self._model_picker_state and not self._command_palette_state + ) + + kb.add('up', filter=_normal_input)(self._tui_history_up) + + kb.add('down', filter=_normal_input)(self._tui_history_down) + + kb.add('c-l')(self._tui_handle_ctrl_l) + + kb.add('c-c')(self._tui_handle_ctrl_c) + + # Ctrl+Shift+C: no binding needed. Terminal emulators (GNOME Terminal, + # iTerm2, kitty, Windows Terminal, etc.) intercept Ctrl+Shift+C before + # the keystroke reaches the application's stdin — prompt_toolkit never + # sees it, and prompt_toolkit's key spec parser doesn't even recognise + # 'c-S-c' anyway (the Shift modifier is meaningless on control-sequence + # keys). #19884 added a handler for this; #19895 patched the resulting + # startup crash with try/except. Both were based on a misreading of how + # terminal key events propagate. Deleting the dead handler outright. + + kb.add('c-q')(self._tui_handle_ctrl_q) + + kb.add('c-d')(self._tui_handle_ctrl_d) + + _modal_prompt_active = Condition( + lambda: bool(self._secret_state or self._sudo_state or self._slash_confirm_state) + ) + + kb.add('escape', filter=_modal_prompt_active, eager=True)(self._tui_handle_escape_modal) + + kb.add('escape', 'escape', filter=~_modal_prompt_active)(self._tui_handle_double_escape) + + kb.add('c-z')(self._tui_handle_ctrl_z) + + # Voice push-to-talk key: configurable via config.yaml (voice.record_key) + # Default: Ctrl+B (avoids conflict with Ctrl+R readline reverse-search). + # Config spellings (ctrl/control/alt/option/opt) are normalized to + # prompt_toolkit's c-x / a-x format via ``normalize_voice_record_key_for_prompt_toolkit`` + # so the same config value binds identically in the TUI and CLI + # (Copilot round-9 review on #19835). ``super``/``win``/``windows`` + # configs silently fall back to the default here since prompt_toolkit + # has no super modifier — log a warning so users notice the + # TUI/CLI split instead of a silent mismatch (round-11). + _raw_key: object = "ctrl+b" + try: + from hermes_cli.config import load_config + from hermes_cli.voice import ( + normalize_voice_record_key_for_prompt_toolkit, + pt_key_to_sequence, + voice_record_key_from_config, + ) + _raw_key = voice_record_key_from_config(load_config()) + _voice_key = normalize_voice_record_key_for_prompt_toolkit(_raw_key) + if ( + isinstance(_raw_key, str) + and _raw_key.strip().lower().split("+", 1)[0].strip() in {"super", "win", "windows"} + and _voice_key == "c-b" + ): + logger.warning( + "voice.record_key %r uses a TUI-only modifier (super/win); " + "CLI fell back to Ctrl+B. Use ctrl+ or alt+ for " + "cross-runtime parity.", + _raw_key, + ) + except Exception: + _voice_key = "c-b" + + # Cache the UI label here — same ``_raw_key`` that drives the + # prompt_toolkit binding below. Every status / placeholder / + # recording-hint render reads this cached value so display can + # never drift from the live keybinding even if the user edits + # voice.record_key mid-session (Copilot round-13 on #19835). + self.set_voice_record_key_cache(_raw_key) + + kb.add(*pt_key_to_sequence(_voice_key))(self._tui_handle_voice_record) + from prompt_toolkit.keys import Keys + + kb.add(Keys.BracketedPaste, eager=True)(self._tui_handle_paste) + + kb.add('c-v')(self._tui_handle_ctrl_v) + + kb.add('escape', 'v')(self._tui_handle_alt_v) + return kb + + def _tui_build_layout(self, kb): + """Build the TUI widgets, Layout and Style; registers wrapper keybindings on ``kb``.""" + from cli import _estimate_tui_input_height, get_skill_bundles, get_skill_commands + # Dynamic prompt: shows Hermes symbol when agent is working, + # or answer prompt when clarify freetext mode is active. + cli_ref = self + + def get_prompt(): + return cli_ref._get_tui_prompt_fragments() + + # Create the input area with multiline (Alt+Enter), autocomplete, and paste handling + from prompt_toolkit.auto_suggest import AutoSuggestFromHistory + from prompt_toolkit.completion import ThreadedCompleter + + + _completer = SlashCommandCompleter( + skill_commands_provider=lambda: get_skill_commands(), + command_filter=cli_ref._command_available, + skill_bundles_provider=lambda: get_skill_bundles(), + ) + input_area = TextArea( + height=Dimension(min=1, max=8, preferred=1), + prompt=get_prompt, + style='class:input-area', + multiline=True, + wrap_lines=True, + read_only=Condition(lambda: bool(cli_ref._command_blocks_input)), + history=FileHistory(str(self._history_file)), + # complete_while_typing fires the completer on every keystroke. The + # completer does blocking work — fuzzy @-file indexing shells out to + # rg/fd (up to a 2s timeout) and path completion hits os.listdir/stat + # — so running it inline would stall the render loop on each key (very + # noticeable on WSL2/slow filesystems). ThreadedCompleter moves it off + # the UI event loop, keeping typing responsive. + completer=ThreadedCompleter(_completer), + complete_while_typing=True, + auto_suggest=SlashCommandAutoSuggest( + history_suggest=AutoSuggestFromHistory(), + completer=_completer, + ), + ) + # Keep prompt_toolkit on its simple tempfile path. Setting + # buffer.tempfile = "prompt.md" triggers its complex-tempfile branch, + # which tries to mkdir() the mkdtemp() directory again and raises + # EEXIST. The suffix keeps markdown highlighting without that bug. + input_area.buffer.tempfile_suffix = '.md' + + # Dynamic height: accounts for both explicit newlines AND visual + # wrapping of long lines so the input area always fits its content. + def _input_height(): + try: + from prompt_toolkit.application import get_app + + doc = input_area.buffer.document + try: + terminal_columns = get_app().output.get_size().columns + except Exception: + terminal_columns = shutil.get_terminal_size((80, 24)).columns + return _estimate_tui_input_height( + doc.lines, + self._get_tui_prompt_text(), + terminal_columns, + ) + except Exception: + return 1 + + input_area.window.height = _input_height + + # Paste collapsing: detect large pastes and save to temp file + self._tui_paste_counter = [0] + self._tui_prev_text_len = [0] + self._tui_prev_newline_count = [0] + self._tui_paste_just_collapsed = [False] + self._skip_paste_collapse = False + + input_area.buffer.on_text_changed += self._tui_on_text_changed + + # --- Input processors for password masking and inline placeholder --- + + # Mask input with '*' when the sudo password prompt is active + input_area.control.input_processors.append( + ConditionalProcessor( + PasswordProcessor(), + filter=Condition( + lambda: bool(cli_ref._sudo_state) or bool(cli_ref._secret_state) + ), + ) + ) + + class _PlaceholderProcessor(Processor): + """Render grayed-out placeholder text inside the input when empty.""" + def __init__(self, get_text): + self._get_text = get_text + + def apply_transformation(self, ti): + if not ti.document.text and ti.lineno == 0: + text = self._get_text() + if text: + # Append after existing fragments (preserves the ❯ prompt) + return Transformation(fragments=ti.fragments + [('class:placeholder', text)]) + return Transformation(fragments=ti.fragments) + + input_area.control.input_processors.append(_PlaceholderProcessor(self._tui_placeholder_text)) + + # Hint line above input: shown only for interactive prompts that need + # extra instructions (sudo countdown, approval navigation, clarify). + # The agent-running interrupt hint is now an inline placeholder above. + + spinner_widget = Window( + content=FormattedTextControl(self._tui_spinner_text), + height=self._tui_spinner_height, + wrap_lines=True, + ) + + # Petdex mascot — right-aligned Kitty placeholder or half-block sprite + # above the prompt. Collapses to height 0 when no pet is enabled. + # The animation thread queues virtual Kitty frames; after_render + # writes them out-of-band while prompt_toolkit owns the placeholder grid. + self._pet_widget = Window( + content=FormattedTextControl(self._pet_fragments), + height=self._pet_widget_height, + align=WindowAlign.RIGHT, + ) + + spacer = Window( + content=FormattedTextControl(self._tui_hint_text), + height=self._tui_hint_height, + ) + + # --- Clarify tool: dynamic display widget for questions + choices --- + + clarify_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_clarify_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._clarify_state is not None), + ) + + # --- Sudo password: display widget --- + + sudo_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_sudo_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._sudo_state is not None), + ) + + secret_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_secret_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._secret_state is not None), + ) + + # --- Dangerous command approval: display widget --- + + approval_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_approval_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._approval_state is not None), + ) + + slash_confirm_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_slash_confirm_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._slash_confirm_state is not None), + ) + + # --- /model picker: display widget --- + + model_picker_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_model_picker_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._model_picker_state is not None), + ) + + # --- Ctrl+P command palette: display widget --- + + command_palette_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_command_palette_display_fragments), + wrap_lines=True, + ), + filter=Condition(lambda: cli_ref._command_palette_state is not None), + ) + + # Horizontal rules above and below the input. + # On narrow/mobile terminals we keep the top separator for structure but + # hide the bottom one to recover a full row for conversation content. + input_rule_top = Window( + char='─', + height=lambda: cli_ref._tui_input_rule_height("top"), + style='class:input-rule', + ) + input_rule_bot = Window( + char='─', + height=lambda: cli_ref._tui_input_rule_height("bottom"), + style='class:input-rule', + ) + + # Image attachment indicator — shows badges like [📎 Image #1] above input + cli_ref = self + + image_bar = Window( + content=FormattedTextControl(self._tui_image_bar_fragments), + height=Condition(lambda: bool(cli_ref._attached_images)), + ) + + # Persistent voice mode status bar (visible only when voice mode is on) + + voice_status_bar = ConditionalContainer( + Window( + FormattedTextControl(self._tui_voice_status_fragments), + height=1, + ), + filter=Condition(lambda: cli_ref._voice_mode), + ) + + status_bar = ConditionalContainer( + Window( + content=FormattedTextControl(lambda: cli_ref._get_status_bar_fragments()), + height=1, + # Prevent fragments that overflow the terminal width from + # wrapping onto a second line, which causes the status bar to + # appear duplicated (one full + one partial row) during long + # sessions, especially on SSH where shutil.get_terminal_size + # may return stale values. _get_status_bar_fragments now reads + # width from prompt_toolkit's own output object, so fragments + # will always fit; wrap_lines=False is the belt-and-suspenders + # guard against any future width mismatch. + wrap_lines=False, + ), + filter=Condition( + lambda: cli_ref._status_bar_visible + and not getattr(cli_ref, "_status_bar_suppressed_after_resize", False) + ), + ) + + # Stash browse panel — appears just above the status bar when the user + # presses Ctrl+S on an empty composer with 2+ stashed drafts. + + self._stash_panel_widget = ConditionalContainer( + Window( + FormattedTextControl(self._get_stash_panel_display_fragments), + wrap_lines=False, + ), + filter=Condition( + lambda: cli_ref._prompt_stash.panel_open + and bool(len(cli_ref._prompt_stash)) + ), + ) + + # Allow wrapper CLIs to register extra keybindings. + self._register_extra_tui_keybindings(kb, input_area=input_area) + + # Layout: interactive prompt widgets + ruled input at bottom. + # The sudo, approval, and clarify widgets appear above the input when + # the corresponding interactive prompt is active. + completions_menu = CompletionsMenu(max_height=12, scroll_offset=1) + + layout = Layout( + HSplit( + self._build_tui_layout_children( + sudo_widget=sudo_widget, + secret_widget=secret_widget, + approval_widget=approval_widget, + slash_confirm_widget=slash_confirm_widget, + clarify_widget=clarify_widget, + model_picker_widget=model_picker_widget, + command_palette_widget=command_palette_widget, + spinner_widget=spinner_widget, + spacer=spacer, + status_bar=status_bar, + input_rule_top=input_rule_top, + image_bar=image_bar, + input_area=input_area, + input_rule_bot=input_rule_bot, + voice_status_bar=voice_status_bar, + completions_menu=completions_menu, + ) + ) + ) + + # Style for the application + self._tui_style_base = { + # Input area / prompt: empty style strings inherit the + # terminal's default foreground/background, so the typed + # text is readable in both light and dark Terminal.app + # color schemes. (Hardcoding a near-white #FFF8DC made + # input invisible on light backgrounds.) + 'input-area': '', + 'placeholder': '#888888 italic', + 'prompt': '', + 'prompt-working': '#888888 italic', + 'hint': '#888888 italic', + 'status-bar': 'bg:#1a1a2e #C0C0C0', + 'status-bar-strong': 'bg:#1a1a2e #FFD700 bold', + 'status-bar-dim': 'bg:#1a1a2e #8B8682', + 'status-bar-good': 'bg:#1a1a2e #8FBC8F bold', + 'status-bar-warn': 'bg:#1a1a2e #FFD700 bold', + 'status-bar-bad': 'bg:#1a1a2e #FF8C00 bold', + 'status-bar-critical': 'bg:#1a1a2e #FF6B6B bold', + 'status-bar-yolo': 'bg:#1a1a2e #FF4444 bold', + 'status-bar-session-title': 'bg:#FFD700 #1a1a2e bold', + # Bronze horizontal rules around the input area + 'input-rule': '#CD7F32', + # Clipboard image attachment badges + 'image-badge': '#87CEEB bold', + 'completion-menu': 'bg:#1a1a2e #FFF8DC', + 'completion-menu.completion': 'bg:#1a1a2e #FFF8DC', + 'completion-menu.completion.current': 'bg:#333355 #FFD700', + 'completion-menu.meta.completion': 'bg:#1a1a2e #888888', + 'completion-menu.meta.completion.current': 'bg:#333355 #FFBF00', + # Clarify question panel + 'clarify-border': '#CD7F32', + 'clarify-title': '#FFD700 bold', + 'clarify-question': '#FFF8DC bold', + 'clarify-choice': '#AAAAAA', + 'clarify-selected': '#FFD700 bold', + 'clarify-active-other': '#FFD700 italic', + 'clarify-answer': '#98FB98', + 'clarify-countdown': '#CD7F32', + # Sudo password panel + 'sudo-prompt': '#FF6B6B bold', + 'sudo-border': '#CD7F32', + 'sudo-title': '#FF6B6B bold', + 'sudo-text': '#FFF8DC', + # Dangerous command approval panel + 'approval-border': '#CD7F32', + 'approval-title': '#FF8C00 bold', + 'approval-desc': '#FFF8DC bold', + 'approval-cmd': '#AAAAAA italic', + 'approval-choice': '#AAAAAA', + 'approval-selected': '#FFD700 bold', + # Voice mode + 'voice-prompt': '#87CEEB', + 'voice-recording': '#FF4444 bold', + 'voice-processing': '#FFA500 italic', + 'voice-status': 'bg:#1a1a2e #87CEEB', + 'voice-status-recording': 'bg:#1a1a2e #FF4444 bold', + } + style = PTStyle.from_dict(self._build_tui_style_dict()) + return (layout, style) diff --git a/hermes_cli/cli_voice_mixin.py b/hermes_cli/cli_voice_mixin.py new file mode 100644 index 0000000000..a91c6ac0d2 --- /dev/null +++ b/hermes_cli/cli_voice_mixin.py @@ -0,0 +1,1022 @@ +"""Voice mode (recording, STT, TTS, full-duplex barge-in) and wake-word listener handlers for the interactive CLI + +Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal +symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin +never imports ``cli`` at module load time (import cycle). +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import tempfile +import threading +import time + +from hermes_constants import is_termux as _is_termux_environment +from typing import Optional + + +class CLIVoiceMixin: + """Voice mode (recording, STT, TTS, full-duplex barge-in) and wake-word listener handlers for the interactive CLI""" + + def _voice_start_recording(self): + """Start capturing audio from the microphone.""" + from cli import _ACCENT, _DIM, _RST, _cprint + if getattr(self, '_should_exit', False): + return + from tools.voice_mode import create_audio_recorder, check_voice_requirements + + reqs = check_voice_requirements() + if not reqs["audio_available"]: + if _is_termux_environment(): + details = reqs.get("details", "") + if "Termux:API Android app is not installed" in details: + raise RuntimeError( + "Termux:API command package detected, but the Android app is missing.\n" + "Install/update the Termux:API Android app, then retry /voice on.\n" + "Fallback: pkg install python-numpy portaudio && python -m pip install sounddevice" + ) + raise RuntimeError( + "Voice mode requires either Termux:API microphone access or Python audio libraries.\n" + "Option 1: pkg install termux-api and install the Termux:API Android app\n" + "Option 2: pkg install python-numpy portaudio && python -m pip install sounddevice" + ) + raise RuntimeError( + "Voice mode requires sounddevice and numpy.\n" + f"Install with: {sys.executable} -m pip install sounddevice numpy" + ) + if not reqs.get("stt_available", reqs.get("stt_key_set")): + raise RuntimeError( + "Voice mode requires an STT provider for transcription.\n" + "Option 1: uv pip install faster-whisper " + "(free, local; `pip install faster-whisper` also works if pip is on PATH)\n" + "Option 2: Set GROQ_API_KEY (free tier)\n" + "Option 3: Set VOICE_TOOLS_OPENAI_KEY (paid)" + ) + + # Prevent double-start from concurrent threads (atomic check-and-set) + with self._voice_lock: + if self._voice_recording: + return + self._voice_recording = True + + # Load silence detection params from config. Shape-safe: a + # hand-edited ``voice: true`` / ``voice: cmd+b`` leaves + # ``load_config()['voice']`` as a non-dict; coerce to {} so + # continuous recording falls back to the documented defaults + # instead of crashing on ``.get()``. + voice_cfg: dict = {} + try: + from hermes_cli.config import load_config + _cfg = load_config().get("voice") + voice_cfg = _cfg if isinstance(_cfg, dict) else {} + except Exception: + pass + + # Recorder creation can fail (no input device, PortAudio init error). + # Reset the flag on failure or _voice_recording stays True forever and + # every future voice start is silently skipped by the guard above. + if self._voice_recorder is None: + try: + self._voice_recorder = create_audio_recorder() + except Exception: + with self._voice_lock: + self._voice_recording = False + raise + + # Apply config-driven silence params (numeric-guarded so YAML + # scalar corruption doesn't break recording start-up). + # + # ``bool`` is explicitly excluded from the numeric check — in + # Python bool is a subclass of int, so a hand-edited + # ``silence_threshold: true`` would otherwise be forwarded as + # ``1`` instead of falling back to the 200 default (Copilot + # round-12 on #19835). + _threshold = voice_cfg.get("silence_threshold") + _duration = voice_cfg.get("silence_duration") + self._voice_recorder._silence_threshold = ( + _threshold if isinstance(_threshold, (int, float)) and not isinstance(_threshold, bool) else 200 + ) + self._voice_recorder._silence_duration = ( + _duration if isinstance(_duration, (int, float)) and not isinstance(_duration, bool) else 3.0 + ) + # voice.max_recording_seconds — hard cap on a single recording's length. + # Same numeric guard as the silence params (bool excluded: a hand-edited + # ``max_recording_seconds: true`` must not become ``1`` — it falls back + # to the documented 120 default, mirroring the silence-param handling). + # An explicit numeric value <= 0 disables the cap. Previously this + # documented key was never read (dead config); wiring it here makes it + # take effect. + _max_rec = voice_cfg.get("max_recording_seconds") + self._voice_recorder._max_recording_seconds = ( + (_max_rec if _max_rec > 0 else 0.0) + if isinstance(_max_rec, (int, float)) and not isinstance(_max_rec, bool) + else 120.0 + ) + + def _on_silence(): + """Called by AudioRecorder when silence is detected after speech.""" + with self._voice_lock: + if not self._voice_recording: + return + _cprint(f"\n{_DIM}Silence detected, auto-stopping...{_RST}") + if hasattr(self, '_app') and self._app: + self._app.invalidate() + self._voice_stop_and_transcribe() + + # Audio cue: single beep BEFORE starting stream (avoid CoreAudio conflict) + if self._voice_beeps_enabled(): + try: + from tools.voice_mode import play_beep + play_beep(frequency=880, count=1) + except Exception: + pass + + try: + self._voice_recorder.start(on_silence_stop=_on_silence) + except Exception: + with self._voice_lock: + self._voice_recording = False + raise + _label = self._voice_record_key_label() + if getattr(self._voice_recorder, "supports_silence_autostop", True): + _recording_hint = f"auto-stops on silence | {_label} to stop & exit continuous" + elif _is_termux_environment(): + _recording_hint = f"Termux:API capture | {_label} to stop" + else: + _recording_hint = f"{_label} to stop" + _cprint(f"\n{_ACCENT}● Recording...{_RST} {_DIM}({_recording_hint}){_RST}") + + # Periodically refresh prompt to update audio level indicator + def _refresh_level(): + while True: + with self._voice_lock: + still_recording = self._voice_recording + if not still_recording: + break + if hasattr(self, '_app') and self._app: + self._app.invalidate() + time.sleep(0.15) + threading.Thread(target=_refresh_level, daemon=True).start() + + def _voice_stt_model(self) -> Optional[str]: + """STT model override from config, or None for the provider default. + + For the local provider, prefer stt.local.model (default ``base``) so the + CLI passes a real model name into the local STT backend. + """ + try: + from hermes_cli.config import load_config + stt_config = load_config().get("stt", {}) + if not isinstance(stt_config, dict): + return None + provider = str(stt_config.get("provider") or "").strip().lower() + if provider == "local": + local_config = stt_config.get("local") or {} + if not isinstance(local_config, dict): + local_config = {} + return local_config.get("model") or "base" + return stt_config.get("model") + except Exception: + return None + + def _voice_stt_provider(self) -> str: + """Configured STT provider name (lowercased), or empty string.""" + try: + from hermes_cli.config import load_config + stt_config = load_config().get("stt", {}) + if not isinstance(stt_config, dict): + return "" + return str(stt_config.get("provider") or "").strip().lower() + except Exception: + return "" + + def _voice_restart_recording_async(self) -> None: + """Restart continuous-mode recording off-thread (start() can block).""" + from cli import _DIM, _RST, _cprint + def _restart_recording(): + try: + self._voice_start_recording() + if hasattr(self, '_app') and self._app: + self._app.invalidate() + except Exception as e: + _cprint(f"{_DIM}Voice auto-restart failed: {e}{_RST}") + threading.Thread(target=_restart_recording, daemon=True).start() + + def _voice_stop_and_transcribe(self): + """Stop recording, transcribe via STT, and queue the transcript as input.""" + from cli import _DIM, _RST, _VoiceInputMessage, _cprint + # Atomic guard: only one thread can enter stop-and-transcribe. + # Set _voice_processing immediately so concurrent Ctrl+B presses + # don't race into the START path while recorder.stop() holds its lock. + with self._voice_lock: + if not self._voice_recording: + return + self._voice_recording = False + self._voice_processing = True + + submitted = False + transcription_failed = False + wav_path = None + try: + if self._voice_recorder is None: + return + + wav_path = self._voice_recorder.stop() + + # Audio cue: double beep after stream stopped (no CoreAudio conflict) + if self._voice_beeps_enabled(): + try: + from tools.voice_mode import play_beep + play_beep(frequency=660, count=2) + except Exception: + pass + + if wav_path is None: + _cprint(f"{_DIM}No speech detected.{_RST}") + return + + # _voice_processing is already True (set atomically above) + if hasattr(self, '_app') and self._app: + self._app.invalidate() + + stt_model = self._voice_stt_model() + if self._voice_stt_provider() == "local": + _cprint( + f"{_DIM}Preparing local STT model '{stt_model}' " + f"(first use may download it from Hugging Face)...{_RST}" + ) + else: + _cprint(f"{_DIM}Transcribing...{_RST}") + + from tools.voice_mode import transcribe_recording + result = transcribe_recording(wav_path, model=stt_model) + + if result.get("success") and result.get("transcript", "").strip(): + transcript = result["transcript"].strip() + from tools.voice_mode import is_voice_stop_phrase + if is_voice_stop_phrase(transcript): + # Bare "stop" (or configured phrase) ends the voice chat + # instead of being sent to the agent. + _cprint(f"{_DIM}Stop phrase detected — ending voice chat.{_RST}") + self._disable_voice_mode() + return + self._attached_images.clear() + if hasattr(self, '_app') and self._app: + self._app.invalidate() + self._pending_input.put(_VoiceInputMessage(transcript)) + submitted = True + elif result.get("success"): + _cprint(f"{_DIM}No speech detected.{_RST}") + else: + error = result.get("error", "Unknown error") + _cprint(f"\n{_DIM}Transcription failed: {error}{_RST}") + transcription_failed = True + + except Exception as e: + _cprint(f"\n{_DIM}Voice processing error: {e}{_RST}") + transcription_failed = wav_path is not None + finally: + with self._voice_lock: + self._voice_processing = False + if hasattr(self, '_app') and self._app: + self._app.invalidate() + # Clean up temp file unless transcription failed. On failure, keep + # the source recording so long dictation is not lost. + try: + if wav_path and os.path.isfile(wav_path): + if transcription_failed: + _cprint(f"{_DIM}Recording preserved at: {wav_path}{_RST}") + else: + os.unlink(wav_path) + except Exception: + pass + + # Track consecutive no-speech cycles to avoid infinite restart loops. + # While the agent is mid-turn or TTS is speaking, the user is + # CORRECTLY silent (waiting/listening) — those cycles must not + # count, or a multi-minute tool run ends the voice chat under + # the user. The stop phrase and barge-in still work during the + # hold (they run on their own paths above). + stop_continuous_restart = False + _tts_done = getattr(self, "_voice_tts_done", None) + _activity_hold = bool( + getattr(self, "_agent_running", False) + or (_tts_done is not None and not _tts_done.is_set()) + ) + if not submitted: + if _activity_hold: + pass # held: keep listening without counting the cycle + else: + self._no_speech_count = getattr(self, '_no_speech_count', 0) + 1 + if self._no_speech_count >= 3: + self._voice_continuous = False + self._no_speech_count = 0 + _cprint(f"{_DIM}No speech detected 3 times, continuous mode stopped.{_RST}") + stop_continuous_restart = True + else: + self._no_speech_count = 0 + + # If no transcript was submitted but continuous mode is active, + # restart recording so the user can keep talking. + # (When transcript IS submitted, process_loop handles restart + # after chat() completes.) + if ( + self._voice_continuous + and not submitted + and not self._voice_recording + and not stop_continuous_restart + ): + self._voice_restart_recording_async() + + def _voice_speak_response_async(self, text: str) -> None: + """Schedule TTS and mark it pending before continuous recording can restart.""" + if not self._voice_tts or not text: + return + self._voice_tts_done.clear() + threading.Thread( + target=self._voice_speak_response, + args=(text,), + daemon=True, + ).start() + # Spoken barge-in must work on the whole-file fallback path too. The + # full-duplex agent-turn listener normally already covers playback + # (armed at turn start in chat()); this arm is an idempotent safety + # net for speak calls outside a chat turn — the listener refuses to + # double-arm via _voice_fd_active. + if self._voice_continuous: + threading.Thread( + target=self._voice_full_duplex_listener, + daemon=True, + ).start() + + def _voice_speak_response(self, text: str): + """Speak the agent's response aloud using TTS (runs in background thread).""" + from cli import _DIM, _RST, _cprint, logger + if not self._voice_tts: + return + self._voice_tts_done.clear() + try: + from tools.tts_tool import text_to_speech_tool + from tools.voice_mode import play_audio_file + + # Strip markdown and non-speech content for cleaner TTS via the + # shared cleaner (tools/tts_text_normalize): markdown, emoji, + # ⋗ blocks, verifier footer, units, newline flattening. + # The TTS tool owns provider request limits and long-form chunking. + try: + from tools.tts_text_normalize import prepare_spoken_text + tts_text = prepare_spoken_text(text, max_chars=None) + except Exception: + # Legacy fallback pipeline — keep voice replies best-effort. + tts_text = re.sub(r'```[\s\S]*?```', ' ', text) # fenced code blocks + tts_text = re.sub(r'\[([^\]]+)\]\([^)]+\)', r'\1', tts_text) # [text](url) -> text + tts_text = re.sub(r'https?://\S+', '', tts_text) # URLs + tts_text = re.sub(r'\*\*(.+?)\*\*', r'\1', tts_text) # bold + tts_text = re.sub(r'\*(.+?)\*', r'\1', tts_text) # italic + tts_text = re.sub(r'`(.+?)`', r'\1', tts_text) # inline code + tts_text = re.sub(r'^#+\s*', '', tts_text, flags=re.MULTILINE) # headers + tts_text = re.sub(r'^\s*[-*]\s+', '', tts_text, flags=re.MULTILINE) # list items + tts_text = re.sub(r'---+', '', tts_text) # horizontal rules + tts_text = re.sub(r'\n{3,}', '\n\n', tts_text) # excessive newlines + tts_text = tts_text.strip() + if not tts_text: + return + self._voice_last_tts_text = tts_text + + # Use MP3 output for CLI playback (afplay doesn't handle OGG well). + # The TTS tool may auto-convert MP3->OGG, but the original MP3 remains. + os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True) + mp3_path = os.path.join( + tempfile.gettempdir(), "hermes_voice", + f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3", + ) + + raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path) + try: + tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {} + except Exception: + tts_result = {} + + # The tool result is authoritative — it may return multiple files + # for long-form chunked output. Play each in order. + play_paths = tts_result.get("file_paths") or [ + tts_result.get("file_path") or mp3_path + ] + for play_path in play_paths if tts_result.get("success") else []: + if os.path.isfile(play_path) and os.path.getsize(play_path) > 0: + play_audio_file(play_path) + # Clean up all generated files (play_paths + mp3_path + ogg variants) + cleanup_paths = set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]) + for path in cleanup_paths: + if os.path.isfile(path): + try: + os.unlink(path) + except OSError: + pass + except Exception as e: + logger.warning("Voice TTS playback failed: %s", e) + _cprint(f"{_DIM}TTS playback failed: {e}{_RST}") + finally: + self._voice_tts_done.set() + + def _voice_full_duplex_listener(self) -> None: + """Full-duplex agent-turn listener: mic live for the WHOLE turn. + + Armed at utterance-submit (chat() start in continuous voice mode) and + disarmed when the turn is fully done (agent finished + TTS played). + Replaces the old per-playback ``_voice_barge_in_monitor``, which only + listened while TTS audio was playing — during LLM generation the mic + was dead, so the user could not interject by voice at all (and the + playback monitor calibrated against its own speaker bleed, making + the trigger unreachable; see tools.voice_mode.full_duplex_listen). + + Phase behaviour: + + * generation (no TTS audio yet): speech interrupts the in-flight + agent turn via ``self.agent.interrupt()`` — the same seam the + typed/Ctrl+C interrupt uses — and the captured utterance is + submitted as the next message. + * playback: speech cuts TTS (pipeline stop event + stop_playback) + and the interruption is captured with pre-roll and submitted. + + The stop phrase ends the voice chat in BOTH phases (a stop during + generation means "stop everything": the turn is already interrupted + at trip time, then ``_voice_submit_barge_utterance`` disables voice + mode). + """ + from cli import _DIM, _RST, _cprint, logger + fd_active = getattr(self, "_voice_fd_active", None) + if fd_active is None: + fd_active = threading.Event() + self._voice_fd_active = fd_active + if fd_active.is_set(): + return # one listener owns the mic for this turn + fd_active.set() + try: + from hermes_cli.config import load_config + voice_cfg = load_config().get("voice") or {} + if not (isinstance(voice_cfg, dict) and voice_cfg.get("barge_in", True)): + return + from tools.voice_mode import ( + full_duplex_listen, + is_audio_output_active, + stop_playback, + ) + + try: + _mult = float(voice_cfg.get("barge_in_threshold_multiplier", 0) or 0) + except (TypeError, ValueError): + _mult = 0.0 + try: + _grace_ms = int(float(voice_cfg.get("barge_in_grace_seconds", 0.5)) * 1000) + except (TypeError, ValueError): + _grace_ms = 500 + + tts_done = getattr(self, "_voice_tts_done", None) + + def _should_stop() -> bool: + if not (getattr(self, "_voice_mode", False) and getattr(self, "_voice_continuous", False)): + return True + if getattr(self, "_agent_running", False): + return False + # Agent finished — keep listening until TTS fully played. + if tts_done is not None and not tts_done.is_set(): + return False + return not is_audio_output_active() + + def _on_trigger(phase: str) -> None: + # Latch BEFORE cutting anything: suppresses process_loop's + # auto-restart until the capture is submitted. + self._voice_barge_capture.set() + self._voice_barge_phase = phase + if phase == "playback": + logger.debug( + "TTS CUT: full-duplex listener tripped during playback" + ) + from tools.tts_streaming import mark_speech_interrupted + mark_speech_interrupted() + _pipe_stop = getattr(self, "_voice_tts_stop", None) + if _pipe_stop is not None: + _pipe_stop.set() + stop_playback() + else: + # Generation phase: no audio to cut — interrupt the + # in-flight agent turn (same seam as typed interrupt). + logger.debug( + "full-duplex listener tripped during generation — " + "interrupting agent turn" + ) + _pipe_stop = getattr(self, "_voice_tts_stop", None) + if _pipe_stop is not None: + _pipe_stop.set() # never let the stale reply speak + try: + if self.agent is not None and getattr(self, "_agent_running", False): + _cprint(f"\n{_DIM}🎤 Voice interjection — interrupting…{_RST}") + self.agent.interrupt() + except Exception as e: + logger.debug("voice interjection interrupt failed: %s", e) + + wav_path = full_duplex_listen( + _should_stop, + is_playing=is_audio_output_active, + on_trigger=_on_trigger, + multiplier=_mult or None, + grace_ms=max(0, _grace_ms), + ) + if wav_path and self._voice_barge_capture.is_set(): + self._voice_submit_barge_utterance(wav_path) + else: + self._voice_barge_capture.clear() + except Exception as e: + self._voice_barge_capture.clear() + logger.debug("Voice full-duplex listener failed: %s", e) + finally: + fd_active.clear() + + def _voice_submit_barge_utterance(self, wav_path: str) -> None: + """Transcribe a barge-captured interruption and queue it as the next turn.""" + from cli import _DIM, _RST, _VoiceInputMessage, _cprint, logger + submitted = False + try: + from tools.voice_mode import transcribe_recording + result = transcribe_recording(wav_path, model=self._voice_stt_model()) + transcript = (result.get("transcript") or "").strip() if result.get("success") else "" + if transcript: + from tools.voice_mode import is_voice_stop_phrase + if is_voice_stop_phrase(transcript): + _cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}") + self._disable_voice_mode() + return + # Fail-closed echo guard (#75780): a playback-phase capture + # has no acoustic echo cancellation, so speaker bleed alone + # can trip the barge trigger. If the transcript is a close + # match for what Hermes just spoke, treat it as self-capture + # instead of queuing it as a user turn. + if getattr(self, "_voice_barge_phase", None) == "playback": + from tools.voice_mode import is_tts_echo + if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")): + logger.debug( + "Dropping playback-phase barge transcript as TTS echo: %r", + transcript, + ) + _cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}") + return + self._pending_input.put(_VoiceInputMessage(transcript)) + submitted = True + elif not result.get("success"): + _cprint(f"\n{_DIM}Transcription failed: {result.get('error', 'Unknown error')}{_RST}") + except Exception as e: + _cprint(f"\n{_DIM}Voice processing error: {e}{_RST}") + finally: + try: + if os.path.isfile(wav_path): + os.unlink(wav_path) + except OSError: + pass + self._voice_barge_capture.clear() + self._voice_barge_phase = None + # No usable transcript: hand the mic back to the normal loop. + if not submitted and self._voice_mode and self._voice_continuous and not self._voice_recording: + self._voice_restart_recording_async() + + def _voice_beeps_enabled(self) -> bool: + """Return whether CLI voice mode should play record start/stop beeps.""" + try: + from hermes_cli.config import load_config + from utils import is_truthy_value + voice_cfg = load_config().get("voice", {}) + if isinstance(voice_cfg, dict): + # is_truthy_value handles quoted YAML strings like "false" + # which bool() would misread as True (#49883). + return is_truthy_value(voice_cfg.get("beep_enabled", True), default=True) + except Exception: + pass + return True + + def _enable_voice_mode(self): + """Enable voice mode after checking requirements.""" + from cli import _ACCENT, _BOLD, _DIM, _RST, _cprint + if self._voice_mode: + _cprint(f"{_DIM}Voice mode is already enabled.{_RST}") + return + + from tools.voice_mode import check_voice_requirements, detect_audio_environment + + # Environment detection -- warn and block in incompatible environments + env_check = detect_audio_environment() + if not env_check["available"]: + _cprint(f"\n{_ACCENT}Voice mode unavailable in this environment:{_RST}") + for warning in env_check["warnings"]: + _cprint(f" {_DIM}{warning}{_RST}") + return + + reqs = check_voice_requirements() + if not reqs["available"]: + _cprint(f"\n{_ACCENT}Voice mode requirements not met:{_RST}") + for line in reqs["details"].split("\n"): + _cprint(f" {_DIM}{line}{_RST}") + if reqs["missing_packages"]: + if _is_termux_environment(): + _cprint(f"\n {_BOLD}Option 1: pkg install termux-api{_RST}") + _cprint(f" {_DIM}Then install/update the Termux:API Android app for microphone capture{_RST}") + _cprint(f" {_BOLD}Option 2: pkg install python-numpy portaudio && python -m pip install sounddevice{_RST}") + else: + _cprint(f"\n {_BOLD}Install: {sys.executable} -m pip install {' '.join(reqs['missing_packages'])}{_RST}") + return + + with self._voice_lock: + self._voice_mode = True + + # Check config for auto_tts (shape-safe — malformed ``voice:`` YAML + # leaves ``voice_config`` as a non-dict, so guard before .get()). + try: + from hermes_cli.config import load_config + _raw_voice = load_config().get("voice") + voice_config = _raw_voice if isinstance(_raw_voice, dict) else {} + if voice_config.get("auto_tts", False): + with self._voice_lock: + self._voice_tts = True + except Exception: + pass + + # Voice mode instruction is injected as a user message prefix (not a + # system prompt change) to avoid invalidating the prompt cache. See + # _voice_message_prefix property and its usage in _process_message(). + + tts_status = " (TTS enabled)" if self._voice_tts else "" + if self._voice_tts: + # Speech output is on from the start — warm the engine now so the + # first spoken reply doesn't pay the model load as dead air. + self._tts_lease_async(True) + # Use the startup-pinned cache so the advertised shortcut always + # matches the live prompt_toolkit binding — reading live config + # here would drift after a mid-session config edit (Copilot + # round-14 on #19835, same class as round-13). + _ptt_display = self._voice_record_key_label() + _cprint(f"\n{_ACCENT}Voice mode enabled{tts_status}{_RST}") + _cprint(f" {_DIM}{_ptt_display} to start/stop recording{_RST}") + # Spoken-stop hint sourced from voice.stop_phrases (first entry); the + # helper returns "" when stop phrases are disabled — show no hint then. + try: + from tools.voice_mode import voice_stop_hint + _stop_hint = voice_stop_hint() + except Exception: + _stop_hint = "" + if _stop_hint: + _cprint(f" {_DIM}{_stop_hint}{_RST}") + _cprint(f" {_DIM}/voice tts to toggle speech output{_RST}") + _cprint(f" {_DIM}/voice off to disable voice mode{_RST}") + + def _typed_voice_stop(self, user_input) -> bool: + """Typed bare stop phrase during an active voice chat ends the chat. + + Saying "stop" ends the voice chat (PR #73106); TYPING the same bare + stop phrase while voice mode is on must behave identically instead of + sending "stop" to the agent as a turn. Guarded on voice mode being ON + — typed "stop" outside voice chat passes through to the agent exactly + as before. Reuses ``is_voice_stop_phrase`` (same config + ``voice.stop_phrases``, same exact-match semantics), so longer typed + messages containing "stop" are never swallowed. + """ + from cli import _DIM, _RST, _cprint + if not isinstance(user_input, str): + return False + with self._voice_lock: + voice_on = self._voice_mode or self._voice_continuous + if not voice_on: + return False + try: + from tools.voice_mode import is_voice_stop_phrase + if not is_voice_stop_phrase(user_input): + return False + except Exception: + return False + _cprint(f"\n{_DIM}Stop phrase typed — ending voice chat.{_RST}") + self._disable_voice_mode() + return True + + def _disable_voice_mode(self): + """Disable voice mode, cancel any active recording, and stop TTS.""" + from cli import _DIM, _RST, _cprint, logger + recorder = None + with self._voice_lock: + if self._voice_recording and self._voice_recorder: + self._voice_recorder.cancel() + self._voice_recording = False + recorder = self._voice_recorder + self._voice_mode = False + self._voice_tts = False + self._voice_continuous = False + + # Speech output is off with the mode — release the TTS engine lease so + # a resident local model (piper/kittentts) is freed once nothing else + # in this process still needs it. + self._tts_lease_async(False) + + # Shut down the persistent audio stream in background + if recorder is not None: + def _bg_shutdown(rec=recorder): + try: + rec.shutdown() + except Exception: + pass + threading.Thread(target=_bg_shutdown, daemon=True).start() + self._voice_recorder = None + + # Stop any active TTS playback (file player + streaming pipeline) + try: + if self._voice_tts_stop is not None: + logger.info("TTS CUT: _disable_voice_mode setting stop event") + self._voice_tts_stop.set() + from tools.voice_mode import stop_playback + stop_playback() + except Exception: + pass + self._voice_tts_done.set() + + _cprint(f"\n{_DIM}Voice mode disabled.{_RST}") + + def _maybe_start_wake_word(self): + """Start the wake-word listener at CLI startup if this surface is eligible.""" + try: + from tools.wake_word import wake_surface_enabled + if not wake_surface_enabled("cli"): + return + except Exception: + return + self._start_wake_word_listener(announce=True) + + def _start_wake_word_listener(self, announce: bool = False) -> bool: + """Build + start the hotword detector. Returns True on success.""" + from cli import _ACCENT, _DIM, _RST, _cprint + try: + from tools.wake_word import ( + check_wake_word_requirements, + load_wake_word_config, + owns_listener, + start_listening, + ) + except Exception as e: + if announce: + _cprint(f"{_DIM}Wake word unavailable: {e}{_RST}") + return False + + if getattr(self, "_wake_word_active", False) and owns_listener(self): + if announce: + _cprint(f"{_DIM}Wake word is already listening.{_RST}") + return True + self._wake_word_active = False + + cfg = load_wake_word_config() + reqs = check_wake_word_requirements(cfg) + if not reqs["available"]: + if announce: + _cprint(f"\n{_ACCENT}Wake word requirements not met:{_RST}") + if reqs.get("hint"): + _cprint(f" {_DIM}{reqs['hint']}{_RST}") + return False + + if announce and not reqs.get("deps_available", True): + # Fresh install: the engine constructor lazy-installs its deps + # (onnxruntime is a large wheel) — tell the user why this is slow. + _cprint(f"{_DIM}Installing wake word engine (first use — this may take a minute)...{_RST}") + + self._wake_start_new_session = bool(cfg.get("start_new_session", True)) + try: + start_listening(self._on_wake_word, owner=self, config=cfg) + except Exception as e: + if announce: + _cprint(f"\n{_DIM}Failed to start wake word: {e}{_RST}") + return False + + self._wake_word_active = True + self._wake_suspended = False + import cli as _cli + _cli._cli_wake_owner = self + self._start_wake_watchdog() + if announce: + _cprint(f"\n{_ACCENT}Wake word listening{_RST} " + f"{_DIM}(say \"{reqs['phrase']}\" — /wake off to stop){_RST}") + return True + + def _stop_wake_word_listener(self, announce: bool = False): + """Stop and tear down the hotword detector.""" + from cli import _DIM, _RST, _cprint + import cli as _cli + was_active = getattr(self, "_wake_word_active", False) + self._wake_word_active = False + self._wake_suspended = False + try: + from tools.wake_word import stop_listening + stop_listening(owner=self) + except Exception: + pass + if _cli._cli_wake_owner is self: + _cli._cli_wake_owner = None + if announce: + if was_active: + _cprint(f"{_DIM}Wake word stopped.{_RST}") + else: + _cprint(f"{_DIM}Wake word is not running.{_RST}") + + def _on_wake_word(self): + """Fired after the detector hears the wake phrase.""" + from cli import _ACCENT, _DIM, _RST, _cprint, logger + if getattr(self, "_should_exit", False): + return + # Ignore wake while a turn is in flight or the mic is already in use. + if self._agent_running or self._voice_recording or getattr(self, "_voice_processing", False): + return + + # Release the mic so STT can capture the command utterance. + try: + from tools.wake_word import pause_listening + if not pause_listening(owner=self): + self._wake_word_active = False + return + except Exception as e: + logger.debug("wake word pause failed: %s", e) + return + self._wake_suspended = True + + # Multi-profile routing: the CLI is a single-profile process, so a + # phrase enrolled by ANOTHER profile can't be routed here — print the + # switch command and re-arm rather than answering as the wrong profile. + try: + from tools.wake_word import get_last_match + _match = get_last_match() + except Exception: + _match = None + if _match and _match[1]: + from tools.wake_word import _active_profile_name + if _match[1] != _active_profile_name(): + _cprint(f"\n{_DIM}Wake phrase for profile '{_match[1]}' — " + f"run: hermes -p {_match[1]}{_RST}") + self._wake_suspended = True # watchdog resumes the listener + return + + _cprint(f"\n{_ACCENT}✦ Wake word detected — listening...{_RST}") + if getattr(self, "_app", None): + try: + self._app.invalidate() + except Exception: + pass + + if getattr(self, "_wake_start_new_session", True): + try: + self.new_session(silent=True) + except Exception as e: + logger.debug("wake word new_session failed: %s", e) + + # Single-utterance capture (not continuous) via the voice pipeline; + # VAD auto-stop transcribes and queues the transcript for process_loop. + with self._voice_lock: + self._voice_mode = True + self._voice_continuous = False + try: + self._voice_start_recording() + except Exception as e: + _cprint(f"{_DIM}Wake capture failed: {e}{_RST}") + + def _start_wake_watchdog(self): + """Resume the paused detector when the CLI returns to a stable idle.""" + from cli import logger + if getattr(self, "_wake_watchdog_started", False): + return + self._wake_watchdog_started = True + + def _loop(): + idle_polls = 0 + try: + while getattr(self, "_wake_word_active", False) and not getattr(self, "_should_exit", False): + time.sleep(0.25) + if not getattr(self, "_wake_suspended", False): + idle_polls = 0 + continue + busy = ( + self._agent_running + or self._voice_recording + or getattr(self, "_voice_processing", False) + or not self._pending_input.empty() + ) + if busy: + idle_polls = 0 + continue + # Require a few consecutive idle polls (~0.75s) so we don't + # resume in the gap between VAD stop and the agent starting. + idle_polls += 1 + if idle_polls >= 3: + idle_polls = 0 + try: + from tools.wake_word import resume_listening + if resume_listening(owner=self): + self._wake_suspended = False + else: + self._wake_word_active = False + except Exception as e: + logger.debug("wake word resume failed: %s", e) + finally: + self._wake_watchdog_started = False + + threading.Thread(target=_loop, daemon=True, name="wake-watchdog").start() + + def _show_wake_word_status(self): + """Show current wake-word listener status.""" + from cli import _ACCENT, _BOLD, _DIM, _RST, _cprint + from tools.wake_word import ( + audio_is_silent, + check_wake_word_requirements, + is_listening, + load_wake_word_config, + owns_listener, + ) + + cfg = load_wake_word_config() + reqs = check_wake_word_requirements(cfg) + owned = owns_listener(self) + state = "LISTENING" if owned and is_listening() else "PAUSED" if owned else "OFF" + + _cprint(f"\n{_BOLD}Wake Word Status{_RST}") + _cprint(f" State: {state}") + _cprint(f" Phrase: \"{reqs['phrase']}\"") + _cprint(f" Provider: {reqs['provider']}") + _cprint(f" Surface: {cfg.get('surface', 'auto')}") + _cprint(f" New session: {'yes' if cfg.get('start_new_session', True) else 'no'}") + if state == "LISTENING" and audio_is_silent(): + _cprint(f" {_ACCENT}⚠ Microphone delivers only silence — the listener can't hear anything.{_RST}") + _cprint(f" {_DIM}On macOS: System Settings > Privacy & Security > Microphone — allow your" + f" terminal/Hermes, then /wake off + /wake on.{_RST}") + if not reqs["available"] and reqs.get("hint"): + _cprint(f" {_DIM}{reqs['hint']}{_RST}") + if not owned: + _cprint(f" {_DIM}Enable with /wake on{_RST}") + + def _tts_lease_async(self, active: bool) -> None: + """Acquire/release this CLI's TTS engine lease in the background. + + The /voice tts toggle (and voice-mode on/off with speech output set) + is the "TTS is about to be needed / no longer needed" signal: + acquiring pre-loads the configured provider so the first reply starts + hot; releasing lets the last-holder path unload resident local models. + Never blocks the toggle and never fails it. + """ + from cli import logger + + def _run(): + try: + from tools.tts_tool import acquire_tts_lease, release_tts_lease + + if active: + acquire_tts_lease("cli:voice-tts") + else: + release_tts_lease("cli:voice-tts") + except Exception as e: + logger.debug("voice: tts lease active=%s failed: %s", active, e) + + threading.Thread(target=_run, name="tts-lease-cli", daemon=True).start() + + def _toggle_voice_tts(self): + """Toggle TTS output for voice mode.""" + from cli import _ACCENT, _DIM, _RST, _cprint + if not self._voice_mode: + _cprint(f"{_DIM}Enable voice mode first: /voice on{_RST}") + return + + with self._voice_lock: + self._voice_tts = not self._voice_tts + status = "enabled" if self._voice_tts else "disabled" + + if self._voice_tts: + from tools.tts_tool import check_tts_requirements + if not check_tts_requirements(): + _cprint(f"{_DIM}Warning: No TTS provider available. Install edge-tts or set API keys.{_RST}") + + # Toggle = warm-up / release signal for the TTS engine (see + # tools.tts_tool.acquire_tts_lease). + self._tts_lease_async(self._voice_tts) + + _cprint(f"{_ACCENT}Voice TTS {status}.{_RST}") + + def _show_voice_status(self): + """Show current voice mode status.""" + from cli import _BOLD, _RST, _cprint + from tools.voice_mode import check_voice_requirements + + reqs = check_voice_requirements() + + _cprint(f"\n{_BOLD}Voice Mode Status{_RST}") + _cprint(f" Mode: {'ON' if self._voice_mode else 'OFF'}") + _cprint(f" TTS: {'ON' if self._voice_tts else 'OFF'}") + _cprint(f" Recording: {'YES' if self._voice_recording else 'no'}") + # Display the startup-pinned label so /voice status always + # matches the live prompt_toolkit binding (Copilot round-14 on + # #19835, same class as round-13). Reading live config here + # would drift after a mid-session config edit. + _cprint(f" Record key: {self._voice_record_key_label()}") + _cprint(f"\n {_BOLD}Requirements:{_RST}") + for line in reqs["details"].split("\n"): + _cprint(f" {line}") diff --git a/tests/cli/test_compress_type_ahead.py b/tests/cli/test_compress_type_ahead.py index a14ed4f8f4..cb4e492f46 100644 --- a/tests/cli/test_compress_type_ahead.py +++ b/tests/cli/test_compress_type_ahead.py @@ -128,7 +128,7 @@ def test_handle_enter_never_gates_on_command_running(): ``_command_blocks_input`` check inside ``handle_enter`` would let a future edit silently drop type-ahead submissions during /compress. """ - cli_path = Path(__file__).resolve().parents[2] / "cli.py" + cli_path = Path(__file__).resolve().parents[2] / "hermes_cli" / "cli_tui_mixin.py" tree = ast.parse(cli_path.read_text(encoding="utf-8")) target = None @@ -136,7 +136,7 @@ def test_handle_enter_never_gates_on_command_running(): if isinstance(node, ast.FunctionDef) and node.name == "_tui_handle_enter": target = node break - assert target is not None, "handle_enter closure not found in cli.py" + assert target is not None, "_tui_handle_enter not found in cli_tui_mixin.py" offenders = [ node.attr diff --git a/tests/cli/test_steer_inline_repaint_34569.py b/tests/cli/test_steer_inline_repaint_34569.py index ed1d7ddd86..2eebff8c68 100644 --- a/tests/cli/test_steer_inline_repaint_34569.py +++ b/tests/cli/test_steer_inline_repaint_34569.py @@ -27,7 +27,7 @@ from pathlib import Path def _load_handle_enter_node() -> ast.FunctionDef: """Extract the ``handle_enter`` nested function node from cli.py.""" - cli_path = Path(__file__).resolve().parents[2] / "cli.py" + cli_path = Path(__file__).resolve().parents[2] / "hermes_cli" / "cli_tui_mixin.py" tree = ast.parse(cli_path.read_text(encoding="utf-8")) target = None @@ -35,7 +35,7 @@ def _load_handle_enter_node() -> ast.FunctionDef: if isinstance(node, ast.FunctionDef) and node.name == "_tui_handle_enter": target = node break - assert target is not None, "handle_enter closure not found in cli.py" + assert target is not None, "_tui_handle_enter not found in cli_tui_mixin.py" return target diff --git a/tests/cli/test_tool_progress_scrollback.py b/tests/cli/test_tool_progress_scrollback.py index 45ef171bd8..e0c04e8181 100644 --- a/tests/cli/test_tool_progress_scrollback.py +++ b/tests/cli/test_tool_progress_scrollback.py @@ -54,8 +54,15 @@ def _make_cli(tool_progress="all", verbose=_UNSET): with patch.object(mod, "get_tool_definitions", return_value=[]), \ patch.dict(mod.__dict__, {"CLI_CONFIG": _clean_config}): if verbose is _UNSET: - return mod.HermesCLI() - return mod.HermesCLI(verbose=verbose) + inst = mod.HermesCLI() + else: + inst = mod.HermesCLI(verbose=verbose) + # patch.dict(sys.modules) above restores the pre-import state on exit, which + # DROPS ``cli`` when this was the first import. The mixin handlers resolve + # ``_cprint`` via a lazy ``from cli import ...``, so ``cli`` must stay + # registered for ``patch.object(_cli_mod, "_cprint")`` to intercept. + sys.modules["cli"] = mod + return inst class TestToolProgressScrollback: