"""Process-wide voice recording + TTS API for the TUI gateway.""" from __future__ import annotations import json import logging import os import re import sys import threading from typing import Any, Callable, Optional # Modifier aliases mirrored from the TUI parser (``ui-tui/src/lib/platform.ts``) # ``_MOD_ALIASES`` table — the contract that removes the cross-runtime # mismatch Copilot flagged in round-9 on #19835. # # ``super``/``win``/``windows`` are intentionally absent: prompt_toolkit # has no super/meta modifier for the Cmd key, so those spellings are # TUI-only. The normalizer below returns the documented default # (``c-b``) for them — a silent fallback was preferred to a hard # startup crash (Copilot round-11). The CLI binding site # (``_register_voice_handler`` in cli.py) logs a warning when that # fallback fires so users see why their TUI-only shortcut isn't # bound in the classic CLI. _VOICE_MOD_ALIASES = { "ctrl": "c-", "control": "c-", "alt": "a-", "option": "a-", "opt": "a-", } # Named keys prompt_toolkit accepts in ``c-`` / ``a-`` form. # Aliases collapse to prompt_toolkit's canonical spelling so the same # config value binds identically in both runtimes (Copilot round-10 on # #19835). _VOICE_NAMED_KEYS = { "space": "space", "spc": "space", "enter": "enter", "return": "enter", "ret": "enter", "tab": "tab", "escape": "escape", "esc": "escape", "backspace": "backspace", "bs": "backspace", "delete": "delete", "del": "delete", } # ``useInputHandlers()`` intercepts these before the voice check runs, # so a binding like ``ctrl+c`` (interrupt), ``ctrl+d`` (quit), or # ``ctrl+l`` (clear screen) would be advertised in /voice status but # never fire push-to-talk — the same blocklist the TUI parser uses. _VOICE_RESERVED_CTRL_CHARS = frozenset({"c", "d", "l"}) # On macOS the classic CLI's prompt_toolkit bindings for copy / exit / # clear also claim ``a-c`` / ``a-d`` / ``a-l`` via the action-modifier # lookup, and hermes-ink reports Alt as ``key.meta`` on many terminals. # Mirror the TUI parser's darwin-only reservation so ``option+c`` etc. # don't bind Alt+C in the CLI while the TUI silently falls back to # Ctrl+B (Copilot round-14 on #19835). _VOICE_RESERVED_ALT_CHARS_MAC = frozenset({"c", "d", "l"}) _DEFAULT_PT_KEY = "c-b" def voice_record_key_from_config(cfg: Any) -> Any: """Shape-safe ``cfg.voice.record_key`` lookup. ``load_config()`` deep-merges raw YAML and preserves scalar overrides, so a hand-edited ``voice: true`` / ``voice: cmd+b`` leaves ``cfg["voice"]`` as a bool/str instead of a dict, and the naive ``.get("voice", {}).get("record_key")`` chain raises AttributeError before voice can even start (Copilot round-11 on #19835). """ voice = cfg.get("voice") if isinstance(cfg, dict) else None return voice.get("record_key") if isinstance(voice, dict) else None def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str: """Coerce ``voice.record_key`` into prompt_toolkit's ``c-x`` / ``a-x`` format. Mirrors the TUI parser contract (``ui-tui/src/lib/platform.ts``) so one config value binds the same shortcut in both runtimes: * non-string / empty / typo'd / bare-char / multi-modifier / reserved ``ctrl+c|d|l`` → documented default ``c-b`` * single-char keys: ``ctrl+o`` → ``c-o`` * named keys: ``ctrl+space`` → ``c-space`` (aliases collapse: ``ctrl+return`` → ``c-enter``) * ``super`` / ``win`` / ``windows`` → ``c-b`` (TUI-only modifiers — prompt_toolkit has no super mod; the CLI binding site is expected to warn when this fallback fires so users see the cross-runtime split, Copilot round-11 on #19835) """ if not isinstance(raw, str): return _DEFAULT_PT_KEY parts = [p.strip() for p in raw.strip().lower().split("+") if p.strip()] # Exactly ``modifier+key``. Multi-modifier chords like ``ctrl+alt+r`` bind # different shortcuts in prompt_toolkit (a-c-r form) and hermes-ink rejects # them; a bare char / named key (no modifier) is refused by the TUI parser. # Both collapse to the documented default so the runtimes agree. if len(parts) != 2: return _DEFAULT_PT_KEY modifier_token, key_token = parts # ``super`` / ``win`` / ``windows`` are TUI-only (prompt_toolkit has # no super modifier, so ``@kb.add(super+b)`` crashes the CLI at # startup). Fall back to the documented default here; the CLI # binding site is expected to log a warning when the configured # value is one of these spellings so users know the TUI+CLI # runtimes diverge on that shortcut (Copilot round-11 on #19835). if modifier_token in {"super", "win", "windows"}: return _DEFAULT_PT_KEY normalized_mod = _VOICE_MOD_ALIASES.get(modifier_token) if not normalized_mod: return _DEFAULT_PT_KEY # Single-char key: reject reserved-ctrl chords that the TUI would # also block at parse time, plus the mac-only alt reservation. if len(key_token) == 1: reserved = ( _VOICE_RESERVED_CTRL_CHARS if normalized_mod == "c-" else _VOICE_RESERVED_ALT_CHARS_MAC if sys.platform == "darwin" else frozenset() ) return _DEFAULT_PT_KEY if key_token in reserved else f"{normalized_mod}{key_token}" # Multi-char key token must be a known named key; typos like # ``ctrl+spcae`` fall back to the default rather than being passed # through as ``c-spcae`` (which prompt_toolkit would reject). named = _VOICE_NAMED_KEYS.get(key_token) return f"{normalized_mod}{named}" if named else _DEFAULT_PT_KEY def pt_key_to_sequence(pt_key: str) -> tuple[str, ...]: """Convert a prompt_toolkit key specifier (e.g. 'c-b' or 'a-v') to a sequence tuple.""" if isinstance(pt_key, str) and pt_key.startswith("a-"): return ("escape", pt_key[2:]) return (pt_key,) def format_voice_record_key_for_status(raw: Any) -> str: """Render ``voice.record_key`` for ``/voice status`` in CLI-friendly form. Mirrors the TUI's ``formatVoiceRecordKey``: returns ``Ctrl+B`` / ``Alt+Space`` / ``Ctrl+Enter``. Malformed configs surface as the documented default so status never advertises a shortcut that won't bind (Copilot round-10 on #19835). """ # The normalizer only ever yields ``c-`` / ``a-`` (or the default ``c-b``). normalized = normalize_voice_record_key_for_prompt_toolkit(raw) prefix = "Alt+" if normalized.startswith("a-") else "Ctrl+" key = normalized[2:] return prefix + key[0].upper() + key[1:] from tools.voice_mode import ( create_audio_recorder, is_voice_stop_phrase, is_whisper_hallucination, play_audio_file, transcribe_recording, ) logger = logging.getLogger(__name__) def _debug(msg: str) -> None: """Emit a debug breadcrumb when HERMES_VOICE_DEBUG=1. Goes to stderr so the TUI gateway wraps it as a gateway.stderr event, which createGatewayEventHandler shows as an Activity line — exactly what we need to diagnose "why didn't the loop auto-restart?" in the user's real terminal without shipping a separate debug RPC. Any OSError / BrokenPipeError is swallowed because this fires from background threads (silence callback, TTS daemon, beep) where a broken stderr pipe must not kill the whole gateway — the main command pipe (stdin+stdout) is what actually matters. """ if os.environ.get("HERMES_VOICE_DEBUG", "").strip() != "1": return try: print(f"[voice] {msg}", file=sys.stderr, flush=True) except (BrokenPipeError, OSError): pass def _beeps_enabled() -> bool: """CLI parity: voice.beep_enabled in config.yaml (default True).""" try: from hermes_cli.config import load_config from utils import is_truthy_value voice_cfg = load_config().get("voice", {}) if isinstance(voice_cfg, dict): # is_truthy_value handles quoted YAML strings like "false" # which bool() would misread as True (#49883). return is_truthy_value(voice_cfg.get("beep_enabled", True), default=True) except Exception: pass return True def _play_beep(frequency: int, count: int = 1) -> None: """Audible cue matching cli.py's record/stop beeps. 880 Hz single-beep on start (cli.py:_voice_start_recording line 7532), 660 Hz double-beep on stop (cli.py:_voice_stop_and_transcribe line 7585). Best-effort — sounddevice failures are silently swallowed so the voice loop never breaks because a speaker was unavailable. """ if not _beeps_enabled(): return try: from tools.voice_mode import play_beep play_beep(frequency=frequency, count=count) except Exception as e: _debug(f"beep {frequency}Hz failed: {e}") def _safe_call(cb: Optional[Callable], *args: Any, warn: Optional[str] = None) -> None: """Invoke an optional consumer callback, swallowing its exceptions. ``warn`` is a ``logger.warning`` format with one ``%s`` slot for the exception; without it failures are silently ignored (status/limit callbacks are fire-and-forget). """ if not cb: return try: cb(*args) except Exception as e: if warn: logger.warning(warn, e) def _transcribe_wav(wav_path: str, fail_msg: str, debug_prefix: Optional[str] = None) -> Optional[str]: """Transcribe ``wav_path``, unlink it, and return the cleaned transcript (or None). transcribe_recording returns {"success": bool, "transcript": str, "error": str?} — NOT {"text": str}. Using the wrong key silently produced empty transcripts even when Groq/local STT returned fine, which masqueraded as "not hearing the user" to the caller. Empty text and Whisper hallucinations are dropped; failures are logged with ``fail_msg``. """ try: result = transcribe_recording(wav_path) success = bool(result.get("success")) text = (result.get("transcript") or "").strip() if debug_prefix: _debug( f"{debug_prefix}: transcribe -> success={success} " f"text={text!r} err={result.get('error')!r}" ) if success and text and not is_whisper_hallucination(text): return text except Exception as e: logger.warning(fail_msg, e) if debug_prefix: _debug(f"{debug_prefix}: transcribe raised {type(e).__name__}: {e}") finally: try: if os.path.isfile(wav_path): os.unlink(wav_path) except Exception: pass return None def _deactivate(on_status: Optional[Callable[[str], None]] = None) -> None: """Mark the continuous loop inactive and (optionally) report ``"idle"``.""" global _continuous_active with _continuous_lock: _continuous_active = False _safe_call(on_status, "idle") # ── Push-to-talk state ─────────────────────────────────────────────── _recorder = None _recorder_lock = threading.Lock() # ── Continuous (VAD) state ─────────────────────────────────────────── _continuous_lock = threading.Lock() _continuous_active = False _continuous_stopping = False _continuous_auto_restart: bool = True _continuous_recorder: Any = None # ── TTS-vs-STT feedback guard ──────────────────────────────────────── # When TTS plays the agent reply over the speakers, the live microphone # picks it up and transcribes the agent's own voice as user input — an # infinite loop the agent happily joins ("Ha, looks like we're in a loop"). # This Event mirrors cli.py:_voice_tts_done: cleared while speak_text is # playing, set while silent. _continuous_on_silence waits on it before # re-arming the recorder, and speak_text itself cancels any live capture # before starting playback so the tail of the previous utterance doesn't # leak into the mic. _tts_playing = threading.Event() _tts_playing.set() # initially "not playing" # ── Silence-count hold (agent busy) ────────────────────────────────── # While the agent is mid-turn (thinking / tool-calling, possibly for # minutes) or TTS is playing, the user is CORRECTLY silent — those cycles # must not count toward the no-speech limit or a long tool run ends the # voice chat under the user (#silence-must-not-end-the-chat). The host # surface (tui_gateway) registers a probe that reports "agent busy"; # TTS-playing is already tracked via _tts_playing above. _voice_busy_probe: Optional[Callable[[], bool]] = None def set_voice_busy_probe(probe: Optional[Callable[[], bool]]) -> None: """Register a callable that returns True while the agent is mid-turn. Called by the hosting surface (tui_gateway registers one that checks every session's ``running`` flag). ``None`` clears it. The probe must be cheap and thread-safe — it runs on the silence- callback thread. """ global _voice_busy_probe _voice_busy_probe = probe def _voice_activity_held() -> bool: """True while silent cycles must NOT count toward the no-speech limit. Held when TTS is playing (the user is listening) or when the registered busy probe reports the agent mid-turn (the user is waiting). Fail-open to "not held" so a broken probe can never make the voice chat immortal. """ if not _tts_playing.is_set(): return True probe = _voice_busy_probe if probe is None: return False try: return bool(probe()) except Exception: return False _continuous_on_transcript: Optional[Callable[[str], None]] = None _continuous_on_status: Optional[Callable[[str], None]] = None _continuous_on_silent_limit: Optional[Callable[[], None]] = None # Explicit user-intent stop signal: fired when the user SAYS a bare stop # phrase ("stop"). Distinct from on_silent_limit (a timeout) so consumers # (TUI, desktop) can end the conversation like a manual stop instead of # reporting "no speech detected". When unset, on_silent_limit fires as a # fallback so older callers still turn voice off. _continuous_on_stop_phrase: Optional[Callable[[str], None]] = None _continuous_no_speech_count = 0 _CONTINUOUS_NO_SPEECH_LIMIT = 3 # ── Push-to-talk API ───────────────────────────────────────────────── def start_recording() -> None: """Begin capturing from the default input device (push-to-talk).""" global _recorder with _recorder_lock: if _recorder is not None and getattr(_recorder, "is_recording", False): return rec = create_audio_recorder() rec.start() _recorder = rec def stop_and_transcribe() -> Optional[str]: """Stop the active push-to-talk recording, transcribe, return text.""" global _recorder with _recorder_lock: rec = _recorder _recorder = None if rec is None: return None wav_path = rec.stop() if not wav_path: return None return _transcribe_wav(wav_path, "voice transcription failed: %s") # ── Continuous (VAD) API ───────────────────────────────────────────── def start_continuous( on_transcript: Callable[[str], None], on_status: Optional[Callable[[str], None]] = None, on_silent_limit: Optional[Callable[[], None]] = None, silence_threshold: int = 200, silence_duration: float = 3.0, auto_restart: bool = True, max_recording_seconds: float = 0.0, on_stop_phrase: Optional[Callable[[str], None]] = None, ) -> bool: """Start a VAD-driven continuous recording loop. ``max_recording_seconds`` is the hard cap on a single recording's length (``voice.max_recording_seconds``); any non-positive or non-numeric value disables the cap, preserving the previous unbounded behaviour. ``on_stop_phrase`` is called with the (stripped) transcript when the user utters a bare voice stop phrase (``voice.stop_phrases``, default "stop"). The loop halts first, so the consumer only needs to reflect "voice off" — exactly like the user pressing the manual stop control. """ global _continuous_active, _continuous_recorder, _continuous_auto_restart global _continuous_on_transcript, _continuous_on_status, _continuous_on_silent_limit global _continuous_on_stop_phrase global _continuous_no_speech_count with _continuous_lock: if _continuous_active: _debug("start_continuous: already active — no-op") return True if _continuous_stopping: _debug("start_continuous: stop/transcribe in progress — busy") return False _continuous_active = True _continuous_auto_restart = auto_restart _continuous_on_transcript = on_transcript _continuous_on_status = on_status _continuous_on_silent_limit = on_silent_limit _continuous_on_stop_phrase = on_stop_phrase if auto_restart: _continuous_no_speech_count = 0 if _continuous_recorder is None: _continuous_recorder = create_audio_recorder() rec = _continuous_recorder rec._silence_threshold = silence_threshold rec._silence_duration = silence_duration # Same numeric-with-bool-excluded guard as the CLI wiring in # cli.py:_voice_start_recording — <= 0 (or garbage) disables the cap. rec._max_recording_seconds = ( max_recording_seconds if isinstance(max_recording_seconds, (int, float)) and not isinstance(max_recording_seconds, bool) and max_recording_seconds > 0 else 0.0 ) _debug( f"start_continuous: begin (threshold={silence_threshold}, duration={silence_duration}s)" ) # CLI parity: single 880 Hz beep *before* opening the stream — placing # the beep after stream.start() on macOS triggers a CoreAudio conflict # (cli.py:7528 comment). _play_beep(frequency=880, count=1) try: rec.start(on_silence_stop=_continuous_on_silence) except Exception as e: logger.error("failed to start continuous recording: %s", e) _debug(f"start_continuous: rec.start raised {type(e).__name__}: {e}") _deactivate() raise _safe_call(on_status, "listening") return True def stop_continuous(force_transcribe: bool = False) -> None: """Stop the active continuous loop and release the microphone. Idempotent — calling while not active is a no-op. If ``force_transcribe`` is True, the recorder stops synchronously, then transcription/cleanup runs on a background thread before reporting ``"idle"``. Otherwise the buffer is discarded. """ global _continuous_active, _continuous_on_transcript, _continuous_stopping global _continuous_on_status, _continuous_on_silent_limit global _continuous_on_stop_phrase global _continuous_recorder, _continuous_no_speech_count with _continuous_lock: if not _continuous_active: return _continuous_active = False rec = _continuous_recorder on_status = _continuous_on_status on_transcript = _continuous_on_transcript on_silent_limit = _continuous_on_silent_limit on_stop_phrase = _continuous_on_stop_phrase auto_restart = _continuous_auto_restart track_no_speech = force_transcribe and not auto_restart _continuous_stopping = rec is not None _continuous_on_transcript = None _continuous_on_status = None _continuous_on_silent_limit = None _continuous_on_stop_phrase = None if not track_no_speech: _continuous_no_speech_count = 0 if rec is not None: if force_transcribe and on_transcript: _safe_call(on_status, "transcribing") try: wav_path = rec.stop() except Exception as e: logger.warning("failed to stop recorder: %s", e) try: rec.cancel() except Exception as cancel_error: logger.warning("failed to cancel recorder: %s", cancel_error) wav_path = None def _transcribe_and_cleanup(): global _continuous_no_speech_count, _continuous_stopping transcript: Optional[str] = None should_halt = False if wav_path: transcript = _transcribe_wav(wav_path, "failed to stop/transcribe recorder: %s") stop_phrase = bool(transcript and is_voice_stop_phrase(transcript)) if stop_phrase: # Bare stop phrase — explicit user intent to end the # voice chat. Never sent to the agent; fire the # dedicated signal so the consumer (TUI / desktop) # ends the conversation instead of silently re-arming # the next capture (with auto_restart=False the CLIENT # drives the loop, so discarding the transcript alone # would leave the conversation running forever). _debug( f"stop_continuous: stop phrase {transcript!r} — ending voice chat" ) stop_text = transcript or "" transcript = None if on_stop_phrase is not None: _safe_call(on_stop_phrase, stop_text) else: _safe_call(on_silent_limit) if transcript: _safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s") if track_no_speech: held = _voice_activity_held() with _continuous_lock: if transcript or stop_phrase: _continuous_no_speech_count = 0 elif held: # Agent busy / TTS playing — the user is # correctly silent; don't count the cycle. _debug( "stop_continuous: silent cycle ignored " "(agent busy or TTS playing)" ) else: _continuous_no_speech_count += 1 should_halt = ( _continuous_no_speech_count >= _CONTINUOUS_NO_SPEECH_LIMIT ) if should_halt: _continuous_no_speech_count = 0 if should_halt: _safe_call(on_silent_limit) _play_beep(frequency=660, count=2) with _continuous_lock: _continuous_stopping = False _safe_call(on_status, "idle") threading.Thread(target=_transcribe_and_cleanup, daemon=True).start() return else: try: # cancel() (not stop()) discards buffered frames — the loop # is over, we don't want to transcribe a half-captured turn. rec.cancel() except Exception as e: logger.warning("failed to cancel recorder: %s", e) with _continuous_lock: _continuous_stopping = False # Audible "recording stopped" cue (CLI parity: same 660 Hz × 2 the # silence-auto-stop path plays). _play_beep(frequency=660, count=2) _safe_call(on_status, "idle") def is_continuous_active() -> bool: """Whether a continuous voice loop is currently running.""" with _continuous_lock: return _continuous_active def _continuous_on_silence() -> None: """AudioRecorder silence callback — runs in a daemon thread. Stops the current capture, transcribes, delivers text via ``on_transcript``, and — if the loop is still active — starts the next capture. Three consecutive silent cycles end the loop. """ global _continuous_active, _continuous_no_speech_count _debug("_continuous_on_silence: fired") with _continuous_lock: if not _continuous_active: _debug("_continuous_on_silence: loop inactive — abort") return rec = _continuous_recorder on_transcript = _continuous_on_transcript on_status = _continuous_on_status on_silent_limit = _continuous_on_silent_limit on_stop_phrase = _continuous_on_stop_phrase if rec is None: _debug("_continuous_on_silence: no recorder — abort") return _safe_call(on_status, "transcribing") wav_path = rec.stop() # Peak RMS is the critical diagnostic when stop() returns None despite # the VAD firing — tells us at a glance whether the mic was too quiet # for SILENCE_RMS_THRESHOLD (200) or the VAD + peak checks disagree. peak_rms = getattr(rec, "_peak_rms", -1) _debug( f"_continuous_on_silence: rec.stop -> {wav_path!r} (peak_rms={peak_rms})" ) # CLI parity: double 660 Hz beep after the stream stops (safe from the # CoreAudio conflict that blocks pre-start beeps). _play_beep(frequency=660, count=2) transcript: Optional[str] = None if wav_path: transcript = _transcribe_wav( wav_path, "continuous transcription failed: %s", "_continuous_on_silence" ) stop_phrase = bool(transcript and is_voice_stop_phrase(transcript)) stop_text = (transcript or "") if stop_phrase else "" if stop_phrase: # User said a bare stop phrase ("stop") — end the voice chat. # Not delivered to the agent; the loop halts and the explicit # on_stop_phrase signal (fallback: on_silent_limit) tells every UI # (TUI, desktop) to end the conversation like a manual stop. _debug(f"_continuous_on_silence: stop phrase {transcript!r} — ending loop") transcript = None # Silent cycle while the agent is mid-turn or TTS is playing: the user # is CORRECTLY quiet (waiting/listening), so the cycle must not count # toward the no-speech limit — a multi-minute tool run would otherwise # end the voice chat under the user. Checked outside the lock (probe # may call into the host surface). _silence_held = (transcript is None and not stop_phrase and _voice_activity_held()) with _continuous_lock: if not _continuous_active: # User stopped us while we were transcribing — discard. _debug("_continuous_on_silence: stopped during transcribe — no restart") return if transcript: _continuous_no_speech_count = 0 elif _silence_held: _debug( "_continuous_on_silence: silent cycle ignored " "(agent busy or TTS playing)" ) elif not stop_phrase: _continuous_no_speech_count += 1 should_halt = stop_phrase or ( _continuous_no_speech_count >= _CONTINUOUS_NO_SPEECH_LIMIT ) no_speech = _continuous_no_speech_count if transcript: _safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s") if should_halt: _debug( "_continuous_on_silence: halting " f"({'stop phrase' if stop_phrase else f'{no_speech} silent cycles'})" ) with _continuous_lock: _continuous_active = False _continuous_no_speech_count = 0 if stop_phrase and on_stop_phrase is not None: # Explicit user-intent stop — distinct from the no-speech timeout # so consumers can report "voice chat ended" instead of "no # speech detected". _safe_call(on_stop_phrase, stop_text) else: _safe_call(on_silent_limit) _safe_call(rec.cancel) _safe_call(on_status, "idle") return # CLI parity (cli.py:10619-10621): wait for any in-flight TTS to # finish before re-arming the mic, then leave a small gap to avoid # catching the tail of the speaker output. Without this the voice # loop becomes a feedback loop — the agent's spoken reply lands # back in the mic and gets re-submitted. if not _tts_playing.is_set(): _debug("_continuous_on_silence: waiting for TTS to finish") _tts_playing.wait(timeout=60) import time as _time _time.sleep(0.3) # User may have stopped the loop during the wait. with _continuous_lock: if not _continuous_active: _debug("_continuous_on_silence: stopped while waiting for TTS") return if _continuous_auto_restart: # Restart for the next turn. _debug(f"_continuous_on_silence: restarting loop (no_speech={no_speech})") _play_beep(frequency=880, count=1) try: rec.start(on_silence_stop=_continuous_on_silence) except Exception as e: logger.error("failed to restart continuous recording: %s", e) _debug(f"_continuous_on_silence: restart raised {type(e).__name__}: {e}") _deactivate(on_status) return _safe_call(on_status, "listening") else: # Do not auto-restart. Clean up state and notify idle. _debug("_continuous_on_silence: auto_restart=False, stopping loop") _deactivate(on_status) # ── TTS API ────────────────────────────────────────────────────────── # Legacy markdown stripper used only when tools.tts_text_normalize is unavailable. _LEGACY_TTS_STRIP = [ (re.compile(r'```[\s\S]*?```'), ' '), # fenced code blocks (re.compile(r'\[([^\]]+)\]\([^)]+\)'), r'\1'), # [text](url) → text (re.compile(r'https?://\S+'), ''), # bare URLs (re.compile(r'\*\*(.+?)\*\*'), r'\1'), # bold (re.compile(r'\*(.+?)\*'), r'\1'), # italic (re.compile(r'`(.+?)`'), r'\1'), # inline code (re.compile(r'^#+\s*', re.MULTILINE), ''), # headers (re.compile(r'^\s*[-*]\s+', re.MULTILINE), ''), # list bullets (re.compile(r'---+'), ''), # horizontal rules (re.compile(r'\n{3,}'), '\n\n'), # excess newlines ] def _speak_text_streaming(text: str, stop_event: Optional[threading.Event] = None) -> bool: """Speak ``text`` via the generic streaming dispatcher; True on success. Bridges the one-shot ``speak_text`` contract onto the shared ``stream_tts_to_speaker`` pipeline (tools.tts_tool): the full reply is fed as a single delta + end-of-text sentinel, and we block until the pipeline's done event fires — same blocking semantics the sync path has, so callers (and the mic re-arm logic in ``speak_text``) see no behavioral difference beyond earlier first audio. ``stop_event`` (optional) is wired straight into the pipeline so external barge-in / stop paths can cut streaming playback — without it the pipeline's stop event was private and speech over this path was uninterruptible (the desktop/TUI fallback-speak hole). """ import queue as _queue import threading as _threading from tools.tts_tool import stream_tts_to_speaker text_queue: "_queue.Queue" = _queue.Queue() text_queue.put(text) text_queue.put(None) # end-of-text sentinel if stop_event is None: stop_event = _threading.Event() done_event = _threading.Event() stream_tts_to_speaker(text_queue, stop_event, done_event) return done_event.is_set() def speak_text(text: str, stop_event: Optional[threading.Event] = None) -> None: """Synthesize ``text`` with the configured TTS provider and play it. While playback is in flight the module-level _tts_playing Event is cleared so the continuous- recording loop knows to wait before re-arming the mic (otherwise the agent's spoken reply feedback-loops through the microphone and the agent ends up replying to itself). """ if not text or not text.strip(): return import tempfile import time # Cancel any live capture before we open the speakers — otherwise the # last ~200ms of the user's turn tail + the first syllables of our TTS # both end up in the next recording window. The continuous loop will # re-arm itself after _tts_playing flips back (see _continuous_on_silence). paused_recording = False with _continuous_lock: if ( _continuous_active and _continuous_recorder is not None and getattr(_continuous_recorder, "is_recording", False) ): try: _continuous_recorder.cancel() paused_recording = True except Exception as e: logger.warning("failed to pause recorder for TTS: %s", e) _tts_playing.clear() _debug(f"speak_text: TTS begin (paused_recording={paused_recording})") try: from tools.tts_tool import text_to_speech_tool # One dispatcher, zero parallel streaming implementations (#58930): # when the configured provider has a chunked streamer registered in # tools.tts_streaming, route the whole reply through the same # stream_tts_to_speaker pipeline the CLI voice mode uses — audio # starts on sentence one instead of after full synthesis. Falls # through to the legacy whole-file path when no streamer resolves. try: from tools.tts_streaming import resolve_streaming_provider from tools.tts_tool import _load_tts_config if ( resolve_streaming_provider(_load_tts_config()) is not None and _speak_text_streaming(text, stop_event) ): return except Exception as e: _debug(f"speak_text: streaming dispatch unavailable ({e}); using sync path") # Shared cleaner (tools/tts_text_normalize): markdown, emoji, # ⋗ blocks, verifier footer, units, newline flattening. # The TTS tool owns provider request limits and long-form chunking. try: from tools.tts_text_normalize import prepare_spoken_text tts_text = prepare_spoken_text(text, max_chars=None) except Exception: # Legacy fallback pipeline — keep speak_text best-effort. tts_text = text for pattern, repl in _LEGACY_TTS_STRIP: tts_text = pattern.sub(repl, tts_text) tts_text = tts_text.strip() if not tts_text: return # MP3 output path, pre-chosen so we can play the MP3 directly even # when text_to_speech_tool auto-converts to OGG for messaging # platforms. afplay's OGG support is flaky, MP3 always works. os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True) mp3_path = os.path.join( tempfile.gettempdir(), "hermes_voice", f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3", ) _debug(f"speak_text: synthesizing {len(tts_text)} chars -> {mp3_path}") raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path) try: tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {} except Exception: tts_result = {} # The tool result is authoritative — it may return multiple files # for long-form chunked output. Play each in order. play_paths = tts_result.get("file_paths") or [ tts_result.get("file_path") or mp3_path ] played_any = False for play_path in play_paths if tts_result.get("success") else []: if os.path.isfile(play_path) and os.path.getsize(play_path) > 0: _debug( f"speak_text: playing {play_path} " f"({os.path.getsize(play_path)} bytes)" ) play_audio_file(play_path) played_any = True for path in set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]): if os.path.isfile(path): try: os.unlink(path) except OSError: pass if not played_any: _debug(f"speak_text: TTS tool produced no audio at {mp3_path}") except Exception as e: logger.warning("Voice TTS playback failed: %s", e) _debug(f"speak_text raised {type(e).__name__}: {e}") finally: _tts_playing.set() _debug("speak_text: TTS done") # Re-arm the mic so the user can answer without pressing Ctrl+B. # Small delay lets the OS flush speaker output and afplay fully # release the audio device before sounddevice re-opens the input. if paused_recording: time.sleep(0.3) with _continuous_lock: if _continuous_active and _continuous_recorder is not None: try: _continuous_recorder.start( on_silence_stop=_continuous_on_silence ) _debug("speak_text: recording resumed after TTS") except Exception as e: logger.warning( "failed to resume recorder after TTS: %s", e )