diff --git a/tools/voice_mode.py b/tools/voice_mode.py index d6a102d3dc..06bc7279bf 100644 --- a/tools/voice_mode.py +++ b/tools/voice_mode.py @@ -1,12 +1,8 @@ -"""Voice Mode -- Push-to-talk audio recording and playback for the CLI. +"""Voice Mode -- push-to-talk recording and playback for the CLI. -Provides audio capture via sounddevice, WAV encoding via stdlib wave, -STT dispatch via tools.transcription_tools, and TTS playback via -sounddevice or system audio players. - -Dependencies (optional): - pip install sounddevice numpy - or: uv sync --extra voice +Audio capture via sounddevice, WAV encoding via stdlib wave, STT dispatch via +tools.transcription_tools, TTS playback via sounddevice or system players. +Optional deps: ``pip install sounddevice numpy`` (or ``uv sync --extra voice``). """ import logging @@ -17,6 +13,7 @@ import shlex import shutil import subprocess import sys +from collections import deque from pathlib import Path import tempfile import threading @@ -40,43 +37,41 @@ from tools.voice_mode_transcript import ( # noqa: F401 - re-exported; tests pat is_tts_echo, voice_stop_hint, ) +from hermes_constants import is_termux as _is_termux_environment + +# ── Recording parameters ── +SAMPLE_RATE = 16000 # Whisper native rate +CHANNELS = 1 +DTYPE = "int16" +SAMPLE_WIDTH = 2 # bytes per sample (int16) +SILENCE_RMS_THRESHOLD = 200 # RMS below this = silence (int16 range 0-32767) +SILENCE_DURATION_SECONDS = 3.0 # continuous silence before auto-stop +_TEMP_DIR = os.path.join(tempfile.gettempdir(), "hermes_voice") + # ── Lazy audio imports ── # Never imported at module level: crashes headless environments (SSH, Docker, # WSL, no PortAudio). def _import_audio(): - """Lazy-import sounddevice and numpy; returns (sd, np). Raises ImportError - or OSError when unavailable (e.g. PortAudio missing on headless servers).""" + """Lazy-import (sounddevice, numpy); raises ImportError/OSError when unavailable.""" import sounddevice as sd import numpy as np return sd, np -def _import_numpy(): - """Lazy-import numpy only — for synthesizing samples where sounddevice - must NOT be imported (see _sounddevice_output_allowed).""" - import numpy as np - return np - - def _sounddevice_output_allowed() -> bool: """Whether sounddevice may be used for audio OUTPUT. False on macOS: initializing PortAudio/CoreAudio for output triggers a - kTCCServiceMediaLibrary prompt even though playback needs no media-library - access, so all output goes through ``afplay`` there. Does NOT affect - *input* (recording), which legitimately needs microphone permission. + kTCCServiceMediaLibrary prompt, so all output goes through ``afplay`` there. + Does NOT affect input (recording), which legitimately needs mic permission. """ return platform.system() != "Darwin" def _play_int16_via_tempfile(audio, sample_rate: int) -> None: - """Write int16 mono PCM to a temp WAV and play it via play_audio_file. - - Used on macOS so tone/beep output goes through ``afplay`` instead of - sounddevice (avoids the TCC media-library prompt). - """ + """Play int16 mono PCM via a temp WAV + play_audio_file (macOS: afplay, no TCC prompt).""" tmp_path = None try: tmp = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) @@ -86,8 +81,7 @@ def _play_int16_via_tempfile(audio, sample_rate: int) -> None: except Exception as e: logger.debug("Tone tempfile playback failed: %s", e) finally: - if tmp_path: - _unlink_quietly(tmp_path) + _unlink_quietly(tmp_path) def _write_wav_frames(dest, frames: bytes, sample_rate: int) -> None: @@ -110,7 +104,6 @@ def _unlink_quietly(path: Optional[str]) -> None: def _audio_available() -> bool: - """Return True if audio libraries can be imported.""" try: _import_audio() return True @@ -119,13 +112,11 @@ def _audio_available() -> bool: def _rms(np, data) -> float: - """Root-mean-square level of an int16 block.""" return float(np.sqrt(np.mean(data.astype(np.float64) ** 2))) def _default_input_samplerate(sd) -> int: - """Preferred capture rate for the default input device; falls back to the - Whisper-friendly 16 kHz constant when the backend exposes no numeric rate.""" + """Default input device rate, else the Whisper-friendly SAMPLE_RATE.""" try: info = sd.query_devices(None, "input") rate = info.get("default_samplerate") if isinstance(info, dict) else getattr(info, "default_samplerate", None) @@ -136,15 +127,12 @@ def _default_input_samplerate(sd) -> int: return SAMPLE_RATE -from hermes_constants import is_termux as _is_termux_environment - - +# ── Environment detection ── def _voice_capture_install_hint() -> str: if _is_termux_environment(): return "pkg install python-numpy portaudio && python -m pip install sounddevice" - # Inside a venv (e.g. the bundled Hermes venv) a bare `pip install` may hit - # whichever Python the shell resolves first (on macOS often a Rosetta - # system Python) — point at the venv's own pip instead. + # Inside a venv a bare `pip install` may hit whichever Python the shell + # resolves first (macOS: often a Rosetta system Python) — use the venv's pip. try: if sys.prefix != getattr(sys, "base_prefix", sys.prefix): pip_in_venv = Path(sys.prefix) / "bin" / "pip" @@ -156,15 +144,11 @@ def _voice_capture_install_hint() -> str: def _portaudio_missing_message() -> str: - """Error text for "sounddevice imports but PortAudio's shared library is - missing" — a pip install can't fix that, so point at the system package.""" + """sounddevice imports but PortAudio's .so is missing — pip can't fix that.""" if _is_termux_environment(): hint = " Termux: pkg install portaudio" else: - hint = ( - " Linux: sudo apt-get install libportaudio2\n" - " macOS: brew install portaudio" - ) + hint = " Linux: sudo apt-get install libportaudio2\n macOS: brew install portaudio" return f"PortAudio system library not found -- install it first:\n{hint}\nThen retry /voice on." @@ -183,14 +167,12 @@ def _run_quiet(cmd: List[str], *, timeout: float, check: bool) -> subprocess.Com """subprocess.run with captured, utf-8-decoded output and no stdin.""" return subprocess.run( cmd, capture_output=True, text=True, encoding='utf-8', errors='replace', - timeout=timeout, check=check, stdin=subprocess.DEVNULL, - ) + timeout=timeout, check=check, stdin=subprocess.DEVNULL) -# Probes for the Termux:API Android app. `pm list packages` is the canonical -# lookup but on some ROMs `pm` isn't on Termux's PATH while `cmd package` is, -# and on others `pm` returns nothing for the calling user even when the app is -# present — so both are tried before concluding the app is missing. +# `pm list packages` is canonical, but on some ROMs `pm` isn't on Termux's PATH +# while `cmd package` is, and on others `pm` returns nothing for the calling +# user even when the app is present — so both are tried. _TERMUX_API_PACKAGE_PROBES = ( ("pm", "list", "packages", "com.termux.api"), ("cmd", "package", "list", "packages", "com.termux.api"), @@ -198,19 +180,16 @@ _TERMUX_API_PACKAGE_PROBES = ( def _termux_api_app_installed() -> bool: - """Return True iff the Termux:API Android app is installed. + """True iff the Termux:API Android app is installed. Any probe reporting ``package:com.termux.api`` is authoritative. If EVERY - probe is inconclusive (binary missing, permission denied, timeout, non-zero - exit) we cannot honestly say the app is missing, so trust the - ``termux-microphone-record`` binary on PATH instead: a false negative here - blocks ``/voice on`` outright, while a false positive only surfaces a - precise runtime error from the binary. If at least one probe ran cleanly - without mentioning the package, the app is genuinely missing. + probe is inconclusive (binary missing, denied, timeout, non-zero exit) we + trust the ``termux-microphone-record`` binary on PATH instead: a false + negative blocks ``/voice on`` outright, a false positive only surfaces a + precise runtime error. One clean probe without the package = genuinely missing. """ if not _is_termux_environment(): return False - inconclusive = False for cmd in _TERMUX_API_PACKAGE_PROBES: try: @@ -223,12 +202,10 @@ def _termux_api_app_installed() -> bool: continue if "package:com.termux.api" in (result.stdout or "").lower(): return True - if inconclusive and shutil.which("termux-microphone-record") is not None: logger.debug( "Termux package-manager probes inconclusive; trusting " - "termux-microphone-record binary on PATH (issue #31015)." - ) + "termux-microphone-record binary on PATH (issue #31015).") return True return False @@ -256,15 +233,13 @@ def _pulse_socket_candidates() -> List[str]: def _pulse_socket_reachable() -> bool: - """True if a PulseAudio/PipeWire socket is reachable on disk. + """True if a PulseAudio/PipeWire socket on disk accepts a connection. - Covers a sound server running locally (e.g. on a remote SSH host) without - ``PULSE_SERVER``/``PIPEWIRE_REMOTE`` set. A socket file must also accept a - connection — a stale socket left by a dead server does not count. + Covers a sound server running locally (e.g. a remote SSH host) without + ``PULSE_SERVER``/``PIPEWIRE_REMOTE`` set; a stale socket of a dead server does not count. """ import socket import stat - for path in _pulse_socket_candidates(): try: if not stat.S_ISSOCK(os.stat(path).st_mode): @@ -294,90 +269,76 @@ def _probe_audio_libraries( even though recording/playback works fine. """ termux_capture = bool(termux_mic_cmd and termux_app_installed) + + def outcome(termux_notice: str, warning: str, *, forwarded_notice: str = "", import_failed: bool = False): + if forwarded_notice and has_forwarded_audio: + notices.append(forwarded_notice) + elif termux_capture: + notices.append(termux_notice) + elif import_failed and termux_mic_cmd and not termux_app_installed: + warnings.append(_TERMUX_APP_MISSING_WARNING) + else: + warnings.append(warning) + try: sd, _ = _import_audio() except ImportError: - if termux_capture: - notices.append("Termux:API microphone recording available (sounddevice not required)") - elif termux_mic_cmd and not termux_app_installed: - warnings.append(_TERMUX_APP_MISSING_WARNING) - else: - warnings.append(f"Audio libraries not installed ({_voice_capture_install_hint()})") - return + return outcome("Termux:API microphone recording available (sounddevice not required)", + f"Audio libraries not installed ({_voice_capture_install_hint()})", import_failed=True) except OSError: - if termux_capture: - notices.append("Termux:API microphone recording available (PortAudio not required)") - elif termux_mic_cmd and not termux_app_installed: - warnings.append(_TERMUX_APP_MISSING_WARNING) - else: - warnings.append(_portaudio_missing_message()) - return - + return outcome("Termux:API microphone recording available (PortAudio not required)", + _portaudio_missing_message(), import_failed=True) try: if sd.query_devices(): return - if has_forwarded_audio: - notices.append("No PortAudio devices detected but host audio forwarding is configured -- continuing") - elif termux_capture: - notices.append("No PortAudio devices detected, but Termux:API microphone capture is available") - else: - warnings.append("No audio input/output devices detected") + outcome("No PortAudio devices detected, but Termux:API microphone capture is available", + "No audio input/output devices detected", + forwarded_notice="No PortAudio devices detected but host audio forwarding is configured -- continuing") except Exception: - if has_forwarded_audio: - notices.append("Audio device query failed but host audio forwarding is configured -- continuing") - elif termux_capture: - notices.append("PortAudio device query failed, but Termux:API microphone capture is available") - else: - warnings.append("Audio subsystem error (PortAudio cannot query devices)") + outcome("PortAudio device query failed, but Termux:API microphone capture is available", + "Audio subsystem error (PortAudio cannot query devices)", + forwarded_notice="Audio device query failed but host audio forwarding is configured -- continuing") def detect_audio_environment() -> dict: - """Detect if the current environment supports audio I/O. + """Detect whether the current environment supports audio I/O. - Returns dict with 'available' (bool), 'warnings' (hard-fail reasons that - block voice mode), and 'notices' (informational, do NOT block). SSH, - containers and WSL normally have no audio devices, but a reachable sound - server (PulseAudio/PipeWire socket or forwarding env vars) is honored. + Returns ``{'available', 'warnings' (hard-fail, block voice mode), + 'notices' (informational)}``. SSH, containers and WSL normally have no + audio devices, but a reachable sound server (PulseAudio/PipeWire socket or + forwarding env vars) is honored. """ warnings: List[str] = [] notices: List[str] = [] termux_mic_cmd = _termux_microphone_command() termux_app_installed = _termux_api_app_installed() has_forwarded_audio = bool( - os.environ.get('PULSE_SERVER') - or os.environ.get('PIPEWIRE_REMOTE') - or _pulse_socket_reachable() - ) + os.environ.get('PULSE_SERVER') or os.environ.get('PIPEWIRE_REMOTE') or _pulse_socket_reachable()) + + def report(notice: str, warning: str) -> None: + (notices if has_forwarded_audio else warnings).append(notice if has_forwarded_audio else warning) if any(os.environ.get(v) for v in ('SSH_CLIENT', 'SSH_TTY', 'SSH_CONNECTION')): - if has_forwarded_audio: - notices.append("Running over SSH with a reachable PulseAudio/PipeWire sound server") - else: - warnings.append( - "Running over SSH -- no audio devices available.\n" - " If a sound server (PulseAudio/PipeWire) is running on this host,\n" - " point Hermes at it, e.g.:\n" - " export XDG_RUNTIME_DIR=/run/user/$(id -u)\n" - " # or: export PULSE_SERVER=unix:$XDG_RUNTIME_DIR/pulse/native" - ) + report("Running over SSH with a reachable PulseAudio/PipeWire sound server", + "Running over SSH -- no audio devices available.\n" + " If a sound server (PulseAudio/PipeWire) is running on this host,\n" + " point Hermes at it, e.g.:\n" + " export XDG_RUNTIME_DIR=/run/user/$(id -u)\n" + " # or: export PULSE_SERVER=unix:$XDG_RUNTIME_DIR/pulse/native") from hermes_constants import is_container if is_container(): - if has_forwarded_audio: - notices.append("Running inside container (Docker/Podman/LXC) with host audio forwarding") - else: - warnings.append( - "Running inside container (Docker/Podman/LXC) -- no audio devices.\n" - " Forward host audio with one of (substitute $XDG_RUNTIME_DIR for your runtime dir,\n" - " typically /run/user/$UID):\n" - " PulseAudio: -v $XDG_RUNTIME_DIR/pulse/native:$XDG_RUNTIME_DIR/pulse/native \\\n" - " -e PULSE_SERVER=unix:$XDG_RUNTIME_DIR/pulse/native\n" - " PipeWire: -e PIPEWIRE_REMOTE=$XDG_RUNTIME_DIR/pipewire-0" - ) + report("Running inside container (Docker/Podman/LXC) with host audio forwarding", + "Running inside container (Docker/Podman/LXC) -- no audio devices.\n" + " Forward host audio with one of (substitute $XDG_RUNTIME_DIR for your runtime dir,\n" + " typically /run/user/$UID):\n" + " PulseAudio: -v $XDG_RUNTIME_DIR/pulse/native:$XDG_RUNTIME_DIR/pulse/native \\\n" + " -e PULSE_SERVER=unix:$XDG_RUNTIME_DIR/pulse/native\n" + " PipeWire: -e PIPEWIRE_REMOTE=$XDG_RUNTIME_DIR/pipewire-0") - # WSL: the PowerShell/Media.SoundPlayer fallback only covers OUTPUT, so - # when it is the only thing available we downgrade to a notice (recording - # guidance stays visible, but TTS-only usage isn't blocked). + # WSL: the PowerShell/Media.SoundPlayer fallback only covers OUTPUT, so when + # it is all that's available downgrade to a notice (recording guidance stays + # visible, TTS-only usage isn't blocked). if _is_wsl2_env(): if has_forwarded_audio: notices.append("Running in WSL with a reachable PulseAudio/PipeWire sound server") @@ -388,50 +349,26 @@ def detect_audio_environment() -> dict: "Voice INPUT (recording) still requires a PulseAudio bridge:\n" " 1. Set PULSE_SERVER=unix:/mnt/wslg/PulseServer\n" " 2. Create ~/.asoundrc pointing ALSA at PulseAudio\n" - " 3. Verify with: arecord -d 3 /tmp/test.wav && aplay /tmp/test.wav" - ) + " 3. Verify with: arecord -d 3 /tmp/test.wav && aplay /tmp/test.wav") else: warnings.append( "Running in WSL -- audio requires a forwarded sound server.\n" " PulseAudio: export PULSE_SERVER=unix:/mnt/wslg/PulseServer\n" " PipeWire: export PIPEWIRE_REMOTE=$XDG_RUNTIME_DIR/pipewire-0\n" - " Then verify: arecord -d 3 /tmp/test.wav && aplay /tmp/test.wav" - ) + " Then verify: arecord -d 3 /tmp/test.wav && aplay /tmp/test.wav") _probe_audio_libraries( - warnings, notices, - has_forwarded_audio=has_forwarded_audio, - termux_mic_cmd=termux_mic_cmd, - termux_app_installed=termux_app_installed, - ) - - return { - "available": not warnings, - "warnings": warnings, - "notices": notices, - } - -# ── Recording parameters ── -SAMPLE_RATE = 16000 # Whisper native rate -CHANNELS = 1 # Mono -DTYPE = "int16" # 16-bit PCM -SAMPLE_WIDTH = 2 # bytes per sample (int16) - -# Silence detection defaults -SILENCE_RMS_THRESHOLD = 200 # RMS below this = silence (int16 range 0-32767) -SILENCE_DURATION_SECONDS = 3.0 # Seconds of continuous silence before auto-stop - -# Temp directory for voice recordings -_TEMP_DIR = os.path.join(tempfile.gettempdir(), "hermes_voice") + warnings, notices, has_forwarded_audio=has_forwarded_audio, + termux_mic_cmd=termux_mic_cmd, termux_app_installed=termux_app_installed) + return {"available": not warnings, "warnings": warnings, "notices": notices} # ── Audio cues (beep tones) ── -_DEFAULT_BEEP_VOLUME = 0.3 # Backward-compatible default (matches prior hardcoded value) +_DEFAULT_BEEP_VOLUME = 0.3 def _get_beep_volume() -> float: - """``voice.beep_volume`` clamped to 0.0-1.0; 0.3 when missing/invalid so the - audio cue never breaks the voice loop on a degenerate config.""" + """``voice.beep_volume`` clamped to 0.0-1.0; 0.3 when missing/invalid.""" raw = _voice_config().get("beep_volume", _DEFAULT_BEEP_VOLUME) try: volume = float(raw) @@ -443,10 +380,9 @@ def _get_beep_volume() -> float: def _sd_play_blocking(sd, audio, sample_rate: int, *, timeout: float, blocksize: int = 0) -> None: - """``sd.play`` then poll until the stream goes idle or *timeout* passes. + """``sd.play`` then poll until idle or *timeout*. - ``sd.wait()`` calls ``Event.wait()`` without a timeout and hangs forever if - the audio device stalls, so poll with a ceiling and force-stop instead. + ``sd.wait()`` has no timeout and hangs forever if the device stalls. """ sd.play(audio, samplerate=sample_rate, blocksize=blocksize) deadline = time.monotonic() + timeout @@ -456,39 +392,34 @@ def _sd_play_blocking(sd, audio, sample_rate: int, *, timeout: float, blocksize: def play_beep(frequency: int = 880, duration: float = 0.12, count: int = 1) -> None: - """Play *count* short beeps of *frequency* Hz (default 880 = A5), *duration* s each. + """Play *count* short beeps of *frequency* Hz, *duration* s each. - The tone is synthesized with numpy only, so the macOS TCC prompt is not - triggered on the synthesis step; on macOS output goes through afplay. + Synthesized with numpy only (no sounddevice import on the synthesis step, + so no macOS TCC prompt); on macOS output goes through afplay. """ try: - np = _import_numpy() + import numpy as np except ImportError: return try: gap = 0.06 # seconds between beeps samples_per_beep = int(SAMPLE_RATE * duration) samples_per_gap = int(SAMPLE_RATE * gap) - beep_volume = _get_beep_volume() parts = [] for i in range(count): t = np.linspace(0, duration, samples_per_beep, endpoint=False) tone = np.sin(2 * np.pi * frequency * t) - # Fade in/out to avoid click artifacts. - fade_len = min(int(SAMPLE_RATE * 0.01), samples_per_beep // 4) + fade_len = min(int(SAMPLE_RATE * 0.01), samples_per_beep // 4) # avoid clicks tone[:fade_len] *= np.linspace(0, 1, fade_len) tone[-fade_len:] *= np.linspace(1, 0, fade_len) parts.append((tone * beep_volume * 32767).astype(np.int16)) if i < count - 1: parts.append(np.zeros(samples_per_gap, dtype=np.int16)) - audio = np.concatenate(parts) - if not _sounddevice_output_allowed(): _play_int16_via_tempfile(audio, SAMPLE_RATE) return - try: sd, _ = _import_audio() except (ImportError, OSError): @@ -499,15 +430,10 @@ def play_beep(frequency: int = 880, duration: float = 0.12, count: int = 1) -> N # ── Thinking sound — calm ambient "blub blub" while the agent works ── -# The agent can think / run tools for minutes with zero audio, which reads as -# "it died"; a quiet repeating pair of soft water-bubble blips fills the gap. -# Synthesized with numpy, scaled by voice.beep_volume, gated by -# voice.thinking_sound (default on). macOS: sounddevice OUTPUT is TCC-gated -# and spawning afplay every second would churn subprocesses, so it is skipped. -# -# The host's *should_play* callback decides when blips are allowed; the -# module-level output ref-count below tracks when real audio (TTS sentences, -# file playback) is actually flowing so hosts have an accurate signal. +# Minutes of silent tool use reads as "it died"; a quiet pair of water-bubble +# blips fills the gap. Scaled by voice.beep_volume, gated by voice.thinking_sound +# (default on). The host's *should_play* callback decides when blips are allowed; +# the output ref-count below tells hosts when real audio is actually flowing. _audio_output_active_count = 0 _audio_output_lock = threading.Lock() @@ -517,15 +443,12 @@ def mark_audio_output_active(active: bool) -> None: """Reference-count real audio output (TTS/file playback). Playback paths bracket their work with ``(True)`` / ``(False)`` so - ``is_audio_output_active()`` reflects whether speech audio is leaving the - speakers RIGHT NOW — unlike per-turn TTS-done events, which stay 'busy' - for a whole turn even while the pipeline silently waits for text. + ``is_audio_output_active()`` reflects speech leaving the speakers RIGHT NOW — + unlike per-turn TTS-done events, which stay 'busy' while waiting for text. """ global _audio_output_active_count with _audio_output_lock: - _audio_output_active_count = max( - 0, _audio_output_active_count + (1 if active else -1) - ) + _audio_output_active_count = max(0, _audio_output_active_count + (1 if active else -1)) def is_audio_output_active() -> bool: @@ -548,18 +471,13 @@ def thinking_sound_enabled() -> bool: def _synth_thinking_blip(np, frequency: float) -> "Any": - """One soft 'blub': short sine with a gentle downward pitch glide and a - smooth attack/decay envelope (no clicks), low-volume.""" + """One soft 'blub': short sine with a downward glide and a click-free envelope.""" duration = 0.16 n = int(SAMPLE_RATE * duration) t = np.linspace(0, duration, n, endpoint=False) - # Downward glide (water-drop feel): freq → 0.72*freq over the blip. - glide = np.linspace(1.0, 0.72, n) + glide = np.linspace(1.0, 0.72, n) # water-drop feel: freq → 0.72*freq phase = 2 * np.pi * np.cumsum(frequency * glide) / SAMPLE_RATE - tone = np.sin(phase) - # Soften harmonics (cheap low-pass feel): add a quieter octave-down sine. - tone = 0.8 * tone + 0.2 * np.sin(phase / 2.0) - # Envelope: quick-but-smooth attack, long exponential-ish decay. + tone = 0.8 * np.sin(phase) + 0.2 * np.sin(phase / 2.0) # octave-down softens harmonics attack = int(0.02 * SAMPLE_RATE) env = np.ones(n) env[:attack] = np.linspace(0.0, 1.0, attack) @@ -569,12 +487,11 @@ def _synth_thinking_blip(np, frequency: float) -> "Any": def _thinking_sound_loop(stop: threading.Event, should_play) -> None: - """Daemon loop: play alternating-pitch blips every ~0.8-1.2s until *stop*. + """Daemon loop: alternating-pitch blips every ~0.8-1.2s until *stop*. - Skips a blip (without stopping) whenever *should_play* returns False — - e.g. TTS audio started flowing or the mic re-armed. macOS: sounddevice - output is TCC-gated, and per-second afplay subprocess churn is worse - than silence, so the loop exits immediately there. + Skips (without stopping) whenever *should_play* returns False. macOS: + sounddevice output is TCC-gated and per-second afplay churn is worse than + silence, so the loop exits immediately there. """ if not _sounddevice_output_allowed(): return @@ -582,11 +499,8 @@ def _thinking_sound_loop(stop: threading.Event, should_play) -> None: sd, np = _import_audio() except (ImportError, OSError): return - import random - - pitches = (392.0, 329.6) # G4 / E4 — calm, low, alternating - blips = [_synth_thinking_blip(np, p) for p in pitches] + blips = [_synth_thinking_blip(np, p) for p in (392.0, 329.6)] # G4 / E4 i = 0 while not stop.is_set(): try: @@ -605,24 +519,19 @@ def _thinking_sound_loop(stop: threading.Event, should_play) -> None: def start_thinking_sound(should_play=None) -> bool: """Start the ambient thinking sound (idempotent). - *should_play* is polled before each blip; return False to skip while - speech audio flows or the mic is capturing. Returns True when the loop - was started (or already running), False when disabled/unavailable. + *should_play* is polled before each blip. Returns True when running + (or already running), False when disabled/unavailable. """ global _thinking_stop if not thinking_sound_enabled(): return False with _thinking_lock: if _thinking_stop is not None and not _thinking_stop.is_set(): - return True # already running + return True stop = threading.Event() _thinking_stop = stop - threading.Thread( - target=_thinking_sound_loop, - args=(stop, should_play), - daemon=True, - name="voice-thinking-sound", - ).start() + threading.Thread(target=_thinking_sound_loop, args=(stop, should_play), + daemon=True, name="voice-thinking-sound").start() return True @@ -635,25 +544,21 @@ def stop_thinking_sound() -> None: stop.set() -# ── Termux Audio Recorder ── +# ── Recorders ── def _new_recording_path(ext: str) -> str: """Timestamped ``recording_*.`` path under _TEMP_DIR (created on demand).""" os.makedirs(_TEMP_DIR, exist_ok=True) - timestamp = time.strftime("%Y%m%d_%H%M%S") - return os.path.join(_TEMP_DIR, f"recording_{timestamp}.{ext}") + return os.path.join(_TEMP_DIR, f"recording_{time.strftime('%Y%m%d_%H%M%S')}.{ext}") -class TermuxAudioRecorder: - """Recorder backend that uses Termux:API microphone capture commands.""" - - supports_silence_autostop = False +class _RecorderBase: + """Lock, recording flag, start time and live RMS shared by both recorder backends.""" def __init__(self) -> None: self._lock = threading.Lock() self._recording = False - self._start_time = 0.0 - self._recording_path: Optional[str] = None - self._current_rms = 0 + self._start_time: float = 0.0 + self._current_rms: int = 0 @property def is_recording(self) -> bool: @@ -667,8 +572,19 @@ class TermuxAudioRecorder: @property def current_rms(self) -> int: + """Current input RMS level (0-32767), updated each audio chunk.""" return self._current_rms + +class TermuxAudioRecorder(_RecorderBase): + """Recorder backend that uses Termux:API microphone capture commands.""" + + supports_silence_autostop = False + + def __init__(self) -> None: + super().__init__() + self._recording_path: Optional[str] = None + def start(self, on_silence_stop=None) -> None: del on_silence_stop # Termux:API does not expose live silence callbacks. mic_cmd = _termux_microphone_command() @@ -676,27 +592,17 @@ class TermuxAudioRecorder: raise RuntimeError( "Termux voice capture requires the termux-api package and app.\n" "Install with: pkg install termux-api\n" - "Then install/update the Termux:API Android app." - ) + "Then install/update the Termux:API Android app.") if not _termux_api_app_installed(): raise RuntimeError( "Termux voice capture requires the Termux:API Android app.\n" - "Install/update the Termux:API app, then retry /voice on." - ) - + "Install/update the Termux:API app, then retry /voice on.") with self._lock: if self._recording: return self._recording_path = _new_recording_path("aac") - - command = [ - mic_cmd, - "-f", self._recording_path, - "-l", "0", - "-e", "aac", - "-r", str(SAMPLE_RATE), - "-c", str(CHANNELS), - ] + command = [mic_cmd, "-f", self._recording_path, "-l", "0", "-e", "aac", + "-r", str(SAMPLE_RATE), "-c", str(CHANNELS)] try: _run_quiet(command, timeout=15, check=True) except subprocess.CalledProcessError as e: @@ -704,7 +610,6 @@ class TermuxAudioRecorder: raise RuntimeError(f"Termux microphone start failed: {details}") from e except Exception as e: raise RuntimeError(f"Termux microphone start failed: {e}") from e - with self._lock: self._start_time = time.monotonic() self._recording = True @@ -713,9 +618,8 @@ class TermuxAudioRecorder: def _stop_termux_recording(self) -> None: mic_cmd = _termux_microphone_command() - if not mic_cmd: - return - _run_quiet([mic_cmd, "-q"], timeout=15, check=False) + if mic_cmd: + _run_quiet([mic_cmd, "-q"], timeout=15, check=False) def _reset_state(self) -> tuple: """Clear recording state under the lock; return (was_recording, path, started_at).""" @@ -730,12 +634,10 @@ class TermuxAudioRecorder: was_recording, path, started_at = self._reset_state() if not was_recording: return None - self._stop_termux_recording() if not path or not os.path.isfile(path): return None - # Discard sub-0.3s taps and empty files. - if time.monotonic() - started_at < 0.3 or os.path.getsize(path) <= 0: + if time.monotonic() - started_at < 0.3 or os.path.getsize(path) <= 0: # sub-0.3s taps / empty _unlink_quietly(path) return None logger.info("Termux voice recording stopped: %s", path) @@ -755,91 +657,55 @@ class TermuxAudioRecorder: self.cancel() -# ── AudioRecorder ── -class AudioRecorder: +class AudioRecorder(_RecorderBase): """Thread-safe audio recorder using sounddevice.InputStream. - Usage:: - - recorder = AudioRecorder() - recorder.start(on_silence_stop=my_callback) - # ... user speaks ... - wav_path = recorder.stop() # returns path to WAV file - # or - recorder.cancel() # discard without saving - - If ``on_silence_stop`` is provided, recording automatically stops when - the user is silent for ``silence_duration`` seconds and calls the callback. + ``start(on_silence_stop=cb)`` ... ``stop()`` returns a WAV path (or None); + ``cancel()`` discards. With ``on_silence_stop`` the recording auto-stops + after ``silence_duration`` seconds of silence following speech. """ supports_silence_autostop = True def __init__(self) -> None: - self._lock = threading.Lock() + super().__init__() self._stream: Any = None self._frames: List[Any] = [] - self._recording = False - self._start_time: float = 0.0 self._sample_rate: int = SAMPLE_RATE self._on_silence_stop = None self._silence_threshold: int = SILENCE_RMS_THRESHOLD self._silence_duration: float = SILENCE_DURATION_SECONDS - self._min_speech_duration: float = 0.3 # Seconds of speech needed to confirm - self._max_dip_tolerance: float = 0.3 # Max dip duration before resetting speech - self._max_wait: float = 15.0 # Max seconds to wait for speech before auto-stop - # Hard cap on total recording length, wired from voice.max_recording_seconds - # by the CLI before each recording. 0 (or unset) = no cap. + self._min_speech_duration: float = 0.3 # seconds of speech needed to confirm + self._max_dip_tolerance: float = 0.3 # max dip before resetting speech + self._max_wait: float = 15.0 # max seconds to wait for speech before auto-stop + # Hard cap on total length, wired from voice.max_recording_seconds by the + # CLI before each recording. 0 (or unset) = no cap. self._max_recording_seconds: float = 0.0 self._peak_rms: int = 0 # for the speech-presence check in stop() - self._current_rms: int = 0 # live level, read by the UI self._reset_detection_state() def _reset_detection_state(self) -> None: - """Reset per-recording silence-detection trackers.""" self._has_spoken = False - self._speech_start: float = 0.0 # When speech attempt began - self._dip_start: float = 0.0 # When current below-threshold dip began + self._speech_start: float = 0.0 + self._dip_start: float = 0.0 self._silence_start: float = 0.0 - self._resume_start: float = 0.0 # Tracks sustained speech after silence starts - self._resume_dip_start: float = 0.0 # Dip tolerance tracker for resume detection + self._resume_start: float = 0.0 # sustained speech after silence started + self._resume_dip_start: float = 0.0 def _max_duration_reached(self, elapsed: float) -> bool: - """Whether the configured hard recording-length cap has elapsed. - - ``voice.max_recording_seconds`` is applied by the CLI before each - recording (see ``HermesCLI._voice_start_recording``). A value <= 0 - (or unset) disables the cap. - """ + """``voice.max_recording_seconds`` cap elapsed (<= 0 / unset disables it).""" cap = self._max_recording_seconds return bool(cap and cap > 0 and elapsed >= cap) - # -- public properties --------------------------------------------------- - - @property - def elapsed_seconds(self) -> float: - if not self._recording: - return 0.0 - return time.monotonic() - self._start_time - - @property - def current_rms(self) -> int: - """Current audio input RMS level (0-32767). Updated each audio chunk.""" - return self._current_rms - - @property - def is_recording(self) -> bool: - """Whether audio recording is currently active.""" - return self._recording - # -- silence detection --------------------------------------------------- def _track_speech(self, rms: int, now: float) -> None: """Advance the speech/dip trackers for one audio block. Speech is confirmed after ``_min_speech_duration`` above threshold, - tolerating dips shorter than ``_max_dip_tolerance`` (micro-pauses - between syllables). After confirmation only SUSTAINED resumed speech - resets the silence timer — brief ambient spikes must not. + tolerating dips shorter than ``_max_dip_tolerance`` (micro-pauses). + After confirmation only SUSTAINED resumed speech resets the silence + timer — brief ambient spikes must not. """ if rms > self._silence_threshold: self._dip_start = 0.0 @@ -847,13 +713,11 @@ class AudioRecorder: self._speech_start = now elif not self._has_spoken and now - self._speech_start >= self._min_speech_duration: self._has_spoken = True - logger.debug("Speech confirmed (%.2fs above threshold)", - now - self._speech_start) + logger.debug("Speech confirmed (%.2fs above threshold)", now - self._speech_start) if not self._has_spoken: self._silence_start = 0.0 else: - # Resumed speech mirrors initial detection: track, tolerate - # short dips, confirm after _min_speech_duration. + # Resumed speech mirrors initial detection: track, tolerate dips, confirm. self._resume_dip_start = 0.0 if self._resume_start == 0.0: self._resume_start = now @@ -861,8 +725,7 @@ class AudioRecorder: self._silence_start = 0.0 self._resume_start = 0.0 elif self._has_spoken: - # Below threshold after confirmed speech: dip-tolerant resume reset. - if self._resume_start > 0: + if self._resume_start > 0: # dip-tolerant resume reset if self._resume_dip_start == 0.0: self._resume_dip_start = now elif now - self._resume_dip_start >= self._max_dip_tolerance: @@ -873,16 +736,13 @@ class AudioRecorder: if self._dip_start == 0.0: self._dip_start = now elif now - self._dip_start >= self._max_dip_tolerance: - logger.debug("Speech attempt reset (dip lasted %.2fs)", - now - self._dip_start) + logger.debug("Speech attempt reset (dip lasted %.2fs)", now - self._dip_start) self._speech_start = 0.0 self._dip_start = 0.0 def _should_auto_stop(self, rms: int, now: float) -> bool: - """Auto-stop when: the user spoke then stayed silent for - ``_silence_duration``; no speech at all for ``_max_wait``; or the - hard ``voice.max_recording_seconds`` cap elapsed (independent of - speech, so a continuous speaker still stops).""" + """Spoke then silent for ``_silence_duration``; no speech for ``_max_wait``; + or the hard cap elapsed (independent of speech).""" elapsed = now - self._start_time if self._has_spoken and rms <= self._silence_threshold: if self._silence_start == 0.0: @@ -894,8 +754,7 @@ class AudioRecorder: logger.info("No speech within %.0fs, auto-stopping", self._max_wait) return True if self._max_duration_reached(elapsed): - logger.info("Max recording length reached (%.0fs), auto-stopping", - self._max_recording_seconds) + logger.info("Max recording length reached (%.0fs), auto-stopping", self._max_recording_seconds) return True return False @@ -915,7 +774,6 @@ class AudioRecorder: threading.Thread(target=_safe_cb, daemon=True).start() def _on_audio_block(self, np, indata) -> None: - """Per-block work for the InputStream callback while recording.""" self._frames.append(indata.copy()) rms = int(_rms(np, indata)) self._current_rms = rms @@ -930,16 +788,13 @@ class AudioRecorder: # -- public methods ------------------------------------------------------ def _ensure_stream(self) -> None: - """Create the audio InputStream once and keep it alive. + """Create the InputStream once and keep it alive. - The stream stays open for the lifetime of the recorder; between - recordings the callback simply discards chunks (``_recording`` is - False). This avoids the CoreAudio bug where closing and re-opening an - ``InputStream`` hangs indefinitely on macOS. + Between recordings the callback simply discards chunks; this avoids the + CoreAudio bug where closing and re-opening an InputStream hangs on macOS. """ if self._stream is not None: - return # already alive - + return sd, np = _import_audio() def _callback(indata, frames, time_info, status): # noqa: ARG001 @@ -948,15 +803,9 @@ class AudioRecorder: if self._recording: self._on_audio_block(np, indata) - # Create stream — may block on CoreAudio (first call only). stream = None - try: - stream = sd.InputStream( - samplerate=self._sample_rate, - channels=CHANNELS, - dtype=DTYPE, - callback=_callback, - ) + try: # may block on CoreAudio (first call only) + stream = sd.InputStream(samplerate=self._sample_rate, channels=CHANNELS, dtype=DTYPE, callback=_callback) stream.start() except Exception as e: if stream is not None: @@ -966,18 +815,14 @@ class AudioRecorder: pass raise RuntimeError( f"Failed to open audio input stream: {e}. " - "Check that a microphone is connected and accessible." - ) from e + "Check that a microphone is connected and accessible.") from e self._stream = stream def start(self, on_silence_stop=None) -> None: - """Start capturing audio from the default input device. + """Start capturing from the default input device. - The InputStream is created once and kept alive across recordings; - later calls reset detection state and toggle frame collection. - *on_silence_stop* is invoked (in a daemon thread, no arguments) when - silence follows speech — use it to auto-stop and transcribe. - Raises ``RuntimeError`` if sounddevice/numpy are not installed. + *on_silence_stop* is invoked (daemon thread, no args) when silence + follows speech. Raises ``RuntimeError`` if sounddevice/numpy are missing. """ try: sd, _ = _import_audio() @@ -986,13 +831,10 @@ class AudioRecorder: except ImportError as e: raise RuntimeError( "Voice mode requires sounddevice and numpy.\n" - f"Install with: {sys.executable} -m pip install sounddevice numpy" - ) from e - + f"Install with: {sys.executable} -m pip install sounddevice numpy") from e with self._lock: if self._recording: - return # already recording - + return self._frames = [] self._start_time = time.monotonic() self._reset_detection_state() @@ -1001,18 +843,15 @@ class AudioRecorder: self._on_silence_stop = on_silence_stop self._sample_rate = _default_input_samplerate(sd) self._ensure_stream() - with self._lock: self._recording = True logger.info("Voice recording started (rate=%d, channels=%d)", self._sample_rate, CHANNELS) def _close_stream_with_timeout(self, timeout: float = 3.0) -> None: - """Close the audio stream with a timeout to prevent CoreAudio hangs.""" + """Close the stream with a timeout to prevent CoreAudio hangs.""" if self._stream is None: return - - stream = self._stream - self._stream = None + stream, self._stream = self._stream, None def _do_close(): try: @@ -1023,49 +862,35 @@ class AudioRecorder: t = threading.Thread(target=_do_close, daemon=True) t.start() - # Poll in short intervals so Ctrl+C is not blocked - deadline = __import__("time").monotonic() + timeout - while t.is_alive() and __import__("time").monotonic() < deadline: + clock = __import__("time") # real clock even when tests patch this module's ``time`` + deadline = clock.monotonic() + timeout + while t.is_alive() and clock.monotonic() < deadline: # short joins keep Ctrl+C responsive t.join(timeout=0.1) if t.is_alive(): logger.warning("Audio stream close timed out after %.1fs — forcing ahead", timeout) def stop(self) -> Optional[str]: - """Stop recording and write captured audio to a WAV file. - - The stream stays alive for reuse — only frame collection stops. - Returns the WAV path, or ``None`` if no usable audio was captured. - """ + """Stop recording (stream stays alive) and return the WAV path, or None if unusable.""" with self._lock: if not self._recording: return None - self._recording = False self._current_rms = 0 - if not self._frames: return None - _, np = _import_audio() audio_data = np.concatenate(self._frames, axis=0) self._frames = [] - elapsed = time.monotonic() - self._start_time logger.info("Voice recording stopped (%.1fs, %d samples)", elapsed, len(audio_data)) - - # Skip very short recordings (< 0.3s of audio) - min_samples = int(self._sample_rate * 0.3) - if len(audio_data) < min_samples: + if len(audio_data) < int(self._sample_rate * 0.3): logger.debug("Recording too short (%d samples), discarding", len(audio_data)) return None - - # Skip silent recordings using peak RMS (not overall average, which - # gets diluted by silence at the end of the recording). + # Peak RMS, not the average (which trailing silence dilutes). if self._peak_rms < SILENCE_RMS_THRESHOLD: logger.info("Recording too quiet (peak RMS=%d < %d), discarding", self._peak_rms, SILENCE_RMS_THRESHOLD) return None - return self._write_wav(audio_data, sample_rate=self._sample_rate) def _discard(self) -> None: @@ -1081,17 +906,14 @@ class AudioRecorder: logger.info("Voice recording cancelled") def shutdown(self) -> None: - """Release the audio stream. Call when voice mode is disabled.""" + """Release the audio stream. Call when voice mode is disabled.""" self._discard() - # Close stream OUTSIDE the lock to avoid deadlock with audio callback - self._close_stream_with_timeout() + self._close_stream_with_timeout() # outside the lock: avoids deadlock with the callback logger.info("AudioRecorder shut down") - # -- private helpers ----------------------------------------------------- - @staticmethod def _write_wav(audio_data, *, sample_rate: int = SAMPLE_RATE) -> str: - """Write numpy int16 audio data to a WAV file; returns the path.""" + """Write numpy int16 audio to a WAV file; returns the path.""" wav_path = _new_recording_path("wav") _write_wav_frames(wav_path, audio_data.tobytes(), sample_rate) logger.info("WAV written: %s (%d bytes)", wav_path, os.path.getsize(wav_path)) @@ -1107,76 +929,53 @@ def create_audio_recorder() -> AudioRecorder | TermuxAudioRecorder: # ── STT dispatch ── def transcribe_recording(wav_path: str, model: Optional[str] = None) -> Dict[str, Any]: - """Transcribe a WAV via ``tools.transcription_tools.transcribe_audio()``, - filtering Whisper hallucinations on silent audio. + """Transcribe a WAV via ``transcribe_audio()``, filtering Whisper hallucinations. Returns dict with ``success``, ``transcript``, and optionally ``error``. """ from tools.transcription_tools import MAX_FILE_SIZE, transcribe_audio result = transcribe_audio(wav_path, model=model, source="voice_mode") - # Only chunk when the provider itself reports "File too large" — local # providers have no upload cap and never return this error. if not result.get("success") and "File too large" in result.get("error", ""): result = _transcribe_wav_in_chunks(wav_path, model=model, max_file_size=MAX_FILE_SIZE) - - # A configured voice-chat stop phrase is checked FIRST and always survives: - # phrases like "bye" or "okay" overlap the hallucination blocklist/repeat - # regex, and swallowing them would make saying "bye" fail to end the chat. + # A configured stop phrase always survives: "bye"/"okay" overlap the + # hallucination blocklist, and swallowing them would make "bye" fail to end the chat. if result.get("success"): raw_transcript = result.get("transcript", "") if is_whisper_hallucination(raw_transcript) and not is_voice_stop_phrase(raw_transcript): logger.info("Filtered Whisper hallucination: %r", result["transcript"]) return {"success": True, "transcript": "", "filtered": True} - - # Providers that flag no_speech failed to hear words, not to transcribe — - # treat like silence so the voice loop re-listens quietly instead of + # no_speech = heard no words, not a failure: re-listen quietly instead of # surfacing "Transcription failed". if result.get("no_speech"): return {"success": True, "transcript": "", "no_speech": True} - return result -def _transcribe_wav_in_chunks( - wav_path: str, - *, - model: Optional[str], - max_file_size: int, -) -> Dict[str, Any]: +def _transcribe_wav_in_chunks(wav_path: str, *, model: Optional[str], max_file_size: int) -> Dict[str, Any]: """Split an oversized WAV into provider-sized chunks and join transcripts.""" from tools.transcription_tools import transcribe_audio chunk_paths: List[str] = [] transcripts: List[str] = [] - try: chunk_paths = _split_wav_for_transcription(wav_path, max_file_size=max_file_size) if not chunk_paths: return {"success": False, "transcript": "", "error": "No audio chunks were created"} - logger.info("Transcribing oversized WAV in %d chunks: %s", len(chunk_paths), wav_path) for index, chunk_path in enumerate(chunk_paths, start=1): result = transcribe_audio(chunk_path, model=model, source="voice_mode") if not result.get("success"): error = result.get("error", "Unknown transcription error") - return { - "success": False, - "transcript": "", - "error": f"Chunk {index}/{len(chunk_paths)} failed: {error}", - } - + return {"success": False, "transcript": "", + "error": f"Chunk {index}/{len(chunk_paths)} failed: {error}"} transcript = result.get("transcript", "").strip() if transcript and not is_whisper_hallucination(transcript): transcripts.append(transcript) - - return { - "success": True, - "transcript": " ".join(transcripts).strip(), - "provider": result.get("provider"), - "chunks": len(chunk_paths), - } + return {"success": True, "transcript": " ".join(transcripts).strip(), + "provider": result.get("provider"), "chunks": len(chunk_paths)} except Exception as e: logger.error("Chunked transcription failed for %s: %s", wav_path, e, exc_info=True) return {"success": False, "transcript": "", "error": f"Chunked transcription failed: {e}"} @@ -1190,31 +989,24 @@ def _split_wav_for_transcription(wav_path: str, *, max_file_size: int) -> List[s os.makedirs(_TEMP_DIR, exist_ok=True) chunk_paths: List[str] = [] header_reserve = 64 * 1024 - with wave.open(wav_path, "rb") as source: params = source.getparams() block_align = max(1, params.nchannels * params.sampwidth) max_data_bytes = max_file_size - header_reserve if max_data_bytes < block_align: raise ValueError("STT max_file_size is too small for WAV chunking") - frames_per_chunk = max(1, max_data_bytes // block_align) index = 0 while True: frames = source.readframes(frames_per_chunk) if not frames: break - index += 1 temp = tempfile.NamedTemporaryFile( prefix=f"{os.path.splitext(os.path.basename(wav_path))[0]}_chunk{index:03d}_", - suffix=".wav", - dir=_TEMP_DIR, - delete=False, - ) + suffix=".wav", dir=_TEMP_DIR, delete=False) chunk_path = temp.name temp.close() - try: with wave.open(chunk_path, "wb") as chunk: chunk.setparams(params._replace(nframes=0)) @@ -1223,17 +1015,20 @@ def _split_wav_for_transcription(wav_path: str, *, max_file_size: int) -> List[s except Exception: _unlink_quietly(chunk_path) raise - return chunk_paths # ── Audio playback (interruptable) ── - -# Global reference to the active playback process so it can be interrupted. -_active_playback: Optional[subprocess.Popen] = None +_active_playback: Optional[subprocess.Popen] = None # so stop_playback can interrupt it _playback_lock = threading.Lock() +def _set_active_playback(proc) -> None: + global _active_playback + with _playback_lock: + _active_playback = proc + + def stop_playback() -> None: """Interrupt the currently playing audio (if any).""" global _active_playback @@ -1246,8 +1041,7 @@ def stop_playback() -> None: logger.info("Audio playback interrupted") except Exception: pass - # Also stop sounddevice playback if active - try: + try: # also stop sounddevice playback if active sd, _ = _import_audio() sd.stop() except Exception: @@ -1255,10 +1049,9 @@ def stop_playback() -> None: def _is_wsl2_env() -> bool: - """True when running inside WSL (Microsoft kernel signature in /proc/version). + """True inside WSL (Microsoft kernel signature in /proc/version); False on any error. - Returns False on any error (non-WSL Linux, Docker, SSH, etc.). A - module-level function so tests can patch it instead of ``builtins.open``. + Module-level so tests can patch it instead of ``builtins.open``. """ try: with open("/proc/version", encoding="utf-8", errors="replace") as _fv: @@ -1267,34 +1060,23 @@ def _is_wsl2_env() -> bool: return False -_is_wsl = _is_wsl2_env - - def _wsl_powershell_tts_available() -> bool: """True when the WSL2 PowerShell TTS playback fallback can be used. - Only covers OUTPUT (Media.SoundPlayer on the Windows host) — it does NOT - make microphone recording work, so callers relaxing the audio-environment - gate must still surface the PulseAudio-bridge guidance for recording. + OUTPUT only (Media.SoundPlayer on the Windows host) — microphone recording + still needs a PulseAudio bridge, so callers must keep surfacing that guidance. """ - return bool( - _is_wsl2_env() - and shutil.which("powershell.exe") - and shutil.which("ffmpeg") - ) + return bool(_is_wsl2_env() and shutil.which("powershell.exe") and shutil.which("ffmpeg")) def play_audio_file(file_path: str) -> bool: - """Play an audio file through the default output device. + """Play an audio file; returns True on success. - WAV files go through ``sounddevice.play()`` when allowed; otherwise system + WAV goes through ``sounddevice.play()`` when allowed; otherwise system players: ``afplay`` (macOS), the WSL2 PowerShell bridge, ``ffplay``, - ``aplay`` (Linux). Interruptible via ``stop_playback()``. Returns True on - success. + ``aplay`` (Linux). Interruptible via ``stop_playback()``. """ - # Ref-count real speaker output for the whole call so the thinking-sound - # loop (and any other ambient cue) knows audio is flowing right now. - mark_audio_output_active(True) + mark_audio_output_active(True) # ref-count real speaker output for the whole call try: return _play_audio_file_impl(file_path) finally: @@ -1302,19 +1084,17 @@ def play_audio_file(file_path: str) -> bool: def _play_wav_via_sounddevice(file_path: str) -> bool: - """Play a WAV through sounddevice; False if the audio libs are unavailable - or playback failed (caller falls through to system players).""" + """Play a WAV through sounddevice; False when unavailable/failed (caller falls through).""" try: sd, np = _import_audio() with wave.open(file_path, "rb") as wf: frames = wf.readframes(wf.getnframes()) audio_data = np.frombuffer(frames, dtype=np.int16) sample_rate = wf.getframerate() - - # WSLg RDP audio needs a warmup to avoid crackling at the start: the - # RDP virtual channel takes ~100 ms to stabilise, and the small default - # blocksize exasperates clock-adjustment jitter (microsoft/wslg#1257). - if _is_wsl(): + # WSLg RDP audio needs a warmup to avoid crackling: the RDP channel takes + # ~100 ms to stabilise and the small default blocksize worsens + # clock-adjustment jitter (microsoft/wslg#1257). + if _is_wsl2_env(): silence_samples = int(0.1 * sample_rate) fade_samples = int(0.1 * sample_rate) fade = np.linspace(0.0, 1.0, fade_samples, dtype=np.float64) @@ -1322,21 +1102,15 @@ def _play_wav_via_sounddevice(file_path: str) -> bool: audio_float[:fade_samples] *= fade tail = np.zeros(int(0.05 * sample_rate), dtype=np.int16) audio_data = np.concatenate([ - np.zeros(silence_samples, dtype=np.int16), - audio_float.astype(np.int16), - tail, - ]) + np.zeros(silence_samples, dtype=np.int16), audio_float.astype(np.int16), tail]) blocksize = 4096 else: blocksize = 0 # default (auto) - - _sd_play_blocking( - sd, audio_data, sample_rate, - timeout=len(audio_data) / sample_rate + 2.0, blocksize=blocksize, - ) + _sd_play_blocking(sd, audio_data, sample_rate, + timeout=len(audio_data) / sample_rate + 2.0, blocksize=blocksize) return True except (ImportError, OSError): - return False # audio libs not available, fall through to system players + return False except Exception as e: logger.debug("sounddevice playback failed: %s", e) return False @@ -1345,12 +1119,11 @@ def _play_wav_via_sounddevice(file_path: str) -> bool: def _wsl_powershell_player_cmd(file_path: str) -> Optional[List[str]]: """Build the WSL2 PowerShell fallback player command, or None. - In WSL without a PulseAudio bridge ffplay/aplay have no device, but - Media.SoundPlayer on the Windows host always does: convert to a - uniquely-named WAV in Windows %TEMP% (so concurrent TTS calls don't - collide) and play it. The WAV is deleted unconditionally, and the ORIGINAL - ffmpeg/powershell exit status is re-raised past that cleanup (rm -f always - exits 0) so the player loop can fall through to the next player. + Without a PulseAudio bridge ffplay/aplay have no device, but Media.SoundPlayer + on the Windows host does: convert to a uniquely-named WAV in Windows %TEMP% + (concurrent TTS calls must not collide) and play it. The WAV is deleted + unconditionally and the ORIGINAL exit status re-raised past that cleanup + (rm -f always exits 0) so the player loop can fall through. """ if not (shutil.which("powershell.exe") and shutil.which("ffmpeg") and _is_wsl2_env()): return None @@ -1372,8 +1145,7 @@ def _wsl_powershell_player_cmd(file_path: str) -> Optional[List[str]]: ps_script = f"(New-Object Media.SoundPlayer '{win_wav_safe}').PlaySync()" ps_cmd = " && ".join([ shlex.join(["ffmpeg", "-i", file_path, "-f", "wav", wsl_wav, "-loglevel", "quiet", "-y"]), - shlex.join(["powershell.exe", "-NoProfile", "-Command", ps_script]), - ]) + shlex.join(["powershell.exe", "-NoProfile", "-Command", ps_script])]) cleanup = shlex.join(["rm", "-f", wsl_wav]) # Full path so the which(cmd[0]) check in the player loop passes. return ["/bin/sh", "-c", f"( {ps_cmd} ); rc=$?; {cleanup}; exit $rc"] @@ -1397,44 +1169,29 @@ def _system_player_candidates(file_path: str) -> List[List[str]]: return players -def _set_active_playback(proc) -> None: - global _active_playback - with _playback_lock: - _active_playback = proc - - def _run_system_player(cmd: List[str]) -> bool: """Run one player to completion (interruptible via stop_playback).""" proc = None try: - # Sibling of the TTS/STT credential scrub: system audio players must - # not inherit gateway tokens / API keys. + # Sibling of the TTS/STT credential scrub: players must not inherit tokens/keys. from tools.environments.local import hermes_subprocess_env - - proc = subprocess.Popen( - cmd, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - stdin=subprocess.DEVNULL, - env=hermes_subprocess_env(inherit_credentials=False), - ) + proc = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + stdin=subprocess.DEVNULL, env=hermes_subprocess_env(inherit_credentials=False)) _set_active_playback(proc) proc.wait(timeout=300) rc = proc.returncode - _set_active_playback(None) if rc == 0: return True - # Non-zero exit: e.g. WSL ffplay/aplay with no audio device, or the - # PowerShell fallback failing. Fall through to the next player. + # e.g. WSL ffplay/aplay with no audio device — fall through to the next player. logger.debug("System player %s exited with code %d, trying next", cmd[0], rc) except subprocess.TimeoutExpired: logger.warning("System player %s timed out, killing process", cmd[0]) if proc is not None: proc.kill() proc.wait() - _set_active_playback(None) except Exception as e: logger.debug("System player %s failed: %s", cmd[0], e) + finally: _set_active_playback(None) return False @@ -1443,16 +1200,12 @@ def _play_audio_file_impl(file_path: str) -> bool: if not os.path.isfile(file_path): logger.warning("Audio file not found: %s", file_path) return False - - # sounddevice output is skipped on macOS (PortAudio/CoreAudio init triggers - # a TCC media-library prompt); afplay handles all formats there instead. + # macOS skips sounddevice output (TCC media-library prompt); afplay handles all formats. if file_path.endswith(".wav") and _sounddevice_output_allowed() and _play_wav_via_sounddevice(file_path): return True - for cmd in _system_player_candidates(file_path): if shutil.which(cmd[0]) and _run_system_player(cmd): return True - logger.warning("No audio player available for %s", file_path) return False @@ -1460,31 +1213,22 @@ def _play_audio_file_impl(file_path: str) -> bool: # ── Full-duplex agent-turn listener ── # One listener for the WHOLE agent turn in continuous voice mode: armed when an # utterance is submitted, disarmed when the turn (response + TTS) is done. It -# replaced per-playback barge monitors that (1) only listened during TTS, so -# the user could not interject during LLM generation, and (2) calibrated their -# noise floor against active speaker bleed with a strict consecutive-block -# counter, making the trigger unreachable for normal speech. This listener -# calibrates against the QUIET room at turn start, freezes that baseline -# through playback, and trips on a windowed majority of blocks. +# calibrates against the QUIET room at turn start, freezes that baseline through +# playback (never against speaker bleed), and trips on a windowed majority of +# blocks — so the user can interject during LLM generation, not just TTS. -# Minimum trigger while TTS audio is flowing. Speaker bleed reaching the mic -# is typically a few hundred RMS (~1000-1400 with loud speakers close to the -# mic), while direct speech at normal distance measures 3000-8000 RMS. +# Minimum trigger while TTS flows: speaker bleed reaching the mic is a few +# hundred RMS (~1000-1400 with loud close speakers); direct speech is 3000-8000. PLAYBACK_MIN_TRIGGER = 1500.0 - -# Absolute trigger ceiling — a noisy room must never push the trigger above -# what normal speech (3000-8000 RMS) can reach. +# A noisy room must never push the trigger above what normal speech can reach. TRIGGER_CEILING = 4000.0 - -# Trigger multiplier over the quiet-room floor (typically 50-300 RMS): 3x -# separates speech from ambient cleanly while staying reachable -# (300 RMS floor * 3 = 900 trigger vs 3000+ RMS speech). +# Multiplier over the quiet-room floor (typically 50-300 RMS): 3x separates +# speech from ambient while staying reachable (300 * 3 = 900 vs 3000+ speech). DEFAULT_BARGE_MULTIPLIER = 3.0 def _vad_log(msg: str) -> None: - """VAD decision-point diagnostic — always logger.debug, plus stderr when - HERMES_VOICE_DEBUG=1 so live hardware tuning doesn't need a log tail.""" + """VAD diagnostic: logger.debug, plus stderr when HERMES_VOICE_DEBUG=1 (live tuning).""" logger.debug(msg) if os.environ.get("HERMES_VOICE_DEBUG", "").strip() == "1": try: @@ -1494,9 +1238,9 @@ def _vad_log(msg: str) -> None: def _capture_until_quiet(stream, np, block: int, pre_roll, *, endpoint_blocks: int, max_blocks: int) -> str: - """Keep reading after a trip until *endpoint_blocks* of quiet (or - *max_blocks*), then write pre-roll + capture to a WAV and return its path. - Playback was cut by the trigger, so plain silence endpointing works.""" + """After a trip, read until *endpoint_blocks* of quiet (or *max_blocks*); write + pre-roll + capture to a WAV and return its path. Playback was cut by the + trigger, so plain silence endpointing works.""" frames: List[Any] = list(pre_roll) quiet = 0 for _ in range(max_blocks): @@ -1508,6 +1252,91 @@ def _capture_until_quiet(stream, np, block: int, pre_roll, *, endpoint_blocks: i return AudioRecorder._write_wav(np.concatenate(frames, axis=0)) +class _BargeDetector: + """Per-block barge-in state machine behind ``full_duplex_listen``.""" + + def __init__(self, np, *, mult: float, calib_blocks: int, trip_blocks: int, grace_blocks: int) -> None: + self._np = np + self.mult = mult + self.calib_blocks = calib_blocks + self.grace_blocks = grace_blocks + self.trip_needed = max(1, int(round(trip_blocks * 0.8))) + self.ambient: deque = deque(maxlen=100) # ~3s of quiet-phase RMS + self.recent_above: deque = deque(maxlen=trip_blocks) + self.quiet_floor = float(SILENCE_RMS_THRESHOLD) + self.floor_locked = False + self.playing_prev = False + self.playback_seen = False + self.grace_remaining = 0 + self.blocks_since_playback = 10_000 + self.block_idx = 0 + + def _floor(self) -> tuple: + """(pct90, floor): 90th percentile of the quiet window; floor never below the silence threshold.""" + seq = self.ambient + pct90 = float(self._np.percentile(list(seq), 90)) if seq else float(SILENCE_RMS_THRESHOLD) + return pct90, max(pct90, float(SILENCE_RMS_THRESHOLD)) + + def _calibrate(self, rms: float, playing: bool) -> bool: + """Pre-playback calibration: the listener arms at utterance submit, before + any TTS exists, so the first calib_blocks sample the actual quiet room — + NOT speaker bleed. Returns True once the floor is locked.""" + if not playing: + self.ambient.append(rms) + if len(self.ambient) >= self.calib_blocks or playing: + pct90, self.quiet_floor = self._floor() + self.floor_locked = True + _vad_log(f"calibrated quiet floor={self.quiet_floor:.0f} " + f"(pct90={pct90:.0f}, {len(self.ambient)} blocks, mult={self.mult:g})") + return self.floor_locked + + def _track_playback(self, playing: bool) -> None: + """Grace only when playback starts after a real gap (>=1s), so inter-sentence + flapping of the audio-active flag can't chain grace windows together and + swallow a genuine interjection.""" + if playing and not self.playing_prev: + if not self.playback_seen or self.blocks_since_playback > 33: + self.grace_remaining = self.grace_blocks + _vad_log(f"playback started (block={self.block_idx}) — grace {self.grace_blocks * 30}ms") + self.playback_seen = True + self.playing_prev = playing + self.blocks_since_playback = 0 if playing else self.blocks_since_playback + 1 + + def feed(self, rms: float, playing: bool) -> Optional[str]: + """Consume one 30ms block; return the phase name when speech trips, else None.""" + self.block_idx += 1 + if not self.floor_locked and not self._calibrate(rms, playing): + return None + self._track_playback(playing) + # Trigger: quiet baseline x multiplier, phase-clamped. + trigger = max(self.quiet_floor * self.mult, + PLAYBACK_MIN_TRIGGER if playing else float(SILENCE_RMS_THRESHOLD) * 2) + trigger = min(trigger, TRIGGER_CEILING) + # Track ambient drift ONLY while nothing is playing (never absorb bleed) and the block isn't speech. + if not playing and rms < trigger: + self.ambient.append(rms) + _, self.quiet_floor = self._floor() + above = rms >= trigger + if above and self.grace_remaining > 0: + _vad_log(f"grace suppression: block={self.block_idx} rms={rms:.0f} " + f"trigger={trigger:.0f} ({self.grace_remaining} blocks left)") + above = False + if self.grace_remaining > 0: + self.grace_remaining -= 1 + self.recent_above.append(above) + phase = "playback" if playing else "generation" + if rms >= trigger * 0.5: + _vad_log(f"block={self.block_idx} rms={rms:.0f} floor={self.quiet_floor:.0f} " + f"trigger={trigger:.0f} above={above} " + f"window={sum(self.recent_above)}/{self.trip_needed} phase={phase}") + if not (above and sum(self.recent_above) >= self.trip_needed): + return None + _vad_log(f"TRIPPED ({phase}): block={self.block_idx} rms={rms:.0f} " + f"floor={self.quiet_floor:.0f} trigger={trigger:.0f} " + f"window={sum(self.recent_above)}/{len(self.recent_above)}") + return phase + + def full_duplex_listen( should_stop: Callable[[], bool], is_playing: Optional[Callable[[], bool]] = None, @@ -1520,148 +1349,47 @@ def full_duplex_listen( endpoint_silence_ms: int = 1250, max_utterance_ms: int = 30_000, ) -> Optional[str]: - """Listen across an ENTIRE agent turn; return the captured interruption. + """Listen across an ENTIRE agent turn; return the captured interruption WAV path. Two phases, decided per 30ms block by *is_playing* (usually - ``is_audio_output_active``): ``generation`` (no TTS) — the first - *calibration_ms* of the quiet room set the noise floor and the trigger is - quiet_floor x *multiplier*; ``playback`` (TTS flowing) — the quiet - baseline is HELD (never recalibrated against speaker bleed), the trigger - is clamped up to ``PLAYBACK_MIN_TRIGGER`` so bleed alone can't trip it, - and a *grace_ms* window after playback starts suppresses onset transients. - - Detection is a windowed majority — >=80% of the last *sustained_ms* of - blocks above trigger (current block included) — so intra-word dips don't - reset progress. On detection ``on_trigger(phase)`` fires, capture - continues from the rolling *pre_roll_ms* buffer until - *endpoint_silence_ms* of quiet, and the WAV path is returned. Returns - ``None`` when *should_stop* ends the turn without speech. + ``is_audio_output_active``): ``generation`` — the first *calibration_ms* of + quiet room set the noise floor, trigger = floor x *multiplier*; + ``playback`` — the quiet baseline is HELD, the trigger is clamped up to + ``PLAYBACK_MIN_TRIGGER`` so bleed alone can't trip it, and *grace_ms* + after playback starts suppresses onset transients. Detection is a windowed + majority (>=80% of the last *sustained_ms* of blocks above trigger) so + intra-word dips don't reset progress. On detection ``on_trigger(phase)`` + fires and capture continues from the rolling *pre_roll_ms* buffer until + *endpoint_silence_ms* of quiet. Returns ``None`` when *should_stop* ends + the turn without speech. """ try: sd, np = _import_audio() except (ImportError, OSError): return None - - from collections import deque - block = int(SAMPLE_RATE * 0.03) # 30ms blocks - calib_blocks = max(1, calibration_ms // 30) - trip_blocks = max(1, sustained_ms // 30) - trip_needed = max(1, int(round(trip_blocks * 0.8))) - grace_blocks = max(0, grace_ms // 30) - endpoint_blocks = max(1, endpoint_silence_ms // 30) - max_blocks = max(1, max_utterance_ms // 30) - mult = float(multiplier) if multiplier else DEFAULT_BARGE_MULTIPLIER - - ambient: "deque[float]" = deque(maxlen=100) # ~3s of quiet-phase RMS + detector = _BargeDetector( + np, mult=float(multiplier) if multiplier else DEFAULT_BARGE_MULTIPLIER, + calib_blocks=max(1, calibration_ms // 30), trip_blocks=max(1, sustained_ms // 30), + grace_blocks=max(0, grace_ms // 30)) pre_roll: deque = deque(maxlen=max(1, pre_roll_ms // 30)) - recent_above: "deque[bool]" = deque(maxlen=trip_blocks) - quiet_floor = float(SILENCE_RMS_THRESHOLD) - floor_locked = False - playing_prev = False - playback_seen = False - grace_remaining = 0 - blocks_since_playback = 10_000 - block_idx = 0 - - def _floor(seq) -> tuple: - """(pct90, floor): 90th percentile of the quiet-phase RMS window, floor never below the silence threshold.""" - pct90 = float(np.percentile(list(seq), 90)) if seq else float(SILENCE_RMS_THRESHOLD) - return pct90, max(pct90, float(SILENCE_RMS_THRESHOLD)) - try: - with sd.InputStream( - samplerate=SAMPLE_RATE, channels=1, dtype="int16", blocksize=block - ) as stream: + with sd.InputStream(samplerate=SAMPLE_RATE, channels=1, dtype="int16", blocksize=block) as stream: while not should_stop(): data, _ = stream.read(block) - rms = _rms(np, data) pre_roll.append(data.copy()) - block_idx += 1 - playing = bool(is_playing()) if is_playing is not None else False - - # Pre-playback calibration: the listener arms at utterance - # submit, before any TTS exists, so the first calib_blocks - # sample the actual quiet room — NOT speaker bleed. - if not floor_locked: - if not playing: - ambient.append(rms) - if len(ambient) >= calib_blocks or playing: - pct90, quiet_floor = _floor(ambient) - floor_locked = True - _vad_log( - f"calibrated quiet floor={quiet_floor:.0f} " - f"(pct90={pct90:.0f}, {len(ambient)} blocks, " - f"mult={mult:g})" - ) - if not floor_locked: - continue - - # Playback phase transitions: grace only when playback starts - # after a real gap (>=1s), so inter-sentence flapping of the - # audio-active flag can't chain grace windows together and - # swallow a genuine interjection. - if playing and not playing_prev: - if not playback_seen or blocks_since_playback > 33: - grace_remaining = grace_blocks - _vad_log( - f"playback started (block={block_idx}) — " - f"grace {grace_blocks * 30}ms" - ) - playback_seen = True - playing_prev = playing - blocks_since_playback = 0 if playing else blocks_since_playback + 1 - - # Trigger: quiet baseline x multiplier, phase-clamped. - trigger = quiet_floor * mult - trigger = max(trigger, PLAYBACK_MIN_TRIGGER if playing else float(SILENCE_RMS_THRESHOLD) * 2) - trigger = min(trigger, TRIGGER_CEILING) - - # Track ambient drift — ONLY while nothing is playing (never - # absorb speaker bleed) and the block isn't speech. - if not playing and rms < trigger: - ambient.append(rms) - _, quiet_floor = _floor(ambient) - - above = rms >= trigger - if above and grace_remaining > 0: - _vad_log( - f"grace suppression: block={block_idx} rms={rms:.0f} " - f"trigger={trigger:.0f} ({grace_remaining} blocks left)" - ) - above = False - if grace_remaining > 0: - grace_remaining -= 1 - - recent_above.append(above) - if rms >= trigger * 0.5: - _vad_log( - f"block={block_idx} rms={rms:.0f} floor={quiet_floor:.0f} " - f"trigger={trigger:.0f} above={above} " - f"window={sum(recent_above)}/{trip_needed} " - f"phase={'playback' if playing else 'generation'}" - ) - - if not (above and sum(recent_above) >= trip_needed): + phase = detector.feed(_rms(np, data), playing) + if phase is None: continue - - phase = "playback" if playing else "generation" - _vad_log( - f"TRIPPED ({phase}): block={block_idx} rms={rms:.0f} " - f"floor={quiet_floor:.0f} trigger={trigger:.0f} " - f"window={sum(recent_above)}/{len(recent_above)}" - ) if on_trigger: try: on_trigger(phase) except Exception as e: logger.debug("full-duplex trigger callback failed: %s", e) - return _capture_until_quiet( stream, np, block, pre_roll, - endpoint_blocks=endpoint_blocks, max_blocks=max_blocks, - ) + endpoint_blocks=max(1, endpoint_silence_ms // 30), max_blocks=max(1, max_utterance_ms // 30)) except Exception as e: logger.debug("Full-duplex listener failed: %s", e) return None @@ -1669,36 +1397,31 @@ def full_duplex_listen( # ── Requirements check ── def _check_plugin_stt_provider(provider: str) -> bool: - """Return True when *provider* resolves to an available STT plugin.""" + """True when *provider* resolves to an available STT plugin.""" key = (provider or "").lower().strip() if not key or key == "none": return False try: from agent.transcription_registry import get_provider from hermes_cli.plugins import _ensure_plugins_discovered - _ensure_plugins_discovered() plugin_provider = get_provider(key) if plugin_provider is None: - # Match the transcription dispatcher: long-lived processes may - # need one refresh after plugins or configuration change. + # Match the transcription dispatcher: long-lived processes may need + # one refresh after plugins or configuration change. _ensure_plugins_discovered(force=True) plugin_provider = get_provider(key) except Exception as exc: # noqa: BLE001 - discovery failure is non-fatal logger.debug("STT plugin requirements check skipped for '%s': %s", key, exc) return False - if plugin_provider is None: return False - try: return bool(plugin_provider.is_available()) except Exception as exc: # noqa: BLE001 - plugins must not break status logger.warning( "STT plugin provider '%s' is_available() raised during requirements " - "check: %s - treating as unavailable", - key, exc, exc_info=True, - ) + "check: %s - treating as unavailable", key, exc, exc_info=True) return False @@ -1715,94 +1438,58 @@ _NATIVE_STT_LABELS = { def check_voice_requirements() -> Dict[str, Any]: - """Check if all voice mode requirements are met. + """Check voice mode requirements. Returns dict with ``available``, ``audio_available``, ``stt_available``, ``missing_packages``, ``details`` and ``environment``. """ from tools.transcription_tools import ( - _get_provider, - _load_stt_config, - _resolve_command_stt_provider_config, - is_stt_enabled, - ) + _get_provider, _load_stt_config, _resolve_command_stt_provider_config, is_stt_enabled) stt_config = _load_stt_config() stt_enabled = is_stt_enabled(stt_config) stt_provider = _get_provider(stt_config) - native_stt_available = stt_provider in _NATIVE_STT_LABELS - command_stt_config = None - plugin_stt_available = False - if stt_enabled and not native_stt_available: - command_stt_config = _resolve_command_stt_provider_config( - stt_provider, stt_config, - ) - if command_stt_config is None: - plugin_stt_available = _check_plugin_stt_provider(stt_provider) - stt_available = stt_enabled and ( - native_stt_available - or command_stt_config is not None - or plugin_stt_available - ) + stt_label = None # "OK (...)" once a native / command / plugin provider resolves + if stt_provider in _NATIVE_STT_LABELS: + stt_label = f"OK ({_NATIVE_STT_LABELS[stt_provider]})" + elif stt_enabled: + if _resolve_command_stt_provider_config(stt_provider, stt_config) is not None: + stt_label = f"OK (command: {stt_provider})" + elif _check_plugin_stt_provider(stt_provider): + stt_label = f"OK (plugin: {stt_provider})" + stt_available = stt_enabled and stt_label is not None - missing: List[str] = [] termux_capture = _termux_voice_capture_available() has_audio = _audio_available() or termux_capture - - if not has_audio: - missing.extend(["sounddevice", "numpy"]) - env_check = detect_audio_environment() - - available = has_audio and stt_available and env_check["available"] - details_parts = [] - - if termux_capture: - details_parts.append("Audio capture: OK (Termux:API microphone)") - elif has_audio: - details_parts.append("Audio capture: OK") - else: - details_parts.append(f"Audio capture: MISSING ({_voice_capture_install_hint()})") - - if not stt_enabled: - details_parts.append("STT provider: DISABLED in config (stt.enabled: false)") - elif stt_provider in _NATIVE_STT_LABELS: - details_parts.append(f"STT provider: OK ({_NATIVE_STT_LABELS[stt_provider]})") - elif command_stt_config is not None: - details_parts.append(f"STT provider: OK (command: {stt_provider})") - elif plugin_stt_available: - details_parts.append(f"STT provider: OK (plugin: {stt_provider})") - else: - details_parts.append( - "STT provider: MISSING (uv pip install faster-whisper — " - "`pip install faster-whisper` also works if pip is on PATH, " - "or set GROQ_API_KEY / VOICE_TOOLS_OPENAI_KEY)" - ) - - for warning in env_check["warnings"]: - details_parts.append(f"Environment: {warning}") - for notice in env_check.get("notices", []): - details_parts.append(f"Environment: {notice}") - + details = [ + "Audio capture: OK (Termux:API microphone)" if termux_capture + else "Audio capture: OK" if has_audio + else f"Audio capture: MISSING ({_voice_capture_install_hint()})", + "STT provider: DISABLED in config (stt.enabled: false)" if not stt_enabled + else f"STT provider: {stt_label}" if stt_label + else ("STT provider: MISSING (uv pip install faster-whisper — " + "`pip install faster-whisper` also works if pip is on PATH, " + "or set GROQ_API_KEY / VOICE_TOOLS_OPENAI_KEY)"), + ] + details += [f"Environment: {w}" for w in env_check["warnings"]] + details += [f"Environment: {n}" for n in env_check.get("notices", [])] return { - "available": available, + "available": has_audio and stt_available and env_check["available"], "audio_available": has_audio, "stt_available": stt_available, - "missing_packages": missing, - "details": "\n".join(details_parts), + "missing_packages": [] if has_audio else ["sounddevice", "numpy"], + "details": "\n".join(details), "environment": env_check, } # ── Temp file cleanup ── def cleanup_temp_recordings(max_age_seconds: int = 3600) -> int: - """Remove ``recording_*.wav`` temp files older than *max_age_seconds* - (default 1 hour); returns the number deleted.""" + """Remove ``recording_*.wav`` temp files older than *max_age_seconds*; returns the count.""" if not os.path.isdir(_TEMP_DIR): return 0 - deleted = 0 now = time.time() - for entry in os.scandir(_TEMP_DIR): if entry.is_file() and entry.name.startswith("recording_") and entry.name.endswith(".wav"): try: @@ -1811,7 +1498,6 @@ def cleanup_temp_recordings(max_age_seconds: int = 3600) -> int: deleted += 1 except OSError: pass - if deleted: logger.debug("Cleaned up %d old voice recordings", deleted) return deleted