From d4a753ea424bfbc03cd7891eaf2eef183edbe33d Mon Sep 17 00:00:00 2001 From: chelsealong Date: Sat, 1 Aug 2026 02:04:43 +0000 Subject: [PATCH] fix(voice): drop playback-phase barge transcripts that echo Hermes' own TTS The full-duplex barge-in listener added in 5081551f0 stays active during TTS playback with no acoustic echo cancellation. On some speaker/mic combinations, TTS bleed alone crosses the barge threshold, gets transcribed, and is queued as the next user turn -- whose reply is then spoken, captured, and queued again, producing an unbounded TTS -> STT -> TTS feedback loop (#75780). Add a fail-closed transcript-level guard: when a barge trip happens during the playback phase, compare the captured transcript against the TTS text Hermes just spoke (tools/voice_mode.is_tts_echo, a language-agnostic character-level similarity ratio). A close match is dropped instead of queued, and the mic is handed back to the normal continuous-listening loop. Generation-phase trips (no TTS playing, so no bleed is possible) are unaffected. --- cli.py | 25 ++++++++ tests/tools/test_voice_cli_integration.py | 73 +++++++++++++++++++++++ tests/tools/test_voice_tts_echo_guard.py | 57 ++++++++++++++++++ tools/voice_mode.py | 38 ++++++++++++ 4 files changed, 193 insertions(+) create mode 100644 tests/tools/test_voice_tts_echo_guard.py diff --git a/cli.py b/cli.py index 26e202711d..c72aef01ae 100644 --- a/cli.py +++ b/cli.py @@ -4723,6 +4723,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._voice_tts_done.set() self._voice_tts_stop = None # active streaming pipeline's stop event self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption + self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780) + self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip # Status bar visibility (toggled via /statusbar) self._status_bar_visible = True @@ -12566,6 +12568,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): tts_text = tts_text.strip() if not tts_text: return + self._voice_last_tts_text = tts_text # Use MP3 output for CLI playback (afplay doesn't handle OGG well). # The TTS tool may auto-convert MP3->OGG, but the original MP3 remains. @@ -12676,6 +12679,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # Latch BEFORE cutting anything: suppresses process_loop's # auto-restart until the capture is submitted. self._voice_barge_capture.set() + self._voice_barge_phase = phase if phase == "playback": logger.debug( "TTS CUT: full-duplex listener tripped during playback" @@ -12733,6 +12737,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): _cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}") self._disable_voice_mode() return + # Fail-closed echo guard (#75780): a playback-phase capture + # has no acoustic echo cancellation, so speaker bleed alone + # can trip the barge trigger. If the transcript is a close + # match for what Hermes just spoke, treat it as self-capture + # instead of queuing it as a user turn. + if getattr(self, "_voice_barge_phase", None) == "playback": + from tools.voice_mode import is_tts_echo + if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")): + logger.debug( + "Dropping playback-phase barge transcript as TTS echo: %r", + transcript, + ) + _cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}") + return self._pending_input.put(_VoiceInputMessage(transcript)) submitted = True elif not result.get("success"): @@ -13892,6 +13910,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # playback (speech cuts TTS), and disarms itself when the turn # is fully done. See _voice_full_duplex_listener. if self._voice_mode and self._voice_continuous: + self._voice_last_tts_text = "" threading.Thread( target=self._voice_full_duplex_listener, daemon=True ).start() @@ -13963,6 +13982,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): def stream_callback(delta: str): if text_queue is not None: text_queue.put(delta) + # Track what's actually being spoken so a playback-phase + # barge capture can be checked against it (echo guard, + # #75780). + self._voice_last_tts_text = (self._voice_last_tts_text or "") + delta # When voice mode is active, prepend a brief instruction so the # model responds concisely. The prefix is API-call-local only — @@ -15230,6 +15253,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._voice_tts_done.set() # Initially "done" (no TTS pending) self._voice_tts_stop = None # active streaming pipeline's stop event self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption + self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780) + self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1": self._install_tool_callbacks() diff --git a/tests/tools/test_voice_cli_integration.py b/tests/tools/test_voice_cli_integration.py index ee207d827f..389419cc10 100644 --- a/tests/tools/test_voice_cli_integration.py +++ b/tests/tools/test_voice_cli_integration.py @@ -413,6 +413,79 @@ class TestVoiceBargeCaptureSubmit: assert not cli._voice_barge_capture.is_set() assert restarted.wait(2.0) # continuous mode resumes listening + def test_playback_phase_echo_of_own_tts_is_dropped(self, tmp_path, monkeypatch): + """#75780: a playback-phase capture that closely matches the TTS + text Hermes just spoke is speaker bleed, not real user speech -- + it must be dropped instead of queued as the next turn, and the mic + handed back so continuous mode keeps listening.""" + cli = _make_voice_cli(_voice_mode=True, _voice_continuous=True) + cli._voice_barge_capture.set() + cli._voice_barge_phase = "playback" + cli._voice_last_tts_text = "네, 방금도 제 답변이 그대로 다시 입력됐어요." + wav = tmp_path / "barge.wav" + wav.write_bytes(b"RIFF") + restarted = threading.Event() + cli._voice_start_recording = lambda: restarted.set() + + monkeypatch.setattr( + "tools.voice_mode.transcribe_recording", + lambda path, model=None: { + "success": True, + "transcript": "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요.", + }, + ) + + cli._voice_submit_barge_utterance(str(wav)) + + assert cli._pending_input.empty() # not queued as a user turn + assert not cli._voice_barge_capture.is_set() + assert restarted.wait(2.0) # mic handed back instead of self-triggering another turn + + def test_playback_phase_genuine_interjection_is_still_queued(self, tmp_path, monkeypatch): + """A real user interjection during playback -- unrelated to the TTS + text -- must still reach the agent.""" + cli = _make_voice_cli() + cli._voice_barge_capture.set() + cli._voice_barge_phase = "playback" + cli._voice_last_tts_text = "The weather today is sunny with a light breeze." + wav = tmp_path / "barge.wav" + wav.write_bytes(b"RIFF") + + monkeypatch.setattr( + "tools.voice_mode.transcribe_recording", + lambda path, model=None: { + "success": True, + "transcript": "actually can you check my calendar for tomorrow", + }, + ) + + cli._voice_submit_barge_utterance(str(wav)) + + queued = cli._pending_input.get_nowait() + from cli import _VoiceInputMessage + assert str(queued) == "actually can you check my calendar for tomorrow" + + def test_generation_phase_transcript_not_echo_checked(self, tmp_path, monkeypatch): + """Generation-phase barges (no TTS playing) are never treated as + echo, even if the transcript happens to match old TTS text.""" + cli = _make_voice_cli() + cli._voice_barge_capture.set() + cli._voice_barge_phase = "generation" + cli._voice_last_tts_text = "stop, do it differently" + wav = tmp_path / "barge.wav" + wav.write_bytes(b"RIFF") + + monkeypatch.setattr( + "tools.voice_mode.transcribe_recording", + lambda path, model=None: {"success": True, "transcript": "stop, do it differently"}, + ) + + cli._voice_submit_barge_utterance(str(wav)) + + queued = cli._pending_input.get_nowait() + from cli import _VoiceInputMessage + assert str(queued) == "stop, do it differently" + # ============================================================================ # Full-duplex agent-turn listener — CLI phase behaviour diff --git a/tests/tools/test_voice_tts_echo_guard.py b/tests/tools/test_voice_tts_echo_guard.py new file mode 100644 index 0000000000..83fe94379e --- /dev/null +++ b/tests/tools/test_voice_tts_echo_guard.py @@ -0,0 +1,57 @@ +"""Tests for the playback-phase TTS-echo guard (#75780). + +Contract: + - `is_tts_echo` flags a barge-in transcript as a likely self-capture of + Hermes' own TTS output when it is a close character-level match for the + text that was just spoken, regardless of language/tokenization. + - A genuine, unrelated user interjection captured during playback must + NOT be flagged, even though it happens to share some words. + - `HermesCLI._voice_submit_barge_utterance` uses this guard ONLY for + playback-phase barge captures (generation-phase speech can't be TTS + bleed, since nothing is playing) and drops the echoed transcript + instead of queuing it as the next user turn. +""" + +from tools.voice_mode import is_tts_echo + + +class TestIsTtsEcho: + def test_near_verbatim_repeat_is_echo(self): + spoken = ( + "맞아요. 사용자가 마이크를 끄는 게 아니라 앱이 제 음성은 " + "에코 제거로 걸러내고 실제 사용자 음성만 끼어들기로 받아야 해요." + ) + transcript = spoken + assert is_tts_echo(transcript, spoken) is True + + def test_repeat_with_leading_stutter_is_echo(self): + spoken = "네, 방금도 제 답변이 그대로 다시 입력됐어요." + transcript = "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요." + assert is_tts_echo(transcript, spoken) is True + + def test_unrelated_interjection_is_not_echo(self): + spoken = "The weather today is sunny with a light breeze from the west." + transcript = "actually can you also check my calendar for tomorrow" + assert is_tts_echo(transcript, spoken) is False + + def test_short_unrelated_reply_is_not_echo(self): + spoken = "I've finished summarizing the document you shared earlier." + transcript = "stop" + assert is_tts_echo(transcript, spoken) is False + + def test_empty_inputs_are_not_echo(self): + assert is_tts_echo("", "hello") is False + assert is_tts_echo("hello", "") is False + assert is_tts_echo("", "") is False + + def test_case_and_whitespace_insensitive(self): + spoken = "Sure, I can help with that right away." + transcript = " SURE, I can help with that right away. " + assert is_tts_echo(transcript, spoken) is True + + def test_custom_threshold_is_honored(self): + spoken = "This is a moderately similar sentence about testing." + transcript = "This is a rather different sentence about coding." + # Lenient threshold treats it as an echo, strict threshold does not. + assert is_tts_echo(transcript, spoken, threshold=0.5) is True + assert is_tts_echo(transcript, spoken, threshold=0.95) is False diff --git a/tools/voice_mode.py b/tools/voice_mode.py index 8d58b01e4f..d3fa07adec 100644 --- a/tools/voice_mode.py +++ b/tools/voice_mode.py @@ -9,6 +9,7 @@ Dependencies (optional): or: uv sync --extra voice """ +import difflib import logging import math import os @@ -1310,6 +1311,43 @@ def is_voice_stop_phrase(transcript: str, stop_phrases: Optional[tuple] = None) return cleaned in stop_phrases +# Similarity ratio (difflib.SequenceMatcher, 0..1) above which a +# playback-phase barge transcript is treated as a self-capture of Hermes' +# own just-spoken TTS rather than genuine user speech. See #75780: the +# full-duplex listener has no acoustic echo cancellation, so speaker bleed +# on the mic can trip the barge trigger and get transcribed nearly +# verbatim from the TTS text, creating a TTS -> STT -> TTS feedback loop. +DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD = 0.6 + + +def _normalize_for_echo_compare(text: str) -> str: + return re.sub(r"\s+", " ", text).strip().lower() + + +def is_tts_echo( + transcript: str, + spoken_text: str, + threshold: float = DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD, +) -> bool: + """Return True when *transcript* looks like a self-capture of *spoken_text*. + + Compares a playback-phase barge-in transcript against the TTS text + Hermes just spoke using a character-level similarity ratio, which works + across languages without word-tokenization. A genuine user interjection + is very unlikely to closely match Hermes' own words, so a high ratio is + a strong signal of speaker-bleed self-capture (fail-closed guard for the + playback-phase full-duplex listener, which has no acoustic echo + cancellation; see #75780). + """ + if not transcript or not spoken_text: + return False + a = _normalize_for_echo_compare(transcript) + b = _normalize_for_echo_compare(spoken_text) + if not a or not b: + return False + return difflib.SequenceMatcher(None, a, b).ratio() >= threshold + + def voice_stop_hint() -> str: """One-line 'Say "stop" to end the voice chat.' hint for voice-mode start.