fix(voice): drop playback-phase barge transcripts that echo Hermes' own TTS

The full-duplex barge-in listener added in 5081551f0 stays active during
TTS playback with no acoustic echo cancellation. On some speaker/mic
combinations, TTS bleed alone crosses the barge threshold, gets
transcribed, and is queued as the next user turn -- whose reply is then
spoken, captured, and queued again, producing an unbounded TTS -> STT ->
TTS feedback loop (#75780).

Add a fail-closed transcript-level guard: when a barge trip happens during
the playback phase, compare the captured transcript against the TTS text
Hermes just spoke (tools/voice_mode.is_tts_echo, a language-agnostic
character-level similarity ratio). A close match is dropped instead of
queued, and the mic is handed back to the normal continuous-listening
loop. Generation-phase trips (no TTS playing, so no bleed is possible)
are unaffected.
This commit is contained in:
chelsealong
2026-08-01 02:04:43 +00:00
committed by kshitij
parent 6ab7528c33
commit d4a753ea42
4 changed files with 193 additions and 0 deletions
+25
View File
@@ -4723,6 +4723,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
self._voice_tts_done.set()
self._voice_tts_stop = None # active streaming pipeline's stop event
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
# Status bar visibility (toggled via /statusbar)
self._status_bar_visible = True
@@ -12566,6 +12568,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
tts_text = tts_text.strip()
if not tts_text:
return
self._voice_last_tts_text = tts_text
# Use MP3 output for CLI playback (afplay doesn't handle OGG well).
# The TTS tool may auto-convert MP3->OGG, but the original MP3 remains.
@@ -12676,6 +12679,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
# Latch BEFORE cutting anything: suppresses process_loop's
# auto-restart until the capture is submitted.
self._voice_barge_capture.set()
self._voice_barge_phase = phase
if phase == "playback":
logger.debug(
"TTS CUT: full-duplex listener tripped during playback"
@@ -12733,6 +12737,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
_cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}")
self._disable_voice_mode()
return
# Fail-closed echo guard (#75780): a playback-phase capture
# has no acoustic echo cancellation, so speaker bleed alone
# can trip the barge trigger. If the transcript is a close
# match for what Hermes just spoke, treat it as self-capture
# instead of queuing it as a user turn.
if getattr(self, "_voice_barge_phase", None) == "playback":
from tools.voice_mode import is_tts_echo
if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")):
logger.debug(
"Dropping playback-phase barge transcript as TTS echo: %r",
transcript,
)
_cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}")
return
self._pending_input.put(_VoiceInputMessage(transcript))
submitted = True
elif not result.get("success"):
@@ -13892,6 +13910,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
# playback (speech cuts TTS), and disarms itself when the turn
# is fully done. See _voice_full_duplex_listener.
if self._voice_mode and self._voice_continuous:
self._voice_last_tts_text = ""
threading.Thread(
target=self._voice_full_duplex_listener, daemon=True
).start()
@@ -13963,6 +13982,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
def stream_callback(delta: str):
if text_queue is not None:
text_queue.put(delta)
# Track what's actually being spoken so a playback-phase
# barge capture can be checked against it (echo guard,
# #75780).
self._voice_last_tts_text = (self._voice_last_tts_text or "") + delta
# When voice mode is active, prepend a brief instruction so the
# model responds concisely. The prefix is API-call-local only —
@@ -15230,6 +15253,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
self._voice_tts_done.set() # Initially "done" (no TTS pending)
self._voice_tts_stop = None # active streaming pipeline's stop event
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1":
self._install_tool_callbacks()
+73
View File
@@ -413,6 +413,79 @@ class TestVoiceBargeCaptureSubmit:
assert not cli._voice_barge_capture.is_set()
assert restarted.wait(2.0) # continuous mode resumes listening
def test_playback_phase_echo_of_own_tts_is_dropped(self, tmp_path, monkeypatch):
"""#75780: a playback-phase capture that closely matches the TTS
text Hermes just spoke is speaker bleed, not real user speech --
it must be dropped instead of queued as the next turn, and the mic
handed back so continuous mode keeps listening."""
cli = _make_voice_cli(_voice_mode=True, _voice_continuous=True)
cli._voice_barge_capture.set()
cli._voice_barge_phase = "playback"
cli._voice_last_tts_text = "네, 방금도 제 답변이 그대로 다시 입력됐어요."
wav = tmp_path / "barge.wav"
wav.write_bytes(b"RIFF")
restarted = threading.Event()
cli._voice_start_recording = lambda: restarted.set()
monkeypatch.setattr(
"tools.voice_mode.transcribe_recording",
lambda path, model=None: {
"success": True,
"transcript": "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요.",
},
)
cli._voice_submit_barge_utterance(str(wav))
assert cli._pending_input.empty() # not queued as a user turn
assert not cli._voice_barge_capture.is_set()
assert restarted.wait(2.0) # mic handed back instead of self-triggering another turn
def test_playback_phase_genuine_interjection_is_still_queued(self, tmp_path, monkeypatch):
"""A real user interjection during playback -- unrelated to the TTS
text -- must still reach the agent."""
cli = _make_voice_cli()
cli._voice_barge_capture.set()
cli._voice_barge_phase = "playback"
cli._voice_last_tts_text = "The weather today is sunny with a light breeze."
wav = tmp_path / "barge.wav"
wav.write_bytes(b"RIFF")
monkeypatch.setattr(
"tools.voice_mode.transcribe_recording",
lambda path, model=None: {
"success": True,
"transcript": "actually can you check my calendar for tomorrow",
},
)
cli._voice_submit_barge_utterance(str(wav))
queued = cli._pending_input.get_nowait()
from cli import _VoiceInputMessage
assert str(queued) == "actually can you check my calendar for tomorrow"
def test_generation_phase_transcript_not_echo_checked(self, tmp_path, monkeypatch):
"""Generation-phase barges (no TTS playing) are never treated as
echo, even if the transcript happens to match old TTS text."""
cli = _make_voice_cli()
cli._voice_barge_capture.set()
cli._voice_barge_phase = "generation"
cli._voice_last_tts_text = "stop, do it differently"
wav = tmp_path / "barge.wav"
wav.write_bytes(b"RIFF")
monkeypatch.setattr(
"tools.voice_mode.transcribe_recording",
lambda path, model=None: {"success": True, "transcript": "stop, do it differently"},
)
cli._voice_submit_barge_utterance(str(wav))
queued = cli._pending_input.get_nowait()
from cli import _VoiceInputMessage
assert str(queued) == "stop, do it differently"
# ============================================================================
# Full-duplex agent-turn listener — CLI phase behaviour
+57
View File
@@ -0,0 +1,57 @@
"""Tests for the playback-phase TTS-echo guard (#75780).
Contract:
- `is_tts_echo` flags a barge-in transcript as a likely self-capture of
Hermes' own TTS output when it is a close character-level match for the
text that was just spoken, regardless of language/tokenization.
- A genuine, unrelated user interjection captured during playback must
NOT be flagged, even though it happens to share some words.
- `HermesCLI._voice_submit_barge_utterance` uses this guard ONLY for
playback-phase barge captures (generation-phase speech can't be TTS
bleed, since nothing is playing) and drops the echoed transcript
instead of queuing it as the next user turn.
"""
from tools.voice_mode import is_tts_echo
class TestIsTtsEcho:
def test_near_verbatim_repeat_is_echo(self):
spoken = (
"맞아요. 사용자가 마이크를 끄는 게 아니라 앱이 제 음성은 "
"에코 제거로 걸러내고 실제 사용자 음성만 끼어들기로 받아야 해요."
)
transcript = spoken
assert is_tts_echo(transcript, spoken) is True
def test_repeat_with_leading_stutter_is_echo(self):
spoken = "네, 방금도 제 답변이 그대로 다시 입력됐어요."
transcript = "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요."
assert is_tts_echo(transcript, spoken) is True
def test_unrelated_interjection_is_not_echo(self):
spoken = "The weather today is sunny with a light breeze from the west."
transcript = "actually can you also check my calendar for tomorrow"
assert is_tts_echo(transcript, spoken) is False
def test_short_unrelated_reply_is_not_echo(self):
spoken = "I've finished summarizing the document you shared earlier."
transcript = "stop"
assert is_tts_echo(transcript, spoken) is False
def test_empty_inputs_are_not_echo(self):
assert is_tts_echo("", "hello") is False
assert is_tts_echo("hello", "") is False
assert is_tts_echo("", "") is False
def test_case_and_whitespace_insensitive(self):
spoken = "Sure, I can help with that right away."
transcript = " SURE, I can help with that right away. "
assert is_tts_echo(transcript, spoken) is True
def test_custom_threshold_is_honored(self):
spoken = "This is a moderately similar sentence about testing."
transcript = "This is a rather different sentence about coding."
# Lenient threshold treats it as an echo, strict threshold does not.
assert is_tts_echo(transcript, spoken, threshold=0.5) is True
assert is_tts_echo(transcript, spoken, threshold=0.95) is False
+38
View File
@@ -9,6 +9,7 @@ Dependencies (optional):
or: uv sync --extra voice
"""
import difflib
import logging
import math
import os
@@ -1310,6 +1311,43 @@ def is_voice_stop_phrase(transcript: str, stop_phrases: Optional[tuple] = None)
return cleaned in stop_phrases
# Similarity ratio (difflib.SequenceMatcher, 0..1) above which a
# playback-phase barge transcript is treated as a self-capture of Hermes'
# own just-spoken TTS rather than genuine user speech. See #75780: the
# full-duplex listener has no acoustic echo cancellation, so speaker bleed
# on the mic can trip the barge trigger and get transcribed nearly
# verbatim from the TTS text, creating a TTS -> STT -> TTS feedback loop.
DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD = 0.6
def _normalize_for_echo_compare(text: str) -> str:
return re.sub(r"\s+", " ", text).strip().lower()
def is_tts_echo(
transcript: str,
spoken_text: str,
threshold: float = DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD,
) -> bool:
"""Return True when *transcript* looks like a self-capture of *spoken_text*.
Compares a playback-phase barge-in transcript against the TTS text
Hermes just spoke using a character-level similarity ratio, which works
across languages without word-tokenization. A genuine user interjection
is very unlikely to closely match Hermes' own words, so a high ratio is
a strong signal of speaker-bleed self-capture (fail-closed guard for the
playback-phase full-duplex listener, which has no acoustic echo
cancellation; see #75780).
"""
if not transcript or not spoken_text:
return False
a = _normalize_for_echo_compare(transcript)
b = _normalize_for_echo_compare(spoken_text)
if not a or not b:
return False
return difflib.SequenceMatcher(None, a, b).ratio() >= threshold
def voice_stop_hint() -> str:
"""One-line 'Say "stop" to end the voice chat.' hint for voice-mode start.