fix(voice): drop playback-phase barge transcripts that echo Hermes' own TTS
The full-duplex barge-in listener added in 5081551f0 stays active during
TTS playback with no acoustic echo cancellation. On some speaker/mic
combinations, TTS bleed alone crosses the barge threshold, gets
transcribed, and is queued as the next user turn -- whose reply is then
spoken, captured, and queued again, producing an unbounded TTS -> STT ->
TTS feedback loop (#75780).
Add a fail-closed transcript-level guard: when a barge trip happens during
the playback phase, compare the captured transcript against the TTS text
Hermes just spoke (tools/voice_mode.is_tts_echo, a language-agnostic
character-level similarity ratio). A close match is dropped instead of
queued, and the mic is handed back to the normal continuous-listening
loop. Generation-phase trips (no TTS playing, so no bleed is possible)
are unaffected.
This commit is contained in:
@@ -4723,6 +4723,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
self._voice_tts_done.set()
|
||||
self._voice_tts_stop = None # active streaming pipeline's stop event
|
||||
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
|
||||
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
|
||||
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
|
||||
|
||||
# Status bar visibility (toggled via /statusbar)
|
||||
self._status_bar_visible = True
|
||||
@@ -12566,6 +12568,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
tts_text = tts_text.strip()
|
||||
if not tts_text:
|
||||
return
|
||||
self._voice_last_tts_text = tts_text
|
||||
|
||||
# Use MP3 output for CLI playback (afplay doesn't handle OGG well).
|
||||
# The TTS tool may auto-convert MP3->OGG, but the original MP3 remains.
|
||||
@@ -12676,6 +12679,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
# Latch BEFORE cutting anything: suppresses process_loop's
|
||||
# auto-restart until the capture is submitted.
|
||||
self._voice_barge_capture.set()
|
||||
self._voice_barge_phase = phase
|
||||
if phase == "playback":
|
||||
logger.debug(
|
||||
"TTS CUT: full-duplex listener tripped during playback"
|
||||
@@ -12733,6 +12737,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
_cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}")
|
||||
self._disable_voice_mode()
|
||||
return
|
||||
# Fail-closed echo guard (#75780): a playback-phase capture
|
||||
# has no acoustic echo cancellation, so speaker bleed alone
|
||||
# can trip the barge trigger. If the transcript is a close
|
||||
# match for what Hermes just spoke, treat it as self-capture
|
||||
# instead of queuing it as a user turn.
|
||||
if getattr(self, "_voice_barge_phase", None) == "playback":
|
||||
from tools.voice_mode import is_tts_echo
|
||||
if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")):
|
||||
logger.debug(
|
||||
"Dropping playback-phase barge transcript as TTS echo: %r",
|
||||
transcript,
|
||||
)
|
||||
_cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}")
|
||||
return
|
||||
self._pending_input.put(_VoiceInputMessage(transcript))
|
||||
submitted = True
|
||||
elif not result.get("success"):
|
||||
@@ -13892,6 +13910,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
# playback (speech cuts TTS), and disarms itself when the turn
|
||||
# is fully done. See _voice_full_duplex_listener.
|
||||
if self._voice_mode and self._voice_continuous:
|
||||
self._voice_last_tts_text = ""
|
||||
threading.Thread(
|
||||
target=self._voice_full_duplex_listener, daemon=True
|
||||
).start()
|
||||
@@ -13963,6 +13982,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
def stream_callback(delta: str):
|
||||
if text_queue is not None:
|
||||
text_queue.put(delta)
|
||||
# Track what's actually being spoken so a playback-phase
|
||||
# barge capture can be checked against it (echo guard,
|
||||
# #75780).
|
||||
self._voice_last_tts_text = (self._voice_last_tts_text or "") + delta
|
||||
|
||||
# When voice mode is active, prepend a brief instruction so the
|
||||
# model responds concisely. The prefix is API-call-local only —
|
||||
@@ -15230,6 +15253,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
||||
self._voice_tts_done.set() # Initially "done" (no TTS pending)
|
||||
self._voice_tts_stop = None # active streaming pipeline's stop event
|
||||
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
|
||||
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
|
||||
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
|
||||
|
||||
if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1":
|
||||
self._install_tool_callbacks()
|
||||
|
||||
@@ -413,6 +413,79 @@ class TestVoiceBargeCaptureSubmit:
|
||||
assert not cli._voice_barge_capture.is_set()
|
||||
assert restarted.wait(2.0) # continuous mode resumes listening
|
||||
|
||||
def test_playback_phase_echo_of_own_tts_is_dropped(self, tmp_path, monkeypatch):
|
||||
"""#75780: a playback-phase capture that closely matches the TTS
|
||||
text Hermes just spoke is speaker bleed, not real user speech --
|
||||
it must be dropped instead of queued as the next turn, and the mic
|
||||
handed back so continuous mode keeps listening."""
|
||||
cli = _make_voice_cli(_voice_mode=True, _voice_continuous=True)
|
||||
cli._voice_barge_capture.set()
|
||||
cli._voice_barge_phase = "playback"
|
||||
cli._voice_last_tts_text = "네, 방금도 제 답변이 그대로 다시 입력됐어요."
|
||||
wav = tmp_path / "barge.wav"
|
||||
wav.write_bytes(b"RIFF")
|
||||
restarted = threading.Event()
|
||||
cli._voice_start_recording = lambda: restarted.set()
|
||||
|
||||
monkeypatch.setattr(
|
||||
"tools.voice_mode.transcribe_recording",
|
||||
lambda path, model=None: {
|
||||
"success": True,
|
||||
"transcript": "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요.",
|
||||
},
|
||||
)
|
||||
|
||||
cli._voice_submit_barge_utterance(str(wav))
|
||||
|
||||
assert cli._pending_input.empty() # not queued as a user turn
|
||||
assert not cli._voice_barge_capture.is_set()
|
||||
assert restarted.wait(2.0) # mic handed back instead of self-triggering another turn
|
||||
|
||||
def test_playback_phase_genuine_interjection_is_still_queued(self, tmp_path, monkeypatch):
|
||||
"""A real user interjection during playback -- unrelated to the TTS
|
||||
text -- must still reach the agent."""
|
||||
cli = _make_voice_cli()
|
||||
cli._voice_barge_capture.set()
|
||||
cli._voice_barge_phase = "playback"
|
||||
cli._voice_last_tts_text = "The weather today is sunny with a light breeze."
|
||||
wav = tmp_path / "barge.wav"
|
||||
wav.write_bytes(b"RIFF")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"tools.voice_mode.transcribe_recording",
|
||||
lambda path, model=None: {
|
||||
"success": True,
|
||||
"transcript": "actually can you check my calendar for tomorrow",
|
||||
},
|
||||
)
|
||||
|
||||
cli._voice_submit_barge_utterance(str(wav))
|
||||
|
||||
queued = cli._pending_input.get_nowait()
|
||||
from cli import _VoiceInputMessage
|
||||
assert str(queued) == "actually can you check my calendar for tomorrow"
|
||||
|
||||
def test_generation_phase_transcript_not_echo_checked(self, tmp_path, monkeypatch):
|
||||
"""Generation-phase barges (no TTS playing) are never treated as
|
||||
echo, even if the transcript happens to match old TTS text."""
|
||||
cli = _make_voice_cli()
|
||||
cli._voice_barge_capture.set()
|
||||
cli._voice_barge_phase = "generation"
|
||||
cli._voice_last_tts_text = "stop, do it differently"
|
||||
wav = tmp_path / "barge.wav"
|
||||
wav.write_bytes(b"RIFF")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"tools.voice_mode.transcribe_recording",
|
||||
lambda path, model=None: {"success": True, "transcript": "stop, do it differently"},
|
||||
)
|
||||
|
||||
cli._voice_submit_barge_utterance(str(wav))
|
||||
|
||||
queued = cli._pending_input.get_nowait()
|
||||
from cli import _VoiceInputMessage
|
||||
assert str(queued) == "stop, do it differently"
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# Full-duplex agent-turn listener — CLI phase behaviour
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Tests for the playback-phase TTS-echo guard (#75780).
|
||||
|
||||
Contract:
|
||||
- `is_tts_echo` flags a barge-in transcript as a likely self-capture of
|
||||
Hermes' own TTS output when it is a close character-level match for the
|
||||
text that was just spoken, regardless of language/tokenization.
|
||||
- A genuine, unrelated user interjection captured during playback must
|
||||
NOT be flagged, even though it happens to share some words.
|
||||
- `HermesCLI._voice_submit_barge_utterance` uses this guard ONLY for
|
||||
playback-phase barge captures (generation-phase speech can't be TTS
|
||||
bleed, since nothing is playing) and drops the echoed transcript
|
||||
instead of queuing it as the next user turn.
|
||||
"""
|
||||
|
||||
from tools.voice_mode import is_tts_echo
|
||||
|
||||
|
||||
class TestIsTtsEcho:
|
||||
def test_near_verbatim_repeat_is_echo(self):
|
||||
spoken = (
|
||||
"맞아요. 사용자가 마이크를 끄는 게 아니라 앱이 제 음성은 "
|
||||
"에코 제거로 걸러내고 실제 사용자 음성만 끼어들기로 받아야 해요."
|
||||
)
|
||||
transcript = spoken
|
||||
assert is_tts_echo(transcript, spoken) is True
|
||||
|
||||
def test_repeat_with_leading_stutter_is_echo(self):
|
||||
spoken = "네, 방금도 제 답변이 그대로 다시 입력됐어요."
|
||||
transcript = "네 방금 네 방금도 제 답변이 그대로 다시 입력됐어요."
|
||||
assert is_tts_echo(transcript, spoken) is True
|
||||
|
||||
def test_unrelated_interjection_is_not_echo(self):
|
||||
spoken = "The weather today is sunny with a light breeze from the west."
|
||||
transcript = "actually can you also check my calendar for tomorrow"
|
||||
assert is_tts_echo(transcript, spoken) is False
|
||||
|
||||
def test_short_unrelated_reply_is_not_echo(self):
|
||||
spoken = "I've finished summarizing the document you shared earlier."
|
||||
transcript = "stop"
|
||||
assert is_tts_echo(transcript, spoken) is False
|
||||
|
||||
def test_empty_inputs_are_not_echo(self):
|
||||
assert is_tts_echo("", "hello") is False
|
||||
assert is_tts_echo("hello", "") is False
|
||||
assert is_tts_echo("", "") is False
|
||||
|
||||
def test_case_and_whitespace_insensitive(self):
|
||||
spoken = "Sure, I can help with that right away."
|
||||
transcript = " SURE, I can help with that right away. "
|
||||
assert is_tts_echo(transcript, spoken) is True
|
||||
|
||||
def test_custom_threshold_is_honored(self):
|
||||
spoken = "This is a moderately similar sentence about testing."
|
||||
transcript = "This is a rather different sentence about coding."
|
||||
# Lenient threshold treats it as an echo, strict threshold does not.
|
||||
assert is_tts_echo(transcript, spoken, threshold=0.5) is True
|
||||
assert is_tts_echo(transcript, spoken, threshold=0.95) is False
|
||||
@@ -9,6 +9,7 @@ Dependencies (optional):
|
||||
or: uv sync --extra voice
|
||||
"""
|
||||
|
||||
import difflib
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
@@ -1310,6 +1311,43 @@ def is_voice_stop_phrase(transcript: str, stop_phrases: Optional[tuple] = None)
|
||||
return cleaned in stop_phrases
|
||||
|
||||
|
||||
# Similarity ratio (difflib.SequenceMatcher, 0..1) above which a
|
||||
# playback-phase barge transcript is treated as a self-capture of Hermes'
|
||||
# own just-spoken TTS rather than genuine user speech. See #75780: the
|
||||
# full-duplex listener has no acoustic echo cancellation, so speaker bleed
|
||||
# on the mic can trip the barge trigger and get transcribed nearly
|
||||
# verbatim from the TTS text, creating a TTS -> STT -> TTS feedback loop.
|
||||
DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD = 0.6
|
||||
|
||||
|
||||
def _normalize_for_echo_compare(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", text).strip().lower()
|
||||
|
||||
|
||||
def is_tts_echo(
|
||||
transcript: str,
|
||||
spoken_text: str,
|
||||
threshold: float = DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD,
|
||||
) -> bool:
|
||||
"""Return True when *transcript* looks like a self-capture of *spoken_text*.
|
||||
|
||||
Compares a playback-phase barge-in transcript against the TTS text
|
||||
Hermes just spoke using a character-level similarity ratio, which works
|
||||
across languages without word-tokenization. A genuine user interjection
|
||||
is very unlikely to closely match Hermes' own words, so a high ratio is
|
||||
a strong signal of speaker-bleed self-capture (fail-closed guard for the
|
||||
playback-phase full-duplex listener, which has no acoustic echo
|
||||
cancellation; see #75780).
|
||||
"""
|
||||
if not transcript or not spoken_text:
|
||||
return False
|
||||
a = _normalize_for_echo_compare(transcript)
|
||||
b = _normalize_for_echo_compare(spoken_text)
|
||||
if not a or not b:
|
||||
return False
|
||||
return difflib.SequenceMatcher(None, a, b).ratio() >= threshold
|
||||
|
||||
|
||||
def voice_stop_hint() -> str:
|
||||
"""One-line 'Say "stop" to end the voice chat.' hint for voice-mode start.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user