fix(voice): drop playback-phase barge transcripts that echo Hermes' own TTS

The full-duplex barge-in listener added in 5081551f0 stays active during
TTS playback with no acoustic echo cancellation. On some speaker/mic
combinations, TTS bleed alone crosses the barge threshold, gets
transcribed, and is queued as the next user turn -- whose reply is then
spoken, captured, and queued again, producing an unbounded TTS -> STT ->
TTS feedback loop (#75780).

Add a fail-closed transcript-level guard: when a barge trip happens during
the playback phase, compare the captured transcript against the TTS text
Hermes just spoke (tools/voice_mode.is_tts_echo, a language-agnostic
character-level similarity ratio). A close match is dropped instead of
queued, and the mic is handed back to the normal continuous-listening
loop. Generation-phase trips (no TTS playing, so no bleed is possible)
are unaffected.
This commit is contained in:
chelsealong
2026-08-01 02:04:43 +00:00
committed by kshitij
parent 6ab7528c33
commit d4a753ea42
4 changed files with 193 additions and 0 deletions
+25
View File
@@ -4723,6 +4723,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
self._voice_tts_done.set()
self._voice_tts_stop = None # active streaming pipeline's stop event
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
# Status bar visibility (toggled via /statusbar)
self._status_bar_visible = True
@@ -12566,6 +12568,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
tts_text = tts_text.strip()
if not tts_text:
return
self._voice_last_tts_text = tts_text
# Use MP3 output for CLI playback (afplay doesn't handle OGG well).
# The TTS tool may auto-convert MP3->OGG, but the original MP3 remains.
@@ -12676,6 +12679,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
# Latch BEFORE cutting anything: suppresses process_loop's
# auto-restart until the capture is submitted.
self._voice_barge_capture.set()
self._voice_barge_phase = phase
if phase == "playback":
logger.debug(
"TTS CUT: full-duplex listener tripped during playback"
@@ -12733,6 +12737,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
_cprint(f"\n{_DIM}Stop phrase detected — ending voice chat.{_RST}")
self._disable_voice_mode()
return
# Fail-closed echo guard (#75780): a playback-phase capture
# has no acoustic echo cancellation, so speaker bleed alone
# can trip the barge trigger. If the transcript is a close
# match for what Hermes just spoke, treat it as self-capture
# instead of queuing it as a user turn.
if getattr(self, "_voice_barge_phase", None) == "playback":
from tools.voice_mode import is_tts_echo
if is_tts_echo(transcript, getattr(self, "_voice_last_tts_text", "")):
logger.debug(
"Dropping playback-phase barge transcript as TTS echo: %r",
transcript,
)
_cprint(f"\n{_DIM}Ignored likely TTS echo (not queued).{_RST}")
return
self._pending_input.put(_VoiceInputMessage(transcript))
submitted = True
elif not result.get("success"):
@@ -13892,6 +13910,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
# playback (speech cuts TTS), and disarms itself when the turn
# is fully done. See _voice_full_duplex_listener.
if self._voice_mode and self._voice_continuous:
self._voice_last_tts_text = ""
threading.Thread(
target=self._voice_full_duplex_listener, daemon=True
).start()
@@ -13963,6 +13982,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
def stream_callback(delta: str):
if text_queue is not None:
text_queue.put(delta)
# Track what's actually being spoken so a playback-phase
# barge capture can be checked against it (echo guard,
# #75780).
self._voice_last_tts_text = (self._voice_last_tts_text or "") + delta
# When voice mode is active, prepend a brief instruction so the
# model responds concisely. The prefix is API-call-local only —
@@ -15230,6 +15253,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
self._voice_tts_done.set() # Initially "done" (no TTS pending)
self._voice_tts_stop = None # active streaming pipeline's stop event
self._voice_barge_capture = threading.Event() # barge monitor is capturing the interruption
self._voice_last_tts_text = "" # most recently spoken TTS text (echo guard, #75780)
self._voice_barge_phase = None # "generation" or "playback" phase of the last barge trip
if os.environ.get("HERMES_DEFER_AGENT_STARTUP") != "1":
self._install_tool_callbacks()