diff --git a/tests/tools/test_voice_tts_echo_guard.py b/tests/tools/test_voice_tts_echo_guard.py index 83fe94379e..d4e8f03a43 100644 --- a/tests/tools/test_voice_tts_echo_guard.py +++ b/tests/tools/test_voice_tts_echo_guard.py @@ -39,6 +39,33 @@ class TestIsTtsEcho: transcript = "stop" assert is_tts_echo(transcript, spoken) is False + def test_short_fragment_of_longer_multi_sentence_reply_is_echo(self): + # Playback-phase captures are cut immediately on trigger and only + # span pre-roll + time-to-silence, so a real self-capture is + # typically a short fragment of a much longer spoken reply, not a + # near-verbatim repeat of the whole thing. A whole-string ratio + # dilutes with the length mismatch and misses this case (#75780 + # review). + spoken = ( + "Sure, here's a summary of what we found. The build failed " + "because of a missing dependency in the lockfile. I've already " + "gone ahead and regenerated it, and the tests are passing " + "again locally. Let me know if you'd like me to open a PR for " + "this or if you want to review the diff first before I do " + "anything else." + ) + transcript = "Sure, here's a summary of what we found." + assert is_tts_echo(transcript, spoken) is True + + def test_short_fragment_from_middle_of_reply_is_echo(self): + spoken = ( + "The deployment finished successfully. All three services " + "came up healthy, and the smoke tests passed without any " + "errors." + ) + transcript = "the smoke tests passed without any errors" + assert is_tts_echo(transcript, spoken) is True + def test_empty_inputs_are_not_echo(self): assert is_tts_echo("", "hello") is False assert is_tts_echo("hello", "") is False diff --git a/tools/voice_mode.py b/tools/voice_mode.py index d3fa07adec..a1e6277bb5 100644 --- a/tools/voice_mode.py +++ b/tools/voice_mode.py @@ -1338,6 +1338,16 @@ def is_tts_echo( a strong signal of speaker-bleed self-capture (fail-closed guard for the playback-phase full-duplex listener, which has no acoustic echo cancellation; see #75780). + + The playback-phase capture is cut immediately when the barge trigger + fires and only spans the pre-roll buffer plus time-to-silence, so for + any spoken reply longer than a clause the transcript is a short + FRAGMENT of `spoken_text`, not a near-verbatim repeat of the whole + thing. A whole-string ratio dilutes towards 0 as `spoken_text` grows + past the fragment's length, so when the whole-string check misses, we + also slide a window sized to the transcript's word count across + `spoken_text` and compare against each window, catching a short + fragment echoed from within a much longer multi-sentence reply. """ if not transcript or not spoken_text: return False @@ -1345,7 +1355,19 @@ def is_tts_echo( b = _normalize_for_echo_compare(spoken_text) if not a or not b: return False - return difflib.SequenceMatcher(None, a, b).ratio() >= threshold + if difflib.SequenceMatcher(None, a, b).ratio() >= threshold: + return True + b_words = b.split(" ") + a_word_count = len(a.split(" ")) + if a_word_count >= len(b_words): + return False + best_ratio = 0.0 + for start in range(0, len(b_words) - a_word_count + 1): + window = " ".join(b_words[start : start + a_word_count]) + ratio = difflib.SequenceMatcher(None, a, window).ratio() + if ratio > best_ratio: + best_ratio = ratio + return best_ratio >= threshold def voice_stop_hint() -> str: