fix(voice): catch short echoed fragments of longer multi-sentence TTS replies
is_tts_echo() compared the captured transcript against the *entire* spoken text with a whole-string similarity ratio, which only scores high when the two strings are close in length. But a playback-phase barge capture is cut immediately when the trigger fires and only spans the pre-roll buffer plus time-to-silence, so a genuine self-capture is typically a short fragment of a longer reply, not a near-verbatim repeat of the whole thing -- for any response longer than a clause, the length-diluted ratio fell below threshold and the echo sailed through ungated. When the whole-string check misses, also slide a window sized to the transcript's word count across the spoken text and compare against each window, so a short fragment echoed from within a much longer multi-sentence reply is still caught. Add regression tests for a short fragment matched at the start and in the middle of a longer reply.
This commit is contained in:
@@ -39,6 +39,33 @@ class TestIsTtsEcho:
|
||||
transcript = "stop"
|
||||
assert is_tts_echo(transcript, spoken) is False
|
||||
|
||||
def test_short_fragment_of_longer_multi_sentence_reply_is_echo(self):
|
||||
# Playback-phase captures are cut immediately on trigger and only
|
||||
# span pre-roll + time-to-silence, so a real self-capture is
|
||||
# typically a short fragment of a much longer spoken reply, not a
|
||||
# near-verbatim repeat of the whole thing. A whole-string ratio
|
||||
# dilutes with the length mismatch and misses this case (#75780
|
||||
# review).
|
||||
spoken = (
|
||||
"Sure, here's a summary of what we found. The build failed "
|
||||
"because of a missing dependency in the lockfile. I've already "
|
||||
"gone ahead and regenerated it, and the tests are passing "
|
||||
"again locally. Let me know if you'd like me to open a PR for "
|
||||
"this or if you want to review the diff first before I do "
|
||||
"anything else."
|
||||
)
|
||||
transcript = "Sure, here's a summary of what we found."
|
||||
assert is_tts_echo(transcript, spoken) is True
|
||||
|
||||
def test_short_fragment_from_middle_of_reply_is_echo(self):
|
||||
spoken = (
|
||||
"The deployment finished successfully. All three services "
|
||||
"came up healthy, and the smoke tests passed without any "
|
||||
"errors."
|
||||
)
|
||||
transcript = "the smoke tests passed without any errors"
|
||||
assert is_tts_echo(transcript, spoken) is True
|
||||
|
||||
def test_empty_inputs_are_not_echo(self):
|
||||
assert is_tts_echo("", "hello") is False
|
||||
assert is_tts_echo("hello", "") is False
|
||||
|
||||
+23
-1
@@ -1338,6 +1338,16 @@ def is_tts_echo(
|
||||
a strong signal of speaker-bleed self-capture (fail-closed guard for the
|
||||
playback-phase full-duplex listener, which has no acoustic echo
|
||||
cancellation; see #75780).
|
||||
|
||||
The playback-phase capture is cut immediately when the barge trigger
|
||||
fires and only spans the pre-roll buffer plus time-to-silence, so for
|
||||
any spoken reply longer than a clause the transcript is a short
|
||||
FRAGMENT of `spoken_text`, not a near-verbatim repeat of the whole
|
||||
thing. A whole-string ratio dilutes towards 0 as `spoken_text` grows
|
||||
past the fragment's length, so when the whole-string check misses, we
|
||||
also slide a window sized to the transcript's word count across
|
||||
`spoken_text` and compare against each window, catching a short
|
||||
fragment echoed from within a much longer multi-sentence reply.
|
||||
"""
|
||||
if not transcript or not spoken_text:
|
||||
return False
|
||||
@@ -1345,7 +1355,19 @@ def is_tts_echo(
|
||||
b = _normalize_for_echo_compare(spoken_text)
|
||||
if not a or not b:
|
||||
return False
|
||||
return difflib.SequenceMatcher(None, a, b).ratio() >= threshold
|
||||
if difflib.SequenceMatcher(None, a, b).ratio() >= threshold:
|
||||
return True
|
||||
b_words = b.split(" ")
|
||||
a_word_count = len(a.split(" "))
|
||||
if a_word_count >= len(b_words):
|
||||
return False
|
||||
best_ratio = 0.0
|
||||
for start in range(0, len(b_words) - a_word_count + 1):
|
||||
window = " ".join(b_words[start : start + a_word_count])
|
||||
ratio = difflib.SequenceMatcher(None, a, window).ratio()
|
||||
if ratio > best_ratio:
|
||||
best_ratio = ratio
|
||||
return best_ratio >= threshold
|
||||
|
||||
|
||||
def voice_stop_hint() -> str:
|
||||
|
||||
Reference in New Issue
Block a user