refactor(tools): unify STT/TTS command-provider config helpers, REST STT flow, dispatch tables
This commit is contained in:
+68
-119
@@ -1,11 +1,8 @@
|
||||
"""Utilities for preparing assistant text for speech synthesis.
|
||||
"""Deterministic cleanup turning assistant Markdown into a spoken script.
|
||||
|
||||
The TTS provider should receive a spoken script, not raw chat Markdown. This
|
||||
module centralises the lightweight, deterministic cleanup used by explicit TTS
|
||||
calls and gateway auto-TTS replies.
|
||||
|
||||
Non-ASCII characters are written as escapes on purpose so the file stays free of
|
||||
invisible/look-alike glyphs.
|
||||
Shared by explicit TTS calls, gateway auto-TTS, voice-mode streaming and the web
|
||||
dashboard. Non-ASCII characters are written as escapes on purpose so the file
|
||||
stays free of invisible/look-alike glyphs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -13,9 +10,9 @@ from __future__ import annotations
|
||||
import html
|
||||
import re
|
||||
|
||||
# Sentinel appended to former heading lines so smooth_whitespace_for_tts can
|
||||
# fold a heading into the sentence that follows it ("Weather, it will be sunny")
|
||||
# rather than leaving a bare "Weather." label that reads abruptly aloud.
|
||||
# Sentinel appended to former heading lines so smooth_whitespace_for_tts folds the
|
||||
# heading into the sentence after it ("Weather, it will be sunny") instead of a bare
|
||||
# "Weather." label.
|
||||
_HEAD = "\x00"
|
||||
|
||||
_MD_CODE_BLOCK_RE = re.compile(r"```[\s\S]*?```")
|
||||
@@ -34,21 +31,22 @@ _MD_HR_RE = re.compile(r"^\s*[-*_]{3,}\s*$", flags=re.MULTILINE)
|
||||
_MD_TABLE_PIPE_RE = re.compile(r"\s*\|\s*")
|
||||
_URL_RE = re.compile(r"https?://\S+")
|
||||
|
||||
# Broad emoji / pictograph cleanup. Voice providers vary a lot here; most read
|
||||
# emojis as awkward labels, so keep the speech script calm and literal.
|
||||
# Unit suffix (regex, after a digit) -> spoken word; km/h variants before the bare "m".
|
||||
_UNIT_WORDS = (
|
||||
(r"km\s*/\s*h", "kilometres per hour"), (r"km/h", "kilometres per hour"),
|
||||
(r"mm", "millimetres"), (r"cm", "centimetres"), (r"m", "metres"),
|
||||
)
|
||||
# Currency prefix (regex) -> spoken word; order matters (NZ$/A$/US$ before bare $).
|
||||
_CURRENCY_WORDS = (
|
||||
(r"NZ\$", "New Zealand dollars", re.IGNORECASE), (r"A\$", "Australian dollars", re.IGNORECASE),
|
||||
(r"US\$", "US dollars", re.IGNORECASE), ("€", "euros", 0), ("£", "pounds", 0), (r"\$", "dollars", 0),
|
||||
)
|
||||
|
||||
# Broad emoji / pictograph cleanup: most voice providers read emojis as awkward labels.
|
||||
_EMOJI_RE = re.compile(
|
||||
"["
|
||||
"\U0001F1E6-\U0001F1FF"
|
||||
"\U0001F300-\U0001F5FF"
|
||||
"\U0001F600-\U0001F64F"
|
||||
"\U0001F680-\U0001F6FF"
|
||||
"\U0001F700-\U0001F77F"
|
||||
"\U0001F780-\U0001F7FF"
|
||||
"\U0001F800-\U0001F8FF"
|
||||
"\U0001F900-\U0001F9FF"
|
||||
"\U0001FA00-\U0001FAFF"
|
||||
"☀-➿"
|
||||
"]+",
|
||||
"[\U0001F1E6-\U0001F1FF\U0001F300-\U0001F5FF\U0001F600-\U0001F64F\U0001F680-\U0001F6FF"
|
||||
"\U0001F700-\U0001F77F\U0001F780-\U0001F7FF\U0001F800-\U0001F8FF\U0001F900-\U0001F9FF"
|
||||
"\U0001FA00-\U0001FAFF☀-➿]+",
|
||||
flags=re.UNICODE,
|
||||
)
|
||||
_VARIATION_SELECTOR_RE = re.compile("[︎️]")
|
||||
@@ -70,34 +68,26 @@ def strip_markdown_for_tts(text: str) -> str:
|
||||
text = _MD_ITALIC_RE.sub(r"\1", text)
|
||||
text = _MD_UNDERSCORE_ITALIC_RE.sub(r"\1", text)
|
||||
text = _MD_STRIKE_RE.sub(r"\1", text)
|
||||
# Mark headings (do not just delete the marker): the whitespace pass folds a
|
||||
# heading into the sentence after it so speech says "Weather, it will be
|
||||
# sunny" instead of a clipped "Weather." then a separate sentence.
|
||||
# Mark headings (do not just delete the marker): see _HEAD.
|
||||
text = _MD_HEADING_LINE_RE.sub(lambda m: m.group(1).rstrip() + _HEAD, text)
|
||||
text = _MD_BLOCKQUOTE_RE.sub("", text)
|
||||
text = _MD_LIST_ITEM_RE.sub("", text)
|
||||
text = _MD_HR_RE.sub("", text)
|
||||
|
||||
# Pipe tables are terrible read aloud. Turn any leftover pipes into pauses
|
||||
# instead of letting a provider speak "vertical bar".
|
||||
# Leftover table pipes become pauses instead of a spoken "vertical bar".
|
||||
text = _MD_TABLE_PIPE_RE.sub("; ", text)
|
||||
return text
|
||||
|
||||
|
||||
def _normalize_temperature_ranges(text: str) -> str:
|
||||
# 11-17 degrees C -> "11 to 17 degrees Celsius" (en/em dash or hyphen).
|
||||
text = re.sub(
|
||||
r"(?<!\w)([-+\u2212]?\d+(?:\.\d+)?)\s*[\u2013\u2014-]\s*([-+\u2212]?\d+(?:\.\d+)?)\s*°\s*C\b",
|
||||
lambda m: f"{m.group(1).replace(chr(0x2212), '-')} to {m.group(2).replace(chr(0x2212), '-')} degrees Celsius",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(?<!\w)([-+\u2212]?\d+(?:\.\d+)?)\s*[\u2013\u2014-]\s*([-+\u2212]?\d+(?:\.\d+)?)\s*°\s*F\b",
|
||||
lambda m: f"{m.group(1).replace(chr(0x2212), '-')} to {m.group(2).replace(chr(0x2212), '-')} degrees Fahrenheit",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
"""``11-17°C`` -> ``11 to 17 degrees Celsius`` (en/em dash or hyphen; unicode minus normalized)."""
|
||||
for unit, word in (("C", "Celsius"), ("F", "Fahrenheit")):
|
||||
text = re.sub(
|
||||
r"(?<!\w)([-+\u2212]?\d+(?:\.\d+)?)\s*[\u2013\u2014-]\s*([-+\u2212]?\d+(?:\.\d+)?)\s*°\s*" + unit + r"\b",
|
||||
lambda m, w=word: f"{m.group(1).replace(chr(0x2212), '-')} to {m.group(2).replace(chr(0x2212), '-')} degrees {w}",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
@@ -112,44 +102,35 @@ def normalize_symbols_for_tts(text: str) -> str:
|
||||
text = text.replace("…", "...") # ellipsis
|
||||
text = _normalize_temperature_ranges(text)
|
||||
|
||||
# Temperatures with a number. Do this before generic degree handling.
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°\s*C\b", r"\1 degrees Celsius", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°\s*F\b", r"\1 degrees Fahrenheit", text, flags=re.IGNORECASE)
|
||||
# Bare units with no leading number ("measured in degrees C").
|
||||
text = re.sub(r"°\s*C\b", "degrees Celsius", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"°\s*F\b", "degrees Fahrenheit", text, flags=re.IGNORECASE)
|
||||
# Any remaining degree symbol (angles, stray cases).
|
||||
# Temperatures with a number first, then bare units ("measured in degrees C"),
|
||||
# then any remaining degree symbol (angles, stray cases).
|
||||
for unit, word in (("C", "Celsius"), ("F", "Fahrenheit")):
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°\s*" + unit + r"\b", r"\1 degrees " + word, text, flags=re.IGNORECASE)
|
||||
for unit, word in (("C", "Celsius"), ("F", "Fahrenheit")):
|
||||
text = re.sub(r"°\s*" + unit + r"\b", "degrees " + word, text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°", r"\1 degrees", text)
|
||||
text = text.replace("°", " degrees")
|
||||
|
||||
# Common weather/travel units.
|
||||
text = re.sub(r"(?<=\d)\s*km\s*/\s*h\b", " kilometres per hour", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*km/h\b", " kilometres per hour", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*mm\b", " millimetres", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*cm\b", " centimetres", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*m\b", " metres", text, flags=re.IGNORECASE)
|
||||
for pattern, word in _UNIT_WORDS:
|
||||
text = re.sub(r"(?<=\d)\s*" + pattern + r"\b", " " + word, text, flags=re.IGNORECASE)
|
||||
|
||||
# Numeric rates only ("5/month" -> "5 per month"). Requiring digit-then-letter
|
||||
# keeps "and/or", "N/A", "TCP/IP" and dates like "2026/06" intact.
|
||||
text = re.sub(r"(?<=\d)\s*/\s*(?=[A-Za-z])", " per ", text)
|
||||
|
||||
# Money and percentages. The integer part must END in a digit so a trailing
|
||||
# comma ("A$50, ...") is not swallowed into the spoken amount.
|
||||
text = re.sub(r"NZ\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 New Zealand dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"A\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 Australian dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"US\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 US dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"€\s*([\d,]*\d(?:\.\d+)?)", r"\1 euros", text)
|
||||
text = re.sub(r"£\s*([\d,]*\d(?:\.\d+)?)", r"\1 pounds", text)
|
||||
text = re.sub(r"\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 dollars", text)
|
||||
# Money and percentages. The integer part must END in a digit so a trailing
|
||||
# comma ("A$50, ...") is not swallowed into the spoken amount. Prefixed
|
||||
# currencies run first so "$" doesn't eat "NZ$".
|
||||
for symbol, word, flags in _CURRENCY_WORDS:
|
||||
text = re.sub(symbol + r"\s*([\d,]*\d(?:\.\d+)?)", r"\1 " + word, text, flags=flags)
|
||||
text = re.sub(r"(?<=\d)\s*%", " percent", text)
|
||||
|
||||
# Operators and separators that commonly leak from formatted answers.
|
||||
text = text.replace("&", " and ")
|
||||
text = re.sub("[•◦▪▫]", " ", text) # bullet glyphs
|
||||
text = text.replace("→", " to ") # ->
|
||||
text = text.replace("⇒", " to ") # =>
|
||||
text = text.replace("≈", " about ") # almost equal
|
||||
text = text.replace("~", " about ")
|
||||
for symbol, word in (("→", " to "), ("⇒", " to "), ("≈", " about "), ("~", " about ")):
|
||||
text = text.replace(symbol, word)
|
||||
|
||||
text = _VARIATION_SELECTOR_RE.sub("", text)
|
||||
text = _EMOJI_RE.sub("", text)
|
||||
@@ -159,10 +140,8 @@ def normalize_symbols_for_tts(text: str) -> str:
|
||||
def smooth_whitespace_for_tts(text: str) -> str:
|
||||
"""Collapse visual formatting into calm spoken paragraphs.
|
||||
|
||||
A former heading line (marked with the _HEAD sentinel) folds into the next
|
||||
content line as a spoken lead-in: "Weather" + "It will be sunny" becomes
|
||||
"Weather, It will be sunny." A heading with no content after it becomes its
|
||||
own short sentence.
|
||||
A _HEAD-marked heading folds into the next content line as a lead-in ("Weather,
|
||||
It will be sunny."); a heading with no content after it becomes its own sentence.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
@@ -182,8 +161,7 @@ def smooth_whitespace_for_tts(text: str) -> str:
|
||||
is_heading = raw_line.rstrip().endswith(_HEAD)
|
||||
line = raw_line.replace(_HEAD, "").strip()
|
||||
if not line:
|
||||
# Hold a pending heading across blank lines so it still folds into
|
||||
# the next real content line; otherwise just collapse the blank.
|
||||
# Hold a pending heading across blank lines so it still folds into the next content line.
|
||||
if pending_heading is None and lines and lines[-1] != "":
|
||||
lines.append("")
|
||||
continue
|
||||
@@ -209,44 +187,30 @@ def smooth_whitespace_for_tts(text: str) -> str:
|
||||
return text.strip()
|
||||
|
||||
|
||||
# Reasoning blocks: models with ``/reasoning show`` enabled emit
|
||||
# ``<think>...</think>`` blocks in the final assistant message. Users want to
|
||||
# SEE reasoning, not hear it read aloud (#34213).
|
||||
# ``/reasoning show`` emits ``<think>...</think>`` in the final message: users want to
|
||||
# SEE reasoning, not hear it. An unterminated block (streaming cut-off) is also silenced.
|
||||
_THINK_BLOCK_RE = re.compile(r"<think[\s>].*?</think>", flags=re.DOTALL | re.IGNORECASE)
|
||||
# An unterminated block (streaming cut-off) should still not be spoken.
|
||||
_THINK_BLOCK_OPEN_RE = re.compile(r"<think[\s>].*\Z", flags=re.DOTALL | re.IGNORECASE)
|
||||
|
||||
# Turn-end file-mutation verifier footer appended by run_agent.py
|
||||
# (``_format_file_mutation_failure_footer``). It's a UI affordance — reading
|
||||
# "warning file mutation verifier, 2 files were NOT modified..." aloud is
|
||||
# noise (#40772). The footer is a ``⚠️ File-mutation verifier:`` header line
|
||||
# followed by indented ``•`` bullet lines; strip the whole block.
|
||||
_VERIFIER_FOOTER_RE = re.compile(
|
||||
r"^\s*⚠️?\s*File-mutation verifier:.*(?:\n[ \t]+•.*)*",
|
||||
flags=re.MULTILINE,
|
||||
)
|
||||
# run_agent.py's turn-end file-mutation verifier footer (a ``⚠️ File-mutation verifier:``
|
||||
# header line plus indented ``•`` bullets) is a UI affordance, not speech.
|
||||
_VERIFIER_FOOTER_RE = re.compile(r"^\s*⚠️?\s*File-mutation verifier:.*(?:\n[ \t]+•.*)*", flags=re.MULTILINE)
|
||||
|
||||
|
||||
def strip_nonspoken_blocks(text: str) -> str:
|
||||
"""Remove blocks that must never reach a speech provider.
|
||||
|
||||
Currently: ``<think>`` reasoning blocks and the end-of-turn
|
||||
file-mutation verifier footer.
|
||||
"""
|
||||
"""Remove ``<think>`` reasoning blocks and the file-mutation verifier footer."""
|
||||
if not text:
|
||||
return ""
|
||||
text = _THINK_BLOCK_RE.sub(" ", text)
|
||||
text = _THINK_BLOCK_OPEN_RE.sub(" ", text)
|
||||
text = _VERIFIER_FOOTER_RE.sub(" ", text)
|
||||
for pattern in (_THINK_BLOCK_RE, _THINK_BLOCK_OPEN_RE, _VERIFIER_FOOTER_RE):
|
||||
text = pattern.sub(" ", text)
|
||||
return text
|
||||
|
||||
|
||||
def flatten_newlines_for_payload(text: str) -> str:
|
||||
"""Collapse newlines into sentence breaks for single-line TTS payloads.
|
||||
|
||||
Some OpenAI-compatible backends (e.g. Kokoro) truncate synthesis at the
|
||||
first newline (#9004). The smoothing pass already terminates each line
|
||||
with punctuation, so newlines can safely become plain spaces.
|
||||
Some OpenAI-compatible backends (e.g. Kokoro) truncate at the first newline; the
|
||||
smoothing pass already terminates each line with punctuation, so this is safe.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
@@ -259,28 +223,20 @@ def flatten_newlines_for_payload(text: str) -> str:
|
||||
|
||||
|
||||
def prepare_spoken_text(text: str, max_chars: int | None = 4000) -> str:
|
||||
"""Return a TTS-friendly script from assistant text.
|
||||
"""Return a TTS-friendly script from assistant text (deterministic cleanup, not a rewrite).
|
||||
|
||||
Deterministic cleanup, not a semantic rewrite: it removes ``<think>``
|
||||
reasoning blocks and the file-mutation verifier footer, removes Markdown,
|
||||
expands common symbols such as a degree-Celsius sign to "degrees Celsius",
|
||||
turns visual line formatting into speakable sentence pauses, and flattens
|
||||
the result to a single line so newline-sensitive providers (Kokoro) speak
|
||||
the whole script.
|
||||
Pipeline: non-spoken blocks > Markdown > symbols/units > line formatting into
|
||||
sentence pauses > single line (for newline-sensitive providers), then ``max_chars``.
|
||||
"""
|
||||
spoken = strip_nonspoken_blocks(text)
|
||||
spoken = strip_markdown_for_tts(spoken)
|
||||
spoken = normalize_symbols_for_tts(spoken)
|
||||
spoken = smooth_whitespace_for_tts(spoken)
|
||||
spoken = flatten_newlines_for_payload(spoken)
|
||||
spoken = text
|
||||
for step in (strip_nonspoken_blocks, strip_markdown_for_tts, normalize_symbols_for_tts,
|
||||
smooth_whitespace_for_tts, flatten_newlines_for_payload):
|
||||
spoken = step(spoken)
|
||||
if max_chars is not None and max_chars > 0 and len(spoken) > max_chars:
|
||||
spoken = spoken[:max_chars].rstrip()
|
||||
return spoken
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# Speech text cleanup (shared by voice-mode streaming and gateway auto-TTS)
|
||||
# ===========================================================================
|
||||
# Legacy regex fallback, only used if the shared normalizer raises.
|
||||
_LEGACY_TTS_STRIP_STEPS = (
|
||||
(re.compile(r'<think[\s>].*?</think>', flags=re.DOTALL), ' '),
|
||||
@@ -300,14 +256,7 @@ _LEGACY_TTS_STRIP_STEPS = (
|
||||
|
||||
|
||||
def _strip_markdown_for_tts(text: str) -> str:
|
||||
"""Prepare text for speech via the shared cleaner in tts_text_normalize.
|
||||
|
||||
One cleaner for every TTS path (tool, gateway auto-TTS, voice-mode
|
||||
streaming, web dashboard): strips <think> blocks, the verifier footer,
|
||||
markdown and emoji; expands units; flattens newlines so newline-sensitive
|
||||
providers (Kokoro) speak the whole script. Falls back to the legacy regex
|
||||
pipeline if the normalizer ever fails.
|
||||
"""
|
||||
"""``prepare_spoken_text`` without a length cap; falls back to the legacy regex pipeline if it raises."""
|
||||
try:
|
||||
return prepare_spoken_text(text, max_chars=None)
|
||||
except Exception:
|
||||
|
||||
Reference in New Issue
Block a user