mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
feat(tts): normalize spoken text (units, symbols, markdown) before synthesis
Auto-TTS previously fed raw chat Markdown and compact symbols straight to the speech provider, so units were read as stray letters and headings or bullets ran together. This routes spoken text through a new normalizer, tools/tts_text_normalize.prepare_spoken_text, that expands units (for example a temperature written with the degree symbol becomes "degrees Celsius") and flattens Markdown into a transcript-like script with sentence pauses. The normalizer is best-effort: if it ever fails the code falls back to the previous markdown-strip behavior, so auto-TTS keeps working. The Telegram voice caption uses the same normalized text. Includes a unit test.
This commit is contained in:
parent
fdc24a975e
commit
a9a9005f31
4 changed files with 319 additions and 6 deletions
|
|
@ -3831,11 +3831,19 @@ class BasePlatformAdapter(ABC):
|
|||
return await self.send(chat_id=chat_id, content=text, reply_to=reply_to, metadata=metadata)
|
||||
|
||||
def prepare_tts_text(self, text: str) -> str:
|
||||
"""Prepare text for TTS. Override to filter tool output, code, etc.
|
||||
"""Prepare a spoken script for TTS.
|
||||
|
||||
Default strips markdown formatting and truncates to 4000 chars.
|
||||
Auto-TTS should not feed raw chat Markdown or compact symbols to the
|
||||
speech provider. It should receive a transcript-like script: headings
|
||||
and bullets flattened into sentence pauses, and units like ``°C``
|
||||
expanded to words such as ``degrees Celsius``.
|
||||
"""
|
||||
return re.sub(r'[*_`#\[\]()]', '', text)[:4000].strip()
|
||||
try:
|
||||
from tools.tts_text_normalize import prepare_spoken_text
|
||||
return prepare_spoken_text(text, max_chars=4000)
|
||||
except Exception:
|
||||
# Keep auto-TTS best-effort if the normalizer ever fails.
|
||||
return re.sub(r'[*_`#\[\]()]', '', text)[:4000].strip()
|
||||
|
||||
async def play_tts(
|
||||
self,
|
||||
|
|
@ -5537,6 +5545,7 @@ class BasePlatformAdapter(ABC):
|
|||
# an explicit ``/voice on|tts`` opt-in OR when ``voice.auto_tts`` is
|
||||
# True globally and no ``/voice off`` has been issued.
|
||||
_tts_path = None
|
||||
_tts_speech_text = None
|
||||
if (self._should_auto_tts_for_chat(event.source.chat_id)
|
||||
and event.message_type == MessageType.VOICE
|
||||
and text_content
|
||||
|
|
@ -5548,6 +5557,7 @@ class BasePlatformAdapter(ABC):
|
|||
speech_text = self.prepare_tts_text(text_content)
|
||||
if not speech_text:
|
||||
raise ValueError("Empty text after markdown cleanup")
|
||||
_tts_speech_text = speech_text
|
||||
tts_result_str = await asyncio.to_thread(
|
||||
text_to_speech_tool, text=speech_text
|
||||
)
|
||||
|
|
@ -5561,12 +5571,13 @@ class BasePlatformAdapter(ABC):
|
|||
if _tts_path and Path(_tts_path).exists():
|
||||
try:
|
||||
telegram_tts_caption = None
|
||||
caption_text = _tts_speech_text or self.prepare_tts_text(text_content)
|
||||
if (
|
||||
self.platform == Platform.TELEGRAM
|
||||
and text_content
|
||||
and text_content[:1024] == text_content
|
||||
and caption_text
|
||||
and caption_text[:1024] == caption_text
|
||||
):
|
||||
telegram_tts_caption = text_content
|
||||
telegram_tts_caption = caption_text
|
||||
tts_result = await self.play_tts(
|
||||
chat_id=event.source.chat_id,
|
||||
audio_path=_tts_path,
|
||||
|
|
|
|||
70
tests/tools/test_tts_text_normalize.py
Normal file
70
tests/tools/test_tts_text_normalize.py
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
from gateway.config import Platform, PlatformConfig
|
||||
from gateway.platforms.base import BasePlatformAdapter
|
||||
from tools.tts_text_normalize import prepare_spoken_text
|
||||
|
||||
|
||||
class _DummyAdapter(BasePlatformAdapter):
|
||||
def __init__(self):
|
||||
super().__init__(PlatformConfig(enabled=True, token="test"), Platform.TELEGRAM)
|
||||
|
||||
async def connect(self):
|
||||
return True
|
||||
|
||||
async def disconnect(self):
|
||||
pass
|
||||
|
||||
async def send(self, chat_id, content, **kwargs):
|
||||
raise AssertionError("not used")
|
||||
|
||||
async def get_chat_info(self, chat_id):
|
||||
return {"id": chat_id, "type": "dm"}
|
||||
|
||||
|
||||
def test_prepare_spoken_text_expands_celsius_and_weather_units():
|
||||
raw = """## Christchurch today\n\n- **Now:** about **14°C**, feels like **14°C**\n- **Wind:** 9 km/h\n- **Rain:** 1.3 mm\n- **Range:** 11\u201317°C\n"""
|
||||
|
||||
spoken = prepare_spoken_text(raw)
|
||||
|
||||
assert "##" not in spoken
|
||||
assert "**" not in spoken
|
||||
assert "14 degrees Celsius" in spoken
|
||||
assert "11 to 17 degrees Celsius" in spoken
|
||||
assert "9 kilometres per hour" in spoken
|
||||
assert "1.3 millimetres" in spoken
|
||||
assert "°C" not in spoken
|
||||
assert "km/h" not in spoken
|
||||
|
||||
|
||||
def test_prepare_spoken_text_flattens_visual_formatting_for_tts():
|
||||
raw = """## Short answer\n\n- [link text](https://example.com) → NZ$120 & 80% likely\n- `inline code` should not keep backticks\n"""
|
||||
|
||||
spoken = prepare_spoken_text(raw)
|
||||
|
||||
assert "Short answer, link text to 120 New Zealand dollars and 80 percent likely" in spoken
|
||||
assert "inline code should not keep backticks" in spoken
|
||||
assert "https://" not in spoken
|
||||
assert "`" not in spoken
|
||||
assert "→" not in spoken
|
||||
assert "&" not in spoken
|
||||
|
||||
|
||||
def test_gateway_auto_tts_preparation_uses_spoken_normalizer():
|
||||
adapter = _DummyAdapter()
|
||||
|
||||
spoken = adapter.prepare_tts_text("## Weather\n- Now: 14°C, wind 9 km/h")
|
||||
|
||||
assert spoken == "Weather, Now: 14 degrees Celsius, wind 9 kilometres per hour."
|
||||
|
||||
|
||||
def test_prepare_spoken_text_polish_edge_cases():
|
||||
# Heading folds into the next sentence as a lead-in, not a bare label.
|
||||
assert prepare_spoken_text("## Weather\nIt will be sunny") == "Weather, It will be sunny."
|
||||
# Bare degree unit (no leading number) still expands.
|
||||
assert "degrees Celsius" in prepare_spoken_text("measured in °C")
|
||||
# Trailing comma is not swallowed into the amount.
|
||||
assert "300 US dollars" in prepare_spoken_text("US$300, next")
|
||||
# Real numeric rates expand, but and/or, N/A, IDs and dates are left intact.
|
||||
assert "5 dollars per month" in prepare_spoken_text("$5/month")
|
||||
assert "and/or" in prepare_spoken_text("choose and/or option")
|
||||
assert "N/A" in prepare_spoken_text("status N/A here")
|
||||
assert "2026/06/02" in prepare_spoken_text("due 2026/06/02 ok")
|
||||
224
tools/tts_text_normalize.py
Normal file
224
tools/tts_text_normalize.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""Utilities for preparing assistant text for speech synthesis.
|
||||
|
||||
The TTS provider should receive a spoken script, not raw chat Markdown. This
|
||||
module centralises the lightweight, deterministic cleanup used by explicit TTS
|
||||
calls and gateway auto-TTS replies.
|
||||
|
||||
Non-ASCII characters are written as escapes on purpose so the file stays free of
|
||||
invisible/look-alike glyphs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
|
||||
# Sentinel appended to former heading lines so smooth_whitespace_for_tts can
|
||||
# fold a heading into the sentence that follows it ("Weather, it will be sunny")
|
||||
# rather than leaving a bare "Weather." label that reads abruptly aloud.
|
||||
_HEAD = "\x00"
|
||||
|
||||
_MD_CODE_BLOCK_RE = re.compile(r"```[\s\S]*?```")
|
||||
_MD_LINK_RE = re.compile(r"\[([^\]]+)\]\((?:[^()]|\([^)]*\))*\)")
|
||||
_MD_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\((?:[^()]|\([^)]*\))*\)")
|
||||
_MD_INLINE_CODE_RE = re.compile(r"`([^`]+)`")
|
||||
_MD_BOLD_RE = re.compile(r"\*\*(.+?)\*\*", flags=re.DOTALL)
|
||||
_MD_UNDERSCORE_BOLD_RE = re.compile(r"__(.+?)__", flags=re.DOTALL)
|
||||
_MD_ITALIC_RE = re.compile(r"(?<!\*)\*(?!\*)(.+?)(?<!\*)\*(?!\*)", flags=re.DOTALL)
|
||||
_MD_UNDERSCORE_ITALIC_RE = re.compile(r"(?<!_)_(?!_)(.+?)(?<!_)_(?!_)", flags=re.DOTALL)
|
||||
_MD_STRIKE_RE = re.compile(r"~~(.+?)~~", flags=re.DOTALL)
|
||||
_MD_HEADING_LINE_RE = re.compile(r"^[ \t]{0,3}#{1,6}[ \t]+(.+?)[ \t]*#*[ \t]*$", flags=re.MULTILINE)
|
||||
_MD_BLOCKQUOTE_RE = re.compile(r"^\s*>\s?", flags=re.MULTILINE)
|
||||
_MD_LIST_ITEM_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+", flags=re.MULTILINE)
|
||||
_MD_HR_RE = re.compile(r"^\s*[-*_]{3,}\s*$", flags=re.MULTILINE)
|
||||
_MD_TABLE_PIPE_RE = re.compile(r"\s*\|\s*")
|
||||
_URL_RE = re.compile(r"https?://\S+")
|
||||
|
||||
# Broad emoji / pictograph cleanup. Voice providers vary a lot here; most read
|
||||
# emojis as awkward labels, so keep the speech script calm and literal.
|
||||
_EMOJI_RE = re.compile(
|
||||
"["
|
||||
"\U0001F1E6-\U0001F1FF"
|
||||
"\U0001F300-\U0001F5FF"
|
||||
"\U0001F600-\U0001F64F"
|
||||
"\U0001F680-\U0001F6FF"
|
||||
"\U0001F700-\U0001F77F"
|
||||
"\U0001F780-\U0001F7FF"
|
||||
"\U0001F800-\U0001F8FF"
|
||||
"\U0001F900-\U0001F9FF"
|
||||
"\U0001FA00-\U0001FAFF"
|
||||
"☀-➿"
|
||||
"]+",
|
||||
flags=re.UNICODE,
|
||||
)
|
||||
_VARIATION_SELECTOR_RE = re.compile("[︎️]")
|
||||
|
||||
|
||||
def strip_markdown_for_tts(text: str) -> str:
|
||||
"""Strip Markdown/Telegram formatting while preserving readable words."""
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
text = html.unescape(str(text))
|
||||
text = _MD_CODE_BLOCK_RE.sub(" ", text)
|
||||
text = _MD_IMAGE_RE.sub(lambda m: f" {m.group(1)} " if m.group(1) else " ", text)
|
||||
text = _MD_LINK_RE.sub(r"\1", text)
|
||||
text = _URL_RE.sub("", text)
|
||||
text = _MD_INLINE_CODE_RE.sub(r"\1", text)
|
||||
text = _MD_BOLD_RE.sub(r"\1", text)
|
||||
text = _MD_UNDERSCORE_BOLD_RE.sub(r"\1", text)
|
||||
text = _MD_ITALIC_RE.sub(r"\1", text)
|
||||
text = _MD_UNDERSCORE_ITALIC_RE.sub(r"\1", text)
|
||||
text = _MD_STRIKE_RE.sub(r"\1", text)
|
||||
# Mark headings (do not just delete the marker): the whitespace pass folds a
|
||||
# heading into the sentence after it so speech says "Weather, it will be
|
||||
# sunny" instead of a clipped "Weather." then a separate sentence.
|
||||
text = _MD_HEADING_LINE_RE.sub(lambda m: m.group(1).rstrip() + _HEAD, text)
|
||||
text = _MD_BLOCKQUOTE_RE.sub("", text)
|
||||
text = _MD_LIST_ITEM_RE.sub("", text)
|
||||
text = _MD_HR_RE.sub("", text)
|
||||
|
||||
# Pipe tables are terrible read aloud. Turn any leftover pipes into pauses
|
||||
# instead of letting a provider speak "vertical bar".
|
||||
text = _MD_TABLE_PIPE_RE.sub("; ", text)
|
||||
return text
|
||||
|
||||
|
||||
def _normalize_temperature_ranges(text: str) -> str:
|
||||
# 11-17 degrees C -> "11 to 17 degrees Celsius" (en/em dash or hyphen).
|
||||
text = re.sub(
|
||||
r"(?<!\w)([-+\u2212]?\d+(?:\.\d+)?)\s*[\u2013\u2014-]\s*([-+\u2212]?\d+(?:\.\d+)?)\s*°\s*C\b",
|
||||
lambda m: f"{m.group(1).replace(chr(0x2212), '-')} to {m.group(2).replace(chr(0x2212), '-')} degrees Celsius",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(?<!\w)([-+\u2212]?\d+(?:\.\d+)?)\s*[\u2013\u2014-]\s*([-+\u2212]?\d+(?:\.\d+)?)\s*°\s*F\b",
|
||||
lambda m: f"{m.group(1).replace(chr(0x2212), '-')} to {m.group(2).replace(chr(0x2212), '-')} degrees Fahrenheit",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
def normalize_symbols_for_tts(text: str) -> str:
|
||||
"""Expand common symbols/shorthand into words a TTS engine reads well."""
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
text = str(text)
|
||||
text = re.sub("[ ]", " ", text) # non-breaking / thin spaces
|
||||
text = text.replace("\u2212", "-") # minus sign
|
||||
text = text.replace("…", "...") # ellipsis
|
||||
text = _normalize_temperature_ranges(text)
|
||||
|
||||
# Temperatures with a number. Do this before generic degree handling.
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°\s*C\b", r"\1 degrees Celsius", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°\s*F\b", r"\1 degrees Fahrenheit", text, flags=re.IGNORECASE)
|
||||
# Bare units with no leading number ("measured in degrees C").
|
||||
text = re.sub(r"°\s*C\b", "degrees Celsius", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"°\s*F\b", "degrees Fahrenheit", text, flags=re.IGNORECASE)
|
||||
# Any remaining degree symbol (angles, stray cases).
|
||||
text = re.sub(r"(?<!\w)([-+]?\d+(?:\.\d+)?)\s*°", r"\1 degrees", text)
|
||||
text = text.replace("°", " degrees")
|
||||
|
||||
# Common weather/travel units.
|
||||
text = re.sub(r"(?<=\d)\s*km\s*/\s*h\b", " kilometres per hour", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*km/h\b", " kilometres per hour", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*mm\b", " millimetres", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*cm\b", " centimetres", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"(?<=\d)\s*m\b", " metres", text, flags=re.IGNORECASE)
|
||||
|
||||
# Numeric rates only ("5/month" -> "5 per month"). Requiring digit-then-letter
|
||||
# keeps "and/or", "N/A", "TCP/IP" and dates like "2026/06" intact.
|
||||
text = re.sub(r"(?<=\d)\s*/\s*(?=[A-Za-z])", " per ", text)
|
||||
|
||||
# Money and percentages. The integer part must END in a digit so a trailing
|
||||
# comma ("A$50, ...") is not swallowed into the spoken amount.
|
||||
text = re.sub(r"NZ\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 New Zealand dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"A\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 Australian dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"US\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 US dollars", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"€\s*([\d,]*\d(?:\.\d+)?)", r"\1 euros", text)
|
||||
text = re.sub(r"£\s*([\d,]*\d(?:\.\d+)?)", r"\1 pounds", text)
|
||||
text = re.sub(r"\$\s*([\d,]*\d(?:\.\d+)?)", r"\1 dollars", text)
|
||||
text = re.sub(r"(?<=\d)\s*%", " percent", text)
|
||||
|
||||
# Operators and separators that commonly leak from formatted answers.
|
||||
text = text.replace("&", " and ")
|
||||
text = re.sub("[•◦▪▫]", " ", text) # bullet glyphs
|
||||
text = text.replace("→", " to ") # ->
|
||||
text = text.replace("⇒", " to ") # =>
|
||||
text = text.replace("≈", " about ") # almost equal
|
||||
text = text.replace("~", " about ")
|
||||
|
||||
text = _VARIATION_SELECTOR_RE.sub("", text)
|
||||
text = _EMOJI_RE.sub("", text)
|
||||
return text
|
||||
|
||||
|
||||
def smooth_whitespace_for_tts(text: str) -> str:
|
||||
"""Collapse visual formatting into calm spoken paragraphs.
|
||||
|
||||
A former heading line (marked with the _HEAD sentinel) folds into the next
|
||||
content line as a spoken lead-in: "Weather" + "It will be sunny" becomes
|
||||
"Weather, It will be sunny." A heading with no content after it becomes its
|
||||
own short sentence.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
raw_lines = text.splitlines()
|
||||
add_sentence_pauses = sum(1 for raw_line in raw_lines if raw_line.replace(_HEAD, "").strip()) > 1
|
||||
lines: list[str] = []
|
||||
pending_heading: str | None = None
|
||||
|
||||
def flush_pending() -> None:
|
||||
nonlocal pending_heading
|
||||
if pending_heading is not None:
|
||||
lines.append(pending_heading.rstrip(".:;,") + ".")
|
||||
pending_heading = None
|
||||
|
||||
for raw_line in raw_lines:
|
||||
is_heading = raw_line.rstrip().endswith(_HEAD)
|
||||
line = raw_line.replace(_HEAD, "").strip()
|
||||
if not line:
|
||||
# Hold a pending heading across blank lines so it still folds into
|
||||
# the next real content line; otherwise just collapse the blank.
|
||||
if pending_heading is None and lines and lines[-1] != "":
|
||||
lines.append("")
|
||||
continue
|
||||
if is_heading:
|
||||
flush_pending()
|
||||
pending_heading = line.rstrip(".:;,")
|
||||
continue
|
||||
if pending_heading is not None:
|
||||
line = f"{pending_heading.rstrip('.:;,')}, {line}"
|
||||
pending_heading = None
|
||||
if add_sentence_pauses and line[-1] not in ".!?;:":
|
||||
line += "."
|
||||
lines.append(line)
|
||||
|
||||
flush_pending()
|
||||
|
||||
text = "\n".join(lines)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
text = re.sub(r"[ \t]{2,}", " ", text)
|
||||
text = re.sub(r"\s+([,.;:!?])", r"\1", text)
|
||||
text = re.sub(r"([,.;:!?])([A-Za-z])", r"\1 \2", text)
|
||||
text = re.sub(r"\.{4,}", "...", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def prepare_spoken_text(text: str, max_chars: int | None = 4000) -> str:
|
||||
"""Return a TTS-friendly script from assistant text.
|
||||
|
||||
Deterministic cleanup, not a semantic rewrite: it removes Markdown, expands
|
||||
common symbols such as a degree-Celsius sign to "degrees Celsius", and turns
|
||||
visual line formatting into speakable sentence pauses.
|
||||
"""
|
||||
spoken = strip_markdown_for_tts(text)
|
||||
spoken = normalize_symbols_for_tts(spoken)
|
||||
spoken = smooth_whitespace_for_tts(spoken)
|
||||
if max_chars is not None and max_chars > 0 and len(spoken) > max_chars:
|
||||
spoken = spoken[:max_chars].rstrip()
|
||||
return spoken
|
||||
|
|
@ -2588,6 +2588,14 @@ def text_to_speech_tool(
|
|||
if not text or not text.strip():
|
||||
return tool_error("Text is required", success=False)
|
||||
|
||||
try:
|
||||
from tools.tts_text_normalize import prepare_spoken_text
|
||||
text = prepare_spoken_text(text, max_chars=None)
|
||||
except Exception:
|
||||
text = text.strip()
|
||||
if not text:
|
||||
return tool_error("Text is empty after TTS cleanup", success=False)
|
||||
|
||||
tts_config = _load_tts_config()
|
||||
provider = _get_provider(tts_config)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue