fix(gateway/weixin): route voice messages through Hermes STT instead of trusting Tencent Cloud's text

When WeChat (Weixin) returns a voice_item.text (Tencent Cloud's STT),
Hermes previously trusted that text as the user-visible message body and
skipped downloading the raw audio. For non-Chinese audio that text is
garbage — the original report was a Russian voice message that came
back as English phonemes — and the user sees nonsense as their own
message. International users on the WeChat gateway effectively can't
use voice.

Two short-circuits in gateway/platforms/weixin.py caused this:

  - _download_voice() returned None whenever voice_item.text was set,
    so the raw SILK/Opus audio was never fetched.
  - _extract_text() returned voice_item.text verbatim as the body,
    so even if the audio had been downloaded, the central STT
    pipeline in gateway/run.py never had a chance to replace it.

Fix both: always download the raw audio (the central pipeline picks
it up via _collect_media()), and skip voice items in _extract_text()
so the body comes from Hermes' own mlx-whisper / whisper.cpp /
faster-whisper transcription instead of Tencent's. Behavior is
unchanged when voice_item.text is absent (the original happy path
where audio was already being downloaded).

Tests:
  - 5 new tests in TestWeixinVoiceAlwaysDownloaded covering both
    functions, the _collect_media integration path, and a
    regression guard for the text-item path.
  - 74/74 in tests/gateway/test_weixin.py pass.
This commit is contained in:
Kewe63 2026-06-16 11:11:47 +03:00 committed by Teknium
parent 1c30c57f11
commit cb1798b43c
2 changed files with 151 additions and 5 deletions

View file

@ -973,9 +973,13 @@ def _extract_text(item_list: List[Dict[str, Any]]) -> str:
return text
for item in item_list:
if item.get("type") == ITEM_VOICE:
voice_text = str((item.get("voice_item") or {}).get("text") or "")
if voice_text:
return voice_text
# #27300: Tencent Cloud's `voice_item.text` is their STT output,
# which is wrong for any non-Chinese audio (the original report
# was a Russian voice message that came back as English
# gibberish). Return empty so the central STT pipeline in
# ``gateway/run.py`` produces the body from the downloaded
# audio instead.
continue
return ""
@ -1659,8 +1663,13 @@ class WeixinAdapter(BasePlatformAdapter):
async def _download_voice(self, item: Dict[str, Any]) -> Optional[str]:
voice_item = item.get("voice_item") or {}
media = voice_item.get("media") or {}
if voice_item.get("text"):
return None
# #27300: previously short-circuited when ``voice_item.text`` was set
# on the assumption that Tencent Cloud's STT was good enough.
# For non-Chinese audio that text is garbage (e.g. a Russian
# message comes back as English phonemes) — we must always
# download the raw audio so ``gateway/run.py``'s central STT
# pipeline can re-transcribe with the user's configured
# mlx-whisper / whisper.cpp / faster-whisper backend.
try:
data = await _download_and_decrypt_media(
self._poll_session,

View file

@ -1205,3 +1205,140 @@ class TestWeixinApiTimeout:
)
)
assert result == {"ret": 0, "msgs": [], "get_updates_buf": "buf-123"}
class TestWeixinVoiceAlwaysDownloaded:
"""Regression tests for #27300: when WeChat (Weixin) returns a
``voice_item.text`` (Tencent Cloud's STT) we must still download
the raw audio and route it through Hermes' own STT pipeline.
Non-Chinese users currently see garbled transcriptions because the
existing code short-circuits in two places: ``_download_voice``
returns ``None`` whenever Tencent provided *any* text (even
incorrect), and ``_extract_text`` returns that text as the message
body. The fix is to always download and never return Tencent's
text the central STT pipeline in ``gateway/run.py`` produces
the actual body from the downloaded audio.
"""
def _make_voice_item(self, text: str = "") -> dict:
"""Build a minimal voice item with media + optional Tencent text."""
return {
"type": weixin.ITEM_VOICE,
"voice_item": {
"text": text,
"media": {
"encrypt_query_param": "q",
"aes_key": "a" * 32,
"full_url": "https://example.invalid/voice.silk",
},
},
}
@pytest.mark.asyncio
async def test_download_voice_returns_path_when_tencent_text_set(self, tmp_path, monkeypatch):
"""#27300 PRIMARY: ``_download_voice`` must not short-circuit on
``voice_item.text``. The audio is needed so Hermes' own STT can
re-transcribe when Tencent's text is in the wrong language.
"""
adapter = _make_adapter()
adapter._cdn_base_url = "https://example.invalid"
adapter._poll_session = Mock()
fake_audio_bytes = b"\\x00\\x01\\x02FAKE_SILK"
monkeypatch.setattr(weixin, "cache_audio_from_bytes",
lambda data, ext: str(tmp_path / f"voice.{ext.lstrip('.')}"))
async def _fake_download(session, *, cdn_base_url, encrypted_query_param,
aes_key_b64, full_url, timeout_seconds):
return fake_audio_bytes
monkeypatch.setattr(weixin, "_download_and_decrypt_media", _fake_download)
item = self._make_voice_item(text="garbled-tencent-transcript")
result = await adapter._download_voice(item)
# Currently broken: returns None when voice_item.text is set.
# After fix: returns a local path so the central STT pipeline
# can pick it up and re-transcribe.
assert result is not None, (
"_download_voice returned None even though raw audio is "
"available — Hermes' STT pipeline needs the audio to handle "
"non-Chinese voice messages (#27300)."
)
assert result.endswith(".silk")
def test_extract_text_does_not_return_tencent_voice_text(self):
"""#27300 SECONDARY: ``_extract_text`` must not return
``voice_item.text`` verbatim. That text is Tencent Cloud's
STT output, which is wrong for non-Chinese audio and the
whole reason #27300 was filed. Returning empty forces the
central STT pipeline's transcript to become the body.
"""
item_list = [self._make_voice_item(text="garbled-tencent-transcript")]
result = weixin._extract_text(item_list)
# Currently broken: returns "garbled-tencent-transcript".
# After fix: returns "" (empty string) so the central pipeline
# transcript replaces it as the user-visible body.
assert result != "garbled-tencent-transcript", (
"_extract_text returned Tencent's text directly — for "
"non-Chinese audio this is garbage; the central STT "
"pipeline's transcript should be the body (#27300)."
)
def test_extract_text_voice_only_returns_empty(self):
"""When the only item is a voice attachment (no text item),
``_extract_text`` should return empty so the central STT
pipeline's transcript becomes the body. Currently returns
Tencent's text which is what the bug is about.
"""
item_list = [self._make_voice_item(text="какой-то текст")]
result = weixin._extract_text(item_list)
assert result == "", (
"Voice-only message: _extract_text should return empty so "
"the central STT pipeline output replaces it as the body."
)
def test_extract_text_still_returns_text_for_text_items(self):
"""Sanity: the fix must not regress the text-item path. A plain
text message should still produce its text body.
"""
item_list = [{
"type": weixin.ITEM_TEXT,
"text_item": {"text": "hello world"},
}]
assert weixin._extract_text(item_list) == "hello world"
@pytest.mark.asyncio
async def test_collect_media_includes_voice_when_tencent_text_set(self, tmp_path, monkeypatch):
"""#27300 INTEGRATION: ``_collect_media`` should add a ``.silk``
path to ``media_paths`` even when Tencent returned text, so the
central STT pipeline can re-transcribe. Currently the
short-circuit in ``_download_voice`` means the audio is never
downloaded, and the message body is whatever Tencent wrote
(garbled for non-Chinese audio).
"""
adapter = _make_adapter()
adapter._cdn_base_url = "https://example.invalid"
adapter._poll_session = Mock()
monkeypatch.setattr(weixin, "cache_audio_from_bytes",
lambda data, ext: str(tmp_path / f"voice.{ext.lstrip('.')}"))
async def _fake_download(session, *, cdn_base_url, encrypted_query_param,
aes_key_b64, full_url, timeout_seconds):
return b"\\x00FAKE"
monkeypatch.setattr(weixin, "_download_and_decrypt_media", _fake_download)
media_paths: list = []
media_types: list = []
item = self._make_voice_item(text="какой-то текст")
await adapter._collect_media(item, media_paths, media_types)
assert len(media_paths) == 1, (
"_collect_media dropped the voice attachment because "
"voice_item.text was set — Hermes' STT never gets a "
"chance to re-transcribe (#27300)."
)
assert media_types == ["audio/silk"]