mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-29 18:46:59 +00:00
fix(gateway/weixin): route voice messages through Hermes STT instead of trusting Tencent Cloud's text
When WeChat (Weixin) returns a voice_item.text (Tencent Cloud's STT),
Hermes previously trusted that text as the user-visible message body and
skipped downloading the raw audio. For non-Chinese audio that text is
garbage — the original report was a Russian voice message that came
back as English phonemes — and the user sees nonsense as their own
message. International users on the WeChat gateway effectively can't
use voice.
Two short-circuits in gateway/platforms/weixin.py caused this:
- _download_voice() returned None whenever voice_item.text was set,
so the raw SILK/Opus audio was never fetched.
- _extract_text() returned voice_item.text verbatim as the body,
so even if the audio had been downloaded, the central STT
pipeline in gateway/run.py never had a chance to replace it.
Fix both: always download the raw audio (the central pipeline picks
it up via _collect_media()), and skip voice items in _extract_text()
so the body comes from Hermes' own mlx-whisper / whisper.cpp /
faster-whisper transcription instead of Tencent's. Behavior is
unchanged when voice_item.text is absent (the original happy path
where audio was already being downloaded).
Tests:
- 5 new tests in TestWeixinVoiceAlwaysDownloaded covering both
functions, the _collect_media integration path, and a
regression guard for the text-item path.
- 74/74 in tests/gateway/test_weixin.py pass.
This commit is contained in:
parent
1c30c57f11
commit
cb1798b43c
2 changed files with 151 additions and 5 deletions
|
|
@ -973,9 +973,13 @@ def _extract_text(item_list: List[Dict[str, Any]]) -> str:
|
|||
return text
|
||||
for item in item_list:
|
||||
if item.get("type") == ITEM_VOICE:
|
||||
voice_text = str((item.get("voice_item") or {}).get("text") or "")
|
||||
if voice_text:
|
||||
return voice_text
|
||||
# #27300: Tencent Cloud's `voice_item.text` is their STT output,
|
||||
# which is wrong for any non-Chinese audio (the original report
|
||||
# was a Russian voice message that came back as English
|
||||
# gibberish). Return empty so the central STT pipeline in
|
||||
# ``gateway/run.py`` produces the body from the downloaded
|
||||
# audio instead.
|
||||
continue
|
||||
return ""
|
||||
|
||||
|
||||
|
|
@ -1659,8 +1663,13 @@ class WeixinAdapter(BasePlatformAdapter):
|
|||
async def _download_voice(self, item: Dict[str, Any]) -> Optional[str]:
|
||||
voice_item = item.get("voice_item") or {}
|
||||
media = voice_item.get("media") or {}
|
||||
if voice_item.get("text"):
|
||||
return None
|
||||
# #27300: previously short-circuited when ``voice_item.text`` was set
|
||||
# on the assumption that Tencent Cloud's STT was good enough.
|
||||
# For non-Chinese audio that text is garbage (e.g. a Russian
|
||||
# message comes back as English phonemes) — we must always
|
||||
# download the raw audio so ``gateway/run.py``'s central STT
|
||||
# pipeline can re-transcribe with the user's configured
|
||||
# mlx-whisper / whisper.cpp / faster-whisper backend.
|
||||
try:
|
||||
data = await _download_and_decrypt_media(
|
||||
self._poll_session,
|
||||
|
|
|
|||
|
|
@ -1205,3 +1205,140 @@ class TestWeixinApiTimeout:
|
|||
)
|
||||
)
|
||||
assert result == {"ret": 0, "msgs": [], "get_updates_buf": "buf-123"}
|
||||
|
||||
|
||||
class TestWeixinVoiceAlwaysDownloaded:
|
||||
"""Regression tests for #27300: when WeChat (Weixin) returns a
|
||||
``voice_item.text`` (Tencent Cloud's STT) we must still download
|
||||
the raw audio and route it through Hermes' own STT pipeline.
|
||||
|
||||
Non-Chinese users currently see garbled transcriptions because the
|
||||
existing code short-circuits in two places: ``_download_voice``
|
||||
returns ``None`` whenever Tencent provided *any* text (even
|
||||
incorrect), and ``_extract_text`` returns that text as the message
|
||||
body. The fix is to always download and never return Tencent's
|
||||
text — the central STT pipeline in ``gateway/run.py`` produces
|
||||
the actual body from the downloaded audio.
|
||||
"""
|
||||
|
||||
def _make_voice_item(self, text: str = "") -> dict:
|
||||
"""Build a minimal voice item with media + optional Tencent text."""
|
||||
return {
|
||||
"type": weixin.ITEM_VOICE,
|
||||
"voice_item": {
|
||||
"text": text,
|
||||
"media": {
|
||||
"encrypt_query_param": "q",
|
||||
"aes_key": "a" * 32,
|
||||
"full_url": "https://example.invalid/voice.silk",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_voice_returns_path_when_tencent_text_set(self, tmp_path, monkeypatch):
|
||||
"""#27300 PRIMARY: ``_download_voice`` must not short-circuit on
|
||||
``voice_item.text``. The audio is needed so Hermes' own STT can
|
||||
re-transcribe when Tencent's text is in the wrong language.
|
||||
"""
|
||||
adapter = _make_adapter()
|
||||
adapter._cdn_base_url = "https://example.invalid"
|
||||
adapter._poll_session = Mock()
|
||||
|
||||
fake_audio_bytes = b"\\x00\\x01\\x02FAKE_SILK"
|
||||
monkeypatch.setattr(weixin, "cache_audio_from_bytes",
|
||||
lambda data, ext: str(tmp_path / f"voice.{ext.lstrip('.')}"))
|
||||
|
||||
async def _fake_download(session, *, cdn_base_url, encrypted_query_param,
|
||||
aes_key_b64, full_url, timeout_seconds):
|
||||
return fake_audio_bytes
|
||||
|
||||
monkeypatch.setattr(weixin, "_download_and_decrypt_media", _fake_download)
|
||||
|
||||
item = self._make_voice_item(text="garbled-tencent-transcript")
|
||||
result = await adapter._download_voice(item)
|
||||
|
||||
# Currently broken: returns None when voice_item.text is set.
|
||||
# After fix: returns a local path so the central STT pipeline
|
||||
# can pick it up and re-transcribe.
|
||||
assert result is not None, (
|
||||
"_download_voice returned None even though raw audio is "
|
||||
"available — Hermes' STT pipeline needs the audio to handle "
|
||||
"non-Chinese voice messages (#27300)."
|
||||
)
|
||||
assert result.endswith(".silk")
|
||||
|
||||
def test_extract_text_does_not_return_tencent_voice_text(self):
|
||||
"""#27300 SECONDARY: ``_extract_text`` must not return
|
||||
``voice_item.text`` verbatim. That text is Tencent Cloud's
|
||||
STT output, which is wrong for non-Chinese audio and the
|
||||
whole reason #27300 was filed. Returning empty forces the
|
||||
central STT pipeline's transcript to become the body.
|
||||
"""
|
||||
item_list = [self._make_voice_item(text="garbled-tencent-transcript")]
|
||||
result = weixin._extract_text(item_list)
|
||||
# Currently broken: returns "garbled-tencent-transcript".
|
||||
# After fix: returns "" (empty string) so the central pipeline
|
||||
# transcript replaces it as the user-visible body.
|
||||
assert result != "garbled-tencent-transcript", (
|
||||
"_extract_text returned Tencent's text directly — for "
|
||||
"non-Chinese audio this is garbage; the central STT "
|
||||
"pipeline's transcript should be the body (#27300)."
|
||||
)
|
||||
|
||||
def test_extract_text_voice_only_returns_empty(self):
|
||||
"""When the only item is a voice attachment (no text item),
|
||||
``_extract_text`` should return empty so the central STT
|
||||
pipeline's transcript becomes the body. Currently returns
|
||||
Tencent's text which is what the bug is about.
|
||||
"""
|
||||
item_list = [self._make_voice_item(text="какой-то текст")]
|
||||
result = weixin._extract_text(item_list)
|
||||
assert result == "", (
|
||||
"Voice-only message: _extract_text should return empty so "
|
||||
"the central STT pipeline output replaces it as the body."
|
||||
)
|
||||
|
||||
def test_extract_text_still_returns_text_for_text_items(self):
|
||||
"""Sanity: the fix must not regress the text-item path. A plain
|
||||
text message should still produce its text body.
|
||||
"""
|
||||
item_list = [{
|
||||
"type": weixin.ITEM_TEXT,
|
||||
"text_item": {"text": "hello world"},
|
||||
}]
|
||||
assert weixin._extract_text(item_list) == "hello world"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_collect_media_includes_voice_when_tencent_text_set(self, tmp_path, monkeypatch):
|
||||
"""#27300 INTEGRATION: ``_collect_media`` should add a ``.silk``
|
||||
path to ``media_paths`` even when Tencent returned text, so the
|
||||
central STT pipeline can re-transcribe. Currently the
|
||||
short-circuit in ``_download_voice`` means the audio is never
|
||||
downloaded, and the message body is whatever Tencent wrote
|
||||
(garbled for non-Chinese audio).
|
||||
"""
|
||||
adapter = _make_adapter()
|
||||
adapter._cdn_base_url = "https://example.invalid"
|
||||
adapter._poll_session = Mock()
|
||||
|
||||
monkeypatch.setattr(weixin, "cache_audio_from_bytes",
|
||||
lambda data, ext: str(tmp_path / f"voice.{ext.lstrip('.')}"))
|
||||
|
||||
async def _fake_download(session, *, cdn_base_url, encrypted_query_param,
|
||||
aes_key_b64, full_url, timeout_seconds):
|
||||
return b"\\x00FAKE"
|
||||
|
||||
monkeypatch.setattr(weixin, "_download_and_decrypt_media", _fake_download)
|
||||
|
||||
media_paths: list = []
|
||||
media_types: list = []
|
||||
item = self._make_voice_item(text="какой-то текст")
|
||||
await adapter._collect_media(item, media_paths, media_types)
|
||||
|
||||
assert len(media_paths) == 1, (
|
||||
"_collect_media dropped the voice attachment because "
|
||||
"voice_item.text was set — Hermes' STT never gets a "
|
||||
"chance to re-transcribe (#27300)."
|
||||
)
|
||||
assert media_types == ["audio/silk"]
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue