hermes-agent/tests/tools/test_voice_mode.py
Solitud1nem 562c8b418f test(voice): drive the micro-pause test from the explicit clock too
Review follow-up: test_micro_pause_tolerance_during_speech was left on real
sleeps, so it kept the wall-clock dependency the rest of this PR removes.
Its three sleeps (50 ms, 50 ms, 60 ms) feed the same monotonic gate at
tools/voice_mode.py:561 that the other two tests were converted for.

It was measured rather than assumed before being left out: 0 failures in 40
runs, because on Windows sleep() rounds up to the timer tick, so sleep(0.05)
really costs ~62.5 ms and the three sleeps clear the 150 ms gate by ~37.5 ms
-- wider than the 15.625 ms GetTickCount64() quantum that made the other two
flake. That margin is an accident of sleep granularity rather than anything
the test guarantees, so it is not worth keeping the bet.

Driving it from fake_clock makes the timeline exact (0.05 + 0.05 + 0.06 =
0.16 against the 0.15 gate, dip 0.05 under the 0.1 tolerance) and drops the
real sleeping from the suite. Mutation-checked: widening the gate at :561
fails the test.
2026-07-28 14:07:21 -07:00

2552 lines
102 KiB
Python

"""Tests for tools.voice_mode -- all mocked, no real microphone or API calls."""
import os
import struct
import time
import wave
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
def _non_wsl_proc_version(real_open):
"""Return an open() shim that makes host WSL detection deterministic."""
def _fake_open(file, *args, **kwargs):
if file == "/proc/version":
from io import StringIO
return StringIO("Linux test-kernel")
return real_open(file, *args, **kwargs)
return _fake_open
# ============================================================================
# Fixtures
# ============================================================================
@pytest.fixture
def sample_wav(tmp_path):
"""Create a minimal valid WAV file (1 second of silence at 16kHz)."""
wav_path = tmp_path / "test.wav"
n_frames = 16000 # 1 second at 16kHz
silence = struct.pack(f"<{n_frames}h", *([0] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(silence)
return str(wav_path)
@pytest.fixture
def temp_voice_dir(tmp_path, monkeypatch):
"""Redirect _TEMP_DIR to a temporary path."""
voice_dir = tmp_path / "hermes_voice"
voice_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(voice_dir))
return voice_dir
@pytest.fixture
def mock_sd(monkeypatch):
"""Mock _import_audio to return (mock_sd, real_np) so lazy imports work."""
mock = MagicMock()
try:
import numpy as real_np
except ImportError:
real_np = MagicMock()
def _fake_import_audio():
return mock, real_np
monkeypatch.setattr("tools.voice_mode._import_audio", _fake_import_audio)
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
return mock
class _FakeTime:
"""Stand-in for the ``time`` module with a monotonic clock the test drives.
Silence detection compares ``time.monotonic()`` deltas against thresholds
of a few dozen milliseconds. Driving those deltas with real ``sleep()``
calls only works when the platform clock is finer-grained than the margin
the test leaves: ``time.monotonic()`` is ``GetTickCount64()`` (15.625 ms
resolution) on Windows until CPython 3.13 moved it to
``QueryPerformanceCounter()``, so a 60 ms sleep can legitimately measure
as 46 ms and land under a 50 ms threshold. Advancing an explicit clock
keeps the arithmetic exact on every platform.
Everything other than ``monotonic`` delegates to the real module, so
``time.sleep``/``time.strftime`` in the code under test keep working.
"""
def __init__(self, real_time, start: float = 1000.0) -> None:
self._real = real_time
self._now = start
def monotonic(self) -> float:
return self._now
def advance(self, seconds: float) -> None:
self._now += seconds
def __getattr__(self, name):
return getattr(self._real, name)
@pytest.fixture
def fake_clock(monkeypatch):
"""Give voice_mode a hand-driven clock.
Patches the name ``time`` inside ``tools.voice_mode`` rather than
``time.monotonic`` itself -- ``voice_mode.time`` *is* the stdlib module,
so setting the attribute on it would swap the clock out from under every
other importer for the duration of the test.
"""
import time as real_time
import tools.voice_mode as voice_mode
clock = _FakeTime(real_time)
monkeypatch.setattr(voice_mode, "time", clock)
return clock
# ============================================================================
# detect_audio_environment — WSL / SSH / Docker detection
# ============================================================================
class TestPulseSocketReachable:
def test_no_env_no_socket(self, monkeypatch):
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.delenv("XDG_RUNTIME_DIR", raising=False)
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is False
def test_stale_socket_file_not_reachable(self, monkeypatch, tmp_path):
"""A socket file with no listener should not count as reachable."""
import socket as _socket
sock_path = tmp_path / "pulse" / "native"
sock_path.parent.mkdir(parents=True)
# Create + bind, then close so the path is a stale socket file.
s = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
s.bind(str(sock_path))
s.close()
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.setenv("XDG_RUNTIME_DIR", str(tmp_path))
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is False
def test_listening_socket_reachable_via_xdg_runtime(self, monkeypatch, tmp_path):
"""A live PulseAudio-style socket under XDG_RUNTIME_DIR is reachable (#35622)."""
import socket as _socket
sock_path = tmp_path / "pulse" / "native"
sock_path.parent.mkdir(parents=True)
server = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
server.bind(str(sock_path))
server.listen(1)
try:
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.setenv("XDG_RUNTIME_DIR", str(tmp_path))
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is True
finally:
server.close()
def test_listening_socket_reachable_via_pulse_server_env(self, monkeypatch, tmp_path):
import socket as _socket
sock_path = tmp_path / "native"
server = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
server.bind(str(sock_path))
server.listen(1)
try:
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.delenv("XDG_RUNTIME_DIR", raising=False)
monkeypatch.setenv("PULSE_SERVER", f"unix:{sock_path}")
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is True
finally:
server.close()
class TestDetectAudioEnvironment:
def test_clean_environment_is_available(self, monkeypatch):
"""No SSH, Docker, or WSL — should be available."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
def test_ssh_blocks_voice(self, monkeypatch):
"""SSH environment without a reachable sound server should block voice mode."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("SSH" in w for w in result["warnings"])
def test_ssh_with_pulse_server_allows_voice(self, monkeypatch):
"""SSH with PULSE_SERVER set should NOT block voice mode (#35622)."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.setenv("PULSE_SERVER", "unix:/run/user/1002/pulse/native")
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("SSH" in n for n in result.get("notices", []))
def test_ssh_with_reachable_pulse_socket_allows_voice(self, monkeypatch):
"""SSH with a reachable PulseAudio socket (no env vars) allows voice (#35622)."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
# User runs `pulseaudio &` locally on the SSH host: the default socket
# is reachable even though PULSE_SERVER is unset.
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("SSH" in n for n in result.get("notices", []))
def test_wsl_without_pulse_blocks_voice(self, monkeypatch, tmp_path):
"""WSL without PULSE_SERVER should block voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("WSL" in w for w in result["warnings"])
assert any("PulseAudio" in w for w in result["warnings"])
def test_wsl_with_pulse_allows_voice(self, monkeypatch, tmp_path):
"""WSL with PULSE_SERVER set should NOT block voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("WSL" in n for n in result.get("notices", []))
def test_wsl_device_query_fails_with_pulse_continues(self, monkeypatch, tmp_path):
"""WSL device query failure should not block if PULSE_SERVER is set."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
mock_sd = MagicMock()
mock_sd.query_devices.side_effect = Exception("device query failed")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (mock_sd, MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert any("device query failed" in n for n in result.get("notices", []))
def test_device_query_fails_without_pulse_blocks(self, monkeypatch):
"""Device query failure without PULSE_SERVER should block."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
mock_sd = MagicMock()
mock_sd.query_devices.side_effect = Exception("device query failed")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (mock_sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("PortAudio" in w for w in result["warnings"])
def test_termux_import_error_shows_termux_install_guidance(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: None)
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("pkg install python-numpy portaudio" in w for w in result["warnings"])
assert any("python -m pip install sounddevice" in w for w in result["warnings"])
def test_termux_api_package_without_android_app_blocks_voice(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("Termux:API Android app is not installed" in w for w in result["warnings"])
def test_docker_with_pulse_server_allows_voice(self, monkeypatch):
"""Docker with PULSE_SERVER set should NOT block voice mode (#21203)."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/run/user/1000/pulse/native")
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("container" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_allows_voice(self, monkeypatch):
"""Docker with PIPEWIRE_REMOTE set should NOT block voice mode (#21203)."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("container" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_and_no_devices_allows_voice(self, monkeypatch):
"""PIPEWIRE_REMOTE should bypass empty PortAudio device lists in Docker."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
sd = MagicMock()
sd.query_devices.return_value = []
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("host audio forwarding" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_and_query_failure_allows_voice(self, monkeypatch):
"""PIPEWIRE_REMOTE should bypass PortAudio query failures in Docker."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
sd = MagicMock()
sd.query_devices.side_effect = RuntimeError("boom")
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("host audio forwarding" in n.lower() for n in result.get("notices", []))
def test_docker_without_audio_forwarding_blocks_voice(self, monkeypatch):
"""Docker without PULSE_SERVER/PIPEWIRE_REMOTE keeps blocking voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("container" in w.lower() for w in result["warnings"])
assert any("PULSE_SERVER" in w or "PIPEWIRE_REMOTE" in w for w in result["warnings"])
def test_termux_api_microphone_allows_voice_without_sounddevice(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: False)
monkeypatch.setattr("tools.voice_mode.shutil.which", lambda cmd: "/data/data/com.termux/files/usr/bin/termux-microphone-record" if cmd == "termux-microphone-record" else None)
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert any("Termux:API microphone recording available" in n for n in result.get("notices", []))
assert result["warnings"] == []
# ============================================================================
# check_voice_requirements
# ============================================================================
class TestCheckVoiceRequirements:
def test_termux_api_capture_counts_as_audio_available(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: False)
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment", lambda: {"available": True, "warnings": [], "notices": ["Termux:API microphone recording available"]})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "openai")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["audio_available"] is True
assert result["missing_packages"] == []
assert "Termux:API microphone" in result["details"]
def test_all_requirements_met(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "openai")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["audio_available"] is True
assert result["stt_available"] is True
assert result["missing_packages"] == []
def test_missing_audio_packages(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: False)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": False, "warnings": ["Audio libraries not installed"]})
monkeypatch.setenv("VOICE_TOOLS_OPENAI_KEY", "sk-test-key")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["audio_available"] is False
assert "sounddevice" in result["missing_packages"]
assert "numpy" in result["missing_packages"]
def test_missing_stt_provider(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "none")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
def test_command_stt_provider_selected(self, monkeypatch):
"""Catch-all branch fires for a selected command provider (not any provider)."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {
"enabled": True,
"provider": "my-custom-stt",
"providers": {
"my-custom-stt": {
"type": "command",
"command": "whisper_cpp {input}",
},
},
},
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["stt_available"] is True
assert "STT provider: OK (command: my-custom-stt)" in result["details"]
def test_unrelated_command_provider_not_confused(self, monkeypatch):
"""Unrelated command provider does NOT make a different selected provider appear OK."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {
"enabled": True,
"provider": "unknown-selected",
"providers": {
"unrelated-command": {
"type": "command",
"command": "whisper_cpp {input}",
},
},
},
)
monkeypatch.setattr(
"agent.transcription_registry.get_provider", lambda p: None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
def test_plugin_stt_provider(self, monkeypatch):
"""Plugin STT provider is recognized."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {"enabled": True, "provider": "my-plugin-stt"},
)
plugin_provider = MagicMock()
plugin_provider.is_available.return_value = True
monkeypatch.setattr(
"agent.transcription_registry.get_provider",
lambda p: plugin_provider if p == "my-plugin-stt" else None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["stt_available"] is True
assert "STT provider: OK (plugin: my-plugin-stt)" in result["details"]
def test_unavailable_plugin_stt_provider(self, monkeypatch):
"""A registered but unavailable plugin does not satisfy requirements."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {"enabled": True, "provider": "my-plugin-stt"},
)
plugin_provider = MagicMock()
plugin_provider.is_available.return_value = False
monkeypatch.setattr(
"agent.transcription_registry.get_provider",
lambda p: plugin_provider if p == "my-plugin-stt" else None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
# ============================================================================
# AudioRecorder
# ============================================================================
class TestCreateAudioRecorder:
def test_termux_uses_termux_audio_recorder_when_api_present(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
from tools.voice_mode import create_audio_recorder, TermuxAudioRecorder
recorder = create_audio_recorder()
assert isinstance(recorder, TermuxAudioRecorder)
assert recorder.supports_silence_autostop is False
def test_termux_without_android_app_falls_back_to_audio_recorder(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: False)
from tools.voice_mode import create_audio_recorder, AudioRecorder
recorder = create_audio_recorder()
assert isinstance(recorder, AudioRecorder)
class TestTermuxAudioRecorder:
def test_start_and_stop_use_termux_microphone_commands(self, monkeypatch, temp_voice_dir):
command_calls = []
output_path = Path(temp_voice_dir) / "recording_20260409_120000.aac"
def fake_run(cmd, **kwargs):
command_calls.append(cmd)
if cmd[1] == "-f":
Path(cmd[2]).write_bytes(b"aac-bytes")
return MagicMock(returncode=0, stdout="", stderr="")
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.time.strftime", lambda fmt: "20260409_120000")
monkeypatch.setattr("tools.voice_mode.subprocess.run", fake_run)
from tools.voice_mode import TermuxAudioRecorder
recorder = TermuxAudioRecorder()
recorder.start()
recorder._start_time = time.monotonic() - 1.0
result = recorder.stop()
assert result == str(output_path)
assert command_calls[0][:2] == ["/data/data/com.termux/files/usr/bin/termux-microphone-record", "-f"]
assert command_calls[1] == ["/data/data/com.termux/files/usr/bin/termux-microphone-record", "-q"]
def test_cancel_removes_partial_termux_recording(self, monkeypatch, temp_voice_dir):
output_path = Path(temp_voice_dir) / "recording_20260409_120000.aac"
def fake_run(cmd, **kwargs):
if cmd[1] == "-f":
Path(cmd[2]).write_bytes(b"aac-bytes")
return MagicMock(returncode=0, stdout="", stderr="")
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.time.strftime", lambda fmt: "20260409_120000")
monkeypatch.setattr("tools.voice_mode.subprocess.run", fake_run)
from tools.voice_mode import TermuxAudioRecorder
recorder = TermuxAudioRecorder()
recorder.start()
recorder.cancel()
assert output_path.exists() is False
assert recorder.is_recording is False
class TestAudioRecorder:
def test_start_raises_without_audio_libs(self, monkeypatch):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="sounddevice and numpy"):
recorder.start()
def test_start_oserror_points_at_portaudio_not_pip(self, monkeypatch):
"""OSError from _import_audio means PortAudio's shared library is
missing — pip can't fix that. The error must point at the system
package, not 'pip install sounddevice numpy' (#18432)."""
def _fail_import():
raise OSError("PortAudio library not found")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("tools.voice_mode._is_termux_environment", lambda: False)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError) as exc_info:
recorder.start()
msg = str(exc_info.value)
assert "PortAudio system library not found" in msg
assert "libportaudio2" in msg
assert "pip install" not in msg
def test_start_oserror_termux_hint(self, monkeypatch):
"""Same OSError path on Termux points at pkg install portaudio."""
def _fail_import():
raise OSError("PortAudio library not found")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("tools.voice_mode._is_termux_environment", lambda: True)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="pkg install portaudio"):
recorder.start()
def test_start_creates_and_starts_stream(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
assert recorder.is_recording is True
mock_sd.InputStream.assert_called_once()
mock_stream.start.assert_called_once()
def test_double_start_is_noop(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
recorder.start() # second call should be noop
assert mock_sd.InputStream.call_count == 1
class TestAudioRecorderStop:
def test_stop_returns_none_when_not_recording(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
assert recorder.stop() is None
def test_stop_writes_wav_file(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder, SAMPLE_RATE
recorder = AudioRecorder()
recorder.start()
# Simulate captured audio frames (1 second of loud audio above RMS threshold)
frame = np.full((SAMPLE_RATE, 1), 1000, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 1000 # Peak RMS above threshold
wav_path = recorder.stop()
assert wav_path is not None
assert os.path.isfile(wav_path)
assert wav_path.endswith(".wav")
assert recorder.is_recording is False
# Verify it is a valid WAV
with wave.open(wav_path, "rb") as wf:
assert wf.getnchannels() == 1
assert wf.getsampwidth() == 2
assert wf.getframerate() == SAMPLE_RATE
def test_stop_returns_none_for_very_short_recording(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# Very short recording (100 samples = ~6ms at 16kHz)
frame = np.zeros((100, 1), dtype="int16")
recorder._frames = [frame]
wav_path = recorder.stop()
assert wav_path is None
def test_stop_returns_none_for_silent_recording(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder, SAMPLE_RATE
recorder = AudioRecorder()
recorder.start()
# 1 second of near-silence (RMS well below threshold)
frame = np.full((SAMPLE_RATE, 1), 10, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 10 # Peak RMS also below threshold
wav_path = recorder.stop()
assert wav_path is None
class TestAudioRecorderCancel:
def test_cancel_discards_frames(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
recorder._frames = [MagicMock()] # simulate captured data
recorder.cancel()
assert recorder.is_recording is False
assert recorder._frames == []
# Stream is kept alive (persistent) — cancel() does NOT close it.
mock_stream.stop.assert_not_called()
mock_stream.close.assert_not_called()
def test_cancel_when_not_recording_is_safe(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.cancel() # should not raise
assert recorder.is_recording is False
class TestAudioRecorderProperties:
def test_elapsed_seconds_when_not_recording(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
assert recorder.elapsed_seconds == 0.0
def test_elapsed_seconds_when_recording(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# Force start time to 1 second ago
recorder._start_time = time.monotonic() - 1.0
elapsed = recorder.elapsed_seconds
assert 0.9 < elapsed < 10.0 # loose upper bound; only the lower bound is the property
recorder.cancel()
# ============================================================================
# transcribe_recording
# ============================================================================
class TestTranscribeRecording:
def test_delegates_to_transcribe_audio(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "hello world",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav", model="whisper-1")
assert result["success"] is True
assert result["transcript"] == "hello world"
mock_transcribe.assert_called_once_with("/tmp/test.wav", model="whisper-1")
def test_filters_whisper_hallucination(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "Thank you.",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is True
assert result["transcript"] == ""
assert result["filtered"] is True
def test_no_speech_failure_maps_to_silent_success(self):
"""Provider "empty transcript" errors are silence, not failure — the
voice loop should re-listen quietly instead of showing an error."""
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "ElevenLabs STT returned empty transcript",
"no_speech": True,
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is True
assert result["transcript"] == ""
assert result["no_speech"] is True
def test_real_failures_still_fail(self):
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "xAI STT API error (HTTP 500): boom",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is False
def test_does_not_filter_real_speech(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "Thank you for helping me with this code.",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["transcript"] == "Thank you for helping me with this code."
assert "filtered" not in result
def test_oversized_wav_is_chunked_and_stitched(self, tmp_path, monkeypatch):
wav_path = tmp_path / "long.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
seen_paths = []
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
# First call is on the original file — simulate remote provider
# rejecting it as too large so chunking kicks in.
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 0.1MB (max 0.1MB)",
}
seen_paths.append(path)
assert model == "base"
assert path != str(wav_path)
assert os.path.getsize(path) <= 70 * 1024
return {
"success": True,
"transcript": f"part {len(seen_paths)}",
"provider": "local",
}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is True
assert result["transcript"] == " ".join(
f"part {i}" for i in range(1, len(seen_paths) + 1)
)
assert result["chunks"] == len(seen_paths)
assert len(seen_paths) > 1
assert all(not os.path.exists(path) for path in seen_paths)
def test_oversized_wav_reports_failing_chunk(self, tmp_path, monkeypatch):
wav_path = tmp_path / "long.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 0.1MB (max 0.1MB)",
}
return {"success": False, "transcript": "", "error": "provider rejected audio"}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is False
assert result["error"].startswith("Chunk 1/")
assert "provider rejected audio" in result["error"]
assert list(temp_dir.iterdir()) == []
def test_trusts_transcribe_audio_skip_chunk_for_local(self, tmp_path, monkeypatch):
"""Local providers never return 'File too large' — no chunking."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "local whisper result",
"provider": "local",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is True
assert result["transcript"] == "local whisper result"
assert "chunks" not in result
mock_transcribe.assert_called_once_with(str(wav_path), model="base")
def test_chunks_when_transcribe_audio_returns_file_too_large(self, tmp_path, monkeypatch):
"""Remote provider rejects large file → chunking fallback."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 30.0MB (max 25MB)",
}
return {
"success": True,
"transcript": f"chunk {call_count - 1}",
"provider": "openai",
}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="whisper-1")
assert result["success"] is True
assert result.get("chunks", 0) > 1
def test_other_error_does_not_trigger_chunk(self, tmp_path, monkeypatch):
"""Non-size errors from transcribe_audio are returned as-is."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "STT is disabled in config.yaml",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is False
assert "STT is disabled" in result["error"]
mock_transcribe.assert_called_once()
class TestWhisperHallucinationFilter:
def test_known_hallucinations(self):
from tools.voice_mode import is_whisper_hallucination
assert is_whisper_hallucination("Thank you.") is True
assert is_whisper_hallucination("thank you") is True
assert is_whisper_hallucination("Thanks for watching.") is True
assert is_whisper_hallucination("Bye.") is True
assert is_whisper_hallucination(" Thank you. ") is True # with whitespace
assert is_whisper_hallucination("you") is True
def test_real_speech_not_filtered(self):
from tools.voice_mode import is_whisper_hallucination
assert is_whisper_hallucination("Hello, how are you?") is False
assert is_whisper_hallucination("Thank you for your help with the project.") is False
assert is_whisper_hallucination("Can you explain this code?") is False
# ============================================================================
# play_audio_file
# ============================================================================
class TestPlayAudioFile:
def test_play_wav_via_sounddevice(self, monkeypatch, sample_wav):
np = pytest.importorskip("numpy")
# Pin to a non-macOS platform: on macOS WAV output deliberately skips
# sounddevice (see TestMacOSAudioOutputPolicy), so this path is only
# exercised off Darwin.
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
mock_sd_obj = MagicMock()
# Simulate stream completing immediately (get_stream().active = False)
mock_stream = MagicMock()
mock_stream.active = False
mock_sd_obj.get_stream.return_value = mock_stream
def _fake_import():
return mock_sd_obj, np
monkeypatch.setattr("tools.voice_mode._import_audio", _fake_import)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is True
mock_sd_obj.play.assert_called_once()
mock_sd_obj.stop.assert_called_once()
def test_returns_false_when_no_player(self, monkeypatch, sample_wav):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("shutil.which", lambda _: None)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is False
def test_returns_false_for_missing_file(self):
from tools.voice_mode import play_audio_file
result = play_audio_file("/nonexistent/file.wav")
assert result is False
# ============================================================================
# macOS output policy (no sounddevice for OUTPUT -> avoids TCC prompt)
# ============================================================================
class TestMacOSAudioOutputPolicy:
def test_output_disallowed_on_macos(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
from tools.voice_mode import _sounddevice_output_allowed
assert _sounddevice_output_allowed() is False
def test_output_allowed_off_macos(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
from tools.voice_mode import _sounddevice_output_allowed
assert _sounddevice_output_allowed() is True
def test_play_audio_file_skips_sounddevice_on_macos(self, monkeypatch, sample_wav):
"""On macOS, WAV playback must not import sounddevice; it routes to afplay."""
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
def _forbidden_import():
raise AssertionError("sounddevice must not be imported for output on macOS")
monkeypatch.setattr("tools.voice_mode._import_audio", _forbidden_import)
popen_cmds = []
class _FakeProc:
returncode = 0
def wait(self, timeout=None):
return 0
def kill(self):
pass
def _fake_popen(cmd, **kwargs):
popen_cmds.append(cmd)
return _FakeProc()
monkeypatch.setattr("shutil.which", lambda exe: f"/usr/bin/{exe}")
monkeypatch.setattr("subprocess.Popen", _fake_popen)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is True
assert popen_cmds, "expected a system player to be invoked"
assert popen_cmds[0][0] == "afplay"
def test_play_beep_routes_through_afplay_on_macos(self, monkeypatch):
"""On macOS, beeps synthesize with numpy but play via the tempfile/afplay path."""
pytest.importorskip("numpy")
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
def _forbidden_import():
raise AssertionError("sounddevice must not be imported for beeps on macOS")
monkeypatch.setattr("tools.voice_mode._import_audio", _forbidden_import)
calls = []
monkeypatch.setattr(
"tools.voice_mode._play_int16_via_tempfile",
lambda audio, sample_rate: calls.append((len(audio), sample_rate)),
)
import tools.voice_mode as vm
vm.play_beep(frequency=880, count=1)
assert len(calls) == 1
n_samples, sample_rate = calls[0]
assert n_samples > 0
assert sample_rate == vm.SAMPLE_RATE
def test_play_beep_uses_sounddevice_off_macos(self, monkeypatch):
"""Off macOS, beeps go straight through sounddevice."""
np = pytest.importorskip("numpy")
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
mock_sd = MagicMock()
mock_stream = MagicMock()
mock_stream.active = False
mock_sd.get_stream.return_value = mock_stream
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (mock_sd, np))
tempfile_calls = []
monkeypatch.setattr(
"tools.voice_mode._play_int16_via_tempfile",
lambda audio, sample_rate: tempfile_calls.append(True),
)
import tools.voice_mode as vm
vm.play_beep(frequency=880, count=1)
mock_sd.play.assert_called_once()
assert not tempfile_calls, "off macOS should not use the tempfile/afplay path"
# ============================================================================
# cleanup_temp_recordings
# ============================================================================
class TestCleanupTempRecordings:
def test_old_files_deleted(self, temp_voice_dir):
# Create an "old" file
old_file = temp_voice_dir / "recording_20240101_000000.wav"
old_file.write_bytes(b"\x00" * 100)
# Set mtime to 2 hours ago
old_mtime = time.time() - 7200
os.utime(str(old_file), (old_mtime, old_mtime))
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 1
assert not old_file.exists()
def test_recent_files_preserved(self, temp_voice_dir):
# Create a "recent" file
recent_file = temp_voice_dir / "recording_20260303_120000.wav"
recent_file.write_bytes(b"\x00" * 100)
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 0
assert recent_file.exists()
def test_nonexistent_dir_returns_zero(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", "/nonexistent/dir")
from tools.voice_mode import cleanup_temp_recordings
assert cleanup_temp_recordings() == 0
def test_non_recording_files_ignored(self, temp_voice_dir):
# Create a file that doesn't match the pattern
other_file = temp_voice_dir / "other_file.txt"
other_file.write_bytes(b"\x00" * 100)
old_mtime = time.time() - 7200
os.utime(str(other_file), (old_mtime, old_mtime))
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 0
assert other_file.exists()
# ============================================================================
# play_beep
# ============================================================================
class TestPlayBeep:
def test_beep_calls_sounddevice_play(self, mock_sd):
np = pytest.importorskip("numpy")
from tools.voice_mode import play_beep
# play_beep uses polling (get_stream) + sd.stop() instead of sd.wait()
mock_stream = MagicMock()
mock_stream.active = False
mock_sd.get_stream.return_value = mock_stream
play_beep(frequency=880, duration=0.1, count=1)
mock_sd.play.assert_called_once()
mock_sd.stop.assert_called()
# Verify audio data is int16 numpy array
audio_arg = mock_sd.play.call_args[0][0]
assert audio_arg.dtype == np.int16
assert len(audio_arg) > 0
def test_beep_double_produces_longer_audio(self, mock_sd):
np = pytest.importorskip("numpy")
from tools.voice_mode import play_beep
play_beep(frequency=660, duration=0.1, count=2)
audio_arg = mock_sd.play.call_args[0][0]
single_beep_samples = int(16000 * 0.1)
# Double beep should be longer than a single beep
assert len(audio_arg) > single_beep_samples
def test_beep_noop_without_audio(self, monkeypatch):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
from tools.voice_mode import play_beep
# Should not raise
play_beep()
def test_beep_handles_playback_error(self, mock_sd):
mock_sd.play.side_effect = Exception("device error")
from tools.voice_mode import play_beep
# Should not raise
play_beep()
# ============================================================================
# Silence detection
# ============================================================================
class TestSilenceDetection:
def test_silence_callback_fires_after_speech_then_silence(self, mock_sd, fake_clock):
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
# Use very short durations for testing
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.05
fired = threading.Event()
def on_silence():
fired.set()
recorder.start(on_silence_stop=on_silence)
# Get the callback function from InputStream constructor
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Simulate sustained speech (multiple loud chunks to exceed min_speech_duration)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
fake_clock.advance(0.06)
callback(loud_frame, 1600, None, None)
assert recorder._has_spoken is True
# Simulate silence
silent_frame = np.zeros((1600, 1), dtype="int16")
callback(silent_frame, 1600, None, None)
# Move past the silence duration, then send another silent frame
fake_clock.advance(0.06)
callback(silent_frame, 1600, None, None)
# The callback should have been fired (it runs on a real thread, so
# this wait is the one place real time is still involved)
assert fired.wait(timeout=5.0) is True
recorder.cancel()
def test_silence_without_speech_does_not_fire(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._silence_duration = 0.02
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Only silence -- no speech detected, so callback should NOT fire
silent_frame = np.zeros((1600, 1), dtype="int16")
for _ in range(5):
callback(silent_frame, 1600, None, None)
time.sleep(0.01)
assert fired.wait(timeout=0.2) is False
recorder.cancel()
def test_micro_pause_tolerance_during_speech(self, mock_sd, fake_clock):
"""Brief dips below threshold during speech should NOT reset speech tracking."""
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.15
recorder._max_dip_tolerance = 0.1
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
loud_frame = np.full((1600, 1), 5000, dtype="int16")
quiet_frame = np.full((1600, 1), 50, dtype="int16")
# Speech chunk 1
callback(loud_frame, 1600, None, None)
fake_clock.advance(0.05)
# Brief micro-pause (dip < max_dip_tolerance)
callback(quiet_frame, 1600, None, None)
fake_clock.advance(0.05)
# Speech resumes -- speech_start should NOT have been reset
callback(loud_frame, 1600, None, None)
assert recorder._speech_start > 0, "Speech start should be preserved across brief dips"
fake_clock.advance(0.06)
# Another speech chunk to exceed min_speech_duration
callback(loud_frame, 1600, None, None)
assert recorder._has_spoken is True, "Speech should be confirmed after tolerating micro-pause"
recorder.cancel()
def test_no_callback_means_no_silence_detection(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start() # no on_silence_stop
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Even with speech then silence, nothing should happen
loud_frame = np.full((1600, 1), 5000, dtype="int16")
silent_frame = np.zeros((1600, 1), dtype="int16")
callback(loud_frame, 1600, None, None)
callback(silent_frame, 1600, None, None)
# No crash, no callback
assert recorder._on_silence_stop is None
recorder.cancel()
# ============================================================================
# Max recording length cap (voice.max_recording_seconds)
# ============================================================================
class TestMaxRecordingCap:
"""The hard cap must auto-stop through the real InputStream-callback
path — not just the predicate — and fire the one-shot callback exactly
once, independent of the silence-detection branches."""
def _get_stream_callback(self, mock_sd):
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
return callback
def test_cap_fires_one_shot_callback_during_continuous_speech(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_sd.InputStream.return_value = MagicMock()
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._max_recording_seconds = 0.1
# Park the other auto-stop branches far away so only the cap can fire:
# loud frames keep the silence branch off, and max_wait covers the
# no-speech branch.
recorder._silence_duration = 60.0
recorder._max_wait = 60.0
fires = []
fired = threading.Event()
def on_stop():
fires.append(1)
fired.set()
recorder.start(on_silence_stop=on_stop)
callback = self._get_stream_callback(mock_sd)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
assert not fired.is_set(), "cap must not fire before the limit elapses"
# Cross the cap while the user is STILL speaking — the silence branch
# can never fire here, so a hit proves the cap path.
time.sleep(0.12)
callback(loud_frame, 1600, None, None)
assert fired.wait(timeout=1.0) is True
# One-shot: the handler cleared _on_silence_stop, further frames past
# the cap must not fire again.
assert recorder._on_silence_stop is None
callback(loud_frame, 1600, None, None)
time.sleep(0.05)
assert len(fires) == 1
recorder.cancel()
def test_disabled_cap_never_fires_on_duration(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_sd.InputStream.return_value = MagicMock()
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._max_recording_seconds = 0.0 # disabled (previous behaviour)
recorder._silence_duration = 60.0
recorder._max_wait = 60.0
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = self._get_stream_callback(mock_sd)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
time.sleep(0.12)
callback(loud_frame, 1600, None, None)
assert fired.wait(timeout=0.2) is False
recorder.cancel()
# ============================================================================
# Playback interrupt
# ============================================================================
class TestPlaybackInterrupt:
"""Verify that TTS playback can be interrupted."""
def test_stop_playback_terminates_process(self):
from tools.voice_mode import stop_playback, _playback_lock
import tools.voice_mode as vm
mock_proc = MagicMock()
mock_proc.poll.return_value = None # process is running
with _playback_lock:
vm._active_playback = mock_proc
stop_playback()
mock_proc.terminate.assert_called_once()
with _playback_lock:
assert vm._active_playback is None
def test_stop_playback_noop_when_nothing_playing(self):
import tools.voice_mode as vm
with vm._playback_lock:
vm._active_playback = None
vm.stop_playback()
def test_play_audio_file_sets_active_playback(self, monkeypatch, sample_wav):
import tools.voice_mode as vm
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
mock_proc = MagicMock()
mock_proc.wait.return_value = 0
mock_popen = MagicMock(return_value=mock_proc)
monkeypatch.setattr("subprocess.Popen", mock_popen)
monkeypatch.setattr("shutil.which", lambda cmd: "/usr/bin/" + cmd)
vm.play_audio_file(sample_wav)
assert mock_popen.called
with vm._playback_lock:
assert vm._active_playback is None
# ============================================================================
# Continuous mode flow
# ============================================================================
class TestContinuousModeFlow:
"""Verify continuous mode: auto-restart after transcription or silence."""
def test_continuous_restart_on_no_speech(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
# First recording: only silence -> stop returns None
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
for _ in range(10):
silence = np.full((1600, 1), 10, dtype="int16")
callback(silence, 1600, None, None)
wav_path = recorder.stop()
assert wav_path is None
# Simulate continuous mode restart
recorder.start()
assert recorder.is_recording is True
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
for _ in range(10):
speech = np.full((1600, 1), 5000, dtype="int16")
callback(speech, 1600, None, None)
wav_path = recorder.stop()
assert wav_path is not None
recorder.cancel()
def test_recorder_reusable_after_stop(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
results = []
for i in range(3):
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
loud = np.full((1600, 1), 5000, dtype="int16")
for _ in range(10):
callback(loud, 1600, None, None)
wav_path = recorder.stop()
results.append(wav_path)
assert all(r is not None for r in results)
assert os.path.isfile(results[-1])
# ============================================================================
# Audio level indicator
# ============================================================================
class TestAudioLevelIndicator:
"""Verify current_rms property updates in real-time for UI feedback."""
def test_rms_updates_with_audio_chunks(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
assert recorder.current_rms == 0
loud = np.full((1600, 1), 5000, dtype="int16")
callback(loud, 1600, None, None)
assert recorder.current_rms == 5000
quiet = np.full((1600, 1), 100, dtype="int16")
callback(quiet, 1600, None, None)
assert recorder.current_rms == 100
recorder.cancel()
def test_peak_rms_tracks_maximum(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
frames = [
np.full((1600, 1), 100, dtype="int16"),
np.full((1600, 1), 8000, dtype="int16"),
np.full((1600, 1), 500, dtype="int16"),
np.full((1600, 1), 3000, dtype="int16"),
]
for frame in frames:
callback(frame, 1600, None, None)
assert recorder._peak_rms == 8000
assert recorder.current_rms == 3000
recorder.cancel()
# ============================================================================
# Configurable silence parameters
# ============================================================================
class TestConfigurableSilenceParams:
"""Verify that silence detection params can be configured."""
def test_custom_threshold_and_duration(self, mock_sd, fake_clock):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
import threading
recorder = AudioRecorder()
recorder._silence_threshold = 5000
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.05
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Audio at RMS 1000 -- below custom threshold (5000)
moderate = np.full((1600, 1), 1000, dtype="int16")
for _ in range(5):
callback(moderate, 1600, None, None)
fake_clock.advance(0.02)
assert recorder._has_spoken is False
assert fired.wait(timeout=0.2) is False
# Now send really loud audio (above 5000 threshold)
very_loud = np.full((1600, 1), 8000, dtype="int16")
callback(very_loud, 1600, None, None)
fake_clock.advance(0.06)
callback(very_loud, 1600, None, None)
assert recorder._has_spoken is True
recorder.cancel()
# ============================================================================
# Bugfix regression tests
# ============================================================================
class TestSubprocessTimeoutKill:
"""Bug: proc.wait(timeout) raised TimeoutExpired but process was not killed."""
def test_timeout_kills_process(self):
import subprocess
proc = subprocess.Popen(["sleep", "600"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
pid = proc.pid
assert proc.poll() is None
try:
proc.wait(timeout=0.1)
except subprocess.TimeoutExpired:
proc.kill()
proc.wait()
assert proc.poll() is not None
assert proc.returncode is not None
class TestStreamLeakOnStartFailure:
"""Bug: stream.start() failure left stream unclosed."""
def test_stream_closed_on_start_failure(self, mock_sd):
mock_stream = MagicMock()
mock_stream.start.side_effect = OSError("Audio device busy")
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="Failed to open audio input stream"):
recorder._ensure_stream()
mock_stream.close.assert_called_once()
class TestSilenceCallbackLock:
"""Bug: _on_silence_stop was read/written without lock in audio callback."""
def test_fire_block_acquires_lock(self):
import inspect
from tools.voice_mode import AudioRecorder
source = inspect.getsource(AudioRecorder._ensure_stream)
# Verify lock is used before reading _on_silence_stop in fire block
assert "with self._lock:" in source
assert "cb = self._on_silence_stop" in source
lock_pos = source.index("with self._lock:")
cb_pos = source.index("cb = self._on_silence_stop")
assert lock_pos < cb_pos
def test_cancel_clears_callback_under_lock(self, mock_sd):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
mock_sd.InputStream.return_value = MagicMock()
cb = lambda: None
recorder.start(on_silence_stop=cb)
assert recorder._on_silence_stop is cb
recorder.cancel()
with recorder._lock:
assert recorder._on_silence_stop is None
# ============================================================================
# listen_for_speech — VAD barge-in monitor
# ============================================================================
class _FakeInputStream:
"""Context-manager InputStream serving a fixed sequence of RMS levels."""
def __init__(self, np, levels):
self._np = np
self._levels = list(levels)
self.reads = 0
def __enter__(self):
return self
def __exit__(self, *args):
return False
def read(self, frames):
level = self._levels[min(self.reads, len(self._levels) - 1)]
self.reads += 1
return self._np.full((frames, 1), level, dtype=self._np.int16), False
class TestListenForSpeech:
"""listen_for_speech: calibration → sustained-speech trigger → barge-in."""
CALIB_BLOCKS = 14 # 400ms / 30ms
TRIP_BLOCKS = 10 # 300ms / 30ms
def _run(self, mock_sd, levels, should_stop=None, **kwargs):
np = pytest.importorskip("numpy")
stream = _FakeInputStream(np, levels)
mock_sd.InputStream.return_value = stream
from tools.voice_mode import listen_for_speech
stops = iter([False] * 200 + [True] * 10_000)
return listen_for_speech(should_stop or (lambda: next(stops)), **kwargs), stream
def test_sustained_speech_triggers(self, mock_sd):
levels = [0] * self.CALIB_BLOCKS + [5000] * 50
heard, _ = self._run(mock_sd, levels)
assert heard is True
def test_brief_spike_does_not_trigger(self, mock_sd):
levels = [0] * self.CALIB_BLOCKS + [5000] * (self.TRIP_BLOCKS - 2) + [0] * 500
heard, _ = self._run(mock_sd, levels)
assert heard is False
def test_should_stop_wins_over_silence(self, mock_sd):
"""TTS finishing (should_stop) ends the monitor without a trigger —
the default _run stopper flips True after 200 silent reads."""
heard, stream = self._run(mock_sd, [0] * 500)
assert heard is False
assert stream.reads <= 201
def test_returns_false_when_audio_unavailable(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._import_audio", MagicMock(side_effect=OSError("no audio")))
from tools.voice_mode import listen_for_speech
assert listen_for_speech(lambda: False) is False
def test_loud_floor_raises_trigger(self, mock_sd):
"""Speaker bleed during calibration bakes into the floor — playback-level
audio after calibration must NOT trip (only louder speech does)."""
levels = [2000] * self.CALIB_BLOCKS + [2000] * 100
heard, _ = self._run(mock_sd, levels)
assert heard is False
class TestListenForSpeechCapture:
"""capture=True: the barge monitor records the interruption with pre-roll,
so the utterance is complete from its first syllable — nothing is lost
between detection and a recorder restart."""
CALIB_BLOCKS = 14 # 400ms / 30ms
LOUD_BLOCKS = 30 # speech: trips after 10, keeps talking
BLOCK = 480 # 16000 * 0.03
def _run(self, mock_sd, monkeypatch, levels, should_stop=None, **kwargs):
np = pytest.importorskip("numpy")
stream = _FakeInputStream(np, levels)
mock_sd.InputStream.return_value = stream
written = {}
monkeypatch.setattr(
"tools.voice_mode.AudioRecorder._write_wav",
staticmethod(lambda audio: written.update(audio=audio) or "/tmp/barge.wav"),
)
from tools.voice_mode import listen_for_speech
stops = iter([False] * 200 + [True] * 10_000)
path = listen_for_speech(
should_stop or (lambda: next(stops)), capture=True, **kwargs
)
return path, written.get("audio"), stream
def test_captured_utterance_includes_speech_onset(self, mock_sd, monkeypatch):
"""Every loud block — including the ones BEFORE detection tripped —
must land in the WAV. That pre-roll is the whole point."""
triggered = []
levels = [0] * self.CALIB_BLOCKS + [5000] * self.LOUD_BLOCKS + [0] * 500
path, audio, _ = self._run(
mock_sd, monkeypatch, levels,
should_stop=lambda: False,
on_trigger=lambda: triggered.append(True),
)
assert path == "/tmp/barge.wav"
assert triggered == [True]
assert int((audio == 5000).sum()) == self.LOUD_BLOCKS * self.BLOCK
def test_no_trip_returns_none(self, mock_sd, monkeypatch):
triggered = []
path, audio, _ = self._run(
mock_sd, monkeypatch, [0] * 500,
on_trigger=lambda: triggered.append(True),
)
assert path is None
assert audio is None
assert triggered == []
def test_returns_none_when_audio_unavailable(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._import_audio", MagicMock(side_effect=OSError("no audio")))
from tools.voice_mode import listen_for_speech
assert listen_for_speech(lambda: False, capture=True) is None
class TestGetBeepVolume:
"""Issue #55908: beep amplitude must come from config.yaml, with safe fallback."""
def _get(self):
from tools.voice_mode import _get_beep_volume
return _get_beep_volume()
def test_default_when_key_missing(self):
with patch("hermes_cli.config.load_config", return_value={"voice": {}}):
assert self._get() == 0.3
def test_default_when_voice_section_missing(self):
with patch("hermes_cli.config.load_config", return_value={}):
assert self._get() == 0.3
def test_custom_value_honored(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 0.6}}):
assert self._get() == 0.6
def test_zero_is_accepted(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 0.0}}):
assert self._get() == 0.0
def test_one_is_accepted(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 1.0}}):
assert self._get() == 1.0
def test_out_of_range_high_clamps_to_default(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 1.5}}):
assert self._get() == 0.3
def test_out_of_range_low_clamps_to_default(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": -0.5}}):
assert self._get() == 0.3
def test_string_numeric_is_coerced(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": "0.7"}}):
assert self._get() == 0.7
def test_non_numeric_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": "loud"}}):
assert self._get() == 0.3
def test_bool_value_falls_back(self):
"""Booleans must not silently pass as 0.0/1.0 (same guard as silence_threshold)."""
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": True}}):
assert self._get() == 0.3
def test_nan_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": float("nan")}}):
assert self._get() == 0.3
def test_load_config_exception_falls_back(self):
with patch("hermes_cli.config.load_config",
side_effect=RuntimeError("broken config")):
assert self._get() == 0.3
def test_voice_section_wrong_type_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": "not-a-dict"}):
assert self._get() == 0.3
class TestPlayBeepVolumeWiring:
"""Issue #55908: play_beep multiplies by the volume returned by _get_beep_volume.
Static wiring check — the behaviour is covered by TestGetBeepVolume above; this
class guards against regressions that re-introduce a hardcoded ``0.3`` literal
at the amplitude line in play_beep (the original bug class).
"""
def test_play_beep_does_not_use_hardcoded_0_3_literal(self):
import inspect
from tools import voice_mode as vm_mod
source = inspect.getsource(vm_mod.play_beep)
# The fix replaces ``tone * 0.3 * 32767`` with ``tone * beep_volume * 32767``
# where beep_volume is the result of _get_beep_volume().
hardcoded = " * 0.3 * 32767"
assert hardcoded not in source, (
"play_beep still contains a hardcoded 0.3 amplitude; use _get_beep_volume()"
)
assert "beep_volume * 32767" in source
assert "_get_beep_volume()" in source
# ============================================================================
# Device-native input sample rate — mics that reject 16 kHz capture
# ============================================================================
class TestDefaultInputSamplerate:
def test_uses_device_default_rate(self):
from tools.voice_mode import _default_input_samplerate
sd = MagicMock()
sd.query_devices.return_value = {"default_samplerate": 44100.0}
assert _default_input_samplerate(sd) == 44100
def test_falls_back_when_query_fails(self):
from tools.voice_mode import SAMPLE_RATE, _default_input_samplerate
sd = MagicMock()
sd.query_devices.side_effect = RuntimeError("no device")
assert _default_input_samplerate(sd) == SAMPLE_RATE
def test_falls_back_on_non_numeric_rate(self):
from tools.voice_mode import SAMPLE_RATE, _default_input_samplerate
sd = MagicMock()
sd.query_devices.return_value = {"default_samplerate": None}
assert _default_input_samplerate(sd) == SAMPLE_RATE
def test_recorder_opens_stream_at_device_rate(self, mock_sd):
mock_sd.query_devices.return_value = {"default_samplerate": 48000.0}
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
assert recorder.is_recording is True
assert mock_sd.InputStream.call_args.kwargs["samplerate"] == 48000
def test_wav_written_at_capture_rate(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_sd.query_devices.return_value = {"default_samplerate": 48000.0}
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# 1 second of loud audio at the device rate (above RMS threshold)
frame = np.full((48000, 1), 1000, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 1000
wav_path = recorder.stop()
assert wav_path is not None
with wave.open(wav_path, "rb") as wf:
assert wf.getframerate() == 48000
class TestWSL2PowerShellFallback:
"""Regression tests for WSL2 PowerShell TTS fallback (issue #17608).
On WSL2 without a PulseAudio bridge, ffplay/aplay have no audio device.
play_audio_file() should insert a PowerShell-based player at the front
of the player list when powershell.exe and ffmpeg are available.
"""
def _fake_check_output(self, responses):
"""Build a subprocess.check_output side_effect from a list of responses."""
it = iter(responses)
def _side_effect(cmd, **kwargs):
return next(it)
return _side_effect
def test_wsl2_powershell_player_inserted_first(self, monkeypatch, sample_wav):
"""When WSL2 is detected and powershell.exe + ffmpeg are available,
a sh -c pipeline must be inserted before ffplay/aplay in the player list."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_players = []
def _capture_popen(cmd, **kw):
captured_players.append(list(cmd))
m = MagicMock()
m.returncode = 0
m.wait = MagicMock(return_value=0)
return m
with patch("tools.voice_mode._is_wsl2_env", return_value=True), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("tools.voice_mode.shutil.which",
side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay", "sh") else (x if x.startswith("/") else None)), \
patch("tools.voice_mode.subprocess.check_output",
side_effect=self._fake_check_output([
b"C:/Temp\r\n",
b"/mnt/c/Temp\n",
b"C:/Temp/hermes.wav\n",
])), \
patch("tools.voice_mode.subprocess.Popen", side_effect=_capture_popen):
vm.play_audio_file(str(sample_wav))
assert captured_players, "No players were tried"
first_cmd = captured_players[0]
assert first_cmd[0] in ("/bin/sh", "sh") and first_cmd[1] == "-c", (
f"Expected sh -c as first player, got {first_cmd}"
)
assert "powershell.exe" in first_cmd[2]
assert "PlaySync" in first_cmd[2]
def test_powershell_pipeline_preserves_real_exit_status(self, sample_wav):
"""Regression (review of #63768): the shell pipeline must preserve
the (ffmpeg && powershell) exit status past the unconditional
cleanup, so a real conversion/playback failure falls through to the
next player instead of being masked by rm -f's always-zero exit."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_cmds = []
def _capture_popen(cmd, **kw):
captured_cmds.append(list(cmd))
m = MagicMock()
# Simulate the PowerShell pipeline failing (nonzero rc), and
# the fallback ffplay succeeding.
if cmd[0] in ("/bin/sh", "sh"):
m.returncode = 1
else:
m.returncode = 0
m.wait = MagicMock(return_value=m.returncode)
return m
with patch("tools.voice_mode._is_wsl2_env", return_value=True), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("tools.voice_mode.shutil.which",
side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay", "sh") else (x if x.startswith("/") else None)), \
patch("tools.voice_mode.subprocess.check_output",
side_effect=self._fake_check_output([
b"C:/Temp\r\n",
b"/mnt/c/Temp\n",
b"C:/Temp/hermes.wav\n",
])), \
patch("tools.voice_mode.subprocess.Popen", side_effect=_capture_popen):
result = vm.play_audio_file(str(sample_wav))
assert result is True, "Must fall through to ffplay and succeed"
assert len(captured_cmds) == 2, (
f"Expected sh pipeline to be tried and fail, then ffplay to be "
f"tried: {captured_cmds}"
)
assert captured_cmds[0][0] in ("/bin/sh", "sh")
assert captured_cmds[1][0] == "ffplay"
# The subshell command must capture and re-exit with $rc, not rely
# on rm -f's exit status.
sh_script = captured_cmds[0][2]
assert "rc=$?" in sh_script and "exit $rc" in sh_script, (
"Shell pipeline must preserve the real exit status past cleanup: " + sh_script
)
def test_wsl2_unique_temp_filename(self, monkeypatch, tmp_path, sample_wav):
"""Two concurrent calls must use different temp WAV filenames."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
filenames = []
def _capture_check_output(cmd, **kwargs):
cmd_str = " ".join(str(c) for c in cmd)
if "TEMP" in cmd_str:
return b"C:\\Temp\r\n"
if "wslpath" in cmd_str and "-u" in cmd_str:
return b"/mnt/c/Temp\n"
if "wslpath" in cmd_str and "-w" in cmd_str:
wsl_path = cmd[-1] if isinstance(cmd[-1], str) else cmd[-1].decode()
filenames.append(wsl_path.split("/")[-1])
return f"C:\\Temp\\{wsl_path.split('/')[-1]}\n".encode()
return b""
def _fake_open(path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux Microsoft WSL2")
return open(path, *args, **kwargs)
with patch("builtins.open", side_effect=_fake_open), \
patch("shutil.which", side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay") else None), \
patch("subprocess.check_output", side_effect=_capture_check_output), \
patch("subprocess.Popen", return_value=MagicMock(returncode=0, wait=lambda **k: 0)), \
patch("tools.voice_mode._playback_lock"), \
patch("tools.voice_mode._active_playback", None):
vm.play_audio_file(str(sample_wav))
vm.play_audio_file(str(sample_wav))
# Regression (review of #63768): the original test made this
# assertion conditional on len(filenames) >= 2, so a broken
# (zero-captured) run passed trivially. Require exactly two.
assert len(filenames) == 2, (
f"Expected exactly 2 captured temp filenames from 2 calls, got "
f"{len(filenames)}: {filenames}"
)
assert filenames[0] != filenames[1], (
"Concurrent TTS calls must use unique temp WAV filenames"
)
def test_non_wsl_skips_powershell_fallback(self, monkeypatch, sample_wav):
"""On non-WSL Linux, the PowerShell player must not be inserted."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_players = []
def _capture_popen(cmd, **kw):
captured_players.append(cmd)
m = MagicMock()
m.returncode = 0
m.wait.return_value = 0
return m
def _fake_open(path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux version 5.15.0-generic #72-Ubuntu")
return open(path, *args, **kwargs)
with patch("builtins.open", side_effect=_fake_open), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("shutil.which", side_effect=lambda x: f"/bin/{x}" if x in ("ffplay", "aplay") else None), \
patch("subprocess.Popen", side_effect=_capture_popen), \
patch("tools.voice_mode._playback_lock"), \
patch("tools.voice_mode._active_playback", None):
vm.play_audio_file(str(sample_wav))
assert captured_players, "No players were tried"
for cmd in captured_players:
assert not (cmd[0] == "sh" and "powershell" in " ".join(str(c) for c in cmd)), (
"PowerShell player must not appear on non-WSL Linux"
)
class TestWSLAudioEnvironmentGate:
"""Regression tests (review of #63768) for detect_audio_environment()'s
WSL gate: when the PowerShell TTS fallback is viable, voice mode must
not be hard-blocked, but the recording/STT PulseAudio-bridge guidance
must still be surfaced (as a non-blocking notice)."""
def _fake_open_wsl(self, path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux version 5.15 Microsoft Standard WSL2")
return open(path, *args, **kwargs)
def test_wsl_no_pulse_but_powershell_available_not_hard_blocked(self, monkeypatch):
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("tools.voice_mode._wsl_powershell_tts_available", return_value=True), \
patch("tools.voice_mode._pulse_socket_reachable", return_value=False), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is True, (
"PowerShell TTS fallback must keep voice mode enabled even "
"without a PulseAudio bridge: " + str(result["warnings"])
)
assert any("PowerShell" in n or "Media.SoundPlayer" in n for n in result["notices"]), (
"The PowerShell fallback path must be mentioned in notices"
)
assert any("recording" in n.lower() or "PulseAudio" in n for n in result["notices"]), (
"The recording/STT PulseAudio caveat must still be surfaced"
)
def test_wsl_no_pulse_no_powershell_still_blocked(self, monkeypatch):
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("tools.voice_mode._wsl_powershell_tts_available", return_value=False), \
patch("tools.voice_mode._pulse_socket_reachable", return_value=False), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is False, (
"Without PulseAudio AND without the PowerShell fallback, WSL "
"must still be hard-blocked as before"
)
def test_wsl_with_pulse_server_unaffected(self, monkeypatch):
"""PULSE_SERVER already configured: existing behavior unchanged."""
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is True
# Merged with #37346: any forwarded sound server (PULSE_SERVER or
# PIPEWIRE_REMOTE) yields the shared reachable-sound-server notice.
assert any(
"PulseAudio" in n and "WSL" in n for n in result["notices"]
)