hermes-agent/tests/tools/test_voice_mode.py
Teknium e6b0344052 test: adapt salvaged voice tests to current main + lint fix
- _FakeProc gains returncode (main's player loop checks proc.returncode)
- WSL gate tests clear SSH_* env vars (main hard-warns over SSH without
  forwarded audio) and accept the merged #37346 forwarded-sound-server
  notice wording
- test_tts_macos_output stubs resolve_streaming_provider so the
  OutputStream setup path actually runs on main's chunked-streamer code
- voice CLI integration tests unwrap the _VoiceInputMessage sentinel
- _is_wsl: explicit encoding + drop unreachable return (ruff PLW1514)
2026-07-28 11:57:37 -07:00

2503 lines
100 KiB
Python

"""Tests for tools.voice_mode -- all mocked, no real microphone or API calls."""
import os
import struct
import time
import wave
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
def _non_wsl_proc_version(real_open):
"""Return an open() shim that makes host WSL detection deterministic."""
def _fake_open(file, *args, **kwargs):
if file == "/proc/version":
from io import StringIO
return StringIO("Linux test-kernel")
return real_open(file, *args, **kwargs)
return _fake_open
# ============================================================================
# Fixtures
# ============================================================================
@pytest.fixture
def sample_wav(tmp_path):
"""Create a minimal valid WAV file (1 second of silence at 16kHz)."""
wav_path = tmp_path / "test.wav"
n_frames = 16000 # 1 second at 16kHz
silence = struct.pack(f"<{n_frames}h", *([0] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(silence)
return str(wav_path)
@pytest.fixture
def temp_voice_dir(tmp_path, monkeypatch):
"""Redirect _TEMP_DIR to a temporary path."""
voice_dir = tmp_path / "hermes_voice"
voice_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(voice_dir))
return voice_dir
@pytest.fixture
def mock_sd(monkeypatch):
"""Mock _import_audio to return (mock_sd, real_np) so lazy imports work."""
mock = MagicMock()
try:
import numpy as real_np
except ImportError:
real_np = MagicMock()
def _fake_import_audio():
return mock, real_np
monkeypatch.setattr("tools.voice_mode._import_audio", _fake_import_audio)
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
return mock
# ============================================================================
# detect_audio_environment — WSL / SSH / Docker detection
# ============================================================================
class TestPulseSocketReachable:
def test_no_env_no_socket(self, monkeypatch):
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.delenv("XDG_RUNTIME_DIR", raising=False)
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is False
def test_stale_socket_file_not_reachable(self, monkeypatch, tmp_path):
"""A socket file with no listener should not count as reachable."""
import socket as _socket
sock_path = tmp_path / "pulse" / "native"
sock_path.parent.mkdir(parents=True)
# Create + bind, then close so the path is a stale socket file.
s = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
s.bind(str(sock_path))
s.close()
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.setenv("XDG_RUNTIME_DIR", str(tmp_path))
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is False
def test_listening_socket_reachable_via_xdg_runtime(self, monkeypatch, tmp_path):
"""A live PulseAudio-style socket under XDG_RUNTIME_DIR is reachable (#35622)."""
import socket as _socket
sock_path = tmp_path / "pulse" / "native"
sock_path.parent.mkdir(parents=True)
server = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
server.bind(str(sock_path))
server.listen(1)
try:
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.setenv("XDG_RUNTIME_DIR", str(tmp_path))
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is True
finally:
server.close()
def test_listening_socket_reachable_via_pulse_server_env(self, monkeypatch, tmp_path):
import socket as _socket
sock_path = tmp_path / "native"
server = _socket.socket(_socket.AF_UNIX, _socket.SOCK_STREAM)
server.bind(str(sock_path))
server.listen(1)
try:
monkeypatch.delenv("PULSE_RUNTIME_PATH", raising=False)
monkeypatch.delenv("XDG_RUNTIME_DIR", raising=False)
monkeypatch.setenv("PULSE_SERVER", f"unix:{sock_path}")
from tools.voice_mode import _pulse_socket_reachable
assert _pulse_socket_reachable() is True
finally:
server.close()
class TestDetectAudioEnvironment:
def test_clean_environment_is_available(self, monkeypatch):
"""No SSH, Docker, or WSL — should be available."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
def test_ssh_blocks_voice(self, monkeypatch):
"""SSH environment without a reachable sound server should block voice mode."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("SSH" in w for w in result["warnings"])
def test_ssh_with_pulse_server_allows_voice(self, monkeypatch):
"""SSH with PULSE_SERVER set should NOT block voice mode (#35622)."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.setenv("PULSE_SERVER", "unix:/run/user/1002/pulse/native")
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("SSH" in n for n in result.get("notices", []))
def test_ssh_with_reachable_pulse_socket_allows_voice(self, monkeypatch):
"""SSH with a reachable PulseAudio socket (no env vars) allows voice (#35622)."""
monkeypatch.setenv("SSH_CLIENT", "1.2.3.4 54321 22")
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
# User runs `pulseaudio &` locally on the SSH host: the default socket
# is reachable even though PULSE_SERVER is unset.
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("SSH" in n for n in result.get("notices", []))
def test_wsl_without_pulse_blocks_voice(self, monkeypatch, tmp_path):
"""WSL without PULSE_SERVER should block voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("WSL" in w for w in result["warnings"])
assert any("PulseAudio" in w for w in result["warnings"])
def test_wsl_with_pulse_allows_voice(self, monkeypatch, tmp_path):
"""WSL with PULSE_SERVER set should NOT block voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("WSL" in n for n in result.get("notices", []))
def test_wsl_device_query_fails_with_pulse_continues(self, monkeypatch, tmp_path):
"""WSL device query failure should not block if PULSE_SERVER is set."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
mock_sd = MagicMock()
mock_sd.query_devices.side_effect = Exception("device query failed")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (mock_sd, MagicMock()))
proc_version = tmp_path / "proc_version"
proc_version.write_text("Linux 5.15.0-microsoft-standard-WSL2")
_real_open = open
def _fake_open(f, *a, **kw):
if f == "/proc/version":
return _real_open(str(proc_version), *a, **kw)
return _real_open(f, *a, **kw)
with patch("builtins.open", side_effect=_fake_open):
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert any("device query failed" in n for n in result.get("notices", []))
def test_device_query_fails_without_pulse_blocks(self, monkeypatch):
"""Device query failure without PULSE_SERVER should block."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
mock_sd = MagicMock()
mock_sd.query_devices.side_effect = Exception("device query failed")
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (mock_sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("PortAudio" in w for w in result["warnings"])
def test_termux_import_error_shows_termux_install_guidance(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: None)
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("pkg install python-numpy portaudio" in w for w in result["warnings"])
assert any("python -m pip install sounddevice" in w for w in result["warnings"])
def test_termux_api_package_without_android_app_blocks_voice(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: False)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("Termux:API Android app is not installed" in w for w in result["warnings"])
def test_docker_with_pulse_server_allows_voice(self, monkeypatch):
"""Docker with PULSE_SERVER set should NOT block voice mode (#21203)."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setenv("PULSE_SERVER", "unix:/run/user/1000/pulse/native")
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("container" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_allows_voice(self, monkeypatch):
"""Docker with PIPEWIRE_REMOTE set should NOT block voice mode (#21203)."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("container" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_and_no_devices_allows_voice(self, monkeypatch):
"""PIPEWIRE_REMOTE should bypass empty PortAudio device lists in Docker."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
sd = MagicMock()
sd.query_devices.return_value = []
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("host audio forwarding" in n.lower() for n in result.get("notices", []))
def test_docker_with_pipewire_remote_and_query_failure_allows_voice(self, monkeypatch):
"""PIPEWIRE_REMOTE should bypass PortAudio query failures in Docker."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.setenv("PIPEWIRE_REMOTE", "/run/user/1000/pipewire-0")
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
sd = MagicMock()
sd.query_devices.side_effect = RuntimeError("boom")
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (sd, MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert result["warnings"] == []
assert any("host audio forwarding" in n.lower() for n in result.get("notices", []))
def test_docker_without_audio_forwarding_blocks_voice(self, monkeypatch):
"""Docker without PULSE_SERVER/PIPEWIRE_REMOTE keeps blocking voice mode."""
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
monkeypatch.setattr("tools.voice_mode._pulse_socket_reachable", lambda: False)
monkeypatch.setattr("hermes_constants.is_container", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is False
assert any("container" in w.lower() for w in result["warnings"])
assert any("PULSE_SERVER" in w or "PIPEWIRE_REMOTE" in w for w in result["warnings"])
def test_termux_api_microphone_allows_voice_without_sounddevice(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.delenv("SSH_CLIENT", raising=False)
monkeypatch.delenv("SSH_TTY", raising=False)
monkeypatch.delenv("SSH_CONNECTION", raising=False)
monkeypatch.setattr("hermes_constants.is_container", lambda: False)
monkeypatch.setattr("tools.voice_mode.shutil.which", lambda cmd: "/data/data/com.termux/files/usr/bin/termux-microphone-record" if cmd == "termux-microphone-record" else None)
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (_ for _ in ()).throw(ImportError("no audio libs")))
monkeypatch.setattr("builtins.open", _non_wsl_proc_version(open))
from tools.voice_mode import detect_audio_environment
result = detect_audio_environment()
assert result["available"] is True
assert any("Termux:API microphone recording available" in n for n in result.get("notices", []))
assert result["warnings"] == []
# ============================================================================
# check_voice_requirements
# ============================================================================
class TestCheckVoiceRequirements:
def test_termux_api_capture_counts_as_audio_available(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: False)
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment", lambda: {"available": True, "warnings": [], "notices": ["Termux:API microphone recording available"]})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "openai")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["audio_available"] is True
assert result["missing_packages"] == []
assert "Termux:API microphone" in result["details"]
def test_all_requirements_met(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "openai")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["audio_available"] is True
assert result["stt_available"] is True
assert result["missing_packages"] == []
def test_missing_audio_packages(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: False)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": False, "warnings": ["Audio libraries not installed"]})
monkeypatch.setenv("VOICE_TOOLS_OPENAI_KEY", "sk-test-key")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["audio_available"] is False
assert "sounddevice" in result["missing_packages"]
assert "numpy" in result["missing_packages"]
def test_missing_stt_provider(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr("tools.transcription_tools._get_provider", lambda cfg: "none")
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
def test_command_stt_provider_selected(self, monkeypatch):
"""Catch-all branch fires for a selected command provider (not any provider)."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {
"enabled": True,
"provider": "my-custom-stt",
"providers": {
"my-custom-stt": {
"type": "command",
"command": "whisper_cpp {input}",
},
},
},
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["stt_available"] is True
assert "STT provider: OK (command: my-custom-stt)" in result["details"]
def test_unrelated_command_provider_not_confused(self, monkeypatch):
"""Unrelated command provider does NOT make a different selected provider appear OK."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {
"enabled": True,
"provider": "unknown-selected",
"providers": {
"unrelated-command": {
"type": "command",
"command": "whisper_cpp {input}",
},
},
},
)
monkeypatch.setattr(
"agent.transcription_registry.get_provider", lambda p: None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
def test_plugin_stt_provider(self, monkeypatch):
"""Plugin STT provider is recognized."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {"enabled": True, "provider": "my-plugin-stt"},
)
plugin_provider = MagicMock()
plugin_provider.is_available.return_value = True
monkeypatch.setattr(
"agent.transcription_registry.get_provider",
lambda p: plugin_provider if p == "my-plugin-stt" else None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is True
assert result["stt_available"] is True
assert "STT provider: OK (plugin: my-plugin-stt)" in result["details"]
def test_unavailable_plugin_stt_provider(self, monkeypatch):
"""A registered but unavailable plugin does not satisfy requirements."""
monkeypatch.setattr("tools.voice_mode._audio_available", lambda: True)
monkeypatch.setattr("tools.voice_mode.detect_audio_environment",
lambda: {"available": True, "warnings": []})
monkeypatch.setattr(
"tools.transcription_tools._load_stt_config",
lambda: {"enabled": True, "provider": "my-plugin-stt"},
)
plugin_provider = MagicMock()
plugin_provider.is_available.return_value = False
monkeypatch.setattr(
"agent.transcription_registry.get_provider",
lambda p: plugin_provider if p == "my-plugin-stt" else None,
)
monkeypatch.setattr(
"hermes_cli.plugins._ensure_plugins_discovered",
lambda force=False: None,
)
from tools.voice_mode import check_voice_requirements
result = check_voice_requirements()
assert result["available"] is False
assert result["stt_available"] is False
assert "STT provider: MISSING" in result["details"]
# ============================================================================
# AudioRecorder
# ============================================================================
class TestCreateAudioRecorder:
def test_termux_uses_termux_audio_recorder_when_api_present(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
from tools.voice_mode import create_audio_recorder, TermuxAudioRecorder
recorder = create_audio_recorder()
assert isinstance(recorder, TermuxAudioRecorder)
assert recorder.supports_silence_autostop is False
def test_termux_without_android_app_falls_back_to_audio_recorder(self, monkeypatch):
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: False)
from tools.voice_mode import create_audio_recorder, AudioRecorder
recorder = create_audio_recorder()
assert isinstance(recorder, AudioRecorder)
class TestTermuxAudioRecorder:
def test_start_and_stop_use_termux_microphone_commands(self, monkeypatch, temp_voice_dir):
command_calls = []
output_path = Path(temp_voice_dir) / "recording_20260409_120000.aac"
def fake_run(cmd, **kwargs):
command_calls.append(cmd)
if cmd[1] == "-f":
Path(cmd[2]).write_bytes(b"aac-bytes")
return MagicMock(returncode=0, stdout="", stderr="")
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.time.strftime", lambda fmt: "20260409_120000")
monkeypatch.setattr("tools.voice_mode.subprocess.run", fake_run)
from tools.voice_mode import TermuxAudioRecorder
recorder = TermuxAudioRecorder()
recorder.start()
recorder._start_time = time.monotonic() - 1.0
result = recorder.stop()
assert result == str(output_path)
assert command_calls[0][:2] == ["/data/data/com.termux/files/usr/bin/termux-microphone-record", "-f"]
assert command_calls[1] == ["/data/data/com.termux/files/usr/bin/termux-microphone-record", "-q"]
def test_cancel_removes_partial_termux_recording(self, monkeypatch, temp_voice_dir):
output_path = Path(temp_voice_dir) / "recording_20260409_120000.aac"
def fake_run(cmd, **kwargs):
if cmd[1] == "-f":
Path(cmd[2]).write_bytes(b"aac-bytes")
return MagicMock(returncode=0, stdout="", stderr="")
monkeypatch.setenv("TERMUX_VERSION", "0.118.3")
monkeypatch.setenv("PREFIX", "/data/data/com.termux/files/usr")
monkeypatch.setattr("tools.voice_mode._termux_microphone_command", lambda: "/data/data/com.termux/files/usr/bin/termux-microphone-record")
monkeypatch.setattr("tools.voice_mode._termux_api_app_installed", lambda: True)
monkeypatch.setattr("tools.voice_mode.time.strftime", lambda fmt: "20260409_120000")
monkeypatch.setattr("tools.voice_mode.subprocess.run", fake_run)
from tools.voice_mode import TermuxAudioRecorder
recorder = TermuxAudioRecorder()
recorder.start()
recorder.cancel()
assert output_path.exists() is False
assert recorder.is_recording is False
class TestAudioRecorder:
def test_start_raises_without_audio_libs(self, monkeypatch):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="sounddevice and numpy"):
recorder.start()
def test_start_oserror_points_at_portaudio_not_pip(self, monkeypatch):
"""OSError from _import_audio means PortAudio's shared library is
missing — pip can't fix that. The error must point at the system
package, not 'pip install sounddevice numpy' (#18432)."""
def _fail_import():
raise OSError("PortAudio library not found")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("tools.voice_mode._is_termux_environment", lambda: False)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError) as exc_info:
recorder.start()
msg = str(exc_info.value)
assert "PortAudio system library not found" in msg
assert "libportaudio2" in msg
assert "pip install" not in msg
def test_start_oserror_termux_hint(self, monkeypatch):
"""Same OSError path on Termux points at pkg install portaudio."""
def _fail_import():
raise OSError("PortAudio library not found")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("tools.voice_mode._is_termux_environment", lambda: True)
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="pkg install portaudio"):
recorder.start()
def test_start_creates_and_starts_stream(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
assert recorder.is_recording is True
mock_sd.InputStream.assert_called_once()
mock_stream.start.assert_called_once()
def test_double_start_is_noop(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
recorder.start() # second call should be noop
assert mock_sd.InputStream.call_count == 1
class TestAudioRecorderStop:
def test_stop_returns_none_when_not_recording(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
assert recorder.stop() is None
def test_stop_writes_wav_file(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder, SAMPLE_RATE
recorder = AudioRecorder()
recorder.start()
# Simulate captured audio frames (1 second of loud audio above RMS threshold)
frame = np.full((SAMPLE_RATE, 1), 1000, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 1000 # Peak RMS above threshold
wav_path = recorder.stop()
assert wav_path is not None
assert os.path.isfile(wav_path)
assert wav_path.endswith(".wav")
assert recorder.is_recording is False
# Verify it is a valid WAV
with wave.open(wav_path, "rb") as wf:
assert wf.getnchannels() == 1
assert wf.getsampwidth() == 2
assert wf.getframerate() == SAMPLE_RATE
def test_stop_returns_none_for_very_short_recording(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# Very short recording (100 samples = ~6ms at 16kHz)
frame = np.zeros((100, 1), dtype="int16")
recorder._frames = [frame]
wav_path = recorder.stop()
assert wav_path is None
def test_stop_returns_none_for_silent_recording(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder, SAMPLE_RATE
recorder = AudioRecorder()
recorder.start()
# 1 second of near-silence (RMS well below threshold)
frame = np.full((SAMPLE_RATE, 1), 10, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 10 # Peak RMS also below threshold
wav_path = recorder.stop()
assert wav_path is None
class TestAudioRecorderCancel:
def test_cancel_discards_frames(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
recorder._frames = [MagicMock()] # simulate captured data
recorder.cancel()
assert recorder.is_recording is False
assert recorder._frames == []
# Stream is kept alive (persistent) — cancel() does NOT close it.
mock_stream.stop.assert_not_called()
mock_stream.close.assert_not_called()
def test_cancel_when_not_recording_is_safe(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.cancel() # should not raise
assert recorder.is_recording is False
class TestAudioRecorderProperties:
def test_elapsed_seconds_when_not_recording(self):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
assert recorder.elapsed_seconds == 0.0
def test_elapsed_seconds_when_recording(self, mock_sd):
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# Force start time to 1 second ago
recorder._start_time = time.monotonic() - 1.0
elapsed = recorder.elapsed_seconds
assert 0.9 < elapsed < 10.0 # loose upper bound; only the lower bound is the property
recorder.cancel()
# ============================================================================
# transcribe_recording
# ============================================================================
class TestTranscribeRecording:
def test_delegates_to_transcribe_audio(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "hello world",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav", model="whisper-1")
assert result["success"] is True
assert result["transcript"] == "hello world"
mock_transcribe.assert_called_once_with("/tmp/test.wav", model="whisper-1")
def test_filters_whisper_hallucination(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "Thank you.",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is True
assert result["transcript"] == ""
assert result["filtered"] is True
def test_no_speech_failure_maps_to_silent_success(self):
"""Provider "empty transcript" errors are silence, not failure — the
voice loop should re-listen quietly instead of showing an error."""
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "ElevenLabs STT returned empty transcript",
"no_speech": True,
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is True
assert result["transcript"] == ""
assert result["no_speech"] is True
def test_real_failures_still_fail(self):
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "xAI STT API error (HTTP 500): boom",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["success"] is False
def test_does_not_filter_real_speech(self):
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "Thank you for helping me with this code.",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording("/tmp/test.wav")
assert result["transcript"] == "Thank you for helping me with this code."
assert "filtered" not in result
def test_oversized_wav_is_chunked_and_stitched(self, tmp_path, monkeypatch):
wav_path = tmp_path / "long.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
seen_paths = []
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
# First call is on the original file — simulate remote provider
# rejecting it as too large so chunking kicks in.
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 0.1MB (max 0.1MB)",
}
seen_paths.append(path)
assert model == "base"
assert path != str(wav_path)
assert os.path.getsize(path) <= 70 * 1024
return {
"success": True,
"transcript": f"part {len(seen_paths)}",
"provider": "local",
}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is True
assert result["transcript"] == " ".join(
f"part {i}" for i in range(1, len(seen_paths) + 1)
)
assert result["chunks"] == len(seen_paths)
assert len(seen_paths) > 1
assert all(not os.path.exists(path) for path in seen_paths)
def test_oversized_wav_reports_failing_chunk(self, tmp_path, monkeypatch):
wav_path = tmp_path / "long.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 0.1MB (max 0.1MB)",
}
return {"success": False, "transcript": "", "error": "provider rejected audio"}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is False
assert result["error"].startswith("Chunk 1/")
assert "provider rejected audio" in result["error"]
assert list(temp_dir.iterdir()) == []
def test_trusts_transcribe_audio_skip_chunk_for_local(self, tmp_path, monkeypatch):
"""Local providers never return 'File too large' — no chunking."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
mock_transcribe = MagicMock(return_value={
"success": True,
"transcript": "local whisper result",
"provider": "local",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is True
assert result["transcript"] == "local whisper result"
assert "chunks" not in result
mock_transcribe.assert_called_once_with(str(wav_path), model="base")
def test_chunks_when_transcribe_audio_returns_file_too_large(self, tmp_path, monkeypatch):
"""Remote provider rejects large file → chunking fallback."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
temp_dir = tmp_path / "chunks"
temp_dir.mkdir()
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", str(temp_dir))
monkeypatch.setattr("tools.transcription_tools.MAX_FILE_SIZE", 70 * 1024)
call_count = 0
def fake_transcribe(path, model=None):
nonlocal call_count
call_count += 1
if call_count == 1:
return {
"success": False,
"transcript": "",
"error": "File too large: 30.0MB (max 25MB)",
}
return {
"success": True,
"transcript": f"chunk {call_count - 1}",
"provider": "openai",
}
with patch("tools.transcription_tools.transcribe_audio", side_effect=fake_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="whisper-1")
assert result["success"] is True
assert result.get("chunks", 0) > 1
def test_other_error_does_not_trigger_chunk(self, tmp_path, monkeypatch):
"""Non-size errors from transcribe_audio are returned as-is."""
wav_path = tmp_path / "record.wav"
n_frames = 50000
audio = struct.pack(f"<{n_frames}h", *([1000] * n_frames))
with wave.open(str(wav_path), "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2)
wf.setframerate(16000)
wf.writeframes(audio)
mock_transcribe = MagicMock(return_value={
"success": False,
"transcript": "",
"error": "STT is disabled in config.yaml",
})
with patch("tools.transcription_tools.transcribe_audio", mock_transcribe):
from tools.voice_mode import transcribe_recording
result = transcribe_recording(str(wav_path), model="base")
assert result["success"] is False
assert "STT is disabled" in result["error"]
mock_transcribe.assert_called_once()
class TestWhisperHallucinationFilter:
def test_known_hallucinations(self):
from tools.voice_mode import is_whisper_hallucination
assert is_whisper_hallucination("Thank you.") is True
assert is_whisper_hallucination("thank you") is True
assert is_whisper_hallucination("Thanks for watching.") is True
assert is_whisper_hallucination("Bye.") is True
assert is_whisper_hallucination(" Thank you. ") is True # with whitespace
assert is_whisper_hallucination("you") is True
def test_real_speech_not_filtered(self):
from tools.voice_mode import is_whisper_hallucination
assert is_whisper_hallucination("Hello, how are you?") is False
assert is_whisper_hallucination("Thank you for your help with the project.") is False
assert is_whisper_hallucination("Can you explain this code?") is False
# ============================================================================
# play_audio_file
# ============================================================================
class TestPlayAudioFile:
def test_play_wav_via_sounddevice(self, monkeypatch, sample_wav):
np = pytest.importorskip("numpy")
# Pin to a non-macOS platform: on macOS WAV output deliberately skips
# sounddevice (see TestMacOSAudioOutputPolicy), so this path is only
# exercised off Darwin.
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
mock_sd_obj = MagicMock()
# Simulate stream completing immediately (get_stream().active = False)
mock_stream = MagicMock()
mock_stream.active = False
mock_sd_obj.get_stream.return_value = mock_stream
def _fake_import():
return mock_sd_obj, np
monkeypatch.setattr("tools.voice_mode._import_audio", _fake_import)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is True
mock_sd_obj.play.assert_called_once()
mock_sd_obj.stop.assert_called_once()
def test_returns_false_when_no_player(self, monkeypatch, sample_wav):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
monkeypatch.setattr("shutil.which", lambda _: None)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is False
def test_returns_false_for_missing_file(self):
from tools.voice_mode import play_audio_file
result = play_audio_file("/nonexistent/file.wav")
assert result is False
# ============================================================================
# macOS output policy (no sounddevice for OUTPUT -> avoids TCC prompt)
# ============================================================================
class TestMacOSAudioOutputPolicy:
def test_output_disallowed_on_macos(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
from tools.voice_mode import _sounddevice_output_allowed
assert _sounddevice_output_allowed() is False
def test_output_allowed_off_macos(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
from tools.voice_mode import _sounddevice_output_allowed
assert _sounddevice_output_allowed() is True
def test_play_audio_file_skips_sounddevice_on_macos(self, monkeypatch, sample_wav):
"""On macOS, WAV playback must not import sounddevice; it routes to afplay."""
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
def _forbidden_import():
raise AssertionError("sounddevice must not be imported for output on macOS")
monkeypatch.setattr("tools.voice_mode._import_audio", _forbidden_import)
popen_cmds = []
class _FakeProc:
returncode = 0
def wait(self, timeout=None):
return 0
def kill(self):
pass
def _fake_popen(cmd, **kwargs):
popen_cmds.append(cmd)
return _FakeProc()
monkeypatch.setattr("shutil.which", lambda exe: f"/usr/bin/{exe}")
monkeypatch.setattr("subprocess.Popen", _fake_popen)
from tools.voice_mode import play_audio_file
result = play_audio_file(sample_wav)
assert result is True
assert popen_cmds, "expected a system player to be invoked"
assert popen_cmds[0][0] == "afplay"
def test_play_beep_routes_through_afplay_on_macos(self, monkeypatch):
"""On macOS, beeps synthesize with numpy but play via the tempfile/afplay path."""
pytest.importorskip("numpy")
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Darwin")
def _forbidden_import():
raise AssertionError("sounddevice must not be imported for beeps on macOS")
monkeypatch.setattr("tools.voice_mode._import_audio", _forbidden_import)
calls = []
monkeypatch.setattr(
"tools.voice_mode._play_int16_via_tempfile",
lambda audio, sample_rate: calls.append((len(audio), sample_rate)),
)
import tools.voice_mode as vm
vm.play_beep(frequency=880, count=1)
assert len(calls) == 1
n_samples, sample_rate = calls[0]
assert n_samples > 0
assert sample_rate == vm.SAMPLE_RATE
def test_play_beep_uses_sounddevice_off_macos(self, monkeypatch):
"""Off macOS, beeps go straight through sounddevice."""
np = pytest.importorskip("numpy")
monkeypatch.setattr("tools.voice_mode.platform.system", lambda: "Linux")
mock_sd = MagicMock()
mock_stream = MagicMock()
mock_stream.active = False
mock_sd.get_stream.return_value = mock_stream
monkeypatch.setattr("tools.voice_mode._import_audio", lambda: (mock_sd, np))
tempfile_calls = []
monkeypatch.setattr(
"tools.voice_mode._play_int16_via_tempfile",
lambda audio, sample_rate: tempfile_calls.append(True),
)
import tools.voice_mode as vm
vm.play_beep(frequency=880, count=1)
mock_sd.play.assert_called_once()
assert not tempfile_calls, "off macOS should not use the tempfile/afplay path"
# ============================================================================
# cleanup_temp_recordings
# ============================================================================
class TestCleanupTempRecordings:
def test_old_files_deleted(self, temp_voice_dir):
# Create an "old" file
old_file = temp_voice_dir / "recording_20240101_000000.wav"
old_file.write_bytes(b"\x00" * 100)
# Set mtime to 2 hours ago
old_mtime = time.time() - 7200
os.utime(str(old_file), (old_mtime, old_mtime))
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 1
assert not old_file.exists()
def test_recent_files_preserved(self, temp_voice_dir):
# Create a "recent" file
recent_file = temp_voice_dir / "recording_20260303_120000.wav"
recent_file.write_bytes(b"\x00" * 100)
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 0
assert recent_file.exists()
def test_nonexistent_dir_returns_zero(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._TEMP_DIR", "/nonexistent/dir")
from tools.voice_mode import cleanup_temp_recordings
assert cleanup_temp_recordings() == 0
def test_non_recording_files_ignored(self, temp_voice_dir):
# Create a file that doesn't match the pattern
other_file = temp_voice_dir / "other_file.txt"
other_file.write_bytes(b"\x00" * 100)
old_mtime = time.time() - 7200
os.utime(str(other_file), (old_mtime, old_mtime))
from tools.voice_mode import cleanup_temp_recordings
deleted = cleanup_temp_recordings(max_age_seconds=3600)
assert deleted == 0
assert other_file.exists()
# ============================================================================
# play_beep
# ============================================================================
class TestPlayBeep:
def test_beep_calls_sounddevice_play(self, mock_sd):
np = pytest.importorskip("numpy")
from tools.voice_mode import play_beep
# play_beep uses polling (get_stream) + sd.stop() instead of sd.wait()
mock_stream = MagicMock()
mock_stream.active = False
mock_sd.get_stream.return_value = mock_stream
play_beep(frequency=880, duration=0.1, count=1)
mock_sd.play.assert_called_once()
mock_sd.stop.assert_called()
# Verify audio data is int16 numpy array
audio_arg = mock_sd.play.call_args[0][0]
assert audio_arg.dtype == np.int16
assert len(audio_arg) > 0
def test_beep_double_produces_longer_audio(self, mock_sd):
np = pytest.importorskip("numpy")
from tools.voice_mode import play_beep
play_beep(frequency=660, duration=0.1, count=2)
audio_arg = mock_sd.play.call_args[0][0]
single_beep_samples = int(16000 * 0.1)
# Double beep should be longer than a single beep
assert len(audio_arg) > single_beep_samples
def test_beep_noop_without_audio(self, monkeypatch):
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
from tools.voice_mode import play_beep
# Should not raise
play_beep()
def test_beep_handles_playback_error(self, mock_sd):
mock_sd.play.side_effect = Exception("device error")
from tools.voice_mode import play_beep
# Should not raise
play_beep()
# ============================================================================
# Silence detection
# ============================================================================
class TestSilenceDetection:
def test_silence_callback_fires_after_speech_then_silence(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
# Use very short durations for testing
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.05
fired = threading.Event()
def on_silence():
fired.set()
recorder.start(on_silence_stop=on_silence)
# Get the callback function from InputStream constructor
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Simulate sustained speech (multiple loud chunks to exceed min_speech_duration)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
time.sleep(0.06)
callback(loud_frame, 1600, None, None)
assert recorder._has_spoken is True
# Simulate silence
silent_frame = np.zeros((1600, 1), dtype="int16")
callback(silent_frame, 1600, None, None)
# Wait a bit past the silence duration, then send another silent frame
time.sleep(0.06)
callback(silent_frame, 1600, None, None)
# The callback should have been fired
assert fired.wait(timeout=1.0) is True
recorder.cancel()
def test_silence_without_speech_does_not_fire(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._silence_duration = 0.02
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Only silence -- no speech detected, so callback should NOT fire
silent_frame = np.zeros((1600, 1), dtype="int16")
for _ in range(5):
callback(silent_frame, 1600, None, None)
time.sleep(0.01)
assert fired.wait(timeout=0.2) is False
recorder.cancel()
def test_micro_pause_tolerance_during_speech(self, mock_sd):
"""Brief dips below threshold during speech should NOT reset speech tracking."""
np = pytest.importorskip("numpy")
import threading
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.15
recorder._max_dip_tolerance = 0.1
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
loud_frame = np.full((1600, 1), 5000, dtype="int16")
quiet_frame = np.full((1600, 1), 50, dtype="int16")
# Speech chunk 1
callback(loud_frame, 1600, None, None)
time.sleep(0.05)
# Brief micro-pause (dip < max_dip_tolerance)
callback(quiet_frame, 1600, None, None)
time.sleep(0.05)
# Speech resumes -- speech_start should NOT have been reset
callback(loud_frame, 1600, None, None)
assert recorder._speech_start > 0, "Speech start should be preserved across brief dips"
time.sleep(0.06)
# Another speech chunk to exceed min_speech_duration
callback(loud_frame, 1600, None, None)
assert recorder._has_spoken is True, "Speech should be confirmed after tolerating micro-pause"
recorder.cancel()
def test_no_callback_means_no_silence_detection(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start() # no on_silence_stop
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Even with speech then silence, nothing should happen
loud_frame = np.full((1600, 1), 5000, dtype="int16")
silent_frame = np.zeros((1600, 1), dtype="int16")
callback(loud_frame, 1600, None, None)
callback(silent_frame, 1600, None, None)
# No crash, no callback
assert recorder._on_silence_stop is None
recorder.cancel()
# ============================================================================
# Max recording length cap (voice.max_recording_seconds)
# ============================================================================
class TestMaxRecordingCap:
"""The hard cap must auto-stop through the real InputStream-callback
path — not just the predicate — and fire the one-shot callback exactly
once, independent of the silence-detection branches."""
def _get_stream_callback(self, mock_sd):
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
return callback
def test_cap_fires_one_shot_callback_during_continuous_speech(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_sd.InputStream.return_value = MagicMock()
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._max_recording_seconds = 0.1
# Park the other auto-stop branches far away so only the cap can fire:
# loud frames keep the silence branch off, and max_wait covers the
# no-speech branch.
recorder._silence_duration = 60.0
recorder._max_wait = 60.0
fires = []
fired = threading.Event()
def on_stop():
fires.append(1)
fired.set()
recorder.start(on_silence_stop=on_stop)
callback = self._get_stream_callback(mock_sd)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
assert not fired.is_set(), "cap must not fire before the limit elapses"
# Cross the cap while the user is STILL speaking — the silence branch
# can never fire here, so a hit proves the cap path.
time.sleep(0.12)
callback(loud_frame, 1600, None, None)
assert fired.wait(timeout=1.0) is True
# One-shot: the handler cleared _on_silence_stop, further frames past
# the cap must not fire again.
assert recorder._on_silence_stop is None
callback(loud_frame, 1600, None, None)
time.sleep(0.05)
assert len(fires) == 1
recorder.cancel()
def test_disabled_cap_never_fires_on_duration(self, mock_sd):
np = pytest.importorskip("numpy")
import threading
mock_sd.InputStream.return_value = MagicMock()
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder._max_recording_seconds = 0.0 # disabled (previous behaviour)
recorder._silence_duration = 60.0
recorder._max_wait = 60.0
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = self._get_stream_callback(mock_sd)
loud_frame = np.full((1600, 1), 5000, dtype="int16")
callback(loud_frame, 1600, None, None)
time.sleep(0.12)
callback(loud_frame, 1600, None, None)
assert fired.wait(timeout=0.2) is False
recorder.cancel()
# ============================================================================
# Playback interrupt
# ============================================================================
class TestPlaybackInterrupt:
"""Verify that TTS playback can be interrupted."""
def test_stop_playback_terminates_process(self):
from tools.voice_mode import stop_playback, _playback_lock
import tools.voice_mode as vm
mock_proc = MagicMock()
mock_proc.poll.return_value = None # process is running
with _playback_lock:
vm._active_playback = mock_proc
stop_playback()
mock_proc.terminate.assert_called_once()
with _playback_lock:
assert vm._active_playback is None
def test_stop_playback_noop_when_nothing_playing(self):
import tools.voice_mode as vm
with vm._playback_lock:
vm._active_playback = None
vm.stop_playback()
def test_play_audio_file_sets_active_playback(self, monkeypatch, sample_wav):
import tools.voice_mode as vm
def _fail_import():
raise ImportError("no sounddevice")
monkeypatch.setattr("tools.voice_mode._import_audio", _fail_import)
mock_proc = MagicMock()
mock_proc.wait.return_value = 0
mock_popen = MagicMock(return_value=mock_proc)
monkeypatch.setattr("subprocess.Popen", mock_popen)
monkeypatch.setattr("shutil.which", lambda cmd: "/usr/bin/" + cmd)
vm.play_audio_file(sample_wav)
assert mock_popen.called
with vm._playback_lock:
assert vm._active_playback is None
# ============================================================================
# Continuous mode flow
# ============================================================================
class TestContinuousModeFlow:
"""Verify continuous mode: auto-restart after transcription or silence."""
def test_continuous_restart_on_no_speech(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
# First recording: only silence -> stop returns None
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
for _ in range(10):
silence = np.full((1600, 1), 10, dtype="int16")
callback(silence, 1600, None, None)
wav_path = recorder.stop()
assert wav_path is None
# Simulate continuous mode restart
recorder.start()
assert recorder.is_recording is True
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
for _ in range(10):
speech = np.full((1600, 1), 5000, dtype="int16")
callback(speech, 1600, None, None)
wav_path = recorder.stop()
assert wav_path is not None
recorder.cancel()
def test_recorder_reusable_after_stop(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
results = []
for i in range(3):
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
loud = np.full((1600, 1), 5000, dtype="int16")
for _ in range(10):
callback(loud, 1600, None, None)
wav_path = recorder.stop()
results.append(wav_path)
assert all(r is not None for r in results)
assert os.path.isfile(results[-1])
# ============================================================================
# Audio level indicator
# ============================================================================
class TestAudioLevelIndicator:
"""Verify current_rms property updates in real-time for UI feedback."""
def test_rms_updates_with_audio_chunks(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
assert recorder.current_rms == 0
loud = np.full((1600, 1), 5000, dtype="int16")
callback(loud, 1600, None, None)
assert recorder.current_rms == 5000
quiet = np.full((1600, 1), 100, dtype="int16")
callback(quiet, 1600, None, None)
assert recorder.current_rms == 100
recorder.cancel()
def test_peak_rms_tracks_maximum(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
frames = [
np.full((1600, 1), 100, dtype="int16"),
np.full((1600, 1), 8000, dtype="int16"),
np.full((1600, 1), 500, dtype="int16"),
np.full((1600, 1), 3000, dtype="int16"),
]
for frame in frames:
callback(frame, 1600, None, None)
assert recorder._peak_rms == 8000
assert recorder.current_rms == 3000
recorder.cancel()
# ============================================================================
# Configurable silence parameters
# ============================================================================
class TestConfigurableSilenceParams:
"""Verify that silence detection params can be configured."""
def test_custom_threshold_and_duration(self, mock_sd):
np = pytest.importorskip("numpy")
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
import threading
recorder = AudioRecorder()
recorder._silence_threshold = 5000
recorder._silence_duration = 0.05
recorder._min_speech_duration = 0.05
fired = threading.Event()
recorder.start(on_silence_stop=lambda: fired.set())
callback = mock_sd.InputStream.call_args.kwargs.get("callback")
if callback is None:
callback = mock_sd.InputStream.call_args[1]["callback"]
# Audio at RMS 1000 -- below custom threshold (5000)
moderate = np.full((1600, 1), 1000, dtype="int16")
for _ in range(5):
callback(moderate, 1600, None, None)
time.sleep(0.02)
assert recorder._has_spoken is False
assert fired.wait(timeout=0.2) is False
# Now send really loud audio (above 5000 threshold)
very_loud = np.full((1600, 1), 8000, dtype="int16")
callback(very_loud, 1600, None, None)
time.sleep(0.06)
callback(very_loud, 1600, None, None)
assert recorder._has_spoken is True
recorder.cancel()
# ============================================================================
# Bugfix regression tests
# ============================================================================
class TestSubprocessTimeoutKill:
"""Bug: proc.wait(timeout) raised TimeoutExpired but process was not killed."""
def test_timeout_kills_process(self):
import subprocess
proc = subprocess.Popen(["sleep", "600"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
pid = proc.pid
assert proc.poll() is None
try:
proc.wait(timeout=0.1)
except subprocess.TimeoutExpired:
proc.kill()
proc.wait()
assert proc.poll() is not None
assert proc.returncode is not None
class TestStreamLeakOnStartFailure:
"""Bug: stream.start() failure left stream unclosed."""
def test_stream_closed_on_start_failure(self, mock_sd):
mock_stream = MagicMock()
mock_stream.start.side_effect = OSError("Audio device busy")
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
with pytest.raises(RuntimeError, match="Failed to open audio input stream"):
recorder._ensure_stream()
mock_stream.close.assert_called_once()
class TestSilenceCallbackLock:
"""Bug: _on_silence_stop was read/written without lock in audio callback."""
def test_fire_block_acquires_lock(self):
import inspect
from tools.voice_mode import AudioRecorder
source = inspect.getsource(AudioRecorder._ensure_stream)
# Verify lock is used before reading _on_silence_stop in fire block
assert "with self._lock:" in source
assert "cb = self._on_silence_stop" in source
lock_pos = source.index("with self._lock:")
cb_pos = source.index("cb = self._on_silence_stop")
assert lock_pos < cb_pos
def test_cancel_clears_callback_under_lock(self, mock_sd):
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
mock_sd.InputStream.return_value = MagicMock()
cb = lambda: None
recorder.start(on_silence_stop=cb)
assert recorder._on_silence_stop is cb
recorder.cancel()
with recorder._lock:
assert recorder._on_silence_stop is None
# ============================================================================
# listen_for_speech — VAD barge-in monitor
# ============================================================================
class _FakeInputStream:
"""Context-manager InputStream serving a fixed sequence of RMS levels."""
def __init__(self, np, levels):
self._np = np
self._levels = list(levels)
self.reads = 0
def __enter__(self):
return self
def __exit__(self, *args):
return False
def read(self, frames):
level = self._levels[min(self.reads, len(self._levels) - 1)]
self.reads += 1
return self._np.full((frames, 1), level, dtype=self._np.int16), False
class TestListenForSpeech:
"""listen_for_speech: calibration → sustained-speech trigger → barge-in."""
CALIB_BLOCKS = 14 # 400ms / 30ms
TRIP_BLOCKS = 10 # 300ms / 30ms
def _run(self, mock_sd, levels, should_stop=None, **kwargs):
np = pytest.importorskip("numpy")
stream = _FakeInputStream(np, levels)
mock_sd.InputStream.return_value = stream
from tools.voice_mode import listen_for_speech
stops = iter([False] * 200 + [True] * 10_000)
return listen_for_speech(should_stop or (lambda: next(stops)), **kwargs), stream
def test_sustained_speech_triggers(self, mock_sd):
levels = [0] * self.CALIB_BLOCKS + [5000] * 50
heard, _ = self._run(mock_sd, levels)
assert heard is True
def test_brief_spike_does_not_trigger(self, mock_sd):
levels = [0] * self.CALIB_BLOCKS + [5000] * (self.TRIP_BLOCKS - 2) + [0] * 500
heard, _ = self._run(mock_sd, levels)
assert heard is False
def test_should_stop_wins_over_silence(self, mock_sd):
"""TTS finishing (should_stop) ends the monitor without a trigger —
the default _run stopper flips True after 200 silent reads."""
heard, stream = self._run(mock_sd, [0] * 500)
assert heard is False
assert stream.reads <= 201
def test_returns_false_when_audio_unavailable(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._import_audio", MagicMock(side_effect=OSError("no audio")))
from tools.voice_mode import listen_for_speech
assert listen_for_speech(lambda: False) is False
def test_loud_floor_raises_trigger(self, mock_sd):
"""Speaker bleed during calibration bakes into the floor — playback-level
audio after calibration must NOT trip (only louder speech does)."""
levels = [2000] * self.CALIB_BLOCKS + [2000] * 100
heard, _ = self._run(mock_sd, levels)
assert heard is False
class TestListenForSpeechCapture:
"""capture=True: the barge monitor records the interruption with pre-roll,
so the utterance is complete from its first syllable — nothing is lost
between detection and a recorder restart."""
CALIB_BLOCKS = 14 # 400ms / 30ms
LOUD_BLOCKS = 30 # speech: trips after 10, keeps talking
BLOCK = 480 # 16000 * 0.03
def _run(self, mock_sd, monkeypatch, levels, should_stop=None, **kwargs):
np = pytest.importorskip("numpy")
stream = _FakeInputStream(np, levels)
mock_sd.InputStream.return_value = stream
written = {}
monkeypatch.setattr(
"tools.voice_mode.AudioRecorder._write_wav",
staticmethod(lambda audio: written.update(audio=audio) or "/tmp/barge.wav"),
)
from tools.voice_mode import listen_for_speech
stops = iter([False] * 200 + [True] * 10_000)
path = listen_for_speech(
should_stop or (lambda: next(stops)), capture=True, **kwargs
)
return path, written.get("audio"), stream
def test_captured_utterance_includes_speech_onset(self, mock_sd, monkeypatch):
"""Every loud block — including the ones BEFORE detection tripped —
must land in the WAV. That pre-roll is the whole point."""
triggered = []
levels = [0] * self.CALIB_BLOCKS + [5000] * self.LOUD_BLOCKS + [0] * 500
path, audio, _ = self._run(
mock_sd, monkeypatch, levels,
should_stop=lambda: False,
on_trigger=lambda: triggered.append(True),
)
assert path == "/tmp/barge.wav"
assert triggered == [True]
assert int((audio == 5000).sum()) == self.LOUD_BLOCKS * self.BLOCK
def test_no_trip_returns_none(self, mock_sd, monkeypatch):
triggered = []
path, audio, _ = self._run(
mock_sd, monkeypatch, [0] * 500,
on_trigger=lambda: triggered.append(True),
)
assert path is None
assert audio is None
assert triggered == []
def test_returns_none_when_audio_unavailable(self, monkeypatch):
monkeypatch.setattr("tools.voice_mode._import_audio", MagicMock(side_effect=OSError("no audio")))
from tools.voice_mode import listen_for_speech
assert listen_for_speech(lambda: False, capture=True) is None
class TestGetBeepVolume:
"""Issue #55908: beep amplitude must come from config.yaml, with safe fallback."""
def _get(self):
from tools.voice_mode import _get_beep_volume
return _get_beep_volume()
def test_default_when_key_missing(self):
with patch("hermes_cli.config.load_config", return_value={"voice": {}}):
assert self._get() == 0.3
def test_default_when_voice_section_missing(self):
with patch("hermes_cli.config.load_config", return_value={}):
assert self._get() == 0.3
def test_custom_value_honored(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 0.6}}):
assert self._get() == 0.6
def test_zero_is_accepted(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 0.0}}):
assert self._get() == 0.0
def test_one_is_accepted(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 1.0}}):
assert self._get() == 1.0
def test_out_of_range_high_clamps_to_default(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": 1.5}}):
assert self._get() == 0.3
def test_out_of_range_low_clamps_to_default(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": -0.5}}):
assert self._get() == 0.3
def test_string_numeric_is_coerced(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": "0.7"}}):
assert self._get() == 0.7
def test_non_numeric_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": "loud"}}):
assert self._get() == 0.3
def test_bool_value_falls_back(self):
"""Booleans must not silently pass as 0.0/1.0 (same guard as silence_threshold)."""
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": True}}):
assert self._get() == 0.3
def test_nan_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": {"beep_volume": float("nan")}}):
assert self._get() == 0.3
def test_load_config_exception_falls_back(self):
with patch("hermes_cli.config.load_config",
side_effect=RuntimeError("broken config")):
assert self._get() == 0.3
def test_voice_section_wrong_type_falls_back(self):
with patch("hermes_cli.config.load_config",
return_value={"voice": "not-a-dict"}):
assert self._get() == 0.3
class TestPlayBeepVolumeWiring:
"""Issue #55908: play_beep multiplies by the volume returned by _get_beep_volume.
Static wiring check — the behaviour is covered by TestGetBeepVolume above; this
class guards against regressions that re-introduce a hardcoded ``0.3`` literal
at the amplitude line in play_beep (the original bug class).
"""
def test_play_beep_does_not_use_hardcoded_0_3_literal(self):
import inspect
from tools import voice_mode as vm_mod
source = inspect.getsource(vm_mod.play_beep)
# The fix replaces ``tone * 0.3 * 32767`` with ``tone * beep_volume * 32767``
# where beep_volume is the result of _get_beep_volume().
hardcoded = " * 0.3 * 32767"
assert hardcoded not in source, (
"play_beep still contains a hardcoded 0.3 amplitude; use _get_beep_volume()"
)
assert "beep_volume * 32767" in source
assert "_get_beep_volume()" in source
# ============================================================================
# Device-native input sample rate — mics that reject 16 kHz capture
# ============================================================================
class TestDefaultInputSamplerate:
def test_uses_device_default_rate(self):
from tools.voice_mode import _default_input_samplerate
sd = MagicMock()
sd.query_devices.return_value = {"default_samplerate": 44100.0}
assert _default_input_samplerate(sd) == 44100
def test_falls_back_when_query_fails(self):
from tools.voice_mode import SAMPLE_RATE, _default_input_samplerate
sd = MagicMock()
sd.query_devices.side_effect = RuntimeError("no device")
assert _default_input_samplerate(sd) == SAMPLE_RATE
def test_falls_back_on_non_numeric_rate(self):
from tools.voice_mode import SAMPLE_RATE, _default_input_samplerate
sd = MagicMock()
sd.query_devices.return_value = {"default_samplerate": None}
assert _default_input_samplerate(sd) == SAMPLE_RATE
def test_recorder_opens_stream_at_device_rate(self, mock_sd):
mock_sd.query_devices.return_value = {"default_samplerate": 48000.0}
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
assert recorder.is_recording is True
assert mock_sd.InputStream.call_args.kwargs["samplerate"] == 48000
def test_wav_written_at_capture_rate(self, mock_sd, temp_voice_dir):
np = pytest.importorskip("numpy")
mock_sd.query_devices.return_value = {"default_samplerate": 48000.0}
mock_stream = MagicMock()
mock_sd.InputStream.return_value = mock_stream
from tools.voice_mode import AudioRecorder
recorder = AudioRecorder()
recorder.start()
# 1 second of loud audio at the device rate (above RMS threshold)
frame = np.full((48000, 1), 1000, dtype="int16")
recorder._frames = [frame]
recorder._peak_rms = 1000
wav_path = recorder.stop()
assert wav_path is not None
with wave.open(wav_path, "rb") as wf:
assert wf.getframerate() == 48000
class TestWSL2PowerShellFallback:
"""Regression tests for WSL2 PowerShell TTS fallback (issue #17608).
On WSL2 without a PulseAudio bridge, ffplay/aplay have no audio device.
play_audio_file() should insert a PowerShell-based player at the front
of the player list when powershell.exe and ffmpeg are available.
"""
def _fake_check_output(self, responses):
"""Build a subprocess.check_output side_effect from a list of responses."""
it = iter(responses)
def _side_effect(cmd, **kwargs):
return next(it)
return _side_effect
def test_wsl2_powershell_player_inserted_first(self, monkeypatch, sample_wav):
"""When WSL2 is detected and powershell.exe + ffmpeg are available,
a sh -c pipeline must be inserted before ffplay/aplay in the player list."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_players = []
def _capture_popen(cmd, **kw):
captured_players.append(list(cmd))
m = MagicMock()
m.returncode = 0
m.wait = MagicMock(return_value=0)
return m
with patch("tools.voice_mode._is_wsl2_env", return_value=True), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("tools.voice_mode.shutil.which",
side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay", "sh") else (x if x.startswith("/") else None)), \
patch("tools.voice_mode.subprocess.check_output",
side_effect=self._fake_check_output([
b"C:/Temp\r\n",
b"/mnt/c/Temp\n",
b"C:/Temp/hermes.wav\n",
])), \
patch("tools.voice_mode.subprocess.Popen", side_effect=_capture_popen):
vm.play_audio_file(str(sample_wav))
assert captured_players, "No players were tried"
first_cmd = captured_players[0]
assert first_cmd[0] in ("/bin/sh", "sh") and first_cmd[1] == "-c", (
f"Expected sh -c as first player, got {first_cmd}"
)
assert "powershell.exe" in first_cmd[2]
assert "PlaySync" in first_cmd[2]
def test_powershell_pipeline_preserves_real_exit_status(self, sample_wav):
"""Regression (review of #63768): the shell pipeline must preserve
the (ffmpeg && powershell) exit status past the unconditional
cleanup, so a real conversion/playback failure falls through to the
next player instead of being masked by rm -f's always-zero exit."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_cmds = []
def _capture_popen(cmd, **kw):
captured_cmds.append(list(cmd))
m = MagicMock()
# Simulate the PowerShell pipeline failing (nonzero rc), and
# the fallback ffplay succeeding.
if cmd[0] in ("/bin/sh", "sh"):
m.returncode = 1
else:
m.returncode = 0
m.wait = MagicMock(return_value=m.returncode)
return m
with patch("tools.voice_mode._is_wsl2_env", return_value=True), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("tools.voice_mode.shutil.which",
side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay", "sh") else (x if x.startswith("/") else None)), \
patch("tools.voice_mode.subprocess.check_output",
side_effect=self._fake_check_output([
b"C:/Temp\r\n",
b"/mnt/c/Temp\n",
b"C:/Temp/hermes.wav\n",
])), \
patch("tools.voice_mode.subprocess.Popen", side_effect=_capture_popen):
result = vm.play_audio_file(str(sample_wav))
assert result is True, "Must fall through to ffplay and succeed"
assert len(captured_cmds) == 2, (
f"Expected sh pipeline to be tried and fail, then ffplay to be "
f"tried: {captured_cmds}"
)
assert captured_cmds[0][0] in ("/bin/sh", "sh")
assert captured_cmds[1][0] == "ffplay"
# The subshell command must capture and re-exit with $rc, not rely
# on rm -f's exit status.
sh_script = captured_cmds[0][2]
assert "rc=$?" in sh_script and "exit $rc" in sh_script, (
"Shell pipeline must preserve the real exit status past cleanup: " + sh_script
)
def test_wsl2_unique_temp_filename(self, monkeypatch, tmp_path, sample_wav):
"""Two concurrent calls must use different temp WAV filenames."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
filenames = []
def _capture_check_output(cmd, **kwargs):
cmd_str = " ".join(str(c) for c in cmd)
if "TEMP" in cmd_str:
return b"C:\\Temp\r\n"
if "wslpath" in cmd_str and "-u" in cmd_str:
return b"/mnt/c/Temp\n"
if "wslpath" in cmd_str and "-w" in cmd_str:
wsl_path = cmd[-1] if isinstance(cmd[-1], str) else cmd[-1].decode()
filenames.append(wsl_path.split("/")[-1])
return f"C:\\Temp\\{wsl_path.split('/')[-1]}\n".encode()
return b""
def _fake_open(path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux Microsoft WSL2")
return open(path, *args, **kwargs)
with patch("builtins.open", side_effect=_fake_open), \
patch("shutil.which", side_effect=lambda x: f"/bin/{x}" if x in ("powershell.exe", "ffmpeg", "ffplay") else None), \
patch("subprocess.check_output", side_effect=_capture_check_output), \
patch("subprocess.Popen", return_value=MagicMock(returncode=0, wait=lambda **k: 0)), \
patch("tools.voice_mode._playback_lock"), \
patch("tools.voice_mode._active_playback", None):
vm.play_audio_file(str(sample_wav))
vm.play_audio_file(str(sample_wav))
# Regression (review of #63768): the original test made this
# assertion conditional on len(filenames) >= 2, so a broken
# (zero-captured) run passed trivially. Require exactly two.
assert len(filenames) == 2, (
f"Expected exactly 2 captured temp filenames from 2 calls, got "
f"{len(filenames)}: {filenames}"
)
assert filenames[0] != filenames[1], (
"Concurrent TTS calls must use unique temp WAV filenames"
)
def test_non_wsl_skips_powershell_fallback(self, monkeypatch, sample_wav):
"""On non-WSL Linux, the PowerShell player must not be inserted."""
from unittest.mock import patch, MagicMock
from tools import voice_mode as vm
captured_players = []
def _capture_popen(cmd, **kw):
captured_players.append(cmd)
m = MagicMock()
m.returncode = 0
m.wait.return_value = 0
return m
def _fake_open(path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux version 5.15.0-generic #72-Ubuntu")
return open(path, *args, **kwargs)
with patch("builtins.open", side_effect=_fake_open), \
patch("tools.voice_mode._import_audio", side_effect=ImportError), \
patch("shutil.which", side_effect=lambda x: f"/bin/{x}" if x in ("ffplay", "aplay") else None), \
patch("subprocess.Popen", side_effect=_capture_popen), \
patch("tools.voice_mode._playback_lock"), \
patch("tools.voice_mode._active_playback", None):
vm.play_audio_file(str(sample_wav))
assert captured_players, "No players were tried"
for cmd in captured_players:
assert not (cmd[0] == "sh" and "powershell" in " ".join(str(c) for c in cmd)), (
"PowerShell player must not appear on non-WSL Linux"
)
class TestWSLAudioEnvironmentGate:
"""Regression tests (review of #63768) for detect_audio_environment()'s
WSL gate: when the PowerShell TTS fallback is viable, voice mode must
not be hard-blocked, but the recording/STT PulseAudio-bridge guidance
must still be surfaced (as a non-blocking notice)."""
def _fake_open_wsl(self, path, *args, **kwargs):
if str(path) == "/proc/version":
import io
return io.StringIO("Linux version 5.15 Microsoft Standard WSL2")
return open(path, *args, **kwargs)
def test_wsl_no_pulse_but_powershell_available_not_hard_blocked(self, monkeypatch):
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("tools.voice_mode._wsl_powershell_tts_available", return_value=True), \
patch("tools.voice_mode._pulse_socket_reachable", return_value=False), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is True, (
"PowerShell TTS fallback must keep voice mode enabled even "
"without a PulseAudio bridge: " + str(result["warnings"])
)
assert any("PowerShell" in n or "Media.SoundPlayer" in n for n in result["notices"]), (
"The PowerShell fallback path must be mentioned in notices"
)
assert any("recording" in n.lower() or "PulseAudio" in n for n in result["notices"]), (
"The recording/STT PulseAudio caveat must still be surfaced"
)
def test_wsl_no_pulse_no_powershell_still_blocked(self, monkeypatch):
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.delenv("PULSE_SERVER", raising=False)
monkeypatch.delenv("PIPEWIRE_REMOTE", raising=False)
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("tools.voice_mode._wsl_powershell_tts_available", return_value=False), \
patch("tools.voice_mode._pulse_socket_reachable", return_value=False), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is False, (
"Without PulseAudio AND without the PowerShell fallback, WSL "
"must still be hard-blocked as before"
)
def test_wsl_with_pulse_server_unaffected(self, monkeypatch):
"""PULSE_SERVER already configured: existing behavior unchanged."""
from unittest.mock import patch
from tools import voice_mode as vm
monkeypatch.setenv("PULSE_SERVER", "unix:/mnt/wslg/PulseServer")
for _ssh_var in ("SSH_CLIENT", "SSH_TTY", "SSH_CONNECTION"):
monkeypatch.delenv(_ssh_var, raising=False)
monkeypatch.setattr("tools.voice_mode._import_audio",
lambda: (MagicMock(), MagicMock()))
with patch("builtins.open", side_effect=self._fake_open_wsl), \
patch("hermes_constants.is_container", return_value=False):
result = vm.detect_audio_environment()
assert result["available"] is True
# Merged with #37346: any forwarded sound server (PULSE_SERVER or
# PIPEWIRE_REMOTE) yields the shared reachable-sound-server notice.
assert any(
"PulseAudio" in n and "WSL" in n for n in result["notices"]
)