mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-24 16:54:43 +00:00
feat(cli): add the /wake command and wake_word config section
Adds the wake_word config block (surface, provider, phrase, sensitivity, and per-engine options) with wake_surface_enabled() so exactly one surface owns the listener and the session it opens. In the CLI the detector runs in-process; on wake it opens a fresh session and captures a single utterance through the existing voice pipeline, with an idle watchdog that re-arms the mic. The /wake [on|off|status] command reports what is configured and what is missing.
This commit is contained in:
parent
8ff79323d8
commit
129178bf09
4 changed files with 255 additions and 1 deletions
198
cli.py
198
cli.py
|
|
@ -1176,6 +1176,11 @@ def _run_cleanup(*, notify_session_finalize: bool = True):
|
|||
# can't skip the reset (#36823). No-op unless the TUI actually ran.
|
||||
_reset_terminal_input_modes_on_exit()
|
||||
|
||||
try:
|
||||
from tools.wake_word import stop_listening as _stop_wake_word
|
||||
_stop_wake_word()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
_cleanup_all_terminals()
|
||||
except Exception:
|
||||
|
|
@ -9352,6 +9357,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
|||
self._handle_skin_command(cmd_original)
|
||||
elif canonical == "voice":
|
||||
self._handle_voice_command(cmd_original)
|
||||
elif canonical == "wake":
|
||||
self._handle_wake_command(cmd_original)
|
||||
elif canonical == "busy":
|
||||
self._handle_busy_command(cmd_original)
|
||||
else:
|
||||
|
|
@ -11467,6 +11474,185 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
|||
|
||||
_cprint(f"\n{_DIM}Voice mode disabled.{_RST}")
|
||||
|
||||
# ── Wake word ("Hey Hermes") ─────────────────────────────────────────
|
||||
#
|
||||
# An always-on hotword listener (tools/wake_word.py) that, on detecting
|
||||
# the wake phrase, starts a fresh session and captures one utterance via
|
||||
# the existing voice pipeline — the "Hey Siri" pattern, fully on-device.
|
||||
#
|
||||
# The detector holds the microphone, so it must be paused while a voice
|
||||
# turn records (two input streams on one device is unreliable). On wake we
|
||||
# pause it and mark the system suspended; a lightweight watchdog resumes it
|
||||
# once the turn finishes and the CLI is idle again — covering every exit
|
||||
# path (transcript submitted, no speech, or transcription error) without
|
||||
# threading resume logic through the voice machinery.
|
||||
|
||||
def _maybe_start_wake_word(self):
|
||||
"""Start the wake-word listener at CLI startup if this surface owns it."""
|
||||
try:
|
||||
from tools.wake_word import wake_surface_enabled
|
||||
if not wake_surface_enabled("cli"):
|
||||
return
|
||||
except Exception:
|
||||
return
|
||||
self._start_wake_word_listener(announce=True)
|
||||
|
||||
def _start_wake_word_listener(self, announce: bool = False) -> bool:
|
||||
"""Build + start the hotword detector. Returns True on success."""
|
||||
if getattr(self, "_wake_word_active", False):
|
||||
if announce:
|
||||
_cprint(f"{_DIM}Wake word is already listening.{_RST}")
|
||||
return True
|
||||
try:
|
||||
from tools.wake_word import (
|
||||
check_wake_word_requirements,
|
||||
load_wake_word_config,
|
||||
start_listening,
|
||||
)
|
||||
except Exception as e:
|
||||
if announce:
|
||||
_cprint(f"{_DIM}Wake word unavailable: {e}{_RST}")
|
||||
return False
|
||||
|
||||
cfg = load_wake_word_config()
|
||||
reqs = check_wake_word_requirements(cfg)
|
||||
if not reqs["available"]:
|
||||
if announce:
|
||||
_cprint(f"\n{_ACCENT}Wake word requirements not met:{_RST}")
|
||||
if reqs.get("hint"):
|
||||
_cprint(f" {_DIM}{reqs['hint']}{_RST}")
|
||||
return False
|
||||
|
||||
self._wake_start_new_session = bool(cfg.get("start_new_session", True))
|
||||
try:
|
||||
start_listening(self._on_wake_word, config=cfg)
|
||||
except Exception as e:
|
||||
if announce:
|
||||
_cprint(f"\n{_DIM}Failed to start wake word: {e}{_RST}")
|
||||
return False
|
||||
|
||||
self._wake_word_active = True
|
||||
self._wake_suspended = False
|
||||
self._start_wake_watchdog()
|
||||
if announce:
|
||||
_cprint(f"\n{_ACCENT}Wake word listening{_RST} "
|
||||
f"{_DIM}(say \"{reqs['phrase']}\" — /wake off to stop){_RST}")
|
||||
return True
|
||||
|
||||
def _stop_wake_word_listener(self, announce: bool = False):
|
||||
"""Stop and tear down the hotword detector."""
|
||||
was_active = getattr(self, "_wake_word_active", False)
|
||||
self._wake_word_active = False
|
||||
self._wake_suspended = False
|
||||
try:
|
||||
from tools.wake_word import stop_listening
|
||||
stop_listening()
|
||||
except Exception:
|
||||
pass
|
||||
if announce:
|
||||
if was_active:
|
||||
_cprint(f"{_DIM}Wake word stopped.{_RST}")
|
||||
else:
|
||||
_cprint(f"{_DIM}Wake word is not running.{_RST}")
|
||||
|
||||
def _on_wake_word(self):
|
||||
"""Fired (on the detector thread) when the wake phrase is heard."""
|
||||
if getattr(self, "_should_exit", False):
|
||||
return
|
||||
# Ignore wake while a turn is in flight or the mic is already in use.
|
||||
if self._agent_running or self._voice_recording or getattr(self, "_voice_processing", False):
|
||||
return
|
||||
|
||||
# Release the mic so STT can capture the command utterance.
|
||||
try:
|
||||
from tools.wake_word import pause_listening
|
||||
pause_listening()
|
||||
except Exception:
|
||||
pass
|
||||
self._wake_suspended = True
|
||||
|
||||
_cprint(f"\n{_ACCENT}✦ Wake word detected — listening...{_RST}")
|
||||
if getattr(self, "_app", None):
|
||||
try:
|
||||
self._app.invalidate()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if getattr(self, "_wake_start_new_session", True):
|
||||
try:
|
||||
self.new_session(silent=True)
|
||||
except Exception as e:
|
||||
logger.debug("wake word new_session failed: %s", e)
|
||||
|
||||
# Single-utterance capture (not continuous) via the voice pipeline;
|
||||
# VAD auto-stop transcribes and queues the transcript for process_loop.
|
||||
with self._voice_lock:
|
||||
self._voice_mode = True
|
||||
self._voice_continuous = False
|
||||
try:
|
||||
self._voice_start_recording()
|
||||
except Exception as e:
|
||||
_cprint(f"{_DIM}Wake capture failed: {e}{_RST}")
|
||||
# Leave _wake_suspended set; the watchdog resumes once idle.
|
||||
|
||||
def _start_wake_watchdog(self):
|
||||
"""Resume the paused detector when the CLI returns to a stable idle."""
|
||||
if getattr(self, "_wake_watchdog_started", False):
|
||||
return
|
||||
self._wake_watchdog_started = True
|
||||
|
||||
def _loop():
|
||||
idle_polls = 0
|
||||
try:
|
||||
while getattr(self, "_wake_word_active", False) and not getattr(self, "_should_exit", False):
|
||||
time.sleep(0.25)
|
||||
if not getattr(self, "_wake_suspended", False):
|
||||
idle_polls = 0
|
||||
continue
|
||||
busy = (
|
||||
self._agent_running
|
||||
or self._voice_recording
|
||||
or getattr(self, "_voice_processing", False)
|
||||
or not self._pending_input.empty()
|
||||
)
|
||||
if busy:
|
||||
idle_polls = 0
|
||||
continue
|
||||
# Require a few consecutive idle polls (~0.75s) so we don't
|
||||
# resume in the gap between VAD stop and the agent starting.
|
||||
idle_polls += 1
|
||||
if idle_polls >= 3:
|
||||
idle_polls = 0
|
||||
try:
|
||||
from tools.wake_word import resume_listening
|
||||
resume_listening()
|
||||
self._wake_suspended = False
|
||||
except Exception as e:
|
||||
logger.debug("wake word resume failed: %s", e)
|
||||
finally:
|
||||
self._wake_watchdog_started = False
|
||||
|
||||
threading.Thread(target=_loop, daemon=True, name="wake-watchdog").start()
|
||||
|
||||
def _show_wake_word_status(self):
|
||||
"""Show current wake-word listener status."""
|
||||
from tools.wake_word import check_wake_word_requirements, load_wake_word_config
|
||||
|
||||
cfg = load_wake_word_config()
|
||||
reqs = check_wake_word_requirements(cfg)
|
||||
active = getattr(self, "_wake_word_active", False)
|
||||
|
||||
_cprint(f"\n{_BOLD}Wake Word Status{_RST}")
|
||||
_cprint(f" State: {'LISTENING' if active else 'OFF'}")
|
||||
_cprint(f" Phrase: \"{reqs['phrase']}\"")
|
||||
_cprint(f" Provider: {reqs['provider']}")
|
||||
_cprint(f" Surface: {cfg.get('surface', 'auto')}")
|
||||
_cprint(f" New session: {'yes' if cfg.get('start_new_session', True) else 'no'}")
|
||||
if not reqs["available"] and reqs.get("hint"):
|
||||
_cprint(f" {_DIM}{reqs['hint']}{_RST}")
|
||||
if not active:
|
||||
_cprint(f" {_DIM}Enable with /wake on{_RST}")
|
||||
|
||||
def _toggle_voice_tts(self):
|
||||
"""Toggle TTS output for voice mode."""
|
||||
if not self._voice_mode:
|
||||
|
|
@ -15501,7 +15687,17 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
|
|||
# Start processing thread
|
||||
process_thread = threading.Thread(target=process_loop, daemon=True)
|
||||
process_thread.start()
|
||||
|
||||
|
||||
# Wake word ("Hey Hermes") — start the always-on hotword listener if
|
||||
# enabled. Off-thread so a first-run engine install never blocks the
|
||||
# prompt; best-effort, so deps/mic/key gaps are surfaced, never fatal.
|
||||
def _wake_startup():
|
||||
try:
|
||||
self._maybe_start_wake_word()
|
||||
except Exception as e:
|
||||
logger.debug("wake-word startup skipped: %s", e)
|
||||
threading.Thread(target=_wake_startup, daemon=True, name="wake-startup").start()
|
||||
|
||||
# Register atexit cleanup so resources are freed even on unexpected exit
|
||||
atexit.register(_run_cleanup)
|
||||
|
||||
|
|
|
|||
|
|
@ -2811,3 +2811,26 @@ class CLICommandsMixin:
|
|||
else:
|
||||
_cprint(f"Unknown voice subcommand: {subcommand}")
|
||||
_cprint("Usage: /voice [on|off|tts|status]")
|
||||
|
||||
def _handle_wake_command(self, command: str):
|
||||
"""Handle /wake [on|off|status] — the 'Hey Hermes' hotword listener."""
|
||||
from cli import _cprint
|
||||
parts = command.strip().split(maxsplit=1)
|
||||
subcommand = parts[1].lower().strip() if len(parts) > 1 else ""
|
||||
|
||||
if subcommand == "on":
|
||||
self._start_wake_word_listener(announce=True)
|
||||
elif subcommand == "off":
|
||||
self._stop_wake_word_listener(announce=True)
|
||||
elif subcommand in ("", "status"):
|
||||
if subcommand == "":
|
||||
# Bare /wake toggles.
|
||||
if getattr(self, "_wake_word_active", False):
|
||||
self._stop_wake_word_listener(announce=True)
|
||||
else:
|
||||
self._start_wake_word_listener(announce=True)
|
||||
else:
|
||||
self._show_wake_word_status()
|
||||
else:
|
||||
_cprint(f"Unknown wake subcommand: {subcommand}")
|
||||
_cprint("Usage: /wake [on|off|status]")
|
||||
|
|
|
|||
|
|
@ -168,6 +168,9 @@ COMMAND_REGISTRY: list[CommandDef] = [
|
|||
subcommands=("kaomoji", "emoji", "unicode", "ascii")),
|
||||
CommandDef("voice", "Toggle voice mode", "Configuration",
|
||||
args_hint="[on|off|tts|status]", subcommands=("on", "off", "tts", "status")),
|
||||
CommandDef("wake", "Toggle the 'Hey Hermes' wake word listener", "Configuration",
|
||||
cli_only=True, args_hint="[on|off|status]",
|
||||
subcommands=("on", "off", "status")),
|
||||
CommandDef("busy", "Control what Enter does while Hermes is working", "Configuration",
|
||||
cli_only=True, args_hint="[queue|steer|interrupt|status]",
|
||||
subcommands=("queue", "steer", "interrupt", "status")),
|
||||
|
|
|
|||
|
|
@ -2256,6 +2256,31 @@ DEFAULT_CONFIG = {
|
|||
"silence_duration": 3.0, # Seconds of silence before auto-stop
|
||||
"barge_in": True, # Stop TTS playback when the user starts talking
|
||||
},
|
||||
|
||||
# "Hey Hermes" hands-free wake word (CLI). Always-on, on-device hotword
|
||||
# detection that starts a fresh voice session — the "Hey Siri" pattern.
|
||||
# Off by default; toggle with /wake or `wake_word.enabled: true`.
|
||||
"wake_word": {
|
||||
"enabled": False,
|
||||
"surface": "auto", # which surface owns the listener / opens the new session: "auto" (the running one) | "cli" | "tui" | "gui"
|
||||
"provider": "openwakeword", # "openwakeword" (free, local) | "porcupine" (premium; needs PORCUPINE_ACCESS_KEY)
|
||||
"phrase": "hey hermes", # cosmetic label only; detection is keyed by the engine model/keyword below
|
||||
"sensitivity": 0.5, # 0.0-1.0 detection threshold (higher = stricter)
|
||||
"start_new_session": True, # start a fresh session on wake vs. continue the current one
|
||||
"openwakeword": {
|
||||
# "hey_hermes" (the bundled, works-out-of-the-box default) OR a
|
||||
# built-in openWakeWord name ("hey_jarvis", "alexa", "hey_mycroft",
|
||||
# ...) OR a path to a custom .onnx/.tflite model for another phrase.
|
||||
# See the wake-word docs for the custom-model training guide.
|
||||
"model": "hey_hermes",
|
||||
"inference_framework": "onnx", # "onnx" | "tflite"
|
||||
},
|
||||
"porcupine": {
|
||||
# Built-in keyword ("jarvis", "computer", "bumblebee", ...) or a path
|
||||
# to a custom .ppn from the Picovoice Console.
|
||||
"keyword": "jarvis",
|
||||
},
|
||||
},
|
||||
|
||||
"human_delay": {
|
||||
"mode": "off",
|
||||
|
|
@ -4152,6 +4177,13 @@ OPTIONAL_ENV_VARS = {
|
|||
"password": True,
|
||||
"category": "tool",
|
||||
},
|
||||
"PORCUPINE_ACCESS_KEY": {
|
||||
"description": "Picovoice access key for the Porcupine 'Hey Hermes' wake word engine (optional; openWakeWord is the free default)",
|
||||
"prompt": "Picovoice access key",
|
||||
"url": "https://console.picovoice.ai/",
|
||||
"password": True,
|
||||
"category": "tool",
|
||||
},
|
||||
"GITHUB_TOKEN": {
|
||||
"description": "GitHub token for Skills Hub (higher API rate limits, skill publish)",
|
||||
"prompt": "GitHub Token",
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue