From 5f43452e91e9f8350426001a936b1d8db16830fa Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Fri, 26 Jun 2026 21:54:45 -0500 Subject: [PATCH] feat(voice): add "Hey Hermes" wake word to start a hands-free session MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds an opt-in, on-device hotword listener for the CLI. With wake_word.enabled (or /wake on), Hermes listens in the background for a wake phrase; on detection it starts a fresh session, captures one utterance through the existing voice pipeline, and answers — the "Hey Siri" pattern. - tools/wake_word.py: provider-pluggable detector (openWakeWord, free local default; Porcupine, premium) over the shared 16 kHz sounddevice capture path. Background daemon thread with pause/resume so it yields the mic during a voice turn. - CLI wiring: startup listener (off-thread), on-wake flow, an idle watchdog that resumes the detector after each turn, cleanup hook, and a /wake [on|off|status] command. - config.yaml wake_word section; PORCUPINE_ACCESS_KEY as an optional secret. Engines lazy-install via the [wake] extra. - Hands a transcript to the input queue exactly like voice mode, so no system-prompt/cache mutation. No new core model tool. - Tests (mocked, no live audio/network) + feature docs. --- cli.py | 197 +++++++- hermes_cli/cli_commands_mixin.py | 23 + hermes_cli/commands.py | 3 + hermes_cli/config.py | 30 ++ pyproject.toml | 10 + tests/tools/test_wake_word.py | 208 +++++++++ tools/lazy_deps.py | 15 + tools/wake_word.py | 431 ++++++++++++++++++ website/docs/user-guide/features/overview.md | 1 + website/docs/user-guide/features/wake-word.md | 148 ++++++ 10 files changed, 1065 insertions(+), 1 deletion(-) create mode 100644 tests/tools/test_wake_word.py create mode 100644 tools/wake_word.py create mode 100644 website/docs/user-guide/features/wake-word.md diff --git a/cli.py b/cli.py index af020b9692bda..8fe26d43a4b27 100644 --- a/cli.py +++ b/cli.py @@ -1177,6 +1177,11 @@ def _run_cleanup(*, notify_session_finalize: bool = True): # can't skip the reset (#36823). No-op unless the TUI actually ran. _reset_terminal_input_modes_on_exit() + try: + from tools.wake_word import stop_listening as _stop_wake_word + _stop_wake_word() + except Exception: + pass try: _cleanup_all_terminals() except Exception: @@ -9919,6 +9924,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._handle_skin_command(cmd_original) elif canonical == "voice": self._handle_voice_command(cmd_original) + elif canonical == "wake": + self._handle_wake_command(cmd_original) elif canonical == "busy": self._handle_busy_command(cmd_original) else: @@ -12161,6 +12168,184 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): _cprint(f"\n{_DIM}Voice mode disabled.{_RST}") + # ── Wake word ("Hey Hermes") ───────────────────────────────────────── + # + # An always-on hotword listener (tools/wake_word.py) that, on detecting + # the wake phrase, starts a fresh session and captures one utterance via + # the existing voice pipeline — the "Hey Siri" pattern, fully on-device. + # + # The detector holds the microphone, so it must be paused while a voice + # turn records (two input streams on one device is unreliable). On wake we + # pause it and mark the system suspended; a lightweight watchdog resumes it + # once the turn finishes and the CLI is idle again — covering every exit + # path (transcript submitted, no speech, or transcription error) without + # threading resume logic through the voice machinery. + + def _maybe_start_wake_word(self): + """Start the wake-word listener at CLI startup if enabled in config.""" + try: + from tools.wake_word import load_wake_word_config + if not load_wake_word_config().get("enabled"): + return + except Exception: + return + self._start_wake_word_listener(announce=True) + + def _start_wake_word_listener(self, announce: bool = False) -> bool: + """Build + start the hotword detector. Returns True on success.""" + if getattr(self, "_wake_word_active", False): + if announce: + _cprint(f"{_DIM}Wake word is already listening.{_RST}") + return True + try: + from tools.wake_word import ( + check_wake_word_requirements, + load_wake_word_config, + start_listening, + ) + except Exception as e: + if announce: + _cprint(f"{_DIM}Wake word unavailable: {e}{_RST}") + return False + + cfg = load_wake_word_config() + reqs = check_wake_word_requirements(cfg) + if not reqs["available"]: + if announce: + _cprint(f"\n{_ACCENT}Wake word requirements not met:{_RST}") + if reqs.get("hint"): + _cprint(f" {_DIM}{reqs['hint']}{_RST}") + return False + + self._wake_start_new_session = bool(cfg.get("start_new_session", True)) + try: + start_listening(self._on_wake_word, config=cfg) + except Exception as e: + if announce: + _cprint(f"\n{_DIM}Failed to start wake word: {e}{_RST}") + return False + + self._wake_word_active = True + self._wake_suspended = False + self._start_wake_watchdog() + if announce: + _cprint(f"\n{_ACCENT}Wake word listening{_RST} " + f"{_DIM}(say \"{reqs['phrase']}\" — /wake off to stop){_RST}") + return True + + def _stop_wake_word_listener(self, announce: bool = False): + """Stop and tear down the hotword detector.""" + was_active = getattr(self, "_wake_word_active", False) + self._wake_word_active = False + self._wake_suspended = False + try: + from tools.wake_word import stop_listening + stop_listening() + except Exception: + pass + if announce: + if was_active: + _cprint(f"{_DIM}Wake word stopped.{_RST}") + else: + _cprint(f"{_DIM}Wake word is not running.{_RST}") + + def _on_wake_word(self): + """Fired (on the detector thread) when the wake phrase is heard.""" + if getattr(self, "_should_exit", False): + return + # Ignore wake while a turn is in flight or the mic is already in use. + if self._agent_running or self._voice_recording or getattr(self, "_voice_processing", False): + return + + # Release the mic so STT can capture the command utterance. + try: + from tools.wake_word import pause_listening + pause_listening() + except Exception: + pass + self._wake_suspended = True + + _cprint(f"\n{_ACCENT}✦ Wake word detected — listening...{_RST}") + if getattr(self, "_app", None): + try: + self._app.invalidate() + except Exception: + pass + + if getattr(self, "_wake_start_new_session", True): + try: + self.new_session(silent=True) + except Exception as e: + logger.debug("wake word new_session failed: %s", e) + + # Single-utterance capture (not continuous) via the voice pipeline; + # VAD auto-stop transcribes and queues the transcript for process_loop. + with self._voice_lock: + self._voice_mode = True + self._voice_continuous = False + try: + self._voice_start_recording() + except Exception as e: + _cprint(f"{_DIM}Wake capture failed: {e}{_RST}") + # Leave _wake_suspended set; the watchdog resumes once idle. + + def _start_wake_watchdog(self): + """Resume the paused detector when the CLI returns to a stable idle.""" + if getattr(self, "_wake_watchdog_started", False): + return + self._wake_watchdog_started = True + + def _loop(): + idle_polls = 0 + try: + while getattr(self, "_wake_word_active", False) and not getattr(self, "_should_exit", False): + time.sleep(0.25) + if not getattr(self, "_wake_suspended", False): + idle_polls = 0 + continue + busy = ( + self._agent_running + or self._voice_recording + or getattr(self, "_voice_processing", False) + or not self._pending_input.empty() + ) + if busy: + idle_polls = 0 + continue + # Require a few consecutive idle polls (~0.75s) so we don't + # resume in the gap between VAD stop and the agent starting. + idle_polls += 1 + if idle_polls >= 3: + idle_polls = 0 + try: + from tools.wake_word import resume_listening + resume_listening() + self._wake_suspended = False + except Exception as e: + logger.debug("wake word resume failed: %s", e) + finally: + self._wake_watchdog_started = False + + threading.Thread(target=_loop, daemon=True, name="wake-watchdog").start() + + def _show_wake_word_status(self): + """Show current wake-word listener status.""" + from tools.wake_word import check_wake_word_requirements, load_wake_word_config + + cfg = load_wake_word_config() + reqs = check_wake_word_requirements(cfg) + active = getattr(self, "_wake_word_active", False) + + _cprint(f"\n{_BOLD}Wake Word Status{_RST}") + _cprint(f" State: {'LISTENING' if active else 'OFF'}") + _cprint(f" Phrase: \"{reqs['phrase']}\"") + _cprint(f" Provider: {reqs['provider']}") + _cprint(f" New session: {'yes' if cfg.get('start_new_session', True) else 'no'}") + if not reqs["available"] and reqs.get("hint"): + _cprint(f" {_DIM}{reqs['hint']}{_RST}") + if not active: + _cprint(f" {_DIM}Enable with /wake on{_RST}") + def _toggle_voice_tts(self): """Toggle TTS output for voice mode.""" if not self._voice_mode: @@ -16325,7 +16510,17 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # Start processing thread process_thread = threading.Thread(target=process_loop, daemon=True) process_thread.start() - + + # Wake word ("Hey Hermes") — start the always-on hotword listener if + # enabled. Off-thread so a first-run engine install never blocks the + # prompt; best-effort, so deps/mic/key gaps are surfaced, never fatal. + def _wake_startup(): + try: + self._maybe_start_wake_word() + except Exception as e: + logger.debug("wake-word startup skipped: %s", e) + threading.Thread(target=_wake_startup, daemon=True, name="wake-startup").start() + # Register atexit cleanup so resources are freed even on unexpected exit atexit.register(_run_cleanup) diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 1891ca268fdb9..f2312296078be 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -3189,3 +3189,26 @@ class CLICommandsMixin: else: _cprint(f"Unknown voice subcommand: {subcommand}") _cprint("Usage: /voice [on|off|tts|status]") + + def _handle_wake_command(self, command: str): + """Handle /wake [on|off|status] — the 'Hey Hermes' hotword listener.""" + from cli import _cprint + parts = command.strip().split(maxsplit=1) + subcommand = parts[1].lower().strip() if len(parts) > 1 else "" + + if subcommand == "on": + self._start_wake_word_listener(announce=True) + elif subcommand == "off": + self._stop_wake_word_listener(announce=True) + elif subcommand in ("", "status"): + if subcommand == "": + # Bare /wake toggles. + if getattr(self, "_wake_word_active", False): + self._stop_wake_word_listener(announce=True) + else: + self._start_wake_word_listener(announce=True) + else: + self._show_wake_word_status() + else: + _cprint(f"Unknown wake subcommand: {subcommand}") + _cprint("Usage: /wake [on|off|status]") diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index 2a5f82cb38870..939c407370d57 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -181,6 +181,9 @@ COMMAND_REGISTRY: list[CommandDef] = [ subcommands=("kaomoji", "emoji", "unicode", "ascii")), CommandDef("voice", "Toggle voice mode", "Configuration", args_hint="[on|off|tts|status]", subcommands=("on", "off", "tts", "status")), + CommandDef("wake", "Toggle the 'Hey Hermes' wake word listener", "Configuration", + cli_only=True, args_hint="[on|off|status]", + subcommands=("on", "off", "status")), CommandDef("busy", "Control what Enter does while Hermes is working", "Configuration", cli_only=True, args_hint="[queue|steer|interrupt|status]", subcommands=("queue", "steer", "interrupt", "status")), diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 2a37a6d63fb9a..bc96780e4f8ca 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -2373,6 +2373,29 @@ DEFAULT_CONFIG = { # surrounding punctuation ignored. Set [] to disable. "stop_phrases": ["stop"], }, + + # "Hey Hermes" hands-free wake word (CLI). Always-on, on-device hotword + # detection that starts a fresh voice session — the "Hey Siri" pattern. + # Off by default; toggle with /wake or `wake_word.enabled: true`. + "wake_word": { + "enabled": False, + "provider": "openwakeword", # "openwakeword" (free, local) | "porcupine" (premium; needs PORCUPINE_ACCESS_KEY) + "phrase": "hey jarvis", # cosmetic label only; detection is keyed by the engine model/keyword below + "sensitivity": 0.5, # 0.0-1.0 detection threshold (higher = stricter) + "start_new_session": True, # start a fresh session on wake vs. continue the current one + "openwakeword": { + # Built-in model name ("hey_jarvis", "alexa", "hey_mycroft", ...) or + # a path to a custom .onnx/.tflite model. Train a "hey hermes" model + # and point this at it — see the wake-word docs. + "model": "hey_jarvis", + "inference_framework": "onnx", # "onnx" | "tflite" + }, + "porcupine": { + # Built-in keyword ("jarvis", "computer", "bumblebee", ...) or a path + # to a custom .ppn from the Picovoice Console. + "keyword": "jarvis", + }, + }, "human_delay": { "mode": "off", @@ -4412,6 +4435,13 @@ OPTIONAL_ENV_VARS = { "password": True, "category": "tool", }, + "PORCUPINE_ACCESS_KEY": { + "description": "Picovoice access key for the Porcupine 'Hey Hermes' wake word engine (optional; openWakeWord is the free default)", + "prompt": "Picovoice access key", + "url": "https://console.picovoice.ai/", + "password": True, + "category": "tool", + }, "GITHUB_TOKEN": { "description": "GitHub token for Skills Hub (higher API rate limits, skill publish)", "prompt": "GitHub Token", diff --git a/pyproject.toml b/pyproject.toml index 02c00a096752f..d757cd0fc9a20 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -175,6 +175,16 @@ voice = [ "sounddevice==0.5.5", "numpy==2.4.3", ] +# "Hey Hermes" wake word — on-device hotword detection. Both engines are +# optional; openWakeWord (ONNX) is the free default, Porcupine the premium +# alternative. Lazy-installed on first /wake; mirrored in tools/lazy_deps.py. +wake = [ + "openwakeword==0.6.0", + "onnxruntime==1.27.0", + "pvporcupine==4.0.3", + "sounddevice==0.5.5", + "numpy==2.4.3", +] honcho = ["honcho-ai==2.2.0"] # Cloud memory providers — opt-in, lazy-installed via tools/lazy_deps.py # (memory.supermemory / memory.mem0) at first use. Exact pins MUST match the diff --git a/tests/tools/test_wake_word.py b/tests/tools/test_wake_word.py new file mode 100644 index 0000000000000..30ece72368bde --- /dev/null +++ b/tests/tools/test_wake_word.py @@ -0,0 +1,208 @@ +"""Tests for tools.wake_word — the "Hey Hermes" hotword detector. + +No live audio or network: the sounddevice import is faked, engines are stubbed, +and lazy-dep availability is monkeypatched. Covers config resolution, engine +dispatch, the requirements probe, the detector fire/cooldown loop, and the +process-wide singleton lifecycle. +""" + +import time +import types + +import pytest + +import tools.wake_word as ww + + +# ── Config helpers ─────────────────────────────────────────────────────── + + +def test_config_defaults_and_clamping(): + assert ww._provider({}) == "openwakeword" + assert ww._provider({"provider": "Porcupine"}) == "porcupine" + assert ww._sensitivity({"sensitivity": 5}) == 1.0 + assert ww._sensitivity({"sensitivity": -1}) == 0.0 + assert ww._sensitivity({"sensitivity": "nope"}) == 0.5 + assert ww.wake_phrase({"phrase": "hey hermes"}) == "hey hermes" + assert ww.wake_phrase({}) == "hey jarvis" + + +def test_looks_like_path(): + assert ww._looks_like_path("models/hey_hermes.onnx") + assert ww._looks_like_path("custom.ppn") + assert not ww._looks_like_path("hey_jarvis") + + +def test_load_wake_word_config_is_a_dict_with_defaults(): + # Wired into DEFAULT_CONFIG, so a real load returns the section shape. + cfg = ww.load_wake_word_config() + assert isinstance(cfg, dict) + assert cfg.get("enabled") is False + assert cfg.get("provider") == "openwakeword" + + +def test_load_wake_word_config_guards_non_dict(monkeypatch): + monkeypatch.setattr( + "hermes_cli.config.load_config", lambda: {"wake_word": "oops"} + ) + assert ww.load_wake_word_config() == {} + + +# ── Engine dispatch ────────────────────────────────────────────────────── + + +def test_build_engine_dispatch(monkeypatch): + monkeypatch.setattr(ww, "_OpenWakeWordEngine", lambda cfg: "oww") + monkeypatch.setattr(ww, "_PorcupineEngine", lambda cfg: "pv") + assert ww._build_engine({"provider": "openwakeword"}) == "oww" + assert ww._build_engine({"provider": "porcupine"}) == "pv" + with pytest.raises(ValueError): + ww._build_engine({"provider": "bogus"}) + + +# ── Requirements probe ─────────────────────────────────────────────────── + + +def test_requirements_openwakeword_available(monkeypatch): + monkeypatch.setattr(ww, "_audio_available", lambda: True) + monkeypatch.setattr("tools.lazy_deps.is_available", lambda f: True) + r = ww.check_wake_word_requirements( + {"provider": "openwakeword", "phrase": "hey hermes"} + ) + assert r["available"] is True + assert r["provider"] == "openwakeword" + assert r["phrase"] == "hey hermes" + + +def test_requirements_porcupine_needs_access_key(monkeypatch): + monkeypatch.delenv("PORCUPINE_ACCESS_KEY", raising=False) + monkeypatch.setattr(ww, "_audio_available", lambda: True) + monkeypatch.setattr("tools.lazy_deps.is_available", lambda f: True) + r = ww.check_wake_word_requirements({"provider": "porcupine"}) + assert r["available"] is False + assert r["access_key_set"] is False + assert "PORCUPINE_ACCESS_KEY" in r["hint"] + + +def test_requirements_unavailable_without_audio(monkeypatch): + monkeypatch.setattr(ww, "_audio_available", lambda: False) + monkeypatch.setattr("tools.lazy_deps.is_available", lambda f: True) + r = ww.check_wake_word_requirements({"provider": "openwakeword"}) + assert r["available"] is False + assert r["audio_available"] is False + + +# ── Detector loop ──────────────────────────────────────────────────────── + + +class _FakeStream: + """Always-readable input stream that yields trivial frames.""" + + def __init__(self, **_kw): + self.closed = False + + def start(self): + pass + + def read(self, n): + time.sleep(0.01) + return [0] * n, False + + def stop(self): + pass + + def close(self): + self.closed = True + + +class _FakeEngine: + frame_length = 4 + + def __init__(self, fire=True): + self._fire = fire + self.closed = False + + def process(self, frame): + return self._fire + + def close(self): + self.closed = True + + +def _fake_audio(monkeypatch): + fake_sd = types.SimpleNamespace(InputStream=lambda **kw: _FakeStream(**kw)) + monkeypatch.setattr(ww, "_import_audio", lambda: (fake_sd, None)) + + +def test_detector_fires_once_under_cooldown(monkeypatch): + _fake_audio(monkeypatch) + calls = [] + eng = _FakeEngine(fire=True) + det = ww.WakeWordDetector(eng, lambda: calls.append(1), cooldown=10.0) + det.start() + time.sleep(0.25) + det.stop() + assert len(calls) == 1 # high cooldown suppresses repeats + assert eng.closed is True + assert det.running is False + + +def test_detector_refires_after_cooldown(monkeypatch): + _fake_audio(monkeypatch) + calls = [] + det = ww.WakeWordDetector(_FakeEngine(fire=True), lambda: calls.append(1), cooldown=0.05) + det.start() + time.sleep(0.3) + det.stop() + assert len(calls) >= 2 + + +def test_detector_no_fire_when_engine_quiet(monkeypatch): + _fake_audio(monkeypatch) + calls = [] + det = ww.WakeWordDetector(_FakeEngine(fire=False), lambda: calls.append(1)) + det.start() + time.sleep(0.15) + det.stop() + assert calls == [] + + +def test_detector_pause_resume(monkeypatch): + _fake_audio(monkeypatch) + det = ww.WakeWordDetector(_FakeEngine(fire=False), lambda: None) + det.start() + time.sleep(0.05) + assert det.running is True + det.pause() + assert det.running is False + det.resume() + time.sleep(0.05) + assert det.running is True + det.stop() + assert det.running is False + + +# ── Singleton lifecycle ────────────────────────────────────────────────── + + +def test_singleton_lifecycle(monkeypatch): + _fake_audio(monkeypatch) + monkeypatch.setattr(ww, "_build_engine", lambda cfg: _FakeEngine(fire=False)) + + assert ww.is_listening() is False + det = ww.start_listening(lambda: None, config={}) + time.sleep(0.05) + assert ww.is_listening() is True + + # Re-entrant start returns the same detector and re-arms it. + det2 = ww.start_listening(lambda: None, config={}) + assert det2 is det + + ww.pause_listening() + assert ww.is_listening() is False + ww.resume_listening() + time.sleep(0.05) + assert ww.is_listening() is True + + ww.stop_listening() + assert ww.is_listening() is False diff --git a/tools/lazy_deps.py b/tools/lazy_deps.py index 0cabac21b2d53..d92ab158b589e 100644 --- a/tools/lazy_deps.py +++ b/tools/lazy_deps.py @@ -137,6 +137,21 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { "numpy==2.4.3", ), + # ─── Wake word ("Hey Hermes") engines ────────────────────────────────── + # Keep in sync with the `wake` extra in pyproject.toml. openWakeWord is the + # free, local default (ONNX runtime); Porcupine is the premium engine. + "wake.openwakeword": ( + "openwakeword==0.6.0", + "onnxruntime==1.27.0", + "sounddevice==0.5.5", + "numpy==2.4.3", + ), + "wake.porcupine": ( + "pvporcupine==4.0.3", + "sounddevice==0.5.5", + "numpy==2.4.3", + ), + # ─── Image generation backends ───────────────────────────────────────── "image.fal": ("fal-client==0.13.1",), diff --git a/tools/wake_word.py b/tools/wake_word.py new file mode 100644 index 0000000000000..6dc473b211c8e --- /dev/null +++ b/tools/wake_word.py @@ -0,0 +1,431 @@ +"""Wake-word ("Hey Hermes") detection — hands-free session trigger for the CLI. + +A lightweight, always-on hotword listener that fires a callback when a wake +phrase is spoken — the "Hey Siri" / "Alexa" pattern. The CLI uses it to start a +fresh voice session without touching the keyboard: say the wake word, Hermes +opens the mic, captures one utterance via the existing voice pipeline, and +answers. + +Two engines, both fully on-device (no audio leaves the machine for detection): + +* **openwakeword** (default, free, no API key) — loads a pretrained or custom + ONNX model. Ships with ``hey_jarvis``, ``alexa``, ``hey_mycroft``, … ; point + ``wake_word.openwakeword.model`` at a custom ``.onnx`` to detect a real + "hey hermes" (training guide in the wake-word docs). +* **porcupine** (premium) — Picovoice's engine. Needs ``PORCUPINE_ACCESS_KEY``; + supports built-in keywords and custom ``.ppn`` files from the Picovoice + Console. + +Audio capture reuses the same 16 kHz mono int16 ``sounddevice`` path as voice +mode. The detector runs on its own daemon thread; callers ``pause()`` it while a +voice turn holds the microphone and ``resume()`` it once the system is idle +again (two input streams on one device is unreliable cross-platform). + +Nothing here mutates agent context or the prompt cache — on wake we hand a plain +string to the caller, exactly like a voice transcript. +""" + +from __future__ import annotations + +import logging +import os +import threading +import time +from typing import Any, Callable, Dict, Optional + +logger = logging.getLogger(__name__) + +# 16 kHz mono int16 — Whisper-native and what both engines expect. +SAMPLE_RATE = 16000 + +# Minimum gap between two consecutive wake fires, so one "hey hermes" can't +# retrigger across several frames while the caller is still reacting. +_FIRE_COOLDOWN_SECONDS = 2.0 + + +# --------------------------------------------------------------------------- +# Config +# --------------------------------------------------------------------------- + +_DEFAULTS: Dict[str, Any] = { + "enabled": False, + "provider": "openwakeword", + "phrase": "hey jarvis", + "sensitivity": 0.5, + "start_new_session": True, +} + + +def load_wake_word_config() -> Dict[str, Any]: + """Return the ``wake_word`` config section, shape-guarded to a dict.""" + try: + from hermes_cli.config import load_config + + cfg = load_config().get("wake_word") + except Exception: + cfg = None + return cfg if isinstance(cfg, dict) else {} + + +def _get(cfg: Dict[str, Any], key: str) -> Any: + val = cfg.get(key, _DEFAULTS.get(key)) + return _DEFAULTS.get(key) if val is None else val + + +def _provider(cfg: Dict[str, Any]) -> str: + return str(_get(cfg, "provider")).strip().lower() or "openwakeword" + + +def _sensitivity(cfg: Dict[str, Any]) -> float: + raw = _get(cfg, "sensitivity") + try: + s = float(raw) + except (TypeError, ValueError): + s = 0.5 + return min(max(s, 0.0), 1.0) + + +def wake_phrase(cfg: Optional[Dict[str, Any]] = None) -> str: + """Human-facing wake phrase label (purely cosmetic; engine keys detection).""" + cfg = cfg if cfg is not None else load_wake_word_config() + return str(_get(cfg, "phrase")) or "hey jarvis" + + +# --------------------------------------------------------------------------- +# Audio capture (lazy — never import sounddevice at module load) +# --------------------------------------------------------------------------- + +def _import_audio(): + import numpy as np + import sounddevice as sd + + return sd, np + + +def _audio_available() -> bool: + try: + _import_audio() + return True + except (ImportError, OSError): + return False + + +# --------------------------------------------------------------------------- +# Engines +# --------------------------------------------------------------------------- + +class _Engine: + """Minimal hotword-engine contract: feed int16 frames, get a bool.""" + + frame_length: int = 1280 # 80 ms at 16 kHz + + def process(self, frame) -> bool: # frame: 1-D int16 ndarray + raise NotImplementedError + + def close(self) -> None: + pass + + +def _looks_like_path(value: str) -> bool: + return ( + os.sep in value + or value.endswith((".onnx", ".tflite", ".ppn")) + or os.path.exists(value) + ) + + +class _OpenWakeWordEngine(_Engine): + """openWakeWord — free, local ONNX hotword detection.""" + + # openWakeWord recommends 80 ms frames (1280 samples) for efficiency. + frame_length = 1280 + + def __init__(self, cfg: Dict[str, Any]): + from tools import lazy_deps + + lazy_deps.ensure("wake.openwakeword", prompt=False) + + import openwakeword + from openwakeword.model import Model + + sub = cfg.get("openwakeword") if isinstance(cfg.get("openwakeword"), dict) else {} + model_ref = str(sub.get("model") or "hey_jarvis").strip() + framework = str(sub.get("inference_framework") or "onnx").strip().lower() + self._threshold = _sensitivity(cfg) + + if _looks_like_path(model_ref): + models = [model_ref] + else: + # Pretrained name (e.g. "hey_jarvis"). Best-effort one-time fetch + # of the bundled models; harmless if already present / offline. + try: + openwakeword.utils.download_models([model_ref]) + except Exception as e: # pragma: no cover - network/path dependent + logger.debug("openwakeword model download skipped: %s", e) + models = [model_ref] + + self._model = Model(wakeword_models=models, inference_framework=framework) + self._labels = list(self._model.models.keys()) + + def process(self, frame) -> bool: + scores = self._model.predict(frame) + return any(score >= self._threshold for score in scores.values()) + + def close(self) -> None: + try: + self._model.reset() + except Exception: + pass + + +class _PorcupineEngine(_Engine): + """Picovoice Porcupine — premium, on-device, needs an access key.""" + + def __init__(self, cfg: Dict[str, Any]): + from tools import lazy_deps + + lazy_deps.ensure("wake.porcupine", prompt=False) + + import pvporcupine + + access_key = (os.getenv("PORCUPINE_ACCESS_KEY") or "").strip() + if not access_key: + raise RuntimeError( + "Porcupine wake word requires PORCUPINE_ACCESS_KEY " + "(get a free key at https://console.picovoice.ai)." + ) + + sub = cfg.get("porcupine") if isinstance(cfg.get("porcupine"), dict) else {} + keyword = str(sub.get("keyword") or "jarvis").strip() + sensitivity = _sensitivity(cfg) + + kwargs: Dict[str, Any] = {"access_key": access_key, "sensitivities": [sensitivity]} + if _looks_like_path(keyword): + kwargs["keyword_paths"] = [keyword] + else: + kwargs["keywords"] = [keyword] + + self._porcupine = pvporcupine.create(**kwargs) + self.frame_length = self._porcupine.frame_length + + def process(self, frame) -> bool: + # pvporcupine wants a plain list/sequence of int16 samples. + return self._porcupine.process(frame) >= 0 + + def close(self) -> None: + try: + self._porcupine.delete() + except Exception: + pass + + +def _build_engine(cfg: Dict[str, Any]) -> _Engine: + provider = _provider(cfg) + if provider == "porcupine": + return _PorcupineEngine(cfg) + if provider in ("openwakeword", "oww", "local"): + return _OpenWakeWordEngine(cfg) + raise ValueError(f"Unknown wake_word provider: {provider!r}") + + +# --------------------------------------------------------------------------- +# Requirements probe (for /wake status + enable path) +# --------------------------------------------------------------------------- + +def check_wake_word_requirements(cfg: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: + """Report whether wake-word detection can run, with a remediation hint.""" + cfg = cfg if cfg is not None else load_wake_word_config() + provider = _provider(cfg) + from tools import lazy_deps + + feature = "wake.porcupine" if provider == "porcupine" else "wake.openwakeword" + deps_ok = lazy_deps.is_available(feature) + audio_ok = _audio_available() + key_ok = True + hint = "" + + if provider == "porcupine" and not (os.getenv("PORCUPINE_ACCESS_KEY") or "").strip(): + key_ok = False + hint = "Set PORCUPINE_ACCESS_KEY (free key at https://console.picovoice.ai)." + elif not deps_ok: + hint = lazy_deps.feature_install_command(feature) or "" + elif not audio_ok: + hint = "Microphone capture needs sounddevice + numpy and a working audio device." + + return { + "available": audio_ok and (deps_ok or lazy_deps._allow_lazy_installs()) and key_ok, + "provider": provider, + "deps_available": deps_ok, + "audio_available": audio_ok, + "access_key_set": key_ok, + "phrase": wake_phrase(cfg), + "hint": hint, + } + + +# --------------------------------------------------------------------------- +# Detector +# --------------------------------------------------------------------------- + +class WakeWordDetector: + """Background hotword listener. Fires ``on_wake()`` when the phrase is heard. + + The engine is built once and kept alive across pause/resume; only the audio + stream + reader thread cycle, so toggling the mic for a voice turn is cheap. + """ + + def __init__(self, engine: _Engine, on_wake: Callable[[], None], + cooldown: float = _FIRE_COOLDOWN_SECONDS): + self.engine = engine + self.on_wake = on_wake + self.cooldown = cooldown + self._thread: Optional[threading.Thread] = None + self._stop = threading.Event() + self._last_fire = 0.0 + self._lock = threading.Lock() + + @property + def running(self) -> bool: + t = self._thread + return t is not None and t.is_alive() + + def start(self) -> None: + """Open the mic and begin listening. Idempotent.""" + with self._lock: + if self._thread is not None and self._thread.is_alive(): + return + self._stop.clear() + self._thread = threading.Thread( + target=self._run, daemon=True, name="wake-word" + ) + self._thread.start() + + # pause/resume keep the engine; stop tears it down. + def pause(self) -> None: + self._halt_thread() + + def resume(self) -> None: + self.start() + + def stop(self) -> None: + self._halt_thread() + self.engine.close() + + def _halt_thread(self) -> None: + with self._lock: + t, self._thread = self._thread, None + if t is not None and t is not threading.current_thread(): + self._stop.set() + t.join(timeout=2.0) + + def _run(self) -> None: + try: + sd, np = _import_audio() + except (ImportError, OSError) as e: + logger.error("wake word: audio libraries unavailable: %s", e) + return + + frame_length = self.engine.frame_length + try: + stream = sd.InputStream( + samplerate=SAMPLE_RATE, + channels=1, + dtype="int16", + blocksize=frame_length, + ) + stream.start() + except Exception as e: + logger.error("wake word: failed to open microphone: %s", e) + return + + logger.debug("wake word: listening (frame=%d)", frame_length) + try: + while not self._stop.is_set(): + try: + data, _overflow = stream.read(frame_length) + except Exception as e: + logger.debug("wake word: stream read error: %s", e) + break + frame = data[:, 0] if getattr(data, "ndim", 1) == 2 else data + try: + fired = self.engine.process(frame) + except Exception as e: + logger.debug("wake word: engine error: %s", e) + continue + if fired: + now = time.monotonic() + if now - self._last_fire >= self.cooldown: + self._last_fire = now + try: + self.on_wake() + except Exception as e: + logger.warning("wake word callback failed: %s", e) + finally: + try: + stream.stop() + stream.close() + except Exception: + pass + logger.debug("wake word: stream closed") + + +# --------------------------------------------------------------------------- +# Process-wide singleton (mirrors hermes_cli.voice's continuous API) +# --------------------------------------------------------------------------- + +_detector: Optional[WakeWordDetector] = None +_detector_lock = threading.Lock() + + +def start_listening( + on_wake: Callable[[], None], + *, + config: Optional[Dict[str, Any]] = None, +) -> WakeWordDetector: + """Build (once) and start the wake-word detector. Idempotent. + + Raises if engine construction fails (missing deps / access key / model); + callers should probe :func:`check_wake_word_requirements` first. + """ + global _detector + with _detector_lock: + if _detector is not None: + _detector.on_wake = on_wake + _detector.resume() + return _detector + cfg = config if config is not None else load_wake_word_config() + engine = _build_engine(cfg) + _detector = WakeWordDetector(engine, on_wake) + _detector.start() + return _detector + + +def pause_listening() -> None: + """Release the microphone without tearing down the engine.""" + with _detector_lock: + det = _detector + if det is not None: + det.pause() + + +def resume_listening() -> None: + """Re-open the microphone after a pause. No-op if not initialised.""" + with _detector_lock: + det = _detector + if det is not None: + det.resume() + + +def stop_listening() -> None: + """Fully stop and discard the detector (closes the engine).""" + global _detector + with _detector_lock: + det, _detector = _detector, None + if det is not None: + det.stop() + + +def is_listening() -> bool: + with _detector_lock: + det = _detector + return det is not None and det.running diff --git a/website/docs/user-guide/features/overview.md b/website/docs/user-guide/features/overview.md index 75f365b38924e..cb3eef2210991 100644 --- a/website/docs/user-guide/features/overview.md +++ b/website/docs/user-guide/features/overview.md @@ -32,6 +32,7 @@ Hermes Agent includes a rich set of capabilities that extend far beyond basic ch ## Media & Web - **[Voice Mode](voice-mode.md)** — Full voice interaction across CLI and messaging platforms. Talk to the agent using your microphone, hear spoken replies, and have live voice conversations in Discord voice channels. +- **[Wake Word](wake-word.md)** — Hands-free "Hey Hermes" trigger for the CLI. An on-device hotword listener starts a fresh voice session when you speak the wake phrase, the "Hey Siri" way. - **[Browser Automation](browser.md)** — Full browser automation with multiple backends: Browserbase cloud, Browser Use cloud, local Chrome/Brave/Chromium/Edge via CDP, or local Chromium. Navigate websites, fill forms, and extract information. - **[Vision & Image Paste](vision.md)** — Multimodal vision support. Paste images from your clipboard into the CLI and ask the agent to analyze, describe, or work with them using any vision-capable model. - **[Image Generation](image-generation.md)** — Generate images from text prompts using FAL.ai. Eleven models supported (FLUX 2 Klein/Pro, GPT-Image 1.5/2, Nano Banana Pro, Ideogram V3, Recraft V4 Pro, Qwen, Z-Image Turbo, Krea V2 Medium/Large); pick one via `hermes tools`. diff --git a/website/docs/user-guide/features/wake-word.md b/website/docs/user-guide/features/wake-word.md new file mode 100644 index 0000000000000..21d1456cb07ef --- /dev/null +++ b/website/docs/user-guide/features/wake-word.md @@ -0,0 +1,148 @@ +--- +sidebar_position: 11 +title: "Wake Word" +description: "Hands-free 'Hey Hermes' wake word — start a voice session by speaking, the 'Hey Siri' way" +--- + +# Wake Word ("Hey Hermes") + +The wake word turns Hermes into a hands-free assistant in the CLI: with one +setting on, Hermes listens in the background for a spoken trigger phrase. Say it, +and Hermes starts a fresh session, opens the microphone, captures your command +via the normal [voice pipeline](/user-guide/features/voice-mode), and answers — +exactly like "Hey Siri" or "Alexa". + +Detection runs **entirely on-device**. The always-on listener only watches for +the wake phrase; no audio leaves your machine until you actually speak a command +to the agent. + +## How it works + +1. With `wake_word.enabled: true` (or after `/wake on`), a lightweight hotword + detector listens on your default microphone. +2. When it hears the wake phrase it pauses itself (freeing the mic), starts a new + session, and records one utterance with voice mode's silence detection. +3. Your speech is transcribed and sent to the agent. After it replies, the + listener resumes automatically and waits for the next wake word. + +It is **off by default** — nothing listens until you turn it on. + +## Engines + +| Engine | Cost | API key | Notes | +|--------|------|---------|-------| +| **openWakeWord** (default) | Free | None | Local ONNX models. Ships with `hey_jarvis`, `alexa`, `hey_mycroft`, … | +| **Porcupine** | Free tier / paid | `PORCUPINE_ACCESS_KEY` | Picovoice engine; built-in keywords + custom `.ppn` files | + +Both are lazy-installed the first time you enable the wake word. To install ahead +of time: + +```bash +uv pip install 'hermes-agent[wake]' # or: pip install 'hermes-agent[wake]' +``` + +## Quick start + +```bash +# In an interactive `hermes` session: +/wake on # start listening (installs the engine on first use) +/wake status # show phrase, provider, and state +/wake off # stop listening +``` + +Or enable it permanently in `~/.hermes/config.yaml`: + +```yaml +wake_word: + enabled: true +``` + +## Configuration + +```yaml +wake_word: + enabled: false + provider: openwakeword # "openwakeword" (free, local) | "porcupine" + phrase: "hey jarvis" # cosmetic label only — detection is keyed by the model/keyword below + sensitivity: 0.5 # 0.0-1.0 — raise to reduce false triggers + start_new_session: true # start a fresh session on wake vs. continue the current one + openwakeword: + model: hey_jarvis # built-in name OR path to a custom .onnx/.tflite + inference_framework: onnx # "onnx" | "tflite" + porcupine: + keyword: jarvis # built-in keyword OR path to a custom .ppn +``` + +`sensitivity`, `phrase`, and `start_new_session` apply to both engines. The +`openwakeword` and `porcupine` blocks select the actual detection model. + +## Using a real "Hey Hermes" + +The bundled openWakeWord models do **not** include "hey hermes" — `hey_jarvis` +is the free, instantly-working default. To detect the literal phrase you supply +your own model and point the config at it: + +### Option A — openWakeWord (free) + +Train a custom model (≈75–90 min on a free/Colab GPU), then drop the `.onnx` +file somewhere and reference it: + +```yaml +wake_word: + enabled: true + provider: openwakeword + phrase: "hey hermes" + openwakeword: + model: ~/.hermes/wakewords/hey_hermes.onnx +``` + +Training references: + +- openWakeWord — +- 2026 training Colab — + +:::tip Pick a distinctive phrase +Wake phrases that don't collide with everyday speech generalize best. Two +syllables with an uncommon word ("hermes" qualifies) beat common words like +"hello" or "stop". +::: + +### Option B — Porcupine (custom keyword in seconds) + +Create a "Hey Hermes" keyword in the [Picovoice Console](https://console.picovoice.ai/), +download the `.ppn`, and: + +```yaml +wake_word: + enabled: true + provider: porcupine + phrase: "hey hermes" + porcupine: + keyword: ~/.hermes/wakewords/hey_hermes.ppn +``` + +Set your access key in `~/.hermes/.env`: + +```bash +PORCUPINE_ACCESS_KEY=your-key-here +``` + +## Requirements + +- A working microphone and the `sounddevice` + `numpy` audio stack (shared with + voice mode). +- An STT provider for transcribing the spoken command — local `faster-whisper` + works out of the box; see [Voice Mode](/user-guide/features/voice-mode) for the + full provider list. +- The wake engine deps (auto-installed, or `hermes-agent[wake]`). + +`/wake status` reports exactly what's missing if the listener won't start. + +## Notes & limits + +- **CLI only.** The wake word lives in the interactive `hermes` CLI, where a + local microphone is available. It does not run in the messaging gateway. +- **One mic at a time.** The detector releases the microphone while a command is + recording and reclaims it once the turn ends, so it won't fight voice capture. +- **Privacy.** Hotword detection is local. Set `sensitivity` higher if you get + false triggers, lower if it misses you.