fix(wake): route desktop control and select input devices

This commit is contained in:
Gille
2026-07-29 14:04:21 -06:00
parent 70411a6152
commit 94d1dff50d
11 changed files with 523 additions and 26 deletions
@@ -22,6 +22,7 @@ import {
setSessions setSessions
} from '@/store/session' } from '@/store/session'
import { dropSessionState, publishSessionState } from '@/store/session-states' import { dropSessionState, publishSessionState } from '@/store/session-states'
import { $wakeWord, resetWakeWordState } from '@/store/wake-word'
import type { SessionInfo } from '@/types/hermes' import type { SessionInfo } from '@/types/hermes'
import type { SubmitTextOptions } from './utils' import type { SubmitTextOptions } from './utils'
@@ -427,6 +428,110 @@ describe('usePromptActions slash session targeting', () => {
}) })
}) })
describe('usePromptActions /wake', () => {
beforeEach(() => {
setSessions(() => [sessionInfo()])
resetWakeWordState()
})
afterEach(() => {
cleanup()
resetWakeWordState()
vi.restoreAllMocks()
})
it('starts the GUI-owned listener through wake.start and never spawns the slash worker', async () => {
const seeds: Record<string, unknown>[] = []
const requestGateway = vi.fn(async (method: string, _params?: Record<string, unknown>, _timeoutMs?: number) => {
if (method === 'wake.start') {
return {
owner_surface: 'gui',
phrase: 'hey hermes',
provider: 'openwakeword',
started: true
} as never
}
if (method === 'wake.status') {
return {
available: true,
configured_surface: 'gui',
enabled: true,
input_device: {
hostapi: 'Windows WASAPI',
name: 'Microphone Array',
selector: 'Microphone Array'
},
listening: true,
owner_surface: 'gui',
phrase: 'hey hermes',
provider: 'openwakeword'
} as never
}
return {} as never
})
let handle: HarnessHandle | null = null
await actRender(
<Harness
onReady={h => (handle = h)}
onSeedState={state => seeds.push(state)}
refreshSessions={async () => undefined}
requestGateway={requestGateway}
/>
)
await handle!.submitText('/wake on')
expect(requestGateway).toHaveBeenCalledWith('wake.start', { persist: true, surface: 'gui' }, 180_000)
expect(requestGateway).toHaveBeenCalledWith('wake.status', {})
expect(requestGateway).not.toHaveBeenCalledWith('slash.exec', expect.anything())
expect(requestGateway).not.toHaveBeenCalledWith('command.dispatch', expect.anything())
expect($wakeWord.get()).toMatchObject({ available: true, enabled: true, listening: true })
expect(renderedSeedTexts(seeds).join('\n')).toContain('Input: Microphone Array (Windows WASAPI)')
})
it('uses gateway truth for a bare toggle and stops through wake.stop', async () => {
let statusCalls = 0
const requestGateway = vi.fn(async (method: string) => {
if (method === 'wake.status') {
statusCalls += 1
return {
available: true,
enabled: statusCalls === 1,
listening: statusCalls === 1,
owner_surface: statusCalls === 1 ? 'gui' : null,
phrase: 'hey hermes',
provider: 'openwakeword'
} as never
}
if (method === 'wake.stop') {
return { disabled_persisted: true, stopped: true } as never
}
return {} as never
})
let handle: HarnessHandle | null = null
await actRender(
<Harness onReady={h => (handle = h)} refreshSessions={async () => undefined} requestGateway={requestGateway} />
)
await handle!.submitText('/wake')
expect(requestGateway.mock.calls.map(([method]) => method)).toEqual(['wake.status', 'wake.stop', 'wake.status'])
expect(requestGateway).toHaveBeenCalledWith('wake.stop', { persist: true })
expect(requestGateway).not.toHaveBeenCalledWith('slash.exec', expect.anything())
expect(requestGateway).not.toHaveBeenCalledWith('command.dispatch', expect.anything())
expect($wakeWord.get()).toMatchObject({ enabled: false, listening: false })
})
})
describe('usePromptActions /compress', () => { describe('usePromptActions /compress', () => {
beforeEach(() => { beforeEach(() => {
setSessions(() => [sessionInfo()]) setSessions(() => [sessionInfo()])
@@ -36,6 +36,15 @@ import {
setYoloActive setYoloActive
} from '@/store/session' } from '@/store/session'
import { $sessionStates } from '@/store/session-states' import { $sessionStates } from '@/store/session-states'
import {
applyWakeStartResult,
applyWakeStatus,
applyWakeStopResult,
type WakeInputDeviceStatus,
type WakeStartResponse,
type WakeStatusResponse,
type WakeStopResponse
} from '@/store/wake-word'
import type { import type {
BrowserManageResponse, BrowserManageResponse,
@@ -60,6 +69,43 @@ import {
// default WS request timeout on large sessions — give it the TUI client's // default WS request timeout on large sessions — give it the TUI client's
// 120s RPC budget (HERMES_TUI_RPC_TIMEOUT_MS default) instead. // 120s RPC budget (HERMES_TUI_RPC_TIMEOUT_MS default) instead.
const SESSION_COMPRESS_TIMEOUT_MS = 120_000 const SESSION_COMPRESS_TIMEOUT_MS = 120_000
const WAKE_START_TIMEOUT_MS = 180_000
const wakeDeviceLabel = (device?: WakeInputDeviceStatus): string => {
if (!device) {
return 'system default'
}
const selector = device.selector
const name = device.name?.trim() || (selector == null ? 'system default' : String(selector))
return device.hostapi?.trim() ? `${name} (${device.hostapi.trim()})` : name
}
const renderWakeStatus = (status: WakeStatusResponse): string => {
const lines = [
'Wake Word Status',
`State: ${status.listening ? 'LISTENING' : 'OFF'}`,
`Phrase: "${status.phrase?.trim() || 'hey hermes'}"`,
`Provider: ${status.provider?.trim() || 'unknown'}`,
`Surface: ${status.owner_surface?.trim() || status.configured_surface?.trim() || 'auto'}`,
`Input: ${wakeDeviceLabel(status.input_device)}`
]
if (status.audio_silent) {
lines.push('Audio: silent')
}
if (status.input_device?.error?.trim()) {
lines.push(`Input error: ${status.input_device.error.trim()}`)
}
if (status.hint?.trim()) {
lines.push(`Hint: ${status.hint.trim()}`)
}
return lines.join('\n')
}
/** Everything a slash handler needs about the invocation it's serving. */ /** Everything a slash handler needs about the invocation it's serving. */
interface SlashActionCtx { interface SlashActionCtx {
@@ -592,6 +638,67 @@ export function useSlashCommand(deps: SlashCommandDeps) {
notify({ kind: 'error', title: copy.yoloTitle, message: copy.yoloToggleFailed }) notify({ kind: 'error', title: copy.yoloTitle, message: copy.yoloToggleFailed })
} }
}, },
// /wake must stay in the gateway process that owns the Desktop wake
// lease. Sending it through slash.exec creates a separate HermesCLI in
// the slash worker, which can claim the machine-wide microphone lock
// while the Desktop UI still reports the GUI listener as off.
wake: async ctx => {
const resolved = await withSlashOutput(ctx)
if (!resolved) {
return
}
const { render: renderSlashOutput } = resolved
const requested = ctx.arg.trim().toLowerCase()
if (requested && !['on', 'off', 'status'].includes(requested)) {
renderSlashOutput('usage: /wake [on|off|status]')
return
}
const status = async (): Promise<WakeStatusResponse> => {
const current = await requestGateway<WakeStatusResponse>('wake.status', {})
applyWakeStatus(current)
return current
}
try {
let action = requested
// Bare /wake is an authoritative toggle. Query the gateway instead
// of trusting a potentially stale renderer cache.
if (!action) {
action = (await status()).listening ? 'off' : 'on'
}
if (action === 'on') {
const started = await requestGateway<WakeStartResponse>(
'wake.start',
{ persist: true, surface: 'gui' },
WAKE_START_TIMEOUT_MS
)
applyWakeStartResult(started)
if (!started?.started) {
renderSlashOutput(
`Failed to start wake word: ${started?.hint?.trim() || started?.reason?.trim() || 'unknown error'}`
)
return
}
} else if (action === 'off') {
applyWakeStopResult(await requestGateway<WakeStopResponse>('wake.stop', { persist: true }))
}
renderSlashOutput(renderWakeStatus(await status()))
} catch (err) {
renderSlashOutput(`error: ${err instanceof Error ? err.message : String(err)}`)
}
},
// /handoff hands this session to a messaging platform. The platform is // /handoff hands this session to a messaging platform. The platform is
// completed inline in the slash popover (backend _handoff_completions), // completed inline in the slash popover (backend _handoff_completions),
// so there is no overlay: `/handoff <platform>` runs the desktop's own // so there is no overlay: `/handoff <platform>` runs the desktop's own
@@ -75,6 +75,14 @@ describe('desktop slash command curation', () => {
expect(isDesktopSlashCommand('/pets')).toBe(false) expect(isDesktopSlashCommand('/pets')).toBe(false)
}) })
it('routes /wake through the desktop wake action instead of the slash worker', () => {
expect(resolveDesktopCommand('/wake')?.surface).toEqual({ kind: 'action', action: 'wake' })
expect(desktopSlashCommandArgumentMode('/wake')).toBe('options')
expect(isDesktopSlashSuggestion('/wake')).toBe(true)
expect(isDesktopSlashCommand('/wake')).toBe(true)
expect(desktopSlashUnavailableMessage('/wake')).toBeNull()
})
it('treats /browser as an executable action command (local-gateway connect)', () => { it('treats /browser as an executable action command (local-gateway connect)', () => {
// /browser used to be terminal-only; it now resolves to a desktop action // /browser used to be terminal-only; it now resolves to a desktop action
// handler that routes browser.manage RPC when the gateway is local. // handler that routes browser.manage RPC when the gateway is local.
@@ -56,6 +56,7 @@ export type DesktopActionId =
| 'profile' | 'profile'
| 'skin' | 'skin'
| 'title' | 'title'
| 'wake'
| 'yolo' | 'yolo'
/** A command fulfilled by opening a desktop overlay picker. */ /** A command fulfilled by opening a desktop overlay picker. */
@@ -168,6 +169,12 @@ const DESKTOP_COMMAND_SPECS: readonly DesktopCommandSpec[] = [
surface: action('branch') surface: action('branch')
}, },
{ name: '/yolo', description: 'Toggle YOLO — auto-approve dangerous commands', surface: action('yolo') }, { name: '/yolo', description: 'Toggle YOLO — auto-approve dangerous commands', surface: action('yolo') },
{
name: '/wake',
description: 'Control the desktop wake-word listener [on|off|status]',
surface: action('wake'),
argumentMode: 'options'
},
{ {
name: '/handoff', name: '/handoff',
description: 'Hand off this session to a messaging platform', description: 'Hand off this session to a messaging platform',
+15 -3
View File
@@ -34,12 +34,14 @@ const INITIAL_WAKE_WORD_STATE: WakeWordState = {
export const $wakeWord = atom<WakeWordState>(INITIAL_WAKE_WORD_STATE) export const $wakeWord = atom<WakeWordState>(INITIAL_WAKE_WORD_STATE)
export interface WakeStatusResponse { export interface WakeStatusResponse {
/** Armed but the mic delivers only silence (macOS backend-permission gap). */ /** Armed but the selected backend input delivers only silence. */
audio_silent?: boolean audio_silent?: boolean
available?: boolean available?: boolean
configured_surface?: string
/** Config truth (wake_word.enabled) — drives post-voice re-arm. */ /** Config truth (wake_word.enabled) — drives post-voice re-arm. */
enabled?: boolean enabled?: boolean
hint?: string hint?: string
input_device?: WakeInputDeviceStatus
listening?: boolean listening?: boolean
owned_by_caller?: boolean owned_by_caller?: boolean
owner_surface?: string | null owner_surface?: string | null
@@ -63,6 +65,16 @@ export interface WakeStopResponse {
stopped?: boolean stopped?: boolean
} }
export interface WakeInputDeviceStatus {
default_samplerate?: number
error?: string
hostapi?: string
hostapi_index?: number
max_input_channels?: number
name?: string
selector?: number | string | null
}
/** Minimal requester shape — satisfied by both `useGatewayRequest`'s /** Minimal requester shape — satisfied by both `useGatewayRequest`'s
* `requestGateway` and the `$gateway` instance wrapper below. */ * `requestGateway` and the `$gateway` instance wrapper below. */
export type WakeRequester = <T>(method: string, params?: Record<string, unknown>) => Promise<T> export type WakeRequester = <T>(method: string, params?: Record<string, unknown>) => Promise<T>
@@ -111,8 +123,8 @@ const noticeFrom = (result: { hint?: string; reason?: string | null } | null | u
export function applyWakeStatus(status: WakeStatusResponse | null | undefined): void { export function applyWakeStatus(status: WakeStatusResponse | null | undefined): void {
const current = $wakeWord.get() const current = $wakeWord.get()
const listening = Boolean(status?.listening) const listening = Boolean(status?.listening)
// "Armed but deaf" (macOS backend without mic permission) keeps its hint // "Armed but deaf" keeps its input-device hint visible in the tooltip even
// visible in the tooltip even though the toggle shows listening. // though the toggle shows listening.
const silent = Boolean(status?.audio_silent) const silent = Boolean(status?.audio_silent)
$wakeWord.set({ $wakeWord.set({
+1
View File
@@ -1464,6 +1464,7 @@ DEFAULT_CONFIG = {
"wake_word": { "wake_word": {
"enabled": False, "enabled": False,
"surface": "auto", # eligible surface: "auto" (first claimant) | "cli" | "tui" | "gui" "surface": "auto", # eligible surface: "auto" (first claimant) | "cli" | "tui" | "gui"
"input_device": None, # PortAudio input device index/name; null uses the process default
"provider": "openwakeword", # "openwakeword" (free, local) | "sherpa" (free, ANY phrase, no training) | "porcupine" (premium; needs PORCUPINE_ACCESS_KEY) "provider": "openwakeword", # "openwakeword" (free, local) | "sherpa" (free, ANY phrase, no training) | "porcupine" (premium; needs PORCUPINE_ACCESS_KEY)
"phrase": "hey hermes", # for "sherpa" this IS the detected phrase (any text works); for other engines it's a cosmetic label — detection is keyed by the model/keyword below "phrase": "hey hermes", # for "sherpa" this IS the detected phrase (any text works); for other engines it's a cosmetic label — detection is keyed by the model/keyword below
"sensitivity": 0.6, # 0.0-1.0 detection threshold, consistent across engines (higher = stricter, fewer false triggers) "sensitivity": 0.6, # 0.0-1.0 detection threshold, consistent across engines (higher = stricter, fewer false triggers)
+57
View File
@@ -1708,6 +1708,63 @@ def test_wake_toggle_persists_enabled_flag_only_on_explicit_gesture(monkeypatch)
server._wake_owner_surface = "" server._wake_owner_surface = ""
def test_wake_status_reports_configured_input_device_and_windows_silence_hint(monkeypatch):
from tools import wake_word
config = {
"enabled": True,
"phrase": "hey hermes",
"provider": "openwakeword",
"surface": "gui",
"input_device": "Microphone Array",
}
device = {
"selector": "Microphone Array",
"name": "Microphone Array",
"hostapi": "Windows WASAPI",
"default_samplerate": 48000.0,
}
transport = types.SimpleNamespace(_closed=False)
monkeypatch.setattr(wake_word, "load_wake_word_config", lambda: config)
monkeypatch.setattr(
wake_word,
"check_wake_word_requirements",
lambda cfg: {
"available": True,
"hint": "",
"phrase": "hey hermes",
"provider": "openwakeword",
},
)
monkeypatch.setattr(wake_word, "get_input_device_status", lambda cfg: device)
monkeypatch.setattr(wake_word, "owns_listener", lambda owner: owner is transport)
monkeypatch.setattr(wake_word, "is_listening", lambda: True)
monkeypatch.setattr(wake_word, "audio_is_silent", lambda: True)
monkeypatch.setattr(
wake_word,
"silent_audio_hint",
lambda details: f"silent input: {details['name']} ({details['hostapi']})",
)
server._wake_owner_transport = transport
server._wake_owner_surface = "gui"
try:
response = server.dispatch(
{"id": "wake-status", "method": "wake.status", "params": {}},
transport=transport,
)
assert response["result"]["configured_surface"] == "gui"
assert response["result"]["input_device"] == device
assert response["result"]["audio_silent"] is True
assert response["result"]["hint"] == (
"silent input: Microphone Array (Windows WASAPI)"
)
finally:
server._wake_owner_transport = None
server._wake_owner_surface = ""
def test_voice_record_start_forwards_max_recording_seconds(monkeypatch): def test_voice_record_start_forwards_max_recording_seconds(monkeypatch):
"""voice.max_recording_seconds must reach start_continuous from the TUI. """voice.max_recording_seconds must reach start_continuous from the TUI.
+59 -1
View File
@@ -25,6 +25,11 @@ import tools.wake_word as ww
def test_config_defaults_and_clamping(): def test_config_defaults_and_clamping():
assert ww._provider({}) == "openwakeword" assert ww._provider({}) == "openwakeword"
assert ww._provider({"provider": "Porcupine"}) == "porcupine" assert ww._provider({"provider": "Porcupine"}) == "porcupine"
assert ww._input_device({}) is None
assert ww._input_device({"input_device": 7}) == 7
assert ww._input_device({"input_device": " Microphone Array "}) == "Microphone Array"
assert ww._input_device({"input_device": ""}) is None
assert ww._input_device({"input_device": False}) is None
assert ww._sensitivity({"sensitivity": 5}) == 1.0 assert ww._sensitivity({"sensitivity": 5}) == 1.0
assert ww._sensitivity({"sensitivity": -1}) == 0.0 assert ww._sensitivity({"sensitivity": -1}) == 0.0
# Invalid input falls back to the configured default, not a hardcoded 0.5. # Invalid input falls back to the configured default, not a hardcoded 0.5.
@@ -842,8 +847,61 @@ class _LoudStream(_FakeStream):
return _Frame([500] * n), False return _Frame([500] * n), False
def test_detector_opens_configured_input_device_and_reports_backend(monkeypatch):
opened = []
def _stream(**kwargs):
opened.append(kwargs)
return _LoudStream(**kwargs)
fake_sd = types.SimpleNamespace(
InputStream=_stream,
query_devices=lambda selector, kind: {
"name": "Microphone Array",
"hostapi": 2,
"max_input_channels": 2,
"default_samplerate": 48000.0,
},
query_hostapis=lambda index: {"name": "Windows WASAPI"},
)
monkeypatch.setattr(ww, "_import_audio", lambda: (fake_sd, None))
det = ww.WakeWordDetector(
_FakeEngine(fire=False),
lambda: None,
input_device="Microphone Array",
)
det.start()
try:
assert opened[0]["device"] == "Microphone Array"
assert det.input_device_details == {
"selector": "Microphone Array",
"name": "Microphone Array",
"hostapi_index": 2,
"hostapi": "Windows WASAPI",
"max_input_channels": 2,
"default_samplerate": 48000.0,
}
finally:
det.stop()
def test_windows_silent_hint_names_selected_device(monkeypatch):
monkeypatch.setattr(ww.sys, "platform", "win32")
hint = ww.silent_audio_hint(
{
"selector": 3,
"name": "Microphone Array",
"hostapi": "Windows WASAPI",
}
)
assert "Microphone Array (Windows WASAPI)" in hint
assert "wake_word.input_device" in hint
assert "macOS" not in hint
def test_detector_flags_silent_stream_and_recovers(monkeypatch): def test_detector_flags_silent_stream_and_recovers(monkeypatch):
"""A stream of zeros sets audio_silent (macOS no-permission mode); audio clears it.""" """A stream of zeros sets audio_silent; audible input clears it."""
monkeypatch.setattr(ww, "_SILENCE_ALERT_SECONDS", 0.001) # trip on the first frame monkeypatch.setattr(ww, "_SILENCE_ALERT_SECONDS", 0.001) # trip on the first frame
stream_cls = {"cls": _SilentStream} stream_cls = {"cls": _SilentStream}
fake_sd = types.SimpleNamespace(InputStream=lambda **kw: stream_cls["cls"](**kw)) fake_sd = types.SimpleNamespace(InputStream=lambda **kw: stream_cls["cls"](**kw))
+126 -17
View File
@@ -57,10 +57,9 @@ _START_TIMEOUT_SECONDS = 5.0
_DEFAULT_CONFIRMATION_FRAMES = 3 _DEFAULT_CONFIRMATION_FRAMES = 3
# Dead-mic detection: an int16 stream whose peak stays at/below this for this # Dead-mic detection: an int16 stream whose peak stays at/below this for this
# many consecutive seconds is flagged as silent. macOS grants the *app* mic # many consecutive seconds is flagged as silent. Desktop push-to-talk and the
# permission per-process — a backend spawned without the entitlement gets a # backend listener use different capture paths, so one can work while the
# "working" CoreAudio stream that delivers zeros forever, so the listener # backend-selected stream is all zeros.
# looks armed but can never hear the phrase.
_SILENCE_PEAK = 10 _SILENCE_PEAK = 10
_SILENCE_ALERT_SECONDS = 10 _SILENCE_ALERT_SECONDS = 10
@@ -76,6 +75,7 @@ class WakeWordInUse(RuntimeError):
_DEFAULTS: Dict[str, Any] = { _DEFAULTS: Dict[str, Any] = {
"enabled": False, "enabled": False,
"surface": "auto", "surface": "auto",
"input_device": None,
"provider": "openwakeword", "provider": "openwakeword",
"phrase": "hey hermes", "phrase": "hey hermes",
"sensitivity": 0.6, "sensitivity": 0.6,
@@ -203,6 +203,17 @@ def _provider(cfg: Dict[str, Any]) -> str:
return str(_get(cfg, "provider")).strip().lower() or "openwakeword" return str(_get(cfg, "provider")).strip().lower() or "openwakeword"
def _input_device(cfg: Dict[str, Any]) -> int | str | None:
"""Configured PortAudio input selector, preserving indices and names."""
raw = _get(cfg, "input_device")
if raw is None or isinstance(raw, bool):
return None
if isinstance(raw, int):
return raw
value = str(raw).strip()
return value or None
def _sensitivity(cfg: Dict[str, Any]) -> float: def _sensitivity(cfg: Dict[str, Any]) -> float:
raw = _get(cfg, "sensitivity") raw = _get(cfg, "sensitivity")
try: try:
@@ -313,6 +324,71 @@ def _audio_available() -> bool:
return False return False
def _describe_input_device(sd, selector: int | str | None) -> Dict[str, Any]:
"""Resolve a PortAudio selector into JSON-safe diagnostics.
Device discovery is diagnostic only. ``InputStream`` remains the authority
on whether the selected device can actually open at the requested format.
"""
details: Dict[str, Any] = {"selector": selector}
try:
info = sd.query_devices(selector, "input")
except Exception as e:
details["error"] = str(e)
return details
if isinstance(info, dict):
name = info.get("name")
if name:
details["name"] = str(name)
channels = info.get("max_input_channels")
if isinstance(channels, (int, float)):
details["max_input_channels"] = int(channels)
rate = info.get("default_samplerate")
if isinstance(rate, (int, float)):
details["default_samplerate"] = float(rate)
hostapi_index = info.get("hostapi")
if isinstance(hostapi_index, (int, float)):
details["hostapi_index"] = int(hostapi_index)
try:
hostapi = sd.query_hostapis(int(hostapi_index))
hostapi_name = hostapi.get("name") if isinstance(hostapi, dict) else None
if hostapi_name:
details["hostapi"] = str(hostapi_name)
except Exception:
pass
return details
def _device_label(details: Dict[str, Any]) -> str:
name = str(details.get("name") or "").strip()
selector = details.get("selector")
label = name or ("system default" if selector is None else str(selector))
hostapi = str(details.get("hostapi") or "").strip()
return f"{label} ({hostapi})" if hostapi else label
def silent_audio_hint(details: Dict[str, Any]) -> str:
"""Platform-specific remediation for an armed stream delivering silence."""
if sys.platform == "darwin":
return (
"Microphone delivers only silence. Grant the Hermes backend "
"microphone access in System Settings > Privacy & Security > "
"Microphone, then toggle the wake word."
)
if sys.platform == "win32":
return (
f"Microphone delivers only silence from {_device_label(details)}. "
"Set wake_word.input_device to a different PortAudio input device, "
"then toggle the wake word."
)
return (
f"Microphone delivers only silence from {_device_label(details)}. "
"Check the selected input device, then toggle the wake word."
)
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Engines # Engines
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -798,20 +874,22 @@ class WakeWordDetector:
def __init__(self, engine: _Engine, on_wake: Callable[[], None], def __init__(self, engine: _Engine, on_wake: Callable[[], None],
cooldown: float = _FIRE_COOLDOWN_SECONDS, cooldown: float = _FIRE_COOLDOWN_SECONDS,
on_failure: Optional[Callable[["WakeWordDetector"], None]] = None): on_failure: Optional[Callable[["WakeWordDetector"], None]] = None,
input_device: int | str | None = None):
self.engine = engine self.engine = engine
self.on_wake = on_wake self.on_wake = on_wake
self.cooldown = cooldown self.cooldown = cooldown
self.on_failure = on_failure self.on_failure = on_failure
self.input_device = input_device
self.input_device_details: Dict[str, Any] = {"selector": input_device}
self._thread: Optional[threading.Thread] = None self._thread: Optional[threading.Thread] = None
self._stop = threading.Event() self._stop = threading.Event()
self._callback_inflight = threading.Event() self._callback_inflight = threading.Event()
self._last_fire = 0.0 self._last_fire = 0.0
self._lock = threading.Lock() self._lock = threading.Lock()
# True when the stream is open but every frame is (near-)silence — the # True when the stream is open but every frame is (near-)silence.
# classic macOS symptom of a backend process without mic permission: # Surfaced via wake.status / /wake status so users can tell "armed"
# CoreAudio "succeeds" and delivers zeros forever. Surfaced via # from "deaf".
# wake.status / /wake status so users can tell "armed" from "deaf".
self.audio_silent = False self.audio_silent = False
self._silent_frames = 0 self._silent_frames = 0
@@ -881,8 +959,19 @@ class WakeWordDetector:
return return
frame_length = self.engine.frame_length frame_length = self.engine.frame_length
self.input_device_details = _describe_input_device(sd, self.input_device)
logger.info(
"wake word: opening microphone device=%s selector=%r hostapi=%s "
"default_rate=%s requested_rate=%d",
self.input_device_details.get("name") or "system default",
self.input_device,
self.input_device_details.get("hostapi") or "unknown",
self.input_device_details.get("default_samplerate") or "unknown",
SAMPLE_RATE,
)
try: try:
stream = sd.InputStream( stream = sd.InputStream(
device=self.input_device,
samplerate=SAMPLE_RATE, samplerate=SAMPLE_RATE,
channels=1, channels=1,
dtype="int16", dtype="int16",
@@ -907,7 +996,7 @@ class WakeWordDetector:
ready.set() ready.set()
failed = False failed = False
# ~seconds of consecutive near-zero frames before we flag the stream # ~seconds of consecutive near-zero frames before we flag the stream
# as silent (macOS no-permission streams deliver zeros forever). # as silent.
silent_alert_frames = max(1, int(_SILENCE_ALERT_SECONDS * SAMPLE_RATE / max(1, frame_length))) silent_alert_frames = max(1, int(_SILENCE_ALERT_SECONDS * SAMPLE_RATE / max(1, frame_length)))
try: try:
while not self._stop.is_set(): while not self._stop.is_set():
@@ -927,10 +1016,9 @@ class WakeWordDetector:
if self._silent_frames == silent_alert_frames: if self._silent_frames == silent_alert_frames:
self.audio_silent = True self.audio_silent = True
logger.warning( logger.warning(
"wake word: mic delivers only silence (peak<=%d for %ds) — " "wake word: mic delivers only silence (peak<=%d for %ds); %s",
"on macOS check System Settings > Privacy & Security > "
"Microphone for the Hermes backend process",
_SILENCE_PEAK, _SILENCE_ALERT_SECONDS, _SILENCE_PEAK, _SILENCE_ALERT_SECONDS,
silent_audio_hint(self.input_device_details),
) )
elif self._silent_frames: elif self._silent_frames:
if self.audio_silent: if self.audio_silent:
@@ -1070,7 +1158,12 @@ def start_listening(
try: try:
cfg = config if config is not None else load_wake_word_config() cfg = config if config is not None else load_wake_word_config()
engine = _build_engine(cfg) engine = _build_engine(cfg)
detector = WakeWordDetector(engine, on_wake, on_failure=_detector_failed) detector = WakeWordDetector(
engine,
on_wake,
on_failure=_detector_failed,
input_device=_input_device(cfg),
)
_detector = detector _detector = detector
_detector_owner = owner _detector_owner = owner
_detector_file_lock = lock_handle _detector_file_lock = lock_handle
@@ -1139,15 +1232,31 @@ def is_listening() -> bool:
def audio_is_silent() -> bool: def audio_is_silent() -> bool:
"""True when the armed stream has delivered only silence (dead mic). """True when the armed stream has delivered only silence (dead mic).
The macOS no-permission failure mode: the stream opens fine but every The stream opens fine but every frame is zeros, so detection can never
frame is zeros, so detection can never fire. Lets status surfaces show fire. Lets status surfaces show "listening but the microphone appears
"listening but the microphone appears silent" instead of a healthy state. silent" instead of a healthy state.
""" """
with _detector_lock: with _detector_lock:
det = _detector det = _detector
return det is not None and det.audio_silent return det is not None and det.audio_silent
def get_input_device_status(cfg: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""Return configured/active PortAudio input diagnostics for status UIs."""
with _detector_lock:
det = _detector
if det is not None:
return dict(det.input_device_details)
cfg = cfg if cfg is not None else load_wake_word_config()
selector = _input_device(cfg)
try:
sd, _ = _import_audio()
except (ImportError, OSError) as e:
return {"selector": selector, "error": str(e)}
return _describe_input_device(sd, selector)
def get_last_match() -> Optional[tuple[str, str]]: def get_last_match() -> Optional[tuple[str, str]]:
"""(matched phrase, profile) of the most recent wake fire, if the engine """(matched phrase, profile) of the most recent wake fire, if the engine
reports per-phrase matches (sherpa multi-profile routing). None otherwise.""" reports per-phrase matches (sherpa multi-profile routing). None otherwise."""
+9 -4
View File
@@ -12616,9 +12616,11 @@ def _(rid, params: dict) -> dict:
from tools.wake_word import ( from tools.wake_word import (
audio_is_silent, audio_is_silent,
check_wake_word_requirements, check_wake_word_requirements,
get_input_device_status,
is_listening, is_listening,
load_wake_word_config, load_wake_word_config,
owns_listener, owns_listener,
silent_audio_hint,
) )
cfg = load_wake_word_config() cfg = load_wake_word_config()
reqs = check_wake_word_requirements(cfg) reqs = check_wake_word_requirements(cfg)
@@ -12627,23 +12629,26 @@ def _(rid, params: dict) -> dict:
owned_by_caller = owns_listener(transport) owned_by_caller = owns_listener(transport)
listening = owned_by_caller and is_listening() listening = owned_by_caller and is_listening()
silent = listening and audio_is_silent() silent = listening and audio_is_silent()
input_device = get_input_device_status(cfg)
hint = reqs.get("hint", "") hint = reqs.get("hint", "")
if input_device.get("error") and not hint:
hint = f"Wake-word input device could not be resolved: {input_device['error']}"
if silent and not hint: if silent and not hint:
hint = ("Microphone delivers only silence — on macOS grant the " hint = silent_audio_hint(input_device)
"Hermes backend mic access (System Settings > Privacy & "
"Security > Microphone), then toggle the wake word.")
return _ok(rid, { return _ok(rid, {
"listening": listening, "listening": listening,
"owned_by_caller": owned_by_caller, "owned_by_caller": owned_by_caller,
"owner_surface": owner_surface if owner is not None else None, "owner_surface": owner_surface if owner is not None else None,
"phrase": reqs["phrase"], "phrase": reqs["phrase"],
"provider": reqs["provider"], "provider": reqs["provider"],
"configured_surface": str(cfg.get("surface") or "auto"),
"input_device": input_device,
"available": reqs["available"], "available": reqs["available"],
"hint": hint, "hint": hint,
# Config truth: clients use this to re-arm after a voice turn # Config truth: clients use this to re-arm after a voice turn
# ("permanent on") without guessing from runtime listener state. # ("permanent on") without guessing from runtime listener state.
"enabled": bool(cfg.get("enabled")), "enabled": bool(cfg.get("enabled")),
# Armed but deaf (macOS permission failure mode) — see hint. # Armed but deaf despite an open stream; see platform-specific hint.
"audio_silent": silent, "audio_silent": silent,
}) })
except Exception as e: except Exception as e:
+29 -1
View File
@@ -20,7 +20,8 @@ to the agent.
## How it works ## How it works
1. With `wake_word.enabled: true` (or after `/wake on`), a lightweight hotword 1. With `wake_word.enabled: true` (or after `/wake on`), a lightweight hotword
detector listens on your default microphone. detector listens on your configured input device, or the process default
microphone when `wake_word.input_device` is unset.
2. When it hears the wake phrase it pauses itself (freeing the mic), starts a new 2. When it hears the wake phrase it pauses itself (freeing the mic), starts a new
session, and records one utterance with voice mode's silence detection. session, and records one utterance with voice mode's silence detection.
3. Your speech is transcribed and sent to the agent. After it replies, the 3. Your speech is transcribed and sent to the agent. After it replies, the
@@ -80,6 +81,7 @@ wake_word:
wake_word: wake_word:
enabled: false enabled: false
surface: auto # eligible surface: "auto" | "cli" | "tui" | "gui" surface: auto # eligible surface: "auto" | "cli" | "tui" | "gui"
input_device: null # PortAudio input index or device-name substring; null = process default
provider: openwakeword # "openwakeword" (free, local) | "sherpa" (free, any phrase) | "porcupine" provider: openwakeword # "openwakeword" (free, local) | "sherpa" (free, any phrase) | "porcupine"
phrase: "hey hermes" # cosmetic label only — detection is keyed by the model/keyword below phrase: "hey hermes" # cosmetic label only — detection is keyed by the model/keyword below
sensitivity: 0.6 # 0.0-1.0 — higher = stricter (fewer false triggers), consistent across all engines sensitivity: 0.6 # 0.0-1.0 — higher = stricter (fewer false triggers), consistent across all engines
@@ -95,6 +97,11 @@ wake_word:
`sensitivity`, `phrase`, and `start_new_session` apply to both engines. The `sensitivity`, `phrase`, and `start_new_session` apply to both engines. The
`openwakeword` and `porcupine` blocks select the actual detection model. `openwakeword` and `porcupine` blocks select the actual detection model.
`input_device` is passed directly to the wake listener's PortAudio
(`sounddevice`) stream. Use either a numeric device index or an unambiguous
device-name substring. This setting only changes wake-word capture; desktop
push-to-talk still uses the desktop application's microphone path.
### Reducing false triggers on ambient speech ### Reducing false triggers on ambient speech
openWakeWord scores one short (~80ms) audio frame at a time, so a stray phoneme openWakeWord scores one short (~80ms) audio frame at a time, so a stray phoneme
@@ -264,6 +271,27 @@ Fix: System Settings → Privacy & Security → Microphone → enable the Hermes
backend (it may appear as your terminal, `python`, or Hermes), then toggle the backend (it may appear as your terminal, `python`, or Hermes), then toggle the
wake word off and on. wake word off and on.
### "Listening" but receives silence (Windows)
Desktop push-to-talk and wake-word capture use different microphone paths.
Push-to-talk uses the desktop application's browser capture, while the
wake-word listener opens a PortAudio stream in the Python backend. One can work
while the other selects a silent or unusable Windows input.
`/wake status` reports the selected input device and Windows audio host API.
When it reports silence, set `wake_word.input_device` to the numeric index or an
unambiguous name of the working PortAudio input, then toggle the wake word:
```bash
hermes config set wake_word.input_device "Microphone Array"
```
Use `null` to return to the process default:
```bash
hermes config set wake_word.input_device null
```
## Notes & limits ## Notes & limits
- **Local surfaces only.** The wake word runs in the CLI, TUI, and desktop GUI — - **Local surfaces only.** The wake word runs in the CLI, TUI, and desktop GUI —