Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions PR_CLIENT_CAPTURE_WAKE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
# feat: client-capture wake word for remote desktop

## Summary

Remote Hermes backends (Docker / headless VM / machine in another room) often
have **no microphone**. Stock wake word opens PortAudio on the **server**, so
the desktop ear fails with:

> Failed to open the wake-word microphone.

This PR keeps **detection on the backend** (openWakeWord / sherpa / porcupine
unchanged) and adds **client capture**: the desktop streams 16 kHz mono int16
PCM over the existing authenticated WebSocket via a new `wake.feed` RPC.

That is the correct product fix for “agent remote + Mac mic hands-free.”

## Design

| Step | Where |
|------|--------|
| Arm ear (`wake.start`, `surface: gui`, `client_capture: true`) | Desktop → backend |
| Backend chooses `capture: client` when preferred / configured | `tools.wake_word.resolve_capture_mode` |
| Engine listens on an in-process PCM queue (no PortAudio device) | `WakeWordDetector(external_audio=True)` |
| Desktop `getUserMedia` → resample → `wake.feed` frames | `apps/desktop/src/lib/wake-client-capture.ts` |
| Phrase detected → `wake.detected` (unchanged) | `tui_gateway` |
| Stop client feeder so voice can take the mic | wiring on `wake.detected` |
| After voice, re-arm + restart feeder | `resumeWakeAfterVoice` |

Config:

```yaml
wake_word:
capture: auto # auto | local | client
```

- **auto** + desktop `client_capture: true` → client mode (remote-friendly)
- **auto** without prefer → local (CLI/TUI unchanged)
- **local** / **client** force the mode

## Test plan

- [x] `pytest tests/tools/test_wake_word.py` — 26 passed
- [ ] Desktop remote to headless backend: ear on → no “Failed to open mic”
- [ ] macOS mic permission prompt once; say “hey hermes” → voice session starts
- [ ] After voice ends, ear re-arms without a manual toggle
- [ ] Local (non-remote) backend with a real mic still works (`capture: local` / auto)
- [ ] CLI `/wake on` still uses local PortAudio (no client_capture)

## Notes for reviewers

- Client capture intentionally does **not** move the ONNX engine into Electron;
only PCM transport moves. Smaller desktop footprint, shared engines with TUI.
- PCM stays on the desktop↔backend WebSocket; no third-party wake API.
- Older desktops without this feeder still get the old local-mic path.

Closes: remote wake on headless hosts (user report: hermes-migrate / vm-1).
6 changes: 5 additions & 1 deletion apps/desktop/src/app/contrib/wiring.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -64,7 +64,7 @@ import {
setMessages
} from '@/store/session'
import { clearSessionTodos, setSessionTodos, todosForHydration } from '@/store/todos'
import { armWakeWord } from '@/store/wake-word'
import { armWakeWord, stopClientCapture } from '@/store/wake-word'
import { isSecondaryWindow } from '@/store/windows'
import { useSkinCommand } from '@/themes/use-skin-command'

Expand Down Expand Up @@ -685,6 +685,10 @@ export function ContribWiring({ children }: { children: ReactNode }) {
if (event.type === 'wake.detected') {
const payload = event.payload as { profile?: null | string; start_new_session?: boolean } | undefined

// Free the Mac mic so voice conversation can open getUserMedia.
// Server already pauses the detector lease; this stops client PCM feed.
stopClientCapture()

// Audible confirmation that the wake registered, before voice capture
// starts. Gated by the shared sound-mute toggle.
playWakeSound()
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -659,7 +659,10 @@ export function useSlashCommand(deps: SlashCommandDeps) {
}

const status = async (): Promise<WakeStatusResponse> => {
const current = await requestGateway<WakeStatusResponse>('wake.status', {})
const current = await requestGateway<WakeStatusResponse>('wake.status', {
client_capture: true,
surface: 'gui'
})
applyWakeStatus(current)

return current
Expand All @@ -677,7 +680,7 @@ export function useSlashCommand(deps: SlashCommandDeps) {
if (action === 'on') {
const started = await requestGateway<WakeStartResponse>(
'wake.start',
{ persist: true, surface: 'gui' },
{ persist: true, surface: 'gui', client_capture: true },
WAKE_START_TIMEOUT_MS
)

Expand Down
208 changes: 208 additions & 0 deletions apps/desktop/src/lib/wake-client-capture.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,208 @@
/**
* Client-side mic capture for remote wake word.
*
* When the backend arms with `capture: "client"`, PortAudio runs on a headless
* VM with no mic. The desktop opens getUserMedia here, resamples to 16 kHz
* mono int16 frames, and pushes them via `wake.feed` so openWakeWord still
* runs server-side without requiring a server sound device.
*/

const TARGET_RATE = 16_000
const DEFAULT_FRAME = 1280 // 80 ms @ 16 kHz — matches tools/wake_word.py

export type WakeFeedRequester = (
method: string,
params?: Record<string, unknown>
) => Promise<unknown>

export interface ClientWakeCaptureOptions {
/** Samples per frame at 16 kHz (from wake.start response). */
frameLength?: number
request: WakeFeedRequester
onError?: (error: Error) => void
}

export interface ClientWakeCaptureHandle {
stop: () => void
readonly active: boolean
}

function downsampleTo16k(input: Float32Array, inputRate: number): Float32Array {
if (inputRate === TARGET_RATE) {
return input
}
if (inputRate <= 0) {
return new Float32Array(0)
}
const ratio = inputRate / TARGET_RATE
const outLen = Math.max(1, Math.floor(input.length / ratio))
const out = new Float32Array(outLen)
for (let i = 0; i < outLen; i++) {
const start = Math.floor(i * ratio)
const end = Math.min(input.length, Math.floor((i + 1) * ratio))
let sum = 0
let count = 0
for (let j = start; j < end; j++) {
sum += input[j] ?? 0
count++
}
out[i] = count > 0 ? sum / count : 0
}
return out
}

function floatToInt16LE(input: Float32Array): ArrayBuffer {
const buf = new ArrayBuffer(input.length * 2)
const view = new DataView(buf)
for (let i = 0; i < input.length; i++) {
const s = Math.max(-1, Math.min(1, input[i] ?? 0))
view.setInt16(i * 2, s < 0 ? s * 0x8000 : s * 0x7fff, true)
}
return buf
}

function bytesToBase64(buf: ArrayBuffer): string {
const bytes = new Uint8Array(buf)
let binary = ''
const chunk = 0x8000
for (let i = 0; i < bytes.length; i += chunk) {
binary += String.fromCharCode(...bytes.subarray(i, i + chunk))
}
return btoa(binary)
}

/**
* Start streaming the default microphone to `wake.feed`.
* Returns a handle whose `stop()` ends tracks + audio graph.
*/
export async function startClientWakeCapture(
options: ClientWakeCaptureOptions
): Promise<ClientWakeCaptureHandle> {
const frameLength = Math.max(160, Math.trunc(options.frameLength || DEFAULT_FRAME))
const audioWindow = window as Window & { webkitAudioContext?: typeof AudioContext }
const AudioContextCtor = window.AudioContext || audioWindow.webkitAudioContext
if (!AudioContextCtor) {
throw new Error('AudioContext unavailable for client wake capture')
}
if (!navigator.mediaDevices?.getUserMedia) {
throw new Error('getUserMedia unavailable for client wake capture')
}

const stream = await navigator.mediaDevices.getUserMedia({
audio: {
channelCount: 1,
echoCancellation: true,
noiseSuppression: true,
autoGainControl: true
},
video: false
})

const context = new AudioContextCtor()
const source = context.createMediaStreamSource(stream)
// ScriptProcessor is deprecated but widely available and simple for PCM export.
// Buffer size 4096 keeps callback rate reasonable on desktop.
const processor = context.createScriptProcessor(4096, 1, 1)
const mute = context.createGain()
mute.gain.value = 0

let pending = new Float32Array(0)
let stopped = false
// Bounded ordered queue of 16 kHz frames. We never drop the frame that is
// currently being sent; under remote latency we drop the oldest queued
// frames so the detector still sees contiguous recent PCM rather than gaps
// from fire-and-forget discard-while-inflight.
const MAX_QUEUED_FRAMES = 24 // ~1.9s at 80 ms/frame
const queue: Float32Array[] = []
let draining = false

const drainQueue = async () => {
if (draining) {
return
}
draining = true
try {
while (!stopped && queue.length > 0) {
const frame = queue.shift()
if (!frame) {
break
}
try {
const pcm = floatToInt16LE(frame)
await options.request('wake.feed', {
pcm: bytesToBase64(pcm),
sample_rate: TARGET_RATE
})
} catch (error) {
options.onError?.(error instanceof Error ? error : new Error(String(error)))
// Keep draining later frames; one failed RPC should not freeze the ear.
}
}
} finally {
draining = false
if (!stopped && queue.length > 0) {
void drainQueue()
}
}
}

const enqueueFrame = (frame: Float32Array) => {
if (stopped) {
return
}
queue.push(frame)
while (queue.length > MAX_QUEUED_FRAMES) {
queue.shift()
}
void drainQueue()
}

processor.onaudioprocess = event => {
if (stopped) {
return
}
const input = event.inputBuffer.getChannelData(0)
const at16k = downsampleTo16k(input, context.sampleRate)
// Append to pending and emit full frames
const merged = new Float32Array(pending.length + at16k.length)
merged.set(pending, 0)
merged.set(at16k, pending.length)
let offset = 0
while (offset + frameLength <= merged.length) {
const frame = merged.subarray(offset, offset + frameLength)
offset += frameLength
enqueueFrame(new Float32Array(frame))
}
pending = merged.subarray(offset)
}

source.connect(processor)
processor.connect(mute)
mute.connect(context.destination)

if (context.state === 'suspended') {
await context.resume().catch(() => undefined)
}

return {
get active() {
return !stopped
},
stop() {
if (stopped) {
return
}
stopped = true
queue.length = 0
try {
processor.disconnect()
source.disconnect()
mute.disconnect()
} catch {
// ignore
}
void context.close().catch(() => undefined)
stream.getTracks().forEach(t => t.stop())
}
}
}
Loading