diff --git a/README.md b/README.md index 6eaebf3..a0ca7dd 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ # EveFlow 2 — Interface vocale JARVIS pour Hermes Agent [![Build](https://img.shields.io/github/actions/workflow/status/R0m1k3/EveFlow/windows-release.yml?style=flat-square)](https://github.com/R0m1k3/EveFlow/actions) -[![Version](https://img.shields.io/badge/version-2.1.0-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases) +[![Version](https://img.shields.io/badge/version-2.2.0-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases) [![License](https://img.shields.io/badge/license-MIT-lightgrey.svg?style=flat-square)](LICENSE) **EveFlow** est un compagnon de bureau Windows qui transforme [Hermes Agent](https://hermes-agent.nousresearch.com/) en assistant vocal à la JARVIS : un noyau holographique réactif au son, une conversation en streaming, les outils, sous-agents, approbations, crons, skills et sessions d'Hermes pilotés depuis un seul HUD. @@ -23,7 +23,7 @@ La version 2 est une réécriture complète : plus de robot 3D, un pipeline voca * **Détection d'activité vocale** (seuil adaptatif, sensibilité et silence de fin réglables) : l'enregistrement s'arrête tout seul quand vous avez fini de parler. * **Mains libres** : le micro se réactive après chaque réponse. * **Modèles intégrés, hors ligne** (sherpa-onnx dans un processus séparé) : reconnaissance Whisper (base, small, large-v3 turbo) ou SenseVoice, synthèse Kokoro v1.0 (voix française Siwis et voix anglaises) ou Piper (Siwis, Tom, UPMC). Les modèles se téléchargent depuis **Paramètres → Modèles locaux** et tournent sur le processeur. -* **Mot d'activation** en mains libres : seules les phrases commençant par « Jarvis » (configurable) partent vers Hermes, le reste est ignoré ; un « Jarvis » seul ouvre une fenêtre d'écoute. +* **Écoute permanente** : un détecteur de mot-clé de 3 Mo (sherpa-onnx, keyword spotting) tourne en continu sur le micro, quasi gratuit en CPU. « Jarvis » (ou n'importe quel mot-clé) ouvre l'écoute, « Jarvis, allume… » envoie directement la commande, et le mot coupe la voix en cours. Alternative : filtre du mot après transcription en mains libres. * **STT externe** : n'importe quelle API `/v1/audio/transcriptions` compatible OpenAI (Qwen3-ASR, Whisper, Speaches, faster-whisper-server, LocalAI, OpenAI). Repli sur la reconnaissance Chromium. * **TTS externe** : API `/v1/audio/speech` compatible OpenAI, voix système Windows ou Google Translate. Lecture phrase par phrase pendant le streaming, préchargement du segment suivant, coupure instantanée. * Raccourcis globaux : `Ctrl+Shift+Espace` (micro), `Ctrl+Shift+J` (afficher/masquer), `Ctrl+Shift+Échap` (couper la voix). diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 17c3304..366ac83 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -9,7 +9,8 @@ | HUD arc-reactor réactif au son | Fait | Canvas 2D optimisé (pas d'ombres, couleurs en cache, 30 fps en veille, arrêt fenêtre masquée) | | Reconnaissance vocale locale | Fait | Whisper base/small/turbo via sherpa-onnx dans un processus utilitaire | | Synthèse vocale locale | Fait | Kokoro v1.0 (voix française Siwis) et Piper fr | -| Mot d'activation | Fait, mode « après transcription » | Filtre « Jarvis … » en mains libres, tolérant aux erreurs de transcription | +| Mot d'activation permanent | Fait (2.2.0) | Keyword spotting sherpa-onnx en continu ; mot-clé libre encodé en BPE ; validé sur audio réel (détection, zéro faux positif sur le test anglais) | +| Mot d'activation après transcription | Fait | Filtre « Jarvis … » en mains libres, tolérant aux erreurs de transcription | | Détection de fin de phrase | Fait | VAD énergétique adaptatif, pré-roll 400 ms | | Hermes : runs, sessions, chat completions | Fait | Transport choisi selon `/v1/capabilities` | | Approbations, steer, stop | Fait | Modales, injection de consigne en cours de run | @@ -21,7 +22,7 @@ Sources : [jarvis-desktop-ai](https://github.com/ccarloshenri/jarvis-desktop-ai), [JarvisAi](https://github.com/PanPenek/JarvisAi), [bertrandmbanwi/Jarvis](https://github.com/bertrandmbanwi/Jarvis), [InterGenJLU/jarvis](https://github.com/InterGenJLU/jarvis), [livekit-wakeword](https://livekit.com/blog/livekit-wakeword), [sherpa-onnx keyword spotting](https://k2-fsa.github.io/sherpa/onnx/kws/index.html), [Hermes Agent features](https://hermes-agent.nousresearch.com/docs/user-guide/features/overview). -1. **Mot d'activation permanent, quasi gratuit en CPU.** Les projets de référence utilisent openWakeWord (« hey jarvis ») ou un modèle de keyword spotting qui écoute en continu, au lieu de transcrire chaque phrase. sherpa-onnx fournit un modèle KWS anglais de 3,3 Mo qui accepte n'importe quel mot-clé sans réentraînement ; « JARVIS » s'encode `▁JA R VI S` avec son modèle BPE (vérifié). C'est le prochain chantier prioritaire : streaming du micro vers le worker, détection en continu, puis capture de la commande. +1. **Mot d'activation permanent, quasi gratuit en CPU.** Les projets de référence utilisent openWakeWord (« hey jarvis ») ou un modèle de keyword spotting qui écoute en continu, au lieu de transcrire chaque phrase. sherpa-onnx fournit un modèle KWS anglais de 3,3 Mo qui accepte n'importe quel mot-clé sans réentraînement ; « JARVIS » s'encode `▁JA R VI S` avec son modèle BPE (vérifié). Livré en 2.2.0 (voir l'étape 1 ci-dessous). 2. **VAD neuronal (Silero) au lieu du seuil d'énergie.** Fin de phrase plus nette (environ 500 ms gagnés) et beaucoup moins de faux départs sur le bruit ambiant. Silero est déjà livré dans sherpa-onnx (`silero_vad.onnx`, 0,6 Mo). 3. **Latence perçue sous la seconde.** Les références visent 1 s entre la fin de parole et le premier mot prononcé : STT rapide, premier token en streaming, TTS phrase par phrase (déjà en place), et un modèle Hermes rapide pour la conversation courante. 4. **Vision d'écran.** Capture d'écran à la demande (« Jarvis, qu'est-ce que je regarde ? ») envoyée à Hermes comme image, ou lecture d'une fenêtre. Hermes accepte déjà les images inline. @@ -31,11 +32,11 @@ Sources : [jarvis-desktop-ai](https://github.com/ccarloshenri/jarvis-desktop-ai) ## Plan proposé -### Étape 1 (courte) : écoute permanente -- Streamer l'audio du micro (16 kHz, blocs de 128 ms) du renderer vers le worker via IPC. -- Dans le worker : `KeywordSpotter` sherpa-onnx (modèle gigaspeech 3,3 Mo, mot-clé configurable encodé automatiquement avec `bpe.model` via un petit encodeur BPE côté Node ou un dictionnaire pré-encodé pour « jarvis », « eve », « hey jarvis », « ok jarvis »). -- À la détection : chime, capture de la commande avec Silero VAD, transcription, envoi. -- Consommation attendue : quelques pourcents d'un cœur, pas de transcription en continu. +### Étape 1 : écoute permanente — livrée en 2.2.0 +- Le renderer garde un seul flux micro (AudioWorklet 16 kHz) et envoie des blocs de 256 ms au processus principal, qui alimente le `KeywordSpotter` sherpa-onnx dans le worker. +- Mots-clés encodés en BPE (table SentencePiece pour les mots courants, repli glouton sur le vocabulaire du modèle), sensibilité réglable (seuil 0,45 → 0,12). +- À la détection : chime, capture de la commande sur le même flux (VAD énergétique, pré-roll 400 ms), transcription locale ou API, envoi à Hermes ; retour automatique à l'écoute. +- Reste à faire : remplacer le VAD énergétique par Silero pour la fin de phrase. ### Étape 2 : vision et actions locales - Outil `capture_screen` (Electron `desktopCapturer`) qui joint une capture à la requête Hermes. diff --git a/electron/preload.ts b/electron/preload.ts index 329ad1c..d48c907 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -14,7 +14,7 @@ import { type WindowMode } from '../shared/ipc'; import type { EveFlowBridge, Unsubscribe } from '../shared/bridge'; -import { VOICE_IPC, type SynthesizeRequest, type SynthesizeResult, type TranscribeRequest, type TranscribeResult, type VoiceDownloadProgress, type VoiceEngineStatus, type VoiceModelStatus } from '../shared/voice'; +import { VOICE_IPC, type KwsDetection, type KwsStartRequest, type SynthesizeRequest, type SynthesizeResult, type TranscribeRequest, type TranscribeResult, type VoiceDownloadProgress, type VoiceEngineStatus, type VoiceModelStatus } from '../shared/voice'; function subscribe(channel: string, callback: (payload: T) => void): Unsubscribe { const listener = (_event: Electron.IpcRendererEvent, payload: T) => callback(payload); @@ -69,7 +69,11 @@ const api: EveFlowBridge = { onProgress: (cb: (progress: VoiceDownloadProgress) => void) => subscribe(VOICE_IPC.modelsProgress, cb), transcribe: (req: TranscribeRequest) => ipcRenderer.invoke(VOICE_IPC.transcribe, req) as Promise, synthesize: (req: SynthesizeRequest) => ipcRenderer.invoke(VOICE_IPC.synthesize, req) as Promise, - unload: (id?: string) => ipcRenderer.invoke(VOICE_IPC.unload, id) as Promise + unload: (id?: string) => ipcRenderer.invoke(VOICE_IPC.unload, id) as Promise, + kwsStart: (req: KwsStartRequest) => ipcRenderer.invoke(VOICE_IPC.kwsStart, req) as Promise<{ accepted: string[]; rejected: string[] }>, + kwsStop: () => ipcRenderer.invoke(VOICE_IPC.kwsStop) as Promise, + kwsAudio: (pcm: Uint8Array, sampleRate: number) => ipcRenderer.send(VOICE_IPC.kwsAudio, pcm, sampleRate), + onKwsDetected: (cb: (detection: KwsDetection) => void) => subscribe(VOICE_IPC.kwsDetected, cb) } }; diff --git a/electron/voice/catalog.ts b/electron/voice/catalog.ts index 2e8045c..b2d9569 100644 --- a/electron/voice/catalog.ts +++ b/electron/voice/catalog.ts @@ -1,6 +1,7 @@ import type { VoiceModelSpec, VoiceSpeaker } from '../../shared/voice'; const ASR = 'https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models'; +const KWS = 'https://github.com/k2-fsa/sherpa-onnx/releases/download/kws-models'; const TTS = 'https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models'; const KOKORO_SPEAKERS: VoiceSpeaker[] = [ @@ -67,6 +68,24 @@ export const VOICE_CATALOG: VoiceModelSpec[] = [ dir: 'sherpa-onnx-sense-voice-zh-en-ja-ko-yue-int8-2024-07-17', files: ['model.int8.onnx', 'tokens.txt'] }, + { + id: 'kws-en', + kind: 'kws', + engine: 'kws-transducer', + name: 'Détecteur de mot-clé (zipformer, 3 Mo)', + description: 'Écoute permanente du mot d’activation, quasi gratuite en CPU. Mot-clé libre (« jarvis », « hey jarvis »…).', + languages: ['en', 'fr'], + sizeMb: 4, + url: `${KWS}/sherpa-onnx-kws-zipformer-gigaspeech-3.3M-2024-01-01.tar.bz2`, + dir: 'sherpa-onnx-kws-zipformer-gigaspeech-3.3M-2024-01-01', + files: [ + 'encoder-epoch-12-avg-2-chunk-16-left-64.int8.onnx', + 'decoder-epoch-12-avg-2-chunk-16-left-64.int8.onnx', + 'joiner-epoch-12-avg-2-chunk-16-left-64.int8.onnx', + 'tokens.txt' + ], + recommended: true + }, { id: 'kokoro-v1', kind: 'tts', diff --git a/electron/voice/engine.ts b/electron/voice/engine.ts index a1e398b..a3a0b00 100644 --- a/electron/voice/engine.ts +++ b/electron/voice/engine.ts @@ -2,13 +2,21 @@ * Host side of the voice worker: spawns the utility process on demand, correlates * requests and responses, restarts the worker if it crashes. */ -import { utilityProcess, type UtilityProcess } from 'electron'; +import { utilityProcess, type UtilityProcess, type WebContents } from 'electron'; +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; import path from 'node:path'; -import type { SynthesizeRequest, SynthesizeResult, TranscribeRequest, TranscribeResult, VoiceEngineStatus } from '../../shared/voice'; +import { VOICE_IPC, type KwsDetection, type KwsStartRequest, type SynthesizeRequest, type SynthesizeResult, type TranscribeRequest, type TranscribeResult, type VoiceEngineStatus } from '../../shared/voice'; +import { buildKeywordsFile, parseTokens } from '../../shared/keywords'; import { findModel } from './catalog'; import { isInstalled, modelDir, modelsDir } from './models'; import { log } from '../logger'; +/** Renderer that receives keyword detections while spotting is active. */ +let kwsSubscriber: WebContents | null = null; +let kwsActive = false; +let kwsRequest: KwsStartRequest | null = null; + interface Pending { resolve: (value: unknown) => void; reject: (err: Error) => void; @@ -34,7 +42,15 @@ function spawn(): UtilityProcess { child = proc; proc.stdout?.on('data', (d: Buffer) => log('DEBUG', 'voice-worker', d.toString().trim())); proc.stderr?.on('data', (d: Buffer) => log('WARN', 'voice-worker', d.toString().trim())); - proc.on('message', (msg: { id: number; ok: boolean; result?: unknown; error?: string }) => { + proc.on('message', (msg: { id?: number; type?: string; ok?: boolean; result?: unknown; error?: string; keyword?: string; at?: number }) => { + if (msg.type === 'kws.detected') { + if (kwsSubscriber && !kwsSubscriber.isDestroyed()) { + kwsSubscriber.send(VOICE_IPC.kwsDetected, { keyword: msg.keyword ?? '', at: msg.at ?? Date.now() } satisfies KwsDetection); + } + log('INFO', 'voice', `wake word detected: ${msg.keyword}`); + return; + } + if (typeof msg.id !== 'number') return; const p = pending.get(msg.id); if (!p) return; pending.delete(msg.id); @@ -48,6 +64,8 @@ function spawn(): UtilityProcess { if (child !== proc) return; child = null; rejectAll('Le moteur vocal local s’est arrêté de façon inattendue.'); + // Keyword spotting survives a worker restart: re-arm on the next audio frame. + if (kwsActive) kwsArmed = false; }); log('INFO', 'voice', 'voice worker started'); return proc; @@ -108,6 +126,60 @@ export async function synthesize(req: SynthesizeRequest): Promise = { 1: 0.45, 2: 0.35, 3: 0.25, 4: 0.18, 5: 0.12 }; +const SENSITIVITY_SCORE: Record = { 1: 1.0, 2: 1.0, 3: 1.2, 4: 1.5, 5: 2.0 }; + +/** Start keyword spotting; the keywords file is derived from the phrases and the model vocabulary. */ +export async function kwsStart(req: KwsStartRequest, sender: WebContents): Promise<{ accepted: string[]; rejected: string[] }> { + const spec = findModel(req.modelId); + if (!spec || spec.kind !== 'kws') throw new Error('Modèle de détection introuvable'); + if (!isInstalled(spec)) throw new Error('Détecteur de mot-clé non installé (Paramètres → Modèles locaux).'); + const dir = modelDir(spec); + const vocab = parseTokens(fs.readFileSync(path.join(dir, 'tokens.txt'), 'utf8')); + const phrases = req.keywords.map((k) => String(k).slice(0, 40)).filter(Boolean).slice(0, 8); + const file = buildKeywordsFile(phrases, vocab); + if (file.accepted.length === 0) throw new Error(`Aucun mot d’activation encodable : ${file.rejected.join(', ')}`); + const hash = createHash('sha1').update(file.content).digest('hex').slice(0, 10); + const keywordsFile = path.join(modelsDir(), `keywords-${hash}.txt`); + fs.writeFileSync(keywordsFile, file.content, 'utf8'); + const sensitivity = Math.min(5, Math.max(1, Math.round(req.sensitivity))) as 1 | 2 | 3 | 4 | 5; + kwsSubscriber = sender; + kwsRequest = req; + await request({ type: 'kws.start', model: { id: spec.id, engine: spec.engine, dir, files: spec.files }, keywordsFile, threshold: SENSITIVITY_THRESHOLD[sensitivity], score: SENSITIVITY_SCORE[sensitivity] }, 60_000); + kwsActive = true; + kwsArmed = true; + log('INFO', 'voice', `keyword spotting on: ${file.accepted.join(', ')} (threshold ${SENSITIVITY_THRESHOLD[sensitivity]})`); + return { accepted: file.accepted, rejected: file.rejected }; +} + +export async function kwsStop(): Promise { + kwsActive = false; + kwsArmed = false; + kwsSubscriber = null; + if (child) await request({ type: 'kws.stop' }, 10_000).catch(() => undefined); +} + +/** Feed 16-bit PCM from the renderer (fire-and-forget). */ +export function kwsFeed(pcm: Uint8Array, sampleRate: number): void { + if (!kwsActive) return; + const proc = spawn(); + if (!kwsArmed) { + // Worker restarted: re-create the spotter before feeding audio. + if (kwsRequest && kwsSubscriber) { + kwsArmed = true; + kwsStart(kwsRequest, kwsSubscriber).catch((err) => log('WARN', 'voice', `kws re-arm failed: ${(err as Error).message}`)); + } + return; + } + const b64 = Buffer.from(pcm.buffer, pcm.byteOffset, pcm.byteLength).toString('base64'); + try { + proc.postMessage({ id: -1, type: 'kws.audio', pcm: b64, sampleRate }); + } catch (err) { + log('WARN', 'voice', `kws feed failed: ${(err as Error).message}`); + } +} + /** Native model memory is only released deterministically by restarting the worker. */ export function unload(_modelId?: string): Promise { stopEngine(); diff --git a/electron/voice/ipc.ts b/electron/voice/ipc.ts index 227864e..a89300f 100644 --- a/electron/voice/ipc.ts +++ b/electron/voice/ipc.ts @@ -1,6 +1,6 @@ import { ipcMain } from 'electron'; -import { VOICE_IPC, type SynthesizeRequest, type TranscribeRequest } from '../../shared/voice'; -import { engineStatus, synthesize, transcribe, unload } from './engine'; +import { VOICE_IPC, type KwsStartRequest, type SynthesizeRequest, type TranscribeRequest } from '../../shared/voice'; +import { engineStatus, kwsFeed, kwsStart, kwsStop, synthesize, transcribe, unload } from './engine'; import { cancelDownload, downloadModel, listModels, removeModel } from './models'; export function registerVoiceIpc(): void { @@ -27,4 +27,13 @@ export function registerVoiceIpc(): void { return synthesize({ ...req, speaker: Number.isFinite(req.speaker) ? req.speaker : 0, speed: Number.isFinite(req.speed) ? req.speed : 1 }); }); ipcMain.handle(VOICE_IPC.unload, (_e, id?: string) => unload(id)); + ipcMain.handle(VOICE_IPC.kwsStart, (event, req: KwsStartRequest) => { + if (!req || !Array.isArray(req.keywords) || typeof req.modelId !== 'string') throw new Error('Requête invalide'); + return kwsStart({ ...req, keywords: req.keywords.filter((k) => typeof k === 'string'), sensitivity: Number(req.sensitivity) || 3 }, event.sender); + }); + ipcMain.handle(VOICE_IPC.kwsStop, () => kwsStop()); + ipcMain.on(VOICE_IPC.kwsAudio, (_e, pcm: unknown, sampleRate: unknown) => { + if (!(pcm instanceof Uint8Array) || pcm.byteLength === 0 || pcm.byteLength > 1024 * 1024) return; + kwsFeed(pcm, typeof sampleRate === 'number' && sampleRate > 0 ? sampleRate : 16000); + }); } diff --git a/electron/voice/worker.ts b/electron/voice/worker.ts index d9d1015..97c7ba1 100644 --- a/electron/voice/worker.ts +++ b/electron/voice/worker.ts @@ -18,7 +18,10 @@ type Request = | { id: number; type: 'status' } | { id: number; type: 'transcribe'; model: ModelRef; wav: Uint8Array | string; language: string } | { id: number; type: 'synthesize'; model: ModelRef; text: string; speaker: number; speed: number } - | { id: number; type: 'unload'; modelId?: string }; + | { id: number; type: 'unload'; modelId?: string } + | { id: number; type: 'kws.start'; model: ModelRef; keywordsFile: string; threshold: number; score: number } + | { id: number; type: 'kws.audio'; pcm: string; sampleRate: number } + | { id: number; type: 'kws.stop' }; type Response = { id: number; ok: true; result: unknown } | { id: number; ok: false; error: string }; @@ -29,6 +32,13 @@ type Sherpa = { decode: (s: unknown) => void; getResult: (s: unknown) => { text: string; lang?: string }; }; + KeywordSpotter: new (config: unknown) => { + createStream: () => KwsStream; + isReady: (s: KwsStream) => boolean; + decode: (s: KwsStream) => void; + reset: (s: KwsStream) => void; + getResult: (s: KwsStream) => { keyword?: string }; + }; OfflineTts: new (config: unknown) => { numSpeakers: number; sampleRate: number; @@ -37,7 +47,10 @@ type Sherpa = { version: string; }; +type KwsStream = { acceptWaveform: (w: { sampleRate: number; samples: Float32Array }) => void }; let sherpa: Sherpa | null = null; +let kws: { spotter: InstanceType; stream: KwsStream } | null = null; +let notify: ((message: unknown) => void) | null = null; let loadError: string | null = null; function loadSherpa(): Sherpa { @@ -225,13 +238,59 @@ function encodeWav(samples: Float32Array, sampleRate: number): Uint8Array { return new Uint8Array(buffer); } +function startKws(model: ModelRef, keywordsFile: string, threshold: number, score: number): void { + const s = loadSherpa(); + kws = null; + const p = (f: string) => path.join(model.dir, f); + const enc = model.files.find((f) => f.startsWith('encoder')) ?? 'encoder.int8.onnx'; + const dec = model.files.find((f) => f.startsWith('decoder')) ?? 'decoder.int8.onnx'; + const join = model.files.find((f) => f.startsWith('joiner')) ?? 'joiner.int8.onnx'; + const spotter = new s.KeywordSpotter({ + featConfig: { sampleRate: 16000, featureDim: 80 }, + modelConfig: { transducer: { encoder: p(enc), decoder: p(dec), joiner: p(join) }, tokens: p('tokens.txt'), numThreads: 1, provider: 'cpu', debug: 0 }, + maxActivePaths: 4, + numTrailingBlanks: 1, + keywordsScore: score, + keywordsThreshold: threshold, + keywordsFile + }); + kws = { spotter, stream: spotter.createStream() }; +} + +function feedKws(pcmBase64: string, sampleRate: number): void { + if (!kws) return; + const bytes = Buffer.from(pcmBase64, 'base64'); + const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, Math.floor(bytes.byteLength / 2)); + let samples: Float32Array = new Float32Array(int16.length); + for (let i = 0; i < int16.length; i++) samples[i] = int16[i] / 32768; + if (sampleRate !== 16000) samples = resampleTo16k(samples, sampleRate); + kws.stream.acceptWaveform({ sampleRate: 16000, samples }); + while (kws.spotter.isReady(kws.stream)) { + kws.spotter.decode(kws.stream); + const result = kws.spotter.getResult(kws.stream); + if (result.keyword) { + kws.spotter.reset(kws.stream); + notify?.({ type: 'kws.detected', keyword: result.keyword, at: Date.now() }); + } + } +} + // ── request handling ─────────────────────────────────────────────────────── function handle(req: Request): unknown { switch (req.type) { + case 'kws.start': + startKws(req.model, req.keywordsFile, req.threshold, req.score); + return { ok: true }; + case 'kws.audio': + feedKws(req.pcm, req.sampleRate); + return { ok: true }; + case 'kws.stop': + kws = null; + return { ok: true }; case 'status': { try { const s = loadSherpa(); - return { available: true, version: s.version, loaded: [...recognizers.keys(), ...synthesizers.keys()] }; + return { available: true, version: s.version, loaded: [...recognizers.keys(), ...synthesizers.keys(), ...(kws ? ['kws'] : [])] }; } catch (err) { return { available: false, error: (err as Error).message, loaded: [] }; } @@ -271,6 +330,7 @@ function handle(req: Request): unknown { } else { recognizers.clear(); synthesizers.clear(); + kws = null; } return { ok: true }; } @@ -290,7 +350,9 @@ function respond(req: Request): Response { // Electron utility process transport, with a child_process fallback for tests. const parentPort = (process as unknown as { parentPort?: { on: (ev: 'message', cb: (e: { data: Request }) => void) => void; postMessage: (m: unknown) => void } }).parentPort; if (parentPort) { + notify = (m) => parentPort.postMessage(m); parentPort.on('message', (event) => parentPort.postMessage(respond(event.data))); } else if (process.send) { + notify = (m) => process.send!(m); process.on('message', (msg: Request) => process.send!(respond(msg))); } diff --git a/package.json b/package.json index 9883b07..7d48ea2 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "eveflow", - "version": "2.1.0", - "releaseVersion": "2.1.0.1", + "version": "2.2.0", + "releaseVersion": "2.2.0", "description": "JARVIS-style desktop HUD for Hermes Agent: voice, streaming runs, scheduled jobs, skills and telemetry", "main": "dist-electron/main.js", "private": true, diff --git a/shared/bridge.ts b/shared/bridge.ts index f9e41f9..b99298e 100644 --- a/shared/bridge.ts +++ b/shared/bridge.ts @@ -12,6 +12,8 @@ import type { WindowMode } from './ipc'; import type { + KwsDetection, + KwsStartRequest, SynthesizeRequest, SynthesizeResult, TranscribeRequest, @@ -72,5 +74,10 @@ export interface EveFlowBridge { transcribe: (req: TranscribeRequest) => Promise; synthesize: (req: SynthesizeRequest) => Promise; unload: (id?: string) => Promise; + kwsStart: (req: KwsStartRequest) => Promise<{ accepted: string[]; rejected: string[] }>; + kwsStop: () => Promise; + /** Fire-and-forget 16-bit PCM frames for the keyword spotter. */ + kwsAudio: (pcm: Uint8Array, sampleRate: number) => void; + onKwsDetected: (cb: (detection: KwsDetection) => void) => Unsubscribe; }; } diff --git a/shared/keywords.ts b/shared/keywords.ts new file mode 100644 index 0000000..48ab244 --- /dev/null +++ b/shared/keywords.ts @@ -0,0 +1,108 @@ +/** + * Encodes wake phrases into the BPE token sequences expected by the sherpa-onnx keyword + * spotter (gigaspeech BPE-500 model). Common phrases use sequences produced by the real + * SentencePiece model; anything else falls back to a greedy longest-match over the vocabulary, + * which is a close approximation for short words. + */ + +const KNOWN: Record = { + 'JARVIS': '▁JA R VI S', + 'HEY JARVIS': '▁HE Y ▁JA R VI S', + 'OK JARVIS': '▁O K ▁JA R VI S', + 'EVE': '▁E VE', + 'HEY EVE': '▁HE Y ▁E VE', + 'COMPUTER': '▁COMP U TER', + 'HEY COMPUTER': '▁HE Y ▁COMP U TER', + 'FRIDAY': '▁F RI DAY', + 'HEY FRIDAY': '▁HE Y ▁F RI DAY', + 'ALFRED': '▁A L F RE D', + 'HERMES': '▁HER ME S', + 'HEY HERMES': '▁HE Y ▁HER ME S', + 'ASSISTANT': '▁AS S IST ANT', + 'OK GOOGLE': '▁O K ▁GO O G LE', + 'ALEXA': '▁A LE X A', + 'SIRI': '▁S I RI', + 'HEY SIRI': '▁HE Y ▁S I RI', + 'NOVA': '▁NO V A', + 'ATLAS': '▁AT LA S', + 'HAL': '▁HA L' +}; + +export function normalizeKeyword(phrase: string): string { + return phrase + .normalize('NFD') + .replace(/[̀-ͯ]/g, '') + .toUpperCase() + .replace(/[^A-Z ]+/g, ' ') + .replace(/\s+/g, ' ') + .trim(); +} + +/** Parse a sherpa tokens.txt ("piece id" per line) into the set of pieces. */ +export function parseTokens(tokensFile: string): Set { + const pieces = new Set(); + for (const line of tokensFile.split(/\r?\n/)) { + const piece = line.trim().split(/\s+/)[0]; + if (piece && !piece.startsWith('<')) pieces.add(piece); + } + return pieces; +} + +function greedyWord(word: string, vocab: Set): string[] | null { + const out: string[] = []; + let i = 0; + let first = true; + while (i < word.length) { + let matched = ''; + for (let len = word.length - i; len >= 1; len--) { + const candidate = (first ? '▁' : '') + word.slice(i, i + len); + if (vocab.has(candidate)) { + matched = candidate; + break; + } + } + if (!matched && first) { + // No word-initial piece: use a bare "▁" if available, then continue without the prefix. + if (vocab.has('▁')) out.push('▁'); + first = false; + continue; + } + if (!matched) return null; + out.push(matched); + i += matched.length - (first ? 1 : 0); + first = false; + } + return out; +} + +/** Encode a phrase into space-separated BPE pieces, or null when it cannot be represented. */ +export function encodeKeyword(phrase: string, vocab: Set): string | null { + const normalized = normalizeKeyword(phrase); + if (!normalized) return null; + if (KNOWN[normalized]) return KNOWN[normalized]; + const pieces: string[] = []; + for (const word of normalized.split(' ')) { + const encoded = greedyWord(word, vocab); + if (!encoded) return null; + pieces.push(...encoded); + } + return pieces.join(' '); +} + +/** Build the keywords file content: one line per phrase, with the readable label. */ +export function buildKeywordsFile(phrases: string[], vocab: Set): { content: string; accepted: string[]; rejected: string[] } { + const lines: string[] = []; + const accepted: string[] = []; + const rejected: string[] = []; + for (const phrase of phrases) { + const encoded = encodeKeyword(phrase, vocab); + const label = normalizeKeyword(phrase).toLowerCase().replace(/ /g, '_'); + if (!encoded || !label) { + rejected.push(phrase); + continue; + } + lines.push(`${encoded} @${label}`); + accepted.push(label); + } + return { content: lines.join('\n') + '\n', accepted, rejected }; +} diff --git a/shared/voice.ts b/shared/voice.ts index 3f0c6a8..f95e40a 100644 --- a/shared/voice.ts +++ b/shared/voice.ts @@ -1,7 +1,7 @@ /** Local voice engine contract (sherpa-onnx in a utility process). Shared by main and renderer. */ -export type VoiceModelKind = 'stt' | 'tts'; -export type VoiceEngineKind = 'whisper' | 'sense-voice' | 'nemo-transducer' | 'kokoro' | 'piper'; +export type VoiceModelKind = 'stt' | 'tts' | 'kws'; +export type VoiceEngineKind = 'whisper' | 'sense-voice' | 'nemo-transducer' | 'kokoro' | 'piper' | 'kws-transducer'; export interface VoiceSpeaker { id: number; @@ -69,6 +69,19 @@ export interface SynthesizeResult { audioSec: number; } +export interface KwsStartRequest { + modelId: string; + /** Wake phrases in plain text (e.g. "jarvis", "hey jarvis"). */ + keywords: string[]; + /** 1 (strict) .. 5 (eager) */ + sensitivity: number; +} + +export interface KwsDetection { + keyword: string; + at: number; +} + export interface VoiceEngineStatus { available: boolean; error?: string; @@ -86,5 +99,9 @@ export const VOICE_IPC = { modelsProgress: 'voice:models:progress', transcribe: 'voice:transcribe', synthesize: 'voice:synthesize', - unload: 'voice:unload' + unload: 'voice:unload', + kwsStart: 'voice:kws:start', + kwsStop: 'voice:kws:stop', + kwsAudio: 'voice:kws:audio', + kwsDetected: 'voice:kws:detected' } as const; diff --git a/src/App.tsx b/src/App.tsx index 723b7cf..9043db0 100644 --- a/src/App.tsx +++ b/src/App.tsx @@ -44,6 +44,18 @@ function useBoot(): boolean { if (api) { disposers.push(useVoiceModels.getState().subscribe()); void useVoiceModels.getState().refresh(); + if (settings.voice.wakeMode === 'kws') void voiceController.startWakeMode(); + let lastWake = JSON.stringify([settings.voice.wakeMode, settings.voice.wakeWord, settings.voice.kwsSensitivity, settings.voice.micDeviceId]); + disposers.push( + useSettings.subscribe((s) => { + const v = s.settings.voice; + const key = JSON.stringify([v.wakeMode, v.wakeWord, v.kwsSensitivity, v.micDeviceId]); + if (key === lastWake) return; + lastWake = key; + if (v.wakeMode === 'kws') void voiceController.restartWakeMode(); + else void voiceController.stopWakeMode(); + }) + ); disposers.push(api.hermes.onPush(handlePush)); disposers.push( api.hotkeys.on((event) => { diff --git a/src/components/hud/CoreStage.tsx b/src/components/hud/CoreStage.tsx index 9b7b550..d110121 100644 --- a/src/components/hud/CoreStage.tsx +++ b/src/components/hud/CoreStage.tsx @@ -35,6 +35,8 @@ export function CoreStage({ compact }: Props) { const transcript = useVoice((s) => s.lastTranscript); const interim = useVoice((s) => s.interim); const voiceError = useVoice((s) => s.error); + const wake = useVoice((s) => s.wake); + const wakeKeywords = useVoice((s) => s.wakeKeywords); const assistantName = useSettings((s) => s.settings.assistantName); const reduceMotion = useSettings((s) => s.settings.ui.reduceMotion); const transport = useHermes((s) => s.transport); @@ -57,6 +59,7 @@ export function CoreStage({ compact }: Props) { else if (hud === 'error') sub = previewText(error ?? voiceError ?? 'erreur', 90); else if (hud === 'speaking') sub = 'synthèse vocale'; else if (transcript) sub = `« ${previewText(transcript, 80)} »`; + else if (wake === 'spotting') sub = `dites « ${wakeKeywords[0] ?? 'jarvis'} »${link === 'online' ? ` · ${transport}` : link === 'offline' ? ' · Hermes hors ligne' : ''}`; else sub = link === 'online' ? `liaison Hermes · ${transport}` : link === 'offline' ? 'Hermes hors ligne' : ''; return ( diff --git a/src/components/settings/ModelsSection.tsx b/src/components/settings/ModelsSection.tsx index 76e3617..ba02ee1 100644 --- a/src/components/settings/ModelsSection.tsx +++ b/src/components/settings/ModelsSection.tsx @@ -1,5 +1,5 @@ import { useEffect } from 'react'; -import { Download, Trash2, X, CheckCircle2, Cpu, Mic, Volume2, AlertTriangle, RefreshCw } from 'lucide-react'; +import { Download, Trash2, X, CheckCircle2, Cpu, Mic, Volume2, AlertTriangle, RefreshCw, Ear } from 'lucide-react'; import type { VoiceModelStatus } from '../../../shared/voice'; import { bridge } from '../../lib/bridge'; import { useShallow } from 'zustand/react/shallow'; @@ -19,20 +19,23 @@ function ModelRow({ model }: { model: VoiceModelStatus }) { const isActive = model.kind === 'stt' ? activeStt === model.id : activeTts === model.id; const busy = !!progress && (progress.phase === 'download' || progress.phase === 'extract'); + const wakeMode = useSettings((s) => s.settings.voice.wakeMode); const activate = () => { if (model.kind === 'stt') update({ voice: { localModel: model.id, provider: 'local' } }); + else if (model.kind === 'kws') update({ voice: { wakeMode: 'kws' } }); else update({ speech: { localModel: model.id, provider: 'local', localSpeaker: model.speakers?.[0]?.id ?? 0 } }); }; + const activeNow = model.kind === 'kws' ? wakeMode === 'kws' : isActive; return ( -
+
{model.name} {model.recommended && recommandé} {model.installed && installé · {formatMb(model.installedBytes)}} {!model.installed && !busy && {model.sizeMb} Mo} - {isActive && model.installed && actif} + {activeNow && model.installed && actif}
{model.description} langues : {model.languages.join(', ')}{model.speakers ? ` · ${model.speakers.length} voix` : ''} @@ -49,7 +52,7 @@ function ModelRow({ model }: { model: VoiceModelStatus }) { ) : model.installed ? ( <> - {!isActive && } + {!activeNow && } ) : ( @@ -72,6 +75,7 @@ export function ModelsSection() { const stt = models.filter((m) => m.kind === 'stt'); const tts = models.filter((m) => m.kind === 'tts'); + const kws = models.filter((m) => m.kind === 'kws'); return ( <>
@@ -91,6 +95,10 @@ export function ModelsSection() {
+
+
Mot d’activation
+
{kws.map((m) => )}
+
Reconnaissance vocale
{stt.map((m) => )}
diff --git a/src/components/settings/SettingsDrawer.tsx b/src/components/settings/SettingsDrawer.tsx index e382133..5d2251e 100644 --- a/src/components/settings/SettingsDrawer.tsx +++ b/src/components/settings/SettingsDrawer.tsx @@ -33,6 +33,33 @@ function Toggle({ on, onChange, label, hint }: { on: boolean; onChange: (v: bool type TestState = { status: 'idle' | 'running' | 'ok' | 'fail'; message: string }; +function WakeStatus() { + const wake = useVoice((s) => s.wake); + const keywords = useVoice((s) => s.wakeKeywords); + const error = useVoice((s) => s.error); + const models = useVoiceModels((s) => s.models); + const download = useVoiceModels((s) => s.download); + const progress = useVoiceModels((s) => s.progress['kws-en']); + const installed = models.find((m) => m.id === 'kws-en')?.installed; + if (installed === false) { + return ( +
+ + Détecteur non installé. + +
+ ); + } + return ( +
+ {wake === 'spotting' ? : wake === 'error' ? : } + {wake === 'spotting' ? `à l’écoute de « ${keywords.join(' », « ')} »` : wake === 'error' ? error ?? 'erreur' : 'démarrage…'} +
+ ); +} + function TestResult({ t }: { t: TestState }) { if (t.status === 'idle') return null; return ( @@ -278,13 +305,39 @@ export function SettingsDrawer({ onClose }: Props) {
{ update({ voice: { handsFree: v } }); useVoice.getState().setHandsFree(v); }} label="Mains libres au démarrage" hint="Le micro se réactive automatiquement après chaque réponse." /> update({ voice: { wakeChime: v } })} label="Signal sonore d’écoute" /> - update({ voice: { wakeWordEnabled: v } })} label="Mot d’activation en mains libres" hint="Seules les phrases commençant par ce mot sont envoyées à Hermes ; le reste est ignoré. Recommandé avec la reconnaissance locale." /> - {settings.voice.wakeWordEnabled && ( -
- - update({ voice: { wakeWord: e.target.value.toLowerCase() } })} /> +
+ + + + {settings.voice.wakeMode === 'kws' + ? 'Détection locale par un modèle de 3 Mo, quasi gratuite en CPU. Dire le mot seul ouvre l’écoute ; dire le mot puis la commande envoie directement. Le mot coupe aussi la voix en cours.' + : settings.voice.wakeMode === 'transcript' + ? 'Chaque phrase est transcrite puis filtrée : plus coûteux, à réserver à la reconnaissance locale.' + : 'Le micro s’active avec le bouton, Ctrl+Shift+Espace ou la boucle mains libres.'} + +
+ {settings.voice.wakeMode !== 'off' && ( +
+
+ + update({ voice: { wakeWord: e.target.value.toLowerCase().slice(0, 40) } })} /> + Prononciation anglaise conseillée (« jarvis », « hey jarvis », « computer », « friday »…). +
+ {settings.voice.wakeMode === 'kws' && ( +
+ + update({ voice: { kwsSensitivity: Number(e.target.value) } })} /> +
+ )}
)} + {settings.voice.wakeMode === 'kws' && ( + + )}
diff --git a/src/services/voice/voiceController.ts b/src/services/voice/voiceController.ts index a8d9471..9307061 100644 --- a/src/services/voice/voiceController.ts +++ b/src/services/voice/voiceController.ts @@ -10,7 +10,9 @@ import { audioBus } from './audioBus'; import { listMicrophones, MicCapture } from './capture'; import { speech } from './speech'; import { BrowserRecognizer, transcribeWav } from './stt'; +import { WakeListener } from './wakeListener'; import type { WavResult } from './wav'; +import { bridge } from '../../lib/bridge'; const SENSITIVITY_RATIO: Record = { 1: 4.5, 2: 3.4, 3: 2.6, 4: 2.0, 5: 1.6 }; const SENSITIVITY_MIN_RMS: Record = { 1: 0.03, 2: 0.02, 3: 0.012, 4: 0.008, 5: 0.005 }; @@ -46,6 +48,8 @@ export function matchWakeWord(text: string, wakeWord: string): { matched: boolea class VoiceController { private capture = new MicCapture(); + private wake = new WakeListener(); + private unsubscribeKws: (() => void) | null = null; /** After a bare wake word, the next utterance is accepted without the wake word. */ private attentionUntil = 0; private browser = new BrowserRecognizer(); @@ -74,17 +78,104 @@ class VoiceController { dispose(): void { this.unsubscribeVoice?.(); this.stop(); + void this.stopWakeMode(); } get isListening(): boolean { return useVoice.getState().phase !== 'off'; } + get wakeActive(): boolean { + return this.wake.isActive; + } + toggle(): void { + if (this.wake.isActive) { + // Always-on mode: the button starts/cancels a command capture on the shared microphone. + if (this.wake.currentPhase === 'command' || this.wake.currentPhase === 'speech') this.wake.cancelCommand(); + else this.onWakeDetected('manuel'); + return; + } if (this.isListening) this.stop(); else void this.start(); } + /** Always-on keyword spotting: the microphone stays open and the main process spots the wake word. */ + async startWakeMode(): Promise { + const api = bridge(); + const voice = useVoice.getState(); + if (!api || this.wake.isActive) return; + const settings = useSettings.getState().settings.voice; + voice.setWake('starting'); + try { + const phrases = [settings.wakeWord || 'jarvis', `hey ${settings.wakeWord || 'jarvis'}`]; + const result = await api.voice.kwsStart({ modelId: 'kws-en', keywords: phrases, sensitivity: settings.kwsSensitivity }); + this.unsubscribeKws?.(); + this.unsubscribeKws = api.voice.onKwsDetected((d) => this.onWakeDetected(d.keyword)); + const sensitivity = Math.min(5, Math.max(1, Math.round(settings.sensitivity))) as 1 | 2 | 3 | 4 | 5; + await this.wake.start({ + deviceId: settings.micDeviceId || undefined, + vad: { silenceMs: settings.silenceMs, speechRatio: SENSITIVITY_RATIO[sensitivity], minRms: SENSITIVITY_MIN_RMS[sensitivity] }, + callbacks: { + onPhase: (phase) => { + const v = useVoice.getState(); + if (phase === 'spotting') { + v.setPhase('off'); + const chat = useChat.getState(); + if (chat.hud === 'listening') chat.setHud(chat.isSending ? 'thinking' : 'idle'); + } else if (phase === 'command') { + v.setPhase('listening'); + useChat.getState().setHud('listening'); + } else if (phase === 'speech') { + v.setPhase('speech'); + useChat.getState().ping(); + } + }, + onLevel: (level) => { + const q = Math.round(level * 20) / 20; + if (q !== useVoice.getState().inputLevel) useVoice.getState().setInputLevel(q); + }, + onUtterance: (wav) => void this.transcribe(wav, true), + onNoSpeech: () => useVoice.getState().setInputLevel(0), + onError: (message) => useVoice.getState().setError(message) + } + }); + voice.setWake('spotting', result.accepted.map((k) => k.replace(/_/g, ' '))); + Log.info('voice', `wake mode on: ${result.accepted.join(', ')}${result.rejected.length ? ` (rejetés : ${result.rejected.join(', ')})` : ''}`); + } catch (err) { + const message = (err as Error).message; + voice.setWake('error'); + voice.setError(message); + useChat.getState().setError(`Mot d’activation : ${message}`); + Log.error('voice', `wake mode failed: ${message}`); + await api.voice.kwsStop().catch(() => undefined); + } + } + + async stopWakeMode(): Promise { + this.unsubscribeKws?.(); + this.unsubscribeKws = null; + this.wake.stop(); + useVoice.getState().setWake('off'); + await bridge()?.voice.kwsStop().catch(() => undefined); + } + + /** Restart spotting with the current settings (wake word, sensitivity, microphone). */ + async restartWakeMode(): Promise { + await this.stopWakeMode(); + if (useSettings.getState().settings.voice.wakeMode === 'kws') await this.startWakeMode(); + } + + private onWakeDetected(keyword: string): void { + if (!this.wake.isActive || this.wake.currentPhase !== 'spotting') return; + if (useChat.getState().isSending) return; + Log.info('voice', `wake: ${keyword}`); + speech.stop(); + useChat.getState().ping(); + this.playChime(true); + this.wake.beginCommand(); + } + setHandsFree(on: boolean): void { useVoice.getState().setHandsFree(on); useSettings.getState().update({ voice: { handsFree: on } }); @@ -113,6 +204,11 @@ class VoiceController { async start(auto = false): Promise { const voice = useVoice.getState(); if (voice.phase !== 'off') return; + if (this.wake.isActive) { + // The always-on listener owns the microphone: open a command capture on it instead. + if (!auto) this.onWakeDetected('manuel'); + return; + } const settings = useSettings.getState().settings.voice; const seq = ++this.startSeq; voice.setError(null); @@ -218,7 +314,7 @@ class VoiceController { } } - private async transcribe(wav: WavResult): Promise { + private async transcribe(wav: WavResult, fromWake = false): Promise { const voice = useVoice.getState(); voice.setPhase('transcribing'); useChat.getState().setHud('thinking'); @@ -229,7 +325,8 @@ class VoiceController { voice.setTranscript(text); voice.setPhase('off'); Log.info('voice', `transcript (${wav.durationSec.toFixed(1)}s): ${text}`); - if (voice.handsFree && settings.wakeWordEnabled && Date.now() > this.attentionUntil) { + const filterByTranscript = settings.wakeMode === 'transcript' || (settings.wakeMode === 'off' && settings.wakeWordEnabled); + if (!fromWake && voice.handsFree && filterByTranscript && Date.now() > this.attentionUntil) { const { matched, rest } = matchWakeWord(text, settings.wakeWord || 'jarvis'); if (!matched) { Log.debug('voice', 'utterance ignored: no wake word'); diff --git a/src/services/voice/wakeListener.ts b/src/services/voice/wakeListener.ts new file mode 100644 index 0000000..ba5a52d --- /dev/null +++ b/src/services/voice/wakeListener.ts @@ -0,0 +1,216 @@ +/** + * Always-on microphone for the keyword spotter. Streams 16 kHz PCM to the main process + * (which feeds the sherpa-onnx spotter); once the wake word is detected it captures the + * following utterance with the energy VAD and hands back a WAV, then resumes spotting. + * One microphone stream, no re-opening between phrases. + */ +import { bridge } from '../../lib/bridge'; +import { Log } from '../../lib/log'; +import { audioBus } from './audioBus'; +import { DEFAULT_VAD, EnergyVad, type VadOptions } from './vad'; +import { buildWav16k, rms, type WavResult } from './wav'; + +const WORKLET_SOURCE = ` +class EveFlowWakeProcessor extends AudioWorkletProcessor { + constructor() { super(); this.buffer = new Float32Array(2048); this.offset = 0; } + process(inputs) { + const channel = inputs[0] && inputs[0][0]; + if (!channel) return true; + let i = 0; + while (i < channel.length) { + const n = Math.min(channel.length - i, this.buffer.length - this.offset); + this.buffer.set(channel.subarray(i, i + n), this.offset); + this.offset += n; i += n; + if (this.offset === this.buffer.length) { + this.port.postMessage(this.buffer, [this.buffer.buffer]); + this.buffer = new Float32Array(2048); this.offset = 0; + } + } + return true; + } +} +registerProcessor('eveflow-wake', EveFlowWakeProcessor); +`; + +export type WakePhase = 'off' | 'spotting' | 'command' | 'speech'; + +export interface WakeCallbacks { + onPhase: (phase: WakePhase) => void; + onLevel?: (level: number) => void; + onUtterance: (wav: WavResult) => void; + onNoSpeech: () => void; + onError: (message: string) => void; +} + +let workletUrl: string | null = null; + +export class WakeListener { + private ctx: AudioContext | null = null; + private stream: MediaStream | null = null; + private node: AudioWorkletNode | null = null; + private phase: WakePhase = 'off'; + private vad: EnergyVad | null = null; + private chunks: Float32Array[] = []; + private totalSamples = 0; + private sampleRate = 16_000; + private pending: Float32Array[] = []; + private pendingSamples = 0; + private callbacks: WakeCallbacks | null = null; + private vadOptions: Partial = {}; + private noSpeechTimer: ReturnType | null = null; + + get isActive(): boolean { + return this.phase !== 'off'; + } + + get currentPhase(): WakePhase { + return this.phase; + } + + async start(options: { deviceId?: string; vad?: Partial; callbacks: WakeCallbacks }): Promise { + if (this.phase !== 'off') return; + this.callbacks = options.callbacks; + this.vadOptions = options.vad ?? {}; + this.stream = await navigator.mediaDevices.getUserMedia({ + audio: { + deviceId: options.deviceId ? { exact: options.deviceId } : undefined, + echoCancellation: true, + noiseSuppression: true, + autoGainControl: true, + channelCount: 1 + } + }); + try { + this.ctx = new AudioContext({ sampleRate: 16_000, latencyHint: 'playback' }); + } catch { + this.ctx = new AudioContext({ latencyHint: 'playback' }); + } + if (this.ctx.state === 'suspended') await this.ctx.resume().catch(() => undefined); + this.sampleRate = this.ctx.sampleRate; + const source = this.ctx.createMediaStreamSource(this.stream); + const analyser = this.ctx.createAnalyser(); + analyser.fftSize = 512; + source.connect(analyser); + audioBus.setInputAnalyser(analyser); + if (!workletUrl) workletUrl = URL.createObjectURL(new Blob([WORKLET_SOURCE], { type: 'application/javascript' })); + await this.ctx.audioWorklet.addModule(workletUrl); + this.node = new AudioWorkletNode(this.ctx, 'eveflow-wake', { numberOfInputs: 1, numberOfOutputs: 0, channelCount: 1 }); + this.node.port.onmessage = (event: MessageEvent) => this.onSamples(event.data); + source.connect(this.node); + this.setPhase('spotting'); + Log.info('wake', `listener started (${this.sampleRate} Hz)`); + } + + /** Called by the controller when the main process reports the wake word (or on manual trigger). */ + beginCommand(): void { + if (this.phase === 'off' || this.phase === 'command' || this.phase === 'speech') return; + this.vad = new EnergyVad({ ...DEFAULT_VAD, ...this.vadOptions, noSpeechTimeoutMs: Number.POSITIVE_INFINITY }); + this.chunks = []; + this.totalSamples = 0; + this.setPhase('command'); + if (this.noSpeechTimer) clearTimeout(this.noSpeechTimer); + this.noSpeechTimer = setTimeout(() => { + if (this.phase === 'command') { + this.resumeSpotting(); + this.callbacks?.onNoSpeech(); + } + }, 8000); + } + + /** Abort a command capture and go back to spotting. */ + cancelCommand(): void { + if (this.phase === 'command' || this.phase === 'speech') this.resumeSpotting(); + } + + stop(): void { + if (this.phase === 'off') return; + if (this.noSpeechTimer) clearTimeout(this.noSpeechTimer); + this.noSpeechTimer = null; + audioBus.setInputAnalyser(null); + if (this.node) { + this.node.port.onmessage = null; + this.node.disconnect(); + this.node = null; + } + this.stream?.getTracks().forEach((t) => t.stop()); + this.stream = null; + void this.ctx?.close().catch(() => undefined); + this.ctx = null; + this.chunks = []; + this.pending = []; + this.pendingSamples = 0; + this.setPhase('off'); + Log.info('wake', 'listener stopped'); + } + + private setPhase(phase: WakePhase): void { + this.phase = phase; + this.callbacks?.onPhase(phase); + } + + private resumeSpotting(): void { + if (this.noSpeechTimer) clearTimeout(this.noSpeechTimer); + this.noSpeechTimer = null; + this.vad = null; + this.chunks = []; + this.totalSamples = 0; + this.setPhase('spotting'); + } + + private onSamples(samples: Float32Array): void { + if (this.phase === 'off') return; + const level = rms(samples); + this.callbacks?.onLevel?.(Math.min(1, level * 6)); + + if (this.phase === 'spotting') { + // Batch ~256 ms of audio per IPC message for the keyword spotter. + this.pending.push(samples); + this.pendingSamples += samples.length; + if (this.pendingSamples >= this.sampleRate * 0.25) this.flushToSpotter(); + return; + } + + // command / speech: collect the utterance + this.chunks.push(samples); + this.totalSamples += samples.length; + if (this.vad && this.phase === 'command') { + // keep only a short pre-roll before speech starts + while (this.chunks.length > 1 && this.totalSamples - this.chunks[0].length > this.sampleRate * 0.4) { + this.totalSamples -= this.chunks.shift()!.length; + } + } + if (!this.vad) return; + const signal = this.vad.feed(level, (samples.length / this.sampleRate) * 1000); + if (signal === 'speech-start') { + this.setPhase('speech'); + if (this.noSpeechTimer) clearTimeout(this.noSpeechTimer); + this.noSpeechTimer = null; + } else if (signal === 'speech-end' || signal === 'max-length') { + const wav = buildWav16k(this.chunks, this.sampleRate); + this.resumeSpotting(); + if (wav.durationSec >= 0.25) this.callbacks?.onUtterance(wav); + else this.callbacks?.onNoSpeech(); + } else if (signal === 'too-short') { + this.resumeSpotting(); + this.callbacks?.onNoSpeech(); + } + } + + private flushToSpotter(): void { + const api = bridge(); + if (!api) return; + const total = this.pendingSamples; + const merged = new Int16Array(total); + let offset = 0; + for (const chunk of this.pending) { + for (let i = 0; i < chunk.length; i++) { + const s = Math.max(-1, Math.min(1, chunk[i])); + merged[offset + i] = s < 0 ? s * 0x8000 : s * 0x7fff; + } + offset += chunk.length; + } + this.pending = []; + this.pendingSamples = 0; + api.voice.kwsAudio(new Uint8Array(merged.buffer), this.sampleRate); + } +} diff --git a/src/state/settings.ts b/src/state/settings.ts index 8255077..ade09a2 100644 --- a/src/state/settings.ts +++ b/src/state/settings.ts @@ -19,6 +19,9 @@ export interface VoiceSettings extends SttConfig { /** Hands-free: only react to utterances starting with this word (local STT recommended). */ wakeWordEnabled: boolean; wakeWord: string; + /** off = push-to-talk / hands-free; transcript = filter after transcription; kws = always-on keyword spotting. */ + wakeMode: 'off' | 'transcript' | 'kws'; + kwsSensitivity: number; // 1..5 } export interface SpeechSettings extends TtsConfig { @@ -83,6 +86,8 @@ export const DEFAULT_SETTINGS: Settings = { wakeChime: true, wakeWordEnabled: false, wakeWord: 'jarvis', + wakeMode: 'off', + kwsSensitivity: 3, localModel: 'whisper-base' }, speech: { @@ -181,6 +186,7 @@ export const useSettings = create((set, get) => ({ } const merged = merge(DEFAULT_SETTINGS, saved); if (!merged.hermesSessionId) merged.hermesSessionId = uid('eveflow'); + if (saved && saved.voice && saved.voice.wakeMode === undefined && saved.voice.wakeWordEnabled) merged.voice.wakeMode = 'transcript'; set({ settings: merged, loaded: true }); persistSet(STORAGE_KEY, merged); }, diff --git a/src/state/voice.ts b/src/state/voice.ts index d3323dd..d4a6c57 100644 --- a/src/state/voice.ts +++ b/src/state/voice.ts @@ -2,6 +2,7 @@ import { create } from 'zustand'; import type { TtsState } from '../services/voice/tts'; export type ListenPhase = 'off' | 'arming' | 'listening' | 'speech' | 'transcribing'; +export type WakeState = 'off' | 'starting' | 'spotting' | 'error'; interface VoiceStore { phase: ListenPhase; @@ -12,6 +13,9 @@ interface VoiceStore { interim: string; error: string | null; micDevices: Array<{ deviceId: string; label: string }>; + wake: WakeState; + wakeKeywords: string[]; + setWake: (state: WakeState, keywords?: string[]) => void; setPhase: (phase: ListenPhase) => void; setInputLevel: (level: number) => void; setTts: (state: TtsState) => void; @@ -31,6 +35,9 @@ export const useVoice = create((set) => ({ interim: '', error: null, micDevices: [], + wake: 'off', + wakeKeywords: [], + setWake: (wake, wakeKeywords) => set(wakeKeywords ? { wake, wakeKeywords } : { wake }), setPhase: (phase) => set({ phase }), setInputLevel: (inputLevel) => set({ inputLevel }), setTts: (tts) => set({ tts }), diff --git a/tests/keywords.test.ts b/tests/keywords.test.ts new file mode 100644 index 0000000..a9d76ef --- /dev/null +++ b/tests/keywords.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from 'vitest'; +import { buildKeywordsFile, encodeKeyword, normalizeKeyword, parseTokens } from '../shared/keywords'; + +const vocab = parseTokens([' 0', '▁ 3', '▁JA 4', 'R 5', 'VI 6', 'S 7', '▁HE 8', 'Y 9', '▁NO 10', 'V 11', 'A 12', '▁MA 13', 'X 14', '▁T 15', 'ON 16', 'Y 17'].join('\n')); + +describe('keywords', () => { + it('normalises accents, case and punctuation', () => { + expect(normalizeKeyword(' Hé, Jarvis ! ')).toBe('HE JARVIS'); + }); + it('uses the known SentencePiece encodings', () => { + expect(encodeKeyword('jarvis', vocab)).toBe('▁JA R VI S'); + expect(encodeKeyword('Hey Jarvis', vocab)).toBe('▁HE Y ▁JA R VI S'); + }); + it('falls back to greedy longest match', () => { + expect(encodeKeyword('max', vocab)).toBe('▁MA X'); + expect(encodeKeyword('tony', vocab)).toBe('▁T ON Y'); + expect(encodeKeyword('zzz', vocab)).toBeNull(); + }); + it('builds a keywords file with labels', () => { + const file = buildKeywordsFile(['jarvis', 'hey jarvis', 'zzz'], vocab); + expect(file.content).toBe('▁JA R VI S @jarvis\n▁HE Y ▁JA R VI S @hey_jarvis\n'); + expect(file.accepted).toEqual(['jarvis', 'hey_jarvis']); + expect(file.rejected).toEqual(['zzz']); + }); +});