diff --git a/README.md b/README.md index e1c4440..6eaebf3 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ # EveFlow 2 — Interface vocale JARVIS pour Hermes Agent [![Build](https://img.shields.io/github/actions/workflow/status/R0m1k3/EveFlow/windows-release.yml?style=flat-square)](https://github.com/R0m1k3/EveFlow/actions) -[![Version](https://img.shields.io/badge/version-2.0.0-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases) +[![Version](https://img.shields.io/badge/version-2.1.0-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases) [![License](https://img.shields.io/badge/license-MIT-lightgrey.svg?style=flat-square)](LICENSE) **EveFlow** est un compagnon de bureau Windows qui transforme [Hermes Agent](https://hermes-agent.nousresearch.com/) en assistant vocal à la JARVIS : un noyau holographique réactif au son, une conversation en streaming, les outils, sous-agents, approbations, crons, skills et sessions d'Hermes pilotés depuis un seul HUD. @@ -22,8 +22,10 @@ La version 2 est une réécriture complète : plus de robot 3D, un pipeline voca * **Capture micro** via AudioWorklet à 16 kHz, sans monitoring du micro dans les haut-parleurs, avec annulation d'écho et réduction de bruit. * **Détection d'activité vocale** (seuil adaptatif, sensibilité et silence de fin réglables) : l'enregistrement s'arrête tout seul quand vous avez fini de parler. * **Mains libres** : le micro se réactive après chaque réponse. -* **STT** : n'importe quelle API `/v1/audio/transcriptions` compatible OpenAI (Qwen3-ASR, Whisper, Speaches, faster-whisper-server, LocalAI, OpenAI). Repli sur la reconnaissance Chromium. -* **TTS** : API `/v1/audio/speech` compatible OpenAI (Kokoro, Piper, OpenAI…), voix système Windows ou Google Translate. Lecture phrase par phrase pendant le streaming, préchargement du segment suivant, coupure instantanée. +* **Modèles intégrés, hors ligne** (sherpa-onnx dans un processus séparé) : reconnaissance Whisper (base, small, large-v3 turbo) ou SenseVoice, synthèse Kokoro v1.0 (voix française Siwis et voix anglaises) ou Piper (Siwis, Tom, UPMC). Les modèles se téléchargent depuis **Paramètres → Modèles locaux** et tournent sur le processeur. +* **Mot d'activation** en mains libres : seules les phrases commençant par « Jarvis » (configurable) partent vers Hermes, le reste est ignoré ; un « Jarvis » seul ouvre une fenêtre d'écoute. +* **STT externe** : n'importe quelle API `/v1/audio/transcriptions` compatible OpenAI (Qwen3-ASR, Whisper, Speaches, faster-whisper-server, LocalAI, OpenAI). Repli sur la reconnaissance Chromium. +* **TTS externe** : API `/v1/audio/speech` compatible OpenAI, voix système Windows ou Google Translate. Lecture phrase par phrase pendant le streaming, préchargement du segment suivant, coupure instantanée. * Raccourcis globaux : `Ctrl+Shift+Espace` (micro), `Ctrl+Shift+J` (afficher/masquer), `Ctrl+Shift+Échap` (couper la voix). ### Hermes, toute la puissance @@ -49,10 +51,12 @@ La version 2 est une réécriture complète : plus de robot 3D, un pipeline voca * Electron 44 (sandbox, contextIsolation, `webSecurity` actif, proxy HTTP en streaming dans le main process) * React 19 + Vite 8 + TypeScript 5.9 + Zustand * Canvas 2D, Web Audio (AudioWorklet, AnalyserNode) +* sherpa-onnx (ONNX Runtime) pour la voix locale, exécuté dans un `utilityProcess` Electron * Vitest pour les tests unitaires (SSE, VAD, WAV, normalisation d'événements Hermes) ``` electron/ processus principal (fenêtre, tray, raccourcis, IPC, webhook, proxy HTTP) +electron/voice catalogue de modèles, téléchargement, worker sherpa-onnx (STT/TTS) shared/ contrat IPC + normalisation des pushs webhook (main + renderer) src/lib transport, SSE, persistance, utilitaires texte src/services hermes/ (client, événements, outils locaux) voice/ (capture, VAD, STT, TTS) diff --git a/electron/main.ts b/electron/main.ts index 271c18e..f518430 100644 --- a/electron/main.ts +++ b/electron/main.ts @@ -8,6 +8,8 @@ import { registerTelemetryIpc } from './ipc/telemetry'; import { getSharedDirectory, registerFilesIpc } from './ipc/files'; import { createMainWindow, getMainWindow, getWindowMode, setWindowMode, toggleWindowVisibility } from './window'; import { getWebhookStatus, startWebhookServer, stopWebhookServer } from './webhook'; +import { registerVoiceIpc } from './voice/ipc'; +import { stopEngine } from './voice/engine'; // Audio playback must never be blocked behind a user gesture (TTS starts on incoming events). app.commandLine.appendSwitch('autoplay-policy', 'no-user-gesture-required'); @@ -149,6 +151,7 @@ app.whenReady().then(async () => { registerTelemetryIpc(); registerFilesIpc(); registerCoreIpc(); + registerVoiceIpc(); createMainWindow(); createTray(); registerShortcuts(); @@ -162,6 +165,7 @@ app.whenReady().then(async () => { app.on('will-quit', () => { globalShortcut.unregisterAll(); abortAllStreams(); + stopEngine(); void stopWebhookServer(); }); diff --git a/electron/preload.ts b/electron/preload.ts index bfd949e..053027a 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -14,6 +14,7 @@ import { type WindowMode } from '../shared/ipc'; import type { EveFlowBridge, Unsubscribe } from '../shared/bridge'; +import { VOICE_IPC, type SynthesizeRequest, type SynthesizeResult, type TranscribeRequest, type TranscribeResult, type VoiceDownloadProgress, type VoiceEngineStatus, type VoiceModelStatus } from '../shared/voice'; function subscribe(channel: string, callback: (payload: T) => void): Unsubscribe { const listener = (_event: Electron.IpcRendererEvent, payload: T) => callback(payload); @@ -57,6 +58,17 @@ const api: EveFlowBridge = { }, hotkeys: { on: (cb: (event: HotkeyEvent) => void) => subscribe(IPC.hotkey, cb) + }, + voice: { + status: () => ipcRenderer.invoke(VOICE_IPC.status) as Promise, + listModels: () => ipcRenderer.invoke(VOICE_IPC.modelsList) as Promise, + downloadModel: (id: string) => ipcRenderer.invoke(VOICE_IPC.modelsDownload, id) as Promise, + cancelDownload: (id: string) => ipcRenderer.invoke(VOICE_IPC.modelsCancel, id) as Promise, + removeModel: (id: string) => ipcRenderer.invoke(VOICE_IPC.modelsRemove, id) as Promise, + onProgress: (cb: (progress: VoiceDownloadProgress) => void) => subscribe(VOICE_IPC.modelsProgress, cb), + transcribe: (req: TranscribeRequest) => ipcRenderer.invoke(VOICE_IPC.transcribe, req) as Promise, + synthesize: (req: SynthesizeRequest) => ipcRenderer.invoke(VOICE_IPC.synthesize, req) as Promise, + unload: (id?: string) => ipcRenderer.invoke(VOICE_IPC.unload, id) as Promise } }; diff --git a/electron/voice/catalog.ts b/electron/voice/catalog.ts new file mode 100644 index 0000000..6a57a90 --- /dev/null +++ b/electron/voice/catalog.ts @@ -0,0 +1,133 @@ +import type { VoiceModelSpec, VoiceSpeaker } from '../../shared/voice'; + +const ASR = 'https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models'; +const TTS = 'https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models'; + +const KOKORO_SPEAKERS: VoiceSpeaker[] = [ + { id: 30, name: 'Siwis (femme, français)', lang: 'fr' }, + { id: 3, name: 'Heart (femme, anglais US)', lang: 'en' }, + { id: 2, name: 'Bella (femme, anglais US)', lang: 'en' }, + { id: 7, name: 'Nova (femme, anglais US)', lang: 'en' }, + { id: 11, name: 'Adam (homme, anglais US)', lang: 'en' }, + { id: 16, name: 'Michael (homme, anglais US)', lang: 'en' }, + { id: 17, name: 'Onyx (homme, anglais US)', lang: 'en' }, + { id: 21, name: 'Emma (femme, anglais UK)', lang: 'en' }, + { id: 24, name: 'Daniel (homme, anglais UK)', lang: 'en' }, + { id: 26, name: 'George (homme, anglais UK)', lang: 'en' } +]; + +export const VOICE_CATALOG: VoiceModelSpec[] = [ + { + id: 'whisper-base', + kind: 'stt', + engine: 'whisper', + name: 'Whisper base (multilingue)', + description: 'Rapide sur CPU, bonne qualité en français pour des commandes courtes.', + languages: ['fr', 'en', 'multi'], + sizeMb: 208, + url: `${ASR}/sherpa-onnx-whisper-base.tar.bz2`, + dir: 'sherpa-onnx-whisper-base', + files: ['base-encoder.int8.onnx', 'base-decoder.int8.onnx', 'base-tokens.txt'], + recommended: true + }, + { + id: 'whisper-small', + kind: 'stt', + engine: 'whisper', + name: 'Whisper small (multilingue)', + description: 'Plus précis que base, environ trois fois plus lent.', + languages: ['fr', 'en', 'multi'], + sizeMb: 640, + url: `${ASR}/sherpa-onnx-whisper-small.tar.bz2`, + dir: 'sherpa-onnx-whisper-small', + files: ['small-encoder.int8.onnx', 'small-decoder.int8.onnx', 'small-tokens.txt'] + }, + { + id: 'whisper-turbo', + kind: 'stt', + engine: 'whisper', + name: 'Whisper large-v3 turbo (multilingue)', + description: 'La meilleure précision ; demande un processeur puissant.', + languages: ['fr', 'en', 'multi'], + sizeMb: 564, + url: `${ASR}/sherpa-onnx-whisper-turbo.tar.bz2`, + dir: 'sherpa-onnx-whisper-turbo', + files: ['turbo-encoder.int8.onnx', 'turbo-decoder.int8.onnx', 'turbo-tokens.txt'] + }, + { + id: 'sense-voice', + kind: 'stt', + engine: 'sense-voice', + name: 'SenseVoice small (zh / en / ja / ko)', + description: 'Très rapide, mais ne comprend pas le français.', + languages: ['en', 'zh', 'ja', 'ko'], + sizeMb: 163, + url: `${ASR}/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-int8-2024-07-17.tar.bz2`, + dir: 'sherpa-onnx-sense-voice-zh-en-ja-ko-yue-int8-2024-07-17', + files: ['model.int8.onnx', 'tokens.txt'] + }, + { + id: 'kokoro-v1', + kind: 'tts', + engine: 'kokoro', + name: 'Kokoro v1.0 multilingue', + description: 'Voix très naturelle, une voix française (Siwis) et de nombreuses voix anglaises. 24 kHz.', + languages: ['fr', 'en', 'multi'], + sizeMb: 349, + url: `${TTS}/kokoro-multi-lang-v1_0.tar.bz2`, + dir: 'kokoro-multi-lang-v1_0', + files: ['model.onnx', 'voices.bin', 'tokens.txt', 'lexicon-us-en.txt', 'lexicon-zh.txt', 'espeak-ng-data/phontab'], + speakers: KOKORO_SPEAKERS, + sampleRate: 24000, + recommended: true + }, + { + id: 'piper-fr-siwis', + kind: 'tts', + engine: 'piper', + name: 'Piper Siwis (femme, français)', + description: 'Léger et instantané, rendu un peu plus mécanique que Kokoro. 22 kHz.', + languages: ['fr'], + sizeMb: 67, + url: `${TTS}/vits-piper-fr_FR-siwis-medium.tar.bz2`, + dir: 'vits-piper-fr_FR-siwis-medium', + files: ['fr_FR-siwis-medium.onnx', 'tokens.txt', 'espeak-ng-data/phontab'], + speakers: [{ id: 0, name: 'Siwis', lang: 'fr' }], + sampleRate: 22050 + }, + { + id: 'piper-fr-tom', + kind: 'tts', + engine: 'piper', + name: 'Piper Tom (homme, français)', + description: 'Voix masculine légère et instantanée. 22 kHz.', + languages: ['fr'], + sizeMb: 67, + url: `${TTS}/vits-piper-fr_FR-tom-medium.tar.bz2`, + dir: 'vits-piper-fr_FR-tom-medium', + files: ['fr_FR-tom-medium.onnx', 'tokens.txt', 'espeak-ng-data/phontab'], + speakers: [{ id: 0, name: 'Tom', lang: 'fr' }], + sampleRate: 22050 + }, + { + id: 'piper-fr-upmc', + kind: 'tts', + engine: 'piper', + name: 'Piper UPMC (Jessica et Pierre, français)', + description: 'Deux voix françaises, une féminine et une masculine. 22 kHz.', + languages: ['fr'], + sizeMb: 80, + url: `${TTS}/vits-piper-fr_FR-upmc-medium.tar.bz2`, + dir: 'vits-piper-fr_FR-upmc-medium', + files: ['fr_FR-upmc-medium.onnx', 'tokens.txt', 'espeak-ng-data/phontab'], + speakers: [ + { id: 0, name: 'Jessica (femme)', lang: 'fr' }, + { id: 1, name: 'Pierre (homme)', lang: 'fr' } + ], + sampleRate: 22050 + } +]; + +export function findModel(id: string): VoiceModelSpec | undefined { + return VOICE_CATALOG.find((m) => m.id === id); +} diff --git a/electron/voice/engine.ts b/electron/voice/engine.ts new file mode 100644 index 0000000..c17f962 --- /dev/null +++ b/electron/voice/engine.ts @@ -0,0 +1,96 @@ +/** + * Host side of the voice worker: spawns the utility process on demand, correlates + * requests and responses, restarts the worker if it crashes. + */ +import { utilityProcess, type UtilityProcess } from 'electron'; +import path from 'node:path'; +import type { SynthesizeRequest, SynthesizeResult, TranscribeRequest, TranscribeResult, VoiceEngineStatus } from '../../shared/voice'; +import { findModel } from './catalog'; +import { isInstalled, modelDir, modelsDir } from './models'; +import { log } from '../logger'; + +interface Pending { + resolve: (value: unknown) => void; + reject: (err: Error) => void; + timer: ReturnType; +} + +let child: UtilityProcess | null = null; +let nextId = 1; +const pending = new Map(); + +function spawn(): UtilityProcess { + if (child) return child; + const script = path.join(__dirname, 'voice-worker.js'); + child = utilityProcess.fork(script, [], { serviceName: 'eveflow-voice', stdio: 'pipe' }); + child.stdout?.on('data', (d: Buffer) => log('DEBUG', 'voice-worker', d.toString().trim())); + child.stderr?.on('data', (d: Buffer) => log('WARN', 'voice-worker', d.toString().trim())); + child.on('message', (msg: { id: number; ok: boolean; result?: unknown; error?: string }) => { + const p = pending.get(msg.id); + if (!p) return; + pending.delete(msg.id); + clearTimeout(p.timer); + if (msg.ok) p.resolve(msg.result); + else p.reject(new Error(msg.error ?? 'erreur moteur vocal')); + }); + child.on('exit', (code) => { + log(code === 0 ? 'INFO' : 'ERROR', 'voice-worker', `exited with code ${code}`); + child = null; + for (const [id, p] of pending) { + clearTimeout(p.timer); + p.reject(new Error('Le moteur vocal local s’est arrêté de façon inattendue.')); + pending.delete(id); + } + }); + log('INFO', 'voice', 'voice worker started'); + return child; +} + +function request(message: Record, timeoutMs = 120_000): Promise { + const proc = spawn(); + const id = nextId++; + return new Promise((resolve, reject) => { + const timer = setTimeout(() => { + pending.delete(id); + reject(new Error('Délai dépassé pour le moteur vocal local.')); + }, timeoutMs); + pending.set(id, { resolve: (v) => resolve(v as T), reject, timer }); + proc.postMessage({ id, ...message }); + }); +} + +function modelRef(modelId: string) { + const spec = findModel(modelId); + if (!spec) throw new Error(`Modèle inconnu : ${modelId}`); + if (!isInstalled(spec)) throw new Error(`Modèle « ${spec.name} » non installé. Téléchargez-le dans Paramètres → Modèles.`); + return { id: spec.id, engine: spec.engine, dir: modelDir(spec), files: spec.files }; +} + +export async function engineStatus(): Promise { + try { + const status = await request>({ type: 'status' }, 20_000); + return { ...status, modelsDir: modelsDir() }; + } catch (err) { + return { available: false, error: (err as Error).message, loaded: [], modelsDir: modelsDir() }; + } +} + +export function transcribe(req: TranscribeRequest): Promise { + return request({ type: 'transcribe', model: modelRef(req.modelId), wav: req.wav, language: req.language }, 180_000); +} + +export function synthesize(req: SynthesizeRequest): Promise { + return request({ type: 'synthesize', model: modelRef(req.modelId), text: req.text, speaker: req.speaker, speed: req.speed }, 180_000); +} + +export function unload(modelId?: string): Promise { + if (!child) return Promise.resolve({ ok: true }); + return request({ type: 'unload', modelId }, 20_000); +} + +export function stopEngine(): void { + if (child) { + child.kill(); + child = null; + } +} diff --git a/electron/voice/ipc.ts b/electron/voice/ipc.ts new file mode 100644 index 0000000..233458b --- /dev/null +++ b/electron/voice/ipc.ts @@ -0,0 +1,22 @@ +import { ipcMain } from 'electron'; +import { VOICE_IPC, type SynthesizeRequest, type TranscribeRequest } from '../../shared/voice'; +import { engineStatus, synthesize, transcribe, unload } from './engine'; +import { cancelDownload, downloadModel, listModels, removeModel } from './models'; + +export function registerVoiceIpc(): void { + ipcMain.handle(VOICE_IPC.status, () => engineStatus()); + ipcMain.handle(VOICE_IPC.modelsList, () => listModels()); + ipcMain.handle(VOICE_IPC.modelsDownload, (event, id: string) => downloadModel(id, event.sender)); + ipcMain.handle(VOICE_IPC.modelsCancel, (_e, id: string) => { + cancelDownload(id); + return true; + }); + ipcMain.handle(VOICE_IPC.modelsRemove, async (_e, id: string) => { + await unload(id).catch(() => undefined); + await removeModel(id); + return listModels(); + }); + ipcMain.handle(VOICE_IPC.transcribe, (_e, req: TranscribeRequest) => transcribe(req)); + ipcMain.handle(VOICE_IPC.synthesize, (_e, req: SynthesizeRequest) => synthesize(req)); + ipcMain.handle(VOICE_IPC.unload, (_e, id?: string) => unload(id)); +} diff --git a/electron/voice/models.ts b/electron/voice/models.ts new file mode 100644 index 0000000..b0fe274 --- /dev/null +++ b/electron/voice/models.ts @@ -0,0 +1,135 @@ +/** + * Model manager: downloads sherpa-onnx model archives (tar.bz2) into userData/models, + * extracts them, verifies required files and reports progress to the renderer. + */ +import { app, type WebContents } from 'electron'; +import fs from 'node:fs'; +import path from 'node:path'; +import { pipeline } from 'node:stream/promises'; +import { Readable, Transform } from 'node:stream'; +import * as tar from 'tar'; +import unbzip2 from 'unbzip2-stream'; +import { VOICE_IPC, type VoiceDownloadProgress, type VoiceModelSpec, type VoiceModelStatus } from '../../shared/voice'; +import { VOICE_CATALOG, findModel } from './catalog'; +import { log } from '../logger'; + +const active = new Map(); + +export function modelsDir(): string { + const dir = path.join(app.getPath('userData'), 'models'); + fs.mkdirSync(dir, { recursive: true }); + return dir; +} + +export function modelDir(spec: VoiceModelSpec): string { + return path.join(modelsDir(), spec.dir); +} + +export function isInstalled(spec: VoiceModelSpec): boolean { + const dir = modelDir(spec); + return spec.files.every((f) => fs.existsSync(path.join(dir, f))); +} + +function dirSize(dir: string): number { + let total = 0; + try { + for (const entry of fs.readdirSync(dir, { withFileTypes: true })) { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) total += dirSize(full); + else total += fs.statSync(full).size; + } + } catch { + /* missing */ + } + return total; +} + +export function listModels(): VoiceModelStatus[] { + return VOICE_CATALOG.map((spec) => { + const installed = isInstalled(spec); + return { + ...spec, + installed, + downloading: active.has(spec.id), + installedBytes: installed ? dirSize(modelDir(spec)) : 0 + }; + }); +} + +export async function removeModel(id: string): Promise { + const spec = findModel(id); + if (!spec) throw new Error(`Modèle inconnu : ${id}`); + await fs.promises.rm(modelDir(spec), { recursive: true, force: true }); + log('INFO', 'voice', `model removed: ${id}`); +} + +export function cancelDownload(id: string): void { + active.get(id)?.abort(); +} + +export async function downloadModel(id: string, sender: WebContents | null): Promise { + const spec = findModel(id); + if (!spec) throw new Error(`Modèle inconnu : ${id}`); + if (active.has(id)) throw new Error('Téléchargement déjà en cours'); + const controller = new AbortController(); + active.set(id, controller); + const report = (progress: Omit) => { + if (sender && !sender.isDestroyed()) sender.send(VOICE_IPC.modelsProgress, { id, ...progress }); + }; + + const tmpDir = path.join(modelsDir(), `.tmp-${spec.id}`); + await fs.promises.rm(tmpDir, { recursive: true, force: true }); + await fs.promises.mkdir(tmpDir, { recursive: true }); + + try { + log('INFO', 'voice', `downloading ${spec.id} from ${spec.url}`); + const response = await fetch(spec.url, { signal: controller.signal, redirect: 'follow' }); + if (!response.ok || !response.body) throw new Error(`HTTP ${response.status} lors du téléchargement`); + const total = Number(response.headers.get('content-length')) || Math.round(spec.sizeMb * 1024 * 1024); + let received = 0; + let lastReport = 0; + const counter = new Transform({ + transform(chunk: Buffer, _enc, cb) { + received += chunk.length; + const now = Date.now(); + if (now - lastReport > 250) { + lastReport = now; + report({ phase: 'download', received, total, percent: Math.min(99, Math.round((received / total) * 100)) }); + } + cb(null, chunk); + } + }); + + const source = Readable.fromWeb(response.body as import('node:stream/web').ReadableStream); + if (spec.url.endsWith('.tar.bz2')) { + report({ phase: 'download', received: 0, total, percent: 0 }); + await pipeline(source, counter, unbzip2(), tar.x({ cwd: tmpDir }), { signal: controller.signal }); + } else { + const target = path.join(tmpDir, spec.dir, path.basename(spec.url)); + await fs.promises.mkdir(path.dirname(target), { recursive: true }); + await pipeline(source, counter, fs.createWriteStream(target), { signal: controller.signal }); + } + report({ phase: 'extract', received: total, total, percent: 99 }); + + const extracted = path.join(tmpDir, spec.dir); + const missing = spec.files.filter((f) => !fs.existsSync(path.join(extracted, f))); + if (missing.length) throw new Error(`Archive incomplète, fichiers manquants : ${missing.join(', ')}`); + + const finalDir = modelDir(spec); + await fs.promises.rm(finalDir, { recursive: true, force: true }); + await fs.promises.rename(extracted, finalDir); + await fs.promises.rm(tmpDir, { recursive: true, force: true }); + report({ phase: 'done', received: total, total, percent: 100 }); + log('INFO', 'voice', `model installed: ${spec.id}`); + } catch (err) { + await fs.promises.rm(tmpDir, { recursive: true, force: true }).catch(() => undefined); + const e = err as Error; + const cancelled = e.name === 'AbortError' || controller.signal.aborted; + report({ phase: cancelled ? 'cancelled' : 'error', received: 0, total: 0, percent: 0, message: cancelled ? 'annulé' : e.message }); + log(cancelled ? 'INFO' : 'ERROR', 'voice', `download ${spec.id} ${cancelled ? 'cancelled' : 'failed: ' + e.message}`); + if (!cancelled) throw new Error(e.message); + } finally { + active.delete(id); + } + return listModels().find((m) => m.id === id)!; +} diff --git a/electron/voice/worker.ts b/electron/voice/worker.ts new file mode 100644 index 0000000..9930b8a --- /dev/null +++ b/electron/voice/worker.ts @@ -0,0 +1,273 @@ +/** + * Voice worker: runs sherpa-onnx (speech recognition and synthesis) in an Electron utility + * process so heavy inference never blocks the main process. Also runnable with + * `child_process.fork` (advanced serialization) for local tests. + */ +import os from 'node:os'; +import path from 'node:path'; +import type { VoiceEngineKind } from '../../shared/voice'; + +interface ModelRef { + id: string; + engine: VoiceEngineKind; + dir: string; + files: string[]; +} + +type Request = + | { id: number; type: 'status' } + | { id: number; type: 'transcribe'; model: ModelRef; wav: Uint8Array; language: string } + | { id: number; type: 'synthesize'; model: ModelRef; text: string; speaker: number; speed: number } + | { id: number; type: 'unload'; modelId?: string }; + +type Response = { id: number; ok: true; result: unknown } | { id: number; ok: false; error: string }; + +// ── sherpa-onnx loading (lazy, so a missing native package is reported, not fatal) ── +type Sherpa = { + OfflineRecognizer: new (config: unknown) => { + createStream: () => { acceptWaveform: (w: { sampleRate: number; samples: Float32Array }) => void }; + decode: (s: unknown) => void; + getResult: (s: unknown) => { text: string; lang?: string }; + }; + OfflineTts: new (config: unknown) => { + numSpeakers: number; + sampleRate: number; + generate: (req: { text: string; sid: number; speed: number }) => { samples: Float32Array; sampleRate: number }; + }; + version: string; +}; + +let sherpa: Sherpa | null = null; +let loadError: string | null = null; + +function loadSherpa(): Sherpa { + if (sherpa) return sherpa; + if (loadError) throw new Error(loadError); + try { + // eslint-disable-next-line @typescript-eslint/no-require-imports + sherpa = require('sherpa-onnx-node') as Sherpa; + return sherpa; + } catch (err) { + loadError = `Module natif sherpa-onnx indisponible : ${(err as Error).message}`; + throw new Error(loadError); + } +} + +const threads = Math.max(2, Math.min(6, Math.floor(os.cpus().length / 2))); + +// ── caches ──────────────────────────────────────────────────────────────── +const recognizers = new Map>(); +const synthesizers = new Map>(); + +function whisperPrefix(model: ModelRef): string { + const encoder = model.files.find((f) => f.includes('-encoder')); + return encoder ? encoder.slice(0, encoder.indexOf('-encoder')) : 'base'; +} + +function getRecognizer(model: ModelRef, language: string) { + const lang = language === 'auto' ? '' : language.split('-')[0].toLowerCase(); + const key = `${model.id}:${lang}`; + const cached = recognizers.get(key); + if (cached) return cached; + const s = loadSherpa(); + const p = (f: string) => path.join(model.dir, f); + let modelConfig: Record; + switch (model.engine) { + case 'whisper': { + const prefix = whisperPrefix(model); + modelConfig = { + whisper: { encoder: p(`${prefix}-encoder.int8.onnx`), decoder: p(`${prefix}-decoder.int8.onnx`), language: lang, task: 'transcribe', tailPaddings: -1 }, + tokens: p(`${prefix}-tokens.txt`) + }; + break; + } + case 'sense-voice': + modelConfig = { senseVoice: { model: p('model.int8.onnx'), language: lang || 'auto', useInverseTextNormalization: 1 }, tokens: p('tokens.txt') }; + break; + case 'nemo-transducer': + modelConfig = { + transducer: { encoder: p('encoder.int8.onnx'), decoder: p('decoder.int8.onnx'), joiner: p('joiner.int8.onnx') }, + tokens: p('tokens.txt'), + modelType: 'nemo_transducer' + }; + break; + default: + throw new Error(`Moteur STT non supporté : ${model.engine}`); + } + for (const [k, v] of Object.entries({ numThreads: threads, provider: 'cpu', debug: 0 })) modelConfig[k] = v; + // Whisper keeps one recognizer per language; other engines ignore the language key. + for (const [k, r] of recognizers) if (k.startsWith(`${model.id}:`)) recognizers.delete(k) && void r; + const recognizer = new s.OfflineRecognizer({ featConfig: { sampleRate: 16000, featureDim: 80 }, modelConfig, decodingMethod: 'greedy_search' }); + recognizers.set(key, recognizer); + return recognizer; +} + +function getSynthesizer(model: ModelRef) { + const cached = synthesizers.get(model.id); + if (cached) return cached; + const s = loadSherpa(); + const p = (f: string) => path.join(model.dir, f); + let ttsModel: Record; + switch (model.engine) { + case 'kokoro': + ttsModel = { + kokoro: { + model: p('model.onnx'), + voices: p('voices.bin'), + tokens: p('tokens.txt'), + dataDir: p('espeak-ng-data'), + lexicon: [p('lexicon-us-en.txt'), p('lexicon-zh.txt')].join(',') + } + }; + break; + case 'piper': { + const onnx = model.files.find((f) => f.endsWith('.onnx')) ?? 'model.onnx'; + ttsModel = { vits: { model: p(onnx), tokens: p('tokens.txt'), dataDir: p('espeak-ng-data') } }; + break; + } + default: + throw new Error(`Moteur TTS non supporté : ${model.engine}`); + } + const tts = new s.OfflineTts({ model: { ...ttsModel, numThreads: threads, provider: 'cpu', debug: 0 }, maxNumSentences: 1 }); + synthesizers.set(model.id, tts); + return tts; +} + +// ── audio helpers ────────────────────────────────────────────────────────── +function decodeWav(bytes: Uint8Array): { samples: Float32Array; sampleRate: number } { + const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength); + if (String.fromCharCode(bytes[0], bytes[1], bytes[2], bytes[3]) !== 'RIFF') throw new Error('WAV invalide'); + let offset = 12; + let sampleRate = 16000; + let channels = 1; + let bits = 16; + let data: { start: number; length: number } | null = null; + while (offset + 8 <= bytes.byteLength) { + const id = String.fromCharCode(bytes[offset], bytes[offset + 1], bytes[offset + 2], bytes[offset + 3]); + const size = view.getUint32(offset + 4, true); + if (id === 'fmt ') { + channels = view.getUint16(offset + 10, true); + sampleRate = view.getUint32(offset + 12, true); + bits = view.getUint16(offset + 22, true); + } else if (id === 'data') { + data = { start: offset + 8, length: Math.min(size, bytes.byteLength - offset - 8) }; + break; + } + offset += 8 + size + (size % 2); + } + if (!data) throw new Error('WAV sans données'); + const bytesPerSample = bits / 8; + const frames = Math.floor(data.length / bytesPerSample / channels); + const samples = new Float32Array(frames); + for (let i = 0; i < frames; i++) { + let sum = 0; + for (let c = 0; c < channels; c++) { + const pos = data.start + (i * channels + c) * bytesPerSample; + sum += bits === 16 ? view.getInt16(pos, true) / 32768 : bits === 32 ? view.getInt32(pos, true) / 2147483648 : (bytes[pos] - 128) / 128; + } + samples[i] = sum / channels; + } + return { samples, sampleRate }; +} + +function resampleTo16k(samples: Float32Array, rate: number): Float32Array { + if (rate === 16000) return samples; + const ratio = rate / 16000; + const out = new Float32Array(Math.round(samples.length / ratio)); + for (let i = 0; i < out.length; i++) { + const pos = i * ratio; + const idx = Math.floor(pos); + const frac = pos - idx; + const a = samples[Math.min(idx, samples.length - 1)]; + const b = samples[Math.min(idx + 1, samples.length - 1)]; + out[i] = a + (b - a) * frac; + } + return out; +} + +function encodeWav(samples: Float32Array, sampleRate: number): Uint8Array { + const buffer = new ArrayBuffer(44 + samples.length * 2); + const view = new DataView(buffer); + const str = (o: number, s: string) => { + for (let i = 0; i < s.length; i++) view.setUint8(o + i, s.charCodeAt(i)); + }; + str(0, 'RIFF'); + view.setUint32(4, 36 + samples.length * 2, true); + str(8, 'WAVE'); + str(12, 'fmt '); + view.setUint32(16, 16, true); + view.setUint16(20, 1, true); + view.setUint16(22, 1, true); + view.setUint32(24, sampleRate, true); + view.setUint32(28, sampleRate * 2, true); + view.setUint16(32, 2, true); + view.setUint16(34, 16, true); + str(36, 'data'); + view.setUint32(40, samples.length * 2, true); + let o = 44; + for (let i = 0; i < samples.length; i++, o += 2) { + const s = Math.max(-1, Math.min(1, samples[i])); + view.setInt16(o, s < 0 ? s * 0x8000 : s * 0x7fff, true); + } + return new Uint8Array(buffer); +} + +// ── request handling ─────────────────────────────────────────────────────── +function handle(req: Request): unknown { + switch (req.type) { + case 'status': { + try { + const s = loadSherpa(); + return { available: true, version: s.version, loaded: [...recognizers.keys(), ...synthesizers.keys()] }; + } catch (err) { + return { available: false, error: (err as Error).message, loaded: [] }; + } + } + case 'transcribe': { + const started = Date.now(); + const { samples, sampleRate } = decodeWav(req.wav); + const pcm = resampleTo16k(samples, sampleRate); + const recognizer = getRecognizer(req.model, req.language); + const stream = recognizer.createStream(); + stream.acceptWaveform({ sampleRate: 16000, samples: pcm }); + recognizer.decode(stream); + const result = recognizer.getResult(stream); + return { text: (result.text ?? '').trim(), language: result.lang, durationMs: Date.now() - started, audioSec: pcm.length / 16000 }; + } + case 'synthesize': { + const started = Date.now(); + const tts = getSynthesizer(req.model); + const sid = Math.max(0, Math.min(tts.numSpeakers - 1, Math.floor(req.speaker))); + const audio = tts.generate({ text: req.text, sid, speed: Math.max(0.5, Math.min(2, req.speed || 1)) }); + return { wav: encodeWav(audio.samples, audio.sampleRate), sampleRate: audio.sampleRate, durationMs: Date.now() - started, audioSec: audio.samples.length / audio.sampleRate }; + } + case 'unload': { + if (req.modelId) { + for (const k of [...recognizers.keys()]) if (k.startsWith(`${req.modelId}:`)) recognizers.delete(k); + synthesizers.delete(req.modelId); + } else { + recognizers.clear(); + synthesizers.clear(); + } + return { ok: true }; + } + default: + throw new Error('requête inconnue'); + } +} + +function respond(req: Request): Response { + try { + return { id: req.id, ok: true, result: handle(req) }; + } catch (err) { + return { id: req.id, ok: false, error: (err as Error).message || String(err) }; + } +} + +// Electron utility process transport, with a child_process fallback for tests. +const parentPort = (process as unknown as { parentPort?: { on: (ev: 'message', cb: (e: { data: Request }) => void) => void; postMessage: (m: unknown) => void } }).parentPort; +if (parentPort) { + parentPort.on('message', (event) => parentPort.postMessage(respond(event.data))); +} else if (process.send) { + process.on('message', (msg: Request) => process.send!(respond(msg))); +} diff --git a/package-lock.json b/package-lock.json index 9036a06..6fcdabc 100644 --- a/package-lock.json +++ b/package-lock.json @@ -14,12 +14,14 @@ "react-dom": "^19.2.8", "react-markdown": "^10.1.0", "remark-gfm": "^4.0.1", + "sherpa-onnx-node": "^1.13.7", "zustand": "^5.0.15" }, "devDependencies": { "@types/node": "^24.0.0", "@types/react": "^19.2.18", "@types/react-dom": "^19.2.0", + "@types/unbzip2-stream": "^1.4.3", "@vitejs/plugin-react": "^6.1.1", "concurrently": "^10.0.5", "cross-env": "^10.0.0", @@ -27,7 +29,9 @@ "electron-builder": "^26.15.3", "esbuild": "^0.28.2", "jsdom": "^30.0.1", + "tar": "^7.5.22", "typescript": "^5.9.3", + "unbzip2-stream": "^1.4.3", "vite": "^8.2.2", "vitest": "^5.0.0", "wait-on": "^9.1.0" @@ -1724,6 +1728,26 @@ "@types/node": "*" } }, + "node_modules/@types/through": { + "version": "0.0.33", + "resolved": "https://registry.npmjs.org/@types/through/-/through-0.0.33.tgz", + "integrity": "sha512-HsJ+z3QuETzP3cswwtzt2vEIiHBk/dCcHGhbmG5X3ecnwFD/lPrMpliGXxSCg03L9AhrdwA4Oz/qfspkDW+xGQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "@types/node": "*" + } + }, + "node_modules/@types/unbzip2-stream": { + "version": "1.4.3", + "resolved": "https://registry.npmjs.org/@types/unbzip2-stream/-/unbzip2-stream-1.4.3.tgz", + "integrity": "sha512-D8X5uuJRISqc8YtwL8jNW2FpPdUOCYXbfD6zNROCTbVXK9nawucxh10tVXE3MPjnHdRA1LvB0zDxVya/lBsnYw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@types/through": "*" + } + }, "node_modules/@types/unist": { "version": "3.0.3", "resolved": "https://registry.npmjs.org/@types/unist/-/unist-3.0.3.tgz", @@ -2259,6 +2283,31 @@ "node": "20 || >=22" } }, + "node_modules/buffer": { + "version": "5.7.1", + "resolved": "https://registry.npmjs.org/buffer/-/buffer-5.7.1.tgz", + "integrity": "sha512-EHcyIPBQ4BSGlvjB16k5KgAJ27CIsHY/2JBmCRReo48y9rQ3MaUzWX3KVlBa4U7MyX02HdVj0K7C3WaB3ju7FQ==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT", + "dependencies": { + "base64-js": "^1.3.1", + "ieee754": "^1.1.13" + } + }, "node_modules/buffer-from": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/buffer-from/-/buffer-from-1.1.2.tgz", @@ -4149,6 +4198,27 @@ "node": ">= 14" } }, + "node_modules/ieee754": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/ieee754/-/ieee754-1.2.1.tgz", + "integrity": "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "BSD-3-Clause" + }, "node_modules/inflight": { "version": "1.0.6", "resolved": "https://registry.npmjs.org/inflight/-/inflight-1.0.6.tgz", @@ -6737,6 +6807,98 @@ "url": "https://github.com/sponsors/ljharb" } }, + "node_modules/sherpa-onnx-darwin-arm64": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-darwin-arm64/-/sherpa-onnx-darwin-arm64-1.13.7.tgz", + "integrity": "sha512-5NCE50hAvr3n2pdett0SgfPBJXaFZE0bqHwbHyiq+IKZ8Ids0l4M0VrG+ImGYIafCwie+oC3uAJ+pKj9xg/k+w==", + "cpu": [ + "arm64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/sherpa-onnx-darwin-x64": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-darwin-x64/-/sherpa-onnx-darwin-x64-1.13.7.tgz", + "integrity": "sha512-N3o+T+wn9WaQmsKV5DD8bTHdo+WN2+sXwmZcGJZiDjtOMR2zFz7uVCZnYCmEAMgvChC+oHcF5RvEEKcRCAu6Pw==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/sherpa-onnx-linux-arm64": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-linux-arm64/-/sherpa-onnx-linux-arm64-1.13.7.tgz", + "integrity": "sha512-TFCVpXyTh69buhOtTS8KIfkRXOVKY4Y1qjAktSItrKS4A0chnnrlXO5bKWoNAPeI6fMxTF/uvMYbYgcvjEMfNg==", + "cpu": [ + "arm64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/sherpa-onnx-linux-x64": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-linux-x64/-/sherpa-onnx-linux-x64-1.13.7.tgz", + "integrity": "sha512-npmxn5WwmAmlthgBhmbZ33t3i2j4mJwQt46dMEb3j7d41y1/uJrjrVAfa/DkvV+vn49ZWfcQ2UEWDipaZBVhuw==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/sherpa-onnx-node": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-node/-/sherpa-onnx-node-1.13.7.tgz", + "integrity": "sha512-0XGV7arGngBCnol0m8OLyqlnaUm19Q1KmetVj1DDBdymXa1upmAHZDwNdN47gjsEhqE5hXUEyc1vRQoXrNhNVg==", + "license": "Apache-2.0", + "optionalDependencies": { + "sherpa-onnx-darwin-arm64": "^1.13.7", + "sherpa-onnx-darwin-x64": "^1.13.7", + "sherpa-onnx-linux-arm64": "^1.13.7", + "sherpa-onnx-linux-x64": "^1.13.7", + "sherpa-onnx-win-ia32": "^1.13.7", + "sherpa-onnx-win-x64": "^1.13.7" + } + }, + "node_modules/sherpa-onnx-win-ia32": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-win-ia32/-/sherpa-onnx-win-ia32-1.13.7.tgz", + "integrity": "sha512-sTwtpxPQ76XLn0giAbvknIDEDKD3XXi2mo2AVROEucf1pIK1DjQl+LjLkalTeFoQqbC4J3xGx/g+xgcHQD1dsw==", + "cpu": [ + "ia32" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/sherpa-onnx-win-x64": { + "version": "1.13.7", + "resolved": "https://registry.npmjs.org/sherpa-onnx-win-x64/-/sherpa-onnx-win-x64-1.13.7.tgz", + "integrity": "sha512-wBV1o+/zgsMrOjfCFIgGrH6S28xq6CqRCLSavCOjTZ6cqr80yGc07DUHxqsHFPZvfoJU+2JF5L2l3gyWFWoWdQ==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ] + }, "node_modules/siginfo": { "version": "2.0.0", "resolved": "https://registry.npmjs.org/siginfo/-/siginfo-2.0.0.tgz", @@ -6999,6 +7161,13 @@ "fs-extra": "^10.0.0" } }, + "node_modules/through": { + "version": "2.3.8", + "resolved": "https://registry.npmjs.org/through/-/through-2.3.8.tgz", + "integrity": "sha512-w89qg7PI8wAdvX60bMDP+bFoD5Dvhm9oLheFp5O4a2QF0cSBGsBX4qZmadPMvVqlLJBBci+WqGGOAPvcDeNSVg==", + "dev": true, + "license": "MIT" + }, "node_modules/tiny-async-pool": { "version": "1.3.0", "resolved": "https://registry.npmjs.org/tiny-async-pool/-/tiny-async-pool-1.3.0.tgz", @@ -7197,6 +7366,17 @@ "node": ">=14.17" } }, + "node_modules/unbzip2-stream": { + "version": "1.4.3", + "resolved": "https://registry.npmjs.org/unbzip2-stream/-/unbzip2-stream-1.4.3.tgz", + "integrity": "sha512-mlExGW4w71ebDJviH16lQLtZS32VKqsSfk80GCfUlwT/4/hNRFsoscrF/c++9xinkMzECL1uL9DDwXqFWkruPg==", + "dev": true, + "license": "MIT", + "dependencies": { + "buffer": "^5.2.1", + "through": "^2.3.8" + } + }, "node_modules/undici": { "version": "7.29.0", "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", diff --git a/package.json b/package.json index 4b4a112..493d7b9 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "eveflow", - "version": "2.0.0", - "releaseVersion": "2.0.0.1", + "version": "2.1.0", + "releaseVersion": "2.1.0", "description": "JARVIS-style desktop HUD for Hermes Agent: voice, streaming runs, scheduled jobs, skills and telemetry", "main": "dist-electron/main.js", "private": true, @@ -53,6 +53,14 @@ "from": "build/icon.png", "to": "icon.png" } + ], + "asarUnpack": [ + "dist-electron/voice-worker.js", + "node_modules/sherpa-onnx-node/**", + "node_modules/sherpa-onnx-win-x64/**", + "node_modules/sherpa-onnx-linux-x64/**", + "node_modules/sherpa-onnx-darwin-arm64/**", + "node_modules/sherpa-onnx-darwin-x64/**" ] }, "dependencies": { @@ -61,12 +69,14 @@ "react-dom": "^19.2.8", "react-markdown": "^10.1.0", "remark-gfm": "^4.0.1", + "sherpa-onnx-node": "^1.13.7", "zustand": "^5.0.15" }, "devDependencies": { "@types/node": "^24.0.0", "@types/react": "^19.2.18", "@types/react-dom": "^19.2.0", + "@types/unbzip2-stream": "^1.4.3", "@vitejs/plugin-react": "^6.1.1", "concurrently": "^10.0.5", "cross-env": "^10.0.0", @@ -74,7 +84,9 @@ "electron-builder": "^26.15.3", "esbuild": "^0.28.2", "jsdom": "^30.0.1", + "tar": "^7.5.22", "typescript": "^5.9.3", + "unbzip2-stream": "^1.4.3", "vite": "^8.2.2", "vitest": "^5.0.0", "wait-on": "^9.1.0" diff --git a/scripts/build-electron.mjs b/scripts/build-electron.mjs index c27863f..2689dc6 100644 --- a/scripts/build-electron.mjs +++ b/scripts/build-electron.mjs @@ -7,13 +7,18 @@ import { rmSync } from 'node:fs'; rmSync('dist-electron', { recursive: true, force: true }); await build({ - entryPoints: ['electron/main.ts', 'electron/preload.ts'], + entryPoints: { + main: 'electron/main.ts', + preload: 'electron/preload.ts', + 'voice-worker': 'electron/voice/worker.ts' + }, outdir: 'dist-electron', bundle: true, platform: 'node', format: 'cjs', target: 'node22', - external: ['electron'], + // sherpa-onnx-node stays external: it is a native addon shipped unpacked from the asar. + external: ['electron', 'sherpa-onnx-node'], sourcemap: false, logLevel: 'info' }); diff --git a/shared/bridge.ts b/shared/bridge.ts index b27186d..c06551c 100644 --- a/shared/bridge.ts +++ b/shared/bridge.ts @@ -11,6 +11,15 @@ import type { WebhookStatus, WindowMode } from './ipc'; +import type { + SynthesizeRequest, + SynthesizeResult, + TranscribeRequest, + TranscribeResult, + VoiceDownloadProgress, + VoiceEngineStatus, + VoiceModelStatus +} from './voice'; export type Unsubscribe = () => void; @@ -51,4 +60,15 @@ export interface EveFlowBridge { hotkeys: { on: (cb: (event: HotkeyEvent) => void) => Unsubscribe; }; + voice: { + status: () => Promise; + listModels: () => Promise; + downloadModel: (id: string) => Promise; + cancelDownload: (id: string) => Promise; + removeModel: (id: string) => Promise; + onProgress: (cb: (progress: VoiceDownloadProgress) => void) => Unsubscribe; + transcribe: (req: TranscribeRequest) => Promise; + synthesize: (req: SynthesizeRequest) => Promise; + unload: (id?: string) => Promise; + }; } diff --git a/shared/voice.ts b/shared/voice.ts new file mode 100644 index 0000000..3f0c6a8 --- /dev/null +++ b/shared/voice.ts @@ -0,0 +1,90 @@ +/** Local voice engine contract (sherpa-onnx in a utility process). Shared by main and renderer. */ + +export type VoiceModelKind = 'stt' | 'tts'; +export type VoiceEngineKind = 'whisper' | 'sense-voice' | 'nemo-transducer' | 'kokoro' | 'piper'; + +export interface VoiceSpeaker { + id: number; + name: string; + lang: string; +} + +export interface VoiceModelSpec { + id: string; + kind: VoiceModelKind; + engine: VoiceEngineKind; + name: string; + description: string; + languages: string[]; + sizeMb: number; + url: string; + /** Folder created by the archive (files are referenced relative to it). */ + dir: string; + /** Files that must exist once installed. */ + files: string[]; + speakers?: VoiceSpeaker[]; + sampleRate?: number; + recommended?: boolean; +} + +export interface VoiceModelStatus extends VoiceModelSpec { + installed: boolean; + downloading: boolean; + installedBytes: number; +} + +export interface VoiceDownloadProgress { + id: string; + phase: 'download' | 'extract' | 'done' | 'error' | 'cancelled'; + received: number; + total: number; + percent: number; + message?: string; +} + +export interface TranscribeRequest { + modelId: string; + wav: Uint8Array; + language: string; // 'fr', 'en', 'auto' +} + +export interface TranscribeResult { + text: string; + language?: string; + durationMs: number; + audioSec: number; +} + +export interface SynthesizeRequest { + modelId: string; + text: string; + speaker: number; + speed: number; +} + +export interface SynthesizeResult { + wav: Uint8Array; + sampleRate: number; + durationMs: number; + audioSec: number; +} + +export interface VoiceEngineStatus { + available: boolean; + error?: string; + version?: string; + loaded: string[]; + modelsDir: string; +} + +export const VOICE_IPC = { + status: 'voice:status', + modelsList: 'voice:models:list', + modelsDownload: 'voice:models:download', + modelsCancel: 'voice:models:cancel', + modelsRemove: 'voice:models:remove', + modelsProgress: 'voice:models:progress', + transcribe: 'voice:transcribe', + synthesize: 'voice:synthesize', + unload: 'voice:unload' +} as const; diff --git a/src/App.tsx b/src/App.tsx index a5d7d6b..92b3b02 100644 --- a/src/App.tsx +++ b/src/App.tsx @@ -10,6 +10,7 @@ import { useChat } from './state/chat'; import { useHermes } from './state/hermes'; import { useSettings } from './state/settings'; import { useVoice } from './state/voice'; +import { useVoiceModels } from './state/voiceModels'; import { ChatPanel } from './components/chat/ChatPanel'; import { PendingRequests } from './components/chat/PendingRequests'; import { CompactWidget } from './components/compact/CompactWidget'; @@ -41,6 +42,8 @@ function useBoot(): boolean { Log.error('app', 'window.eveflow missing inside Electron'); } if (api) { + disposers.push(useVoiceModels.getState().subscribe()); + void useVoiceModels.getState().refresh(); disposers.push(api.hermes.onPush(handlePush)); disposers.push( api.hotkeys.on((event) => { diff --git a/src/components/settings/ModelsSection.tsx b/src/components/settings/ModelsSection.tsx new file mode 100644 index 0000000..3212dd2 --- /dev/null +++ b/src/components/settings/ModelsSection.tsx @@ -0,0 +1,101 @@ +import { useEffect } from 'react'; +import { Download, Trash2, X, CheckCircle2, Cpu, Mic, Volume2, AlertTriangle, RefreshCw } from 'lucide-react'; +import type { VoiceModelStatus } from '../../../shared/voice'; +import { bridge } from '../../lib/bridge'; +import { useVoiceModels } from '../../state/voiceModels'; +import { useSettings } from '../../state/settings'; + +function formatMb(bytes: number): string { + return bytes >= 1024 * 1024 * 1024 ? `${(bytes / 1024 ** 3).toFixed(2)} Go` : `${Math.round(bytes / 1024 / 1024)} Mo`; +} + +function ModelRow({ model }: { model: VoiceModelStatus }) { + const progress = useVoiceModels((s) => s.progress[model.id]); + const { download, cancel, remove } = useVoiceModels(); + const settings = useSettings((s) => s.settings); + const update = useSettings((s) => s.update); + const isActive = model.kind === 'stt' ? settings.voice.localModel === model.id : settings.speech.localModel === model.id; + const busy = !!progress && (progress.phase === 'download' || progress.phase === 'extract'); + + const activate = () => { + if (model.kind === 'stt') update({ voice: { localModel: model.id, provider: 'local' } }); + else update({ speech: { localModel: model.id, provider: 'local', localSpeaker: model.speakers?.[0]?.id ?? 0 } }); + }; + + return ( +
+
+
+ {model.name} + {model.recommended && recommandé} + {model.installed && installé · {formatMb(model.installedBytes)}} + {!model.installed && !busy && {model.sizeMb} Mo} + {isActive && model.installed && actif} +
+ {model.description} + langues : {model.languages.join(', ')}{model.speakers ? ` · ${model.speakers.length} voix` : ''} + {busy && ( +
+
+ {progress.phase === 'extract' ? 'extraction…' : `${progress.percent}% · ${formatMb(progress.received)} / ${formatMb(progress.total)}`} +
+ )} + {progress?.phase === 'error' && échec : {progress.message}} +
+
+ {busy ? ( + + ) : model.installed ? ( + <> + {!isActive && } + + + ) : ( + + )} +
+
+ ); +} + +export function ModelsSection() { + const { models, engine, error, refresh, checkEngine } = useVoiceModels(); + useEffect(() => { + void refresh(); + void checkEngine(); + }, [refresh, checkEngine]); + + if (!bridge()) return
Les modèles locaux ne sont disponibles que dans l’application Electron.
; + + const stt = models.filter((m) => m.kind === 'stt'); + const tts = models.filter((m) => m.kind === 'tts'); + return ( + <> +
+
Moteur local (sherpa-onnx)
+ {engine === null ? ( + vérification… + ) : engine.available ? ( +
moteur prêt · version {engine.version}{engine.loaded.length ? ` · chargé : ${engine.loaded.join(', ')}` : ''}
+ ) : ( +
{engine.error}
+ )} + + Les modèles tournent sur le processeur, sans connexion, dans un processus séparé. Dossier : {engine?.modelsDir} + + {error &&
{error}
} +
+ +
+
+
+
Reconnaissance vocale
+
{stt.map((m) => )}
+
+
+
Synthèse vocale
+
{tts.map((m) => )}
+
+ + ); +} diff --git a/src/components/settings/SettingsDrawer.tsx b/src/components/settings/SettingsDrawer.tsx index dcc99b9..2933ebf 100644 --- a/src/components/settings/SettingsDrawer.tsx +++ b/src/components/settings/SettingsDrawer.tsx @@ -10,8 +10,10 @@ import { encodeWav } from '../../services/voice/wav'; import { useHermes } from '../../state/hermes'; import { DEFAULT_SETTINGS, useSettings, type HudTheme } from '../../state/settings'; import { useVoice } from '../../state/voice'; +import { installedModels, useVoiceModels } from '../../state/voiceModels'; +import { ModelsSection } from './ModelsSection'; -type Section = 'general' | 'hermes' | 'voice' | 'speech' | 'webhook' | 'ui'; +type Section = 'general' | 'hermes' | 'voice' | 'speech' | 'models' | 'webhook' | 'ui'; interface Props { onClose: () => void; @@ -37,6 +39,8 @@ export function SettingsDrawer({ onClose }: Props) { const reset = useSettings((s) => s.reset); const hermesStore = useHermes(); const micDevices = useVoice((s) => s.micDevices); + const voiceModels = useVoiceModels((s) => s.models); + const refreshModels = useVoiceModels((s) => s.refresh); const [section, setSection] = useState
('hermes'); const [hermesTest, setHermesTest] = useState({ status: 'idle', message: '' }); const [sttTest, setSttTest] = useState({ status: 'idle', message: '' }); @@ -49,8 +53,9 @@ export function SettingsDrawer({ onClose }: Props) { refresh(); speechSynthesis.addEventListener?.('voiceschanged', refresh); void listMicrophones().then((d) => useVoice.getState().setMicDevices(d)); + void refreshModels(); return () => speechSynthesis.removeEventListener?.('voiceschanged', refresh); - }, []); + }, [refreshModels]); const testHermes = async () => { setHermesTest({ status: 'running', message: 'connexion…' }); @@ -113,6 +118,7 @@ export function SettingsDrawer({ onClose }: Props) { ['hermes', 'Hermes'], ['voice', 'Micro / STT'], ['speech', 'Voix / TTS'], + ['models', 'Modèles locaux'], ['webhook', 'Webhook'], ['general', 'Général'], ['ui', 'Interface'] @@ -195,10 +201,23 @@ export function SettingsDrawer({ onClose }: Props) {
+ {settings.voice.provider === 'local' && ( +
+ + {installedModels(voiceModels, 'stt').length === 0 ? ( + Aucun modèle installé. { e.preventDefault(); setSection('models'); }}>Téléchargez Whisper base dans « Modèles locaux ». + ) : ( + + )} +
+ )} {settings.voice.provider === 'openai-compatible' && ( <>
@@ -255,8 +274,15 @@ export function SettingsDrawer({ onClose }: Props) {
{ update({ voice: { handsFree: v } }); useVoice.getState().setHandsFree(v); }} label="Mains libres au démarrage" hint="Le micro se réactive automatiquement après chaque réponse." /> update({ voice: { wakeChime: v } })} label="Signal sonore d’écoute" /> + update({ voice: { wakeWordEnabled: v } })} label="Mot d’activation en mains libres" hint="Seules les phrases commençant par ce mot sont envoyées à Hermes ; le reste est ignoré. Recommandé avec la reconnaissance locale." /> + {settings.voice.wakeWordEnabled && ( +
+ + update({ voice: { wakeWord: e.target.value.toLowerCase() } })} /> +
+ )}
- +
@@ -267,6 +293,7 @@ export function SettingsDrawer({ onClose }: Props) {
{ + const m = voiceModels.find((x) => x.id === e.target.value); + update({ speech: { localModel: e.target.value, localSpeaker: m?.speakers?.[0]?.id ?? 0 } }); + }}> + {installedModels(voiceModels, 'tts').map((m) => )} + +
+
+ + +
+ + ) + )} {settings.speech.provider === 'system' && (
@@ -331,6 +385,8 @@ export function SettingsDrawer({ onClose }: Props) {
)} + {section === 'models' && } + {section === 'webhook' && (
update({ webhook: { enabled: v } })} label="Serveur webhook local" hint="Hermes (crons, gateway, scripts) peut pousser des messages vers EveFlow." /> diff --git a/src/services/voice/stt.ts b/src/services/voice/stt.ts index d4181de..e11c73b 100644 --- a/src/services/voice/stt.ts +++ b/src/services/voice/stt.ts @@ -1,7 +1,8 @@ import { Log } from '../../lib/log'; +import { bridge } from '../../lib/bridge'; import { httpFetch } from '../../lib/transport'; -export type SttProvider = 'openai-compatible' | 'browser'; +export type SttProvider = 'openai-compatible' | 'browser' | 'local'; export interface SttConfig { provider: SttProvider; @@ -10,6 +11,8 @@ export interface SttConfig { model: string; language: string; // BCP-47, e.g. fr-FR prompt?: string; + /** Catalog id of the local sherpa-onnx model (provider 'local'). */ + localModel: string; } export function sttEndpoint(apiUrl: string): string { @@ -20,8 +23,20 @@ export function sttEndpoint(apiUrl: string): string { return `${base}/v1/audio/transcriptions`; } -/** Transcribe a WAV buffer through any OpenAI-compatible `/v1/audio/transcriptions` endpoint. */ +/** Transcribe with the in-app sherpa-onnx engine (runs in the Electron main process). */ +export async function transcribeLocal(wav: Uint8Array, config: SttConfig): Promise { + const api = bridge(); + if (!api) throw new Error('La reconnaissance locale nécessite l’application Electron.'); + if (!config.localModel) throw new Error('Aucun modèle de reconnaissance local sélectionné.'); + const started = Date.now(); + const result = await api.voice.transcribe({ modelId: config.localModel, wav, language: config.language.split('-')[0] || 'auto' }); + Log.info('stt', `local transcription in ${Date.now() - started} ms (engine ${result.durationMs} ms, audio ${result.audioSec.toFixed(1)}s)`); + return result.text.trim(); +} + +/** Transcribe a WAV buffer with the configured provider. */ export async function transcribeWav(wav: Uint8Array, config: SttConfig): Promise { + if (config.provider === 'local') return transcribeLocal(wav, config); if (!config.apiUrl.trim()) throw new Error("URL de l'API de transcription non configurée."); const endpoint = sttEndpoint(config.apiUrl); const fields: Record = { diff --git a/src/services/voice/tts.ts b/src/services/voice/tts.ts index cd301e3..57e6f90 100644 --- a/src/services/voice/tts.ts +++ b/src/services/voice/tts.ts @@ -4,11 +4,12 @@ * and the legacy Google Translate endpoint (no key, online only). */ import { Log } from '../../lib/log'; +import { bridge } from '../../lib/bridge'; import { chunkForSpeech, cleanForSpeech, extractSentences } from '../../lib/text'; import { httpFetch } from '../../lib/transport'; import { audioBus } from './audioBus'; -export type TtsProvider = 'openai-compatible' | 'system' | 'google-free' | 'off'; +export type TtsProvider = 'openai-compatible' | 'system' | 'google-free' | 'local' | 'off'; export interface TtsConfig { provider: TtsProvider; @@ -21,6 +22,10 @@ export interface TtsConfig { systemVoice: string; language: string; volume: number; + /** Catalog id of the local sherpa-onnx voice model (provider 'local'). */ + localModel: string; + /** Speaker id inside the local model. */ + localSpeaker: number; } export type TtsState = 'idle' | 'loading' | 'speaking'; @@ -159,7 +164,8 @@ export class TtsEngine { } private prefetch(text: string, gen: number): Promise { - const task = this.config.provider === 'google-free' ? this.fetchGoogle(text) : this.fetchOpenAi(text); + const task = + this.config.provider === 'google-free' ? this.fetchGoogle(text) : this.config.provider === 'local' ? this.fetchLocal(text) : this.fetchOpenAi(text); return task .then(async (bytes) => { if (gen !== this.generation || !bytes) return null; @@ -199,6 +205,20 @@ export class TtsEngine { return res.binary; } + private async fetchLocal(text: string): Promise { + const api = bridge(); + if (!api) throw new Error('La synthèse locale nécessite l’application Electron.'); + if (!this.config.localModel) throw new Error('Aucun modèle de voix local sélectionné.'); + const result = await api.voice.synthesize({ + modelId: this.config.localModel, + text, + speaker: this.config.localSpeaker, + speed: Math.max(0.5, Math.min(2, this.config.speed || 1)) + }); + Log.debug('tts', `local synthesis ${result.durationMs} ms for ${result.audioSec.toFixed(1)}s of audio`); + return result.wav; + } + private async fetchGoogle(text: string): Promise { const url = `https://translate.google.com/translate_tts?ie=UTF-8&tl=${encodeURIComponent(this.config.language.split('-')[0] || 'fr')}&client=tw-ob&q=${encodeURIComponent(text.slice(0, 200))}`; const res = await httpFetch({ diff --git a/src/services/voice/voiceController.ts b/src/services/voice/voiceController.ts index ac04ce0..287cb02 100644 --- a/src/services/voice/voiceController.ts +++ b/src/services/voice/voiceController.ts @@ -15,8 +15,39 @@ import type { WavResult } from './wav'; const SENSITIVITY_RATIO: Record = { 1: 4.5, 2: 3.4, 3: 2.6, 4: 2.0, 5: 1.6 }; const SENSITIVITY_MIN_RMS: Record = { 1: 0.03, 2: 0.02, 3: 0.012, 4: 0.008, 5: 0.005 }; +/** Levenshtein distance, used to tolerate transcription slips in the wake word ("javis" for "jarvis"). */ +function editDistance(a: string, b: string): number { + const dp = Array.from({ length: a.length + 1 }, (_, i) => [i, ...new Array(b.length).fill(0)]); + for (let j = 1; j <= b.length; j++) dp[0][j] = j; + for (let i = 1; i <= a.length; i++) { + for (let j = 1; j <= b.length; j++) { + dp[i][j] = Math.min(dp[i - 1][j] + 1, dp[i][j - 1] + 1, dp[i - 1][j - 1] + (a[i - 1] === b[j - 1] ? 0 : 1)); + } + } + return dp[a.length][b.length]; +} + +/** Compare the start of an utterance with the wake word, tolerant to accents, case, punctuation and small slips. */ +export function matchWakeWord(text: string, wakeWord: string): { matched: boolean; rest: string } { + const norm = (v: string) => v.normalize('NFD').replace(/[\u0300-\u036f]/g, '').toLowerCase(); + const target = norm(wakeWord).split(/[\s\p{P}]+/u).filter(Boolean); + if (target.length === 0) return { matched: true, rest: text.trim() }; + const original = text.trim().replace(/^[\s\p{P}]+/u, ''); + const words = original.split(/\s+/); + const head = words.slice(0, target.length).map((w) => norm(w).replace(/[\p{P}]+/gu, '')); + if (head.length < target.length) return { matched: false, rest: '' }; + const joinedHead = head.join(' '); + const joinedTarget = target.join(' '); + const tolerance = joinedTarget.length >= 6 ? 2 : joinedTarget.length >= 4 ? 1 : 0; + if (editDistance(joinedHead, joinedTarget) > tolerance) return { matched: false, rest: '' }; + const rest = words.slice(target.length).join(' ').replace(/^[\s\p{P}]+/u, ''); + return { matched: true, rest }; +} + class VoiceController { private capture = new MicCapture(); + /** After a bare wake word, the next utterance is accepted without the wake word. */ + private attentionUntil = 0; private browser = new BrowserRecognizer(); private handsFreeTimer: ReturnType | null = null; private stopping = false; @@ -169,10 +200,28 @@ class VoiceController { this.playChime(false); const settings = useSettings.getState().settings.voice; try { - const text = await transcribeWav(wav.bytes, settings); + let text = await transcribeWav(wav.bytes, settings); voice.setTranscript(text); voice.setPhase('off'); Log.info('voice', `transcript (${wav.durationSec.toFixed(1)}s): ${text}`); + if (voice.handsFree && settings.wakeWordEnabled && Date.now() > this.attentionUntil) { + const { matched, rest } = matchWakeWord(text, settings.wakeWord || 'jarvis'); + if (!matched) { + Log.debug('voice', 'utterance ignored: no wake word'); + voice.setTranscript(''); + useChat.getState().setHud('idle'); + this.scheduleHandsFree(200); + return; + } + if (!rest.trim()) { + this.attentionUntil = Date.now() + 10_000; + useChat.getState().setHud('listening'); + speech.say('Oui ?'); + return; + } + text = rest; + } + this.attentionUntil = 0; if (text.trim()) { await sendMessage(text, [], 'voice'); } else { diff --git a/src/state/settings.ts b/src/state/settings.ts index de82b7b..8255077 100644 --- a/src/state/settings.ts +++ b/src/state/settings.ts @@ -16,6 +16,9 @@ export interface VoiceSettings extends SttConfig { silenceMs: number; sensitivity: number; // 1 (low) .. 5 (high) wakeChime: boolean; + /** Hands-free: only react to utterances starting with this word (local STT recommended). */ + wakeWordEnabled: boolean; + wakeWord: string; } export interface SpeechSettings extends TtsConfig { @@ -77,7 +80,10 @@ export const DEFAULT_SETTINGS: Settings = { micDeviceId: '', silenceMs: 900, sensitivity: 3, - wakeChime: true + wakeChime: true, + wakeWordEnabled: false, + wakeWord: 'jarvis', + localModel: 'whisper-base' }, speech: { provider: 'openai-compatible', @@ -90,6 +96,8 @@ export const DEFAULT_SETTINGS: Settings = { systemVoice: '', language: 'fr-FR', volume: 1, + localModel: 'kokoro-v1', + localSpeaker: 30, autoSpeak: true, speakIncoming: true }, diff --git a/src/state/voiceModels.ts b/src/state/voiceModels.ts new file mode 100644 index 0000000..b847f2c --- /dev/null +++ b/src/state/voiceModels.ts @@ -0,0 +1,91 @@ +import { create } from 'zustand'; +import type { VoiceDownloadProgress, VoiceEngineStatus, VoiceModelStatus } from '../../shared/voice'; +import { bridge } from '../lib/bridge'; +import { Log } from '../lib/log'; + +interface VoiceModelsStore { + models: VoiceModelStatus[]; + progress: Record; + engine: VoiceEngineStatus | null; + error: string | null; + refresh: () => Promise; + checkEngine: () => Promise; + download: (id: string) => Promise; + cancel: (id: string) => Promise; + remove: (id: string) => Promise; + subscribe: () => () => void; +} + +export const useVoiceModels = create((set, get) => ({ + models: [], + progress: {}, + engine: null, + error: null, + + refresh: async () => { + const api = bridge(); + if (!api) return; + try { + set({ models: await api.voice.listModels(), error: null }); + } catch (err) { + set({ error: (err as Error).message }); + } + }, + checkEngine: async () => { + const api = bridge(); + if (!api) return; + try { + set({ engine: await api.voice.status() }); + } catch (err) { + set({ engine: { available: false, error: (err as Error).message, loaded: [], modelsDir: '' } }); + } + }, + download: async (id) => { + const api = bridge(); + if (!api) return; + set((s) => ({ progress: { ...s.progress, [id]: { id, phase: 'download', received: 0, total: 0, percent: 0 } }, error: null })); + try { + await api.voice.downloadModel(id); + } catch (err) { + const message = (err as Error).message; + Log.error('voice', `download ${id} failed: ${message}`); + set({ error: message }); + } finally { + await get().refresh(); + } + }, + cancel: async (id) => { + await bridge()?.voice.cancelDownload(id); + }, + remove: async (id) => { + const api = bridge(); + if (!api) return; + try { + set({ models: await api.voice.removeModel(id) }); + } catch (err) { + set({ error: (err as Error).message }); + } + }, + subscribe: () => { + const api = bridge(); + if (!api) return () => undefined; + return api.voice.onProgress((p) => { + set((s) => { + const progress = { ...s.progress, [p.id]: p }; + if (p.phase === 'done' || p.phase === 'error' || p.phase === 'cancelled') { + setTimeout(() => { + set((s2) => { + const next = { ...s2.progress }; + delete next[p.id]; + return { progress: next }; + }); + void get().refresh(); + }, 1500); + } + return { progress }; + }); + }); + } +})); + +export const installedModels = (models: VoiceModelStatus[], kind: 'stt' | 'tts') => models.filter((m) => m.kind === kind && m.installed); diff --git a/tests/wakeWord.test.ts b/tests/wakeWord.test.ts new file mode 100644 index 0000000..4cc99c0 --- /dev/null +++ b/tests/wakeWord.test.ts @@ -0,0 +1,16 @@ +import { describe, expect, it } from 'vitest'; +import { matchWakeWord } from '../src/services/voice/voiceController'; + +describe('matchWakeWord', () => { + it('accepts the wake word with punctuation, accents and small slips', () => { + expect(matchWakeWord('Jarvis, quelle heure est-il ?', 'jarvis')).toEqual({ matched: true, rest: 'quelle heure est-il ?' }); + expect(matchWakeWord('Javis allume la lumière', 'jarvis')).toEqual({ matched: true, rest: 'allume la lumière' }); + expect(matchWakeWord('JARVIS.', 'jarvis')).toEqual({ matched: true, rest: '' }); + expect(matchWakeWord('Hé Jarvis, bonjour', 'hé jarvis')).toEqual({ matched: true, rest: 'bonjour' }); + }); + it('rejects unrelated speech', () => { + expect(matchWakeWord('Il fait beau aujourd’hui', 'jarvis').matched).toBe(false); + expect(matchWakeWord('Service client', 'jarvis').matched).toBe(false); + expect(matchWakeWord('', 'jarvis').matched).toBe(false); + }); +});