Files
EveFlow/shared/voice.ts
T
Claude 0d284fea9e feat: always-on wake word with sherpa-onnx keyword spotting (v2.2.0)
- Catalog: 3.3 MB zipformer keyword-spotting model (kws-en).
- shared/keywords: BPE encoding of wake phrases (SentencePiece table for common words,
  greedy longest-match fallback over the model vocabulary) and keywords file builder.
- Worker: KeywordSpotter stream fed with 16-bit PCM, detections pushed as unsolicited
  messages; engine derives the keywords file, maps sensitivity to threshold/score,
  forwards detections to the renderer and re-arms after a worker restart.
- Renderer: WakeListener keeps one microphone stream, batches 256 ms frames to the
  spotter and captures the command on the same stream after detection (pre-roll, VAD),
  then resumes spotting; the wake word also interrupts speech. Settings: wake mode
  (off / always-on / transcript filter), keyword, sensitivity, status and one-click
  model download; HUD caption shows the active keyword.
- Validated: detection in the worker (fork) and through the real Electron IPC path on
  Kokoro audio, no false positive on an English recording; 23 unit tests.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017Wn5VX9HNbJ7N54hR24u9Y
2026-09-03 18:21:35 +00:00

108 lines
2.4 KiB
TypeScript

/** Local voice engine contract (sherpa-onnx in a utility process). Shared by main and renderer. */
export type VoiceModelKind = 'stt' | 'tts' | 'kws';
export type VoiceEngineKind = 'whisper' | 'sense-voice' | 'nemo-transducer' | 'kokoro' | 'piper' | 'kws-transducer';
export interface VoiceSpeaker {
id: number;
name: string;
lang: string;
}
export interface VoiceModelSpec {
id: string;
kind: VoiceModelKind;
engine: VoiceEngineKind;
name: string;
description: string;
languages: string[];
sizeMb: number;
url: string;
/** Folder created by the archive (files are referenced relative to it). */
dir: string;
/** Files that must exist once installed. */
files: string[];
speakers?: VoiceSpeaker[];
sampleRate?: number;
recommended?: boolean;
}
export interface VoiceModelStatus extends VoiceModelSpec {
installed: boolean;
downloading: boolean;
installedBytes: number;
}
export interface VoiceDownloadProgress {
id: string;
phase: 'download' | 'extract' | 'done' | 'error' | 'cancelled';
received: number;
total: number;
percent: number;
message?: string;
}
export interface TranscribeRequest {
modelId: string;
wav: Uint8Array;
language: string; // 'fr', 'en', 'auto'
}
export interface TranscribeResult {
text: string;
language?: string;
durationMs: number;
audioSec: number;
}
export interface SynthesizeRequest {
modelId: string;
text: string;
speaker: number;
speed: number;
}
export interface SynthesizeResult {
wav: Uint8Array;
sampleRate: number;
durationMs: number;
audioSec: number;
}
export interface KwsStartRequest {
modelId: string;
/** Wake phrases in plain text (e.g. "jarvis", "hey jarvis"). */
keywords: string[];
/** 1 (strict) .. 5 (eager) */
sensitivity: number;
}
export interface KwsDetection {
keyword: string;
at: number;
}
export interface VoiceEngineStatus {
available: boolean;
error?: string;
version?: string;
loaded: string[];
modelsDir: string;
}
export const VOICE_IPC = {
status: 'voice:status',
modelsList: 'voice:models:list',
modelsDownload: 'voice:models:download',
modelsCancel: 'voice:models:cancel',
modelsRemove: 'voice:models:remove',
modelsProgress: 'voice:models:progress',
transcribe: 'voice:transcribe',
synthesize: 'voice:synthesize',
unload: 'voice:unload',
kwsStart: 'voice:kws:start',
kwsStop: 'voice:kws:stop',
kwsAudio: 'voice:kws:audio',
kwsDetected: 'voice:kws:detected'
} as const;