Files
EveFlow/src/services/voice/stt.ts
T
Claude eefc36aca8 feat: local offline voice models (sherpa-onnx) with model manager and wake word (v2.1.0)
- electron/voice: model catalog (Whisper base/small/turbo, SenseVoice, Kokoro v1.0,
  Piper fr), streaming tar.bz2 downloader with progress, sherpa-onnx worker running in
  an Electron utilityProcess (transcribe / synthesize / status / unload), IPC + bridge.
- Renderer: 'local' providers for STT and TTS, Settings → Modèles locaux (download,
  progress, delete, activate), speaker selection, hands-free wake word with a tolerant
  matcher (accents, punctuation, edit distance) and attention window after a bare wake word.
- Packaging: native addon and worker unpacked from the asar; release 2.1.0.

Validated on Linux: Kokoro (fr) → Whisper base round trip, Piper (fr) → Whisper round trip.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017Wn5VX9HNbJ7N54hR24u9Y
2026-09-03 17:43:52 +00:00

126 lines
4.7 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { Log } from '../../lib/log';
import { bridge } from '../../lib/bridge';
import { httpFetch } from '../../lib/transport';
export type SttProvider = 'openai-compatible' | 'browser' | 'local';
export interface SttConfig {
provider: SttProvider;
apiUrl: string;
apiKey: string;
model: string;
language: string; // BCP-47, e.g. fr-FR
prompt?: string;
/** Catalog id of the local sherpa-onnx model (provider 'local'). */
localModel: string;
}
export function sttEndpoint(apiUrl: string): string {
const base = apiUrl.trim().replace(/\/+$/, '');
if (base.endsWith('/transcriptions')) return base;
if (base.endsWith('/audio')) return `${base}/transcriptions`;
if (/\/v\d+$/.test(base)) return `${base}/audio/transcriptions`;
return `${base}/v1/audio/transcriptions`;
}
/** Transcribe with the in-app sherpa-onnx engine (runs in the Electron main process). */
export async function transcribeLocal(wav: Uint8Array, config: SttConfig): Promise<string> {
const api = bridge();
if (!api) throw new Error('La reconnaissance locale nécessite l’application Electron.');
if (!config.localModel) throw new Error('Aucun modèle de reconnaissance local sélectionné.');
const started = Date.now();
const result = await api.voice.transcribe({ modelId: config.localModel, wav, language: config.language.split('-')[0] || 'auto' });
Log.info('stt', `local transcription in ${Date.now() - started} ms (engine ${result.durationMs} ms, audio ${result.audioSec.toFixed(1)}s)`);
return result.text.trim();
}
/** Transcribe a WAV buffer with the configured provider. */
export async function transcribeWav(wav: Uint8Array, config: SttConfig): Promise<string> {
if (config.provider === 'local') return transcribeLocal(wav, config);
if (!config.apiUrl.trim()) throw new Error("URL de l'API de transcription non configurée.");
const endpoint = sttEndpoint(config.apiUrl);
const fields: Record<string, string> = {
model: config.model || 'whisper-1',
response_format: 'json'
};
if (config.language) fields.language = config.language.split('-')[0];
if (config.prompt) fields.prompt = config.prompt;
const headers: Record<string, string> = {};
if (config.apiKey.trim()) headers.Authorization = `Bearer ${config.apiKey.trim()}`;
Log.info('stt', `POST ${endpoint}`, { model: fields.model, bytes: wav.length });
const res = await httpFetch({
url: endpoint,
method: 'POST',
headers,
multipart: { fields, file: { name: 'audio.wav', type: 'audio/wav', data: wav } },
timeoutMs: 90_000
});
if (!res.ok) {
let detail = res.text ?? '';
try {
const parsed = JSON.parse(detail) as { error?: { message?: string }; message?: string; detail?: string };
detail = parsed.error?.message ?? parsed.message ?? parsed.detail ?? detail;
} catch {
/* raw text */
}
throw new Error(`Transcription refusée (HTTP ${res.status}) : ${detail.slice(0, 240)}`);
}
const text = res.text ?? '';
try {
const parsed = JSON.parse(text) as { text?: string; transcript?: string; segments?: Array<{ text: string }> };
const value = parsed.text ?? parsed.transcript ?? parsed.segments?.map((s) => s.text).join(' ');
if (typeof value === 'string') return value.trim();
} catch {
if (text.trim()) return text.trim();
}
throw new Error('Réponse de transcription invalide.');
}
/** Web Speech API fallback (Chromium routes it to Google; works only online, quality varies). */
export class BrowserRecognizer {
private recognition: SpeechRecognition | null = null;
static isSupported(): boolean {
return typeof window !== 'undefined' && !!(window.SpeechRecognition || window.webkitSpeechRecognition);
}
start(lang: string, onResult: (text: string) => void, onError: (message: string) => void, onEnd: () => void): void {
const Ctor = window.SpeechRecognition || window.webkitSpeechRecognition;
if (!Ctor) {
onError("La reconnaissance vocale du navigateur n'est pas disponible.");
return;
}
this.stop();
const rec = new Ctor();
rec.lang = lang;
rec.continuous = false;
rec.interimResults = false;
rec.maxAlternatives = 1;
rec.onresult = (event) => {
const text = event.results[0]?.[0]?.transcript ?? '';
onResult(text);
};
rec.onerror = (event) => onError(event.error === 'not-allowed' ? 'Micro refusé par le navigateur.' : `Reconnaissance: ${event.error}`);
rec.onend = () => {
this.recognition = null;
onEnd();
};
this.recognition = rec;
try {
rec.start();
} catch (err) {
onError((err as Error).message);
}
}
stop(): void {
try {
this.recognition?.stop();
} catch {
/* ignore */
}
this.recognition = null;
}
}