Compare commits

..
10 Commits
6 changed files with 769 additions and 55 deletions

No files matched your search

+3 -12
View File
@@ -1,7 +1,7 @@
# 🚀 EveFlow — Compagnon de Bureau Rétro-Futuriste 3D
[![Windows Version](https://img.shields.io/badge/OS-Windows-blue.svg?style=flat-square&logo=windows)](https://github.com/R0m1k3/EveFlow)
[![Version](https://img.shields.io/badge/version-1.0.3-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases/tag/v1.0.3)
[![Version](https://img.shields.io/badge/version-1.0.8-brightgreen.svg?style=flat-square)](https://github.com/R0m1k3/EveFlow/releases/tag/v1.0.8)
[![License](https://img.shields.io/badge/license-MIT-lightgrey.svg?style=flat-square)](LICENSE)
**EveFlow** est un compagnon de bureau Windows immersif haut de gamme, combinant une esthétique cyberpunk rétro-futuriste soignée et des technologies d'intelligence artificielle avancées. Il intègre un assistant virtuel en 3D nommé **Eve**, animé en temps réel avec des expressions émotionnelles dynamiques et synchronisé avec des services de synthèse vocale (TTS/STT) locaux et des architectures multi-agents.
@@ -12,16 +12,7 @@
Découvrez le cockpit de contrôle immersif et le mode widget compact d'EveFlow :
<table>
<tr>
<td align="center"><strong>Cockpit Principal (Mode Discussion & Télémétrie)</strong></td>
<td align="center"><strong>Widget Flottant (Mode Compact Transparent)</strong></td>
</tr>
<tr>
<td><img src="public/screenshots/cockpit_view.png" width="500" alt="EveFlow Cockpit View"/></td>
<td><img src="public/screenshots/compact_widget.png" width="300" alt="EveFlow Compact Widget"/></td>
</tr>
</table>
---
@@ -83,7 +74,7 @@ Pour générer l'exécutable d'installation Windows (`.exe`) autonome :
```bash
npm run dist
```
*L'exécutable d'installation NSIS (`EveFlow Setup 1.0.3.exe`) et la build décompressée seront générés dans le dossier `./out/`.*
*L'exécutable d'installation NSIS (`EveFlow Setup 1.0.8.exe`) et la build décompressée seront générés dans le dossier `./out/`.*
---
+15
View File
@@ -267,6 +267,21 @@ app.whenReady().then(() => {
}
);
// Autoriser l'accès aux périphériques de capture (microphone) sous Electron
session.defaultSession.setPermissionRequestHandler((webContents, permission, callback) => {
if (permission === 'media') {
return callback(true);
}
callback(false);
});
session.defaultSession.setPermissionCheckHandler((webContents, permission, requestingOrigin, details) => {
if (permission === 'media') {
return true;
}
return false;
});
startWebhookServer();
createWindow();
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "eveflow",
"version": "1.0.3",
"version": "1.0.8",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "eveflow",
"version": "1.0.3",
"version": "1.0.8",
"dependencies": {
"@react-three/drei": "^9.105.0",
"@react-three/fiber": "^8.16.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "eveflow",
"version": "1.0.3",
"version": "1.0.8",
"description": "Retro-futuristic Windows companion with 3D Eve bot, local TTS/STT, and multi-agent connectivity",
"main": "main.js",
"scripts": {
+286 -6
View File
@@ -892,7 +892,18 @@ export const App: React.FC = () => {
const [selectedVoice, setSelectedVoice] = useState<string>('');
const [ttsRate, setTtsRate] = useState<number>(1.15);
const [sttLang, setSttLang] = useState<string>('fr-FR');
const [ttsProvider, setTtsProvider] = useState<'system' | 'google-free'>('google-free');
const [ttsProvider, setTtsProvider] = useState<'system' | 'google-free' | 'openai-tts'>('google-free');
// Configuration personnalisée de l'API Speech-To-Text (ASR) et Text-To-Speech (TTS)
const [sttProvider, setSttProvider] = useState<'browser' | 'qwen3-asr'>('browser');
const [sttApiUrl, setSttApiUrl] = useState<string>('http://127.0.0.1:8000/v1');
const [sttApiKey, setSttApiKey] = useState<string>('');
const [sttModel, setSttModel] = useState<string>('Qwen/Qwen3-ASR-0.6B');
const [ttsApiUrl, setTtsApiUrl] = useState<string>('http://127.0.0.1:8000/v1');
const [ttsApiKey, setTtsApiKey] = useState<string>('');
const [ttsModel, setTtsModel] = useState<string>('tts-1');
const activeAvatar = AVATAR_PROFILES[avatarId];
const hermesUnreadRuns = hermesRunHistory.filter(run => !run.read).length;
const hermesSyncLabel = hermesSyncState === 'offline'
@@ -1136,6 +1147,8 @@ export const App: React.FC = () => {
// --- INITIALISATION ---
useEffect(() => {
const service = new AudioService();
service.updateSTTConfig(sttProvider, sttApiUrl, sttApiKey, sttModel);
service.updateTTSConfig(ttsApiUrl, ttsApiKey, ttsModel);
audioServiceRef.current = service;
// Charger la configuration depuis le store persistant (IPC fichier > localStorage)
@@ -1159,11 +1172,40 @@ export const App: React.FC = () => {
});
persistRead('eveflow_tts_provider').then(savedProvider => {
if (savedProvider === 'system' || savedProvider === 'google-free') {
setTtsProvider(savedProvider as 'system' | 'google-free');
if (savedProvider === 'system' || savedProvider === 'google-free' || savedProvider === 'openai-tts') {
setTtsProvider(savedProvider as 'system' | 'google-free' | 'openai-tts');
}
});
persistRead('eveflow_stt_provider').then(val => {
if (val === 'browser' || val === 'qwen3-asr') setSttProvider(val as 'browser' | 'qwen3-asr');
});
persistRead('eveflow_stt_api_url').then(val => {
if (val) setSttApiUrl(val);
});
persistRead('eveflow_stt_api_key').then(val => {
if (val) setSttApiKey(val);
});
persistRead('eveflow_stt_model').then(val => {
if (val) {
if (val === 'qwen3-asr') {
setSttModel('Qwen/Qwen3-ASR-0.6B');
persistWrite('eveflow_stt_model', 'Qwen/Qwen3-ASR-0.6B');
} else {
setSttModel(val);
}
}
});
persistRead('eveflow_tts_api_url').then(val => {
if (val) setTtsApiUrl(val);
});
persistRead('eveflow_tts_api_key').then(val => {
if (val) setTtsApiKey(val);
});
persistRead('eveflow_tts_model').then(val => {
if (val) setTtsModel(val);
});
persistRead('eveflow_avatar_id').then(savedAvatar => {
if (savedAvatar && savedAvatar in AVATAR_PROFILES) {
setAvatarId(savedAvatar as EveAvatar);
@@ -1221,6 +1263,49 @@ export const App: React.FC = () => {
return undefined;
}, []);
// Synchroniser la configuration de AudioService avec l'état React
useEffect(() => {
if (audioServiceRef.current) {
audioServiceRef.current.updateSTTConfig(sttProvider, sttApiUrl, sttApiKey, sttModel);
}
}, [sttProvider, sttApiUrl, sttApiKey, sttModel]);
useEffect(() => {
if (audioServiceRef.current) {
audioServiceRef.current.updateTTSConfig(ttsApiUrl, ttsApiKey, ttsModel);
}
}, [ttsApiUrl, ttsApiKey, ttsModel]);
const handleSttProviderChange = (val: 'browser' | 'qwen3-asr') => {
setSttProvider(val);
persistWrite('eveflow_stt_provider', val);
};
const handleSttApiUrlChange = (val: string) => {
setSttApiUrl(val);
persistWrite('eveflow_stt_api_url', val);
};
const handleSttApiKeyChange = (val: string) => {
setSttApiKey(val);
persistWrite('eveflow_stt_api_key', val);
};
const handleSttModelChange = (val: string) => {
setSttModel(val);
persistWrite('eveflow_stt_model', val);
};
const handleTtsApiUrlChange = (val: string) => {
setTtsApiUrl(val);
persistWrite('eveflow_tts_api_url', val);
};
const handleTtsApiKeyChange = (val: string) => {
setTtsApiKey(val);
persistWrite('eveflow_tts_api_key', val);
};
const handleTtsModelChange = (val: string) => {
setTtsModel(val);
persistWrite('eveflow_tts_model', val);
};
// Défilement automatique
useEffect(() => {
messagesEndRef.current?.scrollIntoView({ behavior: 'smooth' });
@@ -1528,12 +1613,18 @@ export const App: React.FC = () => {
const handleStartVocalRecord = () => {
if (!audioServiceRef.current) return;
setErrorMessage(null);
audioServiceRef.current.stopSpeaking();
setIsSpeaking(false);
audioServiceRef.current.startListening(
(text) => {
handleSendMessage(text);
if (text.trim()) {
handleSendMessage(text);
} else {
setErrorMessage("Aucune parole n'a été détectée. Veuillez parler distinctement près du micro.");
setCurrentEmotion('sad');
}
},
() => {
setIsListening(true);
@@ -1544,6 +1635,8 @@ export const App: React.FC = () => {
},
(err) => {
setIsListening(false);
setErrorMessage(`Erreur de reconnaissance vocale : ${err}`);
setCurrentEmotion('sad');
console.error(err);
},
sttLang
@@ -2069,7 +2162,7 @@ export const App: React.FC = () => {
</div>
{/* B. TTS / STT */}
{/* B. TTS (Synthese Vocale / Voice Reader) */}
<div className="settings-card">
<h3 className="label-title" style={{ display: 'flex', alignItems: 'center', gap: '6px' }}>
<Volume2 style={{ width: '14px', height: '14px' }} /> Voix de {activeAvatar.name} (Text-to-Speech)
@@ -2078,7 +2171,7 @@ export const App: React.FC = () => {
<div style={{ display: 'flex', flexDirection: 'column', gap: '12px' }}>
<div>
<label className="label-title">Type de Synthèse Vocale</label>
<div className="grid-2" style={{ gap: '10px', marginTop: '4px' }}>
<div className="grid-3" style={{ gap: '10px', marginTop: '4px' }}>
<button
onClick={() => {
setTtsProvider('google-free');
@@ -2113,6 +2206,23 @@ export const App: React.FC = () => {
>
🔌 Voix Système (Local)
</button>
<button
onClick={() => {
setTtsProvider('openai-tts');
persistWrite('eveflow_tts_provider', 'openai-tts');
}}
className="neon-btn"
style={{
padding: '8px 12px',
fontSize: '11px',
background: ttsProvider === 'openai-tts' ? '' : 'transparent',
border: ttsProvider === 'openai-tts' ? 'none' : '1px solid #cbd5e1',
color: ttsProvider === 'openai-tts' ? 'white' : '#64748b',
boxShadow: ttsProvider === 'openai-tts' ? '' : 'none'
}}
>
🌐 API Personnalisée
</button>
</div>
</div>
@@ -2143,6 +2253,63 @@ export const App: React.FC = () => {
</div>
)}
{ttsProvider === 'openai-tts' && (
<div style={{ display: 'flex', flexDirection: 'column', gap: '12px' }}>
<div>
<label className="label-title">URL de l'API TTS</label>
<input
type="text"
value={ttsApiUrl}
onChange={(e) => handleTtsApiUrlChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="Ex: http://127.0.0.1:8000/v1"
/>
<span style={{ display: 'block', marginTop: '5px', fontSize: '10px', fontFamily: 'var(--font-mono)', color: 'var(--text-secondary)', lineHeight: 1.5 }}>
ℹ️ L'endpoint <code>/v1/audio/speech</code> sera utilisé pour la synthèse.
</span>
</div>
<div className="grid-2">
<div>
<label className="label-title">Modèle TTS</label>
<input
type="text"
value={ttsModel}
onChange={(e) => handleTtsModelChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="tts-1"
/>
</div>
<div>
<label className="label-title">Clé API (Optionnel)</label>
<input
type="password"
value={ttsApiKey}
onChange={(e) => handleTtsApiKeyChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="Bearer token (API_KEY)"
/>
</div>
</div>
<div>
<label className="label-title">Nom de la Voix (ex: alloy, onyx, nova)</label>
<input
type="text"
value={selectedVoice}
onChange={(e) => {
setSelectedVoice(e.target.value);
persistWrite('eveflow_voice', e.target.value);
}}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="alloy"
/>
</div>
</div>
)}
<div className="grid-2">
<div>
<label className="label-title">Vitesse: {ttsRate.toFixed(2)}</label>
@@ -2185,6 +2352,98 @@ export const App: React.FC = () => {
</button>
</div>
</div>
{/* C. STT (Reconnaissance Vocale) */}
<div className="settings-card">
<h3 className="label-title" style={{ display: 'flex', alignItems: 'center', gap: '6px' }}>
<Mic style={{ width: '14px', height: '14px' }} /> Reconnaissance Vocale (Speech-to-Text)
</h3>
<div style={{ display: 'flex', flexDirection: 'column', gap: '12px' }}>
<div>
<label className="label-title">Moteur de Reconnaissance Vocale</label>
<div className="grid-2" style={{ gap: '10px', marginTop: '4px' }}>
<button
onClick={() => handleSttProviderChange('browser')}
className="neon-btn"
style={{
padding: '8px 12px',
fontSize: '11px',
background: sttProvider === 'browser' ? '' : 'transparent',
border: sttProvider === 'browser' ? 'none' : '1px solid #cbd5e1',
color: sttProvider === 'browser' ? 'white' : '#64748b',
boxShadow: sttProvider === 'browser' ? '' : 'none'
}}
>
🟢 Moteur Navigateur (Local)
</button>
<button
onClick={() => handleSttProviderChange('qwen3-asr')}
className="neon-btn"
style={{
padding: '8px 12px',
fontSize: '11px',
background: sttProvider === 'qwen3-asr' ? '' : 'transparent',
border: sttProvider === 'qwen3-asr' ? 'none' : '1px solid #cbd5e1',
color: sttProvider === 'qwen3-asr' ? 'white' : '#64748b',
boxShadow: sttProvider === 'qwen3-asr' ? '' : 'none'
}}
>
🌐 Qwen3 ASR (API)
</button>
</div>
</div>
{sttProvider === 'browser' && (
<div style={{ padding: '8px 12px', backgroundColor: 'var(--accent-light)', border: '1px solid rgba(0,212,245,0.2)', borderRadius: '10px', fontSize: '11px', fontFamily: 'var(--font-mono)', color: 'var(--text-neon)', lineHeight: 1.5 }}>
ℹ️ **Mode Navigateur Actif** : Utilise l'API de reconnaissance vocale standard intégrée à votre système (locale, rapide et gratuite).
</div>
)}
{sttProvider === 'qwen3-asr' && (
<div style={{ display: 'flex', flexDirection: 'column', gap: '12px' }}>
<div>
<label className="label-title">URL de l'API STT</label>
<input
type="text"
value={sttApiUrl}
onChange={(e) => handleSttApiUrlChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="Ex: http://127.0.0.1:8000/v1"
/>
<span style={{ display: 'block', marginTop: '5px', fontSize: '10px', fontFamily: 'var(--font-mono)', color: 'var(--text-secondary)', lineHeight: 1.5 }}>
ℹ️ L'endpoint <code>/v1/audio/transcriptions</code> sera utilisé pour la transcription.
</span>
</div>
<div className="grid-2">
<div>
<label className="label-title">Modèle ASR</label>
<input
type="text"
value={sttModel}
onChange={(e) => handleSttModelChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="qwen3-asr"
/>
</div>
<div>
<label className="label-title">Clé API (Optionnel)</label>
<input
type="password"
value={sttApiKey}
onChange={(e) => handleSttApiKeyChange(e.target.value)}
className="glow-input"
style={{ padding: '8px 12px', fontSize: '12px' }}
placeholder="Bearer token"
/>
</div>
</div>
</div>
)}
</div>
</div>
</section>
)}
@@ -2284,6 +2543,27 @@ export const App: React.FC = () => {
<div ref={compactMessagesEndRef} />
</div>
{errorMessage && (
<div style={{
backgroundColor: 'rgba(239, 68, 68, 0.12)',
border: '1px solid rgba(239, 68, 68, 0.35)',
borderRadius: '8px',
color: '#ff6b6b',
padding: '6px 10px',
fontSize: '10px',
display: 'flex',
gap: '6px',
alignItems: 'center',
boxShadow: '0 0 10px rgba(239, 68, 68, 0.05)'
}}>
<AlertCircle style={{ width: '13px', height: '13px', flexShrink: 0, color: '#ef4444' }} />
<div style={{ flex: 1, overflow: 'hidden', textOverflow: 'ellipsis', whiteSpace: 'nowrap' }} title={errorMessage}>
<strong style={{ textTransform: 'uppercase', marginRight: '4px' }}>DÉFAILLANCE:</strong>
{errorMessage}
</div>
</div>
)}
{/* Barre de saisie compacte */}
<div style={{ display: 'flex', gap: '6px', alignItems: 'center', width: '100%', paddingBottom: '2px' }}>
<input
+462 -34
View File
@@ -55,10 +55,34 @@ export class AudioService {
private recognition: any = null;
private isListeningActive = false;
private currentAudioElement: HTMLAudioElement | null = null;
private audioQueue: { text: string; voiceName?: string; rate: number; pitch: number; provider: 'system' | 'google-free' }[] = [];
private audioQueue: { text: string; voiceName?: string; rate: number; pitch: number; provider: 'system' | 'google-free' | 'openai-tts' }[] = [];
private isPlayingQueue = false;
private activeReject: ((err: any) => void) | null = null;
// ASR/STT & TTS Custom API settings
private sttProvider: 'browser' | 'qwen3-asr' = 'browser';
private sttApiUrl = 'http://127.0.0.1:8000/v1';
private sttApiKey = '';
private sttModel = 'Qwen/Qwen3-ASR-0.6B';
// Propriétés pour l'enregistrement WAV direct
private audioContext: AudioContext | null = null;
private scriptProcessor: ScriptProcessorNode | null = null;
private micStream: MediaStream | null = null;
private leftChannel: Float32Array[] = [];
private recordingLength = 0;
private sampleRate = 44100;
// Callbacks pour l'écoute
private activeOnResult: ((text: string) => void) | null = null;
private activeOnEnd: (() => void) | null = null;
private activeOnError: ((err: string) => void) | null = null;
private activeLang = 'fr-FR';
private ttsApiUrl = 'http://127.0.0.1:8000/v1';
private ttsApiKey = '';
private ttsModel = 'tts-1';
public get isSpeakingActive(): boolean {
return this.isPlayingQueue || this.audioQueue.length > 0;
}
@@ -89,11 +113,39 @@ export class AudioService {
}
}
private logInfo(tag: string, message: string, data?: any) {
console.log(`[${tag}] ${message}`, data);
if ((window as any).electronAPI?.writeLog) {
(window as any).electronAPI.writeLog({
ts: new Date().toISOString(),
level: 'INFO',
tag,
message,
data: data ? { message: data.message, stack: data.stack, ...data } : undefined
});
}
}
constructor() {
this.ttsSynth = window.speechSynthesis;
this.initSTT();
}
// Permet de mettre à jour dynamiquement la configuration de la reconnaissance vocale
public updateSTTConfig(provider: 'browser' | 'qwen3-asr', apiUrl: string, apiKey: string, model: string) {
this.sttProvider = provider;
this.sttApiUrl = apiUrl;
this.sttApiKey = apiKey;
this.sttModel = model;
}
// Permet de mettre à jour dynamiquement la configuration du lecteur de voix personnalisé
public updateTTSConfig(apiUrl: string, apiKey: string, model: string) {
this.ttsApiUrl = apiUrl;
this.ttsApiKey = apiKey;
this.ttsModel = model;
}
// Initialisation de la reconnaissance vocale locale et gratuite (Web Speech API)
private initSTT() {
const SpeechRecognition = (window as any).SpeechRecognition || (window as any).webkitSpeechRecognition;
@@ -339,6 +391,98 @@ export class AudioService {
}
}
// Synthese vocale personnalisee via API OpenAI /v1/audio/speech
private async speakCustomAPI(text: string, voiceName?: string, rate: number = 1.0): Promise<void> {
const cleanText = this._cleanForTTS(text);
if (!cleanText.trim()) return;
if (!this.ttsApiUrl) {
throw new Error("L'URL de l'API TTS n'est pas configurée.");
}
const base = this.ttsApiUrl.replace(/\/$/, '');
let endpoint: string;
if (base.endsWith('/speech')) {
endpoint = base;
} else if (base.endsWith('/v1')) {
endpoint = `${base}/audio/speech`;
} else {
endpoint = `${base}/v1/audio/speech`;
}
const body = {
model: this.ttsModel || 'tts-1',
input: cleanText,
voice: voiceName || 'alloy'
};
const headers: Record<string, string> = {
'Content-Type': 'application/json'
};
if (this.ttsApiKey) {
headers['Authorization'] = `Bearer ${this.ttsApiKey.trim()}`;
}
const response = await fetch(endpoint, {
method: 'POST',
headers,
body: JSON.stringify(body)
});
if (!response.ok) {
const rawErr = await response.text().catch(() => '(no body)');
let errMsg = `Erreur TTS API: ${response.status}`;
try {
const parsed = JSON.parse(rawErr);
errMsg = parsed.error?.message || parsed.message || rawErr;
} catch {}
throw new Error(errMsg);
}
const blob = await response.blob();
const url = URL.createObjectURL(blob);
await new Promise<void>((resolve, reject) => {
const audio = new Audio(url);
this.currentAudioElement = audio;
audio.playbackRate = rate;
this.activeReject = reject;
const timeout = setTimeout(() => {
if (this.currentAudioElement === audio) this.currentAudioElement = null;
this.activeReject = null;
audio.pause();
reject(new Error("playback_stopped"));
}, 60000); // 60 secondes max pour une portion
audio.onended = () => {
clearTimeout(timeout);
if (this.currentAudioElement === audio) this.currentAudioElement = null;
this.activeReject = null;
URL.revokeObjectURL(url);
resolve();
};
audio.onerror = () => {
clearTimeout(timeout);
if (this.currentAudioElement === audio) this.currentAudioElement = null;
this.activeReject = null;
URL.revokeObjectURL(url);
const code = audio.error ? audio.error.code : 'UNKNOWN';
const msg = audio.error ? audio.error.message : 'No message';
reject(new Error(`Erreur de lecture audio API (code: ${code}, msg: ${msg})`));
};
audio.play().catch((err) => {
clearTimeout(timeout);
if (this.currentAudioElement === audio) this.currentAudioElement = null;
this.activeReject = null;
URL.revokeObjectURL(url);
reject(err);
});
});
}
// File de traitement asynchrone des phrases de synthèse vocale (Zero Trust & Zero Dependency)
private async processQueue(): Promise<void> {
if (this.isPlayingQueue) return;
@@ -351,6 +495,8 @@ export class AudioService {
try {
if (item.provider === 'google-free') {
await this.speakGoogleFree(item.text, item.rate);
} else if (item.provider === 'openai-tts') {
await this.speakCustomAPI(item.text, item.voiceName, item.rate);
} else {
await new Promise<void>((resolve, reject) => {
if (!this.ttsSynth) { reject('TTS non supporté'); return; }
@@ -371,7 +517,7 @@ export class AudioService {
break;
}
this.logWarn("AudioService", "Échec de la lecture de la phrase dans la file", err);
if (item.provider === 'google-free') {
if (item.provider === 'google-free' || item.provider === 'openai-tts') {
// Repli automatique transparent sur la voix locale hors-ligne pour ce segment précis en cas de défaillance
try {
await new Promise<void>((resolve, reject) => {
@@ -406,7 +552,7 @@ export class AudioService {
voiceName?: string,
rate: number = 1.0,
pitch: number = 1.1,
provider: 'system' | 'google-free' = 'google-free'
provider: 'system' | 'google-free' | 'openai-tts' = 'google-free'
): Promise<void> {
this.stopSpeaking();
this.queueSentence(text, voiceName, rate, pitch, provider);
@@ -423,7 +569,7 @@ export class AudioService {
voiceName?: string,
rate: number = 1.0,
pitch: number = 1.1,
provider: 'system' | 'google-free' = 'google-free'
provider: 'system' | 'google-free' | 'openai-tts' = 'google-free'
): void {
const clean = text.trim();
if (!clean) return;
@@ -463,6 +609,280 @@ export class AudioService {
}
}
// Démarrer l'écoute vocale API via AudioContext (WAV Recorder)
private async startListeningAPI(
onResult: (text: string) => void,
onStart: () => void,
onEnd: () => void,
onError: (err: string) => void,
lang: string
) {
if (this.isListeningActive) {
this.stopListening();
}
// Assigner les callbacks pour y avoir accès à l'arrêt
this.activeOnResult = onResult;
this.activeOnEnd = onEnd;
this.activeOnError = onError;
this.activeLang = lang;
try {
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
this.micStream = stream;
this.leftChannel = [];
this.recordingLength = 0;
const AudioCtx = window.AudioContext || (window as any).webkitAudioContext;
this.audioContext = new AudioCtx();
// Assurer que l'AudioContext est actif (non suspendu)
if (this.audioContext.state === 'suspended') {
this.logWarn("STT", "AudioContext initialisé à l'état suspendu. Tentative de reprise...");
await this.audioContext.resume();
}
this.sampleRate = this.audioContext.sampleRate;
this.logInfo("STT", "AudioContext démarré avec succès", {
state: this.audioContext.state,
sampleRate: this.sampleRate
});
const source = this.audioContext.createMediaStreamSource(stream);
// bufferSize de 2048, 1 canal d'entrée, 1 de sortie
this.scriptProcessor = this.audioContext.createScriptProcessor(2048, 1, 1);
this.scriptProcessor.onaudioprocess = (e) => {
if (!this.isListeningActive) return;
const inputData = e.inputBuffer.getChannelData(0);
// Cloner les échantillons
this.leftChannel.push(new Float32Array(inputData));
this.recordingLength += inputData.length;
};
source.connect(this.scriptProcessor);
this.scriptProcessor.connect(this.audioContext.destination);
this.isListeningActive = true;
onStart();
} catch (err: any) {
this.isListeningActive = false;
this.logError("STT", "Impossible d'accéder au microphone", err);
onError(err.message || "Impossible d'accéder au microphone.");
}
}
// Arrêter proprement la capture AudioContext et exporter en WAV
private stopListeningAPI() {
if (!this.isListeningActive) return;
this.isListeningActive = false;
// Déconnecter le processeur
if (this.scriptProcessor) {
this.scriptProcessor.disconnect();
this.scriptProcessor.onaudioprocess = null;
this.scriptProcessor = null;
}
// Arrêter le flux micro
if (this.micStream) {
this.micStream.getTracks().forEach(track => track.stop());
this.micStream = null;
}
// Fermer l'audio context
if (this.audioContext) {
this.audioContext.close();
this.audioContext = null;
}
const onEndCallback = this.activeOnEnd;
const onResultCallback = this.activeOnResult;
const onErrorCallback = this.activeOnError;
const lang = this.activeLang;
if (onEndCallback) onEndCallback();
if (this.recordingLength === 0) {
if (onErrorCallback) onErrorCallback("Aucune donnée audio capturée");
return;
}
// Exporter le tampon audio en fichier WAV
const wavBlob = this.exportWAV(this.leftChannel, this.recordingLength, this.sampleRate);
this.logInfo("STT", "Arrêt de l'écoute, exportation WAV...", {
recordingLength: this.recordingLength,
sampleRate: this.sampleRate,
wavSize: wavBlob.size
});
this.transcribeAudioWithAPI(wavBlob, lang)
.then(text => {
if (onResultCallback) onResultCallback(text);
})
.catch(err => {
this.logError("STT", "Erreur de transcription", err);
if (onErrorCallback) onErrorCallback(err.message || "Erreur de transcription de l'API");
});
}
// Fonctions utilitaires pour générer le conteneur WAV PCM 16-bit
private exportWAV(channelData: Float32Array[], recordingLength: number, sampleRate: number): Blob {
const originalBuffer = this.mergeBuffers(channelData, recordingLength);
// Rééchantillonner à 16000 Hz pour la compatibilité avec l'API ASR (Whisper / Qwen3-ASR)
const targetSampleRate = 16000;
const resampledBuffer = this.downsampleBuffer(originalBuffer, sampleRate, targetSampleRate);
const wavBuffer = new ArrayBuffer(44 + resampledBuffer.length * 2);
const view = new DataView(wavBuffer);
/* Identifiant RIFF */
this.writeString(view, 0, 'RIFF');
/* Taille du fichier */
view.setUint32(4, 36 + resampledBuffer.length * 2, true);
/* Type RIFF */
this.writeString(view, 8, 'WAVE');
/* Identifiant format fmt */
this.writeString(view, 12, 'fmt ');
/* Longueur du chunk format */
view.setUint32(16, 16, true);
/* Format d'encodage (1 = PCM non compressé) */
view.setUint16(20, 1, true);
/* Nombre de canaux (1 = Mono) */
view.setUint16(22, 1, true);
/* Fréquence d'échantillonnage */
view.setUint32(24, targetSampleRate, true);
/* Débit d'octets (targetSampleRate * blockAlign) */
view.setUint32(28, targetSampleRate * 2, true);
/* Block align (1 canal * 2 octets par échantillon) */
view.setUint16(32, 2, true);
/* Bits par échantillon (16 bits) */
view.setUint16(34, 16, true);
/* Identifiant de données data */
this.writeString(view, 36, 'data');
/* Longueur des données audio */
view.setUint32(40, resampledBuffer.length * 2, true);
this.floatTo16BitPCM(view, 44, resampledBuffer);
return new Blob([view], { type: 'audio/wav' });
}
private downsampleBuffer(buffer: Float32Array, originalSampleRate: number, targetSampleRate: number): Float32Array {
if (originalSampleRate === targetSampleRate) {
return buffer;
}
const sampleRateRatio = originalSampleRate / targetSampleRate;
const newLength = Math.round(buffer.length / sampleRateRatio);
const result = new Float32Array(newLength);
let offsetResult = 0;
let offsetBuffer = 0;
while (offsetResult < result.length) {
const nextOffsetBuffer = Math.round((offsetResult + 1) * sampleRateRatio);
// Moyennage simple (box filter) pour éviter l'aliasing
let accum = 0;
let count = 0;
for (let i = offsetBuffer; i < nextOffsetBuffer && i < buffer.length; i++) {
accum += buffer[i];
count++;
}
if (count > 0) {
result[offsetResult] = accum / count;
} else {
result[offsetResult] = buffer[Math.min(offsetBuffer, buffer.length - 1)];
}
offsetResult++;
offsetBuffer = nextOffsetBuffer;
}
return result;
}
private mergeBuffers(channelData: Float32Array[], recordingLength: number): Float32Array {
const result = new Float32Array(recordingLength);
let offset = 0;
for (let i = 0; i < channelData.length; i++) {
result.set(channelData[i], offset);
offset += channelData[i].length;
}
return result;
}
private writeString(view: DataView, offset: number, string: string) {
for (let i = 0; i < string.length; i++) {
view.setUint8(offset + i, string.charCodeAt(i));
}
}
private floatTo16BitPCM(output: DataView, offset: number, input: Float32Array) {
for (let i = 0; i < input.length; i++, offset += 2) {
const s = Math.max(-1, Math.min(1, input[i]));
output.setInt16(offset, s < 0 ? s * 0x8000 : s * 0x7FFF, true);
}
}
// Requête API Speech-To-Text compatible OpenAI /v1/audio/transcriptions
private async transcribeAudioWithAPI(audioBlob: Blob, lang: string): Promise<string> {
if (!this.sttApiUrl) {
throw new Error("L'URL de l'API STT n'est pas configurée.");
}
const base = this.sttApiUrl.replace(/\/$/, '');
let endpoint: string;
if (base.endsWith('/transcriptions')) {
endpoint = base;
} else if (base.endsWith('/v1')) {
endpoint = `${base}/audio/transcriptions`;
} else {
endpoint = `${base}/v1/audio/transcriptions`;
}
const formData = new FormData();
formData.append('file', audioBlob, 'audio.wav');
formData.append('model', this.sttModel || 'Qwen/Qwen3-ASR-0.6B');
if (lang) {
const isoLang = lang.split('-')[0];
formData.append('language', isoLang);
}
const headers: Record<string, string> = {};
if (this.sttApiKey) {
headers['Authorization'] = `Bearer ${this.sttApiKey.trim()}`;
}
const response = await fetch(endpoint, {
method: 'POST',
headers,
body: formData
});
if (!response.ok) {
const rawErr = await response.text().catch(() => '(no body)');
let errMsg = `Erreur STT API: ${response.status}`;
try {
const parsed = JSON.parse(rawErr);
errMsg = parsed.error?.message || parsed.message || rawErr;
} catch {}
throw new Error(errMsg);
}
const data = await response.json();
if (data && typeof data.text === 'string') {
return data.text;
}
throw new Error("Format de réponse de l'API STT invalide.");
}
// Lancer l'écoute vocale (STT - Speech-To-Text)
public startListening(
onResult: (text: string) => void,
@@ -471,48 +891,56 @@ export class AudioService {
onError: (err: string) => void,
lang: string = 'fr-FR'
) {
if (!this.recognition) {
onError("STT non supporté");
return;
}
if (this.sttProvider === 'qwen3-asr') {
this.startListeningAPI(onResult, onStart, onEnd, onError, lang);
} else {
if (!this.recognition) {
onError("STT non supporté par ce navigateur");
return;
}
if (this.isListeningActive) {
this.recognition.stop();
}
if (this.isListeningActive) {
this.recognition.stop();
}
this.recognition.lang = lang;
this.isListeningActive = true;
this.recognition.lang = lang;
this.isListeningActive = true;
this.recognition.onstart = () => {
onStart();
};
this.recognition.onstart = () => {
onStart();
};
this.recognition.onresult = (event: any) => {
const resultText = event.results[0][0].transcript;
onResult(resultText);
};
this.recognition.onresult = (event: any) => {
const resultText = event.results[0][0].transcript;
onResult(resultText);
};
this.recognition.onerror = (event: any) => {
onError(event.error);
};
this.recognition.onerror = (event: any) => {
onError(event.error);
};
this.recognition.onend = () => {
this.isListeningActive = false;
onEnd();
};
this.recognition.onend = () => {
this.isListeningActive = false;
onEnd();
};
try {
this.recognition.start();
} catch (e: any) {
onError(e.message || "Erreur de démarrage de l'écoute");
try {
this.recognition.start();
} catch (e: any) {
onError(e.message || "Erreur de démarrage de l'écoute");
}
}
}
// Arrêter l'écoute vocale
public stopListening() {
if (this.recognition && this.isListeningActive) {
this.recognition.stop();
this.isListeningActive = false;
if (this.sttProvider === 'qwen3-asr') {
this.stopListeningAPI();
} else {
if (this.recognition && this.isListeningActive) {
this.recognition.stop();
this.isListeningActive = false;
}
}
}
}