From 1e72e829658e58b0594320324ff17e8e621b3bee Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 13 Jun 2026 15:45:33 +0000 Subject: [PATCH] =?UTF-8?q?Qualit=C3=A9=20de=20transcription=20:=20audio?= =?UTF-8?q?=20propre=20pour=20le=20moteur=20(corrige=20les=20erreurs)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit L'AGC sur-amplifiait (jusqu'à x60) l'audio envoyé au moteur -> signal déformé/bruité -> mauvaise reconnaissance. Désormais : - Le VAD détecte sur l'audio amplifié, mais le segment stocke l'audio BRUT - Le segment est normalisé proprement (mise à l'échelle linéaire unique vers crête ~22000, sans saturation) juste avant le moteur - Segment max ramené de 25 s à 15 s pour un retour plus rapide Validé bout en bout : micro faible -> détecté -> audio propre -> transcrit. https://claude.ai/code/session_01YHMp3EKzr4s6o8w1ygxuUe --- app/main.py | 23 ++++++++++++++++------- app/segmenter.py | 18 ++++++++++-------- 2 files changed, 26 insertions(+), 15 deletions(-) diff --git a/app/main.py b/app/main.py index 05f5397..7b8f690 100644 --- a/app/main.py +++ b/app/main.py @@ -189,19 +189,28 @@ def pcm_to_wav(pcm: bytes) -> bytes: return buf.getvalue() -QUIET_PEAK = 9000 # niveau crête en dessous duquel on amplifie (échelle int16) +NORM_TARGET_PEAK = 22000 # niveau crête visé (≈0,67 pleine échelle), sans saturer +NORM_MAX_GAIN = 25.0 # amplification maximale +NORM_NOISE_FLOOR = 80 # en dessous : silence, on ne touche pas def boost_quiet_audio(pcm: bytes) -> tuple[bytes, int]: - """Amplifie les segments trop faibles (micros Bluetooth mains-libres...). + """Normalise proprement le niveau du segment pour le moteur ASR. - Retourne (audio, niveau crête d'origine). + Mise à l'échelle linéaire unique du segment entier vers un niveau cible + confortable, SANS saturation (gain = cible / crête), bien meilleure pour + la reconnaissance qu'une amplification par à-coups. Renvoie (audio, crête). """ peak = audioop.max(pcm, 2) - if 0 < peak < QUIET_PEAK: - factor = min(6.0, 26000 / peak) - return audioop.mul(pcm, 2, factor), peak - return pcm, peak + if peak < NORM_NOISE_FLOOR: + return pcm, peak + gain = min(NORM_MAX_GAIN, NORM_TARGET_PEAK / peak) + if gain <= 1.05: + return pcm, peak + try: + return audioop.mul(pcm, 2, gain), peak + except Exception: + return pcm, peak async def _post_transcription(data: dict, wav: bytes) -> httpx.Response: diff --git a/app/segmenter.py b/app/segmenter.py index e350279..f273c00 100644 --- a/app/segmenter.py +++ b/app/segmenter.py @@ -22,7 +22,7 @@ PREROLL_FRAMES = 10 # 300 ms conservées avant le déclenchement TRIGGER_RATIO = 0.6 # part de trames "parole" du pré-roll pour démarrer SILENCE_END_MS = 700 # silence qui clôt un segment MIN_SPEECH_MS = 300 # en dessous, le segment est ignoré (bruit) -MAX_SEGMENT_S = 25 # coupe forcée pour garder une latence raisonnable +MAX_SEGMENT_S = 15 # coupe forcée pour garder une latence raisonnable # Contrôle de gain automatique appliqué AVANT la détection de parole : sans lui, # un micro faible (casque...) reste sous le seuil du VAD et aucune phrase n'est @@ -83,10 +83,12 @@ class SpeechSegmenter: self._pending.extend(data) segments = [] while len(self._pending) >= FRAME_BYTES: - frame = bytes(self._pending[:FRAME_BYTES]) + raw = bytes(self._pending[:FRAME_BYTES]) del self._pending[:FRAME_BYTES] - frame = self._agc.process(frame) # amplifie avant le VAD - seg = self._process_frame(frame) + # Le VAD travaille sur l'audio amplifié (détection fiable même micro + # faible) ; le segment stocke l'audio BRUT (transcription propre). + amplified = self._agc.process(raw) + seg = self._process_frame(raw, amplified) if seg is not None: segments.append(seg) return segments @@ -98,12 +100,12 @@ class SpeechSegmenter: self._pending.clear() return seg - def _process_frame(self, frame: bytes) -> Segment | None: - is_speech = self._vad.is_speech(frame, SAMPLE_RATE) + def _process_frame(self, raw: bytes, amplified: bytes) -> Segment | None: + is_speech = self._vad.is_speech(amplified, SAMPLE_RATE) self._frame_index += 1 if not self._triggered: - self._ring.append((frame, is_speech)) + self._ring.append((raw, is_speech)) voiced = sum(1 for _, s in self._ring if s) if len(self._ring) == self._ring.maxlen and voiced >= TRIGGER_RATIO * self._ring.maxlen: self._triggered = True @@ -114,7 +116,7 @@ class SpeechSegmenter: self._ring.clear() return None - self._segment.extend(frame) + self._segment.extend(raw) if is_speech: self._speech_frames += 1 self._silence_frames = 0