Filtrage des interjections chinoises de Qwen3-ASR (嗯, 咳咳...)

https://claude.ai/code/session_01YHMp3EKzr4s6o8w1ygxuUe
This commit is contained in:
Claude committed 2026-06-12 17:42:33 +00:00
1 parent 9b083579e1
commit 21866ec334
1 file changed
+20
+20
View File
@@ -5,6 +5,7 @@ import hmac
import io
import json
import os
import re
import secrets
import time
import wave
@@ -26,6 +27,21 @@ ASR_MODEL = os.environ.get("ASR_MODEL", "Qwen/Qwen3-ASR-1.7B")
ASR_API_KEY = os.environ.get("ASR_API_KEY", "sk-local")
ASR_LANGUAGE = os.environ.get("ASR_LANGUAGE", "").strip()
# Qwen3-ASR transcrit les sons non verbaux (hmm, toux...) en interjections
# chinoises et vLLM ignore le paramètre language : si la langue configurée
# n'est pas une langue CJK, on retire ces caractères des transcriptions.
CJK_RE = re.compile(r"[ -〿㐀-䶿一-鿿豈-﫿＀-￯]+")
STRIP_CJK = ASR_LANGUAGE.lower() not in ("", "zh", "ja", "ko", "yue")
def clean_transcript(text: str) -> str:
if STRIP_CJK and CJK_RE.search(text):
text = CJK_RE.sub(" ", text)
text = re.sub(r"\s+", " ", text)
text = text.lstrip(" 。,、!?.,;!?") # ponctuation orpheline en tête
text = text.rstrip(" 。,、!?") # ponctuation chinoise en queue
return text.strip()
LIVEFLOW_USER = os.environ.get("LIVEFLOW_USER", "admin")
LIVEFLOW_PASSWORD = os.environ.get("LIVEFLOW_PASSWORD", "admin")
SESSION_TTL = 7 * 24 * 3600 # 7 jours
@@ -241,6 +257,10 @@ async def ws_transcribe(ws: WebSocket):
print(f"[réunion {meeting_id}] ERREUR ASR : {exc}", flush=True)
await ws.send_json({"type": "error", "message": f"Transcription échouée : {exc}"})
continue
cleaned = clean_transcript(text)
if cleaned != text:
print(f"[réunion {meeting_id}] filtré (CJK) : {text[:60]!r} -> {cleaned[:60]!r}", flush=True)
text = cleaned
print(f"[réunion {meeting_id}] texte : {text[:80]!r}", flush=True)
if not text:
continue