diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..b247af0 --- /dev/null +++ b/.env.example @@ -0,0 +1,9 @@ +# Code langue ISO forcé pour la transcription ("fr", "en"...). +# Laisser vide pour la détection automatique de la langue. +ASR_LANGUAGE= + +# Clé API transmise au moteur ASR (sans importance en local). +ASR_API_KEY=sk-local + +# Jeton Hugging Face, seulement si le téléchargement du modèle l'exige. +HF_TOKEN= diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..381da27 --- /dev/null +++ b/.gitignore @@ -0,0 +1,4 @@ +.env +__pycache__/ +*.pyc +*.db diff --git a/Caddyfile b/Caddyfile new file mode 100644 index 0000000..94fa285 --- /dev/null +++ b/Caddyfile @@ -0,0 +1,14 @@ +{ + # Certificats auto-signés locaux : indispensable pour que le navigateur + # (surtout sur téléphone) autorise l'accès au micro (getUserMedia). + local_certs +} + +:443 { + tls internal + reverse_proxy app:8000 +} + +:80 { + redir https://{host}{uri} permanent +} diff --git a/README.md b/README.md new file mode 100644 index 0000000..c77a8c6 --- /dev/null +++ b/README.md @@ -0,0 +1,81 @@ +# 🎙️ LiveFlow + +Transcription de réunions **en direct**, **100 % locale**, dans Docker. +Capture du micro depuis un navigateur (PC ou téléphone), transcription par +**Qwen3-ASR** sur GPU NVIDIA, affichage du texte au fil de l'eau, historique +et export (TXT, Markdown, SRT, JSON). + +Architecture détaillée : voir [ARCHITECTURE.md](ARCHITECTURE.md) (proposition 2 retenue). + +``` +Navigateur (micro, HTTPS/WebSocket) + → Caddy (TLS local) + → app FastAPI (VAD + segments + SQLite + interface web) + → conteneur ASR Qwen3-ASR via vLLM (API compatible OpenAI) +``` + +## Prérequis + +- Docker + Docker Compose +- GPU NVIDIA avec pilotes récents et le + [NVIDIA Container Toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html) + (`nvidia-ctk runtime configure --runtime=docker`) +- ~6 Go de VRAM libres pour Qwen3-ASR-1.7B (ajuster `--gpu-memory-utilization`) + +## Démarrage + +```bash +docker compose up -d --build +``` + +Au premier lancement, le conteneur `asr` télécharge le modèle depuis +Hugging Face (quelques Go, mis en cache dans un volume). Suivre avec : + +```bash +docker compose logs -f asr +``` + +Puis ouvrir **https://\** depuis un PC ou un téléphone du +réseau local, accepter le certificat auto-signé, autoriser le micro, et +cliquer sur **Démarrer**. + +> ⚠️ Le HTTPS est obligatoire : les navigateurs n'autorisent l'accès au micro +> (`getUserMedia`) qu'en HTTPS (ou sur `localhost`). Caddy génère +> automatiquement un certificat local auto-signé. Pour supprimer +> l'avertissement du navigateur, vous pouvez installer la CA racine de Caddy +> sur vos appareils : `docker compose cp caddy:/data/caddy/pki/authorities/local/root.crt .` + +## Configuration + +Variables d'environnement (fichier `.env` à la racine, voir `.env.example`) : + +| Variable | Défaut | Rôle | +|---|---|---| +| `ASR_LANGUAGE` | *(vide)* | Code langue forcé (`fr`, `en`…) ; vide = détection auto | +| `ASR_API_KEY` | `sk-local` | Clé envoyée au moteur ASR (inutile en local) | +| `HF_TOKEN` | *(vide)* | Jeton Hugging Face si nécessaire au téléchargement | + +## Changer de moteur de transcription + +L'app ne parle au moteur que via l'API OpenAI (`/v1/audio/transcriptions`). +Pour changer de moteur, remplacer le service `asr` dans `docker-compose.yml` +et ajuster `ASR_MODEL`. Exemples : + +```yaml +# CPU uniquement — Parakeet-TDT-0.6B-v3 (rapide, 25 langues dont FR) +asr: + image: ghcr.io/groxaxo/parakeet-tdt-0.6b-v3-fastapi-openai:latest + +# Écosystème Whisper — Speaches (faster-whisper, CPU/GPU) +asr: + image: ghcr.io/speaches-ai/speaches:latest-cuda +``` + +## Récupérer une transcription + +Depuis l'interface (boutons TXT / MD / SRT / JSON / Copier) ou en direct via l'API : + +```bash +curl -k https:///api/meetings # liste des réunions +curl -k https:///api/meetings/1/export?format=txt # export texte +``` diff --git a/app/Dockerfile b/app/Dockerfile new file mode 100644 index 0000000..8a02a6d --- /dev/null +++ b/app/Dockerfile @@ -0,0 +1,11 @@ +FROM python:3.12-slim + +WORKDIR /srv + +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY . . + +EXPOSE 8000 +CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/app/main.py b/app/main.py new file mode 100644 index 0000000..9ff84d5 --- /dev/null +++ b/app/main.py @@ -0,0 +1,247 @@ +import asyncio +import io +import json +import os +import wave +from contextlib import asynccontextmanager +from datetime import datetime, timezone + +import aiosqlite +import httpx +from fastapi import FastAPI, HTTPException, WebSocket, WebSocketDisconnect +from fastapi.responses import JSONResponse, Response +from fastapi.staticfiles import StaticFiles + +from segmenter import SAMPLE_RATE, Segment, SpeechSegmenter + +DB_PATH = os.environ.get("DB_PATH", "/data/liveflow.db") +ASR_BASE_URL = os.environ.get("ASR_BASE_URL", "http://asr:8000/v1").rstrip("/") +ASR_MODEL = os.environ.get("ASR_MODEL", "Qwen/Qwen3-ASR-1.7B") +ASR_API_KEY = os.environ.get("ASR_API_KEY", "sk-local") +ASR_LANGUAGE = os.environ.get("ASR_LANGUAGE", "").strip() + +db: aiosqlite.Connection | None = None +http: httpx.AsyncClient | None = None + + +@asynccontextmanager +async def lifespan(app: FastAPI): + global db, http + os.makedirs(os.path.dirname(DB_PATH), exist_ok=True) + db = await aiosqlite.connect(DB_PATH) + db.row_factory = aiosqlite.Row + await db.executescript( + """ + CREATE TABLE IF NOT EXISTS meetings ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + title TEXT NOT NULL, + created_at TEXT NOT NULL + ); + CREATE TABLE IF NOT EXISTS segments ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + meeting_id INTEGER NOT NULL REFERENCES meetings(id) ON DELETE CASCADE, + t0 REAL NOT NULL, + t1 REAL NOT NULL, + text TEXT NOT NULL + ); + """ + ) + await db.execute("PRAGMA foreign_keys = ON") + await db.commit() + http = httpx.AsyncClient(timeout=120) + yield + await http.aclose() + await db.close() + + +app = FastAPI(title="LiveFlow", lifespan=lifespan) + + +def pcm_to_wav(pcm: bytes) -> bytes: + buf = io.BytesIO() + with wave.open(buf, "wb") as w: + w.setnchannels(1) + w.setsampwidth(2) + w.setframerate(SAMPLE_RATE) + w.writeframes(pcm) + return buf.getvalue() + + +async def transcribe(pcm: bytes) -> str: + data = {"model": ASR_MODEL} + if ASR_LANGUAGE: + data["language"] = ASR_LANGUAGE + resp = await http.post( + f"{ASR_BASE_URL}/audio/transcriptions", + headers={"Authorization": f"Bearer {ASR_API_KEY}"}, + data=data, + files={"file": ("segment.wav", pcm_to_wav(pcm), "audio/wav")}, + ) + resp.raise_for_status() + return resp.json().get("text", "").strip() + + +# ---------------------------------------------------------------- WebSocket + +@app.websocket("/ws") +async def ws_transcribe(ws: WebSocket): + await ws.accept() + + # Premier message : {"type": "start", "title": "..."} + try: + start = json.loads(await ws.receive_text()) + assert start.get("type") == "start" + except Exception: + await ws.close(code=4000) + return + + title = (start.get("title") or "").strip() or datetime.now().strftime("Réunion du %d/%m/%Y %H:%M") + cur = await db.execute( + "INSERT INTO meetings (title, created_at) VALUES (?, ?)", + (title, datetime.now(timezone.utc).isoformat()), + ) + meeting_id = cur.lastrowid + await db.commit() + await ws.send_json({"type": "ready", "meeting_id": meeting_id, "title": title}) + + segmenter = SpeechSegmenter() + queue: asyncio.Queue[Segment | None] = asyncio.Queue() + + async def worker(): + """Transcrit les segments dans l'ordre et pousse le texte au client.""" + while True: + seg = await queue.get() + if seg is None: + return + try: + text = await transcribe(seg.pcm) + except Exception as exc: + await ws.send_json({"type": "error", "message": f"Transcription échouée : {exc}"}) + continue + if not text: + continue + await db.execute( + "INSERT INTO segments (meeting_id, t0, t1, text) VALUES (?, ?, ?, ?)", + (meeting_id, seg.t0, seg.t1, text), + ) + await db.commit() + await ws.send_json({"type": "segment", "t0": seg.t0, "t1": seg.t1, "text": text}) + + worker_task = asyncio.create_task(worker()) + try: + while True: + msg = await ws.receive() + if msg["type"] == "websocket.disconnect": + break + if msg.get("bytes") is not None: + for seg in segmenter.feed(msg["bytes"]): + queue.put_nowait(seg) + elif msg.get("text"): + control = json.loads(msg["text"]) + if control.get("type") == "stop": + if (last := segmenter.flush()) is not None: + queue.put_nowait(last) + queue.put_nowait(None) + await worker_task + worker_task = None + await ws.send_json({"type": "done", "meeting_id": meeting_id}) + break + except WebSocketDisconnect: + pass + finally: + if worker_task is not None: + # Déconnexion brutale : on transcrit quand même ce qui restait. + if (last := segmenter.flush()) is not None: + queue.put_nowait(last) + queue.put_nowait(None) + try: + await worker_task + except Exception: + pass + + +# --------------------------------------------------------------------- API + +@app.get("/api/meetings") +async def list_meetings(): + rows = await db.execute_fetchall( + """ + SELECT m.id, m.title, m.created_at, COUNT(s.id) AS segments, + COALESCE(MAX(s.t1), 0) AS duration + FROM meetings m LEFT JOIN segments s ON s.meeting_id = m.id + GROUP BY m.id ORDER BY m.id DESC + """ + ) + return [dict(r) for r in rows] + + +async def get_meeting_or_404(meeting_id: int) -> dict: + cur = await db.execute("SELECT * FROM meetings WHERE id = ?", (meeting_id,)) + row = await cur.fetchone() + if row is None: + raise HTTPException(404, "Réunion introuvable") + return dict(row) + + +@app.get("/api/meetings/{meeting_id}") +async def get_meeting(meeting_id: int): + meeting = await get_meeting_or_404(meeting_id) + rows = await db.execute_fetchall( + "SELECT t0, t1, text FROM segments WHERE meeting_id = ? ORDER BY id", (meeting_id,) + ) + meeting["segments"] = [dict(r) for r in rows] + return meeting + + +@app.delete("/api/meetings/{meeting_id}") +async def delete_meeting(meeting_id: int): + await get_meeting_or_404(meeting_id) + await db.execute("DELETE FROM meetings WHERE id = ?", (meeting_id,)) + await db.commit() + return JSONResponse({"ok": True}) + + +def fmt_ts(seconds: float, srt: bool = False) -> str: + h, rem = divmod(int(seconds), 3600) + m, s = divmod(rem, 60) + if srt: + ms = int(round((seconds - int(seconds)) * 1000)) + return f"{h:02}:{m:02}:{s:02},{ms:03}" + return f"{h:02}:{m:02}:{s:02}" + + +@app.get("/api/meetings/{meeting_id}/export") +async def export_meeting(meeting_id: int, format: str = "txt"): + meeting = await get_meeting(meeting_id) + segs = meeting["segments"] + title = meeting["title"] + # Les en-têtes HTTP n'acceptent que l'ASCII : on translittère le titre. + safe = "".join( + c if c.isascii() and (c.isalnum() or c in " -_") else "_" for c in title + ).strip() or "reunion" + + if format == "json": + body, mime, ext = json.dumps(meeting, ensure_ascii=False, indent=2), "application/json", "json" + elif format == "md": + lines = [f"# {title}", ""] + lines += [f"**[{fmt_ts(s['t0'])}]** {s['text']}" for s in segs] + body, mime, ext = "\n\n".join(lines) + "\n", "text/markdown", "md" + elif format == "srt": + blocks = [ + f"{i}\n{fmt_ts(s['t0'], srt=True)} --> {fmt_ts(s['t1'], srt=True)}\n{s['text']}" + for i, s in enumerate(segs, 1) + ] + body, mime, ext = "\n\n".join(blocks) + "\n", "application/x-subrip", "srt" + elif format == "txt": + body, mime, ext = "\n".join(s["text"] for s in segs) + "\n", "text/plain", "txt" + else: + raise HTTPException(400, "Format inconnu (txt, md, srt, json)") + + return Response( + content=body, + media_type=f"{mime}; charset=utf-8", + headers={"Content-Disposition": f'attachment; filename="{safe}.{ext}"'}, + ) + + +app.mount("/", StaticFiles(directory="static", html=True), name="static") diff --git a/app/requirements.txt b/app/requirements.txt new file mode 100644 index 0000000..f018be3 --- /dev/null +++ b/app/requirements.txt @@ -0,0 +1,5 @@ +fastapi~=0.115 +uvicorn[standard]~=0.34 +httpx~=0.28 +aiosqlite~=0.21 +webrtcvad-wheels~=2.0 diff --git a/app/segmenter.py b/app/segmenter.py new file mode 100644 index 0000000..4c8bb89 --- /dev/null +++ b/app/segmenter.py @@ -0,0 +1,100 @@ +"""Découpage du flux micro en segments de parole via WebRTC VAD. + +Le client envoie du PCM 16 bits mono 16 kHz. On analyse des trames de 30 ms : +un segment démarre quand la parole domine la fenêtre récente (avec un +pré-roll pour ne pas couper le début de phrase) et se termine après un +silence prolongé ou une durée maximale. +""" + +from collections import deque +from dataclasses import dataclass + +import webrtcvad + +SAMPLE_RATE = 16000 +FRAME_MS = 30 +FRAME_BYTES = SAMPLE_RATE * FRAME_MS // 1000 * 2 # 960 octets + +VAD_AGGRESSIVENESS = 2 +PREROLL_FRAMES = 10 # 300 ms conservées avant le déclenchement +TRIGGER_RATIO = 0.6 # part de trames "parole" du pré-roll pour démarrer +SILENCE_END_MS = 700 # silence qui clôt un segment +MIN_SPEECH_MS = 300 # en dessous, le segment est ignoré (bruit) +MAX_SEGMENT_S = 25 # coupe forcée pour garder une latence raisonnable + + +@dataclass +class Segment: + pcm: bytes + t0: float # secondes depuis le début de la réunion + t1: float + + +class SpeechSegmenter: + def __init__(self): + self._vad = webrtcvad.Vad(VAD_AGGRESSIVENESS) + self._pending = bytearray() + self._ring: deque[tuple[bytes, bool]] = deque(maxlen=PREROLL_FRAMES) + self._frame_index = 0 + self._triggered = False + self._segment = bytearray() + self._segment_start_frame = 0 + self._silence_frames = 0 + self._speech_frames = 0 + + def feed(self, data: bytes) -> list[Segment]: + """Ajoute de l'audio brut et renvoie les segments terminés.""" + self._pending.extend(data) + segments = [] + while len(self._pending) >= FRAME_BYTES: + frame = bytes(self._pending[:FRAME_BYTES]) + del self._pending[:FRAME_BYTES] + seg = self._process_frame(frame) + if seg is not None: + segments.append(seg) + return segments + + def flush(self) -> Segment | None: + """Clôt le segment en cours (fin d'enregistrement).""" + seg = self._finish_segment() if self._triggered else None + self._ring.clear() + self._pending.clear() + return seg + + def _process_frame(self, frame: bytes) -> Segment | None: + is_speech = self._vad.is_speech(frame, SAMPLE_RATE) + self._frame_index += 1 + + if not self._triggered: + self._ring.append((frame, is_speech)) + voiced = sum(1 for _, s in self._ring if s) + if len(self._ring) == self._ring.maxlen and voiced >= TRIGGER_RATIO * self._ring.maxlen: + self._triggered = True + self._segment_start_frame = self._frame_index - len(self._ring) + self._segment = bytearray(b"".join(f for f, _ in self._ring)) + self._speech_frames = voiced + self._silence_frames = 0 + self._ring.clear() + return None + + self._segment.extend(frame) + if is_speech: + self._speech_frames += 1 + self._silence_frames = 0 + else: + self._silence_frames += 1 + + too_long = len(self._segment) >= MAX_SEGMENT_S * SAMPLE_RATE * 2 + ended = self._silence_frames * FRAME_MS >= SILENCE_END_MS + if ended or too_long: + return self._finish_segment() + return None + + def _finish_segment(self) -> Segment | None: + self._triggered = False + seg, self._segment = self._segment, bytearray() + if self._speech_frames * FRAME_MS < MIN_SPEECH_MS: + return None + t0 = self._segment_start_frame * FRAME_MS / 1000 + t1 = self._frame_index * FRAME_MS / 1000 + return Segment(pcm=bytes(seg), t0=t0, t1=t1) diff --git a/app/static/app.js b/app/static/app.js new file mode 100644 index 0000000..d824b90 --- /dev/null +++ b/app/static/app.js @@ -0,0 +1,208 @@ +const $ = (id) => document.getElementById(id); + +const state = { + recording: false, + ws: null, + audioContext: null, + workletNode: null, + mediaStream: null, + sendBuffer: [], + sendBufferSamples: 0, + timerInterval: null, + startedAt: null, + currentMeetingId: null, +}; + +const BATCH_SAMPLES = 4096; // ~256 ms de PCM 16 kHz par message WebSocket + +// ----------------------------------------------------------- enregistrement + +async function startRecording() { + try { + state.mediaStream = await navigator.mediaDevices.getUserMedia({ + audio: { echoCancellation: true, noiseSuppression: true, autoGainControl: true }, + }); + } catch (err) { + alert("Accès au micro refusé ou indisponible.\nVérifiez que la page est bien servie en HTTPS.\n" + err); + return; + } + + const proto = location.protocol === 'https:' ? 'wss' : 'ws'; + state.ws = new WebSocket(`${proto}://${location.host}/ws`); + state.ws.onopen = () => state.ws.send(JSON.stringify({ type: 'start', title: $('title').value })); + state.ws.onmessage = onServerMessage; + state.ws.onclose = () => { if (state.recording) stopRecording(true); }; + + state.audioContext = new AudioContext(); + await state.audioContext.audioWorklet.addModule('worklet.js'); + const source = state.audioContext.createMediaStreamSource(state.mediaStream); + state.workletNode = new AudioWorkletNode(state.audioContext, 'pcm-downsampler'); + state.workletNode.port.onmessage = (e) => queuePcm(new Int16Array(e.data)); + source.connect(state.workletNode); + + state.recording = true; + state.startedAt = Date.now(); + state.timerInterval = setInterval(updateTimer, 500); + $('record-btn').textContent = '■ Arrêter'; + $('record-btn').classList.add('recording'); + $('title').disabled = true; + setStatus('rec', 'Enregistrement…'); + clearTranscript(); +} + +function queuePcm(samples) { + if (!state.recording || !state.ws || state.ws.readyState !== WebSocket.OPEN) return; + state.sendBuffer.push(samples); + state.sendBufferSamples += samples.length; + if (state.sendBufferSamples >= BATCH_SAMPLES) flushPcm(); +} + +function flushPcm() { + if (!state.sendBufferSamples) return; + const out = new Int16Array(state.sendBufferSamples); + let off = 0; + for (const chunk of state.sendBuffer) { out.set(chunk, off); off += chunk.length; } + state.sendBuffer = []; + state.sendBufferSamples = 0; + if (state.ws && state.ws.readyState === WebSocket.OPEN) state.ws.send(out.buffer); +} + +function stopRecording(abrupt = false) { + state.recording = false; + clearInterval(state.timerInterval); + if (state.workletNode) state.workletNode.disconnect(); + if (state.mediaStream) state.mediaStream.getTracks().forEach((t) => t.stop()); + if (state.audioContext) state.audioContext.close(); + + if (!abrupt && state.ws && state.ws.readyState === WebSocket.OPEN) { + flushPcm(); + state.ws.send(JSON.stringify({ type: 'stop' })); + setStatus('busy', 'Finalisation…'); + } else { + setStatus('idle', 'Prêt'); + } + + $('record-btn').textContent = '● Démarrer'; + $('record-btn').classList.remove('recording'); + $('title').disabled = false; +} + +function onServerMessage(event) { + const msg = JSON.parse(event.data); + if (msg.type === 'ready') { + state.currentMeetingId = msg.meeting_id; + $('transcript-title').textContent = msg.title; + showExportBar(msg.meeting_id); + } else if (msg.type === 'segment') { + appendSegment(msg); + if (state.recording) setStatus('rec', 'Enregistrement…'); + } else if (msg.type === 'error') { + setStatus('error', 'Erreur ASR'); + console.error(msg.message); + } else if (msg.type === 'done') { + setStatus('idle', 'Terminé ✓'); + state.ws.close(); + state.ws = null; + loadMeetings(); + } +} + +// ----------------------------------------------------------------- affichage + +function setStatus(cls, text) { + const el = $('status'); + el.className = 'badge ' + cls; + el.textContent = text; +} + +function updateTimer() { + const s = Math.floor((Date.now() - state.startedAt) / 1000); + $('timer').textContent = + String(Math.floor(s / 60)).padStart(2, '0') + ':' + String(s % 60).padStart(2, '0'); +} + +function fmtTs(seconds) { + const s = Math.floor(seconds); + const h = Math.floor(s / 3600), m = Math.floor((s % 3600) / 60), sec = s % 60; + return (h ? String(h).padStart(2, '0') + ':' : '') + + String(m).padStart(2, '0') + ':' + String(sec).padStart(2, '0'); +} + +function clearTranscript() { + $('transcript').innerHTML = ''; +} + +function appendSegment(seg) { + const div = document.createElement('div'); + div.className = 'segment'; + div.innerHTML = `${fmtTs(seg.t0)}`; + div.querySelector('.text').textContent = seg.text; + $('transcript').appendChild(div); + $('transcript').scrollTop = $('transcript').scrollHeight; +} + +function showExportBar(meetingId) { + $('export-bar').classList.remove('hidden'); + for (const fmt of ['txt', 'md', 'srt', 'json']) { + $('export-' + fmt).href = `/api/meetings/${meetingId}/export?format=${fmt}`; + } +} + +// ----------------------------------------------------------------- réunions + +async function loadMeetings() { + const meetings = await (await fetch('/api/meetings')).json(); + const ul = $('meeting-list'); + ul.innerHTML = ''; + for (const m of meetings) { + const li = document.createElement('li'); + const date = new Date(m.created_at).toLocaleString('fr-FR', { dateStyle: 'short', timeStyle: 'short' }); + li.innerHTML = `${date} · ${fmtTs(m.duration)}`; + li.querySelector('.m-title').textContent = m.title; + li.onclick = () => openMeeting(m.id); + if (m.id === state.currentMeetingId) li.classList.add('active'); + ul.appendChild(li); + } +} + +async function openMeeting(id) { + if (state.recording) return; + const meeting = await (await fetch(`/api/meetings/${id}`)).json(); + state.currentMeetingId = id; + $('transcript-title').textContent = meeting.title; + clearTranscript(); + if (meeting.segments.length === 0) { + $('transcript').innerHTML = '

Aucun texte pour cette réunion.

'; + } else { + meeting.segments.forEach(appendSegment); + } + showExportBar(id); + loadMeetings(); +} + +async function deleteCurrentMeeting() { + if (!state.currentMeetingId || state.recording) return; + if (!confirm('Supprimer définitivement cette réunion et sa transcription ?')) return; + await fetch(`/api/meetings/${state.currentMeetingId}`, { method: 'DELETE' }); + state.currentMeetingId = null; + $('transcript-title').textContent = 'Transcription'; + clearTranscript(); + $('transcript').innerHTML = '

Réunion supprimée.

'; + $('export-bar').classList.add('hidden'); + loadMeetings(); +} + +async function copyTranscript() { + const text = [...document.querySelectorAll('#transcript .segment .text')] + .map((el) => el.textContent).join('\n'); + await navigator.clipboard.writeText(text); + $('copy-btn').textContent = '✓ Copié'; + setTimeout(() => ($('copy-btn').textContent = '📋 Copier'), 1500); +} + +// --------------------------------------------------------------------- init + +$('record-btn').onclick = () => (state.recording ? stopRecording() : startRecording()); +$('copy-btn').onclick = copyTranscript; +$('delete-btn').onclick = deleteCurrentMeeting; +loadMeetings(); diff --git a/app/static/index.html b/app/static/index.html new file mode 100644 index 0000000..0339993 --- /dev/null +++ b/app/static/index.html @@ -0,0 +1,49 @@ + + + + + + LiveFlow — Transcription de réunions + + + +
+

🎙️ LiveFlow

+ Prêt +
+ +
+ + +
+
+ + + 00:00 +
+ +
+
+

Transcription

+ +
+
+

Appuyez sur « Démarrer » et parlez : le texte apparaît ici au fil de l'eau.

+
+
+
+
+ + + + diff --git a/app/static/style.css b/app/static/style.css new file mode 100644 index 0000000..6708f4d --- /dev/null +++ b/app/static/style.css @@ -0,0 +1,141 @@ +* { box-sizing: border-box; margin: 0; } + +:root { + --bg: #0f1117; + --panel: #181b24; + --panel-2: #1f2330; + --text: #e8eaf0; + --muted: #8b91a3; + --accent: #4f7cff; + --rec: #e5484d; +} + +body { + background: var(--bg); + color: var(--text); + font-family: system-ui, -apple-system, "Segoe UI", Roboto, sans-serif; + height: 100vh; + display: flex; + flex-direction: column; +} + +header { + display: flex; + align-items: center; + gap: 12px; + padding: 14px 20px; + background: var(--panel); + border-bottom: 1px solid #262b3a; +} + +header h1 { font-size: 1.2rem; } + +.badge { + font-size: 0.78rem; + padding: 3px 10px; + border-radius: 999px; + background: var(--panel-2); + color: var(--muted); +} +.badge.rec { background: #3a181a; color: #ff8589; } +.badge.busy { background: #2a2410; color: #ffd166; } +.badge.error { background: #3a181a; color: #ff8589; } + +.layout { display: flex; flex: 1; min-height: 0; } + +aside { + width: 260px; + background: var(--panel); + border-right: 1px solid #262b3a; + padding: 16px; + overflow-y: auto; +} +aside h2 { font-size: 0.85rem; text-transform: uppercase; color: var(--muted); margin-bottom: 10px; } +aside ul { list-style: none; padding: 0; } +aside li { + padding: 8px 10px; + border-radius: 8px; + cursor: pointer; + display: flex; + flex-direction: column; + gap: 2px; +} +aside li:hover { background: var(--panel-2); } +aside li.active { background: #20294a; } +.m-title { font-size: 0.9rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.m-meta { font-size: 0.74rem; color: var(--muted); } + +main { flex: 1; display: flex; flex-direction: column; min-width: 0; padding: 20px; gap: 16px; } + +.controls { display: flex; gap: 12px; align-items: center; } +.controls input { + flex: 1; + background: var(--panel-2); + border: 1px solid #2c3245; + color: var(--text); + padding: 10px 14px; + border-radius: 10px; + font-size: 0.95rem; +} +.controls input:focus { outline: none; border-color: var(--accent); } + +button, .controls a { + border: none; + border-radius: 10px; + padding: 10px 18px; + font-size: 0.95rem; + cursor: pointer; + background: var(--panel-2); + color: var(--text); +} +button.record { background: var(--accent); font-weight: 600; min-width: 130px; } +button.record.recording { background: var(--rec); } +#timer { font-variant-numeric: tabular-nums; color: var(--muted); min-width: 48px; } + +#transcript-panel { + flex: 1; + display: flex; + flex-direction: column; + background: var(--panel); + border: 1px solid #262b3a; + border-radius: 14px; + min-height: 0; +} +.panel-head { + display: flex; + align-items: center; + justify-content: space-between; + gap: 10px; + padding: 14px 18px; + border-bottom: 1px solid #262b3a; +} +.panel-head h2 { font-size: 1rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +#export-bar { display: flex; gap: 8px; align-items: center; } +#export-bar a, #export-bar button { + font-size: 0.8rem; + padding: 6px 10px; + border-radius: 8px; + background: var(--panel-2); + color: var(--text); + text-decoration: none; +} +#export-bar.hidden { display: none; } + +#transcript { flex: 1; overflow-y: auto; padding: 18px; } +.placeholder { color: var(--muted); font-style: italic; } +.segment { display: flex; gap: 12px; margin-bottom: 12px; line-height: 1.55; } +.segment .ts { + color: var(--muted); + font-size: 0.78rem; + font-variant-numeric: tabular-nums; + padding-top: 3px; + flex-shrink: 0; +} + +@media (max-width: 720px) { + .layout { flex-direction: column; } + aside { width: 100%; max-height: 30vh; border-right: none; border-bottom: 1px solid #262b3a; } + .controls { flex-wrap: wrap; } + .controls input { width: 100%; flex: none; } +} diff --git a/app/static/worklet.js b/app/static/worklet.js new file mode 100644 index 0000000..1db2637 --- /dev/null +++ b/app/static/worklet.js @@ -0,0 +1,42 @@ +// Rééchantillonne le micro (fréquence native du navigateur, ex. 48 kHz) +// vers du PCM Int16 mono 16 kHz, attendu par le serveur. +class PCMDownsampler extends AudioWorkletProcessor { + constructor() { + super(); + this.targetRate = 16000; + this.ratio = sampleRate / this.targetRate; + this.buffer = new Float32Array(0); + this.readPos = 0; + } + + process(inputs) { + const input = inputs[0] && inputs[0][0]; + if (!input) return true; + + const buf = new Float32Array(this.buffer.length + input.length); + buf.set(this.buffer); + buf.set(input, this.buffer.length); + + const outLen = Math.floor((buf.length - 1 - this.readPos) / this.ratio); + if (outLen > 0) { + const out = new Int16Array(outLen); + let pos = this.readPos; + for (let i = 0; i < outLen; i++) { + const i0 = Math.floor(pos); + const frac = pos - i0; + const s = buf[i0] * (1 - frac) + buf[i0 + 1] * frac; + out[i] = Math.max(-32768, Math.min(32767, Math.round(s * 32767))); + pos += this.ratio; + } + this.readPos = pos; + this.port.postMessage(out.buffer, [out.buffer]); + } + + const keep = Math.floor(this.readPos); + this.buffer = buf.slice(keep); + this.readPos -= keep; + return true; + } +} + +registerProcessor('pcm-downsampler', PCMDownsampler); diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..575bf36 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,53 @@ +services: + caddy: + image: caddy:2-alpine + restart: unless-stopped + ports: + - "80:80" + - "443:443" + volumes: + - ./Caddyfile:/etc/caddy/Caddyfile:ro + - caddy-data:/data + - caddy-config:/config + depends_on: + - app + + app: + build: ./app + restart: unless-stopped + environment: + - ASR_BASE_URL=http://asr:8000/v1 + - ASR_MODEL=Qwen/Qwen3-ASR-1.7B + - ASR_API_KEY=${ASR_API_KEY:-sk-local} + # Code langue ISO ("fr", "en"...) ou vide pour la détection automatique + - ASR_LANGUAGE=${ASR_LANGUAGE:-} + - DB_PATH=/data/liveflow.db + volumes: + - liveflow-data:/data + depends_on: + - asr + + asr: + image: vllm/vllm-openai:latest + restart: unless-stopped + command: >- + --model Qwen/Qwen3-ASR-1.7B + --gpu-memory-utilization 0.70 + environment: + - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} + volumes: + - hf-cache:/root/.cache/huggingface + ipc: host + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: 1 + capabilities: [gpu] + +volumes: + caddy-data: + caddy-config: + liveflow-data: + hf-cache: