diff --git a/README.md b/README.md index 1af1ccf..8c0d3fe 100644 --- a/README.md +++ b/README.md @@ -115,6 +115,23 @@ npm run dev # http://localhost:5173 | `web_search` | Recherche web (DuckDuckGo / SearxNG) | désactivé | | `run_shell` | Exécuter une commande **(sensible)** | désactivé | +## Intelligence augmentée + +- **Plan-puis-exécute** : les demandes complexes sont décomposées en 3-5 étapes + affichées dans le fil ; l'agent (ou le moteur code) suit le plan. +- **Auto-critique « Qualité + »** (Configuration → Intelligence) : la réponse + est relue et révisée avant d'être finalisée. +- **Mémoire long-terme (RAG)** : chaque échange est vectorisé (`/api/embed`) + et les souvenirs pertinents des anciennes sessions sont réinjectés en + contexte. Nécessite un modèle d'embedding installé (ex. + `ollama pull nomic-embed-text`) — sinon désactivé silencieusement. +- **Vérification HTML** : liens locaux cassés et balises déséquilibrées sont + détectés après chaque génération ; le moteur code fait une passe + d'auto-correction. +- **Benchmark intégré** (Configuration → Benchmark) : 5 mini-épreuves notées + /100 (appel d'outil, code exécutable, consignes, extraction JSON, format) + pour comparer objectivement tes modèles installés. + ## Tirer le meilleur des petits modèles Loki est conçu pour qu'un modèle local modeste se comporte comme un bon agent : diff --git a/backend/app/agent_config.py b/backend/app/agent_config.py index bdf4aa2..e81dc85 100644 --- a/backend/app/agent_config.py +++ b/backend/app/agent_config.py @@ -60,6 +60,10 @@ PROFILE_FIELDS = { "confirm_shell", "think", "code_model", + "plan_mode", + "self_review", + "rag_enabled", + "embed_model", *GENERATION_FIELDS, } @@ -95,6 +99,13 @@ DEFAULT_CONFIG: dict = { # Modèle utilisé par le moteur code : "auto" = meilleur modèle code installé # (qwen-coder, deepseek-coder…), sinon le modèle de chat courant. "code_model": "auto", + # Plan-puis-exécute : décompose les demandes complexes en étapes. + "plan_mode": True, + # Auto-critique : une passe de relecture/révision avant la réponse finale. + "self_review": False, + # Mémoire long-terme (RAG) entre sessions, via un modèle d'embedding. + "rag_enabled": True, + "embed_model": "auto", } diff --git a/backend/app/bench.py b/backend/app/bench.py new file mode 100644 index 0000000..1d0b12b --- /dev/null +++ b/backend/app/bench.py @@ -0,0 +1,206 @@ +"""Benchmark intégré : évalue objectivement chaque modèle installé. + +Cinq mini-épreuves (~30-60 s au total) qui mesurent ce qui compte pour Loki : +appel d'outil, code exécutable, respect des consignes, extraction JSON, +respect d'un format. Score /100, stocké en base et affiché dans l'UI. +""" +from __future__ import annotations + +import json +import re +import subprocess +import sys +import tempfile +import time +from typing import AsyncIterator + +import httpx + +from . import db +from .ollama_client import ollama + +BENCH_KEY = "bench" # config[bench] = {model: {score, details, at}} + + +async def _ask(model: str, prompt: str, *, system: str = "", + tools: list | None = None, num_predict: int = 400) -> dict: + """Un appel modèle ; renvoie {text, tool_calls}.""" + messages = [] + if system: + messages.append({"role": "system", "content": system}) + messages.append({"role": "user", "content": prompt}) + text, calls = "", [] + async for chunk in ollama.chat( + model, messages, tools=tools, + options={"temperature": 0, "num_predict": num_predict}, stream=True, + ): + msg = chunk.get("message", {}) + text += msg.get("content", "") + if msg.get("tool_calls"): + calls.extend(msg["tool_calls"]) + if chunk.get("done"): + break + return {"text": text.strip(), "tool_calls": calls} + + +def _extract_code(text: str) -> str: + m = re.search(r"```(?:python)?\s*(.*?)```", text, re.S) + return (m.group(1) if m else text).strip() + + +def _run_python(code: str, test: str) -> bool: + """Exécute code+test dans un sous-processus isolé (timeout 8 s).""" + with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f: + f.write(code + "\n" + test) + path = f.name + try: + proc = subprocess.run( + [sys.executable, "-I", path], + capture_output=True, timeout=8, + ) + return proc.returncode == 0 + except (subprocess.SubprocessError, OSError): + return False + + +# ── Les 5 épreuves (score 0-20 chacune) ────────────────────────────────── +async def _task_tool_call(model: str) -> tuple[int, str]: + tools = [{ + "type": "function", + "function": { + "name": "write_file", + "description": "Écrire un fichier", + "parameters": { + "type": "object", + "properties": { + "path": {"type": "string"}, + "content": {"type": "string"}, + }, + "required": ["path", "content"], + }, + }, + }] + try: + r = await _ask( + model, + "Crée le fichier bonjour.txt contenant exactement le texte : salut", + system="Utilise l'outil write_file pour créer le fichier demandé.", + tools=tools, num_predict=200, + ) + except httpx.HTTPStatusError as exc: + if "does not support tools" in exc.response.text.lower(): + return 0, "outils non supportés par ce modèle" + raise + for tc in r["tool_calls"]: + fn = tc.get("function", {}) + if fn.get("name") == "write_file": + args = fn.get("arguments") or {} + if isinstance(args, str): + try: + args = json.loads(args) + except json.JSONDecodeError: + return 8, "appel d'outil aux arguments illisibles" + ok_path = "bonjour" in str(args.get("path", "")).lower() + ok_content = "salut" in str(args.get("content", "")).lower() + score = 10 + 5 * ok_path + 5 * ok_content + return score, "appel d'outil correct" if score == 20 else "appel partiel" + return 0, "aucun appel d'outil émis" + + +async def _task_code(model: str) -> tuple[int, str]: + r = await _ask( + model, + "Écris une fonction Python `somme_pairs(nombres)` qui renvoie la somme " + "des nombres pairs de la liste. Réponds UNIQUEMENT avec le code.", + num_predict=300, + ) + code = _extract_code(r["text"]) + if "def somme_pairs" not in code: + return 0, "fonction absente" + test = ( + "assert somme_pairs([1,2,3,4]) == 6\n" + "assert somme_pairs([]) == 0\n" + "assert somme_pairs([7,9]) == 0\n" + ) + return (20, "code correct (3/3 tests)") if _run_python(code, test) \ + else (6, "code présent mais tests échoués") + + +async def _task_instruction(model: str) -> tuple[int, str]: + r = await _ask( + model, + "Quelle est la capitale de la France ? Réponds en 3 mots maximum.", + num_predict=30, + ) + text = r["text"] + has_answer = "paris" in text.lower() + short = len(text.split()) <= 6 + score = 12 * has_answer + 8 * short + return score, f"réponse « {text[:40]} »" + + +async def _task_json(model: str) -> tuple[int, str]: + r = await _ask( + model, + 'Extrait les informations en JSON strict {"nom": ..., "ville": ...} ' + "depuis : « Marie habite à Lyon ». Réponds UNIQUEMENT avec le JSON.", + num_predict=80, + ) + m = re.search(r"\{.*\}", r["text"], re.S) + if not m: + return 0, "pas de JSON" + try: + data = json.loads(m.group(0)) + except json.JSONDecodeError: + return 5, "JSON invalide" + ok_nom = "marie" in str(data.get("nom", "")).lower() + ok_ville = "lyon" in str(data.get("ville", "")).lower() + return 10 + 5 * ok_nom + 5 * ok_ville, "extraction correcte" \ + if ok_nom and ok_ville else (10 + 5 * ok_nom + 5 * ok_ville, "extraction partielle") + + +async def _task_format(model: str) -> tuple[int, str]: + r = await _ask( + model, + "Liste exactement 3 fruits, un par ligne, chaque ligne préfixée par « - ».", + num_predict=60, + ) + lines = [l for l in r["text"].splitlines() if l.strip().startswith("-")] + if len(lines) == 3: + return 20, "format exact" + if len(lines) >= 2: + return 10, f"{len(lines)} lignes au lieu de 3" + return 0, "format non respecté" + + +TASKS = [ + ("Appel d'outil", _task_tool_call), + ("Code exécutable", _task_code), + ("Consigne courte", _task_instruction), + ("Extraction JSON", _task_json), + ("Respect du format", _task_format), +] + + +async def run_bench(model: str) -> AsyncIterator[dict]: + """Exécute les 5 épreuves en streamant la progression, stocke le score.""" + total = 0 + details = [] + for name, fn in TASKS: + yield {"type": "task_start", "task": name} + try: + score, detail = await fn(model) + except (httpx.HTTPError, OSError) as exc: + score, detail = 0, f"erreur : {str(exc)[:80]}" + total += score + details.append({"task": name, "score": score, "detail": detail}) + yield {"type": "task_done", "task": name, "score": score, "detail": detail} + + results = db.get_config_value(BENCH_KEY) or {} + results[model] = {"score": total, "details": details, "at": time.time()} + db.set_config_value(BENCH_KEY, results) + yield {"type": "done", "score": total, "details": details} + + +def get_scores() -> dict: + return db.get_config_value(BENCH_KEY) or {} diff --git a/backend/app/enhance.py b/backend/app/enhance.py new file mode 100644 index 0000000..bebb9bd --- /dev/null +++ b/backend/app/enhance.py @@ -0,0 +1,105 @@ +"""Boosters de qualité pour petits modèles : plan-puis-exécute et auto-critique. + +- make_plan : décompose une demande complexe en 3-5 étapes courtes. Un petit + modèle qui suit un plan écrit réussit bien mieux qu'en improvisant. +- self_review : une passe de critique éclair sur la réponse, puis une révision + si des défauts sont trouvés (activable : coûte un peu de latence). +""" +from __future__ import annotations + +import logging +import re + +import httpx + +from .ollama_client import ollama + +logger = logging.getLogger(__name__) + +_PLAN_PROMPT = ( + "Découpe la demande en 3 à 5 étapes courtes et concrètes, une par ligne, " + "numérotées « 1. », « 2. »… Pas d'introduction, pas de conclusion, " + "UNIQUEMENT les étapes, en français." +) + +_CRITIQUE_PROMPT = ( + "Tu es un relecteur exigeant. Voici une demande et la réponse d'un " + "assistant. Si la réponse est correcte et complète, réponds exactement " + "PARFAIT. Sinon, liste au plus 3 défauts concrets (erreurs, oublis, " + "incohérences), un par ligne." +) + +_REVISE_PROMPT = ( + "Réécris la réponse en corrigeant les défauts listés. Donne UNIQUEMENT la " + "réponse finale corrigée, sans commentaire sur la révision." +) + + +def needs_plan(message: str) -> bool: + """Une demande assez longue/composée mérite un plan explicite.""" + if len(message) < 120: + return False + connectors = len(re.findall( + r"\b(puis|ensuite|après|avec|ainsi que|et aussi|également)\b", + message, re.I, + )) + return len(message) > 240 or connectors >= 2 + + +async def _ask(model: str, system: str, user: str, *, num_predict: int) -> str: + text = "" + async for chunk in ollama.chat( + model, + [{"role": "system", "content": system}, + {"role": "user", "content": user}], + options={"temperature": 0.2, "num_predict": num_predict}, + stream=True, + ): + text += chunk.get("message", {}).get("content", "") + if chunk.get("done"): + break + return text.strip() + + +async def make_plan(model: str, message: str) -> list[str]: + """Renvoie la liste des étapes (vide si échec — jamais bloquant).""" + try: + raw = await _ask(model, _PLAN_PROMPT, message[:1200], num_predict=220) + except (httpx.HTTPError, OSError) as exc: + logger.warning("Plan impossible : %s", exc) + return [] + steps = [] + for line in raw.splitlines(): + line = line.strip() + m = re.match(r"^\d+[.)]\s*(.+)$", line) + if m: + steps.append(m.group(1).strip()) + return steps[:5] if len(steps) >= 2 else [] + + +async def self_review(model: str, request: str, answer: str) -> str | None: + """Critique puis révise la réponse. None si rien à corriger / échec.""" + if len(answer) < 80: + return None + try: + critique = await _ask( + model, + _CRITIQUE_PROMPT, + f"Demande :\n{request[:800]}\n\nRéponse :\n{answer[:2500]}", + num_predict=180, + ) + if not critique or "PARFAIT" in critique.upper()[:40]: + return None + + revised = await _ask( + model, + _REVISE_PROMPT, + f"Demande :\n{request[:800]}\n\nRéponse initiale :\n{answer[:2500]}" + f"\n\nDéfauts :\n{critique[:600]}", + num_predict=1500, + ) + # Garde-fou : une révision vide ou minuscule ne remplace rien. + return revised if len(revised) > len(answer) // 3 else None + except (httpx.HTTPError, OSError) as exc: + logger.warning("Auto-critique impossible : %s", exc) + return None diff --git a/backend/app/main.py b/backend/app/main.py index de4f301..d4aa9a5 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -12,14 +12,15 @@ from fastapi.middleware.cors import CORSMiddleware from fastapi.responses import FileResponse from fastapi.staticfiles import StaticFiles -from . import coder, db +from . import coder, db, rag from .config import settings -from .routes import chat, config, files, models, sessions, shell, system +from .routes import benchmark, chat, config, files, models, sessions, shell, system @asynccontextmanager async def lifespan(_: FastAPI): db.init_db() + rag.init_table() # Workspace en dépôt git : requis pour les commits du moteur code (Aider). coder.ensure_git(settings.workspace_dir) yield @@ -45,6 +46,7 @@ app.include_router(files.router) app.include_router(config.router) app.include_router(shell.router) app.include_router(system.router) +app.include_router(benchmark.router) @app.get("/api/health") diff --git a/backend/app/ollama_client.py b/backend/app/ollama_client.py index 6fb64e0..afbea21 100644 --- a/backend/app/ollama_client.py +++ b/backend/app/ollama_client.py @@ -65,6 +65,15 @@ class OllamaClient: resp.raise_for_status() return resp.json().get("models", []) + async def embed(self, model: str, texts: list[str]) -> list[list[float]]: + """Vecteurs d'embedding pour une liste de textes (/api/embed).""" + async with httpx.AsyncClient(timeout=30.0, follow_redirects=True) as client: + resp = await client.post( + f"{self.host}/api/embed", json={"model": model, "input": texts} + ) + resp.raise_for_status() + return resp.json().get("embeddings", []) + async def ps(self) -> list[dict]: """Modèles actuellement chargés et leur répartition VRAM/CPU (/api/ps).""" async with httpx.AsyncClient(timeout=5.0, follow_redirects=True) as client: diff --git a/backend/app/rag.py b/backend/app/rag.py new file mode 100644 index 0000000..d736beb --- /dev/null +++ b/backend/app/rag.py @@ -0,0 +1,142 @@ +"""Mémoire long-terme (RAG) : l'agent se souvient des anciennes sessions. + +Chaque échange (question + réponse) est vectorisé via /api/embed d'Ollama et +stocké en SQLite. À chaque nouveau message, on recherche les souvenirs les +plus proches (cosinus) dans les AUTRES sessions et on les injecte en contexte. + +Tout est best-effort : sans modèle d'embedding installé, le RAG se désactive +silencieusement (aucun impact sur le chat). +""" +from __future__ import annotations + +import json +import logging +import math +import time +import uuid + +import httpx + +from . import db +from .ollama_client import ollama + +logger = logging.getLogger(__name__) + +# Modèles d'embedding reconnus, par ordre de préférence. +_EMBED_HINTS = ("nomic-embed", "mxbai-embed", "bge-", "snowflake-arctic-embed", + "all-minilm", "embed") + +_TOP_K = 3 +_MIN_SCORE = 0.45 +_MAX_MEMORIES = 2000 # au-delà, on élague les plus anciens + +_embed_model_cache: dict = {"value": None, "checked_at": 0.0} + + +def init_table() -> None: + with db._LOCK, db._connect() as conn: + conn.execute( + """ + CREATE TABLE IF NOT EXISTS memories ( + id TEXT PRIMARY KEY, + session_id TEXT NOT NULL, + content TEXT NOT NULL, + embedding TEXT NOT NULL, + created_at REAL NOT NULL + ) + """ + ) + + +async def resolve_embed_model(preference: str | None = None) -> str | None: + """Trouve le modèle d'embedding à utiliser (None = RAG indisponible).""" + if preference and preference != "auto": + return preference + + # Cache 60 s pour ne pas marteler /api/tags. + now = time.time() + if now - _embed_model_cache["checked_at"] < 60: + return _embed_model_cache["value"] + + value = None + try: + for m in await ollama.list_models(): + name = (m.get("name") or "").lower() + if any(h in name for h in _EMBED_HINTS): + value = m["name"] + break + except (httpx.HTTPError, OSError): + value = None + + _embed_model_cache.update(value=value, checked_at=now) + return value + + +def _cosine(a: list[float], b: list[float]) -> float: + dot = sum(x * y for x, y in zip(a, b)) + na = math.sqrt(sum(x * x for x in a)) + nb = math.sqrt(sum(x * x for x in b)) + return dot / (na * nb) if na and nb else 0.0 + + +async def index_exchange( + sid: str, user_text: str, assistant_text: str, *, embed_model: str | None +) -> None: + """Indexe un échange terminé (tâche d'arrière-plan, best-effort).""" + model = await resolve_embed_model(embed_model) + if not model: + return + content = f"Q: {user_text[:500]}\nR: {assistant_text[:800]}" + try: + vectors = await ollama.embed(model, [content]) + if not vectors: + return + with db._LOCK, db._connect() as conn: + conn.execute( + "INSERT INTO memories (id, session_id, content, embedding, created_at)" + " VALUES (?, ?, ?, ?, ?)", + (uuid.uuid4().hex, sid, content, + json.dumps(vectors[0]), time.time()), + ) + # Élagage des souvenirs les plus anciens. + conn.execute( + "DELETE FROM memories WHERE id IN (" + " SELECT id FROM memories ORDER BY created_at DESC" + f" LIMIT -1 OFFSET {_MAX_MEMORIES})" + ) + except (httpx.HTTPError, OSError) as exc: + logger.warning("Indexation RAG impossible : %s", exc) + + +async def recall( + sid: str, query: str, *, embed_model: str | None +) -> list[str]: + """Souvenirs pertinents issus des AUTRES sessions (top-k, score minimal).""" + model = await resolve_embed_model(embed_model) + if not model: + return [] + try: + vectors = await ollama.embed(model, [query[:800]]) + if not vectors: + return [] + qvec = vectors[0] + + with db._LOCK, db._connect() as conn: + rows = conn.execute( + "SELECT content, embedding FROM memories WHERE session_id != ?", + (sid,), + ).fetchall() + + scored = [] + for row in rows: + try: + score = _cosine(qvec, json.loads(row["embedding"])) + except (ValueError, TypeError): + continue + if score >= _MIN_SCORE: + scored.append((score, row["content"])) + scored.sort(reverse=True) + return [c for _, c in scored[:_TOP_K]] + except (httpx.HTTPError, OSError) as exc: + logger.warning("Rappel RAG impossible : %s", exc) + return [] diff --git a/backend/app/routes/benchmark.py b/backend/app/routes/benchmark.py new file mode 100644 index 0000000..f4e246a --- /dev/null +++ b/backend/app/routes/benchmark.py @@ -0,0 +1,35 @@ +"""Routes du benchmark de modèles.""" +from __future__ import annotations + +import json + +from fastapi import APIRouter +from fastapi.responses import StreamingResponse +from pydantic import BaseModel + +from .. import bench + +router = APIRouter(prefix="/api/bench", tags=["bench"]) + + +class BenchRequest(BaseModel): + model: str + + +@router.get("") +async def scores() -> dict: + return {"scores": bench.get_scores()} + + +@router.post("") +async def run(req: BenchRequest) -> StreamingResponse: + async def event_stream(): + async for ev in bench.run_bench(req.model): + etype = ev.pop("type") + yield f"event: {etype}\ndata: {json.dumps(ev, ensure_ascii=False)}\n\n" + + return StreamingResponse( + event_stream(), + media_type="text/event-stream", + headers={"Cache-Control": "no-cache, no-transform", "X-Accel-Buffering": "no"}, + ) diff --git a/backend/app/routes/chat.py b/backend/app/routes/chat.py index 9fd480d..4102889 100644 --- a/backend/app/routes/chat.py +++ b/backend/app/routes/chat.py @@ -18,7 +18,8 @@ from fastapi import APIRouter, HTTPException from fastapi.responses import StreamingResponse from pydantic import BaseModel -from .. import agent_config, coder, db, memory, router as msg_router +from .. import agent_config, coder, db, enhance, memory, rag, router as msg_router +from ..tools import check_html, _safe_path from ..agent import run_agent from ..config import settings @@ -36,33 +37,92 @@ def _sse(event: str, data: dict) -> str: return f"event: {event}\ndata: {json.dumps(data, ensure_ascii=False)}\n\n" -async def _code_stream(req: ChatRequest, model: str): - """Chemin « moteur code » : Aider travaille, on garde le flux SSE vivant.""" - instruction = req.content - yield _sse("tool_call", {"name": "code_task", "args": {"instruction": instruction}}) - +async def _run_aider_keepalive(instruction: str, model: str): + """Lance Aider dans un thread en gardant le flux SSE vivant.""" task = asyncio.create_task( asyncio.to_thread(coder.run_code_task, instruction, model, None) ) - # Keepalive SSE pendant le travail (peut durer plusieurs minutes). while not task.done(): await asyncio.sleep(10) if not task.done(): + yield None # signal keepalive + yield await task + + +async def _code_stream( + req: ChatRequest, model: str, *, extra: str = "", plan: list[str] | None = None +): + """Chemin « moteur code » : Aider + vérification HTML avec auto-correction.""" + instruction = req.content + (extra or "") + yield _sse("tool_call", {"name": "code_task", "args": {"instruction": req.content}}) + + result = None + async for item in _run_aider_keepalive(instruction, model): + if item is None: yield ": keepalive\n\n" - result = await task + else: + result = item status = "ok" if result.get("ok") else "error" record = { "name": "code_task", - "args": {"instruction": instruction}, + "args": {"instruction": req.content}, "summary": result.get("summary", "terminé"), "status": status, } tools_meta = [record] yield _sse("tool_result", record) + all_files = list(result.get("files") or []) + + # Vérification des pages HTML produites + une passe d'auto-correction. + html_issues: list[str] = [] + for f in all_files: + if f.lower().endswith((".html", ".htm")): + try: + issues = check_html(_safe_path(f)) + except Exception: + issues = [] + if issues: + html_issues.append(f"{f} : " + " ; ".join(issues)) + + if html_issues and result.get("ok"): + yield _sse("tool_call", {"name": "html_check", "args": {"path": "vérification"}}) + yield _sse("tool_result", { + "name": "html_check", "args": {"path": "vérification"}, + "summary": " | ".join(html_issues)[:200], "status": "error", + }) + tools_meta.append({ + "name": "html_check", "args": {}, + "summary": " | ".join(html_issues)[:200], "status": "error", + }) + fix_instruction = ( + "Corrige ces problèmes détectés dans les fichiers HTML, sans rien " + "casser d'autre :\n" + "\n".join(html_issues) + ) + fix = None + async for item in _run_aider_keepalive(fix_instruction, model): + if item is None: + yield ": keepalive\n\n" + else: + fix = item + fix_rec = { + "name": "code_task", + "args": {"instruction": "auto-correction HTML"}, + "summary": fix.get("summary", "terminé"), + "status": "ok" if fix.get("ok") else "error", + } + tools_meta.append(fix_rec) + yield _sse("tool_call", {"name": "code_task", "args": fix_rec["args"]}) + yield _sse("tool_result", fix_rec) + for f in fix.get("files") or []: + if f not in all_files: + all_files.append(f) + if fix.get("text"): + result["text"] = (result.get("text") or "") + "\n\n" + fix["text"] + # Cartes par fichier modifié (réutilise le rendu write_file de l'UI). - for f in result.get("files") or []: + for f in all_files: file_rec = { "name": "write_file", "args": {"path": f}, @@ -79,10 +139,10 @@ async def _code_stream(req: ChatRequest, model: str): if text: yield _sse("token", {"content": text}) - db.add_message( - req.session_id, "assistant", text, model, - meta={"tools": tools_meta, "engine": "code"}, - ) + meta: dict = {"tools": tools_meta, "engine": "code"} + if plan: + meta["plan"] = plan + db.add_message(req.session_id, "assistant", text, model, meta=meta) yield _sse("done", {"content": text, "tools": tools_meta}) @@ -113,22 +173,61 @@ async def chat(req: ChatRequest) -> StreamingResponse: # Mémoire compressée : système + résumé des anciens tours + messages récents. convo = memory.build_convo(req.session_id, cfg["system_prompt"]) + # Mémoire long-terme (RAG) : souvenirs pertinents des autres sessions. + if cfg.get("rag_enabled", True): + memories = await rag.recall( + req.session_id, req.content, embed_model=cfg.get("embed_model") + ) + if memories: + convo.insert(1, { + "role": "system", + "content": "Souvenirs pertinents d'anciennes sessions :\n" + + "\n---\n".join(memories), + }) + # Moteur code : choisit le meilleur modèle code installé (config "auto"). code_model = ( await coder.pick_code_model(model, cfg.get("code_model")) if use_code else model ) + # Plan-puis-exécute : les demandes complexes sont décomposées d'abord. + plan: list[str] = [] + if cfg.get("plan_mode", True) and (use_code or enhance.needs_plan(req.content)): + plan = await enhance.make_plan(model, req.content) + async def event_stream(): yield _sse("start", {"model": model, "engine": "code" if use_code else "agent"}) + if plan: + yield _sse("plan", {"steps": plan}) + # Chemin « moteur code » : Aider gère la tâche de bout en bout. if use_code: - async for chunk in _code_stream(req, code_model): + instruction_plan = ( + "\n\nPlan à suivre :\n" + + "\n".join(f"{i+1}. {s}" for i, s in enumerate(plan)) + if plan else "" + ) + async for chunk in _code_stream(req, code_model, extra=instruction_plan, plan=plan): yield chunk asyncio.create_task(memory.maybe_summarize(req.session_id, model)) + if cfg.get("rag_enabled", True): + last = db.list_messages(req.session_id) + answer = last[-1]["content"] if last else "" + asyncio.create_task(rag.index_exchange( + req.session_id, req.content, answer, + embed_model=cfg.get("embed_model"), + )) return + if plan: + convo.append({ + "role": "system", + "content": "Plan à suivre pour cette demande :\n" + + "\n".join(f"{i+1}. {s}" for i, s in enumerate(plan)), + }) + final_content = "" tools_meta: list[dict] = [] stats_meta: dict | None = None @@ -198,6 +297,20 @@ async def chat(req: ChatRequest) -> StreamingResponse: with suppress(asyncio.CancelledError): await producer + # Auto-critique : relecture éclair puis révision (option « Qualité + »). + if ( + cfg.get("self_review", False) + and final_content + and not error_message + and not tools_meta + ): + yield _sse("status", {"message": "Relecture de la réponse…"}) + revised = await enhance.self_review(model, req.content, final_content) + if revised: + final_content = revised + yield _sse("revision", {"content": revised}) + yield _sse("notice", {"message": "Réponse révisée après auto-critique ✓"}) + if final_content or tools_meta: meta: dict = {} if tools_meta: @@ -206,6 +319,8 @@ async def chat(req: ChatRequest) -> StreamingResponse: meta["stats"] = stats_meta if thinking_meta: meta["thinking"] = thinking_meta + if plan: + meta["plan"] = plan db.add_message( req.session_id, "assistant", @@ -222,8 +337,13 @@ async def chat(req: ChatRequest) -> StreamingResponse: "error": error_message or None, }, ) - # Compression de l'historique en arrière-plan (sans bloquer la réponse). + # Tâches d'arrière-plan : compression de l'historique + mémoire RAG. asyncio.create_task(memory.maybe_summarize(req.session_id, model)) + if cfg.get("rag_enabled", True) and final_content: + asyncio.create_task(rag.index_exchange( + req.session_id, req.content, final_content, + embed_model=cfg.get("embed_model"), + )) return StreamingResponse( event_stream(), diff --git a/backend/app/routes/config.py b/backend/app/routes/config.py index 480125c..8baade3 100644 --- a/backend/app/routes/config.py +++ b/backend/app/routes/config.py @@ -22,6 +22,10 @@ class ConfigPatch(BaseModel): confirm_shell: bool | None = None think: bool | None = None code_model: str | None = None + plan_mode: bool | None = None + self_review: bool | None = None + rag_enabled: bool | None = None + embed_model: str | None = None @router.get("") diff --git a/backend/app/tools.py b/backend/app/tools.py index 4b59ee3..afa68c6 100644 --- a/backend/app/tools.py +++ b/backend/app/tools.py @@ -50,8 +50,40 @@ def read_file(path: str) -> dict: return {"ok": True, "content": content, "summary": summary} +def check_html(target: str) -> list[str]: + """Contrôles rapides d'une page HTML : références locales et balises. + + Renvoie une liste de problèmes (vide = OK). + """ + issues: list[str] = [] + try: + with open(target, "r", encoding="utf-8", errors="replace") as f: + content = f.read() + except OSError: + return issues + + base_dir = os.path.dirname(target) + + # Références locales cassées (href/src vers un fichier absent). + for _, ref in re.findall(r"""(href|src)=["']([^"'#]+)["']""", content, re.I): + if re.match(r"^(https?:|data:|mailto:|tel:|//|javascript:)", ref, re.I): + continue + ref_path = os.path.normpath(os.path.join(base_dir, ref.split("?")[0])) + if not os.path.exists(ref_path): + issues.append(f"référence cassée : {ref}") + + # Équilibre des balises structurantes. + for tag in ("html", "head", "body", "div", "section", "script", "style"): + opened = len(re.findall(rf"<{tag}[\s>]", content, re.I)) + closed = len(re.findall(rf"", content, re.I)) + if opened != closed: + issues.append(f"balise <{tag}> : {opened} ouverte(s) / {closed} fermée(s)") + + return issues[:6] + + def _verify_written(target: str) -> str | None: - """Vérification syntaxique immédiate après écriture (py/json). + """Vérification immédiate après écriture (py/json/html). Renvoyer l'erreur au modèle tout de suite lui permet de se corriger dans le même tour, au lieu de livrer un fichier cassé. @@ -65,6 +97,10 @@ def _verify_written(target: str) -> str | None: _json.loads(content) elif ext == ".py": compile(content, target, "exec") + elif ext in (".html", ".htm"): + problems = check_html(target) + if problems: + return " ; ".join(problems) except SyntaxError as exc: return f"SyntaxError ligne {exc.lineno}: {exc.msg}" except ValueError as exc: diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts index 7ff3759..04bd897 100644 --- a/frontend/src/api/client.ts +++ b/frontend/src/api/client.ts @@ -89,7 +89,13 @@ export interface Message { role: "user" | "assistant"; content: string; model?: string; - meta?: { tools?: ToolCall[]; stats?: MessageStats; thinking?: string } | null; + meta?: { + tools?: ToolCall[]; + stats?: MessageStats; + thinking?: string; + plan?: string[]; + engine?: string; + } | null; created_at: number; } @@ -105,6 +111,75 @@ export interface AgentConfig { tools: Record; confirm_shell: boolean; think: boolean; + code_model: string; + plan_mode: boolean; + self_review: boolean; + rag_enabled: boolean; + embed_model: string; +} + +// ── Benchmark de modèles ───────────────────────────────────────────────── +export interface BenchDetail { + task: string; + score: number; + detail: string; +} + +export interface BenchResult { + score: number; + details: BenchDetail[]; + at: number; +} + +export async function getBenchScores(): Promise> { + const res = await fetch("/api/bench"); + return (await res.json()).scores; +} + +/** Lance le benchmark d'un modèle en streamant la progression. */ +export async function runBench( + model: string, + onProgress: (task: string, score: number | null, detail?: string) => void +): Promise { + const res = await fetch("/api/bench", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ model }), + }); + if (!res.body) return null; + + const reader = res.body.getReader(); + const decoder = new TextDecoder(); + let buffer = ""; + let final: BenchResult | null = null; + + while (true) { + const { done, value } = await reader.read(); + if (done) break; + buffer += decoder.decode(value, { stream: true }); + const events = buffer.split("\n\n"); + buffer = events.pop() ?? ""; + for (const block of events) { + let event = ""; + let data = ""; + for (const line of block.split("\n")) { + if (line.startsWith("event: ")) event = line.slice(7).trim(); + else if (line.startsWith("data: ")) data += line.slice(6); + } + if (!data) continue; + try { + const payload = JSON.parse(data); + if (event === "task_start") onProgress(payload.task, null); + else if (event === "task_done") + onProgress(payload.task, payload.score, payload.detail); + else if (event === "done") + final = { score: payload.score, details: payload.details, at: Date.now() / 1000 }; + } catch { + /* bloc partiel */ + } + } + } + return final; } export async function runShell( @@ -204,6 +279,8 @@ export async function streamChat( onToolConfirm: (command: string) => void; onStatus: (msg: string) => void; onNotice: (msg: string) => void; + onPlan?: (steps: string[]) => void; + onRevision?: (content: string) => void; onDone: (full: string) => void; onError: (msg: string) => void; onAbort?: () => void; @@ -265,6 +342,8 @@ export async function streamChat( try { const payload = JSON.parse(dataLines.join("\n")); if (event === "token") handlers.onToken(payload.content); + else if (event === "plan") handlers.onPlan?.(payload.steps); + else if (event === "revision") handlers.onRevision?.(payload.content); else if (event === "thinking") handlers.onThinking(payload.content); else if (event === "status") handlers.onStatus(payload.message); else if (event === "notice") handlers.onNotice(payload.message); diff --git a/frontend/src/panels/ChatPanel.tsx b/frontend/src/panels/ChatPanel.tsx index 0b66c09..f615a01 100644 --- a/frontend/src/panels/ChatPanel.tsx +++ b/frontend/src/panels/ChatPanel.tsx @@ -17,6 +17,7 @@ export function ChatPanel() { streamStatus, streamNotice, streamTools, + streamPlan, sendMessage, currentSessionId, config, @@ -37,7 +38,7 @@ export function ChatPanel() { // Auto-scroll vers le bas à chaque token / message. useEffect(() => { scrollRef.current?.scrollTo({ top: scrollRef.current.scrollHeight }); - }, [messages, streamContent, streamThinking, streamTools]); + }, [messages, streamContent, streamThinking, streamTools, streamPlan]); const submit = () => { if (!draft.trim() || streaming) return; @@ -101,7 +102,7 @@ export function ChatPanel() { role: "assistant", content: streamContent, model: selectedModel, - meta: { tools: streamTools }, + meta: { tools: streamTools, plan: streamPlan }, created_at: Date.now() / 1000, }} pending @@ -173,6 +174,33 @@ export function ChatPanel() { ); } +/** Plan d'exécution affiché avant le travail de l'agent. */ +function PlanCard({ steps }: { steps: string[] }) { + return ( +
+
+ PLAN + + {steps.length} étape{steps.length > 1 ? "s" : ""} + +
+
    + {steps.map((s, i) => ( +
  1. + + {i + 1} + + {s} +
  2. + ))} +
+
+ ); +} + function Bubble({ msg, pending, @@ -227,6 +255,9 @@ function Bubble({ text={thinking ?? msg.meta?.thinking ?? ""} live={!!pending} /> + {(msg.meta?.plan?.length ?? 0) > 0 && ( + + )} {(msg.meta?.tools ?? []).map((t: ToolCall, i: number) => ( ))} diff --git a/frontend/src/panels/SettingsView.tsx b/frontend/src/panels/SettingsView.tsx index 565d2d8..ee6d12c 100644 --- a/frontend/src/panels/SettingsView.tsx +++ b/frontend/src/panels/SettingsView.tsx @@ -1,7 +1,7 @@ import { useEffect, useState } from "react"; import { useStore } from "../store/useStore"; -import { deleteModel, pullModel } from "../api/client"; -import type { AgentConfig } from "../api/client"; +import { deleteModel, getBenchScores, pullModel, runBench } from "../api/client"; +import type { AgentConfig, BenchResult } from "../api/client"; import { DownloadIcon, RefreshIcon } from "../components/Icon"; const TOOL_DESC: Record = { @@ -376,6 +376,38 @@ export function SettingsView() { )} + + +
+ INTELLIGENCE +
+ {[ + ["plan_mode", "Plan-puis-exécute", + "Décompose les demandes complexes en étapes"] as const, + ["self_review", "Auto-critique (Qualité +)", + "Relit et révise la réponse avant de la donner"] as const, + ["rag_enabled", "Mémoire long-terme (RAG)", + "Se souvient des anciennes sessions (modèle d'embedding requis)"] as const, + ].map(([key, label, desc], i) => ( +
0 ? "border-t-2 border-line-soft" : "" + }`} + > + + {label} + {desc} + + set(key, !draft[key])} + /> +
+ ))} +
+ + @@ -403,6 +435,108 @@ export function SettingsView() { ); } +/** Benchmark : évalue le modèle sélectionné sur 5 mini-épreuves. */ +function BenchCard() { + const selectedModel = useStore((s) => s.selectedModel); + const [scores, setScores] = useState>({}); + const [running, setRunning] = useState(false); + const [progress, setProgress] = useState< + { task: string; score: number | null; detail?: string }[] + >([]); + + useEffect(() => { + getBenchScores().then(setScores).catch(() => {}); + }, []); + + const launch = async () => { + if (!selectedModel || running) return; + setRunning(true); + setProgress([]); + try { + const result = await runBench(selectedModel, (task, score, detail) => { + setProgress((p) => { + const rest = p.filter((x) => x.task !== task); + return [...rest, { task, score, detail }]; + }); + }); + if (result) setScores((s) => ({ ...s, [selectedModel]: result })); + } finally { + setRunning(false); + } + }; + + const current = scores[selectedModel]; + + return ( + +
+
BENCHMARK
+ +
+ +
+ 5 mini-épreuves (outils, code, consignes, JSON, format) — score /100 + pour {selectedModel || "—"}. +
+ + {running && ( +
+ {progress.map((p) => ( +
+ {p.task} + {p.score === null ? ( + … + ) : ( + = 14 ? "text-ok" : p.score >= 8 ? "text-ink" : "text-warn"}> + {p.score}/20 + + )} +
+ ))} +
+ )} + + {!running && current && ( +
+
+ Score + = 70 ? "text-ok" : current.score >= 45 ? "text-ink" : "text-warn" + }`} + > + {current.score}/100 + +
+ {current.details.map((d) => ( +
+ {d.task} + {d.score}/20 +
+ ))} +
+ )} + + {Object.keys(scores).length > 1 && ( +
+ Autres :{" "} + {Object.entries(scores) + .filter(([m]) => m !== selectedModel) + .map(([m, r]) => `${m} (${r.score})`) + .join(" · ")} +
+ )} +
+ ); +} + function Card({ children, className = "", diff --git a/frontend/src/store/useStore.ts b/frontend/src/store/useStore.ts index 84c6757..89d7b4f 100644 --- a/frontend/src/store/useStore.ts +++ b/frontend/src/store/useStore.ts @@ -40,6 +40,7 @@ interface LokiState { streamStatus: string; streamNotice: string | null; streamTools: ToolCall[]; // appels d'outils de la réponse en cours + streamPlan: string[]; // plan de la réponse en cours fileTree: FileNode[]; previewPath: string | null; @@ -88,6 +89,7 @@ export const useStore = create((set, get) => ({ streamStatus: "", streamNotice: null, streamTools: [], + streamPlan: [], fileTree: [], previewPath: null, previewContent: "", @@ -234,6 +236,7 @@ export const useStore = create((set, get) => ({ streamStatus: "", streamNotice: null, streamTools: [], + streamPlan: [], pendingShell: null, }); void get().refreshSessions(); @@ -305,6 +308,8 @@ export const useStore = create((set, get) => ({ onToken: (t) => set({ streamContent: get().streamContent + t }), onThinking: (t) => set({ streamThinking: get().streamThinking + t }), onStatus: (message) => set({ streamStatus: message }), + onPlan: (steps) => set({ streamPlan: steps }), + onRevision: (content) => set({ streamContent: content }), onNotice: (message) => set({ streamNotice: message }), onToolCall: (call) => set({ streamTools: [...get().streamTools, call] }),