From cf1444f7c0d385b124f0a1b816dd6b50657fc56c Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 7 Jul 2026 09:17:46 +0000 Subject: [PATCH] =?UTF-8?q?Niveau=20sup=C3=A9rieur=20:=20m=C3=A9moire=20co?= =?UTF-8?q?mpress=C3=A9e,=20outils=20chirurgicaux,=20mod=C3=A8le=20code=20?= =?UTF-8?q?auto?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Faire d'un petit modèle un bon agent : - memory.py : contexte = invite + résumé des anciens tours + 10 derniers messages ; résumé régénéré en arrière-plan après chaque réponse (colonne summary sur sessions, migration douce). Contexte court = modèle concentré et qui tient sur le GPU. - tools.py : edit_file (recherche/remplacement exact, erreurs pédagogiques, unicité exigée), grep_search (regex bornée, dossiers ignorés), et vérification syntaxique auto (.py/.json) après chaque écriture — l'erreur revient au modèle qui se corrige dans le même tour. - coder.pick_code_model : les tâches de code vont au meilleur modèle code installé (qwen-coder, deepseek-coder…) via config code_model=auto. - Branché dans la route chat (mémoire + résolution du modèle code) et la boucle agent (code_task) ; UI : descriptions et glyphes des nouveaux outils. Tests : edit/grep/vérification, compression mémoire (19 msgs -> 12 dont résumé), pick auto/explicite, chat HTTP de bout en bout, build front. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01SVay7z3y7q2gEe54ByAE6N --- README.md | 19 +++ backend/app/agent.py | 3 +- backend/app/agent_config.py | 9 +- backend/app/coder.py | 38 ++++++ backend/app/db.py | 11 ++ backend/app/memory.py | 90 ++++++++++++++ backend/app/routes/chat.py | 17 ++- backend/app/routes/config.py | 1 + backend/app/tools.py | 170 ++++++++++++++++++++++++++- frontend/src/components/ToolCard.tsx | 7 +- frontend/src/panels/SettingsView.tsx | 2 + 11 files changed, 356 insertions(+), 11 deletions(-) create mode 100644 backend/app/memory.py diff --git a/README.md b/README.md index 190b71a..1af1ccf 100644 --- a/README.md +++ b/README.md @@ -115,6 +115,25 @@ npm run dev # http://localhost:5173 | `web_search` | Recherche web (DuckDuckGo / SearxNG) | désactivé | | `run_shell` | Exécuter une commande **(sensible)** | désactivé | +## Tirer le meilleur des petits modèles + +Loki est conçu pour qu'un modèle local modeste se comporte comme un bon agent : + +- **Mémoire compressée** : au-delà d'un seuil, les anciens tours sont résumés + en arrière-plan et le modèle ne reçoit que « invite + résumé + 10 derniers + messages ». Contexte court = modèle concentré, et qui reste sur le GPU. +- **Outils chirurgicaux** : `edit_file` (recherche/remplacement exact — pas de + réécriture intégrale ratée), `grep_search` (trouver avant de modifier), et + `write_file` par morceaux (overwrite/append). +- **Auto-vérification** : après chaque écriture, la syntaxe (`.py`, `.json`) + est contrôlée ; l'erreur est renvoyée immédiatement au modèle, qui se + corrige dans le même tour. +- **Bon modèle au bon poste** : les tâches de code sont confiées au meilleur + modèle code installé (`qwen-coder`, `deepseek-coder`…), automatiquement + (config `code_model: auto`), même si tu discutes avec un généraliste. +- **Récupération d'appels d'outils malformés** : arguments réparés ou + redemandés, modèles sans function-calling détectés et gérés. + ## Moteur code (façon Claude Code) — routage automatique Loki embarque [Aider](https://aider.chat) (Apache-2.0, version figée) comme diff --git a/backend/app/agent.py b/backend/app/agent.py index 15fbf5c..b3e1936 100644 --- a/backend/app/agent.py +++ b/backend/app/agent.py @@ -264,10 +264,11 @@ async def run_agent( # code_task : délégué au moteur code (Aider), long -> thread. if name == "code_task": + code_model = await coder.pick_code_model(model) result = await asyncio.to_thread( coder.run_code_task, args.get("instruction", ""), - model, + code_model, args.get("files") or [], ) summary = result.get("summary", "terminé") diff --git a/backend/app/agent_config.py b/backend/app/agent_config.py index 12542cd..bdf4aa2 100644 --- a/backend/app/agent_config.py +++ b/backend/app/agent_config.py @@ -30,13 +30,16 @@ DEFAULT_SYSTEM_PROMPT = ( # Outils disponibles. Les sensibles (web_search, run_shell) sont désactivés # par défaut, conformément à la maquette. AVAILABLE_TOOLS = [ - "read_file", "write_file", "list_dir", "code_task", "web_search", "run_shell", + "read_file", "write_file", "edit_file", "list_dir", "grep_search", + "code_task", "web_search", "run_shell", ] SENSITIVE_TOOLS = {"run_shell"} DEFAULT_TOOL_STATE = { "read_file": True, "write_file": True, + "edit_file": True, "list_dir": True, + "grep_search": True, "code_task": True, "web_search": False, "run_shell": False, @@ -56,6 +59,7 @@ PROFILE_FIELDS = { "tools", "confirm_shell", "think", + "code_model", *GENERATION_FIELDS, } @@ -88,6 +92,9 @@ DEFAULT_CONFIG: dict = { # Mode réflexion des modèles « thinking ». Désactiver (False) évite qu'un # modèle ne renvoie que du raisonnement sans réponse finale. "think": True, + # Modèle utilisé par le moteur code : "auto" = meilleur modèle code installé + # (qwen-coder, deepseek-coder…), sinon le modèle de chat courant. + "code_model": "auto", } diff --git a/backend/app/coder.py b/backend/app/coder.py index 734b3da..de5982c 100644 --- a/backend/app/coder.py +++ b/backend/app/coder.py @@ -39,6 +39,44 @@ def available() -> bool: return False +# Familles spécialisées code, par ordre de préférence. +_CODE_MODEL_HINTS = ( + "qwen3-coder", "qwen2.5-coder", "deepseek-coder", "codestral", "devstral", + "codegemma", "codellama", "starcoder", "coder", +) + + +async def pick_code_model(current: str, preference: str | None = None) -> str: + """Choisit le modèle pour les tâches de code. + + - préférence explicite (config code_model != "auto") -> respectée ; + - sinon, si un modèle spécialisé code est installé, on le prend (le plus + gros d'abord) : un petit modèle code bat un généraliste sur ce terrain ; + - sinon, on garde le modèle courant. + """ + if preference and preference != "auto": + return preference + + from .ollama_client import ollama + try: + installed = await ollama.list_models() + except Exception: + return current + + candidates: list[tuple[int, int, str]] = [] # (rang_hint, -taille, nom) + for m in installed: + name = (m.get("name") or "").lower() + for rank, hint in enumerate(_CODE_MODEL_HINTS): + if hint in name: + candidates.append((rank, -(m.get("size") or 0), m["name"])) + break + + if not candidates: + return current + candidates.sort() + return candidates[0][2] + + def run_code_task( instruction: str, model: str, diff --git a/backend/app/db.py b/backend/app/db.py index 832c430..290d0b0 100644 --- a/backend/app/db.py +++ b/backend/app/db.py @@ -54,6 +54,10 @@ def init_db() -> None: cols = {r["name"] for r in conn.execute("PRAGMA table_info(messages)")} if "meta" not in cols: conn.execute("ALTER TABLE messages ADD COLUMN meta TEXT") + # Migration douce : résumé de conversation (mémoire compressée). + scols = {r["name"] for r in conn.execute("PRAGMA table_info(sessions)")} + if "summary" not in scols: + conn.execute("ALTER TABLE sessions ADD COLUMN summary TEXT") # Table clé/valeur pour la configuration de l'agent. conn.execute( "CREATE TABLE IF NOT EXISTS config (key TEXT PRIMARY KEY, value TEXT)" @@ -166,6 +170,13 @@ def list_messages_for_model(sid: str) -> list[dict]: ] +def set_session_summary(sid: str, summary: str) -> None: + with _LOCK, _connect() as conn: + conn.execute( + "UPDATE sessions SET summary = ? WHERE id = ?", (summary, sid) + ) + + # ── Configuration (clé/valeur JSON) ────────────────────────────────────── def get_config_value(key: str) -> dict | None: with _LOCK, _connect() as conn: diff --git a/backend/app/memory.py b/backend/app/memory.py new file mode 100644 index 0000000..b68c488 --- /dev/null +++ b/backend/app/memory.py @@ -0,0 +1,90 @@ +"""Mémoire de conversation compressée — le vrai levier des petits modèles. + +Un petit modèle se noie dans un long historique : il oublie la consigne, part +en boucle, et un grand num_ctx le fait déborder du GPU. On garde donc : + [invite système] + [résumé compact des anciens tours] + [N derniers messages] + +Le résumé est régénéré en arrière-plan (après la réponse, sans latence pour +l'utilisateur) dès que l'historique dépasse le seuil. +""" +from __future__ import annotations + +import logging + +import httpx + +from . import db +from .ollama_client import ollama + +logger = logging.getLogger(__name__) + +# Nombre de messages récents passés tels quels au modèle. +KEEP_RECENT = 10 +# Au-delà de ce total, les anciens tours sont compressés dans le résumé. +SUMMARIZE_AFTER = KEEP_RECENT + 6 + +_SUMMARY_PROMPT = ( + "Résume la conversation ci-dessous en français, en 10 lignes maximum. " + "Conserve impérativement : l'objectif de l'utilisateur, les décisions " + "prises, les fichiers créés/modifiés et leur rôle, et les points encore " + "ouverts. Réponds UNIQUEMENT par le résumé." +) + + +def build_convo(sid: str, system_prompt: str) -> list[dict]: + """Construit le contexte : système + résumé éventuel + messages récents.""" + convo: list[dict] = [{"role": "system", "content": system_prompt}] + + session = db.get_session(sid) or {} + summary = (session.get("summary") or "").strip() + messages = db.list_messages_for_model(sid) + + if summary and len(messages) > KEEP_RECENT: + convo.append( + { + "role": "system", + "content": f"Résumé des échanges précédents :\n{summary}", + } + ) + convo += messages[-KEEP_RECENT:] + else: + convo += messages + return convo + + +async def maybe_summarize(sid: str, model: str) -> None: + """Compresse les anciens tours dans le résumé (tâche d'arrière-plan).""" + try: + messages = db.list_messages_for_model(sid) + if len(messages) <= SUMMARIZE_AFTER: + return + + session = db.get_session(sid) or {} + previous = (session.get("summary") or "").strip() + old = messages[:-KEEP_RECENT] + + transcript = "\n".join( + f"[{m['role']}] {m['content'][:600]}" for m in old + )[-8000:] + if previous: + transcript = f"[résumé existant] {previous}\n{transcript}" + + text = "" + async for chunk in ollama.chat( + model, + [ + {"role": "system", "content": _SUMMARY_PROMPT}, + {"role": "user", "content": transcript}, + ], + options={"temperature": 0.2, "num_predict": 350}, + stream=True, + ): + text += chunk.get("message", {}).get("content", "") + if chunk.get("done"): + break + + if text.strip(): + db.set_session_summary(sid, text.strip()) + except (httpx.HTTPError, OSError) as exc: + # Best-effort : un échec de résumé ne doit jamais gêner le chat. + logger.warning("Résumé de session %s impossible : %s", sid, exc) diff --git a/backend/app/routes/chat.py b/backend/app/routes/chat.py index 77c55fa..9fd480d 100644 --- a/backend/app/routes/chat.py +++ b/backend/app/routes/chat.py @@ -18,7 +18,7 @@ from fastapi import APIRouter, HTTPException from fastapi.responses import StreamingResponse from pydantic import BaseModel -from .. import agent_config, coder, db, router as msg_router +from .. import agent_config, coder, db, memory, router as msg_router from ..agent import run_agent from ..config import settings @@ -110,16 +110,23 @@ async def chat(req: ChatRequest) -> StreamingResponse: and await msg_router.is_code_task(req.content, model) ) - convo = [{"role": "system", "content": cfg["system_prompt"]}] - convo += db.list_messages_for_model(req.session_id) + # Mémoire compressée : système + résumé des anciens tours + messages récents. + convo = memory.build_convo(req.session_id, cfg["system_prompt"]) + + # Moteur code : choisit le meilleur modèle code installé (config "auto"). + code_model = ( + await coder.pick_code_model(model, cfg.get("code_model")) + if use_code else model + ) async def event_stream(): yield _sse("start", {"model": model, "engine": "code" if use_code else "agent"}) # Chemin « moteur code » : Aider gère la tâche de bout en bout. if use_code: - async for chunk in _code_stream(req, model): + async for chunk in _code_stream(req, code_model): yield chunk + asyncio.create_task(memory.maybe_summarize(req.session_id, model)) return final_content = "" @@ -215,6 +222,8 @@ async def chat(req: ChatRequest) -> StreamingResponse: "error": error_message or None, }, ) + # Compression de l'historique en arrière-plan (sans bloquer la réponse). + asyncio.create_task(memory.maybe_summarize(req.session_id, model)) return StreamingResponse( event_stream(), diff --git a/backend/app/routes/config.py b/backend/app/routes/config.py index 3c9e999..480125c 100644 --- a/backend/app/routes/config.py +++ b/backend/app/routes/config.py @@ -21,6 +21,7 @@ class ConfigPatch(BaseModel): tools: dict[str, bool] | None = None confirm_shell: bool | None = None think: bool | None = None + code_model: str | None = None @router.get("") diff --git a/backend/app/tools.py b/backend/app/tools.py index 75cac6e..4b59ee3 100644 --- a/backend/app/tools.py +++ b/backend/app/tools.py @@ -50,6 +50,30 @@ def read_file(path: str) -> dict: return {"ok": True, "content": content, "summary": summary} +def _verify_written(target: str) -> str | None: + """Vérification syntaxique immédiate après écriture (py/json). + + Renvoyer l'erreur au modèle tout de suite lui permet de se corriger dans + le même tour, au lieu de livrer un fichier cassé. + """ + ext = os.path.splitext(target)[1].lower() + try: + with open(target, "r", encoding="utf-8", errors="replace") as f: + content = f.read() + if ext == ".json": + import json as _json + _json.loads(content) + elif ext == ".py": + compile(content, target, "exec") + except SyntaxError as exc: + return f"SyntaxError ligne {exc.lineno}: {exc.msg}" + except ValueError as exc: + return f"JSON invalide : {exc}" + except OSError: + return None + return None + + def write_file(path: str, content: str, mode: str = "overwrite") -> dict: target = _safe_path(path) if mode not in {"overwrite", "append"}: @@ -60,7 +84,96 @@ def write_file(path: str, content: str, mode: str = "overwrite") -> dict: f.write(content) lines = len(content.splitlines()) verb = "complété" if mode == "append" else "modifié" if existed else "écrit" - return {"ok": True, "summary": f"{verb} · {lines} lignes", "lines": lines} + result = {"ok": True, "summary": f"{verb} · {lines} lignes", "lines": lines} + problem = _verify_written(target) + if problem: + result["verification"] = problem + result["summary"] += f" · ⚠ {problem}" + return result + + +def edit_file(path: str, search: str, replace: str) -> dict: + """Modification chirurgicale : remplace un extrait exact du fichier. + + Bien plus fiable que réécrire tout le fichier avec un petit modèle : + seul le fragment visé change, le reste est garanti intact. + """ + target = _safe_path(path) + if not os.path.isfile(target): + raise ToolError(f"fichier introuvable : {path}") + if not search: + raise ToolError("search vide : fournis l'extrait exact à remplacer") + with open(target, "r", encoding="utf-8", errors="replace") as f: + content = f.read() + + count = content.count(search) + if count == 0: + preview = search.strip().splitlines()[0][:60] if search.strip() else "" + raise ToolError( + f"extrait introuvable dans {path} (cherché : « {preview}… »). " + "Relis le fichier avec read_file et copie l'extrait EXACT." + ) + if count > 1: + raise ToolError( + f"extrait présent {count} fois dans {path} : ajoute du contexte " + "autour pour le rendre unique." + ) + + with open(target, "w", encoding="utf-8") as f: + f.write(content.replace(search, replace, 1)) + + delta = len(replace.splitlines()) - len(search.splitlines()) + result = {"ok": True, "summary": f"remplacé · {delta:+d} ligne(s)"} + problem = _verify_written(target) + if problem: + result["verification"] = problem + result["summary"] += f" · ⚠ {problem}" + return result + + +_GREP_SKIP_DIRS = {".git", "node_modules", "__pycache__", ".venv", "dist"} +_MAX_GREP_MATCHES = 50 + + +def grep_search(pattern: str, path: str = ".") -> dict: + """Recherche un motif (regex) dans les fichiers du workspace.""" + if not pattern: + raise ToolError("pattern vide") + try: + rx = re.compile(pattern, re.I) + except re.error as exc: + raise ToolError(f"regex invalide : {exc}") + + root = _safe_path(path) + matches: list[str] = [] + files_hit: set[str] = set() + for dirpath, dirnames, filenames in os.walk(root): + dirnames[:] = [d for d in dirnames if d not in _GREP_SKIP_DIRS] + for fname in sorted(filenames): + full = os.path.join(dirpath, fname) + if os.path.getsize(full) > 1_000_000: + continue + rel = os.path.relpath(full, _workspace_root()) + try: + with open(full, "r", encoding="utf-8", errors="replace") as f: + for lineno, line in enumerate(f, 1): + if rx.search(line): + matches.append(f"{rel}:{lineno}: {line.strip()[:160]}") + files_hit.add(rel) + if len(matches) >= _MAX_GREP_MATCHES: + break + except OSError: + continue + if len(matches) >= _MAX_GREP_MATCHES: + break + if len(matches) >= _MAX_GREP_MATCHES: + break + + summary = ( + f"{len(matches)} correspondance(s) dans {len(files_hit)} fichier(s)" + if matches else "aucune correspondance" + ) + return {"ok": True, "matches": matches, "summary": summary} def list_dir(path: str = ".") -> dict: @@ -183,7 +296,9 @@ def run_shell(command: str, timeout: int = 60) -> dict: TOOL_IMPL = { "read_file": read_file, "write_file": write_file, + "edit_file": edit_file, "list_dir": list_dir, + "grep_search": grep_search, "web_search": web_search, "run_shell": run_shell, } @@ -228,6 +343,29 @@ TOOL_DEFINITIONS = [ }, }, }, + { + "type": "function", + "function": { + "name": "edit_file", + "description": ( + "Modifier UN extrait précis d'un fichier existant (recherche/" + "remplacement exact). Préférable à write_file pour toute " + "modification partielle : le reste du fichier reste intact." + ), + "parameters": { + "type": "object", + "properties": { + "path": {"type": "string", "description": "Chemin relatif au workspace"}, + "search": { + "type": "string", + "description": "Extrait EXACT à remplacer (copie fidèle, unique dans le fichier)", + }, + "replace": {"type": "string", "description": "Nouveau texte"}, + }, + "required": ["path", "search", "replace"], + }, + }, + }, { "type": "function", "function": { @@ -241,6 +379,24 @@ TOOL_DEFINITIONS = [ }, }, }, + { + "type": "function", + "function": { + "name": "grep_search", + "description": ( + "Chercher un motif (regex, insensible à la casse) dans tous les " + "fichiers du workspace. Renvoie fichier:ligne:texte." + ), + "parameters": { + "type": "object", + "properties": { + "pattern": {"type": "string", "description": "Motif à chercher"}, + "path": {"type": "string", "description": "Sous-répertoire (défaut : racine)"}, + }, + "required": ["pattern"], + }, + }, + }, { "type": "function", "function": { @@ -314,11 +470,21 @@ def _normalize_args(name: str, args: dict | None) -> dict: if alias in args: args["content"] = args[alias] break - elif name in {"read_file", "list_dir"} and "path" not in args: + elif name in {"read_file", "list_dir", "edit_file"} and "path" not in args: for alias in ("file_path", "filepath", "filename", "file", "dir", "directory"): if args.get(alias): args["path"] = args[alias] break + if name == "edit_file": + if "search" not in args and "old" in args: + args["search"] = args.pop("old") + if "replace" not in args and "new" in args: + args["replace"] = args.pop("new") + elif name == "grep_search" and "pattern" not in args: + for alias in ("query", "search", "text", "regex"): + if args.get(alias): + args["pattern"] = args[alias] + break return args diff --git a/frontend/src/components/ToolCard.tsx b/frontend/src/components/ToolCard.tsx index dc0bedd..2c2abe0 100644 --- a/frontend/src/components/ToolCard.tsx +++ b/frontend/src/components/ToolCard.tsx @@ -10,6 +10,7 @@ export function ToolCard({ call }: { call: ToolCall }) { const mainArg = (call.args?.path as string) ?? (call.args?.query as string) ?? + (call.args?.pattern as string) ?? (call.args?.command as string) ?? (call.args?.instruction as string); const argPreview = @@ -19,7 +20,7 @@ export function ToolCard({ call }: { call: ToolCall }) { ? "(…)" : "()"; - const write = call.name === "write_file"; + const write = call.name === "write_file" || call.name === "edit_file"; return (
@@ -71,7 +72,7 @@ function ToolGlyph({ name }: { name: string }) { strokeLinecap: "round" as const, strokeLinejoin: "round" as const, }; - if (name === "write_file") + if (name === "write_file" || name === "edit_file") return ( @@ -89,7 +90,7 @@ function ToolGlyph({ name }: { name: string }) { ); - if (name === "web_search") + if (name === "web_search" || name === "grep_search") return ( diff --git a/frontend/src/panels/SettingsView.tsx b/frontend/src/panels/SettingsView.tsx index fbc5623..565d2d8 100644 --- a/frontend/src/panels/SettingsView.tsx +++ b/frontend/src/panels/SettingsView.tsx @@ -7,7 +7,9 @@ import { DownloadIcon, RefreshIcon } from "../components/Icon"; const TOOL_DESC: Record = { read_file: "Lire un fichier du projet", write_file: "Créer / modifier un fichier", + edit_file: "Modification chirurgicale (recherche/remplacement)", list_dir: "Lister un répertoire", + grep_search: "Chercher dans les fichiers", code_task: "Moteur code (multi-fichiers, git)", web_search: "Recherche web", run_shell: "Exécuter une commande",