mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-12 01:37:06 +02:00
Merge pull request #2 from R0m1k3/claude/ajean-loki-container-fork-aep9r2
Claude/ajean loki container fork aep9r2
This commit is contained in:
260 files changed
+39234
-18986
No files matched your search
+8
-24
@@ -1,26 +1,10 @@
|
||||
# Contrôle de version
|
||||
.git
|
||||
.gitignore
|
||||
|
||||
# Python
|
||||
**/__pycache__
|
||||
**/*.pyc
|
||||
backend/.venv
|
||||
backend/venv
|
||||
|
||||
# Node / build
|
||||
frontend/node_modules
|
||||
frontend/dist
|
||||
**/.vite
|
||||
frontend/tsconfig.tsbuildinfo
|
||||
|
||||
# Données & workspace de runtime (montés en volume)
|
||||
data
|
||||
workspace
|
||||
|
||||
# Environnement / IDE
|
||||
.github
|
||||
data/
|
||||
models/
|
||||
docs/
|
||||
dist/
|
||||
*.md
|
||||
!README.md
|
||||
.env
|
||||
.env.local
|
||||
.idea
|
||||
.vscode
|
||||
**/.DS_Store
|
||||
.env.*
|
||||
+16
-23
@@ -1,26 +1,19 @@
|
||||
# ── Loki — configuration ────────────────────────────────────────────────
|
||||
# Hôte Ollama. En Docker, host.docker.internal pointe vers la machine hôte.
|
||||
# En local (hors Docker), utilise http://localhost:11434
|
||||
OLLAMA_HOST=http://host.docker.internal:11434
|
||||
# ── Loki — configuration du build/run Docker ────────────────────────────
|
||||
# (fork d'AJEAN : https://github.com/nathaninline/ajean)
|
||||
|
||||
# Modèle sélectionné par défaut au démarrage
|
||||
DEFAULT_MODEL=llama3.1:8b
|
||||
# Port de l'interface web (identique dedans/dehors).
|
||||
LOKI_WEB_PORT=8090
|
||||
|
||||
# Dossier de travail de l'agent (monté en volume dans le conteneur)
|
||||
WORKSPACE_DIR=/workspace
|
||||
# Architectures CUDA compilées (build local uniquement).
|
||||
# 75=Turing (GTX 16xx / RTX 20xx), 86=Ampere (RTX 30xx), 89=Ada (RTX 40xx).
|
||||
# Restreindre à sa carte divise le temps de compilation.
|
||||
CUDA_ARCHS=75;86;89
|
||||
|
||||
# Base SQLite (sessions, messages, config)
|
||||
DATA_DIR=/data
|
||||
|
||||
# Port de l'application (identique dedans/dehors). 8717 par défaut.
|
||||
PORT=8717
|
||||
|
||||
# Optionnel : URL d'une instance SearxNG pour l'outil web_search.
|
||||
# Si vide, web_search utilise DuckDuckGo (sans clé d'API).
|
||||
SEARX_URL=
|
||||
|
||||
# Optionnel : VRAM du GPU (en Mo) pour l'auto-réglage, si Ollama tourne sur une
|
||||
# autre machine (la détection nvidia-smi ne voit pas un GPU distant).
|
||||
# Ex. 12000 pour une carte 12 Go. 0 = détection automatique.
|
||||
GPU_VRAM_MB=0
|
||||
GPU_NAME=
|
||||
# Optionnel — semé au PREMIER démarrage seulement (modifiable ensuite dans
|
||||
# l'UI, qui garde la main) :
|
||||
# Modèle initial : nom d'un .gguf présent dans ./models ou chemin complet.
|
||||
LOKI_MODEL=
|
||||
# Taille de contexte (défaut moteur : 32768)
|
||||
LOKI_CTX=
|
||||
# Couches déportées sur le GPU (défaut : 999 = tout)
|
||||
LOKI_NGL=
|
||||
@@ -0,0 +1 @@
|
||||
*.go text eol=lf
|
||||
@@ -0,0 +1,51 @@
|
||||
# CI : build + vet + gofmt + tests sur chaque push/PR.
|
||||
name: ci
|
||||
on:
|
||||
push:
|
||||
branches: ["**"]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: gofmt
|
||||
run: test -z "$(gofmt -l .)" || (gofmt -l . && exit 1)
|
||||
- run: go vet ./...
|
||||
- run: go build ./...
|
||||
- run: go test ./...
|
||||
# staticcheck : le projet est à zéro avertissement, et doit le rester.
|
||||
# Les règles écartées et leur raison vivent dans staticcheck.conf.
|
||||
- name: staticcheck
|
||||
run: go run honnef.co/go/tools/cmd/staticcheck@latest ./...
|
||||
- name: cross-compile (cibles release Linux + Windows)
|
||||
run: |
|
||||
# darwin n'est pas ici : depuis v0.5.4 l'icône de barre de menus passe
|
||||
# par Cocoa (CGO) → les cibles macOS se compilent sur un runner Apple
|
||||
# (job « macos » ci-dessous, comme dans release.yml).
|
||||
for t in linux/amd64 linux/arm64 windows/amd64 windows/arm64; do
|
||||
GOOS=${t%/*} GOARCH=${t#*/} CGO_ENABLED=0 go build -o /dev/null ./cmd/loki
|
||||
done
|
||||
|
||||
macos:
|
||||
runs-on: macos-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- run: go vet ./...
|
||||
# Les tests tournent AUSSI ici : certains chemins ne s'exécutent que sur un
|
||||
# vrai système de fichiers Unix (extraction des liens symboliques des
|
||||
# archives llama.cpp, que Windows résout par une copie de repli).
|
||||
- run: go test ./...
|
||||
- name: build macOS (CGO, arm64 + amd64)
|
||||
run: |
|
||||
CGO_ENABLED=1 GOOS=darwin GOARCH=arm64 go build -o /dev/null ./cmd/loki
|
||||
CGO_ENABLED=1 GOOS=darwin GOARCH=amd64 \
|
||||
CGO_CFLAGS="-arch x86_64" CGO_LDFLAGS="-arch x86_64" \
|
||||
go build -o /dev/null ./cmd/loki
|
||||
@@ -16,6 +16,16 @@ jobs:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# L'étape llama.cpp CUDA a besoin de place : l'image devel pèse ~8 Go et
|
||||
# un runner ubuntu-latest n'offre que ~14 Go libres. On purge ce qui est
|
||||
# préinstallé et inutile ici (SDK Android, .NET, Haskell…).
|
||||
- name: Libérer l'espace disque du runner
|
||||
run: |
|
||||
sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc \
|
||||
/usr/local/.ghcup /opt/hostedtoolcache/CodeQL
|
||||
sudo docker image prune -af
|
||||
df -h /
|
||||
|
||||
- name: Setup Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
@@ -43,7 +53,11 @@ jobs:
|
||||
push: true
|
||||
build-args: |
|
||||
LOKI_VERSION=${{ github.sha }}
|
||||
CUDA_ARCHS=75;86;89
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
# Cache limité : les couches CUDA (plusieurs Go) satureraient le
|
||||
# plafond de 10 Go du cache GitHub Actions — on ne cache que le mode
|
||||
# min (couches finales), le gain principal restant le cache apt/go.
|
||||
cache-from: type=gha
|
||||
cache-to: type=gha,mode=max
|
||||
cache-to: type=gha,mode=min
|
||||
@@ -0,0 +1,52 @@
|
||||
# Build macOS à la demande, SANS rien publier : produit Loki.app en artifact
|
||||
# téléchargeable depuis l'onglet Actions (visible seulement par le dépôt).
|
||||
# Sert à tester le bundle avant de décider d'une vraie release (release.yml).
|
||||
name: mac-build
|
||||
on:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: macos-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: build + bundle Loki.app (arm64)
|
||||
run: |
|
||||
mkdir -p dist
|
||||
CGO_ENABLED=1 GOOS=darwin GOARCH=arm64 \
|
||||
go build -trimpath -ldflags="-s -w" -o dist/loki-macos-arm ./cmd/loki
|
||||
mkdir -p /tmp/loki.iconset
|
||||
sips -s format png 'cmd/loki/icon.ico' --out /tmp/icon.png >/dev/null 2>&1 || true
|
||||
for s in 16 32 128 256 512; do
|
||||
sips -z $s $s /tmp/icon.png --out "/tmp/loki.iconset/icon_${s}x${s}.png" >/dev/null 2>&1 || true
|
||||
done
|
||||
iconutil -c icns /tmp/loki.iconset -o /tmp/loki.icns >/dev/null 2>&1 || true
|
||||
app="dist/Loki.app"
|
||||
mkdir -p "$app/Contents/MacOS" "$app/Contents/Resources"
|
||||
cp dist/loki-macos-arm "$app/Contents/MacOS/loki"
|
||||
chmod +x "$app/Contents/MacOS/loki"
|
||||
cp /tmp/loki.icns "$app/Contents/Resources/loki.icns" 2>/dev/null || true
|
||||
printf '%s\n' \
|
||||
'<?xml version="1.0" encoding="UTF-8"?>' \
|
||||
'<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">' \
|
||||
'<plist version="1.0"><dict>' \
|
||||
'<key>CFBundleName</key><string>Loki</string>' \
|
||||
'<key>CFBundleDisplayName</key><string>Loki</string>' \
|
||||
'<key>CFBundleExecutable</key><string>loki</string>' \
|
||||
'<key>CFBundleIdentifier</key><string>link.loki.app</string>' \
|
||||
'<key>CFBundleIconFile</key><string>loki</string>' \
|
||||
'<key>CFBundlePackageType</key><string>APPL</string>' \
|
||||
'<key>LSMinimumSystemVersion</key><string>11.0</string>' \
|
||||
'<key>LSUIElement</key><true/>' \
|
||||
'<key>NSHighResolutionCapable</key><true/>' \
|
||||
'</dict></plist>' > "$app/Contents/Info.plist"
|
||||
codesign --force --deep -s - "$app" || true
|
||||
(cd dist && zip -qry loki-macos-arm.zip Loki.app && rm -rf Loki.app)
|
||||
ls -l dist
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: loki-macos-arm
|
||||
path: dist/loki-macos-arm.zip
|
||||
@@ -0,0 +1,138 @@
|
||||
# Release : sur tag vX.Y.Z, compile les 6 cibles et publie la release GitHub
|
||||
# avec les binaires (loki-linux, loki-macos, loki-windows.exe et leurs
|
||||
# variantes -arm) ET le fichier SHA256SUMS que `loki update` vérifie avant
|
||||
# d'installer. Ces noms doivent rester en phase avec assetNameFor (sys_update.go),
|
||||
# sinon plus aucune machine ne se met à jour.
|
||||
# ⚠️ NE PAS uploader de binaires à la main en plus : ce workflow écrase les
|
||||
# assets du même nom (softprops remplace), et des sommes qui ne correspondent
|
||||
# plus aux binaires font échouer `loki update` (vécu sur v0.4.7).
|
||||
#
|
||||
# Deux jobs de build : macOS a besoin d'un vrai runner Apple (l'icône de barre de
|
||||
# menus passe par Cocoa → CGO, pas cross-compilable depuis Linux) ; Linux et
|
||||
# Windows restent CGO_ENABLED=0 sur ubuntu.
|
||||
name: release
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
jobs:
|
||||
macos:
|
||||
runs-on: macos-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: build macOS (CGO, arm64 + amd64)
|
||||
run: |
|
||||
mkdir -p dist
|
||||
# arm64 = natif ; amd64 = cross avec le SDK universel (-arch x86_64).
|
||||
CGO_ENABLED=1 GOOS=darwin GOARCH=arm64 \
|
||||
go build -trimpath -ldflags="-s -w" -o dist/loki-macos-arm ./cmd/loki
|
||||
CGO_ENABLED=1 GOOS=darwin GOARCH=amd64 \
|
||||
CGO_CFLAGS="-arch x86_64" CGO_LDFLAGS="-arch x86_64" \
|
||||
go build -trimpath -ldflags="-s -w" -o dist/loki-macos ./cmd/loki
|
||||
file dist/loki-macos*
|
||||
- name: bundles Loki.app
|
||||
run: |
|
||||
# Un binaire nu double-cliqué ouvrirait le Terminal. On publie donc
|
||||
# aussi un vrai bundle Loki.app (zippé pour garder le bit exécutable) :
|
||||
# au clic, le Finder lance Contents/MacOS/loki, qui détecte le bundle
|
||||
# et bascule en mode « application » (UI web + navigateur + icône dans
|
||||
# la barre de menus), comme le double-clic sur loki.exe sous Windows.
|
||||
# LSUIElement : app de barre de menus, aucune icône dans le Dock.
|
||||
# L'icône part du PNG 1024 généré avec le reste de la marque
|
||||
# (cmd/loki/icon.png, voir tools/gen-icon) : plus de conversion depuis
|
||||
# le .ico, dont l'échec silencieux donnait une app sans icône.
|
||||
mkdir -p /tmp/loki.iconset
|
||||
for s in 16 32 128 256 512; do
|
||||
sips -z $s $s 'cmd/loki/icon.png' --out "/tmp/loki.iconset/icon_${s}x${s}.png" >/dev/null
|
||||
done
|
||||
iconutil -c icns /tmp/loki.iconset -o /tmp/loki.icns
|
||||
v="${GITHUB_REF_NAME#v}"
|
||||
for bin in loki-macos loki-macos-arm; do
|
||||
app="dist/bundle-$bin/Loki.app"
|
||||
mkdir -p "$app/Contents/MacOS" "$app/Contents/Resources"
|
||||
cp "dist/$bin" "$app/Contents/MacOS/loki"
|
||||
chmod +x "$app/Contents/MacOS/loki"
|
||||
cp /tmp/loki.icns "$app/Contents/Resources/loki.icns" 2>/dev/null || true
|
||||
printf '%s\n' \
|
||||
'<?xml version="1.0" encoding="UTF-8"?>' \
|
||||
'<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">' \
|
||||
'<plist version="1.0"><dict>' \
|
||||
'<key>CFBundleName</key><string>Loki</string>' \
|
||||
'<key>CFBundleDisplayName</key><string>Loki</string>' \
|
||||
'<key>CFBundleExecutable</key><string>loki</string>' \
|
||||
'<key>CFBundleIdentifier</key><string>link.loki.app</string>' \
|
||||
'<key>CFBundleIconFile</key><string>loki</string>' \
|
||||
'<key>CFBundlePackageType</key><string>APPL</string>' \
|
||||
"<key>CFBundleShortVersionString</key><string>$v</string>" \
|
||||
"<key>CFBundleVersion</key><string>$v</string>" \
|
||||
'<key>LSMinimumSystemVersion</key><string>11.0</string>' \
|
||||
'<key>LSUIElement</key><true/>' \
|
||||
'<key>NSHighResolutionCapable</key><true/>' \
|
||||
'</dict></plist>' > "$app/Contents/Info.plist"
|
||||
# Signature ad-hoc : sans elle, macOS refuse purement et simplement
|
||||
# de lancer un binaire arm64 non signé (pas juste un avertissement).
|
||||
codesign --force --deep -s - "$app" || true
|
||||
(cd "dist/bundle-$bin" && zip -qry "../$bin.zip" Loki.app)
|
||||
rm -rf "dist/bundle-$bin"
|
||||
done
|
||||
ls -l dist
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: macos
|
||||
path: dist/*
|
||||
|
||||
release:
|
||||
needs: macos
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: build Linux + Windows
|
||||
run: |
|
||||
mkdir -p dist
|
||||
# Noms d'asset : loki-<os>[-arm][.exe]. Volontairement lisibles, et
|
||||
# volontairement DIFFÉRENTS du schéma loki-<GOOS>-<GOARCH> des 0.7 :
|
||||
# leur mise à jour automatique ne trouve alors aucun asset et échoue
|
||||
# sans rien remplacer, au lieu d'installer un binaire 0.8 sur une
|
||||
# machine encore agencée en 0.7 (voir updateAssetName, sys_update.go).
|
||||
for t in linux/amd64:loki-linux linux/arm64:loki-linux-arm \
|
||||
windows/amd64:loki-windows.exe windows/arm64:loki-windows-arm.exe; do
|
||||
target=${t%%:*}; name=${t#*:}
|
||||
os=${target%/*}; arch=${target#*/}; ldflags="-s -w"
|
||||
# Windows : sous-système CONSOLE, pour que `cmd` attende la fin du
|
||||
# programme et que redirections, tubes et commandes interactives
|
||||
# fonctionnent. La console créée au double-clic est refermée par le
|
||||
# programme lui-même (voir setupConsole, sys_console_windows.go).
|
||||
GOOS=$os GOARCH=$arch CGO_ENABLED=0 \
|
||||
go build -trimpath -ldflags="$ldflags" -o "dist/$name" ./cmd/loki
|
||||
done
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: macos
|
||||
path: dist
|
||||
- name: SHA256SUMS
|
||||
run: |
|
||||
chmod +x dist/loki-macos dist/loki-macos-arm
|
||||
# Tous les assets commencent par « loki- », zips compris : un seul
|
||||
# motif suffit. Ajouter « *.zip » les compterait DEUX fois, et un
|
||||
# SHA256SUMS à doublons est exactement ce que `loki update` vérifie.
|
||||
(cd dist && sha256sum loki-* > SHA256SUMS)
|
||||
ls -l dist && cat dist/SHA256SUMS
|
||||
- name: publier la release
|
||||
uses: softprops/action-gh-release@v2
|
||||
with:
|
||||
files: dist/*
|
||||
# Notes rédigées à la main, versionnées avec le code : le tag emporte
|
||||
# donc ses notes, au lieu de publier une liste de commits auto-générée
|
||||
# qu'il fallait ensuite remplacer à la main dans l'interface GitHub.
|
||||
# ⚠️ Réécrire RELEASE_NOTES.md AVANT de poser le tag.
|
||||
body_path: RELEASE_NOTES.md
|
||||
generate_release_notes: false
|
||||
+10
-22
@@ -1,27 +1,15 @@
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.venv/
|
||||
venv/
|
||||
*.egg-info/
|
||||
# Binaires compilés
|
||||
/loki
|
||||
/loki.exe
|
||||
/dist/
|
||||
|
||||
# Node
|
||||
node_modules/
|
||||
frontend/dist/
|
||||
.vite/
|
||||
# Artefacts de build Go
|
||||
*.exe
|
||||
*.test
|
||||
*.out
|
||||
|
||||
# Data & workspace (runtime)
|
||||
data/*.db
|
||||
data/*.sqlite3
|
||||
workspace/*
|
||||
!workspace/.gitkeep
|
||||
|
||||
# Env / IDE
|
||||
.env
|
||||
.env.local
|
||||
# OS / éditeurs
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
.idea/
|
||||
.vscode/
|
||||
|
||||
# TS build cache
|
||||
frontend/tsconfig.tsbuildinfo
|
||||
+79
-35
@@ -1,47 +1,91 @@
|
||||
# ── Étape 1 : build du frontend ─────────────────────────────────────────
|
||||
FROM node:20-alpine AS frontend
|
||||
WORKDIR /app/frontend
|
||||
COPY frontend/package.json frontend/package-lock.json* ./
|
||||
RUN npm install
|
||||
COPY frontend/ ./
|
||||
RUN npm run build
|
||||
# ── Loki — image GPU autonome (fork d'AJEAN, https://github.com/nathaninline/ajean)
|
||||
#
|
||||
# Trois étapes :
|
||||
# 1. compilation de llama.cpp avec CUDA (mêmes flags que backend_build.go,
|
||||
# sauf GGML_NATIVE=OFF : l'image est bâtie sur un runner GitHub, pas sur
|
||||
# la machine qui l'exécutera — des instructions CPU « natives » du runner
|
||||
# provoqueraient un Illegal instruction ailleurs) ;
|
||||
# 2. compilation du binaire Go `loki` (UI embarquée via go:embed) ;
|
||||
# 3. runtime CUDA léger : llama-server + loki + entrypoint.
|
||||
#
|
||||
# Pré-requis d'exécution : NVIDIA Container Toolkit sur l'hôte.
|
||||
# docker build -t loki --build-arg CUDA_ARCHS=86 .
|
||||
# CUDA_ARCHS : 75=Turing, 86=Ampere, 89=Ada — restreindre à sa carte divise
|
||||
# le temps de build et la taille de l'image.
|
||||
|
||||
# ── Étape 2 : runtime backend (sert aussi le front statique) ────────────
|
||||
FROM python:3.12-slim AS runtime
|
||||
WORKDIR /app
|
||||
# ── Étape 1 : llama.cpp CUDA ────────────────────────────────────────────
|
||||
FROM nvidia/cuda:12.6.3-devel-ubuntu24.04 AS llamacpp
|
||||
|
||||
ENV PYTHONUNBUFFERED=1 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
WORKSPACE_DIR=/workspace \
|
||||
DATA_DIR=/data \
|
||||
PORT=8717
|
||||
|
||||
# curl pour le HEALTHCHECK ; git pour le moteur code (commits Aider) ;
|
||||
# nodejs/npm pour les serveurs MCP lancés via npx (Playwright, Context7…)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends curl git nodejs npm \
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git cmake build-essential ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY backend/requirements.txt ./
|
||||
RUN pip install --no-cache-dir -r requirements.txt mcp-server-fetch
|
||||
# Épingler LLAMACPP_REF sur un tag (ex. b4991) pour des builds reproductibles.
|
||||
ARG LLAMACPP_REF=master
|
||||
ARG CUDA_ARCHS=75;86;89
|
||||
|
||||
COPY backend/ ./backend/
|
||||
# Frontend compilé servi en statique par FastAPI
|
||||
COPY --from=frontend /app/frontend/dist ./backend/static
|
||||
RUN git clone --depth 1 --branch "${LLAMACPP_REF}" \
|
||||
https://github.com/ggml-org/llama.cpp /src/llama.cpp
|
||||
|
||||
RUN mkdir -p /workspace /data
|
||||
RUN cmake -S /src/llama.cpp -B /src/llama.cpp/build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DGGML_CUDA=ON \
|
||||
-DGGML_CUDA_F16=ON \
|
||||
-DGGML_NATIVE=OFF \
|
||||
-DCMAKE_CUDA_ARCHITECTURES="${CUDA_ARCHS}" \
|
||||
-DLLAMA_CURL=OFF \
|
||||
-DLLAMA_BUILD_TESTS=OFF \
|
||||
-DLLAMA_BUILD_EXAMPLES=OFF \
|
||||
-DLLAMA_BUILD_UI=OFF \
|
||||
-DLLAMA_USE_PREBUILT_UI=OFF \
|
||||
&& cmake --build /src/llama.cpp/build --target llama-server -j"$(nproc)" \
|
||||
&& mkdir -p /opt/llama.cpp \
|
||||
&& find /src/llama.cpp/build -name 'llama-server' -type f -exec cp {} /opt/llama.cpp/ \; \
|
||||
&& find /src/llama.cpp/build -name '*.so*' -exec cp -P {} /opt/llama.cpp/ \;
|
||||
|
||||
# Marqueur de version (git sha court), injecté par le workflow de build.
|
||||
# ── Étape 2 : binaire Go loki ───────────────────────────────────────────
|
||||
FROM golang:1.25 AS gobuild
|
||||
WORKDIR /src
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
COPY cmd/ cmd/
|
||||
COPY internal/ internal/
|
||||
COPY tools/ tools/
|
||||
RUN CGO_ENABLED=0 go build -trimpath -ldflags "-s -w" -o /out/loki ./cmd/loki
|
||||
|
||||
# ── Étape 3 : runtime ───────────────────────────────────────────────────
|
||||
FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 AS runtime
|
||||
|
||||
# Marqueur de build (sha court), injecté par le workflow GitHub.
|
||||
ARG LOKI_VERSION=dev
|
||||
ENV LOKI_VERSION=${LOKI_VERSION}
|
||||
LABEL org.opencontainers.image.revision="${LOKI_VERSION}" \
|
||||
org.opencontainers.image.source="https://github.com/R0m1k3/Loki" \
|
||||
org.opencontainers.image.description="Loki — fork conteneurisé d'AJEAN (github.com/nathaninline/ajean, MIT)"
|
||||
|
||||
# NB : on tourne en root par défaut pour rester compatible avec les volumes
|
||||
# montés d'Unraid (/mnt/user/appdata/...), souvent détenus par root. Pour
|
||||
# durcir, surcharge l'utilisateur côté compose (ex. user: "99:100").
|
||||
# curl pour le HEALTHCHECK ; git pour les outils de l'agent ; nodejs/npm pour
|
||||
# les serveurs MCP lancés via npx ; libgomp1 pour llama-server ; tini en PID 1.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl git libgomp1 tini nodejs npm \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
EXPOSE 8717
|
||||
WORKDIR /app/backend
|
||||
COPY --from=llamacpp /opt/llama.cpp /opt/llama.cpp
|
||||
COPY --from=gobuild /out/loki /usr/local/bin/loki
|
||||
COPY docker-entrypoint.sh /usr/local/bin/docker-entrypoint.sh
|
||||
RUN chmod +x /usr/local/bin/docker-entrypoint.sh && mkdir -p /data /models
|
||||
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=15s --retries=3 \
|
||||
CMD curl -fsS "http://localhost:${PORT}/api/health" || exit 1
|
||||
ENV LOKI_CONTAINER=1 \
|
||||
LOKI_HOME=/data \
|
||||
LOKI_MODEL_DIRS=/models \
|
||||
LD_LIBRARY_PATH=/opt/llama.cpp \
|
||||
NVIDIA_VISIBLE_DEVICES=all \
|
||||
NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
|
||||
CMD ["sh", "-c", "uvicorn app.main:app --host 0.0.0.0 --port ${PORT}"]
|
||||
# 8090 : interface web (seule à exposer). Le moteur (8080) reste interne au
|
||||
# conteneur — il n'est pas authentifié par défaut.
|
||||
EXPOSE 8090
|
||||
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=20s --retries=3 \
|
||||
CMD curl -fsS "http://localhost:${LOKI_WEB_PORT:-8090}/" >/dev/null || exit 1
|
||||
|
||||
WORKDIR /data
|
||||
ENTRYPOINT ["tini", "--", "docker-entrypoint.sh"]
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Jean contributors
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,27 @@
|
||||
# Avis d'attribution
|
||||
|
||||
**Loki est un fork de [AJEAN](https://github.com/nathaninline/ajean)**, créé
|
||||
par [nathaninline](https://github.com/nathaninline) et publié sous licence MIT.
|
||||
L'essentiel du code de ce dépôt — moteur d'assistant IA en Go, interface web
|
||||
embarquée, mémoire persistante, accès internet, outils MCP, accès distant
|
||||
chiffré — est l'œuvre du projet AJEAN. Merci à son auteur.
|
||||
|
||||
La licence d'origine est conservée à l'identique dans [`LICENSE`](LICENSE).
|
||||
|
||||
## Modifications apportées par le fork
|
||||
|
||||
- **Renommage** AJEAN → Loki (binaire `loki`, `LOKI_HOME`, `/etc/loki`,
|
||||
units `loki-engine` / `loki-ui`).
|
||||
- **Conteneurisation** : image Docker autonome (llama.cpp compilé CUDA +
|
||||
binaire Go), `docker-compose` et variante Unraid, publication GHCR.
|
||||
- **Supervision sans systemd** (`sys_service_container.go`) : en conteneur,
|
||||
le moteur est piloté par fichier PID au lieu de systemctl — indispensable
|
||||
pour que l'UI puisse redémarrer le moteur (changement de modèle).
|
||||
- **`loki config get/set`** : lecture/écriture non interactive de la
|
||||
configuration, utilisée par l'entrypoint Docker.
|
||||
|
||||
## Services externes
|
||||
|
||||
Le tunnel d'accès distant continue de pointer vers **ajean.link**, le relais
|
||||
opéré par l'auteur d'AJEAN ; le catalogue de modèles intégré est également
|
||||
servi par ajean.link. Ces services appartiennent au projet amont.
|
||||
@@ -1,309 +1,121 @@
|
||||
# Loki — Agent IA local sur Ollama
|
||||
# Loki — assistant IA local en conteneur (fork d'AJEAN)
|
||||
|
||||
Loki est un atelier d'agent IA **100 % local**, conçu pour se connecter à
|
||||
[Ollama](https://ollama.com) et travailler en mode agentique : un tchat, des
|
||||
outils (lecture/écriture de fichiers, aperçu HTML en direct…) et un workspace
|
||||
de fichiers. Pensé pour un déploiement **Docker** simple.
|
||||
> **Loki est un fork de [AJEAN](https://github.com/nathaninline/ajean)** de
|
||||
> [nathaninline](https://github.com/nathaninline), sous licence MIT — voir
|
||||
> [`NOTICE.md`](NOTICE.md) et [`LICENSE`](LICENSE). L'essentiel du code et des
|
||||
> fonctionnalités vient d'AJEAN ; ce fork le rebaptise et le fait tourner dans
|
||||
> **un conteneur Docker GPU autonome**, là où l'amont s'installe en binaire +
|
||||
> systemd sur la machine hôte.
|
||||
|
||||
Le thème visuel (« atelier café », sombre et chaleureux, accent ambre) est
|
||||
décliné fidèlement depuis la maquette d'origine.
|
||||
Loki fait tourner un modèle de langage **100 % en local** : tchat, mémoire
|
||||
persistante, accès internet, outils MCP, agent (shell, fichiers), accès distant
|
||||
chiffré — serveur d'inférence llama.cpp compris, dans une seule image.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Frontend (React + Vite + TS + Tailwind)
|
||||
│ HTTP + SSE
|
||||
Backend (FastAPI, Python)
|
||||
├── /api/status, /api/models, /api/models/pull (Ollama)
|
||||
├── /api/sessions (CRUD sessions)
|
||||
├── /api/chat (boucle agentique + outils, streaming SSE)
|
||||
├── /api/files (arborescence + contenu du workspace)
|
||||
│ Outils agent : read_file · write_file · list_dir (confinés au workspace)
|
||||
│ httpx │ volume
|
||||
Ollama (:11434) /workspace + /data (SQLite)
|
||||
┌────────────────── conteneur loki ──────────────────┐
|
||||
│ loki web (UI + API, port 8090, premier plan) │
|
||||
│ │ pilote (fichier PID — pas de systemd) │
|
||||
│ loki serve ──exec──► llama-server (CUDA, :8080) │
|
||||
│ ▲ modèles .gguf │
|
||||
│ /data (config, bbolt, mémoire, workspace) │
|
||||
│ /models (GGUF déposés à la main) │
|
||||
└────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
Un seul conteneur : le backend FastAPI sert l'API **et** le frontend compilé.
|
||||
Loki se connecte à un **Ollama existant** via `OLLAMA_HOST`.
|
||||
- **Un seul conteneur** : l'UI joint le moteur sur `localhost` (contrainte
|
||||
héritée de l'amont), les deux partagent donc le même conteneur.
|
||||
- **Sans systemd** : l'amont pilote le moteur via systemctl ; en conteneur,
|
||||
Loki bascule automatiquement sur une supervision par fichier PID
|
||||
(`internal/loki/sys_service_container.go`). Changer de modèle depuis l'UI
|
||||
redémarre le moteur normalement.
|
||||
- Le moteur (port 8080, non authentifié par défaut) **n'est pas exposé** ;
|
||||
seule l'UI (8090) l'est.
|
||||
|
||||
## Démarrage rapide (Docker)
|
||||
## Démarrage rapide (Docker, GPU NVIDIA)
|
||||
|
||||
Pré-requis : pilote NVIDIA + [NVIDIA Container Toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html).
|
||||
|
||||
```bash
|
||||
cp .env.example .env # ajuste OLLAMA_HOST si besoin
|
||||
docker compose up --build
|
||||
cp .env.example .env # CUDA_ARCHS=86 pour une RTX 30xx, etc.
|
||||
docker compose up --build # 30-45 min : compilation CUDA de llama.cpp
|
||||
```
|
||||
|
||||
Application disponible sur http://localhost:8717
|
||||
|
||||
> **Ollama** : par défaut Loki vise `http://host.docker.internal:11434`
|
||||
> (un Ollama installé sur la machine hôte). Pour embarquer Ollama dans la
|
||||
> stack : `docker compose --profile ollama up` puis règle
|
||||
> `OLLAMA_HOST=http://ollama:11434`.
|
||||
Interface : http://localhost:8090 — télécharge un modèle depuis le catalogue
|
||||
intégré (onglet modèles), il démarre tout seul.
|
||||
|
||||
## Installation sur Unraid
|
||||
|
||||
L'image est **construite et publiée automatiquement par GitHub Actions** sur GHCR
|
||||
(`ghcr.io/r0m1k3/loki:latest`) à chaque push sur `main`. Aucun build ni `git`
|
||||
n'est nécessaire sur Unraid — un simple `pull`. Compose prêt à l'emploi :
|
||||
[`docker-compose.unraid.yml`](docker-compose.unraid.yml).
|
||||
L'image est construite et publiée par GitHub Actions sur GHCR
|
||||
(`ghcr.io/r0m1k3/loki:latest`) à chaque push sur `main` — aucun build sur
|
||||
Unraid. Compose prêt à l'emploi : [`docker-compose.unraid.yml`](docker-compose.unraid.yml).
|
||||
|
||||
1. Dans le terminal Unraid, crée les dossiers de données :
|
||||
1. Installe le plugin **Nvidia Driver** (Apps) et vérifie `nvidia-smi`.
|
||||
2. Crée les dossiers :
|
||||
```bash
|
||||
mkdir -p /mnt/user/appdata/loki/workspace /mnt/user/appdata/loki/data
|
||||
mkdir -p /mnt/user/appdata/loki/data /mnt/user/appdata/loki/models
|
||||
```
|
||||
2. Installe le plugin **Compose Manager** (Apps), crée une nouvelle stack, et
|
||||
colle le contenu de `docker-compose.unraid.yml`.
|
||||
3. **Adapte `OLLAMA_HOST`** : mets l'IP de ton serveur Unraid où tourne le
|
||||
conteneur Ollama, p. ex. `http://192.168.1.10:11434`.
|
||||
4. **Compose Up**. Loki est accessible sur `http://<ip-unraid>:8717`
|
||||
(change le port à gauche du mapping `8717:8080` s'il est déjà pris).
|
||||
3. Plugin **Compose Manager** → nouvelle stack → colle
|
||||
`docker-compose.unraid.yml` → **Compose Up**.
|
||||
4. Interface : `http://<ip-unraid>:8090`.
|
||||
|
||||
> La première publication de l'image prend quelques minutes (le temps que le
|
||||
> workflow GitHub se termine). Si l'image est privée, rends le package **public**
|
||||
> une fois (GitHub → Packages → `loki` → Package settings → Change visibility).
|
||||
>
|
||||
> Les modèles déjà présents dans ton Ollama sont **détectés automatiquement** ;
|
||||
> tu peux en télécharger d'autres depuis l'onglet Configuration.
|
||||
> Mise à jour : Compose Down/Up (avec pull), ou `docker compose pull`.
|
||||
## Configuration
|
||||
|
||||
## Développement (sans Docker)
|
||||
Tout se règle **dans l'UI** (modèle, contexte, presets…) et survit aux
|
||||
redémarrages (volume `/data`). Variables d'environnement du conteneur :
|
||||
|
||||
| Variable | Rôle | Défaut |
|
||||
|---|---|---|
|
||||
| `LOKI_WEB_PORT` | port de l'UI | `8090` |
|
||||
| `LOKI_MODEL` | modèle initial (semé au 1er boot seulement) | — |
|
||||
| `LOKI_CTX` | taille de contexte initiale | `32768` |
|
||||
| `LOKI_NGL` | couches GPU initiales | `999` (tout) |
|
||||
| `LOKI_HOME` | données (volume) | `/data` |
|
||||
| `LOKI_MODEL_DIRS` | dossiers .gguf additionnels | `/models` |
|
||||
|
||||
En CLI dans le conteneur : `docker exec -it loki loki status` (aussi :
|
||||
`logs`, `restart`, `config`, `bench`, `test`…).
|
||||
|
||||
## Fonctionnalités (héritées d'AJEAN)
|
||||
|
||||
- **Tchat** avec streaming, raisonnement visible, pièces jointes, vision
|
||||
(selon modèle), export de conversations.
|
||||
- **Mémoire persistante** (`memory off|ondemand|always`).
|
||||
- **Accès internet** : recherche + lecture de pages, moteur Go intégré ou
|
||||
[Crawl4AI](https://github.com/unclecode/crawl4ai) pour les pages JS.
|
||||
- **Agent** : shell, fichiers, workspace (`agent on`).
|
||||
- **Serveurs MCP** : Node.js est inclus dans l'image pour les serveurs `npx`.
|
||||
- **Presets** de configuration par modèle, bench, auto-détection GPU.
|
||||
- **Accès distant chiffré** via le relais [ajean.link](https://ajean.link)
|
||||
(service opéré par l'auteur de l'amont).
|
||||
- **API OpenAI-compatible** exposable (`network on`, protégée par clé).
|
||||
|
||||
## Différences avec l'amont
|
||||
|
||||
| | AJEAN (amont) | Loki (ce fork) |
|
||||
|---|---|---|
|
||||
| Installation | binaire + `sudo ajean install` (systemd) | `docker compose up` |
|
||||
| Moteur llama.cpp | compilé sur la machine (`ajean llamacpp install`) | précompilé CUDA dans l'image |
|
||||
| Supervision moteur | systemd / launchd / PID (Windows) | fichier PID (`LOKI_CONTAINER=1`) |
|
||||
| Configuration initiale | `ajean edit` ($EDITOR) | entrypoint + `loki config set` |
|
||||
| Mise à jour | `ajean update` (binaire GitHub) | `docker compose pull` |
|
||||
|
||||
Le reste — UI, mémoire, outils, protocole — est celui d'AJEAN. Pour récupérer
|
||||
les évolutions de l'amont :
|
||||
|
||||
**Backend**
|
||||
```bash
|
||||
cd backend
|
||||
pip install -r requirements.txt
|
||||
OLLAMA_HOST=http://localhost:11434 uvicorn app.main:app --reload --port 8080
|
||||
git fetch upstream && git merge upstream/main # conflits de renommage à arbitrer
|
||||
```
|
||||
|
||||
**Frontend** (proxy `/api` → `:8080`)
|
||||
```bash
|
||||
cd frontend
|
||||
npm install
|
||||
npm run dev # http://localhost:5173
|
||||
```
|
||||
## Build sans GPU / autres accélérateurs
|
||||
|
||||
## Configuration (variables d'environnement)
|
||||
L'image par défaut cible CUDA. Pour un essai CPU, remplace dans le
|
||||
`Dockerfile` les bases `nvidia/cuda:*` par `ubuntu:24.04` et retire les flags
|
||||
`-DGGML_CUDA=*` (llama.cpp bascule en CPU). Vulkan/ROCm : adapter les flags
|
||||
comme le fait l'amont (`internal/loki/backend_build.go`).
|
||||
|
||||
| Variable | Défaut | Rôle |
|
||||
| --------------- | ----------------------------------- | ----------------------------- |
|
||||
| `OLLAMA_HOST` | `http://host.docker.internal:11434` | URL de l'instance Ollama |
|
||||
| `DEFAULT_MODEL` | `gemma4:12b` | Modèle sélectionné au démarrage |
|
||||
| `WORKSPACE_DIR` | `/workspace` | Dossier de travail de l'agent |
|
||||
| `DATA_DIR` | `/data` | Base SQLite (sessions + config) |
|
||||
| `PORT` | `8717` | Port de l'application (dedans = dehors) |
|
||||
| `SEARX_URL` | *(vide)* | Instance SearxNG pour `web_search` (sinon DuckDuckGo) |
|
||||
## Licence
|
||||
|
||||
## Utilisation
|
||||
|
||||
1. Vérifie la pastille **Ollama** (verte = connecté) en haut à droite, et choisis
|
||||
un modèle qui supporte le *function calling* (profil fourni : `gemma4:12b`).
|
||||
2. Décris une tâche dans le tchat, p. ex. *« Crée une landing page pour un café
|
||||
nommé Café Lumière, avec menu et horaires »*.
|
||||
3. L'agent lit/écrit des fichiers dans le **workspace** ; chaque appel d'outil
|
||||
s'affiche dans le fil, et l'aperçu HTML apparaît à droite (onglets **Aperçu /
|
||||
Code / Logs**).
|
||||
4. Règle le comportement dans **Configuration** (modèle, température/top-p/top-k,
|
||||
jetons max, outils actifs, invite système).
|
||||
|
||||
## Outils de l'agent
|
||||
|
||||
| Outil | Rôle | Par défaut |
|
||||
| ------------- | -------------------------------------- | ---------- |
|
||||
| `read_file` | Lire un fichier du workspace | activé |
|
||||
| `write_file` | Créer / modifier un fichier | activé |
|
||||
| `list_dir` | Lister un répertoire | activé |
|
||||
| `web_search` | Recherche web (DuckDuckGo / SearxNG) | désactivé |
|
||||
| `run_shell` | Exécuter une commande **(sensible)** | désactivé |
|
||||
|
||||
## Préchargement des modèles (réponses instantanées)
|
||||
|
||||
Ollama décharge un modèle de la VRAM après quelques minutes d'inactivité : le
|
||||
message suivant paie alors un rechargement complet (lent). Loki évite ça :
|
||||
|
||||
- **Préchargement à la sélection** : choisir un modèle le charge immédiatement
|
||||
en VRAM (`/api/models/warm`).
|
||||
- **Maintien au chaud** : chaque requête envoie un `keep_alive` (défaut 30 min,
|
||||
réglable dans Configuration → Intelligence : de « décharger aussitôt » à
|
||||
« toujours »).
|
||||
- **Préchargement au démarrage** du modèle par défaut (en arrière-plan).
|
||||
- **Indicateur d'état** : la pastille du sélecteur de modèle est verte quand le
|
||||
modèle est chargé sur GPU, orange sur CPU, blanche s'il reste à charger.
|
||||
|
||||
## Performance Ollama (recommandé)
|
||||
|
||||
Réglages **côté serveur Ollama** qui rendent Loki nettement plus fluide
|
||||
(variables d'environnement du service/conteneur Ollama, redémarrage requis) :
|
||||
|
||||
| Variable | Valeur conseillée | Effet |
|
||||
| -------- | ----------------- | ----- |
|
||||
| `OLLAMA_KEEP_ALIVE` | `30m` (ou `-1`) | Durée de rétention par défaut d'un modèle en VRAM. `-1` = jamais déchargé (machine dédiée). Doit être ≥ au `keep_alive` configuré dans Loki. |
|
||||
| `OLLAMA_MAX_LOADED_MODELS` | `2` | Autorise le modèle de chat **et** le modèle d'embedding (RAG) à résider ensemble en VRAM — supprime les allers-retours de chargement à chaque message. |
|
||||
| `OLLAMA_NUM_PARALLEL` | `1` | Loki est mono-utilisateur ; chaque slot parallèle multiplie la VRAM du cache KV. |
|
||||
| `OLLAMA_FLASH_ATTENTION` | `1` | Attention plus rapide et plus sobre en VRAM. |
|
||||
| `OLLAMA_KV_CACHE_TYPE` | `q8_0` | Cache KV quantifié : ~moitié de VRAM en moins, marge pour un 12B sur 12 Go. |
|
||||
|
||||
Exemple pour un Ollama en Docker (service `ollama` du compose) :
|
||||
|
||||
```yaml
|
||||
ollama:
|
||||
image: ollama/ollama
|
||||
environment:
|
||||
- OLLAMA_KEEP_ALIVE=30m
|
||||
- OLLAMA_MAX_LOADED_MODELS=2
|
||||
- OLLAMA_NUM_PARALLEL=1
|
||||
- OLLAMA_FLASH_ATTENTION=1
|
||||
- OLLAMA_KV_CACHE_TYPE=q8_0
|
||||
```
|
||||
|
||||
Côté Loki, tout est déjà optimisé : pool HTTP keep-alive partagé vers Ollama,
|
||||
options runner identiques sur tous les appels (pas de rechargement du modèle en
|
||||
plein message), `keep_alive` systématique (chat **et** embeddings), préparation
|
||||
du contexte (RAG + plan) en parallèle après l'ouverture du flux, et caches
|
||||
courts (`/api/tags`, stats GPU).
|
||||
|
||||
## Modes d'exécution (Plan / Build / Yolo)
|
||||
|
||||
Un sélecteur dans le composer contrôle le niveau d'autonomie de l'agent :
|
||||
- **Plan** 🔍 — lecture seule (`read_file`, `list_dir`, `grep_search`) : l'agent
|
||||
analyse et propose sans jamais modifier de fichier ni exécuter de commande.
|
||||
- **Build** 🔨 — normal : écrit les fichiers, `run_shell` demande confirmation.
|
||||
- **Yolo** ⚡ — autonomie maximale : approuve tout, y compris le shell.
|
||||
|
||||
## Panneau Git & Diff
|
||||
|
||||
Le workspace étant un dépôt git (chaque action de l'agent = un commit), l'onglet
|
||||
**Git** du panneau de droite montre l'historique des commits, le **diff coloré**
|
||||
de chacun, et un bouton **↶ Annuler** (revert) pour défaire une modification en
|
||||
un clic.
|
||||
|
||||
## Projets (répertoires de travail)
|
||||
|
||||
Chaque session peut travailler dans un **projet** : un sous-dossier du
|
||||
workspace choisi via le chip 📁 du composer (« + Nouveau projet » pour en
|
||||
créer un). L'agent, le shell, le moteur code et l'arborescence sont confinés
|
||||
au projet ; chaque projet a son propre dépôt git (historique et revert
|
||||
indépendants). Session sans projet = racine du workspace.
|
||||
|
||||
## Intelligence augmentée
|
||||
|
||||
- **Plan-puis-exécute** : les demandes complexes sont décomposées en 3-5 étapes
|
||||
affichées dans le fil ; l'agent (ou le moteur code) suit le plan.
|
||||
- **Auto-critique « Qualité + »** (Configuration → Intelligence) : la réponse
|
||||
est relue et révisée avant d'être finalisée.
|
||||
- **Mémoire long-terme (RAG)** : chaque échange est vectorisé (`/api/embed`)
|
||||
et les souvenirs pertinents des anciennes sessions sont réinjectés en
|
||||
contexte. Nécessite un modèle d'embedding installé (ex.
|
||||
`ollama pull nomic-embed-text`) — sinon désactivé silencieusement.
|
||||
- **Vérification HTML** : liens locaux cassés et balises déséquilibrées sont
|
||||
détectés après chaque génération ; le moteur code fait une passe
|
||||
d'auto-correction.
|
||||
- **Benchmark intégré** (Configuration → Benchmark) : 5 mini-épreuves notées
|
||||
/100 (appel d'outil, code exécutable, consignes, extraction JSON, format)
|
||||
pour comparer objectivement tes modèles installés.
|
||||
|
||||
## Serveurs MCP (outils professionnels)
|
||||
|
||||
Configuration → **Serveurs MCP** : catalogue préconfiguré, tout désactivé par
|
||||
défaut (un serveur inactif ne coûte rien — aucun process, aucun outil dans le
|
||||
prompt du modèle).
|
||||
|
||||
| Serveur | Apport |
|
||||
| ------- | ------ |
|
||||
| Playwright | l'agent pilote un vrai navigateur : teste ses pages, lit la console |
|
||||
| Context7 | documentation à jour de n'importe quelle librairie |
|
||||
| Fetch | lecture propre d'URL (markdown) |
|
||||
| SearxNG | vraie recherche web (URL d'instance requise) |
|
||||
| Personnalisé | n'importe quel serveur MCP (commande stdio ou URL) |
|
||||
|
||||
Chaque carte a un bouton **Tester** (connexion d'essai + liste des outils
|
||||
découverts). Un serveur en panne n'interrompt jamais le chat : notice dans le
|
||||
fil, nouvelle tentative au message suivant.
|
||||
|
||||
## Skills automatiques
|
||||
|
||||
Cinq méthodes expertes (débogage systématique, création web, refactor sûr,
|
||||
analyse de données, rédaction structurée) sont injectées automatiquement selon
|
||||
la tâche détectée — sélection lexicale instantanée, une seule à la fois, badge
|
||||
« 📘 Méthode : … » dans le fil. Désactivable dans Configuration → Intelligence.
|
||||
|
||||
## Tirer le meilleur des petits modèles
|
||||
|
||||
Loki est conçu pour qu'un modèle local modeste se comporte comme un bon agent :
|
||||
|
||||
- **Mémoire compressée** : au-delà d'un seuil, les anciens tours sont résumés
|
||||
en arrière-plan et le modèle ne reçoit que « invite + résumé + 10 derniers
|
||||
messages ». Contexte court = modèle concentré, et qui reste sur le GPU.
|
||||
- **Outils chirurgicaux** : `edit_file` (recherche/remplacement exact — pas de
|
||||
réécriture intégrale ratée), `grep_search` (trouver avant de modifier), et
|
||||
`write_file` par morceaux (overwrite/append).
|
||||
- **Auto-vérification** : après chaque écriture, la syntaxe (`.py`, `.json`)
|
||||
est contrôlée ; l'erreur est renvoyée immédiatement au modèle, qui se
|
||||
corrige dans le même tour.
|
||||
- **Bon modèle au bon poste** : les tâches de code sont confiées au meilleur
|
||||
modèle code installé (`qwen-coder`, `deepseek-coder`…), automatiquement
|
||||
(config `code_model: auto`), même si tu discutes avec un généraliste.
|
||||
- **Récupération d'appels d'outils malformés** : arguments réparés ou
|
||||
redemandés, modèles sans function-calling détectés et gérés.
|
||||
|
||||
## Moteur code (façon Claude Code) — routage automatique
|
||||
|
||||
Loki embarque [Aider](https://aider.chat) (Apache-2.0, version figée) comme
|
||||
**moteur code** : édition multi-fichiers fiable (formats diff/search-replace,
|
||||
efficaces même avec de petits modèles), repo map, et **commits git
|
||||
automatiques** dans le workspace.
|
||||
|
||||
**C'est invisible** : un routeur classe chaque message.
|
||||
- Tâche de code détectée (heuristique lexicale instantanée) → le message part
|
||||
au **moteur code** ; le fil affiche la carte `code_task`, les fichiers
|
||||
modifiés et le commit.
|
||||
- Sinon → **boucle agent** classique ; et l'agent peut lui-même déléguer au
|
||||
moteur via l'outil `code_task` quand il juge qu'il faut coder.
|
||||
|
||||
Désactivable dans Configuration → Outils (`code_task`). Le workspace est
|
||||
auto-initialisé en dépôt git : chaque modification de code = un commit
|
||||
(historique et rollback via `git log` / `git revert` dans le workspace).
|
||||
|
||||
## Profil GPU fourni
|
||||
|
||||
Loki est préconfiguré pour une **RTX 3060 12 Go** avec `gemma4:12b` en Q4 :
|
||||
contexte 8192, sortie 4096 jetons, batch 256, GPU principal 0 et les 49 couches
|
||||
du modèle sur le GPU. Les paramètres de génération et de cache/contexte sont
|
||||
enregistrés séparément pour chaque modèle.
|
||||
|
||||
La quantification du cache KV reste globale dans Ollama. Pour économiser environ
|
||||
la moitié de sa VRAM, démarre Ollama avec `OLLAMA_FLASH_ATTENTION=1` et
|
||||
`OLLAMA_KV_CACHE_TYPE=q8_0`. Un redémarrage d'Ollama est requis. Le conteneur
|
||||
doit également exposer le GPU (`--gpus=all`) ; Loki ne peut pas contourner une
|
||||
configuration Docker sans accès CUDA.
|
||||
|
||||
## Sécurité
|
||||
|
||||
- **Confinement** : toutes les opérations fichier (`read_file`, `write_file`,
|
||||
`list_dir`) et `run_shell` sont strictement confinées au `WORKSPACE_DIR`. Toute
|
||||
tentative de sortie (`../`, chemin absolu) est rejetée.
|
||||
- **`run_shell`** est désactivé par défaut. Une fois activé, chaque commande
|
||||
proposée par l'agent demande une **validation explicite** dans l'interface
|
||||
(option *confirm_shell*, activée par défaut) avant exécution.
|
||||
- Le conteneur tourne en **utilisateur non-root** et expose un **HEALTHCHECK**.
|
||||
- Loki est conçu pour un usage **local** : n'expose pas le port publiquement sans
|
||||
ajouter ta propre couche d'authentification.
|
||||
|
||||
## Feuille de route
|
||||
|
||||
- [x] **Phase 1** — Socle + design system fidèle au thème, layout 3 panneaux
|
||||
- [x] **Phase 2** — Connexion Ollama : statut, liste des modèles, pull avec progression, sélecteur
|
||||
- [x] **Phase 3** — Chat streaming (SSE) + persistance des sessions (SQLite)
|
||||
- [x] **Phase 4** — Boucle agentique & outils fichiers (read/write/list), confinés au workspace, rendu des appels d'outils dans le fil
|
||||
- [x] **Phase 5** — Aperçu HTML live + onglets Code/Logs + arborescence du workspace
|
||||
- [x] **Phase 6** — Configuration complète (génération, toggles d'outils, invite système)
|
||||
- [x] **Phase 7** — Outils avancés : `web_search` (DuckDuckGo/SearxNG) et `run_shell` avec validation utilisateur
|
||||
- [x] **Phase 8** — Durcissement & documentation Docker
|
||||
|
||||
## Structure
|
||||
|
||||
```
|
||||
backend/ FastAPI : routes Ollama, client httpx, config
|
||||
frontend/ React : design system (tailwind.config), panneaux, store Zustand
|
||||
workspace/ fichiers créés par l'agent (monté en volume)
|
||||
data/ base SQLite (monté en volume)
|
||||
```
|
||||
MIT — © les contributeurs d'AJEAN (« Jean contributors ») pour le code amont,
|
||||
voir [`LICENSE`](LICENSE) et [`NOTICE.md`](NOTICE.md).
|
||||
@@ -0,0 +1,19 @@
|
||||
La grande nouveauté de cette version : AJEAN sait enfin voir les images.
|
||||
|
||||
## La vision, configurable par preset
|
||||
|
||||
Jusqu'ici rien dans l'interface ne permettait de donner des yeux à un modèle. Il fallait glisser l'option `--mmproj` à la main dans la configuration brute, et son chemin n'était même pas résolu comme celui du modèle. L'éditeur de preset gagne un champ **Vision** : tu y choisis le projecteur multimodal (fichier `mmproj`) qui accompagne le modèle, et c'est tout. Au démarrage du moteur, AJEAN le charge tout seul.
|
||||
|
||||
Le champ ne liste que les projecteurs (les fichiers dont le nom contient `mmproj`), pas les modèles de plusieurs Go, pour que le choix reste lisible. Et s'il te manque un projecteur, le champ « télécharger un modèle » accepte aussi un lien vers un `mmproj` : une fois récupéré, il se place directement dans le champ Vision.
|
||||
|
||||
## Les images arrivent vraiment au modèle
|
||||
|
||||
Avant, une image collée dans le chat était simplement déposée comme un fichier dans le dossier de travail, à charge pour le modèle de l'ouvrir avec ses outils (ce qui ne donnait qu'un tas d'octets illisibles). Désormais, quand un projecteur est configuré, l'image part au modèle en contenu multimodal : il la voit, et peut en parler.
|
||||
|
||||
## Mise à jour
|
||||
|
||||
```
|
||||
ajean update
|
||||
```
|
||||
|
||||
Non vérifié sur cette version : le résultat final dépend d'un modèle vision et de son projecteur compatibles (Qwen2.5-VL, Gemma 3, etc.). À noter, l'image reste dans l'historique de la conversation et repart au moteur à chaque tour tant que la conversation dure.
|
||||
Whitespace-only changes.
@@ -1,517 +0,0 @@
|
||||
"""Boucle agentique : tool-calling itératif au-dessus d'Ollama.
|
||||
|
||||
L'agent dialogue avec le modèle ; quand celui-ci demande un outil, on l'exécute,
|
||||
on réinjecte le résultat, et on reboucle jusqu'à une réponse finale (ou la
|
||||
limite d'itérations). La fonction est un générateur asynchrone d'événements
|
||||
relayés tels quels au client via SSE.
|
||||
|
||||
Événements émis :
|
||||
token {content} — fragment de texte de l'agent
|
||||
tool_call {name, args} — début d'exécution d'un outil
|
||||
tool_result {name, args, summary, status}
|
||||
final {content, tools} — réponse complète + récap des outils
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
from typing import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from . import coder
|
||||
from .ollama_client import OllamaError, ollama
|
||||
from .tools import TOOL_DEFINITIONS, ToolError, run_tool
|
||||
|
||||
MAX_ITERATIONS = 6
|
||||
MAX_TOOL_REPAIR_ATTEMPTS = 2
|
||||
|
||||
# Coupe-circuit de réflexion : au-delà de cette taille de pensée SANS aucun
|
||||
# contenu ni appel d'outil, on interrompt la génération en cours au lieu
|
||||
# d'attendre l'épuisement du budget num_predict (minutes sur un gros modèle).
|
||||
_MAX_THINKING_CHARS = 12_000
|
||||
|
||||
# Élagage : au-delà de cette taille, un résultat d'outil des itérations
|
||||
# passées est tronqué. Les gros payloads (MCP 8 Ko, shell 4 Ko) saturaient le
|
||||
# contexte en un seul tour long — Ollama tronquait alors silencieusement le
|
||||
# DÉBUT de la conversation, faisant « oublier » la consigne au modèle.
|
||||
_PRUNE_KEEP_CHARS = 350
|
||||
|
||||
|
||||
def _prune_old_tool_results(convo: list[dict], before_index: int) -> None:
|
||||
"""Compacte les résultats d'outils déjà consommés par le modèle.
|
||||
|
||||
Seuls les messages ``tool`` antérieurs à ``before_index`` (donc traités
|
||||
lors d'une itération précédente) sont tronqués ; le dernier lot reste
|
||||
intact, c'est celui auquel le modèle répond.
|
||||
"""
|
||||
for msg in convo[:before_index]:
|
||||
content = msg.get("content", "")
|
||||
if msg.get("role") == "tool" and len(content) > _PRUNE_KEEP_CHARS:
|
||||
msg["content"] = (
|
||||
content[:_PRUNE_KEEP_CHARS]
|
||||
+ "… [résultat archivé — déjà traité, ne pas redemander]"
|
||||
)
|
||||
|
||||
|
||||
def _tools_not_supported(exc: OllamaError) -> bool:
|
||||
"""Détecte un modèle incapable de function calling.
|
||||
|
||||
Deux cas :
|
||||
- Ollama refuse explicitement (« does not support tools ») ;
|
||||
- Ollama n'arrive pas à dériver un parseur d'appels d'outils du template
|
||||
du modèle (« Unable to generate parser for this template ») — fréquent
|
||||
sur des modèles exotiques dont le template Jinja lève une exception.
|
||||
Dans les deux cas, on retombe sur une conversation simple, sans outils.
|
||||
"""
|
||||
message = str(exc).lower()
|
||||
return (
|
||||
"does not support tools" in message
|
||||
or "does not support tool" in message
|
||||
or "unable to generate parser for this template" in message
|
||||
or "automatic parser generation failed" in message
|
||||
)
|
||||
|
||||
|
||||
def _thinking_not_supported(exc: OllamaError) -> bool:
|
||||
"""Le modèle (ou son template) refuse le paramètre ``think``."""
|
||||
message = str(exc).lower()
|
||||
return "does not support thinking" in message or "thinking is not supported" in message
|
||||
|
||||
|
||||
def _invalid_tool_arguments(exc: OllamaError) -> bool:
|
||||
message = str(exc).lower()
|
||||
return (
|
||||
"invalid tool call arguments" in message
|
||||
or "unexpected end of json" in message
|
||||
or "failed to parse tool" in message
|
||||
)
|
||||
|
||||
|
||||
def _parse_args(raw) -> dict:
|
||||
if isinstance(raw, dict):
|
||||
return raw
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
return {}
|
||||
|
||||
|
||||
# Marqueur d'avancement du plan émis par le modèle (« ✅ Étape 2 terminée »).
|
||||
# Tolérant : coche/croix optionnelle, mot « étape » optionnel, numéro requis.
|
||||
_STEP_DONE = re.compile(
|
||||
r"(?:✅|✔|☑|\[x\])\s*(?:étape|etape|step)?\s*(\d{1,2})"
|
||||
r"|(?:étape|etape|step)\s*(\d{1,2})\s*(?:terminée|terminee|faite|ok|✅|✔)",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _scan_plan_done(text: str, plan_len: int, already: set[int]) -> list[int]:
|
||||
"""Indices (0-based) de nouvelles étapes annoncées terminées dans ``text``."""
|
||||
fresh: list[int] = []
|
||||
for m in _STEP_DONE.finditer(text):
|
||||
num = m.group(1) or m.group(2)
|
||||
idx = int(num) - 1
|
||||
if 0 <= idx < plan_len and idx not in already:
|
||||
already.add(idx)
|
||||
fresh.append(idx)
|
||||
return fresh
|
||||
|
||||
|
||||
async def run_agent(
|
||||
model: str,
|
||||
convo: list[dict],
|
||||
*,
|
||||
options: dict | None = None,
|
||||
enabled_tools: list[str] | None = None,
|
||||
confirm_shell: bool = True,
|
||||
think: bool = True,
|
||||
keep_alive: str | None = None,
|
||||
mcp_tools: list[dict] | None = None,
|
||||
plan: list[str] | None = None,
|
||||
) -> AsyncIterator[dict]:
|
||||
# enabled_tools=None -> tous les outils ; liste vide -> aucun outil.
|
||||
if enabled_tools is None:
|
||||
tools = list(TOOL_DEFINITIONS)
|
||||
elif enabled_tools:
|
||||
tools = [
|
||||
t for t in TOOL_DEFINITIONS if t["function"]["name"] in enabled_tools
|
||||
]
|
||||
else:
|
||||
tools = []
|
||||
# Outils MCP des serveurs activés (déjà résolus par l'appelant).
|
||||
tools.extend(mcp_tools or [])
|
||||
if not tools:
|
||||
tools = None
|
||||
collected: list[dict] = []
|
||||
text_parts: list[str] = []
|
||||
thinking_parts: list[str] = []
|
||||
active_tools = tools
|
||||
tool_fallback_used = False
|
||||
tool_repair_attempts = 0
|
||||
request_options = dict(options or {})
|
||||
# On n'envoie ``think`` que pour le DÉSACTIVER (False) ; laissé à None, le
|
||||
# modèle garde son comportement par défaut. Repli si le modèle le refuse.
|
||||
request_think: bool | None = None if think else False
|
||||
# Mode réflexion : la pensée consomme le budget num_predict. Sans marge,
|
||||
# le modèle « pense » tout son quota et s'arrête sans agir ni répondre.
|
||||
if request_think is None and think:
|
||||
request_options["num_predict"] = max(
|
||||
int(request_options.get("num_predict") or 0), 6144
|
||||
)
|
||||
# Compteur d'itérations où le modèle n'a produit QUE du raisonnement.
|
||||
thinking_only_strikes = 0
|
||||
|
||||
# Suivi de l'avancement du plan : étapes déjà annoncées terminées.
|
||||
plan_len = len(plan or [])
|
||||
plan_done: set[int] = set()
|
||||
|
||||
# Métriques cumulées sur tous les appels Ollama du tour agentique : Ollama
|
||||
# les renvoie dans le chunk final (done=true) de chaque génération.
|
||||
stats = {"eval_count": 0, "eval_duration": 0, "prompt_eval_count": 0}
|
||||
|
||||
def _accumulate(chunk: dict) -> None:
|
||||
stats["eval_count"] += chunk.get("eval_count") or 0
|
||||
stats["eval_duration"] += chunk.get("eval_duration") or 0
|
||||
stats["prompt_eval_count"] += chunk.get("prompt_eval_count") or 0
|
||||
|
||||
# Index du début du dernier lot de résultats d'outils (à préserver).
|
||||
last_batch_start = 0
|
||||
|
||||
try:
|
||||
for _ in range(MAX_ITERATIONS):
|
||||
content_buf = ""
|
||||
thinking_buf = ""
|
||||
thinking_status_sent = False
|
||||
tool_calls: list[dict] = []
|
||||
|
||||
# Compacte les résultats d'outils des itérations antérieures :
|
||||
# garde le contexte court, la consigne système jamais tronquée.
|
||||
if last_batch_start:
|
||||
_prune_old_tool_results(convo, last_batch_start)
|
||||
|
||||
# Un modèle peut savoir discuter sans supporter les outils. Ollama
|
||||
# refuse alors la requête entière : on retente une fois en chat simple.
|
||||
while True:
|
||||
try:
|
||||
async for chunk in ollama.chat(
|
||||
model,
|
||||
convo,
|
||||
tools=active_tools,
|
||||
options=request_options,
|
||||
think=request_think,
|
||||
keep_alive=keep_alive,
|
||||
stream=True,
|
||||
):
|
||||
msg = chunk.get("message", {})
|
||||
token = msg.get("content", "")
|
||||
thinking = msg.get("thinking", "")
|
||||
if thinking:
|
||||
thinking_buf += thinking
|
||||
# Diffuse le raisonnement en direct pour l'afficher
|
||||
# dans le panneau repliable du chat.
|
||||
yield {"type": "thinking", "content": thinking}
|
||||
if not thinking_status_sent:
|
||||
thinking_status_sent = True
|
||||
yield {"type": "status", "message": "Réflexion…"}
|
||||
# Pensée interminable sans production : on coupe la
|
||||
# génération maintenant — la relance (plus bas)
|
||||
# remet le modèle au travail immédiatement.
|
||||
if (
|
||||
len(thinking_buf) > _MAX_THINKING_CHARS
|
||||
and not content_buf
|
||||
and not tool_calls
|
||||
):
|
||||
break
|
||||
if token:
|
||||
content_buf += token
|
||||
yield {"type": "token", "content": token}
|
||||
# Coche les étapes du plan annoncées terminées, en
|
||||
# direct. Scan borné : uniquement quand le token
|
||||
# porte un marqueur plausible.
|
||||
if plan_len and any(
|
||||
c in token for c in ("✅", "✔", "☑", "tape", "step", "]")
|
||||
):
|
||||
for idx in _scan_plan_done(content_buf, plan_len, plan_done):
|
||||
yield {"type": "plan_step", "index": idx, "status": "done"}
|
||||
if msg.get("tool_calls"):
|
||||
tool_calls.extend(msg["tool_calls"])
|
||||
if chunk.get("done"):
|
||||
_accumulate(chunk)
|
||||
break
|
||||
break
|
||||
except OllamaError as exc:
|
||||
if (
|
||||
active_tools
|
||||
and tool_repair_attempts < MAX_TOOL_REPAIR_ATTEMPTS
|
||||
and not content_buf
|
||||
and not tool_calls
|
||||
and _invalid_tool_arguments(exc)
|
||||
):
|
||||
tool_repair_attempts += 1
|
||||
request_options["num_predict"] = max(
|
||||
int(request_options.get("num_predict", 0)), 4096
|
||||
)
|
||||
thinking_buf = ""
|
||||
# NB : on réinjecte ce rappel en `user`, pas en `system`.
|
||||
# Beaucoup de templates (Gemma, Mistral…) lèvent
|
||||
# « System message must be at the beginning » dès qu'un
|
||||
# message system apparaît ailleurs qu'en tête, ce qui
|
||||
# ferait échouer toute la requête en 400.
|
||||
convo.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"L'appel d'outil précédent contenait un JSON "
|
||||
"tronqué. Réessaie immédiatement avec des arguments "
|
||||
"JSON valides. Pour un fichier long, utilise write_file "
|
||||
"en plusieurs appels : overwrite puis append, avec des "
|
||||
"morceaux courts et complets."
|
||||
),
|
||||
}
|
||||
)
|
||||
yield {
|
||||
"type": "notice",
|
||||
"message": (
|
||||
"Appel d'outil tronqué : nouvelle tentative "
|
||||
f"{tool_repair_attempts}/{MAX_TOOL_REPAIR_ATTEMPTS}."
|
||||
),
|
||||
}
|
||||
continue
|
||||
if (
|
||||
active_tools
|
||||
and not tool_fallback_used
|
||||
and not content_buf
|
||||
and not tool_calls
|
||||
and _tools_not_supported(exc)
|
||||
):
|
||||
active_tools = None
|
||||
tool_fallback_used = True
|
||||
yield {
|
||||
"type": "notice",
|
||||
"message": (
|
||||
"Ce modèle ne supporte pas les outils ; "
|
||||
"réponse en mode conversation simple."
|
||||
),
|
||||
}
|
||||
continue
|
||||
if request_think is not None and _thinking_not_supported(exc):
|
||||
# Le modèle n'accepte pas qu'on désactive sa réflexion :
|
||||
# on retire le paramètre et on relance.
|
||||
request_think = None
|
||||
continue
|
||||
raise
|
||||
|
||||
# Itération « réflexion seule » : ni contenu, ni appel d'outil —
|
||||
# le modèle a brûlé sa génération à penser. Sans relance, la
|
||||
# boucle s'arrêtait là et la tâche restait inachevée.
|
||||
if thinking_buf and not content_buf.strip() and not tool_calls:
|
||||
thinking_parts.append(thinking_buf)
|
||||
thinking_only_strikes += 1
|
||||
if thinking_only_strikes == 1:
|
||||
convo.append({
|
||||
"role": "user",
|
||||
"content": (
|
||||
"Tu n'as produit que du raisonnement, sans réponse "
|
||||
"ni action. Continue la tâche MAINTENANT : appelle "
|
||||
"l'outil suivant ou donne ta réponse finale, sans "
|
||||
"réfléchir davantage."
|
||||
),
|
||||
})
|
||||
yield {"type": "status", "message": "Relance après réflexion…"}
|
||||
continue
|
||||
# Deuxième fois : la réflexion est coupée pour finir la tâche.
|
||||
request_think = False
|
||||
yield {
|
||||
"type": "notice",
|
||||
"message": (
|
||||
"Réflexion désactivée pour ce tour : le modèle "
|
||||
"n'avançait plus."
|
||||
),
|
||||
}
|
||||
continue
|
||||
|
||||
assistant_turn: dict = {"role": "assistant", "content": content_buf}
|
||||
if thinking_buf:
|
||||
# Gardée pour l'affichage (panneau repliable), mais JAMAIS
|
||||
# renvoyée au modèle : la re-soumettre gonflait le contexte à
|
||||
# chaque itération des longues tâches.
|
||||
thinking_parts.append(thinking_buf)
|
||||
if tool_calls:
|
||||
assistant_turn["tool_calls"] = tool_calls
|
||||
convo.append(assistant_turn)
|
||||
if content_buf.strip():
|
||||
text_parts.append(content_buf.strip())
|
||||
|
||||
if not tool_calls:
|
||||
break
|
||||
|
||||
# Les résultats du lot qui suit commencent ici : ils restent
|
||||
# intacts au prochain tour, les précédents seront compactés.
|
||||
last_batch_start = len(convo)
|
||||
|
||||
# Exécution des outils demandés, puis réinjection des résultats.
|
||||
awaiting_confirmation = False
|
||||
for tc in tool_calls:
|
||||
fn = tc.get("function", {})
|
||||
name = fn.get("name", "")
|
||||
args = _parse_args(fn.get("arguments"))
|
||||
|
||||
yield {"type": "tool_call", "name": name, "args": args}
|
||||
|
||||
# run_shell est sensible : on demande validation au lieu d'exécuter.
|
||||
if name == "run_shell" and confirm_shell:
|
||||
command = args.get("command", "")
|
||||
record = {
|
||||
"name": name,
|
||||
"args": args,
|
||||
"summary": "validation requise",
|
||||
"status": "pending",
|
||||
}
|
||||
collected.append(record)
|
||||
yield {"type": "tool_confirm", "name": name, "command": command}
|
||||
convo.append(
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_name": name,
|
||||
"content": json.dumps(
|
||||
{
|
||||
"ok": False,
|
||||
"status": "pending",
|
||||
"message": "Commande en attente de validation "
|
||||
"de l'utilisateur. N'exécute rien d'autre.",
|
||||
},
|
||||
ensure_ascii=False,
|
||||
),
|
||||
}
|
||||
)
|
||||
awaiting_confirmation = True
|
||||
continue
|
||||
|
||||
# code_task : délégué au moteur code (Aider), long -> thread.
|
||||
if name == "code_task":
|
||||
from .tools import active_root
|
||||
code_model = await coder.pick_code_model(model)
|
||||
result = await asyncio.to_thread(
|
||||
coder.run_code_task,
|
||||
args.get("instruction", ""),
|
||||
code_model,
|
||||
args.get("files") or [],
|
||||
active_root(),
|
||||
)
|
||||
summary = result.get("summary", "terminé")
|
||||
status = "ok" if result.get("ok") else "error"
|
||||
record = {"name": name, "args": {"instruction": args.get("instruction", "")},
|
||||
"summary": summary, "status": status}
|
||||
collected.append(record)
|
||||
yield {"type": "tool_result", **record}
|
||||
convo.append({
|
||||
"role": "tool",
|
||||
"tool_name": name,
|
||||
"content": json.dumps(
|
||||
{k: result.get(k) for k in ("ok", "summary", "files", "text")},
|
||||
ensure_ascii=False,
|
||||
),
|
||||
})
|
||||
continue
|
||||
|
||||
# Outils MCP : dispatch asynchrone vers le serveur concerné.
|
||||
if name.startswith("mcp_"):
|
||||
from .mcp_client import manager as mcp_manager
|
||||
mcp_res = await mcp_manager.call_tool(name, args)
|
||||
status = "ok" if mcp_res["ok"] else "error"
|
||||
record = {"name": name, "args": args,
|
||||
"summary": mcp_res["summary"], "status": status}
|
||||
collected.append(record)
|
||||
yield {"type": "tool_result", **record}
|
||||
convo.append({
|
||||
"role": "tool",
|
||||
"tool_name": name,
|
||||
"content": json.dumps(
|
||||
{"ok": mcp_res["ok"], "content": mcp_res["content"]},
|
||||
ensure_ascii=False,
|
||||
),
|
||||
})
|
||||
continue
|
||||
|
||||
try:
|
||||
result = run_tool(name, args)
|
||||
summary = result.get("summary", "terminé")
|
||||
status = result.get("_status", "ok")
|
||||
except ToolError as exc:
|
||||
result = {"ok": False, "error": str(exc)}
|
||||
summary = str(exc)
|
||||
status = "error"
|
||||
|
||||
record = {"name": name, "args": args, "summary": summary, "status": status}
|
||||
collected.append(record)
|
||||
yield {"type": "tool_result", **record}
|
||||
|
||||
convo.append(
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_name": name,
|
||||
"content": json.dumps(result, ensure_ascii=False),
|
||||
}
|
||||
)
|
||||
|
||||
# Une commande shell attend une validation : on interrompt la boucle.
|
||||
if awaiting_confirmation:
|
||||
# Laisse le modèle conclure son tour (message d'attente).
|
||||
final_chunk = ""
|
||||
async for chunk in ollama.chat(
|
||||
model, convo, options=request_options,
|
||||
think=request_think, keep_alive=keep_alive, stream=True
|
||||
):
|
||||
tok = chunk.get("message", {}).get("content", "")
|
||||
if tok:
|
||||
final_chunk += tok
|
||||
yield {"type": "token", "content": tok}
|
||||
if chunk.get("done"):
|
||||
_accumulate(chunk)
|
||||
break
|
||||
if final_chunk.strip():
|
||||
text_parts.append(final_chunk.strip())
|
||||
break
|
||||
except OllamaError as exc:
|
||||
# Échec signalé par Ollama (HTTP ou ligne d'erreur dans le flux) :
|
||||
# souvent un débordement mémoire / contexte trop grand. On l'expose.
|
||||
yield {"type": "error", "message": f"Ollama : {exc}"}
|
||||
return
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
yield {
|
||||
"type": "error",
|
||||
"message": f"Impossible de joindre Ollama ({ollama.host}) : {exc}",
|
||||
}
|
||||
return
|
||||
|
||||
final_content = "\n\n".join(text_parts).strip()
|
||||
if not final_content and not collected:
|
||||
yield {
|
||||
"type": "error",
|
||||
"message": (
|
||||
"Le modèle a terminé sans renvoyer de texte (il n'a produit que "
|
||||
"du raisonnement). Désactive « Mode réflexion » dans les Réglages, "
|
||||
"ou essaie un modèle de chat plus récent."
|
||||
),
|
||||
}
|
||||
return
|
||||
|
||||
eval_secs = stats["eval_duration"] / 1e9
|
||||
final_stats = {
|
||||
"eval_count": stats["eval_count"],
|
||||
"prompt_eval_count": stats["prompt_eval_count"],
|
||||
"tokens_per_sec": (
|
||||
round(stats["eval_count"] / eval_secs, 1) if eval_secs > 0 else None
|
||||
),
|
||||
}
|
||||
yield {
|
||||
"type": "final",
|
||||
"content": final_content,
|
||||
"tools": collected,
|
||||
"stats": final_stats,
|
||||
"thinking": "\n\n".join(thinking_parts).strip(),
|
||||
}
|
||||
@@ -1,309 +0,0 @@
|
||||
"""Profil de configuration de l'agent : invite système, génération, outils.
|
||||
|
||||
Persisté en base sous la clé `agent`. Fournit les valeurs par défaut et la
|
||||
fusion avec ce qui est stocké, pour rester robuste aux montées de version.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from . import db
|
||||
|
||||
CONFIG_KEY = "agent"
|
||||
MODEL_PROFILES_KEY = "model_profiles"
|
||||
PROFILE_STATE_KEY = "model_profiles_state"
|
||||
PROFILE_VERSION = 7
|
||||
|
||||
DEFAULT_SYSTEM_PROMPT = (
|
||||
"Tu es Loki, un assistant de développement local agentique. Tu disposes "
|
||||
"d'outils pour lire, écrire et lister des fichiers dans le workspace, et "
|
||||
"d'un moteur code (code_task) pour toute création ou modification de code "
|
||||
"multi-fichiers : privilégie code_task pour les tâches de programmation. "
|
||||
"Utilise les outils pour accomplir les tâches concrètement, puis réponds "
|
||||
"de façon concise en français. Pour MODIFIER un fichier existant, utilise "
|
||||
"edit_file (search/replace) plutôt que de tout réécrire : lis d'abord le "
|
||||
"fichier avec read_file, puis copie dans `search` l'extrait EXACT à changer "
|
||||
"(quelques lignes suffisent, l'indentation est tolérée). N'emploie "
|
||||
"write_file en overwrite QUE pour créer un nouveau fichier ou en cas de "
|
||||
"refonte complète. Quand tu appelles write_file, fournis toujours `path` et "
|
||||
"`content` ; pour un fichier long, appelle write_file en plusieurs morceaux "
|
||||
"(overwrite puis append) afin de toujours produire un JSON valide. Tu ne "
|
||||
"peux écrire QUE dans le workspace : jamais de chemin absolu ni de `../` "
|
||||
"qui en sortent. Après avoir écrit un fichier, propose un aperçu. "
|
||||
"Formate TOUJOURS tes réponses en Markdown : titres, listes, gras pour les "
|
||||
"points clés, tableaux si pertinent, et surtout des blocs de code avec le "
|
||||
"langage indiqué (```python, ```html…) pour tout extrait de code ou commande."
|
||||
)
|
||||
|
||||
# Outils disponibles. Les sensibles (web_search, run_shell) sont désactivés
|
||||
# par défaut, conformément à la maquette.
|
||||
AVAILABLE_TOOLS = [
|
||||
"read_file", "write_file", "edit_file", "list_dir", "grep_search",
|
||||
"run_check", "code_task", "web_search", "run_shell",
|
||||
]
|
||||
SENSITIVE_TOOLS = {"run_shell"}
|
||||
DEFAULT_TOOL_STATE = {
|
||||
"read_file": True,
|
||||
"write_file": True,
|
||||
"edit_file": True,
|
||||
"list_dir": True,
|
||||
"grep_search": True,
|
||||
"run_check": True,
|
||||
"code_task": True,
|
||||
"web_search": False,
|
||||
"run_shell": False,
|
||||
}
|
||||
|
||||
GENERATION_FIELDS = {
|
||||
"temperature",
|
||||
"top_p",
|
||||
"top_k",
|
||||
"max_tokens",
|
||||
"num_ctx",
|
||||
"num_gpu",
|
||||
"num_batch",
|
||||
}
|
||||
PROFILE_FIELDS = {
|
||||
"system_prompt",
|
||||
"tools",
|
||||
"confirm_shell",
|
||||
"think",
|
||||
"code_model",
|
||||
"plan_mode",
|
||||
"self_review",
|
||||
"rag_enabled",
|
||||
"embed_model",
|
||||
"memory_mode",
|
||||
"skills_enabled",
|
||||
"ponytail",
|
||||
"keep_alive",
|
||||
*GENERATION_FIELDS,
|
||||
}
|
||||
|
||||
DEFAULT_GENERATION: dict = {
|
||||
"temperature": 0.7,
|
||||
"top_p": 0.9,
|
||||
"top_k": 40,
|
||||
"max_tokens": 2048,
|
||||
# 16k : les tâches longues (recherche, multi-fichiers) saturaient 4-8k en
|
||||
# un seul tour et Ollama tronquait la consigne. Le cache KV d'un 16k reste
|
||||
# raisonnable (~1-3 Go selon modèle) ; réduis num_ctx dans Configuration
|
||||
# si la VRAM déborde, ou active OLLAMA_KV_CACHE_TYPE=q8_0 côté Ollama.
|
||||
"num_ctx": 16384,
|
||||
"num_gpu": -1,
|
||||
"num_batch": 256,
|
||||
}
|
||||
|
||||
RTX_3060_GEMMA4_PROFILE: dict = {
|
||||
**DEFAULT_GENERATION,
|
||||
"max_tokens": 4096,
|
||||
"num_ctx": 16384,
|
||||
# num_gpu = -1 : laisse Ollama placer le plus de couches possible sur le GPU
|
||||
# (auto-fit, comme `ollama run`). Forcer un nombre de couches qui ne tient pas
|
||||
# en VRAM fait basculer toute l'inférence sur le CPU.
|
||||
"num_gpu": -1,
|
||||
}
|
||||
|
||||
DEFAULT_CONFIG: dict = {
|
||||
"system_prompt": DEFAULT_SYSTEM_PROMPT,
|
||||
**DEFAULT_GENERATION,
|
||||
"tools": dict(DEFAULT_TOOL_STATE),
|
||||
# Demander une validation utilisateur avant toute commande shell.
|
||||
"confirm_shell": True,
|
||||
# Mode réflexion des modèles « thinking ». Désactiver (False) évite qu'un
|
||||
# modèle ne renvoie que du raisonnement sans réponse finale.
|
||||
"think": True,
|
||||
# Modèle utilisé par le moteur code : "auto" = meilleur modèle code installé
|
||||
# (qwen-coder, deepseek-coder…), sinon le modèle de chat courant.
|
||||
"code_model": "auto",
|
||||
# Plan-puis-exécute : décompose les demandes complexes en étapes.
|
||||
"plan_mode": True,
|
||||
# Auto-critique : une passe de relecture/révision avant la réponse finale.
|
||||
"self_review": False,
|
||||
# Mémoire long-terme (RAG) ENTRE sessions, via un modèle d'embedding.
|
||||
# Désactivée par défaut : chaque discussion ne se souvient que d'elle-même
|
||||
# (résumé + messages récents). Sinon une ancienne demande sans rapport (ex.
|
||||
# « appli sport ») ressurgit dans une nouvelle discussion (ex. « jeu
|
||||
# d'échecs ») et embrouille les petits modèles. Réactivable dans Réglages.
|
||||
"rag_enabled": False,
|
||||
"embed_model": "auto",
|
||||
# Mémoire en notes Markdown écrites par l'agent (voir memory_notes.py).
|
||||
# "ondemand" : les outils existent, mais rien n'est injecté sans que le
|
||||
# modèle le demande — aucun risque de voir ressurgir une demande sans
|
||||
# rapport, contrairement à l'ancien RAG vectoriel.
|
||||
"memory_mode": "ondemand",
|
||||
# Skills : méthodes expertes injectées automatiquement selon la tâche.
|
||||
"skills_enabled": True,
|
||||
# Ponytail : méthode « code minimal » (anti sur-ingénierie) injectée pour
|
||||
# toute tâche de code. Voir github.com/DietrichGebert/ponytail.
|
||||
"ponytail": True,
|
||||
# Durée de maintien du modèle en VRAM (préchargement). "0" = décharge
|
||||
# aussitôt, "-1" = jamais, "30m" = 30 minutes.
|
||||
"keep_alive": "30m",
|
||||
}
|
||||
|
||||
|
||||
def _default_generation(model: str | None) -> dict:
|
||||
if model and model.split(":", 1)[0].lower() == "gemma4":
|
||||
return dict(RTX_3060_GEMMA4_PROFILE)
|
||||
return dict(DEFAULT_GENERATION)
|
||||
|
||||
|
||||
def _migrate_profiles() -> None:
|
||||
state = db.get_config_value(PROFILE_STATE_KEY) or {}
|
||||
if state.get("version", 0) >= PROFILE_VERSION:
|
||||
return
|
||||
profiles = db.get_config_value(MODEL_PROFILES_KEY) or {}
|
||||
gemma_profile = {
|
||||
**RTX_3060_GEMMA4_PROFILE,
|
||||
**profiles.get("gemma4:12b", {}),
|
||||
}
|
||||
if gemma_profile.get("max_tokens", 0) <= 2048:
|
||||
gemma_profile["max_tokens"] = 4096
|
||||
profiles["gemma4:12b"] = gemma_profile
|
||||
# v4 : un ancien profil pouvait forcer num_gpu sur un nombre de couches codé
|
||||
# en dur (ex. 49), ce qui basculait l'inférence sur le CPU quand ça ne tenait
|
||||
# pas en VRAM. On repasse en auto (-1) pour laisser Ollama placer les couches.
|
||||
for prof in profiles.values():
|
||||
if prof.get("num_gpu", -1) is not None and prof.get("num_gpu", -1) > 0:
|
||||
prof["num_gpu"] = -1
|
||||
# v6 : contexte 16k par défaut. On ne touche qu'aux profils restés sur un
|
||||
# ancien défaut (4096/8192) — une valeur personnalisée est respectée.
|
||||
for prof in profiles.values():
|
||||
if prof.get("num_ctx") in (4096, 8192):
|
||||
prof["num_ctx"] = 16384
|
||||
db.set_config_value(MODEL_PROFILES_KEY, profiles)
|
||||
|
||||
# v7 : la mémoire inter-sessions (RAG) faisait ressurgir d'anciennes
|
||||
# demandes sans rapport dans une nouvelle discussion. On la désactive une
|
||||
# fois sur les installs existantes ; réactivable manuellement dans Réglages.
|
||||
stored = db.get_config_value(CONFIG_KEY)
|
||||
if stored and stored.get("rag_enabled"):
|
||||
stored["rag_enabled"] = False
|
||||
db.set_config_value(CONFIG_KEY, stored)
|
||||
|
||||
db.set_config_value(PROFILE_STATE_KEY, {"version": PROFILE_VERSION})
|
||||
|
||||
|
||||
def _clean_tools(value: dict | None, fallback: dict | None = None) -> dict:
|
||||
fallback = fallback or DEFAULT_TOOL_STATE
|
||||
return {
|
||||
name: bool((value or {}).get(name, fallback.get(name, DEFAULT_TOOL_STATE[name])))
|
||||
for name in AVAILABLE_TOOLS
|
||||
}
|
||||
|
||||
|
||||
def get_config(model: str | None = None) -> dict:
|
||||
"""Configuration complète, avec surcharge sauvegardée par modèle."""
|
||||
_migrate_profiles()
|
||||
stored = db.get_config_value(CONFIG_KEY) or {}
|
||||
cfg = {**DEFAULT_CONFIG, **stored}
|
||||
cfg["tools"] = _clean_tools(stored.get("tools"))
|
||||
if model:
|
||||
profiles = db.get_config_value(MODEL_PROFILES_KEY) or {}
|
||||
profile = profiles.get(model, {})
|
||||
cfg.update({**_default_generation(model), **profile})
|
||||
cfg["tools"] = _clean_tools(profile.get("tools"), cfg["tools"])
|
||||
return cfg
|
||||
|
||||
|
||||
def save_config(patch: dict, model: str | None = None) -> dict:
|
||||
"""Sauvegarde tous les réglages, globalement ou pour un modèle."""
|
||||
clean = {k: v for k, v in patch.items() if v is not None}
|
||||
cfg = {**get_config(model), **clean}
|
||||
if "tools" in patch and patch["tools"]:
|
||||
cfg["tools"] = _clean_tools(patch["tools"], cfg["tools"])
|
||||
|
||||
if model:
|
||||
profiles = db.get_config_value(MODEL_PROFILES_KEY) or {}
|
||||
profiles[model] = {field: cfg[field] for field in PROFILE_FIELDS}
|
||||
db.set_config_value(MODEL_PROFILES_KEY, profiles)
|
||||
else:
|
||||
db.set_config_value(CONFIG_KEY, {field: cfg[field] for field in PROFILE_FIELDS})
|
||||
return get_config(model)
|
||||
|
||||
|
||||
def runner_options(cfg: dict) -> dict:
|
||||
"""Sous-ensemble d'options qui détermine l'identité du runner Ollama.
|
||||
|
||||
Ollama choisit son runner (processus de chargement du modèle) d'après
|
||||
num_ctx / num_batch / num_gpu. TOUT appel au même modèle (plan, résumé,
|
||||
agent…) doit envoyer ces mêmes valeurs, sinon Ollama recharge le modèle en
|
||||
plein milieu d'un message — la cause principale des lenteurs observées.
|
||||
"""
|
||||
opts: dict = {"num_batch": cfg["num_batch"]}
|
||||
# num_gpu n'est transmis que si l'utilisateur force explicitement un nombre
|
||||
# de couches (≥ 0). En -1 (défaut), on laisse Ollama auto-ajuster l'offload
|
||||
# GPU comme `ollama run` ; lui imposer une valeur peut le forcer sur le CPU.
|
||||
if cfg.get("num_gpu", -1) >= 0:
|
||||
opts["num_gpu"] = cfg["num_gpu"]
|
||||
opts["main_gpu"] = 0
|
||||
# num_ctx n'est envoyé que s'il est défini (> 0), sinon défaut du modèle.
|
||||
if cfg.get("num_ctx"):
|
||||
opts["num_ctx"] = cfg["num_ctx"]
|
||||
return opts
|
||||
|
||||
|
||||
def ollama_options(cfg: dict) -> dict:
|
||||
"""Traduit la config en options de génération Ollama."""
|
||||
return {
|
||||
**runner_options(cfg),
|
||||
"temperature": cfg["temperature"],
|
||||
"top_p": cfg["top_p"],
|
||||
"top_k": cfg["top_k"],
|
||||
"num_predict": cfg["max_tokens"],
|
||||
}
|
||||
|
||||
|
||||
# ── Presets : jeux de réglages nommés ────────────────────────────────────
|
||||
PRESETS_KEY = "config_presets"
|
||||
_PRESET_NAME = re.compile(r"^[\w \-]{1,40}$")
|
||||
|
||||
|
||||
def list_presets() -> list[str]:
|
||||
return sorted((db.get_config_value(PRESETS_KEY) or {}).keys())
|
||||
|
||||
|
||||
def save_preset(name: str, model: str | None = None) -> list[str]:
|
||||
"""Fige la configuration courante sous ce nom (écrase si déjà pris)."""
|
||||
name = (name or "").strip()
|
||||
if not _PRESET_NAME.match(name):
|
||||
raise ValueError("nom de preset invalide (lettres, chiffres, espaces, -)")
|
||||
presets = db.get_config_value(PRESETS_KEY) or {}
|
||||
cfg = get_config(model)
|
||||
presets[name] = {field: cfg[field] for field in PROFILE_FIELDS if field in cfg}
|
||||
db.set_config_value(PRESETS_KEY, presets)
|
||||
return sorted(presets)
|
||||
|
||||
|
||||
def apply_preset(name: str, model: str | None = None) -> dict | None:
|
||||
"""Applique un preset ; None s'il n'existe pas."""
|
||||
preset = (db.get_config_value(PRESETS_KEY) or {}).get((name or "").strip())
|
||||
if preset is None:
|
||||
return None
|
||||
return save_config(dict(preset), model)
|
||||
|
||||
|
||||
def delete_preset(name: str) -> list[str]:
|
||||
presets = db.get_config_value(PRESETS_KEY) or {}
|
||||
presets.pop((name or "").strip(), None)
|
||||
db.set_config_value(PRESETS_KEY, presets)
|
||||
return sorted(presets)
|
||||
|
||||
|
||||
def enabled_tool_names(cfg: dict) -> list[str]:
|
||||
"""Outils réellement proposables au modèle.
|
||||
|
||||
Un outil activé dans la config mais INUTILISABLE (dépendance absente) ne
|
||||
doit pas apparaître dans le contexte : sinon le modèle l'appelle, échoue,
|
||||
et gâche un tour. Mieux vaut qu'il ne sache pas qu'il existe et emploie
|
||||
directement write_file / edit_file.
|
||||
"""
|
||||
names = [name for name, on in cfg["tools"].items() if on]
|
||||
|
||||
if "code_task" in names:
|
||||
from . import coder
|
||||
if not coder.available():
|
||||
names.remove("code_task")
|
||||
|
||||
return names
|
||||
@@ -1,267 +0,0 @@
|
||||
"""Auto-réglage : exploite au mieux chaque modèle selon le GPU disponible.
|
||||
|
||||
Principe :
|
||||
1. on détecte la VRAM du GPU (nvidia-smi, sinon rocm-smi, sinon override env) ;
|
||||
2. on lit les métadonnées du modèle via Ollama (/api/show + /api/tags) :
|
||||
contexte max, architecture (couches, têtes KV, dimension), taille sur disque ;
|
||||
3. on calcule la fenêtre de contexte (num_ctx) la plus grande qui tient en VRAM,
|
||||
via une estimation du cache KV, puis un nombre de jetons de sortie cohérent.
|
||||
|
||||
Tout est best-effort : si une info manque, on retombe sur des paliers prudents.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import subprocess
|
||||
|
||||
import httpx
|
||||
|
||||
from .config import settings
|
||||
from .ollama_client import ollama
|
||||
|
||||
# Paliers de contexte « ronds » proposés (bornés par le contexte du modèle).
|
||||
CTX_STEPS = [2048, 4096, 8192, 12288, 16384, 24576, 32768, 49152, 65536, 131072]
|
||||
|
||||
# Marges VRAM (Mo) : OS/driver + buffers de calcul d'Ollama.
|
||||
VRAM_OVERHEAD_MB = 1024
|
||||
VRAM_COMPUTE_BUFFER_MB = 768
|
||||
|
||||
|
||||
# ── Détection GPU ────────────────────────────────────────────────────────
|
||||
def detect_gpu() -> dict:
|
||||
"""Renvoie {available, name, vram_total_mb, source}."""
|
||||
# 1) Override explicite (env) — prioritaire, utile en conteneur sans GPU.
|
||||
if settings.gpu_vram_mb > 0:
|
||||
return {
|
||||
"available": True,
|
||||
"name": settings.gpu_name or "GPU (configuré)",
|
||||
"vram_total_mb": settings.gpu_vram_mb,
|
||||
"source": "env",
|
||||
}
|
||||
|
||||
# 2) NVIDIA
|
||||
if shutil.which("nvidia-smi"):
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["nvidia-smi",
|
||||
"--query-gpu=name,memory.total",
|
||||
"--format=csv,noheader,nounits"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
if out.returncode == 0 and out.stdout.strip():
|
||||
line = out.stdout.strip().splitlines()[0]
|
||||
name, total = [p.strip() for p in line.split(",")]
|
||||
return {
|
||||
"available": True,
|
||||
"name": name,
|
||||
"vram_total_mb": int(float(total)),
|
||||
"source": "nvidia-smi",
|
||||
}
|
||||
except (OSError, ValueError, subprocess.SubprocessError):
|
||||
pass
|
||||
|
||||
# 3) AMD (ROCm) — best-effort
|
||||
if shutil.which("rocm-smi"):
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["rocm-smi", "--showmeminfo", "vram", "--csv"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
if out.returncode == 0:
|
||||
# Cherche le plus grand entier (octets) -> Mo
|
||||
nums = [int(x) for x in out.stdout.replace(",", " ").split()
|
||||
if x.isdigit()]
|
||||
if nums:
|
||||
return {
|
||||
"available": True,
|
||||
"name": "GPU AMD",
|
||||
"vram_total_mb": max(nums) // (1024 * 1024),
|
||||
"source": "rocm-smi",
|
||||
}
|
||||
except (OSError, ValueError, subprocess.SubprocessError):
|
||||
pass
|
||||
|
||||
return {"available": False, "name": "CPU", "vram_total_mb": 0, "source": "none"}
|
||||
|
||||
|
||||
# ── Métadonnées modèle ───────────────────────────────────────────────────
|
||||
async def model_profile(model: str) -> dict:
|
||||
"""Extrait contexte max, archi (couches/têtes/dim), taille disque (Mo)."""
|
||||
profile = {
|
||||
"context_length": None,
|
||||
"block_count": None,
|
||||
"head_count": None,
|
||||
"head_count_kv": None,
|
||||
"embedding_length": None,
|
||||
"size_mb": None,
|
||||
"parameter_size": None,
|
||||
"quantization": None,
|
||||
}
|
||||
try:
|
||||
show = await ollama.show(model)
|
||||
except (httpx.HTTPError, OSError):
|
||||
return profile
|
||||
|
||||
info = show.get("model_info", {}) or {}
|
||||
arch = info.get("general.architecture", "")
|
||||
|
||||
def pick(*suffixes):
|
||||
for s in suffixes:
|
||||
key = f"{arch}.{s}" if arch else s
|
||||
if key in info:
|
||||
return info[key]
|
||||
for k, v in info.items():
|
||||
if k.endswith(s):
|
||||
return v
|
||||
return None
|
||||
|
||||
profile["context_length"] = pick("context_length")
|
||||
profile["block_count"] = pick("block_count")
|
||||
profile["head_count"] = pick("attention.head_count")
|
||||
profile["head_count_kv"] = pick("attention.head_count_kv")
|
||||
profile["embedding_length"] = pick("embedding_length")
|
||||
|
||||
details = show.get("details", {}) or {}
|
||||
profile["parameter_size"] = details.get("parameter_size")
|
||||
profile["quantization"] = details.get("quantization_level")
|
||||
|
||||
# Taille sur disque via /api/tags
|
||||
try:
|
||||
for m in await ollama.list_models():
|
||||
if m.get("name") == model:
|
||||
profile["size_mb"] = int(m.get("size", 0)) // (1024 * 1024)
|
||||
break
|
||||
except (httpx.HTTPError, OSError):
|
||||
pass
|
||||
|
||||
return profile
|
||||
|
||||
|
||||
def _kv_bytes_per_token(p: dict) -> int | None:
|
||||
"""Estimation des octets de cache KV par jeton (KV en f16)."""
|
||||
blocks = p.get("block_count")
|
||||
kv_heads = p.get("head_count_kv")
|
||||
heads = p.get("head_count")
|
||||
emb = p.get("embedding_length")
|
||||
if not all((blocks, kv_heads, heads, emb)):
|
||||
return None
|
||||
head_dim = emb / heads
|
||||
# 2 (clé+valeur) * couches * têtes_kv * dim_tête * 2 octets (f16)
|
||||
return int(2 * blocks * kv_heads * head_dim * 2)
|
||||
|
||||
|
||||
def _round_ctx(candidate: int, ctx_max: int | None) -> int:
|
||||
cap = ctx_max or CTX_STEPS[-1]
|
||||
best = CTX_STEPS[0]
|
||||
for step in CTX_STEPS:
|
||||
if step <= candidate and step <= cap:
|
||||
best = step
|
||||
# Si le modèle plafonne bas, respecte son contexte max.
|
||||
return min(best, cap)
|
||||
|
||||
|
||||
# ── Recommandation ───────────────────────────────────────────────────────
|
||||
async def recommend(model: str) -> dict:
|
||||
gpu = detect_gpu()
|
||||
prof = await model_profile(model)
|
||||
ctx_max = prof.get("context_length")
|
||||
|
||||
rationale: list[str] = []
|
||||
|
||||
if not gpu["available"]:
|
||||
# GPU non détecté CÔTÉ LOKI : Ollama tourne probablement sur une autre
|
||||
# machine. On ne force surtout PAS num_ctx (0 = défaut du modèle), sinon
|
||||
# Ollama chargerait une instance distincte qui déborderait sur le CPU.
|
||||
num_ctx = 0
|
||||
rationale.append(
|
||||
"GPU non détecté côté Loki (Ollama distant ?) — contexte laissé au "
|
||||
"défaut du modèle pour rester sur le GPU. Déclare GPU_VRAM_MB pour "
|
||||
"un réglage précis."
|
||||
)
|
||||
else:
|
||||
vram = gpu["vram_total_mb"]
|
||||
model_mb = prof.get("size_mb") or _fallback_model_mb(prof)
|
||||
budget = vram - model_mb - VRAM_OVERHEAD_MB - VRAM_COMPUTE_BUFFER_MB
|
||||
kv_per_tok = _kv_bytes_per_token(prof)
|
||||
|
||||
if budget <= 256:
|
||||
num_ctx = _round_ctx(2048, ctx_max)
|
||||
rationale.append("VRAM insuffisante pour les poids — contexte minimal.")
|
||||
elif kv_per_tok:
|
||||
tokens_fit = int(budget * 1024 * 1024 / kv_per_tok)
|
||||
num_ctx = _round_ctx(tokens_fit, ctx_max)
|
||||
rationale.append(
|
||||
f"{vram} Mo VRAM − {model_mb} Mo poids → budget KV "
|
||||
f"{budget} Mo (~{kv_per_tok // 1024} Ko/jeton)."
|
||||
)
|
||||
else:
|
||||
# Archi inconnue : paliers selon le budget VRAM restant.
|
||||
num_ctx = _round_ctx(_tier_ctx(budget), ctx_max)
|
||||
rationale.append("Archi modèle incomplète — estimation par paliers.")
|
||||
|
||||
# Jetons de sortie : moitié du contexte, borné. Si num_ctx=0 (auto), on
|
||||
# garde une valeur raisonnable sans la réduire.
|
||||
max_tokens = 2048 if num_ctx == 0 else max(512, min(num_ctx // 2, 8192))
|
||||
|
||||
return {
|
||||
"gpu": gpu,
|
||||
"model": model,
|
||||
"model_profile": {
|
||||
"context_length": ctx_max,
|
||||
"parameter_size": prof.get("parameter_size"),
|
||||
"quantization": prof.get("quantization"),
|
||||
"size_mb": prof.get("size_mb"),
|
||||
},
|
||||
"recommended": {"num_ctx": num_ctx, "max_tokens": max_tokens},
|
||||
"rationale": " ".join(rationale),
|
||||
}
|
||||
|
||||
|
||||
async def placement(model: str) -> dict:
|
||||
"""Lit /api/ps : le modèle est-il chargé sur GPU, CPU, ou un mix ?"""
|
||||
try:
|
||||
loaded = await ollama.ps()
|
||||
except (httpx.HTTPError, OSError):
|
||||
return {"loaded": False}
|
||||
|
||||
for m in loaded:
|
||||
if m.get("name") == model or m.get("model") == model:
|
||||
size = m.get("size", 0) or 0
|
||||
size_vram = m.get("size_vram", 0) or 0
|
||||
if size <= 0:
|
||||
where = "inconnu"
|
||||
elif size_vram >= size * 0.99:
|
||||
where = "gpu"
|
||||
elif size_vram <= size * 0.01:
|
||||
where = "cpu"
|
||||
else:
|
||||
where = "mixte"
|
||||
pct = int(size_vram / size * 100) if size else 0
|
||||
return {"loaded": True, "where": where, "gpu_percent": pct,
|
||||
"size_mb": size // (1024 * 1024)}
|
||||
return {"loaded": False}
|
||||
|
||||
|
||||
def _fallback_model_mb(prof: dict) -> int:
|
||||
"""Estime la taille des poids si /api/tags n'a rien donné."""
|
||||
ps = (prof.get("parameter_size") or "").upper().replace("B", "")
|
||||
try:
|
||||
billions = float(ps)
|
||||
except ValueError:
|
||||
billions = 8.0
|
||||
# ~0.6 Go/milliard en Q4, approximation prudente.
|
||||
return int(billions * 600)
|
||||
|
||||
|
||||
def _tier_ctx(budget_mb: int) -> int:
|
||||
if budget_mb >= 12000:
|
||||
return 32768
|
||||
if budget_mb >= 8000:
|
||||
return 16384
|
||||
if budget_mb >= 5000:
|
||||
return 12288
|
||||
if budget_mb >= 3000:
|
||||
return 8192
|
||||
if budget_mb >= 1500:
|
||||
return 4096
|
||||
return 2048
|
||||
@@ -1,222 +0,0 @@
|
||||
"""Benchmark intégré : évalue objectivement chaque modèle installé.
|
||||
|
||||
Cinq mini-épreuves (~30-60 s au total) qui mesurent ce qui compte pour Loki :
|
||||
appel d'outil, code exécutable, respect des consignes, extraction JSON,
|
||||
respect d'un format. Score /100, stocké en base et affiché dans l'UI.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from typing import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from . import db
|
||||
from .ollama_client import OllamaError, ollama
|
||||
|
||||
BENCH_KEY = "bench" # config[bench] = {model: {score, details, at}}
|
||||
|
||||
|
||||
async def _ask(model: str, prompt: str, *, system: str = "",
|
||||
tools: list | None = None, num_predict: int = 400) -> dict:
|
||||
"""Un appel modèle ; renvoie {text, tool_calls}."""
|
||||
messages = []
|
||||
if system:
|
||||
messages.append({"role": "system", "content": system})
|
||||
messages.append({"role": "user", "content": prompt})
|
||||
text, calls = "", []
|
||||
async for chunk in ollama.chat(
|
||||
model, messages, tools=tools,
|
||||
options={"temperature": 0, "num_predict": num_predict}, stream=True,
|
||||
):
|
||||
msg = chunk.get("message", {})
|
||||
text += msg.get("content", "")
|
||||
if msg.get("tool_calls"):
|
||||
calls.extend(msg["tool_calls"])
|
||||
if chunk.get("done"):
|
||||
break
|
||||
return {"text": text.strip(), "tool_calls": calls}
|
||||
|
||||
|
||||
def _extract_code(text: str) -> str:
|
||||
m = re.search(r"```(?:python)?\s*(.*?)```", text, re.S)
|
||||
return (m.group(1) if m else text).strip()
|
||||
|
||||
|
||||
def _run_python(code: str, test: str) -> bool:
|
||||
"""Exécute code+test dans un sous-processus isolé (timeout 8 s)."""
|
||||
with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f:
|
||||
f.write(code + "\n" + test)
|
||||
path = f.name
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[sys.executable, "-I", path],
|
||||
capture_output=True, timeout=8,
|
||||
)
|
||||
return proc.returncode == 0
|
||||
except (subprocess.SubprocessError, OSError):
|
||||
return False
|
||||
|
||||
|
||||
# ── Les 5 épreuves (score 0-20 chacune) ──────────────────────────────────
|
||||
async def _task_tool_call(model: str) -> tuple[int, str]:
|
||||
tools = [{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "write_file",
|
||||
"description": "Écrire un fichier",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string"},
|
||||
"content": {"type": "string"},
|
||||
},
|
||||
"required": ["path", "content"],
|
||||
},
|
||||
},
|
||||
}]
|
||||
try:
|
||||
r = await _ask(
|
||||
model,
|
||||
"Crée le fichier bonjour.txt contenant exactement le texte : salut",
|
||||
system="Utilise l'outil write_file pour créer le fichier demandé.",
|
||||
tools=tools, num_predict=200,
|
||||
)
|
||||
except httpx.HTTPStatusError as exc:
|
||||
if "does not support tools" in exc.response.text.lower():
|
||||
return 0, "outils non supportés par ce modèle"
|
||||
raise
|
||||
for tc in r["tool_calls"]:
|
||||
fn = tc.get("function", {})
|
||||
if fn.get("name") == "write_file":
|
||||
args = fn.get("arguments") or {}
|
||||
if isinstance(args, str):
|
||||
try:
|
||||
args = json.loads(args)
|
||||
except json.JSONDecodeError:
|
||||
return 8, "appel d'outil aux arguments illisibles"
|
||||
ok_path = "bonjour" in str(args.get("path", "")).lower()
|
||||
ok_content = "salut" in str(args.get("content", "")).lower()
|
||||
score = 10 + 5 * ok_path + 5 * ok_content
|
||||
return score, "appel d'outil correct" if score == 20 else "appel partiel"
|
||||
return 0, "aucun appel d'outil émis"
|
||||
|
||||
|
||||
async def _task_code(model: str) -> tuple[int, str]:
|
||||
r = await _ask(
|
||||
model,
|
||||
"Écris une fonction Python `somme_pairs(nombres)` qui renvoie la somme "
|
||||
"des nombres pairs de la liste. Réponds UNIQUEMENT avec le code.",
|
||||
num_predict=300,
|
||||
)
|
||||
code = _extract_code(r["text"])
|
||||
if "def somme_pairs" not in code:
|
||||
return 0, "fonction absente"
|
||||
test = (
|
||||
"assert somme_pairs([1,2,3,4]) == 6\n"
|
||||
"assert somme_pairs([]) == 0\n"
|
||||
"assert somme_pairs([7,9]) == 0\n"
|
||||
)
|
||||
return (20, "code correct (3/3 tests)") if _run_python(code, test) \
|
||||
else (6, "code présent mais tests échoués")
|
||||
|
||||
|
||||
async def _task_instruction(model: str) -> tuple[int, str]:
|
||||
r = await _ask(
|
||||
model,
|
||||
"Quelle est la capitale de la France ? Réponds en 3 mots maximum.",
|
||||
num_predict=30,
|
||||
)
|
||||
text = r["text"]
|
||||
has_answer = "paris" in text.lower()
|
||||
short = len(text.split()) <= 6
|
||||
score = 12 * has_answer + 8 * short
|
||||
return score, f"réponse « {text[:40]} »"
|
||||
|
||||
|
||||
async def _task_json(model: str) -> tuple[int, str]:
|
||||
r = await _ask(
|
||||
model,
|
||||
'Extrait les informations en JSON strict {"nom": ..., "ville": ...} '
|
||||
"depuis : « Marie habite à Lyon ». Réponds UNIQUEMENT avec le JSON.",
|
||||
num_predict=80,
|
||||
)
|
||||
m = re.search(r"\{.*\}", r["text"], re.S)
|
||||
if not m:
|
||||
return 0, "pas de JSON"
|
||||
try:
|
||||
data = json.loads(m.group(0))
|
||||
except json.JSONDecodeError:
|
||||
return 5, "JSON invalide"
|
||||
ok_nom = "marie" in str(data.get("nom", "")).lower()
|
||||
ok_ville = "lyon" in str(data.get("ville", "")).lower()
|
||||
score = 10 + 5 * ok_nom + 5 * ok_ville
|
||||
detail = "extraction correcte" if ok_nom and ok_ville else "extraction partielle"
|
||||
return score, detail
|
||||
|
||||
|
||||
async def _task_format(model: str) -> tuple[int, str]:
|
||||
r = await _ask(
|
||||
model,
|
||||
"Liste exactement 3 fruits, un par ligne, chaque ligne préfixée par « - ».",
|
||||
num_predict=60,
|
||||
)
|
||||
lines = [l for l in r["text"].splitlines() if l.strip().startswith("-")]
|
||||
if len(lines) == 3:
|
||||
return 20, "format exact"
|
||||
if len(lines) >= 2:
|
||||
return 10, f"{len(lines)} lignes au lieu de 3"
|
||||
return 0, "format non respecté"
|
||||
|
||||
|
||||
TASKS = [
|
||||
("Appel d'outil", _task_tool_call),
|
||||
("Code exécutable", _task_code),
|
||||
("Consigne courte", _task_instruction),
|
||||
("Extraction JSON", _task_json),
|
||||
("Respect du format", _task_format),
|
||||
]
|
||||
|
||||
|
||||
async def run_bench(model: str) -> AsyncIterator[dict]:
|
||||
"""Exécute les 5 épreuves en streamant la progression, stocke le score."""
|
||||
total = 0
|
||||
details = []
|
||||
for name, fn in TASKS:
|
||||
yield {"type": "task_start", "task": name}
|
||||
task = asyncio.create_task(fn(model))
|
||||
try:
|
||||
while not task.done():
|
||||
done, _ = await asyncio.wait({task}, timeout=10)
|
||||
if not done:
|
||||
# Empêche OpenResty/Nginx de fermer le SSE pendant une longue
|
||||
# génération d'un gros modèle.
|
||||
yield {"type": "heartbeat", "task": name}
|
||||
score, detail = await task
|
||||
except (OllamaError, httpx.HTTPError, OSError) as exc:
|
||||
score, detail = 0, f"erreur : {str(exc)[:80]}"
|
||||
except Exception as exc:
|
||||
# Une épreuve défaillante ne doit pas couper silencieusement le SSE :
|
||||
# elle vaut zéro et les autres épreuves continuent.
|
||||
score, detail = 0, f"épreuve interrompue : {str(exc)[:80]}"
|
||||
finally:
|
||||
if not task.done():
|
||||
task.cancel()
|
||||
total += score
|
||||
details.append({"task": name, "score": score, "detail": detail})
|
||||
yield {"type": "task_done", "task": name, "score": score, "detail": detail}
|
||||
|
||||
results = db.get_config_value(BENCH_KEY) or {}
|
||||
results[model] = {"score": total, "details": details, "at": time.time()}
|
||||
db.set_config_value(BENCH_KEY, results)
|
||||
yield {"type": "done", "score": total, "details": details}
|
||||
|
||||
|
||||
def get_scores() -> dict:
|
||||
return db.get_config_value(BENCH_KEY) or {}
|
||||
@@ -1,160 +0,0 @@
|
||||
"""Moteur code : enveloppe Aider pour l'édition multi-fichiers fiable.
|
||||
|
||||
Aider apporte ce qui fait la force de Claude Code/Codex : formats d'édition
|
||||
diff/search-replace robustes (même avec de petits modèles), repo map, et
|
||||
commits git automatiques dans le workspace.
|
||||
|
||||
L'API Python d'Aider n'étant pas officiellement stable, la version est FIGÉE
|
||||
dans requirements.txt (aider-chat==0.86.2) et tout l'import est local à la
|
||||
fonction (l'app démarre même si Aider est absent).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import threading
|
||||
|
||||
from .config import settings
|
||||
|
||||
# Aider utilise le répertoire courant : un seul run à la fois.
|
||||
_RUN_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def _confine(root: str, files: list[str] | None) -> list[str]:
|
||||
"""Résout les fichiers demandés en restant confiné au workspace.
|
||||
|
||||
Écarte silencieusement tout chemin absolu ou toute remontée `../` qui
|
||||
sortirait du workspace : le moteur code ne doit jamais toucher au disque
|
||||
hors du workspace de la discussion.
|
||||
"""
|
||||
confined: list[str] = []
|
||||
for f in files or []:
|
||||
target = os.path.abspath(os.path.join(root, f))
|
||||
if target == root or target.startswith(root + os.sep):
|
||||
confined.append(target)
|
||||
return confined
|
||||
|
||||
|
||||
def ensure_git(root: str) -> None:
|
||||
"""Initialise un dépôt git dans le workspace (requis pour les commits Aider)."""
|
||||
os.makedirs(root, exist_ok=True)
|
||||
if os.path.isdir(os.path.join(root, ".git")):
|
||||
return
|
||||
subprocess.run(["git", "init", "-q"], cwd=root, check=False)
|
||||
subprocess.run(["git", "config", "user.email", "loki@local"], cwd=root, check=False)
|
||||
subprocess.run(["git", "config", "user.name", "Loki"], cwd=root, check=False)
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""Aider est-il installé ?"""
|
||||
try:
|
||||
import aider # noqa: F401
|
||||
return True
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
|
||||
# Familles spécialisées code, par ordre de préférence.
|
||||
_CODE_MODEL_HINTS = (
|
||||
"qwen3-coder", "qwen2.5-coder", "deepseek-coder", "codestral", "devstral",
|
||||
"codegemma", "codellama", "starcoder", "coder",
|
||||
)
|
||||
|
||||
|
||||
async def pick_code_model(current: str, preference: str | None = None) -> str:
|
||||
"""Choisit le modèle pour les tâches de code.
|
||||
|
||||
- préférence explicite (config code_model != "auto") -> respectée ;
|
||||
- sinon, si un modèle spécialisé code est installé, on le prend (le plus
|
||||
gros d'abord) : un petit modèle code bat un généraliste sur ce terrain ;
|
||||
- sinon, on garde le modèle courant.
|
||||
"""
|
||||
if preference and preference != "auto":
|
||||
return preference
|
||||
|
||||
from .ollama_client import ollama
|
||||
try:
|
||||
installed = await ollama.list_models_cached()
|
||||
except Exception:
|
||||
return current
|
||||
|
||||
candidates: list[tuple[int, int, str]] = [] # (rang_hint, -taille, nom)
|
||||
for m in installed:
|
||||
name = (m.get("name") or "").lower()
|
||||
for rank, hint in enumerate(_CODE_MODEL_HINTS):
|
||||
if hint in name:
|
||||
candidates.append((rank, -(m.get("size") or 0), m["name"]))
|
||||
break
|
||||
|
||||
if not candidates:
|
||||
return current
|
||||
candidates.sort()
|
||||
return candidates[0][2]
|
||||
|
||||
|
||||
def run_code_task(
|
||||
instruction: str,
|
||||
model: str,
|
||||
files: list[str] | None = None,
|
||||
root: str | None = None,
|
||||
) -> dict:
|
||||
"""Exécute une tâche de code via Aider (synchrone — lancer dans un thread).
|
||||
|
||||
Renvoie {ok, text, files, commit, commit_message, summary}.
|
||||
"""
|
||||
if not available():
|
||||
return {
|
||||
"ok": False,
|
||||
"summary": "moteur code indisponible (aider non installé)",
|
||||
"text": "", "files": [], "commit": None,
|
||||
}
|
||||
|
||||
from aider.coders import Coder
|
||||
from aider.io import InputOutput
|
||||
from aider.models import Model
|
||||
|
||||
root = os.path.abspath(root or settings.workspace_dir)
|
||||
ensure_git(root)
|
||||
|
||||
# Aider parle à Ollama via litellm : on pointe vers notre instance.
|
||||
os.environ["OLLAMA_API_BASE"] = settings.ollama_host
|
||||
|
||||
with _RUN_LOCK:
|
||||
prev_cwd = os.getcwd()
|
||||
os.chdir(root)
|
||||
try:
|
||||
io = InputOutput(yes=True, pretty=False, fancy_input=False)
|
||||
coder = Coder.create(
|
||||
main_model=Model(f"ollama_chat/{model}"),
|
||||
io=io,
|
||||
fnames=_confine(root, files),
|
||||
auto_commits=True,
|
||||
stream=False,
|
||||
use_git=True,
|
||||
suggest_shell_commands=False,
|
||||
detect_urls=False,
|
||||
)
|
||||
text = coder.run(instruction) or ""
|
||||
edited = sorted(coder.aider_edited_files or [])
|
||||
commit = getattr(coder, "last_aider_commit_hash", None)
|
||||
commit_msg = getattr(coder, "last_aider_commit_message", None)
|
||||
except Exception as exc: # aider peut lever des erreurs variées
|
||||
return {
|
||||
"ok": False,
|
||||
"summary": f"échec moteur code : {exc}",
|
||||
"text": "", "files": [], "commit": None,
|
||||
}
|
||||
finally:
|
||||
os.chdir(prev_cwd)
|
||||
|
||||
summary = f"{len(edited)} fichier(s) modifié(s)"
|
||||
if commit:
|
||||
summary += f" · commit {str(commit)[:7]}"
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text.strip(),
|
||||
"files": edited,
|
||||
"commit": commit,
|
||||
"commit_message": commit_msg,
|
||||
"summary": summary,
|
||||
}
|
||||
@@ -1,26 +0,0 @@
|
||||
"""Configuration de l'application, chargée depuis l'environnement."""
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""Réglages globaux de Loki (surchargés par variables d'environnement)."""
|
||||
|
||||
ollama_host: str = "http://host.docker.internal:11434"
|
||||
default_model: str = "gemma4:12b"
|
||||
workspace_dir: str = "/workspace"
|
||||
data_dir: str = "/data"
|
||||
port: int = 8080
|
||||
|
||||
# Marqueur de build injecté à la construction de l'image (git sha court).
|
||||
# Permet de vérifier que l'image déployée est bien à jour.
|
||||
loki_version: str = "dev"
|
||||
|
||||
# Override manuel de la VRAM (Mo) si la détection GPU échoue dans le
|
||||
# conteneur (utile quand Ollama tourne sur une autre machine).
|
||||
gpu_vram_mb: int = 0
|
||||
gpu_name: str = ""
|
||||
|
||||
model_config = SettingsConfigDict(env_file=".env", extra="ignore")
|
||||
|
||||
|
||||
settings = Settings()
|
||||
@@ -1,208 +0,0 @@
|
||||
"""Persistance SQLite : sessions et messages.
|
||||
|
||||
On utilise sqlite3 de la bibliothèque standard (zéro dépendance). Les écritures
|
||||
sont rapides ; un verrou protège l'accès concurrent depuis les routes async.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
|
||||
from .config import settings
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
_DB_PATH = os.path.join(settings.data_dir, "loki.db")
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
os.makedirs(settings.data_dir, exist_ok=True)
|
||||
conn = sqlite3.connect(_DB_PATH, check_same_thread=False)
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
|
||||
def init_db() -> None:
|
||||
"""Crée les tables si elles n'existent pas."""
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.executescript(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS sessions (
|
||||
id TEXT PRIMARY KEY,
|
||||
title TEXT NOT NULL,
|
||||
model TEXT,
|
||||
created_at REAL NOT NULL,
|
||||
updated_at REAL NOT NULL
|
||||
);
|
||||
CREATE TABLE IF NOT EXISTS messages (
|
||||
id TEXT PRIMARY KEY,
|
||||
session_id TEXT NOT NULL REFERENCES sessions(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL,
|
||||
content TEXT NOT NULL,
|
||||
model TEXT,
|
||||
meta TEXT,
|
||||
created_at REAL NOT NULL
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_messages_session
|
||||
ON messages(session_id, created_at);
|
||||
"""
|
||||
)
|
||||
# Migration douce : ajoute la colonne meta aux bases antérieures.
|
||||
cols = {r["name"] for r in conn.execute("PRAGMA table_info(messages)")}
|
||||
if "meta" not in cols:
|
||||
conn.execute("ALTER TABLE messages ADD COLUMN meta TEXT")
|
||||
# Migration douce : résumé de conversation (mémoire compressée).
|
||||
scols = {r["name"] for r in conn.execute("PRAGMA table_info(sessions)")}
|
||||
if "summary" not in scols:
|
||||
conn.execute("ALTER TABLE sessions ADD COLUMN summary TEXT")
|
||||
# Migration douce : projet (sous-dossier de travail) de la session.
|
||||
if "project" not in scols:
|
||||
conn.execute("ALTER TABLE sessions ADD COLUMN project TEXT")
|
||||
# Table clé/valeur pour la configuration de l'agent.
|
||||
conn.execute(
|
||||
"CREATE TABLE IF NOT EXISTS config (key TEXT PRIMARY KEY, value TEXT)"
|
||||
)
|
||||
|
||||
|
||||
def _now() -> float:
|
||||
return time.time()
|
||||
|
||||
|
||||
# ── Sessions ─────────────────────────────────────────────────────────────
|
||||
def create_session(
|
||||
title: str, model: str | None, project: str | None = None
|
||||
) -> dict:
|
||||
sid = uuid.uuid4().hex
|
||||
now = _now()
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO sessions (id, title, model, project, created_at, updated_at)"
|
||||
" VALUES (?, ?, ?, ?, ?, ?)",
|
||||
(sid, title, model, project, now, now),
|
||||
)
|
||||
return {"id": sid, "title": title, "model": model, "project": project,
|
||||
"created_at": now, "updated_at": now, "message_count": 0}
|
||||
|
||||
|
||||
def set_session_project(sid: str, project: str | None) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"UPDATE sessions SET project = ? WHERE id = ?", (project, sid)
|
||||
)
|
||||
|
||||
|
||||
def list_sessions() -> list[dict]:
|
||||
with _LOCK, _connect() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT s.*, COUNT(m.id) AS message_count
|
||||
FROM sessions s
|
||||
LEFT JOIN messages m ON m.session_id = s.id
|
||||
GROUP BY s.id
|
||||
ORDER BY s.updated_at DESC
|
||||
"""
|
||||
).fetchall()
|
||||
return [dict(r) for r in rows]
|
||||
|
||||
|
||||
def get_session(sid: str) -> dict | None:
|
||||
with _LOCK, _connect() as conn:
|
||||
row = conn.execute("SELECT * FROM sessions WHERE id = ?", (sid,)).fetchone()
|
||||
return dict(row) if row else None
|
||||
|
||||
|
||||
def rename_session(sid: str, title: str) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"UPDATE sessions SET title = ?, updated_at = ? WHERE id = ?",
|
||||
(title, _now(), sid),
|
||||
)
|
||||
|
||||
|
||||
def delete_session(sid: str) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute("DELETE FROM messages WHERE session_id = ?", (sid,))
|
||||
conn.execute("DELETE FROM sessions WHERE id = ?", (sid,))
|
||||
|
||||
|
||||
def touch_session(sid: str) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"UPDATE sessions SET updated_at = ? WHERE id = ?", (_now(), sid)
|
||||
)
|
||||
|
||||
|
||||
# ── Messages ─────────────────────────────────────────────────────────────
|
||||
def add_message(
|
||||
sid: str,
|
||||
role: str,
|
||||
content: str,
|
||||
model: str | None,
|
||||
meta: dict | None = None,
|
||||
) -> dict:
|
||||
mid = uuid.uuid4().hex
|
||||
now = _now()
|
||||
meta_json = json.dumps(meta, ensure_ascii=False) if meta else None
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO messages (id, session_id, role, content, model, meta, created_at)"
|
||||
" VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||
(mid, sid, role, content, model, meta_json, now),
|
||||
)
|
||||
conn.execute(
|
||||
"UPDATE sessions SET updated_at = ? WHERE id = ?", (now, sid)
|
||||
)
|
||||
return {"id": mid, "session_id": sid, "role": role, "content": content,
|
||||
"model": model, "meta": meta, "created_at": now}
|
||||
|
||||
|
||||
def _row_to_message(row: sqlite3.Row) -> dict:
|
||||
msg = dict(row)
|
||||
msg["meta"] = json.loads(msg["meta"]) if msg.get("meta") else None
|
||||
return msg
|
||||
|
||||
|
||||
def list_messages(sid: str) -> list[dict]:
|
||||
with _LOCK, _connect() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT * FROM messages WHERE session_id = ? ORDER BY created_at",
|
||||
(sid,),
|
||||
).fetchall()
|
||||
return [_row_to_message(r) for r in rows]
|
||||
|
||||
|
||||
def list_messages_for_model(sid: str) -> list[dict]:
|
||||
"""Historique épuré (role/content) destiné au contexte du modèle."""
|
||||
return [
|
||||
{"role": m["role"], "content": m["content"]}
|
||||
for m in list_messages(sid)
|
||||
]
|
||||
|
||||
|
||||
def set_session_summary(sid: str, summary: str) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"UPDATE sessions SET summary = ? WHERE id = ?", (summary, sid)
|
||||
)
|
||||
|
||||
|
||||
# ── Configuration (clé/valeur JSON) ──────────────────────────────────────
|
||||
def get_config_value(key: str) -> dict | None:
|
||||
with _LOCK, _connect() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT value FROM config WHERE key = ?", (key,)
|
||||
).fetchone()
|
||||
return json.loads(row["value"]) if row else None
|
||||
|
||||
|
||||
def set_config_value(key: str, value: dict) -> None:
|
||||
payload = json.dumps(value, ensure_ascii=False)
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO config (key, value) VALUES (?, ?)"
|
||||
" ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
||||
(key, payload),
|
||||
)
|
||||
@@ -1,169 +0,0 @@
|
||||
"""Boosters de qualité pour petits modèles : plan-puis-exécute et auto-critique.
|
||||
|
||||
- make_plan : décompose une demande complexe en 3-5 étapes courtes. Un petit
|
||||
modèle qui suit un plan écrit réussit bien mieux qu'en improvisant.
|
||||
- self_review : une passe de critique éclair sur la réponse, puis une révision
|
||||
si des défauts sont trouvés (activable : coûte un peu de latence).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
|
||||
import httpx
|
||||
|
||||
from .ollama_client import OllamaError, ollama
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_PLAN_PROMPT = (
|
||||
"Découpe la demande en 3 à 5 étapes courtes et concrètes, une par ligne, "
|
||||
"numérotées « 1. », « 2. »… Pas d'introduction, pas de conclusion, "
|
||||
"UNIQUEMENT les étapes, en français."
|
||||
)
|
||||
|
||||
_PLAN_CODE_PROMPT = (
|
||||
"Tu es architecte logiciel. Établis un plan d'IMPLÉMENTATION concret pour "
|
||||
"cette application, en 3 à 5 étapes courtes (une par ligne, numérotées "
|
||||
"« 1. », « 2. »…).\n"
|
||||
"- Étape 1 = ARCHITECTURE : privilégie UN SEUL fichier autonome (ex. "
|
||||
"index.html avec le CSS et le JS intégrés, sans dépendance externe) ; ne "
|
||||
"sépare en plusieurs fichiers que si c'est vraiment indispensable.\n"
|
||||
"- Étapes suivantes = construire UNE fonctionnalité concrète et testable à "
|
||||
"la fois (structure/affichage, puis interactions, puis logique).\n"
|
||||
"Reste RÉALISTE et réalisable en une passe : pas de dépendance externe, pas "
|
||||
"de bibliothèque à installer, pas de « moteur IA » complexe si ce n'est pas "
|
||||
"explicitement demandé — une logique simple en JavaScript suffit. "
|
||||
"Pas d'introduction ni de conclusion, UNIQUEMENT les étapes, en français."
|
||||
)
|
||||
|
||||
_CRITIQUE_PROMPT = (
|
||||
"Tu es un relecteur exigeant. Voici une demande et la réponse d'un "
|
||||
"assistant. Si la réponse est correcte et complète, réponds exactement "
|
||||
"PARFAIT. Sinon, liste au plus 3 défauts concrets (erreurs, oublis, "
|
||||
"incohérences), un par ligne."
|
||||
)
|
||||
|
||||
_REVISE_PROMPT = (
|
||||
"Réécris la réponse en corrigeant les défauts listés. Donne UNIQUEMENT la "
|
||||
"réponse finale corrigée, sans commentaire sur la révision."
|
||||
)
|
||||
|
||||
|
||||
def needs_plan(message: str) -> bool:
|
||||
"""Une demande assez longue/composée mérite un plan explicite."""
|
||||
if len(message) < 120:
|
||||
return False
|
||||
connectors = len(re.findall(
|
||||
r"\b(puis|ensuite|après|avec|ainsi que|et aussi|également)\b",
|
||||
message, re.I,
|
||||
))
|
||||
return len(message) > 240 or connectors >= 2
|
||||
|
||||
|
||||
async def _ask(
|
||||
model: str,
|
||||
system: str,
|
||||
user: str,
|
||||
*,
|
||||
num_predict: int,
|
||||
options: dict | None = None,
|
||||
keep_alive: str | None = None,
|
||||
) -> str:
|
||||
"""Appel court au modèle.
|
||||
|
||||
``options`` doit contenir les options runner (num_ctx, num_batch…) du chat
|
||||
principal : un appel avec des options divergentes force Ollama à recharger
|
||||
le modèle en plein message.
|
||||
"""
|
||||
messages = [{"role": "system", "content": system},
|
||||
{"role": "user", "content": user}]
|
||||
opts = {**(options or {}), "temperature": 0.2, "num_predict": num_predict}
|
||||
# think=False : un appel utilitaire (plan, critique) ne doit jamais
|
||||
# « réfléchir » — sur un modèle thinking, la pensée dévore le budget et
|
||||
# multiplie la latence. Repli sans le paramètre si le modèle le refuse.
|
||||
think: bool | None = False
|
||||
while True:
|
||||
text = ""
|
||||
try:
|
||||
async for chunk in ollama.chat(
|
||||
model, messages, options=opts, think=think,
|
||||
keep_alive=keep_alive, stream=True,
|
||||
):
|
||||
text += chunk.get("message", {}).get("content", "")
|
||||
if chunk.get("done"):
|
||||
break
|
||||
return text.strip()
|
||||
except OllamaError as exc:
|
||||
if think is False and "think" in str(exc).lower():
|
||||
think = None
|
||||
continue
|
||||
raise
|
||||
|
||||
|
||||
async def make_plan(
|
||||
model: str,
|
||||
message: str,
|
||||
*,
|
||||
code: bool = False,
|
||||
options: dict | None = None,
|
||||
keep_alive: str | None = None,
|
||||
) -> list[str]:
|
||||
"""Renvoie la liste des étapes (vide si échec — jamais bloquant).
|
||||
|
||||
``code=True`` bascule sur un plan d'ARCHITECTURE/implémentation concret
|
||||
(fichiers, structure, fonctionnalités) plutôt qu'une liste d'objectifs.
|
||||
"""
|
||||
try:
|
||||
raw = await _ask(
|
||||
model, _PLAN_CODE_PROMPT if code else _PLAN_PROMPT,
|
||||
message[:1200], num_predict=260,
|
||||
options=options, keep_alive=keep_alive,
|
||||
)
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
logger.warning("Plan impossible : %s", exc)
|
||||
return []
|
||||
steps = []
|
||||
for line in raw.splitlines():
|
||||
line = line.strip()
|
||||
m = re.match(r"^\d+[.)]\s*(.+)$", line)
|
||||
if m:
|
||||
steps.append(m.group(1).strip())
|
||||
return steps[:5] if len(steps) >= 2 else []
|
||||
|
||||
|
||||
async def self_review(
|
||||
model: str,
|
||||
request: str,
|
||||
answer: str,
|
||||
*,
|
||||
options: dict | None = None,
|
||||
keep_alive: str | None = None,
|
||||
) -> str | None:
|
||||
"""Critique puis révise la réponse. None si rien à corriger / échec."""
|
||||
if len(answer) < 80:
|
||||
return None
|
||||
try:
|
||||
critique = await _ask(
|
||||
model,
|
||||
_CRITIQUE_PROMPT,
|
||||
f"Demande :\n{request[:800]}\n\nRéponse :\n{answer[:2500]}",
|
||||
num_predict=180,
|
||||
options=options, keep_alive=keep_alive,
|
||||
)
|
||||
if not critique or "PARFAIT" in critique.upper()[:40]:
|
||||
return None
|
||||
|
||||
revised = await _ask(
|
||||
model,
|
||||
_REVISE_PROMPT,
|
||||
f"Demande :\n{request[:800]}\n\nRéponse initiale :\n{answer[:2500]}"
|
||||
f"\n\nDéfauts :\n{critique[:600]}",
|
||||
num_predict=1500,
|
||||
options=options, keep_alive=keep_alive,
|
||||
)
|
||||
# Garde-fou : une révision vide ou minuscule ne remplace rien.
|
||||
return revised if len(revised) > len(answer) // 3 else None
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
logger.warning("Auto-critique impossible : %s", exc)
|
||||
return None
|
||||
@@ -1,122 +0,0 @@
|
||||
"""Point d'entrée FastAPI de Loki.
|
||||
|
||||
Sert l'API (/api/*) et, en production, le frontend React compilé (static/).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import FileResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
|
||||
from . import agent_config, coder, db, rag
|
||||
from .config import settings
|
||||
from .ollama_client import ollama
|
||||
from .routes import (
|
||||
benchmark, chat, config, files, git, mcp, models, projects, sessions,
|
||||
shell, system,
|
||||
)
|
||||
|
||||
|
||||
async def _warm_default_model() -> None:
|
||||
"""Précharge le modèle par défaut en VRAM au démarrage (best-effort)."""
|
||||
import asyncio
|
||||
import logging
|
||||
|
||||
await asyncio.sleep(2) # laisse le service démarrer
|
||||
try:
|
||||
cfg = agent_config.get_config(settings.default_model)
|
||||
models.start_model_warm(
|
||||
settings.default_model, cfg.get("keep_alive", "30m")
|
||||
)
|
||||
logging.getLogger(__name__).info(
|
||||
"Préchargement du modèle %s lancé", settings.default_model
|
||||
)
|
||||
except Exception as exc: # best-effort
|
||||
logging.getLogger(__name__).info(
|
||||
"Préchargement au démarrage ignoré : %s", exc
|
||||
)
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_: FastAPI):
|
||||
import asyncio
|
||||
|
||||
db.init_db()
|
||||
rag.init_table()
|
||||
# Workspace en dépôt git : requis pour les commits du moteur code (Aider).
|
||||
coder.ensure_git(settings.workspace_dir)
|
||||
# Préchargement du modèle par défaut, sans bloquer le démarrage.
|
||||
asyncio.create_task(_warm_default_model())
|
||||
yield
|
||||
# Ferme le pool HTTP partagé vers Ollama et les sessions MCP.
|
||||
await ollama.aclose()
|
||||
from .mcp_client import manager as mcp_manager
|
||||
await mcp_manager.aclose()
|
||||
|
||||
|
||||
app = FastAPI(
|
||||
title="Loki", description="Agent IA local sur Ollama", lifespan=lifespan
|
||||
)
|
||||
|
||||
# En dev, le front tourne sur Vite (5173). On autorise le CORS large ;
|
||||
# en prod le front est servi par le même origin, donc sans impact.
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=["*"],
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
app.include_router(models.router)
|
||||
app.include_router(sessions.router)
|
||||
app.include_router(chat.router)
|
||||
app.include_router(files.router)
|
||||
app.include_router(config.router)
|
||||
app.include_router(shell.router)
|
||||
app.include_router(system.router)
|
||||
app.include_router(benchmark.router)
|
||||
app.include_router(git.router)
|
||||
app.include_router(mcp.router)
|
||||
app.include_router(projects.router)
|
||||
|
||||
|
||||
@app.get("/api/health")
|
||||
async def health() -> dict:
|
||||
return {"status": "ok", "service": "loki", "version": settings.loki_version}
|
||||
|
||||
|
||||
@app.get("/api/version")
|
||||
async def version() -> dict:
|
||||
"""Marqueur de build — pour vérifier que l'image déployée est à jour."""
|
||||
return {"version": settings.loki_version}
|
||||
|
||||
|
||||
# ── Service du frontend compilé (présent uniquement en image Docker) ─────
|
||||
_STATIC_DIR = os.path.join(os.path.dirname(__file__), "..", "static")
|
||||
if os.path.isdir(_STATIC_DIR):
|
||||
app.mount(
|
||||
"/assets",
|
||||
StaticFiles(directory=os.path.join(_STATIC_DIR, "assets")),
|
||||
name="assets",
|
||||
)
|
||||
|
||||
@app.get("/{full_path:path}")
|
||||
async def spa_fallback(full_path: str):
|
||||
"""Fichier statique racine s'il existe (favicon…), sinon index.html.
|
||||
|
||||
Sans ce test, /favicon.svg recevait index.html : aucun favicon ne
|
||||
s'affichait en production.
|
||||
"""
|
||||
static_root = os.path.abspath(_STATIC_DIR)
|
||||
candidate = os.path.abspath(os.path.join(static_root, full_path))
|
||||
if (
|
||||
full_path
|
||||
and candidate.startswith(static_root + os.sep)
|
||||
and os.path.isfile(candidate)
|
||||
):
|
||||
return FileResponse(candidate)
|
||||
return FileResponse(os.path.join(static_root, "index.html"))
|
||||
@@ -1,341 +0,0 @@
|
||||
"""Client MCP : catalogue préconfiguré + sessions vers les serveurs activés.
|
||||
|
||||
Un serveur désactivé n'est jamais démarré et n'expose aucun outil au modèle
|
||||
(chaque outil injecté coûte du contexte). Connexion lazy au premier message,
|
||||
session réutilisée ensuite. Toute panne est non bloquante pour le chat.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
import shlex
|
||||
from contextlib import AsyncExitStack
|
||||
|
||||
from mcp import ClientSession, StdioServerParameters
|
||||
from mcp.client.stdio import stdio_client
|
||||
|
||||
from . import db
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MCP_KEY = "mcp"
|
||||
|
||||
# Catalogue embarqué. command=None => serveur "custom" (commande utilisateur).
|
||||
CATALOG: dict[str, dict] = {
|
||||
"playwright": {
|
||||
"label": "Playwright (navigateur)",
|
||||
"description": "Pilote un vrai navigateur : naviguer, cliquer, lire "
|
||||
"la console, captures. L'agent teste réellement ses pages.",
|
||||
"command": ["npx", "@playwright/mcp@latest", "--headless"],
|
||||
"url_param": False,
|
||||
"env_params": [],
|
||||
# Limite le nombre d'outils injectés dans le prompt.
|
||||
"expose": [
|
||||
"browser_navigate", "browser_click", "browser_type",
|
||||
"browser_snapshot", "browser_console_messages",
|
||||
"browser_take_screenshot",
|
||||
],
|
||||
},
|
||||
"context7": {
|
||||
"label": "Context7 (documentation)",
|
||||
"description": "Documentation à jour de n'importe quelle librairie ou "
|
||||
"framework (React, FastAPI, Tailwind…).",
|
||||
"command": ["npx", "-y", "@upstash/context7-mcp"],
|
||||
"url_param": False,
|
||||
"env_params": [],
|
||||
"expose": None,
|
||||
},
|
||||
"fetch": {
|
||||
"label": "Fetch (lecture web)",
|
||||
"description": "Lit proprement n'importe quelle URL (markdown épuré).",
|
||||
"command": ["python", "-m", "mcp_server_fetch"],
|
||||
"url_param": False,
|
||||
"env_params": [],
|
||||
"expose": None,
|
||||
},
|
||||
"searxng": {
|
||||
"label": "SearxNG (recherche web)",
|
||||
"description": "Vraie recherche web via une instance SearxNG "
|
||||
"(renseigner SEARXNG_URL).",
|
||||
"command": ["npx", "-y", "mcp-searxng"],
|
||||
"url_param": False,
|
||||
"env_params": ["SEARXNG_URL"],
|
||||
"expose": None,
|
||||
},
|
||||
"custom": {
|
||||
"label": "Personnalisé",
|
||||
"description": "N'importe quel serveur MCP : colle une commande "
|
||||
"(stdio) ou une URL (streamable HTTP).",
|
||||
"command": None,
|
||||
"url_param": True,
|
||||
"env_params": [],
|
||||
"expose": None,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def get_mcp_state() -> dict:
|
||||
"""État activé/params de chaque serveur du catalogue (défaut : désactivé)."""
|
||||
stored = db.get_config_value(MCP_KEY) or {}
|
||||
return {
|
||||
sid: {
|
||||
"enabled": bool(stored.get(sid, {}).get("enabled", False)),
|
||||
"params": dict(stored.get(sid, {}).get("params", {})),
|
||||
}
|
||||
for sid in CATALOG
|
||||
}
|
||||
|
||||
|
||||
def set_mcp_state(sid: str, *, enabled: bool, params: dict) -> dict:
|
||||
if sid not in CATALOG:
|
||||
raise KeyError(sid)
|
||||
state = get_mcp_state()
|
||||
state[sid] = {"enabled": enabled, "params": dict(params)}
|
||||
db.set_config_value(MCP_KEY, state)
|
||||
return state
|
||||
|
||||
|
||||
_CALL_TIMEOUT = 30.0
|
||||
# Généreux : le premier lancement d'un serveur npx (Playwright, Context7…)
|
||||
# télécharge le paquet — souvent bien plus de 20 s. Les démarrages suivants
|
||||
# sont instantanés (cache npm).
|
||||
_CONNECT_TIMEOUT = 90.0
|
||||
_MAX_RESULT_CHARS = 8000
|
||||
|
||||
|
||||
def _safe_tool_name(name: str) -> str:
|
||||
"""Nom d'outil compatible function-calling (lettres/chiffres/underscore).
|
||||
|
||||
Les noms MCP peuvent contenir des tirets (« resolve-library-id ») que les
|
||||
grammaires de tool-calling et les modèles mélangent avec des underscores —
|
||||
source de « Tool not found ». On expose une version assainie et on garde
|
||||
la correspondance vers le vrai nom.
|
||||
"""
|
||||
return re.sub(r"[^a-zA-Z0-9_]", "_", name)
|
||||
|
||||
|
||||
class _ServerConn:
|
||||
"""Session vivante vers un serveur MCP (process stdio + handshake)."""
|
||||
|
||||
def __init__(self, sid: str) -> None:
|
||||
self.sid = sid
|
||||
self.stack = AsyncExitStack()
|
||||
self.session: ClientSession | None = None
|
||||
self.tools: list[dict] = [] # définitions format Ollama
|
||||
# nom exposé au modèle -> vrai nom d'outil MCP
|
||||
self.name_map: dict[str, str] = {}
|
||||
|
||||
async def start(self) -> None:
|
||||
entry = CATALOG[self.sid]
|
||||
params = get_mcp_state()[self.sid]["params"]
|
||||
spec = params.get("command", "").strip()
|
||||
if entry["command"] is None and spec.startswith(("http://", "https://")):
|
||||
# Serveur "custom" en streamable HTTP (URL collée par l'utilisateur).
|
||||
from mcp.client.streamable_http import streamablehttp_client
|
||||
read, write, _ = await self.stack.enter_async_context(
|
||||
streamablehttp_client(spec)
|
||||
)
|
||||
else:
|
||||
if entry["command"] is None:
|
||||
# Serveur "custom" : commande stdio saisie par l'utilisateur.
|
||||
command = shlex.split(spec)
|
||||
if not command:
|
||||
raise ValueError("commande du serveur personnalisé vide")
|
||||
else:
|
||||
command = list(entry["command"])
|
||||
# Paramètres obligatoires (ex. SEARXNG_URL) : refus clair AVANT le
|
||||
# lancement, plutôt qu'un échec cryptique à chaque appel d'outil.
|
||||
missing = [k for k in entry["env_params"] if not params.get(k)]
|
||||
if missing:
|
||||
raise ValueError(
|
||||
f"{', '.join(missing)} requis — renseigne ce champ dans la "
|
||||
"carte du serveur (Configuration → Serveurs MCP)"
|
||||
)
|
||||
env = {k: params[k] for k in entry["env_params"] if params.get(k)}
|
||||
server = StdioServerParameters(
|
||||
command=command[0], args=command[1:], env=env or None
|
||||
)
|
||||
read, write = await self.stack.enter_async_context(
|
||||
stdio_client(server)
|
||||
)
|
||||
self.session = await self.stack.enter_async_context(
|
||||
ClientSession(read, write)
|
||||
)
|
||||
await asyncio.wait_for(self.session.initialize(), _CONNECT_TIMEOUT)
|
||||
listed = await asyncio.wait_for(self.session.list_tools(), _CONNECT_TIMEOUT)
|
||||
expose = entry.get("expose")
|
||||
self.tools = []
|
||||
self.name_map = {}
|
||||
for t in listed.tools:
|
||||
if expose is not None and t.name not in expose:
|
||||
continue
|
||||
exposed = f"mcp_{self.sid}_{_safe_tool_name(t.name)}"
|
||||
self.name_map[exposed] = t.name
|
||||
self.tools.append({
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": exposed,
|
||||
"description": (t.description or t.name)[:400],
|
||||
"parameters": t.inputSchema
|
||||
or {"type": "object", "properties": {}},
|
||||
},
|
||||
})
|
||||
|
||||
async def close(self) -> None:
|
||||
try:
|
||||
await self.stack.aclose()
|
||||
except Exception: # process déjà mort : sans importance
|
||||
pass
|
||||
|
||||
|
||||
class McpManager:
|
||||
"""Sessions MCP lazy + dispatch d'appels d'outils, jamais bloquant."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._conns: dict[str, _ServerConn] = {}
|
||||
self._errors: dict[str, str] = {}
|
||||
self._notices: list[str] = []
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def _ensure(self, sid: str) -> _ServerConn | None:
|
||||
if sid in self._conns:
|
||||
return self._conns[sid]
|
||||
try:
|
||||
conn = _ServerConn(sid)
|
||||
await conn.start()
|
||||
except Exception as exc:
|
||||
if isinstance(exc, (asyncio.TimeoutError, TimeoutError)):
|
||||
msg = (
|
||||
"délai de démarrage dépassé — premier lancement d'un "
|
||||
"serveur npx (téléchargement) ou paquet absent de l'image. "
|
||||
"Réessaie au prochain message."
|
||||
)
|
||||
else:
|
||||
msg = str(exc)[:300] or exc.__class__.__name__
|
||||
logger.warning("Serveur MCP %s indisponible : %s", sid, msg)
|
||||
self._errors[sid] = msg
|
||||
self._notices.append(
|
||||
f"Serveur MCP « {CATALOG[sid]['label']} » indisponible : {msg}"
|
||||
)
|
||||
return None
|
||||
self._errors.pop(sid, None)
|
||||
self._conns[sid] = conn
|
||||
return conn
|
||||
|
||||
async def tool_definitions(self) -> list[dict]:
|
||||
"""Outils des serveurs activés (connexion lazy, pannes ignorées)."""
|
||||
defs: list[dict] = []
|
||||
async with self._lock:
|
||||
state = get_mcp_state()
|
||||
# Ferme les serveurs désactivés entre-temps.
|
||||
for sid in [s for s in self._conns if not state[s]["enabled"]]:
|
||||
await self._conns.pop(sid).close()
|
||||
for sid, st in state.items():
|
||||
if not st["enabled"]:
|
||||
continue
|
||||
conn = await self._ensure(sid)
|
||||
if conn:
|
||||
defs.extend(conn.tools)
|
||||
return defs
|
||||
|
||||
def _resolve(self, prefixed_name: str) -> tuple[str, str] | None:
|
||||
"""(sid, vrai nom d'outil) depuis le nom exposé au modèle.
|
||||
|
||||
Résolution par table de correspondance, avec tolérance : les modèles
|
||||
confondent parfois tirets et underscores dans les noms d'outils.
|
||||
"""
|
||||
wanted = _safe_tool_name(prefixed_name)
|
||||
for sid, conn in self._conns.items():
|
||||
for exposed, real in conn.name_map.items():
|
||||
if exposed == prefixed_name or exposed == wanted:
|
||||
return sid, real
|
||||
# Repli : découpage mcp_<sid>_<outil> (serveur pas encore connecté).
|
||||
try:
|
||||
_, sid, tool = prefixed_name.split("_", 2)
|
||||
return sid, tool
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
async def call_tool(self, prefixed_name: str, args: dict) -> dict:
|
||||
resolved = self._resolve(prefixed_name)
|
||||
if resolved is None:
|
||||
return {"ok": False, "content": "", "summary": "nom d'outil invalide"}
|
||||
sid, tool = resolved
|
||||
async with self._lock:
|
||||
conn = self._conns.get(sid) or await self._ensure(sid)
|
||||
if conn is None or conn.session is None:
|
||||
return {"ok": False, "content": "",
|
||||
"summary": f"serveur MCP {sid} indisponible"}
|
||||
# Serveur (re)connecté après le repli : re-résout via sa table.
|
||||
if conn.name_map:
|
||||
tool = conn.name_map.get(prefixed_name) or conn.name_map.get(
|
||||
_safe_tool_name(prefixed_name), tool
|
||||
)
|
||||
try:
|
||||
result = await asyncio.wait_for(
|
||||
conn.session.call_tool(tool, args or {}), _CALL_TIMEOUT
|
||||
)
|
||||
except Exception as exc:
|
||||
# Session probablement morte : on la ferme, retry au prochain tour.
|
||||
async with self._lock:
|
||||
dead = self._conns.pop(sid, None)
|
||||
if dead:
|
||||
await dead.close()
|
||||
return {"ok": False, "content": "",
|
||||
"summary": f"échec MCP : {str(exc)[:200]}"}
|
||||
parts = [
|
||||
c.text for c in result.content
|
||||
if getattr(c, "type", "") == "text" and getattr(c, "text", "")
|
||||
]
|
||||
content = "\n".join(parts)[:_MAX_RESULT_CHARS]
|
||||
ok = not bool(getattr(result, "isError", False))
|
||||
return {
|
||||
"ok": ok,
|
||||
"content": content,
|
||||
"summary": (content.splitlines()[0][:120] if content else "terminé")
|
||||
if ok else (content[:120] or "erreur outil MCP"),
|
||||
}
|
||||
|
||||
def statuses(self) -> dict[str, dict]:
|
||||
state = get_mcp_state()
|
||||
out = {}
|
||||
for sid, st in state.items():
|
||||
if sid in self._conns:
|
||||
s = "connected"
|
||||
elif sid in self._errors:
|
||||
s = "error"
|
||||
else:
|
||||
s = "inactive"
|
||||
out[sid] = {
|
||||
"state": s if st["enabled"] else "inactive",
|
||||
"error": self._errors.get(sid),
|
||||
"tools": len(self._conns[sid].tools) if sid in self._conns else 0,
|
||||
}
|
||||
return out
|
||||
|
||||
def notices(self) -> list[str]:
|
||||
out, self._notices = self._notices, []
|
||||
return out
|
||||
|
||||
async def test_server(self, sid: str) -> dict:
|
||||
"""Connexion d'essai indépendante (n'altère pas les sessions)."""
|
||||
conn = _ServerConn(sid)
|
||||
try:
|
||||
await conn.start()
|
||||
return {"ok": True,
|
||||
"tools": [d["function"]["name"] for d in conn.tools],
|
||||
"error": None}
|
||||
except Exception as exc:
|
||||
return {"ok": False, "tools": [],
|
||||
"error": str(exc)[:300] or exc.__class__.__name__}
|
||||
finally:
|
||||
await conn.close()
|
||||
|
||||
async def aclose(self) -> None:
|
||||
for conn in list(self._conns.values()):
|
||||
await conn.close()
|
||||
self._conns.clear()
|
||||
|
||||
|
||||
manager = McpManager()
|
||||
@@ -1,110 +0,0 @@
|
||||
"""Mémoire de conversation compressée — le vrai levier des petits modèles.
|
||||
|
||||
Un petit modèle se noie dans un long historique : il oublie la consigne, part
|
||||
en boucle, et un grand num_ctx le fait déborder du GPU. On garde donc :
|
||||
[invite système] + [résumé compact des anciens tours] + [N derniers messages]
|
||||
|
||||
Le résumé est régénéré en arrière-plan (après la réponse, sans latence pour
|
||||
l'utilisateur) dès que l'historique dépasse le seuil.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import httpx
|
||||
|
||||
from . import db
|
||||
from .ollama_client import OllamaError, ollama
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Nombre de messages récents passés tels quels au modèle.
|
||||
KEEP_RECENT = 10
|
||||
# Au-delà de ce total, les anciens tours sont compressés dans le résumé.
|
||||
SUMMARIZE_AFTER = KEEP_RECENT + 6
|
||||
|
||||
_SUMMARY_PROMPT = (
|
||||
"Résume la conversation ci-dessous en français, en 10 lignes maximum. "
|
||||
"Conserve impérativement : l'objectif de l'utilisateur, les décisions "
|
||||
"prises, les fichiers créés/modifiés et leur rôle, et les points encore "
|
||||
"ouverts. Réponds UNIQUEMENT par le résumé."
|
||||
)
|
||||
|
||||
|
||||
def build_convo(sid: str, system_prompt: str) -> list[dict]:
|
||||
"""Construit le contexte : système + résumé éventuel + messages récents.
|
||||
|
||||
Le résumé est fusionné DANS l'invite système plutôt qu'ajouté comme second
|
||||
message système : de nombreux templates (Gemma, Mistral…) rejettent tout
|
||||
message système qui n'est pas le premier de la liste.
|
||||
"""
|
||||
session = db.get_session(sid) or {}
|
||||
summary = (session.get("summary") or "").strip()
|
||||
messages = db.list_messages_for_model(sid)
|
||||
|
||||
if summary and len(messages) > KEEP_RECENT:
|
||||
system_prompt = (
|
||||
f"{system_prompt}\n\nRésumé des échanges précédents :\n{summary}"
|
||||
)
|
||||
messages = messages[-KEEP_RECENT:]
|
||||
|
||||
return [{"role": "system", "content": system_prompt}, *messages]
|
||||
|
||||
|
||||
async def maybe_summarize(
|
||||
sid: str,
|
||||
model: str,
|
||||
*,
|
||||
options: dict | None = None,
|
||||
keep_alive: str | None = None,
|
||||
) -> None:
|
||||
"""Compresse les anciens tours dans le résumé (tâche d'arrière-plan).
|
||||
|
||||
``options`` doit reprendre les options runner du chat (num_ctx…) pour ne
|
||||
pas déclencher un rechargement du modèle après chaque réponse.
|
||||
"""
|
||||
try:
|
||||
messages = db.list_messages_for_model(sid)
|
||||
if len(messages) <= SUMMARIZE_AFTER:
|
||||
return
|
||||
|
||||
session = db.get_session(sid) or {}
|
||||
previous = (session.get("summary") or "").strip()
|
||||
old = messages[:-KEEP_RECENT]
|
||||
|
||||
transcript = "\n".join(
|
||||
f"[{m['role']}] {m['content'][:600]}" for m in old
|
||||
)[-8000:]
|
||||
if previous:
|
||||
transcript = f"[résumé existant] {previous}\n{transcript}"
|
||||
|
||||
messages = [
|
||||
{"role": "system", "content": _SUMMARY_PROMPT},
|
||||
{"role": "user", "content": transcript},
|
||||
]
|
||||
opts = {**(options or {}), "temperature": 0.2, "num_predict": 350}
|
||||
# think=False : le résumé d'arrière-plan ne doit pas « réfléchir »
|
||||
# (latence ×3 sur un modèle thinking). Repli si paramètre refusé.
|
||||
think: bool | None = False
|
||||
while True:
|
||||
text = ""
|
||||
try:
|
||||
async for chunk in ollama.chat(
|
||||
model, messages, options=opts, think=think,
|
||||
keep_alive=keep_alive, stream=True,
|
||||
):
|
||||
text += chunk.get("message", {}).get("content", "")
|
||||
if chunk.get("done"):
|
||||
break
|
||||
break
|
||||
except OllamaError as exc:
|
||||
if think is False and "think" in str(exc).lower():
|
||||
think = None
|
||||
continue
|
||||
raise
|
||||
|
||||
if text.strip():
|
||||
db.set_session_summary(sid, text.strip())
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
# Best-effort : un échec de résumé ne doit jamais gêner le chat.
|
||||
logger.warning("Résumé de session %s impossible : %s", sid, exc)
|
||||
@@ -1,154 +0,0 @@
|
||||
"""Mémoire en notes Markdown : l'agent écrit lui-même ce qu'il veut retenir.
|
||||
|
||||
Repris d'AJEAN (github.com/nathaninline/jean) : des notes lisibles rangées
|
||||
dans ``$DATA_DIR/MEMORY/``, que le modèle consulte et enregistre EXPLICITEMENT.
|
||||
|
||||
Face au RAG vectoriel, trois avantages décisifs :
|
||||
- aucun modèle d'embedding requis (marche dès l'installation) ;
|
||||
- le contenu est inspectable et modifiable à la main ;
|
||||
- rien n'est mémorisé « à l'insu » : pas de vieille demande sans rapport qui
|
||||
ressurgit au milieu d'une nouvelle discussion.
|
||||
|
||||
Trois modes (config ``memory_mode``) :
|
||||
off — rien du tout ;
|
||||
ondemand — les outils memory_search / memory_save sont proposés au modèle ;
|
||||
always — idem, plus l'injection automatique des notes pertinentes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import unicodedata
|
||||
|
||||
from .config import settings
|
||||
|
||||
MODES = ("off", "ondemand", "always")
|
||||
|
||||
# Bornes : une note reste une note, pas une archive.
|
||||
_MAX_NOTE_CHARS = 4000
|
||||
_MAX_NOTES = 200
|
||||
_EXCERPT_CHARS = 400
|
||||
|
||||
|
||||
def _dir() -> str:
|
||||
path = os.path.join(os.path.abspath(settings.data_dir), "MEMORY")
|
||||
os.makedirs(path, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
def _slug(title: str) -> str:
|
||||
"""Nom de fichier sûr dérivé du titre (confiné au dossier MEMORY)."""
|
||||
text = unicodedata.normalize("NFKD", title or "").encode("ascii", "ignore").decode()
|
||||
text = re.sub(r"[^a-zA-Z0-9]+", "-", text).strip("-").lower()
|
||||
return (text or "note")[:60]
|
||||
|
||||
|
||||
def _tokens(text: str) -> set[str]:
|
||||
"""Mots significatifs, sans accents ni casse (>= 3 lettres)."""
|
||||
text = unicodedata.normalize("NFKD", text or "").encode("ascii", "ignore").decode()
|
||||
return {w for w in re.findall(r"[a-z0-9]{3,}", text.lower())}
|
||||
|
||||
|
||||
def save_note(title: str, content: str) -> dict:
|
||||
"""Crée ou remplace une note. Renvoie {ok, summary}."""
|
||||
title = (title or "").strip()
|
||||
content = (content or "").strip()
|
||||
if not title:
|
||||
return {"ok": False, "summary": "titre vide"}
|
||||
if not content:
|
||||
return {"ok": False, "summary": "contenu vide"}
|
||||
|
||||
content = content[:_MAX_NOTE_CHARS]
|
||||
path = os.path.join(_dir(), f"{_slug(title)}.md")
|
||||
existed = os.path.isfile(path)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write(f"# {title}\n\n{content}\n")
|
||||
|
||||
_prune()
|
||||
verb = "mise à jour" if existed else "enregistrée"
|
||||
return {"ok": True, "summary": f"note {verb} : {title}"}
|
||||
|
||||
|
||||
def _prune() -> None:
|
||||
"""Garde les notes les plus récentes (borne dure, jamais de purge totale)."""
|
||||
files = [os.path.join(_dir(), n) for n in os.listdir(_dir()) if n.endswith(".md")]
|
||||
if len(files) <= _MAX_NOTES:
|
||||
return
|
||||
files.sort(key=lambda p: os.path.getmtime(p), reverse=True)
|
||||
for path in files[_MAX_NOTES:]:
|
||||
try:
|
||||
os.remove(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _read_all() -> list[tuple[str, str]]:
|
||||
"""[(titre, corps)] de toutes les notes."""
|
||||
out = []
|
||||
try:
|
||||
names = sorted(os.listdir(_dir()))
|
||||
except OSError:
|
||||
return out
|
||||
for name in names:
|
||||
if not name.endswith(".md"):
|
||||
continue
|
||||
try:
|
||||
with open(os.path.join(_dir(), name), encoding="utf-8", errors="replace") as f:
|
||||
raw = f.read()
|
||||
except OSError:
|
||||
continue
|
||||
first, _, body = raw.partition("\n")
|
||||
title = first.lstrip("# ").strip() or name[:-3]
|
||||
out.append((title, body.strip()))
|
||||
return out
|
||||
|
||||
|
||||
def search_notes(query: str, limit: int = 3) -> list[dict]:
|
||||
"""Notes les plus proches de la requête (recouvrement de mots-clés)."""
|
||||
wanted = _tokens(query)
|
||||
if not wanted:
|
||||
return []
|
||||
scored: list[tuple[int, str, str]] = []
|
||||
for title, body in _read_all():
|
||||
score = len(wanted & _tokens(f"{title} {body}"))
|
||||
if score:
|
||||
scored.append((score, title, body))
|
||||
scored.sort(key=lambda item: -item[0])
|
||||
return [
|
||||
{"title": title, "content": body[:_EXCERPT_CHARS]}
|
||||
for _, title, body in scored[:limit]
|
||||
]
|
||||
|
||||
|
||||
def list_notes() -> list[dict]:
|
||||
"""Inventaire des notes (pour l'UI / l'inspection)."""
|
||||
notes = []
|
||||
for title, body in _read_all():
|
||||
path = os.path.join(_dir(), f"{_slug(title)}.md")
|
||||
notes.append({
|
||||
"title": title,
|
||||
"chars": len(body),
|
||||
"updated_at": os.path.getmtime(path) if os.path.isfile(path) else time.time(),
|
||||
})
|
||||
return sorted(notes, key=lambda n: -n["updated_at"])
|
||||
|
||||
|
||||
def delete_note(title: str) -> bool:
|
||||
path = os.path.join(_dir(), f"{_slug(title)}.md")
|
||||
if not os.path.isfile(path):
|
||||
return False
|
||||
os.remove(path)
|
||||
return True
|
||||
|
||||
|
||||
def recall_block(message: str) -> str | None:
|
||||
"""Bloc de contexte à injecter en mode ``always`` (None si rien de net)."""
|
||||
hits = search_notes(message, limit=2)
|
||||
if not hits:
|
||||
return None
|
||||
body = "\n\n".join(f"## {h['title']}\n{h['content']}" for h in hits)
|
||||
return (
|
||||
"Notes que tu avais enregistrées et qui semblent liées à cette demande "
|
||||
"(ignore-les si elles ne s'appliquent pas) :\n" + body
|
||||
)
|
||||
@@ -1,251 +0,0 @@
|
||||
"""Client HTTP léger pour l'API Ollama.
|
||||
|
||||
On utilise httpx directement (plutôt que le SDK) pour garder le contrôle
|
||||
total sur le streaming et n'embarquer aucune dépendance superflue.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import time
|
||||
from typing import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from .config import settings
|
||||
|
||||
|
||||
class OllamaError(RuntimeError):
|
||||
"""Erreur renvoyée par Ollama (statut HTTP ≥ 400 ou champ ``error`` dans le flux).
|
||||
|
||||
Ollama signale certains échecs *au milieu* d'un flux streaming (HTTP 200)
|
||||
via une ligne JSON ``{"error": "..."}`` — typiquement un débordement mémoire
|
||||
ou un contexte trop grand. On lève alors cette exception pour que l'appelant
|
||||
la remonte à l'utilisateur au lieu de l'avaler silencieusement.
|
||||
"""
|
||||
|
||||
|
||||
# Connexion rapide à échouer si Ollama est injoignable, mais lecture sans limite :
|
||||
# une génération longue (ou un chargement de modèle sur CPU) ne doit pas couper.
|
||||
_STREAM_TIMEOUT = httpx.Timeout(connect=10.0, read=None, write=30.0, pool=10.0)
|
||||
|
||||
# Un chat déclenche plusieurs appels Ollama (routage, embed, plan, agent…) :
|
||||
# un pool keep-alive partagé évite un handshake TCP à chaque appel.
|
||||
_LIMITS = httpx.Limits(
|
||||
max_connections=20, max_keepalive_connections=10, keepalive_expiry=30.0
|
||||
)
|
||||
|
||||
|
||||
async def _raise_for_stream_status(resp: httpx.Response) -> None:
|
||||
"""Lève une ``OllamaError`` détaillée si la réponse streaming est en erreur.
|
||||
|
||||
Sur une réponse en flux, ``raise_for_status`` n'inclut pas le corps ; on le
|
||||
lit explicitement pour exposer le message d'Ollama (modèle absent, etc.).
|
||||
"""
|
||||
if resp.status_code < 400:
|
||||
return
|
||||
body = await resp.aread()
|
||||
detail = body.decode(errors="replace").strip()
|
||||
try:
|
||||
detail = json.loads(detail).get("error", detail)
|
||||
except (json.JSONDecodeError, AttributeError):
|
||||
pass
|
||||
raise OllamaError(f"Ollama a renvoyé {resp.status_code} : {detail[:500]}")
|
||||
|
||||
|
||||
class OllamaClient:
|
||||
"""Enveloppe asynchrone autour de l'API REST d'Ollama."""
|
||||
|
||||
def __init__(self, host: str | None = None) -> None:
|
||||
self.host = (host or settings.ollama_host).rstrip("/")
|
||||
self._client: httpx.AsyncClient | None = None
|
||||
# Cache court de /api/tags : la liste des modèles installés change
|
||||
# rarement mais est consultée par plusieurs modules à chaque message.
|
||||
self._tags_at = 0.0
|
||||
self._tags: list[dict] = []
|
||||
|
||||
def _http(self) -> httpx.AsyncClient:
|
||||
"""Client partagé (pool keep-alive), créé paresseusement.
|
||||
|
||||
Un seul worker uvicorn / une seule boucle : la création lazy est sûre.
|
||||
Le garde ``is_closed`` recrée le client si un arrêt l'a fermé.
|
||||
"""
|
||||
if self._client is None or self._client.is_closed:
|
||||
self._client = httpx.AsyncClient(
|
||||
timeout=_STREAM_TIMEOUT, follow_redirects=True, limits=_LIMITS
|
||||
)
|
||||
return self._client
|
||||
|
||||
async def aclose(self) -> None:
|
||||
"""Ferme le pool partagé (appelé au shutdown de l'app)."""
|
||||
if self._client is not None and not self._client.is_closed:
|
||||
await self._client.aclose()
|
||||
|
||||
async def ping(self) -> dict:
|
||||
"""Vérifie la connexion et renvoie la version d'Ollama."""
|
||||
resp = await self._http().get(f"{self.host}/api/version", timeout=5.0)
|
||||
resp.raise_for_status()
|
||||
return resp.json()
|
||||
|
||||
async def list_models(self) -> list[dict]:
|
||||
"""Liste les modèles installés localement (/api/tags), sans cache."""
|
||||
resp = await self._http().get(f"{self.host}/api/tags", timeout=10.0)
|
||||
resp.raise_for_status()
|
||||
models = resp.json().get("models", [])
|
||||
self._tags, self._tags_at = models, time.monotonic()
|
||||
return models
|
||||
|
||||
async def list_models_cached(self, ttl: float = 30.0) -> list[dict]:
|
||||
"""Comme ``list_models`` mais avec un cache court partagé.
|
||||
|
||||
Utilisé par les chemins chauds (routage code, résolution embed, route
|
||||
/api/models) pour ne pas marteler /api/tags à chaque message.
|
||||
"""
|
||||
if self._tags and time.monotonic() - self._tags_at < ttl:
|
||||
return self._tags
|
||||
return await self.list_models()
|
||||
|
||||
def invalidate_tags_cache(self) -> None:
|
||||
"""Force un rafraîchissement après un pull ou une suppression de modèle."""
|
||||
self._tags_at = 0.0
|
||||
self._tags = []
|
||||
|
||||
async def embed(
|
||||
self, model: str, texts: list[str], keep_alive: str = "30m"
|
||||
) -> list[list[float]]:
|
||||
"""Vecteurs d'embedding pour une liste de textes (/api/embed).
|
||||
|
||||
``keep_alive`` long : le modèle d'embedding est minuscule (<0,5 Go) et
|
||||
sollicité à chaque message (recall + indexation) — le laisser chargé
|
||||
évite un aller-retour VRAM permanent avec le modèle de chat.
|
||||
"""
|
||||
resp = await self._http().post(
|
||||
f"{self.host}/api/embed",
|
||||
json={"model": model, "input": texts, "keep_alive": keep_alive},
|
||||
timeout=30.0,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp.json().get("embeddings", [])
|
||||
|
||||
async def ps(self) -> list[dict]:
|
||||
"""Modèles actuellement chargés et leur répartition VRAM/CPU (/api/ps)."""
|
||||
resp = await self._http().get(f"{self.host}/api/ps", timeout=5.0)
|
||||
resp.raise_for_status()
|
||||
return resp.json().get("models", [])
|
||||
|
||||
async def show(self, name: str) -> dict:
|
||||
"""Métadonnées détaillées d'un modèle (/api/show)."""
|
||||
resp = await self._http().post(
|
||||
f"{self.host}/api/show", json={"name": name}, timeout=15.0
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp.json()
|
||||
|
||||
async def delete_model(self, name: str) -> dict:
|
||||
"""Supprime un modèle installé (/api/delete)."""
|
||||
resp = await self._http().request(
|
||||
"DELETE", f"{self.host}/api/delete", json={"model": name}, timeout=30.0
|
||||
)
|
||||
resp.raise_for_status()
|
||||
self.invalidate_tags_cache()
|
||||
return resp.json() if resp.content else {"status": "success"}
|
||||
|
||||
async def pull_model(self, name: str) -> AsyncIterator[dict]:
|
||||
"""Télécharge un modèle en streamant la progression (/api/pull)."""
|
||||
async with self._http().stream(
|
||||
"POST",
|
||||
f"{self.host}/api/pull",
|
||||
json={"name": name},
|
||||
timeout=_STREAM_TIMEOUT,
|
||||
) as resp:
|
||||
await _raise_for_stream_status(resp)
|
||||
async for line in resp.aiter_lines():
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
chunk = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(chunk, dict) and chunk.get("error"):
|
||||
raise OllamaError(str(chunk["error"]))
|
||||
yield chunk
|
||||
self.invalidate_tags_cache()
|
||||
|
||||
async def warm(
|
||||
self, model: str, keep_alive: str = "30m", options: dict | None = None
|
||||
) -> dict:
|
||||
"""Précharge un modèle en VRAM sans générer (/api/generate sans prompt).
|
||||
|
||||
Le paramètre keep_alive fixe la durée de rétention en mémoire. Les
|
||||
`options` (num_ctx, num_batch, num_gpu…) DOIVENT correspondre à celles du
|
||||
chat : sinon Ollama chargerait un runner distinct puis en rechargerait un
|
||||
autre au premier message — un double chargement très coûteux pour un gros
|
||||
modèle.
|
||||
"""
|
||||
payload: dict = {"model": model, "keep_alive": keep_alive, "stream": False}
|
||||
if options:
|
||||
payload["options"] = options
|
||||
# Le chargement se fait désormais en tâche de fond côté API Loki. On lui
|
||||
# laisse jusqu'à dix minutes pour les gros modèles ou un stockage lent.
|
||||
timeout = httpx.Timeout(connect=10.0, read=600.0, write=30.0, pool=10.0)
|
||||
for attempt in range(2):
|
||||
resp = await self._http().post(
|
||||
f"{self.host}/api/generate", json=payload, timeout=timeout
|
||||
)
|
||||
try:
|
||||
# Inclut le corps JSON d'Ollama dans l'erreur (OOM, runner…),
|
||||
# contrairement à raise_for_status qui ne montrait que « 500 ».
|
||||
await _raise_for_stream_status(resp)
|
||||
except OllamaError:
|
||||
if resp.status_code >= 500 and attempt == 0:
|
||||
await asyncio.sleep(2)
|
||||
continue
|
||||
raise
|
||||
return resp.json()
|
||||
raise OllamaError("préchargement interrompu sans réponse")
|
||||
|
||||
async def chat(
|
||||
self,
|
||||
model: str,
|
||||
messages: list[dict],
|
||||
*,
|
||||
tools: list[dict] | None = None,
|
||||
options: dict | None = None,
|
||||
think: bool | None = None,
|
||||
keep_alive: str | None = None,
|
||||
stream: bool = True,
|
||||
) -> AsyncIterator[dict]:
|
||||
"""Conversation avec le modèle, en streaming token par token."""
|
||||
payload: dict = {"model": model, "messages": messages, "stream": stream}
|
||||
if tools:
|
||||
payload["tools"] = tools
|
||||
if options:
|
||||
payload["options"] = options
|
||||
# think=False désactive le raisonnement des modèles « thinking ».
|
||||
if think is not None:
|
||||
payload["think"] = think
|
||||
# keep_alive : durée de maintien du modèle en VRAM après la réponse.
|
||||
if keep_alive is not None:
|
||||
payload["keep_alive"] = keep_alive
|
||||
|
||||
async with self._http().stream(
|
||||
"POST", f"{self.host}/api/chat", json=payload, timeout=_STREAM_TIMEOUT
|
||||
) as resp:
|
||||
await _raise_for_stream_status(resp)
|
||||
async for line in resp.aiter_lines():
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
chunk = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
# Ligne partielle / non-JSON : on l'ignore plutôt que de
|
||||
# faire planter tout le flux.
|
||||
continue
|
||||
# Échec en cours de génération (OOM, contexte trop grand…) :
|
||||
# Ollama l'émet dans le flux avec HTTP 200. On le remonte.
|
||||
if isinstance(chunk, dict) and chunk.get("error"):
|
||||
raise OllamaError(str(chunk["error"]))
|
||||
yield chunk
|
||||
|
||||
|
||||
ollama = OllamaClient()
|
||||
@@ -1,155 +0,0 @@
|
||||
"""Mémoire long-terme (RAG) : l'agent se souvient des anciennes sessions.
|
||||
|
||||
Chaque échange (question + réponse) est vectorisé via /api/embed d'Ollama et
|
||||
stocké en SQLite. À chaque nouveau message, on recherche les souvenirs les
|
||||
plus proches (cosinus) dans les AUTRES sessions et on les injecte en contexte.
|
||||
|
||||
Tout est best-effort : sans modèle d'embedding installé, le RAG se désactive
|
||||
silencieusement (aucun impact sur le chat).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import time
|
||||
import uuid
|
||||
|
||||
import httpx
|
||||
|
||||
from . import db
|
||||
from .ollama_client import ollama
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Modèles d'embedding reconnus, par ordre de préférence.
|
||||
_EMBED_HINTS = ("nomic-embed", "mxbai-embed", "bge-", "snowflake-arctic-embed",
|
||||
"all-minilm", "embed")
|
||||
|
||||
_TOP_K = 3
|
||||
# Seuil de similarité volontairement élevé : une mémoire n'est rappelée que si
|
||||
# elle est FORTEMENT proche de la demande. Trop bas (0.45), un « jeu d'échecs »
|
||||
# rappelait une « appli sport » et déroutait le modèle.
|
||||
_MIN_SCORE = 0.6
|
||||
_MAX_MEMORIES = 2000 # au-delà, on élague les plus anciens
|
||||
|
||||
_embed_model_cache: dict = {"value": None, "checked_at": 0.0}
|
||||
|
||||
|
||||
def init_table() -> None:
|
||||
with db._LOCK, db._connect() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS memories (
|
||||
id TEXT PRIMARY KEY,
|
||||
session_id TEXT NOT NULL,
|
||||
content TEXT NOT NULL,
|
||||
embedding TEXT NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
async def resolve_embed_model(preference: str | None = None) -> str | None:
|
||||
"""Trouve le modèle d'embedding à utiliser (None = RAG indisponible)."""
|
||||
if preference and preference != "auto":
|
||||
return preference
|
||||
|
||||
# Cache 60 s pour ne pas marteler /api/tags.
|
||||
now = time.time()
|
||||
if now - _embed_model_cache["checked_at"] < 60:
|
||||
return _embed_model_cache["value"]
|
||||
|
||||
value = None
|
||||
try:
|
||||
for m in await ollama.list_models_cached():
|
||||
name = (m.get("name") or "").lower()
|
||||
if any(h in name for h in _EMBED_HINTS):
|
||||
value = m["name"]
|
||||
break
|
||||
except (httpx.HTTPError, OSError):
|
||||
value = None
|
||||
|
||||
_embed_model_cache.update(value=value, checked_at=now)
|
||||
return value
|
||||
|
||||
|
||||
def _cosine(a: list[float], b: list[float]) -> float:
|
||||
dot = sum(x * y for x, y in zip(a, b))
|
||||
na = math.sqrt(sum(x * x for x in a))
|
||||
nb = math.sqrt(sum(x * x for x in b))
|
||||
return dot / (na * nb) if na and nb else 0.0
|
||||
|
||||
|
||||
def _score_rows(qvec: list[float], rows: list) -> list[str]:
|
||||
"""Scoring cosinus sur toutes les mémoires — CPU pur, à lancer via to_thread.
|
||||
|
||||
Jusqu'à _MAX_MEMORIES vecteurs : la boucle Python bloquerait l'event loop
|
||||
(et donc tous les SSE en cours) pendant plusieurs dizaines de ms.
|
||||
"""
|
||||
scored: list[tuple[float, str]] = []
|
||||
for row in rows:
|
||||
try:
|
||||
score = _cosine(qvec, json.loads(row["embedding"]))
|
||||
except (ValueError, TypeError):
|
||||
continue
|
||||
if score >= _MIN_SCORE:
|
||||
scored.append((score, row["content"]))
|
||||
scored.sort(reverse=True)
|
||||
return [c for _, c in scored[:_TOP_K]]
|
||||
|
||||
|
||||
async def index_exchange(
|
||||
sid: str, user_text: str, assistant_text: str, *, embed_model: str | None
|
||||
) -> None:
|
||||
"""Indexe un échange terminé (tâche d'arrière-plan, best-effort)."""
|
||||
model = await resolve_embed_model(embed_model)
|
||||
if not model:
|
||||
return
|
||||
content = f"Q: {user_text[:500]}\nR: {assistant_text[:800]}"
|
||||
try:
|
||||
vectors = await ollama.embed(model, [content])
|
||||
if not vectors:
|
||||
return
|
||||
with db._LOCK, db._connect() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO memories (id, session_id, content, embedding, created_at)"
|
||||
" VALUES (?, ?, ?, ?, ?)",
|
||||
(uuid.uuid4().hex, sid, content,
|
||||
json.dumps(vectors[0]), time.time()),
|
||||
)
|
||||
# Élagage des souvenirs les plus anciens.
|
||||
conn.execute(
|
||||
"DELETE FROM memories WHERE id IN ("
|
||||
" SELECT id FROM memories ORDER BY created_at DESC"
|
||||
f" LIMIT -1 OFFSET {_MAX_MEMORIES})"
|
||||
)
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
logger.warning("Indexation RAG impossible : %s", exc)
|
||||
|
||||
|
||||
async def recall(
|
||||
sid: str, query: str, *, embed_model: str | None
|
||||
) -> list[str]:
|
||||
"""Souvenirs pertinents issus des AUTRES sessions (top-k, score minimal)."""
|
||||
model = await resolve_embed_model(embed_model)
|
||||
if not model:
|
||||
return []
|
||||
try:
|
||||
vectors = await ollama.embed(model, [query[:800]])
|
||||
if not vectors:
|
||||
return []
|
||||
qvec = vectors[0]
|
||||
|
||||
with db._LOCK, db._connect() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT content, embedding FROM memories WHERE session_id != ?",
|
||||
(sid,),
|
||||
).fetchall()
|
||||
|
||||
return await asyncio.to_thread(_score_rows, qvec, rows)
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
logger.warning("Rappel RAG impossible : %s", exc)
|
||||
return []
|
||||
@@ -1,69 +0,0 @@
|
||||
"""Routeur automatique : moteur code (Aider) ou boucle agent classique ?
|
||||
|
||||
Invisible pour l'utilisateur : chaque message est classé par une heuristique
|
||||
lexicale instantanée. Les cas ambigus partent vers la boucle agent, qui garde
|
||||
l'outil `code_task` en secours — plus d'appel LLM bloquant avant le premier
|
||||
token (l'ancien micro-classifieur coûtait un aller-retour modèle complet et
|
||||
chargeait un runner divergent).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
# Vocabulaire fortement lié au code / au développement.
|
||||
_STRONG = re.compile(
|
||||
r"\b(code|coder?|script|fonction|classe|refactor|bug|d[ée]bug|html|css|"
|
||||
r"javascript|js|python|typescript|react|api|composant|page|site|landing|"
|
||||
r"formulaire|d[ée]veloppe|impl[ée]mente|programme|widget|frontend|backend)\b",
|
||||
re.I,
|
||||
)
|
||||
# Extensions de fichiers mentionnées explicitement.
|
||||
_FILE_EXT = re.compile(
|
||||
r"\.(py|js|ts|tsx|jsx|html?|css|json|md|sh|sql|ya?ml|go|rs|java|php|vue)\b", re.I
|
||||
)
|
||||
# Verbes d'action de création/modification.
|
||||
_ACTION = re.compile(
|
||||
r"\b(cr[ée]e[rs]?|modifie[rs]?|corrige[rs]?|ajoute[rs]?|[ée]cri[st]|refai[st]|"
|
||||
r"am[ée]liore[rs]?|fixe?|update|change[rs]?|construis|g[ée]n[èe]re[rs]?)\b",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def score_code_task(message: str) -> int:
|
||||
"""Score heuristique : >= 3 -> code, <= 2 -> agent."""
|
||||
score = 0
|
||||
if _STRONG.search(message):
|
||||
score += 2
|
||||
if _FILE_EXT.search(message):
|
||||
score += 2
|
||||
if _ACTION.search(message):
|
||||
score += 1
|
||||
if "```" in message:
|
||||
score += 2
|
||||
return score
|
||||
|
||||
|
||||
def is_code_task(message: str) -> bool:
|
||||
"""Décision : heuristique pure, aucune requête modèle."""
|
||||
return score_code_task(message) >= 3
|
||||
|
||||
|
||||
# Formulations de « suite de travail » : courtes, sans vocabulaire code
|
||||
# explicite, mais qui prolongent clairement la tâche en cours.
|
||||
_FOLLOWUP = re.compile(
|
||||
r"\b(continue[rs]?|reprend[s]?|poursui[st]|termine[rs]?|finis|"
|
||||
r"rajoute[rs]?|enl[èe]ve[rs]?|retire[rs]?|supprime[rs]?|d[ée]place[rs]?|"
|
||||
r"agrandi[st]|r[ée]dui[st]|remet[s]?|remplace[rs]?|inverse[rs]?|"
|
||||
r"plut[ôo]t|aussi|encore|pareil|m[êe]me chose|comme avant)\b",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def is_code_followup(message: str) -> bool:
|
||||
"""Reprise probable d'un travail de code en cours.
|
||||
|
||||
À n'utiliser que si le tour précédent de la session était du code : un
|
||||
« ajoute un bouton rouge » isolé n'est pas du code, mais après un
|
||||
code_task, si.
|
||||
"""
|
||||
return bool(_FOLLOWUP.search(message)) or score_code_task(message) >= 1
|
||||
Whitespace-only changes.
@@ -1,42 +0,0 @@
|
||||
"""Routes du benchmark de modèles."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
from fastapi import APIRouter
|
||||
from fastapi.responses import StreamingResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import bench
|
||||
|
||||
router = APIRouter(prefix="/api/bench", tags=["bench"])
|
||||
|
||||
|
||||
class BenchRequest(BaseModel):
|
||||
model: str
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def scores() -> dict:
|
||||
return {"scores": bench.get_scores()}
|
||||
|
||||
|
||||
@router.post("")
|
||||
async def run(req: BenchRequest) -> StreamingResponse:
|
||||
async def event_stream():
|
||||
try:
|
||||
async for ev in bench.run_bench(req.model):
|
||||
etype = ev.pop("type")
|
||||
yield f"event: {etype}\ndata: {json.dumps(ev, ensure_ascii=False)}\n\n"
|
||||
except Exception as exc:
|
||||
payload = json.dumps(
|
||||
{"message": f"benchmark interrompu : {str(exc)[:200]}"},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
yield f"event: error\ndata: {payload}\n\n"
|
||||
|
||||
return StreamingResponse(
|
||||
event_stream(),
|
||||
media_type="text/event-stream",
|
||||
headers={"Cache-Control": "no-cache, no-transform", "X-Accel-Buffering": "no"},
|
||||
)
|
||||
@@ -1,746 +0,0 @@
|
||||
"""Route de conversation : routage automatique agent / moteur code + SSE.
|
||||
|
||||
Flux :
|
||||
1. on enregistre le message utilisateur ;
|
||||
2. le routeur classe la demande : tâche de code -> moteur code (Aider),
|
||||
sinon -> boucle agentique classique (qui peut elle-même appeler code_task) ;
|
||||
3. on relaie tokens et événements d'outils au client (SSE) ;
|
||||
4. on enregistre la réponse finale de l'assistant + le récap des outils.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from contextlib import suppress
|
||||
|
||||
import httpx
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from fastapi.responses import StreamingResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import (
|
||||
agent_config, coder, db, enhance, memory, memory_notes, rag, skills, tools,
|
||||
)
|
||||
from .. import router as msg_router
|
||||
from ..tools import check_html, _safe_path
|
||||
from ..agent import run_agent
|
||||
from ..config import settings
|
||||
from ..ollama_client import ollama
|
||||
|
||||
router = APIRouter(prefix="/api", tags=["chat"])
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ChatRequest(BaseModel):
|
||||
session_id: str
|
||||
content: str
|
||||
model: str | None = None
|
||||
# Mode d'exécution : "plan" (lecture seule), "build" (normal), "yolo" (auto).
|
||||
mode: str = "build"
|
||||
|
||||
|
||||
# Outils autorisés en mode Plan : lecture/analyse uniquement.
|
||||
_READONLY_TOOLS = {"read_file", "list_dir", "grep_search", "run_check"}
|
||||
|
||||
|
||||
def _apply_mode(cfg: dict, mode: str) -> dict:
|
||||
"""Adapte la config au mode d'exécution choisi dans le composer."""
|
||||
cfg = dict(cfg)
|
||||
if mode == "plan":
|
||||
# Lecture seule : aucune écriture, aucun shell, aucun moteur code.
|
||||
cfg["tools"] = {
|
||||
name: (on and name in _READONLY_TOOLS)
|
||||
for name, on in cfg["tools"].items()
|
||||
}
|
||||
cfg["plan_mode"] = True
|
||||
elif mode == "yolo":
|
||||
# Autonomie maximale : plus de validation shell.
|
||||
cfg["confirm_shell"] = False
|
||||
# "build" : comportement par défaut (inchangé).
|
||||
return cfg
|
||||
|
||||
|
||||
async def _placement_notice(model: str) -> str | None:
|
||||
"""Avertit si le modèle tourne (au moins en partie) hors du GPU.
|
||||
|
||||
C'est LA cause d'un débit ridicule (quelques jetons/s) : Ollama a placé
|
||||
tout ou partie des couches en RAM. Loki ne peut pas le corriger — le
|
||||
placement appartient à Ollama — mais il peut le NOMMER, au lieu de
|
||||
laisser croire à un blocage de l'application.
|
||||
"""
|
||||
try:
|
||||
loaded = await ollama.ps()
|
||||
except (httpx.HTTPError, OSError):
|
||||
return None
|
||||
|
||||
for m in loaded:
|
||||
if (m.get("name") or m.get("model")) != model:
|
||||
continue
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
if size <= 0 or vram >= size * 0.99:
|
||||
return None
|
||||
pct = int(vram / size * 100)
|
||||
in_ram = (size - vram) / 1e9
|
||||
where = "entièrement en RAM" if pct == 0 else f"{pct} % en VRAM"
|
||||
return (
|
||||
f"⚠️ {model} tourne {where} ({in_ram:.1f} Go hors GPU) : "
|
||||
"le débit sera de quelques jetons par seconde. Réduis le contexte, "
|
||||
"prends un modèle plus petit, ou répartis sur tes GPU côté Ollama."
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _loading_message(model: str, seconds: float) -> str:
|
||||
"""Message d'attente avant le premier jeton (chargement du modèle)."""
|
||||
waited = int(seconds)
|
||||
delay = f"{waited // 60} min" if waited >= 60 else f"{waited} s"
|
||||
text = f"Chargement de {model} en mémoire… ({delay})"
|
||||
if waited >= 90:
|
||||
# On CONSTATE l'attente sans présumer de la cause : selon la machine,
|
||||
# un gros modèle peut se répartir sur plusieurs GPU, ou déborder sur
|
||||
# le CPU. Le panneau Matériel donne le placement réel.
|
||||
text += " — placement réel visible dans Réglages › Matériel."
|
||||
return text
|
||||
|
||||
|
||||
def _merge_system(convo: list[dict], extras: list[str]) -> None:
|
||||
"""Garantit UN SEUL message système, en première position.
|
||||
|
||||
Beaucoup de templates Ollama (Gemma, Mistral…) contiennent un garde
|
||||
« {% if role == 'system' and not loop.first %}{{ raise_exception(...) }} » :
|
||||
tout message système qui n'est pas le premier fait échouer la requête
|
||||
entière en 400 (« System message must be at the beginning »), y compris
|
||||
la génération du parseur d'outils.
|
||||
|
||||
On fusionne donc, dans l'ordre : l'invite système, les éventuels systèmes
|
||||
égarés (résumé de session…) puis les consignes du tour. Modifie ``convo``
|
||||
sur place.
|
||||
"""
|
||||
head = ""
|
||||
stray: list[str] = []
|
||||
rest: list[dict] = []
|
||||
for i, msg in enumerate(convo):
|
||||
if msg.get("role") == "system":
|
||||
text = (msg.get("content") or "").strip()
|
||||
if i == 0:
|
||||
head = text
|
||||
elif text:
|
||||
stray.append(text)
|
||||
else:
|
||||
rest.append(msg)
|
||||
|
||||
blocks = [b for b in (head, *stray, *(e.strip() for e in extras if e)) if b]
|
||||
convo[:] = ([{"role": "system", "content": "\n\n".join(blocks)}] if blocks else []) + rest
|
||||
|
||||
|
||||
def _sse(event: str, data: dict) -> str:
|
||||
return f"event: {event}\ndata: {json.dumps(data, ensure_ascii=False)}\n\n"
|
||||
|
||||
|
||||
def _prev_was_code(history: list[dict]) -> bool:
|
||||
"""Le dernier tour assistant de la session était-il un travail de code ?"""
|
||||
for m in reversed(history):
|
||||
if m["role"] != "assistant":
|
||||
continue
|
||||
meta = m.get("meta") or {}
|
||||
if meta.get("engine") == "code":
|
||||
return True
|
||||
tools = meta.get("tools") or []
|
||||
return any(
|
||||
t.get("name") in ("code_task", "write_file", "edit_file")
|
||||
for t in tools
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
def _session_code_context(history: list[dict]) -> tuple[str, list[str]]:
|
||||
"""Récap compact du travail en cours + fichiers touchés dans la session.
|
||||
|
||||
Le moteur code ne reçoit que le message courant : sur une reprise
|
||||
(« corrige les bugs »), sans ce récap il ignore quel fichier, quel projet
|
||||
et quelle demande d'origine. Les fichiers touchés servent aussi de cible
|
||||
par défaut pour Aider.
|
||||
"""
|
||||
root = tools.active_root()
|
||||
files: list[str] = []
|
||||
for m in history:
|
||||
if m["role"] != "assistant":
|
||||
continue
|
||||
for t in (m.get("meta") or {}).get("tools") or []:
|
||||
candidates: list[str] = []
|
||||
path = (t.get("args") or {}).get("path")
|
||||
if t.get("name") in ("write_file", "edit_file") and path:
|
||||
candidates.append(str(path))
|
||||
for f in t.get("files") or []:
|
||||
candidates.append(str(f))
|
||||
for c in candidates:
|
||||
rel = c.replace("\\", "/").lstrip("./")
|
||||
if rel not in files and os.path.isfile(os.path.join(root, rel)):
|
||||
files.append(rel)
|
||||
|
||||
user_msgs = [m["content"].strip() for m in history if m["role"] == "user"]
|
||||
lines: list[str] = []
|
||||
if user_msgs:
|
||||
lines.append(f"- Demande initiale : {user_msgs[0][:200]}")
|
||||
for prev in user_msgs[-2:]:
|
||||
if prev != user_msgs[0]:
|
||||
lines.append(f"- Puis : {prev[:200]}")
|
||||
if files:
|
||||
lines.append(f"- Fichiers déjà créés/modifiés : {', '.join(files[:8])}")
|
||||
recap = (
|
||||
"Contexte de la session (travail en cours) :\n" + "\n".join(lines)
|
||||
if lines else ""
|
||||
)
|
||||
return recap, files
|
||||
|
||||
|
||||
def _workspace_listing(limit: int = 40) -> list[str]:
|
||||
"""Chemins relatifs des fichiers de la racine active (aperçu compact)."""
|
||||
root = tools.active_root()
|
||||
out: list[str] = []
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
dirnames[:] = [d for d in dirnames if not d.startswith(".")]
|
||||
for name in sorted(filenames):
|
||||
if name.startswith("."):
|
||||
continue
|
||||
rel = os.path.relpath(os.path.join(dirpath, name), root)
|
||||
out.append(rel.replace("\\", "/"))
|
||||
if len(out) >= limit:
|
||||
return out
|
||||
return out
|
||||
|
||||
|
||||
_FILE_MENTION = re.compile(r"[\w][\w./\\-]*\.[a-z0-9]{1,5}\b", re.I)
|
||||
|
||||
|
||||
def _mentioned_files(text: str) -> list[str]:
|
||||
"""Fichiers du workspace explicitement cités dans le message.
|
||||
|
||||
Transmis au moteur code pour qu'Aider travaille directement sur les bons
|
||||
fichiers au lieu de deviner via la repo map.
|
||||
"""
|
||||
root = tools.active_root()
|
||||
out: list[str] = []
|
||||
for raw in _FILE_MENTION.findall(text):
|
||||
rel = raw.replace("\\", "/").lstrip("./")
|
||||
if os.path.isfile(os.path.join(root, rel)) and rel not in out:
|
||||
out.append(rel)
|
||||
return out[:8]
|
||||
|
||||
|
||||
async def _run_aider_keepalive(
|
||||
instruction: str, model: str, files: list[str] | None = None
|
||||
):
|
||||
"""Lance Aider dans un thread en gardant le flux SSE vivant."""
|
||||
# Racine résolue AVANT le thread : la contextvar projet ne suit pas
|
||||
# dans asyncio.to_thread.
|
||||
root = tools.active_root()
|
||||
task = asyncio.create_task(
|
||||
asyncio.to_thread(coder.run_code_task, instruction, model, files, root)
|
||||
)
|
||||
while not task.done():
|
||||
await asyncio.sleep(10)
|
||||
if not task.done():
|
||||
yield None # signal keepalive
|
||||
yield await task
|
||||
|
||||
|
||||
async def _code_stream(
|
||||
req: ChatRequest,
|
||||
model: str,
|
||||
*,
|
||||
extra: str = "",
|
||||
plan: list[str] | None = None,
|
||||
files: list[str] | None = None,
|
||||
):
|
||||
"""Chemin « moteur code » : Aider + vérification HTML avec auto-correction."""
|
||||
instruction = req.content + (extra or "")
|
||||
yield _sse("tool_call", {"name": "code_task", "args": {"instruction": req.content}})
|
||||
|
||||
result = None
|
||||
async for item in _run_aider_keepalive(instruction, model, files):
|
||||
if item is None:
|
||||
yield ": keepalive\n\n"
|
||||
else:
|
||||
result = item
|
||||
|
||||
status = "ok" if result.get("ok") else "error"
|
||||
record = {
|
||||
"name": "code_task",
|
||||
"args": {"instruction": req.content},
|
||||
"summary": result.get("summary", "terminé"),
|
||||
"status": status,
|
||||
}
|
||||
tools_meta = [record]
|
||||
yield _sse("tool_result", record)
|
||||
|
||||
all_files = list(result.get("files") or [])
|
||||
|
||||
# Vérification des pages HTML produites + une passe d'auto-correction.
|
||||
html_issues: list[str] = []
|
||||
for f in all_files:
|
||||
if f.lower().endswith((".html", ".htm")):
|
||||
try:
|
||||
issues = check_html(_safe_path(f))
|
||||
except Exception:
|
||||
issues = []
|
||||
if issues:
|
||||
html_issues.append(f"{f} : " + " ; ".join(issues))
|
||||
|
||||
if html_issues and result.get("ok"):
|
||||
yield _sse("tool_call", {"name": "html_check", "args": {"path": "vérification"}})
|
||||
yield _sse("tool_result", {
|
||||
"name": "html_check", "args": {"path": "vérification"},
|
||||
"summary": " | ".join(html_issues)[:200], "status": "error",
|
||||
})
|
||||
tools_meta.append({
|
||||
"name": "html_check", "args": {},
|
||||
"summary": " | ".join(html_issues)[:200], "status": "error",
|
||||
})
|
||||
fix_instruction = (
|
||||
"Corrige ces problèmes détectés dans les fichiers HTML, sans rien "
|
||||
"casser d'autre :\n" + "\n".join(html_issues)
|
||||
)
|
||||
fix = None
|
||||
async for item in _run_aider_keepalive(fix_instruction, model):
|
||||
if item is None:
|
||||
yield ": keepalive\n\n"
|
||||
else:
|
||||
fix = item
|
||||
fix_rec = {
|
||||
"name": "code_task",
|
||||
"args": {"instruction": "auto-correction HTML"},
|
||||
"summary": fix.get("summary", "terminé"),
|
||||
"status": "ok" if fix.get("ok") else "error",
|
||||
}
|
||||
tools_meta.append(fix_rec)
|
||||
yield _sse("tool_call", {"name": "code_task", "args": fix_rec["args"]})
|
||||
yield _sse("tool_result", fix_rec)
|
||||
for f in fix.get("files") or []:
|
||||
if f not in all_files:
|
||||
all_files.append(f)
|
||||
if fix.get("text"):
|
||||
result["text"] = (result.get("text") or "") + "\n\n" + fix["text"]
|
||||
|
||||
# Cartes par fichier modifié (réutilise le rendu write_file de l'UI).
|
||||
for f in all_files:
|
||||
file_rec = {
|
||||
"name": "write_file",
|
||||
"args": {"path": f},
|
||||
"summary": "modifié par le moteur code",
|
||||
"status": "ok",
|
||||
}
|
||||
tools_meta.append(file_rec)
|
||||
yield _sse("tool_call", {"name": "write_file", "args": {"path": f}})
|
||||
yield _sse("tool_result", file_rec)
|
||||
|
||||
text = result.get("text") or (
|
||||
"" if result.get("ok") else f"⚠️ {result.get('summary', 'échec du moteur code')}"
|
||||
)
|
||||
if text:
|
||||
yield _sse("token", {"content": text})
|
||||
|
||||
meta: dict = {"tools": tools_meta, "engine": "code"}
|
||||
if plan:
|
||||
meta["plan"] = plan
|
||||
db.add_message(req.session_id, "assistant", text, model, meta=meta)
|
||||
yield _sse("done", {"content": text, "tools": tools_meta})
|
||||
|
||||
|
||||
@router.post("/chat")
|
||||
async def chat(req: ChatRequest) -> StreamingResponse:
|
||||
session = db.get_session(req.session_id)
|
||||
if not session:
|
||||
raise HTTPException(404, "session introuvable")
|
||||
|
||||
model = req.model or session.get("model") or settings.default_model
|
||||
cfg = _apply_mode(agent_config.get_config(model), req.mode)
|
||||
|
||||
# Projet de la session : re-racine outils, shell, Aider et aides de
|
||||
# contexte pour tout le tour. Projet disparu -> retour racine + notice.
|
||||
project = session.get("project") or None
|
||||
project_missing = False
|
||||
if project:
|
||||
proj_dir = os.path.join(os.path.abspath(settings.workspace_dir), project)
|
||||
if not os.path.isdir(proj_dir):
|
||||
project_missing, project = True, None
|
||||
tools.set_project(project)
|
||||
|
||||
history = db.list_messages(req.session_id)
|
||||
|
||||
# Premier message : titre la session avec un extrait.
|
||||
if not history:
|
||||
title = req.content.strip().split("\n")[0][:60] or "Nouvelle session"
|
||||
db.rename_session(req.session_id, title)
|
||||
|
||||
db.add_message(req.session_id, "user", req.content, None)
|
||||
|
||||
# Routage automatique : moteur code si la demande est une tâche de code —
|
||||
# ou la SUITE d'un travail de code (« ajoute un bouton », « continue »…),
|
||||
# que l'heuristique seule classerait à tort en discussion.
|
||||
prev_code = _prev_was_code(history)
|
||||
use_code = (
|
||||
cfg["tools"].get("code_task", True)
|
||||
and coder.available()
|
||||
and (
|
||||
msg_router.is_code_task(req.content)
|
||||
or (prev_code and msg_router.is_code_followup(req.content))
|
||||
)
|
||||
)
|
||||
|
||||
# Mémoire compressée : système + résumé des anciens tours + messages récents.
|
||||
convo = memory.build_convo(req.session_id, cfg["system_prompt"])
|
||||
|
||||
# Options runner partagées par TOUS les appels au modèle de chat (plan,
|
||||
# résumé, agent) : indispensables pour qu'Ollama garde le même runner.
|
||||
run_opts = agent_config.runner_options(cfg)
|
||||
keep = cfg.get("keep_alive", "30m")
|
||||
|
||||
async def event_stream():
|
||||
yield _sse("start", {"model": model, "engine": "code" if use_code else "agent"})
|
||||
|
||||
if project_missing:
|
||||
yield _sse("notice", {"message": (
|
||||
"Projet de la session introuvable sur le disque — retour au "
|
||||
"workspace."
|
||||
)})
|
||||
|
||||
# Préparation du contexte APRÈS le start SSE et en PARALLÈLE : rappel
|
||||
# RAG, plan et choix du modèle code partent ensemble au lieu de
|
||||
# s'enchaîner en bloquant le premier token.
|
||||
want_rag = cfg.get("rag_enabled", False)
|
||||
# Un plan « from scratch » n'a de sens que pour une NOUVELLE construction.
|
||||
# Pour une correction ou la suite d'un travail (l'appli existe déjà,
|
||||
# session code en cours, formulation de suivi), il est hors-sujet et
|
||||
# perturbe : on ne replanifie pas l'architecture à chaque message.
|
||||
is_modification = prev_code or bool(_workspace_listing())
|
||||
new_build = use_code and not is_modification
|
||||
want_plan = cfg.get("plan_mode", True) and (
|
||||
new_build or (not use_code and enhance.needs_plan(req.content))
|
||||
)
|
||||
if want_rag or want_plan or use_code:
|
||||
yield _sse("status", {"message": "Préparation du contexte…"})
|
||||
|
||||
from ..mcp_client import manager as mcp_manager
|
||||
|
||||
# La préparation peut déclencher un CHARGEMENT de modèle (plan, embed)
|
||||
# qui dure plusieurs minutes sur un gros modèle. Sans battement de
|
||||
# cœur, le silence SSE fait couper la connexion par le reverse proxy.
|
||||
prep = asyncio.ensure_future(asyncio.gather(
|
||||
rag.recall(req.session_id, req.content, embed_model=cfg.get("embed_model"))
|
||||
if want_rag else asyncio.sleep(0, result=[]),
|
||||
enhance.make_plan(model, req.content, code=use_code,
|
||||
options=run_opts, keep_alive=keep)
|
||||
if want_plan else asyncio.sleep(0, result=[]),
|
||||
coder.pick_code_model(model, cfg.get("code_model"))
|
||||
if use_code else asyncio.sleep(0, result=model),
|
||||
mcp_manager.tool_definitions(),
|
||||
))
|
||||
while True:
|
||||
try:
|
||||
memories, plan, code_model, mcp_tools = await asyncio.wait_for(
|
||||
asyncio.shield(prep), timeout=10.0
|
||||
)
|
||||
break
|
||||
except TimeoutError:
|
||||
yield _sse("ping", {"status": "waiting"})
|
||||
|
||||
# Pannes MCP éventuelles : notice non bloquante dans le fil.
|
||||
for mcp_notice in mcp_manager.notices():
|
||||
yield _sse("notice", {"message": mcp_notice})
|
||||
|
||||
# Placement du modèle : dire tout de suite si l'inférence est hors GPU,
|
||||
# au lieu de laisser l'utilisateur attribuer la lenteur à Loki.
|
||||
placement = await _placement_notice(model)
|
||||
if placement:
|
||||
yield _sse("notice", {"message": placement})
|
||||
|
||||
# ── Consignes additionnelles du tour ──────────────────────────────
|
||||
# Elles sont COLLECTÉES ici puis fusionnées dans l'UNIQUE message
|
||||
# système (voir _merge_system plus bas). Beaucoup de templates
|
||||
# (Gemma, Mistral…) lèvent « System message must be at the beginning »
|
||||
# dès qu'un second message system apparaît : les empiler faisait
|
||||
# échouer toute la requête en 400.
|
||||
extras: list[str] = []
|
||||
|
||||
# La demande touche-t-elle au code / aux fichiers ? Calculé AVANT les
|
||||
# injections : tout ce qui suit est conditionné à cette réponse.
|
||||
is_code_like = use_code or msg_router.is_code_task(req.content)
|
||||
needs_workspace = is_code_like or prev_code
|
||||
|
||||
if memories:
|
||||
extras.append(
|
||||
"Souvenirs pertinents d'anciennes sessions :\n"
|
||||
+ "\n---\n".join(memories)
|
||||
)
|
||||
|
||||
# État du workspace : indispensable pour travailler sur les fichiers,
|
||||
# mais inutile pour une simple discussion. On l'injectait à CHAQUE
|
||||
# message — un « bonjour » partait avec des milliers de jetons de
|
||||
# contexte, donc une longue phase de traitement avant le 1er jeton.
|
||||
listing = _workspace_listing() if needs_workspace else []
|
||||
if listing:
|
||||
recap, session_files = _session_code_context(history)
|
||||
parts = ["Fichiers du workspace : " + ", ".join(listing)]
|
||||
if session_files:
|
||||
parts.append(
|
||||
"Fichiers de la tâche en cours : " + ", ".join(session_files[:8])
|
||||
)
|
||||
extras.append("\n".join(parts))
|
||||
|
||||
# Session code restée en chemin agent : pousse le modèle à AGIR sur
|
||||
# les fichiers au lieu de décrire les changements — cause fréquente de
|
||||
# « l'agent s'arrête sans rien modifier » sur une reprise de code.
|
||||
if prev_code and not use_code:
|
||||
extras.append(
|
||||
"Cette session travaille sur du code existant du workspace. "
|
||||
"Pour toute demande de modification ou d'ajout : AGIS avec "
|
||||
"les outils — code_task pour un changement multi-fichiers, "
|
||||
"edit_file pour un changement ciblé, write_file pour un "
|
||||
"nouveau fichier. Lis le fichier concerné avant de le "
|
||||
"modifier, puis modifie-le RÉELLEMENT. Ne colle JAMAIS le "
|
||||
"code corrigé dans ta réponse sans l'avoir écrit dans le "
|
||||
"fichier."
|
||||
)
|
||||
|
||||
# Skill : méthode experte injectée pour ce tour (jamais persistée).
|
||||
if cfg.get("skills_enabled", True):
|
||||
skill = skills.pick_skill(req.content)
|
||||
if skill:
|
||||
extras.append("Méthode à suivre pour cette tâche :\n" + skill["body"])
|
||||
yield _sse("notice", {"message": f"📘 Méthode : {skill['title']}"})
|
||||
|
||||
# Ponytail : méthode « code minimal » injectée pour toute tâche de code
|
||||
# (les deux chemins). Contre la sur-ingénierie qui casse les rendus.
|
||||
if cfg.get("ponytail", True) and is_code_like:
|
||||
extras.append(skills.PONYTAIL_GUIDANCE)
|
||||
|
||||
# Appli web : contraintes dures (fichier autonome, zéro dépendance
|
||||
# externe, rendu réel) pour que ça marche vraiment hors-ligne.
|
||||
want_web = is_code_like and skills.is_web_task(req.content)
|
||||
if want_web:
|
||||
extras.append(skills.WEBAPP_GUIDANCE)
|
||||
|
||||
# Mémoire en notes : en mode « always », on injecte les notes liées à la
|
||||
# demande. Rien n'est deviné — ce sont des notes que l'agent a
|
||||
# lui-même écrites, et le bloc dit explicitement de les ignorer si
|
||||
# elles ne s'appliquent pas.
|
||||
memory_mode = cfg.get("memory_mode", "ondemand")
|
||||
# Mode off : les outils n'existent pas pour le modèle (principe AJEAN).
|
||||
# En mode Plan (lecture seule), la consultation reste permise mais pas
|
||||
# l'écriture d'une note.
|
||||
memory_tools = (
|
||||
[] if memory_mode == "off"
|
||||
else ["memory_search"] if req.mode == "plan"
|
||||
else ["memory_search", "memory_save"]
|
||||
)
|
||||
if memory_mode == "always":
|
||||
block = memory_notes.recall_block(req.content)
|
||||
if block:
|
||||
extras.append(block)
|
||||
|
||||
if plan:
|
||||
yield _sse("plan", {"steps": plan})
|
||||
|
||||
# Chemin « moteur code » : Aider gère la tâche de bout en bout.
|
||||
if use_code:
|
||||
instruction_plan = (
|
||||
"\n\nPlan à suivre :\n"
|
||||
+ "\n".join(f"{i+1}. {s}" for i, s in enumerate(plan))
|
||||
if plan else ""
|
||||
)
|
||||
# Reprise : Aider ne voit que le message courant — on lui donne le
|
||||
# récap de session et, à défaut de fichiers cités, ceux déjà
|
||||
# touchés (« corrige les bugs » => il ouvre le bon fichier).
|
||||
recap, session_files = _session_code_context(history)
|
||||
extra = instruction_plan
|
||||
if recap:
|
||||
extra = f"\n\n{recap}" + extra
|
||||
# Le moteur code (Aider) ne voit pas convo : on lui redonne Ponytail
|
||||
# (et les contraintes web) directement dans la consigne.
|
||||
if want_web:
|
||||
extra = "\n\n" + skills.WEBAPP_GUIDANCE + extra
|
||||
if cfg.get("ponytail", True):
|
||||
extra = "\n\n" + skills.PONYTAIL_GUIDANCE + extra
|
||||
code_files = _mentioned_files(req.content) or session_files[:8]
|
||||
async for chunk in _code_stream(
|
||||
req, code_model, extra=extra, plan=plan, files=code_files,
|
||||
):
|
||||
yield chunk
|
||||
asyncio.create_task(memory.maybe_summarize(
|
||||
req.session_id, model, options=run_opts, keep_alive=keep,
|
||||
))
|
||||
if cfg.get("rag_enabled", False):
|
||||
last = db.list_messages(req.session_id)
|
||||
answer = last[-1]["content"] if last else ""
|
||||
asyncio.create_task(rag.index_exchange(
|
||||
req.session_id, req.content, answer,
|
||||
embed_model=cfg.get("embed_model"),
|
||||
))
|
||||
return
|
||||
|
||||
if plan:
|
||||
extras.append(
|
||||
"Plan à suivre pour cette demande, étape par étape :\n"
|
||||
+ "\n".join(f"{i+1}. {s}" for i, s in enumerate(plan))
|
||||
+ "\n\nTraite les étapes DANS L'ORDRE. Dès qu'une étape est "
|
||||
"réellement accomplie, écris sur une ligne seule "
|
||||
"« ✅ Étape N terminée » (N = son numéro) avant de passer à "
|
||||
"la suivante. N'annonce jamais une étape terminée à l'avance."
|
||||
)
|
||||
|
||||
# Fusion : un SEUL message système, en tête. Indispensable pour les
|
||||
# templates qui refusent tout system ailleurs qu'en première position.
|
||||
_merge_system(convo, extras)
|
||||
|
||||
final_content = ""
|
||||
tools_meta: list[dict] = []
|
||||
stats_meta: dict | None = None
|
||||
thinking_meta = ""
|
||||
error_message = ""
|
||||
|
||||
queue: asyncio.Queue[dict | None] = asyncio.Queue()
|
||||
|
||||
async def produce_events() -> None:
|
||||
try:
|
||||
async for event in run_agent(
|
||||
model,
|
||||
convo,
|
||||
options=agent_config.ollama_options(cfg),
|
||||
enabled_tools=agent_config.enabled_tool_names(cfg) + memory_tools,
|
||||
confirm_shell=cfg.get("confirm_shell", True),
|
||||
think=cfg.get("think", True),
|
||||
keep_alive=cfg.get("keep_alive", "30m"),
|
||||
mcp_tools=mcp_tools,
|
||||
plan=plan,
|
||||
):
|
||||
await queue.put(event)
|
||||
except Exception as exc:
|
||||
logger.exception("Échec inattendu du flux de chat")
|
||||
await queue.put(
|
||||
{
|
||||
"type": "error",
|
||||
"message": f"Erreur interne du chat : {exc}",
|
||||
}
|
||||
)
|
||||
finally:
|
||||
await queue.put(None)
|
||||
|
||||
producer = asyncio.create_task(produce_events())
|
||||
started = time.monotonic()
|
||||
got_event = False
|
||||
try:
|
||||
while True:
|
||||
try:
|
||||
ev = await asyncio.wait_for(queue.get(), timeout=15.0)
|
||||
except TimeoutError:
|
||||
# Empêche OpenResty/Nginx/Cloudflare de fermer le SSE pendant
|
||||
# le chargement parfois long d'un modèle Ollama.
|
||||
yield _sse("ping", {"status": "waiting"})
|
||||
# …et DIT ce qui se passe : sans ça, l'interface restait sur
|
||||
# « Connexion à Ollama… » pendant tout le chargement d'un
|
||||
# gros modèle, sans moyen de distinguer lenteur et blocage.
|
||||
if not got_event:
|
||||
yield _sse("status", {
|
||||
"message": _loading_message(
|
||||
model, time.monotonic() - started
|
||||
)
|
||||
})
|
||||
continue
|
||||
|
||||
got_event = True
|
||||
if ev is None:
|
||||
break
|
||||
|
||||
etype = ev.pop("type")
|
||||
if etype in (
|
||||
"token",
|
||||
"thinking",
|
||||
"status",
|
||||
"notice",
|
||||
"tool_call",
|
||||
"tool_result",
|
||||
"tool_confirm",
|
||||
"plan_step",
|
||||
):
|
||||
yield _sse(etype, ev)
|
||||
elif etype == "error":
|
||||
error_message = ev.get("message", "Erreur Ollama inconnue")
|
||||
yield _sse("error", {"message": error_message})
|
||||
elif etype == "final":
|
||||
final_content = ev["content"]
|
||||
tools_meta = ev["tools"]
|
||||
stats_meta = ev.get("stats")
|
||||
thinking_meta = ev.get("thinking", "")
|
||||
finally:
|
||||
if not producer.done():
|
||||
producer.cancel()
|
||||
with suppress(asyncio.CancelledError):
|
||||
await producer
|
||||
|
||||
# Auto-critique : relecture éclair puis révision (option « Qualité + »).
|
||||
if (
|
||||
cfg.get("self_review", False)
|
||||
and final_content
|
||||
and not error_message
|
||||
and not tools_meta
|
||||
):
|
||||
yield _sse("status", {"message": "Relecture de la réponse…"})
|
||||
revised = await enhance.self_review(
|
||||
model, req.content, final_content,
|
||||
options=run_opts, keep_alive=keep,
|
||||
)
|
||||
if revised:
|
||||
final_content = revised
|
||||
yield _sse("revision", {"content": revised})
|
||||
yield _sse("notice", {"message": "Réponse révisée après auto-critique ✓"})
|
||||
|
||||
if final_content or tools_meta:
|
||||
meta: dict = {}
|
||||
if tools_meta:
|
||||
meta["tools"] = tools_meta
|
||||
if stats_meta:
|
||||
meta["stats"] = stats_meta
|
||||
if thinking_meta:
|
||||
meta["thinking"] = thinking_meta
|
||||
if plan:
|
||||
meta["plan"] = plan
|
||||
db.add_message(
|
||||
req.session_id,
|
||||
"assistant",
|
||||
final_content,
|
||||
model,
|
||||
meta=meta or None,
|
||||
)
|
||||
yield _sse(
|
||||
"done",
|
||||
{
|
||||
"content": final_content,
|
||||
"tools": tools_meta,
|
||||
"stats": stats_meta,
|
||||
"error": error_message or None,
|
||||
},
|
||||
)
|
||||
# Tâches d'arrière-plan : compression de l'historique + mémoire RAG.
|
||||
asyncio.create_task(memory.maybe_summarize(
|
||||
req.session_id, model, options=run_opts, keep_alive=keep,
|
||||
))
|
||||
if cfg.get("rag_enabled", False) and final_content:
|
||||
asyncio.create_task(rag.index_exchange(
|
||||
req.session_id, req.content, final_content,
|
||||
embed_model=cfg.get("embed_model"),
|
||||
))
|
||||
|
||||
return StreamingResponse(
|
||||
event_stream(),
|
||||
media_type="text/event-stream",
|
||||
headers={
|
||||
"Cache-Control": "no-cache, no-transform",
|
||||
"X-Accel-Buffering": "no",
|
||||
},
|
||||
)
|
||||
@@ -1,86 +0,0 @@
|
||||
"""Routes de lecture et mise à jour de la configuration de l'agent."""
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import agent_config
|
||||
|
||||
router = APIRouter(prefix="/api/config", tags=["config"])
|
||||
|
||||
|
||||
class ConfigPatch(BaseModel):
|
||||
"""Champs modifiables. DOIT couvrir agent_config.PROFILE_FIELDS.
|
||||
|
||||
Pydantic ignore silencieusement un champ non déclaré : un oubli ici fait
|
||||
« sauter » le réglage sans la moindre erreur (c'était le cas de
|
||||
keep_alive, skills_enabled et ponytail).
|
||||
"""
|
||||
|
||||
system_prompt: str | None = None
|
||||
temperature: float | None = None
|
||||
top_p: float | None = None
|
||||
top_k: int | None = None
|
||||
max_tokens: int | None = None
|
||||
num_ctx: int | None = None
|
||||
num_gpu: int | None = None
|
||||
num_batch: int | None = None
|
||||
tools: dict[str, bool] | None = None
|
||||
confirm_shell: bool | None = None
|
||||
think: bool | None = None
|
||||
code_model: str | None = None
|
||||
plan_mode: bool | None = None
|
||||
self_review: bool | None = None
|
||||
rag_enabled: bool | None = None
|
||||
embed_model: str | None = None
|
||||
memory_mode: str | None = None
|
||||
skills_enabled: bool | None = None
|
||||
ponytail: bool | None = None
|
||||
keep_alive: str | None = None
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def get_config(model: str | None = None) -> dict:
|
||||
return {
|
||||
"config": agent_config.get_config(model),
|
||||
"available_tools": agent_config.AVAILABLE_TOOLS,
|
||||
}
|
||||
|
||||
|
||||
@router.put("")
|
||||
async def put_config(patch: ConfigPatch, model: str | None = None) -> dict:
|
||||
cfg = agent_config.save_config(patch.model_dump(exclude_none=True), model)
|
||||
return {"config": cfg}
|
||||
|
||||
|
||||
# ── Presets : jeux de réglages nommés, commutables (idée reprise d'AJEAN) ──
|
||||
class PresetBody(BaseModel):
|
||||
name: str
|
||||
|
||||
|
||||
@router.get("/presets")
|
||||
async def get_presets() -> dict:
|
||||
return {"presets": agent_config.list_presets()}
|
||||
|
||||
|
||||
@router.post("/presets")
|
||||
async def post_preset(body: PresetBody, model: str | None = None) -> dict:
|
||||
"""Enregistre la configuration courante sous un nom."""
|
||||
try:
|
||||
return {"presets": agent_config.save_preset(body.name, model)}
|
||||
except ValueError as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
|
||||
|
||||
@router.post("/presets/apply")
|
||||
async def apply_preset(body: PresetBody, model: str | None = None) -> dict:
|
||||
"""Applique un preset à la configuration courante."""
|
||||
cfg = agent_config.apply_preset(body.name, model)
|
||||
if cfg is None:
|
||||
raise HTTPException(404, "preset introuvable")
|
||||
return {"config": cfg}
|
||||
|
||||
|
||||
@router.delete("/presets")
|
||||
async def remove_preset(name: str) -> dict:
|
||||
return {"presets": agent_config.delete_preset(name)}
|
||||
@@ -1,105 +0,0 @@
|
||||
"""Routes du workspace : arborescence, contenu, téléchargement, suppression.
|
||||
|
||||
Toutes les routes acceptent un paramètre optionnel ``project`` : la requête
|
||||
est alors re-racinée sur ``workspace/<projet>`` (même confinement).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from fastapi.responses import FileResponse
|
||||
|
||||
from .. import tools as tools_mod
|
||||
from ..tools import ToolError, _safe_path
|
||||
|
||||
router = APIRouter(prefix="/api/files", tags=["files"])
|
||||
|
||||
|
||||
def _activate(project: str | None) -> None:
|
||||
"""Pose le projet actif pour la requête (400 si nom invalide)."""
|
||||
try:
|
||||
tools_mod.set_project(project or None)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
|
||||
|
||||
def _tree(path: str, root: str) -> list[dict]:
|
||||
"""Arborescence triée (dossiers d'abord) de la racine active."""
|
||||
items = []
|
||||
for name in sorted(os.listdir(path)):
|
||||
if name.startswith("."):
|
||||
continue
|
||||
full = os.path.join(path, name)
|
||||
rel = os.path.relpath(full, root)
|
||||
if os.path.isdir(full):
|
||||
items.append({"name": name, "path": rel, "type": "dir",
|
||||
"children": _tree(full, root)})
|
||||
else:
|
||||
items.append({"name": name, "path": rel, "type": "file",
|
||||
"size": os.path.getsize(full)})
|
||||
# Dossiers avant fichiers
|
||||
items.sort(key=lambda x: (x["type"] != "dir", x["name"]))
|
||||
return items
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_files(project: str | None = None) -> dict:
|
||||
_activate(project)
|
||||
root = tools_mod.active_root()
|
||||
return {"tree": _tree(root, root)}
|
||||
|
||||
|
||||
@router.get("/content")
|
||||
async def file_content(path: str, project: str | None = None) -> dict:
|
||||
_activate(project)
|
||||
try:
|
||||
target = _safe_path(path)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
if not os.path.isfile(target):
|
||||
raise HTTPException(404, "fichier introuvable")
|
||||
with open(target, "r", encoding="utf-8", errors="replace") as f:
|
||||
return {"path": path, "content": f.read()}
|
||||
|
||||
|
||||
@router.get("/download")
|
||||
async def download_file(path: str, project: str | None = None) -> FileResponse:
|
||||
"""Télécharge un fichier en conservant le confinement au workspace."""
|
||||
_activate(project)
|
||||
try:
|
||||
target = _safe_path(path)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
if not os.path.isfile(target):
|
||||
raise HTTPException(404, "fichier introuvable")
|
||||
return FileResponse(
|
||||
target,
|
||||
filename=os.path.basename(target),
|
||||
media_type="application/octet-stream",
|
||||
)
|
||||
|
||||
|
||||
@router.delete("")
|
||||
async def delete_file(path: str, project: str | None = None) -> dict:
|
||||
"""Supprime un fichier ou un dossier (récursif) de la racine active.
|
||||
|
||||
Même confinement que le téléchargement (`_safe_path`) ; la racine active
|
||||
est refusée. Les dotfiles (.git…) ne sont jamais listés par
|
||||
l'arborescence, donc inaccessibles depuis l'UI.
|
||||
"""
|
||||
_activate(project)
|
||||
try:
|
||||
target = _safe_path(path)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
if os.path.abspath(target) == tools_mod.active_root():
|
||||
raise HTTPException(400, "suppression de la racine refusée")
|
||||
if os.path.isdir(target):
|
||||
shutil.rmtree(target)
|
||||
elif os.path.isfile(target):
|
||||
os.remove(target)
|
||||
else:
|
||||
raise HTTPException(404, "fichier introuvable")
|
||||
return {"deleted": path}
|
||||
@@ -1,106 +0,0 @@
|
||||
"""Routes Git du workspace : historique, diff, retour arrière.
|
||||
|
||||
Le workspace est un dépôt git (Aider commite chaque modification). Ces routes
|
||||
donnent à l'UI un panneau Git : voir les commits, leur diff, et annuler une
|
||||
modification. Toutes les commandes sont exécutées DANS le workspace.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import tools
|
||||
from ..coder import ensure_git
|
||||
from ..config import settings
|
||||
|
||||
router = APIRouter(prefix="/api/git", tags=["git"])
|
||||
|
||||
|
||||
def _git(
|
||||
*args: str, timeout: int = 15, project: str | None = None
|
||||
) -> subprocess.CompletedProcess:
|
||||
"""Exécute une commande git dans le workspace ou le projet demandé."""
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
if project:
|
||||
if not tools.PROJECT_NAME.match(project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
root = os.path.join(root, project)
|
||||
ensure_git(root)
|
||||
return subprocess.run(
|
||||
["git", *args],
|
||||
cwd=root,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=timeout,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/log")
|
||||
async def git_log(limit: int = 40, project: str | None = None) -> dict:
|
||||
"""Historique des commits (hash court, message, auteur, date, nb fichiers)."""
|
||||
fmt = "%h%x1f%s%x1f%an%x1f%ar%x1f%H"
|
||||
proc = _git(
|
||||
"log", f"-{max(1, min(limit, 200))}", f"--pretty=format:{fmt}",
|
||||
project=project,
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
# Dépôt sans commit encore.
|
||||
return {"commits": []}
|
||||
|
||||
commits = []
|
||||
for line in proc.stdout.splitlines():
|
||||
parts = line.split("\x1f")
|
||||
if len(parts) != 5:
|
||||
continue
|
||||
short, subject, author, when, full = parts
|
||||
# Nombre de fichiers touchés par ce commit.
|
||||
stat = _git("show", "--stat", "--oneline", "--format=", full,
|
||||
project=project)
|
||||
files = [l for l in stat.stdout.splitlines() if "|" in l]
|
||||
commits.append({
|
||||
"hash": short,
|
||||
"full_hash": full,
|
||||
"subject": subject,
|
||||
"author": author,
|
||||
"when": when,
|
||||
"files_changed": len(files),
|
||||
})
|
||||
return {"commits": commits}
|
||||
|
||||
|
||||
@router.get("/diff")
|
||||
async def git_diff(hash: str | None = None, project: str | None = None) -> dict:
|
||||
"""Diff d'un commit (hash) ou des modifications non commitées si absent."""
|
||||
if hash:
|
||||
if not hash.replace("-", "").isalnum():
|
||||
raise HTTPException(400, "hash invalide")
|
||||
proc = _git("show", "--no-color", hash, project=project)
|
||||
else:
|
||||
proc = _git("diff", "--no-color", "HEAD", project=project)
|
||||
if proc.returncode != 0:
|
||||
raise HTTPException(404, "diff introuvable")
|
||||
# Borne la taille pour ne pas noyer l'UI.
|
||||
return {"diff": proc.stdout[:200_000]}
|
||||
|
||||
|
||||
class RevertRequest(BaseModel):
|
||||
hash: str
|
||||
project: str | None = None
|
||||
|
||||
|
||||
@router.post("/revert")
|
||||
async def git_revert(req: RevertRequest) -> dict:
|
||||
"""Annule un commit en créant un commit inverse (git revert)."""
|
||||
h = req.hash.strip()
|
||||
if not h.replace("-", "").isalnum():
|
||||
raise HTTPException(400, "hash invalide")
|
||||
proc = _git("revert", "--no-edit", h, timeout=30, project=req.project)
|
||||
if proc.returncode != 0:
|
||||
# Conflit ou commit introuvable : on nettoie un éventuel revert en cours.
|
||||
_git("revert", "--abort", project=req.project)
|
||||
detail = (proc.stderr or proc.stdout).strip()[:300]
|
||||
raise HTTPException(409, f"revert impossible : {detail}")
|
||||
return {"ok": True, "message": f"commit {h[:7]} annulé"}
|
||||
@@ -1,54 +0,0 @@
|
||||
"""Routes de configuration des serveurs MCP (catalogue + toggles)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from ..mcp_client import CATALOG, get_mcp_state, manager, set_mcp_state
|
||||
|
||||
router = APIRouter(prefix="/api/mcp", tags=["mcp"])
|
||||
|
||||
|
||||
def _payload() -> dict:
|
||||
state = get_mcp_state()
|
||||
statuses = manager.statuses()
|
||||
return {
|
||||
"servers": [
|
||||
{
|
||||
"id": sid,
|
||||
"label": entry["label"],
|
||||
"description": entry["description"],
|
||||
"url_param": entry["url_param"],
|
||||
"env_params": entry["env_params"],
|
||||
"enabled": state[sid]["enabled"],
|
||||
"params": state[sid]["params"],
|
||||
**statuses[sid],
|
||||
}
|
||||
for sid, entry in CATALOG.items()
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_servers() -> dict:
|
||||
return _payload()
|
||||
|
||||
|
||||
class McpUpdate(BaseModel):
|
||||
enabled: bool
|
||||
params: dict = {}
|
||||
|
||||
|
||||
@router.put("/{sid}")
|
||||
async def update_server(sid: str, req: McpUpdate) -> dict:
|
||||
if sid not in CATALOG:
|
||||
raise HTTPException(404, "serveur inconnu")
|
||||
set_mcp_state(sid, enabled=req.enabled, params=req.params)
|
||||
return _payload()
|
||||
|
||||
|
||||
@router.post("/{sid}/test")
|
||||
async def test_server(sid: str) -> dict:
|
||||
if sid not in CATALOG:
|
||||
raise HTTPException(404, "serveur inconnu")
|
||||
return await manager.test_server(sid)
|
||||
@@ -1,197 +0,0 @@
|
||||
"""Routes liées à Ollama : statut de connexion, modèles, téléchargement."""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
import httpx
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
from fastapi.responses import StreamingResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import agent_config
|
||||
from ..config import settings
|
||||
from ..ollama_client import ollama
|
||||
|
||||
router = APIRouter(prefix="/api", tags=["ollama"])
|
||||
|
||||
# État en mémoire des préchargements. Loki est lancé avec un worker unique ; ce
|
||||
# suivi permet au navigateur d'interroger une route courte pendant que le long
|
||||
# chargement Ollama continue sans maintenir la requête HTTP initiale ouverte.
|
||||
_warm_states: dict[str, dict[str, str]] = {}
|
||||
_warm_tasks: dict[str, asyncio.Task[None]] = {}
|
||||
|
||||
|
||||
@router.get("/status")
|
||||
async def status() -> dict:
|
||||
"""État de la connexion Ollama (point vert/rouge de la barre supérieure)."""
|
||||
try:
|
||||
version = await ollama.ping()
|
||||
return {
|
||||
"connected": True,
|
||||
"host": ollama.host,
|
||||
"version": version.get("version"),
|
||||
"default_model": settings.default_model,
|
||||
}
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
return {
|
||||
"connected": False,
|
||||
"host": ollama.host,
|
||||
"error": str(exc),
|
||||
"default_model": settings.default_model,
|
||||
}
|
||||
|
||||
|
||||
@router.get("/models")
|
||||
async def list_models() -> dict:
|
||||
"""Modèles installés localement, formatés pour le sélecteur de l'UI."""
|
||||
try:
|
||||
raw = await ollama.list_models_cached()
|
||||
except (httpx.HTTPError, OSError):
|
||||
return {"models": []}
|
||||
|
||||
models = []
|
||||
for m in raw:
|
||||
details = m.get("details", {})
|
||||
size_go = round(m.get("size", 0) / 1_000_000_000, 1)
|
||||
models.append(
|
||||
{
|
||||
"name": m.get("name"),
|
||||
"size_go": size_go,
|
||||
"parameter_size": details.get("parameter_size"),
|
||||
"quantization": details.get("quantization_level"),
|
||||
"family": details.get("family"),
|
||||
}
|
||||
)
|
||||
return {"models": models, "default": settings.default_model}
|
||||
|
||||
|
||||
class WarmRequest(BaseModel):
|
||||
name: str
|
||||
keep_alive: str = "30m"
|
||||
|
||||
|
||||
async def _placement_of(name: str) -> dict:
|
||||
"""Où le modèle vient-il d'être chargé ? (GPU / CPU / mixte), via /api/ps."""
|
||||
try:
|
||||
for m in await ollama.ps():
|
||||
if m.get("name") == name or m.get("model") == name:
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
if size <= 0:
|
||||
return {}
|
||||
pct = int(vram / size * 100)
|
||||
where = "gpu" if pct >= 99 else "cpu" if pct <= 1 else "mixte"
|
||||
return {"processor": where, "gpu_percent": str(pct)}
|
||||
except (httpx.HTTPError, OSError):
|
||||
pass
|
||||
return {}
|
||||
|
||||
|
||||
async def _warm_in_background(name: str, keep_alive: str, options: dict | None) -> None:
|
||||
try:
|
||||
await ollama.warm(name, keep_alive, options)
|
||||
state = {"state": "loaded"}
|
||||
state.update(await _placement_of(name))
|
||||
_warm_states[name] = state
|
||||
except Exception as exc:
|
||||
_warm_states[name] = {
|
||||
"state": "error",
|
||||
"error": f"préchargement impossible : {str(exc)[:500]}",
|
||||
}
|
||||
|
||||
|
||||
def start_model_warm(name: str, keep_alive: str) -> None:
|
||||
"""Démarre au plus une tâche de préchargement par modèle.
|
||||
|
||||
Les options de génération du modèle (num_ctx, num_batch, num_gpu…) sont
|
||||
envoyées au préchargement pour qu'Ollama charge exactement le runner que le
|
||||
chat utilisera — évite un rechargement complet au premier message.
|
||||
"""
|
||||
current = _warm_tasks.get(name)
|
||||
if current and not current.done():
|
||||
return
|
||||
cfg = agent_config.get_config(name)
|
||||
options = agent_config.ollama_options(cfg)
|
||||
_warm_states[name] = {"state": "loading"}
|
||||
task = asyncio.create_task(_warm_in_background(name, keep_alive, options))
|
||||
_warm_tasks[name] = task
|
||||
|
||||
def forget(done: asyncio.Task[None]) -> None:
|
||||
if _warm_tasks.get(name) is done:
|
||||
_warm_tasks.pop(name, None)
|
||||
|
||||
task.add_done_callback(forget)
|
||||
|
||||
|
||||
@router.post("/models/warm", status_code=202)
|
||||
async def warm_model(req: WarmRequest) -> dict:
|
||||
"""Démarre le préchargement sans exposer sa durée au reverse proxy."""
|
||||
name = req.name.strip()
|
||||
if not name:
|
||||
raise HTTPException(400, "nom de modèle vide")
|
||||
start_model_warm(name, req.keep_alive)
|
||||
return {"warming": name, "state": _warm_states[name]["state"]}
|
||||
|
||||
|
||||
@router.get("/models/warm/status")
|
||||
async def warm_status(name: str = Query(min_length=1)) -> dict:
|
||||
"""État court du préchargement : idle, loading, loaded ou error."""
|
||||
return _warm_states.get(name.strip(), {"state": "idle"})
|
||||
|
||||
|
||||
@router.get("/models/loaded")
|
||||
async def loaded_models() -> dict:
|
||||
"""Modèles actuellement chargés en mémoire + placement GPU/CPU (/api/ps)."""
|
||||
try:
|
||||
loaded = await ollama.ps()
|
||||
except (httpx.HTTPError, OSError):
|
||||
return {"loaded": []}
|
||||
result = []
|
||||
for m in loaded:
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
result.append({
|
||||
"name": m.get("name") or m.get("model"),
|
||||
"on_gpu": bool(size and vram >= size * 0.99),
|
||||
"gpu_percent": int(vram / size * 100) if size else 0,
|
||||
})
|
||||
return {"loaded": result}
|
||||
|
||||
|
||||
class PullRequest(BaseModel):
|
||||
name: str
|
||||
|
||||
|
||||
class DeleteModelRequest(BaseModel):
|
||||
name: str
|
||||
|
||||
|
||||
@router.delete("/models")
|
||||
async def delete_model(req: DeleteModelRequest) -> dict:
|
||||
"""Supprime explicitement un modèle de l'instance Ollama."""
|
||||
if not req.name.strip():
|
||||
raise HTTPException(400, "nom de modèle vide")
|
||||
try:
|
||||
await ollama.delete_model(req.name.strip())
|
||||
except httpx.HTTPStatusError as exc:
|
||||
detail = exc.response.text[:500] or str(exc)
|
||||
raise HTTPException(502, f"Ollama : {detail}") from exc
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
raise HTTPException(502, f"Ollama injoignable : {exc}") from exc
|
||||
return {"deleted": req.name.strip()}
|
||||
|
||||
|
||||
@router.post("/models/pull")
|
||||
async def pull_model(req: PullRequest) -> StreamingResponse:
|
||||
"""Télécharge un modèle en streamant la progression (SSE)."""
|
||||
|
||||
async def event_stream():
|
||||
try:
|
||||
async for chunk in ollama.pull_model(req.name):
|
||||
yield f"data: {json.dumps(chunk)}\n\n"
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
yield f"data: {json.dumps({'error': str(exc)})}\n\n"
|
||||
yield "data: [DONE]\n\n"
|
||||
|
||||
return StreamingResponse(event_stream(), media_type="text/event-stream")
|
||||
@@ -1,60 +0,0 @@
|
||||
"""Routes des projets : sous-dossiers de premier niveau du workspace."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import coder, tools
|
||||
from ..config import settings
|
||||
|
||||
router = APIRouter(prefix="/api/projects", tags=["projects"])
|
||||
|
||||
|
||||
def _root() -> str:
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
return root
|
||||
|
||||
|
||||
def _count_files(path: str) -> int:
|
||||
total = 0
|
||||
for dirpath, dirnames, filenames in os.walk(path):
|
||||
dirnames[:] = [d for d in dirnames if not d.startswith(".")]
|
||||
total += sum(1 for f in filenames if not f.startswith("."))
|
||||
return total
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_projects() -> dict:
|
||||
root = _root()
|
||||
projects = []
|
||||
root_files = 0
|
||||
for name in sorted(os.listdir(root)):
|
||||
full = os.path.join(root, name)
|
||||
if name.startswith("."):
|
||||
continue
|
||||
if os.path.isdir(full):
|
||||
projects.append({"name": name, "files": _count_files(full)})
|
||||
else:
|
||||
root_files += 1
|
||||
return {"projects": projects, "root_files": root_files}
|
||||
|
||||
|
||||
class CreateProject(BaseModel):
|
||||
name: str
|
||||
|
||||
|
||||
@router.post("", status_code=201)
|
||||
async def create_project(req: CreateProject) -> dict:
|
||||
name = req.name.strip()
|
||||
if not tools.PROJECT_NAME.match(name):
|
||||
raise HTTPException(400, "nom de projet invalide (a-z, 0-9, - et _)")
|
||||
target = os.path.join(_root(), name)
|
||||
if os.path.exists(target):
|
||||
raise HTTPException(400, "ce projet existe déjà")
|
||||
os.makedirs(target)
|
||||
# Dépôt git par projet : commits Aider + onglet Git propres au projet.
|
||||
coder.ensure_git(target)
|
||||
return {"name": name}
|
||||
@@ -1,60 +0,0 @@
|
||||
"""Routes de gestion des sessions de conversation."""
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import db, tools
|
||||
|
||||
router = APIRouter(prefix="/api/sessions", tags=["sessions"])
|
||||
|
||||
|
||||
class CreateSession(BaseModel):
|
||||
title: str = "Nouvelle session"
|
||||
model: str | None = None
|
||||
project: str | None = None
|
||||
|
||||
|
||||
class UpdateSession(BaseModel):
|
||||
title: str | None = None
|
||||
project: str | None = None # "" = retour à la racine du workspace
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def get_sessions() -> dict:
|
||||
return {"sessions": db.list_sessions()}
|
||||
|
||||
|
||||
@router.post("")
|
||||
async def post_session(req: CreateSession) -> dict:
|
||||
if req.project and not tools.PROJECT_NAME.match(req.project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
return db.create_session(req.title, req.model, req.project or None)
|
||||
|
||||
|
||||
@router.get("/{sid}")
|
||||
async def get_one(sid: str) -> dict:
|
||||
session = db.get_session(sid)
|
||||
if not session:
|
||||
raise HTTPException(404, "session introuvable")
|
||||
return {"session": session, "messages": db.list_messages(sid)}
|
||||
|
||||
|
||||
@router.patch("/{sid}")
|
||||
async def patch_session(sid: str, req: UpdateSession) -> dict:
|
||||
if not db.get_session(sid):
|
||||
raise HTTPException(404, "session introuvable")
|
||||
if req.title is not None:
|
||||
db.rename_session(sid, req.title)
|
||||
if req.project is not None:
|
||||
project = req.project or None
|
||||
if project and not tools.PROJECT_NAME.match(project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
db.set_session_project(sid, project)
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@router.delete("/{sid}")
|
||||
async def remove_session(sid: str) -> dict:
|
||||
db.delete_session(sid)
|
||||
return {"ok": True}
|
||||
@@ -1,33 +0,0 @@
|
||||
"""Route d'exécution d'une commande shell validée par l'utilisateur.
|
||||
|
||||
La boucle agentique n'exécute jamais run_shell elle-même quand la confirmation
|
||||
est active : elle émet un événement `tool_confirm`. Le client affiche la
|
||||
commande, et c'est seulement après clic explicite de l'utilisateur que cette
|
||||
route exécute la commande (confinée au workspace).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from ..tools import ToolError, run_shell
|
||||
|
||||
router = APIRouter(prefix="/api/shell", tags=["shell"])
|
||||
|
||||
|
||||
class ShellRequest(BaseModel):
|
||||
command: str
|
||||
|
||||
|
||||
@router.post("/run")
|
||||
async def run(req: ShellRequest) -> dict:
|
||||
"""Exécute la commande validée et renvoie sa sortie."""
|
||||
try:
|
||||
result = run_shell(req.command)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
return {
|
||||
"command": req.command,
|
||||
"exit_code": result["exit_code"],
|
||||
"output": result["output"],
|
||||
}
|
||||
@@ -1,232 +0,0 @@
|
||||
"""Statistiques système temps réel : CPU, RAM, GPU/VRAM (barre supérieure)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import shutil
|
||||
import time
|
||||
|
||||
import httpx
|
||||
import psutil
|
||||
from fastapi import APIRouter
|
||||
|
||||
from ..config import settings
|
||||
from ..ollama_client import ollama
|
||||
|
||||
router = APIRouter(prefix="/api/system", tags=["system"])
|
||||
|
||||
_NVIDIA_SMI = shutil.which("nvidia-smi")
|
||||
|
||||
# Cache court : le front interroge /stats en continu ; relancer un sous-processus
|
||||
# nvidia-smi à chaque tick charge la machine qui héberge aussi Ollama.
|
||||
# TTL volontairement > à la cadence de sondage, sinon le cache n'absorbe rien.
|
||||
_GPU_CACHE_TTL = 12.0
|
||||
_gpu_cache: dict = {"at": 0.0, "value": None}
|
||||
# Même logique pour la vue matériel (sous-processus nvidia-smi plus lourd).
|
||||
_HW_CACHE_TTL = 12.0
|
||||
_hw_cache: dict = {"at": 0.0, "value": None}
|
||||
|
||||
|
||||
async def _gpu_stats() -> dict | None:
|
||||
"""Utilisation GPU/VRAM via nvidia-smi ; None si absent (pas de GPU NVIDIA).
|
||||
|
||||
Résultat mis en cache ~5 s pour limiter les sous-processus.
|
||||
"""
|
||||
if not _NVIDIA_SMI:
|
||||
return None
|
||||
if time.monotonic() - _gpu_cache["at"] < _GPU_CACHE_TTL:
|
||||
return _gpu_cache["value"]
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
_NVIDIA_SMI,
|
||||
"--query-gpu=utilization.gpu,memory.used,memory.total,name",
|
||||
"--format=csv,noheader,nounits",
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
out, _ = await asyncio.wait_for(proc.communicate(), timeout=3)
|
||||
line = out.decode().strip().splitlines()[0]
|
||||
util, used, total, name = (p.strip() for p in line.split(","))
|
||||
value = {
|
||||
"name": name,
|
||||
"util_pct": float(util),
|
||||
"vram_used_mb": float(used),
|
||||
"vram_total_mb": float(total),
|
||||
}
|
||||
except Exception:
|
||||
value = None
|
||||
_gpu_cache.update(at=time.monotonic(), value=value)
|
||||
return value
|
||||
|
||||
|
||||
@router.get("/stats")
|
||||
async def stats() -> dict:
|
||||
"""CPU %, RAM et GPU/VRAM courants."""
|
||||
cpu_pct = psutil.cpu_percent(interval=None)
|
||||
mem = psutil.virtual_memory()
|
||||
gpu = await _gpu_stats()
|
||||
return {
|
||||
"cpu_pct": cpu_pct,
|
||||
"ram_used_go": round(mem.used / 1_000_000_000, 1),
|
||||
"ram_total_go": round(mem.total / 1_000_000_000, 1),
|
||||
"ram_pct": mem.percent,
|
||||
"gpu": gpu,
|
||||
}
|
||||
|
||||
|
||||
# Battement groupé : absorbe les rafales (plusieurs onglets, reconnexions)
|
||||
# sans multiplier les sous-processus ni les connexions vers Ollama.
|
||||
_PULSE_TTL = 2.5
|
||||
_pulse_cache: dict = {"at": 0.0, "value": None}
|
||||
|
||||
|
||||
async def _loaded_models() -> list[dict]:
|
||||
"""Modèles chargés en mémoire + placement GPU (vide si Ollama muet)."""
|
||||
try:
|
||||
loaded = await ollama.ps()
|
||||
except (httpx.HTTPError, OSError):
|
||||
return []
|
||||
out = []
|
||||
for m in loaded:
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
out.append({
|
||||
"name": m.get("name") or m.get("model"),
|
||||
"on_gpu": bool(size and vram >= size * 0.99),
|
||||
"gpu_percent": int(vram / size * 100) if size else 0,
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
@router.get("/pulse")
|
||||
async def pulse() -> dict:
|
||||
"""Battement unique : statut Ollama + ressources + modèles chargés.
|
||||
|
||||
Remplace trois sondages séparés (statut, stats, modèles chargés) par UNE
|
||||
requête : moins de réveils, moins de sous-processus et moins de connexions
|
||||
vers Ollama quand l'application est simplement ouverte sans être utilisée.
|
||||
"""
|
||||
if time.monotonic() - _pulse_cache["at"] < _PULSE_TTL and _pulse_cache["value"]:
|
||||
return _pulse_cache["value"]
|
||||
|
||||
# Les deux appels Ollama partent ensemble : latence = le plus lent, pas la
|
||||
# somme (important quand Ollama est sur une autre machine).
|
||||
version, loaded = await asyncio.gather(
|
||||
ollama.ping(), _loaded_models(), return_exceptions=True
|
||||
)
|
||||
connected = not isinstance(version, BaseException)
|
||||
|
||||
mem = psutil.virtual_memory()
|
||||
value = {
|
||||
"status": {
|
||||
"connected": connected,
|
||||
"host": ollama.host,
|
||||
"version": version.get("version") if connected else None,
|
||||
"default_model": settings.default_model,
|
||||
**({} if connected else {"error": str(version)[:200]}),
|
||||
},
|
||||
"stats": {
|
||||
"cpu_pct": psutil.cpu_percent(interval=None),
|
||||
"ram_used_go": round(mem.used / 1_000_000_000, 1),
|
||||
"ram_total_go": round(mem.total / 1_000_000_000, 1),
|
||||
"ram_pct": mem.percent,
|
||||
"gpu": await _gpu_stats(),
|
||||
},
|
||||
"loaded": [] if isinstance(loaded, BaseException) else loaded,
|
||||
}
|
||||
_pulse_cache.update(at=time.monotonic(), value=value)
|
||||
return value
|
||||
|
||||
|
||||
async def _all_local_gpus() -> list[dict]:
|
||||
"""Tous les GPU NVIDIA visibles depuis le CONTENEUR Loki (peut être vide).
|
||||
|
||||
Mis en cache : le panneau Matériel se rafraîchit en continu tant qu'il est
|
||||
ouvert, et chaque appel lançait un sous-processus nvidia-smi.
|
||||
"""
|
||||
if not _NVIDIA_SMI:
|
||||
return []
|
||||
if time.monotonic() - _hw_cache["at"] < _HW_CACHE_TTL:
|
||||
return _hw_cache["value"] or []
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
_NVIDIA_SMI,
|
||||
"--query-gpu=index,name,memory.total,memory.used,utilization.gpu",
|
||||
"--format=csv,noheader,nounits",
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
out, _ = await asyncio.wait_for(proc.communicate(), timeout=3)
|
||||
gpus = []
|
||||
for line in out.decode().strip().splitlines():
|
||||
idx, name, total, used, util = (p.strip() for p in line.split(","))
|
||||
gpus.append({
|
||||
"index": int(idx), "name": name,
|
||||
"vram_total_mb": float(total), "vram_used_mb": float(used),
|
||||
"util_pct": float(util),
|
||||
})
|
||||
_hw_cache.update(at=time.monotonic(), value=gpus)
|
||||
return gpus
|
||||
except Exception:
|
||||
_hw_cache.update(at=time.monotonic(), value=[])
|
||||
return []
|
||||
|
||||
|
||||
@router.get("/hardware")
|
||||
async def hardware() -> dict:
|
||||
"""Vue matériel : GPU vu par Loki vs GPU réellement utilisé par Ollama."""
|
||||
# 1) Ce que voit le conteneur Loki (nvidia-smi) + override éventuel.
|
||||
local_gpus = await _all_local_gpus()
|
||||
override = None
|
||||
if settings.gpu_vram_mb > 0:
|
||||
override = {
|
||||
"name": settings.gpu_name or "GPU déclaré (GPU_VRAM_MB)",
|
||||
"vram_total_mb": settings.gpu_vram_mb,
|
||||
}
|
||||
|
||||
# 2) Ce qu'Ollama utilise réellement (modèles chargés + placement VRAM/CPU).
|
||||
ollama_info: dict = {"host": ollama.host, "connected": False, "running": []}
|
||||
try:
|
||||
version = await ollama.ping()
|
||||
ollama_info["connected"] = True
|
||||
ollama_info["version"] = version.get("version")
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
ollama_info["error"] = str(exc)[:200]
|
||||
|
||||
if ollama_info["connected"]:
|
||||
try:
|
||||
for m in await ollama.ps():
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
if size <= 0:
|
||||
where = "inconnu"
|
||||
elif vram >= size * 0.99:
|
||||
where = "GPU"
|
||||
elif vram <= size * 0.01:
|
||||
where = "CPU"
|
||||
else:
|
||||
where = "mixte"
|
||||
ollama_info["running"].append({
|
||||
"name": m.get("name") or m.get("model"),
|
||||
"processor": where,
|
||||
"gpu_percent": int(vram / size * 100) if size else 0,
|
||||
"size_mb": round(size / 1_000_000),
|
||||
"vram_mb": round(vram / 1_000_000),
|
||||
})
|
||||
except (httpx.HTTPError, OSError):
|
||||
pass
|
||||
|
||||
# Ollama est-il sur la même machine que Loki ? (heuristique sur l'hôte)
|
||||
host = ollama.host.lower()
|
||||
is_local = any(h in host for h in ("localhost", "127.0.0.1", "host.docker.internal"))
|
||||
|
||||
return {
|
||||
"loki_gpus": local_gpus,
|
||||
"gpu_override": override,
|
||||
"ollama": ollama_info,
|
||||
"ollama_is_local": is_local,
|
||||
"note": (
|
||||
"Loki ne voit pas directement le GPU d'un Ollama distant : il déduit "
|
||||
"le placement (GPU/CPU) depuis les modèles chargés. Déclare GPU_VRAM_MB "
|
||||
"pour l'auto-réglage si Ollama tourne sur une autre machine."
|
||||
),
|
||||
}
|
||||
@@ -1,117 +0,0 @@
|
||||
"""Skills : méthodes expertes injectées automatiquement selon la tâche.
|
||||
|
||||
Sélection lexicale instantanée (aucun appel LLM) : au plus UNE skill par
|
||||
message, injectée en message système pour ce tour uniquement.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
_SKILLS_DIR = os.path.join(os.path.dirname(__file__), "..", "skills")
|
||||
|
||||
# Ponytail (https://github.com/DietrichGebert/ponytail) : philosophie de code
|
||||
# « paresseux » (anti sur-ingénierie). Adaptée en méthode transverse injectée
|
||||
# pour toute tâche de code — indépendante du sélecteur de skill (mono-skill).
|
||||
PONYTAIL_TITLE = "Ponytail · code minimal"
|
||||
PONYTAIL_GUIDANCE = (
|
||||
"Méthode Ponytail — code minimal, anti sur-ingénierie. Avant d'écrire du "
|
||||
"code, descends l'échelle de décision et arrête-toi au premier échelon qui "
|
||||
"suffit :\n"
|
||||
"1. NE PAS coder ce qui n'est pas explicitement demandé (YAGNI) ;\n"
|
||||
"2. RÉUTILISER l'existant (fichiers/fonctions déjà présents) ;\n"
|
||||
"3. UTILISER les fonctions natives du langage / du navigateur ;\n"
|
||||
"4. en DERNIER recours seulement, écrire le minimum de code nécessaire.\n"
|
||||
"Livre la solution la plus simple qui fonctionne : aucune dépendance ni "
|
||||
"bibliothèque à installer, aucune abstraction prématurée, aucune "
|
||||
"fonctionnalité en plus non demandée (pas de « moteur IA », d'options ou de "
|
||||
"configuration superflues). Préfère un seul fichier clair à une "
|
||||
"architecture élaborée."
|
||||
)
|
||||
|
||||
# Seuil : nombre minimal de mots-clés distincts trouvés dans le message.
|
||||
_MIN_HITS = 2
|
||||
|
||||
|
||||
def _load() -> dict[str, dict]:
|
||||
out: dict[str, dict] = {}
|
||||
if not os.path.isdir(_SKILLS_DIR):
|
||||
return out
|
||||
for fname in sorted(os.listdir(_SKILLS_DIR)):
|
||||
if not fname.endswith(".md"):
|
||||
continue
|
||||
with open(os.path.join(_SKILLS_DIR, fname), encoding="utf-8") as f:
|
||||
raw = f.read()
|
||||
m = re.match(r"^---\n(.*?)\n---\n(.*)$", raw, re.S)
|
||||
if not m:
|
||||
continue
|
||||
meta: dict[str, str] = {}
|
||||
for line in m.group(1).splitlines():
|
||||
if ":" in line:
|
||||
key, _, value = line.partition(":")
|
||||
meta[key.strip()] = value.strip()
|
||||
keywords = [
|
||||
k.strip().lower()
|
||||
for k in meta.get("keywords", "").split(",")
|
||||
if k.strip()
|
||||
]
|
||||
out[meta.get("name", fname[:-3])] = {
|
||||
"name": meta.get("name", fname[:-3]),
|
||||
"title": meta.get("title", fname[:-3]),
|
||||
"keywords": keywords,
|
||||
"body": m.group(2).strip(),
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
_WEB_RE = re.compile(
|
||||
r"\b(html|css|javascript|js|page|site|web|appli|application|interface|"
|
||||
r"bouton|formulaire|canvas|animation|jeu|game|échiquier|echiquier|"
|
||||
r"dashboard|landing|maquette|ui|front)\b",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def is_web_task(message: str) -> bool:
|
||||
"""La demande porte-t-elle sur une page / appli web ?"""
|
||||
return bool(_WEB_RE.search(message or ""))
|
||||
|
||||
|
||||
# Contraintes « appli web » : le sandbox de prévisualisation est HORS-LIGNE et
|
||||
# isolé (pas de CDN, pas de WASM téléchargé). Sans ces règles, les modèles
|
||||
# promettent des libs (Stockfish, React…) qui ne se chargent jamais, éclatent
|
||||
# le code en plusieurs fichiers incohérents, puis déclarent « ça marche ».
|
||||
WEBAPP_GUIDANCE = (
|
||||
"Contraintes STRICTES pour une page/appli web :\n"
|
||||
"1. UN SEUL fichier index.html AUTONOME : tout le CSS dans <style>, tout le "
|
||||
"JS dans <script> à la fin du <body>. Pas de fichiers .css/.js séparés.\n"
|
||||
"2. AUCUNE ressource externe : pas de CDN, pas de <script src=\"https://…\">, "
|
||||
"pas de bibliothèque à télécharger (jQuery, React, chess.js, Stockfish, "
|
||||
".wasm…). Le sandbox est hors-ligne : ça ne se charge JAMAIS. Si une "
|
||||
"fonctionnalité a besoin d'une lib, écris une version simple toi-même en "
|
||||
"JavaScript natif.\n"
|
||||
"3. Le rendu doit s'afficher AU CHARGEMENT : construis réellement le DOM "
|
||||
"dans un script qui s'exécute (ex. génère les 64 cases de l'échiquier en "
|
||||
"boucle). Pas de fonction jamais appelée, pas de conteneur laissé vide.\n"
|
||||
"4. Ne prétends JAMAIS qu'une fonctionnalité marche sans l'avoir codée, et "
|
||||
"ne « simule » pas un moteur en le faisant passer pour réel : dis "
|
||||
"clairement ce qui est réel et ce qui ne l'est pas.\n"
|
||||
"5. Design sobre, lisible et moderne — mais la priorité est que ça "
|
||||
"FONCTIONNE vraiment."
|
||||
)
|
||||
|
||||
|
||||
ALL_SKILLS: dict[str, dict] = _load()
|
||||
|
||||
|
||||
def pick_skill(message: str) -> dict | None:
|
||||
"""Meilleure skill pour ce message, ou None si rien d'assez net."""
|
||||
low = message.lower()
|
||||
best, best_hits = None, 0
|
||||
for skill in ALL_SKILLS.values():
|
||||
hits = sum(1 for kw in skill["keywords"] if kw in low)
|
||||
if hits > best_hits:
|
||||
best, best_hits = skill, hits
|
||||
if best is None or best_hits < _MIN_HITS:
|
||||
return None
|
||||
return {"name": best["name"], "title": best["title"], "body": best["body"]}
|
||||
@@ -1,944 +0,0 @@
|
||||
"""Outils de l'agent, exécutés côté serveur et confinés au workspace.
|
||||
|
||||
Chaque outil expose :
|
||||
- une définition JSON (format function-calling Ollama/OpenAI) ;
|
||||
- une implémentation Python qui renvoie un dict {ok, summary, ...}.
|
||||
|
||||
Toutes les opérations fichier sont strictement confinées à WORKSPACE_DIR :
|
||||
toute tentative de sortie (../, chemin absolu hors workspace) est rejetée.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from contextvars import ContextVar
|
||||
|
||||
import httpx
|
||||
|
||||
from .config import settings
|
||||
|
||||
|
||||
class ToolError(Exception):
|
||||
"""Erreur d'exécution d'un outil (message destiné au modèle)."""
|
||||
|
||||
|
||||
# Projet actif pour la requête en cours : re-racine tous les outils sur
|
||||
# workspace/<projet>. None = racine du workspace (comportement historique).
|
||||
_ACTIVE_PROJECT: ContextVar[str | None] = ContextVar("loki_project", default=None)
|
||||
|
||||
PROJECT_NAME = re.compile(r"^[a-z0-9][a-z0-9_-]{0,40}$")
|
||||
|
||||
|
||||
def set_project(name: str | None) -> None:
|
||||
"""Fixe le projet actif de la requête (None = racine)."""
|
||||
if name is not None and not PROJECT_NAME.match(name):
|
||||
raise ToolError(f"nom de projet invalide : {name!r}")
|
||||
_ACTIVE_PROJECT.set(name)
|
||||
|
||||
|
||||
def active_root() -> str:
|
||||
"""Racine effective (workspace ou projet), créée si nécessaire."""
|
||||
return _workspace_root()
|
||||
|
||||
|
||||
def _workspace_root() -> str:
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
project = _ACTIVE_PROJECT.get()
|
||||
if project:
|
||||
root = os.path.join(root, project)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
return root
|
||||
|
||||
|
||||
def _safe_path(rel: str) -> str:
|
||||
"""Résout un chemin relatif en restant confiné au workspace."""
|
||||
root = _workspace_root()
|
||||
target = os.path.abspath(os.path.join(root, rel or "."))
|
||||
if target != root and not target.startswith(root + os.sep):
|
||||
raise ToolError(f"chemin hors du workspace refusé : {rel}")
|
||||
return target
|
||||
|
||||
|
||||
# ── Implémentations ──────────────────────────────────────────────────────
|
||||
# Fenêtrage de lecture : un gros fichier entier engloutit le contexte du
|
||||
# modèle. Au-delà du seuil, on renvoie une fenêtre + la marche à suivre.
|
||||
_READ_WINDOW_LINES = 200
|
||||
_READ_MAX_CHARS = 12_000
|
||||
|
||||
|
||||
def read_file(path: str, start_line: int = 1) -> dict:
|
||||
target = _safe_path(path)
|
||||
if os.path.isdir(target):
|
||||
raise ToolError(
|
||||
f"{path} est un dossier — utilise list_dir pour voir son contenu"
|
||||
)
|
||||
if not os.path.isfile(target):
|
||||
raise ToolError(f"fichier introuvable : {path}")
|
||||
with open(target, "r", encoding="utf-8", errors="replace") as f:
|
||||
lines = f.read().splitlines()
|
||||
total = len(lines)
|
||||
if total == 0:
|
||||
return {"ok": True, "content": "", "summary": "fichier vide (0 octet)"}
|
||||
|
||||
start = max(1, int(start_line or 1))
|
||||
window = lines[start - 1 : start - 1 + _READ_WINDOW_LINES]
|
||||
content = "\n".join(window)
|
||||
truncated_by_chars = False
|
||||
if len(content) > _READ_MAX_CHARS:
|
||||
content = content[:_READ_MAX_CHARS]
|
||||
truncated_by_chars = True
|
||||
|
||||
end = start + len(window) - 1
|
||||
if start == 1 and end >= total and not truncated_by_chars:
|
||||
return {"ok": True, "content": content, "summary": f"{total} lignes lues"}
|
||||
|
||||
# Fenêtre partielle : le modèle sait où il en est et comment continuer.
|
||||
note = (
|
||||
f"[fichier {path} : {total} lignes — fenêtre {start}-{end}. "
|
||||
f"Pour la suite : read_file(path, start_line={end + 1}). "
|
||||
"Pour cibler un passage précis : grep_search puis edit_file.]"
|
||||
)
|
||||
return {
|
||||
"ok": True,
|
||||
"content": content + "\n" + note,
|
||||
"summary": f"lignes {start}-{end} sur {total}",
|
||||
}
|
||||
|
||||
|
||||
def check_html(target: str) -> list[str]:
|
||||
"""Contrôles rapides d'une page HTML : références locales et balises.
|
||||
|
||||
Renvoie une liste de problèmes (vide = OK).
|
||||
"""
|
||||
issues: list[str] = []
|
||||
try:
|
||||
with open(target, "r", encoding="utf-8", errors="replace") as f:
|
||||
content = f.read()
|
||||
except OSError:
|
||||
return issues
|
||||
|
||||
base_dir = os.path.dirname(target)
|
||||
|
||||
# Références href/src : cassées (fichier local absent) ou externes
|
||||
# (CDN/WASM qui ne se chargeront pas dans un sandbox hors-ligne).
|
||||
for attr, ref in re.findall(r"""(href|src)=["']([^"'#]+)["']""", content, re.I):
|
||||
low = ref.lower()
|
||||
if low.startswith(("data:", "mailto:", "tel:", "javascript:")):
|
||||
continue
|
||||
if re.match(r"^(https?:)?//", ref, re.I):
|
||||
base = low.split("?")[0]
|
||||
if attr.lower() == "src" and base.endswith((".js", ".mjs", ".wasm")):
|
||||
issues.append(f"script externe (hors-ligne : ne se charge pas) : {ref}")
|
||||
elif attr.lower() == "href" and base.endswith(".css"):
|
||||
issues.append(f"style externe (hors-ligne : ne se charge pas) : {ref}")
|
||||
continue
|
||||
ref_path = os.path.normpath(os.path.join(base_dir, ref.split("?")[0]))
|
||||
if not os.path.exists(ref_path):
|
||||
issues.append(f"référence cassée : {ref}")
|
||||
|
||||
# Équilibre des balises structurantes.
|
||||
for tag in ("html", "head", "body", "div", "section", "script", "style"):
|
||||
opened = len(re.findall(rf"<{tag}[\s>]", content, re.I))
|
||||
closed = len(re.findall(rf"</{tag}>", content, re.I))
|
||||
if opened != closed:
|
||||
issues.append(f"balise <{tag}> : {opened} ouverte(s) / {closed} fermée(s)")
|
||||
|
||||
return issues[:6]
|
||||
|
||||
|
||||
def _verify_written(target: str) -> str | None:
|
||||
"""Vérification immédiate après écriture (py/json/html).
|
||||
|
||||
Renvoyer l'erreur au modèle tout de suite lui permet de se corriger dans
|
||||
le même tour, au lieu de livrer un fichier cassé.
|
||||
"""
|
||||
ext = os.path.splitext(target)[1].lower()
|
||||
try:
|
||||
with open(target, "r", encoding="utf-8", errors="replace") as f:
|
||||
content = f.read()
|
||||
if ext == ".json":
|
||||
import json as _json
|
||||
_json.loads(content)
|
||||
elif ext == ".py":
|
||||
compile(content, target, "exec")
|
||||
elif ext in (".html", ".htm"):
|
||||
problems = check_html(target)
|
||||
if problems:
|
||||
return " ; ".join(problems)
|
||||
elif ext in (".js", ".mjs"):
|
||||
node = shutil.which("node")
|
||||
if node:
|
||||
proc = subprocess.run(
|
||||
[node, "--check", target], capture_output=True, text=True,
|
||||
timeout=15,
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
return (proc.stderr or proc.stdout)[:300]
|
||||
except SyntaxError as exc:
|
||||
return f"SyntaxError ligne {exc.lineno}: {exc.msg}"
|
||||
except ValueError as exc:
|
||||
return f"JSON invalide : {exc}"
|
||||
except OSError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def write_file(path: str, content: str, mode: str = "overwrite") -> dict:
|
||||
target = _safe_path(path)
|
||||
if mode not in {"overwrite", "append"}:
|
||||
raise ToolError("mode write_file invalide : utilise overwrite ou append")
|
||||
os.makedirs(os.path.dirname(target) or _workspace_root(), exist_ok=True)
|
||||
existed = os.path.isfile(target)
|
||||
with open(target, "a" if mode == "append" else "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
lines = len(content.splitlines())
|
||||
verb = "complété" if mode == "append" else "modifié" if existed else "écrit"
|
||||
result = {"ok": True, "summary": f"{verb} · {lines} lignes", "lines": lines}
|
||||
problem = _verify_written(target)
|
||||
if problem:
|
||||
result["verification"] = problem
|
||||
result["summary"] += f" · ⚠ {problem}"
|
||||
return result
|
||||
|
||||
|
||||
def _leading_ws(line: str) -> str:
|
||||
"""Renvoie l'indentation (blancs de gauche) d'une ligne."""
|
||||
return line[: len(line) - len(line.lstrip())]
|
||||
|
||||
|
||||
def _reindent(search_lines: list[str], window: list[str], replace: str) -> list[str]:
|
||||
"""Réaligne le texte de remplacement sur l'indentation réelle du fichier.
|
||||
|
||||
Quand la correspondance a été trouvée en tolérant l'indentation (le modèle
|
||||
a copié l'extrait « à plat »), on réapplique au remplacement le décalage
|
||||
d'indentation observé entre le fichier et la recherche, pour ne pas casser
|
||||
la mise en forme (Python surtout).
|
||||
"""
|
||||
src_indent = next((_leading_ws(s) for s in search_lines if s.strip()), "")
|
||||
file_indent = next((_leading_ws(w) for w in window if w.strip()), "")
|
||||
replace_lines = replace.splitlines()
|
||||
if file_indent == src_indent:
|
||||
return replace_lines
|
||||
out: list[str] = []
|
||||
for line in replace_lines:
|
||||
if not line.strip():
|
||||
out.append(line)
|
||||
elif src_indent and line.startswith(src_indent):
|
||||
out.append(file_indent + line[len(src_indent):])
|
||||
elif not src_indent:
|
||||
out.append(file_indent + line)
|
||||
else:
|
||||
out.append(line)
|
||||
return out
|
||||
|
||||
|
||||
def _apply_edit(content: str, search: str, replace: str) -> tuple[str, str]:
|
||||
"""Applique un remplacement search→replace, du plus strict au plus tolérant.
|
||||
|
||||
1. correspondance exacte (unique) ;
|
||||
2. correspondance ligne à ligne en ignorant les espaces de fin / de début
|
||||
(indentation) — cas le plus fréquent où un petit modèle recopie l'extrait
|
||||
sans reproduire fidèlement les blancs.
|
||||
|
||||
Renvoie (nouveau_contenu, note). Lève ToolError si introuvable ou ambigu.
|
||||
"""
|
||||
# 1. Correspondance exacte.
|
||||
count = content.count(search)
|
||||
if count == 1:
|
||||
return content.replace(search, replace, 1), ""
|
||||
if count > 1:
|
||||
raise ToolError(
|
||||
f"extrait présent {count} fois : ajoute du contexte "
|
||||
"autour pour le rendre unique."
|
||||
)
|
||||
|
||||
# 2. Correspondance tolérante (ligne à ligne, espaces normalisés).
|
||||
file_lines = content.splitlines(keepends=True)
|
||||
search_lines = search.splitlines()
|
||||
if not any(s.strip() for s in search_lines):
|
||||
raise ToolError("extrait vide après normalisation")
|
||||
norm_search = [s.strip() for s in search_lines]
|
||||
n = len(search_lines)
|
||||
hits = [
|
||||
i
|
||||
for i in range(len(file_lines) - n + 1)
|
||||
if [w.strip() for w in file_lines[i:i + n]] == norm_search
|
||||
]
|
||||
if not hits:
|
||||
raise ToolError("introuvable")
|
||||
if len(hits) > 1:
|
||||
raise ToolError(
|
||||
f"extrait présent {len(hits)} fois : ajoute du contexte "
|
||||
"autour pour le rendre unique."
|
||||
)
|
||||
|
||||
i = hits[0]
|
||||
window = file_lines[i:i + n]
|
||||
newline = "\r\n" if window and window[0].endswith("\r\n") else "\n"
|
||||
adjusted = _reindent(search_lines, window, replace)
|
||||
rep_text = newline.join(adjusted)
|
||||
if window and window[-1].endswith("\n"):
|
||||
rep_text += newline
|
||||
new_content = "".join(file_lines[:i]) + rep_text + "".join(file_lines[i + n:])
|
||||
return new_content, "correspondance tolérante (indentation/espaces ignorés)"
|
||||
|
||||
|
||||
def edit_file(path: str, search: str, replace: str) -> dict:
|
||||
"""Modification chirurgicale : remplace un extrait du fichier.
|
||||
|
||||
Bien plus fiable que réécrire tout le fichier avec un petit modèle : seul le
|
||||
fragment visé change, le reste est garanti intact. La correspondance tolère
|
||||
les différences d'espaces / d'indentation, pour ne pas bloquer quand le
|
||||
modèle recopie l'extrait de façon approximative.
|
||||
"""
|
||||
target = _safe_path(path)
|
||||
if not os.path.isfile(target):
|
||||
raise ToolError(f"fichier introuvable : {path}")
|
||||
if not search:
|
||||
raise ToolError("search vide : fournis l'extrait à remplacer")
|
||||
with open(target, "r", encoding="utf-8", errors="replace") as f:
|
||||
content = f.read()
|
||||
|
||||
try:
|
||||
new_content, note = _apply_edit(content, search, replace)
|
||||
except ToolError as exc:
|
||||
reason = str(exc)
|
||||
if reason == "introuvable":
|
||||
preview = search.strip().splitlines()[0][:60] if search.strip() else ""
|
||||
raise ToolError(
|
||||
f"extrait introuvable dans {path} (cherché : « {preview}… »). "
|
||||
"Relis le fichier avec read_file et copie l'extrait EXACT, "
|
||||
"ou utilise write_file pour réécrire le fichier."
|
||||
) from exc
|
||||
raise ToolError(f"{reason.rstrip('.')} dans {path}.") from exc
|
||||
|
||||
with open(target, "w", encoding="utf-8") as f:
|
||||
f.write(new_content)
|
||||
|
||||
delta = len(replace.splitlines()) - len(search.splitlines())
|
||||
summary = f"remplacé · {delta:+d} ligne(s)"
|
||||
if note:
|
||||
summary += f" · {note}"
|
||||
result = {"ok": True, "summary": summary}
|
||||
problem = _verify_written(target)
|
||||
if problem:
|
||||
result["verification"] = problem
|
||||
result["summary"] += f" · ⚠ {problem}"
|
||||
return result
|
||||
|
||||
|
||||
def memory_search(query: str) -> dict:
|
||||
"""Cherche dans les notes que l'agent a lui-même enregistrées."""
|
||||
from . import memory_notes
|
||||
|
||||
hits = memory_notes.search_notes(query, limit=3)
|
||||
if not hits:
|
||||
return {"ok": True, "notes": [], "summary": "aucune note correspondante"}
|
||||
return {
|
||||
"ok": True,
|
||||
"notes": hits,
|
||||
"summary": f"{len(hits)} note(s) trouvée(s)",
|
||||
}
|
||||
|
||||
|
||||
def memory_save(title: str, content: str) -> dict:
|
||||
"""Enregistre une note durable (préférence, décision, fait à retenir)."""
|
||||
from . import memory_notes
|
||||
|
||||
result = memory_notes.save_note(title, content)
|
||||
if not result["ok"]:
|
||||
raise ToolError(result["summary"])
|
||||
return result
|
||||
|
||||
|
||||
_GREP_SKIP_DIRS = {".git", "node_modules", "__pycache__", ".venv", "dist"}
|
||||
_MAX_GREP_MATCHES = 50
|
||||
|
||||
|
||||
def grep_search(pattern: str, path: str = ".") -> dict:
|
||||
"""Recherche un motif (regex) dans les fichiers du workspace."""
|
||||
if not pattern:
|
||||
raise ToolError("pattern vide")
|
||||
try:
|
||||
rx = re.compile(pattern, re.I)
|
||||
except re.error as exc:
|
||||
raise ToolError(f"regex invalide : {exc}")
|
||||
|
||||
root = _safe_path(path)
|
||||
matches: list[str] = []
|
||||
files_hit: set[str] = set()
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
dirnames[:] = [d for d in dirnames if d not in _GREP_SKIP_DIRS]
|
||||
for fname in sorted(filenames):
|
||||
full = os.path.join(dirpath, fname)
|
||||
if os.path.getsize(full) > 1_000_000:
|
||||
continue
|
||||
rel = os.path.relpath(full, _workspace_root())
|
||||
try:
|
||||
with open(full, "r", encoding="utf-8", errors="replace") as f:
|
||||
for lineno, line in enumerate(f, 1):
|
||||
if rx.search(line):
|
||||
matches.append(f"{rel}:{lineno}: {line.strip()[:160]}")
|
||||
files_hit.add(rel)
|
||||
if len(matches) >= _MAX_GREP_MATCHES:
|
||||
break
|
||||
except OSError:
|
||||
continue
|
||||
if len(matches) >= _MAX_GREP_MATCHES:
|
||||
break
|
||||
if len(matches) >= _MAX_GREP_MATCHES:
|
||||
break
|
||||
|
||||
summary = (
|
||||
f"{len(matches)} correspondance(s) dans {len(files_hit)} fichier(s)"
|
||||
if matches else "aucune correspondance"
|
||||
)
|
||||
return {"ok": True, "matches": matches, "summary": summary}
|
||||
|
||||
|
||||
def list_dir(path: str = ".") -> dict:
|
||||
target = _safe_path(path)
|
||||
if not os.path.isdir(target):
|
||||
raise ToolError(f"répertoire introuvable : {path}")
|
||||
entries = []
|
||||
for name in sorted(os.listdir(target)):
|
||||
full = os.path.join(target, name)
|
||||
entries.append({"name": name, "type": "dir" if os.path.isdir(full) else "file"})
|
||||
return {
|
||||
"ok": True,
|
||||
"entries": entries,
|
||||
"summary": f"{len(entries)} élément(s)",
|
||||
}
|
||||
|
||||
|
||||
def web_search(query: str, max_results: int = 5) -> dict:
|
||||
"""Recherche web (DuckDuckGo HTML, sans clé d'API).
|
||||
|
||||
Optionnellement, si SEARX_URL est défini, interroge une instance SearxNG.
|
||||
Renvoie une liste de résultats {title, url, snippet}.
|
||||
"""
|
||||
query = (query or "").strip()
|
||||
if not query:
|
||||
raise ToolError("requête de recherche vide")
|
||||
|
||||
searx = os.environ.get("SEARX_URL")
|
||||
try:
|
||||
if searx:
|
||||
results = _search_searx(searx, query, max_results)
|
||||
else:
|
||||
results = _search_duckduckgo(query, max_results)
|
||||
except httpx.HTTPError as exc:
|
||||
raise ToolError(f"recherche web indisponible : {exc}") from exc
|
||||
|
||||
summary = f"{len(results)} résultat(s)" if results else "aucun résultat"
|
||||
return {"ok": True, "results": results, "summary": summary}
|
||||
|
||||
|
||||
def _search_searx(base: str, query: str, n: int) -> list[dict]:
|
||||
with httpx.Client(timeout=10.0) as client:
|
||||
r = client.get(
|
||||
base.rstrip("/") + "/search",
|
||||
params={"q": query, "format": "json"},
|
||||
)
|
||||
r.raise_for_status()
|
||||
data = r.json().get("results", [])[:n]
|
||||
return [
|
||||
{"title": d.get("title", ""), "url": d.get("url", ""),
|
||||
"snippet": d.get("content", "")}
|
||||
for d in data
|
||||
]
|
||||
|
||||
|
||||
_DDG_RESULT = re.compile(
|
||||
r'<a[^>]*class="result__a"[^>]*href="([^"]+)"[^>]*>(.*?)</a>'
|
||||
r'.*?class="result__snippet"[^>]*>(.*?)</a>',
|
||||
re.DOTALL,
|
||||
)
|
||||
_TAGS = re.compile(r"<[^>]+>")
|
||||
|
||||
|
||||
def _clean(text: str) -> str:
|
||||
return html.unescape(_TAGS.sub("", text)).strip()
|
||||
|
||||
|
||||
def _search_duckduckgo(query: str, n: int) -> list[dict]:
|
||||
with httpx.Client(timeout=10.0, follow_redirects=True) as client:
|
||||
r = client.post(
|
||||
"https://html.duckduckgo.com/html/",
|
||||
data={"q": query},
|
||||
headers={"User-Agent": "Mozilla/5.0 (Loki agent)"},
|
||||
)
|
||||
r.raise_for_status()
|
||||
results = []
|
||||
for url, title, snippet in _DDG_RESULT.findall(r.text)[:n]:
|
||||
results.append({
|
||||
"title": _clean(title),
|
||||
"url": html.unescape(url),
|
||||
"snippet": _clean(snippet),
|
||||
})
|
||||
return results
|
||||
|
||||
|
||||
# Lignes porteuses de signal dans une sortie de commande en échec.
|
||||
_ERROR_LINE = re.compile(
|
||||
r"error|erreur|fail|except|traceback|fatal|warn|undefined|cannot|"
|
||||
r"not found|introuvable|refus|denied|invalid|missing|panic",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _dedupe_lines(lines: list[str]) -> list[str]:
|
||||
"""Compacte les répétitions consécutives (« ligne ×N »)."""
|
||||
out: list[str] = []
|
||||
for line in lines:
|
||||
if out:
|
||||
base, _, count = out[-1].partition(" ×")
|
||||
if base == line:
|
||||
n = int(count) if count.isdigit() else 1
|
||||
out[-1] = f"{line} ×{n + 1}"
|
||||
continue
|
||||
out.append(line)
|
||||
return out
|
||||
|
||||
|
||||
def _compact_output(output: str, exit_code: int) -> str:
|
||||
"""Filtre la sortie shell façon rtk : le signal, pas le déroulé.
|
||||
|
||||
- succès : dernières lignes seulement (le détail n'apporte rien) ;
|
||||
- échec : lignes d'erreur + fin de sortie, dédupliquées.
|
||||
Tronquer bêtement à N caractères gardait le bruit et coupait l'erreur.
|
||||
"""
|
||||
lines = [l.rstrip() for l in output.splitlines() if l.strip()]
|
||||
lines = _dedupe_lines(lines)
|
||||
if exit_code == 0:
|
||||
kept = lines[-12:]
|
||||
text = "\n".join(kept)
|
||||
if len(lines) > 12:
|
||||
text = f"[…{len(lines) - 12} lignes omises]\n" + text
|
||||
return text[:1200]
|
||||
|
||||
error_lines = [l for l in lines if _ERROR_LINE.search(l)]
|
||||
tail = lines[-10:]
|
||||
kept = error_lines[:20] + [l for l in tail if l not in error_lines[:20]]
|
||||
text = "\n".join(kept)
|
||||
if len(lines) > len(kept):
|
||||
text = f"[…sortie filtrée : {len(kept)}/{len(lines)} lignes]\n" + text
|
||||
return text[:2500]
|
||||
|
||||
|
||||
# Pseudo-fichiers absolus inoffensifs, tolérés dans les commandes shell.
|
||||
_ALLOWED_ABS = ("/dev/null", "/dev/stdout", "/dev/stderr", "/dev/zero", "/dev/tty")
|
||||
# Jetons ressemblant à un chemin (absolu, ~ ou contenant ../).
|
||||
_PATH_TOKEN = re.compile(r"""(?:^|[\s=:><|&(])((?:~|/)[^\s'"|&;><)]*|[^\s'"|&;><)]*\.\.[^\s'"|&;><)]*)""")
|
||||
|
||||
|
||||
def _guard_shell(command: str) -> None:
|
||||
"""Refuse toute commande qui référence un chemin hors du workspace.
|
||||
|
||||
Barrière de confinement (best-effort) : le shell est trop puissant pour être
|
||||
totalement bridé, mais on bloque les cas concrets d'évasion — chemins
|
||||
absolus hors workspace (`/config/...`, `~/...`) et remontées `../` qui
|
||||
sortent du workspace. Combiné à la validation utilisateur, ça empêche le
|
||||
modèle d'écrire ailleurs que dans son workspace.
|
||||
"""
|
||||
# Racine autorisée = base du workspace (pas le sous-dossier projet) : ainsi
|
||||
# `ls /workspace` reste permis même quand la session cible un projet.
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
for token in _PATH_TOKEN.findall(command):
|
||||
token = token.strip()
|
||||
if not token:
|
||||
continue
|
||||
# Jeton commençant par `//` = autorité d'URL (http://, ws://…) ou chemin
|
||||
# réseau, jamais une évasion du système de fichiers : on l'ignore.
|
||||
if token.startswith("//"):
|
||||
continue
|
||||
if token.startswith("~"):
|
||||
raise ToolError(
|
||||
f"chemin hors du workspace refusé : {token}. "
|
||||
"Utilise uniquement des chemins relatifs au workspace."
|
||||
)
|
||||
if token.startswith("/"):
|
||||
if any(token == a or token.startswith(a + "/") for a in _ALLOWED_ABS):
|
||||
continue
|
||||
resolved = os.path.abspath(token)
|
||||
if resolved != root and not resolved.startswith(root + os.sep):
|
||||
raise ToolError(
|
||||
f"chemin absolu hors du workspace refusé : {token}. "
|
||||
"Utilise uniquement des chemins relatifs au workspace."
|
||||
)
|
||||
elif ".." in token.split("/"):
|
||||
resolved = os.path.abspath(os.path.join(root, token))
|
||||
if resolved != root and not resolved.startswith(root + os.sep):
|
||||
raise ToolError(
|
||||
f"remontée hors du workspace refusée : {token}. "
|
||||
"Reste dans le workspace."
|
||||
)
|
||||
|
||||
|
||||
def run_shell(command: str, timeout: int = 60) -> dict:
|
||||
"""Exécute une commande shell dans le workspace (outil sensible).
|
||||
|
||||
L'exécution effective n'a lieu qu'après validation utilisateur (gérée par
|
||||
la boucle agentique / la route /api/shell). Confinée au workspace.
|
||||
"""
|
||||
command = (command or "").strip()
|
||||
if not command:
|
||||
raise ToolError("commande vide")
|
||||
_guard_shell(command)
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
command,
|
||||
shell=True,
|
||||
cwd=_workspace_root(),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=timeout,
|
||||
)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise ToolError(f"délai dépassé ({timeout}s)") from exc
|
||||
|
||||
out = (proc.stdout or "") + (proc.stderr or "")
|
||||
# Filtrage signal/bruit (façon rtk) plutôt que troncature aveugle.
|
||||
out = _compact_output(out, proc.returncode)
|
||||
status = "ok" if proc.returncode == 0 else "error"
|
||||
return {
|
||||
"ok": proc.returncode == 0,
|
||||
"exit_code": proc.returncode,
|
||||
"output": out,
|
||||
"summary": f"code {proc.returncode}",
|
||||
"_status": status,
|
||||
}
|
||||
|
||||
|
||||
def run_check(path: str) -> dict:
|
||||
"""Vérification STATIQUE d'un fichier de code — n'exécute jamais rien.
|
||||
|
||||
.py -> py_compile ; .js/.mjs -> node --check (si node présent) ;
|
||||
.html -> check_html ; .json -> parse. Autres types : ok sans contrôle.
|
||||
"""
|
||||
target = _safe_path(path)
|
||||
if not os.path.isfile(target):
|
||||
raise ToolError(f"fichier introuvable : {path}")
|
||||
ext = os.path.splitext(target)[1].lower()
|
||||
issues: list[str] = []
|
||||
|
||||
if ext == ".py":
|
||||
import py_compile
|
||||
try:
|
||||
py_compile.compile(target, doraise=True)
|
||||
except py_compile.PyCompileError as exc:
|
||||
issues.append(str(exc.msg)[:500])
|
||||
elif ext in (".js", ".mjs"):
|
||||
node = shutil.which("node")
|
||||
if node:
|
||||
proc = subprocess.run(
|
||||
[node, "--check", target], capture_output=True, text=True,
|
||||
timeout=15,
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
issues.append((proc.stderr or proc.stdout)[:500])
|
||||
elif ext in (".html", ".htm"):
|
||||
issues.extend(check_html(target))
|
||||
elif ext == ".json":
|
||||
try:
|
||||
with open(target, encoding="utf-8") as f:
|
||||
json.load(f)
|
||||
except json.JSONDecodeError as exc:
|
||||
issues.append(f"JSON invalide : {exc}")
|
||||
|
||||
ok = not issues
|
||||
return {
|
||||
"ok": ok,
|
||||
"issues": issues,
|
||||
"summary": "aucun problème" if ok else f"{len(issues)} problème(s)",
|
||||
"_status": "ok" if ok else "error",
|
||||
}
|
||||
|
||||
|
||||
# ── Registre & définitions exposées au modèle ────────────────────────────
|
||||
TOOL_IMPL = {
|
||||
"read_file": read_file,
|
||||
"write_file": write_file,
|
||||
"edit_file": edit_file,
|
||||
"list_dir": list_dir,
|
||||
"grep_search": grep_search,
|
||||
"web_search": web_search,
|
||||
"run_shell": run_shell,
|
||||
"run_check": run_check,
|
||||
"memory_search": memory_search,
|
||||
"memory_save": memory_save,
|
||||
}
|
||||
|
||||
TOOL_DEFINITIONS = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "read_file",
|
||||
"description": (
|
||||
"Lire le contenu d'un fichier du workspace. Les gros fichiers "
|
||||
"sont renvoyés par fenêtres de 200 lignes : utilise start_line "
|
||||
"pour lire la suite, ou grep_search pour cibler un passage."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string", "description": "Chemin relatif au workspace"},
|
||||
"start_line": {
|
||||
"type": "integer",
|
||||
"description": "Première ligne de la fenêtre (défaut 1)",
|
||||
},
|
||||
},
|
||||
"required": ["path"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "write_file",
|
||||
"description": "Créer ou modifier un fichier du workspace.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string", "description": "Chemin relatif au workspace"},
|
||||
"content": {
|
||||
"type": "string",
|
||||
"description": "Contenu complet ou morceau court du fichier",
|
||||
},
|
||||
"mode": {
|
||||
"type": "string",
|
||||
"enum": ["overwrite", "append"],
|
||||
"description": (
|
||||
"overwrite pour le premier morceau, append pour les suivants"
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["path", "content"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "edit_file",
|
||||
"description": (
|
||||
"Modifier UN extrait précis d'un fichier existant (recherche/"
|
||||
"remplacement exact). Préférable à write_file pour toute "
|
||||
"modification partielle : le reste du fichier reste intact."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string", "description": "Chemin relatif au workspace"},
|
||||
"search": {
|
||||
"type": "string",
|
||||
"description": "Extrait EXACT à remplacer (copie fidèle, unique dans le fichier)",
|
||||
},
|
||||
"replace": {"type": "string", "description": "Nouveau texte"},
|
||||
},
|
||||
"required": ["path", "search", "replace"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "list_dir",
|
||||
"description": "Lister le contenu d'un répertoire du workspace.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string", "description": "Répertoire (défaut : racine)"}
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "grep_search",
|
||||
"description": (
|
||||
"Chercher un motif (regex, insensible à la casse) dans tous les "
|
||||
"fichiers du workspace. Renvoie fichier:ligne:texte."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pattern": {"type": "string", "description": "Motif à chercher"},
|
||||
"path": {"type": "string", "description": "Sous-répertoire (défaut : racine)"},
|
||||
},
|
||||
"required": ["pattern"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "code_task",
|
||||
"description": (
|
||||
"Confier une tâche de programmation au moteur code (édition "
|
||||
"multi-fichiers fiable, commits git). À utiliser pour créer ou "
|
||||
"modifier du code."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"instruction": {
|
||||
"type": "string",
|
||||
"description": "La tâche de code, précise et complète",
|
||||
},
|
||||
"files": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"},
|
||||
"description": "Fichiers concernés (optionnel)",
|
||||
},
|
||||
},
|
||||
"required": ["instruction"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "web_search",
|
||||
"description": "Rechercher sur le web et renvoyer les meilleurs résultats.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "Termes de recherche"}
|
||||
},
|
||||
"required": ["query"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "run_shell",
|
||||
"description": (
|
||||
"Exécuter une commande shell dans le workspace. Outil sensible :"
|
||||
" l'utilisateur doit valider la commande avant exécution."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"command": {"type": "string", "description": "Commande à exécuter"}
|
||||
},
|
||||
"required": ["command"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "run_check",
|
||||
"description": (
|
||||
"Vérifier statiquement un fichier de code du workspace "
|
||||
"(syntaxe Python/JS/JSON, structure HTML). N'exécute rien."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"path": {"type": "string", "description": "Chemin relatif au workspace"}
|
||||
},
|
||||
"required": ["path"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "memory_search",
|
||||
"description": (
|
||||
"Chercher dans tes notes durables (préférences de l'utilisateur, "
|
||||
"décisions, faits appris lors d'anciennes discussions)."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "Ce que tu cherches"}
|
||||
},
|
||||
"required": ["query"],
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "memory_save",
|
||||
"description": (
|
||||
"Enregistrer une note durable, réutilisable dans les prochaines "
|
||||
"discussions. À réserver aux informations stables et utiles "
|
||||
"(préférence, choix technique, contrainte). Jamais le détail "
|
||||
"d'une tâche en cours."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"title": {"type": "string", "description": "Titre court et parlant"},
|
||||
"content": {"type": "string", "description": "La note, concise"},
|
||||
},
|
||||
"required": ["title", "content"],
|
||||
},
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _normalize_args(name: str, args: dict | None) -> dict:
|
||||
args = dict(args or {})
|
||||
if name == "write_file":
|
||||
if "path" not in args:
|
||||
for alias in ("file_path", "filepath", "filename", "file", "name"):
|
||||
if args.get(alias):
|
||||
args["path"] = args[alias]
|
||||
break
|
||||
if "content" not in args:
|
||||
for alias in ("text", "body", "data", "contents"):
|
||||
if alias in args:
|
||||
args["content"] = args[alias]
|
||||
break
|
||||
elif name in {"read_file", "list_dir", "edit_file"} and "path" not in args:
|
||||
for alias in ("file_path", "filepath", "filename", "file", "dir", "directory"):
|
||||
if args.get(alias):
|
||||
args["path"] = args[alias]
|
||||
break
|
||||
if name == "edit_file":
|
||||
if "search" not in args and "old" in args:
|
||||
args["search"] = args.pop("old")
|
||||
if "replace" not in args and "new" in args:
|
||||
args["replace"] = args.pop("new")
|
||||
elif name == "grep_search" and "pattern" not in args:
|
||||
for alias in ("query", "search", "text", "regex"):
|
||||
if args.get(alias):
|
||||
args["pattern"] = args[alias]
|
||||
break
|
||||
return args
|
||||
|
||||
|
||||
def run_tool(name: str, args: dict) -> dict:
|
||||
"""Exécute un outil par son nom ; lève ToolError si inconnu/invalide."""
|
||||
impl = TOOL_IMPL.get(name)
|
||||
if impl is None:
|
||||
raise ToolError(f"outil inconnu : {name}")
|
||||
args = _normalize_args(name, args)
|
||||
if name == "write_file":
|
||||
missing = [key for key in ("path", "content") if key not in args]
|
||||
if missing:
|
||||
raise ToolError(
|
||||
"arguments invalides pour write_file : "
|
||||
f"{', '.join(missing)} requis. Utilise par exemple "
|
||||
'{"path":"index.html","content":"...","mode":"overwrite"}.'
|
||||
)
|
||||
try:
|
||||
return impl(**args)
|
||||
except ToolError:
|
||||
raise
|
||||
except TypeError as exc:
|
||||
raise ToolError(f"arguments invalides pour {name} : {exc}") from exc
|
||||
except OSError as exc:
|
||||
raise ToolError(f"erreur système ({name}) : {exc}") from exc
|
||||
@@ -1,2 +0,0 @@
|
||||
[pytest]
|
||||
asyncio_mode = auto
|
||||
@@ -1,13 +0,0 @@
|
||||
# Aider fige ses dépendances partagées (fastapi, pydantic, httpx…) : il est
|
||||
# listé en premier et le reste est aligné sur ses versions pour éviter tout
|
||||
# conflit de résolution pip.
|
||||
aider-chat==0.86.2
|
||||
fastapi==0.128.8
|
||||
uvicorn[standard]==0.34.0
|
||||
httpx==0.28.1
|
||||
pydantic==2.12.5
|
||||
pydantic-settings==2.7.1
|
||||
psutil==7.2.2
|
||||
mcp>=1.9
|
||||
pytest>=8.3
|
||||
pytest-asyncio>=0.25
|
||||
@@ -1,11 +0,0 @@
|
||||
---
|
||||
name: analyse-donnees
|
||||
title: Analyse de données
|
||||
keywords: csv, json, données, tableau, statistique, moyenne, analyse, colonnes, tri, filtre, graphique, export
|
||||
---
|
||||
Méthode d'analyse de données :
|
||||
1. EXAMINER : lis un échantillon du fichier réel (read_file) AVANT tout traitement. Identifie séparateur, encodage, en-têtes, types de colonnes.
|
||||
2. VALIDER : repère valeurs manquantes, doublons, incohérences de type. Signale-les explicitement au lieu de les masquer.
|
||||
3. TRANSFORMER : script clair et borné (pas de dépendance exotique) ; garde les données d'origine intactes, écris le résultat dans un nouveau fichier.
|
||||
4. PRÉSENTER : résumé chiffré (compte, min/max, moyennes pertinentes) + limites de l'analyse (données ignorées, hypothèses faites).
|
||||
Interdit : supposer le format sans avoir lu le fichier ; modifier les données sources ; présenter des chiffres sans dire comment ils sont calculés.
|
||||
@@ -1,11 +0,0 @@
|
||||
---
|
||||
name: creation-web
|
||||
title: Création web
|
||||
keywords: site, page, html, css, landing, formulaire, portfolio, interface, responsive, boutique, vitrine, menu, header, footer, animation
|
||||
---
|
||||
Méthode de création web, dans cet ordre :
|
||||
1. STRUCTURE : écris d'abord le HTML sémantique COMPLET (header, main, sections, footer). Contenu réel, pas de lorem ipsum si le sujet est connu.
|
||||
2. STYLE : CSS cohérent — palette limitée (3-4 couleurs), typographie lisible, espacements réguliers. Mobile-first, responsive (flexbox/grid, max-width sur images).
|
||||
3. INTERACTIVITÉ : JavaScript minimal et sans dépendance externe. Chaque interaction doit fonctionner hors ligne.
|
||||
4. VÉRIFIER : run_check sur chaque fichier produit ; contrôle les liens internes. Si les outils navigateur (mcp_playwright_*) sont disponibles, ouvre la page et lis la console pour vérifier qu'elle est propre.
|
||||
Interdit : livrer sans vérification ; référencer des images ou CDN externes non demandés ; produire un fichier tronqué (utilise write_file en plusieurs morceaux).
|
||||
@@ -1,12 +0,0 @@
|
||||
---
|
||||
name: debogage-systematique
|
||||
title: Débogage systématique
|
||||
keywords: bug, plante, erreur, exception, traceback, crash, échoue, marche pas, fonctionne pas, cassé, debug, débogue, corrige le bug, ne s'affiche pas, undefined, null, NaN
|
||||
---
|
||||
Méthode de débogage à suivre STRICTEMENT, étape par étape :
|
||||
1. REPRODUIRE : identifie l'entrée exacte et le comportement observé vs attendu. Si le message d'erreur est fourni, cite-le et pars de là.
|
||||
2. LOCALISER : lis le code concerné (read_file / grep_search) AVANT toute modification. Trouve la ligne qui produit le symptôme.
|
||||
3. HYPOTHÈSE : formule UNE cause précise. Vérifie-la en lisant le code, pas en devinant.
|
||||
4. CORRIGER : modification minimale et ciblée (edit_file). Ne réécris pas tout le fichier. Ne corrige qu'une cause à la fois.
|
||||
5. VÉRIFIER : relis le code modifié (run_check) et explique pourquoi le symptôme disparaît. Signale tout autre problème repéré sans le corriger.
|
||||
Interdit : proposer une correction sans avoir lu le code ; corriger plusieurs choses à la fois ; conclure « ça devrait marcher » sans vérification.
|
||||
@@ -1,10 +0,0 @@
|
||||
---
|
||||
name: redaction-structuree
|
||||
title: Rédaction structurée
|
||||
keywords: rédige, écris un texte, article, documentation, readme, rapport, résumé, lettre, mail, présentation, plan
|
||||
---
|
||||
Méthode de rédaction :
|
||||
1. PLAN : annonce la structure (titres/sections) avant de rédiger. Adapte la longueur à la demande — court par défaut.
|
||||
2. RÉDACTION : paragraphes courts, une idée par paragraphe, français précis. Concret d'abord, généralités interdites. Markdown propre (titres, listes, gras parcimonieux).
|
||||
3. RELECTURE : traque répétitions, phrases creuses, incohérences de ton. Coupe tout ce qui n'apporte rien.
|
||||
Interdit : remplissage, formules toutes faites (« dans un monde où… »), conclusion qui répète l'introduction.
|
||||
@@ -1,11 +0,0 @@
|
||||
---
|
||||
name: refactor-sur
|
||||
title: Refactor sûr
|
||||
keywords: refactor, refactorise, réorganise, nettoie, simplifie, renomme, découpe, extrait, duplication, dette
|
||||
---
|
||||
Méthode de refactoring, comportement STRICTEMENT identique avant/après :
|
||||
1. COMPRENDRE : lis TOUT le code concerné et ses usages (grep_search sur chaque symbole touché) avant de modifier quoi que ce soit.
|
||||
2. PETITS PAS : une seule transformation à la fois (renommage, extraction, déplacement). Jamais plusieurs changements mélangés.
|
||||
3. VÉRIFIER après CHAQUE pas : run_check sur les fichiers modifiés ; re-grep pour confirmer qu'aucun usage n'est resté sur l'ancien nom.
|
||||
4. RÉCAPITULER : liste ce qui a changé et pourquoi le comportement est inchangé.
|
||||
Interdit : changer le comportement ou l'API publique sans le signaler ; renommer sans vérifier tous les usages ; réécrire un fichier entier quand edit_file suffit.
|
||||
Whitespace-only changes.
@@ -1,124 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from app import agent # noqa: E402
|
||||
|
||||
|
||||
def _convo():
|
||||
return [
|
||||
{"role": "system", "content": "consigne"},
|
||||
{"role": "user", "content": "fais la tâche"},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_relance_apres_reflexion_seule(monkeypatch):
|
||||
"""1er appel : réflexion seule -> relance ; 2e appel : réponse finale."""
|
||||
calls: list[list[dict]] = []
|
||||
|
||||
async def fake_chat(model, convo, **kwargs):
|
||||
calls.append([dict(m) for m in convo])
|
||||
if len(calls) == 1:
|
||||
yield {"message": {"thinking": "hmm, je réfléchis longuement…"},
|
||||
"done": False}
|
||||
yield {"message": {}, "done": True}
|
||||
else:
|
||||
yield {"message": {"content": "réponse finale"}, "done": True}
|
||||
|
||||
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
||||
events = [
|
||||
e async for e in agent.run_agent("test", _convo(), enabled_tools=[])
|
||||
]
|
||||
|
||||
final = [e for e in events if e["type"] == "final"]
|
||||
assert final and "réponse finale" in final[0]["content"]
|
||||
# La relance a bien injecté la consigne de reprise.
|
||||
assert any(
|
||||
"Continue la tâche" in m["content"]
|
||||
for m in calls[1] if m["role"] == "user"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reflexion_pas_renvoyee_au_modele(monkeypatch):
|
||||
"""La pensée est affichée mais jamais réinjectée dans l'historique."""
|
||||
calls: list[list[dict]] = []
|
||||
|
||||
async def fake_chat(model, convo, **kwargs):
|
||||
calls.append([dict(m) for m in convo])
|
||||
if len(calls) == 1:
|
||||
yield {"message": {"thinking": "je planifie",
|
||||
"content": "étape 1",
|
||||
"tool_calls": [{"function": {
|
||||
"name": "list_dir",
|
||||
"arguments": {"path": "."}}}]},
|
||||
"done": True}
|
||||
else:
|
||||
yield {"message": {"content": "terminé"}, "done": True}
|
||||
|
||||
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
||||
events = [
|
||||
e async for e in agent.run_agent("test", _convo(), enabled_tools=None)
|
||||
]
|
||||
|
||||
final = [e for e in events if e["type"] == "final"]
|
||||
assert final and "terminé" in final[0]["content"]
|
||||
assert "je planifie" in final[0]["thinking"] # gardée pour l'UI
|
||||
# Aucun message assistant réinjecté ne contient la clé thinking.
|
||||
assistant_turns = [m for m in calls[1] if m["role"] == "assistant"]
|
||||
assert assistant_turns
|
||||
assert all("thinking" not in m for m in assistant_turns)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_coupe_circuit_pensee_interminable(monkeypatch):
|
||||
"""Pensée sans fin -> génération coupée en vol, puis relance qui aboutit."""
|
||||
calls: list[int] = []
|
||||
|
||||
async def fake_chat(model, convo, **kwargs):
|
||||
calls.append(1)
|
||||
if len(calls) == 1:
|
||||
# Flux de pensée « infini » : jamais de done, jamais de contenu.
|
||||
for _ in range(10_000):
|
||||
yield {"message": {"thinking": "x" * 200}, "done": False}
|
||||
else:
|
||||
yield {"message": {"content": "réponse après coupe"}, "done": True}
|
||||
|
||||
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
||||
events = [
|
||||
e async for e in agent.run_agent("test", _convo(), enabled_tools=[])
|
||||
]
|
||||
|
||||
final = [e for e in events if e["type"] == "final"]
|
||||
assert final and "réponse après coupe" in final[0]["content"]
|
||||
assert len(calls) == 2 # coupé puis relancé, pas d'épuisement du flux
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reflexion_coupee_apres_deux_impasses(monkeypatch):
|
||||
"""Deux itérations de pensée pure -> think désactivé, tâche finie."""
|
||||
seen_think: list = []
|
||||
|
||||
async def fake_chat(model, convo, think=None, **kwargs):
|
||||
seen_think.append(think)
|
||||
if len(seen_think) <= 2:
|
||||
yield {"message": {"thinking": "boucle de pensée"}, "done": True}
|
||||
else:
|
||||
yield {"message": {"content": "fini sans réfléchir"}, "done": True}
|
||||
|
||||
monkeypatch.setattr(agent.ollama, "chat", fake_chat)
|
||||
events = [
|
||||
e async for e in agent.run_agent(
|
||||
"test", _convo(), enabled_tools=[], think=True
|
||||
)
|
||||
]
|
||||
|
||||
final = [e for e in events if e["type"] == "final"]
|
||||
assert final and "fini sans réfléchir" in final[0]["content"]
|
||||
assert seen_think[-1] is False # think coupé pour l'appel final
|
||||
assert any(e["type"] == "notice" for e in events)
|
||||
@@ -1,62 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from app import agent, db, mcp_client # noqa: E402
|
||||
|
||||
db.init_db()
|
||||
|
||||
|
||||
def test_prune_compacte_les_anciens_resultats():
|
||||
big = "x" * 5000
|
||||
convo = [
|
||||
{"role": "system", "content": "consigne"},
|
||||
{"role": "user", "content": "tâche"},
|
||||
{"role": "assistant", "content": "", "tool_calls": [{}]},
|
||||
{"role": "tool", "tool_name": "read_file", "content": big},
|
||||
{"role": "assistant", "content": "", "tool_calls": [{}]},
|
||||
{"role": "tool", "tool_name": "grep_search", "content": big},
|
||||
]
|
||||
# Le dernier lot (index 5) commence à 5 : seul l'index 3 est compacté.
|
||||
agent._prune_old_tool_results(convo, before_index=5)
|
||||
assert len(convo[3]["content"]) < 500
|
||||
assert "archivé" in convo[3]["content"]
|
||||
assert convo[5]["content"] == big # lot courant intact
|
||||
assert convo[0]["content"] == "consigne" # système intact
|
||||
|
||||
|
||||
def test_prune_ignore_les_petits_resultats():
|
||||
convo = [{"role": "tool", "tool_name": "list_dir", "content": "court"}]
|
||||
agent._prune_old_tool_results(convo, before_index=1)
|
||||
assert convo[0]["content"] == "court"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_searxng_sans_url_erreur_claire():
|
||||
mgr = mcp_client.McpManager()
|
||||
try:
|
||||
result = await mgr.test_server("searxng")
|
||||
assert result["ok"] is False
|
||||
assert "SEARXNG_URL" in (result["error"] or "")
|
||||
assert "requis" in (result["error"] or "")
|
||||
finally:
|
||||
await mgr.aclose()
|
||||
|
||||
|
||||
def test_migration_v6_monte_les_anciens_defauts():
|
||||
from app import agent_config
|
||||
profiles = db.get_config_value(agent_config.MODEL_PROFILES_KEY) or {}
|
||||
profiles["testmodel:7b"] = {"num_ctx": 8192}
|
||||
profiles["custom:7b"] = {"num_ctx": 5000}
|
||||
db.set_config_value(agent_config.MODEL_PROFILES_KEY, profiles)
|
||||
db.set_config_value(agent_config.PROFILE_STATE_KEY, {"version": 5})
|
||||
|
||||
agent_config._migrate_profiles()
|
||||
|
||||
migrated = db.get_config_value(agent_config.MODEL_PROFILES_KEY)
|
||||
assert migrated["testmodel:7b"]["num_ctx"] == 16384 # ancien défaut monté
|
||||
assert migrated["custom:7b"]["num_ctx"] == 5000 # valeur perso respectée
|
||||
@@ -1,39 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from app import db # noqa: E402
|
||||
from app import mcp_client # noqa: E402
|
||||
|
||||
db.init_db()
|
||||
|
||||
|
||||
def test_catalog_has_preconfigured_servers():
|
||||
for sid in ("playwright", "context7", "fetch", "searxng", "custom"):
|
||||
assert sid in mcp_client.CATALOG
|
||||
assert mcp_client.CATALOG[sid]["label"]
|
||||
|
||||
|
||||
def test_state_defaults_disabled():
|
||||
state = mcp_client.get_mcp_state()
|
||||
assert set(state) == set(mcp_client.CATALOG)
|
||||
assert all(not s["enabled"] for s in state.values())
|
||||
|
||||
|
||||
def test_toggle_persists():
|
||||
mcp_client.set_mcp_state("fetch", enabled=True, params={})
|
||||
assert mcp_client.get_mcp_state()["fetch"]["enabled"] is True
|
||||
mcp_client.set_mcp_state("fetch", enabled=False, params={})
|
||||
assert mcp_client.get_mcp_state()["fetch"]["enabled"] is False
|
||||
|
||||
|
||||
def test_custom_params_persist():
|
||||
mcp_client.set_mcp_state(
|
||||
"custom", enabled=False, params={"command": "npx -y some-mcp"}
|
||||
)
|
||||
assert (
|
||||
mcp_client.get_mcp_state()["custom"]["params"]["command"]
|
||||
== "npx -y some-mcp"
|
||||
)
|
||||
@@ -1,96 +0,0 @@
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import textwrap
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from app import db, mcp_client # noqa: E402
|
||||
|
||||
db.init_db()
|
||||
|
||||
# Serveur MCP minimal : un outil "echo" + un outil au nom À TIRETS
|
||||
# (comme Context7 « resolve-library-id »).
|
||||
_FAKE_SERVER = textwrap.dedent("""
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
mcp = FastMCP("fake")
|
||||
|
||||
@mcp.tool()
|
||||
def echo(text: str) -> str:
|
||||
\"\"\"Répète le texte fourni.\"\"\"
|
||||
return "echo:" + text
|
||||
|
||||
@mcp.tool(name="dash-tool-name")
|
||||
def dash_tool(text: str) -> str:
|
||||
\"\"\"Outil au nom à tirets.\"\"\"
|
||||
return "dash:" + text
|
||||
|
||||
mcp.run()
|
||||
""")
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def fake_server_cmd(tmp_path):
|
||||
path = tmp_path / "fake_mcp.py"
|
||||
path.write_text(_FAKE_SERVER, encoding="utf-8")
|
||||
return [sys.executable, str(path)]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tools_exposed_and_called(fake_server_cmd, monkeypatch):
|
||||
monkeypatch.setitem(
|
||||
mcp_client.CATALOG, "fake",
|
||||
{"label": "Fake", "description": "", "command": fake_server_cmd,
|
||||
"url_param": False, "env_params": [], "expose": None},
|
||||
)
|
||||
mcp_client.set_mcp_state("fake", enabled=True, params={})
|
||||
mgr = mcp_client.McpManager()
|
||||
try:
|
||||
defs = await mgr.tool_definitions()
|
||||
names = [d["function"]["name"] for d in defs]
|
||||
assert "mcp_fake_echo" in names
|
||||
# Nom à tirets exposé assaini (compatibilité function-calling).
|
||||
assert "mcp_fake_dash_tool_name" in names
|
||||
result = await mgr.call_tool("mcp_fake_echo", {"text": "bonjour"})
|
||||
assert result["ok"] is True
|
||||
assert "echo:bonjour" in result["content"]
|
||||
# Appel via le nom assaini -> résolu vers le vrai nom à tirets.
|
||||
dash = await mgr.call_tool("mcp_fake_dash_tool_name", {"text": "x"})
|
||||
assert dash["ok"] is True and "dash:x" in dash["content"]
|
||||
# Tolérance : le modèle répond avec des tirets au lieu d'underscores.
|
||||
mixed = await mgr.call_tool("mcp_fake_dash-tool-name", {"text": "y"})
|
||||
assert mixed["ok"] is True and "dash:y" in mixed["content"]
|
||||
finally:
|
||||
await mgr.aclose()
|
||||
mcp_client.set_mcp_state("fake", enabled=False, params={})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disabled_server_exposes_nothing():
|
||||
mgr = mcp_client.McpManager()
|
||||
try:
|
||||
assert await mgr.tool_definitions() == []
|
||||
finally:
|
||||
await mgr.aclose()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_broken_server_never_raises(monkeypatch):
|
||||
monkeypatch.setitem(
|
||||
mcp_client.CATALOG, "broken",
|
||||
{"label": "Broken", "description": "",
|
||||
"command": [sys.executable, "-c", "import sys; sys.exit(3)"],
|
||||
"url_param": False, "env_params": [], "expose": None},
|
||||
)
|
||||
mcp_client.set_mcp_state("broken", enabled=True, params={})
|
||||
mgr = mcp_client.McpManager()
|
||||
try:
|
||||
assert await mgr.tool_definitions() == []
|
||||
assert mgr.statuses()["broken"]["state"] == "error"
|
||||
assert any("broken" in n.lower() or "Broken" in n for n in mgr.notices())
|
||||
finally:
|
||||
await mgr.aclose()
|
||||
mcp_client.set_mcp_state("broken", enabled=False, params={})
|
||||
@@ -1,62 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from app import db, tools # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
db.init_db()
|
||||
_ROOT = os.path.abspath(settings.workspace_dir)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_project():
|
||||
tools.set_project(None)
|
||||
yield
|
||||
tools.set_project(None)
|
||||
|
||||
|
||||
def test_racine_par_defaut():
|
||||
assert tools.active_root() == _ROOT
|
||||
|
||||
|
||||
def test_set_project_reracine():
|
||||
tools.set_project("demo")
|
||||
root = tools.active_root()
|
||||
assert root == os.path.join(_ROOT, "demo")
|
||||
assert os.path.isdir(root) # créée à la volée
|
||||
|
||||
|
||||
def test_confinement_conserve():
|
||||
tools.set_project("demo")
|
||||
with pytest.raises(tools.ToolError):
|
||||
tools._safe_path("../hors-projet")
|
||||
|
||||
|
||||
def test_nom_projet_invalide():
|
||||
for bad in ("../x", "UPPER", "a b", "", "x" * 50):
|
||||
with pytest.raises(tools.ToolError):
|
||||
tools.set_project(bad)
|
||||
|
||||
|
||||
def test_session_porte_son_projet():
|
||||
s = db.create_session("t", None, project="demo")
|
||||
assert db.get_session(s["id"])["project"] == "demo"
|
||||
db.set_session_project(s["id"], None)
|
||||
assert db.get_session(s["id"])["project"] is None
|
||||
|
||||
|
||||
def test_aides_contexte_suivent_le_projet():
|
||||
from app.routes.chat import _mentioned_files, _workspace_listing
|
||||
tools.set_project("ctxdemo")
|
||||
root = tools.active_root()
|
||||
with open(os.path.join(root, "app.py"), "w", encoding="utf-8") as f:
|
||||
f.write("x = 1")
|
||||
assert "app.py" in _workspace_listing()
|
||||
assert _mentioned_files("corrige app.py") == ["app.py"]
|
||||
tools.set_project(None)
|
||||
assert _mentioned_files("corrige app.py") == [] # absent de la racine
|
||||
@@ -1,46 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from fastapi.testclient import TestClient # noqa: E402
|
||||
|
||||
from app.main import app # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
client = TestClient(app)
|
||||
_ROOT = os.path.abspath(settings.workspace_dir)
|
||||
|
||||
|
||||
def test_creation_et_liste():
|
||||
r = client.post("/api/projects", json={"name": "routedemo"})
|
||||
assert r.status_code == 201
|
||||
assert os.path.isdir(os.path.join(_ROOT, "routedemo", ".git"))
|
||||
names = [p["name"] for p in client.get("/api/projects").json()["projects"]]
|
||||
assert "routedemo" in names
|
||||
|
||||
|
||||
def test_nom_invalide_400():
|
||||
assert client.post("/api/projects", json={"name": "../x"}).status_code == 400
|
||||
assert client.post("/api/projects", json={"name": "Demo"}).status_code == 400
|
||||
|
||||
|
||||
def test_existant_400():
|
||||
client.post("/api/projects", json={"name": "dup"})
|
||||
assert client.post("/api/projects", json={"name": "dup"}).status_code == 400
|
||||
|
||||
|
||||
def test_files_re_racine():
|
||||
client.post("/api/projects", json={"name": "scoped"})
|
||||
with open(os.path.join(_ROOT, "scoped", "a.txt"), "w") as f:
|
||||
f.write("x")
|
||||
with open(os.path.join(_ROOT, "racine.txt"), "w") as f:
|
||||
f.write("y")
|
||||
tree = client.get("/api/files", params={"project": "scoped"}).json()["tree"]
|
||||
names = [n["name"] for n in tree]
|
||||
assert "a.txt" in names and "racine.txt" not in names
|
||||
|
||||
|
||||
def test_files_projet_invalide_400():
|
||||
assert client.get("/api/files", params={"project": "../x"}).status_code == 400
|
||||
@@ -1,96 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from app import router # noqa: E402
|
||||
from app.routes.chat import ( # noqa: E402
|
||||
_mentioned_files,
|
||||
_prev_was_code,
|
||||
_session_code_context,
|
||||
_workspace_listing,
|
||||
)
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
|
||||
def test_message_code_explicite():
|
||||
assert router.is_code_task("crée une page html avec un formulaire") is True
|
||||
|
||||
|
||||
def test_message_banal_pas_code():
|
||||
assert router.is_code_task("quelle heure est-il ?") is False
|
||||
|
||||
|
||||
def test_followup_courts():
|
||||
# Suites typiques d'un travail de code, sans vocabulaire code explicite.
|
||||
assert router.is_code_followup("ajoute un bouton rouge en haut") is True
|
||||
assert router.is_code_followup("continue") is True
|
||||
assert router.is_code_followup("enlève le menu et agrandis le titre") is True
|
||||
assert router.is_code_followup("merci beaucoup !") is False
|
||||
|
||||
|
||||
def test_prev_was_code_via_engine():
|
||||
history = [
|
||||
{"role": "user", "content": "crée un site", "meta": None},
|
||||
{"role": "assistant", "content": "fait", "meta": {"engine": "code"}},
|
||||
]
|
||||
assert _prev_was_code(history) is True
|
||||
|
||||
|
||||
def test_prev_was_code_via_write_file():
|
||||
history = [
|
||||
{"role": "assistant", "content": "ok",
|
||||
"meta": {"tools": [{"name": "write_file", "args": {}}]}},
|
||||
]
|
||||
assert _prev_was_code(history) is True
|
||||
|
||||
|
||||
def test_prev_was_code_discussion():
|
||||
history = [
|
||||
{"role": "assistant", "content": "voici la réponse", "meta": None},
|
||||
]
|
||||
assert _prev_was_code(history) is False
|
||||
|
||||
|
||||
def test_mentioned_files_existants_seulement():
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(os.path.join(root, "index.html"), "w", encoding="utf-8") as f:
|
||||
f.write("<html></html>")
|
||||
found = _mentioned_files("modifie index.html et style.css")
|
||||
assert found == ["index.html"] # style.css n'existe pas
|
||||
|
||||
|
||||
def test_session_code_context_recap_et_fichiers():
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(os.path.join(root, "jeu.html"), "w", encoding="utf-8") as f:
|
||||
f.write("<html>jeu</html>")
|
||||
history = [
|
||||
{"role": "user", "content": "crée un jeu snake dans jeu.html",
|
||||
"meta": None},
|
||||
{"role": "assistant", "content": "fait",
|
||||
"meta": {"engine": "code",
|
||||
"tools": [{"name": "write_file",
|
||||
"args": {"path": "jeu.html"}}]}},
|
||||
{"role": "user", "content": "il y a des bugs", "meta": None},
|
||||
]
|
||||
recap, files = _session_code_context(history)
|
||||
assert "crée un jeu snake" in recap
|
||||
assert files == ["jeu.html"]
|
||||
|
||||
|
||||
def test_session_code_context_vide_sans_historique():
|
||||
recap, files = _session_code_context([])
|
||||
assert recap == ""
|
||||
assert files == []
|
||||
|
||||
|
||||
def test_workspace_listing_contient_fichiers():
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(os.path.join(root, "jeu.html"), "w", encoding="utf-8") as f:
|
||||
f.write("x")
|
||||
listing = _workspace_listing()
|
||||
assert "jeu.html" in listing
|
||||
@@ -1,43 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from app import tools # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
# Le workspace effectif peut avoir été fixé par un autre fichier de test
|
||||
# importé avant celui-ci : on écrit là où les outils lisent réellement.
|
||||
_WS = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(_WS, exist_ok=True)
|
||||
|
||||
|
||||
def _write(name: str, content: str) -> str:
|
||||
path = os.path.join(_WS, name)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
return name
|
||||
|
||||
|
||||
def test_python_valide():
|
||||
rel = _write("ok.py", "x = 1\nprint(x)\n")
|
||||
assert tools.run_check(rel)["ok"] is True
|
||||
|
||||
|
||||
def test_python_casse():
|
||||
rel = _write("ko.py", "def broken(:\n")
|
||||
result = tools.run_check(rel)
|
||||
assert result["ok"] is False
|
||||
assert result["issues"]
|
||||
|
||||
|
||||
def test_json_casse():
|
||||
rel = _write("ko.json", "{invalid")
|
||||
assert tools.run_check(rel)["ok"] is False
|
||||
|
||||
|
||||
def test_fichier_inconnu_type():
|
||||
rel = _write("notes.txt", "bonjour")
|
||||
result = tools.run_check(rel)
|
||||
assert result["ok"] is True # type non vérifiable = pas d'erreur
|
||||
@@ -1,26 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from app import skills # noqa: E402
|
||||
|
||||
|
||||
def test_five_skills_loaded():
|
||||
assert len(skills.ALL_SKILLS) == 5
|
||||
assert all(s["title"] and s["body"] for s in skills.ALL_SKILLS.values())
|
||||
|
||||
|
||||
def test_bug_message_picks_debug():
|
||||
s = skills.pick_skill("mon script plante avec une erreur TypeError au démarrage")
|
||||
assert s and s["name"] == "debogage-systematique"
|
||||
|
||||
|
||||
def test_web_message_picks_web():
|
||||
s = skills.pick_skill("crée une page html responsive pour ma boutique")
|
||||
assert s and s["name"] == "creation-web"
|
||||
|
||||
|
||||
def test_banal_message_picks_nothing():
|
||||
assert skills.pick_skill("bonjour, quelle heure est-il ?") is None
|
||||
@@ -1,69 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from app import tools # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
_WS = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(_WS, exist_ok=True)
|
||||
|
||||
|
||||
# ── read_file fenêtré ────────────────────────────────────────────────────
|
||||
|
||||
def _write(name: str, content: str) -> str:
|
||||
with open(os.path.join(_WS, name), "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
return name
|
||||
|
||||
|
||||
def test_read_file_petit_entier():
|
||||
rel = _write("petit.txt", "a\nb\nc\n")
|
||||
result = tools.read_file(rel)
|
||||
assert result["content"] == "a\nb\nc"
|
||||
assert "3 lignes" in result["summary"]
|
||||
|
||||
|
||||
def test_read_file_gros_fenetre():
|
||||
rel = _write("gros.txt", "\n".join(f"ligne {i}" for i in range(1, 501)))
|
||||
result = tools.read_file(rel)
|
||||
assert "ligne 200" in result["content"]
|
||||
assert "ligne 201" not in result["content"].replace("start_line=201", "")
|
||||
assert "start_line=201" in result["content"] # marche à suivre
|
||||
assert "1-200 sur 500" in result["summary"]
|
||||
|
||||
|
||||
def test_read_file_fenetre_suivante():
|
||||
rel = _write("gros2.txt", "\n".join(f"ligne {i}" for i in range(1, 501)))
|
||||
result = tools.read_file(rel, start_line=201)
|
||||
assert "ligne 201" in result["content"]
|
||||
assert "201-400 sur 500" in result["summary"]
|
||||
|
||||
|
||||
# ── sortie shell filtrée ─────────────────────────────────────────────────
|
||||
|
||||
def test_compact_succes_garde_la_fin():
|
||||
out = "\n".join(f"étape {i}" for i in range(1, 101))
|
||||
compact = tools._compact_output(out, 0)
|
||||
assert "étape 100" in compact
|
||||
assert "étape 1\n" not in compact
|
||||
assert "lignes omises" in compact
|
||||
|
||||
|
||||
def test_compact_echec_garde_les_erreurs():
|
||||
out = "\n".join(
|
||||
["compilation démarrée"]
|
||||
+ [f"module {i} ok" for i in range(50)]
|
||||
+ ["ERROR: variable x undefined", "build failed"]
|
||||
)
|
||||
compact = tools._compact_output(out, 1)
|
||||
assert "ERROR: variable x undefined" in compact
|
||||
assert "build failed" in compact
|
||||
assert "module 3 ok" not in compact
|
||||
|
||||
|
||||
def test_dedupe_repetitions():
|
||||
lines = ["warn: deprecated"] * 5 + ["fin"]
|
||||
assert tools._dedupe_lines(lines) == ["warn: deprecated ×5", "fin"]
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 2.7 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 7.7 KiB |
@@ -0,0 +1,17 @@
|
||||
// Point d'entrée du binaire loki. Tout le code vit dans internal/loki ; ce
|
||||
// dossier ne porte que le main() et les ressources Windows (.syso, icône,
|
||||
// versioninfo) qui doivent résider dans le dossier du package main.
|
||||
//
|
||||
// Les directives ci-dessous embarquent les métadonnées Windows (éditeur, version,
|
||||
// description) dans le .exe pour réduire les faux positifs antivirus. Régénère
|
||||
// les .syso après avoir bumpé la version : `go generate ./...`
|
||||
// (nécessite : go install github.com/josephspurrier/goversioninfo/cmd/goversioninfo@latest)
|
||||
//
|
||||
//go:generate go run github.com/R0m1k3/Loki/tools/gen-icon icon.ico
|
||||
//go:generate goversioninfo -64 -icon=icon.ico -o resource_windows_amd64.syso versioninfo.json
|
||||
//go:generate goversioninfo -64 -arm -icon=icon.ico -o resource_windows_arm64.syso versioninfo.json
|
||||
package main
|
||||
|
||||
import "github.com/R0m1k3/Loki/internal/loki"
|
||||
|
||||
func main() { loki.Main() }
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"FixedFileInfo": {
|
||||
"FileVersion": {
|
||||
"Major": 0,
|
||||
"Minor": 9,
|
||||
"Patch": 4,
|
||||
"Build": 0
|
||||
},
|
||||
"ProductVersion": {
|
||||
"Major": 0,
|
||||
"Minor": 9,
|
||||
"Patch": 4,
|
||||
"Build": 0
|
||||
},
|
||||
"FileFlagsMask": "3f",
|
||||
"FileFlags": "00",
|
||||
"FileOS": "040004",
|
||||
"FileType": "01",
|
||||
"FileSubType": "00"
|
||||
},
|
||||
"StringFileInfo": {
|
||||
"CompanyName": "Loki contributors",
|
||||
"FileDescription": "Loki — gestionnaire mono-binaire de serveurs llama.cpp auto-hébergés",
|
||||
"InternalName": "loki",
|
||||
"LegalCopyright": "Copyright (c) 2026 Loki contributors. MIT License.",
|
||||
"OriginalFilename": "loki.exe",
|
||||
"ProductName": "Loki",
|
||||
"ProductVersion": "0.9.4",
|
||||
"Comments": "https://github.com/R0m1k3/Loki — projet open source (MIT)"
|
||||
},
|
||||
"VarFileInfo": {
|
||||
"Translation": {
|
||||
"LangID": "040C",
|
||||
"CharsetID": "04B0"
|
||||
}
|
||||
}
|
||||
}
|
||||
Whitespace-only changes.
+26
-29
@@ -1,45 +1,42 @@
|
||||
# ───────────────────────────────────────────────────────────────────────────
|
||||
# Loki — docker-compose pour Unraid (image préconstruite par GitHub)
|
||||
# Loki — docker-compose pour Unraid (image préconstruite par GitHub Actions)
|
||||
#
|
||||
# Aucun build ni git nécessaire sur Unraid : l'image est publiée automatiquement
|
||||
# sur GHCR par GitHub Actions à chaque push sur main.
|
||||
# Loki est un fork conteneurisé d'AJEAN (github.com/nathaninline/ajean, MIT) :
|
||||
# moteur llama.cpp CUDA + interface web dans UN conteneur autonome.
|
||||
# Aucun build ni git nécessaire sur Unraid — l'image est publiée sur GHCR à
|
||||
# chaque push sur main.
|
||||
#
|
||||
# MISE EN PLACE :
|
||||
# 1) Crée les dossiers de données (terminal Unraid) :
|
||||
# mkdir -p /mnt/user/appdata/loki/workspace /mnt/user/appdata/loki/data
|
||||
# 2) Plugin « Compose Manager » -> nouvelle stack -> colle ce fichier.
|
||||
# 3) Adapte OLLAMA_HOST avec l'IP de ton serveur Unraid (Ollama sur :11434).
|
||||
# 4) Compose Up. Interface : http://<ip-unraid>:8717
|
||||
# 1) Plugin « Nvidia Driver » (Apps) installé, GPU visible (nvidia-smi).
|
||||
# 2) Crée les dossiers de données (terminal Unraid) :
|
||||
# mkdir -p /mnt/user/appdata/loki/data /mnt/user/appdata/loki/models
|
||||
# 3) Plugin « Compose Manager » -> nouvelle stack -> colle ce fichier.
|
||||
# 4) Compose Up. Interface : http://<ip-unraid>:8090
|
||||
#
|
||||
# Mise à jour : Compose Down puis Up (l'option « pull image » récupère la
|
||||
# dernière version), ou « docker compose pull » avant Up.
|
||||
# Les modèles se téléchargent directement depuis l'UI (catalogue intégré),
|
||||
# ou dépose tes .gguf dans /mnt/user/appdata/loki/models.
|
||||
#
|
||||
# NB : si l'image ghcr.io/r0m1k3/loki est privée, rends-la publique une fois
|
||||
# (GitHub -> Packages -> loki -> Package settings -> Change visibility),
|
||||
# ou connecte Unraid à GHCR avec un token.
|
||||
# Mise à jour : Compose Down puis Up avec « pull image », ou
|
||||
# « docker compose pull » avant Up.
|
||||
# ───────────────────────────────────────────────────────────────────────────
|
||||
services:
|
||||
loki:
|
||||
image: ghcr.io/r0m1k3/loki:latest
|
||||
container_name: loki
|
||||
runtime: nvidia
|
||||
ports:
|
||||
# Un seul numéro de port, identique dedans/dehors (évite tout décalage).
|
||||
# Pour changer de port : modifie 8717 AUX DEUX endroits + PORT ci-dessous.
|
||||
- "8717:8717"
|
||||
# Pour changer de port : modifie 8090 à gauche ET LOKI_WEB_PORT… ou
|
||||
# seulement à gauche (ex. "8717:8090") si tu veux garder l'ancien port.
|
||||
- "8090:8090"
|
||||
environment:
|
||||
# >>> À ADAPTER : IP de ton serveur Unraid (Ollama écoute sur :11434) <<<
|
||||
- OLLAMA_HOST=http://192.168.1.10:11434
|
||||
- DEFAULT_MODEL=gemma4:12b
|
||||
- WORKSPACE_DIR=/workspace
|
||||
- DATA_DIR=/data
|
||||
- PORT=8717
|
||||
# Optionnel : instance SearxNG pour l'outil web_search (sinon DuckDuckGo)
|
||||
- SEARX_URL=
|
||||
- NVIDIA_VISIBLE_DEVICES=all
|
||||
# Optionnel : modèle initial (sinon, choisis-le dans l'UI au premier
|
||||
# lancement). Ex. LOKI_MODEL=Qwen3-14B-Q4_K_M.gguf après dépôt du
|
||||
# fichier dans /mnt/user/appdata/loki/models.
|
||||
- LOKI_MODEL=
|
||||
volumes:
|
||||
- /mnt/user/appdata/loki/workspace:/workspace
|
||||
# /data : config, base, mémoire, workspace, modèles téléchargés par l'UI
|
||||
- /mnt/user/appdata/loki/data:/data
|
||||
extra_hosts:
|
||||
# Permet aussi OLLAMA_HOST=http://host.docker.internal:11434
|
||||
- "host.docker.internal:host-gateway"
|
||||
# Le HEALTHCHECK est défini dans l'image (utilise automatiquement le bon port).
|
||||
# /models : dossier de .gguf déposés à la main (gros fichiers)
|
||||
- /mnt/user/appdata/loki/models:/models
|
||||
restart: unless-stopped
|
||||
+23
-40
@@ -1,47 +1,33 @@
|
||||
# Loki — fork conteneurisé d'AJEAN (https://github.com/nathaninline/ajean, MIT)
|
||||
# Un seul service : le conteneur embarque le moteur llama.cpp (CUDA) ET l'UI.
|
||||
#
|
||||
# cp .env.example .env # ajuste si besoin
|
||||
# docker compose up --build # build local (30-45 min : compilation CUDA)
|
||||
#
|
||||
# Interface : http://localhost:8090
|
||||
# Pré-requis GPU : NVIDIA Container Toolkit sur l'hôte.
|
||||
services:
|
||||
loki:
|
||||
build: .
|
||||
build:
|
||||
context: .
|
||||
args:
|
||||
# Restreins à ta carte pour un build plus court : 75=Turing (GTX 16xx/RTX 20xx),
|
||||
# 86=Ampere (RTX 30xx), 89=Ada (RTX 40xx).
|
||||
CUDA_ARCHS: ${CUDA_ARCHS:-75;86;89}
|
||||
container_name: loki
|
||||
ports:
|
||||
# Un seul numéro de port, identique dedans/dehors.
|
||||
- "${PORT:-8717}:${PORT:-8717}"
|
||||
- "${LOKI_WEB_PORT:-8090}:${LOKI_WEB_PORT:-8090}"
|
||||
environment:
|
||||
- OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434}
|
||||
- DEFAULT_MODEL=${DEFAULT_MODEL:-gemma4:12b}
|
||||
- WORKSPACE_DIR=/workspace
|
||||
- DATA_DIR=/data
|
||||
- PORT=${PORT:-8717}
|
||||
# Optionnel : instance SearxNG pour web_search (sinon DuckDuckGo)
|
||||
- SEARX_URL=${SEARX_URL:-}
|
||||
- LOKI_WEB_PORT=${LOKI_WEB_PORT:-8090}
|
||||
# Semé UNE fois au premier démarrage (modifiable ensuite dans l'UI) :
|
||||
- LOKI_MODEL=${LOKI_MODEL:-}
|
||||
- LOKI_CTX=${LOKI_CTX:-}
|
||||
- LOKI_NGL=${LOKI_NGL:-}
|
||||
volumes:
|
||||
- ./workspace:/workspace
|
||||
# /data = LOKI_HOME : config, base bbolt, mémoire, workspace, modèles téléchargés
|
||||
- ./data:/data
|
||||
extra_hosts:
|
||||
# Permet d'atteindre un Ollama installé sur la machine hôte
|
||||
- "host.docker.internal:host-gateway"
|
||||
# Le HEALTHCHECK est défini dans l'image (utilise automatiquement $PORT).
|
||||
restart: unless-stopped
|
||||
|
||||
# ── Ollama optionnel ──────────────────────────────────────────────────
|
||||
# Par défaut, Loki se connecte à un Ollama déjà présent sur l'hôte.
|
||||
# Pour embarquer Ollama dans la stack, lance : docker compose --profile ollama up
|
||||
# et règle OLLAMA_HOST=http://ollama:11434
|
||||
ollama:
|
||||
image: ollama/ollama:latest
|
||||
container_name: loki-ollama
|
||||
profiles: ["ollama"]
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=all
|
||||
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
- OLLAMA_FLASH_ATTENTION=1
|
||||
- OLLAMA_KV_CACHE_TYPE=q8_0
|
||||
ports:
|
||||
- "11434:11434"
|
||||
volumes:
|
||||
- ollama-models:/root/.ollama
|
||||
# Accès GPU NVIDIA : sans ceci, l'Ollama embarqué tourne en CPU (très lent).
|
||||
# Nécessite le NVIDIA Container Toolkit installé sur l'hôte. Si tu n'as pas
|
||||
# de GPU NVIDIA, commente tout le bloc `deploy:` ci-dessous.
|
||||
# /models : dossier(s) de modèles GGUF supplémentaires (LOKI_MODEL_DIRS)
|
||||
- ./models:/models
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
@@ -50,6 +36,3 @@ services:
|
||||
count: all
|
||||
capabilities: ["gpu"]
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
ollama-models:
|
||||
@@ -0,0 +1,34 @@
|
||||
#!/bin/sh
|
||||
# Entrypoint du conteneur Loki : sème la configuration au premier démarrage,
|
||||
# lance le moteur (supervision PID, sans systemd), puis l'UI au premier plan.
|
||||
set -eu
|
||||
|
||||
: "${LOKI_HOME:=/data}"
|
||||
mkdir -p "$LOKI_HOME"
|
||||
|
||||
# Le chemin du moteur est imposé par l'image : on le (re)pose à chaque boot,
|
||||
# une mise à jour de l'image ne doit pas laisser un BIN obsolète en base.
|
||||
loki config set "BIN=/opt/llama.cpp/llama-server"
|
||||
|
||||
# Les autres clés ne sont semées QUE si absentes : ce que l'utilisateur règle
|
||||
# ensuite dans l'UI (modèle, contexte…) est conservé d'un redémarrage à l'autre.
|
||||
seed() { # seed CLÉ VALEUR
|
||||
[ -n "$2" ] || return 0
|
||||
cur=$(loki config get "$1")
|
||||
[ -n "$cur" ] || loki config set "$1=$2"
|
||||
}
|
||||
seed HOST "${LOKI_ENGINE_HOST:-127.0.0.1}"
|
||||
seed PORT "${LOKI_ENGINE_PORT:-8080}"
|
||||
seed MODEL "${LOKI_MODEL:-}"
|
||||
seed CTX "${LOKI_CTX:-}"
|
||||
seed NGL "${LOKI_NGL:-}"
|
||||
|
||||
# Moteur : démarrage best-effort. Sans MODEL configuré, on laisse l'UI monter
|
||||
# quand même — le moteur partira au premier modèle choisi/téléchargé dans l'UI.
|
||||
if [ -n "$(loki config get MODEL)" ]; then
|
||||
loki start || echo "[entrypoint] moteur non démarré (voir loki logs) — l'UI reste accessible"
|
||||
else
|
||||
echo "[entrypoint] aucun MODEL configuré — télécharge un modèle depuis l'UI pour démarrer le moteur"
|
||||
fi
|
||||
|
||||
exec loki web "${LOKI_WEB_PORT:-8090}"
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,954 +0,0 @@
|
||||
# Projets + aperçu réductible — Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** l'agent travaille dans des projets (sous-dossiers du workspace) choisis par session depuis le composer, et le panneau d'aperçu se replie.
|
||||
|
||||
**Architecture:** une contextvar dans `tools.py` re-racine `_safe_path`/`_workspace_root` sur `workspace/<projet>` pour toute la requête (chat, outils, shell, Aider) ; les routes fichiers/git prennent un paramètre `project` optionnel et posent la même contextvar. La session porte son projet (colonne SQLite). Frontend : chip 📁 dans le composer, store re-racine les appels fichiers.
|
||||
|
||||
**Tech Stack:** FastAPI, SQLite (migrations douces ALTER TABLE), contextvars, React/zustand.
|
||||
|
||||
**Spec:** `docs/superpowers/specs/2026-07-19-projets-apercu-design.md`
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Nom de projet : `^[a-z0-9][a-z0-9_-]{0,40}$` — sinon 400.
|
||||
- Session sans projet (NULL) = racine du workspace, comportement actuel intact.
|
||||
- Confinement `_safe_path` inchangé dans sa logique (re-raciné seulement).
|
||||
- Un dépôt git PAR projet (`ensure_git` à la création et au premier usage git).
|
||||
- Projet disparu du disque → chat retombe sur la racine + notice SSE.
|
||||
- UI française, styles existants. Tests : `python -m pytest` depuis `backend/`.
|
||||
- Après chaque tâche frontend : `cd frontend && npx tsc -b` passe.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Racine active (contextvar) + colonne project + routes sessions
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/app/tools.py` (contextvar + validation nom)
|
||||
- Modify: `backend/app/db.py` (migration + create/set project)
|
||||
- Modify: `backend/app/routes/sessions.py` (create + PATCH avec project)
|
||||
- Test: `backend/tests/test_projects.py`
|
||||
|
||||
**Interfaces:**
|
||||
- Produces: `tools.set_project(name: str | None) -> None` (valide le nom, pose la contextvar ; `ToolError` si nom invalide) ; `tools.active_root() -> str` (racine effective, créée si absente) ; `tools.PROJECT_NAME` (regex compilée) ; `db.create_session(title, model, project=None)` ; `db.set_session_project(sid, project: str | None)` ; PATCH `/api/sessions/{sid}` accepte `{title?, project?}` (`project: ""` = retour racine).
|
||||
|
||||
- [ ] **Step 1: Écrire les tests qui échouent**
|
||||
|
||||
`backend/tests/test_projects.py` :
|
||||
|
||||
```python
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from app import db, tools # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
db.init_db()
|
||||
_ROOT = os.path.abspath(settings.workspace_dir)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_project():
|
||||
tools.set_project(None)
|
||||
yield
|
||||
tools.set_project(None)
|
||||
|
||||
|
||||
def test_racine_par_defaut():
|
||||
assert tools.active_root() == _ROOT
|
||||
|
||||
|
||||
def test_set_project_reracine():
|
||||
tools.set_project("demo")
|
||||
root = tools.active_root()
|
||||
assert root == os.path.join(_ROOT, "demo")
|
||||
assert os.path.isdir(root) # créée à la volée
|
||||
|
||||
|
||||
def test_confinement_conserve():
|
||||
tools.set_project("demo")
|
||||
with pytest.raises(tools.ToolError):
|
||||
tools._safe_path("../hors-projet")
|
||||
|
||||
|
||||
def test_nom_projet_invalide():
|
||||
for bad in ("../x", "UPPER", "a b", "", "x" * 50):
|
||||
with pytest.raises(tools.ToolError):
|
||||
tools.set_project(bad)
|
||||
|
||||
|
||||
def test_session_porte_son_projet():
|
||||
s = db.create_session("t", None, project="demo")
|
||||
assert db.get_session(s["id"])["project"] == "demo"
|
||||
db.set_session_project(s["id"], None)
|
||||
assert db.get_session(s["id"])["project"] is None
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Vérifier l'échec**
|
||||
|
||||
Run: `cd backend && python -m pytest tests/test_projects.py -q`
|
||||
Expected: FAIL — `AttributeError: module 'app.tools' has no attribute 'set_project'`
|
||||
|
||||
- [ ] **Step 3: tools.py — contextvar**
|
||||
|
||||
En tête de `backend/app/tools.py` (après l'import `settings`) :
|
||||
|
||||
```python
|
||||
from contextvars import ContextVar
|
||||
|
||||
# Projet actif pour la requête en cours : re-racine tous les outils sur
|
||||
# workspace/<projet>. None = racine du workspace (comportement historique).
|
||||
_ACTIVE_PROJECT: ContextVar[str | None] = ContextVar("loki_project", default=None)
|
||||
|
||||
PROJECT_NAME = re.compile(r"^[a-z0-9][a-z0-9_-]{0,40}$")
|
||||
|
||||
|
||||
def set_project(name: str | None) -> None:
|
||||
"""Fixe le projet actif de la requête (None = racine)."""
|
||||
if name is not None and not PROJECT_NAME.match(name):
|
||||
raise ToolError(f"nom de projet invalide : {name!r}")
|
||||
_ACTIVE_PROJECT.set(name)
|
||||
|
||||
|
||||
def active_root() -> str:
|
||||
"""Racine effective (workspace ou projet), créée si nécessaire."""
|
||||
return _workspace_root()
|
||||
```
|
||||
|
||||
Remplacer `_workspace_root` :
|
||||
|
||||
```python
|
||||
def _workspace_root() -> str:
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
project = _ACTIVE_PROJECT.get()
|
||||
if project:
|
||||
root = os.path.join(root, project)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
return root
|
||||
```
|
||||
|
||||
(`_safe_path` ne change pas : il s'appuie sur `_workspace_root()`.)
|
||||
|
||||
- [ ] **Step 4: db.py — colonne + accesseurs**
|
||||
|
||||
Dans `init_db`, à côté de la migration `summary` :
|
||||
|
||||
```python
|
||||
if "project" not in scols:
|
||||
conn.execute("ALTER TABLE sessions ADD COLUMN project TEXT")
|
||||
```
|
||||
|
||||
`create_session` — ajouter le paramètre et la colonne (adapter l'INSERT
|
||||
existant) :
|
||||
|
||||
```python
|
||||
def create_session(title: str, model: str | None, project: str | None = None) -> dict:
|
||||
```
|
||||
|
||||
et inclure `project` dans l'INSERT et le dict renvoyé. Ajouter :
|
||||
|
||||
```python
|
||||
def set_session_project(sid: str, project: str | None) -> None:
|
||||
with _LOCK, _connect() as conn:
|
||||
conn.execute(
|
||||
"UPDATE sessions SET project = ? WHERE id = ?", (project, sid)
|
||||
)
|
||||
```
|
||||
|
||||
Vérifier que `list_sessions`/`get_session` renvoient la colonne (elles font
|
||||
`dict(row)` — la colonne suit automatiquement).
|
||||
|
||||
- [ ] **Step 5: sessions.py — create + PATCH**
|
||||
|
||||
```python
|
||||
class CreateSession(BaseModel):
|
||||
title: str = "Nouvelle session"
|
||||
model: str | None = None
|
||||
project: str | None = None
|
||||
|
||||
|
||||
class UpdateSession(BaseModel):
|
||||
title: str | None = None
|
||||
project: str | None = None # "" = retour à la racine du workspace
|
||||
```
|
||||
|
||||
```python
|
||||
@router.post("")
|
||||
async def post_session(req: CreateSession) -> dict:
|
||||
if req.project and not tools.PROJECT_NAME.match(req.project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
return db.create_session(req.title, req.model, req.project or None)
|
||||
|
||||
|
||||
@router.patch("/{sid}")
|
||||
async def patch_session(sid: str, req: UpdateSession) -> dict:
|
||||
if not db.get_session(sid):
|
||||
raise HTTPException(404, "session introuvable")
|
||||
if req.title is not None:
|
||||
db.rename_session(sid, req.title)
|
||||
if req.project is not None:
|
||||
project = req.project or None
|
||||
if project and not tools.PROJECT_NAME.match(project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
db.set_session_project(sid, project)
|
||||
return {"ok": True}
|
||||
```
|
||||
|
||||
(import : `from .. import db, tools`)
|
||||
|
||||
- [ ] **Step 6: Vérifier + commit**
|
||||
|
||||
Run: `cd backend && python -m pytest tests/ -q` — tout PASS.
|
||||
|
||||
```bash
|
||||
git add backend/app/tools.py backend/app/db.py backend/app/routes/sessions.py backend/tests/test_projects.py
|
||||
git commit -m "feat(projets): racine active par contextvar + projet par session"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 2: Routes /api/projects + files/git re-racinés
|
||||
|
||||
**Files:**
|
||||
- Create: `backend/app/routes/projects.py`
|
||||
- Modify: `backend/app/routes/files.py` (param `project` sur les 4 routes)
|
||||
- Modify: `backend/app/routes/git.py` (param `project` sur les 3 routes)
|
||||
- Modify: `backend/app/main.py` (include projects.router)
|
||||
- Test: `backend/tests/test_projects_routes.py`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `tools.set_project`, `tools.active_root`, `tools.PROJECT_NAME`, `coder.ensure_git` (existant).
|
||||
- Produces: `GET /api/projects` → `{projects: [{name, files}], root_files}` ; `POST /api/projects {name}` → `{name}` (201) ; toutes les routes files/git acceptent `?project=`.
|
||||
|
||||
- [ ] **Step 1: Tests qui échouent**
|
||||
|
||||
`backend/tests/test_projects_routes.py` :
|
||||
|
||||
```python
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
os.environ.setdefault("DATA_DIR", tempfile.mkdtemp())
|
||||
os.environ.setdefault("WORKSPACE_DIR", tempfile.mkdtemp())
|
||||
|
||||
from fastapi.testclient import TestClient # noqa: E402
|
||||
|
||||
from app import tools # noqa: E402
|
||||
from app.main import app # noqa: E402
|
||||
from app.config import settings # noqa: E402
|
||||
|
||||
client = TestClient(app)
|
||||
_ROOT = os.path.abspath(settings.workspace_dir)
|
||||
|
||||
|
||||
def test_creation_et_liste():
|
||||
r = client.post("/api/projects", json={"name": "demo"})
|
||||
assert r.status_code == 201
|
||||
assert os.path.isdir(os.path.join(_ROOT, "demo", ".git"))
|
||||
names = [p["name"] for p in client.get("/api/projects").json()["projects"]]
|
||||
assert "demo" in names
|
||||
|
||||
|
||||
def test_nom_invalide_400():
|
||||
assert client.post("/api/projects", json={"name": "../x"}).status_code == 400
|
||||
assert client.post("/api/projects", json={"name": "Demo"}).status_code == 400
|
||||
|
||||
|
||||
def test_existant_400():
|
||||
client.post("/api/projects", json={"name": "dup"})
|
||||
assert client.post("/api/projects", json={"name": "dup"}).status_code == 400
|
||||
|
||||
|
||||
def test_files_re_racine():
|
||||
client.post("/api/projects", json={"name": "scoped"})
|
||||
with open(os.path.join(_ROOT, "scoped", "a.txt"), "w") as f:
|
||||
f.write("x")
|
||||
with open(os.path.join(_ROOT, "racine.txt"), "w") as f:
|
||||
f.write("y")
|
||||
tree = client.get("/api/files", params={"project": "scoped"}).json()["tree"]
|
||||
names = [n["name"] for n in tree]
|
||||
assert "a.txt" in names and "racine.txt" not in names
|
||||
|
||||
|
||||
def test_files_projet_invalide_400():
|
||||
assert client.get("/api/files", params={"project": "../x"}).status_code == 400
|
||||
```
|
||||
|
||||
Run: `cd backend && python -m pytest tests/test_projects_routes.py -q`
|
||||
Expected: FAIL (404 sur /api/projects).
|
||||
|
||||
Note : `TestClient` requiert `httpx` (déjà présent). Si `starlette` réclame
|
||||
un extra, `python -m pip install "httpx<1"` est déjà satisfait.
|
||||
|
||||
- [ ] **Step 2: routes/projects.py**
|
||||
|
||||
```python
|
||||
"""Routes des projets : sous-dossiers de premier niveau du workspace."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from .. import coder, tools
|
||||
from ..config import settings
|
||||
|
||||
router = APIRouter(prefix="/api/projects", tags=["projects"])
|
||||
|
||||
|
||||
def _root() -> str:
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
return root
|
||||
|
||||
|
||||
def _count_files(path: str) -> int:
|
||||
total = 0
|
||||
for dirpath, dirnames, filenames in os.walk(path):
|
||||
dirnames[:] = [d for d in dirnames if not d.startswith(".")]
|
||||
total += sum(1 for f in filenames if not f.startswith("."))
|
||||
return total
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_projects() -> dict:
|
||||
root = _root()
|
||||
projects = []
|
||||
root_files = 0
|
||||
for name in sorted(os.listdir(root)):
|
||||
full = os.path.join(root, name)
|
||||
if name.startswith("."):
|
||||
continue
|
||||
if os.path.isdir(full):
|
||||
projects.append({"name": name, "files": _count_files(full)})
|
||||
else:
|
||||
root_files += 1
|
||||
return {"projects": projects, "root_files": root_files}
|
||||
|
||||
|
||||
class CreateProject(BaseModel):
|
||||
name: str
|
||||
|
||||
|
||||
@router.post("", status_code=201)
|
||||
async def create_project(req: CreateProject) -> dict:
|
||||
name = req.name.strip()
|
||||
if not tools.PROJECT_NAME.match(name):
|
||||
raise HTTPException(400, "nom de projet invalide (a-z, 0-9, - et _)")
|
||||
target = os.path.join(_root(), name)
|
||||
if os.path.exists(target):
|
||||
raise HTTPException(400, "ce projet existe déjà")
|
||||
os.makedirs(target)
|
||||
# Dépôt git par projet : commits Aider + onglet Git propres au projet.
|
||||
coder.ensure_git(target)
|
||||
return {"name": name}
|
||||
```
|
||||
|
||||
`main.py` : ajouter `projects` à l'import des routes et
|
||||
`app.include_router(projects.router)`.
|
||||
|
||||
- [ ] **Step 3: files.py — param project**
|
||||
|
||||
Chaque route gagne `project: str | None = None` et commence par :
|
||||
|
||||
```python
|
||||
try:
|
||||
tools_mod.set_project(project or None)
|
||||
except ToolError as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
```
|
||||
|
||||
avec en tête du fichier `from .. import tools as tools_mod` (l'import existant
|
||||
`from ..tools import ToolError, _safe_path` reste). Puis remplacer chaque
|
||||
usage de `os.path.abspath(settings.workspace_dir)` par `tools_mod.active_root()` :
|
||||
- `list_files` : `root = tools_mod.active_root()` (le `os.makedirs` est déjà
|
||||
fait par `active_root`).
|
||||
- `_tree` : signature `_tree(path: str, root: str)` — le `rel` se calcule
|
||||
contre `root` passé en argument (adapter l'appel récursif et l'appel
|
||||
depuis `list_files`).
|
||||
- `delete_file` : le refus de racine devient
|
||||
`if os.path.abspath(target) == tools_mod.active_root():`.
|
||||
- `file_content` / `download_file` : seulement `set_project` en tête
|
||||
(`_safe_path` suit la contextvar).
|
||||
|
||||
IMPORTANT : remettre `tools_mod.set_project(None)` n'est pas nécessaire —
|
||||
la contextvar est par-tâche asyncio, chaque requête FastAPI a son contexte.
|
||||
|
||||
- [ ] **Step 4: git.py — param project**
|
||||
|
||||
`_git` gagne `project: str | None = None` :
|
||||
|
||||
```python
|
||||
def _git(*args: str, timeout: int = 15, project: str | None = None) -> subprocess.CompletedProcess:
|
||||
root = os.path.abspath(settings.workspace_dir)
|
||||
if project:
|
||||
if not tools.PROJECT_NAME.match(project):
|
||||
raise HTTPException(400, "nom de projet invalide")
|
||||
root = os.path.join(root, project)
|
||||
ensure_git(root)
|
||||
return subprocess.run(
|
||||
["git", *args], cwd=root, ...
|
||||
)
|
||||
```
|
||||
|
||||
(reprendre les kwargs existants de l'appel `subprocess.run` du fichier ;
|
||||
imports : `from .. import tools`, `from fastapi import HTTPException` déjà
|
||||
présent ou à ajouter, `import os`). Les trois routes (`git_log`, `git_diff`,
|
||||
`git_revert`) gagnent `project: str | None = None` et le propagent à chaque
|
||||
appel `_git(..., project=project)`.
|
||||
|
||||
- [ ] **Step 5: Vérifier + commit**
|
||||
|
||||
Run: `cd backend && python -m pytest tests/ -q` — tout PASS.
|
||||
|
||||
```bash
|
||||
git add backend/app/routes/projects.py backend/app/routes/files.py backend/app/routes/git.py backend/app/main.py backend/tests/test_projects_routes.py
|
||||
git commit -m "feat(projets): routes /api/projects + files et git re-racinés"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Chat re-raciné (contexte, Aider, notice projet disparu)
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/app/routes/chat.py`
|
||||
- Modify: `backend/app/coder.py` (`run_code_task` paramètre `root`)
|
||||
- Modify: `backend/app/agent.py` (dispatch `code_task` passe la racine)
|
||||
- Test: `backend/tests/test_projects.py` (ajout)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `tools.set_project`, `tools.active_root` (Task 1).
|
||||
- Produces: `coder.run_code_task(instruction, model, files=None, root: str | None = None)`.
|
||||
|
||||
- [ ] **Step 1: Test qui échoue (aides de contexte re-racinées)**
|
||||
|
||||
Ajouter à `backend/tests/test_projects.py` :
|
||||
|
||||
```python
|
||||
def test_aides_contexte_suivent_le_projet():
|
||||
from app.routes.chat import _mentioned_files, _workspace_listing
|
||||
tools.set_project("ctxdemo")
|
||||
root = tools.active_root()
|
||||
with open(os.path.join(root, "app.py"), "w", encoding="utf-8") as f:
|
||||
f.write("x = 1")
|
||||
assert "app.py" in _workspace_listing()
|
||||
assert _mentioned_files("corrige app.py") == ["app.py"]
|
||||
tools.set_project(None)
|
||||
assert _mentioned_files("corrige app.py") == [] # absent de la racine
|
||||
```
|
||||
|
||||
Run: `cd backend && python -m pytest tests/test_projects.py -q`
|
||||
Expected: FAIL — `_workspace_listing`/`_mentioned_files` lisent encore
|
||||
`settings.workspace_dir`.
|
||||
|
||||
- [ ] **Step 2: chat.py — racine active partout**
|
||||
|
||||
Dans `chat()` juste après la récupération de la session :
|
||||
|
||||
```python
|
||||
# Projet de la session : re-racine outils, shell, Aider et aides de
|
||||
# contexte pour tout le tour. Projet disparu -> retour racine + notice.
|
||||
project = session.get("project") or None
|
||||
project_missing = False
|
||||
if project:
|
||||
proj_dir = os.path.join(os.path.abspath(settings.workspace_dir), project)
|
||||
if not os.path.isdir(proj_dir):
|
||||
project_missing, project = True, None
|
||||
tools.set_project(project)
|
||||
```
|
||||
|
||||
avec `from .. import tools` dans les imports du fichier. Dans
|
||||
`event_stream()`, après le `yield _sse("start", …)` :
|
||||
|
||||
```python
|
||||
if project_missing:
|
||||
yield _sse("notice", {"message":
|
||||
"Projet de la session introuvable sur le disque — retour au workspace."})
|
||||
```
|
||||
|
||||
Remplacer dans `_session_code_context`, `_workspace_listing` et
|
||||
`_mentioned_files` chaque `root = os.path.abspath(settings.workspace_dir)`
|
||||
par `root = tools.active_root()`.
|
||||
|
||||
Chemin code : passer la racine à Aider —
|
||||
|
||||
```python
|
||||
async for chunk in _code_stream(
|
||||
req, code_model, extra=extra, plan=plan, files=code_files,
|
||||
):
|
||||
```
|
||||
|
||||
`_code_stream` → `_run_aider_keepalive(instruction, model, files)` →
|
||||
|
||||
```python
|
||||
task = asyncio.create_task(
|
||||
asyncio.to_thread(
|
||||
coder.run_code_task, instruction, model, files, tools.active_root()
|
||||
)
|
||||
)
|
||||
```
|
||||
|
||||
(`active_root()` évalué AVANT le to_thread : la contextvar ne suit pas dans
|
||||
le thread.)
|
||||
|
||||
- [ ] **Step 3: coder.py — paramètre root**
|
||||
|
||||
```python
|
||||
def run_code_task(
|
||||
instruction: str,
|
||||
model: str,
|
||||
files: list[str] | None = None,
|
||||
root: str | None = None,
|
||||
) -> dict:
|
||||
```
|
||||
|
||||
et remplacer `root = os.path.abspath(settings.workspace_dir)` par
|
||||
`root = os.path.abspath(root or settings.workspace_dir)` (le reste — ensure_git,
|
||||
chdir, fnames — utilise déjà `root`).
|
||||
|
||||
- [ ] **Step 4: agent.py — code_task scoped**
|
||||
|
||||
Dans le dispatch `code_task` de `run_agent`, remplacer l'appel :
|
||||
|
||||
```python
|
||||
if name == "code_task":
|
||||
from .tools import active_root
|
||||
code_model = await coder.pick_code_model(model)
|
||||
result = await asyncio.to_thread(
|
||||
coder.run_code_task,
|
||||
args.get("instruction", ""),
|
||||
code_model,
|
||||
args.get("files") or [],
|
||||
active_root(),
|
||||
)
|
||||
```
|
||||
|
||||
- [ ] **Step 5: Vérifier + commit**
|
||||
|
||||
Run: `cd backend && python -m pytest tests/ -q` — tout PASS ;
|
||||
`python -m compileall -q app`.
|
||||
|
||||
```bash
|
||||
git add backend/app/routes/chat.py backend/app/coder.py backend/app/agent.py backend/tests/test_projects.py
|
||||
git commit -m "feat(projets): chat, aides de contexte et Aider suivent le projet de la session"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 4: Frontend — API client + store projets
|
||||
|
||||
**Files:**
|
||||
- Modify: `frontend/src/api/client.ts`
|
||||
- Modify: `frontend/src/store/useStore.ts`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: routes Tasks 1-2.
|
||||
- Produces: `listProjects(): Promise<{projects: {name: string; files: number}[]; root_files: number}>` ; `createProject(name): Promise<void>` ; `setSessionProject(id, project: string | null): Promise<void>` ; store : `currentProject(): string | null` (getter), `setProject(name: string | null)`, `projects: {name: string; files: number}[]`, `refreshProjects()`.
|
||||
|
||||
- [ ] **Step 1: client.ts**
|
||||
|
||||
Type `Session` : ajouter `project?: string | null;`. Ajouter :
|
||||
|
||||
```typescript
|
||||
export async function listProjects(): Promise<{
|
||||
projects: { name: string; files: number }[];
|
||||
root_files: number;
|
||||
}> {
|
||||
const res = await fetch("/api/projects");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function createProject(name: string): Promise<void> {
|
||||
const res = await fetch("/api/projects", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "création du projet impossible");
|
||||
}
|
||||
|
||||
export async function setSessionProject(
|
||||
id: string,
|
||||
project: string | null
|
||||
): Promise<void> {
|
||||
const res = await fetch(`/api/sessions/${id}`, {
|
||||
method: "PATCH",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ project: project ?? "" }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "changement de projet impossible");
|
||||
}
|
||||
```
|
||||
|
||||
Re-raciner les appels fichiers — signatures :
|
||||
|
||||
```typescript
|
||||
const projQuery = (project?: string | null) =>
|
||||
project ? `&project=${encodeURIComponent(project)}` : "";
|
||||
|
||||
export async function listFiles(project?: string | null): Promise<FileNode[]> {
|
||||
const query = project ? `?project=${encodeURIComponent(project)}` : "";
|
||||
const res = await fetch(`/api/files${query}`);
|
||||
return (await res.json()).tree;
|
||||
}
|
||||
|
||||
export async function fileContent(
|
||||
path: string,
|
||||
project?: string | null
|
||||
): Promise<string> {
|
||||
const res = await fetch(
|
||||
`/api/files/content?path=${encodeURIComponent(path)}${projQuery(project)}`
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
return (await res.json()).content;
|
||||
}
|
||||
|
||||
export function downloadFile(path: string, project?: string | null): void {
|
||||
const link = document.createElement("a");
|
||||
link.href = `/api/files/download?path=${encodeURIComponent(path)}${projQuery(project)}`;
|
||||
link.download = path.split(/[\\/]/).pop() ?? "fichier";
|
||||
document.body.appendChild(link);
|
||||
link.click();
|
||||
link.remove();
|
||||
}
|
||||
|
||||
export async function deleteFile(
|
||||
path: string,
|
||||
project?: string | null
|
||||
): Promise<void> {
|
||||
const res = await fetch(
|
||||
`/api/files?path=${encodeURIComponent(path)}${projQuery(project)}`,
|
||||
{ method: "DELETE" }
|
||||
);
|
||||
if (!res.ok) {
|
||||
throw await apiError(res, `suppression impossible (${res.status})`);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Idem routes git utilisées par PreviewPanel (`getGitLog`, `getGitDiff`,
|
||||
`revertCommit`) : paramètre optionnel `project` ajouté en query string de la
|
||||
même façon.
|
||||
|
||||
`createSession` : paramètre `project?: string | null` transmis dans le body.
|
||||
|
||||
- [ ] **Step 2: useStore.ts**
|
||||
|
||||
État : `projects: { name: string; files: number }[]` (init `[]`). Actions :
|
||||
|
||||
```typescript
|
||||
refreshProjects: async () => {
|
||||
try {
|
||||
const { projects } = await listProjects();
|
||||
set({ projects });
|
||||
} catch { /* backend indisponible */ }
|
||||
},
|
||||
|
||||
currentProject: () => {
|
||||
const s = get().sessions.find((x) => x.id === get().currentSessionId);
|
||||
return s?.project ?? null;
|
||||
},
|
||||
|
||||
setProject: async (name) => {
|
||||
let sid = get().currentSessionId;
|
||||
if (!sid) {
|
||||
const s = await createSession(get().selectedModel || undefined, name);
|
||||
set({ currentSessionId: s.id, messages: [] });
|
||||
sid = s.id;
|
||||
} else {
|
||||
await setSessionProject(sid, name);
|
||||
}
|
||||
await get().refreshSessions();
|
||||
await get().refreshFiles();
|
||||
set({ previewPath: null, previewContent: "" });
|
||||
},
|
||||
```
|
||||
|
||||
(`createSession(model?, project?)` : étendre la fonction client existante.)
|
||||
`refreshFiles` passe le projet : `listFiles(get().currentProject())`.
|
||||
`openPreview` : `fileContent(path, get().currentProject())`.
|
||||
`removeFile` : `deleteFile(path, get().currentProject())`.
|
||||
`newSession` : `createSession(model, get().currentProject())` — héritage.
|
||||
`openSession` : après chargement, `await get().refreshFiles()` (l'arbre suit
|
||||
la session ouverte).
|
||||
Les composants qui appellent `downloadFile(path)` passent aussi
|
||||
`useStore.getState().currentProject()` (LeftPanel, ActivityViews,
|
||||
PreviewPanel).
|
||||
Types : ajouter `projects`, `refreshProjects`, `currentProject`, `setProject`
|
||||
à `LokiState`.
|
||||
|
||||
- [ ] **Step 3: Vérifier + commit**
|
||||
|
||||
Run: `cd frontend && npx tsc -b` — OK (les composants utilisant downloadFile
|
||||
sont ajustés dans cette tâche pour compiler).
|
||||
|
||||
```bash
|
||||
git add frontend/src/api/client.ts frontend/src/store/useStore.ts frontend/src/panels/LeftPanel.tsx frontend/src/panels/ActivityViews.tsx frontend/src/panels/PreviewPanel.tsx
|
||||
git commit -m "feat(projets): client API + store re-racinés par projet"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 5: Chip projet dans le composer + aperçu réductible
|
||||
|
||||
**Files:**
|
||||
- Create: `frontend/src/components/ProjectChip.tsx`
|
||||
- Modify: `frontend/src/panels/ChatPanel.tsx` (rendu du chip à côté du ModeSelector)
|
||||
- Modify: `frontend/src/panels/PreviewPanel.tsx` (repli)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: store Task 4 (`projects`, `refreshProjects`, `currentProject`, `setProject`).
|
||||
|
||||
- [ ] **Step 1: ProjectChip.tsx**
|
||||
|
||||
```tsx
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { createProject } from "../api/client";
|
||||
|
||||
/** Sélecteur de projet du composer : 📁 <projet> + menu (liste, création). */
|
||||
export function ProjectChip() {
|
||||
const { projects, refreshProjects, currentProject, setProject } = useStore();
|
||||
const [open, setOpen] = useState(false);
|
||||
const [creating, setCreating] = useState(false);
|
||||
const [draft, setDraft] = useState("");
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const rootRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
const active = currentProject();
|
||||
|
||||
useEffect(() => {
|
||||
if (open) void refreshProjects();
|
||||
}, [open, refreshProjects]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
const onDown = (e: MouseEvent) => {
|
||||
if (rootRef.current && !rootRef.current.contains(e.target as Node)) {
|
||||
setOpen(false);
|
||||
setCreating(false);
|
||||
setError(null);
|
||||
}
|
||||
};
|
||||
document.addEventListener("mousedown", onDown);
|
||||
return () => document.removeEventListener("mousedown", onDown);
|
||||
}, [open]);
|
||||
|
||||
const choose = async (name: string | null) => {
|
||||
setOpen(false);
|
||||
await setProject(name);
|
||||
};
|
||||
|
||||
const create = async () => {
|
||||
const name = draft.trim();
|
||||
if (!name) return;
|
||||
try {
|
||||
await createProject(name);
|
||||
setCreating(false);
|
||||
setDraft("");
|
||||
await choose(name);
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "création impossible");
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div ref={rootRef} className="relative">
|
||||
<button
|
||||
onClick={() => setOpen((o) => !o)}
|
||||
className="flex h-8 items-center gap-1.5 border-2 border-line bg-base px-2.5 text-[12px] text-ink-2"
|
||||
title="Projet de travail de cette session"
|
||||
>
|
||||
📁 <span className="max-w-[140px] truncate">{active ?? "workspace"}</span>
|
||||
</button>
|
||||
|
||||
{open && (
|
||||
<div
|
||||
className="absolute bottom-[calc(100%+6px)] left-0 z-30 w-[240px] border-[3px] border-line bg-card shadow-hard"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
<button
|
||||
onClick={() => choose(null)}
|
||||
className={`block w-full px-3 py-2 text-left text-[13px] text-ink-2 hover:bg-base ${
|
||||
active === null ? "bg-base font-bold" : ""
|
||||
}`}
|
||||
>
|
||||
workspace (racine)
|
||||
</button>
|
||||
{projects.map((p) => (
|
||||
<button
|
||||
key={p.name}
|
||||
onClick={() => choose(p.name)}
|
||||
className={`block w-full border-t-2 border-line-soft px-3 py-2 text-left text-[13px] text-ink-2 hover:bg-base ${
|
||||
active === p.name ? "bg-base font-bold" : ""
|
||||
}`}
|
||||
>
|
||||
📁 {p.name}
|
||||
<span className="ml-1.5 text-[11px] text-muted-2">
|
||||
{p.files} fichier{p.files > 1 ? "s" : ""}
|
||||
</span>
|
||||
</button>
|
||||
))}
|
||||
<div className="border-t-2 border-line-soft p-2">
|
||||
{creating ? (
|
||||
<input
|
||||
autoFocus
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === "Enter") void create();
|
||||
if (e.key === "Escape") {
|
||||
setCreating(false);
|
||||
setError(null);
|
||||
}
|
||||
}}
|
||||
placeholder="nom-du-projet"
|
||||
className="w-full border-2 border-line bg-base px-2 py-1 text-[12px] text-ink outline-none"
|
||||
/>
|
||||
) : (
|
||||
<button
|
||||
onClick={() => setCreating(true)}
|
||||
className="w-full text-left text-[13px] font-bold text-accent"
|
||||
>
|
||||
+ Nouveau projet
|
||||
</button>
|
||||
)}
|
||||
{error && <div className="mt-1 text-[11px] text-warn">{error}</div>}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: ChatPanel — rendu**
|
||||
|
||||
Localiser le rendu du sélecteur de mode dans le composer (composant local
|
||||
`ModeSelector`, utilisé vers le bas du composer). Ajouter `<ProjectChip />`
|
||||
juste à côté (même conteneur flex), avec
|
||||
`import { ProjectChip } from "../components/ProjectChip";`.
|
||||
|
||||
- [ ] **Step 3: PreviewPanel — repli**
|
||||
|
||||
Dans `PreviewPanel` :
|
||||
|
||||
```tsx
|
||||
const [collapsed, setCollapsed] = useState(
|
||||
() => window.localStorage.getItem("loki.preview.collapsed") === "1"
|
||||
);
|
||||
const toggleCollapsed = () => {
|
||||
setCollapsed((c) => {
|
||||
window.localStorage.setItem("loki.preview.collapsed", c ? "0" : "1");
|
||||
return !c;
|
||||
});
|
||||
};
|
||||
```
|
||||
|
||||
Rendu replié — AVANT le rendu normal :
|
||||
|
||||
```tsx
|
||||
if (collapsed) {
|
||||
return (
|
||||
<div className="flex w-9 flex-none flex-col items-center border-l-[3px] border-line bg-panel pt-3">
|
||||
<button
|
||||
onClick={toggleCollapsed}
|
||||
title="Déplier l'aperçu"
|
||||
className="text-[15px] text-muted-2 hover:text-accent"
|
||||
>
|
||||
⇤
|
||||
</button>
|
||||
<div className="mt-3 rotate-90 whitespace-nowrap text-[10px] font-bold tracking-wide text-muted-3">
|
||||
APERÇU
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
Bouton de repli dans la barre d'onglets existante (à droite des Tabs) :
|
||||
|
||||
```tsx
|
||||
<div className="flex-1" />
|
||||
<button
|
||||
onClick={toggleCollapsed}
|
||||
title="Replier l'aperçu"
|
||||
className="px-2 text-[15px] text-muted-2 hover:text-accent"
|
||||
>
|
||||
⇥
|
||||
</button>
|
||||
```
|
||||
|
||||
(replié = composant retourne tôt → aucun contenu/iframe monté, conforme spec.)
|
||||
|
||||
- [ ] **Step 4: Vérifier + commit**
|
||||
|
||||
Run: `cd frontend && npx tsc -b && npm run build` — OK.
|
||||
|
||||
```bash
|
||||
git add frontend/src/components/ProjectChip.tsx frontend/src/panels/ChatPanel.tsx frontend/src/panels/PreviewPanel.tsx
|
||||
git commit -m "feat(ui): chip projet dans le composer + aperçu réductible"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 6: Vérification bout-en-bout + docs + push
|
||||
|
||||
**Files:**
|
||||
- Modify: `README.md`
|
||||
|
||||
- [ ] **Step 1: README**
|
||||
|
||||
Section après « Panneau Git & Diff » :
|
||||
|
||||
```markdown
|
||||
## Projets (répertoires de travail)
|
||||
|
||||
Chaque session peut travailler dans un **projet** : un sous-dossier du
|
||||
workspace choisi via le chip 📁 du composer (« + Nouveau projet » pour en
|
||||
créer un). L'agent, le shell, le moteur code et l'arborescence sont confinés
|
||||
au projet ; chaque projet a son propre dépôt git (historique et revert
|
||||
indépendants). Session sans projet = racine du workspace.
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Suite complète**
|
||||
|
||||
```bash
|
||||
cd backend && python -m pytest tests/ -q && python -m compileall -q app
|
||||
cd ../frontend && npx tsc -b && npm run build
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Smoke local (uvicorn, sans Ollama)**
|
||||
|
||||
```bash
|
||||
cd backend && python -m uvicorn app.main:app --port 8199 & # env WORKSPACE_DIR/DATA_DIR temporaires
|
||||
curl -s -X POST http://localhost:8199/api/projects -H "Content-Type: application/json" -d '{"name":"demo"}'
|
||||
curl -s http://localhost:8199/api/projects
|
||||
curl -s -X POST http://localhost:8199/api/projects -H "Content-Type: application/json" -d '{"name":"../x"}' -o /dev/null -w "%{http_code}" # 400
|
||||
SID=$(curl -s -X POST http://localhost:8199/api/sessions -H "Content-Type: application/json" -d '{"title":"t","project":"demo"}' | python -c "import sys,json;print(json.load(sys.stdin)['id'])")
|
||||
curl -s -X PATCH http://localhost:8199/api/sessions/$SID -H "Content-Type: application/json" -d '{"project":""}' # retour racine
|
||||
curl -s "http://localhost:8199/api/files?project=demo"
|
||||
```
|
||||
|
||||
Expected : création 201 + `.git` ; liste contient demo ; 400 sur nom
|
||||
invalide ; PATCH ok ; arbre du projet vide (pas les fichiers racine).
|
||||
|
||||
- [ ] **Step 4: Commit + push**
|
||||
|
||||
```bash
|
||||
git add README.md
|
||||
git commit -m "docs: projets par session"
|
||||
git push
|
||||
```
|
||||
|
||||
Vérif UI après déploiement : chip 📁 → créer `demo` → message « crée
|
||||
hello.txt » → fichier dans workspace/demo/ ; session B sans projet → racine ;
|
||||
onglet Git montre l'historique du projet actif ; bouton ⇥ replie l'aperçu,
|
||||
état conservé au rechargement.
|
||||
@@ -1,135 +0,0 @@
|
||||
# Loki — MCP intégré + Skills automatiques + Boucle code vérifiée
|
||||
|
||||
**Date** : 2026-07-18
|
||||
**Objectif** : augmenter l'intelligence effective des modèles locaux (~17 Go,
|
||||
répartis sur 5060 Ti + 3060 12 Go) sans changer de modèle, en améliorant le
|
||||
harnais : outils professionnels via MCP, méthodes expertes injectées (skills),
|
||||
et vérification réelle du code produit.
|
||||
|
||||
**Choix utilisateur validés** :
|
||||
- Cas d'usage : tout (code + raisonnement + multi-étapes).
|
||||
- Compromis : équilibré — passes supplémentaires uniquement pour le code
|
||||
(1 passe de correction max).
|
||||
- Un seul gros modèle ; pas d'architecture deux-modèles.
|
||||
- MCP : catalogue préconfiguré des meilleurs serveurs, toggles on/off,
|
||||
désactivé = zéro coût (aucun outil dans le prompt, aucun process lancé).
|
||||
- Skills : déclenchement automatique (routeur lexical), invisible, désactivable.
|
||||
|
||||
---
|
||||
|
||||
## 1. Client MCP intégré
|
||||
|
||||
### Backend — `backend/app/mcp_client.py`
|
||||
|
||||
- SDK officiel Python `mcp` (ajout à `requirements.txt`).
|
||||
- Transports : stdio (commande locale, ex. `npx …`) et streamable-http (URL).
|
||||
- **Gestionnaire** (singleton, même pattern que `ollama_client`) :
|
||||
- Catalogue embarqué (dict Python) + état activé/désactivé et paramètres
|
||||
(clé API, URL) persistés en DB (`db.get/set_config_value`, clé `mcp`).
|
||||
- Connexion **lazy** : un serveur activé n'est démarré qu'au premier message
|
||||
qui suit ; la session MCP (process + handshake) est réutilisée ensuite.
|
||||
- Serveur désactivé : process jamais lancé, outils jamais exposés.
|
||||
- `list_tools()` par serveur → conversion schéma MCP → schéma
|
||||
function-calling Ollama ; préfixe `mcp_<serveur>_<outil>`.
|
||||
- `call_tool(name, args)` avec timeout 30 s par appel.
|
||||
- Panne (crash process, handshake KO, timeout répété) : le serveur est
|
||||
marqué en erreur, ses outils retirés, notice SSE dans le fil ; nouvelle
|
||||
tentative au message suivant. Un MCP cassé ne bloque jamais le chat.
|
||||
- `aclose()` au shutdown (lifespan `main.py`).
|
||||
|
||||
### Catalogue préconfiguré (désactivés par défaut)
|
||||
|
||||
| ID | Serveur | Commande / transport | Apport |
|
||||
|---|---|---|---|
|
||||
| `playwright` | Playwright MCP | `npx @playwright/mcp@latest --headless` | navigateur réel : naviguer, cliquer, lire la console, screenshot |
|
||||
| `context7` | Context7 | `npx -y @upstash/context7-mcp` | doc à jour de toute librairie/framework |
|
||||
| `fetch` | Fetch | `python -m mcp_server_fetch` | lecture propre d'URL (markdown) |
|
||||
| `searxng` | SearxNG | `npx -y mcp-searxng` + `SEARXNG_URL` | vraie recherche web (nécessite instance SearxNG) |
|
||||
| `custom` | Personnalisé | commande ou URL saisie par l'utilisateur | n'importe quel serveur MCP |
|
||||
|
||||
- Filtre d'outils par serveur (liste `expose` optionnelle dans le catalogue)
|
||||
pour limiter le nombre d'outils injectés (ex. Playwright : navigate, click,
|
||||
type, snapshot, console, screenshot — pas les 25+ outils complets).
|
||||
|
||||
### Intégration boucle agent — `backend/app/agent.py` + `tools.py`
|
||||
|
||||
- Les outils MCP actifs sont ajoutés à la liste d'outils envoyée au modèle
|
||||
(après les outils natifs) et dispatchés vers `mcp_client.call_tool`.
|
||||
- Résultats tronqués à une taille max (~8 000 caractères) avant retour au
|
||||
modèle (protège le contexte).
|
||||
- Rendu frontend : ToolCards existantes (aucun composant nouveau requis).
|
||||
|
||||
### UI — Configuration → onglet « MCP » (`SettingsView.tsx`)
|
||||
|
||||
- Une carte par serveur du catalogue : nom, description, toggle, statut
|
||||
(● connecté / ○ inactif / ⚠ erreur + message), nombre d'outils exposés,
|
||||
champs paramètres (URL/clé) si requis.
|
||||
- Carte « Personnalisé » : champ commande ou URL + toggle.
|
||||
- Routes : `GET /api/mcp` (état), `PUT /api/mcp/{id}` (toggle + params),
|
||||
`POST /api/mcp/{id}/test` (connexion d'essai, renvoie outils découverts).
|
||||
|
||||
### Dockerfile
|
||||
|
||||
- Ajout Node.js LTS (requis npx) + `pip install mcp mcp-server-fetch`.
|
||||
- Image : les serveurs npx sont téléchargés au premier lancement (cache
|
||||
volume npm optionnel).
|
||||
|
||||
## 2. Skills automatiques
|
||||
|
||||
### Bibliothèque — `backend/skills/*.md`
|
||||
|
||||
Fichiers markdown en français, livrés avec l'app :
|
||||
- `debogage-systematique.md` — reproduire → isoler → hypothèse → corriger → vérifier.
|
||||
- `creation-web.md` — structure sémantique → style → interactivité → vérification (liens, console, responsive).
|
||||
- `refactor-sur.md` — comprendre → tests/garde-fous → petits pas → vérifier à chaque pas.
|
||||
- `analyse-donnees.md` — examiner le format → valider → transformer → présenter.
|
||||
- `redaction-structuree.md` — plan → rédaction → relecture ciblée.
|
||||
|
||||
Format : frontmatter (`name`, `description`, `keywords`) + corps injecté tel quel.
|
||||
|
||||
### Routeur de skills — `backend/app/skills.py`
|
||||
|
||||
- Sélection **lexicale instantanée** (regex/mots-clés, même style que
|
||||
`router.py`) : au plus UNE skill par message ; aucun appel LLM.
|
||||
- Injection : message système supplémentaire
|
||||
(`"Méthode à suivre pour cette tâche :\n<corps>"`) inséré après l'invite
|
||||
système, pour ce tour uniquement (non persisté dans l'historique).
|
||||
- Événement SSE `skill` `{name, title}` → badge dans le fil
|
||||
(« 📘 Skill : Débogage systématique »).
|
||||
- Toggle global `skills_enabled` (défaut : activé) dans la config agent
|
||||
(Configuration → Intelligence).
|
||||
|
||||
## 3. Boucle code vérifiée
|
||||
|
||||
- Nouvel outil natif `run_check` (`tools.py`) : détection par extension —
|
||||
`.py` → `python -m py_compile` (analyse seule, JAMAIS d'exécution — toute
|
||||
exécution passe par `run_shell` et sa validation utilisateur) ;
|
||||
`.js/.ts` → `node --check` ; `.html` → `check_html` existant ; JSON → parse.
|
||||
Sortie = erreurs compactées.
|
||||
- Boucle agent : après un `write_file`/`edit_file` sur du code, le harnais
|
||||
exécute `run_check` automatiquement et renvoie les erreurs au modèle dans le
|
||||
même tour — **1 passe de correction maximum** (choix « équilibré »).
|
||||
- Si le serveur MCP Playwright est actif et que la tâche a produit du HTML :
|
||||
la skill `creation-web` oriente le modèle vers une vérification réelle dans
|
||||
le navigateur (console + rendu) au lieu du check statique seul.
|
||||
|
||||
## 4. Hors périmètre
|
||||
|
||||
- Architecture deux-modèles (architecte/codeur) — rejetée par l'utilisateur.
|
||||
- Best-of-N / juge — trop coûteux.
|
||||
- Marketplace/édition de skills dans l'UI (v2 possible ; v1 = fichiers livrés).
|
||||
|
||||
## 5. Vérification
|
||||
|
||||
1. **MCP** : activer Fetch → demander « résume cette page <url> » → ToolCard
|
||||
`mcp_fetch_fetch` + résumé correct. Désactiver → l'outil disparaît du
|
||||
prompt (vérifiable via logs). Serveur cassé (commande invalide) → notice,
|
||||
chat fonctionnel.
|
||||
2. **Playwright** : « crée une page X puis vérifie-la dans le navigateur » →
|
||||
navigation + console lue + correction éventuelle.
|
||||
3. **Skills** : message « mon script plante avec TypeError » → badge skill
|
||||
débogage ; message banal → aucun badge.
|
||||
4. **run_check** : demander un script Python avec bug volontaire induit →
|
||||
l'erreur est détectée et corrigée dans le même tour (1 passe).
|
||||
5. **Perf** : aucun serveur MCP actif → latence premier token inchangée
|
||||
(mesure avant/après).
|
||||
@@ -1,99 +0,0 @@
|
||||
# Loki — Projets (répertoires de travail) + aperçu réductible
|
||||
|
||||
**Date** : 2026-07-19
|
||||
**Objectif** : permettre à l'agent de travailler dans des projets entiers —
|
||||
un projet = un répertoire choisi depuis la carte du chat — et pouvoir replier
|
||||
le panneau d'aperçu.
|
||||
|
||||
**Choix validés** : projet = sous-dossier de premier niveau du workspace
|
||||
(pas de chemin arbitraire) ; portée par session ; édition de la mémoire
|
||||
écartée du périmètre.
|
||||
|
||||
---
|
||||
|
||||
## 1. Projets — backend
|
||||
|
||||
### Modèle
|
||||
- Un projet = un sous-dossier de premier niveau de `WORKSPACE_DIR`
|
||||
(ex. `workspace/jeu-snake/`). Nom validé : `^[a-z0-9][a-z0-9_-]{0,40}$`.
|
||||
- Session sans projet (`NULL`) = racine du workspace — compatibilité totale
|
||||
avec les sessions existantes.
|
||||
- Colonne `project TEXT` ajoutée à `sessions` (migration douce
|
||||
`ALTER TABLE`, comme `summary`/`meta`).
|
||||
|
||||
### Racine active par requête (contextvar)
|
||||
- `tools.py` : `_ACTIVE_ROOT: ContextVar[str | None]` + fonctions
|
||||
`set_project(name | None)` / lecture dans `_workspace_root()`. `_safe_path`
|
||||
inchangé dans sa logique de confinement — simplement re-raciné sur
|
||||
`workspace/<projet>` quand un projet est actif.
|
||||
- `routes/chat.py` : au début de `chat()`, `tools.set_project(session["project"])`
|
||||
(après validation : dossier existant, sinon retombe sur la racine + notice
|
||||
SSE « projet introuvable, retour au workspace »).
|
||||
- Les aides de contexte suivent la même racine : `_workspace_listing`,
|
||||
`_mentioned_files`, `_session_code_context` (fichiers vérifiés sous la
|
||||
racine active).
|
||||
- `run_shell` : cwd = racine active (déjà `_workspace_root()`).
|
||||
|
||||
### Moteur code (Aider) par projet
|
||||
- `coder.run_code_task` reçoit la racine active (cwd du process Aider).
|
||||
- `coder.ensure_git(dir)` appelé à la création d'un projet → un dépôt git
|
||||
par projet, historique/revert propres. L'onglet Git du panneau droit opère
|
||||
sur le projet de la session courante (routes git prennent `project`).
|
||||
|
||||
### Routes
|
||||
- `GET /api/projects` → `{projects: [{name, files: int}], root_files: int}`
|
||||
(sous-dossiers de premier niveau, dotfiles exclus).
|
||||
- `POST /api/projects {name}` → mkdir + `ensure_git` ; 400 si nom invalide
|
||||
ou existant.
|
||||
- `PATCH /api/sessions/{sid}` accepte `project: str | null` (en plus de
|
||||
`title`).
|
||||
- `GET /api/files`, `/api/files/content`, `/api/files/download`,
|
||||
`DELETE /api/files` : paramètre optionnel `project` — même re-racinage,
|
||||
même confinement.
|
||||
- Routes git (`/api/git/*`) : paramètre optionnel `project`.
|
||||
|
||||
## 2. Sélecteur projet — frontend (carte du chat)
|
||||
|
||||
- **Chip « 📁 <projet> »** dans le composer, à côté du sélecteur de mode
|
||||
Plan/Build/Yolo. Affiche `workspace` si aucun projet.
|
||||
- Clic → menu (même style que les menus existants) : liste des projets,
|
||||
entrée active cochée, **« + Nouveau projet »** avec input inline
|
||||
(Enter crée + sélectionne, Escape annule).
|
||||
- Sélection → `PATCH /api/sessions/{sid} {project}` → store met à jour la
|
||||
session courante → `refreshFiles()` re-racine LeftPanel / FilesView /
|
||||
PreviewPanel (le store passe `project` de la session courante aux appels
|
||||
fichiers).
|
||||
- Nouvelle session : hérite du projet de la session courante (envoyé au
|
||||
`POST /api/sessions`).
|
||||
- Session sans projet : comportement actuel inchangé.
|
||||
|
||||
## 3. Aperçu réductible — frontend
|
||||
|
||||
- Bouton de repli dans l'en-tête du `PreviewPanel` (à côté des onglets) :
|
||||
replie le panneau en **barre verticale fine (36 px)** ne contenant qu'un
|
||||
bouton de réouverture (icône ⇤ pivotée) et l'indicateur d'onglet actif.
|
||||
- État replié persisté en `localStorage` (`loki.preview.collapsed`), la
|
||||
largeur l'est déjà (`loki.preview.width`).
|
||||
- Replié, le panneau ne rend pas son contenu (pas d'iframe HTML vivante).
|
||||
|
||||
## 4. Hors périmètre
|
||||
- Édition de la mémoire (RAG/résumés) — écartée par l'utilisateur.
|
||||
- Chemins hors workspace, suppression/renommage de projets depuis l'UI
|
||||
(suppression possible via la corbeille de l'arborescence).
|
||||
- Projet imbriqué (sous-sous-dossier comme projet).
|
||||
|
||||
## 5. Vérification
|
||||
1. `POST /api/projects {"name":"demo"}` → dossier + `.git` créés ;
|
||||
nom invalide → 400.
|
||||
2. Session A projet `demo`, session B sans projet : fichiers créés par
|
||||
l'agent de A atterrissent dans `workspace/demo/`, ceux de B à la racine ;
|
||||
l'arborescence gauche suit la session ouverte.
|
||||
3. `curl "…/api/files?project=demo"` → arbre du projet seul ;
|
||||
`?project=../x` → 400.
|
||||
4. Reprise : « corrige les bugs » dans une session projet → Aider travaille
|
||||
dans `workspace/demo/`, commit dans le git du projet, onglet Git montre
|
||||
l'historique du projet.
|
||||
5. Aperçu : bouton replie → barre fine, contenu démonté ; réouverture
|
||||
restaure l'onglet et la largeur ; état conservé après rechargement.
|
||||
6. pytest : contextvar (re-racinage + confinement), routes projects,
|
||||
migration colonne, héritage projet à la création de session.
|
||||
BIN
Binary file not shown.
|
After Width: | Height: | Size: 118 KiB |
@@ -1,19 +0,0 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fr">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<link rel="icon" type="image/svg+xml" href="/favicon.svg" />
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com" />
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
|
||||
<link
|
||||
href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&display=swap"
|
||||
rel="stylesheet"
|
||||
/>
|
||||
<title>Loki - Agent local</title>
|
||||
</head>
|
||||
<body>
|
||||
<div id="root"></div>
|
||||
<script type="module" src="/src/main.tsx"></script>
|
||||
</body>
|
||||
</html>
|
||||
Generated
-4273
File diff suppressed because it is too large.
Load diff
@@ -1,28 +0,0 @@
|
||||
{
|
||||
"name": "loki-frontend",
|
||||
"private": true,
|
||||
"version": "0.1.0",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
"build": "tsc -b && vite build",
|
||||
"preview": "vite preview"
|
||||
},
|
||||
"dependencies": {
|
||||
"react": "^18.3.1",
|
||||
"react-dom": "^18.3.1",
|
||||
"react-markdown": "^10.1.0",
|
||||
"remark-gfm": "^4.0.1",
|
||||
"zustand": "^5.0.2"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/react": "^18.3.18",
|
||||
"@types/react-dom": "^18.3.5",
|
||||
"@vitejs/plugin-react": "^4.3.4",
|
||||
"autoprefixer": "^10.4.20",
|
||||
"postcss": "^8.4.49",
|
||||
"tailwindcss": "^3.4.17",
|
||||
"typescript": "^5.7.2",
|
||||
"vite": "^6.0.7"
|
||||
}
|
||||
}
|
||||
@@ -1,6 +0,0 @@
|
||||
export default {
|
||||
plugins: {
|
||||
tailwindcss: {},
|
||||
autoprefixer: {},
|
||||
},
|
||||
};
|
||||
@@ -1,6 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 64 64">
|
||||
<rect width="64" height="64" rx="10" fill="#18181b"/>
|
||||
<rect x="14" y="14" width="38" height="38" rx="7" fill="#ff5436"/>
|
||||
<rect x="11" y="11" width="38" height="38" rx="7" fill="#ff5436" stroke="#ffffff" stroke-width="6"/>
|
||||
<rect x="25" y="25" width="12" height="12" fill="#ffffff"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 358 B |
@@ -1,73 +0,0 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { Sidebar, type View } from "./components/Sidebar";
|
||||
import { TopBar } from "./components/TopBar";
|
||||
import { ChatPanel } from "./panels/ChatPanel";
|
||||
import { PreviewPanel } from "./panels/PreviewPanel";
|
||||
import { SettingsView } from "./panels/SettingsView";
|
||||
import { HistoryView, ToolsView } from "./panels/ActivityViews";
|
||||
import { useStore } from "./store/useStore";
|
||||
|
||||
export default function App() {
|
||||
const [view, setView] = useState<View>("chat");
|
||||
const { refreshPulse, refreshModels, refreshSessions } = useStore();
|
||||
|
||||
// Sondage frugal. Trois principes, contre l'app qui chauffait le CPU à vide :
|
||||
// - UNE seule requête groupée (/api/system/pulse) au lieu de trois ;
|
||||
// - RIEN quand l'onglet est caché (l'utilisateur ne regarde pas) ;
|
||||
// - cadence adaptative : réactif pendant le travail, lent au repos.
|
||||
// Le serveur, lui, ne tourne jamais en boucle : sans navigateur, zéro CPU.
|
||||
useEffect(() => {
|
||||
refreshPulse();
|
||||
refreshModels();
|
||||
refreshSessions();
|
||||
|
||||
let timer: number | undefined;
|
||||
const tick = () => {
|
||||
// Onglet caché : on ne sonde pas du tout, on repassera au retour.
|
||||
if (document.hidden) return schedule();
|
||||
refreshPulse().finally(schedule);
|
||||
};
|
||||
const schedule = () => {
|
||||
if (document.hidden) {
|
||||
timer = window.setTimeout(schedule, 30000);
|
||||
return;
|
||||
}
|
||||
// Pendant un flux, les indicateurs (VRAM, GPU) doivent suivre.
|
||||
const busy = useStore.getState().streaming;
|
||||
timer = window.setTimeout(tick, busy ? 4000 : 20000);
|
||||
};
|
||||
schedule();
|
||||
|
||||
// Retour sur l'onglet : rafraîchit tout de suite plutôt que d'attendre.
|
||||
const onVisible = () => {
|
||||
if (!document.hidden) {
|
||||
window.clearTimeout(timer);
|
||||
tick();
|
||||
}
|
||||
};
|
||||
document.addEventListener("visibilitychange", onVisible);
|
||||
return () => {
|
||||
window.clearTimeout(timer);
|
||||
document.removeEventListener("visibilitychange", onVisible);
|
||||
};
|
||||
}, [refreshPulse, refreshModels, refreshSessions]);
|
||||
|
||||
return (
|
||||
<div className="flex h-full flex-col bg-base text-ink">
|
||||
<TopBar />
|
||||
<div className="flex min-h-0 flex-1">
|
||||
<Sidebar active={view} onChange={setView} />
|
||||
|
||||
{view === "chat" && (
|
||||
<>
|
||||
<ChatPanel />
|
||||
<PreviewPanel />
|
||||
</>
|
||||
)}
|
||||
{view === "history" && <HistoryView onOpen={() => setView("chat")} />}
|
||||
{view === "tools" && <ToolsView onSettings={() => setView("settings")} />}
|
||||
{view === "settings" && <SettingsView />}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,795 +0,0 @@
|
||||
/** Petit client API typé pour le backend Loki. */
|
||||
|
||||
export interface OllamaStatus {
|
||||
connected: boolean;
|
||||
host: string;
|
||||
version?: string;
|
||||
error?: string;
|
||||
default_model: string;
|
||||
}
|
||||
|
||||
export interface OllamaModel {
|
||||
name: string;
|
||||
size_go: number;
|
||||
parameter_size?: string;
|
||||
quantization?: string;
|
||||
family?: string;
|
||||
}
|
||||
|
||||
export async function getStatus(): Promise<OllamaStatus> {
|
||||
const res = await fetch("/api/status");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export interface GpuStats {
|
||||
name: string;
|
||||
util_pct: number;
|
||||
vram_used_mb: number;
|
||||
vram_total_mb: number;
|
||||
}
|
||||
|
||||
export interface SystemStats {
|
||||
cpu_pct: number;
|
||||
ram_used_go: number;
|
||||
ram_total_go: number;
|
||||
ram_pct: number;
|
||||
gpu: GpuStats | null;
|
||||
}
|
||||
|
||||
export async function getSystemStats(): Promise<SystemStats> {
|
||||
const res = await fetch("/api/system/stats");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
/** Battement groupé : statut + ressources + modèles chargés en UNE requête. */
|
||||
export interface Pulse {
|
||||
status: OllamaStatus;
|
||||
stats: SystemStats;
|
||||
loaded: LoadedModel[];
|
||||
}
|
||||
|
||||
export async function getPulse(): Promise<Pulse> {
|
||||
const res = await fetch("/api/system/pulse");
|
||||
if (!res.ok) throw new Error(`pulse ${res.status}`);
|
||||
return res.json();
|
||||
}
|
||||
|
||||
// ── Git du workspace ─────────────────────────────────────────────────────
|
||||
export interface GitCommit {
|
||||
hash: string;
|
||||
full_hash: string;
|
||||
subject: string;
|
||||
author: string;
|
||||
when: string;
|
||||
files_changed: number;
|
||||
}
|
||||
|
||||
const projQuery = (project?: string | null) =>
|
||||
project ? `&project=${encodeURIComponent(project)}` : "";
|
||||
|
||||
export async function getGitLog(project?: string | null): Promise<GitCommit[]> {
|
||||
try {
|
||||
const res = await fetch(`/api/git/log?limit=40${projQuery(project)}`);
|
||||
return (await res.json()).commits;
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getGitDiff(
|
||||
hash?: string,
|
||||
project?: string | null
|
||||
): Promise<string> {
|
||||
const params = new URLSearchParams();
|
||||
if (hash) params.set("hash", hash);
|
||||
if (project) params.set("project", project);
|
||||
const qs = params.toString();
|
||||
const res = await fetch(`/api/git/diff${qs ? `?${qs}` : ""}`);
|
||||
if (!res.ok) return "";
|
||||
return (await res.json()).diff;
|
||||
}
|
||||
|
||||
export async function revertCommit(
|
||||
hash: string,
|
||||
project?: string | null
|
||||
): Promise<{ message: string }> {
|
||||
const res = await fetch("/api/git/revert", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ hash, project: project ?? null }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "revert impossible");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export interface HardwareInfo {
|
||||
loki_gpus: {
|
||||
index: number;
|
||||
name: string;
|
||||
vram_total_mb: number;
|
||||
vram_used_mb: number;
|
||||
util_pct: number;
|
||||
}[];
|
||||
gpu_override: { name: string; vram_total_mb: number } | null;
|
||||
ollama: {
|
||||
host: string;
|
||||
connected: boolean;
|
||||
version?: string;
|
||||
error?: string;
|
||||
running: {
|
||||
name: string;
|
||||
processor: string;
|
||||
gpu_percent: number;
|
||||
size_mb: number;
|
||||
vram_mb: number;
|
||||
}[];
|
||||
};
|
||||
ollama_is_local: boolean;
|
||||
note: string;
|
||||
}
|
||||
|
||||
export async function getHardware(): Promise<HardwareInfo> {
|
||||
const res = await fetch("/api/system/hardware");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function getVersion(): Promise<string> {
|
||||
try {
|
||||
const res = await fetch("/api/version");
|
||||
return (await res.json()).version ?? "?";
|
||||
} catch {
|
||||
return "?";
|
||||
}
|
||||
}
|
||||
|
||||
export interface LoadedModel {
|
||||
name: string;
|
||||
on_gpu: boolean;
|
||||
gpu_percent: number;
|
||||
}
|
||||
|
||||
async function apiError(res: Response, fallback: string): Promise<Error> {
|
||||
try {
|
||||
const payload = await res.json();
|
||||
return new Error(payload?.detail ?? payload?.error ?? fallback);
|
||||
} catch {
|
||||
return new Error(fallback);
|
||||
}
|
||||
}
|
||||
|
||||
interface WarmStatus {
|
||||
state: "idle" | "loading" | "loaded" | "error";
|
||||
error?: string;
|
||||
processor?: "gpu" | "cpu" | "mixte";
|
||||
gpu_percent?: string;
|
||||
}
|
||||
|
||||
const wait = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
/**
|
||||
* Lance le préchargement en arrière-plan puis suit son état. Chaque requête
|
||||
* reste courte afin qu'un reverse proxy ne puisse plus interrompre le warm-up.
|
||||
*/
|
||||
export async function warmModel(
|
||||
name: string,
|
||||
keepAlive = "30m"
|
||||
): Promise<WarmStatus> {
|
||||
const res = await fetch("/api/models/warm", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name, keep_alive: keepAlive }),
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw await apiError(res, `préchargement refusé (${res.status})`);
|
||||
}
|
||||
|
||||
// Backoff progressif : réactif au début (modèle déjà chargé), espacé ensuite
|
||||
// pour ne pas marteler l'API pendant un long chargement.
|
||||
const deadline = Date.now() + 10 * 60 * 1000;
|
||||
let delay = 1000;
|
||||
while (Date.now() < deadline) {
|
||||
const statusRes = await fetch(
|
||||
`/api/models/warm/status?name=${encodeURIComponent(name)}`,
|
||||
{ cache: "no-store" }
|
||||
);
|
||||
if (!statusRes.ok) {
|
||||
throw await apiError(statusRes, `suivi du préchargement refusé (${statusRes.status})`);
|
||||
}
|
||||
const status = (await statusRes.json()) as WarmStatus;
|
||||
if (status.state === "loaded") return status;
|
||||
if (status.state === "error") {
|
||||
throw new Error(status.error ?? "préchargement impossible");
|
||||
}
|
||||
await wait(delay);
|
||||
delay = Math.min(delay * 1.5, 5000);
|
||||
}
|
||||
throw new Error("préchargement toujours en cours après 10 minutes");
|
||||
}
|
||||
|
||||
export async function getLoadedModels(): Promise<LoadedModel[]> {
|
||||
try {
|
||||
const res = await fetch("/api/models/loaded");
|
||||
return (await res.json()).loaded;
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getModels(): Promise<{
|
||||
models: OllamaModel[];
|
||||
default: string;
|
||||
}> {
|
||||
const res = await fetch("/api/models");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function deleteModel(name: string): Promise<void> {
|
||||
const res = await fetch("/api/models", {
|
||||
method: "DELETE",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const payload = await res.json().catch(() => null);
|
||||
throw new Error(payload?.detail ?? `suppression refusée (${res.status})`);
|
||||
}
|
||||
}
|
||||
|
||||
export interface Session {
|
||||
id: string;
|
||||
title: string;
|
||||
model?: string;
|
||||
project?: string | null;
|
||||
created_at: number;
|
||||
updated_at: number;
|
||||
message_count?: number;
|
||||
}
|
||||
|
||||
export interface ToolCall {
|
||||
name: string;
|
||||
args: Record<string, unknown>;
|
||||
summary?: string;
|
||||
status?: "ok" | "error" | "running" | "pending";
|
||||
}
|
||||
|
||||
export interface MessageStats {
|
||||
eval_count: number; // jetons générés
|
||||
prompt_eval_count: number; // jetons du prompt
|
||||
tokens_per_sec: number | null; // vitesse de génération
|
||||
}
|
||||
|
||||
export interface Message {
|
||||
id: string;
|
||||
session_id: string;
|
||||
role: "user" | "assistant";
|
||||
content: string;
|
||||
model?: string;
|
||||
meta?: {
|
||||
tools?: ToolCall[];
|
||||
stats?: MessageStats;
|
||||
thinking?: string;
|
||||
plan?: string[];
|
||||
engine?: string;
|
||||
} | null;
|
||||
created_at: number;
|
||||
}
|
||||
|
||||
export interface AgentConfig {
|
||||
system_prompt: string;
|
||||
temperature: number;
|
||||
top_p: number;
|
||||
top_k: number;
|
||||
max_tokens: number;
|
||||
num_ctx: number;
|
||||
num_gpu: number;
|
||||
num_batch: number;
|
||||
tools: Record<string, boolean>;
|
||||
confirm_shell: boolean;
|
||||
think: boolean;
|
||||
code_model: string;
|
||||
plan_mode: boolean;
|
||||
self_review: boolean;
|
||||
rag_enabled: boolean;
|
||||
embed_model: string;
|
||||
skills_enabled: boolean;
|
||||
ponytail: boolean;
|
||||
memory_mode: "off" | "ondemand" | "always";
|
||||
keep_alive: string;
|
||||
}
|
||||
|
||||
// ── Benchmark de modèles ─────────────────────────────────────────────────
|
||||
export interface BenchDetail {
|
||||
task: string;
|
||||
score: number;
|
||||
detail: string;
|
||||
}
|
||||
|
||||
export interface BenchResult {
|
||||
score: number;
|
||||
details: BenchDetail[];
|
||||
at: number;
|
||||
}
|
||||
|
||||
export async function getBenchScores(): Promise<Record<string, BenchResult>> {
|
||||
const res = await fetch("/api/bench");
|
||||
return (await res.json()).scores;
|
||||
}
|
||||
|
||||
/** Lance le benchmark d'un modèle en streamant la progression. */
|
||||
export async function runBench(
|
||||
model: string,
|
||||
onProgress: (task: string, score: number | null, detail?: string) => void
|
||||
): Promise<BenchResult | null> {
|
||||
let res: Response;
|
||||
try {
|
||||
res = await fetch("/api/bench", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ model }),
|
||||
});
|
||||
} catch (err) {
|
||||
const raw = err instanceof Error ? err.message : "connexion impossible";
|
||||
throw new Error(
|
||||
/network error|failed to fetch|load failed/i.test(raw)
|
||||
? "connexion au benchmark interrompue par le réseau ou le reverse proxy"
|
||||
: raw
|
||||
);
|
||||
}
|
||||
if (!res.ok) {
|
||||
throw await apiError(res, `benchmark refusé (${res.status})`);
|
||||
}
|
||||
if (!res.body) throw new Error("le serveur n'a pas renvoyé de progression");
|
||||
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
let final: BenchResult | null = null;
|
||||
|
||||
const dispatch = (block: string) => {
|
||||
let event = "message";
|
||||
const dataLines: string[] = [];
|
||||
for (const line of block.replace(/\r\n/g, "\n").split("\n")) {
|
||||
if (line.startsWith("event:")) event = line.slice(6).trim();
|
||||
else if (line.startsWith("data:")) dataLines.push(line.slice(5).trimStart());
|
||||
}
|
||||
if (dataLines.length === 0) return;
|
||||
const payload = JSON.parse(dataLines.join("\n"));
|
||||
if (event === "task_start") onProgress(payload.task, null);
|
||||
else if (event === "task_done")
|
||||
onProgress(payload.task, payload.score, payload.detail);
|
||||
else if (event === "error")
|
||||
throw new Error(payload.message ?? "le benchmark a échoué");
|
||||
else if (event === "done")
|
||||
final = { score: payload.score, details: payload.details, at: Date.now() / 1000 };
|
||||
};
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buffer += decoder.decode(value, { stream: true }).replace(/\r\n/g, "\n");
|
||||
const events = buffer.split("\n\n");
|
||||
buffer = events.pop() ?? "";
|
||||
for (const block of events) {
|
||||
if (block.trim()) dispatch(block);
|
||||
}
|
||||
}
|
||||
buffer += decoder.decode();
|
||||
if (buffer.trim()) dispatch(buffer);
|
||||
} catch (err) {
|
||||
const raw = err instanceof Error ? err.message : "connexion interrompue";
|
||||
if (/network error|failed to fetch|load failed/i.test(raw)) {
|
||||
throw new Error("flux du benchmark interrompu par le reverse proxy");
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
if (!final) throw new Error("le benchmark s'est interrompu avant le résultat");
|
||||
return final;
|
||||
}
|
||||
|
||||
export async function runShell(
|
||||
command: string
|
||||
): Promise<{ command: string; exit_code: number; output: string }> {
|
||||
const res = await fetch("/api/shell/run", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ command }),
|
||||
});
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function getConfig(model?: string): Promise<{
|
||||
config: AgentConfig;
|
||||
available_tools: string[];
|
||||
}> {
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config${query}`);
|
||||
return res.json();
|
||||
}
|
||||
|
||||
// ── Presets de configuration ─────────────────────────────────────────────
|
||||
export async function listPresets(): Promise<string[]> {
|
||||
const res = await fetch("/api/config/presets");
|
||||
return (await res.json()).presets;
|
||||
}
|
||||
|
||||
export async function savePreset(name: string, model?: string): Promise<string[]> {
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config/presets${query}`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "preset refusé");
|
||||
return (await res.json()).presets;
|
||||
}
|
||||
|
||||
export async function applyPreset(
|
||||
name: string,
|
||||
model?: string
|
||||
): Promise<AgentConfig> {
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config/presets/apply${query}`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "preset introuvable");
|
||||
return (await res.json()).config;
|
||||
}
|
||||
|
||||
export async function deletePreset(name: string): Promise<string[]> {
|
||||
const res = await fetch(`/api/config/presets?name=${encodeURIComponent(name)}`, {
|
||||
method: "DELETE",
|
||||
});
|
||||
return (await res.json()).presets;
|
||||
}
|
||||
|
||||
export async function saveConfig(
|
||||
patch: Partial<AgentConfig>,
|
||||
model?: string
|
||||
): Promise<AgentConfig> {
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config${query}`, {
|
||||
method: "PUT",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(patch),
|
||||
});
|
||||
return (await res.json()).config;
|
||||
}
|
||||
|
||||
export interface FileNode {
|
||||
name: string;
|
||||
path: string;
|
||||
type: "dir" | "file";
|
||||
size?: number;
|
||||
children?: FileNode[];
|
||||
}
|
||||
|
||||
export async function listFiles(project?: string | null): Promise<FileNode[]> {
|
||||
const query = project ? `?project=${encodeURIComponent(project)}` : "";
|
||||
const res = await fetch(`/api/files${query}`);
|
||||
return (await res.json()).tree;
|
||||
}
|
||||
|
||||
export async function fileContent(
|
||||
path: string,
|
||||
project?: string | null
|
||||
): Promise<string> {
|
||||
const res = await fetch(
|
||||
`/api/files/content?path=${encodeURIComponent(path)}${projQuery(project)}`
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
return (await res.json()).content;
|
||||
}
|
||||
|
||||
export async function deleteFile(
|
||||
path: string,
|
||||
project?: string | null
|
||||
): Promise<void> {
|
||||
const res = await fetch(
|
||||
`/api/files?path=${encodeURIComponent(path)}${projQuery(project)}`,
|
||||
{ method: "DELETE" }
|
||||
);
|
||||
if (!res.ok) {
|
||||
throw await apiError(res, `suppression impossible (${res.status})`);
|
||||
}
|
||||
}
|
||||
|
||||
export function downloadFile(path: string, project?: string | null): void {
|
||||
const link = document.createElement("a");
|
||||
link.href = `/api/files/download?path=${encodeURIComponent(path)}${projQuery(project)}`;
|
||||
link.download = path.split(/[\\/]/).pop() ?? "fichier";
|
||||
document.body.appendChild(link);
|
||||
link.click();
|
||||
link.remove();
|
||||
}
|
||||
|
||||
export interface McpServer {
|
||||
id: string;
|
||||
label: string;
|
||||
description: string;
|
||||
url_param: boolean;
|
||||
env_params: string[];
|
||||
enabled: boolean;
|
||||
params: Record<string, string>;
|
||||
state: "inactive" | "connected" | "error";
|
||||
error: string | null;
|
||||
tools: number;
|
||||
}
|
||||
|
||||
export async function listMcp(): Promise<McpServer[]> {
|
||||
const res = await fetch("/api/mcp");
|
||||
return (await res.json()).servers;
|
||||
}
|
||||
|
||||
export async function updateMcp(
|
||||
id: string,
|
||||
enabled: boolean,
|
||||
params: Record<string, string>
|
||||
): Promise<McpServer[]> {
|
||||
const res = await fetch(`/api/mcp/${id}`, {
|
||||
method: "PUT",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ enabled, params }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "mise à jour MCP impossible");
|
||||
return (await res.json()).servers;
|
||||
}
|
||||
|
||||
export async function testMcp(
|
||||
id: string
|
||||
): Promise<{ ok: boolean; tools: string[]; error: string | null }> {
|
||||
const res = await fetch(`/api/mcp/${id}/test`, { method: "POST" });
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function listSessions(): Promise<Session[]> {
|
||||
const res = await fetch("/api/sessions");
|
||||
return (await res.json()).sessions;
|
||||
}
|
||||
|
||||
export async function createSession(
|
||||
model?: string,
|
||||
project?: string | null
|
||||
): Promise<Session> {
|
||||
const res = await fetch("/api/sessions", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
title: "Nouvelle session",
|
||||
model,
|
||||
project: project ?? null,
|
||||
}),
|
||||
});
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function listProjects(): Promise<{
|
||||
projects: { name: string; files: number }[];
|
||||
root_files: number;
|
||||
}> {
|
||||
const res = await fetch("/api/projects");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function createProject(name: string): Promise<void> {
|
||||
const res = await fetch("/api/projects", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "création du projet impossible");
|
||||
}
|
||||
|
||||
export async function setSessionProject(
|
||||
id: string,
|
||||
project: string | null
|
||||
): Promise<void> {
|
||||
const res = await fetch(`/api/sessions/${id}`, {
|
||||
method: "PATCH",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ project: project ?? "" }),
|
||||
});
|
||||
if (!res.ok) throw await apiError(res, "changement de projet impossible");
|
||||
}
|
||||
|
||||
export async function getSession(
|
||||
id: string
|
||||
): Promise<{ session: Session; messages: Message[] }> {
|
||||
const res = await fetch(`/api/sessions/${id}`);
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function deleteSession(id: string): Promise<void> {
|
||||
await fetch(`/api/sessions/${id}`, { method: "DELETE" });
|
||||
}
|
||||
|
||||
export async function renameSession(id: string, title: string): Promise<void> {
|
||||
const res = await fetch(`/api/sessions/${id}`, {
|
||||
method: "PATCH",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ title }),
|
||||
});
|
||||
if (!res.ok) {
|
||||
throw await apiError(res, `renommage impossible (${res.status})`);
|
||||
}
|
||||
}
|
||||
|
||||
/** Envoie un message et streame la réponse de l'agent via SSE. */
|
||||
export async function streamChat(
|
||||
body: { session_id: string; content: string; model?: string; mode?: string },
|
||||
handlers: {
|
||||
onToken: (t: string) => void;
|
||||
onThinking: (t: string) => void;
|
||||
onToolCall: (call: ToolCall) => void;
|
||||
onToolResult: (call: ToolCall) => void;
|
||||
onToolConfirm: (command: string) => void;
|
||||
onStatus: (msg: string) => void;
|
||||
onNotice: (msg: string) => void;
|
||||
onPlan?: (steps: string[]) => void;
|
||||
onPlanStep?: (index: number) => void;
|
||||
onRevision?: (content: string) => void;
|
||||
onDone: (full: string) => void;
|
||||
onError: (msg: string) => void;
|
||||
onAbort?: () => void;
|
||||
},
|
||||
signal?: AbortSignal
|
||||
): Promise<void> {
|
||||
let res: Response;
|
||||
try {
|
||||
res = await fetch("/api/chat", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(body),
|
||||
signal,
|
||||
});
|
||||
} catch (err) {
|
||||
if (err instanceof DOMException && err.name === "AbortError") {
|
||||
handlers.onAbort?.();
|
||||
return;
|
||||
}
|
||||
const raw = err instanceof Error ? err.message : "serveur Loki injoignable";
|
||||
handlers.onError(
|
||||
/network error|failed to fetch/i.test(raw)
|
||||
? "Serveur Loki injoignable. Vérifiez le reverse proxy et le conteneur."
|
||||
: raw
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (!res.ok) {
|
||||
let message = `requête refusée (${res.status})`;
|
||||
try {
|
||||
const payload = await res.json();
|
||||
message = payload.detail ?? payload.error ?? message;
|
||||
} catch {
|
||||
/* réponse non JSON */
|
||||
}
|
||||
handlers.onError(message);
|
||||
return;
|
||||
}
|
||||
if (!res.body) {
|
||||
handlers.onError("pas de flux de réponse");
|
||||
return;
|
||||
}
|
||||
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
let terminal = false;
|
||||
let failed = false;
|
||||
|
||||
const dispatch = (raw: string) => {
|
||||
const block = raw.replace(/\r\n/g, "\n");
|
||||
let event = "message";
|
||||
const dataLines: string[] = [];
|
||||
for (const line of block.split("\n")) {
|
||||
if (line.startsWith("event:")) event = line.slice(6).trim();
|
||||
else if (line.startsWith("data:")) dataLines.push(line.slice(5).trimStart());
|
||||
}
|
||||
if (dataLines.length === 0) return;
|
||||
try {
|
||||
const payload = JSON.parse(dataLines.join("\n"));
|
||||
if (event === "token") handlers.onToken(payload.content);
|
||||
else if (event === "plan") handlers.onPlan?.(payload.steps);
|
||||
else if (event === "plan_step") handlers.onPlanStep?.(payload.index);
|
||||
else if (event === "revision") handlers.onRevision?.(payload.content);
|
||||
else if (event === "thinking") handlers.onThinking(payload.content);
|
||||
else if (event === "status") handlers.onStatus(payload.message);
|
||||
else if (event === "notice") handlers.onNotice(payload.message);
|
||||
else if (event === "tool_call")
|
||||
handlers.onToolCall({ ...payload, status: "running" });
|
||||
else if (event === "tool_result") handlers.onToolResult(payload);
|
||||
else if (event === "tool_confirm") handlers.onToolConfirm(payload.command);
|
||||
else if (event === "error") {
|
||||
failed = true;
|
||||
handlers.onError(payload.message);
|
||||
} else if (event === "done") {
|
||||
terminal = true;
|
||||
if (payload.error) {
|
||||
if (!failed) handlers.onError(payload.error);
|
||||
failed = true;
|
||||
} else if (!failed) {
|
||||
handlers.onDone(payload.content);
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
if (!failed) {
|
||||
failed = true;
|
||||
handlers.onError("réponse illisible reçue du serveur");
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buffer += decoder.decode(value, { stream: true }).replace(/\r\n/g, "\n");
|
||||
const events = buffer.split("\n\n");
|
||||
buffer = events.pop() ?? "";
|
||||
for (const block of events) {
|
||||
if (block.trim()) dispatch(block);
|
||||
}
|
||||
}
|
||||
buffer += decoder.decode();
|
||||
if (buffer.trim()) dispatch(buffer);
|
||||
} catch (err) {
|
||||
if (!failed) {
|
||||
if (err instanceof DOMException && err.name === "AbortError") {
|
||||
handlers.onAbort?.();
|
||||
return;
|
||||
}
|
||||
failed = true;
|
||||
const raw = err instanceof Error ? err.message : "connexion interrompue";
|
||||
handlers.onError(
|
||||
/network error|failed to fetch/i.test(raw)
|
||||
? "Connexion interrompue pendant le chargement du modèle. Vérifiez OLLAMA_HOST et le délai du reverse proxy."
|
||||
: raw
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (!terminal && !failed) {
|
||||
handlers.onError("le serveur a fermé la réponse avant sa fin");
|
||||
}
|
||||
}
|
||||
|
||||
/** Télécharge un modèle en streamant la progression via SSE. */
|
||||
export async function pullModel(
|
||||
name: string,
|
||||
onProgress: (status: string, percent: number) => void
|
||||
): Promise<void> {
|
||||
const res = await fetch("/api/models/pull", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ name }),
|
||||
});
|
||||
if (!res.body) return;
|
||||
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n\n");
|
||||
buffer = lines.pop() ?? "";
|
||||
for (const line of lines) {
|
||||
const data = line.replace(/^data: /, "").trim();
|
||||
if (!data || data === "[DONE]") continue;
|
||||
try {
|
||||
const chunk = JSON.parse(data);
|
||||
const percent =
|
||||
chunk.total && chunk.completed
|
||||
? Math.round((chunk.completed / chunk.total) * 100)
|
||||
: 0;
|
||||
onProgress(chunk.status ?? "", percent);
|
||||
} catch {
|
||||
/* ligne partielle, ignorée */
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,63 +0,0 @@
|
||||
import {
|
||||
ChatIcon,
|
||||
ClockIcon,
|
||||
FilesIcon,
|
||||
NodesIcon,
|
||||
SettingsIcon,
|
||||
} from "./Icon";
|
||||
|
||||
export type View = "chat" | "history" | "files" | "tools" | "settings";
|
||||
|
||||
const items: { id: View; label: string; icon: React.ReactNode }[] = [
|
||||
{ id: "chat", label: "Chat", icon: <ChatIcon /> },
|
||||
{ id: "history", label: "Historique", icon: <ClockIcon /> },
|
||||
{ id: "files", label: "Fichiers", icon: <FilesIcon /> },
|
||||
{ id: "tools", label: "Outils", icon: <NodesIcon /> },
|
||||
];
|
||||
|
||||
const btn = (on: boolean) =>
|
||||
`flex h-10 w-10 items-center justify-center border ${
|
||||
on
|
||||
? "border-white bg-accent text-white shadow-accent-soft"
|
||||
: "border-chrome-3 text-on-dark-2 hover:text-ink"
|
||||
}`;
|
||||
|
||||
/** Barre d'activité verticale (60px) sombre à gauche. */
|
||||
export function ActivityBar({
|
||||
active,
|
||||
onChange,
|
||||
}: {
|
||||
active: View;
|
||||
onChange: (v: View) => void;
|
||||
}) {
|
||||
return (
|
||||
<div className="flex w-[60px] flex-none flex-col items-center gap-2 border-r border-line bg-bar py-3">
|
||||
{items.map((it) => (
|
||||
<button
|
||||
key={it.id}
|
||||
onClick={() => onChange(it.id)}
|
||||
className={btn(active === it.id)}
|
||||
title={it.label}
|
||||
aria-label={it.label}
|
||||
>
|
||||
{it.icon}
|
||||
</button>
|
||||
))}
|
||||
|
||||
<div className="flex-1" />
|
||||
|
||||
<button
|
||||
onClick={() => onChange("settings")}
|
||||
className={btn(active === "settings")}
|
||||
title="Configuration"
|
||||
aria-label="Configuration"
|
||||
>
|
||||
<SettingsIcon />
|
||||
</button>
|
||||
|
||||
<div className="font-pixel flex h-[34px] w-[34px] items-center justify-center border border-line bg-line-soft text-[11px] text-ink-3">
|
||||
M
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,79 +0,0 @@
|
||||
import { useStore } from "../store/useStore";
|
||||
import { downloadFile, type FileNode } from "../api/client";
|
||||
import { DownloadIcon, TrashIcon } from "./Icon";
|
||||
|
||||
/** Arborescence du workspace : ouvrir dans l'aperçu, télécharger, supprimer. */
|
||||
export function FileTree({
|
||||
nodes,
|
||||
depth = 0,
|
||||
}: {
|
||||
nodes: FileNode[];
|
||||
depth?: number;
|
||||
}) {
|
||||
const { openPreview, previewPath, removeFile, currentProject } = useStore();
|
||||
|
||||
const confirmDelete = (n: FileNode) => {
|
||||
const msg =
|
||||
n.type === "dir"
|
||||
? `Supprimer le dossier ${n.path} et tout son contenu ?`
|
||||
: `Supprimer ${n.path} ?`;
|
||||
if (window.confirm(msg)) void removeFile(n.path);
|
||||
};
|
||||
|
||||
return (
|
||||
<>
|
||||
{nodes.map((n) => {
|
||||
const active = n.type === "file" && n.path === previewPath;
|
||||
return (
|
||||
<div key={n.path}>
|
||||
<div
|
||||
onClick={() => n.type === "file" && openPreview(n.path)}
|
||||
className={`group mb-[2px] flex items-center gap-2 rounded-card px-2 py-[5px] text-[13.5px] ${
|
||||
n.type === "file"
|
||||
? active
|
||||
? "cursor-pointer border border-accent-line bg-accent-ghost text-ink"
|
||||
: "cursor-pointer border border-transparent text-ink-3 hover:bg-accent-ghost"
|
||||
: "text-muted-2"
|
||||
}`}
|
||||
style={{ paddingLeft: 8 + depth * 14 }}
|
||||
>
|
||||
{n.type === "dir" ? (
|
||||
<span className="text-muted-3">▾</span>
|
||||
) : (
|
||||
<span
|
||||
className={`h-1.5 w-1.5 rounded-sm ${
|
||||
active ? "bg-accent" : "bg-muted-3"
|
||||
}`}
|
||||
/>
|
||||
)}
|
||||
<span className="min-w-0 flex-1 truncate">{n.name}</span>
|
||||
{n.type === "file" && (
|
||||
<button
|
||||
onClick={(event) => {
|
||||
event.stopPropagation();
|
||||
downloadFile(n.path, currentProject());
|
||||
}}
|
||||
className="hidden flex-none text-muted-3 hover:text-accent group-hover:block"
|
||||
title={`Télécharger ${n.name}`}
|
||||
>
|
||||
<DownloadIcon size={13} />
|
||||
</button>
|
||||
)}
|
||||
<button
|
||||
onClick={(event) => {
|
||||
event.stopPropagation();
|
||||
confirmDelete(n);
|
||||
}}
|
||||
className="hidden flex-none text-muted-3 hover:text-warn group-hover:block"
|
||||
title={`Supprimer ${n.name}`}
|
||||
>
|
||||
<TrashIcon size={13} />
|
||||
</button>
|
||||
</div>
|
||||
{n.children && <FileTree nodes={n.children} depth={depth + 1} />}
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
</>
|
||||
);
|
||||
}
|
||||
@@ -1,126 +0,0 @@
|
||||
/** Icônes SVG (style line, stroke) reprises du visuel. */
|
||||
type Props = { size?: number; className?: string };
|
||||
|
||||
const base = (size: number, className?: string) => ({
|
||||
width: size,
|
||||
height: size,
|
||||
viewBox: "0 0 24 24",
|
||||
fill: "none",
|
||||
stroke: "currentColor",
|
||||
strokeWidth: 1.7,
|
||||
strokeLinecap: "round" as const,
|
||||
strokeLinejoin: "round" as const,
|
||||
className,
|
||||
});
|
||||
|
||||
export const ChatIcon = ({ size = 19, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<path d="M4 6h16v10H10l-4 3v-3H4z" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const ClockIcon = ({ size = 19, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<circle cx="12" cy="12" r="8.5" />
|
||||
<path d="M12 7.5V12l3 1.8" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const FilesIcon = ({ size = 19, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<path d="M4 7h6l2 2h8v9H4z" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const NodesIcon = ({ size = 19, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<circle cx="7" cy="7" r="2.4" />
|
||||
<circle cx="17" cy="7" r="2.4" />
|
||||
<circle cx="7" cy="17" r="2.4" />
|
||||
<circle cx="17" cy="17" r="2.4" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const SettingsIcon = ({ size = 18, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<line x1="4" y1="8" x2="20" y2="8" />
|
||||
<line x1="4" y1="13" x2="20" y2="13" />
|
||||
<line x1="4" y1="18" x2="20" y2="18" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const ChevronDown = ({ size = 13, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={2}>
|
||||
<path d="m6 9 6 6 6-6" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const PlusIcon = ({ size = 13, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={2.2}>
|
||||
<path d="M12 5v14M5 12h14" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const RefreshIcon = ({ size = 15, className }: Props) => (
|
||||
<svg {...base(size, className)}>
|
||||
<path d="M21 12a9 9 0 1 1-3-6.7M21 4v4h-4" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const SendIcon = ({ size = 15, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={2.2}>
|
||||
<path d="M5 12h14M13 6l6 6-6 6" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const ClipIcon = ({ size = 16, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={1.8}>
|
||||
<path d="M21 11.5 12 20a5 5 0 0 1-7-7l8.5-8.5a3.5 3.5 0 0 1 5 5L10 16" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const DownloadIcon = ({ size = 14, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={1.9}>
|
||||
<path d="M12 4v11m0 0 4-4m-4 4-4-4M5 20h14" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const TrashIcon = ({ size = 14, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={1.9}>
|
||||
<path d="M4 7h16M10 7V5h4v2M9 7l.8 13h4.4L15 7M6.5 7l.7 13h9.6l.7-13" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const CopyIcon = ({ size = 14, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={1.7}>
|
||||
<rect x="9" y="9" width="11" height="11" rx="2" />
|
||||
<path d="M5 15V5a2 2 0 0 1 2-2h8" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
export const CheckIcon = ({ size = 14, className }: Props) => (
|
||||
<svg {...base(size, className)} strokeWidth={2.3}>
|
||||
<path d="m5 12 4.5 4.5L19 7" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
/** Logo Loki : carré teinté à filet accent + petit carré accent (maquette). */
|
||||
export const LokiMark = ({ size = 30 }: { size?: number }) => (
|
||||
<div
|
||||
className="grid place-items-center border border-accent bg-accent-ghost"
|
||||
style={{
|
||||
width: size,
|
||||
height: size,
|
||||
borderRadius: Math.round(size * 0.27),
|
||||
}}
|
||||
>
|
||||
<div
|
||||
className="bg-accent"
|
||||
style={{
|
||||
width: size * 0.3,
|
||||
height: size * 0.3,
|
||||
borderRadius: 2,
|
||||
}}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
@@ -1,151 +0,0 @@
|
||||
import { useState } from "react";
|
||||
import ReactMarkdown, { type Components } from "react-markdown";
|
||||
import remarkGfm from "remark-gfm";
|
||||
import { CheckIcon, CopyIcon } from "./Icon";
|
||||
|
||||
function copyWithTextarea(value: string) {
|
||||
const textarea = document.createElement("textarea");
|
||||
textarea.value = value;
|
||||
textarea.setAttribute("readonly", "");
|
||||
textarea.style.position = "fixed";
|
||||
textarea.style.left = "-9999px";
|
||||
textarea.style.top = "0";
|
||||
document.body.appendChild(textarea);
|
||||
textarea.select();
|
||||
const ok = document.execCommand("copy");
|
||||
textarea.remove();
|
||||
if (!ok) throw new Error("copy failed");
|
||||
}
|
||||
|
||||
function CodeBlock({ lang, code }: { lang: string; code: string }) {
|
||||
const [copied, setCopied] = useState(false);
|
||||
const [failed, setFailed] = useState(false);
|
||||
|
||||
const copy = async () => {
|
||||
try {
|
||||
if (navigator.clipboard?.writeText) {
|
||||
await navigator.clipboard.writeText(code);
|
||||
} else {
|
||||
copyWithTextarea(code);
|
||||
}
|
||||
setCopied(true);
|
||||
setFailed(false);
|
||||
setTimeout(() => setCopied(false), 1500);
|
||||
} catch {
|
||||
try {
|
||||
copyWithTextarea(code);
|
||||
setCopied(true);
|
||||
setFailed(false);
|
||||
setTimeout(() => setCopied(false), 1500);
|
||||
} catch {
|
||||
setFailed(true);
|
||||
setTimeout(() => setFailed(false), 1800);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="my-2 overflow-hidden border border-line bg-card-deep shadow-hard-sm" style={{ borderRadius: 7 }}>
|
||||
<div className="flex items-center justify-between border-b border-chrome-2 px-3 py-1.5">
|
||||
<span className="font-mono text-[10.5px] uppercase tracking-wide text-on-dark-2">
|
||||
{lang || "code"}
|
||||
</span>
|
||||
<button
|
||||
type="button"
|
||||
onClick={copy}
|
||||
className="flex items-center gap-1 rounded-md px-1.5 py-0.5 text-[11px] text-on-dark-2 transition-colors hover:text-ink"
|
||||
>
|
||||
{copied ? <CheckIcon size={13} /> : <CopyIcon size={13} />}
|
||||
{failed ? "Erreur" : copied ? "Copié" : "Copier"}
|
||||
</button>
|
||||
</div>
|
||||
<pre className="scr m-0 overflow-auto p-3 font-mono text-[12px] leading-relaxed text-on-dark">
|
||||
<code>{code}</code>
|
||||
</pre>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
const components: Components = {
|
||||
pre: ({ children }) => <>{children}</>,
|
||||
code({ className, children }) {
|
||||
const text = String(children).replace(/\n$/, "");
|
||||
const match = /language-(\w+)/.exec(className || "");
|
||||
if (match || text.includes("\n")) {
|
||||
return <CodeBlock lang={match?.[1] ?? ""} code={text} />;
|
||||
}
|
||||
return (
|
||||
<code className="rounded bg-sunken px-1 py-0.5 font-mono text-[12px] text-accent-1">
|
||||
{children}
|
||||
</code>
|
||||
);
|
||||
},
|
||||
h1: ({ children }) => (
|
||||
<h1 className="mb-2 mt-3 text-[17px] font-bold text-ink first:mt-0">
|
||||
{children}
|
||||
</h1>
|
||||
),
|
||||
h2: ({ children }) => (
|
||||
<h2 className="mb-2 mt-3 text-[15px] font-bold text-ink first:mt-0">
|
||||
{children}
|
||||
</h2>
|
||||
),
|
||||
h3: ({ children }) => (
|
||||
<h3 className="mb-1.5 mt-2.5 text-[13.5px] font-bold text-ink first:mt-0">
|
||||
{children}
|
||||
</h3>
|
||||
),
|
||||
p: ({ children }) => <p className="my-2 first:mt-0 last:mb-0">{children}</p>,
|
||||
ul: ({ children }) => (
|
||||
<ul className="my-2 list-disc space-y-1 pl-5 marker:text-muted-3">
|
||||
{children}
|
||||
</ul>
|
||||
),
|
||||
ol: ({ children }) => (
|
||||
<ol className="my-2 list-decimal space-y-1 pl-5 marker:text-muted-3">
|
||||
{children}
|
||||
</ol>
|
||||
),
|
||||
li: ({ children }) => <li className="leading-relaxed">{children}</li>,
|
||||
a: ({ href, children }) => (
|
||||
<a
|
||||
href={href}
|
||||
target="_blank"
|
||||
rel="noreferrer"
|
||||
className="text-accent underline decoration-accent/40 underline-offset-2 hover:decoration-accent"
|
||||
>
|
||||
{children}
|
||||
</a>
|
||||
),
|
||||
strong: ({ children }) => (
|
||||
<strong className="font-semibold text-ink">{children}</strong>
|
||||
),
|
||||
em: ({ children }) => <em className="italic">{children}</em>,
|
||||
blockquote: ({ children }) => (
|
||||
<blockquote className="my-2 border-l border-line-strong pl-3 text-muted">
|
||||
{children}
|
||||
</blockquote>
|
||||
),
|
||||
hr: () => <hr className="my-3 border-line-soft" />,
|
||||
table: ({ children }) => (
|
||||
<div className="my-2 overflow-x-auto">
|
||||
<table className="w-full border-collapse text-[12.5px]">{children}</table>
|
||||
</div>
|
||||
),
|
||||
th: ({ children }) => (
|
||||
<th className="border border-line-strong bg-card-deep px-2.5 py-1.5 text-left font-semibold text-ink-2">
|
||||
{children}
|
||||
</th>
|
||||
),
|
||||
td: ({ children }) => (
|
||||
<td className="border border-line-soft px-2.5 py-1.5">{children}</td>
|
||||
),
|
||||
};
|
||||
|
||||
export function MessageContent({ text }: { text: string }) {
|
||||
return (
|
||||
<ReactMarkdown remarkPlugins={[remarkGfm]} components={components}>
|
||||
{text}
|
||||
</ReactMarkdown>
|
||||
);
|
||||
}
|
||||
@@ -1,124 +0,0 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { ChevronDown } from "./Icon";
|
||||
|
||||
/** Sélecteur de modèle Ollama (chip orange, pastille = état de chargement).
|
||||
|
||||
variant "composer" : taille alignée sur les chips du composer, menu vers
|
||||
le haut. */
|
||||
export function ModelSelector({
|
||||
variant = "top",
|
||||
}: {
|
||||
variant?: "top" | "composer";
|
||||
}) {
|
||||
const {
|
||||
models,
|
||||
selectedModel,
|
||||
setSelectedModel,
|
||||
loadedModels,
|
||||
warmingModel,
|
||||
warmError,
|
||||
} = useStore();
|
||||
const [open, setOpen] = useState(false);
|
||||
const ref = useRef<HTMLDivElement>(null);
|
||||
|
||||
const loadedInfo = (name: string) =>
|
||||
loadedModels.find((m) => m.name === name);
|
||||
const selectedLoaded = loadedInfo(selectedModel);
|
||||
|
||||
useEffect(() => {
|
||||
const onClick = (e: MouseEvent) => {
|
||||
if (ref.current && !ref.current.contains(e.target as Node)) setOpen(false);
|
||||
};
|
||||
document.addEventListener("mousedown", onClick);
|
||||
return () => document.removeEventListener("mousedown", onClick);
|
||||
}, []);
|
||||
|
||||
return (
|
||||
<div className="relative" ref={ref}>
|
||||
<button
|
||||
onClick={() => setOpen((v) => !v)}
|
||||
className={
|
||||
variant === "composer"
|
||||
? "flex h-8 min-w-0 max-w-[220px] items-center gap-1.5 border border-line bg-accent px-2.5 text-[13px] text-white"
|
||||
: "flex h-[34px] min-w-0 max-w-[260px] items-center gap-2 border border-white bg-accent px-3 text-white shadow-accent-soft"
|
||||
}
|
||||
title={selectedModel || undefined}
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
<span
|
||||
className={`h-2.5 w-2.5 border border-line ${
|
||||
warmingModel === selectedModel
|
||||
? "bg-accent animate-pulse"
|
||||
: selectedLoaded
|
||||
? selectedLoaded.on_gpu
|
||||
? "bg-ok"
|
||||
: "bg-warn"
|
||||
: "bg-white"
|
||||
}`}
|
||||
title={
|
||||
warmError
|
||||
? warmError
|
||||
: warmingModel === selectedModel
|
||||
? "préchargement en cours…"
|
||||
: selectedLoaded
|
||||
? selectedLoaded.on_gpu
|
||||
? "chargé sur GPU"
|
||||
: `chargé (${selectedLoaded.gpu_percent}% GPU)`
|
||||
: "à charger"
|
||||
}
|
||||
/>
|
||||
<span className="min-w-0 truncate text-[14px] leading-none">{selectedModel || "—"}</span>
|
||||
<ChevronDown size={12} className="flex-none" />
|
||||
</button>
|
||||
|
||||
{open && (
|
||||
<div
|
||||
className={`absolute z-20 w-64 border border-line bg-card p-1 shadow-hard ${
|
||||
variant === "composer"
|
||||
? "bottom-[calc(100%+6px)] left-0"
|
||||
: "right-0 top-11"
|
||||
}`}
|
||||
>
|
||||
{models.length === 0 && (
|
||||
<div className="px-3 py-2 text-xs text-muted-2">
|
||||
Aucun modèle installé
|
||||
</div>
|
||||
)}
|
||||
{models.map((m) => (
|
||||
<button
|
||||
key={m.name}
|
||||
onClick={() => {
|
||||
setSelectedModel(m.name);
|
||||
setOpen(false);
|
||||
}}
|
||||
className={`flex w-full items-center gap-2 px-2 py-2 text-left hover:bg-base ${
|
||||
m.name === selectedModel ? "bg-base" : ""
|
||||
}`}
|
||||
>
|
||||
<span
|
||||
className={`h-2 w-2 border border-line ${
|
||||
m.name === selectedModel ? "bg-accent" : "bg-card"
|
||||
}`}
|
||||
/>
|
||||
<span className="min-w-0 flex-1 truncate text-xs text-ink-2" title={m.name}>
|
||||
{m.name}
|
||||
</span>
|
||||
{loadedInfo(m.name) && (
|
||||
<span
|
||||
className={`h-1.5 w-1.5 border border-line ${
|
||||
loadedInfo(m.name)!.on_gpu ? "bg-ok" : "bg-warn"
|
||||
}`}
|
||||
title={loadedInfo(m.name)!.on_gpu ? "chargé GPU" : "chargé CPU"}
|
||||
/>
|
||||
)}
|
||||
{m.size_go > 0 && (
|
||||
<span className="text-[10px] text-muted-2">{m.size_go} Go</span>
|
||||
)}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,118 +0,0 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { createProject } from "../api/client";
|
||||
|
||||
/** Sélecteur de projet du composer : 📁 <projet> + menu (liste, création). */
|
||||
export function ProjectChip() {
|
||||
const { projects, refreshProjects, currentProject, setProject } = useStore();
|
||||
const [open, setOpen] = useState(false);
|
||||
const [creating, setCreating] = useState(false);
|
||||
const [draft, setDraft] = useState("");
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const rootRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
const active = currentProject();
|
||||
|
||||
useEffect(() => {
|
||||
if (open) void refreshProjects();
|
||||
}, [open, refreshProjects]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
const onDown = (e: MouseEvent) => {
|
||||
if (rootRef.current && !rootRef.current.contains(e.target as Node)) {
|
||||
setOpen(false);
|
||||
setCreating(false);
|
||||
setError(null);
|
||||
}
|
||||
};
|
||||
document.addEventListener("mousedown", onDown);
|
||||
return () => document.removeEventListener("mousedown", onDown);
|
||||
}, [open]);
|
||||
|
||||
const choose = async (name: string | null) => {
|
||||
setOpen(false);
|
||||
await setProject(name);
|
||||
};
|
||||
|
||||
const create = async () => {
|
||||
const name = draft.trim();
|
||||
if (!name) return;
|
||||
try {
|
||||
await createProject(name);
|
||||
setCreating(false);
|
||||
setDraft("");
|
||||
await choose(name);
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "création impossible");
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div ref={rootRef} className="relative">
|
||||
<button
|
||||
onClick={() => setOpen((o) => !o)}
|
||||
className="flex h-8 items-center gap-1.5 border border-line bg-base px-2.5 text-[12px] text-ink-2"
|
||||
title="Projet de travail de cette session"
|
||||
>
|
||||
📁 <span className="max-w-[140px] truncate">{active ?? "workspace"}</span>
|
||||
</button>
|
||||
|
||||
{open && (
|
||||
<div
|
||||
className="absolute bottom-[calc(100%+6px)] left-0 z-30 w-[240px] border border-line bg-card shadow-hard"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
<button
|
||||
onClick={() => choose(null)}
|
||||
className={`block w-full px-3 py-2 text-left text-[13px] text-ink-2 hover:bg-base ${
|
||||
active === null ? "bg-base font-bold" : ""
|
||||
}`}
|
||||
>
|
||||
workspace (racine)
|
||||
</button>
|
||||
{projects.map((p) => (
|
||||
<button
|
||||
key={p.name}
|
||||
onClick={() => choose(p.name)}
|
||||
className={`block w-full border-t border-line-soft px-3 py-2 text-left text-[13px] text-ink-2 hover:bg-base ${
|
||||
active === p.name ? "bg-base font-bold" : ""
|
||||
}`}
|
||||
>
|
||||
📁 {p.name}
|
||||
<span className="ml-1.5 text-[11px] text-muted-2">
|
||||
{p.files} fichier{p.files > 1 ? "s" : ""}
|
||||
</span>
|
||||
</button>
|
||||
))}
|
||||
<div className="border-t border-line-soft p-2">
|
||||
{creating ? (
|
||||
<input
|
||||
autoFocus
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === "Enter") void create();
|
||||
if (e.key === "Escape") {
|
||||
setCreating(false);
|
||||
setError(null);
|
||||
}
|
||||
}}
|
||||
placeholder="nom-du-projet"
|
||||
className="w-full border border-line bg-base px-2 py-1 text-[12px] text-ink outline-none"
|
||||
/>
|
||||
) : (
|
||||
<button
|
||||
onClick={() => setCreating(true)}
|
||||
className="w-full text-left text-[13px] font-bold text-accent"
|
||||
>
|
||||
+ Nouveau projet
|
||||
</button>
|
||||
)}
|
||||
{error && <div className="mt-1 text-[11px] text-warn">{error}</div>}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,174 +0,0 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { ChevronDown, PlusIcon, TrashIcon } from "./Icon";
|
||||
import { relTime } from "../lib/time";
|
||||
|
||||
/** Sélecteur de session de la barre supérieure : titre courant + menu déroulant
|
||||
* (nouvelle session, liste, renommage inline, suppression). */
|
||||
export function SessionMenu() {
|
||||
const {
|
||||
sessions,
|
||||
currentSessionId,
|
||||
streamingSessionId,
|
||||
newSession,
|
||||
openSession,
|
||||
removeSession,
|
||||
renameSession,
|
||||
} = useStore();
|
||||
const [open, setOpen] = useState(false);
|
||||
const [editingId, setEditingId] = useState<string | null>(null);
|
||||
const [draft, setDraft] = useState("");
|
||||
const rootRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
const current = sessions.find((s) => s.id === currentSessionId);
|
||||
|
||||
// Fermeture au clic extérieur + Escape.
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
const onDown = (e: MouseEvent) => {
|
||||
if (rootRef.current && !rootRef.current.contains(e.target as Node)) {
|
||||
setOpen(false);
|
||||
setEditingId(null);
|
||||
}
|
||||
};
|
||||
const onKey = (e: KeyboardEvent) => {
|
||||
if (e.key === "Escape") {
|
||||
setOpen(false);
|
||||
setEditingId(null);
|
||||
}
|
||||
};
|
||||
document.addEventListener("mousedown", onDown);
|
||||
document.addEventListener("keydown", onKey);
|
||||
return () => {
|
||||
document.removeEventListener("mousedown", onDown);
|
||||
document.removeEventListener("keydown", onKey);
|
||||
};
|
||||
}, [open]);
|
||||
|
||||
const commitRename = async (id: string) => {
|
||||
const title = draft.trim();
|
||||
setEditingId(null);
|
||||
if (title) await renameSession(id, title);
|
||||
};
|
||||
|
||||
return (
|
||||
<div ref={rootRef} className="relative">
|
||||
<button
|
||||
onClick={() => setOpen((o) => !o)}
|
||||
className="flex cursor-pointer items-center gap-2 text-[13px] text-on-dark hover:text-ink"
|
||||
title="Sessions"
|
||||
>
|
||||
<span className="max-w-[220px] truncate">
|
||||
{current?.title ?? "Nouvelle session"}
|
||||
</span>
|
||||
<ChevronDown className="text-on-dark-3" />
|
||||
</button>
|
||||
|
||||
{open && (
|
||||
<div
|
||||
className="absolute left-0 top-[calc(100%+9px)] z-30 w-[300px] border border-line bg-chrome-2 shadow-hard"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
<button
|
||||
onClick={async () => {
|
||||
setOpen(false);
|
||||
await newSession();
|
||||
}}
|
||||
className="flex w-full cursor-pointer items-center gap-2 border-b border-chrome-3 px-3 py-2.5 text-[13px] font-bold text-accent hover:bg-chrome-3"
|
||||
>
|
||||
<PlusIcon />
|
||||
Nouvelle session
|
||||
</button>
|
||||
|
||||
<div className="scr max-h-[320px] overflow-auto">
|
||||
{sessions.length === 0 && (
|
||||
<div className="px-3 py-4 text-center text-[12px] text-on-dark-3">
|
||||
Aucune session enregistrée.
|
||||
</div>
|
||||
)}
|
||||
{sessions.map((s) => {
|
||||
const active = s.id === currentSessionId;
|
||||
const working = s.id === streamingSessionId;
|
||||
return (
|
||||
<div
|
||||
key={s.id}
|
||||
onClick={() => {
|
||||
if (editingId === s.id) return;
|
||||
setOpen(false);
|
||||
void openSession(s.id);
|
||||
}}
|
||||
className={`group cursor-pointer border-b border-chrome-3 px-3 py-2 last:border-b-0 ${
|
||||
active ? "bg-chrome-3" : "hover:bg-chrome-3"
|
||||
}`}
|
||||
>
|
||||
<div className="flex items-center gap-1.5">
|
||||
{editingId === s.id ? (
|
||||
<input
|
||||
autoFocus
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onClick={(e) => e.stopPropagation()}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === "Enter") void commitRename(s.id);
|
||||
if (e.key === "Escape") {
|
||||
e.stopPropagation();
|
||||
setEditingId(null);
|
||||
}
|
||||
}}
|
||||
onBlur={() => setEditingId(null)}
|
||||
className="min-w-0 flex-1 border border-line bg-card-deep px-1.5 py-0.5 text-[13px] text-ink outline-none"
|
||||
/>
|
||||
) : (
|
||||
<span
|
||||
className={`min-w-0 flex-1 truncate text-[13px] ${
|
||||
active ? "text-ink" : "text-ink-3"
|
||||
}`}
|
||||
>
|
||||
{s.title}
|
||||
</span>
|
||||
)}
|
||||
{working && (
|
||||
<span
|
||||
className="flex-none border border-line bg-accent px-1 py-0.5 text-[9px] leading-none text-white"
|
||||
title="Session en cours de travail"
|
||||
>
|
||||
EN COURS
|
||||
</span>
|
||||
)}
|
||||
<button
|
||||
onClick={(e) => {
|
||||
e.stopPropagation();
|
||||
setEditingId(s.id);
|
||||
setDraft(s.title);
|
||||
}}
|
||||
className="hidden flex-none text-[11px] text-on-dark-3 hover:text-ink group-hover:block"
|
||||
title="Renommer la session"
|
||||
>
|
||||
✎
|
||||
</button>
|
||||
<button
|
||||
onClick={(e) => {
|
||||
e.stopPropagation();
|
||||
if (window.confirm("Supprimer cette session ?")) {
|
||||
void removeSession(s.id);
|
||||
}
|
||||
}}
|
||||
className="hidden flex-none text-on-dark-3 hover:text-accent group-hover:block"
|
||||
title="Supprimer la session"
|
||||
>
|
||||
<TrashIcon size={12} />
|
||||
</button>
|
||||
</div>
|
||||
<div className="mt-0.5 text-[11px] text-on-dark-3">
|
||||
{relTime(s.updated_at)} · {s.message_count ?? 0} message
|
||||
{(s.message_count ?? 0) > 1 ? "s" : ""}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,211 +0,0 @@
|
||||
import { useStore } from "../store/useStore";
|
||||
import {
|
||||
ChatIcon,
|
||||
ClockIcon,
|
||||
LokiMark,
|
||||
NodesIcon,
|
||||
PlusIcon,
|
||||
SettingsIcon,
|
||||
} from "./Icon";
|
||||
|
||||
export type View = "chat" | "history" | "tools" | "settings";
|
||||
|
||||
// Les fichiers ne sont plus une vue : ils vivent dans l'onglet « Fichiers »
|
||||
// du panneau droit, à côté de Code.
|
||||
const NAV: { id: View; label: string; icon: React.ReactNode }[] = [
|
||||
{ id: "chat", label: "Discussion", icon: <ChatIcon /> },
|
||||
{ id: "history", label: "Historique", icon: <ClockIcon /> },
|
||||
{ id: "tools", label: "Outils", icon: <NodesIcon /> },
|
||||
];
|
||||
|
||||
/** Nombre de discussions récentes listées sous « Discussion ». */
|
||||
const RECENT = 5;
|
||||
|
||||
/** Barre latérale (264 px) : marque, navigation, état Ollama, compte. */
|
||||
export function Sidebar({
|
||||
active,
|
||||
onChange,
|
||||
}: {
|
||||
active: View;
|
||||
onChange: (v: View) => void;
|
||||
}) {
|
||||
const {
|
||||
status,
|
||||
selectedModel,
|
||||
systemStats,
|
||||
sessions,
|
||||
currentSessionId,
|
||||
streamingSessionId,
|
||||
newSession,
|
||||
openSession,
|
||||
} = useStore();
|
||||
const gpu = systemStats?.gpu ?? null;
|
||||
const recent = sessions.slice(0, RECENT);
|
||||
|
||||
const startChat = () => {
|
||||
void newSession();
|
||||
onChange("chat");
|
||||
};
|
||||
|
||||
const pick = (id: string) => {
|
||||
void openSession(id);
|
||||
onChange("chat");
|
||||
};
|
||||
|
||||
return (
|
||||
<aside className="flex w-[264px] flex-none flex-col border-r border-line bg-panel px-3.5 py-4">
|
||||
{/* Marque */}
|
||||
<div className="flex items-center gap-2.5 px-1.5 pb-4">
|
||||
<LokiMark size={26} />
|
||||
<span className="text-[16px] font-medium tracking-tight text-ink">Loki</span>
|
||||
<span className="ml-auto rounded-full border border-line px-2 py-0.5 text-[11px] text-muted-3">
|
||||
local
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<div className="mx-1.5 mb-3.5 h-px bg-line" />
|
||||
|
||||
<nav className="flex flex-col gap-[3px]">
|
||||
<NavItem
|
||||
id="chat"
|
||||
label={NAV[0].label}
|
||||
icon={NAV[0].icon}
|
||||
active={active === "chat"}
|
||||
onClick={() => onChange("chat")}
|
||||
/>
|
||||
|
||||
{/* Discussions récentes, directement sous « Discussion ». */}
|
||||
<div className="mb-1 ml-2 flex flex-col gap-px border-l border-line pl-2">
|
||||
<button
|
||||
onClick={startChat}
|
||||
className="flex items-center gap-1.5 rounded-card px-2 py-1.5 text-left text-[13px] text-accent hover:bg-accent-ghost"
|
||||
>
|
||||
<PlusIcon />
|
||||
Nouvelle discussion
|
||||
</button>
|
||||
{recent.map((s) => (
|
||||
<button
|
||||
key={s.id}
|
||||
onClick={() => pick(s.id)}
|
||||
title={s.title}
|
||||
className={`flex items-center gap-1.5 truncate rounded-card px-2 py-1.5 text-left text-[13px] ${
|
||||
s.id === currentSessionId
|
||||
? "bg-accent-ghost text-ink"
|
||||
: "text-muted hover:bg-accent-ghost hover:text-ink"
|
||||
}`}
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate">{s.title}</span>
|
||||
{s.id === streamingSessionId && (
|
||||
<span className="h-1.5 w-1.5 flex-none rounded-full bg-accent" />
|
||||
)}
|
||||
</button>
|
||||
))}
|
||||
{recent.length === 0 && (
|
||||
<span className="px-2 py-1.5 text-[12.5px] text-muted-3">
|
||||
Aucune discussion
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{NAV.slice(1).map((it) => (
|
||||
<NavItem
|
||||
key={it.id}
|
||||
{...it}
|
||||
active={active === it.id}
|
||||
onClick={() => onChange(it.id)}
|
||||
/>
|
||||
))}
|
||||
</nav>
|
||||
|
||||
<div className="font-pixel px-2.5 pb-2 pt-5 text-[10px] text-label">
|
||||
Configuration
|
||||
</div>
|
||||
<NavItem
|
||||
id="settings"
|
||||
label="Modèle & génération"
|
||||
icon={<SettingsIcon />}
|
||||
active={active === "settings"}
|
||||
onClick={() => onChange("settings")}
|
||||
/>
|
||||
|
||||
<div className="flex-1" />
|
||||
|
||||
{/* État Ollama + GPU */}
|
||||
<div className="rounded-card border border-line p-3">
|
||||
<div className="mb-2 flex items-center gap-2">
|
||||
<span
|
||||
className={`h-1.5 w-1.5 rounded-full ${
|
||||
status?.connected ? "bg-accent" : "bg-warn"
|
||||
}`}
|
||||
/>
|
||||
<span className="text-[12.5px] text-ink-3">
|
||||
{status?.connected ? "Ollama connecté" : "Ollama déconnecté"}
|
||||
</span>
|
||||
</div>
|
||||
<div className="mb-2.5 truncate font-mono text-[12px] text-muted-3">
|
||||
{selectedModel || "aucun modèle"}
|
||||
</div>
|
||||
{gpu && (
|
||||
<>
|
||||
<div className="h-1 overflow-hidden rounded-sm bg-line-soft">
|
||||
<span
|
||||
className="block h-full bg-accent"
|
||||
style={{
|
||||
width: `${Math.min(
|
||||
100,
|
||||
(gpu.vram_used_mb / Math.max(gpu.vram_total_mb, 1)) * 100
|
||||
)}%`,
|
||||
}}
|
||||
/>
|
||||
</div>
|
||||
<div className="mt-2 text-[11.5px] text-muted-3">
|
||||
VRAM {(gpu.vram_used_mb / 1000).toFixed(1)} /{" "}
|
||||
{(gpu.vram_total_mb / 1000).toFixed(1)} Go · GPU{" "}
|
||||
{Math.round(gpu.util_pct)} %
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Compte local */}
|
||||
<div className="mt-3 flex items-center gap-2.5 rounded-card border border-line px-3 py-2.5">
|
||||
<span className="grid h-[30px] w-[30px] flex-none place-items-center rounded-full bg-line-soft text-[12px] text-ink-3">
|
||||
M
|
||||
</span>
|
||||
<span className="min-w-0">
|
||||
<span className="block text-[13.5px] text-ink">Compte local</span>
|
||||
<span className="block truncate font-mono text-[11.5px] text-muted-3">
|
||||
{status?.host || "workspace"}
|
||||
</span>
|
||||
</span>
|
||||
</div>
|
||||
</aside>
|
||||
);
|
||||
}
|
||||
|
||||
function NavItem({
|
||||
label,
|
||||
icon,
|
||||
active,
|
||||
onClick,
|
||||
}: {
|
||||
id: View;
|
||||
label: string;
|
||||
icon: React.ReactNode;
|
||||
active: boolean;
|
||||
onClick: () => void;
|
||||
}) {
|
||||
return (
|
||||
<button
|
||||
onClick={onClick}
|
||||
className={`flex w-full items-center gap-2.5 rounded-card border px-2.5 py-2 text-left text-[14px] ${
|
||||
active
|
||||
? "border-accent-line bg-accent-ghost text-ink"
|
||||
: "border-transparent text-ink-3 hover:bg-accent-ghost hover:text-ink"
|
||||
}`}
|
||||
>
|
||||
{icon}
|
||||
{label}
|
||||
</button>
|
||||
);
|
||||
}
|
||||
@@ -1,114 +0,0 @@
|
||||
import type { ToolCall } from "../api/client";
|
||||
|
||||
/** Carte d'appel d'outil rendue dans le fil (fidèle à la maquette). */
|
||||
export function ToolCard({ call }: { call: ToolCall }) {
|
||||
const running = call.status === "running";
|
||||
const error = call.status === "error";
|
||||
const pending = call.status === "pending";
|
||||
|
||||
// Argument principal affiché entre parenthèses selon l'outil.
|
||||
const mainArg =
|
||||
(call.args?.path as string) ??
|
||||
(call.args?.query as string) ??
|
||||
(call.args?.pattern as string) ??
|
||||
(call.args?.command as string) ??
|
||||
(call.args?.instruction as string);
|
||||
const argPreview =
|
||||
typeof mainArg === "string"
|
||||
? `("${mainArg.length > 48 ? mainArg.slice(0, 48) + "…" : mainArg}")`
|
||||
: Object.keys(call.args ?? {}).length
|
||||
? "(…)"
|
||||
: "()";
|
||||
|
||||
const write = call.name === "write_file" || call.name === "edit_file";
|
||||
|
||||
return (
|
||||
<div className="mb-[11px] overflow-hidden border border-line bg-card shadow-hard-sm" style={{ borderRadius: 7 }}>
|
||||
<div className="flex items-center gap-2.5 px-3 py-2.5">
|
||||
<span
|
||||
className={`flex h-[26px] w-[26px] items-center justify-center border border-line ${
|
||||
write ? "bg-accent text-white" : "bg-card-deep text-ink"
|
||||
}`}
|
||||
>
|
||||
<ToolGlyph name={call.name} />
|
||||
</span>
|
||||
<span className="text-[14px] text-ink">{call.name}</span>
|
||||
<span className="text-[13px] text-muted-3">{argPreview}</span>
|
||||
<div className="flex-1" />
|
||||
{running ? (
|
||||
<span className="flex items-center gap-1.5 text-[13px] text-muted-2">
|
||||
<span className="h-2.5 w-2.5 animate-spin rounded-full border border-muted-3 border-t-accent" />
|
||||
en cours
|
||||
</span>
|
||||
) : pending ? (
|
||||
<span className="text-[13px] text-accent">⏸ à valider</span>
|
||||
) : (
|
||||
<span
|
||||
className={`flex items-center gap-1.5 text-[13px] ${
|
||||
error ? "text-warn" : "text-ok"
|
||||
}`}
|
||||
>
|
||||
{error ? "✕ échec" : "✓ terminé"}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{call.summary && !running && (
|
||||
<div className="border-t border-line-soft px-3 py-2 text-[13px] text-muted-2">
|
||||
→ {call.summary}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function ToolGlyph({ name }: { name: string }) {
|
||||
const common = {
|
||||
width: 13,
|
||||
height: 13,
|
||||
viewBox: "0 0 24 24",
|
||||
fill: "none",
|
||||
stroke: "currentColor",
|
||||
strokeWidth: 1.9,
|
||||
strokeLinecap: "round" as const,
|
||||
strokeLinejoin: "round" as const,
|
||||
};
|
||||
if (name === "write_file" || name === "edit_file")
|
||||
return (
|
||||
<svg {...common}>
|
||||
<path d="M5 19h14M7 14l9-9 3 3-9 9-4 1z" />
|
||||
</svg>
|
||||
);
|
||||
if (name === "list_dir")
|
||||
return (
|
||||
<svg {...common}>
|
||||
<path d="M4 7h6l2 2h8v9H4z" />
|
||||
</svg>
|
||||
);
|
||||
if (name === "code_task")
|
||||
return (
|
||||
<svg {...common}>
|
||||
<path d="m8 6-5 6 5 6M16 6l5 6-5 6" />
|
||||
</svg>
|
||||
);
|
||||
if (name === "web_search" || name === "grep_search")
|
||||
return (
|
||||
<svg {...common}>
|
||||
<circle cx="11" cy="11" r="7" />
|
||||
<path d="m21 21-4.3-4.3" />
|
||||
</svg>
|
||||
);
|
||||
if (name === "run_shell")
|
||||
return (
|
||||
<svg {...common}>
|
||||
<path d="m6 9 3 3-3 3M13 15h5" />
|
||||
<rect x="2" y="4" width="20" height="16" rx="2" />
|
||||
</svg>
|
||||
);
|
||||
// read_file (défaut)
|
||||
return (
|
||||
<svg {...common}>
|
||||
<path d="M7 4h7l4 4v12H7z" />
|
||||
<path d="M14 4v4h4" />
|
||||
</svg>
|
||||
);
|
||||
}
|
||||
@@ -1,69 +0,0 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { getVersion } from "../api/client";
|
||||
import { SessionMenu } from "./SessionMenu";
|
||||
|
||||
/** Barre supérieure : conversation courante, ressources, statut Ollama. */
|
||||
export function TopBar() {
|
||||
const status = useStore((s) => s.status);
|
||||
const stats = useStore((s) => s.systemStats);
|
||||
const connected = status?.connected ?? false;
|
||||
const [version, setVersion] = useState("");
|
||||
|
||||
useEffect(() => {
|
||||
getVersion().then(setVersion);
|
||||
}, []);
|
||||
|
||||
return (
|
||||
<div className="flex h-[54px] flex-none items-center gap-3.5 border-b border-line bg-bar px-4">
|
||||
{/* La marque vit désormais dans la barre latérale : on ne garde ici que
|
||||
le contexte de la conversation en cours. */}
|
||||
<SessionMenu />
|
||||
{version && version !== "dev" && (
|
||||
<span
|
||||
className="text-[12px] text-muted-3"
|
||||
title={`build ${version}`}
|
||||
>
|
||||
{version.slice(0, 7)}
|
||||
</span>
|
||||
)}
|
||||
|
||||
<div className="flex-1" />
|
||||
|
||||
{/* Stats système temps réel : CPU, RAM, GPU/VRAM */}
|
||||
{stats && (
|
||||
<div className="flex h-8 items-center gap-2.5 border border-chrome-3 bg-chrome-2 px-[11px] text-[13px] text-on-dark">
|
||||
<span>CPU {stats.cpu_pct.toFixed(0)}%</span>
|
||||
<span className="text-on-dark-3">·</span>
|
||||
<span>RAM {stats.ram_pct.toFixed(0)}%</span>
|
||||
{stats.gpu && (
|
||||
<>
|
||||
<span className="text-on-dark-3">·</span>
|
||||
<span>GPU {stats.gpu.util_pct.toFixed(0)}%</span>
|
||||
<span className="text-on-dark-3">
|
||||
VRAM {(stats.gpu.vram_used_mb / 1024).toFixed(1)}/
|
||||
{(stats.gpu.vram_total_mb / 1024).toFixed(1)}G
|
||||
</span>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Statut Ollama */}
|
||||
<div
|
||||
className="flex h-8 items-center gap-2 border border-chrome-3 bg-chrome-2 px-[11px]"
|
||||
title={connected ? status?.host : status?.error}
|
||||
>
|
||||
<span
|
||||
className={`h-[11px] w-[11px] border border-line ${
|
||||
connected ? "bg-ok" : "bg-warn"
|
||||
}`}
|
||||
/>
|
||||
<span className="text-[13px] text-on-dark">Ollama</span>
|
||||
<span className="text-[13px] text-on-dark-3">
|
||||
{connected ? status?.host.replace(/^https?:\/\//, "") : "déconnecté"}
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,56 +0,0 @@
|
||||
@tailwind base;
|
||||
@tailwind components;
|
||||
@tailwind utilities;
|
||||
|
||||
html,
|
||||
body,
|
||||
#root {
|
||||
height: 100%;
|
||||
margin: 0;
|
||||
}
|
||||
|
||||
body {
|
||||
background: #f7f8fa;
|
||||
color: #1c2536;
|
||||
font-family: "Inter", ui-sans-serif, system-ui, sans-serif;
|
||||
-webkit-font-smoothing: antialiased;
|
||||
}
|
||||
|
||||
/* Titres : léger resserrement, comme dans la maquette Nocturne. */
|
||||
h1,
|
||||
h2,
|
||||
h3 {
|
||||
letter-spacing: -0.025em;
|
||||
}
|
||||
|
||||
/* Ex-police pixel : « kicker » de la maquette (petites capitales espacées). */
|
||||
.font-pixel {
|
||||
font-family: "Inter", ui-sans-serif, system-ui, sans-serif;
|
||||
font-weight: 500;
|
||||
letter-spacing: 0.14em;
|
||||
text-transform: uppercase;
|
||||
}
|
||||
|
||||
/* Barres de défilement discrètes (règles .scr de la maquette) */
|
||||
.scr::-webkit-scrollbar {
|
||||
width: 10px;
|
||||
height: 10px;
|
||||
}
|
||||
.scr::-webkit-scrollbar-thumb {
|
||||
background: #c3c9d2;
|
||||
border-radius: 6px;
|
||||
border: 3px solid transparent;
|
||||
background-clip: content-box;
|
||||
}
|
||||
.scr::-webkit-scrollbar-track {
|
||||
background: transparent;
|
||||
}
|
||||
|
||||
/* Focus visible accentué, sélection teintée — repris de la maquette. */
|
||||
:focus-visible {
|
||||
outline: 2px solid #2f4a7a;
|
||||
outline-offset: 2px;
|
||||
}
|
||||
::selection {
|
||||
background: rgba(47, 74, 122, 0.32);
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
/** Date relative courte en français (« il y a 5 min »). */
|
||||
export function relTime(ts: number): string {
|
||||
const diff = Date.now() / 1000 - ts;
|
||||
if (diff < 60) return "à l'instant";
|
||||
if (diff < 3600) return `il y a ${Math.floor(diff / 60)} min`;
|
||||
if (diff < 86400) return `il y a ${Math.floor(diff / 3600)} h`;
|
||||
return `il y a ${Math.floor(diff / 86400)} j`;
|
||||
}
|
||||
@@ -1,10 +0,0 @@
|
||||
import React from "react";
|
||||
import ReactDOM from "react-dom/client";
|
||||
import App from "./App";
|
||||
import "./index.css";
|
||||
|
||||
ReactDOM.createRoot(document.getElementById("root")!).render(
|
||||
<React.StrictMode>
|
||||
<App />
|
||||
</React.StrictMode>
|
||||
);
|
||||
@@ -1,94 +0,0 @@
|
||||
import { useEffect } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
|
||||
export function HistoryView({ onOpen }: { onOpen: () => void }) {
|
||||
const { sessions, refreshSessions, openSession, removeSession } = useStore();
|
||||
|
||||
useEffect(() => {
|
||||
refreshSessions();
|
||||
}, [refreshSessions]);
|
||||
|
||||
return (
|
||||
<Page title="Historique" subtitle="Toutes les conversations enregistrées localement.">
|
||||
<div className="grid grid-cols-1 gap-3 lg:grid-cols-2 xl:grid-cols-3">
|
||||
{sessions.map((session) => (
|
||||
<div key={session.id} className="border border-line bg-card p-4 shadow-hard">
|
||||
<div className="truncate text-[15px] font-semibold text-ink">{session.title}</div>
|
||||
<div className="mt-1 text-[12px] text-muted-2">
|
||||
{session.message_count ?? 0} message(s) · {new Date(session.updated_at * 1000).toLocaleString("fr-FR")}
|
||||
</div>
|
||||
<div className="mt-4 flex gap-2">
|
||||
<button
|
||||
onClick={async () => {
|
||||
await openSession(session.id);
|
||||
onOpen();
|
||||
}}
|
||||
className="border border-line bg-accent px-3 py-1.5 text-[12px] font-bold text-white"
|
||||
>
|
||||
Ouvrir
|
||||
</button>
|
||||
<button
|
||||
onClick={() => removeSession(session.id)}
|
||||
className="border border-line bg-base px-3 py-1.5 text-[12px] text-warn"
|
||||
>
|
||||
Supprimer
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
{sessions.length === 0 && <Empty text="Aucune conversation enregistrée." />}
|
||||
</div>
|
||||
</Page>
|
||||
);
|
||||
}
|
||||
|
||||
export function ToolsView({ onSettings }: { onSettings: () => void }) {
|
||||
const { config, availableTools, refreshConfig, selectedModel } = useStore();
|
||||
|
||||
useEffect(() => {
|
||||
refreshConfig();
|
||||
}, [refreshConfig, selectedModel]);
|
||||
|
||||
return (
|
||||
<Page title="Outils" subtitle={`Capacités proposées à ${selectedModel || "l’agent"}.`}>
|
||||
<div className="grid grid-cols-1 gap-3 md:grid-cols-2 xl:grid-cols-3">
|
||||
{availableTools.map((name) => {
|
||||
const enabled = config?.tools[name] ?? false;
|
||||
return (
|
||||
<div key={name} className="border border-line bg-card p-4 shadow-hard">
|
||||
<div className="flex items-center justify-between gap-3">
|
||||
<span className="font-mono text-[14px] text-ink">{name}</span>
|
||||
<span className={`border border-line px-2 py-1 text-[10px] ${enabled ? "bg-ok text-white" : "bg-base text-muted-2"}`}>
|
||||
{enabled ? "ACTIF" : "INACTIF"}
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
<button
|
||||
onClick={onSettings}
|
||||
className="mt-5 border border-line bg-accent px-4 py-2 text-[13px] font-bold text-white shadow-hard"
|
||||
>
|
||||
Configurer les outils
|
||||
</button>
|
||||
</Page>
|
||||
);
|
||||
}
|
||||
|
||||
function Page({ title, subtitle, children }: { title: string; subtitle: string; children: React.ReactNode }) {
|
||||
return (
|
||||
<div className="scr min-w-0 flex-1 overflow-auto bg-base p-7">
|
||||
<div className="mx-auto max-w-[1100px]">
|
||||
<h1 className="m-0 text-xl font-bold text-ink">{title}</h1>
|
||||
<p className="mb-6 mt-1 text-[13px] text-muted-2">{subtitle}</p>
|
||||
{children}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Empty({ text }: { text: string }) {
|
||||
return <div className="border border-line bg-card p-8 text-center text-muted-2">{text}</div>;
|
||||
}
|
||||
|
||||
@@ -1,549 +0,0 @@
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { ChevronDown, LokiMark, SendIcon } from "../components/Icon";
|
||||
import { ToolCard } from "../components/ToolCard";
|
||||
import { MessageContent } from "../components/MessageContent";
|
||||
import { ProjectChip } from "../components/ProjectChip";
|
||||
import { ModelSelector } from "../components/ModelSelector";
|
||||
import type { Message, ToolCall } from "../api/client";
|
||||
|
||||
/** Panneau central : barre de contexte, fil de conversation, composer. */
|
||||
export function ChatPanel() {
|
||||
const {
|
||||
selectedModel,
|
||||
messages,
|
||||
streaming,
|
||||
streamingSessionId,
|
||||
streamContent,
|
||||
streamThinking,
|
||||
streamStatus,
|
||||
streamNotice,
|
||||
streamTools,
|
||||
streamPlan,
|
||||
streamPlanDone,
|
||||
sendMessage,
|
||||
currentSessionId,
|
||||
config,
|
||||
pendingShell,
|
||||
approveShell,
|
||||
rejectShell,
|
||||
sessions,
|
||||
stopStreaming,
|
||||
} = useStore();
|
||||
|
||||
const activeTools = config
|
||||
? Object.values(config.tools).filter(Boolean).length
|
||||
: 0;
|
||||
|
||||
const [draft, setDraft] = useState("");
|
||||
const scrollRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
// Auto-scroll vers le bas à chaque token / message.
|
||||
useEffect(() => {
|
||||
scrollRef.current?.scrollTo({ top: scrollRef.current.scrollHeight });
|
||||
}, [messages, streamContent, streamThinking, streamTools, streamPlan, streamPlanDone]);
|
||||
|
||||
const submit = () => {
|
||||
if (!draft.trim() || streaming) return;
|
||||
sendMessage(draft);
|
||||
setDraft("");
|
||||
};
|
||||
|
||||
const onKeyDown = (e: React.KeyboardEvent) => {
|
||||
if (e.key === "Enter" && !e.shiftKey) {
|
||||
e.preventDefault();
|
||||
submit();
|
||||
}
|
||||
};
|
||||
|
||||
const showingStreaming = streaming && currentSessionId === streamingSessionId;
|
||||
const workingSession = sessions.find((s) => s.id === streamingSessionId);
|
||||
const empty = messages.length === 0 && !showingStreaming;
|
||||
|
||||
return (
|
||||
<div className="flex min-w-0 flex-1 flex-col bg-base">
|
||||
{/* Barre de contexte */}
|
||||
<div className="flex h-[46px] flex-none items-center gap-2 border-b border-line bg-panel px-4">
|
||||
<Chip>⚑ Invite système</Chip>
|
||||
<Chip>
|
||||
<span className="h-2 w-2 border border-line bg-ok" />
|
||||
{activeTools} outil{activeTools > 1 ? "s" : ""}
|
||||
</Chip>
|
||||
<Chip>
|
||||
Temp{" "}
|
||||
<b className="text-accent">{config ? config.temperature.toFixed(1) : "—"}</b>
|
||||
</Chip>
|
||||
<div className="flex-1" />
|
||||
<span className="text-[13px] text-muted-3">
|
||||
{currentSessionId
|
||||
? `${messages.length} message${messages.length > 1 ? "s" : ""}`
|
||||
: "aucune session"}
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{/* Messages */}
|
||||
<div ref={scrollRef} className="scr flex-1 overflow-auto px-7 py-6">
|
||||
{empty ? (
|
||||
<Welcome
|
||||
sessionCount={sessions.length}
|
||||
onPick={(text) => setDraft(text)}
|
||||
/>
|
||||
) : (
|
||||
<div className="mx-auto flex max-w-[680px] flex-col gap-5">
|
||||
{messages.map((m) => (
|
||||
<Bubble key={m.id} msg={m} />
|
||||
))}
|
||||
{showingStreaming && (
|
||||
<Bubble
|
||||
msg={{
|
||||
id: "stream",
|
||||
session_id: "",
|
||||
role: "assistant",
|
||||
content: streamContent,
|
||||
model: selectedModel,
|
||||
meta: { tools: streamTools, plan: streamPlan },
|
||||
created_at: Date.now() / 1000,
|
||||
}}
|
||||
pending
|
||||
pendingStatus={streamStatus}
|
||||
notice={streamNotice}
|
||||
thinking={streamThinking}
|
||||
planDone={streamPlanDone}
|
||||
/>
|
||||
)}
|
||||
{/* La validation shell persiste APRÈS la fin du flux : l'agent
|
||||
termine son tour en attendant l'utilisateur, donc le streaming
|
||||
s'arrête — la carte doit rester tant qu'on n'a pas tranché. */}
|
||||
{pendingShell && (
|
||||
<ShellConfirm
|
||||
command={pendingShell}
|
||||
onApprove={approveShell}
|
||||
onReject={rejectShell}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Composer */}
|
||||
<div className="flex-none border-t border-line bg-panel px-7 pb-[18px] pt-3.5">
|
||||
<div className="mx-auto max-w-[680px]">
|
||||
<div className="border border-line bg-card p-3 shadow-hard" style={{ borderRadius: 8 }}>
|
||||
<textarea
|
||||
rows={1}
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onKeyDown={onKeyDown}
|
||||
placeholder="Envoyer un message à l'agent…"
|
||||
className="min-h-[40px] w-full resize-none bg-transparent text-[14px] leading-relaxed text-ink outline-none placeholder:text-muted-3"
|
||||
/>
|
||||
<div className="mt-1.5 flex items-center gap-2">
|
||||
<ModeSelector />
|
||||
<ProjectChip />
|
||||
<ModelSelector variant="composer" />
|
||||
{streaming && !showingStreaming ? (
|
||||
<span className="min-w-0 flex-1 truncate text-[13px] text-accent">
|
||||
Travail en cours : {workingSession?.title ?? "session ouverte"}
|
||||
</span>
|
||||
) : (
|
||||
<>
|
||||
<div className="flex-1" />
|
||||
<span className="text-[13px] text-muted-3">⏎ envoyer · ⇧⏎ ligne</span>
|
||||
</>
|
||||
)}
|
||||
<button
|
||||
onClick={streaming ? stopStreaming : submit}
|
||||
disabled={!streaming && !draft.trim()}
|
||||
className={`flex h-[38px] items-center gap-1.5 border border-line px-4 text-[14px] text-white shadow-hard-accent disabled:opacity-40 ${
|
||||
streaming ? "bg-warn" : "bg-accent"
|
||||
}`}
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
{streaming ? "ARRÊTER" : "ENVOYER"}
|
||||
{!streaming && <SendIcon />}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/** Pistes proposées sur l'écran d'accueil (maquette « Loki App »). */
|
||||
const STARTERS = [
|
||||
["📄", "Résumer un dossier", "Résume le contenu du dossier "],
|
||||
["🐞", "Corriger un bug", "Corrige le bug suivant : "],
|
||||
["✏️", "Refactorer un fichier", "Refactore le fichier "],
|
||||
["⌨️", "Écrire un script", "Écris un script qui "],
|
||||
["🔍", "Chercher dans le code", "Cherche dans le code "],
|
||||
["📊", "Analyser un CSV", "Analyse le fichier CSV "],
|
||||
["🌿", "Préparer un commit", "Prépare un commit pour "],
|
||||
["📥", "Générer un rapport", "Génère un rapport sur "],
|
||||
] as const;
|
||||
|
||||
/** Écran d'accueil : accroche, pistes cliquables, renvoi vers l'historique. */
|
||||
function Welcome({
|
||||
sessionCount,
|
||||
onPick,
|
||||
}: {
|
||||
sessionCount: number;
|
||||
onPick: (text: string) => void;
|
||||
}) {
|
||||
return (
|
||||
<div className="flex h-full flex-col items-center justify-center px-7 py-10 text-center">
|
||||
<LokiMark size={44} />
|
||||
<h1 className="mt-6 text-[34px] font-medium text-ink">
|
||||
Bienvenue dans Loki
|
||||
</h1>
|
||||
<p className="mt-3 max-w-[52ch] text-[15.5px] leading-relaxed text-muted">
|
||||
Choisissez une piste ci-dessous, ou écrivez directement votre demande.
|
||||
Tout s'exécute sur cette machine, aucune donnée ne sort.
|
||||
</p>
|
||||
|
||||
<div className="mt-8 flex max-w-[760px] flex-wrap justify-center gap-2.5">
|
||||
{STARTERS.map(([icon, label, prefill]) => (
|
||||
<button
|
||||
key={label}
|
||||
onClick={() => onPick(prefill)}
|
||||
className="flex items-center gap-2 whitespace-nowrap rounded-full border border-line px-3.5 py-2 text-[13.5px] text-ink-2 hover:border-accent hover:bg-accent-ghost"
|
||||
>
|
||||
<span aria-hidden>{icon}</span>
|
||||
{label}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
|
||||
{sessionCount > 0 && (
|
||||
<div className="mt-9 flex items-center gap-3 text-[12.5px] text-muted-3">
|
||||
<span className="h-px w-11 bg-line" />
|
||||
{sessionCount} conversation{sessionCount > 1 ? "s" : ""} enregistrée
|
||||
{sessionCount > 1 ? "s" : ""} localement
|
||||
<span className="h-px w-11 bg-line" />
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/** Plan d'exécution affiché avant le travail de l'agent. */
|
||||
const MODES = [
|
||||
{ id: "plan", label: "Plan", icon: "🔍", desc: "Lecture seule : analyse et propose, sans rien modifier" },
|
||||
{ id: "build", label: "Build", icon: "🔨", desc: "Normal : écrit les fichiers, confirme les commandes shell" },
|
||||
{ id: "yolo", label: "Yolo", icon: "⚡", desc: "Auto : approuve tout, y compris le shell" },
|
||||
] as const;
|
||||
|
||||
/** Sélecteur de mode d'exécution (Plan / Build / Yolo) dans le composer. */
|
||||
function ModeSelector() {
|
||||
const { mode, setMode } = useStore();
|
||||
const [open, setOpen] = useState(false);
|
||||
const ref = useRef<HTMLDivElement>(null);
|
||||
|
||||
useEffect(() => {
|
||||
const onClick = (e: MouseEvent) => {
|
||||
if (ref.current && !ref.current.contains(e.target as Node)) setOpen(false);
|
||||
};
|
||||
document.addEventListener("mousedown", onClick);
|
||||
return () => document.removeEventListener("mousedown", onClick);
|
||||
}, []);
|
||||
|
||||
const current = MODES.find((m) => m.id === mode) ?? MODES[1];
|
||||
|
||||
return (
|
||||
<div className="relative" ref={ref}>
|
||||
<button
|
||||
onClick={() => setOpen((v) => !v)}
|
||||
className={`flex h-8 items-center gap-1.5 border border-line px-2.5 text-[13px] ${
|
||||
mode === "plan"
|
||||
? "bg-info text-white"
|
||||
: mode === "yolo"
|
||||
? "bg-accent text-white"
|
||||
: "bg-card text-ink-2"
|
||||
}`}
|
||||
title={current.desc}
|
||||
>
|
||||
<span>{current.icon}</span>
|
||||
<span>{current.label}</span>
|
||||
<ChevronDown size={11} />
|
||||
</button>
|
||||
{open && (
|
||||
<div className="absolute bottom-10 left-0 z-20 w-64 border border-line bg-card p-1 shadow-hard">
|
||||
{MODES.map((m) => (
|
||||
<button
|
||||
key={m.id}
|
||||
onClick={() => {
|
||||
setMode(m.id);
|
||||
setOpen(false);
|
||||
}}
|
||||
className={`flex w-full flex-col items-start gap-0.5 px-2 py-2 text-left hover:bg-base ${
|
||||
m.id === mode ? "bg-base" : ""
|
||||
}`}
|
||||
>
|
||||
<span className="text-[13px] text-ink">
|
||||
{m.icon} {m.label}
|
||||
</span>
|
||||
<span className="text-[11px] leading-tight text-muted-2">{m.desc}</span>
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function PlanCard({
|
||||
steps,
|
||||
done,
|
||||
live,
|
||||
}: {
|
||||
steps: string[];
|
||||
done?: number[];
|
||||
live?: boolean;
|
||||
}) {
|
||||
const doneSet = new Set(done ?? []);
|
||||
// Étape active = la première non validée, uniquement pendant le stream.
|
||||
const activeIdx = live
|
||||
? steps.findIndex((_, i) => !doneSet.has(i))
|
||||
: -1;
|
||||
const doneCount = doneSet.size;
|
||||
|
||||
return (
|
||||
<div
|
||||
className="mb-[11px] overflow-hidden border border-line bg-card shadow-hard-sm"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
<div className="flex items-center gap-2 border-b border-line-soft px-3 py-2">
|
||||
<span className="font-pixel text-[9px] text-accent">PLAN</span>
|
||||
<span className="text-[12px] text-muted-2">
|
||||
{doneCount > 0
|
||||
? `${doneCount}/${steps.length} validée${doneCount > 1 ? "s" : ""}`
|
||||
: `${steps.length} étape${steps.length > 1 ? "s" : ""}`}
|
||||
</span>
|
||||
</div>
|
||||
<ol className="m-0 list-none px-3 py-2">
|
||||
{steps.map((s, i) => {
|
||||
const isDone = doneSet.has(i);
|
||||
const isActive = i === activeIdx;
|
||||
return (
|
||||
<li
|
||||
key={i}
|
||||
className={`flex gap-2 py-[3px] text-[13px] ${
|
||||
isDone ? "text-muted-2" : "text-ink-2"
|
||||
}`}
|
||||
>
|
||||
<span
|
||||
className={`flex h-[18px] w-[18px] flex-none items-center justify-center border text-[11px] ${
|
||||
isDone
|
||||
? "border-ok bg-ok text-white"
|
||||
: isActive
|
||||
? "border-accent bg-base text-accent"
|
||||
: "border-line bg-base text-ink"
|
||||
}`}
|
||||
>
|
||||
{isDone ? "✓" : isActive ? "…" : i + 1}
|
||||
</span>
|
||||
<span className={`min-w-0 ${isDone ? "line-through" : ""}`}>
|
||||
{s}
|
||||
</span>
|
||||
</li>
|
||||
);
|
||||
})}
|
||||
</ol>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Bubble({
|
||||
msg,
|
||||
pending,
|
||||
pendingStatus,
|
||||
notice,
|
||||
thinking,
|
||||
planDone,
|
||||
}: {
|
||||
msg: Message;
|
||||
pending?: boolean;
|
||||
pendingStatus?: string;
|
||||
notice?: string | null;
|
||||
thinking?: string;
|
||||
planDone?: number[];
|
||||
}) {
|
||||
const time = new Date(msg.created_at * 1000).toLocaleTimeString("fr-FR", {
|
||||
hour: "2-digit",
|
||||
minute: "2-digit",
|
||||
});
|
||||
|
||||
if (msg.role === "user") {
|
||||
return (
|
||||
<div className="flex gap-3">
|
||||
<div className="font-pixel flex h-[34px] w-[34px] flex-none items-center justify-center border border-line bg-card-deep text-[11px] text-ink">
|
||||
M
|
||||
</div>
|
||||
<div className="flex-1">
|
||||
<div className="mb-1.5 flex items-baseline gap-2">
|
||||
<span className="text-[14px] text-ink">VOUS</span>
|
||||
<span className="text-[13px] text-muted-3">{time}</span>
|
||||
</div>
|
||||
<div className="border border-line bg-card px-[14px] py-3 text-[14px] leading-snug text-ink shadow-hard-sm whitespace-pre-wrap" style={{ borderRadius: 7 }}>
|
||||
{msg.content}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="flex gap-3">
|
||||
<div className="grid h-[34px] w-[34px] flex-none place-items-center rounded-[9px] border border-accent bg-accent-ghost">
|
||||
<div className="bg-accent" style={{ width: 10, height: 10, borderRadius: 2 }} />
|
||||
</div>
|
||||
<div className="min-w-0 flex-1">
|
||||
<div className="mb-2 flex items-baseline gap-2">
|
||||
<span className="text-[14px] text-ink">LOKI</span>
|
||||
<span className="text-[13px] text-muted-3">
|
||||
{msg.model ? `${msg.model} · ` : ""}
|
||||
{time}
|
||||
</span>
|
||||
</div>
|
||||
<ReasoningPanel
|
||||
text={thinking ?? msg.meta?.thinking ?? ""}
|
||||
live={!!pending}
|
||||
/>
|
||||
{(msg.meta?.plan?.length ?? 0) > 0 && (
|
||||
<PlanCard steps={msg.meta!.plan!} done={planDone} live={!!pending} />
|
||||
)}
|
||||
{(msg.meta?.tools ?? []).map((t: ToolCall, i: number) => (
|
||||
<ToolCard key={i} call={t} />
|
||||
))}
|
||||
{notice && (
|
||||
<div className="mb-2 border border-line bg-card px-3 py-2 text-[13px] text-warn">
|
||||
{notice}
|
||||
</div>
|
||||
)}
|
||||
{(msg.content || pending) && (
|
||||
<div className="text-[14px] leading-[1.5] text-ink-2">
|
||||
{msg.content ? (
|
||||
<MessageContent text={msg.content} />
|
||||
) : (
|
||||
<span className="text-muted-2">{pendingStatus || "Génération…"}</span>
|
||||
)}
|
||||
{pending && (
|
||||
<span className="ml-0.5 inline-block h-3.5 w-[7px] animate-pulse bg-accent align-middle" />
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
{!pending && msg.meta?.stats && (
|
||||
<div className="mt-2 flex items-center gap-2.5 text-[12px] text-muted-3">
|
||||
{msg.meta.stats.tokens_per_sec != null && (
|
||||
<span className="text-accent-2">
|
||||
{msg.meta.stats.tokens_per_sec} tok/s
|
||||
</span>
|
||||
)}
|
||||
<span>{msg.meta.stats.eval_count} jetons</span>
|
||||
{msg.meta.stats.prompt_eval_count > 0 && (
|
||||
<span>· {msg.meta.stats.prompt_eval_count} en entrée</span>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/** Panneau « Raisonnement » : repliable et redimensionnable (poignée en bas). */
|
||||
function ReasoningPanel({ text, live }: { text: string; live: boolean }) {
|
||||
const [open, setOpen] = useState(live);
|
||||
const bodyRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
useEffect(() => {
|
||||
if (live) setOpen(true);
|
||||
}, [live]);
|
||||
|
||||
useEffect(() => {
|
||||
if (open && live) {
|
||||
bodyRef.current?.scrollTo({ top: bodyRef.current.scrollHeight });
|
||||
}
|
||||
}, [text, open, live]);
|
||||
|
||||
if (!text) return null;
|
||||
|
||||
return (
|
||||
<div className="mb-[11px] overflow-hidden border border-line bg-card shadow-hard-sm" style={{ borderRadius: 7 }}>
|
||||
<button
|
||||
onClick={() => setOpen((o) => !o)}
|
||||
className="flex w-full items-center gap-2 px-3 py-2 text-left"
|
||||
>
|
||||
<ChevronDown
|
||||
size={12}
|
||||
className={`text-ink-2 transition-transform ${open ? "" : "-rotate-90"}`}
|
||||
/>
|
||||
<span className="text-[13px] font-medium text-ink">Raisonnement</span>
|
||||
{live && (
|
||||
<span className="flex items-center gap-1 text-[12px] text-accent">
|
||||
<span className="h-2 w-2 animate-pulse border border-line bg-accent" />
|
||||
en cours…
|
||||
</span>
|
||||
)}
|
||||
<span className="ml-auto text-[12px] text-muted-3">
|
||||
{open ? "réduire" : "afficher"}
|
||||
</span>
|
||||
</button>
|
||||
{open && (
|
||||
<div
|
||||
ref={bodyRef}
|
||||
className="scr max-h-[200px] min-h-[60px] resize-y overflow-auto whitespace-pre-wrap border-t border-line-soft bg-card-deep px-3 py-2 text-[12.5px] leading-relaxed text-on-dark-2"
|
||||
>
|
||||
{text}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function ShellConfirm({
|
||||
command,
|
||||
onApprove,
|
||||
onReject,
|
||||
}: {
|
||||
command: string;
|
||||
onApprove: () => void;
|
||||
onReject: () => void;
|
||||
}) {
|
||||
return (
|
||||
<div className="ml-[46px] overflow-hidden border border-line bg-card shadow-hard" style={{ borderRadius: 7 }}>
|
||||
<div className="flex items-center gap-2 border-b border-line px-3 py-2.5">
|
||||
<span className="text-[14px] text-accent">run_shell</span>
|
||||
<span className="text-[13px] text-muted-2">· commande sensible à valider</span>
|
||||
</div>
|
||||
<div className="px-3 py-3">
|
||||
<pre className="m-0 mb-3 overflow-auto whitespace-pre-wrap border border-line bg-card-deep px-3 py-2.5 text-[12.5px] text-on-dark">
|
||||
$ {command}
|
||||
</pre>
|
||||
<div className="flex items-center gap-2">
|
||||
<button
|
||||
onClick={onApprove}
|
||||
className="flex h-[32px] items-center gap-1.5 border border-line bg-accent px-3.5 text-[13px] text-white shadow-hard-accent"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
Approuver & exécuter
|
||||
</button>
|
||||
<button
|
||||
onClick={onReject}
|
||||
className="flex h-[32px] items-center border border-line bg-card px-3.5 text-[13px] text-ink-2"
|
||||
style={{ borderRadius: 7 }}
|
||||
>
|
||||
Refuser
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Chip({ children }: { children: React.ReactNode }) {
|
||||
return (
|
||||
<div className="flex h-7 items-center gap-1.5 border border-line bg-card px-2.5 text-[13px] text-ink-2">
|
||||
{children}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,414 +0,0 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import {
|
||||
downloadFile,
|
||||
getGitDiff,
|
||||
getGitLog,
|
||||
revertCommit,
|
||||
type GitCommit,
|
||||
type ToolCall,
|
||||
} from "../api/client";
|
||||
import { DownloadIcon, TrashIcon } from "../components/Icon";
|
||||
import { FileTree } from "../components/FileTree";
|
||||
|
||||
type TabId = "preview" | "code" | "files" | "logs" | "git";
|
||||
|
||||
/** Panneau droit : onglets Aperçu / Code / Fichiers / Logs / Git. */
|
||||
export function PreviewPanel() {
|
||||
const { previewPath, previewContent, messages, streamTools, removeFile } =
|
||||
useStore();
|
||||
const [tab, setTab] = useState<TabId>("preview");
|
||||
const [width, setWidth] = useState(() => {
|
||||
const saved = Number(window.localStorage.getItem("loki.preview.width"));
|
||||
return Number.isFinite(saved) && saved >= 300 ? saved : 452;
|
||||
});
|
||||
|
||||
const updateWidth = (next: number) => {
|
||||
const bounded = Math.max(300, Math.min(next, window.innerWidth - 520));
|
||||
setWidth(bounded);
|
||||
window.localStorage.setItem("loki.preview.width", String(bounded));
|
||||
};
|
||||
|
||||
const startResize = (event: React.PointerEvent<HTMLDivElement>) => {
|
||||
event.preventDefault();
|
||||
const startX = event.clientX;
|
||||
const startWidth = width;
|
||||
const onMove = (move: PointerEvent) =>
|
||||
updateWidth(startWidth + startX - move.clientX);
|
||||
const onUp = () => {
|
||||
window.removeEventListener("pointermove", onMove);
|
||||
window.removeEventListener("pointerup", onUp);
|
||||
};
|
||||
window.addEventListener("pointermove", onMove);
|
||||
window.addEventListener("pointerup", onUp);
|
||||
};
|
||||
|
||||
const isHtml = previewPath ? /\.html?$/.test(previewPath) : false;
|
||||
|
||||
const [collapsed, setCollapsed] = useState(
|
||||
() => window.localStorage.getItem("loki.preview.collapsed") === "1"
|
||||
);
|
||||
const toggleCollapsed = () => {
|
||||
setCollapsed((c) => {
|
||||
window.localStorage.setItem("loki.preview.collapsed", c ? "0" : "1");
|
||||
return !c;
|
||||
});
|
||||
};
|
||||
|
||||
// Journal d'activité : tous les appels d'outils de la session + en cours.
|
||||
const logs: ToolCall[] = [
|
||||
...messages.flatMap((m) => m.meta?.tools ?? []),
|
||||
...streamTools,
|
||||
];
|
||||
|
||||
// Replié : barre fine, aucun contenu monté (pas d'iframe vivante).
|
||||
if (collapsed) {
|
||||
return (
|
||||
<div className="flex w-9 flex-none flex-col items-center border-l border-line bg-panel pt-3">
|
||||
<button
|
||||
onClick={toggleCollapsed}
|
||||
title="Déplier l'aperçu"
|
||||
className="text-[15px] text-muted-2 hover:text-accent"
|
||||
>
|
||||
⇤
|
||||
</button>
|
||||
<div className="mt-3 rotate-90 whitespace-nowrap text-[10px] font-bold tracking-wide text-muted-3">
|
||||
APERÇU
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div
|
||||
className="relative flex flex-none flex-col border-l border-line bg-panel"
|
||||
style={{ width }}
|
||||
>
|
||||
<div
|
||||
role="separator"
|
||||
aria-label="Redimensionner le panneau d'aperçu"
|
||||
aria-orientation="vertical"
|
||||
tabIndex={0}
|
||||
onPointerDown={startResize}
|
||||
onKeyDown={(event) => {
|
||||
if (event.key === "ArrowLeft") updateWidth(width + 20);
|
||||
if (event.key === "ArrowRight") updateWidth(width - 20);
|
||||
}}
|
||||
className="absolute -left-[6px] top-0 z-20 h-full w-[9px] cursor-col-resize bg-transparent hover:bg-accent/50 focus:bg-accent/50 focus:outline-none"
|
||||
title="Glisser pour redimensionner"
|
||||
/>
|
||||
{/* Onglets */}
|
||||
<div className="flex h-[46px] flex-none items-center gap-1.5 border-b border-line bg-base px-3">
|
||||
<Tab active={tab === "preview"} onClick={() => setTab("preview")}>
|
||||
Aperçu
|
||||
</Tab>
|
||||
<Tab active={tab === "code"} onClick={() => setTab("code")}>
|
||||
Code
|
||||
</Tab>
|
||||
<Tab active={tab === "files"} onClick={() => setTab("files")}>
|
||||
Fichiers
|
||||
</Tab>
|
||||
<Tab active={tab === "logs"} onClick={() => setTab("logs")}>
|
||||
Logs
|
||||
<span className="font-pixel ml-1.5 border border-line bg-accent px-1 py-0.5 text-[8px] text-white">
|
||||
{logs.length}
|
||||
</span>
|
||||
</Tab>
|
||||
<Tab active={tab === "git"} onClick={() => setTab("git")}>
|
||||
Git
|
||||
</Tab>
|
||||
<div className="flex-1" />
|
||||
<button
|
||||
onClick={toggleCollapsed}
|
||||
title="Replier l'aperçu"
|
||||
className="px-2 text-[15px] text-muted-2 hover:text-accent"
|
||||
>
|
||||
⇥
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* URL bar */}
|
||||
<div className="flex flex-none items-center gap-2 px-3 py-2.5">
|
||||
<div className="flex h-8 flex-1 items-center gap-2 border border-line bg-card px-[11px]">
|
||||
<span className="text-[13px]">🔒</span>
|
||||
<span className="truncate text-[13px] text-muted-2">
|
||||
{previewPath ? `workspace/${previewPath}` : "workspace/"}
|
||||
</span>
|
||||
</div>
|
||||
{previewPath && (
|
||||
<>
|
||||
<button
|
||||
onClick={() => downloadFile(previewPath, useStore.getState().currentProject())}
|
||||
className="flex h-8 items-center gap-1.5 border border-line bg-card px-2.5 text-[12px] text-accent"
|
||||
title={`Télécharger ${previewPath}`}
|
||||
>
|
||||
<DownloadIcon size={13} />
|
||||
Télécharger
|
||||
</button>
|
||||
<button
|
||||
onClick={() => {
|
||||
if (window.confirm(`Supprimer ${previewPath} ?`)) {
|
||||
void removeFile(previewPath);
|
||||
}
|
||||
}}
|
||||
className="flex h-8 items-center gap-1.5 border border-line bg-card px-2.5 text-[12px] text-warn"
|
||||
title={`Supprimer ${previewPath}`}
|
||||
>
|
||||
<TrashIcon size={13} />
|
||||
Supprimer
|
||||
</button>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Onglet Git */}
|
||||
{tab === "git" ? (
|
||||
<GitPanel />
|
||||
) : tab === "files" ? (
|
||||
<FilesTab />
|
||||
) : /* Onglet Logs (fond sombre) */
|
||||
tab === "logs" ? (
|
||||
<div className="scr mx-3 mb-3 flex-1 overflow-auto border border-line bg-card-deep p-3 text-[12.5px]">
|
||||
{logs.length === 0 ? (
|
||||
<div className="py-10 text-center text-on-dark-3">
|
||||
Aucune activité d'outil pour cette session.
|
||||
</div>
|
||||
) : (
|
||||
logs.map((l, i) => (
|
||||
<div
|
||||
key={i}
|
||||
className="flex items-center gap-2 border-b border-chrome-2 py-[7px] last:border-0"
|
||||
>
|
||||
<span
|
||||
className={
|
||||
l.status === "error" || l.status === "pending"
|
||||
? "text-accent"
|
||||
: l.status === "running"
|
||||
? "text-on-dark-2"
|
||||
: "text-ok"
|
||||
}
|
||||
>
|
||||
{l.status === "error"
|
||||
? "✕"
|
||||
: l.status === "pending"
|
||||
? "⏸"
|
||||
: l.status === "running"
|
||||
? "…"
|
||||
: "✓"}
|
||||
</span>
|
||||
<span className="text-on-dark">{l.name}</span>
|
||||
<span className="flex-1 truncate text-on-dark-3">
|
||||
{(l.args?.path as string) ??
|
||||
(l.args?.query as string) ??
|
||||
(l.args?.command as string) ??
|
||||
""}
|
||||
</span>
|
||||
<span className="text-on-dark-3">{l.summary}</span>
|
||||
</div>
|
||||
))
|
||||
)}
|
||||
</div>
|
||||
) : (
|
||||
/* Viewport Aperçu / Code (fond clair) */
|
||||
<div className="scr mx-3 mb-3 flex-1 overflow-auto border border-line bg-panel shadow-hard">
|
||||
{!previewPath ? (
|
||||
<Empty>
|
||||
L'aperçu s'affichera ici dès que l'agent générera un fichier (clique
|
||||
aussi un fichier à gauche).
|
||||
</Empty>
|
||||
) : tab === "code" ? (
|
||||
<pre className="m-0 whitespace-pre-wrap p-4 text-[12px] leading-relaxed text-ink">
|
||||
{previewContent}
|
||||
</pre>
|
||||
) : isHtml ? (
|
||||
<iframe
|
||||
title="aperçu"
|
||||
srcDoc={previewContent}
|
||||
className="h-full w-full border-0 bg-white"
|
||||
// allow-scripts : sans ça, le JS de la page générée ne s'exécute pas
|
||||
// (animations, canvas…). On garde une origine opaque (pas
|
||||
// d'allow-same-origin) pour que le script ne puisse pas atteindre Loki.
|
||||
sandbox="allow-scripts"
|
||||
/>
|
||||
) : (
|
||||
<pre className="m-0 whitespace-pre-wrap p-4 text-[12px] leading-relaxed text-ink">
|
||||
{previewContent}
|
||||
</pre>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/** Onglet Git : historique des commits, diff colorisé, retour arrière. */
|
||||
/** Onglet Fichiers : arborescence du workspace, à côté de Code. */
|
||||
function FilesTab() {
|
||||
const { fileTree, refreshFiles } = useStore();
|
||||
|
||||
useEffect(() => {
|
||||
refreshFiles();
|
||||
}, [refreshFiles]);
|
||||
|
||||
return (
|
||||
<div className="scr mx-3 mb-3 flex-1 overflow-auto rounded-card border border-line bg-panel p-2">
|
||||
{fileTree.length === 0 ? (
|
||||
<div className="px-2 py-8 text-center text-[13px] text-muted-2">
|
||||
Le workspace est vide.
|
||||
<br />
|
||||
L'agent créera des fichiers ici.
|
||||
</div>
|
||||
) : (
|
||||
<FileTree nodes={fileTree} />
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function GitPanel() {
|
||||
const refreshFiles = useStore((s) => s.refreshFiles);
|
||||
const currentProject = useStore((s) => s.currentProject);
|
||||
const project = currentProject();
|
||||
const [commits, setCommits] = useState<GitCommit[]>([]);
|
||||
const [selected, setSelected] = useState<string | null>(null);
|
||||
const [diff, setDiff] = useState("");
|
||||
const [busy, setBusy] = useState<string | null>(null);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const load = () => getGitLog(project).then(setCommits).catch(() => {});
|
||||
useEffect(() => {
|
||||
load();
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [project]);
|
||||
|
||||
const openDiff = async (hash: string) => {
|
||||
setSelected(hash);
|
||||
setDiff("Chargement…");
|
||||
setDiff((await getGitDiff(hash, project)) || "(diff vide)");
|
||||
};
|
||||
|
||||
const doRevert = async (hash: string) => {
|
||||
if (!window.confirm(`Annuler le commit ${hash.slice(0, 7)} ? (crée un commit inverse)`))
|
||||
return;
|
||||
setBusy(hash);
|
||||
setError(null);
|
||||
try {
|
||||
await revertCommit(hash, project);
|
||||
await load();
|
||||
await refreshFiles();
|
||||
if (selected === hash) setSelected(null);
|
||||
} catch (e) {
|
||||
setError(e instanceof Error ? e.message : "revert impossible");
|
||||
} finally {
|
||||
setBusy(null);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="scr mx-3 mb-3 flex flex-1 flex-col overflow-hidden border border-line bg-card">
|
||||
{error && (
|
||||
<div className="border-b border-line bg-warn/10 px-3 py-2 text-[12px] text-warn">
|
||||
{error}
|
||||
</div>
|
||||
)}
|
||||
{/* Liste des commits */}
|
||||
<div className="scr max-h-[45%] overflow-auto border-b border-line">
|
||||
{commits.length === 0 ? (
|
||||
<div className="p-4 text-center text-[12px] text-muted-2">
|
||||
Aucun commit. L'agent commitera chaque modification de code ici.
|
||||
</div>
|
||||
) : (
|
||||
commits.map((c) => (
|
||||
<div
|
||||
key={c.full_hash}
|
||||
className={`flex items-center gap-2 border-b border-line-soft px-3 py-2 last:border-0 ${
|
||||
selected === c.hash ? "bg-base" : ""
|
||||
}`}
|
||||
>
|
||||
<button
|
||||
onClick={() => openDiff(c.hash)}
|
||||
className="min-w-0 flex-1 text-left"
|
||||
title={c.subject}
|
||||
>
|
||||
<div className="truncate text-[13px] text-ink">{c.subject}</div>
|
||||
<div className="text-[11px] text-muted-2">
|
||||
{c.hash} · {c.when} · {c.files_changed} fichier
|
||||
{c.files_changed > 1 ? "s" : ""}
|
||||
</div>
|
||||
</button>
|
||||
<button
|
||||
onClick={() => doRevert(c.hash)}
|
||||
disabled={busy === c.hash}
|
||||
className="flex-none border border-line bg-card px-2 py-1 text-[11px] text-warn disabled:opacity-40"
|
||||
title="Annuler ce commit (revert)"
|
||||
>
|
||||
{busy === c.hash ? "…" : "↶ Annuler"}
|
||||
</button>
|
||||
</div>
|
||||
))
|
||||
)}
|
||||
</div>
|
||||
{/* Diff du commit sélectionné */}
|
||||
<div className="scr flex-1 overflow-auto bg-card-deep p-3">
|
||||
{selected ? (
|
||||
<DiffView diff={diff} />
|
||||
) : (
|
||||
<div className="py-8 text-center text-[12px] text-on-dark-3">
|
||||
Clique un commit pour voir ses modifications.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/** Rendu coloré d'un diff unifié. */
|
||||
function DiffView({ diff }: { diff: string }) {
|
||||
return (
|
||||
<pre className="m-0 whitespace-pre-wrap text-[11.5px] leading-relaxed">
|
||||
{diff.split("\n").map((line, i) => {
|
||||
let color = "text-on-dark-2";
|
||||
if (line.startsWith("+") && !line.startsWith("+++")) color = "text-ok";
|
||||
else if (line.startsWith("-") && !line.startsWith("---")) color = "text-accent";
|
||||
else if (line.startsWith("@@")) color = "text-info";
|
||||
else if (/^(commit|Author|Date|diff|index)/.test(line))
|
||||
color = "text-on-dark-3";
|
||||
return (
|
||||
<div key={i} className={color}>
|
||||
{line || " "}
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
</pre>
|
||||
);
|
||||
}
|
||||
|
||||
function Empty({ children }: { children: React.ReactNode }) {
|
||||
return (
|
||||
<div className="flex h-full flex-col items-center justify-center gap-2 p-8 text-center">
|
||||
<div className="font-pixel text-[10px] text-label">AUCUN APERÇU</div>
|
||||
<div className="max-w-[240px] text-[13px] text-muted-2">{children}</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Tab({
|
||||
children,
|
||||
active,
|
||||
onClick,
|
||||
}: {
|
||||
children: React.ReactNode;
|
||||
active?: boolean;
|
||||
onClick?: () => void;
|
||||
}) {
|
||||
return (
|
||||
<button
|
||||
onClick={onClick}
|
||||
className={`flex h-[30px] items-center border border-line px-[13px] text-[13px] ${
|
||||
active ? "bg-card-deep text-ink" : "bg-card text-muted"
|
||||
}`}
|
||||
>
|
||||
{children}
|
||||
</button>
|
||||
);
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
Loaded 100 of 260 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user