Go: decoupe des 3 gros fichiers - web_server (mux/cmd) + web_api (handlers REST) + web_chat (SSE), backend_llamacpp (commandes) + backend_build (machinerie cmake), chat_internet (client crawl) + chat_internet_tools (outils du modele) ; deplacement pur, zero changement de code

This commit is contained in:
nathaninline committed 2026-07-22 11:05:02 +02:00
1 parent 76f0d963f9
commit b4b30cd47a
7 files changed
+1691 -1645

No files matched your search

+646
View File
@@ -0,0 +1,646 @@
// backend_build.go — machinerie de compilation de llama.cpp : détection du
// plan de build (CUDA/ROCm/Metal/Vulkan/CPU), cmake, suivi de progression, logs.
package jean
import (
"bufio"
"fmt"
"io"
"os"
"os/exec"
"path/filepath"
"regexp"
"runtime"
"sort"
"strings"
"sync"
"time"
)
func detectBuildPlan() buildPlan {
p := buildPlan{backend: "cpu", jobs: numJobs()}
// Flags communs : Release + tuning natif pour la machine de build.
// (libcurl est activé d'office par llama.cpp ; LLAMA_CURL est déprécié.)
p.flags = []string{
"-DCMAKE_BUILD_TYPE=Release",
"-DGGML_NATIVE=ON",
// L'UI web embarquée de llama-server exige npm (ou un téléchargement
// d'assets pré-compilés depuis HuggingFace) pour générer un service-worker
// PWA — une dépendance lourde qui casse le build sur une machine sans node.
// jean fournit sa propre UI, donc on la désactive : build plus rapide et
// sans dépendance réseau/npm. BUILD_UI=OFF coupe npm ; USE_PREBUILT_UI=OFF
// coupe le téléchargement d'assets pré-compilés depuis HuggingFace (qui
// échoue sur un réseau restreint et fait planter l'embed). Sur un checkout
// neuf le dist est vide → llama-server embarque une UI vide sans erreur.
// Voir scripts/ui-assets.cmake côté llama.cpp.
"-DLLAMA_BUILD_UI=OFF",
"-DLLAMA_USE_PREBUILT_UI=OFF",
}
// Sur Windows, le générateur CMake par défaut est « NMake Makefiles », qui
// suppose un Developer Command Prompt MSVC. On force le générateur Visual
// Studio : il localise le toolchain MSVC tout seul via le registre, sans
// vcvars, depuis un shell ordinaire.
if runtime.GOOS == "windows" {
p.gen = msvcGenerator()
p.genArch = "x64"
if runtime.GOARCH == "arm64" {
p.genArch = "ARM64"
}
}
if runtime.GOOS == "darwin" {
// Metal est activé par défaut sur Apple Silicon ; on l'explicite.
p.backend = "metal"
p.flags = append(p.flags, "-DGGML_METAL=ON")
return p
}
// CUDA : nvcc présent ET un GPU NVIDIA visible.
if nvcc := findNvcc(); nvcc != "" && hasNvidiaGPU() {
p.backend = "cuda"
p.cudaCXX = nvcc
// NB : on n'active PAS GGML_CUDA_FA_ALL_QUANTS — il compile les kernels
// Flash-Attention pour toutes les combinaisons de quant (des centaines de
// .cu), ce qui explose le temps de build pour un gain d'inférence marginal.
p.flags = append(p.flags, "-DGGML_CUDA=ON", "-DGGML_CUDA_F16=ON")
if arch := detectCudaArch(); arch != "" {
p.cudaArch = arch
p.flags = append(p.flags, "-DCMAKE_CUDA_ARCHITECTURES="+arch)
}
return p
}
// AMD ROCm / HIP.
if hasTool("hipcc") || isDir("/opt/rocm") {
p.backend = "hip"
p.flags = append(p.flags, "-DGGML_HIP=ON")
return p
}
// Vulkan (GPU générique) — utile sur Intel/AMD sans ROCm.
if hasTool("glslc") && (isFile("/usr/lib/x86_64-linux-gnu/libvulkan.so.1") || hasTool("vulkaninfo")) {
p.backend = "vulkan"
p.flags = append(p.flags, "-DGGML_VULKAN=ON")
return p
}
return p // CPU
}
// buildLlamacpp configures and builds the llama-server target. It handles the
// "relocated checkout" gotcha: a build/ whose CMake cache was generated under a
// different source path can't reconfigure in place, so we wipe it. `clean`
// forces a from-scratch build regardless.
func buildLlamacpp(repo string, p buildPlan, clean bool) error {
build := filepath.Join(repo, "build")
if clean || cacheStale(build, repo) {
if isDir(build) {
fmt.Printf("%s reconfiguration propre (suppression de build/)\n", dim("[info]"))
old := build + ".old"
_ = os.RemoveAll(old)
if err := os.Rename(build, old); err != nil {
_ = os.RemoveAll(build) // dernier recours
}
}
}
// nvcc doit être dans le PATH et exposé via CUDACXX pour la config CMake.
env := ""
if p.backend == "cuda" && p.cudaCXX != "" {
cudaBin := filepath.Dir(p.cudaCXX)
parts := []string{
"CUDACXX=" + p.cudaCXX,
"PATH=" + cudaBin + string(os.PathListSeparator) + os.Getenv("PATH"),
}
// L'intégration MSBuild CUDA (générateur Visual Studio) résout
// CudaToolkitDir depuis CUDA_PATH / CUDA_PATH_Vx_y. L'installeur les pose
// dans l'environnement persistant, mais pas dans ce process déjà lancé —
// on les réinjecte sinon le configure échoue sur « CUDA Toolkit directory '' ».
parts = append(parts, cudaPathEnv(filepath.Dir(cudaBin))...)
env = strings.Join(parts, "\x00")
}
cfgArgs := []string{"-B", "build", "-S", "."}
if p.gen != "" {
cfgArgs = append(cfgArgs, "-G", p.gen)
if p.genArch != "" {
cfgArgs = append(cfgArgs, "-A", p.genArch)
}
}
cfgArgs = append(cfgArgs, p.flags...)
cfgLog := filepath.Join(repo, "configure.log")
if err := runBuildStep("cmake configure", repo, env, "cmake", cfgLog, cfgArgs...); err != nil {
hintMissingBuildDep(p, cfgLog)
return fmt.Errorf("configuration CMake échouée: %w", err)
}
buildArgs := []string{"--build", "build", "--config", "Release",
"-j", fmt.Sprintf("%d", p.jobs), "--target", "llama-server"}
// Générateur Visual Studio : MSBuild réaffiche par défaut la ligne de commande
// nvcc complète de chaque kernel (des pavés illisibles). On le passe en
// verbosité minimale via les args natifs après « -- ».
if strings.HasPrefix(p.gen, "Visual Studio") {
buildArgs = append(buildArgs, "--", "/nologo", "/verbosity:minimal")
}
if err := runBuildStep("cmake build", repo, env, "cmake", filepath.Join(repo, "build.log"), buildArgs...); err != nil {
return fmt.Errorf("compilation échouée: %w", err)
}
return nil
}
// hintMissingBuildDep scanne le log de configuration CMake à la recherche de
// dépendances manquantes CONNUES et affiche un indice d'installation adapté à la
// distribution, plutôt que de laisser l'utilisateur face à l'erreur CMake brute.
// Best-effort : silencieux si rien de reconnu. (Issue #6 : backend Vulkan qui
// échoue sur « Could not find ... SPIRV-Headers ».)
func hintMissingBuildDep(p buildPlan, cfgLog string) {
data, err := os.ReadFile(cfgLog)
if err != nil {
return
}
log := string(data)
// Backend Vulkan : les en-têtes SPIR-V (paquet SPIRV-Headers) sont requis par
// la config CMake de ggml-vulkan, mais absents par défaut sur beaucoup de
// distros même quand glslc/libvulkan sont là.
if p.backend == "vulkan" && strings.Contains(log, "SPIRV-Headers") {
fmt.Printf("\n%s dépendance manquante pour le backend %s : les en-têtes SPIR-V (paquet « SPIRV-Headers ») sont introuvables.\n",
yellow("[dépendance]"), green("Vulkan"))
if cmd := pkgInstallHint("spirv-headers"); cmd != "" {
fmt.Printf(" installe-les puis relance %s : %s\n", bold("jean llamacpp install"), bold(cmd))
} else {
fmt.Printf(" installe le paquet de développement « SPIRV-Headers » de ta distribution, puis relance %s.\n", bold("jean llamacpp install"))
}
}
}
// pkgInstallHint renvoie la commande d'installation d'un paquet adaptée au
// gestionnaire de paquets présent sur la machine (best-effort ; "" si aucun
// gestionnaire connu n'est trouvé). Sert uniquement à afficher un indice — on
// n'exécute rien automatiquement.
func pkgInstallHint(pkg string) string {
for _, m := range []struct{ bin, cmd string }{
{"pacman", "sudo pacman -S " + pkg},
{"apt-get", "sudo apt-get install -y " + pkg},
{"dnf", "sudo dnf install -y " + pkg},
{"zypper", "sudo zypper install -y " + pkg},
{"brew", "brew install " + pkg},
} {
if _, err := exec.LookPath(m.bin); err == nil {
return m.cmd
}
}
return ""
}
// cacheStale reports whether build/CMakeCache.txt was generated for a different
// source directory than `repo` (the relocated-checkout case).
func cacheStale(build, repo string) bool {
cache := filepath.Join(build, "CMakeCache.txt")
b, err := os.ReadFile(cache)
if err != nil {
return false // pas de cache => configure neuf, rien à nettoyer
}
absRepo, _ := filepath.Abs(repo)
for _, line := range strings.Split(string(b), "\n") {
// CMAKE_HOME_DIRECTORY pointe vers le source dir d'origine.
if strings.HasPrefix(line, "CMAKE_HOME_DIRECTORY:") {
if i := strings.IndexByte(line, '='); i >= 0 {
home := strings.TrimSpace(line[i+1:])
return home != "" && home != absRepo
}
}
}
return false
}
// ---------------------------------------------------------------------------
// Sondes matérielles
// ---------------------------------------------------------------------------
// findNvcc returns the path to nvcc from PATH or a /usr/local/cuda* install,
// preferring the highest version.
func findNvcc() string {
if p, err := exec.LookPath("nvcc"); err == nil {
return p
}
if runtime.GOOS == "windows" {
// CUDA_PATH est posé par l'installeur officiel.
if cp := os.Getenv("CUDA_PATH"); cp != "" {
if p := filepath.Join(cp, "bin", "nvcc.exe"); isFile(p) {
return p
}
}
// Layout standard : …\NVIDIA GPU Computing Toolkit\CUDA\v12.x\bin\nvcc.exe
for _, base := range []string{os.Getenv("ProgramFiles"), `C:\Program Files`} {
if base == "" {
continue
}
matches, _ := filepath.Glob(filepath.Join(base, "NVIDIA GPU Computing Toolkit", "CUDA", "v*", "bin", "nvcc.exe"))
if len(matches) > 0 {
sort.Strings(matches) // v12.2 < v12.8 → on prend le plus récent
return matches[len(matches)-1]
}
}
return ""
}
if p := "/usr/local/cuda/bin/nvcc"; isFile(p) {
return p
}
matches, _ := filepath.Glob("/usr/local/cuda-*/bin/nvcc")
if len(matches) > 0 {
sort.Strings(matches) // cuda-12.2 < cuda-12.8 lexicographiquement → on prend le dernier
return matches[len(matches)-1]
}
return ""
}
func hasNvidiaGPU() bool {
if !hasTool("nvidia-smi") {
return false
}
out, err := hideCmd(exec.Command("nvidia-smi", "-L")).Output()
return err == nil && strings.Contains(string(out), "GPU")
}
// detectCudaArch queries every GPU's compute capability via nvidia-smi and
// returns them as CMake-style arch codes (e.g. "8.6" → "86"), deduped and
// joined with ';'. Empty when the driver is too old to report it (CMake then
// falls back to native detection).
func detectCudaArch() string {
out, err := hideCmd(exec.Command("nvidia-smi", "--query-gpu=compute_cap", "--format=csv,noheader")).Output()
if err != nil {
return ""
}
seen := map[string]bool{}
var archs []string
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
cap := strings.TrimSpace(line)
if cap == "" || strings.Contains(strings.ToLower(cap), "not supported") {
continue
}
code := strings.ReplaceAll(cap, ".", "") // "12.0" → "120"
if code != "" && !seen[code] {
seen[code] = true
archs = append(archs, code)
}
}
return strings.Join(archs, ";")
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
func numJobs() int {
n := runtime.NumCPU()
if n < 1 {
return 1
}
return n
}
func isFile(p string) bool {
fi, err := os.Stat(p)
return err == nil && !fi.IsDir()
}
func isDir(p string) bool {
fi, err := os.Stat(p)
return err == nil && fi.IsDir()
}
// llamaServerBin returns the path to the built llama-server binary under repo,
// probing the layouts the different CMake generators emit: the Visual Studio
// multi-config generator nests it under build/bin/Release/ and Windows adds a
// .exe suffix, whereas the Unix Makefiles generator drops it in build/bin/.
// Returns "" when no binary is found.
func llamaServerBin(repo string) string {
ext := ""
if runtime.GOOS == "windows" {
ext = ".exe"
}
for _, rel := range []string{
filepath.Join("build", "bin", "Release", "llama-server"+ext),
filepath.Join("build", "bin", "llama-server"+ext),
filepath.Join("build", "Release", "llama-server"+ext),
filepath.Join("build", "llama-server"+ext),
} {
if p := filepath.Join(repo, rel); isFile(p) {
return p
}
}
return ""
}
func hasTool(name string) bool {
_, err := exec.LookPath(name)
return err == nil
}
func requireTools(tools ...string) error {
missing := missingTools(tools)
if len(missing) == 0 {
return nil
}
// Tentative d'installation automatique (winget sur Windows, apt/brew/dnf sur
// Unix). On rafraîchit ensuite le PATH du process car un installeur système
// écrit le PATH machine sans toucher l'environnement déjà chargé.
fmt.Printf("%s outils manquants: %s — installation automatique…\n", yellow("[info]"), strings.Join(missing, ", "))
for _, t := range missing {
if err := autoInstallTool(t); err != nil {
fmt.Printf(" %s %s: %v\n", dim("•"), t, err)
}
}
refreshToolPath()
if still := missingTools(tools); len(still) > 0 {
return fmt.Errorf("outils toujours manquants après tentative d'install: %s — installe-les à la main puis réessaie", strings.Join(still, ", "))
}
fmt.Printf("%s outils installés.\n", green("✓"))
return nil
}
func missingTools(tools []string) []string {
var missing []string
for _, t := range tools {
if !hasTool(t) {
missing = append(missing, t)
}
}
return missing
}
// gitOutput runs a git command in `dir` and returns trimmed stdout (or "").
func gitOutput(dir string, args ...string) string {
cmd := exec.Command("git", args...)
cmd.Dir = dir
out, err := cmd.Output()
if err != nil {
return ""
}
return strings.TrimSpace(string(out))
}
// runStep runs a command in `dir` streaming output live to the terminal.
func runStep(name, dir, bin string, args ...string) error {
return runStepEnv(name, dir, "", bin, args...)
}
// runStepEnv is runStep with optional extra env vars (NUL-separated KEY=VAL
// pairs in `extraEnv`, which override existing ones).
func runStepEnv(name, dir, extraEnv, bin string, args ...string) error {
fmt.Printf("\n%s %s %s\n", cyan("▶"), name, dim(strings.Join(args, " ")))
cmd := exec.Command(bin, args...)
cmd.Dir = dir
cmd.Stdout = os.Stdout
cmd.Stderr = os.Stderr
cmd.Stdin = os.Stdin
if extraEnv != "" {
env := os.Environ()
for _, kv := range strings.Split(extraEnv, "\x00") {
if kv == "" {
continue
}
env = upsertEnv(env, kv)
}
cmd.Env = env
}
return cmd.Run()
}
// runBuildStep runs a compile step while keeping the terminal clean: the full
// output goes to logPath, and the screen shows only a single self-rewriting
// progress line (spinner + compiled-file count) plus any real compiler
// diagnostics. The hundreds of per-file nvcc/cl command echoes are hidden. On
// failure the tail of the log is printed so the actual error is never lost.
func runBuildStep(name, dir, extraEnv, bin, logPath string, args ...string) error {
fmt.Printf("\n%s %s\n", cyan("▶"), name)
cmd := exec.Command(bin, args...)
cmd.Dir = dir
if extraEnv != "" {
env := os.Environ()
for _, kv := range strings.Split(extraEnv, "\x00") {
if kv != "" {
env = upsertEnv(env, kv)
}
}
cmd.Env = env
}
var logf *os.File
if logPath != "" {
if f, err := os.Create(logPath); err == nil {
logf = f
defer logf.Close()
}
}
pr, pw := io.Pipe()
cmd.Stdout = pw
cmd.Stderr = pw
if err := cmd.Start(); err != nil {
return err
}
frames := []rune{'⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'}
var (
mu sync.Mutex
count int
label = "préparation…"
fi int
)
clearLine := func() {
if colorOn {
fmt.Print("\r\033[K")
}
}
// draw redessine la ligne d'état ; appelé par une horloge pour rester animé
// même quand un seul gros fichier compile pendant plusieurs minutes.
draw := func() {
if !colorOn {
return
}
mu.Lock()
fi = (fi + 1) % len(frames)
fmt.Printf("\r\033[K %c %s", frames[fi], label)
mu.Unlock()
}
done := make(chan struct{})
go func() {
sc := bufio.NewScanner(pr)
sc.Buffer(make([]byte, 1<<20), 1<<20) // les échos de commande sont énormes
for sc.Scan() {
line := sc.Text()
if logf != nil {
fmt.Fprintln(logf, line)
}
mu.Lock()
if f := compiledFile(line); f != "" {
count++
label = fmt.Sprintf("compilation… %d fichiers %s", count, dim("("+f+")"))
mu.Unlock()
continue
}
if p := phaseLabel(line); p != "" {
label = p
}
mu.Unlock()
// On ne fait remonter que les vraies ERREURS (les warnings MSVC/linker
// d'un projet tiers sont du bruit ; ils restent dans le log). Les CMake
// Error de la phase configure sont aussi affichés.
if reBuildError.MatchString(line) || strings.HasPrefix(strings.TrimSpace(line), "CMake Error") {
mu.Lock()
clearLine()
fmt.Println(" " + strings.TrimSpace(line))
mu.Unlock()
}
}
close(done)
}()
// Horloge d'animation, indépendante du flux de sortie.
stop := make(chan struct{})
tickerDone := make(chan struct{})
go func() {
defer close(tickerDone)
t := time.NewTicker(120 * time.Millisecond)
defer t.Stop()
for {
select {
case <-stop:
return
case <-t.C:
draw()
}
}
}()
err := cmd.Wait()
_ = pw.Close()
<-done
close(stop)
<-tickerDone
clearLine()
if err == nil && count > 0 {
fmt.Printf(" %s %d fichiers compilés\n", green("✓"), count)
}
if err != nil && logPath != "" {
fmt.Printf("%s étape échouée — log complet : %s\n", yellow("[err]"), logPath)
printLogTail(logPath, 30)
}
return err
}
// phaseLabel maps a non-compile output line to a short status label, or "" to
// leave the current label unchanged. Keeps the spinner informative during the
// CMake configure phase and the final link.
func phaseLabel(line string) string {
t := strings.TrimSpace(line)
switch {
case strings.HasPrefix(t, "-- "):
return "configuration… " + truncLabel(strings.TrimPrefix(t, "-- "), 50)
case strings.Contains(t, "Linking") || strings.Contains(t, "Build files have been written"):
return "édition de liens…"
}
return ""
}
func truncLabel(s string, n int) string {
s = strings.TrimSpace(s)
if len(s) > n {
return s[:n-1] + "…"
}
return s
}
var (
// MSBuild (Windows) : « Compiling CUDA source file …\foo.cu… » ou nom de
// source seul « foo.cpp » imprimé par cl.
reCompilingCUDA = regexp.MustCompile(`Compiling .*?([\w.\-]+\.cu)\b`)
reBareSource = regexp.MustCompile(`^[\w.\-]+\.(c|cc|cpp|cxx|cu|cuh)$`)
// Make / Ninja (Linux, macOS) : « [ 45%] Building CXX object …/foo.cpp.o » ou
// « [12/345] Building CUDA object …/foo.cu.o ».
reBuildingObj = regexp.MustCompile(`Building (?:C|CXX|CUDA|ASM)\w* object .*?/([^/]+?)\.o(?:bj)?\b`)
// Vraies erreurs : « foo.cpp(12): error C2065 » (MSVC), « foo.cpp:12:5: error: »
// (gcc/clang), « LINK : fatal error LNK1104 ». On exige le « : » devant le
// mot-clé pour ne PAS matcher les flags type -D_CRT_SECURE_NO_WARNINGS dans les
// lignes de commande. Les warnings (bruit d'un projet tiers) sont exclus.
reBuildError = regexp.MustCompile(`(?i):\s*(fatal error|error)\b`)
)
// compiledFile returns the source filename a build line announces compiling, or
// "" if the line isn't a compile-progress marker. Handles both the MSBuild
// (Windows) and Make/Ninja (Unix) output formats.
func compiledFile(line string) string {
t := strings.TrimSpace(line)
if m := reCompilingCUDA.FindStringSubmatch(t); m != nil {
return m[1]
}
if m := reBuildingObj.FindStringSubmatch(t); m != nil {
return m[1]
}
if reBareSource.MatchString(t) {
return t
}
return ""
}
// printLogTail prints the last n lines of the log file (best-effort).
func printLogTail(path string, n int) {
b, err := os.ReadFile(path)
if err != nil {
return
}
lines := strings.Split(strings.TrimRight(string(b), "\n"), "\n")
if len(lines) > n {
lines = lines[len(lines)-n:]
}
for _, l := range lines {
fmt.Println(" " + dim(l))
}
}
// upsertEnv replaces KEY=… in env if present, else appends kv (kv is "KEY=VAL").
func upsertEnv(env []string, kv string) []string {
key := kv
if i := strings.IndexByte(kv, '='); i >= 0 {
key = kv[:i]
}
for i, e := range env {
if strings.HasPrefix(e, key+"=") {
env[i] = kv
return env
}
}
return append(env, kv)
}
func planLabel(p buildPlan) string {
switch p.backend {
case "cuda":
arch := p.cudaArch
if arch == "" {
arch = "native"
}
return green("CUDA") + dim(" (arch="+arch+", nvcc="+p.cudaCXX+")")
case "hip":
return green("ROCm/HIP")
case "metal":
return green("Metal")
case "vulkan":
return green("Vulkan")
default:
return yellow("CPU") + dim(" (aucun accélérateur détecté)")
}
}
func printPlan(p buildPlan, repo string) {
fmt.Printf("\n%s configuration du build\n", bold("•"))
fmt.Printf(" backend : %s\n", planLabel(p))
fmt.Printf(" jobs : %d\n", p.jobs)
fmt.Printf(" flags : %s\n", dim(strings.Join(p.flags, " ")))
}
-635
View File
@@ -1,18 +1,10 @@
package jean
import (
"bufio"
"fmt"
"io"
"os"
"os/exec"
"path/filepath"
"regexp"
"runtime"
"sort"
"strings"
"sync"
"time"
)
// backend_llamacpp.go — gestion du backend llama.cpp (clone, build, mise à jour).
@@ -341,630 +333,3 @@ func llamacppStatus(args []string) error {
// detectBuildPlan probes the machine and returns the CMake flags for the best
// available accelerator. Order of preference: CUDA → ROCm/HIP → Metal (macOS)
// → Vulkan → CPU.
func detectBuildPlan() buildPlan {
p := buildPlan{backend: "cpu", jobs: numJobs()}
// Flags communs : Release + tuning natif pour la machine de build.
// (libcurl est activé d'office par llama.cpp ; LLAMA_CURL est déprécié.)
p.flags = []string{
"-DCMAKE_BUILD_TYPE=Release",
"-DGGML_NATIVE=ON",
// L'UI web embarquée de llama-server exige npm (ou un téléchargement
// d'assets pré-compilés depuis HuggingFace) pour générer un service-worker
// PWA — une dépendance lourde qui casse le build sur une machine sans node.
// jean fournit sa propre UI, donc on la désactive : build plus rapide et
// sans dépendance réseau/npm. BUILD_UI=OFF coupe npm ; USE_PREBUILT_UI=OFF
// coupe le téléchargement d'assets pré-compilés depuis HuggingFace (qui
// échoue sur un réseau restreint et fait planter l'embed). Sur un checkout
// neuf le dist est vide → llama-server embarque une UI vide sans erreur.
// Voir scripts/ui-assets.cmake côté llama.cpp.
"-DLLAMA_BUILD_UI=OFF",
"-DLLAMA_USE_PREBUILT_UI=OFF",
}
// Sur Windows, le générateur CMake par défaut est « NMake Makefiles », qui
// suppose un Developer Command Prompt MSVC. On force le générateur Visual
// Studio : il localise le toolchain MSVC tout seul via le registre, sans
// vcvars, depuis un shell ordinaire.
if runtime.GOOS == "windows" {
p.gen = msvcGenerator()
p.genArch = "x64"
if runtime.GOARCH == "arm64" {
p.genArch = "ARM64"
}
}
if runtime.GOOS == "darwin" {
// Metal est activé par défaut sur Apple Silicon ; on l'explicite.
p.backend = "metal"
p.flags = append(p.flags, "-DGGML_METAL=ON")
return p
}
// CUDA : nvcc présent ET un GPU NVIDIA visible.
if nvcc := findNvcc(); nvcc != "" && hasNvidiaGPU() {
p.backend = "cuda"
p.cudaCXX = nvcc
// NB : on n'active PAS GGML_CUDA_FA_ALL_QUANTS — il compile les kernels
// Flash-Attention pour toutes les combinaisons de quant (des centaines de
// .cu), ce qui explose le temps de build pour un gain d'inférence marginal.
p.flags = append(p.flags, "-DGGML_CUDA=ON", "-DGGML_CUDA_F16=ON")
if arch := detectCudaArch(); arch != "" {
p.cudaArch = arch
p.flags = append(p.flags, "-DCMAKE_CUDA_ARCHITECTURES="+arch)
}
return p
}
// AMD ROCm / HIP.
if hasTool("hipcc") || isDir("/opt/rocm") {
p.backend = "hip"
p.flags = append(p.flags, "-DGGML_HIP=ON")
return p
}
// Vulkan (GPU générique) — utile sur Intel/AMD sans ROCm.
if hasTool("glslc") && (isFile("/usr/lib/x86_64-linux-gnu/libvulkan.so.1") || hasTool("vulkaninfo")) {
p.backend = "vulkan"
p.flags = append(p.flags, "-DGGML_VULKAN=ON")
return p
}
return p // CPU
}
// buildLlamacpp configures and builds the llama-server target. It handles the
// "relocated checkout" gotcha: a build/ whose CMake cache was generated under a
// different source path can't reconfigure in place, so we wipe it. `clean`
// forces a from-scratch build regardless.
func buildLlamacpp(repo string, p buildPlan, clean bool) error {
build := filepath.Join(repo, "build")
if clean || cacheStale(build, repo) {
if isDir(build) {
fmt.Printf("%s reconfiguration propre (suppression de build/)\n", dim("[info]"))
old := build + ".old"
_ = os.RemoveAll(old)
if err := os.Rename(build, old); err != nil {
_ = os.RemoveAll(build) // dernier recours
}
}
}
// nvcc doit être dans le PATH et exposé via CUDACXX pour la config CMake.
env := ""
if p.backend == "cuda" && p.cudaCXX != "" {
cudaBin := filepath.Dir(p.cudaCXX)
parts := []string{
"CUDACXX=" + p.cudaCXX,
"PATH=" + cudaBin + string(os.PathListSeparator) + os.Getenv("PATH"),
}
// L'intégration MSBuild CUDA (générateur Visual Studio) résout
// CudaToolkitDir depuis CUDA_PATH / CUDA_PATH_Vx_y. L'installeur les pose
// dans l'environnement persistant, mais pas dans ce process déjà lancé —
// on les réinjecte sinon le configure échoue sur « CUDA Toolkit directory '' ».
parts = append(parts, cudaPathEnv(filepath.Dir(cudaBin))...)
env = strings.Join(parts, "\x00")
}
cfgArgs := []string{"-B", "build", "-S", "."}
if p.gen != "" {
cfgArgs = append(cfgArgs, "-G", p.gen)
if p.genArch != "" {
cfgArgs = append(cfgArgs, "-A", p.genArch)
}
}
cfgArgs = append(cfgArgs, p.flags...)
cfgLog := filepath.Join(repo, "configure.log")
if err := runBuildStep("cmake configure", repo, env, "cmake", cfgLog, cfgArgs...); err != nil {
hintMissingBuildDep(p, cfgLog)
return fmt.Errorf("configuration CMake échouée: %w", err)
}
buildArgs := []string{"--build", "build", "--config", "Release",
"-j", fmt.Sprintf("%d", p.jobs), "--target", "llama-server"}
// Générateur Visual Studio : MSBuild réaffiche par défaut la ligne de commande
// nvcc complète de chaque kernel (des pavés illisibles). On le passe en
// verbosité minimale via les args natifs après « -- ».
if strings.HasPrefix(p.gen, "Visual Studio") {
buildArgs = append(buildArgs, "--", "/nologo", "/verbosity:minimal")
}
if err := runBuildStep("cmake build", repo, env, "cmake", filepath.Join(repo, "build.log"), buildArgs...); err != nil {
return fmt.Errorf("compilation échouée: %w", err)
}
return nil
}
// hintMissingBuildDep scanne le log de configuration CMake à la recherche de
// dépendances manquantes CONNUES et affiche un indice d'installation adapté à la
// distribution, plutôt que de laisser l'utilisateur face à l'erreur CMake brute.
// Best-effort : silencieux si rien de reconnu. (Issue #6 : backend Vulkan qui
// échoue sur « Could not find ... SPIRV-Headers ».)
func hintMissingBuildDep(p buildPlan, cfgLog string) {
data, err := os.ReadFile(cfgLog)
if err != nil {
return
}
log := string(data)
// Backend Vulkan : les en-têtes SPIR-V (paquet SPIRV-Headers) sont requis par
// la config CMake de ggml-vulkan, mais absents par défaut sur beaucoup de
// distros même quand glslc/libvulkan sont là.
if p.backend == "vulkan" && strings.Contains(log, "SPIRV-Headers") {
fmt.Printf("\n%s dépendance manquante pour le backend %s : les en-têtes SPIR-V (paquet « SPIRV-Headers ») sont introuvables.\n",
yellow("[dépendance]"), green("Vulkan"))
if cmd := pkgInstallHint("spirv-headers"); cmd != "" {
fmt.Printf(" installe-les puis relance %s : %s\n", bold("jean llamacpp install"), bold(cmd))
} else {
fmt.Printf(" installe le paquet de développement « SPIRV-Headers » de ta distribution, puis relance %s.\n", bold("jean llamacpp install"))
}
}
}
// pkgInstallHint renvoie la commande d'installation d'un paquet adaptée au
// gestionnaire de paquets présent sur la machine (best-effort ; "" si aucun
// gestionnaire connu n'est trouvé). Sert uniquement à afficher un indice — on
// n'exécute rien automatiquement.
func pkgInstallHint(pkg string) string {
for _, m := range []struct{ bin, cmd string }{
{"pacman", "sudo pacman -S " + pkg},
{"apt-get", "sudo apt-get install -y " + pkg},
{"dnf", "sudo dnf install -y " + pkg},
{"zypper", "sudo zypper install -y " + pkg},
{"brew", "brew install " + pkg},
} {
if _, err := exec.LookPath(m.bin); err == nil {
return m.cmd
}
}
return ""
}
// cacheStale reports whether build/CMakeCache.txt was generated for a different
// source directory than `repo` (the relocated-checkout case).
func cacheStale(build, repo string) bool {
cache := filepath.Join(build, "CMakeCache.txt")
b, err := os.ReadFile(cache)
if err != nil {
return false // pas de cache => configure neuf, rien à nettoyer
}
absRepo, _ := filepath.Abs(repo)
for _, line := range strings.Split(string(b), "\n") {
// CMAKE_HOME_DIRECTORY pointe vers le source dir d'origine.
if strings.HasPrefix(line, "CMAKE_HOME_DIRECTORY:") {
if i := strings.IndexByte(line, '='); i >= 0 {
home := strings.TrimSpace(line[i+1:])
return home != "" && home != absRepo
}
}
}
return false
}
// ---------------------------------------------------------------------------
// Sondes matérielles
// ---------------------------------------------------------------------------
// findNvcc returns the path to nvcc from PATH or a /usr/local/cuda* install,
// preferring the highest version.
func findNvcc() string {
if p, err := exec.LookPath("nvcc"); err == nil {
return p
}
if runtime.GOOS == "windows" {
// CUDA_PATH est posé par l'installeur officiel.
if cp := os.Getenv("CUDA_PATH"); cp != "" {
if p := filepath.Join(cp, "bin", "nvcc.exe"); isFile(p) {
return p
}
}
// Layout standard : …\NVIDIA GPU Computing Toolkit\CUDA\v12.x\bin\nvcc.exe
for _, base := range []string{os.Getenv("ProgramFiles"), `C:\Program Files`} {
if base == "" {
continue
}
matches, _ := filepath.Glob(filepath.Join(base, "NVIDIA GPU Computing Toolkit", "CUDA", "v*", "bin", "nvcc.exe"))
if len(matches) > 0 {
sort.Strings(matches) // v12.2 < v12.8 → on prend le plus récent
return matches[len(matches)-1]
}
}
return ""
}
if p := "/usr/local/cuda/bin/nvcc"; isFile(p) {
return p
}
matches, _ := filepath.Glob("/usr/local/cuda-*/bin/nvcc")
if len(matches) > 0 {
sort.Strings(matches) // cuda-12.2 < cuda-12.8 lexicographiquement → on prend le dernier
return matches[len(matches)-1]
}
return ""
}
func hasNvidiaGPU() bool {
if !hasTool("nvidia-smi") {
return false
}
out, err := hideCmd(exec.Command("nvidia-smi", "-L")).Output()
return err == nil && strings.Contains(string(out), "GPU")
}
// detectCudaArch queries every GPU's compute capability via nvidia-smi and
// returns them as CMake-style arch codes (e.g. "8.6" → "86"), deduped and
// joined with ';'. Empty when the driver is too old to report it (CMake then
// falls back to native detection).
func detectCudaArch() string {
out, err := hideCmd(exec.Command("nvidia-smi", "--query-gpu=compute_cap", "--format=csv,noheader")).Output()
if err != nil {
return ""
}
seen := map[string]bool{}
var archs []string
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
cap := strings.TrimSpace(line)
if cap == "" || strings.Contains(strings.ToLower(cap), "not supported") {
continue
}
code := strings.ReplaceAll(cap, ".", "") // "12.0" → "120"
if code != "" && !seen[code] {
seen[code] = true
archs = append(archs, code)
}
}
return strings.Join(archs, ";")
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
func numJobs() int {
n := runtime.NumCPU()
if n < 1 {
return 1
}
return n
}
func isFile(p string) bool {
fi, err := os.Stat(p)
return err == nil && !fi.IsDir()
}
func isDir(p string) bool {
fi, err := os.Stat(p)
return err == nil && fi.IsDir()
}
// llamaServerBin returns the path to the built llama-server binary under repo,
// probing the layouts the different CMake generators emit: the Visual Studio
// multi-config generator nests it under build/bin/Release/ and Windows adds a
// .exe suffix, whereas the Unix Makefiles generator drops it in build/bin/.
// Returns "" when no binary is found.
func llamaServerBin(repo string) string {
ext := ""
if runtime.GOOS == "windows" {
ext = ".exe"
}
for _, rel := range []string{
filepath.Join("build", "bin", "Release", "llama-server"+ext),
filepath.Join("build", "bin", "llama-server"+ext),
filepath.Join("build", "Release", "llama-server"+ext),
filepath.Join("build", "llama-server"+ext),
} {
if p := filepath.Join(repo, rel); isFile(p) {
return p
}
}
return ""
}
func hasTool(name string) bool {
_, err := exec.LookPath(name)
return err == nil
}
func requireTools(tools ...string) error {
missing := missingTools(tools)
if len(missing) == 0 {
return nil
}
// Tentative d'installation automatique (winget sur Windows, apt/brew/dnf sur
// Unix). On rafraîchit ensuite le PATH du process car un installeur système
// écrit le PATH machine sans toucher l'environnement déjà chargé.
fmt.Printf("%s outils manquants: %s — installation automatique…\n", yellow("[info]"), strings.Join(missing, ", "))
for _, t := range missing {
if err := autoInstallTool(t); err != nil {
fmt.Printf(" %s %s: %v\n", dim("•"), t, err)
}
}
refreshToolPath()
if still := missingTools(tools); len(still) > 0 {
return fmt.Errorf("outils toujours manquants après tentative d'install: %s — installe-les à la main puis réessaie", strings.Join(still, ", "))
}
fmt.Printf("%s outils installés.\n", green("✓"))
return nil
}
func missingTools(tools []string) []string {
var missing []string
for _, t := range tools {
if !hasTool(t) {
missing = append(missing, t)
}
}
return missing
}
// gitOutput runs a git command in `dir` and returns trimmed stdout (or "").
func gitOutput(dir string, args ...string) string {
cmd := exec.Command("git", args...)
cmd.Dir = dir
out, err := cmd.Output()
if err != nil {
return ""
}
return strings.TrimSpace(string(out))
}
// runStep runs a command in `dir` streaming output live to the terminal.
func runStep(name, dir, bin string, args ...string) error {
return runStepEnv(name, dir, "", bin, args...)
}
// runStepEnv is runStep with optional extra env vars (NUL-separated KEY=VAL
// pairs in `extraEnv`, which override existing ones).
func runStepEnv(name, dir, extraEnv, bin string, args ...string) error {
fmt.Printf("\n%s %s %s\n", cyan("▶"), name, dim(strings.Join(args, " ")))
cmd := exec.Command(bin, args...)
cmd.Dir = dir
cmd.Stdout = os.Stdout
cmd.Stderr = os.Stderr
cmd.Stdin = os.Stdin
if extraEnv != "" {
env := os.Environ()
for _, kv := range strings.Split(extraEnv, "\x00") {
if kv == "" {
continue
}
env = upsertEnv(env, kv)
}
cmd.Env = env
}
return cmd.Run()
}
// runBuildStep runs a compile step while keeping the terminal clean: the full
// output goes to logPath, and the screen shows only a single self-rewriting
// progress line (spinner + compiled-file count) plus any real compiler
// diagnostics. The hundreds of per-file nvcc/cl command echoes are hidden. On
// failure the tail of the log is printed so the actual error is never lost.
func runBuildStep(name, dir, extraEnv, bin, logPath string, args ...string) error {
fmt.Printf("\n%s %s\n", cyan("▶"), name)
cmd := exec.Command(bin, args...)
cmd.Dir = dir
if extraEnv != "" {
env := os.Environ()
for _, kv := range strings.Split(extraEnv, "\x00") {
if kv != "" {
env = upsertEnv(env, kv)
}
}
cmd.Env = env
}
var logf *os.File
if logPath != "" {
if f, err := os.Create(logPath); err == nil {
logf = f
defer logf.Close()
}
}
pr, pw := io.Pipe()
cmd.Stdout = pw
cmd.Stderr = pw
if err := cmd.Start(); err != nil {
return err
}
frames := []rune{'⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'}
var (
mu sync.Mutex
count int
label = "préparation…"
fi int
)
clearLine := func() {
if colorOn {
fmt.Print("\r\033[K")
}
}
// draw redessine la ligne d'état ; appelé par une horloge pour rester animé
// même quand un seul gros fichier compile pendant plusieurs minutes.
draw := func() {
if !colorOn {
return
}
mu.Lock()
fi = (fi + 1) % len(frames)
fmt.Printf("\r\033[K %c %s", frames[fi], label)
mu.Unlock()
}
done := make(chan struct{})
go func() {
sc := bufio.NewScanner(pr)
sc.Buffer(make([]byte, 1<<20), 1<<20) // les échos de commande sont énormes
for sc.Scan() {
line := sc.Text()
if logf != nil {
fmt.Fprintln(logf, line)
}
mu.Lock()
if f := compiledFile(line); f != "" {
count++
label = fmt.Sprintf("compilation… %d fichiers %s", count, dim("("+f+")"))
mu.Unlock()
continue
}
if p := phaseLabel(line); p != "" {
label = p
}
mu.Unlock()
// On ne fait remonter que les vraies ERREURS (les warnings MSVC/linker
// d'un projet tiers sont du bruit ; ils restent dans le log). Les CMake
// Error de la phase configure sont aussi affichés.
if reBuildError.MatchString(line) || strings.HasPrefix(strings.TrimSpace(line), "CMake Error") {
mu.Lock()
clearLine()
fmt.Println(" " + strings.TrimSpace(line))
mu.Unlock()
}
}
close(done)
}()
// Horloge d'animation, indépendante du flux de sortie.
stop := make(chan struct{})
tickerDone := make(chan struct{})
go func() {
defer close(tickerDone)
t := time.NewTicker(120 * time.Millisecond)
defer t.Stop()
for {
select {
case <-stop:
return
case <-t.C:
draw()
}
}
}()
err := cmd.Wait()
_ = pw.Close()
<-done
close(stop)
<-tickerDone
clearLine()
if err == nil && count > 0 {
fmt.Printf(" %s %d fichiers compilés\n", green("✓"), count)
}
if err != nil && logPath != "" {
fmt.Printf("%s étape échouée — log complet : %s\n", yellow("[err]"), logPath)
printLogTail(logPath, 30)
}
return err
}
// phaseLabel maps a non-compile output line to a short status label, or "" to
// leave the current label unchanged. Keeps the spinner informative during the
// CMake configure phase and the final link.
func phaseLabel(line string) string {
t := strings.TrimSpace(line)
switch {
case strings.HasPrefix(t, "-- "):
return "configuration… " + truncLabel(strings.TrimPrefix(t, "-- "), 50)
case strings.Contains(t, "Linking") || strings.Contains(t, "Build files have been written"):
return "édition de liens…"
}
return ""
}
func truncLabel(s string, n int) string {
s = strings.TrimSpace(s)
if len(s) > n {
return s[:n-1] + "…"
}
return s
}
var (
// MSBuild (Windows) : « Compiling CUDA source file …\foo.cu… » ou nom de
// source seul « foo.cpp » imprimé par cl.
reCompilingCUDA = regexp.MustCompile(`Compiling .*?([\w.\-]+\.cu)\b`)
reBareSource = regexp.MustCompile(`^[\w.\-]+\.(c|cc|cpp|cxx|cu|cuh)$`)
// Make / Ninja (Linux, macOS) : « [ 45%] Building CXX object …/foo.cpp.o » ou
// « [12/345] Building CUDA object …/foo.cu.o ».
reBuildingObj = regexp.MustCompile(`Building (?:C|CXX|CUDA|ASM)\w* object .*?/([^/]+?)\.o(?:bj)?\b`)
// Vraies erreurs : « foo.cpp(12): error C2065 » (MSVC), « foo.cpp:12:5: error: »
// (gcc/clang), « LINK : fatal error LNK1104 ». On exige le « : » devant le
// mot-clé pour ne PAS matcher les flags type -D_CRT_SECURE_NO_WARNINGS dans les
// lignes de commande. Les warnings (bruit d'un projet tiers) sont exclus.
reBuildError = regexp.MustCompile(`(?i):\s*(fatal error|error)\b`)
)
// compiledFile returns the source filename a build line announces compiling, or
// "" if the line isn't a compile-progress marker. Handles both the MSBuild
// (Windows) and Make/Ninja (Unix) output formats.
func compiledFile(line string) string {
t := strings.TrimSpace(line)
if m := reCompilingCUDA.FindStringSubmatch(t); m != nil {
return m[1]
}
if m := reBuildingObj.FindStringSubmatch(t); m != nil {
return m[1]
}
if reBareSource.MatchString(t) {
return t
}
return ""
}
// printLogTail prints the last n lines of the log file (best-effort).
func printLogTail(path string, n int) {
b, err := os.ReadFile(path)
if err != nil {
return
}
lines := strings.Split(strings.TrimRight(string(b), "\n"), "\n")
if len(lines) > n {
lines = lines[len(lines)-n:]
}
for _, l := range lines {
fmt.Println(" " + dim(l))
}
}
// upsertEnv replaces KEY=… in env if present, else appends kv (kv is "KEY=VAL").
func upsertEnv(env []string, kv string) []string {
key := kv
if i := strings.IndexByte(kv, '='); i >= 0 {
key = kv[:i]
}
for i, e := range env {
if strings.HasPrefix(e, key+"=") {
env[i] = kv
return env
}
}
return append(env, kv)
}
func planLabel(p buildPlan) string {
switch p.backend {
case "cuda":
arch := p.cudaArch
if arch == "" {
arch = "native"
}
return green("CUDA") + dim(" (arch="+arch+", nvcc="+p.cudaCXX+")")
case "hip":
return green("ROCm/HIP")
case "metal":
return green("Metal")
case "vulkan":
return green("Vulkan")
default:
return yellow("CPU") + dim(" (aucun accélérateur détecté)")
}
}
func printPlan(p buildPlan, repo string) {
fmt.Printf("\n%s configuration du build\n", bold("•"))
fmt.Printf(" backend : %s\n", planLabel(p))
fmt.Printf(" jobs : %d\n", p.jobs)
fmt.Printf(" flags : %s\n", dim(strings.Join(p.flags, " ")))
}
-299
View File
@@ -540,302 +540,3 @@ func duckduckgoSearch(query string, limit int) ([]searchResult, error) {
}
// ─── définitions d'outils (schémas OpenAI, comme llm_client.go) ────────────────────
func webSearchTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_search",
Description: "Recherche sur le web via DuckDuckGo. Renvoie une liste classée de {title, url, snippet}. " +
"À utiliser quand l'utilisateur pose une question sans URL, cherche un outil/une bibliothèque, " +
"ou demande une information récente. À enchaîner avec web_open + web_read sur le meilleur résultat.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"query": map[string]any{"type": "string", "description": "Requête (langage naturel ou mots-clés)"},
"limit": map[string]any{"type": "integer", "description": "Nb max de résultats (défaut 8, max 20)"},
},
"required": []string{"query"},
},
}}
}
func webOpenTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_open",
Description: "Récupère une URL et renvoie SEULEMENT les métadonnées (taille, nb de lignes, plan des titres). " +
"Ne renvoie PAS le contenu. Toujours appeler ceci d'abord avant de lire. Résultat en cache 10 min " +
"— les web_read / web_grep suivants le réutilisent.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL complète à récupérer"},
"refresh": map[string]any{"type": "boolean", "description": "Ignore le cache et re-fetch. Défaut false."},
"actions": map[string]any{"type": "array", "items": map[string]any{"type": "string"},
"description": "Snippets JS à exécuter sur la page AVANT extraction (déplier des sections, cliquer 'voir plus', etc.)."},
"dismiss_popups": map[string]any{"type": "boolean", "description": "Ferme auto les bandeaux cookies/overlays. Défaut true."},
"wait_for": map[string]any{"type": "string", "description": "Sélecteur CSS ou expr JS à attendre après les actions."},
},
"required": []string{"url"},
},
}}
}
func webReadTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_read",
Description: "Lit une plage de lignes d'une URL déjà ouverte avec web_open. Coût en tokens prévisible. " +
"Lignes 1-indexées, préfixées par leur numéro.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL précédemment ouverte avec web_open"},
"offset": map[string]any{"type": "integer", "description": "Ligne de départ (1-indexée, défaut 1)"},
"limit": map[string]any{"type": "integer", "description": "Nb de lignes (défaut 80, max 500)"},
},
"required": []string{"url"},
},
}}
}
func webGrepTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_grep",
Description: "Recherche regex dans une URL déjà ouverte avec web_open. Renvoie les lignes correspondantes " +
"avec contexte et numéros. Idéal quand la page est longue et qu'on connaît un mot-clé.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL précédemment ouverte avec web_open"},
"pattern": map[string]any{"type": "string", "description": "Motif regex (insensible à la casse)"},
"context": map[string]any{"type": "integer", "description": "Lignes de contexte autour de chaque match. Défaut 2."},
"max_matches": map[string]any{"type": "integer", "description": "Plafond de matches renvoyés. Défaut 30."},
},
"required": []string{"url", "pattern"},
},
}}
}
// ─── exécution des outils (appelée par le dispatch de llm_client.go) ───────────────
func toolWebSearch(args map[string]any) string {
query, _ := args["query"].(string)
limit := 8
if v, ok := args["limit"].(float64); ok {
limit = int(v)
}
if limit < 1 {
limit = 1
}
if limit > 20 {
limit = 20
}
results, err := duckduckgoSearch(query, limit)
if err != nil {
return "❌ Recherche échouée : " + err.Error()
}
if len(results) == 0 {
return fmt.Sprintf("Aucun résultat pour « %s »", query)
}
var b strings.Builder
fmt.Fprintf(&b, "# Recherche : %s\n%d résultat(s) DuckDuckGo\n\n", query, len(results))
for i, r := range results {
fmt.Fprintf(&b, "%d. %s\n %s\n %s\n\n", i+1, r.Title, r.URL, r.Snippet)
}
return strings.TrimRight(b.String(), "\n")
}
func toolWebOpen(args map[string]any) string {
u, _ := args["url"].(string)
opts := fetchOptions{dismissPopups: true}
if v, ok := args["refresh"].(bool); ok {
opts.force = v
}
if v, ok := args["dismiss_popups"].(bool); ok {
opts.dismissPopups = v
}
if v, ok := args["wait_for"].(string); ok {
opts.waitFor = v
}
if arr, ok := args["actions"].([]any); ok {
for _, a := range arr {
if s, ok := a.(string); ok {
opts.actions = append(opts.actions, s)
}
}
}
entry, err := getPage(u, opts)
if err != nil {
return "❌ " + err.Error()
}
total := len(entry.lines)
chars := total
for _, l := range entry.lines {
chars += len(l)
}
return fmt.Sprintf("# Ouvert : %s\nTotal : %d lignes, %s (%d caractères)\nEn cache 10 min. Utilise web_read ou web_grep pour lire.\n\n## Plan (n° de ligne des titres)\n```\n%s\n```",
entry.url, total, formatBytes(chars), chars, extractOutline(entry.lines))
}
func toolWebRead(args map[string]any) string {
u, _ := args["url"].(string)
entry := findCached(u)
if entry == nil {
return fmt.Sprintf("❌ Page absente du cache. Appelle d'abord web_open(\"%s\").", u)
}
total := len(entry.lines)
offset := 1
if v, ok := args["offset"].(float64); ok {
offset = int(v)
}
if offset < 1 {
offset = 1
}
limit := 80
if v, ok := args["limit"].(float64); ok {
limit = int(v)
}
if limit < 1 {
limit = 1
}
if limit > 500 {
limit = 500
}
start := offset - 1
if start > total {
start = total
}
end := start + limit
if end > total {
end = total
}
slice := entry.lines[start:end]
remaining := total - end
tail := " (fin de page)"
if remaining > 0 {
tail = fmt.Sprintf(" (%d de plus en dessous)", remaining)
}
return fmt.Sprintf("# %s\nLignes %d–%d sur %d%s\n\n```\n%s\n```",
entry.url, offset, end, total, tail, formatLines(slice, offset))
}
func toolWebGrep(args map[string]any) string {
u, _ := args["url"].(string)
pattern, _ := args["pattern"].(string)
entry := findCached(u)
if entry == nil {
return fmt.Sprintf("❌ Page absente du cache. Appelle d'abord web_open(\"%s\").", u)
}
re, err := regexp.Compile("(?i)" + pattern)
if err != nil {
return "❌ Regex invalide : " + err.Error()
}
ctx := 2
if v, ok := args["context"].(float64); ok {
ctx = int(v)
}
if ctx < 0 {
ctx = 0
}
maxMatches := 30
if v, ok := args["max_matches"].(float64); ok {
maxMatches = int(v)
}
if maxMatches < 1 {
maxMatches = 1
}
lines := entry.lines
var matchIdx []int
for i := 0; i < len(lines) && len(matchIdx) < maxMatches; i++ {
if re.MatchString(lines[i]) {
matchIdx = append(matchIdx, i)
}
}
if len(matchIdx) == 0 {
return fmt.Sprintf("# %s\nAucun match pour /%s/i", entry.url, pattern)
}
// Fusionne les fenêtres de contexte qui se chevauchent.
type rng struct{ s, e int }
var ranges []rng
for _, i := range matchIdx {
s := i - ctx
if s < 0 {
s = 0
}
e := i + ctx
if e > len(lines)-1 {
e = len(lines) - 1
}
if n := len(ranges); n > 0 && s <= ranges[n-1].e+1 {
if e > ranges[n-1].e {
ranges[n-1].e = e
}
} else {
ranges = append(ranges, rng{s, e})
}
}
var blocks []string
for _, r := range ranges {
blocks = append(blocks, "```\n"+formatLines(lines[r.s:r.e+1], r.s+1)+"\n```")
}
capped := ""
if len(matchIdx) == maxMatches {
capped = fmt.Sprintf(" (plafonné à %d)", maxMatches)
}
return fmt.Sprintf("# %s\n%d match(es) pour /%s/i%s\n\n%s",
entry.url, len(matchIdx), pattern, capped, strings.Join(blocks, "\n\n---\n\n"))
}
// ─── CLI : jean internet [on|off|status|url <url>] ──────────────────────────
func cmdInternet(args []string) error {
sub := ""
if len(args) > 0 {
sub = args[0]
}
switch sub {
case "on":
if crawl4aiURL() == "" {
return fmt.Errorf("configure d'abord l'URL : jean internet url <url>")
}
if err := setInternetEnabled(true); err != nil {
return err
}
fmt.Println(green("[ok]") + " accès internet activé — l'IA dispose de web_search/web_open/web_read/web_grep (si le mode agent est actif)")
case "off":
if err := setInternetEnabled(false); err != nil {
return err
}
fmt.Println(green("[ok]") + " accès internet désactivé")
case "url":
if len(args) < 2 {
return fmt.Errorf("usage: jean internet url <url> (ex: http://localhost:11235)")
}
u := strings.TrimRight(strings.TrimSpace(args[1]), "/")
if err := SetConfigKey("CRAWL4AI_URL", u); err != nil {
return err
}
reachMu.Lock()
reachURL = "" // invalide le cache de reachability
reachMu.Unlock()
fmt.Printf("%s serveur Crawl4AI : %s\n", green("[ok]"), bold(u))
case "", "status", "list":
state := dim("off")
if internetEnabled() {
state = green("on")
}
fmt.Printf("%s état: %s\n", cyan("Accès internet"), state)
u := crawl4aiURL()
if u == "" {
fmt.Printf(" serveur : %s — configure : jean internet url <url>\n", dim("(non configuré)"))
return nil
}
reach := red("injoignable")
if crawlReachable() {
reach = green("joignable")
}
fmt.Printf(" serveur : %s (%s)\n", bold(u), reach)
fmt.Printf(" outils : web_search, web_open, web_read, web_grep\n")
default:
return fmt.Errorf("usage: jean internet [on|off|status|url <url>]")
}
return nil
}
+308
View File
@@ -0,0 +1,308 @@
// chat_internet_tools.go — les 4 outils web exposés au modèle (web_search,
// web_open, web_read, web_grep) + la sous-commande `jean internet`.
package jean
import (
"fmt"
"regexp"
"strings"
)
func webSearchTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_search",
Description: "Recherche sur le web via DuckDuckGo. Renvoie une liste classée de {title, url, snippet}. " +
"À utiliser quand l'utilisateur pose une question sans URL, cherche un outil/une bibliothèque, " +
"ou demande une information récente. À enchaîner avec web_open + web_read sur le meilleur résultat.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"query": map[string]any{"type": "string", "description": "Requête (langage naturel ou mots-clés)"},
"limit": map[string]any{"type": "integer", "description": "Nb max de résultats (défaut 8, max 20)"},
},
"required": []string{"query"},
},
}}
}
func webOpenTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_open",
Description: "Récupère une URL et renvoie SEULEMENT les métadonnées (taille, nb de lignes, plan des titres). " +
"Ne renvoie PAS le contenu. Toujours appeler ceci d'abord avant de lire. Résultat en cache 10 min " +
"— les web_read / web_grep suivants le réutilisent.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL complète à récupérer"},
"refresh": map[string]any{"type": "boolean", "description": "Ignore le cache et re-fetch. Défaut false."},
"actions": map[string]any{"type": "array", "items": map[string]any{"type": "string"},
"description": "Snippets JS à exécuter sur la page AVANT extraction (déplier des sections, cliquer 'voir plus', etc.)."},
"dismiss_popups": map[string]any{"type": "boolean", "description": "Ferme auto les bandeaux cookies/overlays. Défaut true."},
"wait_for": map[string]any{"type": "string", "description": "Sélecteur CSS ou expr JS à attendre après les actions."},
},
"required": []string{"url"},
},
}}
}
func webReadTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_read",
Description: "Lit une plage de lignes d'une URL déjà ouverte avec web_open. Coût en tokens prévisible. " +
"Lignes 1-indexées, préfixées par leur numéro.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL précédemment ouverte avec web_open"},
"offset": map[string]any{"type": "integer", "description": "Ligne de départ (1-indexée, défaut 1)"},
"limit": map[string]any{"type": "integer", "description": "Nb de lignes (défaut 80, max 500)"},
},
"required": []string{"url"},
},
}}
}
func webGrepTool() Tool {
return Tool{Type: "function", Function: ToolFunction{
Name: "web_grep",
Description: "Recherche regex dans une URL déjà ouverte avec web_open. Renvoie les lignes correspondantes " +
"avec contexte et numéros. Idéal quand la page est longue et qu'on connaît un mot-clé.",
Parameters: map[string]any{
"type": "object",
"properties": map[string]any{
"url": map[string]any{"type": "string", "description": "URL précédemment ouverte avec web_open"},
"pattern": map[string]any{"type": "string", "description": "Motif regex (insensible à la casse)"},
"context": map[string]any{"type": "integer", "description": "Lignes de contexte autour de chaque match. Défaut 2."},
"max_matches": map[string]any{"type": "integer", "description": "Plafond de matches renvoyés. Défaut 30."},
},
"required": []string{"url", "pattern"},
},
}}
}
// ─── exécution des outils (appelée par le dispatch de llm_client.go) ───────────────
func toolWebSearch(args map[string]any) string {
query, _ := args["query"].(string)
limit := 8
if v, ok := args["limit"].(float64); ok {
limit = int(v)
}
if limit < 1 {
limit = 1
}
if limit > 20 {
limit = 20
}
results, err := duckduckgoSearch(query, limit)
if err != nil {
return "❌ Recherche échouée : " + err.Error()
}
if len(results) == 0 {
return fmt.Sprintf("Aucun résultat pour « %s »", query)
}
var b strings.Builder
fmt.Fprintf(&b, "# Recherche : %s\n%d résultat(s) DuckDuckGo\n\n", query, len(results))
for i, r := range results {
fmt.Fprintf(&b, "%d. %s\n %s\n %s\n\n", i+1, r.Title, r.URL, r.Snippet)
}
return strings.TrimRight(b.String(), "\n")
}
func toolWebOpen(args map[string]any) string {
u, _ := args["url"].(string)
opts := fetchOptions{dismissPopups: true}
if v, ok := args["refresh"].(bool); ok {
opts.force = v
}
if v, ok := args["dismiss_popups"].(bool); ok {
opts.dismissPopups = v
}
if v, ok := args["wait_for"].(string); ok {
opts.waitFor = v
}
if arr, ok := args["actions"].([]any); ok {
for _, a := range arr {
if s, ok := a.(string); ok {
opts.actions = append(opts.actions, s)
}
}
}
entry, err := getPage(u, opts)
if err != nil {
return "❌ " + err.Error()
}
total := len(entry.lines)
chars := total
for _, l := range entry.lines {
chars += len(l)
}
return fmt.Sprintf("# Ouvert : %s\nTotal : %d lignes, %s (%d caractères)\nEn cache 10 min. Utilise web_read ou web_grep pour lire.\n\n## Plan (n° de ligne des titres)\n```\n%s\n```",
entry.url, total, formatBytes(chars), chars, extractOutline(entry.lines))
}
func toolWebRead(args map[string]any) string {
u, _ := args["url"].(string)
entry := findCached(u)
if entry == nil {
return fmt.Sprintf("❌ Page absente du cache. Appelle d'abord web_open(\"%s\").", u)
}
total := len(entry.lines)
offset := 1
if v, ok := args["offset"].(float64); ok {
offset = int(v)
}
if offset < 1 {
offset = 1
}
limit := 80
if v, ok := args["limit"].(float64); ok {
limit = int(v)
}
if limit < 1 {
limit = 1
}
if limit > 500 {
limit = 500
}
start := offset - 1
if start > total {
start = total
}
end := start + limit
if end > total {
end = total
}
slice := entry.lines[start:end]
remaining := total - end
tail := " (fin de page)"
if remaining > 0 {
tail = fmt.Sprintf(" (%d de plus en dessous)", remaining)
}
return fmt.Sprintf("# %s\nLignes %d–%d sur %d%s\n\n```\n%s\n```",
entry.url, offset, end, total, tail, formatLines(slice, offset))
}
func toolWebGrep(args map[string]any) string {
u, _ := args["url"].(string)
pattern, _ := args["pattern"].(string)
entry := findCached(u)
if entry == nil {
return fmt.Sprintf("❌ Page absente du cache. Appelle d'abord web_open(\"%s\").", u)
}
re, err := regexp.Compile("(?i)" + pattern)
if err != nil {
return "❌ Regex invalide : " + err.Error()
}
ctx := 2
if v, ok := args["context"].(float64); ok {
ctx = int(v)
}
if ctx < 0 {
ctx = 0
}
maxMatches := 30
if v, ok := args["max_matches"].(float64); ok {
maxMatches = int(v)
}
if maxMatches < 1 {
maxMatches = 1
}
lines := entry.lines
var matchIdx []int
for i := 0; i < len(lines) && len(matchIdx) < maxMatches; i++ {
if re.MatchString(lines[i]) {
matchIdx = append(matchIdx, i)
}
}
if len(matchIdx) == 0 {
return fmt.Sprintf("# %s\nAucun match pour /%s/i", entry.url, pattern)
}
// Fusionne les fenêtres de contexte qui se chevauchent.
type rng struct{ s, e int }
var ranges []rng
for _, i := range matchIdx {
s := i - ctx
if s < 0 {
s = 0
}
e := i + ctx
if e > len(lines)-1 {
e = len(lines) - 1
}
if n := len(ranges); n > 0 && s <= ranges[n-1].e+1 {
if e > ranges[n-1].e {
ranges[n-1].e = e
}
} else {
ranges = append(ranges, rng{s, e})
}
}
var blocks []string
for _, r := range ranges {
blocks = append(blocks, "```\n"+formatLines(lines[r.s:r.e+1], r.s+1)+"\n```")
}
capped := ""
if len(matchIdx) == maxMatches {
capped = fmt.Sprintf(" (plafonné à %d)", maxMatches)
}
return fmt.Sprintf("# %s\n%d match(es) pour /%s/i%s\n\n%s",
entry.url, len(matchIdx), pattern, capped, strings.Join(blocks, "\n\n---\n\n"))
}
// ─── CLI : jean internet [on|off|status|url <url>] ──────────────────────────
func cmdInternet(args []string) error {
sub := ""
if len(args) > 0 {
sub = args[0]
}
switch sub {
case "on":
if crawl4aiURL() == "" {
return fmt.Errorf("configure d'abord l'URL : jean internet url <url>")
}
if err := setInternetEnabled(true); err != nil {
return err
}
fmt.Println(green("[ok]") + " accès internet activé — l'IA dispose de web_search/web_open/web_read/web_grep (si le mode agent est actif)")
case "off":
if err := setInternetEnabled(false); err != nil {
return err
}
fmt.Println(green("[ok]") + " accès internet désactivé")
case "url":
if len(args) < 2 {
return fmt.Errorf("usage: jean internet url <url> (ex: http://localhost:11235)")
}
u := strings.TrimRight(strings.TrimSpace(args[1]), "/")
if err := SetConfigKey("CRAWL4AI_URL", u); err != nil {
return err
}
reachMu.Lock()
reachURL = "" // invalide le cache de reachability
reachMu.Unlock()
fmt.Printf("%s serveur Crawl4AI : %s\n", green("[ok]"), bold(u))
case "", "status", "list":
state := dim("off")
if internetEnabled() {
state = green("on")
}
fmt.Printf("%s état: %s\n", cyan("Accès internet"), state)
u := crawl4aiURL()
if u == "" {
fmt.Printf(" serveur : %s — configure : jean internet url <url>\n", dim("(non configuré)"))
return nil
}
reach := red("injoignable")
if crawlReachable() {
reach = green("joignable")
}
fmt.Printf(" serveur : %s (%s)\n", bold(u), reach)
fmt.Printf(" outils : web_search, web_open, web_read, web_grep\n")
default:
return fmt.Errorf("usage: jean internet [on|off|status|url <url>]")
}
return nil
}
+602
View File
@@ -0,0 +1,602 @@
// web_api.go — handlers REST /api/* (statut, config, presets, modèles,
// mémoire, agent, clés, bench…) du serveur web local.
package jean
import (
"encoding/json"
"net"
"net/http"
"os"
"os/exec"
"strconv"
"strings"
)
func handlePing(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, map[string]any{"ok": true, "service": "jean", "version": Version})
}
// handleStatus reports service state cross-platform via serviceIsActive
// (systemd sous Linux, supervision par PID-file sous Windows — voir sys_service_*.go).
func handleStatus(w http.ResponseWriter, r *http.Request) {
active := serviceIsActive()
state := "inactive"
if active {
state = "active"
}
health := false
if active {
health = healthCheck()
}
ctx := 32768
if v := ReadConfig()["CTX"]; v != "" {
if n, err := strconv.Atoi(v); err == nil && n > 0 {
ctx = n
}
}
sendJSON(w, 200, map[string]any{
"state": state,
"active": active,
"health": health,
"port": LLMPort(),
"ctx": ctx,
"version": Version,
})
}
func handleVram(w http.ResponseWriter, r *http.Request) {
out, err := hideCmd(exec.Command("nvidia-smi",
"--query-gpu=name,memory.used,memory.total,utilization.gpu,temperature.gpu",
"--format=csv,noheader,nounits")).Output()
gpus := []map[string]any{}
if err == nil {
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
parts := strings.Split(line, ",")
if len(parts) != 5 {
continue
}
for i := range parts {
parts[i] = strings.TrimSpace(parts[i])
}
used, _ := strconv.Atoi(parts[1])
total, _ := strconv.Atoi(parts[2])
util, _ := strconv.Atoi(parts[3])
temp, _ := strconv.Atoi(parts[4])
gpus = append(gpus, map[string]any{
"name": parts[0], "used": used, "total": total, "util": util, "temp": temp,
})
}
}
sendJSON(w, 200, gpus)
}
func handleConfigEnv(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, ReadConfig())
}
// handleBackends scans JEAN_HOME/backends/<name>/ for a llama-server binary,
// trying common build subpaths (build/bin, build-sm120/bin, bin, .).
// Returns [{name, path}].
func handleBackends(w http.ResponseWriter, r *http.Request) {
root := JeanHome() + "/backends"
entries, err := os.ReadDir(root)
if err != nil {
sendJSON(w, 200, []map[string]any{})
return
}
subpaths := []string{
"build/bin/llama-server", "build-sm120/bin/llama-server",
"build/llama-server", "bin/llama-server", "llama-server",
// Layout du générateur Visual Studio (multi-config) + suffixe .exe Windows.
"build/bin/Release/llama-server.exe", "build/bin/llama-server.exe",
"build/bin/Release/llama-server", "llama-server.exe",
}
out := []map[string]any{}
for _, e := range entries {
// e can be a directory or a symlink to one; either is fine.
name := e.Name()
if strings.HasPrefix(name, ".") {
continue
}
for _, sp := range subpaths {
p := root + "/" + name + "/" + sp
if fi, err := os.Stat(p); err == nil && !fi.IsDir() {
out = append(out, map[string]any{"name": name, "path": p})
break
}
}
}
sendJSON(w, 200, out)
}
// handleModels lists *.gguf files in JEAN_HOME (size in bytes) for the preset
// editor's model picker.
func handleModels(w http.ResponseWriter, r *http.Request) {
entries, err := os.ReadDir(JeanHome())
if err != nil {
sendJSON(w, 200, []map[string]any{})
return
}
out := []map[string]any{}
for _, e := range entries {
if e.IsDir() || !strings.HasSuffix(strings.ToLower(e.Name()), ".gguf") {
continue
}
info, _ := e.Info()
size := int64(0)
if info != nil {
size = info.Size()
}
out = append(out, map[string]any{"name": e.Name(), "size": size})
}
sendJSON(w, 200, out)
}
func handlePresets(w http.ResponseWriter, r *http.Request) {
list, err := ListPresets()
if err != nil {
sendJSON(w, 500, map[string]any{"error": err.Error()})
return
}
store := loadBenchStore()
out := []map[string]any{}
for _, p := range list {
item := map[string]any{"id": p.ID, "name": p.Name, "active": p.Active}
if content, err := ReadPreset(p.ID); err == nil {
if q := detectQuant(content); q != "" {
item["quant"] = q
}
if r := presetReasoning(content); reasoningActive(r) {
item["reasoning"] = strings.ToLower(r)
}
}
if sb, ok := store[p.ID]; ok {
item["bench"] = map[string]any{
"prefill": sb.Result.PromptPerSecond,
"decode": sb.Result.PredictedPerSec,
"at": sb.At,
}
}
out = append(out, item)
}
sendJSON(w, 200, out)
}
func handlePreset(w http.ResponseWriter, r *http.Request) {
id := strings.TrimSpace(r.URL.Query().Get("id"))
if id == "" {
// new preset → seed from current config.env so users can tweak rather than start blank
b, _ := os.ReadFile(confPath())
sendJSON(w, 200, map[string]any{"id": "", "name": "", "content": string(b)})
return
}
content, err := ReadPreset(id)
if err != nil {
sendJSON(w, 404, map[string]any{"error": "not found"})
return
}
sendJSON(w, 200, map[string]any{"id": id, "name": presetDisplayName(content, id), "content": content})
}
// presetSaveReq is the preset editor payload. `id` identifies an existing
// preset to update ("" creates a new one); `name` is the display name.
type presetSaveReq struct {
ID string `json:"id"`
Name string `json:"name"`
Content string `json:"content"`
DeleteModel bool `json:"deleteModel"`
}
// saveReq is the skill editor payload (skills keep name-as-identity + rename).
type saveReq struct {
Name string `json:"name"`
Old string `json:"old"`
Content string `json:"content"`
}
func handlePresetSave(w http.ResponseWriter, r *http.Request) {
var req presetSaveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
newID, err := SavePreset(req.ID, req.Name, req.Content)
if err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "id": newID, "name": req.Name})
}
func handlePresetDelete(w http.ResponseWriter, r *http.Request) {
var req presetSaveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
// Capture the referenced model before the preset file disappears, so we can
// optionally delete the .gguf alongside it.
model := ""
if req.DeleteModel {
if content, err := ReadPreset(req.ID); err == nil {
model = modelFromPresetContent(content)
}
}
if err := DeletePreset(req.ID); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
modelDeleted, modelErr := "", ""
if req.DeleteModel && model != "" {
if err := deleteModelFile(model); err != nil {
modelErr = err.Error()
} else {
modelDeleted = model
}
}
sendJSON(w, 200, map[string]any{"ok": true, "modelDeleted": modelDeleted, "modelError": modelErr})
}
// handleAgent renvoie l'état du mode agent ET la liste des pages mémoire (que
// l'IA gère via les outils mem_*) — un seul aller-retour pour l'UI. La clé
// "skills" est conservée en miroir de "pages" pour l'ancien portail ajean.link.
func handleAgent(w http.ResponseWriter, r *http.Request) {
pages := MemList()
out := []map[string]any{}
for _, p := range pages {
out = append(out, map[string]any{"name": p.Name, "desc": p.Title})
}
sendJSON(w, 200, map[string]any{"enabled": agentEnabled(), "tool_limit": toolLimitEnabled(), "compact": compactEnabled(), "mem_mode": string(memMode()), "pages": out, "skills": out})
}
// handleMemoryMode lit/écrit le mode mémoire (off / ondemand / always).
//
// GET → {mode}
// POST {mode} → persiste MEM_MODE
func handleMemoryMode(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Mode string `json:"mode"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
// On normalise via memMode() en réinjectant la valeur : toute entrée
// inconnue retombe sur "always", donc on valide en passant par le parseur.
m := MemAlways
switch MemMode(strings.ToLower(strings.TrimSpace(req.Mode))) {
case MemOff:
m = MemOff
case MemOnDemand:
m = MemOnDemand
case MemAlways:
m = MemAlways
}
if err := setMemMode(m); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
sendJSON(w, 200, map[string]any{"ok": true, "mode": string(memMode())})
}
// handleToolLimitToggle active/désactive le plafond d'appels d'outils par tour
// (config.env TOOL_LIMIT). On=limité (défaut), off=quasi illimité.
func handleToolLimitToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
val := ""
if !req.On {
val = "off"
}
if err := SetConfigKey("TOOL_LIMIT", val); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "tool_limit": toolLimitEnabled()})
}
// handleCompactToggle active/désactive le compactage automatique du contexte
// (config.env COMPACT). On=compacte (défaut), off=jamais.
func handleCompactToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
val := ""
if !req.On {
val = "off"
}
if err := SetConfigKey("COMPACT", val); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "compact": compactEnabled()})
}
func handleAgentToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
if err := setAgentEnabled(req.On); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "enabled": agentEnabled()})
}
// handleInternet pilote l'accès web de l'IA (serveur Crawl4AI).
//
// GET → {enabled, url, reachable}
// POST {enabled, url} → enregistre CRAWL4AI_URL + le drapeau .internet_enabled
//
// handleAPIKey expose et pilote la clé d'accès à l'endpoint compatible OpenAI
// (llama-server /v1). GET renvoie l'état ; POST {action:"generate"|"set"|"clear",
// key?} l'écrit puis redémarre le service (llama-server lit --api-key au lancement).
func handleAPIKey(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Action string `json:"action"`
Key string `json:"key"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
var key string
switch req.Action {
case "generate":
key = genAPIKey()
case "set":
key = strings.TrimSpace(req.Key)
case "clear":
key = ""
default:
sendJSON(w, 400, map[string]any{"ok": false, "error": "action inconnue"})
return
}
if err := writeAPIKey(key); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
// La clé n'est appliquée qu'au (re)démarrage de llama-server.
if serviceIsActive() {
_ = serviceAction("restart")
}
}
k := readAPIKey()
sendJSON(w, 200, map[string]any{
"ok": true,
"set": k != "",
"key": k,
"masked": maskAPIKey(k),
"port": LLMPort(),
"host": localIP(),
// Accès OpenAI PUBLIC via ajean.link (passthrough SNI, VPS aveugle) : si
// activé, l'URL publique est https://<machine>.oai.ajean.link/v1.
"oai_public": oaiPublicEnabled(),
"machine": machineID(),
})
}
// handleOAIPublic pilote le drapeau d'accès OpenAI public (exposition via
// ajean.link). GET renvoie l'état ; POST {enabled} l'active/coupe en direct
// (aucun redémarrage : le démux du tunnel relit le drapeau à chaque connexion).
func handleOAIPublic(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Enabled *bool `json:"enabled"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if req.Enabled != nil {
if err := setOAIPublic(*req.Enabled); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
}
sendJSON(w, 200, map[string]any{
"ok": true,
"enabled": oaiPublicEnabled(),
"machine": machineID(),
})
}
// localIP best-effort renvoie l'IPv4 LAN primaire de la machine (l'IP source du
// trafic sortant), ou "localhost" à défaut. Sert à annoncer l'endpoint OpenAI
// avec une adresse correcte sur le réseau local MÊME quand l'UI est atteinte via
// le tunnel ajean.link (où location.hostname serait le domaine du relais, faux).
func localIP() string {
conn, err := net.Dial("udp", "8.8.8.8:80")
if err != nil {
return "localhost"
}
defer conn.Close()
if a, ok := conn.LocalAddr().(*net.UDPAddr); ok && a.IP != nil {
return a.IP.String()
}
return "localhost"
}
func handleInternet(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Enabled *bool `json:"enabled"`
URL *string `json:"url"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
if req.URL != nil {
u := strings.TrimRight(strings.TrimSpace(*req.URL), "/")
if err := SetConfigKey("CRAWL4AI_URL", u); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
reachMu.Lock()
reachURL = "" // invalide le cache de reachability
reachMu.Unlock()
}
if req.Enabled != nil {
if err := setInternetEnabled(*req.Enabled); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
}
sendJSON(w, 200, map[string]any{
"ok": true,
"enabled": internetEnabled(),
"url": crawl4aiURL(),
"reachable": crawlReachable(),
})
}
// handleMem / handleMemSave / handleMemDelete : éditeur web des pages mémoire
// (MEMORY/<nom>.md). Payload partagé saveReq (name/old/content) ; "name" = nom
// de fichier de la page.
func handleMem(w http.ResponseWriter, r *http.Request) {
name := strings.TrimSpace(r.URL.Query().Get("name"))
if name == "" {
sendJSON(w, 200, map[string]any{"name": "", "content": "# nouvelle page\n\nNote ici ce que jean doit retenir entre les sessions.\n"})
return
}
c := MemContent(name)
if c == "" {
sendJSON(w, 404, map[string]any{"error": "not found"})
return
}
sendJSON(w, 200, map[string]any{"name": name, "content": c})
}
func handleMemSave(w http.ResponseWriter, r *http.Request) {
var req saveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if err := MemSave(req.Name, req.Old, req.Content); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "name": req.Name})
}
func handleMemDelete(w http.ResponseWriter, r *http.Request) {
var req saveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if err := MemDelete(req.Name); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleSwitch(w http.ResponseWriter, r *http.Request) {
var req struct {
N int `json:"n"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
list, err := ListPresets()
if err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
if req.N < 1 || req.N > len(list) {
sendJSON(w, 400, map[string]any{"ok": false, "error": "index hors limites"})
return
}
target := list[req.N-1]
if err := SwitchToPreset(target.Path); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "preset": target.Name})
}
// svcHandler returns an HTTP handler that triggers a start/stop/restart through
// the cross-platform serviceAction (systemd sous Linux, supervision PID-file
// sous Windows — voir sys_service_*.go). C'est ce qui permet à un client distant
// de relancer Jean.
func svcHandler(action string) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
err := serviceAction(action)
msg := "ok"
if err != nil {
msg = err.Error()
}
sendJSON(w, 200, map[string]any{"ok": err == nil, "out": msg})
}
}
// handleChat is the SSE proxy with tool-calling. The HTTP handler writes raw
// data: lines matching what the embedded JS expects (delta.content,
// delta.reasoning_content, delta.tool_used).
// handleBench runs `runBench` synchronously. Long enough (~30-60s) that we
// rely on the client side to show a spinner / disable the button.
func handleBench(w http.ResponseWriter, r *http.Request) {
nPrompt, nPredict := 2000, 300
if v := r.URL.Query().Get("prompt"); v != "" {
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
nPrompt = parsed
}
}
if v := r.URL.Query().Get("n"); v != "" {
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
nPredict = parsed
}
}
res, err := runBench(nPrompt, nPredict)
if err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "result": res})
}
// handleBenchLast returns the most recent persisted benchmark, or {ok:false}
// when none has been run yet.
func handleBenchLast(w http.ResponseWriter, r *http.Request) {
sb := loadLastBench()
if sb == nil {
sendJSON(w, 200, map[string]any{"ok": false})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "result": sb.Result, "model": sb.Model, "at": sb.At})
}
// chatReq est le corps d'une requête de chat (commun au chat clair et au chat E2E).
type chatReq struct {
Messages []Message `json:"messages"`
Temperature float64 `json:"temperature"`
// Optional per-request override of the agent mode (used by ajean.link
// agents, which carry their own toggle). nil = inherit the machine's
// global config. Tools/Skills sont conservés pour la rétro-compat des
// anciens clients relais : l'un OU l'autre à true active le mode agent.
Agent *bool `json:"agent"`
Tools *bool `json:"tools"`
Skills *bool `json:"skills"`
// Override par requête de l'accès internet (outils web). nil = config machine.
Internet *bool `json:"internet"`
// Taille réelle du contexte au tour précédent (usage.prompt_tokens + tokens
// générés), rapportée par le client qui l'affiche déjà. Sert à décider du
// compactage sur le VRAI décompte plutôt qu'une estimation. 0 = inconnu.
CtxUsed int `json:"ctx_used"`
// Nouveau modèle « conversation serveur » : Message = texte du tour à lancer
// (via /api/chat/send) ; From = dernier Seq déjà vu par le client (le flux
// d'abonnement rejoue Log[From:] puis suit le direct).
Message string `json:"message"`
From int `json:"from"`
}
// capsFromBody dérive les capacités du tour à partir des overrides éventuels du
// corps de requête (agents ajean.link portant leurs propres toggles), sinon la
// config machine.
+135
View File
@@ -0,0 +1,135 @@
// web_chat.go — endpoints de chat du serveur web local : envoi/stop/reset,
// flux SSE d'abonnement à la conversation serveur (voir chat_conversation.go).
package jean
import (
"context"
"encoding/json"
"net/http"
"strings"
"sync"
"time"
)
func capsFromBody(body chatReq) Caps {
caps := globalCaps()
if body.Agent != nil {
caps.Agent = *body.Agent
} else if body.Tools != nil || body.Skills != nil {
caps.Agent = (body.Tools != nil && *body.Tools) || (body.Skills != nil && *body.Skills)
}
if body.Internet != nil {
caps.Internet = *body.Internet && crawlReachable()
}
return caps
}
// sseHeartbeat garde la réponse SSE active en écrivant un commentaire (`: ping`,
// ignoré par le parseur côté navigateur, aucun contenu donc rien à chiffrer)
// toutes les ~15 s. Sans ça, un long silence (exécution d'outil en mode agent,
// gros prefill) laisse la réponse inactive et un proxy intermédiaire (Cloudflare,
// ~100 s) la coupe → le fetch navigateur échoue (« Load failed »). Retourne un
// mutex à partager avec l'émetteur (writes concurrents sur le même w) et une
// fonction d'arrêt à différer.
func sseHeartbeat(w http.ResponseWriter, flusher http.Flusher) (*sync.Mutex, func()) {
mu := &sync.Mutex{}
done := make(chan struct{})
go func() {
// 4 s (et non 15) : borne le temps qu'un dernier bout de flux peut rester
// coincé dans un buffer proxy (Cloudflare) faute d'octets pour le pousser.
t := time.NewTicker(4 * time.Second)
defer t.Stop()
for {
select {
case <-done:
return
case <-t.C:
mu.Lock()
_, err := w.Write([]byte(": ping\n\n"))
if flusher != nil {
flusher.Flush()
}
mu.Unlock()
if err != nil {
return
}
}
}
}()
return mu, func() { close(done) }
}
// runChatStream est désormais un pur ABONNÉ au journal de la conversation serveur :
// il rejoue Log[body.From:] puis suit le direct, jusqu'à ce que la connexion (ctx)
// se ferme. La GÉNÉRATION est lancée séparément par /api/chat/send dans une
// goroutine détachée — fermer le navigateur n'arrête donc plus rien. Partagé par
// handleChat (clair) et handleE2EChat (chiffré).
func runChatStream(ctx context.Context, body chatReq, emit func(map[string]any) bool) {
conv.Subscribe(ctx, body.From, emit)
}
// handleChatSend ajoute un message et lance la génération en arrière-plan. Réponse
// req/resp (les événements arrivent par le flux d'abonnement). Passe par le proxy
// tunnel /api/e2e/req pour app.ajean.link — aucun code E2E spécifique requis.
func handleChatSend(w http.ResponseWriter, r *http.Request) {
var body chatReq
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if strings.TrimSpace(body.Message) == "" {
sendJSON(w, 400, map[string]any{"ok": false, "error": "message vide"})
return
}
if err := conv.StartTurn(body.Message, capsFromBody(body), body.Temperature); err != nil {
// 409 = occupé (génération en cours) ; 503 = modèle pas prêt.
code := 503
if err == ErrBusy {
code = 409
}
sendJSON(w, code, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatStop(w http.ResponseWriter, r *http.Request) {
conv.Stop()
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatReset(w http.ResponseWriter, r *http.Request) {
conv.Reset()
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatState(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, conv.state())
}
func handleChat(w http.ResponseWriter, r *http.Request) {
var body chatReq
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
http.Error(w, err.Error(), 400)
return
}
w.Header().Set("Content-Type", "text/event-stream")
w.Header().Set("Cache-Control", "no-cache, no-transform")
w.Header().Set("X-Accel-Buffering", "no")
flusher, _ := w.(http.Flusher)
mu, stop := sseHeartbeat(w, flusher)
defer stop()
emit := func(obj map[string]any) bool {
b, _ := json.Marshal(map[string]any{"choices": []any{map[string]any{"delta": obj}}})
mu.Lock()
defer mu.Unlock()
if _, err := w.Write([]byte("data: " + string(b) + "\n\n")); err != nil {
return false
}
if flusher != nil {
flusher.Flush()
}
return true
}
runChatStream(r.Context(), body, emit)
}
-711
View File
@@ -2,7 +2,6 @@ package jean
import (
"bufio"
"context"
"embed"
"encoding/json"
"fmt"
@@ -219,713 +218,3 @@ func sendJSON(w http.ResponseWriter, code int, v any) {
// handlePing is a lightweight authenticated endpoint a client hits to verify
// connectivity AND that its key is valid (200 = bonne clé, 401 = mauvaise clé).
func handlePing(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, map[string]any{"ok": true, "service": "jean", "version": Version})
}
// handleStatus reports service state cross-platform via serviceIsActive
// (systemd sous Linux, supervision par PID-file sous Windows — voir sys_service_*.go).
func handleStatus(w http.ResponseWriter, r *http.Request) {
active := serviceIsActive()
state := "inactive"
if active {
state = "active"
}
health := false
if active {
health = healthCheck()
}
ctx := 32768
if v := ReadConfig()["CTX"]; v != "" {
if n, err := strconv.Atoi(v); err == nil && n > 0 {
ctx = n
}
}
sendJSON(w, 200, map[string]any{
"state": state,
"active": active,
"health": health,
"port": LLMPort(),
"ctx": ctx,
"version": Version,
})
}
func handleVram(w http.ResponseWriter, r *http.Request) {
out, err := hideCmd(exec.Command("nvidia-smi",
"--query-gpu=name,memory.used,memory.total,utilization.gpu,temperature.gpu",
"--format=csv,noheader,nounits")).Output()
gpus := []map[string]any{}
if err == nil {
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
parts := strings.Split(line, ",")
if len(parts) != 5 {
continue
}
for i := range parts {
parts[i] = strings.TrimSpace(parts[i])
}
used, _ := strconv.Atoi(parts[1])
total, _ := strconv.Atoi(parts[2])
util, _ := strconv.Atoi(parts[3])
temp, _ := strconv.Atoi(parts[4])
gpus = append(gpus, map[string]any{
"name": parts[0], "used": used, "total": total, "util": util, "temp": temp,
})
}
}
sendJSON(w, 200, gpus)
}
func handleConfigEnv(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, ReadConfig())
}
// handleBackends scans JEAN_HOME/backends/<name>/ for a llama-server binary,
// trying common build subpaths (build/bin, build-sm120/bin, bin, .).
// Returns [{name, path}].
func handleBackends(w http.ResponseWriter, r *http.Request) {
root := JeanHome() + "/backends"
entries, err := os.ReadDir(root)
if err != nil {
sendJSON(w, 200, []map[string]any{})
return
}
subpaths := []string{
"build/bin/llama-server", "build-sm120/bin/llama-server",
"build/llama-server", "bin/llama-server", "llama-server",
// Layout du générateur Visual Studio (multi-config) + suffixe .exe Windows.
"build/bin/Release/llama-server.exe", "build/bin/llama-server.exe",
"build/bin/Release/llama-server", "llama-server.exe",
}
out := []map[string]any{}
for _, e := range entries {
// e can be a directory or a symlink to one; either is fine.
name := e.Name()
if strings.HasPrefix(name, ".") {
continue
}
for _, sp := range subpaths {
p := root + "/" + name + "/" + sp
if fi, err := os.Stat(p); err == nil && !fi.IsDir() {
out = append(out, map[string]any{"name": name, "path": p})
break
}
}
}
sendJSON(w, 200, out)
}
// handleModels lists *.gguf files in JEAN_HOME (size in bytes) for the preset
// editor's model picker.
func handleModels(w http.ResponseWriter, r *http.Request) {
entries, err := os.ReadDir(JeanHome())
if err != nil {
sendJSON(w, 200, []map[string]any{})
return
}
out := []map[string]any{}
for _, e := range entries {
if e.IsDir() || !strings.HasSuffix(strings.ToLower(e.Name()), ".gguf") {
continue
}
info, _ := e.Info()
size := int64(0)
if info != nil {
size = info.Size()
}
out = append(out, map[string]any{"name": e.Name(), "size": size})
}
sendJSON(w, 200, out)
}
func handlePresets(w http.ResponseWriter, r *http.Request) {
list, err := ListPresets()
if err != nil {
sendJSON(w, 500, map[string]any{"error": err.Error()})
return
}
store := loadBenchStore()
out := []map[string]any{}
for _, p := range list {
item := map[string]any{"id": p.ID, "name": p.Name, "active": p.Active}
if content, err := ReadPreset(p.ID); err == nil {
if q := detectQuant(content); q != "" {
item["quant"] = q
}
if r := presetReasoning(content); reasoningActive(r) {
item["reasoning"] = strings.ToLower(r)
}
}
if sb, ok := store[p.ID]; ok {
item["bench"] = map[string]any{
"prefill": sb.Result.PromptPerSecond,
"decode": sb.Result.PredictedPerSec,
"at": sb.At,
}
}
out = append(out, item)
}
sendJSON(w, 200, out)
}
func handlePreset(w http.ResponseWriter, r *http.Request) {
id := strings.TrimSpace(r.URL.Query().Get("id"))
if id == "" {
// new preset → seed from current config.env so users can tweak rather than start blank
b, _ := os.ReadFile(confPath())
sendJSON(w, 200, map[string]any{"id": "", "name": "", "content": string(b)})
return
}
content, err := ReadPreset(id)
if err != nil {
sendJSON(w, 404, map[string]any{"error": "not found"})
return
}
sendJSON(w, 200, map[string]any{"id": id, "name": presetDisplayName(content, id), "content": content})
}
// presetSaveReq is the preset editor payload. `id` identifies an existing
// preset to update ("" creates a new one); `name` is the display name.
type presetSaveReq struct {
ID string `json:"id"`
Name string `json:"name"`
Content string `json:"content"`
DeleteModel bool `json:"deleteModel"`
}
// saveReq is the skill editor payload (skills keep name-as-identity + rename).
type saveReq struct {
Name string `json:"name"`
Old string `json:"old"`
Content string `json:"content"`
}
func handlePresetSave(w http.ResponseWriter, r *http.Request) {
var req presetSaveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
newID, err := SavePreset(req.ID, req.Name, req.Content)
if err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "id": newID, "name": req.Name})
}
func handlePresetDelete(w http.ResponseWriter, r *http.Request) {
var req presetSaveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
// Capture the referenced model before the preset file disappears, so we can
// optionally delete the .gguf alongside it.
model := ""
if req.DeleteModel {
if content, err := ReadPreset(req.ID); err == nil {
model = modelFromPresetContent(content)
}
}
if err := DeletePreset(req.ID); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
modelDeleted, modelErr := "", ""
if req.DeleteModel && model != "" {
if err := deleteModelFile(model); err != nil {
modelErr = err.Error()
} else {
modelDeleted = model
}
}
sendJSON(w, 200, map[string]any{"ok": true, "modelDeleted": modelDeleted, "modelError": modelErr})
}
// handleAgent renvoie l'état du mode agent ET la liste des pages mémoire (que
// l'IA gère via les outils mem_*) — un seul aller-retour pour l'UI. La clé
// "skills" est conservée en miroir de "pages" pour l'ancien portail ajean.link.
func handleAgent(w http.ResponseWriter, r *http.Request) {
pages := MemList()
out := []map[string]any{}
for _, p := range pages {
out = append(out, map[string]any{"name": p.Name, "desc": p.Title})
}
sendJSON(w, 200, map[string]any{"enabled": agentEnabled(), "tool_limit": toolLimitEnabled(), "compact": compactEnabled(), "mem_mode": string(memMode()), "pages": out, "skills": out})
}
// handleMemoryMode lit/écrit le mode mémoire (off / ondemand / always).
//
// GET → {mode}
// POST {mode} → persiste MEM_MODE
func handleMemoryMode(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Mode string `json:"mode"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
// On normalise via memMode() en réinjectant la valeur : toute entrée
// inconnue retombe sur "always", donc on valide en passant par le parseur.
m := MemAlways
switch MemMode(strings.ToLower(strings.TrimSpace(req.Mode))) {
case MemOff:
m = MemOff
case MemOnDemand:
m = MemOnDemand
case MemAlways:
m = MemAlways
}
if err := setMemMode(m); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
sendJSON(w, 200, map[string]any{"ok": true, "mode": string(memMode())})
}
// handleToolLimitToggle active/désactive le plafond d'appels d'outils par tour
// (config.env TOOL_LIMIT). On=limité (défaut), off=quasi illimité.
func handleToolLimitToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
val := ""
if !req.On {
val = "off"
}
if err := SetConfigKey("TOOL_LIMIT", val); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "tool_limit": toolLimitEnabled()})
}
// handleCompactToggle active/désactive le compactage automatique du contexte
// (config.env COMPACT). On=compacte (défaut), off=jamais.
func handleCompactToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
val := ""
if !req.On {
val = "off"
}
if err := SetConfigKey("COMPACT", val); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "compact": compactEnabled()})
}
func handleAgentToggle(w http.ResponseWriter, r *http.Request) {
var req struct {
On bool `json:"on"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
if err := setAgentEnabled(req.On); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "enabled": agentEnabled()})
}
// handleInternet pilote l'accès web de l'IA (serveur Crawl4AI).
//
// GET → {enabled, url, reachable}
// POST {enabled, url} → enregistre CRAWL4AI_URL + le drapeau .internet_enabled
//
// handleAPIKey expose et pilote la clé d'accès à l'endpoint compatible OpenAI
// (llama-server /v1). GET renvoie l'état ; POST {action:"generate"|"set"|"clear",
// key?} l'écrit puis redémarre le service (llama-server lit --api-key au lancement).
func handleAPIKey(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Action string `json:"action"`
Key string `json:"key"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
var key string
switch req.Action {
case "generate":
key = genAPIKey()
case "set":
key = strings.TrimSpace(req.Key)
case "clear":
key = ""
default:
sendJSON(w, 400, map[string]any{"ok": false, "error": "action inconnue"})
return
}
if err := writeAPIKey(key); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
// La clé n'est appliquée qu'au (re)démarrage de llama-server.
if serviceIsActive() {
_ = serviceAction("restart")
}
}
k := readAPIKey()
sendJSON(w, 200, map[string]any{
"ok": true,
"set": k != "",
"key": k,
"masked": maskAPIKey(k),
"port": LLMPort(),
"host": localIP(),
// Accès OpenAI PUBLIC via ajean.link (passthrough SNI, VPS aveugle) : si
// activé, l'URL publique est https://<machine>.oai.ajean.link/v1.
"oai_public": oaiPublicEnabled(),
"machine": machineID(),
})
}
// handleOAIPublic pilote le drapeau d'accès OpenAI public (exposition via
// ajean.link). GET renvoie l'état ; POST {enabled} l'active/coupe en direct
// (aucun redémarrage : le démux du tunnel relit le drapeau à chaque connexion).
func handleOAIPublic(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Enabled *bool `json:"enabled"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if req.Enabled != nil {
if err := setOAIPublic(*req.Enabled); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
}
sendJSON(w, 200, map[string]any{
"ok": true,
"enabled": oaiPublicEnabled(),
"machine": machineID(),
})
}
// localIP best-effort renvoie l'IPv4 LAN primaire de la machine (l'IP source du
// trafic sortant), ou "localhost" à défaut. Sert à annoncer l'endpoint OpenAI
// avec une adresse correcte sur le réseau local MÊME quand l'UI est atteinte via
// le tunnel ajean.link (où location.hostname serait le domaine du relais, faux).
func localIP() string {
conn, err := net.Dial("udp", "8.8.8.8:80")
if err != nil {
return "localhost"
}
defer conn.Close()
if a, ok := conn.LocalAddr().(*net.UDPAddr); ok && a.IP != nil {
return a.IP.String()
}
return "localhost"
}
func handleInternet(w http.ResponseWriter, r *http.Request) {
if r.Method == http.MethodPost {
var req struct {
Enabled *bool `json:"enabled"`
URL *string `json:"url"`
}
_ = json.NewDecoder(r.Body).Decode(&req)
if req.URL != nil {
u := strings.TrimRight(strings.TrimSpace(*req.URL), "/")
if err := SetConfigKey("CRAWL4AI_URL", u); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
reachMu.Lock()
reachURL = "" // invalide le cache de reachability
reachMu.Unlock()
}
if req.Enabled != nil {
if err := setInternetEnabled(*req.Enabled); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
}
}
sendJSON(w, 200, map[string]any{
"ok": true,
"enabled": internetEnabled(),
"url": crawl4aiURL(),
"reachable": crawlReachable(),
})
}
// handleMem / handleMemSave / handleMemDelete : éditeur web des pages mémoire
// (MEMORY/<nom>.md). Payload partagé saveReq (name/old/content) ; "name" = nom
// de fichier de la page.
func handleMem(w http.ResponseWriter, r *http.Request) {
name := strings.TrimSpace(r.URL.Query().Get("name"))
if name == "" {
sendJSON(w, 200, map[string]any{"name": "", "content": "# nouvelle page\n\nNote ici ce que jean doit retenir entre les sessions.\n"})
return
}
c := MemContent(name)
if c == "" {
sendJSON(w, 404, map[string]any{"error": "not found"})
return
}
sendJSON(w, 200, map[string]any{"name": name, "content": c})
}
func handleMemSave(w http.ResponseWriter, r *http.Request) {
var req saveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if err := MemSave(req.Name, req.Old, req.Content); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "name": req.Name})
}
func handleMemDelete(w http.ResponseWriter, r *http.Request) {
var req saveReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if err := MemDelete(req.Name); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleSwitch(w http.ResponseWriter, r *http.Request) {
var req struct {
N int `json:"n"`
}
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
list, err := ListPresets()
if err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
if req.N < 1 || req.N > len(list) {
sendJSON(w, 400, map[string]any{"ok": false, "error": "index hors limites"})
return
}
target := list[req.N-1]
if err := SwitchToPreset(target.Path); err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "preset": target.Name})
}
// svcHandler returns an HTTP handler that triggers a start/stop/restart through
// the cross-platform serviceAction (systemd sous Linux, supervision PID-file
// sous Windows — voir sys_service_*.go). C'est ce qui permet à un client distant
// de relancer Jean.
func svcHandler(action string) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
err := serviceAction(action)
msg := "ok"
if err != nil {
msg = err.Error()
}
sendJSON(w, 200, map[string]any{"ok": err == nil, "out": msg})
}
}
// handleChat is the SSE proxy with tool-calling. The HTTP handler writes raw
// data: lines matching what the embedded JS expects (delta.content,
// delta.reasoning_content, delta.tool_used).
// handleBench runs `runBench` synchronously. Long enough (~30-60s) that we
// rely on the client side to show a spinner / disable the button.
func handleBench(w http.ResponseWriter, r *http.Request) {
nPrompt, nPredict := 2000, 300
if v := r.URL.Query().Get("prompt"); v != "" {
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
nPrompt = parsed
}
}
if v := r.URL.Query().Get("n"); v != "" {
if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 {
nPredict = parsed
}
}
res, err := runBench(nPrompt, nPredict)
if err != nil {
sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "result": res})
}
// handleBenchLast returns the most recent persisted benchmark, or {ok:false}
// when none has been run yet.
func handleBenchLast(w http.ResponseWriter, r *http.Request) {
sb := loadLastBench()
if sb == nil {
sendJSON(w, 200, map[string]any{"ok": false})
return
}
sendJSON(w, 200, map[string]any{"ok": true, "result": sb.Result, "model": sb.Model, "at": sb.At})
}
// chatReq est le corps d'une requête de chat (commun au chat clair et au chat E2E).
type chatReq struct {
Messages []Message `json:"messages"`
Temperature float64 `json:"temperature"`
// Optional per-request override of the agent mode (used by ajean.link
// agents, which carry their own toggle). nil = inherit the machine's
// global config. Tools/Skills sont conservés pour la rétro-compat des
// anciens clients relais : l'un OU l'autre à true active le mode agent.
Agent *bool `json:"agent"`
Tools *bool `json:"tools"`
Skills *bool `json:"skills"`
// Override par requête de l'accès internet (outils web). nil = config machine.
Internet *bool `json:"internet"`
// Taille réelle du contexte au tour précédent (usage.prompt_tokens + tokens
// générés), rapportée par le client qui l'affiche déjà. Sert à décider du
// compactage sur le VRAI décompte plutôt qu'une estimation. 0 = inconnu.
CtxUsed int `json:"ctx_used"`
// Nouveau modèle « conversation serveur » : Message = texte du tour à lancer
// (via /api/chat/send) ; From = dernier Seq déjà vu par le client (le flux
// d'abonnement rejoue Log[From:] puis suit le direct).
Message string `json:"message"`
From int `json:"from"`
}
// capsFromBody dérive les capacités du tour à partir des overrides éventuels du
// corps de requête (agents ajean.link portant leurs propres toggles), sinon la
// config machine.
func capsFromBody(body chatReq) Caps {
caps := globalCaps()
if body.Agent != nil {
caps.Agent = *body.Agent
} else if body.Tools != nil || body.Skills != nil {
caps.Agent = (body.Tools != nil && *body.Tools) || (body.Skills != nil && *body.Skills)
}
if body.Internet != nil {
caps.Internet = *body.Internet && crawlReachable()
}
return caps
}
// sseHeartbeat garde la réponse SSE active en écrivant un commentaire (`: ping`,
// ignoré par le parseur côté navigateur, aucun contenu donc rien à chiffrer)
// toutes les ~15 s. Sans ça, un long silence (exécution d'outil en mode agent,
// gros prefill) laisse la réponse inactive et un proxy intermédiaire (Cloudflare,
// ~100 s) la coupe → le fetch navigateur échoue (« Load failed »). Retourne un
// mutex à partager avec l'émetteur (writes concurrents sur le même w) et une
// fonction d'arrêt à différer.
func sseHeartbeat(w http.ResponseWriter, flusher http.Flusher) (*sync.Mutex, func()) {
mu := &sync.Mutex{}
done := make(chan struct{})
go func() {
// 4 s (et non 15) : borne le temps qu'un dernier bout de flux peut rester
// coincé dans un buffer proxy (Cloudflare) faute d'octets pour le pousser.
t := time.NewTicker(4 * time.Second)
defer t.Stop()
for {
select {
case <-done:
return
case <-t.C:
mu.Lock()
_, err := w.Write([]byte(": ping\n\n"))
if flusher != nil {
flusher.Flush()
}
mu.Unlock()
if err != nil {
return
}
}
}
}()
return mu, func() { close(done) }
}
// runChatStream est désormais un pur ABONNÉ au journal de la conversation serveur :
// il rejoue Log[body.From:] puis suit le direct, jusqu'à ce que la connexion (ctx)
// se ferme. La GÉNÉRATION est lancée séparément par /api/chat/send dans une
// goroutine détachée — fermer le navigateur n'arrête donc plus rien. Partagé par
// handleChat (clair) et handleE2EChat (chiffré).
func runChatStream(ctx context.Context, body chatReq, emit func(map[string]any) bool) {
conv.Subscribe(ctx, body.From, emit)
}
// handleChatSend ajoute un message et lance la génération en arrière-plan. Réponse
// req/resp (les événements arrivent par le flux d'abonnement). Passe par le proxy
// tunnel /api/e2e/req pour app.ajean.link — aucun code E2E spécifique requis.
func handleChatSend(w http.ResponseWriter, r *http.Request) {
var body chatReq
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()})
return
}
if strings.TrimSpace(body.Message) == "" {
sendJSON(w, 400, map[string]any{"ok": false, "error": "message vide"})
return
}
if err := conv.StartTurn(body.Message, capsFromBody(body), body.Temperature); err != nil {
// 409 = occupé (génération en cours) ; 503 = modèle pas prêt.
code := 503
if err == ErrBusy {
code = 409
}
sendJSON(w, code, map[string]any{"ok": false, "error": err.Error()})
return
}
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatStop(w http.ResponseWriter, r *http.Request) {
conv.Stop()
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatReset(w http.ResponseWriter, r *http.Request) {
conv.Reset()
sendJSON(w, 200, map[string]any{"ok": true})
}
func handleChatState(w http.ResponseWriter, r *http.Request) {
sendJSON(w, 200, conv.state())
}
func handleChat(w http.ResponseWriter, r *http.Request) {
var body chatReq
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
http.Error(w, err.Error(), 400)
return
}
w.Header().Set("Content-Type", "text/event-stream")
w.Header().Set("Cache-Control", "no-cache, no-transform")
w.Header().Set("X-Accel-Buffering", "no")
flusher, _ := w.(http.Flusher)
mu, stop := sseHeartbeat(w, flusher)
defer stop()
emit := func(obj map[string]any) bool {
b, _ := json.Marshal(map[string]any{"choices": []any{map[string]any{"delta": obj}}})
mu.Lock()
defer mu.Unlock()
if _, err := w.Write([]byte("data: " + string(b) + "\n\n")); err != nil {
return false
}
if flusher != nil {
flusher.Flush()
}
return true
}
runChatStream(r.Context(), body, emit)
}