From b1890ed6f7e1245f58350f16bc19e4c31cf07ce3 Mon Sep 17 00:00:00 2001 From: nathaninline Date: Wed, 22 Jul 2026 19:22:08 +0200 Subject: [PATCH] backend llama.cpp : mode binaires precompiles officiels (aucune compilation) Alternative rapide a la compilation locale : telechargement des binaires officiels de ggml-org/llama.cpp publies a chaque release (~2 min au lieu de ~40 de build CUDA). Le variant est choisi selon la machine (CUDA Windows avec selection de la version selon le pilote + cudart, Vulkan, Metal, ROCm, CPU), extrait dans backends/llama.cpp-prebuilt et BIN repointe dessus. - backend_prebuilt.go : detection release/variant, telechargement avec progression, extraction zip/tar.gz anti zip-slip, marqueur VERSION - UI : bouton 'binaires precompiles' + 'compiler (sources)', le statut distingue le mode actif (precompile vs build local), 'verifier les maj' compare a la derniere release officielle en mode precompile - CLI : 'jean llamacpp prebuilt' - le build local reste sur le disque : bascule possible via l'editeur de preset Limites : builds generiques (pas de tuning natif) ; pas de build CUDA officiel Linux (variant Vulkan) ; la compilation locale reste requise pour les forks. --- internal/jean/backend_llamacpp.go | 16 +- internal/jean/backend_prebuilt.go | 451 +++++++++++++++++++++++++ internal/jean/ui/index.html | 45 ++- internal/jean/ui/src/index.tmpl.html | 3 +- internal/jean/ui/src/js/10-llamacpp.js | 42 ++- internal/jean/web_llamacpp.go | 75 ++++ internal/jean/web_server.go | 12 +- 7 files changed, 628 insertions(+), 16 deletions(-) create mode 100644 internal/jean/backend_prebuilt.go diff --git a/internal/jean/backend_llamacpp.go b/internal/jean/backend_llamacpp.go index 43e7828..3576774 100644 --- a/internal/jean/backend_llamacpp.go +++ b/internal/jean/backend_llamacpp.go @@ -43,8 +43,22 @@ func cmdLlamacpp(args []string) error { return llamacppUpdate(args) case "status", "info", "": return llamacppStatus(args) + case "prebuilt": + // Binaires officiels précompilés (aucune compilation) — voir backend_prebuilt.go. + bin, err := prebuiltInstall( + func(s string) { fmt.Println(" " + s) }, + func(s string) { fmt.Printf("%s %s\n", cyan("▶"), s) }, + ) + if err != nil { + return err + } + if err := SetConfigKey("BIN", bin); err != nil { + return fmt.Errorf("binaires installés mais échec écriture BIN dans config.env: %w", err) + } + fmt.Printf("%s BIN mis à jour dans %s — %s pour appliquer\n", green("✓"), confPath(), bold("jean restart")) + return nil default: - return fmt.Errorf("sous-commande inconnue: %s (install | update | status)", sub) + return fmt.Errorf("sous-commande inconnue: %s (install | update | prebuilt | status)", sub) } } diff --git a/internal/jean/backend_prebuilt.go b/internal/jean/backend_prebuilt.go new file mode 100644 index 0000000..ef877ea --- /dev/null +++ b/internal/jean/backend_prebuilt.go @@ -0,0 +1,451 @@ +// backend_prebuilt.go — backend llama.cpp SANS compilation : télécharge les +// binaires officiels précompilés publiés à chaque release de ggml-org/llama.cpp +// (zip Windows, tar.gz macOS/Linux), choisit l'asset adapté à la machine +// (CUDA / ROCm / Vulkan / CPU), l'extrait dans backends/llama.cpp-prebuilt et +// pointe BIN dessus. ~2 minutes au lieu d'une compilation complète. +// +// Limites assumées : builds génériques (pas de tuning natif), et pas de build +// CUDA officiel pour Linux (on retombe sur Vulkan — la compilation locale +// reste la voie CUDA sous Linux). La compilation locale reste indispensable +// pour les forks (ex. PrismML). +package jean + +import ( + "archive/tar" + "archive/zip" + "compress/gzip" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "os/exec" + "path/filepath" + "regexp" + "runtime" + "strconv" + "strings" + "time" +) + +const llamaReleasesAPI = "https://api.github.com/repos/ggml-org/llama.cpp/releases/latest" + +type ghAsset struct { + Name string `json:"name"` + URL string `json:"browser_download_url"` + Size int64 `json:"size"` +} + +func prebuiltDir() string { + return filepath.Join(JeanHome(), "backends", "llama.cpp-prebuilt") +} + +// prebuiltVersion lit le marqueur VERSION du dossier prebuilt : "tag cudaVer" +// (cudaVer vide hors CUDA Windows). +func prebuiltVersion() (tag, cudaVer string) { + b, err := os.ReadFile(filepath.Join(prebuiltDir(), "VERSION")) + if err != nil { + return "", "" + } + f := strings.Fields(strings.TrimSpace(string(b))) + if len(f) > 0 { + tag = f[0] + } + if len(f) > 1 { + cudaVer = f[1] + } + return +} + +// prebuiltServerBin localise llama-server(.exe) sous le dossier prebuilt +// (l'arborescence interne des archives officielles varie : racine, build/bin…). +func prebuiltServerBin() string { + want := "llama-server" + if runtime.GOOS == "windows" { + want += ".exe" + } + var found string + _ = filepath.WalkDir(prebuiltDir(), func(p string, d os.DirEntry, err error) error { + if err != nil || d.IsDir() { + return nil + } + if d.Name() == want { + found = p + return filepath.SkipAll + } + return nil + }) + return found +} + +// fetchLlamaLatest interroge l'API GitHub pour la dernière release officielle. +func fetchLlamaLatest() (string, []ghAsset, error) { + req, err := http.NewRequest("GET", llamaReleasesAPI, nil) + if err != nil { + return "", nil, err + } + req.Header.Set("Accept", "application/vnd.github+json") + client := &http.Client{Timeout: 30 * time.Second} + resp, err := client.Do(req) + if err != nil { + return "", nil, err + } + defer resp.Body.Close() + if resp.StatusCode != 200 { + return "", nil, fmt.Errorf("GitHub API : HTTP %d", resp.StatusCode) + } + var rel struct { + TagName string `json:"tag_name"` + Assets []ghAsset `json:"assets"` + } + if err := json.NewDecoder(resp.Body).Decode(&rel); err != nil { + return "", nil, err + } + if rel.TagName == "" { + return "", nil, fmt.Errorf("release invalide (tag vide)") + } + return rel.TagName, rel.Assets, nil +} + +// driverCudaVersion renvoie la version CUDA max supportée par le pilote NVIDIA +// (bandeau de nvidia-smi : « CUDA Version: 12.8 »), ou 0 si inconnue. +func driverCudaVersion() float64 { + out, err := hideCmd(exec.Command("nvidia-smi")).Output() + if err != nil { + return 0 + } + m := regexp.MustCompile(`CUDA Version:\s*([0-9]+\.[0-9]+)`).FindSubmatch(out) + if m == nil { + return 0 + } + v, _ := strconv.ParseFloat(string(m[1]), 64) + return v +} + +// assetMatch garde les assets dont le nom contient TOUS les fragments. +func assetMatch(assets []ghAsset, frags ...string) []ghAsset { + var out []ghAsset + for _, a := range assets { + ok := true + for _, f := range frags { + if !strings.Contains(a.Name, f) { + ok = false + break + } + } + if ok { + out = append(out, a) + } + } + return out +} + +var reCudaAssetVer = regexp.MustCompile(`cuda-([0-9]+\.[0-9]+)`) + +// pickPrebuilt choisit l'asset principal (+ cudart pour CUDA Windows) adapté à +// la machine, et renvoie un label lisible du variant retenu. +func pickPrebuilt(assets []ghAsset) (main *ghAsset, cudart *ghAsset, label, cudaVer string, err error) { + pickOne := func(list []ghAsset) *ghAsset { + if len(list) == 0 { + return nil + } + return &list[0] + } + switch runtime.GOOS { + case "windows": + if runtime.GOARCH == "arm64" { + main = pickOne(assetMatch(assets, "llama-", "bin-win-cpu-arm64")) + label = "CPU (Windows arm64)" + break + } + if hasNvidiaGPU() { + // Plusieurs versions CUDA publiées (ex. 12.4 et 13.3) : on prend la plus + // haute supportée par le pilote (sinon la plus basse, la plus compatible). + // NB : les archives cudart-llama-bin-win-cuda-… contiennent aussi ces + // fragments — on les écarte explicitement du choix du binaire principal. + var cand []ghAsset + for _, a := range assetMatch(assets, "llama-", "bin-win-cuda-", "-x64.zip") { + if !strings.HasPrefix(a.Name, "cudart") { + cand = append(cand, a) + } + } + maxV := driverCudaVersion() + var best *ghAsset + bestV := 0.0 + for i := range cand { + m := reCudaAssetVer.FindStringSubmatch(cand[i].Name) + if m == nil { + continue + } + v, _ := strconv.ParseFloat(m[1], 64) + ok := maxV == 0 && (best == nil || v < bestV) || // pilote inconnu → la plus basse + maxV > 0 && v <= maxV && v > bestV // sinon la plus haute compatible + if ok { + best = &cand[i] + bestV = v + cudaVer = m[1] + } + } + if best != nil { + main = best + cudart = pickOne(assetMatch(assets, "cudart-", "win-cuda-"+cudaVer)) + label = "CUDA " + cudaVer + " (Windows x64)" + break + } + } + if a := pickOne(assetMatch(assets, "llama-", "bin-win-vulkan-x64")); a != nil { + main, label = a, "Vulkan (Windows x64)" + break + } + main = pickOne(assetMatch(assets, "llama-", "bin-win-cpu-x64")) + label = "CPU (Windows x64)" + case "darwin": + arch := "x64" + if runtime.GOARCH == "arm64" { + arch = "arm64" + } + main = pickOne(assetMatch(assets, "llama-", "bin-macos-"+arch)) + label = "Metal (macOS " + arch + ")" + default: // linux + arch := "x64" + if runtime.GOARCH == "arm64" { + arch = "arm64" + } + if hasTool("hipcc") || isDir("/opt/rocm") { + if a := pickOne(assetMatch(assets, "llama-", "bin-ubuntu-rocm-", arch)); a != nil { + main, label = a, "ROCm (Linux "+arch+")" + break + } + } + // Pas de build CUDA officiel pour Linux : sur GPU NVIDIA le variant Vulkan + // fonctionne via le pilote (moins optimal que le build CUDA local). + if hasNvidiaGPU() || hasTool("vulkaninfo") { + if a := pickOne(assetMatch(assets, "llama-", "bin-ubuntu-vulkan-"+arch)); a != nil { + main, label = a, "Vulkan (Linux "+arch+")" + if hasNvidiaGPU() { + label += " — pas de build CUDA officiel Linux ; compile localement pour du CUDA natif" + } + break + } + } + main = pickOne(assetMatch(assets, "llama-", "bin-ubuntu-"+arch+".tar.gz")) + label = "CPU (Linux " + arch + ")" + } + if main == nil { + return nil, nil, "", "", fmt.Errorf("aucun binaire précompilé adapté à cette machine dans la release officielle") + } + return main, cudart, label, cudaVer, nil +} + +// prebuiltInstall télécharge et installe (ou met à jour) les binaires +// précompilés. logf reçoit chaque ligne de log ; phasef la phase courante. +// Renvoie le chemin du binaire installé. +func prebuiltInstall(logf, phasef func(string)) (string, error) { + phasef("récupération de la dernière release llama.cpp…") + tag, assets, err := fetchLlamaLatest() + if err != nil { + return "", fmt.Errorf("impossible d'interroger les releases llama.cpp : %w", err) + } + curTag, curCuda := prebuiltVersion() + main, cudart, label, cudaVer, err := pickPrebuilt(assets) + if err != nil { + return "", err + } + logf(fmt.Sprintf("release %s — variant retenu : %s", tag, label)) + + if curTag == tag && prebuiltServerBin() != "" { + logf("déjà à jour (" + tag + ")") + return prebuiltServerBin(), nil + } + + dir := prebuiltDir() + if err := os.MkdirAll(dir, 0o755); err != nil { + return "", err + } + + // cudart (DLLs runtime CUDA, ~400 Mo) : seulement si absent ou si la version + // CUDA du variant a changé depuis la dernière installation. + if cudart != nil { + haveDLL, _ := filepath.Glob(filepath.Join(dir, "**", "cudart64*.dll")) + if len(haveDLL) == 0 { + haveDLL, _ = filepath.Glob(filepath.Join(dir, "cudart64*.dll")) + } + if len(haveDLL) > 0 && curCuda == cudaVer { + logf("cudart " + cudaVer + " déjà présent — téléchargement évité") + cudart = nil + } + } + + for _, a := range []*ghAsset{main, cudart} { + if a == nil { + continue + } + phasef(fmt.Sprintf("téléchargement de %s (%d Mo)…", a.Name, a.Size/1_000_000)) + tmp := filepath.Join(dir, a.Name+".part") + if err := downloadWithProgress(a.URL, tmp, a.Size, logf); err != nil { + _ = os.Remove(tmp) + return "", fmt.Errorf("téléchargement de %s : %w", a.Name, err) + } + phasef("extraction de " + a.Name + "…") + if err := extractArchive(tmp, dir); err != nil { + _ = os.Remove(tmp) + return "", fmt.Errorf("extraction de %s : %w", a.Name, err) + } + _ = os.Remove(tmp) + } + + bin := prebuiltServerBin() + if bin == "" { + return "", fmt.Errorf("archives extraites mais llama-server introuvable sous %s", dir) + } + if runtime.GOOS != "windows" { + _ = os.Chmod(bin, 0o755) + } + if err := os.WriteFile(filepath.Join(dir, "VERSION"), []byte(tag+" "+cudaVer+"\n"), 0o644); err != nil { + return "", err + } + logf("binaire installé : " + bin + " (release " + tag + ")") + return bin, nil +} + +// downloadWithProgress télécharge url vers dest en journalisant la progression +// par tranches de ~25 Mo. +func downloadWithProgress(url, dest string, total int64, logf func(string)) error { + client := &http.Client{Timeout: 0} + resp, err := client.Get(url) + if err != nil { + return err + } + defer resp.Body.Close() + if resp.StatusCode != 200 { + return fmt.Errorf("HTTP %d", resp.StatusCode) + } + if total <= 0 { + total = resp.ContentLength + } + f, err := os.Create(dest) + if err != nil { + return err + } + defer f.Close() + buf := make([]byte, 1<<20) + var done, lastLog int64 + for { + n, rerr := resp.Body.Read(buf) + if n > 0 { + if _, werr := f.Write(buf[:n]); werr != nil { + return werr + } + done += int64(n) + if done-lastLog >= 25<<20 { + lastLog = done + if total > 0 { + logf(fmt.Sprintf("⬇ %d / %d Mo (%d%%)", done/1_000_000, total/1_000_000, done*100/total)) + } else { + logf(fmt.Sprintf("⬇ %d Mo", done/1_000_000)) + } + } + } + if rerr == io.EOF { + return nil + } + if rerr != nil { + return rerr + } + } +} + +// extractArchive extrait un .zip ou un .tar.gz dans dir, en refusant toute +// entrée qui s'échapperait du dossier (zip-slip). +func extractArchive(path, dir string) error { + safe := func(name string) (string, error) { + p := filepath.Join(dir, filepath.FromSlash(name)) + if rel, err := filepath.Rel(dir, p); err != nil || strings.HasPrefix(rel, "..") { + return "", fmt.Errorf("entrée d'archive suspecte : %s", name) + } + return p, nil + } + if strings.HasSuffix(path, ".zip") || strings.HasSuffix(path, ".zip.part") { + zr, err := zip.OpenReader(path) + if err != nil { + return err + } + defer zr.Close() + for _, f := range zr.File { + p, err := safe(f.Name) + if err != nil { + return err + } + if f.FileInfo().IsDir() { + if err := os.MkdirAll(p, 0o755); err != nil { + return err + } + continue + } + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + return err + } + rc, err := f.Open() + if err != nil { + return err + } + w, err := os.Create(p) + if err != nil { + rc.Close() + return err + } + _, err = io.Copy(w, rc) + rc.Close() + w.Close() + if err != nil { + return err + } + } + return nil + } + // tar.gz + f, err := os.Open(path) + if err != nil { + return err + } + defer f.Close() + gz, err := gzip.NewReader(f) + if err != nil { + return err + } + defer gz.Close() + tr := tar.NewReader(gz) + for { + h, err := tr.Next() + if err == io.EOF { + return nil + } + if err != nil { + return err + } + p, err := safe(h.Name) + if err != nil { + return err + } + switch h.Typeflag { + case tar.TypeDir: + if err := os.MkdirAll(p, 0o755); err != nil { + return err + } + case tar.TypeReg: + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + return err + } + w, err := os.Create(p) + if err != nil { + return err + } + if _, err := io.Copy(w, tr); err != nil { + w.Close() + return err + } + w.Close() + _ = os.Chmod(p, os.FileMode(h.Mode)&0o777) + } + } +} diff --git a/internal/jean/ui/index.html b/internal/jean/ui/index.html index 5874b93..efbf4c4 100644 --- a/internal/jean/ui/index.html +++ b/internal/jean/ui/index.html @@ -397,7 +397,8 @@ button:disabled{opacity:.5;cursor:not-allowed}
Backend llama.cpp
…
- + + @@ -1624,11 +1625,20 @@ async function loadLlamacpp(){ const p = s.plan; parts.push('accélérateur détecté : '+p.backend.toUpperCase()+''+(p.arch?' (arch '+p.arch+')':'')+' · '+p.jobs+' jobs'); } + // Mode actif : binaires officiels précompilés vs build local. + const pb = s.prebuilt || {}; + if(pb.in_use){ + parts.push('mode : ⚡ binaires précompilés officiels'+(pb.tag?' ('+pb.tag+')':'')); + lcBadge('précompilé'); + } else if(s.installed && s.in_use){ + parts.push('mode : 🔧 compilé depuis les sources'+(pb.tag?' (précompilé '+pb.tag+' aussi installé)':'')); + } st.innerHTML = parts.join('
'); document.getElementById('lc-install').style.display = s.installed ? 'none' : ''; - document.getElementById('lc-check').style.display = s.installed ? '' : 'none'; - document.getElementById('lc-update').style.display = (s.installed && (s.behind>0 || !s.bin)) ? '' : 'none'; - document.getElementById('lc-rebuild').style.display = s.installed ? '' : 'none'; + document.getElementById('lc-check').style.display = (s.installed || pb.in_use) ? '' : 'none'; + document.getElementById('lc-update').style.display = (s.installed && !pb.in_use && (s.behind>0 || !s.bin)) ? '' : 'none'; + document.getElementById('lc-rebuild').style.display = (s.installed && !pb.in_use) ? '' : 'none'; + document.getElementById('lc-prebuilt').textContent = pb.in_use ? '⚡ maj binaires précompilés' : '⚡ binaires précompilés'; // Job en cours (page rechargée pendant un build) → on raccroche le polling. if(s.job && s.job.exists && s.job.running && !lcPoll){ document.getElementById('lc-details').open = true; @@ -1640,6 +1650,20 @@ async function lcCheck(){ const btn = document.getElementById('lc-check'); const msg = document.getElementById('lc-check-msg'); btn.disabled = true; btn.textContent = '⏳ vérification…'; + // Mode précompilé : on compare à la dernière release officielle, pas au git. + if(lcState && lcState.prebuilt && lcState.prebuilt.in_use){ + try{ + const r = await jpost('/api/llamacpp/prebuilt/check'); + if(!r.ok){ msg.innerHTML = ''+(r.error||'erreur')+''; return; } + if(r.update){ + msg.innerHTML = 'nouvelle release officielle : '+r.latest+' (installée : '+(r.current||'aucune')+')
'+String(r.variant||'').replace(/[<>&]/g,'')+' · ~'+r.size_mb+' Mo'; + lcBadge('maj dispo'); + } else { + msg.innerHTML = '✓ à jour ('+r.current+')'; + } + } finally { btn.disabled = false; btn.textContent = '↥ vérifier les maj'; } + return; + } try{ const r = await jpost('/api/llamacpp/check'); if(!r.ok){ msg.innerHTML = ''+(r.error||'erreur')+''; return; } @@ -1663,6 +1687,17 @@ async function lcInstall(){ lcStartPolling(); } +async function lcPrebuilt(){ + const inUse = lcState && lcState.prebuilt && lcState.prebuilt.in_use; + const msg = inUse + ? 'Télécharger la dernière release officielle précompilée de llama.cpp (~2 min, aucune compilation). Le service sera arrêté pendant le remplacement puis redémarré.' + : 'Télécharger les binaires OFFICIELS précompilés de llama.cpp au lieu de compiler (~2 min). Le variant est choisi automatiquement (CUDA / Vulkan / Metal / CPU) et BIN pointera dessus. Le build local, s\'il existe, reste sur le disque — tu peux y revenir via l\'éditeur de preset. Note : builds génériques (pas de tuning natif) ; sous Linux il n\'existe pas de build CUDA officiel (variant Vulkan).'; + if(!await askConfirm(msg, {title: inUse ? 'Mettre à jour les binaires précompilés ?' : 'Passer aux binaires précompilés ?', okText: inUse ? 'Mettre à jour' : 'Télécharger'})) return; + const r = await jpost('/api/llamacpp/prebuilt', {}); + if(!r.ok){ toast('erreur : '+(r.error||'')); return; } + lcStartPolling(); +} + async function lcUpdate(clean){ const msg = clean ? 'Recompiler llama.cpp from scratch (build/ supprimé). Le service sera arrêté pendant le build puis redémarré.' @@ -1674,7 +1709,7 @@ async function lcUpdate(clean){ } function lcSetButtons(disabled){ - ['lc-install','lc-check','lc-update','lc-rebuild'].forEach(id=>{ + ['lc-install','lc-check','lc-update','lc-rebuild','lc-prebuilt'].forEach(id=>{ const b=document.getElementById(id); if(b) b.disabled = disabled; }); } diff --git a/internal/jean/ui/src/index.tmpl.html b/internal/jean/ui/src/index.tmpl.html index f00ed94..0b12bd2 100644 --- a/internal/jean/ui/src/index.tmpl.html +++ b/internal/jean/ui/src/index.tmpl.html @@ -119,7 +119,8 @@
Backend llama.cpp
…
- + + diff --git a/internal/jean/ui/src/js/10-llamacpp.js b/internal/jean/ui/src/js/10-llamacpp.js index 54520c8..02f30e0 100644 --- a/internal/jean/ui/src/js/10-llamacpp.js +++ b/internal/jean/ui/src/js/10-llamacpp.js @@ -31,11 +31,20 @@ async function loadLlamacpp(){ const p = s.plan; parts.push('accélérateur détecté : '+p.backend.toUpperCase()+''+(p.arch?' (arch '+p.arch+')':'')+' · '+p.jobs+' jobs'); } + // Mode actif : binaires officiels précompilés vs build local. + const pb = s.prebuilt || {}; + if(pb.in_use){ + parts.push('mode : ⚡ binaires précompilés officiels'+(pb.tag?' ('+pb.tag+')':'')); + lcBadge('précompilé'); + } else if(s.installed && s.in_use){ + parts.push('mode : 🔧 compilé depuis les sources'+(pb.tag?' (précompilé '+pb.tag+' aussi installé)':'')); + } st.innerHTML = parts.join('
'); document.getElementById('lc-install').style.display = s.installed ? 'none' : ''; - document.getElementById('lc-check').style.display = s.installed ? '' : 'none'; - document.getElementById('lc-update').style.display = (s.installed && (s.behind>0 || !s.bin)) ? '' : 'none'; - document.getElementById('lc-rebuild').style.display = s.installed ? '' : 'none'; + document.getElementById('lc-check').style.display = (s.installed || pb.in_use) ? '' : 'none'; + document.getElementById('lc-update').style.display = (s.installed && !pb.in_use && (s.behind>0 || !s.bin)) ? '' : 'none'; + document.getElementById('lc-rebuild').style.display = (s.installed && !pb.in_use) ? '' : 'none'; + document.getElementById('lc-prebuilt').textContent = pb.in_use ? '⚡ maj binaires précompilés' : '⚡ binaires précompilés'; // Job en cours (page rechargée pendant un build) → on raccroche le polling. if(s.job && s.job.exists && s.job.running && !lcPoll){ document.getElementById('lc-details').open = true; @@ -47,6 +56,20 @@ async function lcCheck(){ const btn = document.getElementById('lc-check'); const msg = document.getElementById('lc-check-msg'); btn.disabled = true; btn.textContent = '⏳ vérification…'; + // Mode précompilé : on compare à la dernière release officielle, pas au git. + if(lcState && lcState.prebuilt && lcState.prebuilt.in_use){ + try{ + const r = await jpost('/api/llamacpp/prebuilt/check'); + if(!r.ok){ msg.innerHTML = ''+(r.error||'erreur')+''; return; } + if(r.update){ + msg.innerHTML = 'nouvelle release officielle : '+r.latest+' (installée : '+(r.current||'aucune')+')
'+String(r.variant||'').replace(/[<>&]/g,'')+' · ~'+r.size_mb+' Mo'; + lcBadge('maj dispo'); + } else { + msg.innerHTML = '✓ à jour ('+r.current+')'; + } + } finally { btn.disabled = false; btn.textContent = '↥ vérifier les maj'; } + return; + } try{ const r = await jpost('/api/llamacpp/check'); if(!r.ok){ msg.innerHTML = ''+(r.error||'erreur')+''; return; } @@ -70,6 +93,17 @@ async function lcInstall(){ lcStartPolling(); } +async function lcPrebuilt(){ + const inUse = lcState && lcState.prebuilt && lcState.prebuilt.in_use; + const msg = inUse + ? 'Télécharger la dernière release officielle précompilée de llama.cpp (~2 min, aucune compilation). Le service sera arrêté pendant le remplacement puis redémarré.' + : 'Télécharger les binaires OFFICIELS précompilés de llama.cpp au lieu de compiler (~2 min). Le variant est choisi automatiquement (CUDA / Vulkan / Metal / CPU) et BIN pointera dessus. Le build local, s\'il existe, reste sur le disque — tu peux y revenir via l\'éditeur de preset. Note : builds génériques (pas de tuning natif) ; sous Linux il n\'existe pas de build CUDA officiel (variant Vulkan).'; + if(!await askConfirm(msg, {title: inUse ? 'Mettre à jour les binaires précompilés ?' : 'Passer aux binaires précompilés ?', okText: inUse ? 'Mettre à jour' : 'Télécharger'})) return; + const r = await jpost('/api/llamacpp/prebuilt', {}); + if(!r.ok){ toast('erreur : '+(r.error||'')); return; } + lcStartPolling(); +} + async function lcUpdate(clean){ const msg = clean ? 'Recompiler llama.cpp from scratch (build/ supprimé). Le service sera arrêté pendant le build puis redémarré.' @@ -81,7 +115,7 @@ async function lcUpdate(clean){ } function lcSetButtons(disabled){ - ['lc-install','lc-check','lc-update','lc-rebuild'].forEach(id=>{ + ['lc-install','lc-check','lc-update','lc-rebuild','lc-prebuilt'].forEach(id=>{ const b=document.getElementById(id); if(b) b.disabled = disabled; }); } diff --git a/internal/jean/web_llamacpp.go b/internal/jean/web_llamacpp.go index 0224fa5..270d8f0 100644 --- a/internal/jean/web_llamacpp.go +++ b/internal/jean/web_llamacpp.go @@ -158,6 +158,14 @@ func handleLlamacpp(w http.ResponseWriter, r *http.Request) { "arch": plan.cudaArch, "jobs": plan.jobs, } + // Mode « binaires précompilés » : installé ? utilisé par la config ? + pbTag, _ := prebuiltVersion() + pbBin := prebuiltServerBin() + out["prebuilt"] = map[string]any{ + "tag": pbTag, + "bin": pbBin, + "in_use": pbBin != "" && samePath(pbBin, cfgBin), + } out["job"] = lcJobSnapshot(0, false) sendJSON(w, 200, out) } @@ -238,6 +246,73 @@ func handleLlamacppUpdate(w http.ResponseWriter, r *http.Request) { sendJSON(w, 200, map[string]any{"ok": true}) } +// handleLlamacppPrebuiltCheck interroge la dernière release officielle de +// llama.cpp et la compare à la version précompilée installée. Synchrone. +func handleLlamacppPrebuiltCheck(w http.ResponseWriter, r *http.Request) { + tag, assets, err := fetchLlamaLatest() + if err != nil { + sendJSON(w, 200, map[string]any{"ok": false, "error": err.Error()}) + return + } + main, _, label, _, err := pickPrebuilt(assets) + if err != nil { + sendJSON(w, 200, map[string]any{"ok": false, "error": err.Error()}) + return + } + cur, _ := prebuiltVersion() + sendJSON(w, 200, map[string]any{ + "ok": true, + "latest": tag, + "current": cur, + "variant": label, + "size_mb": main.Size / 1_000_000, + "update": cur != tag || prebuiltServerBin() == "", + }) +} + +// handleLlamacppPrebuilt lance le job de téléchargement / mise à jour des +// binaires précompilés officiels (pas de compilation). +func handleLlamacppPrebuilt(w http.ResponseWriter, r *http.Request) { + if err := startLcJob("prebuilt", lcRunPrebuilt); err != nil { + sendJSON(w, 409, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true}) +} + +// lcRunPrebuilt : télécharge les binaires officiels, pointe BIN dessus, en +// stoppant le service pendant le remplacement (binaire verrouillé en cours +// d'exécution) puis en le relançant. +func lcRunPrebuilt() { + svcWasUp := serviceIsActive() + if svcWasUp { + lcPhase("arrêt du service le temps de l'installation…") + if err := serviceAction("stop"); err != nil { + lcAppend("[warn] impossible d'arrêter le service : " + err.Error()) + } + } + bin, err := prebuiltInstall(lcAppend, lcPhase) + if err == nil { + if serr := SetConfigKey("BIN", bin); serr != nil { + err = fmt.Errorf("binaires installés mais échec écriture BIN dans config.env : %w", serr) + } else { + lcAppend("BIN mis à jour dans " + confPath()) + } + } + if svcWasUp { + lcPhase("redémarrage du service…") + if serr := serviceAction("start"); serr != nil { + lcAppend("[warn] redémarrage du service échoué : " + serr.Error()) + } + } + if err != nil { + lcFail(err) + return + } + tag, _ := prebuiltVersion() + lcDone("binaires précompilés installés (" + tag + ")") +} + // handleLlamacppJob renvoie l'état du job courant + les lignes de log depuis // l'offset absolu ?from=N (le client mémorise `next` et enchaîne). func handleLlamacppJob(w http.ResponseWriter, r *http.Request) { diff --git a/internal/jean/web_server.go b/internal/jean/web_server.go index 8209c35..be5125d 100644 --- a/internal/jean/web_server.go +++ b/internal/jean/web_server.go @@ -88,11 +88,13 @@ func newWebMux() *http.ServeMux { api("/api/models/download", handleModelDownload) api("/api/models/download/status", handleModelDownloadStatus) api("/api/backends", handleBackends) - api("/api/llamacpp", handleLlamacpp) // statut du backend llama.cpp - api("/api/llamacpp/check", handleLlamacppCheck) // git fetch + retard sur origin - api("/api/llamacpp/install", handleLlamacppInstall) // job : clone + build + BIN - api("/api/llamacpp/update", handleLlamacppUpdate) // job : pull + rebuild + restart - api("/api/llamacpp/job", handleLlamacppJob) // progression + logs du job + api("/api/llamacpp", handleLlamacpp) // statut du backend llama.cpp + api("/api/llamacpp/check", handleLlamacppCheck) // git fetch + retard sur origin + api("/api/llamacpp/install", handleLlamacppInstall) // job : clone + build + BIN + api("/api/llamacpp/update", handleLlamacppUpdate) // job : pull + rebuild + restart + api("/api/llamacpp/job", handleLlamacppJob) // progression + logs du job + api("/api/llamacpp/prebuilt", handleLlamacppPrebuilt) // job : binaires officiels précompilés + api("/api/llamacpp/prebuilt/check", handleLlamacppPrebuiltCheck) // dernière release officielle vs installée api("/api/presets", handlePresets) api("/api/preset", handlePreset) api("/api/preset/save", handlePresetSave)