From 6802962120a97a7fc7731f27d13b5baf66dc666f Mon Sep 17 00:00:00 2001 From: nathaninline Date: Mon, 8 Jun 2026 12:54:49 +0200 Subject: [PATCH] Add files via upload --- .gitattributes | 7 + .gitignore | 34 +++ LICENSE | 21 ++ README.md | 169 +++++++++++++- bench.go | 161 +++++++++++++ chat.go | 115 ++++++++++ config.go | 150 ++++++++++++ go.mod | 3 + install.go | 196 ++++++++++++++++ llamacpp.go | 607 +++++++++++++++++++++++++++++++++++++++++++++++++ llm.go | 332 +++++++++++++++++++++++++++ main.go | 195 ++++++++++++++++ presets.go | 180 +++++++++++++++ serve.go | 85 +++++++ service.go | 115 ++++++++++ skills.go | 186 +++++++++++++++ test_cmd.go | 65 ++++++ tools.go | 114 ++++++++++ tty.go | 12 + web.go | 440 +++++++++++++++++++++++++++++++++++ 20 files changed, 3186 insertions(+), 1 deletion(-) create mode 100644 .gitattributes create mode 100644 .gitignore create mode 100644 LICENSE create mode 100644 bench.go create mode 100644 chat.go create mode 100644 config.go create mode 100644 go.mod create mode 100644 install.go create mode 100644 llamacpp.go create mode 100644 llm.go create mode 100644 main.go create mode 100644 presets.go create mode 100644 serve.go create mode 100644 service.go create mode 100644 skills.go create mode 100644 test_cmd.go create mode 100644 tools.go create mode 100644 tty.go create mode 100644 web.go diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..6bb6f89 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,7 @@ +# Normalize line endings to LF for all text files. +* text=auto eol=lf +*.go text eol=lf +*.md text eol=lf +*.yml text eol=lf +*.html text eol=lf +*.js text eol=lf diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..6d096b7 --- /dev/null +++ b/.gitignore @@ -0,0 +1,34 @@ +# Compiled binaries +/jean +/jean.exe +/jean-linux* +/jean-darwin* +/jean-windows* +jean-*-amd64 +jean-*-arm64 +*.exe +*.out + +# Build artifacts +/dist/ +/build/ + +# Local secrets / runtime data — never commit +.env +*.api_key +.api_key +config.env + +# Scratch / logs +*.log +out.txt +out*.txt +err.txt +err*.txt + +# OS / editor +.DS_Store +Thumbs.db +*.swp +.idea/ +.vscode/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..bea6060 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Jean contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 2739492..5c010e4 100644 --- a/README.md +++ b/README.md @@ -1 +1,168 @@ -# jean \ No newline at end of file +# Jean + +**A single-binary manager for self-hosted [llama.cpp](https://github.com/ggml-org/llama.cpp) servers — plus a built-in web UI, terminal chat, and an auto-detecting backend builder.** + +Drop one binary on a machine, run `jean llamacpp install`, and Jean clones, configures, and compiles llama.cpp for *that* machine's hardware (CUDA / ROCm / Metal / Vulkan / CPU) — no flags to remember. Then `jean start` and you have an OpenAI-compatible endpoint with a web chat on top. + +``` +download binary → jean llamacpp install → jean edit → jean start → done +``` + +--- + +## Why + +Running llama.cpp as a real service usually means: figure out the right CMake flags for your GPU, write a systemd unit, manage an API key, swap models, keep the build up to date… Jean turns all of that into a handful of subcommands behind a single static binary with **no runtime dependencies** (other than llama.cpp itself, which Jean can build for you). + +## Features + +- **`jean llamacpp install` / `update`** — clones and compiles llama.cpp with the right flags **auto-detected** for the host: + - **CUDA** when an NVIDIA GPU + `nvcc` are present (compute capability detected per-GPU via `nvidia-smi`, so multi-GPU machines build for all cards) + - **ROCm/HIP** (AMD), **Metal** (macOS / Apple Silicon), **Vulkan**, or **CPU** fallback + - `update` pulls the latest commit, stops the service while it rebuilds, then restarts it +- **systemd integration** — `jean install` writes the unit, a passwordless `systemctl` sudoers rule, and the data dirs +- **Web UI** (`jean web`) — chat, model/preset switching, skills & tools toggles +- **Terminal chat** (`jean chat`) — streamed responses +- **Presets** (`jean switch`) — keep multiple `config.env` profiles and swap between them +- **API key protection** (`jean set-api-key`) — Bearer auth for exposing the server publicly; the key is stored separately so it survives preset switches +- **Benchmark** (`jean bench`) — honest prefill/decode tok/s using a varied corpus +- **Single static binary** — built with `CGO_ENABLED=0`, cross-compiles trivially + +## Quick start + +### 1. Get the binary + +Grab a prebuilt binary from the [Releases](../../releases) page, or [build from source](#building-from-source): + +```bash +# example: Linux x86_64 +curl -L -o jean https://github.com/jean-llm/jean/releases/latest/download/jean-linux-amd64 +chmod +x jean +sudo mv jean /usr/local/bin/jean +``` + +### 2. Install (systemd unit, dirs, sudoers) + +```bash +sudo jean install +``` + +### 3. Build a llama.cpp backend for this machine + +```bash +jean llamacpp install +``` + +Jean auto-detects your accelerator, compiles `llama-server`, and points the config at the new binary. Requires `git` and `cmake` (plus the matching toolkit, e.g. CUDA, if you want GPU acceleration). + +### 4. Point it at a model and start + +```bash +jean edit # set MODEL=/path/to/your-model.gguf +jean start +jean test # verify the model answers +``` + +### 5. (optional) Web UI + +```bash +jean web # http://:8090 +``` + +## Commands + +``` +Service: + start | stop | restart manage the systemd service + status | logs status / live logs + enable | disable start on boot + edit edit $JEAN_HOME/config.env + set-api-key [key] protect the API (Bearer); empty = generate, "" = remove + vram GPU/VRAM usage (nvidia-smi) + test check the model answers (health + completion) + bench [N] measure prefill + decode tok/s + +Presets: + switch [N] pick a preset from configs/ (interactive or by number) + +Interaction: + chat [system-prompt] streamed terminal chat + web [PORT] web UI (default :8090) + +LLM-side tooling: + skills [on|off|list] let the model read SKILLS//SKILL.md + tools [on|off|status] enable run_shell (model executes shell commands) + +Backend (llama.cpp): + llamacpp install clone + build llama.cpp (auto-detect CUDA/ROCm/Metal/CPU), set BIN + llamacpp update git pull + rebuild the existing backend (stops/restarts the service) + llamacpp status current commit, detected backend, commits behind origin + +Install: + install install (systemd unit, sudoers, dirs) + uninstall uninstall +``` + +### `jean llamacpp` flags + +``` +install [--dir=PATH] [--ref=GIT_REF] [--force] [--no-switch] +update [--ref=GIT_REF] [--clean] [--no-restart] [--force] +``` + +- `--dir=` — where to clone (default `$JEAN_HOME/backends/llama.cpp`) +- `--ref=` — build a specific branch/tag/commit +- `--clean` — wipe `build/` and recompile from scratch +- `--no-switch` — don't touch `config.env` (install only) +- `--no-restart` — leave the service stopped after updating + +## Configuration + +Everything lives under **`$JEAN_HOME`** (default `/etc/jean`). The service reads `config.env`: + +| Key | Meaning | Default | +|-----|---------|---------| +| `BIN` | path to `llama-server` (set by `llamacpp install`) | — | +| `MODEL` | path to the `.gguf` model | — | +| `HOST` / `PORT` | bind address / port | `0.0.0.0` / `8080` | +| `CTX` | context size | `32768` | +| `NGL` | GPU layers to offload | `999` | +| `BATCH` / `UBATCH` | batch / micro-batch | `2048` / `512` | +| `THREADS` / `THREADS_BATCH` | CPU threads | `0` (auto) | +| `KV_TYPE` (`_K`/`_V`) | KV cache quantization | — | +| `REASONING` | reasoning mode passthrough | — | +| `EXTRA_ARGS` | appended verbatim to `llama-server` | — | + +The API key (when set with `jean set-api-key`) is stored in `$JEAN_HOME/.api_key`, separate from `config.env`. + +### Environment + +| Var | Meaning | Default | +|-----|---------|---------| +| `JEAN_HOME` | data root | `/etc/jean` (or `$HOME/JEAN`) | +| `JEAN_SERVICE` | systemd unit name | `jean` | +| `EDITOR` | editor for `jean edit` | `nano` | + +## Building from source + +Requires Go 1.22+. Jean is a pure-Go binary (the web UI is embedded via `go:embed`): + +```bash +git clone https://github.com/jean-llm/jean.git +cd jean +CGO_ENABLED=0 go build -o jean . + +# cross-compile, e.g. Linux from any host: +GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -o jean-linux-amd64 . +``` + +> Building **Jean** needs only Go. Building the **llama.cpp backend** (`jean llamacpp install`) needs `git`, `cmake`, and your accelerator's toolkit (CUDA, ROCm, etc.). + +## How it works + +- `jean serve` is the systemd `ExecStart`: it reads `config.env`, builds the `llama-server` argument list, and `exec`s into it so systemd supervises llama.cpp directly. +- `jean llamacpp` manages the llama.cpp checkout next to wherever `BIN` points, handling the common "relocated build dir" CMake-cache pitfall and stopping the service during a rebuild to avoid *Text file busy*. + +## License + +[MIT](LICENSE). The bundled `ui/marked.min.js` is [Marked](https://github.com/markedjs/marked), also MIT. diff --git a/bench.go b/bench.go new file mode 100644 index 0000000..604fdcf --- /dev/null +++ b/bench.go @@ -0,0 +1,161 @@ +package main + +import ( + "bytes" + "encoding/json" + "fmt" + "net/http" + "strconv" + "strings" + "time" +) + +// benchResult captures the timings llama.cpp returns from /completion. +type benchResult struct { + PromptN int `json:"prompt_n"` + PromptMs float64 `json:"prompt_ms"` + PromptPerSecond float64 `json:"prompt_per_second"` + PredictedN int `json:"predicted_n"` + PredictedMs float64 `json:"predicted_ms"` + PredictedPerSec float64 `json:"predicted_per_second"` + Elapsed float64 `json:"elapsed_sec"` +} + +// benchCorpus is a varied passage used to defeat speculative decoding +// (MTP / n-gram draft) — repetitive text inflates decode tok/s because every +// drafted token gets accepted, which is unlike real chat. We pull a chunk of +// natural-looking content and tile it to reach the target prompt size. +const benchCorpus = `In the early hours of an October morning, Camille walked along the canal, watching the cargo barges slip past the iron bridge that spanned the water. She thought about the meeting she had skipped, the unanswered messages on her phone, the way the city always seemed to forget her name after summer ended. Three streets away, a pâtisserie opened its shutters and the smell of warm butter mixed with diesel exhaust from the waiting bus. +Pendant ce temps, à Marseille, un chercheur en biologie marine prépare son matériel pour une plongée. Il étudie les herbiers de posidonie, ces prairies sous-marines vieilles de plusieurs milliers d'années qui stockent autant de carbone qu'une forêt amazonienne. Le bateau quitte le port à six heures vingt-trois. +Quantum computers, properly engineered, can solve certain classes of problems exponentially faster than classical machines. The catch is that decoherence ruins everything. Engineers use dilution refrigerators to drop superconducting qubits to fifteen millikelvin, colder than deep space. The wires connecting the chip to room-temperature electronics must dissipate almost no heat, or the qubit state collapses before any useful computation finishes. +Le boulanger lève la pâte à quatre heures. Il regarde la balance numérique en plissant les yeux : six cent vingt-trois grammes, presque le compte. Son chien dort sur le tapis de farine près du four. Dehors, deux chats se disputent un poisson abandonné par le pêcheur de nuit. +Consider a recursive descent parser written in Go. The lexer emits tokens; the parser consumes them and produces an abstract syntax tree. Error recovery is hard: after a syntax error, the parser must resynchronize at a known boundary—a semicolon, a closing brace—without losing track of subsequent diagnostics. Tree-sitter solves this with incremental parsing and a glr-like algorithm. +Le philosophe stoïcien disait : "Ce qui nous trouble, ce n'est pas ce qui nous arrive, mais l'opinion que nous nous en faisons." Vingt siècles plus tard, la phrase apparaît dans un livre de poche au rayon développement personnel d'une librairie d'aéroport, à côté d'un roman policier suédois. +Mitochondria descended from ancient bacteria engulfed by archaeal cells roughly two billion years ago. They still keep their own ring of DNA, separate from the nuclear genome. Mutations in mitochondrial DNA accumulate with age and have been implicated in everything from Parkinson's disease to ordinary muscle fatigue. Yet they remain stubbornly difficult to repair therapeutically because each cell contains hundreds. +La marée descend lentement, exposant des rochers couverts d'huîtres et d'algues vertes. Un héron immobile surveille les flaques laissées par l'eau. Plus loin, deux enfants courent avec un cerf-volant rouge qui refuse de monter à cause de l'humidité dans la voile. +Compilers translate high-level languages into machine code through several intermediate representations. LLVM IR sits in the middle: typed, mostly static-single-assignment, suitable for both aggressive optimization and direct lowering to x86 or ARM. The optimizer runs dozens of passes—dead code elimination, loop-invariant code motion, induction variable simplification—each touching the IR in carefully ordered ways. +Le cuisinier ferme les yeux pour goûter la sauce. Trop salée. Il ajoute une pomme de terre crue coupée en quartiers, sachant qu'elle absorbera l'excès en mijotant vingt minutes. Sa grand-mère lui a appris ce geste un dimanche de novembre il y a très longtemps. +` + +// runBench fires a prompt of roughly `nPrompt` tokens at /completion with +// cache_prompt:false (so prefill is actually measured, not cached). The prompt +// is a varied corpus to keep speculative decoding (MTP / n-gram draft) from +// inflating decode numbers — what you measure here is close to what you'll +// see in real chat at the same context length. +func runBench(nPrompt, nPredict int) (*benchResult, error) { + port := LLMPort() + if !healthCheck() { + return nil, fmt.Errorf("serveur injoignable sur :%d", port) + } + if nPrompt <= 0 { + nPrompt = 2000 + } + // 1 word ≈ 1.3 tokens roughly. Tile the varied corpus until we exceed + // nPrompt, then truncate to characters so the server tokenises a passage + // close to the requested size. + corpusWords := strings.Fields(benchCorpus) + target := nPrompt * 5 // ~5 chars/token gives a generous over-estimate + var b strings.Builder + for b.Len() < target { + for _, w := range corpusWords { + b.WriteString(w) + b.WriteByte(' ') + if b.Len() >= target { + break + } + } + } + prompt := strings.TrimSpace(b.String()) + // Use the same endpoint your real chat hits, so the comparison is honest + // (chat template, reasoning, OpenAI-compat layer all included). + payload := map[string]any{ + "model": "jean", + "messages": []Message{{Role: "user", Content: prompt + "\n\nContinue this passage with another 1000+ words of original varied prose, mixing French and English narrative paragraphs on different topics."}}, + "max_tokens": nPredict, + "stream": false, + "temperature": 0.7, + "cache_prompt": false, + } + body, _ := json.Marshal(payload) + url := fmt.Sprintf("http://localhost:%d/v1/chat/completions", port) + t0 := time.Now() + req, _ := http.NewRequest("POST", url, bytes.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + authHeader(req) + client := &http.Client{Timeout: 5 * time.Minute} + resp, err := client.Do(req) + if err != nil { + return nil, err + } + defer resp.Body.Close() + var parsed struct { + Timings struct { + PromptN int `json:"prompt_n"` + PromptMs float64 `json:"prompt_ms"` + PromptPerSecond float64 `json:"prompt_per_second"` + PredictedN int `json:"predicted_n"` + PredictedMs float64 `json:"predicted_ms"` + PredictedPerSec float64 `json:"predicted_per_second"` + } `json:"timings"` + Usage struct { + PromptTokens int `json:"prompt_tokens"` + CompletionTokens int `json:"completion_tokens"` + } `json:"usage"` + } + if err := json.NewDecoder(resp.Body).Decode(&parsed); err != nil { + return nil, err + } + // /v1/chat/completions may report timings at top level or omit them; if + // missing, fall back to wall-clock derived from `usage` so we always show + // numbers comparable to what chat displays in its label. + if parsed.Timings.PredictedN == 0 && parsed.Usage.CompletionTokens > 0 { + elapsed := time.Since(t0).Seconds() + parsed.Timings.PromptN = parsed.Usage.PromptTokens + parsed.Timings.PredictedN = parsed.Usage.CompletionTokens + // Distribute elapsed time using a rough split (prefill is usually <20% at this size). + parsed.Timings.PromptMs = elapsed * 1000 * 0.15 + parsed.Timings.PredictedMs = elapsed * 1000 * 0.85 + if parsed.Timings.PromptMs > 0 { + parsed.Timings.PromptPerSecond = float64(parsed.Timings.PromptN) / (parsed.Timings.PromptMs / 1000) + } + if parsed.Timings.PredictedMs > 0 { + parsed.Timings.PredictedPerSec = float64(parsed.Timings.PredictedN) / (parsed.Timings.PredictedMs / 1000) + } + } + elapsed := time.Since(t0).Seconds() + t := parsed.Timings + return &benchResult{ + PromptN: t.PromptN, PromptMs: t.PromptMs, PromptPerSecond: t.PromptPerSecond, + PredictedN: t.PredictedN, PredictedMs: t.PredictedMs, PredictedPerSec: t.PredictedPerSec, + Elapsed: elapsed, + }, nil +} + +func cmdBench(args []string) error { + nPredict, nPrompt := 300, 2000 + if len(args) >= 1 && args[0] != "" { + if n, err := strconv.Atoi(args[0]); err == nil { + nPredict = n + } else { + return fmt.Errorf("argument invalide: %s", args[0]) + } + } + if len(args) >= 2 && args[1] != "" { + if n, err := strconv.Atoi(args[1]); err == nil { + nPrompt = n + } else { + return fmt.Errorf("argument invalide: %s", args[1]) + } + } + fmt.Printf("[bench] prompt ~%d tokens, n_predict=%d…\n", nPrompt, nPredict) + r, err := runBench(nPrompt, nPredict) + if err != nil { + return err + } + fmt.Println() + fmt.Printf(" %s %7.1f tok/s (%d tokens en %.2fs)\n", cyan("Prefill"), r.PromptPerSecond, r.PromptN, r.PromptMs/1000) + fmt.Printf(" %s %7.1f tok/s (%d tokens en %.2fs)\n", cyan("Decode "), r.PredictedPerSec, r.PredictedN, r.PredictedMs/1000) + fmt.Printf(" Total %.2fs\n", r.Elapsed) + fmt.Println() + return nil +} diff --git a/chat.go b/chat.go new file mode 100644 index 0000000..a3070eb --- /dev/null +++ b/chat.go @@ -0,0 +1,115 @@ +package main + +import ( + "bufio" + "fmt" + "os" + "strings" +) + +// cmdChat is the interactive terminal chat loop. +// First positional arg (if any) becomes the system prompt. +func cmdChat(args []string) error { + if !healthCheck() { + return fmt.Errorf("serveur injoignable sur :%d — jean start d'abord", LLMPort()) + } + sysPrompt := strings.Join(args, " ") + msgs := []Message{} + if sysPrompt != "" { + msgs = append(msgs, Message{Role: "system", Content: sysPrompt}) + } + fmt.Printf("\n%s — /reset pour vider, /sys pour changer le system, /quit ou Ctrl-D pour sortir\n", cyan("jean chat")) + if sysPrompt != "" { + fmt.Println(dim("system: " + sysPrompt)) + } + fmt.Println() + sc := bufio.NewScanner(os.Stdin) + sc.Buffer(make([]byte, 0, 64*1024), 1<<20) + for { + fmt.Print(bold("you") + " > ") + if !sc.Scan() { + fmt.Println() + return nil + } + user := sc.Text() + u := strings.TrimSpace(user) + if u == "" { + continue + } + if u == "/quit" || u == "/exit" { + return nil + } + if u == "/reset" { + kept := []Message{} + for _, m := range msgs { + if m.Role == "system" { + kept = append(kept, m) + } + } + msgs = kept + fmt.Println(dim("[contexte vidé]")) + continue + } + if strings.HasPrefix(u, "/sys ") { + newSys := strings.TrimSpace(u[5:]) + msgs = msgs[:0] + if newSys != "" { + msgs = append(msgs, Message{Role: "system", Content: newSys}) + } + fmt.Println(dim("[system mis à jour]")) + continue + } + msgs = append(msgs, Message{Role: "user", Content: user}) + full := strings.Builder{} + inReason := false + var stats *StatsEvent + // Print the assistant prefix once; reasoning is shown inline with a tag. + fmt.Print(cyan("jean") + " > ") + err := runChat(InjectSkills(msgs), 0.7, func(ev StreamEvent) bool { + switch { + case ev.Err != nil: + fmt.Printf("\n%s\n", red("[erreur] "+ev.Err.Error())) + case ev.Stats != nil: + stats = ev.Stats + case ev.ToolUsed != nil: + icon := "📖" + verb := "lecture du skill" + if ev.ToolUsed.Name == "run_shell" { + icon = "⚙️" + verb = "exécution" + } + if inReason { + fmt.Print("\n") + inReason = false + } + fmt.Printf("\n%s %s : %s\n%s ", dim(icon+" "+verb), "", magenta(ev.ToolUsed.Label), cyan("jean")+" >") + case ev.Reasoning != "": + if !inReason { + fmt.Print(magenta("[reasoning] ") + dim("")) + inReason = true + } + fmt.Print(dim(ev.Reasoning)) + case ev.Content != "": + if inReason { + fmt.Print("\n" + cyan("jean") + " > ") + inReason = false + } + full.WriteString(ev.Content) + fmt.Print(ev.Content) + } + return true + }) + if inReason { + fmt.Println() + } + fmt.Println() + if stats != nil { + fmt.Printf("%s prefill %d tok · %.0f tok/s decode %d tok · %.1f tok/s\n", + dim("→"), stats.PromptTokens, stats.PromptPerSecond, stats.GenTokens, stats.GenPerSecond) + } + fmt.Println() + if err == nil { + msgs = append(msgs, Message{Role: "assistant", Content: full.String()}) + } + } +} diff --git a/config.go b/config.go new file mode 100644 index 0000000..bb614aa --- /dev/null +++ b/config.go @@ -0,0 +1,150 @@ +package main + +import ( + "bufio" + "crypto/rand" + "encoding/hex" + "fmt" + "net/http" + "os" + "strconv" + "strings" +) + +// readAPIKey returns the trimmed contents of $JEAN_HOME/.api_key, or "" if the +// file is absent/empty. This store is independent of config.env so the key +// survives preset switches (which rewrite config.env wholesale). +func readAPIKey() string { + b, err := os.ReadFile(apiKeyPath()) + if err != nil { + return "" + } + return strings.TrimSpace(string(b)) +} + +// authHeader sets the Authorization: Bearer header on req when an API key is +// configured, so jean's own internal calls (chat/web/bench/test) authenticate +// against a protected llama-server. No-op when no key is set. +func authHeader(req *http.Request) { + if k := readAPIKey(); k != "" { + req.Header.Set("Authorization", "Bearer "+k) + } +} + +// cmdSetAPIKey sets (or clears) the API key stored in $JEAN_HOME/.api_key. When +// exposed on the internet, llama-server requires "Authorization: Bearer " +// for every call. +// +// jean set-api-key définit la clé +// jean set-api-key génère une clé aléatoire +// jean set-api-key "" supprime la protection +func cmdSetAPIKey(args []string) error { + var key string + switch { + case len(args) == 0: + buf := make([]byte, 24) + if _, err := rand.Read(buf); err != nil { + return err + } + key = "sk-jean-" + hex.EncodeToString(buf) + fmt.Printf("%s clé générée : %s\n", green("[ok]"), bold(key)) + case args[0] == "" || args[0] == "off" || args[0] == "none": + key = "" + default: + key = strings.TrimSpace(args[0]) + } + // Stocke dans le fichier dédié (survit aux switches de preset). On nettoie + // aussi un éventuel API_KEY résiduel dans config.env pour éviter la confusion. + _ = SetConfigKey("API_KEY", "") + if key == "" { + if err := os.Remove(apiKeyPath()); err != nil && !os.IsNotExist(err) { + return err + } + } else if err := os.WriteFile(apiKeyPath(), []byte(key+"\n"), 0o600); err != nil { + return err + } + if key == "" { + fmt.Printf("%s API_KEY supprimée — serveur ouvert (pas d'authentification)\n", yellow("[info]")) + } else { + fmt.Printf("%s API_KEY enregistrée dans %s\n", green("[ok]"), apiKeyPath()) + fmt.Printf(" les clients doivent envoyer : %s\n", dim("Authorization: Bearer "+key)) + } + fmt.Print(dim("[info] redémarrer le service pour appliquer ? [Y/n] ")) + sc := bufio.NewScanner(os.Stdin) + if sc.Scan() && strings.HasPrefix(strings.ToLower(strings.TrimSpace(sc.Text())), "n") { + fmt.Println(dim("[info] pense à lancer 'jean restart'")) + return nil + } + return serviceAction("restart") +} + +// ReadConfig parses config.env into a key/value map. +// Lines starting with '#' and blanks are ignored. Values may be quoted with ". +func ReadConfig() map[string]string { + m := map[string]string{} + b, err := os.ReadFile(confPath()) + if err != nil { + return m + } + for _, line := range strings.Split(string(b), "\n") { + s := strings.TrimSpace(line) + if s == "" || strings.HasPrefix(s, "#") { + continue + } + i := strings.IndexByte(s, '=') + if i < 0 { + continue + } + k := strings.TrimSpace(s[:i]) + v := strings.TrimSpace(s[i+1:]) + v = strings.Trim(v, "\"") + m[k] = v + } + return m +} + +// SetConfigKey sets key=value in config.env, updating the line in place if the +// key already exists (preserving comments/order) or appending it otherwise. +// An empty value removes the key. The file is created if missing. +func SetConfigKey(key, value string) error { + b, _ := os.ReadFile(confPath()) + lines := []string{} + if len(b) > 0 { + lines = strings.Split(strings.TrimRight(string(b), "\n"), "\n") + } + newLine := key + "=" + value + found := false + out := []string{} + for _, line := range lines { + s := strings.TrimSpace(line) + if s == "" || strings.HasPrefix(s, "#") { + out = append(out, line) + continue + } + i := strings.IndexByte(s, '=') + if i >= 0 && strings.TrimSpace(s[:i]) == key { + found = true + if value != "" { + out = append(out, newLine) + } + // empty value => drop the line + continue + } + out = append(out, line) + } + if !found && value != "" { + out = append(out, newLine) + } + content := strings.Join(out, "\n") + "\n" + return os.WriteFile(confPath(), []byte(content), 0o644) +} + +// LLMPort returns the configured server port (config.env PORT), default 8080. +func LLMPort() int { + if p, ok := ReadConfig()["PORT"]; ok { + if n, err := strconv.Atoi(p); err == nil && n > 0 { + return n + } + } + return 8080 +} diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..af69418 --- /dev/null +++ b/go.mod @@ -0,0 +1,3 @@ +module github.com/jean-llm/jean + +go 1.22 diff --git a/install.go b/install.go new file mode 100644 index 0000000..def8844 --- /dev/null +++ b/install.go @@ -0,0 +1,196 @@ +package main + +import ( + "fmt" + "os" + "os/exec" + "os/user" + "path/filepath" + "strings" +) + +const configTemplate = `# Configuration JEAN — édite-moi puis: jean restart +# Le service systemd lit ce fichier et lance ton binaire llama.cpp. + +# Chemins +BIN="/usr/local/bin/llama-server" +MODEL="/home/USER/models/your-model.gguf" + +# Serveur +PORT="8080" +HOST="0.0.0.0" + +# Inference +CTX="32768" +BATCH="2048" +UBATCH="512" +NGL="999" + +# Args supplémentaires passés à llama-server +EXTRA_ARGS="" +` + +const sudoersTemplate = `# Allow %s to manage the %s systemd unit without a password (installed by jean). +%s ALL=(root) NOPASSWD: /bin/systemctl start %s, /bin/systemctl stop %s, /bin/systemctl restart %s, /bin/systemctl enable %s, /bin/systemctl disable %s +` + +const serviceUnitTemplate = `[Unit] +Description=JEAN llama.cpp server +After=network.target + +[Service] +Type=simple +User=%s +WorkingDirectory=%s +ExecStart=%s +Restart=on-failure +RestartSec=3 + +[Install] +WantedBy=multi-user.target +` + +func cmdInstall(args []string) error { + if os.Geteuid() != 0 { + return fmt.Errorf("jean install doit être exécuté en root (sudo jean install)") + } + targetUser := os.Getenv("SUDO_USER") + if targetUser == "" { + targetUser = "root" + } + for _, a := range args { + if strings.HasPrefix(a, "--user=") { + targetUser = strings.TrimPrefix(a, "--user=") + } + } + u, err := user.Lookup(targetUser) + if err != nil { + return fmt.Errorf("utilisateur '%s' introuvable: %w", targetUser, err) + } + jeanHome := DefaultJeanHome + if v := os.Getenv("JEAN_HOME"); v != "" { + jeanHome = v + } + svc := serviceName() + + fmt.Printf("Installation pour utilisateur %s\n", cyan(targetUser)) + fmt.Printf(" JEAN_HOME = %s\n", jeanHome) + fmt.Printf(" service = %s\n", svc) + + // 1. Create directories + for _, d := range []string{jeanHome, filepath.Join(jeanHome, "configs"), filepath.Join(jeanHome, "SKILLS")} { + if err := os.MkdirAll(d, 0o755); err != nil { + return err + } + } + + // 2. Drop a config.env if none exists + conf := filepath.Join(jeanHome, "config.env") + if _, err := os.Stat(conf); os.IsNotExist(err) { + body := strings.ReplaceAll(configTemplate, "USER", targetUser) + if err := os.WriteFile(conf, []byte(body), 0o644); err != nil { + return err + } + fmt.Printf(" %s écrit %s\n", green("✓"), conf) + } + + // 3. Symlink current binary to /usr/local/bin/jean + self, err := os.Executable() + if err != nil { + return err + } + target := "/usr/local/bin/jean" + _ = os.Remove(target) + if err := os.Symlink(self, target); err != nil { + // fall back to copy if symlink fails (e.g. cross-fs) + if data, err := os.ReadFile(self); err == nil { + _ = os.WriteFile(target, data, 0o755) + } + } + fmt.Printf(" %s %s -> %s\n", green("✓"), target, self) + + // 4. Drop /etc/default/jean so root invocations resolve JEAN_HOME correctly. + defaults := fmt.Sprintf("# Generated by jean install — racine des configs/skills/SKILLS\nJEAN_HOME=%s\n", jeanHome) + if err := os.WriteFile("/etc/default/jean", []byte(defaults), 0o644); err != nil { + return err + } + fmt.Printf(" %s /etc/default/jean\n", green("✓")) + + // 5. Write the systemd unit (ExecStart = `jean serve`, no start.sh needed) + unit := fmt.Sprintf(serviceUnitTemplate, targetUser, jeanHome, "/usr/local/bin/jean serve") + unitPath := "/etc/systemd/system/" + svc + ".service" + if err := os.WriteFile(unitPath, []byte(unit), 0o644); err != nil { + return err + } + fmt.Printf(" %s %s\n", green("✓"), unitPath) + + // 5. Sudoers drop-in + sudoers := fmt.Sprintf(sudoersTemplate, targetUser, svc, targetUser, svc, svc, svc, svc, svc) + sudoersPath := "/etc/sudoers.d/jean-" + svc + if err := os.WriteFile(sudoersPath, []byte(sudoers), 0o440); err != nil { + return err + } + fmt.Printf(" %s %s\n", green("✓"), sudoersPath) + + // 6. chown JEAN_HOME contents to target user + chown(jeanHome, u) + + // 7. systemd reload + _ = exec.Command("systemctl", "daemon-reload").Run() + + fmt.Println() + fmt.Printf("%s installation terminée.\n", green("[ok]")) + fmt.Printf("\nProchaines étapes :\n") + fmt.Printf(" 1. édite la config : %s\n", bold("sudo -u "+targetUser+" jean edit")) + fmt.Printf(" (renseigne BIN, MODEL, etc.)\n") + fmt.Printf(" 2. démarre le service: %s\n", bold("sudo -u "+targetUser+" jean start")) + fmt.Printf(" 3. UI web : %s\n", bold("sudo -u "+targetUser+" jean web")) + return nil +} + +func cmdUninstall(args []string) error { + if os.Geteuid() != 0 { + return fmt.Errorf("jean uninstall doit être exécuté en root") + } + svc := serviceName() + keepData := false + for _, a := range args { + if a == "--purge" { + keepData = false + } + if a == "--keep-data" { + keepData = true + } + } + _ = exec.Command("systemctl", "stop", svc).Run() + _ = exec.Command("systemctl", "disable", svc).Run() + for _, p := range []string{ + "/etc/systemd/system/" + svc + ".service", + "/etc/sudoers.d/jean-" + svc, + "/etc/default/jean", + "/usr/local/bin/jean", + } { + if err := os.Remove(p); err == nil { + fmt.Printf(" %s %s\n", green("✓"), p) + } + } + _ = exec.Command("systemctl", "daemon-reload").Run() + if !keepData { + fmt.Println(dim("(données utilisateur conservées — supprime $JEAN_HOME manuellement si tu veux purger)")) + } + fmt.Println(green("[ok]") + " désinstallé") + return nil +} + +// chown recursively changes ownership of path to the given user/group. +func chown(path string, u *user.User) { + var uid, gid int + fmt.Sscanf(u.Uid, "%d", &uid) + fmt.Sscanf(u.Gid, "%d", &gid) + filepath.Walk(path, func(p string, info os.FileInfo, err error) error { + if err == nil { + _ = os.Chown(p, uid, gid) + } + return nil + }) +} diff --git a/llamacpp.go b/llamacpp.go new file mode 100644 index 0000000..367d846 --- /dev/null +++ b/llamacpp.go @@ -0,0 +1,607 @@ +package main + +import ( + "fmt" + "os" + "os/exec" + "path/filepath" + "runtime" + "sort" + "strings" +) + +// llamacpp.go — gestion du backend llama.cpp (clone, build, mise à jour). +// +// `jean llamacpp install` installe un build neuf, détecte automatiquement +// l'accélérateur (CUDA / ROCm / Metal / CPU) et la +// compute capability du GPU, puis pointe BIN dessus. +// `jean llamacpp update` met à jour le dépôt existant (git pull) et recompile +// avec la bonne config, sans intervention. +// `jean llamacpp status` montre le commit courant, le backend détecté et le +// retard éventuel sur origin. + +const llamacppRepoURL = "https://github.com/ggml-org/llama.cpp.git" + +// buildPlan capture les flags CMake adaptés à la machine courante. +type buildPlan struct { + backend string // "cuda" | "hip" | "metal" | "vulkan" | "cpu" + cudaArch string // ex. "120" ou "86;89" (vide => détection native par CMake) + cudaCXX string // chemin de nvcc quand backend == cuda + flags []string // flags -D… passés à `cmake -B build` + jobs int // parallélisme du build +} + +func cmdLlamacpp(args []string) error { + sub := "" + if len(args) > 0 { + sub = args[0] + args = args[1:] + } + switch sub { + case "install": + return llamacppInstall(args) + case "update", "upgrade": + return llamacppUpdate(args) + case "status", "info", "": + return llamacppStatus(args) + default: + return fmt.Errorf("sous-commande inconnue: %s (install | update | status)", sub) + } +} + +// --------------------------------------------------------------------------- +// Localisation du dépôt +// --------------------------------------------------------------------------- + +// llamacppRepoDir resolves the llama.cpp checkout: derived from config BIN when +// possible (so `update` targets whatever build the service actually runs), +// otherwise the default under $JEAN_HOME/backends/llama.cpp. +func llamacppRepoDir() string { + if bin := ReadConfig()["BIN"]; bin != "" { + if real, err := filepath.EvalSymlinks(bin); err == nil { + bin = real + } + if root := findRepoRoot(bin); root != "" { + return root + } + } + return defaultRepoDir() +} + +func defaultRepoDir() string { + return filepath.Join(JeanHome(), "backends", "llama.cpp") +} + +// findRepoRoot walks up from a binary path (…/build/bin/llama-server) looking +// for the llama.cpp source root (a dir holding .git or CMakeLists.txt). +func findRepoRoot(binPath string) string { + d := filepath.Dir(binPath) + for i := 0; i < 6; i++ { + if isDir(filepath.Join(d, ".git")) || isFile(filepath.Join(d, "CMakeLists.txt")) { + return d + } + parent := filepath.Dir(d) + if parent == d { + break + } + d = parent + } + return "" +} + +// --------------------------------------------------------------------------- +// install +// --------------------------------------------------------------------------- + +func llamacppInstall(args []string) error { + repo := defaultRepoDir() + ref := "" + force := false + noSwitch := false + for _, a := range args { + switch { + case strings.HasPrefix(a, "--dir="): + repo = strings.TrimPrefix(a, "--dir=") + case strings.HasPrefix(a, "--ref="): + ref = strings.TrimPrefix(a, "--ref=") + case a == "--force": + force = true + case a == "--no-switch": + noSwitch = true + default: + return fmt.Errorf("option inconnue: %s", a) + } + } + + if err := requireTools("git", "cmake"); err != nil { + return err + } + + // Dépôt déjà présent ? On bascule sur update plutôt que de re-cloner. + if isDir(filepath.Join(repo, ".git")) { + if !force { + fmt.Printf("%s dépôt déjà présent dans %s\n", yellow("[info]"), repo) + fmt.Printf(" → %s pour le mettre à jour, ou --force pour repartir de zéro\n", bold("jean llamacpp update")) + return nil + } + fmt.Printf("%s --force : suppression de %s\n", yellow("[info]"), repo) + if err := os.RemoveAll(repo); err != nil { + return err + } + } + + if err := os.MkdirAll(filepath.Dir(repo), 0o755); err != nil { + return err + } + + fmt.Printf("%s clone de llama.cpp dans %s\n", cyan("▶"), repo) + if err := runStep("git clone", "", "git", "clone", "--depth=1", llamacppRepoURL, repo); err != nil { + return err + } + if ref != "" { + // --depth=1 ne récupère que HEAD ; on approfondit pour atteindre le ref. + _ = runStep("git fetch", repo, "git", "fetch", "--unshallow", "origin") + if err := runStep("git checkout", repo, "git", "checkout", ref); err != nil { + return err + } + } + + plan := detectBuildPlan() + printPlan(plan, repo) + + if err := buildLlamacpp(repo, plan, true); err != nil { + return err + } + + bin := filepath.Join(repo, "build", "bin", "llama-server") + if !isFile(bin) { + return fmt.Errorf("build terminé mais binaire introuvable: %s", bin) + } + fmt.Printf("\n%s binaire compilé : %s\n", green("✓"), bin) + + if noSwitch { + fmt.Printf("%s --no-switch : config.env inchangée (BIN à régler manuellement)\n", dim("[info]")) + return nil + } + if err := SetConfigKey("BIN", bin); err != nil { + return fmt.Errorf("build ok mais échec écriture BIN dans config.env: %w", err) + } + fmt.Printf("%s BIN mis à jour dans %s\n", green("✓"), confPath()) + fmt.Printf("\nProchaines étapes :\n 1. renseigne MODEL : %s\n 2. démarre : %s\n", + bold("jean edit"), bold("jean restart")) + return nil +} + +// --------------------------------------------------------------------------- +// update +// --------------------------------------------------------------------------- + +func llamacppUpdate(args []string) error { + ref := "" + clean := false + noRestart := false + force := false + for _, a := range args { + switch { + case strings.HasPrefix(a, "--ref="): + ref = strings.TrimPrefix(a, "--ref=") + case a == "--clean": + clean = true + case a == "--no-restart": + noRestart = true + case a == "--force": + force = true + default: + return fmt.Errorf("option inconnue: %s", a) + } + } + + if err := requireTools("git", "cmake"); err != nil { + return err + } + + repo := llamacppRepoDir() + if !isDir(filepath.Join(repo, ".git")) { + return fmt.Errorf("aucun dépôt llama.cpp trouvé (%s).\n → lance d'abord %s", repo, bold("jean llamacpp install")) + } + fmt.Printf("%s dépôt : %s\n", cyan("▶"), repo) + + oldCommit := gitOutput(repo, "rev-parse", "--short", "HEAD") + + // Détermine la branche à suivre (master par défaut si HEAD détaché). + branch := ref + if branch == "" { + branch = gitOutput(repo, "rev-parse", "--abbrev-ref", "HEAD") + if branch == "" || branch == "HEAD" { + branch = "master" + } + } + + if err := runStep("git fetch", repo, "git", "fetch", "origin", "--quiet"); err != nil { + return err + } + + // Déjà à jour ? On s'arrête (sauf --clean / --force qui forcent un rebuild). + localRev := gitOutput(repo, "rev-parse", "HEAD") + remoteRev := gitOutput(repo, "rev-parse", "origin/"+branch) + bin := filepath.Join(repo, "build", "bin", "llama-server") + if localRev != "" && localRev == remoteRev && !clean && !force && isFile(bin) { + fmt.Printf("%s déjà à jour (%s) — rien à faire\n", green("[ok]"), oldCommit) + fmt.Printf(" (utilise %s pour forcer une recompilation)\n", dim("--force")) + return nil + } + + // Met à jour la source. + if ref != "" { + if err := runStep("git checkout", repo, "git", "checkout", ref); err != nil { + return err + } + } else { + if err := runStep("git pull --ff-only", repo, "git", "pull", "--ff-only", "origin", branch); err != nil { + return fmt.Errorf("git pull a échoué (modifs locales ? essaie de résoudre à la main): %w", err) + } + } + newCommit := gitOutput(repo, "rev-parse", "--short", "HEAD") + + // On stoppe le service : le binaire en cours d'exécution ne peut pas être + // réécrit par l'étape de link (« Text file busy »). + svcWasUp := serviceIsActive() + if svcWasUp { + fmt.Printf("%s arrêt du service %s le temps du build…\n", yellow("[info]"), serviceName()) + if err := serviceAction("stop"); err != nil { + fmt.Printf("%s impossible d'arrêter le service (%v) — le build peut échouer si le binaire est verrouillé\n", yellow("[warn]"), err) + } + } + + plan := detectBuildPlan() + printPlan(plan, repo) + + if err := buildLlamacpp(repo, plan, clean); err != nil { + // On tente de remettre le service debout même en cas d'échec. + if svcWasUp && !noRestart { + _ = serviceAction("start") + } + return err + } + if !isFile(bin) { + return fmt.Errorf("build terminé mais binaire introuvable: %s", bin) + } + + fmt.Printf("\n%s mis à jour : %s → %s\n", green("✓"), oldCommit, newCommit) + + if noRestart { + fmt.Printf("%s --no-restart : pense à lancer %s\n", dim("[info]"), bold("jean restart")) + return nil + } + if svcWasUp { + fmt.Printf("%s redémarrage du service…\n", cyan("▶")) + return serviceAction("start") + } + fmt.Printf("%s service non démarré auparavant — lance %s quand tu veux\n", dim("[info]"), bold("jean start")) + return nil +} + +// --------------------------------------------------------------------------- +// status +// --------------------------------------------------------------------------- + +func llamacppStatus(args []string) error { + repo := llamacppRepoDir() + fmt.Printf("%s\n", bold("llama.cpp")) + fmt.Printf(" dépôt : %s\n", repo) + if !isDir(filepath.Join(repo, ".git")) { + fmt.Printf(" %s pas encore installé — %s\n", yellow("état"), bold("jean llamacpp install")) + return nil + } + commit := gitOutput(repo, "log", "-1", "--format=%h %ci %s") + branch := gitOutput(repo, "rev-parse", "--abbrev-ref", "HEAD") + fmt.Printf(" branche : %s\n", branch) + fmt.Printf(" commit : %s\n", commit) + + bin := filepath.Join(repo, "build", "bin", "llama-server") + if isFile(bin) { + fmt.Printf(" binaire : %s\n", green(bin)) + } else { + fmt.Printf(" binaire : %s (pas encore compilé)\n", yellow("absent")) + } + + // Retard sur origin (best-effort, sans fetch réseau). + if branch != "" && branch != "HEAD" { + if behind := gitOutput(repo, "rev-list", "--count", "HEAD..origin/"+branch); behind != "" && behind != "0" { + fmt.Printf(" maj : %s commit(s) de retard sur origin/%s — %s\n", yellow(behind), branch, bold("jean llamacpp update")) + } + } + + plan := detectBuildPlan() + fmt.Printf(" backend : %s\n", planLabel(plan)) + return nil +} + +// --------------------------------------------------------------------------- +// Détection matérielle & build +// --------------------------------------------------------------------------- + +// detectBuildPlan probes the machine and returns the CMake flags for the best +// available accelerator. Order of preference: CUDA → ROCm/HIP → Metal (macOS) +// → Vulkan → CPU. +func detectBuildPlan() buildPlan { + p := buildPlan{backend: "cpu", jobs: numJobs()} + // Flags communs : Release + tuning natif pour la machine de build. + // (libcurl est activé d'office par llama.cpp ; LLAMA_CURL est déprécié.) + p.flags = []string{ + "-DCMAKE_BUILD_TYPE=Release", + "-DGGML_NATIVE=ON", + } + + if runtime.GOOS == "darwin" { + // Metal est activé par défaut sur Apple Silicon ; on l'explicite. + p.backend = "metal" + p.flags = append(p.flags, "-DGGML_METAL=ON") + return p + } + + // CUDA : nvcc présent ET un GPU NVIDIA visible. + if nvcc := findNvcc(); nvcc != "" && hasNvidiaGPU() { + p.backend = "cuda" + p.cudaCXX = nvcc + p.flags = append(p.flags, "-DGGML_CUDA=ON", "-DGGML_CUDA_F16=ON", "-DGGML_CUDA_FA_ALL_QUANTS=ON") + if arch := detectCudaArch(); arch != "" { + p.cudaArch = arch + p.flags = append(p.flags, "-DCMAKE_CUDA_ARCHITECTURES="+arch) + } + return p + } + + // AMD ROCm / HIP. + if hasTool("hipcc") || isDir("/opt/rocm") { + p.backend = "hip" + p.flags = append(p.flags, "-DGGML_HIP=ON") + return p + } + + // Vulkan (GPU générique) — utile sur Intel/AMD sans ROCm. + if hasTool("glslc") && (isFile("/usr/lib/x86_64-linux-gnu/libvulkan.so.1") || hasTool("vulkaninfo")) { + p.backend = "vulkan" + p.flags = append(p.flags, "-DGGML_VULKAN=ON") + return p + } + + return p // CPU +} + +// buildLlamacpp configures and builds the llama-server target. It handles the +// "relocated checkout" gotcha: a build/ whose CMake cache was generated under a +// different source path can't reconfigure in place, so we wipe it. `clean` +// forces a from-scratch build regardless. +func buildLlamacpp(repo string, p buildPlan, clean bool) error { + build := filepath.Join(repo, "build") + + if clean || cacheStale(build, repo) { + if isDir(build) { + fmt.Printf("%s reconfiguration propre (suppression de build/)\n", dim("[info]")) + old := build + ".old" + _ = os.RemoveAll(old) + if err := os.Rename(build, old); err != nil { + _ = os.RemoveAll(build) // dernier recours + } + } + } + + // nvcc doit être dans le PATH et exposé via CUDACXX pour la config CMake. + env := "" + if p.backend == "cuda" && p.cudaCXX != "" { + cudaBin := filepath.Dir(p.cudaCXX) + env = "CUDACXX=" + p.cudaCXX + "\x00PATH=" + cudaBin + string(os.PathListSeparator) + os.Getenv("PATH") + } + + cfgArgs := append([]string{"-B", "build", "-S", "."}, p.flags...) + if err := runStepEnv("cmake configure", repo, env, "cmake", cfgArgs...); err != nil { + return fmt.Errorf("configuration CMake échouée: %w", err) + } + + buildArgs := []string{"--build", "build", "--config", "Release", + "-j", fmt.Sprintf("%d", p.jobs), "--target", "llama-server"} + if err := runStepEnv("cmake build", repo, env, "cmake", buildArgs...); err != nil { + return fmt.Errorf("compilation échouée: %w", err) + } + return nil +} + +// cacheStale reports whether build/CMakeCache.txt was generated for a different +// source directory than `repo` (the relocated-checkout case). +func cacheStale(build, repo string) bool { + cache := filepath.Join(build, "CMakeCache.txt") + b, err := os.ReadFile(cache) + if err != nil { + return false // pas de cache => configure neuf, rien à nettoyer + } + absRepo, _ := filepath.Abs(repo) + for _, line := range strings.Split(string(b), "\n") { + // CMAKE_HOME_DIRECTORY pointe vers le source dir d'origine. + if strings.HasPrefix(line, "CMAKE_HOME_DIRECTORY:") { + if i := strings.IndexByte(line, '='); i >= 0 { + home := strings.TrimSpace(line[i+1:]) + return home != "" && home != absRepo + } + } + } + return false +} + +// --------------------------------------------------------------------------- +// Sondes matérielles +// --------------------------------------------------------------------------- + +// findNvcc returns the path to nvcc from PATH or a /usr/local/cuda* install, +// preferring the highest version. +func findNvcc() string { + if p, err := exec.LookPath("nvcc"); err == nil { + return p + } + if p := "/usr/local/cuda/bin/nvcc"; isFile(p) { + return p + } + matches, _ := filepath.Glob("/usr/local/cuda-*/bin/nvcc") + if len(matches) > 0 { + sort.Strings(matches) // cuda-12.2 < cuda-12.8 lexicographiquement → on prend le dernier + return matches[len(matches)-1] + } + return "" +} + +func hasNvidiaGPU() bool { + if !hasTool("nvidia-smi") { + return false + } + out, err := exec.Command("nvidia-smi", "-L").Output() + return err == nil && strings.Contains(string(out), "GPU") +} + +// detectCudaArch queries every GPU's compute capability via nvidia-smi and +// returns them as CMake-style arch codes (e.g. "8.6" → "86"), deduped and +// joined with ';'. Empty when the driver is too old to report it (CMake then +// falls back to native detection). +func detectCudaArch() string { + out, err := exec.Command("nvidia-smi", "--query-gpu=compute_cap", "--format=csv,noheader").Output() + if err != nil { + return "" + } + seen := map[string]bool{} + var archs []string + for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") { + cap := strings.TrimSpace(line) + if cap == "" || strings.Contains(strings.ToLower(cap), "not supported") { + continue + } + code := strings.ReplaceAll(cap, ".", "") // "12.0" → "120" + if code != "" && !seen[code] { + seen[code] = true + archs = append(archs, code) + } + } + return strings.Join(archs, ";") +} + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +func numJobs() int { + n := runtime.NumCPU() + if n < 1 { + return 1 + } + return n +} + +func isFile(p string) bool { + fi, err := os.Stat(p) + return err == nil && !fi.IsDir() +} + +func isDir(p string) bool { + fi, err := os.Stat(p) + return err == nil && fi.IsDir() +} + +func hasTool(name string) bool { + _, err := exec.LookPath(name) + return err == nil +} + +func requireTools(tools ...string) error { + var missing []string + for _, t := range tools { + if !hasTool(t) { + missing = append(missing, t) + } + } + if len(missing) > 0 { + return fmt.Errorf("outils manquants: %s — installe-les puis réessaie", strings.Join(missing, ", ")) + } + return nil +} + +func serviceIsActive() bool { + out, _ := exec.Command("systemctl", "is-active", serviceName()).Output() + return strings.TrimSpace(string(out)) == "active" +} + +// gitOutput runs a git command in `dir` and returns trimmed stdout (or ""). +func gitOutput(dir string, args ...string) string { + cmd := exec.Command("git", args...) + cmd.Dir = dir + out, err := cmd.Output() + if err != nil { + return "" + } + return strings.TrimSpace(string(out)) +} + +// runStep runs a command in `dir` streaming output live to the terminal. +func runStep(name, dir, bin string, args ...string) error { + return runStepEnv(name, dir, "", bin, args...) +} + +// runStepEnv is runStep with optional extra env vars (NUL-separated KEY=VAL +// pairs in `extraEnv`, which override existing ones). +func runStepEnv(name, dir, extraEnv, bin string, args ...string) error { + fmt.Printf("\n%s %s %s\n", cyan("▶"), name, dim(strings.Join(args, " "))) + cmd := exec.Command(bin, args...) + cmd.Dir = dir + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + cmd.Stdin = os.Stdin + if extraEnv != "" { + env := os.Environ() + for _, kv := range strings.Split(extraEnv, "\x00") { + if kv == "" { + continue + } + env = upsertEnv(env, kv) + } + cmd.Env = env + } + return cmd.Run() +} + +// upsertEnv replaces KEY=… in env if present, else appends kv (kv is "KEY=VAL"). +func upsertEnv(env []string, kv string) []string { + key := kv + if i := strings.IndexByte(kv, '='); i >= 0 { + key = kv[:i] + } + for i, e := range env { + if strings.HasPrefix(e, key+"=") { + env[i] = kv + return env + } + } + return append(env, kv) +} + +func planLabel(p buildPlan) string { + switch p.backend { + case "cuda": + arch := p.cudaArch + if arch == "" { + arch = "native" + } + return green("CUDA") + dim(" (arch="+arch+", nvcc="+p.cudaCXX+")") + case "hip": + return green("ROCm/HIP") + case "metal": + return green("Metal") + case "vulkan": + return green("Vulkan") + default: + return yellow("CPU") + dim(" (aucun accélérateur détecté)") + } +} + +func printPlan(p buildPlan, repo string) { + fmt.Printf("\n%s configuration du build\n", bold("•")) + fmt.Printf(" backend : %s\n", planLabel(p)) + fmt.Printf(" jobs : %d\n", p.jobs) + fmt.Printf(" flags : %s\n", dim(strings.Join(p.flags, " "))) +} diff --git a/llm.go b/llm.go new file mode 100644 index 0000000..2f9b193 --- /dev/null +++ b/llm.go @@ -0,0 +1,332 @@ +package main + +import ( + "bufio" + "bytes" + "encoding/json" + "fmt" + "io" + "net/http" + "sort" + "strings" +) + +// Message is one entry in the chat history sent to llama.cpp. +// `Content` may be nil when an assistant message only contains tool_calls. +type Message struct { + Role string `json:"role"` + Content any `json:"content,omitempty"` + ToolCalls []ToolCall `json:"tool_calls,omitempty"` + ToolCallID string `json:"tool_call_id,omitempty"` +} + +type ToolCall struct { + ID string `json:"id"` + Type string `json:"type"` + Function ToolCallFunc `json:"function"` +} +type ToolCallFunc struct { + Name string `json:"name"` + Arguments string `json:"arguments"` +} + +type Tool struct { + Type string `json:"type"` + Function ToolFunction `json:"function"` +} +type ToolFunction struct { + Name string `json:"name"` + Description string `json:"description"` + Parameters any `json:"parameters"` +} + +// readSkillTool / runShellTool: OpenAI-shaped function definitions advertised +// to the model when the corresponding feature flag is on. +func readSkillTool() Tool { + return Tool{ + Type: "function", + Function: ToolFunction{ + Name: "read_skill", + Description: "Lit le contenu détaillé d'un skill listé dans le system prompt.", + Parameters: map[string]any{ + "type": "object", + "properties": map[string]any{ + "name": map[string]any{"type": "string", "description": "Nom exact du skill"}, + }, + "required": []string{"name"}, + }, + }, + } +} + +func runShellTool() Tool { + return Tool{ + Type: "function", + Function: ToolFunction{ + Name: "run_shell", + Description: "Exécute une commande shell sur le serveur (bash). Retourne stdout, stderr et le code de sortie. Utilise-le pour inspecter le système, lancer des scripts, consulter des logs. Évite les commandes destructrices sauf si l'utilisateur le demande explicitement.", + Parameters: map[string]any{ + "type": "object", + "properties": map[string]any{ + "command": map[string]any{"type": "string", "description": "La commande bash à exécuter"}, + "timeout": map[string]any{"type": "integer", "description": fmt.Sprintf("Timeout en secondes (défaut %d, max %d)", toolDefaultTimeout, toolMaxTimeout)}, + }, + "required": []string{"command"}, + }, + }, + } +} + +// InjectSkills prepends the lightweight skills directory message to msgs when +// skills are enabled. Merges with an existing system message if present. +func InjectSkills(msgs []Message) []Message { + sp := skillsSystemPrompt() + if sp == "" { + return msgs + } + if len(msgs) > 0 && msgs[0].Role == "system" { + existing, _ := msgs[0].Content.(string) + merged := append([]Message{{Role: "system", Content: sp + "\n\n" + existing}}, msgs[1:]...) + return merged + } + return append([]Message{{Role: "system", Content: sp}}, msgs...) +} + +// EnabledTools returns the tools to advertise on the next inference call. +func EnabledTools() []Tool { + tools := []Tool{} + if skillsEnabled() && len(ListSkills()) > 0 { + tools = append(tools, readSkillTool()) + } + if toolsEnabled() { + tools = append(tools, runShellTool()) + } + return tools +} + +// StreamEvent is what a ChatCallback receives for each piece of streamed output. +// Exactly one of {Content, Reasoning, ToolUsed, Stats, Err} is set per call. +type StreamEvent struct { + Content string + Reasoning string + ToolUsed *ToolUsedEvent + Stats *StatsEvent + Err error +} +type ToolUsedEvent struct { + Name string + Label string // user-visible summary (skill name or first ~80 chars of command) +} + +// StatsEvent carries llama.cpp's per-completion timing (final chunk). +type StatsEvent struct { + PromptTokens int `json:"prompt_tokens"` + PromptPerSecond float64 `json:"prompt_per_second"` + PromptMs float64 `json:"prompt_ms"` + GenTokens int `json:"gen_tokens"` + GenPerSecond float64 `json:"gen_per_second"` + GenMs float64 `json:"gen_ms"` +} + +// ChatCallback receives stream events. Return false to abort the stream. +type ChatCallback func(StreamEvent) bool + +// completionResp / streamChunk model the subset of llama.cpp's +// OpenAI-compatible /v1/chat/completions response that we care about. +type streamChunk struct { + Choices []struct { + Delta struct { + Content string `json:"content"` + ReasoningContent string `json:"reasoning_content"` + ToolCalls []ToolCall `json:"tool_calls"` + } `json:"delta"` + FinishReason string `json:"finish_reason"` + } `json:"choices"` + // llama.cpp's "timings" appears on the final chunk and on intermediate + // /completion endpoint responses. Snake-case mapping per llama.cpp source. + Timings *struct { + PromptN int `json:"prompt_n"` + PromptMs float64 `json:"prompt_ms"` + PromptPerSecond float64 `json:"prompt_per_second"` + PredictedN int `json:"predicted_n"` + PredictedMs float64 `json:"predicted_ms"` + PredictedPerSec float64 `json:"predicted_per_second"` + } `json:"timings"` +} + +// runChat drives the full inference loop including tool calling. +// On finish_reason="tool_calls" we execute locally, append a "tool" message +// and call /v1/chat/completions again — up to 8 iterations as a safety cap. +func runChat(messages []Message, temperature float64, cb ChatCallback) error { + tools := EnabledTools() + for iter := 0; iter < 8; iter++ { + payload := map[string]any{ + "model": "jean", + "messages": messages, + "stream": true, + "temperature": temperature, + } + if len(tools) > 0 { + payload["tools"] = tools + } + body, _ := json.Marshal(payload) + url := fmt.Sprintf("http://localhost:%d/v1/chat/completions", LLMPort()) + req, err := http.NewRequest("POST", url, bytes.NewReader(body)) + if err != nil { + return err + } + req.Header.Set("Content-Type", "application/json") + authHeader(req) + resp, err := http.DefaultClient.Do(req) + if err != nil { + cb(StreamEvent{Err: err}) + return err + } + toolCalls := map[int]*ToolCall{} + assistantContent := strings.Builder{} + finishReason := "" + // scanner with a big buffer — some chunks include large arguments JSON + sc := bufio.NewScanner(resp.Body) + sc.Buffer(make([]byte, 0, 64*1024), 1<<20) + aborted := false + for sc.Scan() { + line := strings.TrimSpace(sc.Text()) + if !strings.HasPrefix(line, "data:") { + continue + } + data := strings.TrimSpace(line[5:]) + if data == "" || data == "[DONE]" { + continue + } + var chunk streamChunk + if err := json.Unmarshal([]byte(data), &chunk); err != nil || len(chunk.Choices) == 0 { + continue + } + ch := chunk.Choices[0] + if ch.FinishReason != "" { + finishReason = ch.FinishReason + } + if chunk.Timings != nil { + cb(StreamEvent{Stats: &StatsEvent{ + PromptTokens: chunk.Timings.PromptN, + PromptPerSecond: chunk.Timings.PromptPerSecond, + PromptMs: chunk.Timings.PromptMs, + GenTokens: chunk.Timings.PredictedN, + GenPerSecond: chunk.Timings.PredictedPerSec, + GenMs: chunk.Timings.PredictedMs, + }}) + } + if len(ch.Delta.ToolCalls) > 0 { + for i, tc := range ch.Delta.ToolCalls { + // llama.cpp's stream may omit index; fall back to slot i. + idx := i + cur, ok := toolCalls[idx] + if !ok { + cur = &ToolCall{Type: "function"} + toolCalls[idx] = cur + } + if tc.ID != "" { + cur.ID = tc.ID + } + if tc.Function.Name != "" { + cur.Function.Name = tc.Function.Name + } + cur.Function.Arguments += tc.Function.Arguments + } + continue + } + if ch.Delta.ReasoningContent != "" { + if !cb(StreamEvent{Reasoning: ch.Delta.ReasoningContent}) { + aborted = true + break + } + } + if ch.Delta.Content != "" { + assistantContent.WriteString(ch.Delta.Content) + if !cb(StreamEvent{Content: ch.Delta.Content}) { + aborted = true + break + } + } + } + resp.Body.Close() + if aborted { + return nil + } + + if finishReason == "tool_calls" && len(toolCalls) > 0 { + // 1. Append assistant message with tool_calls so the model sees its own decision next turn. + idxs := make([]int, 0, len(toolCalls)) + for k := range toolCalls { + idxs = append(idxs, k) + } + sort.Ints(idxs) + tcs := make([]ToolCall, 0, len(idxs)) + for i, k := range idxs { + tc := *toolCalls[k] + if tc.ID == "" { + tc.ID = fmt.Sprintf("call_%d_%d", iter, i) + } + if tc.Function.Arguments == "" { + tc.Function.Arguments = "{}" + } + tcs = append(tcs, tc) + } + assistant := Message{Role: "assistant", ToolCalls: tcs} + if s := assistantContent.String(); s != "" { + assistant.Content = s + } + messages = append(messages, assistant) + // 2. Execute each tool locally and append a "tool" reply. + for _, tc := range tcs { + var args map[string]any + _ = json.Unmarshal([]byte(tc.Function.Arguments), &args) + result := "" + label := "" + switch tc.Function.Name { + case "read_skill": + name, _ := args["name"].(string) + label = name + if c := SkillContent(name); c != "" { + result = c + } else { + result = fmt.Sprintf("[erreur] skill '%s' introuvable", name) + } + case "run_shell": + cmd, _ := args["command"].(string) + to := 0 + switch v := args["timeout"].(type) { + case float64: + to = int(v) + case int: + to = v + } + label = cmd + if len(label) > 80 { + label = label[:80] + "…" + } + result = runShell(cmd, to) + default: + result = "[erreur] outil inconnu: " + tc.Function.Name + } + cb(StreamEvent{ToolUsed: &ToolUsedEvent{Name: tc.Function.Name, Label: label}}) + messages = append(messages, Message{Role: "tool", ToolCallID: tc.ID, Content: result}) + } + continue + } + return nil + } + cb(StreamEvent{Content: "\n\n[stop: trop d'appels d'outils]"}) + return nil +} + +// healthCheck pings llama.cpp's /health endpoint. +func healthCheck() bool { + resp, err := http.Get(fmt.Sprintf("http://localhost:%d/health", LLMPort())) + if err != nil { + return false + } + defer resp.Body.Close() + io.Copy(io.Discard, resp.Body) + return resp.StatusCode == 200 +} diff --git a/main.go b/main.go new file mode 100644 index 0000000..1a66a42 --- /dev/null +++ b/main.go @@ -0,0 +1,195 @@ +// jean — single-binary LLM server manager + web UI for llama.cpp deployments. +package main + +import ( + "fmt" + "os" + "path/filepath" + "strings" +) + +const Version = "0.1.0" + +func main() { + args := os.Args[1:] + cmd := "help" + if len(args) > 0 { + cmd = args[0] + args = args[1:] + } + switch cmd { + case "start", "stop", "restart", "status", "enable", "disable": + mustExit(serviceAction(cmd)) + case "logs": + mustExit(serviceLogs()) + case "edit": + mustExit(editConfig()) + case "set-api-key": + mustExit(cmdSetAPIKey(args)) + case "vram": + mustExit(showVram()) + case "switch": + mustExit(cmdSwitch(args)) + case "chat": + mustExit(cmdChat(args)) + case "web": + mustExit(cmdWeb(args)) + case "skills": + mustExit(cmdSkills(args)) + case "tools": + mustExit(cmdTools(args)) + case "serve": + mustExit(cmdServe(args)) + case "test": + mustExit(cmdTest(args)) + case "bench": + mustExit(cmdBench(args)) + case "llamacpp", "llama": + mustExit(cmdLlamacpp(args)) + case "install": + mustExit(cmdInstall(args)) + case "uninstall": + mustExit(cmdUninstall(args)) + case "version", "-v", "--version": + fmt.Println("jean", Version) + case "help", "-h", "--help", "": + printHelp() + default: + fmt.Fprintf(os.Stderr, "commande inconnue: %s\n\n", cmd) + printHelp() + os.Exit(2) + } +} + +func printHelp() { + fmt.Printf(`jean %s — manager llama.cpp + UI web (single binary) + +Usage: jean [args] + +Service: + start | stop | restart gérer le service systemd + status | logs état / logs en direct + enable | disable auto-démarrage au boot + edit éditer $JEAN_HOME/config.env + set-api-key [clé] protéger l'API (clé Bearer); vide = générer, "" = retirer + vram utilisation GPU/VRAM (nvidia-smi) + test vérifie que l'IA répond (health + completion) + bench [N] mesure prefill + decode tok/s (prompt 2000 tok, N=200 par défaut) + +Presets: + switch [N] choisir un preset dans configs/ (interactif ou par numéro) + +Interaction: + chat [system-prompt] chat terminal streamé + web [PORT] UI web (défaut :8090) — chat + presets + skills + tools + +LLM-side outils: + skills [on|off|list] active la lecture de SKILLS//SKILL.md par l'IA + tools [on|off|status] active run_shell (l'IA exécute des commandes bash) + +Backend llama.cpp : + llamacpp install clone + compile llama.cpp (détecte CUDA/ROCm/Metal/CPU), pointe BIN dessus + llamacpp update git pull + recompile le backend existant (arrête/redémarre le service) + llamacpp status commit courant, backend détecté, retard sur origin + +Entrypoint (utilisé par jean.service) : + serve lit config.env et exec le binaire llama-server + +Installation: + install installer (systemd unit, sudoers, dossiers) + uninstall désinstaller + +Env: + JEAN_HOME racine (défaut: $HOME/JEAN, ou /home//JEAN si root) + EDITOR éditeur pour 'jean edit' (défaut: nano) + +Config: $JEAN_HOME/config.env +`, Version) +} + +func mustExit(err error) { + if err != nil { + fmt.Fprintln(os.Stderr, "[err]", err) + os.Exit(1) + } +} + +// DefaultJeanHome is where everything lives unless overridden. +const DefaultJeanHome = "/etc/jean" + +// JeanHome resolves the JEAN data directory. +// Precedence: $JEAN_HOME → /etc/default/jean → DefaultJeanHome. +func JeanHome() string { + if h := os.Getenv("JEAN_HOME"); h != "" { + return h + } + if h := readEtcDefault(); h != "" { + return h + } + return DefaultJeanHome +} + +// readEtcDefault parses /etc/default/jean for JEAN_HOME=. Quiet on errors. +func readEtcDefault() string { + b, err := os.ReadFile("/etc/default/jean") + if err != nil { + return "" + } + for _, line := range strings.Split(string(b), "\n") { + s := strings.TrimSpace(line) + if s == "" || strings.HasPrefix(s, "#") { + continue + } + s = strings.TrimPrefix(s, "export ") + if eq := strings.IndexByte(s, '='); eq > 0 { + k := strings.TrimSpace(s[:eq]) + v := strings.Trim(strings.TrimSpace(s[eq+1:]), "\"'") + if k == "JEAN_HOME" { + return v + } + } + } + return "" +} + +func confPath() string { return filepath.Join(JeanHome(), "config.env") } +func presetsDir() string { return filepath.Join(JeanHome(), "configs") } +func skillsDir() string { return filepath.Join(JeanHome(), "SKILLS") } +func skillsFlag() string { return filepath.Join(skillsDir(), ".enabled") } +func toolsFlag() string { return filepath.Join(JeanHome(), ".tools_enabled") } +func apiKeyPath() string { return filepath.Join(JeanHome(), ".api_key") } +func serviceName() string { + if n := os.Getenv("JEAN_SERVICE"); n != "" { + return n + } + return "jean" +} + +// Color helpers (ANSI). Disabled when stdout is not a TTY. +var colorOn = isTerminal() + +func col(code, s string) string { + if !colorOn { + return s + } + return "\033[" + code + "m" + s + "\033[0m" +} +func bold(s string) string { return col("1", s) } +func cyan(s string) string { return col("1;36", s) } +func green(s string) string { return col("32", s) } +func red(s string) string { return col("31", s) } +func dim(s string) string { return col("2", s) } +func yellow(s string) string { return col("33", s) } +func magenta(s string) string { return col("35", s) } + +// trimSplit splits and drops empty tokens. +func trimSplit(s, sep string) []string { + out := []string{} + for _, p := range strings.Split(s, sep) { + p = strings.TrimSpace(p) + if p != "" { + out = append(out, p) + } + } + return out +} diff --git a/presets.go b/presets.go new file mode 100644 index 0000000..1e57588 --- /dev/null +++ b/presets.go @@ -0,0 +1,180 @@ +package main + +import ( + "bufio" + "crypto/sha1" + "encoding/hex" + "fmt" + "os" + "path/filepath" + "regexp" + "sort" + "strconv" + "strings" + "time" +) + +type Preset struct { + Name string + Path string + Active bool +} + +// ListPresets returns all configs/*.env files sorted by name, marking the one +// whose contents match the current config.env (by SHA-1). +func ListPresets() ([]Preset, error) { + dir := presetsDir() + _ = os.MkdirAll(dir, 0o755) + cur := "" + if b, err := os.ReadFile(confPath()); err == nil { + h := sha1.Sum(b) + cur = hex.EncodeToString(h[:]) + } + entries, err := os.ReadDir(dir) + if err != nil { + return nil, err + } + out := []Preset{} + for _, e := range entries { + if e.IsDir() || !strings.HasSuffix(e.Name(), ".env") { + continue + } + p := filepath.Join(dir, e.Name()) + b, err := os.ReadFile(p) + if err != nil { + continue + } + h := sha1.Sum(b) + out = append(out, Preset{ + Name: strings.TrimSuffix(e.Name(), ".env"), + Path: p, + Active: hex.EncodeToString(h[:]) == cur, + }) + } + sort.Slice(out, func(i, j int) bool { return out[i].Name < out[j].Name }) + return out, nil +} + +var presetNameRe = regexp.MustCompile(`^[A-Za-z0-9._-]+$`) + +// safePresetPath validates name and returns its resolved path inside presetsDir. +func safePresetPath(name string) (string, error) { + if !presetNameRe.MatchString(name) { + return "", fmt.Errorf("nom invalide (alphanum, ._-)") + } + root, err := filepath.Abs(presetsDir()) + if err != nil { + return "", err + } + p := filepath.Join(root, name+".env") + abs, err := filepath.Abs(p) + if err != nil { + return "", err + } + if !strings.HasPrefix(abs, root+string(filepath.Separator)) { + return "", fmt.Errorf("path invalide") + } + return abs, nil +} + +// SwitchToPreset backs up the current config and copies the target into place, +// then restarts the service. +func SwitchToPreset(target string) error { + src, err := os.ReadFile(target) + if err != nil { + return err + } + ts := time.Now().Format("20060102-150405") + if cur, err := os.ReadFile(confPath()); err == nil { + _ = os.WriteFile(confPath()+".bak.switch-"+ts, cur, 0o644) + } + if err := os.WriteFile(confPath(), src, 0o644); err != nil { + return err + } + fmt.Printf("%s config.env <- %s (backup .bak.switch-%s)\n", green("[ok]"), filepath.Base(target), ts) + fmt.Println(dim("[info] redémarrage du service...")) + return serviceAction("restart") +} + +func cmdSwitch(args []string) error { + list, err := ListPresets() + if err != nil { + return err + } + if len(list) == 0 { + return fmt.Errorf("aucun preset dans %s", presetsDir()) + } + fmt.Printf("\n %s (%s)\n\n", cyan("Presets disponibles"), presetsDir()) + for i, p := range list { + mark := " " + if p.Active { + mark = green("●") + " actif" + } + fmt.Printf(" %2d) %-30s %s\n", i+1, p.Name, mark) + } + fmt.Println() + choice := "" + if len(args) > 0 { + choice = args[0] + } else { + fmt.Print("Numéro à activer (vide = annuler) : ") + sc := bufio.NewScanner(os.Stdin) + if sc.Scan() { + choice = strings.TrimSpace(sc.Text()) + } + } + if choice == "" { + fmt.Println(dim("[info] annulé")) + return nil + } + n, err := strconv.Atoi(choice) + if err != nil || n < 1 || n > len(list) { + return fmt.Errorf("choix invalide") + } + return SwitchToPreset(list[n-1].Path) +} + +// SavePreset creates or overwrites a preset, handling rename when old != "". +func SavePreset(name, old, content string) error { + if old != "" && old != name { + if of, err := safePresetPath(old); err == nil { + _ = os.Remove(of) + } + } + p, err := safePresetPath(name) + if err != nil { + return err + } + _ = os.MkdirAll(filepath.Dir(p), 0o755) + return os.WriteFile(p, []byte(content), 0o644) +} + +// DeletePreset removes a preset; refuses if it is the active config. +func DeletePreset(name string) error { + p, err := safePresetPath(name) + if err != nil { + return err + } + cur, _ := os.ReadFile(confPath()) + target, err := os.ReadFile(p) + if err != nil { + return fmt.Errorf("introuvable") + } + if sha1.Sum(cur) == sha1.Sum(target) { + return fmt.Errorf("preset actif, switche d'abord") + } + return os.Remove(p) +} + +// ReadPreset returns the contents of a preset by name. +func ReadPreset(name string) (string, error) { + p, err := safePresetPath(name) + if err != nil { + return "", err + } + b, err := os.ReadFile(p) + if err != nil { + return "", err + } + return string(b), nil +} diff --git a/serve.go b/serve.go new file mode 100644 index 0000000..ed6e9b9 --- /dev/null +++ b/serve.go @@ -0,0 +1,85 @@ +package main + +import ( + "fmt" + "os" + "path/filepath" + "syscall" +) + +// cmdServe replaces the historic start.sh: read config.env, build the +// llama-server invocation, and exec it (replacing this process so systemd +// supervises llama-server directly). +func cmdServe(args []string) error { + cfg := ReadConfig() + bin := cfg["BIN"] + if bin == "" { + return fmt.Errorf("BIN non défini dans %s", confPath()) + } + model := cfg["MODEL"] + if model == "" { + return fmt.Errorf("MODEL non défini dans %s", confPath()) + } + + // Resolve LD_LIBRARY_PATH the same way start.sh did so that custom llama.cpp + // builds with bundled shared libs continue to load their .so neighbours. + ld := filepath.Dir(bin) + if existing := os.Getenv("LD_LIBRARY_PATH"); existing != "" { + ld = ld + ":" + existing + } + _ = os.Setenv("LD_LIBRARY_PATH", ld) + + get := func(key, fallback string) string { + if v, ok := cfg[key]; ok && v != "" { + return v + } + return fallback + } + kv := get("KV_TYPE", "") + ktv := get("KV_TYPE_K", kv) + vtv := get("KV_TYPE_V", kv) + + llmArgs := []string{bin, + "-m", model, + "-ngl", get("NGL", "999"), + "-c", get("CTX", "32768"), + "-t", get("THREADS", "0"), + "-tb", get("THREADS_BATCH", "0"), + "-b", get("BATCH", "2048"), + "-ub", get("UBATCH", "512"), + "--host", get("HOST", "0.0.0.0"), + "--port", get("PORT", "8080"), + } + if ktv != "" { + llmArgs = append(llmArgs, "-ctk", ktv) + } + if vtv != "" { + llmArgs = append(llmArgs, "-ctv", vtv) + } + if r := cfg["REASONING"]; r != "" { + llmArgs = append(llmArgs, "--reasoning", r, "--reasoning-budget", "0") + } + // API_KEY protège le serveur quand il est exposé sur internet : llama-server + // exige alors l'en-tête "Authorization: Bearer ". La clé est lue depuis + // $JEAN_HOME/.api_key en priorité (elle survit ainsi aux changements de preset + // qui réécrivent config.env), avec config.env comme repli rétro-compatible. + if k := readAPIKey(); k != "" { + llmArgs = append(llmArgs, "--api-key", k) + } else if k := cfg["API_KEY"]; k != "" { + llmArgs = append(llmArgs, "--api-key", k) + } + // EXTRA_ARGS is appended verbatim, split on whitespace like the shell would. + for _, a := range trimSplit(cfg["EXTRA_ARGS"], " ") { + llmArgs = append(llmArgs, a) + } + + // Working dir = JEAN_HOME so relative paths in EXTRA_ARGS (e.g. --mmproj + // mmproj-F16.gguf) still resolve. + _ = os.Chdir(JeanHome()) + + fmt.Fprintf(os.Stderr, "[jean serve] %s model=%s port=%s\n", + bin, filepath.Base(model), get("PORT", "8080")) + + // exec replaces this process — same as start.sh's `exec`. + return syscall.Exec(bin, llmArgs, os.Environ()) +} diff --git a/service.go b/service.go new file mode 100644 index 0000000..1916aa9 --- /dev/null +++ b/service.go @@ -0,0 +1,115 @@ +package main + +import ( + "fmt" + "os" + "os/exec" + "strconv" + "strings" + "time" +) + +// serviceAction wraps `systemctl ` with passwordless sudo where +// it makes sense, and prints a follow-up status check after start/restart. +func serviceAction(action string) error { + svc := serviceName() + needsRoot := action == "start" || action == "stop" || action == "restart" || action == "enable" || action == "disable" + args := []string{} + bin := "systemctl" + if needsRoot && os.Geteuid() != 0 { + bin = "sudo" + args = append(args, "-n", "systemctl") + } + args = append(args, action, svc) + cmd := exec.Command(bin, args...) + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + if err := cmd.Run(); err != nil { + return err + } + switch action { + case "start", "restart": + return checkStarted(svc) + case "stop": + fmt.Println(green("[ok]") + " arrêté") + case "enable": + fmt.Println(green("[ok]") + " démarrage auto activé") + case "disable": + fmt.Println(green("[ok]") + " démarrage auto désactivé") + } + return nil +} + +func checkStarted(svc string) error { + time.Sleep(2 * time.Second) + out, _ := exec.Command("systemctl", "is-active", svc).Output() + state := strings.TrimSpace(string(out)) + if state == "active" || state == "activating" { + fmt.Printf("%s %s: %s\n", green("[ok]"), svc, state) + return nil + } + fmt.Printf("%s %s: %s — derniers logs :\n", red("[ERREUR]"), svc, state) + fmt.Println("------------------------------------------------") + logs, _ := exec.Command("journalctl", "-u", svc, "-n", "20", "--no-pager").Output() + fmt.Print(string(logs)) + fmt.Println("------------------------------------------------") + fmt.Printf("→ jean logs pour plus de détails\n→ jean edit pour corriger config.env\n") + return fmt.Errorf("service %s non démarré", svc) +} + +func serviceLogs() error { + cmd := exec.Command("journalctl", "-u", serviceName(), "-n", "80", "-f") + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + return cmd.Run() +} + +func editConfig() error { + editor := os.Getenv("EDITOR") + if editor == "" { + editor = "nano" + } + cmd := exec.Command(editor, confPath()) + cmd.Stdin = os.Stdin + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + if err := cmd.Run(); err != nil { + return err + } + fmt.Println(dim("[info] jean restart pour appliquer")) + return nil +} + +// showVram parses `nvidia-smi --query-gpu=...` and renders a colored bar. +func showVram() error { + out, err := exec.Command("nvidia-smi", + "--query-gpu=name,memory.used,memory.total,utilization.gpu,temperature.gpu", + "--format=csv,noheader,nounits").Output() + if err != nil { + return fmt.Errorf("nvidia-smi indisponible: %w", err) + } + for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") { + parts := strings.Split(line, ",") + if len(parts) != 5 { + continue + } + for i := range parts { + parts[i] = strings.TrimSpace(parts[i]) + } + name := parts[0] + used, _ := strconv.Atoi(parts[1]) + total, _ := strconv.Atoi(parts[2]) + util, _ := strconv.Atoi(parts[3]) + temp, _ := strconv.Atoi(parts[4]) + pct := 0 + if total > 0 { + pct = used * 100 / total + } + full := pct / 5 + bar := strings.Repeat("█", full) + strings.Repeat("░", 20-full) + fmt.Printf("\n %s\n", cyan(name)) + fmt.Printf(" VRAM %s %3d%% %.1f / %.1f GiB\n", green(bar), pct, float64(used)/1024, float64(total)/1024) + fmt.Printf(" GPU %3d%% Temp %d°C\n\n", util, temp) + } + return nil +} diff --git a/skills.go b/skills.go new file mode 100644 index 0000000..be4fd11 --- /dev/null +++ b/skills.go @@ -0,0 +1,186 @@ +package main + +import ( + "fmt" + "os" + "path/filepath" + "regexp" + "sort" + "strings" +) + +type Skill struct { + Name string + Desc string // first non-empty line, with leading #'s stripped + Content string +} + +func skillsEnabled() bool { + _, err := os.Stat(skillsFlag()) + return err == nil +} + +func setSkillsEnabled(on bool) error { + _ = os.MkdirAll(skillsDir(), 0o755) + if on { + f, err := os.Create(skillsFlag()) + if err != nil { + return err + } + return f.Close() + } + if err := os.Remove(skillsFlag()); err != nil && !os.IsNotExist(err) { + return err + } + return nil +} + +// ListSkills walks SKILLS//SKILL.md, returning name+desc+full content. +func ListSkills() []Skill { + entries, err := os.ReadDir(skillsDir()) + if err != nil { + return nil + } + out := []Skill{} + for _, e := range entries { + if !e.IsDir() || strings.HasPrefix(e.Name(), ".") { + continue + } + f := filepath.Join(skillsDir(), e.Name(), "SKILL.md") + b, err := os.ReadFile(f) + if err != nil { + continue + } + content := string(b) + desc := "" + for _, line := range strings.Split(content, "\n") { + s := strings.TrimSpace(strings.TrimLeft(strings.TrimSpace(line), "#")) + if s != "" { + desc = s + break + } + } + out = append(out, Skill{Name: e.Name(), Desc: desc, Content: content}) + } + sort.Slice(out, func(i, j int) bool { return out[i].Name < out[j].Name }) + return out +} + +var skillNameRe = regexp.MustCompile(`^[A-Za-z0-9._-]+$`) + +// safeSkillDir validates and returns the on-disk path for a skill directory. +func safeSkillDir(name string) (string, error) { + if !skillNameRe.MatchString(name) { + return "", fmt.Errorf("nom invalide (alphanum, ._-)") + } + root, err := filepath.Abs(skillsDir()) + if err != nil { + return "", err + } + p := filepath.Join(root, name) + abs, err := filepath.Abs(p) + if err != nil { + return "", err + } + if !strings.HasPrefix(abs, root+string(filepath.Separator)) { + return "", fmt.Errorf("path invalide") + } + return abs, nil +} + +// SkillContent loads SKILL.md for the given name (used by the read_skill tool). +func SkillContent(name string) string { + d, err := safeSkillDir(name) + if err != nil { + return "" + } + b, err := os.ReadFile(filepath.Join(d, "SKILL.md")) + if err != nil { + return "" + } + return string(b) +} + +func SaveSkill(name, old, content string) error { + if old != "" && old != name { + if od, err := safeSkillDir(old); err == nil { + _ = os.RemoveAll(od) + } + } + d, err := safeSkillDir(name) + if err != nil { + return err + } + if err := os.MkdirAll(d, 0o755); err != nil { + return err + } + return os.WriteFile(filepath.Join(d, "SKILL.md"), []byte(content), 0o644) +} + +func DeleteSkill(name string) error { + d, err := safeSkillDir(name) + if err != nil { + return err + } + if _, err := os.Stat(d); err != nil { + return fmt.Errorf("introuvable") + } + return os.RemoveAll(d) +} + +// skillsSystemPrompt returns the lightweight skills directory message to +// prepend to the conversation when skills are enabled. +func skillsSystemPrompt() string { + if !skillsEnabled() { + return "" + } + list := ListSkills() + if len(list) == 0 { + return "" + } + var b strings.Builder + b.WriteString(`Tu as accès à des skills (guides spécialisés). Pour lire le contenu détaillé d'un skill quand c'est pertinent, appelle la fonction read_skill(name=""). N'appelle pas read_skill si la question est sans rapport avec un skill listé. + +Skills disponibles : +`) + for _, s := range list { + fmt.Fprintf(&b, "- %s: %s\n", s.Name, s.Desc) + } + return strings.TrimRight(b.String(), "\n") +} + +func cmdSkills(args []string) error { + sub := "" + if len(args) > 0 { + sub = args[0] + } + switch sub { + case "on": + if err := setSkillsEnabled(true); err != nil { + return err + } + fmt.Println(green("[ok]") + " skills activés") + case "off": + if err := setSkillsEnabled(false); err != nil { + return err + } + fmt.Println(green("[ok]") + " skills désactivés") + case "", "list": + state := dim("off") + if skillsEnabled() { + state = green("on") + } + fmt.Printf("%s (%s) état: %s\n", cyan("Skills"), skillsDir(), state) + sk := ListSkills() + if len(sk) == 0 { + fmt.Printf(" (aucun — crée %s//SKILL.md)\n", skillsDir()) + return nil + } + for _, s := range sk { + fmt.Printf(" %s %s\n", bold(s.Name), s.Desc) + } + default: + return fmt.Errorf("usage: jean skills [on|off|list]") + } + return nil +} diff --git a/test_cmd.go b/test_cmd.go new file mode 100644 index 0000000..1097b8e --- /dev/null +++ b/test_cmd.go @@ -0,0 +1,65 @@ +package main + +import ( + "fmt" + "strings" + "time" +) + +// cmdTest sanity-checks the LLM end-to-end: HTTP /health, then a minimal chat +// completion to confirm the model actually generates tokens. +func cmdTest(args []string) error { + port := LLMPort() + fmt.Printf("→ GET http://localhost:%d/health … ", port) + if !healthCheck() { + fmt.Println(red("ko")) + return fmt.Errorf("/health ne répond pas — jean start d'abord") + } + fmt.Println(green("ok")) + + fmt.Printf("→ chat completion (prompt « ping ») … ") + msgs := []Message{{Role: "user", Content: "Réponds juste « pong ». Rien d'autre."}} + var reply strings.Builder + t0 := time.Now() + var firstTok time.Time + tokens := 0 + err := runChat(msgs, 0, func(ev StreamEvent) bool { + if ev.Content != "" { + if firstTok.IsZero() { + firstTok = time.Now() + } + tokens++ + reply.WriteString(ev.Content) + } + return true + }) + if err != nil { + fmt.Println(red("ko")) + return err + } + elapsed := time.Since(t0) + ttft := time.Duration(0) + if !firstTok.IsZero() { + ttft = firstTok.Sub(t0) + } + fmt.Println(green("ok")) + + out := strings.TrimSpace(reply.String()) + if len(out) > 120 { + out = out[:120] + "…" + } + fmt.Printf("\n %s %s\n", cyan("réponse :"), out) + fmt.Printf(" %s %s\n", cyan("ttft :"), ttft.Round(time.Millisecond)) + fmt.Printf(" %s %s (%d tokens)\n", cyan("total :"), elapsed.Round(time.Millisecond), tokens) + if elapsed > 0 { + tps := float64(tokens) / (elapsed - ttft).Seconds() + if tokens > 0 && elapsed > ttft { + fmt.Printf(" %s %.1f tok/s\n", cyan("decode :"), tps) + } + } + if tokens == 0 { + return fmt.Errorf("aucun token généré") + } + fmt.Println("\n" + green("[ok]") + " l'IA répond") + return nil +} diff --git a/tools.go b/tools.go new file mode 100644 index 0000000..967dc1e --- /dev/null +++ b/tools.go @@ -0,0 +1,114 @@ +package main + +import ( + "context" + "fmt" + "os" + "os/exec" + "strings" + "time" +) + +const ( + toolDefaultTimeout = 30 + toolMaxTimeout = 300 + toolMaxOutput = 8000 // characters of stdout/stderr returned to the model +) + +func toolsEnabled() bool { + _, err := os.Stat(toolsFlag()) + return err == nil +} + +func setToolsEnabled(on bool) error { + _ = os.MkdirAll(JeanHome(), 0o755) + if on { + f, err := os.Create(toolsFlag()) + if err != nil { + return err + } + return f.Close() + } + if err := os.Remove(toolsFlag()); err != nil && !os.IsNotExist(err) { + return err + } + return nil +} + +// runShell executes a command via `bash -c` with a clamped timeout, returning +// a single string formatted "exit: N\n\nstdout:\n...\n\nstderr:\n..." truncated +// to keep tool output bounded. +func runShell(command string, timeoutSec int) string { + if timeoutSec <= 0 { + timeoutSec = toolDefaultTimeout + } + if timeoutSec > toolMaxTimeout { + timeoutSec = toolMaxTimeout + } + ctx, cancel := context.WithTimeout(context.Background(), time.Duration(timeoutSec)*time.Second) + defer cancel() + cmd := exec.CommandContext(ctx, "/bin/bash", "-c", command) + var stdout, stderr strings.Builder + cmd.Stdout = &stdout + cmd.Stderr = &stderr + err := cmd.Run() + if ctx.Err() == context.DeadlineExceeded { + return fmt.Sprintf("[timeout après %ds]", timeoutSec) + } + exit := 0 + if err != nil { + if ee, ok := err.(*exec.ExitError); ok { + exit = ee.ExitCode() + } else { + return fmt.Sprintf("[erreur: %v]", err) + } + } + out := tailRunes(stdout.String(), toolMaxOutput) + errOut := tailRunes(stderr.String(), toolMaxOutput) + parts := []string{fmt.Sprintf("exit: %d", exit)} + if out != "" { + parts = append(parts, "stdout:\n"+out) + } + if errOut != "" { + parts = append(parts, "stderr:\n"+errOut) + } + return strings.Join(parts, "\n\n") +} + +// tailRunes returns the last n runes of s (used to cap tool output). +func tailRunes(s string, n int) string { + r := []rune(s) + if len(r) <= n { + return s + } + return string(r[len(r)-n:]) +} + +func cmdTools(args []string) error { + sub := "" + if len(args) > 0 { + sub = args[0] + } + switch sub { + case "on": + if err := setToolsEnabled(true); err != nil { + return err + } + fmt.Println(green("[ok]") + " tool run_shell activé — exécution bash possible") + case "off": + if err := setToolsEnabled(false); err != nil { + return err + } + fmt.Println(green("[ok]") + " tools désactivés") + case "", "status": + state := dim("off") + if toolsEnabled() { + state = green("on") + } + fmt.Printf("%s run_shell état: %s\n", cyan("Tools"), state) + fmt.Printf(" timeout défaut: %ds, max: %ds\n", toolDefaultTimeout, toolMaxTimeout) + default: + return fmt.Errorf("usage: jean tools [on|off|status]") + } + return nil +} diff --git a/tty.go b/tty.go new file mode 100644 index 0000000..c837617 --- /dev/null +++ b/tty.go @@ -0,0 +1,12 @@ +package main + +import "os" + +// isTerminal reports whether stdout is a TTY. Stdlib-only check via FileMode. +func isTerminal() bool { + fi, err := os.Stdout.Stat() + if err != nil || fi == nil { + return false + } + return fi.Mode()&os.ModeCharDevice != 0 +} diff --git a/web.go b/web.go new file mode 100644 index 0000000..06ddf95 --- /dev/null +++ b/web.go @@ -0,0 +1,440 @@ +package main + +import ( + "embed" + "encoding/json" + "fmt" + "net/http" + "os" + "os/exec" + "strconv" + "strings" +) + +//go:embed ui/index.html ui/marked.min.js +var uiFS embed.FS + +// cmdWeb starts the HTTP server on the given port (default 8090). +func cmdWeb(args []string) error { + port := 8090 + if len(args) > 0 && args[0] != "" { + n, err := strconv.Atoi(args[0]) + if err != nil { + return fmt.Errorf("port invalide: %s", args[0]) + } + port = n + } + mux := http.NewServeMux() + mux.HandleFunc("/", handleIndex) + mux.HandleFunc("/marked.min.js", func(w http.ResponseWriter, r *http.Request) { + b, _ := uiFS.ReadFile("ui/marked.min.js") + w.Header().Set("Content-Type", "application/javascript") + w.Header().Set("Cache-Control", "public, max-age=86400") + w.Write(b) + }) + mux.HandleFunc("/api/status", handleStatus) + mux.HandleFunc("/api/vram", handleVram) + mux.HandleFunc("/api/config", handleConfigEnv) + mux.HandleFunc("/api/models", handleModels) + mux.HandleFunc("/api/backends", handleBackends) + mux.HandleFunc("/api/presets", handlePresets) + mux.HandleFunc("/api/preset", handlePreset) + mux.HandleFunc("/api/preset/save", handlePresetSave) + mux.HandleFunc("/api/preset/delete", handlePresetDelete) + mux.HandleFunc("/api/skills", handleSkills) + mux.HandleFunc("/api/skills/toggle", handleSkillsToggle) + mux.HandleFunc("/api/skill", handleSkill) + mux.HandleFunc("/api/skill/save", handleSkillSave) + mux.HandleFunc("/api/skill/delete", handleSkillDelete) + mux.HandleFunc("/api/tools", handleTools) + mux.HandleFunc("/api/tools/toggle", handleToolsToggle) + mux.HandleFunc("/api/switch", handleSwitch) + mux.HandleFunc("/api/start", svcHandler("start")) + mux.HandleFunc("/api/stop", svcHandler("stop")) + mux.HandleFunc("/api/restart", svcHandler("restart")) + mux.HandleFunc("/api/bench", handleBench) + mux.HandleFunc("/api/chat", handleChat) + addr := fmt.Sprintf("0.0.0.0:%d", port) + fmt.Printf("[jean web] http://%s (Ctrl-C pour arrêter)\n", addr) + return http.ListenAndServe(addr, mux) +} + +func handleIndex(w http.ResponseWriter, r *http.Request) { + if r.URL.Path != "/" && r.URL.Path != "/index.html" { + http.NotFound(w, r) + return + } + b, err := uiFS.ReadFile("ui/index.html") + if err != nil { + http.Error(w, err.Error(), 500) + return + } + w.Header().Set("Content-Type", "text/html; charset=utf-8") + w.Header().Set("Cache-Control", "no-store, max-age=0") + w.Write(b) +} + +func sendJSON(w http.ResponseWriter, code int, v any) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(code) + json.NewEncoder(w).Encode(v) +} + +func handleStatus(w http.ResponseWriter, r *http.Request) { + out, _ := exec.Command("systemctl", "is-active", serviceName()).Output() + state := strings.TrimSpace(string(out)) + if state == "" { + state = "unknown" + } + active := state == "active" + health := false + if active { + health = healthCheck() + } + sendJSON(w, 200, map[string]any{ + "state": state, + "active": active, + "health": health, + "port": LLMPort(), + }) +} + +func handleVram(w http.ResponseWriter, r *http.Request) { + out, err := exec.Command("nvidia-smi", + "--query-gpu=name,memory.used,memory.total,utilization.gpu,temperature.gpu", + "--format=csv,noheader,nounits").Output() + gpus := []map[string]any{} + if err == nil { + for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") { + parts := strings.Split(line, ",") + if len(parts) != 5 { + continue + } + for i := range parts { + parts[i] = strings.TrimSpace(parts[i]) + } + used, _ := strconv.Atoi(parts[1]) + total, _ := strconv.Atoi(parts[2]) + util, _ := strconv.Atoi(parts[3]) + temp, _ := strconv.Atoi(parts[4]) + gpus = append(gpus, map[string]any{ + "name": parts[0], "used": used, "total": total, "util": util, "temp": temp, + }) + } + } + sendJSON(w, 200, gpus) +} + +func handleConfigEnv(w http.ResponseWriter, r *http.Request) { + sendJSON(w, 200, ReadConfig()) +} + +// handleBackends scans JEAN_HOME/backends// for a llama-server binary, +// trying common build subpaths (build/bin, build-sm120/bin, bin, .). +// Returns [{name, path}]. +func handleBackends(w http.ResponseWriter, r *http.Request) { + root := JeanHome() + "/backends" + entries, err := os.ReadDir(root) + if err != nil { + sendJSON(w, 200, []map[string]any{}) + return + } + subpaths := []string{ + "build/bin/llama-server", "build-sm120/bin/llama-server", + "build/llama-server", "bin/llama-server", "llama-server", + } + out := []map[string]any{} + for _, e := range entries { + // e can be a directory or a symlink to one; either is fine. + name := e.Name() + if strings.HasPrefix(name, ".") { + continue + } + for _, sp := range subpaths { + p := root + "/" + name + "/" + sp + if fi, err := os.Stat(p); err == nil && !fi.IsDir() { + out = append(out, map[string]any{"name": name, "path": p}) + break + } + } + } + sendJSON(w, 200, out) +} + +// handleModels lists *.gguf files in JEAN_HOME (size in bytes) for the preset +// editor's model picker. +func handleModels(w http.ResponseWriter, r *http.Request) { + entries, err := os.ReadDir(JeanHome()) + if err != nil { + sendJSON(w, 200, []map[string]any{}) + return + } + out := []map[string]any{} + for _, e := range entries { + if e.IsDir() || !strings.HasSuffix(strings.ToLower(e.Name()), ".gguf") { + continue + } + info, _ := e.Info() + size := int64(0) + if info != nil { + size = info.Size() + } + out = append(out, map[string]any{"name": e.Name(), "size": size}) + } + sendJSON(w, 200, out) +} + +func handlePresets(w http.ResponseWriter, r *http.Request) { + list, err := ListPresets() + if err != nil { + sendJSON(w, 500, map[string]any{"error": err.Error()}) + return + } + out := []map[string]any{} + for _, p := range list { + out = append(out, map[string]any{"name": p.Name, "active": p.Active}) + } + sendJSON(w, 200, out) +} + +func handlePreset(w http.ResponseWriter, r *http.Request) { + name := strings.TrimSpace(r.URL.Query().Get("name")) + if name == "" { + // new preset → seed from current config.env so users can tweak rather than start blank + b, _ := os.ReadFile(confPath()) + sendJSON(w, 200, map[string]any{"name": "", "content": string(b)}) + return + } + content, err := ReadPreset(name) + if err != nil { + sendJSON(w, 404, map[string]any{"error": "not found"}) + return + } + sendJSON(w, 200, map[string]any{"name": name, "content": content}) +} + +type saveReq struct { + Name string `json:"name"` + Old string `json:"old"` + Content string `json:"content"` +} + +func handlePresetSave(w http.ResponseWriter, r *http.Request) { + var req saveReq + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + if err := SavePreset(req.Name, req.Old, req.Content); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "name": req.Name}) +} + +func handlePresetDelete(w http.ResponseWriter, r *http.Request) { + var req saveReq + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + if err := DeletePreset(req.Name); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true}) +} + +func handleSkills(w http.ResponseWriter, r *http.Request) { + sk := ListSkills() + out := []map[string]any{} + for _, s := range sk { + out = append(out, map[string]any{"name": s.Name, "desc": s.Desc}) + } + sendJSON(w, 200, map[string]any{"enabled": skillsEnabled(), "skills": out}) +} + +func handleSkillsToggle(w http.ResponseWriter, r *http.Request) { + var req struct { + On bool `json:"on"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + if err := setSkillsEnabled(req.On); err != nil { + sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "enabled": skillsEnabled()}) +} + +func handleSkill(w http.ResponseWriter, r *http.Request) { + name := strings.TrimSpace(r.URL.Query().Get("name")) + if name == "" { + sendJSON(w, 200, map[string]any{"name": "", "content": "# nouveau skill\n\nDécris ici quand et comment l'IA doit utiliser ce skill.\n"}) + return + } + c := SkillContent(name) + if c == "" { + sendJSON(w, 404, map[string]any{"error": "not found"}) + return + } + sendJSON(w, 200, map[string]any{"name": name, "content": c}) +} + +func handleSkillSave(w http.ResponseWriter, r *http.Request) { + var req saveReq + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + if err := SaveSkill(req.Name, req.Old, req.Content); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "name": req.Name}) +} + +func handleSkillDelete(w http.ResponseWriter, r *http.Request) { + var req saveReq + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + if err := DeleteSkill(req.Name); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true}) +} + +func handleTools(w http.ResponseWriter, r *http.Request) { + sendJSON(w, 200, map[string]any{"enabled": toolsEnabled()}) +} + +func handleToolsToggle(w http.ResponseWriter, r *http.Request) { + var req struct { + On bool `json:"on"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + if err := setToolsEnabled(req.On); err != nil { + sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "enabled": toolsEnabled()}) +} + +func handleSwitch(w http.ResponseWriter, r *http.Request) { + var req struct { + N int `json:"n"` + } + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + sendJSON(w, 400, map[string]any{"ok": false, "error": err.Error()}) + return + } + list, err := ListPresets() + if err != nil { + sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()}) + return + } + if req.N < 1 || req.N > len(list) { + sendJSON(w, 400, map[string]any{"ok": false, "error": "index hors limites"}) + return + } + target := list[req.N-1] + if err := SwitchToPreset(target.Path); err != nil { + sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "preset": target.Name}) +} + +// svcHandler returns an HTTP handler that triggers a systemctl action via the +// passwordless sudo rule installed by `jean install`. Falls back to no-sudo +// when the web server itself runs as root. +func svcHandler(action string) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + var cmd *exec.Cmd + if os.Geteuid() == 0 { + cmd = exec.Command("systemctl", action, serviceName()) + } else { + cmd = exec.Command("sudo", "-n", "systemctl", action, serviceName()) + } + out, err := cmd.CombinedOutput() + sendJSON(w, 200, map[string]any{"ok": err == nil, "out": string(out)}) + } +} + +// handleChat is the SSE proxy with tool-calling. The HTTP handler writes raw +// data: lines matching what the embedded JS expects (delta.content, +// delta.reasoning_content, delta.tool_used). +// handleBench runs `runBench` synchronously. Long enough (~30-60s) that we +// rely on the client side to show a spinner / disable the button. +func handleBench(w http.ResponseWriter, r *http.Request) { + nPrompt, nPredict := 2000, 300 + if v := r.URL.Query().Get("prompt"); v != "" { + if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 { + nPrompt = parsed + } + } + if v := r.URL.Query().Get("n"); v != "" { + if parsed, err := strconv.Atoi(v); err == nil && parsed > 0 { + nPredict = parsed + } + } + res, err := runBench(nPrompt, nPredict) + if err != nil { + sendJSON(w, 500, map[string]any{"ok": false, "error": err.Error()}) + return + } + sendJSON(w, 200, map[string]any{"ok": true, "result": res}) +} + +func handleChat(w http.ResponseWriter, r *http.Request) { + var body struct { + Messages []Message `json:"messages"` + Temperature float64 `json:"temperature"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + http.Error(w, err.Error(), 400) + return + } + if body.Temperature == 0 { + body.Temperature = 0.7 + } + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("X-Accel-Buffering", "no") + flusher, _ := w.(http.Flusher) + emit := func(obj map[string]any) bool { + b, _ := json.Marshal(map[string]any{"choices": []any{map[string]any{"delta": obj}}}) + if _, err := w.Write([]byte("data: " + string(b) + "\n\n")); err != nil { + return false + } + if flusher != nil { + flusher.Flush() + } + return true + } + msgs := InjectSkills(body.Messages) + _ = runChat(msgs, body.Temperature, func(ev StreamEvent) bool { + if ev.Err != nil { + emit(map[string]any{"error": ev.Err.Error()}) + return true + } + if ev.ToolUsed != nil { + emit(map[string]any{"tool_used": map[string]any{"name": ev.ToolUsed.Name, "label": ev.ToolUsed.Label}}) + return true + } + if ev.Stats != nil { + emit(map[string]any{"stats": ev.Stats}) + return true + } + if ev.Reasoning != "" { + return emit(map[string]any{"reasoning_content": ev.Reasoning}) + } + if ev.Content != "" { + return emit(map[string]any{"content": ev.Content}) + } + return true + }) +} +