mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Moteur : port déjà occupé refusé, cache KV lent signalé
Deux reprises d'AJEAN 0.15.9, côté `loki serve`. PORT OCCUPÉ Un llama-server orphelin (un stop qui n'a pas tué, un relancement trop rapide) peut encore tenir le port quand le suivant démarre. Deux moteurs sur le même port se partagent alors les requêtes au hasard : VRAM saturée, réponses du mauvais modèle. waitPortFree tente une connexion TCP (fiable même en SO_REUSEADDR, là où un Listen de test réussirait), laisse 5 s à un moteur qu'on vient d'arrêter pour libérer, puis refuse avec un message qui dit quoi faire. argValue lit la DERNIÈRE occurrence de --port/--host, comme llama-server, puisque EXTRA_ARGS peut les surcharger. CACHE KV LENT llama.cpp CUDA, compilé avec ses options par défaut (FA_ALL_QUANTS=OFF), n'accélère en Flash-Attention que f16, bf16, q8_0 et q4_0 — et seulement à l'identique pour K et V. C'est le cas du moteur de l'image server-cuda et de ceux installés par OCI. q5_1 ou un couple mixte q8_0/q4_0 marchent mais sont reconvertis en f16 à chaque pas. L'amont ne prévient que pour son binaire précompilé ; chez nous tous les moteurs sont concernés, donc l'avertissement part toujours dans loki-engine.log. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b893d1a81f
commit
20375d0e4c
2 files changed
+107
-1
No files matched your search
@@ -4,6 +4,7 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
@@ -370,10 +371,84 @@ func cmdServe(args []string) error {
|
||||
// mmproj-F16.gguf) still resolve.
|
||||
_ = os.Chdir(LokiHome())
|
||||
|
||||
// Port déjà pris (souvent un llama-server orphelin qu'un stop n'a pas pu
|
||||
// tuer) : deux moteurs sur un même port se partagent les requêtes au hasard,
|
||||
// VRAM saturée et réponses du mauvais modèle. On refuse en clair.
|
||||
host, port := argValue(llmArgs, "--host"), argValue(llmArgs, "--port")
|
||||
if err := waitPortFree(host, port, 5*time.Second); err != nil {
|
||||
return err
|
||||
}
|
||||
warnSlowKV(ktv, vtv)
|
||||
|
||||
fmt.Fprintf(os.Stderr, "[loki serve] %s model=%s port=%s\n",
|
||||
bin, filepath.Base(model), get("PORT", "8080"))
|
||||
bin, filepath.Base(model), port)
|
||||
|
||||
// Hand off to the llama-server process. On Unix this replaces the current
|
||||
// process (exec); on Windows it runs as a child and waits. See sys_platform_*.go.
|
||||
return execServer(bin, llmArgs)
|
||||
}
|
||||
|
||||
// argValue renvoie la valeur de la DERNIÈRE occurrence de flag (llama-server
|
||||
// retient la dernière : EXTRA_ARGS peut surcharger --port).
|
||||
func argValue(args []string, flag string) string {
|
||||
v := ""
|
||||
for i := 0; i+1 < len(args); i++ {
|
||||
if args[i] == flag {
|
||||
v = args[i+1]
|
||||
}
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// waitPortFree vérifie que personne n'écoute déjà sur host:port. On tente une
|
||||
// connexion TCP (fiable même si l'occupant a bindé en SO_REUSEADDR, où un
|
||||
// Listen de test pourrait réussir). Un moteur qu'on vient d'arrêter peut tenir
|
||||
// le port quelques instants : on réessaie jusqu'à wait avant de conclure.
|
||||
// Repris d'AJEAN 0.15.9.
|
||||
func waitPortFree(host, port string, wait time.Duration) error {
|
||||
if port == "" {
|
||||
return nil
|
||||
}
|
||||
h := strings.Trim(host, "[]")
|
||||
if h == "" || h == "0.0.0.0" || h == "::" {
|
||||
h = "127.0.0.1"
|
||||
}
|
||||
addr := net.JoinHostPort(h, port)
|
||||
deadline := time.Now().Add(wait)
|
||||
for {
|
||||
c, err := net.DialTimeout("tcp", addr, 500*time.Millisecond)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
c.Close()
|
||||
if time.Now().After(deadline) {
|
||||
return fmt.Errorf("le port %s est déjà utilisé (sans doute un ancien llama-server encore actif) : "+
|
||||
"« loki stop », docker restart loki, ou change PORT", port)
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// fastKV : combinaisons K/V que llama.cpp CUDA accélère en Flash-Attention
|
||||
// avec ses options de compilation par défaut (GGML_CUDA_FA_ALL_QUANTS=OFF) —
|
||||
// c'est le cas du moteur de l'image server-cuda comme de ceux tirés par OCI.
|
||||
// Les autres (q5_*, q4_1, combinaisons mixtes) marchent mais sont reconverties
|
||||
// en f16 à chaque pas : la génération ralentit nettement (AJEAN 0.15.9).
|
||||
var fastKV = map[string]bool{"f16|f16": true, "q8_0|q8_0": true, "q4_0|q4_0": true, "bf16|bf16": true}
|
||||
|
||||
func slowKV(k, v string) bool {
|
||||
if k == "" {
|
||||
k = "f16"
|
||||
}
|
||||
if v == "" {
|
||||
v = "f16"
|
||||
}
|
||||
return !fastKV[k+"|"+v]
|
||||
}
|
||||
|
||||
func warnSlowKV(k, v string) {
|
||||
if slowKV(k, v) {
|
||||
fmt.Fprintf(os.Stderr, "[loki serve] avertissement : cache KV %s/%s non accéléré par llama.cpp CUDA "+
|
||||
"(converti en f16 à chaque pas, lent). Préfère q8_0/q8_0, q4_0/q4_0 ou f16.\n", k, v)
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,10 @@
|
||||
package loki
|
||||
|
||||
import (
|
||||
"net"
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestHasAnyFlag(t *testing.T) {
|
||||
@@ -91,3 +93,32 @@ func TestNGLArgs(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestArgValueEtPortLibre(t *testing.T) {
|
||||
if got := argValue([]string{"--port", "8080", "-c", "1", "--port", "9000"}, "--port"); got != "9000" {
|
||||
t.Fatalf("la dernière occurrence doit gagner, got %q", got)
|
||||
}
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, port, _ := net.SplitHostPort(ln.Addr().String())
|
||||
if waitPortFree("0.0.0.0", port, 600*time.Millisecond) == nil {
|
||||
t.Fatal("un port occupé doit être refusé")
|
||||
}
|
||||
ln.Close()
|
||||
if err := waitPortFree("0.0.0.0", port, time.Second); err != nil {
|
||||
t.Fatalf("un port libre doit passer : %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlowKV(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
k, v string
|
||||
lent bool
|
||||
}{{"", "", false}, {"q8_0", "q8_0", false}, {"q4_0", "q4_0", false}, {"q8_0", "q4_0", true}, {"q5_1", "q5_1", true}} {
|
||||
if slowKV(c.k, c.v) != c.lent {
|
||||
t.Errorf("%s/%s : lent attendu %v", c.k, c.v, c.lent)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user