diff --git a/internal/jean/backend_buildfilter_test.go b/internal/jean/backend_buildfilter_test.go index 90df685..f1d93b1 100644 --- a/internal/jean/backend_buildfilter_test.go +++ b/internal/jean/backend_buildfilter_test.go @@ -8,12 +8,12 @@ func TestCompiledFile(t *testing.T) { ` ggml-threading.cpp`: "ggml-threading.cpp", ` ggml-quants.c`: "ggml-quants.c", // Make (Linux) et Ninja. - `[ 45%] Building CXX object src/CMakeFiles/llama.dir/llama.cpp.o`: "llama.cpp", + `[ 45%] Building CXX object src/CMakeFiles/llama.dir/llama.cpp.o`: "llama.cpp", `[123/456] Building CUDA object ggml/src/ggml-cuda/CMakeFiles/ggml-cuda.dir/acc.cu.o`: "acc.cu", // La ligne de commande nvcc géante ne doit PAS être prise pour un fichier. ` C:\...\nvcc.exe -x cu ... -o ggml-cuda.dir\Release\acc.obj "C:\...\acc.cu"`: "", - `Building Custom Rule C:/ProgramData/jean/...`: "", - `-- UI: running npm install`: "", + `Building Custom Rule C:/ProgramData/jean/...`: "", + `-- UI: running npm install`: "", } for in, want := range cases { if got := compiledFile(in); got != want { diff --git a/internal/jean/llm_bench.go b/internal/jean/llm_bench.go index 2c2b8c0..dfe647e 100644 --- a/internal/jean/llm_bench.go +++ b/internal/jean/llm_bench.go @@ -14,13 +14,13 @@ import ( // benchResult captures the timings llama.cpp returns from /completion. type benchResult struct { - PromptN int `json:"prompt_n"` - PromptMs float64 `json:"prompt_ms"` - PromptPerSecond float64 `json:"prompt_per_second"` - PredictedN int `json:"predicted_n"` - PredictedMs float64 `json:"predicted_ms"` - PredictedPerSec float64 `json:"predicted_per_second"` - Elapsed float64 `json:"elapsed_sec"` + PromptN int `json:"prompt_n"` + PromptMs float64 `json:"prompt_ms"` + PromptPerSecond float64 `json:"prompt_per_second"` + PredictedN int `json:"predicted_n"` + PredictedMs float64 `json:"predicted_ms"` + PredictedPerSec float64 `json:"predicted_per_second"` + Elapsed float64 `json:"elapsed_sec"` } // benchCorpus is a varied passage used to defeat speculative decoding @@ -71,12 +71,12 @@ func runBench(nPrompt, nPredict int) (*benchResult, error) { // Use the same endpoint your real chat hits, so the comparison is honest // (chat template, reasoning, OpenAI-compat layer all included). payload := map[string]any{ - "model": "jean", - "messages": []Message{{Role: "user", Content: prompt + "\n\nContinue this passage with another 1000+ words of original varied prose, mixing French and English narrative paragraphs on different topics."}}, - "max_tokens": nPredict, - "stream": false, - "temperature": 0.7, - "cache_prompt": false, + "model": "jean", + "messages": []Message{{Role: "user", Content: prompt + "\n\nContinue this passage with another 1000+ words of original varied prose, mixing French and English narrative paragraphs on different topics."}}, + "max_tokens": nPredict, + "stream": false, + "temperature": 0.7, + "cache_prompt": false, } body, _ := json.Marshal(payload) url := fmt.Sprintf("http://localhost:%d/v1/chat/completions", port) @@ -92,12 +92,12 @@ func runBench(nPrompt, nPredict int) (*benchResult, error) { defer resp.Body.Close() var parsed struct { Timings struct { - PromptN int `json:"prompt_n"` - PromptMs float64 `json:"prompt_ms"` - PromptPerSecond float64 `json:"prompt_per_second"` - PredictedN int `json:"predicted_n"` - PredictedMs float64 `json:"predicted_ms"` - PredictedPerSec float64 `json:"predicted_per_second"` + PromptN int `json:"prompt_n"` + PromptMs float64 `json:"prompt_ms"` + PromptPerSecond float64 `json:"prompt_per_second"` + PredictedN int `json:"predicted_n"` + PredictedMs float64 `json:"predicted_ms"` + PredictedPerSec float64 `json:"predicted_per_second"` } `json:"timings"` Usage struct { PromptTokens int `json:"prompt_tokens"` diff --git a/internal/jean/relay_link.go b/internal/jean/relay_link.go index 710312b..c5bf90a 100644 --- a/internal/jean/relay_link.go +++ b/internal/jean/relay_link.go @@ -466,4 +466,3 @@ func demuxTunnelStream(stream net.Conn, httpLn, oaiLn *chanListener) { } httpLn.push(pc) } - diff --git a/internal/jean/sys_proc_windows.go b/internal/jean/sys_proc_windows.go index 0cd6943..6d99d5d 100644 --- a/internal/jean/sys_proc_windows.go +++ b/internal/jean/sys_proc_windows.go @@ -17,9 +17,9 @@ import ( func relaunchDetachedApp() { const ( - detachedProcess = 0x00000008 - createNoWindow = 0x08000000 - createNewProcGrp = 0x00000200 + detachedProcess = 0x00000008 + createNoWindow = 0x08000000 + createNewProcGrp = 0x00000200 ) exe, err := os.Executable() if err != nil { diff --git a/internal/jean/sys_tray_windows.go b/internal/jean/sys_tray_windows.go index 72f6850..15e47b6 100644 --- a/internal/jean/sys_tray_windows.go +++ b/internal/jean/sys_tray_windows.go @@ -105,17 +105,17 @@ func trayIcon() []byte { // Conteneur ICO : ICONDIR (6) + 1 ICONDIRENTRY (16) + PNG. var ico bytes.Buffer - binary.Write(&ico, binary.LittleEndian, uint16(0)) // réservé - binary.Write(&ico, binary.LittleEndian, uint16(1)) // type = icône - binary.Write(&ico, binary.LittleEndian, uint16(1)) // nombre d'images - ico.WriteByte(n) // largeur - ico.WriteByte(n) // hauteur - ico.WriteByte(0) // couleurs - ico.WriteByte(0) // réservé - binary.Write(&ico, binary.LittleEndian, uint16(1)) // plans - binary.Write(&ico, binary.LittleEndian, uint16(32)) // bits/pixel - binary.Write(&ico, binary.LittleEndian, uint32(len(p))) // taille données - binary.Write(&ico, binary.LittleEndian, uint32(6+16)) // offset données + binary.Write(&ico, binary.LittleEndian, uint16(0)) // réservé + binary.Write(&ico, binary.LittleEndian, uint16(1)) // type = icône + binary.Write(&ico, binary.LittleEndian, uint16(1)) // nombre d'images + ico.WriteByte(n) // largeur + ico.WriteByte(n) // hauteur + ico.WriteByte(0) // couleurs + ico.WriteByte(0) // réservé + binary.Write(&ico, binary.LittleEndian, uint16(1)) // plans + binary.Write(&ico, binary.LittleEndian, uint16(32)) // bits/pixel + binary.Write(&ico, binary.LittleEndian, uint32(len(p))) // taille données + binary.Write(&ico, binary.LittleEndian, uint32(6+16)) // offset données ico.Write(p) return ico.Bytes() } diff --git a/internal/jean/web_server.go b/internal/jean/web_server.go index 0fd092d..09e3ef6 100644 --- a/internal/jean/web_server.go +++ b/internal/jean/web_server.go @@ -116,12 +116,12 @@ func newWebMux() *http.ServeMux { api("/api/restart", svcHandler("restart")) api("/api/bench", handleBench) api("/api/bench/last", handleBenchLast) - api("/api/chat", handleChat) // flux d'ABONNEMENT (SSE) : rejoue + suit le fil - api("/api/chat/send", handleChatSend) // envoie un message (lance la génération détachée) - api("/api/chat/stop", handleChatStop) // interrompt la génération en cours - api("/api/chat/reset", handleChatReset) // nouvelle conversation (pour tous les appareils) - api("/api/chat/state", handleChatState) // instantané léger {seq, generating, ctx_used} - api("/api/e2e/chat", handleE2EChat) // même flux mais chiffré E2E (boîte noire via le relais) + api("/api/chat", handleChat) // flux d'ABONNEMENT (SSE) : rejoue + suit le fil + api("/api/chat/send", handleChatSend) // envoie un message (lance la génération détachée) + api("/api/chat/stop", handleChatStop) // interrompt la génération en cours + api("/api/chat/reset", handleChatReset) // nouvelle conversation (pour tous les appareils) + api("/api/chat/state", handleChatState) // instantané léger {seq, generating, ctx_used} + api("/api/e2e/chat", handleE2EChat) // même flux mais chiffré E2E (boîte noire via le relais) return mux } @@ -537,6 +537,7 @@ func handleAgentToggle(w http.ResponseWriter, r *http.Request) { // // GET → {enabled, url, reachable} // POST {enabled, url} → enregistre CRAWL4AI_URL + le drapeau .internet_enabled +// // handleAPIKey expose et pilote la clé d'accès à l'endpoint compatible OpenAI // (llama-server /v1). GET renvoie l'état ; POST {action:"generate"|"set"|"clear", // key?} l'écrit puis redémarre le service (llama-server lit --api-key au lancement).