diff --git a/main.go b/main.go index 7b4b903..51125bd 100644 --- a/main.go +++ b/main.go @@ -15,7 +15,7 @@ import ( "strings" ) -const Version = "0.2.13" +const Version = "0.2.14" func main() { // Migration one-shot des anciens skills (SKILLS//SKILL.md) vers la diff --git a/resource_windows_amd64.syso b/resource_windows_amd64.syso index 23aa4f2..063847b 100644 Binary files a/resource_windows_amd64.syso and b/resource_windows_amd64.syso differ diff --git a/resource_windows_arm64.syso b/resource_windows_arm64.syso index 2804c2a..6167438 100644 Binary files a/resource_windows_arm64.syso and b/resource_windows_arm64.syso differ diff --git a/tools.go b/tools.go index 123f10a..f066a60 100644 --- a/tools.go +++ b/tools.go @@ -53,6 +53,7 @@ func baseSystemPrompt(caps Caps) string { b.WriteString("- " + l + "\n") } b.WriteString("\nFor anything about the system or files, use bash instead of guessing. Act immediately — call the right tool, then answer. Never end your turn after only thinking. Be concise.\n") + b.WriteString("Before answering any question about yourself or this machine, always call mem_search first — even trivial-seeming ones. Testing with a tool never replaces this: memory may hold context the tool won't reveal. Search memory, then verify, then answer.\n") b.WriteString("\nDate: " + time.Now().Format("2006-01-02")) return b.String() } diff --git a/ui/index.html b/ui/index.html index 5bde241..46e0add 100644 --- a/ui/index.html +++ b/ui/index.html @@ -1,18 +1,28 @@ - + + + + + + jean
@@ -400,13 +443,19 @@ async function jpost(u,b){ const r=await jfetch(u,{method:'POST',headers:{'Conte async function loadStatus(){ const s=await jget('/api/status'); const el=document.getElementById('status-svc'); - el.className='statuspill '+(s.active?'ok':'err'); - el.innerHTML=''+(s.active?'active':s.state)+' :'+s.port+''; + // Trois états : service coupé (err) · service actif mais modèle pas encore + // chargé (loading, llama-server renvoie 503) · modèle prêt (ok). + let cls='err', txt=s.state; + if(s.active && s.health){ cls='ok'; txt='prêt'; } + else if(s.active){ cls='loading'; txt='chargement…'; } + el.className='statuspill '+cls; + el.innerHTML=''+txt+' :'+s.port+''; + MODEL_READY = !!(s.active && s.health); if(s.ctx){ CTX_MAX=s.ctx; updateCtxMeter(); } } // Compteur de contexte : CTX_USED estimé via les stats serveur (prefill+decode // du dernier tour ≈ taille du prochain prompt). À 90% on propose de compacter. -let CTX_MAX=0, CTX_USED=0; +let CTX_MAX=0, CTX_USED=0, MODEL_READY=false; function setCtxUsed(n){ CTX_USED=n||0; updateCtxMeter(); } function updateCtxMeter(){ if(!CTX_MAX) return; @@ -993,7 +1042,7 @@ async function send(){ let reasonEl=null, contentEl=null, pendingToolEl=null, fullContent='', fullReason='', toolMsgs=null; let turnCollapsibles=[]; // bulles reasoning/tool à replier dès la réponse finale let tokCount=0, reasonTokCount=0, t0=performance.now(), firstTok=0, statsTimer=null; - let serverStats=null; + let serverStats=null, streamErr=false; const updateStats=(final)=>{ if(!contentEl && !reasonEl) return; const el = contentEl || reasonEl; @@ -1038,6 +1087,19 @@ async function send(){ try{ const o=JSON.parse(data); const d=(o.choices&&o.choices[0]&&o.choices[0].delta)||{}; + if(d.error){ + // Erreur serveur (ex : modèle en cours de chargement → 503). On + // l'affiche, on retire le message user de l'historique et on le + // remet dans l'input pour renvoyer d'un clic une fois prêt. + contentEl=null; reasonEl=null; + const eb=addMsg('assistant',''); eb.classList.add('errmsg'); + renderBody(eb, d.error); + if(msgs.length && msgs[msgs.length-1].role==='user'){ ta.value=msgs[msgs.length-1].content; msgs.pop(); } + saveChat(); + streamErr=true; + try{ reader.cancel(); }catch(_){} + break; + } if(d.stats){ serverStats = d.stats; } // Tool-turn messages (assistant tool_calls + tool results) sent at the // end of the stream. On les garde pour les réinjecter dans msgs AVANT @@ -1087,10 +1149,14 @@ async function send(){ } }catch(e){} } + if(streamErr) break; } + if(streamErr) break; + } + if(!streamErr){ + if(toolMsgs) msgs.push(...toolMsgs); + msgs.push({role:'assistant',content:fullContent}); saveChat(); } - if(toolMsgs) msgs.push(...toolMsgs); - msgs.push({role:'assistant',content:fullContent}); saveChat(); // Le KV cache s'accumule : chaque tour ajoute les tokens NOUVELLEMENT traités // (prompt_tokens, le préfixe caché n'est pas recompté) + ceux générés. La // taille réelle du contexte = somme cumulée, donc monotone croissante. diff --git a/versioninfo.json b/versioninfo.json index 099ecaf..5491f8a 100644 --- a/versioninfo.json +++ b/versioninfo.json @@ -3,13 +3,13 @@ "FileVersion": { "Major": 0, "Minor": 2, - "Patch": 13, + "Patch": 14, "Build": 0 }, "ProductVersion": { "Major": 0, "Minor": 2, - "Patch": 13, + "Patch": 14, "Build": 0 }, "FileFlagsMask": "3f", @@ -25,7 +25,7 @@ "LegalCopyright": "Copyright (c) 2026 Jean contributors. MIT License.", "OriginalFilename": "jean.exe", "ProductName": "Jean", - "ProductVersion": "0.2.13", + "ProductVersion": "0.2.14", "Comments": "https://github.com/nathaninline/jean — projet open source (MIT)" }, "VarFileInfo": { diff --git a/web.go b/web.go index f60e3bd..45748d1 100644 --- a/web.go +++ b/web.go @@ -678,6 +678,13 @@ func sseHeartbeat(w http.ResponseWriter, flusher http.Flusher) (*sync.Mutex, fun // runChatStream exécute le chat et pousse chaque événement (delta) via emit, qui // renvoie false pour interrompre. Partagé par handleChat et handleE2EChat. func runChatStream(ctx context.Context, body chatReq, emit func(map[string]any) bool) { + // Garde-fou : si le modèle n'est pas encore chargé, llama-server répond 503 + // ("loading model") et le tour partirait dans le vide (aucune réponse, l'user + // renvoie en boucle). On renvoie une erreur explicite affichée dans le chat. + if !healthCheck() { + emit(map[string]any{"error": "⏳ Le modèle est encore en train de charger — patiente quelques secondes puis renvoie ton message."}) + return + } if body.Temperature == 0 { body.Temperature = 0.7 }