Auto-réglage GPU : détection, num_ctx optimal et bouton Réglage auto

Backend :
- agent_config : num_ctx dans la config, envoyé à Ollama (0 = défaut modèle)
- ollama_client : méthode ps() (placement GPU/CPU via /api/ps)
- autotune : placement() lit size vs size_vram pour savoir si le modèle est sur GPU
- routes/config : POST /api/config/auto (détecte GPU + modèle, calcule et applique
  num_ctx/max_tokens, renvoie la détection et le placement)

Frontend :
- client/store : autoTune(), état tuning + résultat
- SettingsView : bouton ⚡ Réglage auto, bannière de détection (GPU/VRAM,
  contexte, placement GPU/CPU), slider Contexte (num_ctx)

Config :
- GPU_VRAM_MB / GPU_NAME pour déclarer la VRAM si Ollama est distant

Tests : recommandation (8B sur 12 Go -> ctx 32768), route /auto + persistance,
transmission num_ctx aux options Ollama, placement via /api/ps.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVay7z3y7q2gEe54ByAE6N
This commit is contained in:
Claude committed 2026-06-30 05:54:30 +00:00
1 parent 98d7bb03fd
commit cbe9bcd514
10 files changed
+231 -5

No files matched your search

+29
View File
@@ -61,10 +61,39 @@ export interface AgentConfig {
top_p: number;
top_k: number;
max_tokens: number;
num_ctx: number;
tools: Record<string, boolean>;
confirm_shell: boolean;
}
export interface AutoTuneResult {
detection: {
gpu: { available: boolean; name: string; vram_total_mb: number; source: string };
model_profile: {
context_length: number | null;
parameter_size: string | null;
quantization: string | null;
size_mb: number | null;
};
recommended: { num_ctx: number; max_tokens: number };
rationale: string;
};
placement: { loaded: boolean; where?: string; gpu_percent?: number };
config: AgentConfig;
}
export async function autoTune(
model: string,
apply = true
): Promise<AutoTuneResult> {
const res = await fetch("/api/config/auto", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ model, apply }),
});
return res.json();
}
export async function runShell(
command: string
): Promise<{ command: string; exit_code: number; output: string }> {
+79 -2
View File
@@ -1,7 +1,7 @@
import { useEffect, useState } from "react";
import { useStore } from "../store/useStore";
import { pullModel } from "../api/client";
import type { AgentConfig } from "../api/client";
import type { AgentConfig, AutoTuneResult } from "../api/client";
import { DownloadIcon, RefreshIcon } from "../components/Icon";
const TOOL_DESC: Record<string, string> = {
@@ -25,6 +25,9 @@ export function SettingsView() {
availableTools,
refreshConfig,
updateConfig,
tuning,
tuneResult,
runAutoTune,
} = useStore();
// Brouillon local édité, synchronisé depuis la config serveur.
@@ -221,7 +224,25 @@ export function SettingsView() {
{/* Génération + Outils */}
<div className="flex flex-col gap-5">
<div className="rounded-card border border-line bg-card p-[18px]">
<div className="mb-4 text-sm font-bold">Génération</div>
<div className="mb-4 flex items-center justify-between">
<div className="text-sm font-bold">Génération</div>
<button
onClick={runAutoTune}
disabled={tuning || !selectedModel}
className="flex h-[26px] items-center gap-1.5 rounded-[7px] border border-[rgba(240,161,92,.4)] bg-[rgba(240,161,92,.10)] px-2.5 text-[11.5px] font-semibold text-accent disabled:opacity-40"
title="Détecte le GPU et le modèle, puis optimise le contexte et les jetons"
>
{tuning ? (
<span className="h-3 w-3 animate-spin rounded-full border-2 border-accent/40 border-t-accent" />
) : (
<span>⚡</span>
)}
{tuning ? "Détection…" : "Réglage auto"}
</button>
</div>
{tuneResult && <TuneBanner result={tuneResult} />}
<Slider
label="Température"
value={draft.temperature}
@@ -257,6 +278,17 @@ export function SettingsView() {
step={128}
fmt={(v) => String(Math.round(v))}
onChange={(v) => set("max_tokens", Math.round(v))}
/>
<Slider
label="Contexte (num_ctx)"
value={draft.num_ctx}
min={0}
max={32768}
step={1024}
fmt={(v) =>
v === 0 ? "auto" : v >= 1024 ? `${v / 1024}K` : String(v)
}
onChange={(v) => set("num_ctx", Math.round(v))}
last
/>
</div>
@@ -352,6 +384,51 @@ export function SettingsView() {
);
}
function TuneBanner({ result }: { result: AutoTuneResult }) {
const { gpu, model_profile, recommended, rationale } = result.detection;
const place = result.placement;
const placeLabel =
place.loaded && place.where
? place.where === "gpu"
? `chargé GPU (${place.gpu_percent}%)`
: place.where === "cpu"
? "chargé CPU ⚠️"
: `mixte GPU ${place.gpu_percent}%`
: "non chargé";
const placeColor =
place.where === "gpu" ? "text-ok" : place.where === "cpu" ? "text-warn" : "text-muted";
return (
<div className="mb-4 rounded-[10px] border border-[rgba(240,161,92,.35)] bg-[rgba(240,161,92,.06)] p-3 text-[11.5px]">
<div className="mb-1.5 flex items-center justify-between">
<span className="font-semibold text-accent">⚡ Optimisé</span>
<span className={`font-mono ${placeColor}`}>{placeLabel}</span>
</div>
<div className="font-mono text-muted">
{gpu.available
? `${gpu.name} · ${(gpu.vram_total_mb / 1024).toFixed(1)} Go VRAM`
: "Aucun GPU détecté"}
{model_profile.context_length
? ` · ctx modèle ${Math.round(model_profile.context_length / 1024)}K`
: ""}
</div>
<div className="mt-1 text-muted-2">
→ contexte <b className="text-ink-2">{recommended.num_ctx}</b> · jetons max{" "}
<b className="text-ink-2">{recommended.max_tokens}</b>
</div>
{rationale && <div className="mt-1 text-muted-3">{rationale}</div>}
{!gpu.available && (
<div className="mt-1.5 text-muted-3">
GPU non détecté dans le conteneur. Si Ollama tourne sur une autre
machine, déclare la VRAM via <code>GPU_VRAM_MB</code> (ex. 12000) dans
le compose pour un réglage précis.
</div>
)}
</div>
);
}
function Slider({
label,
value,
+21
View File
@@ -1,5 +1,6 @@
import { create } from "zustand";
import {
autoTune,
createSession,
deleteSession,
getConfig,
@@ -13,6 +14,7 @@ import {
saveConfig,
streamChat,
type AgentConfig,
type AutoTuneResult,
type FileNode,
type Message,
type OllamaModel,
@@ -43,6 +45,10 @@ interface LokiState {
refreshConfig: () => Promise<void>;
updateConfig: (patch: Partial<AgentConfig>) => Promise<void>;
tuning: boolean;
tuneResult: AutoTuneResult | null;
runAutoTune: () => Promise<void>;
pendingShell: string | null; // commande shell en attente de validation
approveShell: () => Promise<void>;
rejectShell: () => Promise<void>;
@@ -114,6 +120,21 @@ export const useStore = create<LokiState>((set, get) => ({
set({ config });
},
tuning: false,
tuneResult: null,
runAutoTune: async () => {
const model = get().selectedModel;
if (!model || get().tuning) return;
set({ tuning: true });
try {
const result = await autoTune(model, true);
set({ tuneResult: result, config: result.config });
} finally {
set({ tuning: false });
}
},
openPreview: async (path) => {
const content = await fileContent(path);
set({ previewPath: path, previewContent: content });