mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Add per-model GPU and context profiles
This commit is contained in:
1 parent
cdbb5f03fa
commit
e24d397d31
12 files changed
+224
-215
No files matched your search
@@ -62,38 +62,12 @@ export interface AgentConfig {
|
||||
top_k: number;
|
||||
max_tokens: number;
|
||||
num_ctx: number;
|
||||
num_gpu: number;
|
||||
num_batch: number;
|
||||
tools: Record<string, boolean>;
|
||||
confirm_shell: boolean;
|
||||
}
|
||||
|
||||
export interface AutoTuneResult {
|
||||
detection: {
|
||||
gpu: { available: boolean; name: string; vram_total_mb: number; source: string };
|
||||
model_profile: {
|
||||
context_length: number | null;
|
||||
parameter_size: string | null;
|
||||
quantization: string | null;
|
||||
size_mb: number | null;
|
||||
};
|
||||
recommended: { num_ctx: number; max_tokens: number };
|
||||
rationale: string;
|
||||
};
|
||||
placement: { loaded: boolean; where?: string; gpu_percent?: number };
|
||||
config: AgentConfig;
|
||||
}
|
||||
|
||||
export async function autoTune(
|
||||
model: string,
|
||||
apply = true
|
||||
): Promise<AutoTuneResult> {
|
||||
const res = await fetch("/api/config/auto", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ model, apply }),
|
||||
});
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function runShell(
|
||||
command: string
|
||||
): Promise<{ command: string; exit_code: number; output: string }> {
|
||||
@@ -105,18 +79,21 @@ export async function runShell(
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function getConfig(): Promise<{
|
||||
export async function getConfig(model?: string): Promise<{
|
||||
config: AgentConfig;
|
||||
available_tools: string[];
|
||||
}> {
|
||||
const res = await fetch("/api/config");
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config${query}`);
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function saveConfig(
|
||||
patch: Partial<AgentConfig>
|
||||
patch: Partial<AgentConfig>,
|
||||
model?: string
|
||||
): Promise<AgentConfig> {
|
||||
const res = await fetch("/api/config", {
|
||||
const query = model ? `?model=${encodeURIComponent(model)}` : "";
|
||||
const res = await fetch(`/api/config${query}`, {
|
||||
method: "PUT",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(patch),
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { pullModel } from "../api/client";
|
||||
import type { AgentConfig, AutoTuneResult } from "../api/client";
|
||||
import type { AgentConfig } from "../api/client";
|
||||
import { DownloadIcon, RefreshIcon } from "../components/Icon";
|
||||
|
||||
const TOOL_DESC: Record<string, string> = {
|
||||
@@ -25,9 +25,6 @@ export function SettingsView() {
|
||||
availableTools,
|
||||
refreshConfig,
|
||||
updateConfig,
|
||||
tuning,
|
||||
tuneResult,
|
||||
runAutoTune,
|
||||
} = useStore();
|
||||
|
||||
// Brouillon local édité, synchronisé depuis la config serveur.
|
||||
@@ -36,7 +33,7 @@ export function SettingsView() {
|
||||
|
||||
useEffect(() => {
|
||||
refreshConfig();
|
||||
}, [refreshConfig]);
|
||||
}, [refreshConfig, selectedModel]);
|
||||
useEffect(() => {
|
||||
if (config) setDraft(config);
|
||||
}, [config]);
|
||||
@@ -224,25 +221,12 @@ export function SettingsView() {
|
||||
{/* Génération + Outils */}
|
||||
<div className="flex flex-col gap-5">
|
||||
<div className="rounded-card border border-line bg-card p-[18px]">
|
||||
<div className="mb-4 flex items-center justify-between">
|
||||
<div className="text-sm font-bold">Génération</div>
|
||||
<button
|
||||
onClick={runAutoTune}
|
||||
disabled={tuning || !selectedModel}
|
||||
className="flex h-[26px] items-center gap-1.5 rounded-[7px] border border-[rgba(240,161,92,.4)] bg-[rgba(240,161,92,.10)] px-2.5 text-[11.5px] font-semibold text-accent disabled:opacity-40"
|
||||
title="Détecte le GPU et le modèle, puis optimise le contexte et les jetons"
|
||||
>
|
||||
{tuning ? (
|
||||
<span className="h-3 w-3 animate-spin rounded-full border-2 border-accent/40 border-t-accent" />
|
||||
) : (
|
||||
<span>⚡</span>
|
||||
)}
|
||||
{tuning ? "Détection…" : "Réglage auto"}
|
||||
</button>
|
||||
<div className="mb-1 text-sm font-bold">Génération</div>
|
||||
<div className="mb-4 text-[11.5px] text-muted-2">
|
||||
Profil enregistré pour <b>{selectedModel || "ce modèle"}</b>.
|
||||
Le contexte détermine directement la taille du cache KV.
|
||||
</div>
|
||||
|
||||
{tuneResult && <TuneBanner result={tuneResult} />}
|
||||
|
||||
<Slider
|
||||
label="Température"
|
||||
value={draft.temperature}
|
||||
@@ -280,17 +264,38 @@ export function SettingsView() {
|
||||
onChange={(v) => set("max_tokens", Math.round(v))}
|
||||
/>
|
||||
<Slider
|
||||
label="Contexte (num_ctx)"
|
||||
label="Cache KV / contexte"
|
||||
value={draft.num_ctx}
|
||||
min={0}
|
||||
min={2048}
|
||||
max={32768}
|
||||
step={1024}
|
||||
fmt={(v) =>
|
||||
v === 0 ? "auto" : v >= 1024 ? `${v / 1024}K` : String(v)
|
||||
}
|
||||
fmt={(v) => `${Math.round(v / 1024)}K`}
|
||||
onChange={(v) => set("num_ctx", Math.round(v))}
|
||||
/>
|
||||
<Slider
|
||||
label="Couches GPU"
|
||||
value={draft.num_gpu}
|
||||
min={-1}
|
||||
max={120}
|
||||
step={1}
|
||||
fmt={(v) => (v < 0 ? "auto" : String(Math.round(v)))}
|
||||
onChange={(v) => set("num_gpu", Math.round(v))}
|
||||
/>
|
||||
<Slider
|
||||
label="Batch"
|
||||
value={draft.num_batch}
|
||||
min={64}
|
||||
max={1024}
|
||||
step={64}
|
||||
fmt={(v) => String(Math.round(v))}
|
||||
onChange={(v) => set("num_batch", Math.round(v))}
|
||||
last
|
||||
/>
|
||||
<div className="mt-3 rounded-lg border border-line-soft bg-base px-3 py-2 text-[11px] leading-relaxed text-muted-2">
|
||||
La précision KV (<code>f16</code>/<code>q8_0</code>) est un
|
||||
réglage global du serveur Ollama et nécessite son redémarrage.
|
||||
Les valeurs ci-dessus sont, elles, propres à chaque modèle.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="rounded-card border border-line bg-card p-[18px]">
|
||||
@@ -384,51 +389,6 @@ export function SettingsView() {
|
||||
);
|
||||
}
|
||||
|
||||
function TuneBanner({ result }: { result: AutoTuneResult }) {
|
||||
const { gpu, model_profile, recommended, rationale } = result.detection;
|
||||
const place = result.placement;
|
||||
|
||||
const placeLabel =
|
||||
place.loaded && place.where
|
||||
? place.where === "gpu"
|
||||
? `chargé GPU (${place.gpu_percent}%)`
|
||||
: place.where === "cpu"
|
||||
? "chargé CPU ⚠️"
|
||||
: `mixte GPU ${place.gpu_percent}%`
|
||||
: "non chargé";
|
||||
const placeColor =
|
||||
place.where === "gpu" ? "text-ok" : place.where === "cpu" ? "text-warn" : "text-muted";
|
||||
|
||||
return (
|
||||
<div className="mb-4 rounded-[10px] border border-[rgba(240,161,92,.35)] bg-[rgba(240,161,92,.06)] p-3 text-[11.5px]">
|
||||
<div className="mb-1.5 flex items-center justify-between">
|
||||
<span className="font-semibold text-accent">⚡ Optimisé</span>
|
||||
<span className={`font-mono ${placeColor}`}>{placeLabel}</span>
|
||||
</div>
|
||||
<div className="font-mono text-muted">
|
||||
{gpu.available
|
||||
? `${gpu.name} · ${(gpu.vram_total_mb / 1024).toFixed(1)} Go VRAM`
|
||||
: "Aucun GPU détecté"}
|
||||
{model_profile.context_length
|
||||
? ` · ctx modèle ${Math.round(model_profile.context_length / 1024)}K`
|
||||
: ""}
|
||||
</div>
|
||||
<div className="mt-1 text-muted-2">
|
||||
→ contexte <b className="text-ink-2">{recommended.num_ctx}</b> · jetons max{" "}
|
||||
<b className="text-ink-2">{recommended.max_tokens}</b>
|
||||
</div>
|
||||
{rationale && <div className="mt-1 text-muted-3">{rationale}</div>}
|
||||
{!gpu.available && (
|
||||
<div className="mt-1.5 text-muted-3">
|
||||
GPU non détecté dans le conteneur. Si Ollama tourne sur une autre
|
||||
machine, déclare la VRAM via <code>GPU_VRAM_MB</code> (ex. 12000) dans
|
||||
le compose pour un réglage précis.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Slider({
|
||||
label,
|
||||
value,
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import { create } from "zustand";
|
||||
import {
|
||||
autoTune,
|
||||
createSession,
|
||||
deleteSession,
|
||||
getConfig,
|
||||
@@ -14,7 +13,6 @@ import {
|
||||
saveConfig,
|
||||
streamChat,
|
||||
type AgentConfig,
|
||||
type AutoTuneResult,
|
||||
type FileNode,
|
||||
type Message,
|
||||
type OllamaModel,
|
||||
@@ -47,10 +45,6 @@ interface LokiState {
|
||||
refreshConfig: () => Promise<void>;
|
||||
updateConfig: (patch: Partial<AgentConfig>) => Promise<void>;
|
||||
|
||||
tuning: boolean;
|
||||
tuneResult: AutoTuneResult | null;
|
||||
runAutoTune: () => Promise<void>;
|
||||
|
||||
pendingShell: string | null; // commande shell en attente de validation
|
||||
approveShell: () => Promise<void>;
|
||||
rejectShell: () => Promise<void>;
|
||||
@@ -115,36 +109,26 @@ export const useStore = create<LokiState>((set, get) => ({
|
||||
},
|
||||
|
||||
refreshConfig: async () => {
|
||||
const { config, available_tools } = await getConfig();
|
||||
const { config, available_tools } = await getConfig(
|
||||
get().selectedModel || undefined
|
||||
);
|
||||
set({ config, availableTools: available_tools });
|
||||
},
|
||||
|
||||
updateConfig: async (patch) => {
|
||||
const config = await saveConfig(patch);
|
||||
const config = await saveConfig(patch, get().selectedModel || undefined);
|
||||
set({ config });
|
||||
},
|
||||
|
||||
tuning: false,
|
||||
tuneResult: null,
|
||||
|
||||
runAutoTune: async () => {
|
||||
const model = get().selectedModel;
|
||||
if (!model || get().tuning) return;
|
||||
set({ tuning: true });
|
||||
try {
|
||||
const result = await autoTune(model, true);
|
||||
set({ tuneResult: result, config: result.config });
|
||||
} finally {
|
||||
set({ tuning: false });
|
||||
}
|
||||
},
|
||||
|
||||
openPreview: async (path) => {
|
||||
const content = await fileContent(path);
|
||||
set({ previewPath: path, previewContent: content });
|
||||
},
|
||||
|
||||
setSelectedModel: (name) => set({ selectedModel: name }),
|
||||
setSelectedModel: (name) => {
|
||||
set({ selectedModel: name });
|
||||
void get().refreshConfig();
|
||||
},
|
||||
|
||||
refreshFiles: async () => {
|
||||
try {
|
||||
@@ -175,6 +159,7 @@ export const useStore = create<LokiState>((set, get) => ({
|
||||
? def
|
||||
: models[0]?.name ?? "";
|
||||
set({ models, selectedModel });
|
||||
await get().refreshConfig();
|
||||
} finally {
|
||||
set({ loadingModels: false });
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user