mirror of
https://github.com/R0m1k3/Loki.git
synced 2026-10-11 17:26:57 +02:00
Panneau Matériel : GPU vu par Loki vs GPU utilisé par Ollama
- routes/system : /api/system/hardware — liste les GPU vus par le conteneur (nvidia-smi, multi-GPU) ou l'override GPU_VRAM_MB, et le placement réel des modèles chargés côté Ollama (GPU/CPU/mixte + VRAM) via /api/ps. - config : réintègre gpu_vram_mb / gpu_name (perdus lors d'une fusion). - UI : carte MATÉRIEL dans Configuration (rafraîchie toutes les 5 s) montrant les deux points de vue et si Ollama est local ou distant. Répond à « l'appli est-elle réservée à la 3060 ? » (non) et permet de voir tout de suite si Ollama charge un modèle sur CPU faute de support GPU. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SVay7z3y7q2gEe54ByAE6N
This commit is contained in:
4 files changed
+243
-2
No files matched your search
@@ -15,6 +15,11 @@ class Settings(BaseSettings):
|
||||
# Permet de vérifier que l'image déployée est bien à jour.
|
||||
loki_version: str = "dev"
|
||||
|
||||
# Override manuel de la VRAM (Mo) si la détection GPU échoue dans le
|
||||
# conteneur (utile quand Ollama tourne sur une autre machine).
|
||||
gpu_vram_mb: int = 0
|
||||
gpu_name: str = ""
|
||||
|
||||
model_config = SettingsConfigDict(env_file=".env", extra="ignore")
|
||||
|
||||
|
||||
|
||||
@@ -4,9 +4,13 @@ from __future__ import annotations
|
||||
import asyncio
|
||||
import shutil
|
||||
|
||||
import httpx
|
||||
import psutil
|
||||
from fastapi import APIRouter
|
||||
|
||||
from ..config import settings
|
||||
from ..ollama_client import ollama
|
||||
|
||||
router = APIRouter(prefix="/api/system", tags=["system"])
|
||||
|
||||
_NVIDIA_SMI = shutil.which("nvidia-smi")
|
||||
@@ -50,3 +54,90 @@ async def stats() -> dict:
|
||||
"ram_pct": mem.percent,
|
||||
"gpu": gpu,
|
||||
}
|
||||
|
||||
|
||||
async def _all_local_gpus() -> list[dict]:
|
||||
"""Tous les GPU NVIDIA visibles depuis le CONTENEUR Loki (peut être vide)."""
|
||||
if not _NVIDIA_SMI:
|
||||
return []
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
_NVIDIA_SMI,
|
||||
"--query-gpu=index,name,memory.total,memory.used,utilization.gpu",
|
||||
"--format=csv,noheader,nounits",
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
out, _ = await asyncio.wait_for(proc.communicate(), timeout=3)
|
||||
gpus = []
|
||||
for line in out.decode().strip().splitlines():
|
||||
idx, name, total, used, util = (p.strip() for p in line.split(","))
|
||||
gpus.append({
|
||||
"index": int(idx), "name": name,
|
||||
"vram_total_mb": float(total), "vram_used_mb": float(used),
|
||||
"util_pct": float(util),
|
||||
})
|
||||
return gpus
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
@router.get("/hardware")
|
||||
async def hardware() -> dict:
|
||||
"""Vue matériel : GPU vu par Loki vs GPU réellement utilisé par Ollama."""
|
||||
# 1) Ce que voit le conteneur Loki (nvidia-smi) + override éventuel.
|
||||
local_gpus = await _all_local_gpus()
|
||||
override = None
|
||||
if settings.gpu_vram_mb > 0:
|
||||
override = {
|
||||
"name": settings.gpu_name or "GPU déclaré (GPU_VRAM_MB)",
|
||||
"vram_total_mb": settings.gpu_vram_mb,
|
||||
}
|
||||
|
||||
# 2) Ce qu'Ollama utilise réellement (modèles chargés + placement VRAM/CPU).
|
||||
ollama_info: dict = {"host": ollama.host, "connected": False, "running": []}
|
||||
try:
|
||||
version = await ollama.ping()
|
||||
ollama_info["connected"] = True
|
||||
ollama_info["version"] = version.get("version")
|
||||
except (httpx.HTTPError, OSError) as exc:
|
||||
ollama_info["error"] = str(exc)[:200]
|
||||
|
||||
if ollama_info["connected"]:
|
||||
try:
|
||||
for m in await ollama.ps():
|
||||
size = m.get("size", 0) or 0
|
||||
vram = m.get("size_vram", 0) or 0
|
||||
if size <= 0:
|
||||
where = "inconnu"
|
||||
elif vram >= size * 0.99:
|
||||
where = "GPU"
|
||||
elif vram <= size * 0.01:
|
||||
where = "CPU"
|
||||
else:
|
||||
where = "mixte"
|
||||
ollama_info["running"].append({
|
||||
"name": m.get("name") or m.get("model"),
|
||||
"processor": where,
|
||||
"gpu_percent": int(vram / size * 100) if size else 0,
|
||||
"size_mb": round(size / 1_000_000),
|
||||
"vram_mb": round(vram / 1_000_000),
|
||||
})
|
||||
except (httpx.HTTPError, OSError):
|
||||
pass
|
||||
|
||||
# Ollama est-il sur la même machine que Loki ? (heuristique sur l'hôte)
|
||||
host = ollama.host.lower()
|
||||
is_local = any(h in host for h in ("localhost", "127.0.0.1", "host.docker.internal"))
|
||||
|
||||
return {
|
||||
"loki_gpus": local_gpus,
|
||||
"gpu_override": override,
|
||||
"ollama": ollama_info,
|
||||
"ollama_is_local": is_local,
|
||||
"note": (
|
||||
"Loki ne voit pas directement le GPU d'un Ollama distant : il déduit "
|
||||
"le placement (GPU/CPU) depuis les modèles chargés. Déclare GPU_VRAM_MB "
|
||||
"pour l'auto-réglage si Ollama tourne sur une autre machine."
|
||||
),
|
||||
}
|
||||
@@ -41,6 +41,37 @@ export async function getSystemStats(): Promise<SystemStats> {
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export interface HardwareInfo {
|
||||
loki_gpus: {
|
||||
index: number;
|
||||
name: string;
|
||||
vram_total_mb: number;
|
||||
vram_used_mb: number;
|
||||
util_pct: number;
|
||||
}[];
|
||||
gpu_override: { name: string; vram_total_mb: number } | null;
|
||||
ollama: {
|
||||
host: string;
|
||||
connected: boolean;
|
||||
version?: string;
|
||||
error?: string;
|
||||
running: {
|
||||
name: string;
|
||||
processor: string;
|
||||
gpu_percent: number;
|
||||
size_mb: number;
|
||||
vram_mb: number;
|
||||
}[];
|
||||
};
|
||||
ollama_is_local: boolean;
|
||||
note: string;
|
||||
}
|
||||
|
||||
export async function getHardware(): Promise<HardwareInfo> {
|
||||
const res = await fetch("/api/system/hardware");
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function getVersion(): Promise<string> {
|
||||
try {
|
||||
const res = await fetch("/api/version");
|
||||
|
||||
@@ -1,7 +1,13 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { useStore } from "../store/useStore";
|
||||
import { deleteModel, getBenchScores, pullModel, runBench } from "../api/client";
|
||||
import type { AgentConfig, BenchResult } from "../api/client";
|
||||
import {
|
||||
deleteModel,
|
||||
getBenchScores,
|
||||
getHardware,
|
||||
pullModel,
|
||||
runBench,
|
||||
} from "../api/client";
|
||||
import type { AgentConfig, BenchResult, HardwareInfo } from "../api/client";
|
||||
import { DownloadIcon, RefreshIcon } from "../components/Icon";
|
||||
|
||||
const TOOL_DESC: Record<string, string> = {
|
||||
@@ -460,6 +466,8 @@ export function SettingsView() {
|
||||
</div>
|
||||
</Card>
|
||||
|
||||
<HardwareCard />
|
||||
|
||||
<BenchCard />
|
||||
</div>
|
||||
</div>
|
||||
@@ -489,6 +497,112 @@ export function SettingsView() {
|
||||
}
|
||||
|
||||
/** Benchmark : évalue le modèle sélectionné sur 5 mini-épreuves. */
|
||||
/** Panneau Matériel : GPU vu par Loki vs GPU réellement utilisé par Ollama. */
|
||||
function HardwareCard() {
|
||||
const [hw, setHw] = useState<HardwareInfo | null>(null);
|
||||
|
||||
const refresh = () => getHardware().then(setHw).catch(() => {});
|
||||
useEffect(() => {
|
||||
refresh();
|
||||
const id = setInterval(refresh, 5000);
|
||||
return () => clearInterval(id);
|
||||
}, []);
|
||||
|
||||
return (
|
||||
<Card>
|
||||
<div className="mb-3.5 flex items-center justify-between">
|
||||
<div className="font-pixel text-[11px] text-ink">MATÉRIEL</div>
|
||||
<button
|
||||
onClick={refresh}
|
||||
className="flex h-7 items-center gap-1.5 border-2 border-line bg-card-soft px-2.5 text-[13px] text-muted-2"
|
||||
>
|
||||
<RefreshIcon size={12} />
|
||||
Rafraîchir
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* GPU vu par le conteneur Loki */}
|
||||
<div className="mb-1 text-[12px] font-bold uppercase tracking-wide text-label">
|
||||
GPU vu par Loki (conteneur)
|
||||
</div>
|
||||
{hw?.loki_gpus.length ? (
|
||||
hw.loki_gpus.map((g) => (
|
||||
<div key={g.index} className="border-2 border-line bg-base px-3 py-2 text-[13px]">
|
||||
<div className="flex items-center justify-between">
|
||||
<span className="text-ink">{g.name}</span>
|
||||
<span className="text-muted-2">
|
||||
{(g.vram_used_mb / 1024).toFixed(1)}/
|
||||
{(g.vram_total_mb / 1024).toFixed(1)} Go · {g.util_pct.toFixed(0)}%
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
))
|
||||
) : hw?.gpu_override ? (
|
||||
<div className="border-2 border-line bg-base px-3 py-2 text-[13px] text-ink">
|
||||
{hw.gpu_override.name} ·{" "}
|
||||
{(hw.gpu_override.vram_total_mb / 1024).toFixed(0)} Go{" "}
|
||||
<span className="text-muted-2">(déclaré)</span>
|
||||
</div>
|
||||
) : (
|
||||
<div className="border-2 border-line bg-base px-3 py-2 text-[12px] text-muted-2">
|
||||
Aucun GPU visible depuis le conteneur — normal si Ollama tourne sur une
|
||||
autre machine. Déclare <code>GPU_VRAM_MB</code> pour l'auto-réglage.
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* GPU utilisé par Ollama */}
|
||||
<div className="mb-1 mt-3 text-[12px] font-bold uppercase tracking-wide text-label">
|
||||
Ollama {hw?.ollama_is_local ? "(local)" : "(distant)"}
|
||||
</div>
|
||||
<div className="border-2 border-line bg-base px-3 py-2 text-[13px]">
|
||||
<div className="mb-1 flex items-center gap-2">
|
||||
<span
|
||||
className={`h-2 w-2 border-2 border-line ${
|
||||
hw?.ollama.connected ? "bg-ok" : "bg-warn"
|
||||
}`}
|
||||
/>
|
||||
<span className="min-w-0 flex-1 truncate text-muted-2" title={hw?.ollama.host}>
|
||||
{hw?.ollama.host}
|
||||
</span>
|
||||
{hw?.ollama.version && (
|
||||
<span className="text-[11px] text-muted-3">v{hw.ollama.version}</span>
|
||||
)}
|
||||
</div>
|
||||
{hw?.ollama.running.length ? (
|
||||
hw.ollama.running.map((m) => (
|
||||
<div key={m.name} className="flex items-center gap-2 py-[2px]">
|
||||
<span className="min-w-0 flex-1 truncate text-ink" title={m.name}>
|
||||
{m.name}
|
||||
</span>
|
||||
<span
|
||||
className={
|
||||
m.processor === "GPU"
|
||||
? "text-ok"
|
||||
: m.processor === "CPU"
|
||||
? "text-warn"
|
||||
: "text-ink"
|
||||
}
|
||||
>
|
||||
{m.processor}
|
||||
{m.processor === "mixte" ? ` ${m.gpu_percent}%` : ""}
|
||||
</span>
|
||||
<span className="text-[11px] text-muted-3">
|
||||
{(m.vram_mb / 1024).toFixed(1)}G VRAM
|
||||
</span>
|
||||
</div>
|
||||
))
|
||||
) : (
|
||||
<div className="text-[12px] text-muted-2">
|
||||
Aucun modèle chargé actuellement.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="mt-2 text-[11px] leading-relaxed text-muted-3">{hw?.note}</div>
|
||||
</Card>
|
||||
);
|
||||
}
|
||||
|
||||
function BenchCard() {
|
||||
const selectedModel = useStore((s) => s.selectedModel);
|
||||
const [scores, setScores] = useState<Record<string, BenchResult>>({});
|
||||
|
||||
Reference in new issue
Block a user