Files
Loki/backend/app/main.py
T
MichaelandClaude Opus 4.8 cb2872c78a perf: fix model reload thrash and cut time-to-first-token
Ollama reloaded the chat model mid-message because plan/summary/router
calls sent divergent runner options (no num_ctx/num_batch) and omitted
keep_alive — the main cause of perceived slowness.

- share one keep-alive httpx.AsyncClient for all Ollama calls
- unify runner options (runner_options) + keep_alive on every model
  call, embeddings included
- drop the blocking LLM routing fallback (pure lexical heuristic)
- run RAG recall + plan + code-model pick in parallel inside the SSE
  stream, after the start event
- RAG cosine scoring off the event loop; cache /api/tags 30s and
  nvidia-smi 5s; frontend polls 2s->5s, warm poll backoff, dedup
  config fetch

feat: working session menu in TopBar (switch/create/rename/delete)
feat: workspace file deletion (DELETE /api/files + UI trash buttons)
docs: recommended Ollama env vars (KEEP_ALIVE, MAX_LOADED_MODELS...)

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-18 13:58:18 +02:00

105 lines
3.2 KiB
Python

"""Point d'entrée FastAPI de Loki.
Sert l'API (/api/*) et, en production, le frontend React compilé (static/).
"""
from __future__ import annotations
import os
from contextlib import asynccontextmanager
from fastapi import FastAPI
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import FileResponse
from fastapi.staticfiles import StaticFiles
from . import agent_config, coder, db, rag
from .config import settings
from .ollama_client import ollama
from .routes import benchmark, chat, config, files, git, models, sessions, shell, system
async def _warm_default_model() -> None:
"""Précharge le modèle par défaut en VRAM au démarrage (best-effort)."""
import asyncio
import logging
await asyncio.sleep(2) # laisse le service démarrer
try:
cfg = agent_config.get_config(settings.default_model)
models.start_model_warm(
settings.default_model, cfg.get("keep_alive", "30m")
)
logging.getLogger(__name__).info(
"Préchargement du modèle %s lancé", settings.default_model
)
except Exception as exc: # best-effort
logging.getLogger(__name__).info(
"Préchargement au démarrage ignoré : %s", exc
)
@asynccontextmanager
async def lifespan(_: FastAPI):
import asyncio
db.init_db()
rag.init_table()
# Workspace en dépôt git : requis pour les commits du moteur code (Aider).
coder.ensure_git(settings.workspace_dir)
# Préchargement du modèle par défaut, sans bloquer le démarrage.
asyncio.create_task(_warm_default_model())
yield
# Ferme le pool HTTP partagé vers Ollama.
await ollama.aclose()
app = FastAPI(
title="Loki", description="Agent IA local sur Ollama", lifespan=lifespan
)
# En dev, le front tourne sur Vite (5173). On autorise le CORS large ;
# en prod le front est servi par le même origin, donc sans impact.
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["*"],
allow_headers=["*"],
)
app.include_router(models.router)
app.include_router(sessions.router)
app.include_router(chat.router)
app.include_router(files.router)
app.include_router(config.router)
app.include_router(shell.router)
app.include_router(system.router)
app.include_router(benchmark.router)
app.include_router(git.router)
@app.get("/api/health")
async def health() -> dict:
return {"status": "ok", "service": "loki", "version": settings.loki_version}
@app.get("/api/version")
async def version() -> dict:
"""Marqueur de build — pour vérifier que l'image déployée est à jour."""
return {"version": settings.loki_version}
# ── Service du frontend compilé (présent uniquement en image Docker) ─────
_STATIC_DIR = os.path.join(os.path.dirname(__file__), "..", "static")
if os.path.isdir(_STATIC_DIR):
app.mount(
"/assets",
StaticFiles(directory=os.path.join(_STATIC_DIR, "assets")),
name="assets",
)
@app.get("/{full_path:path}")
async def spa_fallback(full_path: str):
"""Renvoie index.html pour toutes les routes (SPA React)."""
index = os.path.join(_STATIC_DIR, "index.html")
return FileResponse(index)