Phase 7 : outils web_search et run_shell avec validation

Backend :
- tools.py : web_search (DuckDuckGo HTML sans clé, ou SearxNG via SEARX_URL) et
  run_shell (confiné au workspace, sortie bornée, timeout)
- agent.py : run_shell sensible -> émet tool_confirm et ne s'exécute pas tant que
  l'utilisateur n'a pas validé (confirm_shell)
- agent_config.py : web_search/run_shell désactivés par défaut, flag confirm_shell
- routes/shell.py : POST /api/shell/run exécute une commande validée
- routes/chat.py : relaie l'événement tool_confirm, passe confirm_shell

Frontend :
- streamChat : événement tool_confirm
- store : pendingShell + approveShell/rejectShell (réinjecte le résultat à l'agent)
- ChatPanel : carte de validation de commande (Approuver / Refuser)
- SettingsView : toggles web_search/run_shell, marqueur sensible, switch confirm_shell
- ToolCard : glyphes web_search/run_shell, statut 'à valider', aperçu d'argument

Tests : run_shell confiné, gate de confirmation (commande dangereuse non exécutée),
route shell, parser DuckDuckGo.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVay7z3y7q2gEe54ByAE6N
This commit is contained in:
MichaelandClaude Opus 4.8 committed 2026-06-29 21:09:27 +00:00
1 parent 2db6beea5c
commit de2befe31e
12 files changed
+398 -26

No files matched your search

+4 -4
View File
@@ -72,10 +72,10 @@ npm run dev # http://localhost:5173
- [x] **Phase 2** — Connexion Ollama : statut, liste des modèles, pull avec progression, sélecteur
- [x] **Phase 3** — Chat streaming (SSE) + persistance des sessions (SQLite)
- [x] **Phase 4** — Boucle agentique & outils fichiers (read/write/list), confinés au workspace, rendu des appels d'outils dans le fil
- [~] **Phase 5** — Aperçu HTML live + onglet Code + arborescence du workspace (fait) · onglet Logs (à venir)
- [ ] **Phase 6** — Configuration complète (génération, toggles d'outils, invite système)
- [ ] **Phase 7** — Outils avancés (web_search, run_shell confirmé, html_preview)
- [ ] **Phase 8** — Hardening & documentation Docker
- [x] **Phase 5** — Aperçu HTML live + onglets Code/Logs + arborescence du workspace
- [x] **Phase 6** — Configuration complète (génération, toggles d'outils, invite système)
- [x] **Phase 7** — Outils avancés : `web_search` (DuckDuckGo/SearxNG) et `run_shell` avec validation utilisateur
- [x] **Phase 8** — Durcissement & documentation Docker
## Structure
+47 -1
View File
@@ -41,6 +41,7 @@ async def run_agent(
*,
options: dict | None = None,
enabled_tools: list[str] | None = None,
confirm_shell: bool = True,
) -> AsyncIterator[dict]:
# enabled_tools=None -> tous les outils ; liste vide -> aucun outil.
if enabled_tools is None:
@@ -83,6 +84,7 @@ async def run_agent(
break
# Exécution des outils demandés, puis réinjection des résultats.
awaiting_confirmation = False
for tc in tool_calls:
fn = tc.get("function", {})
name = fn.get("name", "")
@@ -90,10 +92,39 @@ async def run_agent(
yield {"type": "tool_call", "name": name, "args": args}
# run_shell est sensible : on demande validation au lieu d'exécuter.
if name == "run_shell" and confirm_shell:
command = args.get("command", "")
record = {
"name": name,
"args": args,
"summary": "validation requise",
"status": "pending",
}
collected.append(record)
yield {"type": "tool_confirm", "name": name, "command": command}
convo.append(
{
"role": "tool",
"name": name,
"content": json.dumps(
{
"ok": False,
"status": "pending",
"message": "Commande en attente de validation "
"de l'utilisateur. N'exécute rien d'autre.",
},
ensure_ascii=False,
),
}
)
awaiting_confirmation = True
continue
try:
result = run_tool(name, args)
summary = result.get("summary", "terminé")
status = "ok"
status = result.get("_status", "ok")
except ToolError as exc:
result = {"ok": False, "error": str(exc)}
summary = str(exc)
@@ -110,6 +141,21 @@ async def run_agent(
"content": json.dumps(result, ensure_ascii=False),
}
)
# Une commande shell attend une validation : on interrompt la boucle.
if awaiting_confirmation:
# Laisse le modèle conclure son tour (message d'attente).
final_chunk = ""
async for chunk in ollama.chat(model, convo, options=options, stream=True):
tok = chunk.get("message", {}).get("content", "")
if tok:
final_chunk += tok
yield {"type": "token", "content": tok}
if chunk.get("done"):
break
if final_chunk.strip():
text_parts.append(final_chunk.strip())
break
except (httpx.HTTPError, OSError) as exc:
yield {"type": "error", "message": str(exc)}
return
+20 -5
View File
@@ -16,8 +16,17 @@ DEFAULT_SYSTEM_PROMPT = (
"concise en français. Après avoir écrit un fichier, propose un aperçu."
)
# Outils réellement disponibles (les sensibles arrivent en phase 7).
AVAILABLE_TOOLS = ["read_file", "write_file", "list_dir"]
# Outils disponibles. Les sensibles (web_search, run_shell) sont désactivés
# par défaut, conformément à la maquette.
AVAILABLE_TOOLS = ["read_file", "write_file", "list_dir", "web_search", "run_shell"]
SENSITIVE_TOOLS = {"run_shell"}
DEFAULT_TOOL_STATE = {
"read_file": True,
"write_file": True,
"list_dir": True,
"web_search": False,
"run_shell": False,
}
DEFAULT_CONFIG: dict = {
"system_prompt": DEFAULT_SYSTEM_PROMPT,
@@ -25,7 +34,9 @@ DEFAULT_CONFIG: dict = {
"top_p": 0.9,
"top_k": 40,
"max_tokens": 2048,
"tools": {name: True for name in AVAILABLE_TOOLS},
"tools": dict(DEFAULT_TOOL_STATE),
# Demander une validation utilisateur avant toute commande shell.
"confirm_shell": True,
}
@@ -34,7 +45,7 @@ def get_config() -> dict:
stored = db.get_config_value(CONFIG_KEY) or {}
cfg = {**DEFAULT_CONFIG, **stored}
cfg["tools"] = {
name: bool(stored.get("tools", {}).get(name, True))
name: bool(stored.get("tools", {}).get(name, DEFAULT_TOOL_STATE[name]))
for name in AVAILABLE_TOOLS
}
return cfg
@@ -45,7 +56,11 @@ def save_config(patch: dict) -> dict:
cfg = {**get_config(), **{k: v for k, v in patch.items() if v is not None}}
if "tools" in patch and patch["tools"]:
cfg["tools"] = {
name: bool(patch["tools"].get(name, cfg["tools"].get(name, True)))
name: bool(
patch["tools"].get(
name, cfg["tools"].get(name, DEFAULT_TOOL_STATE[name])
)
)
for name in AVAILABLE_TOOLS
}
db.set_config_value(CONFIG_KEY, cfg)
+2 -1
View File
@@ -14,7 +14,7 @@ from fastapi.staticfiles import StaticFiles
from . import db
from .config import settings
from .routes import chat, config, files, models, sessions
from .routes import chat, config, files, models, sessions, shell
@asynccontextmanager
@@ -41,6 +41,7 @@ app.include_router(sessions.router)
app.include_router(chat.router)
app.include_router(files.router)
app.include_router(config.router)
app.include_router(shell.router)
@app.get("/api/health")
+3 -8
View File
@@ -60,16 +60,11 @@ async def chat(req: ChatRequest) -> StreamingResponse:
convo,
options=agent_config.ollama_options(cfg),
enabled_tools=agent_config.enabled_tool_names(cfg),
confirm_shell=cfg.get("confirm_shell", True),
):
etype = ev.pop("type")
if etype == "token":
yield _sse("token", ev)
elif etype == "tool_call":
yield _sse("tool_call", ev)
elif etype == "tool_result":
yield _sse("tool_result", ev)
elif etype == "error":
yield _sse("error", ev)
if etype in ("token", "tool_call", "tool_result", "tool_confirm", "error"):
yield _sse(etype, ev)
elif etype == "final":
final_content = ev["content"]
tools_meta = ev["tools"]
+33
View File
@@ -0,0 +1,33 @@
"""Route d'exécution d'une commande shell validée par l'utilisateur.
La boucle agentique n'exécute jamais run_shell elle-même quand la confirmation
est active : elle émet un événement `tool_confirm`. Le client affiche la
commande, et c'est seulement après clic explicite de l'utilisateur que cette
route exécute la commande (confinée au workspace).
"""
from __future__ import annotations
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
from ..tools import ToolError, run_shell
router = APIRouter(prefix="/api/shell", tags=["shell"])
class ShellRequest(BaseModel):
command: str
@router.post("/run")
async def run(req: ShellRequest) -> dict:
"""Exécute la commande validée et renvoie sa sortie."""
try:
result = run_shell(req.command)
except ToolError as exc:
raise HTTPException(400, str(exc))
return {
"command": req.command,
"exit_code": result["exit_code"],
"output": result["output"],
}
+139
View File
@@ -9,7 +9,12 @@ toute tentative de sortie (../, chemin absolu hors workspace) est rejetée.
"""
from __future__ import annotations
import html
import os
import re
import subprocess
import httpx
from .config import settings
@@ -71,11 +76,114 @@ def list_dir(path: str = ".") -> dict:
}
def web_search(query: str, max_results: int = 5) -> dict:
"""Recherche web (DuckDuckGo HTML, sans clé d'API).
Optionnellement, si SEARX_URL est défini, interroge une instance SearxNG.
Renvoie une liste de résultats {title, url, snippet}.
"""
query = (query or "").strip()
if not query:
raise ToolError("requête de recherche vide")
searx = os.environ.get("SEARX_URL")
try:
if searx:
results = _search_searx(searx, query, max_results)
else:
results = _search_duckduckgo(query, max_results)
except httpx.HTTPError as exc:
raise ToolError(f"recherche web indisponible : {exc}") from exc
summary = f"{len(results)} résultat(s)" if results else "aucun résultat"
return {"ok": True, "results": results, "summary": summary}
def _search_searx(base: str, query: str, n: int) -> list[dict]:
with httpx.Client(timeout=10.0) as client:
r = client.get(
base.rstrip("/") + "/search",
params={"q": query, "format": "json"},
)
r.raise_for_status()
data = r.json().get("results", [])[:n]
return [
{"title": d.get("title", ""), "url": d.get("url", ""),
"snippet": d.get("content", "")}
for d in data
]
_DDG_RESULT = re.compile(
r'<a[^>]*class="result__a"[^>]*href="([^"]+)"[^>]*>(.*?)</a>'
r'.*?class="result__snippet"[^>]*>(.*?)</a>',
re.DOTALL,
)
_TAGS = re.compile(r"<[^>]+>")
def _clean(text: str) -> str:
return html.unescape(_TAGS.sub("", text)).strip()
def _search_duckduckgo(query: str, n: int) -> list[dict]:
with httpx.Client(timeout=10.0, follow_redirects=True) as client:
r = client.post(
"https://html.duckduckgo.com/html/",
data={"q": query},
headers={"User-Agent": "Mozilla/5.0 (Loki agent)"},
)
r.raise_for_status()
results = []
for url, title, snippet in _DDG_RESULT.findall(r.text)[:n]:
results.append({
"title": _clean(title),
"url": html.unescape(url),
"snippet": _clean(snippet),
})
return results
def run_shell(command: str, timeout: int = 60) -> dict:
"""Exécute une commande shell dans le workspace (outil sensible).
L'exécution effective n'a lieu qu'après validation utilisateur (gérée par
la boucle agentique / la route /api/shell). Confinée au workspace.
"""
command = (command or "").strip()
if not command:
raise ToolError("commande vide")
try:
proc = subprocess.run(
command,
shell=True,
cwd=_workspace_root(),
capture_output=True,
text=True,
timeout=timeout,
)
except subprocess.TimeoutExpired as exc:
raise ToolError(f"délai dépassé ({timeout}s)") from exc
out = (proc.stdout or "") + (proc.stderr or "")
out = out[:4000] # borne la taille renvoyée au modèle
status = "ok" if proc.returncode == 0 else "error"
return {
"ok": proc.returncode == 0,
"exit_code": proc.returncode,
"output": out,
"summary": f"code {proc.returncode}",
"_status": status,
}
# ── Registre & définitions exposées au modèle ────────────────────────────
TOOL_IMPL = {
"read_file": read_file,
"write_file": write_file,
"list_dir": list_dir,
"web_search": web_search,
"run_shell": run_shell,
}
TOOL_DEFINITIONS = [
@@ -121,6 +229,37 @@ TOOL_DEFINITIONS = [
},
},
},
{
"type": "function",
"function": {
"name": "web_search",
"description": "Rechercher sur le web et renvoyer les meilleurs résultats.",
"parameters": {
"type": "object",
"properties": {
"query": {"type": "string", "description": "Termes de recherche"}
},
"required": ["query"],
},
},
},
{
"type": "function",
"function": {
"name": "run_shell",
"description": (
"Exécuter une commande shell dans le workspace. Outil sensible :"
" l'utilisateur doit valider la commande avant exécution."
),
"parameters": {
"type": "object",
"properties": {
"command": {"type": "string", "description": "Commande à exécuter"}
},
"required": ["command"],
},
},
},
]
+15 -1
View File
@@ -42,7 +42,7 @@ export interface ToolCall {
name: string;
args: Record<string, unknown>;
summary?: string;
status?: "ok" | "error" | "running";
status?: "ok" | "error" | "running" | "pending";
}
export interface Message {
@@ -62,6 +62,18 @@ export interface AgentConfig {
top_k: number;
max_tokens: number;
tools: Record<string, boolean>;
confirm_shell: boolean;
}
export async function runShell(
command: string
): Promise<{ command: string; exit_code: number; output: string }> {
const res = await fetch("/api/shell/run", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ command }),
});
return res.json();
}
export async function getConfig(): Promise<{
@@ -134,6 +146,7 @@ export async function streamChat(
onToken: (t: string) => void;
onToolCall: (call: ToolCall) => void;
onToolResult: (call: ToolCall) => void;
onToolConfirm: (command: string) => void;
onDone: (full: string) => void;
onError: (msg: string) => void;
}
@@ -173,6 +186,7 @@ export async function streamChat(
else if (event === "tool_call")
handlers.onToolCall({ ...payload, status: "running" });
else if (event === "tool_result") handlers.onToolResult(payload);
else if (event === "tool_confirm") handlers.onToolConfirm(payload.command);
else if (event === "done") handlers.onDone(payload.content);
else if (event === "error") handlers.onError(payload.message);
} catch {
+24 -3
View File
@@ -4,11 +4,16 @@ import type { ToolCall } from "../api/client";
export function ToolCard({ call }: { call: ToolCall }) {
const running = call.status === "running";
const error = call.status === "error";
const pending = call.status === "pending";
// Argument principal affiché entre parenthèses (path le plus souvent).
// Argument principal affiché entre parenthèses selon l'outil.
const mainArg =
(call.args?.path as string) ??
(call.args?.query as string) ??
(call.args?.command as string);
const argPreview =
typeof call.args?.path === "string"
? `("${call.args.path}")`
typeof mainArg === "string"
? `("${mainArg.length > 48 ? mainArg.slice(0, 48) + "…" : mainArg}")`
: Object.keys(call.args ?? {}).length
? "(…)"
: "()";
@@ -32,6 +37,8 @@ export function ToolCard({ call }: { call: ToolCall }) {
<span className="h-2.5 w-2.5 animate-spin rounded-full border-2 border-muted-3 border-t-accent" />
en cours
</span>
) : pending ? (
<span className="font-mono text-[11px] text-warn">⏸ à valider</span>
) : (
<span
className={`flex items-center gap-1.5 font-mono text-[11px] ${
@@ -74,6 +81,20 @@ function ToolGlyph({ name }: { name: string }) {
<path d="M4 7h6l2 2h8v9H4z" />
</svg>
);
if (name === "web_search")
return (
<svg {...common}>
<circle cx="11" cy="11" r="7" />
<path d="m21 21-4.3-4.3" />
</svg>
);
if (name === "run_shell")
return (
<svg {...common}>
<path d="m6 9 3 3-3 3M13 15h5" />
<rect x="2" y="4" width="20" height="16" rx="2" />
</svg>
);
// read_file (défaut)
return (
<svg {...common}>
+50
View File
@@ -15,6 +15,9 @@ export function ChatPanel() {
sendMessage,
currentSessionId,
config,
pendingShell,
approveShell,
rejectShell,
} = useStore();
const activeTools = config
@@ -100,6 +103,13 @@ export function ChatPanel() {
pending
/>
)}
{pendingShell && (
<ShellConfirm
command={pendingShell}
onApprove={approveShell}
onReject={rejectShell}
/>
)}
</div>
)}
</div>
@@ -201,6 +211,46 @@ function Bubble({ msg, pending }: { msg: Message; pending?: boolean }) {
);
}
function ShellConfirm({
command,
onApprove,
onReject,
}: {
command: string;
onApprove: () => void;
onReject: () => void;
}) {
return (
<div className="ml-[43px] overflow-hidden rounded-[11px] border border-[#4a3a2a] bg-card-deep">
<div className="flex items-center gap-2 border-b border-line-soft px-[13px] py-2.5">
<span className="font-mono text-[12.5px] font-semibold text-warn">
run_shell
</span>
<span className="text-xs text-muted-2">· commande sensible à valider</span>
</div>
<div className="px-[13px] py-3">
<pre className="m-0 mb-3 overflow-auto whitespace-pre-wrap rounded-lg bg-sunken px-3 py-2.5 font-mono text-[12px] text-ink-2">
$ {command}
</pre>
<div className="flex items-center gap-2">
<button
onClick={onApprove}
className="flex h-[30px] items-center gap-1.5 rounded-lg bg-accent px-3.5 text-[12.5px] font-bold text-[#1f1b16]"
>
Approuver &amp; exécuter
</button>
<button
onClick={onReject}
className="flex h-[30px] items-center rounded-lg border border-line-strong px-3.5 text-[12.5px] font-semibold text-ink-3"
>
Refuser
</button>
</div>
</div>
</div>
);
}
function Chip({ children }: { children: React.ReactNode }) {
return (
<div className="flex h-[25px] items-center gap-1.5 rounded-[7px] border border-line-strong bg-card-soft px-[9px] text-[11.5px] text-ink-3">
+28 -3
View File
@@ -8,7 +8,10 @@ const TOOL_DESC: Record<string, string> = {
read_file: "Lire un fichier du projet",
write_file: "Créer / modifier un fichier",
list_dir: "Lister un répertoire",
web_search: "Recherche web",
run_shell: "Exécuter une commande",
};
const SENSITIVE = new Set(["run_shell"]);
/** Vue Configuration (Frame 2) — Modèle & génération, outils, invite système. */
export function SettingsView() {
@@ -279,6 +282,9 @@ export function SettingsView() {
</span>
<span className="flex-1 text-xs text-muted-2">
{TOOL_DESC[name] ?? ""}
{SENSITIVE.has(name) && (
<span className="text-warn"> · sensible</span>
)}
</span>
<button
onClick={() =>
@@ -298,9 +304,28 @@ export function SettingsView() {
</div>
);
})}
<div className="mt-1 border-t border-line-soft pt-2.5 font-mono text-[11px] text-muted-3">
web_search · run_shell — arrivent bientôt (sensibles)
</div>
{draft.tools.run_shell && (
<div className="mt-2 flex items-center gap-3 rounded-[10px] border border-line-soft bg-base px-3 py-2.5">
<span className="flex-1 text-xs text-ink-3">
Demander une validation avant chaque commande shell
</span>
<button
onClick={() => set("confirm_shell", !draft.confirm_shell)}
className="relative h-5 w-[34px] rounded-full transition-colors"
style={{
background: draft.confirm_shell ? "#f0a15c" : "#3a342c",
}}
>
<span
className="absolute top-0.5 h-4 w-4 rounded-full transition-all"
style={{
background: draft.confirm_shell ? "#1f1b16" : "#6f675c",
left: draft.confirm_shell ? 16 : 2,
}}
/>
</button>
</div>
)}
</div>
</div>
</div>
+33
View File
@@ -9,6 +9,7 @@ import {
fileContent,
listFiles,
listSessions,
runShell,
saveConfig,
streamChat,
type AgentConfig,
@@ -42,6 +43,10 @@ interface LokiState {
refreshConfig: () => Promise<void>;
updateConfig: (patch: Partial<AgentConfig>) => Promise<void>;
pendingShell: string | null; // commande shell en attente de validation
approveShell: () => Promise<void>;
rejectShell: () => Promise<void>;
openPreview: (path: string) => Promise<void>;
setSelectedModel: (name: string) => void;
refreshStatus: () => Promise<void>;
@@ -72,6 +77,32 @@ export const useStore = create<LokiState>((set, get) => ({
previewContent: "",
config: null,
availableTools: [],
pendingShell: null,
approveShell: async () => {
const cmd = get().pendingShell;
if (!cmd) return;
set({ pendingShell: null });
let report: string;
try {
const r = await runShell(cmd);
report =
`J'ai validé la commande \`${cmd}\` (code ${r.exit_code}).\n` +
`Sortie :\n\`\`\`\n${r.output || "(vide)"}\n\`\`\``;
} catch {
report = `Échec de l'exécution de \`${cmd}\`.`;
}
await get().refreshFiles();
// On renvoie le résultat à l'agent pour qu'il poursuive.
await get().sendMessage(report);
},
rejectShell: async () => {
const cmd = get().pendingShell;
if (!cmd) return;
set({ pendingShell: null });
await get().sendMessage(`J'ai refusé la commande \`${cmd}\`. N'exécute pas cette commande.`);
},
refreshConfig: async () => {
const { config, available_tools } = await getConfig();
@@ -171,6 +202,7 @@ export const useStore = create<LokiState>((set, get) => ({
streaming: true,
streamContent: "",
streamTools: [],
pendingShell: null,
});
await streamChat(
@@ -190,6 +222,7 @@ export const useStore = create<LokiState>((set, get) => ({
}
set({ streamTools: tools });
},
onToolConfirm: (command) => set({ pendingShell: command }),
onDone: async () => {
// Repère un fichier HTML écrit pour l'afficher automatiquement.
const writtenHtml = [...get().streamTools]