Analyseur de texte riche pour les informations adhérents

Sous-ensemble de Markdown — gras, italique, listes, liens, sous-titres —
analysé par `src/lib/rich-text.ts` puis rendu en éléments React, jamais en
HTML brut : même principe que le journal d'activité. Un éditeur WYSIWYG
aurait imposé de stocker du HTML tiers puis de l'assainir, soit une
seconde dépendance et une seconde surface d'attaque.

Deux rendus, un seul analyseur : `richTextNodes()` pour l'écran,
`richTextToEmailHtml()` pour le message. L'aperçu du formulaire passera
par le premier, il sera donc fidèle par construction.

`safeHttpUrl()` remplace le `webUrl()` privé d'`email-templates.ts` : un
lien hors http/https perd sa cible et ne garde que son libellé.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014QsLjRAnuLwivqxCbM4WeP
This commit is contained in:
Claude committed 2026-09-14 15:28:37 +00:00
1 parent 3df3e55974
commit 1701fc2834
3 files changed
+487 -10

No files matched your search

+3 -10
View File
@@ -1,3 +1,5 @@
import { safeHttpUrl } from "@/lib/rich-text";
export type EmailBrand = {
associationName: string;
address?: string;
@@ -45,15 +47,6 @@ function lines(value: unknown) {
return esc(value).replace(/\r?\n/g, "<br>");
}
function webUrl(value: string) {
try {
const url = new URL(value);
return url.protocol === "http:" || url.protocol === "https:" ? url.href : "";
} catch {
return "";
}
}
function paragraphs(value: string) {
return value
.split(/\r?\n\s*\r?\n/)
@@ -100,7 +93,7 @@ function emailShell({ title, content, baseUrl, brand }: { title: string; content
}
function cta(url: string, label: string) {
const safe = webUrl(url);
const safe = safeHttpUrl(url);
if (!safe || !label) return "";
return `<table role="presentation" width="300" cellpadding="0" cellspacing="0" border="0" style="width:300px;border-collapse:collapse;"><tr><td align="center" bgcolor="#2C6FB3" style="background:#2C6FB3;padding:16px 0;border-radius:8px;font-family:Arial,Helvetica,sans-serif;font-size:17px;font-weight:bold;"><a href="${esc(safe)}" target="_blank" style="display:block;color:#FFFFFF;text-decoration:none;">${esc(label)}</a></td></tr></table>`;
}
+318
View File
@@ -0,0 +1,318 @@
import type { ReactNode } from "react";
import { createElement, Fragment } from "react";
/**
* Sous-ensemble de Markdown pour les informations de l'association.
*
* Module **pur**, verrouillé par `tests/rich-text.test.ts`, bâti sur le même
* principe que `src/lib/activity.ts` : on ne stocke jamais de HTML et on ne
* rend jamais de HTML brut. Le texte saisi est analysé ici, puis rendu en
* éléments React côté site — ce que React échappe. Un éditeur WYSIWYG
* produirait du HTML, qu'il faudrait stocker puis assainir : deuxième
* dépendance, deuxième surface d'attaque.
*
* Grammaire : `**gras**`, `*italique*`, `- puce`, `1. numéro`,
* `[texte](https://…)`, `## sous-titre`, ligne vide = paragraphe.
* Un saut de ligne simple reste un saut de ligne.
*
* Deux rendus, un seul analyseur : `richTextNodes()` pour l'écran,
* `richTextToEmailHtml()` pour le message. L'aperçu du formulaire passe par
* `richTextNodes()`, il est donc fidèle par construction.
*/
export type Inline =
| { kind: "text"; value: string }
| { kind: "break" }
| { kind: "strong"; children: Inline[] }
| { kind: "em"; children: Inline[] }
| { kind: "link"; href: string; children: Inline[] };
export type Block =
| { kind: "paragraph"; children: Inline[] }
| { kind: "heading"; children: Inline[] }
| { kind: "bullets"; items: Inline[][] }
| { kind: "ordered"; items: Inline[][] };
/**
* Seuls `http:` et `https:` passent. Un `javascript:` ou un `data:` rendrait
* exécutable un lien écrit par un tiers ; il est rendu comme du texte.
*/
export function safeHttpUrl(value: string): string {
try {
const url = new URL(String(value ?? "").trim());
return url.protocol === "http:" || url.protocol === "https:" ? url.href : "";
} catch {
return "";
}
}
function parseInline(text: string): Inline[] {
const nodes: Inline[] = [];
let buffer = "";
let index = 0;
const flush = () => {
if (buffer) nodes.push({ kind: "text", value: buffer });
buffer = "";
};
while (index < text.length) {
// `**` avant `*` : sans cet ordre, « **gras** » serait lu comme un
// italique vide suivi du mot.
if (text.startsWith("**", index)) {
const end = text.indexOf("**", index + 2);
if (end > index + 2) {
flush();
nodes.push({ kind: "strong", children: parseInline(text.slice(index + 2, end)) });
index = end + 2;
continue;
}
}
if (text[index] === "*") {
const end = text.indexOf("*", index + 1);
if (end > index + 1) {
flush();
nodes.push({ kind: "em", children: parseInline(text.slice(index + 1, end)) });
index = end + 1;
continue;
}
}
if (text[index] === "[") {
const close = text.indexOf("]", index + 1);
if (close > index && text[close + 1] === "(") {
const paren = text.indexOf(")", close + 2);
if (paren > close + 1) {
const label = text.slice(index + 1, close);
const href = safeHttpUrl(text.slice(close + 2, paren));
flush();
// Lien refusé : le libellé reste, la cible disparaît.
if (href) nodes.push({ kind: "link", href, children: parseInline(label) });
else nodes.push(...parseInline(label));
index = paren + 1;
continue;
}
}
}
buffer += text[index];
index += 1;
}
flush();
return nodes;
}
function parseInlineLines(lines: string[]): Inline[] {
const nodes: Inline[] = [];
lines.forEach((line, index) => {
if (index > 0) nodes.push({ kind: "break" });
nodes.push(...parseInline(line.trim()));
});
return nodes;
}
export function parseRichText(source: string): Block[] {
const lines = String(source ?? "").replace(/\r\n?/g, "\n").split("\n");
const blocks: Block[] = [];
let paragraph: string[] = [];
const flushParagraph = () => {
if (paragraph.length === 0) return;
blocks.push({ kind: "paragraph", children: parseInlineLines(paragraph) });
paragraph = [];
};
const pushItem = (kind: "bullets" | "ordered", item: Inline[]) => {
flushParagraph();
const last = blocks[blocks.length - 1];
if (last && last.kind === kind) last.items.push(item);
else blocks.push({ kind, items: [item] } as Block);
};
for (const raw of lines) {
const line = raw.trimEnd();
if (!line.trim()) {
flushParagraph();
continue;
}
const heading = /^##\s+(.*)$/.exec(line);
if (heading) {
flushParagraph();
blocks.push({ kind: "heading", children: parseInline(heading[1].trim()) });
continue;
}
// L'espace après le tiret est obligatoire : « *italique* » en début de
// ligne ne doit pas devenir une puce.
const bullet = /^[-*]\s+(.*)$/.exec(line);
if (bullet) {
pushItem("bullets", parseInline(bullet[1].trim()));
continue;
}
const ordered = /^\d+[.)]\s+(.*)$/.exec(line);
if (ordered) {
pushItem("ordered", parseInline(ordered[1].trim()));
continue;
}
paragraph.push(line);
}
flushParagraph();
return blocks;
}
// ---- Rendu écran ----
function inlineNodes(nodes: Inline[]): ReactNode[] {
return nodes.map((node, index) => {
switch (node.kind) {
case "break":
return createElement("br", { key: index });
case "strong":
return createElement("strong", { key: index }, inlineNodes(node.children));
case "em":
return createElement("em", { key: index }, inlineNodes(node.children));
case "link":
return createElement(
"a",
{ key: index, href: node.href, target: "_blank", rel: "noopener noreferrer" },
inlineNodes(node.children),
);
default:
return createElement(Fragment, { key: index }, node.value);
}
});
}
/**
* Rend le texte en nœuds React. Rien n'est jamais passé à
* `dangerouslySetInnerHTML` : une balise écrite dans la saisie s'affiche
* littéralement au lieu d'être interprétée.
*/
export function richTextNodes(source: string): ReactNode {
return parseRichText(source).map((block, index) => {
switch (block.kind) {
case "heading":
return createElement("h4", { key: index }, inlineNodes(block.children));
case "bullets":
return createElement(
"ul",
{ key: index },
block.items.map((item, position) => createElement("li", { key: position }, inlineNodes(item))),
);
case "ordered":
return createElement(
"ol",
{ key: index },
block.items.map((item, position) => createElement("li", { key: position }, inlineNodes(item))),
);
default:
return createElement("p", { key: index }, inlineNodes(block.children));
}
});
}
// ---- Rendu e-mail ----
const BODY_STYLE = "font-family:Arial,Helvetica,sans-serif;font-size:15px;line-height:25px;color:#5d5447;";
const HEADING_STYLE = "font-family:Georgia,'Times New Roman',serif;font-size:19px;line-height:26px;color:#26201a;";
function esc(value: string) {
return String(value ?? "")
.replace(/&/g, "&amp;")
.replace(/</g, "&lt;")
.replace(/>/g, "&gt;")
.replace(/"/g, "&quot;");
}
function inlineHtml(nodes: Inline[]): string {
return nodes
.map((node) => {
switch (node.kind) {
case "break":
return "<br>";
case "strong":
return `<strong style="color:#26201a;">${inlineHtml(node.children)}</strong>`;
case "em":
return `<em>${inlineHtml(node.children)}</em>`;
case "link":
return `<a href="${esc(node.href)}" target="_blank" style="color:#2C6FB3;">${inlineHtml(node.children)}</a>`;
default:
return esc(node.value);
}
})
.join("");
}
/**
* HTML à styles en ligne pour le gabarit d'e-mail. Sûr parce que le seul
* producteur est l'analyseur ci-dessus : aucune balise de la saisie n'arrive
* jusqu'ici, tout passe par `esc()`.
*/
export function richTextToEmailHtml(source: string): string {
return parseRichText(source)
.map((block) => {
switch (block.kind) {
case "heading":
return `<div style="${HEADING_STYLE}margin:0 0 12px;">${inlineHtml(block.children)}</div>`;
case "bullets":
case "ordered": {
const tag = block.kind === "bullets" ? "ul" : "ol";
const items = block.items.map((item) => `<li style="margin-bottom:4px;">${inlineHtml(item)}</li>`).join("");
return `<${tag} style="${BODY_STYLE}margin:0 0 18px;padding-left:22px;">${items}</${tag}>`;
}
default:
return `<div style="${BODY_STYLE}margin:0 0 18px;">${inlineHtml(block.children)}</div>`;
}
})
.join("");
}
// ---- Rendus texte ----
function inlinePlain(nodes: Inline[]): string {
return nodes
.map((node) => {
switch (node.kind) {
case "break":
return "\n";
case "link": {
const label = inlinePlain(node.children);
return label === node.href ? node.href : `${label} (${node.href})`;
}
case "strong":
case "em":
return inlinePlain(node.children);
default:
return node.value;
}
})
.join("");
}
/** Repli `text/plain` du message : même contenu, sans balisage. */
export function richTextToPlain(source: string): string {
return parseRichText(source)
.map((block) => {
switch (block.kind) {
case "bullets":
return block.items.map((item) => `- ${inlinePlain(item)}`).join("\n");
case "ordered":
return block.items.map((item, index) => `${index + 1}. ${inlinePlain(item)}`).join("\n");
default:
return inlinePlain(block.children);
}
})
.join("\n\n");
}
/** Résumé d'une ligne, pour les listes du back-office. */
export function richTextExcerpt(source: string, max = 160): string {
const plain = richTextToPlain(source).replace(/\s+/g, " ").trim();
if (plain.length <= max) return plain;
const cut = plain.slice(0, max);
const lastSpace = cut.lastIndexOf(" ");
return `${(lastSpace > max * 0.6 ? cut.slice(0, lastSpace) : cut).trimEnd()}…`;
}
+166
View File
@@ -0,0 +1,166 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import {
parseRichText,
richTextExcerpt,
richTextNodes,
richTextToEmailHtml,
richTextToPlain,
safeHttpUrl,
} from "../src/lib/rich-text";
/** Sérialise l'arbre React pour inspecter ce qui a réellement été construit. */
function serialize(source: string) {
return JSON.stringify(richTextNodes(source));
}
test("les paragraphes sont séparés par une ligne vide", () => {
const blocks = parseRichText("Premier.\n\nSecond.");
assert.equal(blocks.length, 2);
assert.equal(blocks[0].kind, "paragraph");
assert.equal(richTextToPlain("Premier.\n\nSecond."), "Premier.\n\nSecond.");
});
test("un saut de ligne simple reste un saut de ligne", () => {
const blocks = parseRichText("Une ligne\nune autre");
assert.equal(blocks.length, 1);
assert.deepEqual(
blocks[0].kind === "paragraph" ? blocks[0].children.map((c) => c.kind) : [],
["text", "break", "text"],
);
assert.ok(richTextToEmailHtml("Une ligne\nune autre").includes("<br>"));
});
test("gras et italique sont reconnus, et imbriqués", () => {
assert.ok(richTextToEmailHtml("**gras**").includes("<strong"));
assert.ok(richTextToEmailHtml("*doux*").includes("<em>doux</em>"));
assert.ok(richTextToEmailHtml("**du *doux* dedans**").includes("<em>doux</em>"));
});
test("une étoile non appariée reste littérale", () => {
assert.equal(richTextToPlain("**gras sans fin"), "**gras sans fin");
assert.equal(richTextToPlain("2 * 3 = 6"), "2 * 3 = 6");
assert.ok(!richTextToEmailHtml("**gras sans fin").includes("<strong"));
});
test("les listes à puces et numérotées regroupent leurs lignes", () => {
const blocks = parseRichText("- un\n- deux\n\n1. a\n2. b");
assert.equal(blocks.length, 2);
assert.equal(blocks[0].kind, "bullets");
assert.equal(blocks[0].kind === "bullets" ? blocks[0].items.length : 0, 2);
assert.equal(blocks[1].kind, "ordered");
const html = richTextToEmailHtml("- un\n- deux");
assert.equal(html.match(/<li/g)?.length, 2);
assert.ok(html.startsWith("<ul"));
});
test("un italique en début de ligne n'est pas confondu avec une puce", () => {
const blocks = parseRichText("*insistons* sur ce point");
assert.equal(blocks[0].kind, "paragraph");
});
test("un sous-titre devient un bloc à part", () => {
const blocks = parseRichText("## Ordre du jour\nPremier point");
assert.equal(blocks.length, 2);
assert.equal(blocks[0].kind, "heading");
assert.ok(richTextToEmailHtml("## Ordre du jour").includes("Georgia"));
});
test("un lien http(s) est rendu, en nouvel onglet et sans fuite d'ouvreur", () => {
const nodes = serialize("[le site](https://pleinr.fr/promotions)");
assert.ok(nodes.includes("https://pleinr.fr/promotions"));
assert.ok(nodes.includes("noopener noreferrer"));
assert.ok(nodes.includes('"target":"_blank"'));
assert.ok(richTextToEmailHtml("[le site](https://pleinr.fr)").includes('href="https://pleinr.fr/"'));
});
test("un lien javascript: ou data: perd sa cible et ne garde que son libellé", () => {
for (const hostile of [
"[clique ici](javascript:alert(1))",
"[clique ici](data:text/html,<script>alert(1)</script>)",
"[clique ici](vbscript:msgbox)",
"[clique ici]( JavaScript:alert(1) )",
]) {
const html = richTextToEmailHtml(hostile);
assert.ok(!html.includes("<a "), `cible conservée pour ${hostile}`);
assert.ok(!/javascript:/i.test(html), `protocole conservé pour ${hostile}`);
assert.ok(html.includes("clique ici"), "le libellé doit rester");
}
assert.equal(safeHttpUrl("javascript:alert(1)"), "");
assert.equal(safeHttpUrl("pas une url"), "");
assert.equal(safeHttpUrl("https://pleinr.fr"), "https://pleinr.fr/");
});
test("un crochet dans le libellé ne permet pas de sortir du lien", () => {
const html = richTextToEmailHtml("[a]b](https://pleinr.fr)");
assert.ok(!html.includes("<a "), "le lien malformé ne doit pas être construit");
assert.ok(html.includes("[a]b]"));
});
test("le HTML écrit dans la saisie est rendu littéralement", () => {
const payloads = [
"<script>alert(1)</script>",
'<img src=x onerror="alert(1)">',
"<svg/onload=alert(1)>",
'<iframe src="javascript:alert(1)"></iframe>',
"<ScRiPt>alert(1)</ScRiPt>",
'<a href="javascript:alert(1)">x</a>',
"<style>body{display:none}</style>",
];
// Les seules balises que le HTML d'e-mail a le droit de contenir sont
// celles que l'analyseur construit lui-même.
const OURS = /<\/?(?:div|strong|em|a|br|ul|ol|li)\b[^>]*>/g;
for (const payload of payloads) {
const html = richTextToEmailHtml(payload);
const leftovers = html.replace(OURS, "");
assert.ok(!leftovers.includes("<"), `balise conservée : ${payload}`);
assert.ok(!leftovers.includes(">"), `balise conservée : ${payload}`);
assert.ok(html.includes("&lt;"), `balise non échappée : ${payload}`);
// « onerror » survit comme texte échappé, et c'est sain : il n'est plus
// dans un contexte d'attribut, donc plus exécutable.
// Côté écran, le texte arrive tel quel dans un nœud React : c'est React
// qui échappe au rendu, la balise n'est jamais construite.
const tree = serialize(payload);
assert.ok(!tree.includes('"type":"script"'), `élément script construit : ${payload}`);
assert.ok(!tree.includes('"type":"img"'), `élément img construit : ${payload}`);
assert.ok(!tree.includes("onError"), `gestionnaire construit : ${payload}`);
}
});
test("le rendu n'emprunte jamais dangerouslySetInnerHTML", () => {
assert.ok(!serialize("**a** *b* [c](https://d.fr)\n- e\n\n## f").includes("dangerouslySetInnerHTML"));
});
test("les guillemets et esperluettes sont échappés dans le HTML d'e-mail", () => {
const html = richTextToEmailHtml('Marie & "Paul" <chez eux>');
assert.ok(html.includes("&amp;"));
assert.ok(html.includes("&quot;"));
assert.ok(html.includes("&lt;chez eux&gt;"));
});
test("le repli texte conserve le contenu et explicite les liens", () => {
assert.equal(
richTextToPlain("Voir **ici** : [les promos](https://pleinr.fr/promotions)"),
"Voir ici : les promos (https://pleinr.fr/promotions)",
);
assert.equal(richTextToPlain("- un\n- deux"), "- un\n- deux");
assert.equal(richTextToPlain("1. un\n2. deux"), "1. un\n2. deux");
});
test("le résumé coupe sur un mot et ajoute des points de suspension", () => {
assert.equal(richTextExcerpt("Court."), "Court.");
const long = richTextExcerpt("Vingt-deux chalets cette année, du 12 au 21 décembre prochain.", 30);
assert.ok(long.endsWith("…"));
assert.ok(long.length <= 31);
assert.ok(!long.includes(" "));
});
test("une saisie vide ne produit aucun bloc", () => {
assert.deepEqual(parseRichText(""), []);
assert.deepEqual(parseRichText(" \n\n "), []);
assert.equal(richTextToEmailHtml(""), "");
assert.equal(richTextExcerpt(""), "");
});