Files
FlowReader/internal/utils/extractor.go
T
MichaelandClaude Opus 4.8 3402e53954 sécurité : correctifs XSS, SSRF, CSWSH, rate-limit et durcissement
- XSS stocké (critique) : sanitisation bluemonday sur tous les endpoints
  d'articles (List, ListByFeed, Favorites, Search, Get), pas seulement Get
- SSRF (élevé) : nouveau utils/safehttp.go (ValidateExternalURL + client durci
  via Dialer.Control, bloque IP privées/loopback/link-local/metadata,
  anti-DNS-rebinding et limite de redirections) appliqué à l'extracteur et au parser
- WebSocket CSWSH (élevé) : politique same-origin + override WS_ALLOWED_ORIGINS
- rate-limiting (moyen) : token-bucket en mémoire sur /auth/*
- token de session retiré du corps JSON (json:"-"), livré uniquement par le
  cookie HttpOnly
- cookie Secure correct derrière un reverse-proxy (X-Forwarded-Proto +
  override COOKIE_SECURE)
- admin : interdiction de supprimer son propre compte
- en-têtes de sécurité (nosniff, X-Frame-Options, Referrer-Policy,
  Permissions-Policy)

Note : backend non compilé localement (pas de toolchain Go) ; à valider via Docker.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-11 13:21:43 +02:00

123 lines
3.0 KiB
Go

package utils
import (
"context"
"fmt"
"net/http"
"strings"
"time"
"github.com/PuerkitoBio/goquery"
)
// ContentExtractor extracts the main text content from a web page.
type ContentExtractor struct {
client *http.Client
}
// NewContentExtractor creates a new extractor instance.
func NewContentExtractor() *ContentExtractor {
return &ContentExtractor{
// SSRF-hardened client: refuses to connect to private/internal addresses.
client: SafeHTTPClient(10 * time.Second),
}
}
// Extract fetches the URL and tries to extract the main article content.
func (e *ContentExtractor) Extract(ctx context.Context, url string) (string, error) {
if url == "" {
return "", fmt.Errorf("empty URL")
}
// Validate up-front (scheme + non-private host) before issuing the request.
if _, err := ValidateExternalURL(url); err != nil {
return "", err
}
req, err := http.NewRequestWithContext(ctx, "GET", url, nil)
if err != nil {
return "", err
}
// Set a common User-Agent to avoid some basic bot detection
req.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
resp, err := e.client.Do(req)
if err != nil {
return "", fmt.Errorf("fetching URL: %w", err)
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return "", fmt.Errorf("unexpected status code: %d", resp.StatusCode)
}
doc, err := goquery.NewDocumentFromReader(resp.Body)
if err != nil {
return "", fmt.Errorf("parsing HTML: %w", err)
}
// 1. Remove noise
doc.Find("script, style, nav, footer, header, aside, .ads, #comments, .sidebar").Remove()
// 2. Try to find the main content container
var content string
// Priority list of selectors for article content
selectors := []string{
"article",
"main",
".article-content",
".post-content",
".entry-content",
".content",
"#main-content",
}
for _, selector := range selectors {
selection := doc.Find(selector)
if selection.Length() > 0 {
// Get the longest piece of text if multiple matches
maxLength := 0
var bestSelection *goquery.Selection
selection.Each(func(i int, s *goquery.Selection) {
textLen := len(strings.TrimSpace(s.Text()))
if textLen > maxLength {
maxLength = textLen
bestSelection = s
}
})
if bestSelection != nil {
content = e.cleanText(bestSelection.Text())
if len(content) > 500 { // Heuristic: good enough
break
}
}
}
}
// 3. Fallback: Body text if no container found or text too short
if len(content) < 500 {
content = e.cleanText(doc.Find("body").Text())
}
// Limit to 10000 characters to avoid huge payloads to AI
if len(content) > 10000 {
content = content[:10000]
}
return content, nil
}
func (e *ContentExtractor) cleanText(text string) string {
lines := strings.Split(text, "\n")
var cleaned []string
for _, line := range lines {
trimmed := strings.TrimSpace(line)
if trimmed != "" {
cleaned = append(cleaned, trimmed)
}
}
return strings.Join(cleaned, "\n")
}