mirror of
https://github.com/R0m1k3/FlowReader.git
synced 2026-10-11 17:28:05 +02:00
Security: - WebSocket events are routed to their owner only (no cross-user leak); hub close is idempotent (fixes double-close panic), adds ping/pong and write deadlines. - Session tokens stored as SHA-256 (migration 008 keeps sessions valid); single-query auth middleware puts the user in the request context. - Client IP only trusts X-Forwarded-For from TRUSTED_PROXIES; rate limiter map is bounded; per-user limit on AI summaries. - Argon2id at OWASP minimum with a concurrency cap; constant-time login for unknown emails; atomic first-admin bootstrap; REGISTRATION_ENABLED. - CSP/HSTS/COOP headers, same-origin guard on mutations, body size limits, wider SSRF denylist, bounded feed/page/AI response reads, generic errors. - Upgrade chi, pgx, x/net, x/text, x/crypto (known CVEs); commit go.sum. Performance: - List endpoints return a plain-text excerpt and reading time instead of full HTML; content is sanitized once at ingest (legacy rows backfilled). - Keyset pagination on (sort_at, id) with matching partial indexes; redundant indexes dropped (migration 007). - Fetcher: bounded worker pool, conditional GET (ETag/Last-Modified), exponential backoff, dedupe before insert, column-safe truncation, retention-aware ingest, per-user refresh coalescing. - Read/favorite/read-all are single ownership-scoped statements. - gzip compression, immutable caching for hashed assets, path-safe SPA handler, server timeouts; expired sessions purged. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
128 lines
3.2 KiB
Go
128 lines
3.2 KiB
Go
package utils
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/PuerkitoBio/goquery"
|
|
)
|
|
|
|
// maxPageBytes bounds how much of a remote page is read.
|
|
const maxPageBytes = 5 << 20
|
|
|
|
// ContentExtractor extracts the main text content from a web page.
|
|
type ContentExtractor struct {
|
|
client *http.Client
|
|
}
|
|
|
|
// NewContentExtractor creates a new extractor instance.
|
|
func NewContentExtractor() *ContentExtractor {
|
|
return &ContentExtractor{
|
|
// SSRF-hardened client: refuses to connect to private/internal addresses.
|
|
client: SafeHTTPClient(10 * time.Second),
|
|
}
|
|
}
|
|
|
|
// Extract fetches the URL and tries to extract the main article content.
|
|
func (e *ContentExtractor) Extract(ctx context.Context, url string) (string, error) {
|
|
if url == "" {
|
|
return "", fmt.Errorf("empty URL")
|
|
}
|
|
|
|
// Validate up-front (scheme + non-private host) before issuing the request.
|
|
if _, err := ValidateExternalURL(url); err != nil {
|
|
return "", err
|
|
}
|
|
|
|
req, err := http.NewRequestWithContext(ctx, "GET", url, nil)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
|
|
// Set a common User-Agent to avoid some basic bot detection
|
|
req.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
|
|
|
|
resp, err := e.client.Do(req)
|
|
if err != nil {
|
|
return "", fmt.Errorf("fetching URL: %w", err)
|
|
}
|
|
defer resp.Body.Close()
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
return "", fmt.Errorf("unexpected status code: %d", resp.StatusCode)
|
|
}
|
|
|
|
if resp.ContentLength > maxPageBytes {
|
|
return "", fmt.Errorf("page too large")
|
|
}
|
|
doc, err := goquery.NewDocumentFromReader(io.LimitReader(resp.Body, maxPageBytes))
|
|
if err != nil {
|
|
return "", fmt.Errorf("parsing HTML: %w", err)
|
|
}
|
|
|
|
// 1. Remove noise
|
|
doc.Find("script, style, nav, footer, header, aside, .ads, #comments, .sidebar").Remove()
|
|
|
|
// 2. Try to find the main content container
|
|
var content string
|
|
|
|
// Priority list of selectors for article content
|
|
selectors := []string{
|
|
"article",
|
|
"main",
|
|
".article-content",
|
|
".post-content",
|
|
".entry-content",
|
|
".content",
|
|
"#main-content",
|
|
}
|
|
|
|
for _, selector := range selectors {
|
|
selection := doc.Find(selector)
|
|
if selection.Length() > 0 {
|
|
// Get the longest piece of text if multiple matches
|
|
maxLength := 0
|
|
var bestSelection *goquery.Selection
|
|
selection.Each(func(i int, s *goquery.Selection) {
|
|
textLen := len(strings.TrimSpace(s.Text()))
|
|
if textLen > maxLength {
|
|
maxLength = textLen
|
|
bestSelection = s
|
|
}
|
|
})
|
|
if bestSelection != nil {
|
|
content = e.cleanText(bestSelection.Text())
|
|
if len(content) > 500 { // Heuristic: good enough
|
|
break
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// 3. Fallback: Body text if no container found or text too short
|
|
if len(content) < 500 {
|
|
content = e.cleanText(doc.Find("body").Text())
|
|
}
|
|
|
|
// Limit to 10000 characters to avoid huge payloads to AI
|
|
content = TruncateRunes(content, 10000)
|
|
|
|
return content, nil
|
|
}
|
|
|
|
func (e *ContentExtractor) cleanText(text string) string {
|
|
lines := strings.Split(text, "\n")
|
|
var cleaned []string
|
|
for _, line := range lines {
|
|
trimmed := strings.TrimSpace(line)
|
|
if trimmed != "" {
|
|
cleaned = append(cleaned, trimmed)
|
|
}
|
|
}
|
|
return strings.Join(cleaned, "\n")
|
|
}
|