mirror of
https://github.com/R0m1k3/FlowReader.git
synced 2026-10-11 17:28:05 +02:00
fix(backend): harden security and speed up feeds and article API
Security: - WebSocket events are routed to their owner only (no cross-user leak); hub close is idempotent (fixes double-close panic), adds ping/pong and write deadlines. - Session tokens stored as SHA-256 (migration 008 keeps sessions valid); single-query auth middleware puts the user in the request context. - Client IP only trusts X-Forwarded-For from TRUSTED_PROXIES; rate limiter map is bounded; per-user limit on AI summaries. - Argon2id at OWASP minimum with a concurrency cap; constant-time login for unknown emails; atomic first-admin bootstrap; REGISTRATION_ENABLED. - CSP/HSTS/COOP headers, same-origin guard on mutations, body size limits, wider SSRF denylist, bounded feed/page/AI response reads, generic errors. - Upgrade chi, pgx, x/net, x/text, x/crypto (known CVEs); commit go.sum. Performance: - List endpoints return a plain-text excerpt and reading time instead of full HTML; content is sanitized once at ingest (legacy rows backfilled). - Keyset pagination on (sort_at, id) with matching partial indexes; redundant indexes dropped (migration 007). - Fetcher: bounded worker pool, conditional GET (ETag/Last-Modified), exponential backoff, dedupe before insert, column-safe truncation, retention-aware ingest, per-user refresh coalescing. - Read/favorite/read-all are single ownership-scoped statements. - gzip compression, immutable caching for hashed assets, path-safe SPA handler, server timeouts; expired sessions purged. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
d037e2be34
commit
03e57e4308
40 files changed
+2231
-1898
No files matched your search
+189
-80
@@ -3,11 +3,15 @@ package parser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/PuerkitoBio/goquery"
|
||||
"github.com/google/uuid"
|
||||
@@ -16,131 +20,228 @@ import (
|
||||
"github.com/mmcdole/gofeed"
|
||||
)
|
||||
|
||||
// ErrNotModified is returned when the server answered 304 to a conditional GET.
|
||||
var ErrNotModified = errors.New("feed not modified")
|
||||
|
||||
// maxFeedBytes bounds how much of a feed document is read.
|
||||
const maxFeedBytes = 10 << 20
|
||||
|
||||
// Column limits from the schema (VARCHAR sizes); values are truncated on a
|
||||
// rune boundary so one oversized item can't abort the whole batch insert.
|
||||
const (
|
||||
maxTitle = 1024
|
||||
maxFeedTitle = 512
|
||||
maxAuthor = 256
|
||||
maxURL = 2048
|
||||
maxGUID = 512 // longer GUIDs are hashed
|
||||
excerptRunes = 320
|
||||
)
|
||||
|
||||
// FeedParser handles RSS/Atom feed parsing.
|
||||
type FeedParser struct {
|
||||
client *http.Client
|
||||
parser *gofeed.Parser
|
||||
client *http.Client
|
||||
parser *gofeed.Parser
|
||||
sanitizer *utils.ContentSanitizer
|
||||
}
|
||||
|
||||
// NewFeedParser creates a new feed parser.
|
||||
func NewFeedParser() *FeedParser {
|
||||
return &FeedParser{
|
||||
// SSRF-hardened client: refuses to connect to private/internal addresses.
|
||||
client: utils.SafeHTTPClient(30 * time.Second),
|
||||
parser: gofeed.NewParser(),
|
||||
client: utils.SafeHTTPClient(30 * time.Second),
|
||||
parser: gofeed.NewParser(),
|
||||
sanitizer: utils.NewContentSanitizer(),
|
||||
}
|
||||
}
|
||||
|
||||
// ParsedFeed contains the parsed feed data.
|
||||
type ParsedFeed struct {
|
||||
Title string
|
||||
Description string
|
||||
SiteURL string
|
||||
ImageURL string
|
||||
Articles []*domain.Article
|
||||
Title string
|
||||
Description string
|
||||
SiteURL string
|
||||
ImageURL string
|
||||
ETag string
|
||||
LastModified string
|
||||
Items []*Item
|
||||
}
|
||||
|
||||
// Parse fetches and parses a feed URL.
|
||||
func (p *FeedParser) Parse(ctx context.Context, feedURL string, feedID uuid.UUID) (*ParsedFeed, error) {
|
||||
// Item is a feed entry whose GUID is known; the (more expensive) article
|
||||
// conversion is deferred until the item is known to be new.
|
||||
type Item struct {
|
||||
GUID string
|
||||
raw *gofeed.Item
|
||||
}
|
||||
|
||||
// Parse fetches and parses a feed, sending the stored validators so an
|
||||
// unchanged feed costs a 304 instead of a full download and parse.
|
||||
func (p *FeedParser) Parse(ctx context.Context, feed *domain.Feed) (*ParsedFeed, error) {
|
||||
// Validate up-front (scheme + non-private host) before issuing the request.
|
||||
if _, err := utils.ValidateExternalURL(feedURL); err != nil {
|
||||
if _, err := utils.ValidateExternalURL(feed.URL); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Create request with context
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feedURL, nil)
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feed.URL, nil)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("creating request: %w", err)
|
||||
}
|
||||
req.Header.Set("User-Agent", "FlowReader/1.0 (RSS Reader)")
|
||||
req.Header.Set("Accept", "application/rss+xml, application/atom+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.5")
|
||||
if feed.ETag != "" {
|
||||
req.Header.Set("If-None-Match", feed.ETag)
|
||||
}
|
||||
if feed.LastModified != "" {
|
||||
req.Header.Set("If-Modified-Since", feed.LastModified)
|
||||
}
|
||||
|
||||
// Fetch the feed
|
||||
resp, err := p.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("fetching feed: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode == http.StatusNotModified {
|
||||
return nil, ErrNotModified
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, fmt.Errorf("unexpected status code: %d", resp.StatusCode)
|
||||
}
|
||||
if resp.ContentLength > maxFeedBytes {
|
||||
return nil, fmt.Errorf("feed too large")
|
||||
}
|
||||
|
||||
// Parse the feed
|
||||
feed, err := p.parser.Parse(resp.Body)
|
||||
parsedDoc, err := p.parser.Parse(io.LimitReader(resp.Body, maxFeedBytes))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing feed: %w", err)
|
||||
}
|
||||
|
||||
// Extract metadata
|
||||
parsed := &ParsedFeed{
|
||||
Title: feed.Title,
|
||||
Description: feed.Description,
|
||||
Title: utils.TruncateRunes(strings.TrimSpace(parsedDoc.Title), maxFeedTitle),
|
||||
Description: utils.PlainText(parsedDoc.Description),
|
||||
SiteURL: httpURL(parsedDoc.Link),
|
||||
ETag: utils.TruncateRunes(resp.Header.Get("ETag"), 512),
|
||||
LastModified: utils.TruncateRunes(resp.Header.Get("Last-Modified"), 128),
|
||||
}
|
||||
if parsedDoc.Image != nil {
|
||||
parsed.ImageURL = httpURL(parsedDoc.Image.URL)
|
||||
}
|
||||
|
||||
if feed.Link != "" {
|
||||
parsed.SiteURL = feed.Link
|
||||
}
|
||||
|
||||
if feed.Image != nil && feed.Image.URL != "" {
|
||||
parsed.ImageURL = feed.Image.URL
|
||||
}
|
||||
|
||||
// Convert items to articles
|
||||
for _, item := range feed.Items {
|
||||
article := &domain.Article{
|
||||
ID: uuid.New(),
|
||||
FeedID: feedID,
|
||||
GUID: getGUID(item),
|
||||
Title: item.Title,
|
||||
seen := make(map[string]struct{}, len(parsedDoc.Items))
|
||||
for _, item := range parsedDoc.Items {
|
||||
guid := getGUID(item)
|
||||
if guid == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
if item.Link != "" {
|
||||
article.URL = item.Link
|
||||
if _, dup := seen[guid]; dup {
|
||||
continue
|
||||
}
|
||||
|
||||
if item.Content != "" {
|
||||
article.Content = item.Content
|
||||
}
|
||||
|
||||
if item.Description != "" {
|
||||
article.Summary = item.Description
|
||||
}
|
||||
|
||||
if item.Author != nil {
|
||||
article.Author = item.Author.Name
|
||||
} else if len(item.Authors) > 0 {
|
||||
article.Author = item.Authors[0].Name
|
||||
}
|
||||
|
||||
if item.Image != nil && item.Image.URL != "" {
|
||||
article.ImageURL = item.Image.URL
|
||||
} else {
|
||||
article.ImageURL = findImage(item)
|
||||
}
|
||||
|
||||
if item.PublishedParsed != nil {
|
||||
article.PublishedAt = item.PublishedParsed
|
||||
} else if item.UpdatedParsed != nil {
|
||||
article.PublishedAt = item.UpdatedParsed
|
||||
}
|
||||
|
||||
article.CreatedAt = time.Now()
|
||||
|
||||
parsed.Articles = append(parsed.Articles, article)
|
||||
seen[guid] = struct{}{}
|
||||
parsed.Items = append(parsed.Items, &Item{GUID: guid, raw: item})
|
||||
}
|
||||
|
||||
return parsed, nil
|
||||
}
|
||||
|
||||
// getGUID returns a unique identifier for the feed item.
|
||||
// ToArticle converts a feed item into a sanitized article ready to insert.
|
||||
func (p *FeedParser) ToArticle(it *Item, feedID uuid.UUID, now time.Time) *domain.Article {
|
||||
item := it.raw
|
||||
content := p.sanitizer.Sanitize(item.Content)
|
||||
summary := p.sanitizer.Sanitize(item.Description)
|
||||
|
||||
plain := utils.PlainText(content)
|
||||
if plain == "" {
|
||||
plain = utils.PlainText(summary)
|
||||
}
|
||||
excerptSrc := utils.PlainText(summary)
|
||||
if excerptSrc == "" {
|
||||
excerptSrc = plain
|
||||
}
|
||||
|
||||
title := strings.TrimSpace(utils.PlainText(item.Title))
|
||||
if title == "" {
|
||||
title = utils.Excerpt(plain, 80)
|
||||
}
|
||||
if title == "" {
|
||||
title = "(sans titre)"
|
||||
}
|
||||
|
||||
article := &domain.Article{
|
||||
ID: uuid.New(),
|
||||
FeedID: feedID,
|
||||
GUID: it.GUID,
|
||||
Title: utils.TruncateRunes(title, maxTitle),
|
||||
URL: httpURL(item.Link),
|
||||
Content: content,
|
||||
Summary: summary,
|
||||
Excerpt: utils.Excerpt(excerptSrc, excerptRunes),
|
||||
WordCount: utils.WordCount(plain),
|
||||
CreatedAt: now,
|
||||
}
|
||||
|
||||
if item.Author != nil {
|
||||
article.Author = item.Author.Name
|
||||
} else if len(item.Authors) > 0 && item.Authors[0] != nil {
|
||||
article.Author = item.Authors[0].Name
|
||||
}
|
||||
article.Author = utils.TruncateRunes(strings.TrimSpace(article.Author), maxAuthor)
|
||||
|
||||
if item.Image != nil && item.Image.URL != "" {
|
||||
article.ImageURL = httpURL(item.Image.URL)
|
||||
}
|
||||
if article.ImageURL == "" {
|
||||
article.ImageURL = httpURL(findImage(item))
|
||||
}
|
||||
|
||||
if item.PublishedParsed != nil {
|
||||
article.PublishedAt = item.PublishedParsed
|
||||
} else if item.UpdatedParsed != nil {
|
||||
article.PublishedAt = item.UpdatedParsed
|
||||
}
|
||||
// Clamp future dates (bad feed clocks) so they don't pin the top of the list.
|
||||
if article.PublishedAt != nil && article.PublishedAt.After(now) {
|
||||
t := now
|
||||
article.PublishedAt = &t
|
||||
}
|
||||
|
||||
return article
|
||||
}
|
||||
|
||||
// PublishedAt returns the item's publication date, if any.
|
||||
func (it *Item) PublishedAt() *time.Time {
|
||||
if it.raw.PublishedParsed != nil {
|
||||
return it.raw.PublishedParsed
|
||||
}
|
||||
return it.raw.UpdatedParsed
|
||||
}
|
||||
|
||||
// httpURL keeps only absolute http(s) URLs within the column limit.
|
||||
func httpURL(raw string) string {
|
||||
raw = strings.TrimSpace(raw)
|
||||
if raw == "" || len(raw) > maxURL {
|
||||
return ""
|
||||
}
|
||||
u, err := url.Parse(raw)
|
||||
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" {
|
||||
return ""
|
||||
}
|
||||
return raw
|
||||
}
|
||||
|
||||
// getGUID returns a unique identifier for the feed item, hashing values too
|
||||
// long to index.
|
||||
func getGUID(item *gofeed.Item) string {
|
||||
if item.GUID != "" {
|
||||
return item.GUID
|
||||
guid := item.GUID
|
||||
if guid == "" {
|
||||
guid = item.Link
|
||||
}
|
||||
if item.Link != "" {
|
||||
return item.Link
|
||||
if guid == "" {
|
||||
guid = item.Title // Last resort fallback
|
||||
}
|
||||
return item.Title // Last resort fallback
|
||||
guid = strings.TrimSpace(guid)
|
||||
if len(guid) > maxGUID {
|
||||
sum := sha256.Sum256([]byte(guid))
|
||||
return "sha256:" + hex.EncodeToString(sum[:])
|
||||
}
|
||||
return guid
|
||||
}
|
||||
|
||||
// findImage attempts to find the best image for a feed item.
|
||||
@@ -172,12 +273,20 @@ func findImage(item *gofeed.Item) string {
|
||||
htmlContent = item.Description
|
||||
}
|
||||
|
||||
if htmlContent != "" {
|
||||
if htmlContent != "" && strings.Contains(htmlContent, "<img") {
|
||||
doc, err := goquery.NewDocumentFromReader(strings.NewReader(htmlContent))
|
||||
if err == nil {
|
||||
if imgURL, exists := doc.Find("img").First().Attr("src"); exists {
|
||||
return imgURL
|
||||
}
|
||||
var found string
|
||||
doc.Find("img").EachWithBreak(func(_ int, s *goquery.Selection) bool {
|
||||
src, _ := s.Attr("src")
|
||||
// Skip tracking pixels.
|
||||
if w, _ := s.Attr("width"); w == "1" {
|
||||
return true
|
||||
}
|
||||
found = src
|
||||
return src == ""
|
||||
})
|
||||
return found
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user