fix(backend): harden security and speed up feeds and article API

Security:
- WebSocket events are routed to their owner only (no cross-user leak);
  hub close is idempotent (fixes double-close panic), adds ping/pong and
  write deadlines.
- Session tokens stored as SHA-256 (migration 008 keeps sessions valid);
  single-query auth middleware puts the user in the request context.
- Client IP only trusts X-Forwarded-For from TRUSTED_PROXIES; rate limiter
  map is bounded; per-user limit on AI summaries.
- Argon2id at OWASP minimum with a concurrency cap; constant-time login
  for unknown emails; atomic first-admin bootstrap; REGISTRATION_ENABLED.
- CSP/HSTS/COOP headers, same-origin guard on mutations, body size limits,
  wider SSRF denylist, bounded feed/page/AI response reads, generic errors.
- Upgrade chi, pgx, x/net, x/text, x/crypto (known CVEs); commit go.sum.

Performance:
- List endpoints return a plain-text excerpt and reading time instead of
  full HTML; content is sanitized once at ingest (legacy rows backfilled).
- Keyset pagination on (sort_at, id) with matching partial indexes;
  redundant indexes dropped (migration 007).
- Fetcher: bounded worker pool, conditional GET (ETag/Last-Modified),
  exponential backoff, dedupe before insert, column-safe truncation,
  retention-aware ingest, per-user refresh coalescing.
- Read/favorite/read-all are single ownership-scoped statements.
- gzip compression, immutable caching for hashed assets, path-safe SPA
  handler, server timeouts; expired sessions purged.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Antigravity AgentandClaude Opus 5.5 committed 2026-10-09 07:34:08 +02:00
1 parent d037e2be34
commit 03e57e4308
40 files changed
+2231 -1898

No files matched your search

+189 -80
View File
@@ -3,11 +3,15 @@ package parser
import (
"context"
"crypto/sha256"
"encoding/hex"
"errors"
"fmt"
"io"
"net/http"
"time"
"net/url"
"strings"
"time"
"github.com/PuerkitoBio/goquery"
"github.com/google/uuid"
@@ -16,131 +20,228 @@ import (
"github.com/mmcdole/gofeed"
)
// ErrNotModified is returned when the server answered 304 to a conditional GET.
var ErrNotModified = errors.New("feed not modified")
// maxFeedBytes bounds how much of a feed document is read.
const maxFeedBytes = 10 << 20
// Column limits from the schema (VARCHAR sizes); values are truncated on a
// rune boundary so one oversized item can't abort the whole batch insert.
const (
maxTitle = 1024
maxFeedTitle = 512
maxAuthor = 256
maxURL = 2048
maxGUID = 512 // longer GUIDs are hashed
excerptRunes = 320
)
// FeedParser handles RSS/Atom feed parsing.
type FeedParser struct {
client *http.Client
parser *gofeed.Parser
client *http.Client
parser *gofeed.Parser
sanitizer *utils.ContentSanitizer
}
// NewFeedParser creates a new feed parser.
func NewFeedParser() *FeedParser {
return &FeedParser{
// SSRF-hardened client: refuses to connect to private/internal addresses.
client: utils.SafeHTTPClient(30 * time.Second),
parser: gofeed.NewParser(),
client: utils.SafeHTTPClient(30 * time.Second),
parser: gofeed.NewParser(),
sanitizer: utils.NewContentSanitizer(),
}
}
// ParsedFeed contains the parsed feed data.
type ParsedFeed struct {
Title string
Description string
SiteURL string
ImageURL string
Articles []*domain.Article
Title string
Description string
SiteURL string
ImageURL string
ETag string
LastModified string
Items []*Item
}
// Parse fetches and parses a feed URL.
func (p *FeedParser) Parse(ctx context.Context, feedURL string, feedID uuid.UUID) (*ParsedFeed, error) {
// Item is a feed entry whose GUID is known; the (more expensive) article
// conversion is deferred until the item is known to be new.
type Item struct {
GUID string
raw *gofeed.Item
}
// Parse fetches and parses a feed, sending the stored validators so an
// unchanged feed costs a 304 instead of a full download and parse.
func (p *FeedParser) Parse(ctx context.Context, feed *domain.Feed) (*ParsedFeed, error) {
// Validate up-front (scheme + non-private host) before issuing the request.
if _, err := utils.ValidateExternalURL(feedURL); err != nil {
if _, err := utils.ValidateExternalURL(feed.URL); err != nil {
return nil, err
}
// Create request with context
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feedURL, nil)
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feed.URL, nil)
if err != nil {
return nil, fmt.Errorf("creating request: %w", err)
}
req.Header.Set("User-Agent", "FlowReader/1.0 (RSS Reader)")
req.Header.Set("Accept", "application/rss+xml, application/atom+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.5")
if feed.ETag != "" {
req.Header.Set("If-None-Match", feed.ETag)
}
if feed.LastModified != "" {
req.Header.Set("If-Modified-Since", feed.LastModified)
}
// Fetch the feed
resp, err := p.client.Do(req)
if err != nil {
return nil, fmt.Errorf("fetching feed: %w", err)
}
defer resp.Body.Close()
if resp.StatusCode == http.StatusNotModified {
return nil, ErrNotModified
}
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("unexpected status code: %d", resp.StatusCode)
}
if resp.ContentLength > maxFeedBytes {
return nil, fmt.Errorf("feed too large")
}
// Parse the feed
feed, err := p.parser.Parse(resp.Body)
parsedDoc, err := p.parser.Parse(io.LimitReader(resp.Body, maxFeedBytes))
if err != nil {
return nil, fmt.Errorf("parsing feed: %w", err)
}
// Extract metadata
parsed := &ParsedFeed{
Title: feed.Title,
Description: feed.Description,
Title: utils.TruncateRunes(strings.TrimSpace(parsedDoc.Title), maxFeedTitle),
Description: utils.PlainText(parsedDoc.Description),
SiteURL: httpURL(parsedDoc.Link),
ETag: utils.TruncateRunes(resp.Header.Get("ETag"), 512),
LastModified: utils.TruncateRunes(resp.Header.Get("Last-Modified"), 128),
}
if parsedDoc.Image != nil {
parsed.ImageURL = httpURL(parsedDoc.Image.URL)
}
if feed.Link != "" {
parsed.SiteURL = feed.Link
}
if feed.Image != nil && feed.Image.URL != "" {
parsed.ImageURL = feed.Image.URL
}
// Convert items to articles
for _, item := range feed.Items {
article := &domain.Article{
ID: uuid.New(),
FeedID: feedID,
GUID: getGUID(item),
Title: item.Title,
seen := make(map[string]struct{}, len(parsedDoc.Items))
for _, item := range parsedDoc.Items {
guid := getGUID(item)
if guid == "" {
continue
}
if item.Link != "" {
article.URL = item.Link
if _, dup := seen[guid]; dup {
continue
}
if item.Content != "" {
article.Content = item.Content
}
if item.Description != "" {
article.Summary = item.Description
}
if item.Author != nil {
article.Author = item.Author.Name
} else if len(item.Authors) > 0 {
article.Author = item.Authors[0].Name
}
if item.Image != nil && item.Image.URL != "" {
article.ImageURL = item.Image.URL
} else {
article.ImageURL = findImage(item)
}
if item.PublishedParsed != nil {
article.PublishedAt = item.PublishedParsed
} else if item.UpdatedParsed != nil {
article.PublishedAt = item.UpdatedParsed
}
article.CreatedAt = time.Now()
parsed.Articles = append(parsed.Articles, article)
seen[guid] = struct{}{}
parsed.Items = append(parsed.Items, &Item{GUID: guid, raw: item})
}
return parsed, nil
}
// getGUID returns a unique identifier for the feed item.
// ToArticle converts a feed item into a sanitized article ready to insert.
func (p *FeedParser) ToArticle(it *Item, feedID uuid.UUID, now time.Time) *domain.Article {
item := it.raw
content := p.sanitizer.Sanitize(item.Content)
summary := p.sanitizer.Sanitize(item.Description)
plain := utils.PlainText(content)
if plain == "" {
plain = utils.PlainText(summary)
}
excerptSrc := utils.PlainText(summary)
if excerptSrc == "" {
excerptSrc = plain
}
title := strings.TrimSpace(utils.PlainText(item.Title))
if title == "" {
title = utils.Excerpt(plain, 80)
}
if title == "" {
title = "(sans titre)"
}
article := &domain.Article{
ID: uuid.New(),
FeedID: feedID,
GUID: it.GUID,
Title: utils.TruncateRunes(title, maxTitle),
URL: httpURL(item.Link),
Content: content,
Summary: summary,
Excerpt: utils.Excerpt(excerptSrc, excerptRunes),
WordCount: utils.WordCount(plain),
CreatedAt: now,
}
if item.Author != nil {
article.Author = item.Author.Name
} else if len(item.Authors) > 0 && item.Authors[0] != nil {
article.Author = item.Authors[0].Name
}
article.Author = utils.TruncateRunes(strings.TrimSpace(article.Author), maxAuthor)
if item.Image != nil && item.Image.URL != "" {
article.ImageURL = httpURL(item.Image.URL)
}
if article.ImageURL == "" {
article.ImageURL = httpURL(findImage(item))
}
if item.PublishedParsed != nil {
article.PublishedAt = item.PublishedParsed
} else if item.UpdatedParsed != nil {
article.PublishedAt = item.UpdatedParsed
}
// Clamp future dates (bad feed clocks) so they don't pin the top of the list.
if article.PublishedAt != nil && article.PublishedAt.After(now) {
t := now
article.PublishedAt = &t
}
return article
}
// PublishedAt returns the item's publication date, if any.
func (it *Item) PublishedAt() *time.Time {
if it.raw.PublishedParsed != nil {
return it.raw.PublishedParsed
}
return it.raw.UpdatedParsed
}
// httpURL keeps only absolute http(s) URLs within the column limit.
func httpURL(raw string) string {
raw = strings.TrimSpace(raw)
if raw == "" || len(raw) > maxURL {
return ""
}
u, err := url.Parse(raw)
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" {
return ""
}
return raw
}
// getGUID returns a unique identifier for the feed item, hashing values too
// long to index.
func getGUID(item *gofeed.Item) string {
if item.GUID != "" {
return item.GUID
guid := item.GUID
if guid == "" {
guid = item.Link
}
if item.Link != "" {
return item.Link
if guid == "" {
guid = item.Title // Last resort fallback
}
return item.Title // Last resort fallback
guid = strings.TrimSpace(guid)
if len(guid) > maxGUID {
sum := sha256.Sum256([]byte(guid))
return "sha256:" + hex.EncodeToString(sum[:])
}
return guid
}
// findImage attempts to find the best image for a feed item.
@@ -172,12 +273,20 @@ func findImage(item *gofeed.Item) string {
htmlContent = item.Description
}
if htmlContent != "" {
if htmlContent != "" && strings.Contains(htmlContent, "<img") {
doc, err := goquery.NewDocumentFromReader(strings.NewReader(htmlContent))
if err == nil {
if imgURL, exists := doc.Find("img").First().Attr("src"); exists {
return imgURL
}
var found string
doc.Find("img").EachWithBreak(func(_ int, s *goquery.Selection) bool {
src, _ := s.Attr("src")
// Skip tracking pixels.
if w, _ := s.Attr("width"); w == "1" {
return true
}
found = src
return src == ""
})
return found
}
}