mirror of
https://github.com/R0m1k3/FlowReader.git
synced 2026-10-12 01:38:16 +02:00
Security: - WebSocket events are routed to their owner only (no cross-user leak); hub close is idempotent (fixes double-close panic), adds ping/pong and write deadlines. - Session tokens stored as SHA-256 (migration 008 keeps sessions valid); single-query auth middleware puts the user in the request context. - Client IP only trusts X-Forwarded-For from TRUSTED_PROXIES; rate limiter map is bounded; per-user limit on AI summaries. - Argon2id at OWASP minimum with a concurrency cap; constant-time login for unknown emails; atomic first-admin bootstrap; REGISTRATION_ENABLED. - CSP/HSTS/COOP headers, same-origin guard on mutations, body size limits, wider SSRF denylist, bounded feed/page/AI response reads, generic errors. - Upgrade chi, pgx, x/net, x/text, x/crypto (known CVEs); commit go.sum. Performance: - List endpoints return a plain-text excerpt and reading time instead of full HTML; content is sanitized once at ingest (legacy rows backfilled). - Keyset pagination on (sort_at, id) with matching partial indexes; redundant indexes dropped (migration 007). - Fetcher: bounded worker pool, conditional GET (ETag/Last-Modified), exponential backoff, dedupe before insert, column-safe truncation, retention-aware ingest, per-user refresh coalescing. - Read/favorite/read-all are single ownership-scoped statements. - gzip compression, immutable caching for hashed assets, path-safe SPA handler, server timeouts; expired sessions purged. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
295 lines
7.8 KiB
Go
295 lines
7.8 KiB
Go
// Package parser provides RSS/Atom feed parsing functionality.
|
|
package parser
|
|
|
|
import (
|
|
"context"
|
|
"crypto/sha256"
|
|
"encoding/hex"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/url"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/PuerkitoBio/goquery"
|
|
"github.com/google/uuid"
|
|
"github.com/michael/flowreader/internal/domain"
|
|
"github.com/michael/flowreader/internal/utils"
|
|
"github.com/mmcdole/gofeed"
|
|
)
|
|
|
|
// ErrNotModified is returned when the server answered 304 to a conditional GET.
|
|
var ErrNotModified = errors.New("feed not modified")
|
|
|
|
// maxFeedBytes bounds how much of a feed document is read.
|
|
const maxFeedBytes = 10 << 20
|
|
|
|
// Column limits from the schema (VARCHAR sizes); values are truncated on a
|
|
// rune boundary so one oversized item can't abort the whole batch insert.
|
|
const (
|
|
maxTitle = 1024
|
|
maxFeedTitle = 512
|
|
maxAuthor = 256
|
|
maxURL = 2048
|
|
maxGUID = 512 // longer GUIDs are hashed
|
|
excerptRunes = 320
|
|
)
|
|
|
|
// FeedParser handles RSS/Atom feed parsing.
|
|
type FeedParser struct {
|
|
client *http.Client
|
|
parser *gofeed.Parser
|
|
sanitizer *utils.ContentSanitizer
|
|
}
|
|
|
|
// NewFeedParser creates a new feed parser.
|
|
func NewFeedParser() *FeedParser {
|
|
return &FeedParser{
|
|
// SSRF-hardened client: refuses to connect to private/internal addresses.
|
|
client: utils.SafeHTTPClient(30 * time.Second),
|
|
parser: gofeed.NewParser(),
|
|
sanitizer: utils.NewContentSanitizer(),
|
|
}
|
|
}
|
|
|
|
// ParsedFeed contains the parsed feed data.
|
|
type ParsedFeed struct {
|
|
Title string
|
|
Description string
|
|
SiteURL string
|
|
ImageURL string
|
|
ETag string
|
|
LastModified string
|
|
Items []*Item
|
|
}
|
|
|
|
// Item is a feed entry whose GUID is known; the (more expensive) article
|
|
// conversion is deferred until the item is known to be new.
|
|
type Item struct {
|
|
GUID string
|
|
raw *gofeed.Item
|
|
}
|
|
|
|
// Parse fetches and parses a feed, sending the stored validators so an
|
|
// unchanged feed costs a 304 instead of a full download and parse.
|
|
func (p *FeedParser) Parse(ctx context.Context, feed *domain.Feed) (*ParsedFeed, error) {
|
|
// Validate up-front (scheme + non-private host) before issuing the request.
|
|
if _, err := utils.ValidateExternalURL(feed.URL); err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feed.URL, nil)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("creating request: %w", err)
|
|
}
|
|
req.Header.Set("User-Agent", "FlowReader/1.0 (RSS Reader)")
|
|
req.Header.Set("Accept", "application/rss+xml, application/atom+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.5")
|
|
if feed.ETag != "" {
|
|
req.Header.Set("If-None-Match", feed.ETag)
|
|
}
|
|
if feed.LastModified != "" {
|
|
req.Header.Set("If-Modified-Since", feed.LastModified)
|
|
}
|
|
|
|
resp, err := p.client.Do(req)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("fetching feed: %w", err)
|
|
}
|
|
defer resp.Body.Close()
|
|
|
|
if resp.StatusCode == http.StatusNotModified {
|
|
return nil, ErrNotModified
|
|
}
|
|
if resp.StatusCode != http.StatusOK {
|
|
return nil, fmt.Errorf("unexpected status code: %d", resp.StatusCode)
|
|
}
|
|
if resp.ContentLength > maxFeedBytes {
|
|
return nil, fmt.Errorf("feed too large")
|
|
}
|
|
|
|
parsedDoc, err := p.parser.Parse(io.LimitReader(resp.Body, maxFeedBytes))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("parsing feed: %w", err)
|
|
}
|
|
|
|
parsed := &ParsedFeed{
|
|
Title: utils.TruncateRunes(strings.TrimSpace(parsedDoc.Title), maxFeedTitle),
|
|
Description: utils.PlainText(parsedDoc.Description),
|
|
SiteURL: httpURL(parsedDoc.Link),
|
|
ETag: utils.TruncateRunes(resp.Header.Get("ETag"), 512),
|
|
LastModified: utils.TruncateRunes(resp.Header.Get("Last-Modified"), 128),
|
|
}
|
|
if parsedDoc.Image != nil {
|
|
parsed.ImageURL = httpURL(parsedDoc.Image.URL)
|
|
}
|
|
|
|
seen := make(map[string]struct{}, len(parsedDoc.Items))
|
|
for _, item := range parsedDoc.Items {
|
|
guid := getGUID(item)
|
|
if guid == "" {
|
|
continue
|
|
}
|
|
if _, dup := seen[guid]; dup {
|
|
continue
|
|
}
|
|
seen[guid] = struct{}{}
|
|
parsed.Items = append(parsed.Items, &Item{GUID: guid, raw: item})
|
|
}
|
|
|
|
return parsed, nil
|
|
}
|
|
|
|
// ToArticle converts a feed item into a sanitized article ready to insert.
|
|
func (p *FeedParser) ToArticle(it *Item, feedID uuid.UUID, now time.Time) *domain.Article {
|
|
item := it.raw
|
|
content := p.sanitizer.Sanitize(item.Content)
|
|
summary := p.sanitizer.Sanitize(item.Description)
|
|
|
|
plain := utils.PlainText(content)
|
|
if plain == "" {
|
|
plain = utils.PlainText(summary)
|
|
}
|
|
excerptSrc := utils.PlainText(summary)
|
|
if excerptSrc == "" {
|
|
excerptSrc = plain
|
|
}
|
|
|
|
title := strings.TrimSpace(utils.PlainText(item.Title))
|
|
if title == "" {
|
|
title = utils.Excerpt(plain, 80)
|
|
}
|
|
if title == "" {
|
|
title = "(sans titre)"
|
|
}
|
|
|
|
article := &domain.Article{
|
|
ID: uuid.New(),
|
|
FeedID: feedID,
|
|
GUID: it.GUID,
|
|
Title: utils.TruncateRunes(title, maxTitle),
|
|
URL: httpURL(item.Link),
|
|
Content: content,
|
|
Summary: summary,
|
|
Excerpt: utils.Excerpt(excerptSrc, excerptRunes),
|
|
WordCount: utils.WordCount(plain),
|
|
CreatedAt: now,
|
|
}
|
|
|
|
if item.Author != nil {
|
|
article.Author = item.Author.Name
|
|
} else if len(item.Authors) > 0 && item.Authors[0] != nil {
|
|
article.Author = item.Authors[0].Name
|
|
}
|
|
article.Author = utils.TruncateRunes(strings.TrimSpace(article.Author), maxAuthor)
|
|
|
|
if item.Image != nil && item.Image.URL != "" {
|
|
article.ImageURL = httpURL(item.Image.URL)
|
|
}
|
|
if article.ImageURL == "" {
|
|
article.ImageURL = httpURL(findImage(item))
|
|
}
|
|
|
|
if item.PublishedParsed != nil {
|
|
article.PublishedAt = item.PublishedParsed
|
|
} else if item.UpdatedParsed != nil {
|
|
article.PublishedAt = item.UpdatedParsed
|
|
}
|
|
// Clamp future dates (bad feed clocks) so they don't pin the top of the list.
|
|
if article.PublishedAt != nil && article.PublishedAt.After(now) {
|
|
t := now
|
|
article.PublishedAt = &t
|
|
}
|
|
|
|
return article
|
|
}
|
|
|
|
// PublishedAt returns the item's publication date, if any.
|
|
func (it *Item) PublishedAt() *time.Time {
|
|
if it.raw.PublishedParsed != nil {
|
|
return it.raw.PublishedParsed
|
|
}
|
|
return it.raw.UpdatedParsed
|
|
}
|
|
|
|
// httpURL keeps only absolute http(s) URLs within the column limit.
|
|
func httpURL(raw string) string {
|
|
raw = strings.TrimSpace(raw)
|
|
if raw == "" || len(raw) > maxURL {
|
|
return ""
|
|
}
|
|
u, err := url.Parse(raw)
|
|
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" {
|
|
return ""
|
|
}
|
|
return raw
|
|
}
|
|
|
|
// getGUID returns a unique identifier for the feed item, hashing values too
|
|
// long to index.
|
|
func getGUID(item *gofeed.Item) string {
|
|
guid := item.GUID
|
|
if guid == "" {
|
|
guid = item.Link
|
|
}
|
|
if guid == "" {
|
|
guid = item.Title // Last resort fallback
|
|
}
|
|
guid = strings.TrimSpace(guid)
|
|
if len(guid) > maxGUID {
|
|
sum := sha256.Sum256([]byte(guid))
|
|
return "sha256:" + hex.EncodeToString(sum[:])
|
|
}
|
|
return guid
|
|
}
|
|
|
|
// findImage attempts to find the best image for a feed item.
|
|
func findImage(item *gofeed.Item) string {
|
|
// 1. Check Enclosures
|
|
for _, enc := range item.Enclosures {
|
|
if strings.HasPrefix(enc.Type, "image/") {
|
|
return enc.URL
|
|
}
|
|
}
|
|
|
|
// 2. Check Media Extensions (media:content, media:thumbnail)
|
|
if media, ok := item.Extensions["media"]; ok {
|
|
if content, ok := media["content"]; ok && len(content) > 0 {
|
|
if url := content[0].Attrs["url"]; url != "" {
|
|
return url
|
|
}
|
|
}
|
|
if thumbnail, ok := media["thumbnail"]; ok && len(thumbnail) > 0 {
|
|
if url := thumbnail[0].Attrs["url"]; url != "" {
|
|
return url
|
|
}
|
|
}
|
|
}
|
|
|
|
// 3. Extract from Content/Description as fallback
|
|
htmlContent := item.Content
|
|
if htmlContent == "" {
|
|
htmlContent = item.Description
|
|
}
|
|
|
|
if htmlContent != "" && strings.Contains(htmlContent, "<img") {
|
|
doc, err := goquery.NewDocumentFromReader(strings.NewReader(htmlContent))
|
|
if err == nil {
|
|
var found string
|
|
doc.Find("img").EachWithBreak(func(_ int, s *goquery.Selection) bool {
|
|
src, _ := s.Attr("src")
|
|
// Skip tracking pixels.
|
|
if w, _ := s.Attr("width"); w == "1" {
|
|
return true
|
|
}
|
|
found = src
|
|
return src == ""
|
|
})
|
|
return found
|
|
}
|
|
}
|
|
|
|
return ""
|
|
}
|