Files
FlowReader/internal/utils/text.go
T
Antigravity AgentandClaude Opus 5.5 03e57e4308 fix(backend): harden security and speed up feeds and article API
Security:
- WebSocket events are routed to their owner only (no cross-user leak);
  hub close is idempotent (fixes double-close panic), adds ping/pong and
  write deadlines.
- Session tokens stored as SHA-256 (migration 008 keeps sessions valid);
  single-query auth middleware puts the user in the request context.
- Client IP only trusts X-Forwarded-For from TRUSTED_PROXIES; rate limiter
  map is bounded; per-user limit on AI summaries.
- Argon2id at OWASP minimum with a concurrency cap; constant-time login
  for unknown emails; atomic first-admin bootstrap; REGISTRATION_ENABLED.
- CSP/HSTS/COOP headers, same-origin guard on mutations, body size limits,
  wider SSRF denylist, bounded feed/page/AI response reads, generic errors.
- Upgrade chi, pgx, x/net, x/text, x/crypto (known CVEs); commit go.sum.

Performance:
- List endpoints return a plain-text excerpt and reading time instead of
  full HTML; content is sanitized once at ingest (legacy rows backfilled).
- Keyset pagination on (sort_at, id) with matching partial indexes;
  redundant indexes dropped (migration 007).
- Fetcher: bounded worker pool, conditional GET (ETag/Last-Modified),
  exponential backoff, dedupe before insert, column-safe truncation,
  retention-aware ingest, per-user refresh coalescing.
- Read/favorite/read-all are single ownership-scoped statements.
- gzip compression, immutable caching for hashed assets, path-safe SPA
  handler, server timeouts; expired sessions purged.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 07:34:08 +02:00

117 lines
2.5 KiB
Go

package utils
import (
"strings"
"unicode"
"unicode/utf8"
"golang.org/x/net/html"
)
// WordsPerMinute is the average silent reading speed for non-fiction
// (Brysbaert, 2019) used to estimate reading time.
const WordsPerMinute = 238
// PlainText converts an HTML fragment to whitespace-collapsed plain text.
// Content of script/style/noscript elements is dropped and entities are decoded.
func PlainText(fragment string) string {
if fragment == "" {
return ""
}
var b strings.Builder
b.Grow(len(fragment) / 2)
z := html.NewTokenizer(strings.NewReader(fragment))
skip := 0
for {
switch z.Next() {
case html.ErrorToken:
return collapseSpaces(b.String())
case html.StartTagToken:
name, _ := z.TagName()
switch string(name) {
case "script", "style", "noscript":
skip++
case "br", "p", "div", "li", "h1", "h2", "h3", "h4", "h5", "h6", "blockquote", "tr":
b.WriteByte(' ')
}
case html.EndTagToken:
name, _ := z.TagName()
switch string(name) {
case "script", "style", "noscript":
if skip > 0 {
skip--
}
case "p", "div", "li", "h1", "h2", "h3", "h4", "h5", "h6", "blockquote", "tr", "td":
b.WriteByte(' ')
}
case html.TextToken:
if skip == 0 {
b.Write(z.Text())
}
}
}
}
func collapseSpaces(s string) string {
var b strings.Builder
b.Grow(len(s))
space := false
for _, r := range s {
if unicode.IsSpace(r) {
space = true
continue
}
if space && b.Len() > 0 {
b.WriteByte(' ')
}
space = false
b.WriteRune(r)
}
return b.String()
}
// WordCount counts whitespace-separated words in plain text.
func WordCount(plain string) int {
return len(strings.Fields(plain))
}
// ReadingMinutes converts a word count into a rounded-up reading time (min 1).
func ReadingMinutes(words int) int {
if words <= 0 {
return 1
}
return (words + WordsPerMinute - 1) / WordsPerMinute
}
// Excerpt returns at most maxRunes runes of plain text, cut on a word
// boundary when possible and suffixed with an ellipsis when truncated.
func Excerpt(plain string, maxRunes int) string {
if utf8.RuneCountInString(plain) <= maxRunes {
return plain
}
cut := TruncateRunes(plain, maxRunes)
if i := strings.LastIndexByte(cut, ' '); i > len(cut)*2/3 {
cut = cut[:i]
}
return strings.TrimRight(cut, " ,;:.-–—") + "…"
}
// TruncateRunes truncates s to at most n runes without splitting a UTF-8
// sequence.
func TruncateRunes(s string, n int) string {
if n <= 0 {
return ""
}
if len(s) <= n {
return s
}
i := 0
for pos := range s {
if i == n {
return s[:pos]
}
i++
}
return s
}