mirror of
https://github.com/R0m1k3/FlowReader.git
synced 2026-10-11 17:28:05 +02:00
feat: implement full content extraction for improved AI summaries
This commit is contained in:
1 parent
eb2c938a2a
commit
b3cbaf741d
2 files changed
+128
No files matched your search
@@ -20,6 +20,7 @@ type ArticleHandler struct {
|
||||
authService *service.AuthService
|
||||
aiService *service.AIService
|
||||
sanitizer *utils.ContentSanitizer
|
||||
extractor *utils.ContentExtractor
|
||||
hub *ws.Hub
|
||||
}
|
||||
|
||||
@@ -31,6 +32,7 @@ func NewArticleHandler(articleRepo domain.ArticleRepository, feedService *servic
|
||||
authService: authService,
|
||||
aiService: aiService,
|
||||
sanitizer: utils.NewContentSanitizer(),
|
||||
extractor: utils.NewContentExtractor(),
|
||||
hub: hub,
|
||||
}
|
||||
}
|
||||
@@ -431,6 +433,14 @@ func (h *ArticleHandler) Summarize(w http.ResponseWriter, r *http.Request) {
|
||||
content = article.Summary
|
||||
}
|
||||
|
||||
// Try to extract full content from URL if available
|
||||
if article.URL != "" {
|
||||
fullContent, err := h.extractor.Extract(r.Context(), article.URL)
|
||||
if err == nil && len(fullContent) > len(content) {
|
||||
content = "--- CONTENU COMPLET EXTRAIT DU SITE WEB ---\n" + fullContent
|
||||
}
|
||||
}
|
||||
|
||||
aiInput := fmt.Sprintf("Titre: %s\n\nContenu: %s", article.Title, content)
|
||||
|
||||
// Summary generation (can be slow, but for this demo/small app we do it synchronously
|
||||
|
||||
Reference in new issue
Block a user