mirror of
https://github.com/R0m1k3/noteflow.git
synced 2026-10-11 17:29:37 +02:00
Merge pull request #75 from R0m1k3/claude/update-rss-feeds-011CV6EZDsWAUqbRHZR1117Q
Fix: Amélioration détection doublons RSS avec titre+date
This commit is contained in:
3 files changed
+210
-5
No files matched your search
@@ -0,0 +1,84 @@
|
|||||||
|
// Script pour nettoyer les doublons et analyser les liens
|
||||||
|
const { getAll, runQuery, initDatabase } = require('../config/database');
|
||||||
|
|
||||||
|
async function cleanupDuplicates() {
|
||||||
|
console.log('\n==================== NETTOYAGE DOUBLONS RSS ====================\n');
|
||||||
|
|
||||||
|
try {
|
||||||
|
await initDatabase();
|
||||||
|
|
||||||
|
// Afficher les articles avec le même titre mais des liens différents
|
||||||
|
console.log('🔍 Recherche de doublons par titre:\n');
|
||||||
|
|
||||||
|
const duplicates = await getAll(`
|
||||||
|
SELECT
|
||||||
|
title,
|
||||||
|
COUNT(*) as count,
|
||||||
|
GROUP_CONCAT(link, '|||') as links,
|
||||||
|
GROUP_CONCAT(pub_date, '|||') as dates
|
||||||
|
FROM rss_articles
|
||||||
|
GROUP BY title
|
||||||
|
HAVING count > 1
|
||||||
|
ORDER BY count DESC
|
||||||
|
LIMIT 10
|
||||||
|
`);
|
||||||
|
|
||||||
|
if (duplicates.length > 0) {
|
||||||
|
console.log(`⚠️ Trouvé ${duplicates.length} titres en double:\n`);
|
||||||
|
|
||||||
|
duplicates.forEach((dup, i) => {
|
||||||
|
console.log(`${i + 1}. "${dup.title}"`);
|
||||||
|
console.log(` Nombre de doublons: ${dup.count}`);
|
||||||
|
|
||||||
|
const links = dup.links.split('|||');
|
||||||
|
const dates = dup.dates.split('|||');
|
||||||
|
|
||||||
|
links.forEach((link, j) => {
|
||||||
|
console.log(` ${j + 1}. ${link.substring(0, 80)}...`);
|
||||||
|
console.log(` Date: ${new Date(dates[j]).toLocaleString('fr-FR')}`);
|
||||||
|
});
|
||||||
|
console.log('');
|
||||||
|
});
|
||||||
|
|
||||||
|
// Proposer le nettoyage
|
||||||
|
console.log('💡 Pour nettoyer, garder uniquement l\'article le plus récent par titre.\n');
|
||||||
|
|
||||||
|
} else {
|
||||||
|
console.log('✓ Aucun doublon trouvé par titre\n');
|
||||||
|
}
|
||||||
|
|
||||||
|
// Analyser les patterns de liens
|
||||||
|
console.log('🔗 Analyse des patterns de liens:\n');
|
||||||
|
|
||||||
|
const sampleLinks = await getAll(`
|
||||||
|
SELECT link, title
|
||||||
|
FROM rss_articles
|
||||||
|
ORDER BY created_at DESC
|
||||||
|
LIMIT 5
|
||||||
|
`);
|
||||||
|
|
||||||
|
sampleLinks.forEach((article, i) => {
|
||||||
|
console.log(`${i + 1}. ${article.title.substring(0, 50)}...`);
|
||||||
|
console.log(` ${article.link}`);
|
||||||
|
|
||||||
|
// Extraire le domaine et les paramètres
|
||||||
|
try {
|
||||||
|
const url = new URL(article.link);
|
||||||
|
console.log(` Domaine: ${url.hostname}`);
|
||||||
|
console.log(` Params: ${url.search}`);
|
||||||
|
} catch (e) {
|
||||||
|
console.log(` ⚠️ URL invalide`);
|
||||||
|
}
|
||||||
|
console.log('');
|
||||||
|
});
|
||||||
|
|
||||||
|
console.log('========================================================\n');
|
||||||
|
|
||||||
|
} catch (error) {
|
||||||
|
console.error('❌ Erreur:', error);
|
||||||
|
}
|
||||||
|
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanupDuplicates();
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
// Script de debug approfondi pour la récupération RSS
|
||||||
|
const Parser = require('rss-parser');
|
||||||
|
const { getAll, getOne, runQuery, initDatabase } = require('../config/database');
|
||||||
|
|
||||||
|
const parser = new Parser({
|
||||||
|
timeout: 10000,
|
||||||
|
headers: {
|
||||||
|
'User-Agent': 'NoteFlow RSS Reader'
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
const TEST_FEED = 'https://news.google.com/rss/search?tbm=nws&q=NBA&oq=NBA&scoring=n&hl=fr&gl=FR&ceid=FR:fr';
|
||||||
|
|
||||||
|
async function debugFetch() {
|
||||||
|
console.log('\n==================== DEBUG RÉCUPÉRATION RSS ====================\n');
|
||||||
|
|
||||||
|
try {
|
||||||
|
await initDatabase();
|
||||||
|
|
||||||
|
console.log('📡 Récupération du flux:', TEST_FEED);
|
||||||
|
console.log('');
|
||||||
|
|
||||||
|
const parsedFeed = await parser.parseURL(TEST_FEED);
|
||||||
|
|
||||||
|
console.log(`✅ Flux récupéré: ${parsedFeed.title}`);
|
||||||
|
console.log(`📊 Nombre d'articles dans le flux: ${parsedFeed.items.length}\n`);
|
||||||
|
|
||||||
|
// Vérifier les 10 premiers articles
|
||||||
|
console.log('🔍 Analyse des 10 premiers articles:\n');
|
||||||
|
|
||||||
|
for (let i = 0; i < Math.min(10, parsedFeed.items.length); i++) {
|
||||||
|
const item = parsedFeed.items[i];
|
||||||
|
|
||||||
|
console.log(`${i + 1}. ${item.title}`);
|
||||||
|
console.log(` Date pubDate: ${item.pubDate || 'N/A'}`);
|
||||||
|
console.log(` Date isoDate: ${item.isoDate || 'N/A'}`);
|
||||||
|
|
||||||
|
// Parser la date
|
||||||
|
const pubDate = item.pubDate || item.isoDate || new Date().toISOString();
|
||||||
|
const parsedDate = new Date(pubDate);
|
||||||
|
console.log(` Date parsée: ${parsedDate.toLocaleString('fr-FR')}`);
|
||||||
|
console.log(` Est valide: ${!isNaN(parsedDate.getTime())}`);
|
||||||
|
|
||||||
|
// Vérifier le lien
|
||||||
|
console.log(` Lien: ${item.link.substring(0, 100)}...`);
|
||||||
|
|
||||||
|
// Vérifier si existe en DB
|
||||||
|
const existing = await getOne('SELECT id, pub_date, created_at FROM rss_articles WHERE link = ?', [item.link]);
|
||||||
|
|
||||||
|
if (existing) {
|
||||||
|
const dbDate = new Date(existing.pub_date);
|
||||||
|
console.log(` ⚠️ EXISTE DÉJÀ en DB (id: ${existing.id})`);
|
||||||
|
console.log(` Date DB: ${dbDate.toLocaleString('fr-FR')}`);
|
||||||
|
} else {
|
||||||
|
console.log(` ✓ NOUVEAU (pas en DB)`);
|
||||||
|
}
|
||||||
|
|
||||||
|
console.log('');
|
||||||
|
}
|
||||||
|
|
||||||
|
// Statistiques DB
|
||||||
|
console.log('📊 Statistiques base de données:\n');
|
||||||
|
|
||||||
|
const totalArticles = await getAll('SELECT COUNT(*) as count FROM rss_articles');
|
||||||
|
console.log(`Total articles en DB: ${totalArticles[0]?.count || 0}`);
|
||||||
|
|
||||||
|
const articlesToday = await getAll(`
|
||||||
|
SELECT COUNT(*) as count
|
||||||
|
FROM rss_articles
|
||||||
|
WHERE DATE(pub_date) = DATE('now')
|
||||||
|
`);
|
||||||
|
console.log(`Articles d'aujourd'hui (13 nov): ${articlesToday[0]?.count || 0}`);
|
||||||
|
|
||||||
|
const articlesYesterday = await getAll(`
|
||||||
|
SELECT COUNT(*) as count
|
||||||
|
FROM rss_articles
|
||||||
|
WHERE DATE(pub_date) = DATE('now', '-1 day')
|
||||||
|
`);
|
||||||
|
console.log(`Articles d'hier (12 nov): ${articlesYesterday[0]?.count || 0}`);
|
||||||
|
|
||||||
|
// Afficher les 5 plus récents en DB
|
||||||
|
console.log('\n📅 5 articles les plus récents en DB:\n');
|
||||||
|
const recent = await getAll(`
|
||||||
|
SELECT title, pub_date, created_at
|
||||||
|
FROM rss_articles
|
||||||
|
ORDER BY pub_date DESC
|
||||||
|
LIMIT 5
|
||||||
|
`);
|
||||||
|
|
||||||
|
recent.forEach((article, i) => {
|
||||||
|
const pubDate = new Date(article.pub_date);
|
||||||
|
const createdDate = new Date(article.created_at);
|
||||||
|
console.log(`${i + 1}. ${article.title}`);
|
||||||
|
console.log(` Publié: ${pubDate.toLocaleString('fr-FR')}`);
|
||||||
|
console.log(` Ajouté: ${createdDate.toLocaleString('fr-FR')}`);
|
||||||
|
console.log('');
|
||||||
|
});
|
||||||
|
|
||||||
|
console.log('========================================================\n');
|
||||||
|
|
||||||
|
} catch (error) {
|
||||||
|
console.error('❌ Erreur:', error);
|
||||||
|
}
|
||||||
|
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
debugFetch();
|
||||||
@@ -114,10 +114,23 @@ async function fetchAllFeeds() {
|
|||||||
try {
|
try {
|
||||||
if (!item.link) continue; // Skip articles sans lien
|
if (!item.link) continue; // Skip articles sans lien
|
||||||
|
|
||||||
// Vérifier si l'article existe déjà
|
// Normaliser la date pour la comparaison
|
||||||
const existing = await getOne('SELECT id FROM rss_articles WHERE link = ?', [item.link]);
|
const pubDate = item.pubDate || item.isoDate || new Date().toISOString();
|
||||||
|
|
||||||
if (!existing) {
|
// Vérifier si l'article existe déjà par lien OU par titre+date
|
||||||
|
// Permet de gérer les liens qui changent (tracking) et les vrais doublons
|
||||||
|
const existingByLink = await getOne(
|
||||||
|
'SELECT id FROM rss_articles WHERE link = ?',
|
||||||
|
[item.link]
|
||||||
|
);
|
||||||
|
|
||||||
|
const existingByTitleDate = await getOne(
|
||||||
|
'SELECT id FROM rss_articles WHERE feed_id = ? AND title = ? AND DATE(pub_date) = DATE(?)',
|
||||||
|
[feed.id, item.title, pubDate]
|
||||||
|
);
|
||||||
|
|
||||||
|
// Ajouter seulement si n'existe ni par lien ni par titre+date
|
||||||
|
if (!existingByLink && !existingByTitleDate) {
|
||||||
await runQuery(
|
await runQuery(
|
||||||
'INSERT INTO rss_articles (feed_id, title, link, description, pub_date, content) VALUES (?, ?, ?, ?, ?, ?)',
|
'INSERT INTO rss_articles (feed_id, title, link, description, pub_date, content) VALUES (?, ?, ?, ?, ?, ?)',
|
||||||
[
|
[
|
||||||
@@ -125,7 +138,7 @@ async function fetchAllFeeds() {
|
|||||||
item.title || 'Sans titre',
|
item.title || 'Sans titre',
|
||||||
item.link,
|
item.link,
|
||||||
item.contentSnippet || item.description || '',
|
item.contentSnippet || item.description || '',
|
||||||
item.pubDate || item.isoDate || new Date().toISOString(),
|
pubDate,
|
||||||
item.content || item['content:encoded'] || ''
|
item.content || item['content:encoded'] || ''
|
||||||
]
|
]
|
||||||
);
|
);
|
||||||
@@ -133,7 +146,7 @@ async function fetchAllFeeds() {
|
|||||||
totalArticles++;
|
totalArticles++;
|
||||||
}
|
}
|
||||||
} catch (articleError) {
|
} catch (articleError) {
|
||||||
// Ignorer les articles en double
|
// Ignorer les articles en double (contrainte UNIQUE sur link)
|
||||||
if (!articleError.message.includes('UNIQUE')) {
|
if (!articleError.message.includes('UNIQUE')) {
|
||||||
logger.debug(`Article ignoré: ${articleError.message}`);
|
logger.debug(`Article ignoré: ${articleError.message}`);
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in new issue
Block a user