feat: add Tiendeo.fr scraper service using Crawl4AI for promotional catalog extraction and parsing.

This commit is contained in:
Michael committed 2025-11-30 01:29:00 +01:00
1 parent 2b974aacf9
commit 607f47c0f0
1 file changed
+15
+15
View File
@@ -212,6 +212,21 @@ async def scrape_catalog_list(crawler: AsyncWebCrawler, enseigne: Enseigne) -> l
if not catalog_id:
continue
# CRITICAL FIX: Filter by enseigne name
# The catalog link or its parent container MUST contain the enseigne name
link_text = link.get_text(strip=True)
parent = link.find_parent(['div', 'section', 'article', 'li'])
parent_text = parent.get_text(strip=True) if parent else ''
# Combined text to search in
combined_text = f"{link_text} {parent_text}".lower()
enseigne_name_lower = enseigne.nom.lower()
# Check if enseigne name is in the link or parent text
if enseigne_name_lower not in combined_text:
logger.debug(f"Skipping catalog (not for {enseigne.nom}): {link_text[:50]}")
continue
seen_urls.add(href)
# Extract title