feat: Implement new product parsers for Stokomani, Auchan, Carrefour, and Gifi, and add a centralized search configuration module.

This commit is contained in:
Michael committed 2025-11-30 20:26:19 +01:00
1 parent e0d73f4d0e
commit b44082aeb4
5 files changed
+74 -49

No files matched your search

+9
View File
@@ -127,6 +127,15 @@ SITE_CONFIGS = {
"requires_proxy": False,
},
# === NOUVEAUX SITES ===
"lincroyable.fr": {
"name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
"product_selector": "div.product-card, div.product-miniature, article",
"product_image_selector": "img.product-image, img[class*='product'], picture img",
"wait_selector": "div.product-card, div.product-miniature, article",
"category": "Discount",
"requires_proxy": False,
},
"bmstores.fr": {
"name": "B&M",
"search_url": "https://bmstores.fr/module/ambjolisearch/jolisearch?s={query}",
+40 -25
View File
@@ -12,55 +12,70 @@ class AuchanParser(BaseParser):
soup = BeautifulSoup(html, "html.parser")
results = []
# Auchan products
# Config: article, div[class*='product-card']
# Auchan products - Ultra Robust Strategy
# The DOM is flat and dynamic. We rely on finding product links first.
# Links usually contain '/p-' in the href.
cards = soup.select("article, div[class*='product-card'], div[class*='list__item']")
links = soup.select("a[href*='/p-']")
seen_urls = set()
# Fallback: try finding links directly
if not cards:
links = soup.select("a[href*='/p-']")
cards = []
for link in links:
parent = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if parent:
cards.append(parent)
else:
cards.append(link.parent)
for card in cards:
for link in links:
try:
link_el = card.select_one("a[href*='/p-'], a")
if not link_el:
href = link.get('href')
if not href or href in seen_urls:
continue
href = link_el.get('href')
if not href or '/p-' not in href:
# Filter out non-product links if any (e.g. facets)
if '/p-' not in href:
continue
seen_urls.add(href)
url = self.make_absolute_url(href)
title_el = card.select_one("h3, div[class*='title'], span[class*='title']")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
# Find the container: usually the link itself or a close parent
# We look for a parent that contains price or image info
container = link
for _ in range(5): # Go up 5 levels max
parent = container.parent
if not parent:
break
container = parent
# Stop if we hit a large container or list item
if container.name == 'article' or (container.get('class') and any('list__item' in c for c in container.get('class'))):
break
# Stop if we find a price element inside this container that isn't the link itself
if container.select_one("span[class*='price'], div[class*='price']"):
break
# Title
title = link.get('title')
if not title:
title_el = container.select_one("h3, div[class*='title'], span[class*='title']")
if title_el:
title = title_el.get_text(strip=True)
if not title:
title = link.get_text(strip=True)
if not title:
continue
# Image
img_url = None
img_el = card.select_one("img")
img_el = container.select_one("img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
# Price
price = None
price_el = card.select_one("div[class*='price'], span[class*='price'], .product-price")
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price")
if price_el:
price = self.parse_price_text(price_el.get_text())
# Fallback price search in text
# Fallback price search in text of container
if not price:
text = card.get_text(" ", strip=True)
text = container.get_text(" ", strip=True)
price = self.parse_price_text(text)
results.append(ProductResult(
+13 -18
View File
@@ -14,16 +14,14 @@ class CarrefourParser(BaseParser):
# Carrefour products
# New container: div containing both image and title link
cards = soup.select("div.product-list-card-plp-grid-new, article")
if not cards:
cards = soup.select("div[class*='product-list-card']") # Fallback
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']")
# If still no cards, try finding by link
# If no cards, try finding by link
if not cards:
links = soup.select("a.c-link.product-card-click-wrapper")
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
cards = []
for link in links:
parent = link.find_parent("div", class_=lambda x: x and "product-list-card" in x)
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
if parent:
cards.append(parent)
else:
@@ -53,21 +51,18 @@ class CarrefourParser(BaseParser):
price = None
# Price is often text node near h3 or in a specific price element
price_el = card.select_one("[class*='price'], .product-card-price")
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']")
if price_el:
price = self.parse_price_text(price_el.get_text())
if not price and title_el:
# Fallback: check siblings of title for text price
price_text = ""
for sibling in title_el.find_next_siblings():
if sibling.name is None: # Text node
price_text += sibling.strip() + " "
elif sibling.name in ['div', 'span']:
price_text += sibling.get_text(strip=True) + " "
if "€" in price_text:
break
price = self.parse_price_text(price_text)
if not price:
# Fallback: check all text in the card for a price pattern
text = card.get_text(" ", strip=True)
# Simple regex for price like 12,99 € or 12.99€
import re
match = re.search(r'(\d+[.,]\d{2})\s*€?', text)
if match:
price = self.parse_price_text(match.group(0))
results.append(ProductResult(
title=title,
+7 -5
View File
@@ -27,12 +27,14 @@ class GifiParser(BaseParser):
href = link_el.get('href')
url = self.make_absolute_url(href)
title = link_el.get_text(strip=True)
# Title: Try specific classes first to avoid getting rating text etc.
title = None
title_el = product.select_one(".pdp-link > a, .link, [class*='name'], [class*='title']")
if title_el:
title = title_el.get_text(strip=True)
if not title:
# Try finding title in nested elements
title_el = product.select_one("[class*='name'], [class*='title']")
if title_el:
title = title_el.get_text(strip=True)
title = link_el.get_text(strip=True)
if not title:
continue
+5 -1
View File
@@ -41,7 +41,11 @@ class StokomaniParser(BaseParser):
# Try specific selector first
img_el = prev_a.select_one("motion-element img, img")
if img_el:
img_url = self._get_image_src(img_el)
# Check for lazy loading attributes explicitly
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
if img_url and " " in img_url:
# Handle srcset: take the first URL
img_url = img_url.split(" ")[0]
if not img_url:
img_url = self.extract_image_url(prev_a)