mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #190 from R0m1k3/antigravity
feat: Implement new product parsers for Stokomani, Auchan, Carrefour,…
This commit is contained in:
5 files changed
+75
-50
No files matched your search
@@ -127,6 +127,15 @@ SITE_CONFIGS = {
|
||||
"requires_proxy": False,
|
||||
},
|
||||
# === NOUVEAUX SITES ===
|
||||
"lincroyable.fr": {
|
||||
"name": "L'Incroyable",
|
||||
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
|
||||
"product_selector": "div.product-card, div.product-miniature, article",
|
||||
"product_image_selector": "img.product-image, img[class*='product'], picture img",
|
||||
"wait_selector": "div.product-card, div.product-miniature, article",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
"bmstores.fr": {
|
||||
"name": "B&M",
|
||||
"search_url": "https://bmstores.fr/module/ambjolisearch/jolisearch?s={query}",
|
||||
|
||||
@@ -12,55 +12,70 @@ class AuchanParser(BaseParser):
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Auchan products
|
||||
# Config: article, div[class*='product-card']
|
||||
# Auchan products - Ultra Robust Strategy
|
||||
# The DOM is flat and dynamic. We rely on finding product links first.
|
||||
# Links usually contain '/p-' in the href.
|
||||
|
||||
cards = soup.select("article, div[class*='product-card'], div[class*='list__item']")
|
||||
links = soup.select("a[href*='/p-']")
|
||||
seen_urls = set()
|
||||
|
||||
# Fallback: try finding links directly
|
||||
if not cards:
|
||||
links = soup.select("a[href*='/p-']")
|
||||
cards = []
|
||||
for link in links:
|
||||
parent = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
if parent:
|
||||
cards.append(parent)
|
||||
else:
|
||||
cards.append(link.parent)
|
||||
|
||||
for card in cards:
|
||||
for link in links:
|
||||
try:
|
||||
link_el = card.select_one("a[href*='/p-'], a")
|
||||
if not link_el:
|
||||
href = link.get('href')
|
||||
if not href or href in seen_urls:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
if not href or '/p-' not in href:
|
||||
|
||||
# Filter out non-product links if any (e.g. facets)
|
||||
if '/p-' not in href:
|
||||
continue
|
||||
|
||||
seen_urls.add(href)
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = card.select_one("h3, div[class*='title'], span[class*='title']")
|
||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||
# Find the container: usually the link itself or a close parent
|
||||
# We look for a parent that contains price or image info
|
||||
container = link
|
||||
for _ in range(5): # Go up 5 levels max
|
||||
parent = container.parent
|
||||
if not parent:
|
||||
break
|
||||
container = parent
|
||||
# Stop if we hit a large container or list item
|
||||
if container.name == 'article' or (container.get('class') and any('list__item' in c for c in container.get('class'))):
|
||||
break
|
||||
# Stop if we find a price element inside this container that isn't the link itself
|
||||
if container.select_one("span[class*='price'], div[class*='price']"):
|
||||
break
|
||||
|
||||
# Title
|
||||
title = link.get('title')
|
||||
if not title:
|
||||
title_el = container.select_one("h3, div[class*='title'], span[class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
if not title:
|
||||
title = link.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image
|
||||
img_url = None
|
||||
img_el = card.select_one("img")
|
||||
img_el = container.select_one("img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
price_el = card.select_one("div[class*='price'], span[class*='price'], .product-price")
|
||||
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
# Fallback price search in text
|
||||
# Fallback price search in text of container
|
||||
if not price:
|
||||
text = card.get_text(" ", strip=True)
|
||||
text = container.get_text(" ", strip=True)
|
||||
price = self.parse_price_text(text)
|
||||
|
||||
results.append(ProductResult(
|
||||
|
||||
@@ -14,16 +14,14 @@ class CarrefourParser(BaseParser):
|
||||
|
||||
# Carrefour products
|
||||
# New container: div containing both image and title link
|
||||
cards = soup.select("div.product-list-card-plp-grid-new, article")
|
||||
if not cards:
|
||||
cards = soup.select("div[class*='product-list-card']") # Fallback
|
||||
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']")
|
||||
|
||||
# If still no cards, try finding by link
|
||||
# If no cards, try finding by link
|
||||
if not cards:
|
||||
links = soup.select("a.c-link.product-card-click-wrapper")
|
||||
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
|
||||
cards = []
|
||||
for link in links:
|
||||
parent = link.find_parent("div", class_=lambda x: x and "product-list-card" in x)
|
||||
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
if parent:
|
||||
cards.append(parent)
|
||||
else:
|
||||
@@ -53,21 +51,18 @@ class CarrefourParser(BaseParser):
|
||||
|
||||
price = None
|
||||
# Price is often text node near h3 or in a specific price element
|
||||
price_el = card.select_one("[class*='price'], .product-card-price")
|
||||
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
if not price and title_el:
|
||||
# Fallback: check siblings of title for text price
|
||||
price_text = ""
|
||||
for sibling in title_el.find_next_siblings():
|
||||
if sibling.name is None: # Text node
|
||||
price_text += sibling.strip() + " "
|
||||
elif sibling.name in ['div', 'span']:
|
||||
price_text += sibling.get_text(strip=True) + " "
|
||||
if "€" in price_text:
|
||||
break
|
||||
price = self.parse_price_text(price_text)
|
||||
if not price:
|
||||
# Fallback: check all text in the card for a price pattern
|
||||
text = card.get_text(" ", strip=True)
|
||||
# Simple regex for price like 12,99 € or 12.99€
|
||||
import re
|
||||
match = re.search(r'(\d+[.,]\d{2})\s*€?', text)
|
||||
if match:
|
||||
price = self.parse_price_text(match.group(0))
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
|
||||
@@ -27,12 +27,14 @@ class GifiParser(BaseParser):
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title = link_el.get_text(strip=True)
|
||||
# Title: Try specific classes first to avoid getting rating text etc.
|
||||
title = None
|
||||
title_el = product.select_one(".pdp-link > a, .link, [class*='name'], [class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
# Try finding title in nested elements
|
||||
title_el = product.select_one("[class*='name'], [class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
title = link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
@@ -41,7 +41,11 @@ class StokomaniParser(BaseParser):
|
||||
# Try specific selector first
|
||||
img_el = prev_a.select_one("motion-element img, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
# Check for lazy loading attributes explicitly
|
||||
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
|
||||
if img_url and " " in img_url:
|
||||
# Handle srcset: take the first URL
|
||||
img_url = img_url.split(" ")[0]
|
||||
|
||||
if not img_url:
|
||||
img_url = self.extract_image_url(prev_a)
|
||||
|
||||
Reference in new issue
Block a user