From b44082aeb4bddc8161db2e0e11314dab9a559b85 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 30 Nov 2025 20:26:19 +0100 Subject: [PATCH] feat: Implement new product parsers for Stokomani, Auchan, Carrefour, and Gifi, and add a centralized search configuration module. --- app/core/search_config.py | 9 ++++ app/services/parsers/auchan_parser.py | 67 +++++++++++++++--------- app/services/parsers/carrefour_parser.py | 31 +++++------ app/services/parsers/gifi_parser.py | 12 +++-- app/services/parsers/stokomani_parser.py | 6 ++- 5 files changed, 75 insertions(+), 50 deletions(-) diff --git a/app/core/search_config.py b/app/core/search_config.py index d2a694f..70bd2bc 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -127,6 +127,15 @@ SITE_CONFIGS = { "requires_proxy": False, }, # === NOUVEAUX SITES === + "lincroyable.fr": { + "name": "L'Incroyable", + "search_url": "https://www.lincroyable.fr/recherche-query={query}/", + "product_selector": "div.product-card, div.product-miniature, article", + "product_image_selector": "img.product-image, img[class*='product'], picture img", + "wait_selector": "div.product-card, div.product-miniature, article", + "category": "Discount", + "requires_proxy": False, + }, "bmstores.fr": { "name": "B&M", "search_url": "https://bmstores.fr/module/ambjolisearch/jolisearch?s={query}", diff --git a/app/services/parsers/auchan_parser.py b/app/services/parsers/auchan_parser.py index 0738e27..a8e9221 100644 --- a/app/services/parsers/auchan_parser.py +++ b/app/services/parsers/auchan_parser.py @@ -12,55 +12,70 @@ class AuchanParser(BaseParser): soup = BeautifulSoup(html, "html.parser") results = [] - # Auchan products - # Config: article, div[class*='product-card'] + # Auchan products - Ultra Robust Strategy + # The DOM is flat and dynamic. We rely on finding product links first. + # Links usually contain '/p-' in the href. - cards = soup.select("article, div[class*='product-card'], div[class*='list__item']") + links = soup.select("a[href*='/p-']") + seen_urls = set() - # Fallback: try finding links directly - if not cards: - links = soup.select("a[href*='/p-']") - cards = [] - for link in links: - parent = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x) - if parent: - cards.append(parent) - else: - cards.append(link.parent) - - for card in cards: + for link in links: try: - link_el = card.select_one("a[href*='/p-'], a") - if not link_el: + href = link.get('href') + if not href or href in seen_urls: continue - - href = link_el.get('href') - if not href or '/p-' not in href: + + # Filter out non-product links if any (e.g. facets) + if '/p-' not in href: continue + seen_urls.add(href) url = self.make_absolute_url(href) - title_el = card.select_one("h3, div[class*='title'], span[class*='title']") - title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) + # Find the container: usually the link itself or a close parent + # We look for a parent that contains price or image info + container = link + for _ in range(5): # Go up 5 levels max + parent = container.parent + if not parent: + break + container = parent + # Stop if we hit a large container or list item + if container.name == 'article' or (container.get('class') and any('list__item' in c for c in container.get('class'))): + break + # Stop if we find a price element inside this container that isn't the link itself + if container.select_one("span[class*='price'], div[class*='price']"): + break + + # Title + title = link.get('title') + if not title: + title_el = container.select_one("h3, div[class*='title'], span[class*='title']") + if title_el: + title = title_el.get_text(strip=True) + if not title: + title = link.get_text(strip=True) if not title: continue + # Image img_url = None - img_el = card.select_one("img") + img_el = container.select_one("img") if img_el: img_url = self._get_image_src(img_el) if img_url: img_url = self.make_absolute_url(img_url) + # Price price = None - price_el = card.select_one("div[class*='price'], span[class*='price'], .product-price") + price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price") if price_el: price = self.parse_price_text(price_el.get_text()) - # Fallback price search in text + # Fallback price search in text of container if not price: - text = card.get_text(" ", strip=True) + text = container.get_text(" ", strip=True) price = self.parse_price_text(text) results.append(ProductResult( diff --git a/app/services/parsers/carrefour_parser.py b/app/services/parsers/carrefour_parser.py index 9781f78..e5e0ffb 100644 --- a/app/services/parsers/carrefour_parser.py +++ b/app/services/parsers/carrefour_parser.py @@ -14,16 +14,14 @@ class CarrefourParser(BaseParser): # Carrefour products # New container: div containing both image and title link - cards = soup.select("div.product-list-card-plp-grid-new, article") - if not cards: - cards = soup.select("div[class*='product-list-card']") # Fallback + cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']") - # If still no cards, try finding by link + # If no cards, try finding by link if not cards: - links = soup.select("a.c-link.product-card-click-wrapper") + links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']") cards = [] for link in links: - parent = link.find_parent("div", class_=lambda x: x and "product-list-card" in x) + parent = link.find_parent("div", class_=lambda x: x and "product" in x) if parent: cards.append(parent) else: @@ -53,21 +51,18 @@ class CarrefourParser(BaseParser): price = None # Price is often text node near h3 or in a specific price element - price_el = card.select_one("[class*='price'], .product-card-price") + price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']") if price_el: price = self.parse_price_text(price_el.get_text()) - if not price and title_el: - # Fallback: check siblings of title for text price - price_text = "" - for sibling in title_el.find_next_siblings(): - if sibling.name is None: # Text node - price_text += sibling.strip() + " " - elif sibling.name in ['div', 'span']: - price_text += sibling.get_text(strip=True) + " " - if "€" in price_text: - break - price = self.parse_price_text(price_text) + if not price: + # Fallback: check all text in the card for a price pattern + text = card.get_text(" ", strip=True) + # Simple regex for price like 12,99 € or 12.99€ + import re + match = re.search(r'(\d+[.,]\d{2})\s*€?', text) + if match: + price = self.parse_price_text(match.group(0)) results.append(ProductResult( title=title, diff --git a/app/services/parsers/gifi_parser.py b/app/services/parsers/gifi_parser.py index 977d8d4..971ff6b 100644 --- a/app/services/parsers/gifi_parser.py +++ b/app/services/parsers/gifi_parser.py @@ -27,12 +27,14 @@ class GifiParser(BaseParser): href = link_el.get('href') url = self.make_absolute_url(href) - title = link_el.get_text(strip=True) + # Title: Try specific classes first to avoid getting rating text etc. + title = None + title_el = product.select_one(".pdp-link > a, .link, [class*='name'], [class*='title']") + if title_el: + title = title_el.get_text(strip=True) + if not title: - # Try finding title in nested elements - title_el = product.select_one("[class*='name'], [class*='title']") - if title_el: - title = title_el.get_text(strip=True) + title = link_el.get_text(strip=True) if not title: continue diff --git a/app/services/parsers/stokomani_parser.py b/app/services/parsers/stokomani_parser.py index 9b65b53..d9d944e 100644 --- a/app/services/parsers/stokomani_parser.py +++ b/app/services/parsers/stokomani_parser.py @@ -41,7 +41,11 @@ class StokomaniParser(BaseParser): # Try specific selector first img_el = prev_a.select_one("motion-element img, img") if img_el: - img_url = self._get_image_src(img_el) + # Check for lazy loading attributes explicitly + img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src') + if img_url and " " in img_url: + # Handle srcset: take the first URL + img_url = img_url.split(" ")[0] if not img_url: img_url = self.extract_image_url(prev_a)