From 078060f155da58d8d80a0ce5eae95873ff75240d Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 30 Nov 2025 19:03:26 +0100 Subject: [PATCH] feat: Add abstract base parser with product result dataclass and implement Stokomani parser for search results. --- app/core/search_config.py | 43 +++++------------- app/services/parsers/auchan_parser.py | 38 ---------------- app/services/parsers/base_parser.py | 2 + app/services/parsers/carrefour_parser.py | 46 ++++++++++++++++---- app/services/parsers/cdiscount_parser.py | 15 ++++--- app/services/parsers/centrakor_parser.py | 45 +++++++++++++++---- app/services/parsers/lafoirfouille_parser.py | 38 ---------------- app/services/parsers/lincroyable_parser.py | 43 ++++++++++++++---- app/services/parsers/stokomani_parser.py | 3 ++ 9 files changed, 136 insertions(+), 137 deletions(-) diff --git a/app/core/search_config.py b/app/core/search_config.py index 23a442a..d2a694f 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -82,19 +82,19 @@ SITE_CONFIGS = { }, "lafoirfouille.fr": { "name": "La Foir'Fouille", - "search_url": "https://www.lafoirfouille.fr/recherche?controller=search&s={query}", - "product_selector": "article.product-miniature a.product-thumbnail, a.product_img_link", - "product_image_selector": "img[src], img[loading='lazy'], picture img", - "wait_selector": "article.product-miniature, .products", + "search_url": "https://www.lafoirfouille.fr/recherche?s={query}", + "product_selector": "div.product-miniature, article, div[class*='product']", + "product_image_selector": "img.product-thumbnail, img[src*='product'], img", + "wait_selector": "div.product-miniature, article, div[class*='product']", "category": "Discount", "requires_proxy": False, }, "cdiscount.com": { "name": "Cdiscount", "search_url": "https://www.cdiscount.com/search/10/{query}.html", - "product_selector": "a.prdtBILnk, a[href*='/f-'][href*='.html']", + "product_selector": "a[class*='sc-'], a.prdtBILnk, a[href*='/f-'][href*='.html']", "product_image_selector": "img.lazy, img[data-src], img[src*='image']", - "wait_selector": ".prdtBILDetails, .prdtBIL", + "wait_selector": "a[class*='sc-'], .prdtBILDetails, .prdtBIL", "category": "E-commerce", "requires_proxy": False, }, @@ -139,37 +139,18 @@ SITE_CONFIGS = { "centrakor.com": { "name": "Centrakor", "search_url": "https://www.centrakor.com/search/{query}", - "product_selector": "a.link.link--block, a.product-item__link", - "product_image_selector": "img.product-card__image, img[loading='lazy'], noscript img", - "wait_selector": ".product-list, .collection__products", + "product_selector": "div.product-item, div.product-card, article", + "product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']", + "wait_selector": "div.product-item, div.product-card, article", "category": "Discount", "requires_proxy": False, }, - "lincroyable.fr": { - "name": "L'Incroyable", - "search_url": "https://www.lincroyable.fr/recherche?query={query}", - "product_selector": "a.product-link, a[href*='/p/']", - "product_image_selector": "img[loading='lazy'], img.product-image__img, img[src*='product']", - "wait_selector": ".product-card, .products-grid", - "category": "Discount", - "requires_proxy": False, - }, - # === GRANDES SURFACES === - "e-leclerc.com": { - "name": "E.Leclerc", - "search_url": "https://www.e.leclerc/recherche?q={query}", - "product_selector": "a.product-card-link", - "product_image_selector": "img.product-card-image, img[class*='product-image']", - "wait_selector": ".product-card-link", - "category": "Grande Surface", - "requires_proxy": False, - }, "auchan.fr": { "name": "Auchan", "search_url": "https://www.auchan.fr/recherche?text={query}", - "product_selector": "article[class*='product-item'] a[href*='/p-'], div[class*='product-item'] a[href*='/p-']", - "product_image_selector": "img[class*='product-thumbnail__img'], img[class*='product-item__image']", - "wait_selector": "article[class*='product-item'], div[class*='product-item']", + "product_selector": "article, div[class*='product-card'], div[class*='list__item']", + "product_image_selector": "img[class*='product'], img[src*='auchan']", + "wait_selector": "article, div[class*='product-card'], div[class*='list__item']", "category": "Grande Surface", "requires_proxy": False, }, diff --git a/app/services/parsers/auchan_parser.py b/app/services/parsers/auchan_parser.py index 3ecafc8..788395b 100644 --- a/app/services/parsers/auchan_parser.py +++ b/app/services/parsers/auchan_parser.py @@ -6,44 +6,6 @@ logger = logging.getLogger(__name__) class AuchanParser(BaseParser): def __init__(self): - super().__init__("auchan.fr", "https://www.auchan.fr") - - def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: - soup = BeautifulSoup(html, "html.parser") - results = [] - - # Auchan products are usually in article or div with class product-item - products = soup.select("article[class*='product-item'], div[class*='product-item']") - - for product in products: - try: - # Link and Title - link_el = product.select_one("a[href*='/p-']") - if not link_el: - continue - - href = link_el.get('href') - url = self.make_absolute_url(href) - - title = link_el.get('title') or link_el.get_text(strip=True) - if not title: - continue - - # Image - img_url = self.extract_image_url(product) - - # Price - price = None - price_el = product.select_one("span[class*='price__value'], div[class*='price--selling'], span[itemprop='price']") - if price_el: - price = self.parse_price_text(price_el.get_text()) - - results.append(ProductResult( - title=title, - url=url, - source="Auchan", - price=price, - currency="EUR", in_stock=True, image_url=img_url, snippet=f"Product from Auchan" diff --git a/app/services/parsers/base_parser.py b/app/services/parsers/base_parser.py index a341f93..c94e879 100644 --- a/app/services/parsers/base_parser.py +++ b/app/services/parsers/base_parser.py @@ -81,6 +81,8 @@ class BaseParser(ABC): """Convert relative URL to absolute URL""" if not url: return "" + if url.startswith("//"): + return "https:" + url if url.startswith("http"): return url return urljoin(base_url or self.base_url, url) diff --git a/app/services/parsers/carrefour_parser.py b/app/services/parsers/carrefour_parser.py index 07f875b..9781f78 100644 --- a/app/services/parsers/carrefour_parser.py +++ b/app/services/parsers/carrefour_parser.py @@ -13,32 +13,62 @@ class CarrefourParser(BaseParser): results = [] # Carrefour products - # Config: a[href*='/p/'] + # New container: div containing both image and title link + cards = soup.select("div.product-list-card-plp-grid-new, article") + if not cards: + cards = soup.select("div[class*='product-list-card']") # Fallback - products = soup.select("article, div[class*='product-card']") - - for product in products: + # If still no cards, try finding by link + if not cards: + links = soup.select("a.c-link.product-card-click-wrapper") + cards = [] + for link in links: + parent = link.find_parent("div", class_=lambda x: x and "product-list-card" in x) + if parent: + cards.append(parent) + else: + cards.append(link.parent.parent) + + for card in cards: try: - link_el = product.select_one("a[href*='/p/'], a[href*='/produit']") + link_el = card.select_one("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']") if not link_el: continue href = link_el.get('href') url = self.make_absolute_url(href) - title_el = product.select_one("[class*='title'], h2, h3") + title_el = card.select_one("h3, h2, [class*='title']") title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) if not title: continue - img_url = self.extract_image_url(product) + img_url = None + img_el = card.select_one("img") + if img_el: + img_url = self._get_image_src(img_el) + if img_url: + img_url = self.make_absolute_url(img_url) price = None - price_el = product.select_one("[class*='price']") + # Price is often text node near h3 or in a specific price element + price_el = card.select_one("[class*='price'], .product-card-price") if price_el: price = self.parse_price_text(price_el.get_text()) + if not price and title_el: + # Fallback: check siblings of title for text price + price_text = "" + for sibling in title_el.find_next_siblings(): + if sibling.name is None: # Text node + price_text += sibling.strip() + " " + elif sibling.name in ['div', 'span']: + price_text += sibling.get_text(strip=True) + " " + if "€" in price_text: + break + price = self.parse_price_text(price_text) + results.append(ProductResult( title=title, url=url, diff --git a/app/services/parsers/cdiscount_parser.py b/app/services/parsers/cdiscount_parser.py index d433a67..4e10db5 100644 --- a/app/services/parsers/cdiscount_parser.py +++ b/app/services/parsers/cdiscount_parser.py @@ -14,14 +14,19 @@ class CdiscountParser(BaseParser): # Cdiscount products # Config: a.prdtBILnk, a[href*='/f-'][href*='.html'] - # We look for the card container + # We look for the card container or the link itself if it acts as a card - cards = soup.select("li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil") + # Try finding cards/links with new classes + cards = soup.select("a[class*='sc-'], li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil") for card in cards: try: # Link - link_el = card.select_one("a.prdtBILnk, a") + if card.name == 'a': + link_el = card + else: + link_el = card.select_one("a.prdtBILnk, a") + if not link_el: continue @@ -33,7 +38,7 @@ class CdiscountParser(BaseParser): # Title title = None - title_el = card.select_one(".prdtBTitle, .prdtBTit") + title_el = card.select_one("h2, .prdtBTitle, .prdtBTit") if title_el: title = title_el.get_text(strip=True) else: @@ -47,7 +52,7 @@ class CdiscountParser(BaseParser): # Price price = None - price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price") + price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price, span[class*='price']") if price_el: price = self.parse_price_text(price_el.get_text()) diff --git a/app/services/parsers/centrakor_parser.py b/app/services/parsers/centrakor_parser.py index fab5d6c..8ef7a0f 100644 --- a/app/services/parsers/centrakor_parser.py +++ b/app/services/parsers/centrakor_parser.py @@ -13,32 +13,59 @@ class CentrakorParser(BaseParser): results = [] # Centrakor products - # Config: a.link.link--block, a.product-item__link + # Config: div.product-item, div.product-card - products = soup.select(".product-item, .product-card") + # Try finding cards first + cards = soup.select("div.product-item, div.product-card, article") - for product in products: + # If no cards found, try finding by link and getting parent + if not cards: + links = soup.select("a[href*='/p/'], a[href*='/produit/'], a.product-item__link") + cards = [] + for link in links: + # Try to find a container div + parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "item" in x)) + if parent: + cards.append(parent) + else: + cards.append(link.parent) # Fallback to immediate parent + + for card in cards: try: - link_el = product.select_one("a.product-item__link, a.link") + link_el = card.select_one("a.product-item__link, a.link, a[href*='/p/']") if not link_el: - continue + # If card is the link itself + if card.name == 'a' and card.get('href'): + link_el = card + else: + continue href = link_el.get('href') url = self.make_absolute_url(href) - title_el = product.select_one(".product-item__name, .product-card__title") - title = title_el.get_text(strip=True) if title_el else None + title_el = card.select_one(".product-item__name, .product-card__title, h3, h2") + title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) if not title: continue - img_url = self.extract_image_url(product) + img_url = None + img_el = card.select_one("img") + if img_el: + img_url = self._get_image_src(img_el) + if img_url: + img_url = self.make_absolute_url(img_url) price = None - price_el = product.select_one(".price, .product-price") + price_el = card.select_one(".price, .product-price, .product-item__price") if price_el: price = self.parse_price_text(price_el.get_text()) + # Fallback price search in text + if not price: + text = card.get_text(" ", strip=True) + price = self.parse_price_text(text) + results.append(ProductResult( title=title, url=url, diff --git a/app/services/parsers/lafoirfouille_parser.py b/app/services/parsers/lafoirfouille_parser.py index 9148530..29e3879 100644 --- a/app/services/parsers/lafoirfouille_parser.py +++ b/app/services/parsers/lafoirfouille_parser.py @@ -7,44 +7,6 @@ logger = logging.getLogger(__name__) class LaFoirFouilleParser(BaseParser): def __init__(self): super().__init__("lafoirfouille.fr", "https://www.lafoirfouille.fr") - - def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: - soup = BeautifulSoup(html, "html.parser") - results = [] - - # La Foir'Fouille products - # Config: article.product-miniature a.product-thumbnail - - products = soup.select("article.product-miniature") - - for product in products: - try: - link_el = product.select_one("a.product-thumbnail, a.product_img_link") - if not link_el: - continue - - href = link_el.get('href') - url = self.make_absolute_url(href) - - title_el = product.select_one(".product-title, h3") - title = title_el.get_text(strip=True) if title_el else None - - if not title: - continue - - img_url = self.extract_image_url(product) - - price = None - price_el = product.select_one(".product-price-and-shipping, .price") - if price_el: - price = self.parse_price_text(price_el.get_text()) - - results.append(ProductResult( - title=title, - url=url, - source="La Foir'Fouille", - price=price, - currency="EUR", in_stock=True, image_url=img_url, snippet=f"Product from La Foir'Fouille" diff --git a/app/services/parsers/lincroyable_parser.py b/app/services/parsers/lincroyable_parser.py index 9e80c0f..1c0397d 100644 --- a/app/services/parsers/lincroyable_parser.py +++ b/app/services/parsers/lincroyable_parser.py @@ -15,30 +15,57 @@ class LIncroyableParser(BaseParser): # L'Incroyable products # Config: a.product-link - products = soup.select(".product-card, .product-miniature") + # Try finding cards first + cards = soup.select("div.product-card, div.product-miniature, article") - for product in products: + # If no cards found, try finding by link and getting parent + if not cards: + links = soup.select("a[href*='/p/'], a[href*='/produit/']") + cards = [] + for link in links: + # Try to find a container div + parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x)) + if parent: + cards.append(parent) + else: + cards.append(link.parent) # Fallback to immediate parent + + for card in cards: try: - link_el = product.select_one("a.product-link, a[href*='/p/']") + link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link") if not link_el: - continue + # If card is the link itself + if card.name == 'a' and card.get('href'): + link_el = card + else: + continue href = link_el.get('href') url = self.make_absolute_url(href) - title_el = product.select_one(".product-title, h3") - title = title_el.get_text(strip=True) if title_el else None + title_el = card.select_one(".product-title, h3, h2, [class*='title']") + title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) if not title: continue - img_url = self.extract_image_url(product) + img_url = None + img_el = card.select_one("img") + if img_el: + img_url = self._get_image_src(img_el) + if img_url: + img_url = self.make_absolute_url(img_url) price = None - price_el = product.select_one(".price, .product-price") + price_el = card.select_one(".price, .product-price, [class*='price']") if price_el: price = self.parse_price_text(price_el.get_text()) + # Fallback price search in text + if not price: + text = card.get_text(" ", strip=True) + price = self.parse_price_text(text) + results.append(ProductResult( title=title, url=url, diff --git a/app/services/parsers/stokomani_parser.py b/app/services/parsers/stokomani_parser.py index f71ac0f..9b65b53 100644 --- a/app/services/parsers/stokomani_parser.py +++ b/app/services/parsers/stokomani_parser.py @@ -46,6 +46,9 @@ class StokomaniParser(BaseParser): if not img_url: img_url = self.extract_image_url(prev_a) + if img_url: + img_url = self.make_absolute_url(img_url) + # Price: Look for text node after the link price = None next_sibling = link.next_sibling