Merge pull request #187 from R0m1k3/antigravity

feat: Add abstract base parser with product result dataclass and impl…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 19:03:52 +01:00
commit 4b6468a3bd
9 files changed
+136 -137

No files matched your search

+12 -31
View File
@@ -82,19 +82,19 @@ SITE_CONFIGS = {
},
"lafoirfouille.fr": {
"name": "La Foir'Fouille",
"search_url": "https://www.lafoirfouille.fr/recherche?controller=search&s={query}",
"product_selector": "article.product-miniature a.product-thumbnail, a.product_img_link",
"product_image_selector": "img[src], img[loading='lazy'], picture img",
"wait_selector": "article.product-miniature, .products",
"search_url": "https://www.lafoirfouille.fr/recherche?s={query}",
"product_selector": "div.product-miniature, article, div[class*='product']",
"product_image_selector": "img.product-thumbnail, img[src*='product'], img",
"wait_selector": "div.product-miniature, article, div[class*='product']",
"category": "Discount",
"requires_proxy": False,
},
"cdiscount.com": {
"name": "Cdiscount",
"search_url": "https://www.cdiscount.com/search/10/{query}.html",
"product_selector": "a.prdtBILnk, a[href*='/f-'][href*='.html']",
"product_selector": "a[class*='sc-'], a.prdtBILnk, a[href*='/f-'][href*='.html']",
"product_image_selector": "img.lazy, img[data-src], img[src*='image']",
"wait_selector": ".prdtBILDetails, .prdtBIL",
"wait_selector": "a[class*='sc-'], .prdtBILDetails, .prdtBIL",
"category": "E-commerce",
"requires_proxy": False,
},
@@ -139,37 +139,18 @@ SITE_CONFIGS = {
"centrakor.com": {
"name": "Centrakor",
"search_url": "https://www.centrakor.com/search/{query}",
"product_selector": "a.link.link--block, a.product-item__link",
"product_image_selector": "img.product-card__image, img[loading='lazy'], noscript img",
"wait_selector": ".product-list, .collection__products",
"product_selector": "div.product-item, div.product-card, article",
"product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']",
"wait_selector": "div.product-item, div.product-card, article",
"category": "Discount",
"requires_proxy": False,
},
"lincroyable.fr": {
"name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche?query={query}",
"product_selector": "a.product-link, a[href*='/p/']",
"product_image_selector": "img[loading='lazy'], img.product-image__img, img[src*='product']",
"wait_selector": ".product-card, .products-grid",
"category": "Discount",
"requires_proxy": False,
},
# === GRANDES SURFACES ===
"e-leclerc.com": {
"name": "E.Leclerc",
"search_url": "https://www.e.leclerc/recherche?q={query}",
"product_selector": "a.product-card-link",
"product_image_selector": "img.product-card-image, img[class*='product-image']",
"wait_selector": ".product-card-link",
"category": "Grande Surface",
"requires_proxy": False,
},
"auchan.fr": {
"name": "Auchan",
"search_url": "https://www.auchan.fr/recherche?text={query}",
"product_selector": "article[class*='product-item'] a[href*='/p-'], div[class*='product-item'] a[href*='/p-']",
"product_image_selector": "img[class*='product-thumbnail__img'], img[class*='product-item__image']",
"wait_selector": "article[class*='product-item'], div[class*='product-item']",
"product_selector": "article, div[class*='product-card'], div[class*='list__item']",
"product_image_selector": "img[class*='product'], img[src*='auchan']",
"wait_selector": "article, div[class*='product-card'], div[class*='list__item']",
"category": "Grande Surface",
"requires_proxy": False,
},
-38
View File
@@ -6,44 +6,6 @@ logger = logging.getLogger(__name__)
class AuchanParser(BaseParser):
def __init__(self):
super().__init__("auchan.fr", "https://www.auchan.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Auchan products are usually in article or div with class product-item
products = soup.select("article[class*='product-item'], div[class*='product-item']")
for product in products:
try:
# Link and Title
link_el = product.select_one("a[href*='/p-']")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title = link_el.get('title') or link_el.get_text(strip=True)
if not title:
continue
# Image
img_url = self.extract_image_url(product)
# Price
price = None
price_el = product.select_one("span[class*='price__value'], div[class*='price--selling'], span[itemprop='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Auchan",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Auchan"
+2
View File
@@ -81,6 +81,8 @@ class BaseParser(ABC):
"""Convert relative URL to absolute URL"""
if not url:
return ""
if url.startswith("//"):
return "https:" + url
if url.startswith("http"):
return url
return urljoin(base_url or self.base_url, url)
+38 -8
View File
@@ -13,32 +13,62 @@ class CarrefourParser(BaseParser):
results = []
# Carrefour products
# Config: a[href*='/p/']
# New container: div containing both image and title link
cards = soup.select("div.product-list-card-plp-grid-new, article")
if not cards:
cards = soup.select("div[class*='product-list-card']") # Fallback
products = soup.select("article, div[class*='product-card']")
for product in products:
# If still no cards, try finding by link
if not cards:
links = soup.select("a.c-link.product-card-click-wrapper")
cards = []
for link in links:
parent = link.find_parent("div", class_=lambda x: x and "product-list-card" in x)
if parent:
cards.append(parent)
else:
cards.append(link.parent.parent)
for card in cards:
try:
link_el = product.select_one("a[href*='/p/'], a[href*='/produit']")
link_el = card.select_one("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one("[class*='title'], h2, h3")
title_el = card.select_one("h3, h2, [class*='title']")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(product)
img_url = None
img_el = card.select_one("img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
price = None
price_el = product.select_one("[class*='price']")
# Price is often text node near h3 or in a specific price element
price_el = card.select_one("[class*='price'], .product-card-price")
if price_el:
price = self.parse_price_text(price_el.get_text())
if not price and title_el:
# Fallback: check siblings of title for text price
price_text = ""
for sibling in title_el.find_next_siblings():
if sibling.name is None: # Text node
price_text += sibling.strip() + " "
elif sibling.name in ['div', 'span']:
price_text += sibling.get_text(strip=True) + " "
if "€" in price_text:
break
price = self.parse_price_text(price_text)
results.append(ProductResult(
title=title,
url=url,
+10 -5
View File
@@ -14,14 +14,19 @@ class CdiscountParser(BaseParser):
# Cdiscount products
# Config: a.prdtBILnk, a[href*='/f-'][href*='.html']
# We look for the card container
# We look for the card container or the link itself if it acts as a card
cards = soup.select("li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil")
# Try finding cards/links with new classes
cards = soup.select("a[class*='sc-'], li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil")
for card in cards:
try:
# Link
link_el = card.select_one("a.prdtBILnk, a")
if card.name == 'a':
link_el = card
else:
link_el = card.select_one("a.prdtBILnk, a")
if not link_el:
continue
@@ -33,7 +38,7 @@ class CdiscountParser(BaseParser):
# Title
title = None
title_el = card.select_one(".prdtBTitle, .prdtBTit")
title_el = card.select_one("h2, .prdtBTitle, .prdtBTit")
if title_el:
title = title_el.get_text(strip=True)
else:
@@ -47,7 +52,7 @@ class CdiscountParser(BaseParser):
# Price
price = None
price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price")
price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price, span[class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
+36 -9
View File
@@ -13,32 +13,59 @@ class CentrakorParser(BaseParser):
results = []
# Centrakor products
# Config: a.link.link--block, a.product-item__link
# Config: div.product-item, div.product-card
products = soup.select(".product-item, .product-card")
# Try finding cards first
cards = soup.select("div.product-item, div.product-card, article")
for product in products:
# If no cards found, try finding by link and getting parent
if not cards:
links = soup.select("a[href*='/p/'], a[href*='/produit/'], a.product-item__link")
cards = []
for link in links:
# Try to find a container div
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "item" in x))
if parent:
cards.append(parent)
else:
cards.append(link.parent) # Fallback to immediate parent
for card in cards:
try:
link_el = product.select_one("a.product-item__link, a.link")
link_el = card.select_one("a.product-item__link, a.link, a[href*='/p/']")
if not link_el:
continue
# If card is the link itself
if card.name == 'a' and card.get('href'):
link_el = card
else:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-item__name, .product-card__title")
title = title_el.get_text(strip=True) if title_el else None
title_el = card.select_one(".product-item__name, .product-card__title, h3, h2")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(product)
img_url = None
img_el = card.select_one("img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
price = None
price_el = product.select_one(".price, .product-price")
price_el = card.select_one(".price, .product-price, .product-item__price")
if price_el:
price = self.parse_price_text(price_el.get_text())
# Fallback price search in text
if not price:
text = card.get_text(" ", strip=True)
price = self.parse_price_text(text)
results.append(ProductResult(
title=title,
url=url,
@@ -7,44 +7,6 @@ logger = logging.getLogger(__name__)
class LaFoirFouilleParser(BaseParser):
def __init__(self):
super().__init__("lafoirfouille.fr", "https://www.lafoirfouille.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# La Foir'Fouille products
# Config: article.product-miniature a.product-thumbnail
products = soup.select("article.product-miniature")
for product in products:
try:
link_el = product.select_one("a.product-thumbnail, a.product_img_link")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-title, h3")
title = title_el.get_text(strip=True) if title_el else None
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".product-price-and-shipping, .price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="La Foir'Fouille",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from La Foir'Fouille"
+35 -8
View File
@@ -15,30 +15,57 @@ class LIncroyableParser(BaseParser):
# L'Incroyable products
# Config: a.product-link
products = soup.select(".product-card, .product-miniature")
# Try finding cards first
cards = soup.select("div.product-card, div.product-miniature, article")
for product in products:
# If no cards found, try finding by link and getting parent
if not cards:
links = soup.select("a[href*='/p/'], a[href*='/produit/']")
cards = []
for link in links:
# Try to find a container div
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x))
if parent:
cards.append(parent)
else:
cards.append(link.parent) # Fallback to immediate parent
for card in cards:
try:
link_el = product.select_one("a.product-link, a[href*='/p/']")
link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link")
if not link_el:
continue
# If card is the link itself
if card.name == 'a' and card.get('href'):
link_el = card
else:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-title, h3")
title = title_el.get_text(strip=True) if title_el else None
title_el = card.select_one(".product-title, h3, h2, [class*='title']")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(product)
img_url = None
img_el = card.select_one("img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
price = None
price_el = product.select_one(".price, .product-price")
price_el = card.select_one(".price, .product-price, [class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
# Fallback price search in text
if not price:
text = card.get_text(" ", strip=True)
price = self.parse_price_text(text)
results.append(ProductResult(
title=title,
url=url,
+3
View File
@@ -46,6 +46,9 @@ class StokomaniParser(BaseParser):
if not img_url:
img_url = self.extract_image_url(prev_a)
if img_url:
img_url = self.make_absolute_url(img_url)
# Price: Look for text node after the link
price = None
next_sibling = link.next_sibling