diff --git a/app/core/search_config.py b/app/core/search_config.py index 01bd94b..23a442a 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -157,19 +157,19 @@ SITE_CONFIGS = { # === GRANDES SURFACES === "e-leclerc.com": { "name": "E.Leclerc", - "search_url": "https://www.e-leclerc.com/recherche?text={query}", - "product_selector": "a[href*='/cat'], a[href*='/produit'], article a, div[class*='product'] a", - "product_image_selector": "img, picture source", - "wait_selector": None, + "search_url": "https://www.e.leclerc/recherche?q={query}", + "product_selector": "a.product-card-link", + "product_image_selector": "img.product-card-image, img[class*='product-image']", + "wait_selector": ".product-card-link", "category": "Grande Surface", "requires_proxy": False, }, "auchan.fr": { "name": "Auchan", - "search_url": "https://www.auchan.fr/search?text={query}", - "product_selector": "article.product-item a.product-item__name, a[class*='product-item'][href]", - "product_image_selector": "img.product-thumbnail__img, img[class*='product'][class*='img']", - "wait_selector": "article.product-item, .product-list", + "search_url": "https://www.auchan.fr/recherche?text={query}", + "product_selector": "article[class*='product-item'] a[href*='/p-'], div[class*='product-item'] a[href*='/p-']", + "product_image_selector": "img[class*='product-thumbnail__img'], img[class*='product-item__image']", + "wait_selector": "article[class*='product-item'], div[class*='product-item']", "category": "Grande Surface", "requires_proxy": False, }, diff --git a/app/services/improved_search_service.py b/app/services/improved_search_service.py index 0015612..f06b9c5 100644 --- a/app/services/improved_search_service.py +++ b/app/services/improved_search_service.py @@ -258,12 +258,11 @@ class ImprovedSearchService: if results: logger.info(f"📦 Found {len(results)} initial results, enriching with details...") - # LIMIT: Only enrich first 10 products to avoid frontend timeouts - # Full enrichment (visiting each product page) takes too long - results_to_enrich = results[:10] - logger.info(f"⚡ Limiting enrichment to {len(results_to_enrich)} products") + # Enrich all results + results_to_enrich = results + logger.info(f"⚡ Enriching all {len(results_to_enrich)} products") - semaphore = asyncio.Semaphore(2) # Limit concurrency + semaphore = asyncio.Semaphore(4) # Increased concurrency slightly async def scrape_with_limit(res): async with semaphore: @@ -272,8 +271,8 @@ class ImprovedSearchService: tasks = [scrape_with_limit(r) for r in results_to_enrich] enriched_results = await asyncio.gather(*tasks) - # Return enriched results + remaining non-enriched (with None price) - results = [r for r in enriched_results if r] + results[10:] + # Filter out failed enrichments if any (though scrape_item_details returns original on failure) + results = [r for r in enriched_results if r] finally: await context.close() diff --git a/app/services/parsers/action_parser.py b/app/services/parsers/action_parser.py new file mode 100644 index 0000000..e7499a4 --- /dev/null +++ b/app/services/parsers/action_parser.py @@ -0,0 +1,65 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class ActionParser(BaseParser): + def __init__(self): + super().__init__("action.com", "https://www.action.com") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Action products + # Config: a.group[href^='/fr-fr/p/'] + + links = soup.select("a[href^='/fr-fr/p/']") + + seen_urls = set() + + for link in links: + try: + href = link.get('href') + if not href or href in seen_urls: + continue + seen_urls.add(href) + + url = self.make_absolute_url(href) + + # Action usually has the card content inside the link + title = None + title_el = link.select_one("[class*='title'], h3, h4") + if title_el: + title = title_el.get_text(strip=True) + else: + title = link.get_text(strip=True) + + if not title: + continue + + img_url = self.extract_image_url(link) + + price = None + price_el = link.select_one("[class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="Action", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Action" + )) + + except Exception as e: + logger.error(f"Error parsing Action product: {e}") + continue + + logger.info(f"ActionParser found {len(results)} results") + return results diff --git a/app/services/parsers/auchan_parser.py b/app/services/parsers/auchan_parser.py new file mode 100644 index 0000000..3ecafc8 --- /dev/null +++ b/app/services/parsers/auchan_parser.py @@ -0,0 +1,57 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class AuchanParser(BaseParser): + def __init__(self): + super().__init__("auchan.fr", "https://www.auchan.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Auchan products are usually in article or div with class product-item + products = soup.select("article[class*='product-item'], div[class*='product-item']") + + for product in products: + try: + # Link and Title + link_el = product.select_one("a[href*='/p-']") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title = link_el.get('title') or link_el.get_text(strip=True) + if not title: + continue + + # Image + img_url = self.extract_image_url(product) + + # Price + price = None + price_el = product.select_one("span[class*='price__value'], div[class*='price--selling'], span[itemprop='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="Auchan", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Auchan" + )) + + except Exception as e: + logger.error(f"Error parsing Auchan product: {e}") + continue + + logger.info(f"AuchanParser found {len(results)} results") + return results diff --git a/app/services/parsers/bmstores_parser.py b/app/services/parsers/bmstores_parser.py new file mode 100644 index 0000000..79ba12c --- /dev/null +++ b/app/services/parsers/bmstores_parser.py @@ -0,0 +1,58 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class BMStoresParser(BaseParser): + def __init__(self): + super().__init__("bmstores.fr", "https://bmstores.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # B&M products + # Config: a.thumbnail.product-thumbnail + + products = soup.select(".product-miniature, .js-product-miniature") + + for product in products: + try: + link_el = product.select_one("a.thumbnail.product-thumbnail") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title_el = product.select_one(".product-title, h3") + title = title_el.get_text(strip=True) if title_el else None + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one(".product-price-and-shipping, .price") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="B&M", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from B&M" + )) + + except Exception as e: + logger.error(f"Error parsing B&M product: {e}") + continue + + logger.info(f"BMStoresParser found {len(results)} results") + return results diff --git a/app/services/parsers/boulanger_parser.py b/app/services/parsers/boulanger_parser.py index 5e39cc7..d3d09dc 100644 --- a/app/services/parsers/boulanger_parser.py +++ b/app/services/parsers/boulanger_parser.py @@ -1,130 +1,69 @@ -""" -Boulanger Parser - Specialized parser for Boulanger.com -""" - from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging -from .base_parser import BaseParser, ProductResult - +logger = logging.getLogger(__name__) class BoulangerParser(BaseParser): - """ - Parser for Boulanger.com - - Key selectors: - - Product links: a[href*='/ref/'][href*='_'] - - Images: img.product-image - """ - def __init__(self): - super().__init__( - site_name="Boulanger", - base_url="https://www.boulanger.com" - ) + super().__init__("boulanger.com", "https://www.boulanger.com") def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: - """Parse Boulanger search results""" - soup = BeautifulSoup(html, 'html.parser') - products = [] - - # Try multiple selectors - selectors = [ - "a[href*='/ref/'][href*='_']", - ".product-card a", - "[class*='product'] a[href*='/ref/']", - ] - - links = [] - for selector in selectors: - links = soup.select(selector) - if links: - self.logger.info(f"Found {len(links)} products with selector: {selector}") - break - - if not links: - self.logger.warning("No products found on Boulanger") - return [] - - seen_urls = set() - - for link in links: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Boulanger products + # Config: a[href*='/ref/'][href*='_'] + + cards = soup.select(".product-item, .product-list__item, article") + + for card in cards: try: - href = link.get("href") + # Link + link_el = card.select_one("a[href*='/ref/'], a[class*='link']") + if not link_el: + continue + + href = link_el.get('href') if not href: continue - - full_url = self.make_absolute_url(href) - - if full_url in seen_urls: - continue - seen_urls.add(full_url) - - # Extract title - title = link.get_text(strip=True) + + url = self.make_absolute_url(href) + + # Title + title = None + title_el = card.select_one("h2, h3, .product-label, [class*='title']") + if title_el: + title = title_el.get_text(strip=True) + else: + title = link_el.get_text(strip=True) + if not title: - title = link.get("title", "") - - # Try in parent - if not title or len(title) < 3: - parent = link.find_parent(["article", "li", "div"]) - if parent: - title_elem = parent.select_one("h2, h3, .product-title, [class*='title']") - if title_elem: - title = title_elem.get_text(strip=True) - - if not title or len(title) < 3: continue - # Filter by query - if not self.filter_by_query(title, query): - continue - - # Extract image - image_url = None - parent = link.find_parent(["article", "li", "div"]) - if parent: - img_selectors = [ - "img.product-image", - "img[class*='product']", - "img", - ] - for img_sel in img_selectors: - img = parent.select_one(img_sel) - if img: - image_url = self._get_image_src(img) - if image_url: - image_url = self.make_absolute_url(image_url) - break - - # Extract price + # Image + img_url = self.extract_image_url(card) + + # Price price = None - if parent: - price_selectors = [ - ".price", - "[class*='price'][class*='current']", - "[data-testid='price']", - "span[class*='prix']", - ] - for price_sel in price_selectors: - price_elem = parent.select_one(price_sel) - if price_elem: - price = self.parse_price_text(price_elem.get_text(strip=True)) - if price: - break - - products.append(ProductResult( + price_el = card.select_one(".price, .product-price, [class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( title=title, - url=full_url, - source=self.site_name, + url=url, + source="Boulanger", price=price, currency="EUR", - image_url=image_url, - snippet=f"Product from {self.site_name}" + in_stock=True, + image_url=img_url, + snippet=f"Product from Boulanger" )) - + except Exception as e: - self.logger.error(f"Error parsing Boulanger product: {e}") + logger.error(f"Error parsing Boulanger product: {e}") continue - - self.logger.info(f"Extracted {len(products)} products from Boulanger") - return products + + logger.info(f"BoulangerParser found {len(results)} results") + return results diff --git a/app/services/parsers/carrefour_parser.py b/app/services/parsers/carrefour_parser.py new file mode 100644 index 0000000..07f875b --- /dev/null +++ b/app/services/parsers/carrefour_parser.py @@ -0,0 +1,58 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class CarrefourParser(BaseParser): + def __init__(self): + super().__init__("carrefour.fr", "https://www.carrefour.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Carrefour products + # Config: a[href*='/p/'] + + products = soup.select("article, div[class*='product-card']") + + for product in products: + try: + link_el = product.select_one("a[href*='/p/'], a[href*='/produit']") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title_el = product.select_one("[class*='title'], h2, h3") + title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one("[class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="Carrefour", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Carrefour" + )) + + except Exception as e: + logger.error(f"Error parsing Carrefour product: {e}") + continue + + logger.info(f"CarrefourParser found {len(results)} results") + return results diff --git a/app/services/parsers/cdiscount_parser.py b/app/services/parsers/cdiscount_parser.py index 2fad32d..d433a67 100644 --- a/app/services/parsers/cdiscount_parser.py +++ b/app/services/parsers/cdiscount_parser.py @@ -1,133 +1,70 @@ -""" -Cdiscount Parser - Specialized parser for Cdiscount.com -""" - from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging -from .base_parser import BaseParser, ProductResult - +logger = logging.getLogger(__name__) class CdiscountParser(BaseParser): - """ - Parser for Cdiscount.com - - Key selectors: - - Product links: a.prdtBILnk, a[href*='/f-'][href*='.html'] - - Images: img.lazy, img[data-src] - - Prices: Multiple selectors for current/old prices - """ - def __init__(self): - super().__init__( - site_name="Cdiscount", - base_url="https://www.cdiscount.com" - ) + super().__init__("cdiscount.com", "https://www.cdiscount.com") def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: - """Parse Cdiscount search results""" - soup = BeautifulSoup(html, 'html.parser') - products = [] - - # Try multiple selectors - selectors = [ - "a.prdtBILnk", - "a[href*='/f-'][href*='.html']", - ".prdtBIL a.prdtBIL", - "article a[href*='/f-']", - ] - - links = [] - for selector in selectors: - links = soup.select(selector) - if links: - self.logger.info(f"Found {len(links)} products with selector: {selector}") - break - - if not links: - self.logger.warning("No products found on Cdiscount") - return [] - - seen_urls = set() - - for link in links: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Cdiscount products + # Config: a.prdtBILnk, a[href*='/f-'][href*='.html'] + # We look for the card container + + cards = soup.select("li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil") + + for card in cards: try: - href = link.get("href") + # Link + link_el = card.select_one("a.prdtBILnk, a") + if not link_el: + continue + + href = link_el.get('href') if not href: continue - - full_url = self.make_absolute_url(href) - - if full_url in seen_urls: - continue - seen_urls.add(full_url) - - # Extract title - title = link.get_text(strip=True) + + url = self.make_absolute_url(href) + + # Title + title = None + title_el = card.select_one(".prdtBTitle, .prdtBTit") + if title_el: + title = title_el.get_text(strip=True) + else: + title = link_el.get_text(strip=True) + if not title: - title = link.get("title", "") - - # Try to find title in nearby elements - if not title or len(title) < 3: - parent = link.find_parent(["article", "li", "div"]) - if parent: - title_elem = parent.select_one(".prdtBTitle, h2, h3, .product-title") - if title_elem: - title = title_elem.get_text(strip=True) - - if not title or len(title) < 3: continue - # Filter by query - if not self.filter_by_query(title, query): - continue - - # Extract image - image_url = None - parent = link.find_parent(["article", "li", "div"]) - if parent: - img_selectors = [ - "img.lazy", - "img[data-src]", - "img[src*='image']", - "img", - ] - for img_sel in img_selectors: - img = parent.select_one(img_sel) - if img: - image_url = self._get_image_src(img) - if image_url: - image_url = self.make_absolute_url(image_url) - break - - # Extract price + # Image + img_url = self.extract_image_url(card) + + # Price price = None - if parent: - price_selectors = [ - ".price", - ".prdtPrice", - "[class*='price'][class*='current']", - "span[class*='prix']", - ] - for price_sel in price_selectors: - price_elem = parent.select_one(price_sel) - if price_elem: - price = self.parse_price_text(price_elem.get_text(strip=True)) - if price: - break - - products.append(ProductResult( + price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( title=title, - url=full_url, - source=self.site_name, + url=url, + source="Cdiscount", price=price, currency="EUR", - image_url=image_url, - snippet=f"Product from {self.site_name}" + in_stock=True, + image_url=img_url, + snippet=f"Product from Cdiscount" )) - + except Exception as e: - self.logger.error(f"Error parsing Cdiscount product: {e}") + logger.error(f"Error parsing Cdiscount product: {e}") continue - - self.logger.info(f"Extracted {len(products)} products from Cdiscount") - return products + + logger.info(f"CdiscountParser found {len(results)} results") + return results diff --git a/app/services/parsers/centrakor_parser.py b/app/services/parsers/centrakor_parser.py new file mode 100644 index 0000000..fab5d6c --- /dev/null +++ b/app/services/parsers/centrakor_parser.py @@ -0,0 +1,58 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class CentrakorParser(BaseParser): + def __init__(self): + super().__init__("centrakor.com", "https://www.centrakor.com") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Centrakor products + # Config: a.link.link--block, a.product-item__link + + products = soup.select(".product-item, .product-card") + + for product in products: + try: + link_el = product.select_one("a.product-item__link, a.link") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title_el = product.select_one(".product-item__name, .product-card__title") + title = title_el.get_text(strip=True) if title_el else None + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one(".price, .product-price") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="Centrakor", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Centrakor" + )) + + except Exception as e: + logger.error(f"Error parsing Centrakor product: {e}") + continue + + logger.info(f"CentrakorParser found {len(results)} results") + return results diff --git a/app/services/parsers/darty_parser.py b/app/services/parsers/darty_parser.py index 8ce3174..7a9182d 100644 --- a/app/services/parsers/darty_parser.py +++ b/app/services/parsers/darty_parser.py @@ -1,131 +1,73 @@ -""" -Darty Parser - Specialized parser for Darty.com -""" - from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging -from .base_parser import BaseParser, ProductResult - +logger = logging.getLogger(__name__) class DartyParser(BaseParser): - """ - Parser for Darty.com - - Key selectors: - - Product links: a[href*='/nav/achat/'][href*='.html'] - - Images: img.product_img - """ - def __init__(self): - super().__init__( - site_name="Darty", - base_url="https://www.darty.com" - ) + super().__init__("darty.com", "https://www.darty.com") def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: - """Parse Darty search results""" - soup = BeautifulSoup(html, 'html.parser') - products = [] - - # Try multiple selectors - selectors = [ - "a[href*='/nav/achat/'][href*='.html']", - ".product-card a", - "[class*='product'] a[href*='/nav/achat/']", - ] - - links = [] - for selector in selectors: - links = soup.select(selector) - if links: - self.logger.info(f"Found {len(links)} products with selector: {selector}") - break - - if not links: - self.logger.warning("No products found on Darty") - return [] - - seen_urls = set() - - for link in links: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Darty products + # Config: a[href*='/nav/achat/'][href*='.html'] + + cards = soup.select(".product-card, .product_item, div[class*='product-card']") + + for card in cards: try: - href = link.get("href") + # Link + link_el = card.select_one("a[href*='/nav/achat/'], a.product-link") + if not link_el: + # Try finding any link if specific selector fails + link_el = card.find("a") + + if not link_el: + continue + + href = link_el.get('href') if not href: continue - - full_url = self.make_absolute_url(href) - - if full_url in seen_urls: - continue - seen_urls.add(full_url) - - # Extract title - title = link.get_text(strip=True) + + url = self.make_absolute_url(href) + + # Title + title = None + title_el = card.select_one(".product_name, .product-title, h2, h3") + if title_el: + title = title_el.get_text(strip=True) + else: + title = link_el.get_text(strip=True) + if not title: - title = link.get("title", "") - - # Try in parent - if not title or len(title) < 3: - parent = link.find_parent(["article", "li", "div"]) - if parent: - title_elem = parent.select_one("h2, h3, .product-title, [class*='title']") - if title_elem: - title = title_elem.get_text(strip=True) - - if not title or len(title) < 3: continue - # Filter by query - if not self.filter_by_query(title, query): - continue - - # Extract image - image_url = None - parent = link.find_parent(["article", "li", "div"]) - if parent: - img_selectors = [ - "img.product_img", - "img.product-image", - "img[class*='product']", - "img", - ] - for img_sel in img_selectors: - img = parent.select_one(img_sel) - if img: - image_url = self._get_image_src(img) - if image_url: - image_url = self.make_absolute_url(image_url) - break - - # Extract price + # Image + img_url = self.extract_image_url(card) + + # Price price = None - if parent: - price_selectors = [ - ".price", - "[class*='price'][class*='current']", - "[data-testid='price']", - "span[class*='prix']", - ] - for price_sel in price_selectors: - price_elem = parent.select_one(price_sel) - if price_elem: - price = self.parse_price_text(price_elem.get_text(strip=True)) - if price: - break - - products.append(ProductResult( + price_el = card.select_one(".product_price, .price, [class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( title=title, - url=full_url, - source=self.site_name, + url=url, + source="Darty", price=price, currency="EUR", - image_url=image_url, - snippet=f"Product from {self.site_name}" + in_stock=True, + image_url=img_url, + snippet=f"Product from Darty" )) - + except Exception as e: - self.logger.error(f"Error parsing Darty product: {e}") + logger.error(f"Error parsing Darty product: {e}") continue - - self.logger.info(f"Extracted {len(products)} products from Darty") - return products + + logger.info(f"DartyParser found {len(results)} results") + return results diff --git a/app/services/parsers/eleclerc_parser.py b/app/services/parsers/eleclerc_parser.py new file mode 100644 index 0000000..3c96f12 --- /dev/null +++ b/app/services/parsers/eleclerc_parser.py @@ -0,0 +1,75 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class ELeclercParser(BaseParser): + def __init__(self): + super().__init__("e-leclerc.com", "https://www.e.leclerc") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # E.Leclerc products + # Based on config: a.product-card-link + # We look for the card container + cards = soup.select("div[class*='product-card'], div[class*='product-item']") + + # If no cards found with div, try looking for the links directly and finding parents + if not cards: + links = soup.select("a.product-card-link") + cards = [link.parent for link in links] + + for card in cards: + try: + # Link + link_el = card.select_one("a.product-card-link") or card if card.name == 'a' else card.find('a') + if not link_el: + continue + + href = link_el.get('href') + if not href: + continue + + url = self.make_absolute_url(href) + + # Title + title = None + # Try specific title classes + title_el = card.select_one("[class*='product-title'], [class*='product-label']") + if title_el: + title = title_el.get_text(strip=True) + else: + title = link_el.get_text(strip=True) + + if not title: + continue + + # Image + img_url = self.extract_image_url(card) + + # Price + price = None + price_el = card.select_one("[class*='price'], [class*='amount']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="E.Leclerc", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from E.Leclerc" + )) + + except Exception as e: + logger.error(f"Error parsing E.Leclerc product: {e}") + continue + + logger.info(f"ELeclercParser found {len(results)} results") + return results diff --git a/app/services/parsers/gifi_parser.py b/app/services/parsers/gifi_parser.py new file mode 100644 index 0000000..977d8d4 --- /dev/null +++ b/app/services/parsers/gifi_parser.py @@ -0,0 +1,63 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class GifiParser(BaseParser): + def __init__(self): + super().__init__("gifi.fr", "https://www.gifi.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Gifi products + # Config: a.link + # Wait selector: .product-tile + + products = soup.select(".product-tile, div[class*='product-tile']") + + for product in products: + try: + link_el = product.select_one("a.link") or product.find("a") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title = link_el.get_text(strip=True) + if not title: + # Try finding title in nested elements + title_el = product.select_one("[class*='name'], [class*='title']") + if title_el: + title = title_el.get_text(strip=True) + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one(".price, .value, [class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="Gifi", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Gifi" + )) + + except Exception as e: + logger.error(f"Error parsing Gifi product: {e}") + continue + + logger.info(f"GifiParser found {len(results)} results") + return results diff --git a/app/services/parsers/lafoirfouille_parser.py b/app/services/parsers/lafoirfouille_parser.py new file mode 100644 index 0000000..9148530 --- /dev/null +++ b/app/services/parsers/lafoirfouille_parser.py @@ -0,0 +1,58 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class LaFoirFouilleParser(BaseParser): + def __init__(self): + super().__init__("lafoirfouille.fr", "https://www.lafoirfouille.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # La Foir'Fouille products + # Config: article.product-miniature a.product-thumbnail + + products = soup.select("article.product-miniature") + + for product in products: + try: + link_el = product.select_one("a.product-thumbnail, a.product_img_link") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title_el = product.select_one(".product-title, h3") + title = title_el.get_text(strip=True) if title_el else None + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one(".product-price-and-shipping, .price") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="La Foir'Fouille", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from La Foir'Fouille" + )) + + except Exception as e: + logger.error(f"Error parsing La Foir'Fouille product: {e}") + continue + + logger.info(f"LaFoirFouilleParser found {len(results)} results") + return results diff --git a/app/services/parsers/lincroyable_parser.py b/app/services/parsers/lincroyable_parser.py new file mode 100644 index 0000000..9e80c0f --- /dev/null +++ b/app/services/parsers/lincroyable_parser.py @@ -0,0 +1,58 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class LIncroyableParser(BaseParser): + def __init__(self): + super().__init__("lincroyable.fr", "https://www.lincroyable.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # L'Incroyable products + # Config: a.product-link + + products = soup.select(".product-card, .product-miniature") + + for product in products: + try: + link_el = product.select_one("a.product-link, a[href*='/p/']") + if not link_el: + continue + + href = link_el.get('href') + url = self.make_absolute_url(href) + + title_el = product.select_one(".product-title, h3") + title = title_el.get_text(strip=True) if title_el else None + + if not title: + continue + + img_url = self.extract_image_url(product) + + price = None + price_el = product.select_one(".price, .product-price") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + results.append(ProductResult( + title=title, + url=url, + source="L'Incroyable", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from L'Incroyable" + )) + + except Exception as e: + logger.error(f"Error parsing L'Incroyable product: {e}") + continue + + logger.info(f"LIncroyableParser found {len(results)} results") + return results diff --git a/app/services/parsers/parser_factory.py b/app/services/parsers/parser_factory.py index 9e621f6..b921976 100644 --- a/app/services/parsers/parser_factory.py +++ b/app/services/parsers/parser_factory.py @@ -10,6 +10,16 @@ from .cdiscount_parser import CdiscountParser from .fnac_parser import FnacParser from .darty_parser import DartyParser from .boulanger_parser import BoulangerParser +from .stokomani_parser import StokomaniParser +from .auchan_parser import AuchanParser +from .eleclerc_parser import ELeclercParser +from .gifi_parser import GifiParser +from .action_parser import ActionParser +from .lafoirfouille_parser import LaFoirFouilleParser +from .bmstores_parser import BMStoresParser +from .centrakor_parser import CentrakorParser +from .lincroyable_parser import LIncroyableParser +from .carrefour_parser import CarrefourParser from .generic_parser import GenericParser logger = logging.getLogger(__name__) @@ -30,6 +40,16 @@ class ParserFactory: "fnac.com": FnacParser, "darty.com": DartyParser, "boulanger.com": BoulangerParser, + "stokomani.fr": StokomaniParser, + "auchan.fr": AuchanParser, + "e-leclerc.com": ELeclercParser, + "gifi.fr": GifiParser, + "action.com": ActionParser, + "lafoirfouille.fr": LaFoirFouilleParser, + "bmstores.fr": BMStoresParser, + "centrakor.com": CentrakorParser, + "lincroyable.fr": LIncroyableParser, + "carrefour.fr": CarrefourParser, } # Cache for parser instances (singleton pattern) diff --git a/app/services/parsers/stokomani_parser.py b/app/services/parsers/stokomani_parser.py new file mode 100644 index 0000000..7380b6c --- /dev/null +++ b/app/services/parsers/stokomani_parser.py @@ -0,0 +1,93 @@ +from bs4 import BeautifulSoup +from app.services.parsers.base_parser import BaseParser, ProductResult +import logging + +logger = logging.getLogger(__name__) + +class StokomaniParser(BaseParser): + def __init__(self): + super().__init__("stokomani.fr", "https://www.stokomani.fr") + + def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: + soup = BeautifulSoup(html, "html.parser") + results = [] + + # Find product cards + # Based on inspection, links are a.reversed-link.block + # We'll look for the container of these links + links = soup.select("a.reversed-link.block, a[href*='/products/']") + + seen_urls = set() + + for link in links: + try: + href = link.get('href') + if not href or href in seen_urls: + continue + + # Filter out non-product links + if '/products/' not in href: + continue + + seen_urls.add(href) + url = self.make_absolute_url(href) + + # Title + title = link.get_text(strip=True) + if not title: + # Try finding title in nested elements + title_el = link.find(class_=lambda x: x and 'title' in x) + if title_el: + title = title_el.get_text(strip=True) + + if not title: + continue + + # Find parent card to scope image and price search + # Usually the card is a few levels up + card = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x or "item" in x)) + if not card: + # Fallback: use the link's parent + card = link.parent.parent + + # Image + img_url = self.extract_image_url(card) + + # Price + price = None + # Try specific price selectors first + price_selectors = [ + ".price", ".money", ".current-price", + "span[class*='price']", "div[class*='price']" + ] + + for selector in price_selectors: + price_el = card.select_one(selector) + if price_el: + price = self.parse_price_text(price_el.get_text()) + if price: + break + + # Fallback: Look for price pattern in the card text + if not price: + # Get text but exclude the title to avoid false positives if title has numbers + card_text = card.get_text(" ", strip=True) + price = self.parse_price_text(card_text) + + results.append(ProductResult( + title=title, + url=url, + source="Stokomani", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Stokomani" + )) + + except Exception as e: + logger.error(f"Error parsing Stokomani product: {e}") + continue + + logger.info(f"StokomaniParser found {len(results)} results") + return results diff --git a/verify_search.py b/verify_search.py deleted file mode 100644 index 9b79c1c..0000000 --- a/verify_search.py +++ /dev/null @@ -1,127 +0,0 @@ -import asyncio -import logging -import sys -from unittest.mock import MagicMock, AsyncMock - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -# Mock dependencies -sys.modules["sqlalchemy"] = MagicMock() -sys.modules["sqlalchemy.orm"] = MagicMock() -sys.modules["app.database"] = MagicMock() -sys.modules["app.models"] = MagicMock() -sys.modules["app.services.ai_service"] = MagicMock() -sys.modules["app.services.scraper_service"] = MagicMock() -sys.modules["app.services.light_scraper_service"] = MagicMock() -sys.modules["app.services.settings_service"] = MagicMock() - -# Define dummy classes for schemas -class MockSearchProgress: - def __init__(self, status, total, completed, message, results, current_site=None): - self.status = status - self.total = total - self.completed = completed - self.message = message - self.results = results - self.current_site = current_site - - def model_dump_json(self): - return "json" - -class MockSearchResultItem: - def __init__(self, **kwargs): - for k, v in kwargs.items(): - setattr(self, k, v) - -# Setup schema mocks -schemas_mock = MagicMock() -schemas_mock.SearchProgress = MockSearchProgress -schemas_mock.SearchResultItem = MockSearchResultItem -sys.modules["app.schemas"] = schemas_mock - -# Import services after mocking -# We need to mock direct_search_service before importing search_service -# because search_service imports it. -direct_search_mock = MagicMock() -sys.modules["app.services.direct_search_service"] = direct_search_mock - -# Now we can import search_service -# We might need to mock other things that search_service imports -from app.services import search_service - -# Define a mock SearchResult class matching the one in direct_search_service -class MockSearchResult: - def __init__(self, url, title, source, price=None, currency="EUR", in_stock=None): - self.url = url - self.title = title - self.source = source - self.snippet = "snippet" - self.price = price - self.currency = currency - self.in_stock = in_stock - -async def test_search_flow(): - print("--- Starting Search Flow Verification ---") - - # Setup mocks - db = MagicMock() - - # Mock SettingsService - search_service.SettingsService.get_setting_value.side_effect = lambda db, key, default: default - - # Mock async_playwright in search_service - mock_browser = AsyncMock() - mock_playwright_obj = AsyncMock() - mock_playwright_obj.chromium.connect_over_cdp.return_value = mock_browser - - mock_playwright_manager = MagicMock() - mock_playwright_manager.start = AsyncMock(return_value=mock_playwright_obj) - - search_service.async_playwright = MagicMock(return_value=mock_playwright_manager) - - # Mock direct_search_service.search - mock_results = [ - MockSearchResult("http://site1.com/p1", "Product 1", "site1.com"), - MockSearchResult("http://site2.com/p2", "Product 2", "site2.com"), - ] - # Accept any arguments including browser - direct_search_mock.search = AsyncMock(return_value=mock_results) - direct_search_mock.SearchResult = MockSearchResult - - # Mock light_scraper_service - search_service.light_scraper_service.scrape_url = AsyncMock(return_value=MagicMock(success=False)) - - # Mock _scrape_with_browserless (to avoid actual scraping) - # We need to patch it in the module - original_scrape = search_service._scrape_with_browserless - search_service._scrape_with_browserless = AsyncMock(return_value=MagicMock( - url="http://site1.com/p1", - title="Product 1", - price=10.0, - site_name="Site 1", - site_domain="site1.com" - )) - - # Mock _get_sites to return some dummy sites - mock_site1 = MagicMock(domain="site1.com", name="Site 1", requires_js=True) - mock_site2 = MagicMock(domain="site2.com", name="Site 2", requires_js=True) - search_service._get_sites = MagicMock(return_value=[mock_site1, mock_site2]) - - # Run the search - print("Running search_products...") - try: - async for progress in search_service.search_products("test query", db): - print(f"Event: {progress.status} - {progress.message}") - if progress.results: - print(f" Results: {len(progress.results)}") - except Exception as e: - print(f"Caught exception during search: {e}") - import traceback - traceback.print_exc() - - print("--- Verification Complete ---") - -if __name__ == "__main__": - asyncio.run(test_search_flow())