Merge pull request #184 from R0m1k3/antigravity

feat: Implement improved search service with centralized configuratio…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 18:06:05 +01:00
commit 668784ca14
17 files changed
+829 -476

No files matched your search

+8 -8
View File
@@ -157,19 +157,19 @@ SITE_CONFIGS = {
# === GRANDES SURFACES ===
"e-leclerc.com": {
"name": "E.Leclerc",
"search_url": "https://www.e-leclerc.com/recherche?text={query}",
"product_selector": "a[href*='/cat'], a[href*='/produit'], article a, div[class*='product'] a",
"product_image_selector": "img, picture source",
"wait_selector": None,
"search_url": "https://www.e.leclerc/recherche?q={query}",
"product_selector": "a.product-card-link",
"product_image_selector": "img.product-card-image, img[class*='product-image']",
"wait_selector": ".product-card-link",
"category": "Grande Surface",
"requires_proxy": False,
},
"auchan.fr": {
"name": "Auchan",
"search_url": "https://www.auchan.fr/search?text={query}",
"product_selector": "article.product-item a.product-item__name, a[class*='product-item'][href]",
"product_image_selector": "img.product-thumbnail__img, img[class*='product'][class*='img']",
"wait_selector": "article.product-item, .product-list",
"search_url": "https://www.auchan.fr/recherche?text={query}",
"product_selector": "article[class*='product-item'] a[href*='/p-'], div[class*='product-item'] a[href*='/p-']",
"product_image_selector": "img[class*='product-thumbnail__img'], img[class*='product-item__image']",
"wait_selector": "article[class*='product-item'], div[class*='product-item']",
"category": "Grande Surface",
"requires_proxy": False,
},
+6 -7
View File
@@ -258,12 +258,11 @@ class ImprovedSearchService:
if results:
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
# LIMIT: Only enrich first 10 products to avoid frontend timeouts
# Full enrichment (visiting each product page) takes too long
results_to_enrich = results[:10]
logger.info(f"⚡ Limiting enrichment to {len(results_to_enrich)} products")
# Enrich all results
results_to_enrich = results
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
semaphore = asyncio.Semaphore(2) # Limit concurrency
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
async def scrape_with_limit(res):
async with semaphore:
@@ -272,8 +271,8 @@ class ImprovedSearchService:
tasks = [scrape_with_limit(r) for r in results_to_enrich]
enriched_results = await asyncio.gather(*tasks)
# Return enriched results + remaining non-enriched (with None price)
results = [r for r in enriched_results if r] + results[10:]
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
results = [r for r in enriched_results if r]
finally:
await context.close()
+65
View File
@@ -0,0 +1,65 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class ActionParser(BaseParser):
def __init__(self):
super().__init__("action.com", "https://www.action.com")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Action products
# Config: a.group[href^='/fr-fr/p/']
links = soup.select("a[href^='/fr-fr/p/']")
seen_urls = set()
for link in links:
try:
href = link.get('href')
if not href or href in seen_urls:
continue
seen_urls.add(href)
url = self.make_absolute_url(href)
# Action usually has the card content inside the link
title = None
title_el = link.select_one("[class*='title'], h3, h4")
if title_el:
title = title_el.get_text(strip=True)
else:
title = link.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(link)
price = None
price_el = link.select_one("[class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Action",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Action"
))
except Exception as e:
logger.error(f"Error parsing Action product: {e}")
continue
logger.info(f"ActionParser found {len(results)} results")
return results
+57
View File
@@ -0,0 +1,57 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class AuchanParser(BaseParser):
def __init__(self):
super().__init__("auchan.fr", "https://www.auchan.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Auchan products are usually in article or div with class product-item
products = soup.select("article[class*='product-item'], div[class*='product-item']")
for product in products:
try:
# Link and Title
link_el = product.select_one("a[href*='/p-']")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title = link_el.get('title') or link_el.get_text(strip=True)
if not title:
continue
# Image
img_url = self.extract_image_url(product)
# Price
price = None
price_el = product.select_one("span[class*='price__value'], div[class*='price--selling'], span[itemprop='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Auchan",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Auchan"
))
except Exception as e:
logger.error(f"Error parsing Auchan product: {e}")
continue
logger.info(f"AuchanParser found {len(results)} results")
return results
+58
View File
@@ -0,0 +1,58 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class BMStoresParser(BaseParser):
def __init__(self):
super().__init__("bmstores.fr", "https://bmstores.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# B&M products
# Config: a.thumbnail.product-thumbnail
products = soup.select(".product-miniature, .js-product-miniature")
for product in products:
try:
link_el = product.select_one("a.thumbnail.product-thumbnail")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-title, h3")
title = title_el.get_text(strip=True) if title_el else None
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".product-price-and-shipping, .price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="B&M",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from B&M"
))
except Exception as e:
logger.error(f"Error parsing B&M product: {e}")
continue
logger.info(f"BMStoresParser found {len(results)} results")
return results
+49 -110
View File
@@ -1,130 +1,69 @@
"""
Boulanger Parser - Specialized parser for Boulanger.com
"""
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
from .base_parser import BaseParser, ProductResult
logger = logging.getLogger(__name__)
class BoulangerParser(BaseParser):
"""
Parser for Boulanger.com
Key selectors:
- Product links: a[href*='/ref/'][href*='_']
- Images: img.product-image
"""
def __init__(self):
super().__init__(
site_name="Boulanger",
base_url="https://www.boulanger.com"
)
super().__init__("boulanger.com", "https://www.boulanger.com")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
"""Parse Boulanger search results"""
soup = BeautifulSoup(html, 'html.parser')
products = []
# Try multiple selectors
selectors = [
"a[href*='/ref/'][href*='_']",
".product-card a",
"[class*='product'] a[href*='/ref/']",
]
links = []
for selector in selectors:
links = soup.select(selector)
if links:
self.logger.info(f"Found {len(links)} products with selector: {selector}")
break
if not links:
self.logger.warning("No products found on Boulanger")
return []
seen_urls = set()
for link in links:
soup = BeautifulSoup(html, "html.parser")
results = []
# Boulanger products
# Config: a[href*='/ref/'][href*='_']
cards = soup.select(".product-item, .product-list__item, article")
for card in cards:
try:
href = link.get("href")
# Link
link_el = card.select_one("a[href*='/ref/'], a[class*='link']")
if not link_el:
continue
href = link_el.get('href')
if not href:
continue
full_url = self.make_absolute_url(href)
if full_url in seen_urls:
continue
seen_urls.add(full_url)
# Extract title
title = link.get_text(strip=True)
url = self.make_absolute_url(href)
# Title
title = None
title_el = card.select_one("h2, h3, .product-label, [class*='title']")
if title_el:
title = title_el.get_text(strip=True)
else:
title = link_el.get_text(strip=True)
if not title:
title = link.get("title", "")
# Try in parent
if not title or len(title) < 3:
parent = link.find_parent(["article", "li", "div"])
if parent:
title_elem = parent.select_one("h2, h3, .product-title, [class*='title']")
if title_elem:
title = title_elem.get_text(strip=True)
if not title or len(title) < 3:
continue
# Filter by query
if not self.filter_by_query(title, query):
continue
# Extract image
image_url = None
parent = link.find_parent(["article", "li", "div"])
if parent:
img_selectors = [
"img.product-image",
"img[class*='product']",
"img",
]
for img_sel in img_selectors:
img = parent.select_one(img_sel)
if img:
image_url = self._get_image_src(img)
if image_url:
image_url = self.make_absolute_url(image_url)
break
# Extract price
# Image
img_url = self.extract_image_url(card)
# Price
price = None
if parent:
price_selectors = [
".price",
"[class*='price'][class*='current']",
"[data-testid='price']",
"span[class*='prix']",
]
for price_sel in price_selectors:
price_elem = parent.select_one(price_sel)
if price_elem:
price = self.parse_price_text(price_elem.get_text(strip=True))
if price:
break
products.append(ProductResult(
price_el = card.select_one(".price, .product-price, [class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=full_url,
source=self.site_name,
url=url,
source="Boulanger",
price=price,
currency="EUR",
image_url=image_url,
snippet=f"Product from {self.site_name}"
in_stock=True,
image_url=img_url,
snippet=f"Product from Boulanger"
))
except Exception as e:
self.logger.error(f"Error parsing Boulanger product: {e}")
logger.error(f"Error parsing Boulanger product: {e}")
continue
self.logger.info(f"Extracted {len(products)} products from Boulanger")
return products
logger.info(f"BoulangerParser found {len(results)} results")
return results
+58
View File
@@ -0,0 +1,58 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class CarrefourParser(BaseParser):
def __init__(self):
super().__init__("carrefour.fr", "https://www.carrefour.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Carrefour products
# Config: a[href*='/p/']
products = soup.select("article, div[class*='product-card']")
for product in products:
try:
link_el = product.select_one("a[href*='/p/'], a[href*='/produit']")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one("[class*='title'], h2, h3")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one("[class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Carrefour",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Carrefour"
))
except Exception as e:
logger.error(f"Error parsing Carrefour product: {e}")
continue
logger.info(f"CarrefourParser found {len(results)} results")
return results
+50 -113
View File
@@ -1,133 +1,70 @@
"""
Cdiscount Parser - Specialized parser for Cdiscount.com
"""
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
from .base_parser import BaseParser, ProductResult
logger = logging.getLogger(__name__)
class CdiscountParser(BaseParser):
"""
Parser for Cdiscount.com
Key selectors:
- Product links: a.prdtBILnk, a[href*='/f-'][href*='.html']
- Images: img.lazy, img[data-src]
- Prices: Multiple selectors for current/old prices
"""
def __init__(self):
super().__init__(
site_name="Cdiscount",
base_url="https://www.cdiscount.com"
)
super().__init__("cdiscount.com", "https://www.cdiscount.com")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
"""Parse Cdiscount search results"""
soup = BeautifulSoup(html, 'html.parser')
products = []
# Try multiple selectors
selectors = [
"a.prdtBILnk",
"a[href*='/f-'][href*='.html']",
".prdtBIL a.prdtBIL",
"article a[href*='/f-']",
]
links = []
for selector in selectors:
links = soup.select(selector)
if links:
self.logger.info(f"Found {len(links)} products with selector: {selector}")
break
if not links:
self.logger.warning("No products found on Cdiscount")
return []
seen_urls = set()
for link in links:
soup = BeautifulSoup(html, "html.parser")
results = []
# Cdiscount products
# Config: a.prdtBILnk, a[href*='/f-'][href*='.html']
# We look for the card container
cards = soup.select("li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil")
for card in cards:
try:
href = link.get("href")
# Link
link_el = card.select_one("a.prdtBILnk, a")
if not link_el:
continue
href = link_el.get('href')
if not href:
continue
full_url = self.make_absolute_url(href)
if full_url in seen_urls:
continue
seen_urls.add(full_url)
# Extract title
title = link.get_text(strip=True)
url = self.make_absolute_url(href)
# Title
title = None
title_el = card.select_one(".prdtBTitle, .prdtBTit")
if title_el:
title = title_el.get_text(strip=True)
else:
title = link_el.get_text(strip=True)
if not title:
title = link.get("title", "")
# Try to find title in nearby elements
if not title or len(title) < 3:
parent = link.find_parent(["article", "li", "div"])
if parent:
title_elem = parent.select_one(".prdtBTitle, h2, h3, .product-title")
if title_elem:
title = title_elem.get_text(strip=True)
if not title or len(title) < 3:
continue
# Filter by query
if not self.filter_by_query(title, query):
continue
# Extract image
image_url = None
parent = link.find_parent(["article", "li", "div"])
if parent:
img_selectors = [
"img.lazy",
"img[data-src]",
"img[src*='image']",
"img",
]
for img_sel in img_selectors:
img = parent.select_one(img_sel)
if img:
image_url = self._get_image_src(img)
if image_url:
image_url = self.make_absolute_url(image_url)
break
# Extract price
# Image
img_url = self.extract_image_url(card)
# Price
price = None
if parent:
price_selectors = [
".price",
".prdtPrice",
"[class*='price'][class*='current']",
"span[class*='prix']",
]
for price_sel in price_selectors:
price_elem = parent.select_one(price_sel)
if price_elem:
price = self.parse_price_text(price_elem.get_text(strip=True))
if price:
break
products.append(ProductResult(
price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=full_url,
source=self.site_name,
url=url,
source="Cdiscount",
price=price,
currency="EUR",
image_url=image_url,
snippet=f"Product from {self.site_name}"
in_stock=True,
image_url=img_url,
snippet=f"Product from Cdiscount"
))
except Exception as e:
self.logger.error(f"Error parsing Cdiscount product: {e}")
logger.error(f"Error parsing Cdiscount product: {e}")
continue
self.logger.info(f"Extracted {len(products)} products from Cdiscount")
return products
logger.info(f"CdiscountParser found {len(results)} results")
return results
+58
View File
@@ -0,0 +1,58 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class CentrakorParser(BaseParser):
def __init__(self):
super().__init__("centrakor.com", "https://www.centrakor.com")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Centrakor products
# Config: a.link.link--block, a.product-item__link
products = soup.select(".product-item, .product-card")
for product in products:
try:
link_el = product.select_one("a.product-item__link, a.link")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-item__name, .product-card__title")
title = title_el.get_text(strip=True) if title_el else None
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".price, .product-price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Centrakor",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Centrakor"
))
except Exception as e:
logger.error(f"Error parsing Centrakor product: {e}")
continue
logger.info(f"CentrakorParser found {len(results)} results")
return results
+53 -111
View File
@@ -1,131 +1,73 @@
"""
Darty Parser - Specialized parser for Darty.com
"""
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
from .base_parser import BaseParser, ProductResult
logger = logging.getLogger(__name__)
class DartyParser(BaseParser):
"""
Parser for Darty.com
Key selectors:
- Product links: a[href*='/nav/achat/'][href*='.html']
- Images: img.product_img
"""
def __init__(self):
super().__init__(
site_name="Darty",
base_url="https://www.darty.com"
)
super().__init__("darty.com", "https://www.darty.com")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
"""Parse Darty search results"""
soup = BeautifulSoup(html, 'html.parser')
products = []
# Try multiple selectors
selectors = [
"a[href*='/nav/achat/'][href*='.html']",
".product-card a",
"[class*='product'] a[href*='/nav/achat/']",
]
links = []
for selector in selectors:
links = soup.select(selector)
if links:
self.logger.info(f"Found {len(links)} products with selector: {selector}")
break
if not links:
self.logger.warning("No products found on Darty")
return []
seen_urls = set()
for link in links:
soup = BeautifulSoup(html, "html.parser")
results = []
# Darty products
# Config: a[href*='/nav/achat/'][href*='.html']
cards = soup.select(".product-card, .product_item, div[class*='product-card']")
for card in cards:
try:
href = link.get("href")
# Link
link_el = card.select_one("a[href*='/nav/achat/'], a.product-link")
if not link_el:
# Try finding any link if specific selector fails
link_el = card.find("a")
if not link_el:
continue
href = link_el.get('href')
if not href:
continue
full_url = self.make_absolute_url(href)
if full_url in seen_urls:
continue
seen_urls.add(full_url)
# Extract title
title = link.get_text(strip=True)
url = self.make_absolute_url(href)
# Title
title = None
title_el = card.select_one(".product_name, .product-title, h2, h3")
if title_el:
title = title_el.get_text(strip=True)
else:
title = link_el.get_text(strip=True)
if not title:
title = link.get("title", "")
# Try in parent
if not title or len(title) < 3:
parent = link.find_parent(["article", "li", "div"])
if parent:
title_elem = parent.select_one("h2, h3, .product-title, [class*='title']")
if title_elem:
title = title_elem.get_text(strip=True)
if not title or len(title) < 3:
continue
# Filter by query
if not self.filter_by_query(title, query):
continue
# Extract image
image_url = None
parent = link.find_parent(["article", "li", "div"])
if parent:
img_selectors = [
"img.product_img",
"img.product-image",
"img[class*='product']",
"img",
]
for img_sel in img_selectors:
img = parent.select_one(img_sel)
if img:
image_url = self._get_image_src(img)
if image_url:
image_url = self.make_absolute_url(image_url)
break
# Extract price
# Image
img_url = self.extract_image_url(card)
# Price
price = None
if parent:
price_selectors = [
".price",
"[class*='price'][class*='current']",
"[data-testid='price']",
"span[class*='prix']",
]
for price_sel in price_selectors:
price_elem = parent.select_one(price_sel)
if price_elem:
price = self.parse_price_text(price_elem.get_text(strip=True))
if price:
break
products.append(ProductResult(
price_el = card.select_one(".product_price, .price, [class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=full_url,
source=self.site_name,
url=url,
source="Darty",
price=price,
currency="EUR",
image_url=image_url,
snippet=f"Product from {self.site_name}"
in_stock=True,
image_url=img_url,
snippet=f"Product from Darty"
))
except Exception as e:
self.logger.error(f"Error parsing Darty product: {e}")
logger.error(f"Error parsing Darty product: {e}")
continue
self.logger.info(f"Extracted {len(products)} products from Darty")
return products
logger.info(f"DartyParser found {len(results)} results")
return results
+75
View File
@@ -0,0 +1,75 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class ELeclercParser(BaseParser):
def __init__(self):
super().__init__("e-leclerc.com", "https://www.e.leclerc")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# E.Leclerc products
# Based on config: a.product-card-link
# We look for the card container
cards = soup.select("div[class*='product-card'], div[class*='product-item']")
# If no cards found with div, try looking for the links directly and finding parents
if not cards:
links = soup.select("a.product-card-link")
cards = [link.parent for link in links]
for card in cards:
try:
# Link
link_el = card.select_one("a.product-card-link") or card if card.name == 'a' else card.find('a')
if not link_el:
continue
href = link_el.get('href')
if not href:
continue
url = self.make_absolute_url(href)
# Title
title = None
# Try specific title classes
title_el = card.select_one("[class*='product-title'], [class*='product-label']")
if title_el:
title = title_el.get_text(strip=True)
else:
title = link_el.get_text(strip=True)
if not title:
continue
# Image
img_url = self.extract_image_url(card)
# Price
price = None
price_el = card.select_one("[class*='price'], [class*='amount']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="E.Leclerc",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from E.Leclerc"
))
except Exception as e:
logger.error(f"Error parsing E.Leclerc product: {e}")
continue
logger.info(f"ELeclercParser found {len(results)} results")
return results
+63
View File
@@ -0,0 +1,63 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class GifiParser(BaseParser):
def __init__(self):
super().__init__("gifi.fr", "https://www.gifi.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Gifi products
# Config: a.link
# Wait selector: .product-tile
products = soup.select(".product-tile, div[class*='product-tile']")
for product in products:
try:
link_el = product.select_one("a.link") or product.find("a")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title = link_el.get_text(strip=True)
if not title:
# Try finding title in nested elements
title_el = product.select_one("[class*='name'], [class*='title']")
if title_el:
title = title_el.get_text(strip=True)
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".price, .value, [class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="Gifi",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Gifi"
))
except Exception as e:
logger.error(f"Error parsing Gifi product: {e}")
continue
logger.info(f"GifiParser found {len(results)} results")
return results
@@ -0,0 +1,58 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class LaFoirFouilleParser(BaseParser):
def __init__(self):
super().__init__("lafoirfouille.fr", "https://www.lafoirfouille.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# La Foir'Fouille products
# Config: article.product-miniature a.product-thumbnail
products = soup.select("article.product-miniature")
for product in products:
try:
link_el = product.select_one("a.product-thumbnail, a.product_img_link")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-title, h3")
title = title_el.get_text(strip=True) if title_el else None
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".product-price-and-shipping, .price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="La Foir'Fouille",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from La Foir'Fouille"
))
except Exception as e:
logger.error(f"Error parsing La Foir'Fouille product: {e}")
continue
logger.info(f"LaFoirFouilleParser found {len(results)} results")
return results
@@ -0,0 +1,58 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class LIncroyableParser(BaseParser):
def __init__(self):
super().__init__("lincroyable.fr", "https://www.lincroyable.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# L'Incroyable products
# Config: a.product-link
products = soup.select(".product-card, .product-miniature")
for product in products:
try:
link_el = product.select_one("a.product-link, a[href*='/p/']")
if not link_el:
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = product.select_one(".product-title, h3")
title = title_el.get_text(strip=True) if title_el else None
if not title:
continue
img_url = self.extract_image_url(product)
price = None
price_el = product.select_one(".price, .product-price")
if price_el:
price = self.parse_price_text(price_el.get_text())
results.append(ProductResult(
title=title,
url=url,
source="L'Incroyable",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from L'Incroyable"
))
except Exception as e:
logger.error(f"Error parsing L'Incroyable product: {e}")
continue
logger.info(f"LIncroyableParser found {len(results)} results")
return results
+20
View File
@@ -10,6 +10,16 @@ from .cdiscount_parser import CdiscountParser
from .fnac_parser import FnacParser
from .darty_parser import DartyParser
from .boulanger_parser import BoulangerParser
from .stokomani_parser import StokomaniParser
from .auchan_parser import AuchanParser
from .eleclerc_parser import ELeclercParser
from .gifi_parser import GifiParser
from .action_parser import ActionParser
from .lafoirfouille_parser import LaFoirFouilleParser
from .bmstores_parser import BMStoresParser
from .centrakor_parser import CentrakorParser
from .lincroyable_parser import LIncroyableParser
from .carrefour_parser import CarrefourParser
from .generic_parser import GenericParser
logger = logging.getLogger(__name__)
@@ -30,6 +40,16 @@ class ParserFactory:
"fnac.com": FnacParser,
"darty.com": DartyParser,
"boulanger.com": BoulangerParser,
"stokomani.fr": StokomaniParser,
"auchan.fr": AuchanParser,
"e-leclerc.com": ELeclercParser,
"gifi.fr": GifiParser,
"action.com": ActionParser,
"lafoirfouille.fr": LaFoirFouilleParser,
"bmstores.fr": BMStoresParser,
"centrakor.com": CentrakorParser,
"lincroyable.fr": LIncroyableParser,
"carrefour.fr": CarrefourParser,
}
# Cache for parser instances (singleton pattern)
+93
View File
@@ -0,0 +1,93 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
logger = logging.getLogger(__name__)
class StokomaniParser(BaseParser):
def __init__(self):
super().__init__("stokomani.fr", "https://www.stokomani.fr")
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Find product cards
# Based on inspection, links are a.reversed-link.block
# We'll look for the container of these links
links = soup.select("a.reversed-link.block, a[href*='/products/']")
seen_urls = set()
for link in links:
try:
href = link.get('href')
if not href or href in seen_urls:
continue
# Filter out non-product links
if '/products/' not in href:
continue
seen_urls.add(href)
url = self.make_absolute_url(href)
# Title
title = link.get_text(strip=True)
if not title:
# Try finding title in nested elements
title_el = link.find(class_=lambda x: x and 'title' in x)
if title_el:
title = title_el.get_text(strip=True)
if not title:
continue
# Find parent card to scope image and price search
# Usually the card is a few levels up
card = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x or "item" in x))
if not card:
# Fallback: use the link's parent
card = link.parent.parent
# Image
img_url = self.extract_image_url(card)
# Price
price = None
# Try specific price selectors first
price_selectors = [
".price", ".money", ".current-price",
"span[class*='price']", "div[class*='price']"
]
for selector in price_selectors:
price_el = card.select_one(selector)
if price_el:
price = self.parse_price_text(price_el.get_text())
if price:
break
# Fallback: Look for price pattern in the card text
if not price:
# Get text but exclude the title to avoid false positives if title has numbers
card_text = card.get_text(" ", strip=True)
price = self.parse_price_text(card_text)
results.append(ProductResult(
title=title,
url=url,
source="Stokomani",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Stokomani"
))
except Exception as e:
logger.error(f"Error parsing Stokomani product: {e}")
continue
logger.info(f"StokomaniParser found {len(results)} results")
return results
-127
View File
@@ -1,127 +0,0 @@
import asyncio
import logging
import sys
from unittest.mock import MagicMock, AsyncMock
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# Mock dependencies
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
sys.modules["app.database"] = MagicMock()
sys.modules["app.models"] = MagicMock()
sys.modules["app.services.ai_service"] = MagicMock()
sys.modules["app.services.scraper_service"] = MagicMock()
sys.modules["app.services.light_scraper_service"] = MagicMock()
sys.modules["app.services.settings_service"] = MagicMock()
# Define dummy classes for schemas
class MockSearchProgress:
def __init__(self, status, total, completed, message, results, current_site=None):
self.status = status
self.total = total
self.completed = completed
self.message = message
self.results = results
self.current_site = current_site
def model_dump_json(self):
return "json"
class MockSearchResultItem:
def __init__(self, **kwargs):
for k, v in kwargs.items():
setattr(self, k, v)
# Setup schema mocks
schemas_mock = MagicMock()
schemas_mock.SearchProgress = MockSearchProgress
schemas_mock.SearchResultItem = MockSearchResultItem
sys.modules["app.schemas"] = schemas_mock
# Import services after mocking
# We need to mock direct_search_service before importing search_service
# because search_service imports it.
direct_search_mock = MagicMock()
sys.modules["app.services.direct_search_service"] = direct_search_mock
# Now we can import search_service
# We might need to mock other things that search_service imports
from app.services import search_service
# Define a mock SearchResult class matching the one in direct_search_service
class MockSearchResult:
def __init__(self, url, title, source, price=None, currency="EUR", in_stock=None):
self.url = url
self.title = title
self.source = source
self.snippet = "snippet"
self.price = price
self.currency = currency
self.in_stock = in_stock
async def test_search_flow():
print("--- Starting Search Flow Verification ---")
# Setup mocks
db = MagicMock()
# Mock SettingsService
search_service.SettingsService.get_setting_value.side_effect = lambda db, key, default: default
# Mock async_playwright in search_service
mock_browser = AsyncMock()
mock_playwright_obj = AsyncMock()
mock_playwright_obj.chromium.connect_over_cdp.return_value = mock_browser
mock_playwright_manager = MagicMock()
mock_playwright_manager.start = AsyncMock(return_value=mock_playwright_obj)
search_service.async_playwright = MagicMock(return_value=mock_playwright_manager)
# Mock direct_search_service.search
mock_results = [
MockSearchResult("http://site1.com/p1", "Product 1", "site1.com"),
MockSearchResult("http://site2.com/p2", "Product 2", "site2.com"),
]
# Accept any arguments including browser
direct_search_mock.search = AsyncMock(return_value=mock_results)
direct_search_mock.SearchResult = MockSearchResult
# Mock light_scraper_service
search_service.light_scraper_service.scrape_url = AsyncMock(return_value=MagicMock(success=False))
# Mock _scrape_with_browserless (to avoid actual scraping)
# We need to patch it in the module
original_scrape = search_service._scrape_with_browserless
search_service._scrape_with_browserless = AsyncMock(return_value=MagicMock(
url="http://site1.com/p1",
title="Product 1",
price=10.0,
site_name="Site 1",
site_domain="site1.com"
))
# Mock _get_sites to return some dummy sites
mock_site1 = MagicMock(domain="site1.com", name="Site 1", requires_js=True)
mock_site2 = MagicMock(domain="site2.com", name="Site 2", requires_js=True)
search_service._get_sites = MagicMock(return_value=[mock_site1, mock_site2])
# Run the search
print("Running search_products...")
try:
async for progress in search_service.search_products("test query", db):
print(f"Event: {progress.status} - {progress.message}")
if progress.results:
print(f" Results: {len(progress.results)}")
except Exception as e:
print(f"Caught exception during search: {e}")
import traceback
traceback.print_exc()
print("--- Verification Complete ---")
if __name__ == "__main__":
asyncio.run(test_search_flow())