mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #184 from R0m1k3/antigravity
feat: Implement improved search service with centralized configuratio…
This commit is contained in:
17 files changed
+829
-476
No files matched your search
@@ -157,19 +157,19 @@ SITE_CONFIGS = {
|
||||
# === GRANDES SURFACES ===
|
||||
"e-leclerc.com": {
|
||||
"name": "E.Leclerc",
|
||||
"search_url": "https://www.e-leclerc.com/recherche?text={query}",
|
||||
"product_selector": "a[href*='/cat'], a[href*='/produit'], article a, div[class*='product'] a",
|
||||
"product_image_selector": "img, picture source",
|
||||
"wait_selector": None,
|
||||
"search_url": "https://www.e.leclerc/recherche?q={query}",
|
||||
"product_selector": "a.product-card-link",
|
||||
"product_image_selector": "img.product-card-image, img[class*='product-image']",
|
||||
"wait_selector": ".product-card-link",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
"auchan.fr": {
|
||||
"name": "Auchan",
|
||||
"search_url": "https://www.auchan.fr/search?text={query}",
|
||||
"product_selector": "article.product-item a.product-item__name, a[class*='product-item'][href]",
|
||||
"product_image_selector": "img.product-thumbnail__img, img[class*='product'][class*='img']",
|
||||
"wait_selector": "article.product-item, .product-list",
|
||||
"search_url": "https://www.auchan.fr/recherche?text={query}",
|
||||
"product_selector": "article[class*='product-item'] a[href*='/p-'], div[class*='product-item'] a[href*='/p-']",
|
||||
"product_image_selector": "img[class*='product-thumbnail__img'], img[class*='product-item__image']",
|
||||
"wait_selector": "article[class*='product-item'], div[class*='product-item']",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
|
||||
@@ -258,12 +258,11 @@ class ImprovedSearchService:
|
||||
if results:
|
||||
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
|
||||
|
||||
# LIMIT: Only enrich first 10 products to avoid frontend timeouts
|
||||
# Full enrichment (visiting each product page) takes too long
|
||||
results_to_enrich = results[:10]
|
||||
logger.info(f"⚡ Limiting enrichment to {len(results_to_enrich)} products")
|
||||
# Enrich all results
|
||||
results_to_enrich = results
|
||||
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
|
||||
|
||||
semaphore = asyncio.Semaphore(2) # Limit concurrency
|
||||
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
|
||||
|
||||
async def scrape_with_limit(res):
|
||||
async with semaphore:
|
||||
@@ -272,8 +271,8 @@ class ImprovedSearchService:
|
||||
tasks = [scrape_with_limit(r) for r in results_to_enrich]
|
||||
enriched_results = await asyncio.gather(*tasks)
|
||||
|
||||
# Return enriched results + remaining non-enriched (with None price)
|
||||
results = [r for r in enriched_results if r] + results[10:]
|
||||
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
|
||||
results = [r for r in enriched_results if r]
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class ActionParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("action.com", "https://www.action.com")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Action products
|
||||
# Config: a.group[href^='/fr-fr/p/']
|
||||
|
||||
links = soup.select("a[href^='/fr-fr/p/']")
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
try:
|
||||
href = link.get('href')
|
||||
if not href or href in seen_urls:
|
||||
continue
|
||||
seen_urls.add(href)
|
||||
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Action usually has the card content inside the link
|
||||
title = None
|
||||
title_el = link.select_one("[class*='title'], h3, h4")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
else:
|
||||
title = link.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(link)
|
||||
|
||||
price = None
|
||||
price_el = link.select_one("[class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Action",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Action"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Action product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"ActionParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,57 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class AuchanParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("auchan.fr", "https://www.auchan.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Auchan products are usually in article or div with class product-item
|
||||
products = soup.select("article[class*='product-item'], div[class*='product-item']")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
# Link and Title
|
||||
link_el = product.select_one("a[href*='/p-']")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title = link_el.get('title') or link_el.get_text(strip=True)
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
price_el = product.select_one("span[class*='price__value'], div[class*='price--selling'], span[itemprop='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Auchan",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Auchan"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Auchan product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"AuchanParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,58 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class BMStoresParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("bmstores.fr", "https://bmstores.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# B&M products
|
||||
# Config: a.thumbnail.product-thumbnail
|
||||
|
||||
products = soup.select(".product-miniature, .js-product-miniature")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a.thumbnail.product-thumbnail")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = product.select_one(".product-title, h3")
|
||||
title = title_el.get_text(strip=True) if title_el else None
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".product-price-and-shipping, .price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="B&M",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from B&M"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing B&M product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"BMStoresParser found {len(results)} results")
|
||||
return results
|
||||
@@ -1,130 +1,69 @@
|
||||
"""
|
||||
Boulanger Parser - Specialized parser for Boulanger.com
|
||||
"""
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
from .base_parser import BaseParser, ProductResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class BoulangerParser(BaseParser):
|
||||
"""
|
||||
Parser for Boulanger.com
|
||||
|
||||
Key selectors:
|
||||
- Product links: a[href*='/ref/'][href*='_']
|
||||
- Images: img.product-image
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(
|
||||
site_name="Boulanger",
|
||||
base_url="https://www.boulanger.com"
|
||||
)
|
||||
super().__init__("boulanger.com", "https://www.boulanger.com")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
"""Parse Boulanger search results"""
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
products = []
|
||||
|
||||
# Try multiple selectors
|
||||
selectors = [
|
||||
"a[href*='/ref/'][href*='_']",
|
||||
".product-card a",
|
||||
"[class*='product'] a[href*='/ref/']",
|
||||
]
|
||||
|
||||
links = []
|
||||
for selector in selectors:
|
||||
links = soup.select(selector)
|
||||
if links:
|
||||
self.logger.info(f"Found {len(links)} products with selector: {selector}")
|
||||
break
|
||||
|
||||
if not links:
|
||||
self.logger.warning("No products found on Boulanger")
|
||||
return []
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Boulanger products
|
||||
# Config: a[href*='/ref/'][href*='_']
|
||||
|
||||
cards = soup.select(".product-item, .product-list__item, article")
|
||||
|
||||
for card in cards:
|
||||
try:
|
||||
href = link.get("href")
|
||||
# Link
|
||||
link_el = card.select_one("a[href*='/ref/'], a[class*='link']")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
if not href:
|
||||
continue
|
||||
|
||||
full_url = self.make_absolute_url(href)
|
||||
|
||||
if full_url in seen_urls:
|
||||
continue
|
||||
seen_urls.add(full_url)
|
||||
|
||||
# Extract title
|
||||
title = link.get_text(strip=True)
|
||||
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = None
|
||||
title_el = card.select_one("h2, h3, .product-label, [class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
else:
|
||||
title = link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
title = link.get("title", "")
|
||||
|
||||
# Try in parent
|
||||
if not title or len(title) < 3:
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
title_elem = parent.select_one("h2, h3, .product-title, [class*='title']")
|
||||
if title_elem:
|
||||
title = title_elem.get_text(strip=True)
|
||||
|
||||
if not title or len(title) < 3:
|
||||
continue
|
||||
|
||||
# Filter by query
|
||||
if not self.filter_by_query(title, query):
|
||||
continue
|
||||
|
||||
# Extract image
|
||||
image_url = None
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
img_selectors = [
|
||||
"img.product-image",
|
||||
"img[class*='product']",
|
||||
"img",
|
||||
]
|
||||
for img_sel in img_selectors:
|
||||
img = parent.select_one(img_sel)
|
||||
if img:
|
||||
image_url = self._get_image_src(img)
|
||||
if image_url:
|
||||
image_url = self.make_absolute_url(image_url)
|
||||
break
|
||||
|
||||
# Extract price
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
if parent:
|
||||
price_selectors = [
|
||||
".price",
|
||||
"[class*='price'][class*='current']",
|
||||
"[data-testid='price']",
|
||||
"span[class*='prix']",
|
||||
]
|
||||
for price_sel in price_selectors:
|
||||
price_elem = parent.select_one(price_sel)
|
||||
if price_elem:
|
||||
price = self.parse_price_text(price_elem.get_text(strip=True))
|
||||
if price:
|
||||
break
|
||||
|
||||
products.append(ProductResult(
|
||||
price_el = card.select_one(".price, .product-price, [class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=full_url,
|
||||
source=self.site_name,
|
||||
url=url,
|
||||
source="Boulanger",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
image_url=image_url,
|
||||
snippet=f"Product from {self.site_name}"
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Boulanger"
|
||||
))
|
||||
|
||||
|
||||
except Exception as e:
|
||||
self.logger.error(f"Error parsing Boulanger product: {e}")
|
||||
logger.error(f"Error parsing Boulanger product: {e}")
|
||||
continue
|
||||
|
||||
self.logger.info(f"Extracted {len(products)} products from Boulanger")
|
||||
return products
|
||||
|
||||
logger.info(f"BoulangerParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,58 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class CarrefourParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("carrefour.fr", "https://www.carrefour.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Carrefour products
|
||||
# Config: a[href*='/p/']
|
||||
|
||||
products = soup.select("article, div[class*='product-card']")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a[href*='/p/'], a[href*='/produit']")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = product.select_one("[class*='title'], h2, h3")
|
||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one("[class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Carrefour",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Carrefour"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Carrefour product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"CarrefourParser found {len(results)} results")
|
||||
return results
|
||||
@@ -1,133 +1,70 @@
|
||||
"""
|
||||
Cdiscount Parser - Specialized parser for Cdiscount.com
|
||||
"""
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
from .base_parser import BaseParser, ProductResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class CdiscountParser(BaseParser):
|
||||
"""
|
||||
Parser for Cdiscount.com
|
||||
|
||||
Key selectors:
|
||||
- Product links: a.prdtBILnk, a[href*='/f-'][href*='.html']
|
||||
- Images: img.lazy, img[data-src]
|
||||
- Prices: Multiple selectors for current/old prices
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(
|
||||
site_name="Cdiscount",
|
||||
base_url="https://www.cdiscount.com"
|
||||
)
|
||||
super().__init__("cdiscount.com", "https://www.cdiscount.com")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
"""Parse Cdiscount search results"""
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
products = []
|
||||
|
||||
# Try multiple selectors
|
||||
selectors = [
|
||||
"a.prdtBILnk",
|
||||
"a[href*='/f-'][href*='.html']",
|
||||
".prdtBIL a.prdtBIL",
|
||||
"article a[href*='/f-']",
|
||||
]
|
||||
|
||||
links = []
|
||||
for selector in selectors:
|
||||
links = soup.select(selector)
|
||||
if links:
|
||||
self.logger.info(f"Found {len(links)} products with selector: {selector}")
|
||||
break
|
||||
|
||||
if not links:
|
||||
self.logger.warning("No products found on Cdiscount")
|
||||
return []
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Cdiscount products
|
||||
# Config: a.prdtBILnk, a[href*='/f-'][href*='.html']
|
||||
# We look for the card container
|
||||
|
||||
cards = soup.select("li.prdtBIL, div.prdtBIL, ul#lpBloc > li, div.js-prdt-bil")
|
||||
|
||||
for card in cards:
|
||||
try:
|
||||
href = link.get("href")
|
||||
# Link
|
||||
link_el = card.select_one("a.prdtBILnk, a")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
if not href:
|
||||
continue
|
||||
|
||||
full_url = self.make_absolute_url(href)
|
||||
|
||||
if full_url in seen_urls:
|
||||
continue
|
||||
seen_urls.add(full_url)
|
||||
|
||||
# Extract title
|
||||
title = link.get_text(strip=True)
|
||||
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = None
|
||||
title_el = card.select_one(".prdtBTitle, .prdtBTit")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
else:
|
||||
title = link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
title = link.get("title", "")
|
||||
|
||||
# Try to find title in nearby elements
|
||||
if not title or len(title) < 3:
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
title_elem = parent.select_one(".prdtBTitle, h2, h3, .product-title")
|
||||
if title_elem:
|
||||
title = title_elem.get_text(strip=True)
|
||||
|
||||
if not title or len(title) < 3:
|
||||
continue
|
||||
|
||||
# Filter by query
|
||||
if not self.filter_by_query(title, query):
|
||||
continue
|
||||
|
||||
# Extract image
|
||||
image_url = None
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
img_selectors = [
|
||||
"img.lazy",
|
||||
"img[data-src]",
|
||||
"img[src*='image']",
|
||||
"img",
|
||||
]
|
||||
for img_sel in img_selectors:
|
||||
img = parent.select_one(img_sel)
|
||||
if img:
|
||||
image_url = self._get_image_src(img)
|
||||
if image_url:
|
||||
image_url = self.make_absolute_url(image_url)
|
||||
break
|
||||
|
||||
# Extract price
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
if parent:
|
||||
price_selectors = [
|
||||
".price",
|
||||
".prdtPrice",
|
||||
"[class*='price'][class*='current']",
|
||||
"span[class*='prix']",
|
||||
]
|
||||
for price_sel in price_selectors:
|
||||
price_elem = parent.select_one(price_sel)
|
||||
if price_elem:
|
||||
price = self.parse_price_text(price_elem.get_text(strip=True))
|
||||
if price:
|
||||
break
|
||||
|
||||
products.append(ProductResult(
|
||||
price_el = card.select_one(".price, .prdtPrice, .prdtPInfo .price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=full_url,
|
||||
source=self.site_name,
|
||||
url=url,
|
||||
source="Cdiscount",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
image_url=image_url,
|
||||
snippet=f"Product from {self.site_name}"
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Cdiscount"
|
||||
))
|
||||
|
||||
|
||||
except Exception as e:
|
||||
self.logger.error(f"Error parsing Cdiscount product: {e}")
|
||||
logger.error(f"Error parsing Cdiscount product: {e}")
|
||||
continue
|
||||
|
||||
self.logger.info(f"Extracted {len(products)} products from Cdiscount")
|
||||
return products
|
||||
|
||||
logger.info(f"CdiscountParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,58 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class CentrakorParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("centrakor.com", "https://www.centrakor.com")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Centrakor products
|
||||
# Config: a.link.link--block, a.product-item__link
|
||||
|
||||
products = soup.select(".product-item, .product-card")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a.product-item__link, a.link")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = product.select_one(".product-item__name, .product-card__title")
|
||||
title = title_el.get_text(strip=True) if title_el else None
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".price, .product-price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Centrakor",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Centrakor"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Centrakor product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"CentrakorParser found {len(results)} results")
|
||||
return results
|
||||
@@ -1,131 +1,73 @@
|
||||
"""
|
||||
Darty Parser - Specialized parser for Darty.com
|
||||
"""
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
from .base_parser import BaseParser, ProductResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class DartyParser(BaseParser):
|
||||
"""
|
||||
Parser for Darty.com
|
||||
|
||||
Key selectors:
|
||||
- Product links: a[href*='/nav/achat/'][href*='.html']
|
||||
- Images: img.product_img
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(
|
||||
site_name="Darty",
|
||||
base_url="https://www.darty.com"
|
||||
)
|
||||
super().__init__("darty.com", "https://www.darty.com")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
"""Parse Darty search results"""
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
products = []
|
||||
|
||||
# Try multiple selectors
|
||||
selectors = [
|
||||
"a[href*='/nav/achat/'][href*='.html']",
|
||||
".product-card a",
|
||||
"[class*='product'] a[href*='/nav/achat/']",
|
||||
]
|
||||
|
||||
links = []
|
||||
for selector in selectors:
|
||||
links = soup.select(selector)
|
||||
if links:
|
||||
self.logger.info(f"Found {len(links)} products with selector: {selector}")
|
||||
break
|
||||
|
||||
if not links:
|
||||
self.logger.warning("No products found on Darty")
|
||||
return []
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Darty products
|
||||
# Config: a[href*='/nav/achat/'][href*='.html']
|
||||
|
||||
cards = soup.select(".product-card, .product_item, div[class*='product-card']")
|
||||
|
||||
for card in cards:
|
||||
try:
|
||||
href = link.get("href")
|
||||
# Link
|
||||
link_el = card.select_one("a[href*='/nav/achat/'], a.product-link")
|
||||
if not link_el:
|
||||
# Try finding any link if specific selector fails
|
||||
link_el = card.find("a")
|
||||
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
if not href:
|
||||
continue
|
||||
|
||||
full_url = self.make_absolute_url(href)
|
||||
|
||||
if full_url in seen_urls:
|
||||
continue
|
||||
seen_urls.add(full_url)
|
||||
|
||||
# Extract title
|
||||
title = link.get_text(strip=True)
|
||||
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = None
|
||||
title_el = card.select_one(".product_name, .product-title, h2, h3")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
else:
|
||||
title = link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
title = link.get("title", "")
|
||||
|
||||
# Try in parent
|
||||
if not title or len(title) < 3:
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
title_elem = parent.select_one("h2, h3, .product-title, [class*='title']")
|
||||
if title_elem:
|
||||
title = title_elem.get_text(strip=True)
|
||||
|
||||
if not title or len(title) < 3:
|
||||
continue
|
||||
|
||||
# Filter by query
|
||||
if not self.filter_by_query(title, query):
|
||||
continue
|
||||
|
||||
# Extract image
|
||||
image_url = None
|
||||
parent = link.find_parent(["article", "li", "div"])
|
||||
if parent:
|
||||
img_selectors = [
|
||||
"img.product_img",
|
||||
"img.product-image",
|
||||
"img[class*='product']",
|
||||
"img",
|
||||
]
|
||||
for img_sel in img_selectors:
|
||||
img = parent.select_one(img_sel)
|
||||
if img:
|
||||
image_url = self._get_image_src(img)
|
||||
if image_url:
|
||||
image_url = self.make_absolute_url(image_url)
|
||||
break
|
||||
|
||||
# Extract price
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
if parent:
|
||||
price_selectors = [
|
||||
".price",
|
||||
"[class*='price'][class*='current']",
|
||||
"[data-testid='price']",
|
||||
"span[class*='prix']",
|
||||
]
|
||||
for price_sel in price_selectors:
|
||||
price_elem = parent.select_one(price_sel)
|
||||
if price_elem:
|
||||
price = self.parse_price_text(price_elem.get_text(strip=True))
|
||||
if price:
|
||||
break
|
||||
|
||||
products.append(ProductResult(
|
||||
price_el = card.select_one(".product_price, .price, [class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=full_url,
|
||||
source=self.site_name,
|
||||
url=url,
|
||||
source="Darty",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
image_url=image_url,
|
||||
snippet=f"Product from {self.site_name}"
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Darty"
|
||||
))
|
||||
|
||||
|
||||
except Exception as e:
|
||||
self.logger.error(f"Error parsing Darty product: {e}")
|
||||
logger.error(f"Error parsing Darty product: {e}")
|
||||
continue
|
||||
|
||||
self.logger.info(f"Extracted {len(products)} products from Darty")
|
||||
return products
|
||||
|
||||
logger.info(f"DartyParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,75 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class ELeclercParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("e-leclerc.com", "https://www.e.leclerc")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# E.Leclerc products
|
||||
# Based on config: a.product-card-link
|
||||
# We look for the card container
|
||||
cards = soup.select("div[class*='product-card'], div[class*='product-item']")
|
||||
|
||||
# If no cards found with div, try looking for the links directly and finding parents
|
||||
if not cards:
|
||||
links = soup.select("a.product-card-link")
|
||||
cards = [link.parent for link in links]
|
||||
|
||||
for card in cards:
|
||||
try:
|
||||
# Link
|
||||
link_el = card.select_one("a.product-card-link") or card if card.name == 'a' else card.find('a')
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
if not href:
|
||||
continue
|
||||
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = None
|
||||
# Try specific title classes
|
||||
title_el = card.select_one("[class*='product-title'], [class*='product-label']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
else:
|
||||
title = link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
price_el = card.select_one("[class*='price'], [class*='amount']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="E.Leclerc",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from E.Leclerc"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing E.Leclerc product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"ELeclercParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,63 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class GifiParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("gifi.fr", "https://www.gifi.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Gifi products
|
||||
# Config: a.link
|
||||
# Wait selector: .product-tile
|
||||
|
||||
products = soup.select(".product-tile, div[class*='product-tile']")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a.link") or product.find("a")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title = link_el.get_text(strip=True)
|
||||
if not title:
|
||||
# Try finding title in nested elements
|
||||
title_el = product.select_one("[class*='name'], [class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".price, .value, [class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Gifi",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Gifi"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Gifi product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"GifiParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,58 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class LaFoirFouilleParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("lafoirfouille.fr", "https://www.lafoirfouille.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# La Foir'Fouille products
|
||||
# Config: article.product-miniature a.product-thumbnail
|
||||
|
||||
products = soup.select("article.product-miniature")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a.product-thumbnail, a.product_img_link")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = product.select_one(".product-title, h3")
|
||||
title = title_el.get_text(strip=True) if title_el else None
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".product-price-and-shipping, .price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="La Foir'Fouille",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from La Foir'Fouille"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing La Foir'Fouille product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"LaFoirFouilleParser found {len(results)} results")
|
||||
return results
|
||||
@@ -0,0 +1,58 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class LIncroyableParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("lincroyable.fr", "https://www.lincroyable.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# L'Incroyable products
|
||||
# Config: a.product-link
|
||||
|
||||
products = soup.select(".product-card, .product-miniature")
|
||||
|
||||
for product in products:
|
||||
try:
|
||||
link_el = product.select_one("a.product-link, a[href*='/p/']")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = product.select_one(".product-title, h3")
|
||||
title = title_el.get_text(strip=True) if title_el else None
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".price, .product-price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="L'Incroyable",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from L'Incroyable"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing L'Incroyable product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"LIncroyableParser found {len(results)} results")
|
||||
return results
|
||||
@@ -10,6 +10,16 @@ from .cdiscount_parser import CdiscountParser
|
||||
from .fnac_parser import FnacParser
|
||||
from .darty_parser import DartyParser
|
||||
from .boulanger_parser import BoulangerParser
|
||||
from .stokomani_parser import StokomaniParser
|
||||
from .auchan_parser import AuchanParser
|
||||
from .eleclerc_parser import ELeclercParser
|
||||
from .gifi_parser import GifiParser
|
||||
from .action_parser import ActionParser
|
||||
from .lafoirfouille_parser import LaFoirFouilleParser
|
||||
from .bmstores_parser import BMStoresParser
|
||||
from .centrakor_parser import CentrakorParser
|
||||
from .lincroyable_parser import LIncroyableParser
|
||||
from .carrefour_parser import CarrefourParser
|
||||
from .generic_parser import GenericParser
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -30,6 +40,16 @@ class ParserFactory:
|
||||
"fnac.com": FnacParser,
|
||||
"darty.com": DartyParser,
|
||||
"boulanger.com": BoulangerParser,
|
||||
"stokomani.fr": StokomaniParser,
|
||||
"auchan.fr": AuchanParser,
|
||||
"e-leclerc.com": ELeclercParser,
|
||||
"gifi.fr": GifiParser,
|
||||
"action.com": ActionParser,
|
||||
"lafoirfouille.fr": LaFoirFouilleParser,
|
||||
"bmstores.fr": BMStoresParser,
|
||||
"centrakor.com": CentrakorParser,
|
||||
"lincroyable.fr": LIncroyableParser,
|
||||
"carrefour.fr": CarrefourParser,
|
||||
}
|
||||
|
||||
# Cache for parser instances (singleton pattern)
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class StokomaniParser(BaseParser):
|
||||
def __init__(self):
|
||||
super().__init__("stokomani.fr", "https://www.stokomani.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Find product cards
|
||||
# Based on inspection, links are a.reversed-link.block
|
||||
# We'll look for the container of these links
|
||||
links = soup.select("a.reversed-link.block, a[href*='/products/']")
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
try:
|
||||
href = link.get('href')
|
||||
if not href or href in seen_urls:
|
||||
continue
|
||||
|
||||
# Filter out non-product links
|
||||
if '/products/' not in href:
|
||||
continue
|
||||
|
||||
seen_urls.add(href)
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = link.get_text(strip=True)
|
||||
if not title:
|
||||
# Try finding title in nested elements
|
||||
title_el = link.find(class_=lambda x: x and 'title' in x)
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Find parent card to scope image and price search
|
||||
# Usually the card is a few levels up
|
||||
card = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x or "item" in x))
|
||||
if not card:
|
||||
# Fallback: use the link's parent
|
||||
card = link.parent.parent
|
||||
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
# Try specific price selectors first
|
||||
price_selectors = [
|
||||
".price", ".money", ".current-price",
|
||||
"span[class*='price']", "div[class*='price']"
|
||||
]
|
||||
|
||||
for selector in price_selectors:
|
||||
price_el = card.select_one(selector)
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
if price:
|
||||
break
|
||||
|
||||
# Fallback: Look for price pattern in the card text
|
||||
if not price:
|
||||
# Get text but exclude the title to avoid false positives if title has numbers
|
||||
card_text = card.get_text(" ", strip=True)
|
||||
price = self.parse_price_text(card_text)
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
source="Stokomani",
|
||||
price=price,
|
||||
currency="EUR",
|
||||
in_stock=True,
|
||||
image_url=img_url,
|
||||
snippet=f"Product from Stokomani"
|
||||
))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing Stokomani product: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"StokomaniParser found {len(results)} results")
|
||||
return results
|
||||
@@ -1,127 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Mock dependencies
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
sys.modules["app.database"] = MagicMock()
|
||||
sys.modules["app.models"] = MagicMock()
|
||||
sys.modules["app.services.ai_service"] = MagicMock()
|
||||
sys.modules["app.services.scraper_service"] = MagicMock()
|
||||
sys.modules["app.services.light_scraper_service"] = MagicMock()
|
||||
sys.modules["app.services.settings_service"] = MagicMock()
|
||||
|
||||
# Define dummy classes for schemas
|
||||
class MockSearchProgress:
|
||||
def __init__(self, status, total, completed, message, results, current_site=None):
|
||||
self.status = status
|
||||
self.total = total
|
||||
self.completed = completed
|
||||
self.message = message
|
||||
self.results = results
|
||||
self.current_site = current_site
|
||||
|
||||
def model_dump_json(self):
|
||||
return "json"
|
||||
|
||||
class MockSearchResultItem:
|
||||
def __init__(self, **kwargs):
|
||||
for k, v in kwargs.items():
|
||||
setattr(self, k, v)
|
||||
|
||||
# Setup schema mocks
|
||||
schemas_mock = MagicMock()
|
||||
schemas_mock.SearchProgress = MockSearchProgress
|
||||
schemas_mock.SearchResultItem = MockSearchResultItem
|
||||
sys.modules["app.schemas"] = schemas_mock
|
||||
|
||||
# Import services after mocking
|
||||
# We need to mock direct_search_service before importing search_service
|
||||
# because search_service imports it.
|
||||
direct_search_mock = MagicMock()
|
||||
sys.modules["app.services.direct_search_service"] = direct_search_mock
|
||||
|
||||
# Now we can import search_service
|
||||
# We might need to mock other things that search_service imports
|
||||
from app.services import search_service
|
||||
|
||||
# Define a mock SearchResult class matching the one in direct_search_service
|
||||
class MockSearchResult:
|
||||
def __init__(self, url, title, source, price=None, currency="EUR", in_stock=None):
|
||||
self.url = url
|
||||
self.title = title
|
||||
self.source = source
|
||||
self.snippet = "snippet"
|
||||
self.price = price
|
||||
self.currency = currency
|
||||
self.in_stock = in_stock
|
||||
|
||||
async def test_search_flow():
|
||||
print("--- Starting Search Flow Verification ---")
|
||||
|
||||
# Setup mocks
|
||||
db = MagicMock()
|
||||
|
||||
# Mock SettingsService
|
||||
search_service.SettingsService.get_setting_value.side_effect = lambda db, key, default: default
|
||||
|
||||
# Mock async_playwright in search_service
|
||||
mock_browser = AsyncMock()
|
||||
mock_playwright_obj = AsyncMock()
|
||||
mock_playwright_obj.chromium.connect_over_cdp.return_value = mock_browser
|
||||
|
||||
mock_playwright_manager = MagicMock()
|
||||
mock_playwright_manager.start = AsyncMock(return_value=mock_playwright_obj)
|
||||
|
||||
search_service.async_playwright = MagicMock(return_value=mock_playwright_manager)
|
||||
|
||||
# Mock direct_search_service.search
|
||||
mock_results = [
|
||||
MockSearchResult("http://site1.com/p1", "Product 1", "site1.com"),
|
||||
MockSearchResult("http://site2.com/p2", "Product 2", "site2.com"),
|
||||
]
|
||||
# Accept any arguments including browser
|
||||
direct_search_mock.search = AsyncMock(return_value=mock_results)
|
||||
direct_search_mock.SearchResult = MockSearchResult
|
||||
|
||||
# Mock light_scraper_service
|
||||
search_service.light_scraper_service.scrape_url = AsyncMock(return_value=MagicMock(success=False))
|
||||
|
||||
# Mock _scrape_with_browserless (to avoid actual scraping)
|
||||
# We need to patch it in the module
|
||||
original_scrape = search_service._scrape_with_browserless
|
||||
search_service._scrape_with_browserless = AsyncMock(return_value=MagicMock(
|
||||
url="http://site1.com/p1",
|
||||
title="Product 1",
|
||||
price=10.0,
|
||||
site_name="Site 1",
|
||||
site_domain="site1.com"
|
||||
))
|
||||
|
||||
# Mock _get_sites to return some dummy sites
|
||||
mock_site1 = MagicMock(domain="site1.com", name="Site 1", requires_js=True)
|
||||
mock_site2 = MagicMock(domain="site2.com", name="Site 2", requires_js=True)
|
||||
search_service._get_sites = MagicMock(return_value=[mock_site1, mock_site2])
|
||||
|
||||
# Run the search
|
||||
print("Running search_products...")
|
||||
try:
|
||||
async for progress in search_service.search_products("test query", db):
|
||||
print(f"Event: {progress.status} - {progress.message}")
|
||||
if progress.results:
|
||||
print(f" Results: {len(progress.results)}")
|
||||
except Exception as e:
|
||||
print(f"Caught exception during search: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
|
||||
print("--- Verification Complete ---")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_search_flow())
|
||||
Reference in new issue
Block a user