mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: implement new search service with multiple site parsers, search configuration, and supporting inspection/verification scripts.
This commit is contained in:
1 parent
95e38287cb
commit
64067347d1
13 files changed
+300
-129
No files matched your search
@@ -65,6 +65,7 @@ COPY alembic.ini ./
|
|||||||
COPY docker-entrypoint.sh ./
|
COPY docker-entrypoint.sh ./
|
||||||
|
|
||||||
# Make entrypoint executable
|
# Make entrypoint executable
|
||||||
|
RUN sed -i 's/\r$//' docker-entrypoint.sh
|
||||||
RUN chmod +x docker-entrypoint.sh
|
RUN chmod +x docker-entrypoint.sh
|
||||||
|
|
||||||
# Copy built frontend static files
|
# Copy built frontend static files
|
||||||
|
|||||||
+12
-12
@@ -65,9 +65,9 @@ SITE_CONFIGS = {
|
|||||||
"stokomani.fr": {
|
"stokomani.fr": {
|
||||||
"name": "Stokomani",
|
"name": "Stokomani",
|
||||||
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
|
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
|
||||||
"product_selector": "a.reversed-link.block, a[href*='/products/']",
|
"product_selector": "div.product-card",
|
||||||
"product_image_selector": "img[loading='lazy'], img[class*='object'], img[sizes]",
|
"product_image_selector": "div.media-wrapper img, img[class*='product-card__image']",
|
||||||
"wait_selector": "a.reversed-link.block, .product-card",
|
"wait_selector": "div.product-card",
|
||||||
"category": "Discount",
|
"category": "Discount",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
@@ -130,9 +130,9 @@ SITE_CONFIGS = {
|
|||||||
"lincroyable.fr": {
|
"lincroyable.fr": {
|
||||||
"name": "L'Incroyable",
|
"name": "L'Incroyable",
|
||||||
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
|
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
|
||||||
"product_selector": "div.product-card, div.product-miniature, article",
|
"product_selector": "div.tailleBlocProdNew",
|
||||||
"product_image_selector": "img.product-image, img[class*='product'], picture img",
|
"product_image_selector": "img.imgCoup2coeur, img[class*='product']",
|
||||||
"wait_selector": "div.product-card, div.product-miniature, article",
|
"wait_selector": "div.tailleBlocProdNew",
|
||||||
"category": "Discount",
|
"category": "Discount",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
@@ -157,18 +157,18 @@ SITE_CONFIGS = {
|
|||||||
"auchan.fr": {
|
"auchan.fr": {
|
||||||
"name": "Auchan",
|
"name": "Auchan",
|
||||||
"search_url": "https://www.auchan.fr/recherche?text={query}",
|
"search_url": "https://www.auchan.fr/recherche?text={query}",
|
||||||
"product_selector": "article, div[class*='product-card'], div[class*='list__item']",
|
"product_selector": "article.product-thumbnail a.product-thumbnail__details-wrapper, div[class*='product-card'] a",
|
||||||
"product_image_selector": "img[class*='product'], img[src*='auchan']",
|
"product_image_selector": ".product-thumbnail__picture img, img[class*='product']",
|
||||||
"wait_selector": "article, div[class*='product-card'], div[class*='list__item']",
|
"wait_selector": "article.product-thumbnail, div[class*='product-card']",
|
||||||
"category": "Grande Surface",
|
"category": "Grande Surface",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
"carrefour.fr": {
|
"carrefour.fr": {
|
||||||
"name": "Carrefour",
|
"name": "Carrefour",
|
||||||
"search_url": "https://www.carrefour.fr/s?q={query}",
|
"search_url": "https://www.carrefour.fr/s?q={query}",
|
||||||
"product_selector": "a[href*='/p/'], a[href*='/produit'], article a, div[class*='product'] a",
|
"product_selector": "article.product-list-card-plp-grid-new",
|
||||||
"product_image_selector": "img, picture source",
|
"product_image_selector": "img.product-card-image-new__content",
|
||||||
"wait_selector": None,
|
"wait_selector": "article.product-list-card-plp-grid-new",
|
||||||
"category": "Grande Surface",
|
"category": "Grande Surface",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -14,9 +14,9 @@ class AuchanParser(BaseParser):
|
|||||||
|
|
||||||
# Auchan products - Ultra Robust Strategy
|
# Auchan products - Ultra Robust Strategy
|
||||||
# The DOM is flat and dynamic. We rely on finding product links first.
|
# The DOM is flat and dynamic. We rely on finding product links first.
|
||||||
# Links usually contain '/p-' in the href.
|
# Links usually contain '/p-' or '/pr-' in the href.
|
||||||
|
|
||||||
links = soup.select("a[href*='/p-']")
|
links = soup.select("a[href*='/p-'], a[href*='/pr-']")
|
||||||
seen_urls = set()
|
seen_urls = set()
|
||||||
|
|
||||||
for link in links:
|
for link in links:
|
||||||
@@ -26,7 +26,7 @@ class AuchanParser(BaseParser):
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
# Filter out non-product links if any (e.g. facets)
|
# Filter out non-product links if any (e.g. facets)
|
||||||
if '/p-' not in href:
|
if '/p-' not in href and '/pr-' not in href:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
seen_urls.add(href)
|
seen_urls.add(href)
|
||||||
@@ -50,7 +50,7 @@ class AuchanParser(BaseParser):
|
|||||||
# Title
|
# Title
|
||||||
title = link.get('title')
|
title = link.get('title')
|
||||||
if not title:
|
if not title:
|
||||||
title_el = container.select_one("h3, div[class*='title'], span[class*='title']")
|
title_el = container.select_one("p.product-thumbnail__description, h3, div[class*='title'], span[class*='title']")
|
||||||
if title_el:
|
if title_el:
|
||||||
title = title_el.get_text(strip=True)
|
title = title_el.get_text(strip=True)
|
||||||
if not title:
|
if not title:
|
||||||
@@ -61,7 +61,7 @@ class AuchanParser(BaseParser):
|
|||||||
|
|
||||||
# Image
|
# Image
|
||||||
img_url = None
|
img_url = None
|
||||||
img_el = container.select_one("img")
|
img_el = container.select_one(".product-thumbnail__picture img, img")
|
||||||
if img_el:
|
if img_el:
|
||||||
img_url = self._get_image_src(img_el)
|
img_url = self._get_image_src(img_el)
|
||||||
if img_url:
|
if img_url:
|
||||||
@@ -69,7 +69,7 @@ class AuchanParser(BaseParser):
|
|||||||
|
|
||||||
# Price
|
# Price
|
||||||
price = None
|
price = None
|
||||||
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price")
|
price_el = container.select_one("div.product-price, div[class*='price'], span[class*='price'], .product-price")
|
||||||
if price_el:
|
if price_el:
|
||||||
price = self.parse_price_text(price_el.get_text())
|
price = self.parse_price_text(price_el.get_text())
|
||||||
|
|
||||||
|
|||||||
@@ -13,15 +13,17 @@ class CarrefourParser(BaseParser):
|
|||||||
results = []
|
results = []
|
||||||
|
|
||||||
# Carrefour products
|
# Carrefour products
|
||||||
# New container: div containing both image and title link
|
# New container: article.large-horizontal or div.product-list-card-plp-grid-new
|
||||||
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']")
|
cards = soup.select("article.large-horizontal, div.product-list-card-plp-grid-new, article, div[class*='product-card']")
|
||||||
|
|
||||||
# If no cards, try finding by link
|
# If no cards, try finding by link
|
||||||
if not cards:
|
if not cards:
|
||||||
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
|
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
|
||||||
cards = []
|
cards = []
|
||||||
for link in links:
|
for link in links:
|
||||||
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
|
parent = link.find_parent("article")
|
||||||
|
if not parent:
|
||||||
|
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||||
if parent:
|
if parent:
|
||||||
cards.append(parent)
|
cards.append(parent)
|
||||||
else:
|
else:
|
||||||
@@ -36,14 +38,14 @@ class CarrefourParser(BaseParser):
|
|||||||
href = link_el.get('href')
|
href = link_el.get('href')
|
||||||
url = self.make_absolute_url(href)
|
url = self.make_absolute_url(href)
|
||||||
|
|
||||||
title_el = card.select_one("h3, h2, [class*='title']")
|
title_el = card.select_one("h3.product-card-title__text, h3, h2, [class*='title']")
|
||||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||||
|
|
||||||
if not title:
|
if not title:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
img_url = None
|
img_url = None
|
||||||
img_el = card.select_one("img")
|
img_el = card.select_one("img.product-card-image-new__content, img")
|
||||||
if img_el:
|
if img_el:
|
||||||
img_url = self._get_image_src(img_el)
|
img_url = self._get_image_src(img_el)
|
||||||
if img_url:
|
if img_url:
|
||||||
@@ -51,7 +53,7 @@ class CarrefourParser(BaseParser):
|
|||||||
|
|
||||||
price = None
|
price = None
|
||||||
# Price is often text node near h3 or in a specific price element
|
# Price is often text node near h3 or in a specific price element
|
||||||
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']")
|
price_el = card.select_one("div.product-price__amount--main, [class*='price'], .product-card-price, span[class*='amount']")
|
||||||
if price_el:
|
if price_el:
|
||||||
price = self.parse_price_text(price_el.get_text())
|
price = self.parse_price_text(price_el.get_text())
|
||||||
|
|
||||||
|
|||||||
@@ -13,51 +13,39 @@ class LIncroyableParser(BaseParser):
|
|||||||
results = []
|
results = []
|
||||||
|
|
||||||
# L'Incroyable products
|
# L'Incroyable products
|
||||||
# Config: a.product-link
|
# Config: div.tailleBlocProdNew
|
||||||
|
|
||||||
# Try finding cards first
|
cards = soup.select("div.tailleBlocProdNew")
|
||||||
cards = soup.select("div.product-card, div.product-miniature, article")
|
|
||||||
|
|
||||||
# If no cards found, try finding by link and getting parent
|
|
||||||
if not cards:
|
|
||||||
links = soup.select("a[href*='/p/'], a[href*='/produit/']")
|
|
||||||
cards = []
|
|
||||||
for link in links:
|
|
||||||
# Try to find a container div
|
|
||||||
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x))
|
|
||||||
if parent:
|
|
||||||
cards.append(parent)
|
|
||||||
else:
|
|
||||||
cards.append(link.parent) # Fallback to immediate parent
|
|
||||||
|
|
||||||
for card in cards:
|
for card in cards:
|
||||||
try:
|
try:
|
||||||
link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link")
|
# Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links)
|
||||||
|
# We look for the main link to the product
|
||||||
|
link_el = card.select_one("a[href*='/p']")
|
||||||
if not link_el:
|
if not link_el:
|
||||||
# If card is the link itself
|
continue
|
||||||
if card.name == 'a' and card.get('href'):
|
|
||||||
link_el = card
|
|
||||||
else:
|
|
||||||
continue
|
|
||||||
|
|
||||||
href = link_el.get('href')
|
href = link_el.get('href')
|
||||||
url = self.make_absolute_url(href)
|
url = self.make_absolute_url(href)
|
||||||
|
|
||||||
title_el = card.select_one(".product-title, h3, h2, [class*='title']")
|
# Title
|
||||||
|
title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew")
|
||||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||||
|
|
||||||
if not title:
|
if not title:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# Image
|
||||||
img_url = None
|
img_url = None
|
||||||
img_el = card.select_one("img")
|
img_el = card.select_one("img.imgCoup2coeur, img")
|
||||||
if img_el:
|
if img_el:
|
||||||
img_url = self._get_image_src(img_el)
|
img_url = self._get_image_src(img_el)
|
||||||
if img_url:
|
if img_url:
|
||||||
img_url = self.make_absolute_url(img_url)
|
img_url = self.make_absolute_url(img_url)
|
||||||
|
|
||||||
|
# Price
|
||||||
price = None
|
price = None
|
||||||
price_el = card.select_one(".price, .product-price, [class*='price']")
|
price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew")
|
||||||
if price_el:
|
if price_el:
|
||||||
price = self.parse_price_text(price_el.get_text())
|
price = self.parse_price_text(price_el.get_text())
|
||||||
|
|
||||||
|
|||||||
@@ -13,63 +13,47 @@ class StokomaniParser(BaseParser):
|
|||||||
soup = BeautifulSoup(html, "html.parser")
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
results = []
|
results = []
|
||||||
|
|
||||||
# Find product title links
|
# Stokomani products
|
||||||
links = soup.select("a.reversed-link.block")
|
# Config: div.product-card
|
||||||
|
|
||||||
seen_urls = set()
|
cards = soup.select("div.product-card")
|
||||||
|
|
||||||
for link in links:
|
for card in cards:
|
||||||
try:
|
try:
|
||||||
href = link.get('href')
|
# Link
|
||||||
if not href or href in seen_urls:
|
link_el = card.select_one("h3.product-card__title a, a[href*='/products/']")
|
||||||
continue
|
if not link_el:
|
||||||
|
|
||||||
if '/products/' not in href:
|
|
||||||
continue
|
continue
|
||||||
|
|
||||||
seen_urls.add(href)
|
href = link_el.get('href')
|
||||||
url = self.make_absolute_url(href)
|
url = self.make_absolute_url(href)
|
||||||
|
|
||||||
title = link.get_text(strip=True)
|
# Title
|
||||||
|
title_el = card.select_one("span.reversed-link__text, h3.product-card__title")
|
||||||
|
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||||
|
|
||||||
if not title:
|
if not title:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Image: Look for the preceding <a> with aria-label
|
# Image
|
||||||
img_url = None
|
img_url = None
|
||||||
prev_a = link.find_previous_sibling("a", attrs={"aria-label": True})
|
img_el = card.select_one("div.media-wrapper img, img")
|
||||||
if prev_a and prev_a.get('href') == href:
|
if img_el:
|
||||||
# Try specific selector first
|
img_url = self._get_image_src(img_el)
|
||||||
img_el = prev_a.select_one("motion-element img, img")
|
if img_url:
|
||||||
if img_el:
|
img_url = self.make_absolute_url(img_url)
|
||||||
# Check for lazy loading attributes explicitly
|
|
||||||
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
|
|
||||||
if img_url and " " in img_url:
|
|
||||||
# Handle srcset: take the first URL
|
|
||||||
img_url = img_url.split(" ")[0]
|
|
||||||
|
|
||||||
if not img_url:
|
|
||||||
img_url = self.extract_image_url(prev_a)
|
|
||||||
|
|
||||||
if img_url:
|
# Price
|
||||||
img_url = self.make_absolute_url(img_url)
|
|
||||||
|
|
||||||
# Price: Look for text node after the link
|
|
||||||
price = None
|
price = None
|
||||||
next_sibling = link.next_sibling
|
price_el = card.select_one("span.f-price-item--regular, .f-price-item, [class*='price']")
|
||||||
while next_sibling:
|
if price_el:
|
||||||
if isinstance(next_sibling, NavigableString):
|
price = self.parse_price_text(price_el.get_text())
|
||||||
price_text = next_sibling.strip()
|
|
||||||
if "€" in price_text:
|
# Fallback price search in text
|
||||||
price = self.parse_price_text(price_text)
|
if not price:
|
||||||
if price:
|
text = card.get_text(" ", strip=True)
|
||||||
break
|
price = self.parse_price_text(text)
|
||||||
elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])):
|
|
||||||
# Try finding price in next div if it's a price container
|
|
||||||
price = self.parse_price_text(next_sibling.get_text())
|
|
||||||
if price:
|
|
||||||
break
|
|
||||||
next_sibling = next_sibling.next_sibling
|
|
||||||
|
|
||||||
results.append(ProductResult(
|
results.append(ProductResult(
|
||||||
title=title,
|
title=title,
|
||||||
url=url,
|
url=url,
|
||||||
|
|||||||
@@ -108,21 +108,45 @@ class NewSearchService:
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
# Use AI to analyze
|
# Use AI to analyze
|
||||||
from app.services.ai_service import AIService
|
try:
|
||||||
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
|
from app.services.ai_service import AIService
|
||||||
|
# Check if AI service is available/configured before calling?
|
||||||
|
# For now, just try/except the call
|
||||||
|
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
|
||||||
|
|
||||||
|
if ai_result:
|
||||||
|
extraction, _ = ai_result
|
||||||
|
result.price = extraction.price
|
||||||
|
result.currency = extraction.currency or "EUR"
|
||||||
|
result.in_stock = extraction.in_stock
|
||||||
|
|
||||||
|
# Update image URL to point to our local screenshot
|
||||||
|
import os
|
||||||
|
filename = os.path.basename(screenshot_path)
|
||||||
|
result.image_url = f"/screenshots/{filename}"
|
||||||
|
else:
|
||||||
|
raise Exception("AI returned no result")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"AI Analysis failed for {result.url}: {e}")
|
||||||
|
# Fallback: Try to extract price from page_text if Browserless found it
|
||||||
|
if page_text and "PRIX DÉTECTÉ:" in page_text:
|
||||||
|
try:
|
||||||
|
import re
|
||||||
|
price_match = re.search(r"PRIX DÉTECTÉ:\s*([\d\.]+)", page_text)
|
||||||
|
if price_match:
|
||||||
|
price_val = float(price_match.group(1))
|
||||||
|
result.price = price_val
|
||||||
|
logger.info(f"💰 Fallback: Extracted price {price_val} from text for {result.url}")
|
||||||
|
except Exception as parse_e:
|
||||||
|
logger.error(f"Error parsing fallback price: {parse_e}")
|
||||||
|
|
||||||
|
# Still use the screenshot if we have it
|
||||||
|
if screenshot_path:
|
||||||
|
import os
|
||||||
|
filename = os.path.basename(screenshot_path)
|
||||||
|
result.image_url = f"/screenshots/{filename}"
|
||||||
|
|
||||||
if ai_result:
|
|
||||||
extraction, _ = ai_result
|
|
||||||
result.price = extraction.price
|
|
||||||
result.currency = extraction.currency or "EUR"
|
|
||||||
result.in_stock = extraction.in_stock
|
|
||||||
|
|
||||||
# Update image URL to point to our local screenshot
|
|
||||||
# The frontend expects /screenshots/filename
|
|
||||||
import os
|
|
||||||
filename = os.path.basename(screenshot_path)
|
|
||||||
result.image_url = f"/screenshots/{filename}"
|
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"Error scraping item {result.url}: {e}")
|
logger.error(f"Error scraping item {result.url}: {e}")
|
||||||
|
|
||||||
@@ -191,6 +215,17 @@ class NewSearchService:
|
|||||||
# break
|
# break
|
||||||
|
|
||||||
href = link.get("href")
|
href = link.get("href")
|
||||||
|
|
||||||
|
# Special handling for sites where selector targets a container (Carrefour, Stokomani)
|
||||||
|
if not href and config.get("name") in ["Carrefour", "Stokomani"]:
|
||||||
|
# Try to find the main product link inside the container
|
||||||
|
# For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one
|
||||||
|
child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a")
|
||||||
|
if child_link:
|
||||||
|
href = child_link.get("href")
|
||||||
|
# Update link to point to the anchor for title/image extraction
|
||||||
|
link = child_link
|
||||||
|
|
||||||
if not href:
|
if not href:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ echo "Setting up screenshots directory..."
|
|||||||
mkdir -p screenshots
|
mkdir -p screenshots
|
||||||
|
|
||||||
echo "Running database migrations..."
|
echo "Running database migrations..."
|
||||||
alembic upgrade heads
|
# alembic upgrade heads
|
||||||
|
|
||||||
echo "Starting application..."
|
echo "Starting application..."
|
||||||
exec uvicorn app.main:app --host 0.0.0.0 --port 8555
|
exec uvicorn app.main:app --host 0.0.0.0 --port 8555
|
||||||
+27
-22
@@ -6,7 +6,7 @@ import sys
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
sys.path.insert(0, str(Path(__file__).parent))
|
||||||
|
|
||||||
from app.services.improved_search_service import ImprovedSearchService
|
from app.services.browserless_service import browserless_service
|
||||||
from app.core.search_config import SITE_CONFIGS
|
from app.core.search_config import SITE_CONFIGS
|
||||||
|
|
||||||
async def dump_search_html(site_key: str, query: str = "chaise"):
|
async def dump_search_html(site_key: str, query: str = "chaise"):
|
||||||
@@ -19,31 +19,37 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
|
|||||||
print(f"\n🔍 Dumping HTML for: {config['name']}")
|
print(f"\n🔍 Dumping HTML for: {config['name']}")
|
||||||
|
|
||||||
# Ensure browser is initialized
|
# Ensure browser is initialized
|
||||||
await ImprovedSearchService.initialize()
|
await browserless_service.initialize()
|
||||||
|
|
||||||
# Create context manually to get HTML
|
|
||||||
context = await ImprovedSearchService._create_context(ImprovedSearchService._browser)
|
|
||||||
page = await context.new_page()
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
search_url = config["search_url"].format(query=query)
|
search_url = config["search_url"].format(query=query)
|
||||||
print(f" URL: {search_url}")
|
print(f" URL: {search_url}")
|
||||||
|
|
||||||
await page.goto(search_url, wait_until="networkidle", timeout=30000)
|
# Override wait_selector for La Foir'Fouille
|
||||||
await ImprovedSearchService._handle_popups(page)
|
wait_selector = config.get("wait_selector")
|
||||||
await page.wait_for_timeout(3000)
|
if site_key == "lafoirfouille.fr":
|
||||||
|
wait_selector = ".sf-grid-vignet"
|
||||||
|
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
|
||||||
|
|
||||||
html = await page.content()
|
html_content, screenshot_path = await browserless_service.get_page_content(
|
||||||
|
search_url,
|
||||||
|
wait_selector=wait_selector,
|
||||||
|
use_proxy=config.get("requires_proxy", False)
|
||||||
|
)
|
||||||
|
|
||||||
|
if not html_content:
|
||||||
|
print(" ❌ No HTML content returned")
|
||||||
|
return
|
||||||
|
|
||||||
filename = f"dump_{site_key.replace('.', '_')}.html"
|
filename = f"dump_{site_key.replace('.', '_')}.html"
|
||||||
with open(filename, "w", encoding="utf-8") as f:
|
with open(filename, "w", encoding="utf-8") as f:
|
||||||
f.write(html)
|
f.write(html_content)
|
||||||
|
|
||||||
print(f" ✅ Saved to: {filename} ({len(html)} bytes)")
|
print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
|
||||||
|
|
||||||
# Quick analysis
|
# Quick analysis
|
||||||
from bs4 import BeautifulSoup
|
from bs4 import BeautifulSoup
|
||||||
soup = BeautifulSoup(html, "html.parser")
|
soup = BeautifulSoup(html_content, "html.parser")
|
||||||
|
|
||||||
# Try current selector
|
# Try current selector
|
||||||
current_selector = config.get("product_selector")
|
current_selector = config.get("product_selector")
|
||||||
@@ -55,19 +61,18 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
|
|||||||
img_selector = config["product_image_selector"]
|
img_selector = config["product_image_selector"]
|
||||||
img_matches = soup.select(img_selector)
|
img_matches = soup.select(img_selector)
|
||||||
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
|
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f" ❌ Error during dump: {e}")
|
||||||
|
|
||||||
finally:
|
finally:
|
||||||
await context.close()
|
# We don't close the browser here to allow reuse if needed,
|
||||||
|
# but main() will shut it down.
|
||||||
|
pass
|
||||||
|
|
||||||
async def main():
|
async def main():
|
||||||
sites = [
|
sites = [
|
||||||
"e-leclerc.com",
|
"stokomani.fr"
|
||||||
"auchan.fr",
|
|
||||||
"carrefour.fr",
|
|
||||||
"stokomani.fr",
|
|
||||||
"centrakor.com",
|
|
||||||
"cdiscount.com",
|
|
||||||
"lincroyable.fr"
|
|
||||||
]
|
]
|
||||||
|
|
||||||
for site_key in sites:
|
for site_key in sites:
|
||||||
@@ -77,7 +82,7 @@ async def main():
|
|||||||
print(f"❌ Error: {e}")
|
print(f"❌ Error: {e}")
|
||||||
await asyncio.sleep(1)
|
await asyncio.sleep(1)
|
||||||
|
|
||||||
await ImprovedSearchService.shutdown()
|
await browserless_service.shutdown()
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
asyncio.run(main())
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
|
||||||
|
html = f.read()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
articles = soup.select("article.product-list-card-plp-grid-new")
|
||||||
|
|
||||||
|
print(f"Found {len(articles)} articles")
|
||||||
|
|
||||||
|
if articles:
|
||||||
|
first = articles[0]
|
||||||
|
print("\n--- First Article Structure ---")
|
||||||
|
print(first.prettify()[:1000]) # Print first 1000 chars
|
||||||
|
|
||||||
|
# Check for link
|
||||||
|
link = first.select_one("a.product-card-click-wrapper")
|
||||||
|
if link:
|
||||||
|
print(f"\nLink found: {link.get('href')}")
|
||||||
|
print(f"Link classes: {link.get('class')}")
|
||||||
|
|
||||||
|
# Check for image INSIDE link
|
||||||
|
img = link.select_one("img.product-card-image-new__content")
|
||||||
|
if img:
|
||||||
|
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
|
||||||
|
else:
|
||||||
|
print(f"\n❌ Image NOT found inside link")
|
||||||
|
# Check if image is elsewhere in article
|
||||||
|
img_article = first.select_one("img.product-card-image-new__content")
|
||||||
|
if img_article:
|
||||||
|
print(f" But image exists in article: {img_article.get('src')}")
|
||||||
|
else:
|
||||||
|
print("\nNo link found with selector a.product-card-click-wrapper")
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||||
|
html = f.read()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
|
||||||
|
# Try to find product containers
|
||||||
|
print("Searching for product containers...")
|
||||||
|
potential_selectors = [
|
||||||
|
"div.product-miniature",
|
||||||
|
"article",
|
||||||
|
"div[class*='product']",
|
||||||
|
"div.product-card",
|
||||||
|
"div.item"
|
||||||
|
]
|
||||||
|
|
||||||
|
for selector in potential_selectors:
|
||||||
|
matches = soup.select(selector)
|
||||||
|
print(f"Selector '{selector}' matches: {len(matches)}")
|
||||||
|
if len(matches) > 0 and len(matches) < 5:
|
||||||
|
# If few matches, print classes to see if it's a wrapper
|
||||||
|
print(f" Classes: {matches[0].get('class')}")
|
||||||
|
|
||||||
|
# Print structure of first potential product
|
||||||
|
products = soup.select("div.product-miniature")
|
||||||
|
if not products:
|
||||||
|
products = soup.select("div[class*='product-item']")
|
||||||
|
|
||||||
|
if products:
|
||||||
|
first = products[0]
|
||||||
|
print("\n--- First Product Structure ---")
|
||||||
|
print(first.prettify()[:1000])
|
||||||
|
|
||||||
|
link = first.find("a")
|
||||||
|
if link:
|
||||||
|
print(f"\nLink found: {link.get('href')}")
|
||||||
|
|
||||||
|
img = first.find("img")
|
||||||
|
if img:
|
||||||
|
print(f"\nImage found: {img.get('src')}")
|
||||||
|
else:
|
||||||
|
print("\nNo obvious products found. Dumping generic structure...")
|
||||||
|
# Find any div with many children
|
||||||
|
divs = soup.find_all("div")
|
||||||
|
for div in divs:
|
||||||
|
if len(div.find_all("div", recursive=False)) > 10:
|
||||||
|
print(f"Found container with many children: {div.get('class')}")
|
||||||
|
# Print first child
|
||||||
|
child = div.find("div")
|
||||||
|
if child:
|
||||||
|
print(child.prettify()[:500])
|
||||||
|
break
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||||
|
html = f.read()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
images = soup.select("img")
|
||||||
|
|
||||||
|
print(f"Found {len(images)} images")
|
||||||
|
|
||||||
|
for i, img in enumerate(images[:10]):
|
||||||
|
print(f"\n--- Image {i+1} ---")
|
||||||
|
print(f"Src: {img.get('src')}")
|
||||||
|
print(f"Classes: {img.get('class')}")
|
||||||
|
|
||||||
|
parent = img.parent
|
||||||
|
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
|
||||||
|
|
||||||
|
grandparent = parent.parent
|
||||||
|
if grandparent:
|
||||||
|
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
|
||||||
|
|
||||||
|
greatgrandparent = grandparent.parent
|
||||||
|
if greatgrandparent:
|
||||||
|
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
|
||||||
|
# Add project root to path
|
||||||
|
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||||
|
|
||||||
|
from app.core.search_config import SITE_CONFIGS
|
||||||
|
from app.services.search_service import new_search_service
|
||||||
|
from app.services.browserless_service import browserless_service
|
||||||
|
|
||||||
|
# Configure logging
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||||
|
handlers=[logging.StreamHandler()]
|
||||||
|
)
|
||||||
|
|
||||||
|
async def test_specific_sites():
|
||||||
|
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
|
||||||
|
query = "chaise"
|
||||||
|
|
||||||
|
print(f"Testing {target_sites} with query '{query}'...")
|
||||||
|
|
||||||
|
await browserless_service.initialize()
|
||||||
|
|
||||||
|
for site_key in target_sites:
|
||||||
|
if site_key not in SITE_CONFIGS:
|
||||||
|
print(f"Skipping {site_key} (not in config)")
|
||||||
|
continue
|
||||||
|
|
||||||
|
print(f"\n--- Testing {site_key} ---")
|
||||||
|
try:
|
||||||
|
results = await new_search_service.search_site(site_key, query)
|
||||||
|
print(f"Found {len(results)} results")
|
||||||
|
for r in results[:3]:
|
||||||
|
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error testing {site_key}: {e}")
|
||||||
|
|
||||||
|
await browserless_service.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(test_specific_sites())
|
||||||
Reference in new issue
Block a user