feat: implement new search service with multiple site parsers, search configuration, and supporting inspection/verification scripts.

This commit is contained in:
Michael committed 2025-11-30 22:04:14 +01:00
1 parent 95e38287cb
commit 64067347d1
13 files changed
+297 -126

No files matched your search

+1
View File
@@ -65,6 +65,7 @@ COPY alembic.ini ./
COPY docker-entrypoint.sh ./ COPY docker-entrypoint.sh ./
# Make entrypoint executable # Make entrypoint executable
RUN sed -i 's/\r$//' docker-entrypoint.sh
RUN chmod +x docker-entrypoint.sh RUN chmod +x docker-entrypoint.sh
# Copy built frontend static files # Copy built frontend static files
+12 -12
View File
@@ -65,9 +65,9 @@ SITE_CONFIGS = {
"stokomani.fr": { "stokomani.fr": {
"name": "Stokomani", "name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}", "search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "a.reversed-link.block, a[href*='/products/']", "product_selector": "div.product-card",
"product_image_selector": "img[loading='lazy'], img[class*='object'], img[sizes]", "product_image_selector": "div.media-wrapper img, img[class*='product-card__image']",
"wait_selector": "a.reversed-link.block, .product-card", "wait_selector": "div.product-card",
"category": "Discount", "category": "Discount",
"requires_proxy": False, "requires_proxy": False,
}, },
@@ -130,9 +130,9 @@ SITE_CONFIGS = {
"lincroyable.fr": { "lincroyable.fr": {
"name": "L'Incroyable", "name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche-query={query}/", "search_url": "https://www.lincroyable.fr/recherche-query={query}/",
"product_selector": "div.product-card, div.product-miniature, article", "product_selector": "div.tailleBlocProdNew",
"product_image_selector": "img.product-image, img[class*='product'], picture img", "product_image_selector": "img.imgCoup2coeur, img[class*='product']",
"wait_selector": "div.product-card, div.product-miniature, article", "wait_selector": "div.tailleBlocProdNew",
"category": "Discount", "category": "Discount",
"requires_proxy": False, "requires_proxy": False,
}, },
@@ -157,18 +157,18 @@ SITE_CONFIGS = {
"auchan.fr": { "auchan.fr": {
"name": "Auchan", "name": "Auchan",
"search_url": "https://www.auchan.fr/recherche?text={query}", "search_url": "https://www.auchan.fr/recherche?text={query}",
"product_selector": "article, div[class*='product-card'], div[class*='list__item']", "product_selector": "article.product-thumbnail a.product-thumbnail__details-wrapper, div[class*='product-card'] a",
"product_image_selector": "img[class*='product'], img[src*='auchan']", "product_image_selector": ".product-thumbnail__picture img, img[class*='product']",
"wait_selector": "article, div[class*='product-card'], div[class*='list__item']", "wait_selector": "article.product-thumbnail, div[class*='product-card']",
"category": "Grande Surface", "category": "Grande Surface",
"requires_proxy": False, "requires_proxy": False,
}, },
"carrefour.fr": { "carrefour.fr": {
"name": "Carrefour", "name": "Carrefour",
"search_url": "https://www.carrefour.fr/s?q={query}", "search_url": "https://www.carrefour.fr/s?q={query}",
"product_selector": "a[href*='/p/'], a[href*='/produit'], article a, div[class*='product'] a", "product_selector": "article.product-list-card-plp-grid-new",
"product_image_selector": "img, picture source", "product_image_selector": "img.product-card-image-new__content",
"wait_selector": None, "wait_selector": "article.product-list-card-plp-grid-new",
"category": "Grande Surface", "category": "Grande Surface",
"requires_proxy": False, "requires_proxy": False,
}, },
+6 -6
View File
@@ -14,9 +14,9 @@ class AuchanParser(BaseParser):
# Auchan products - Ultra Robust Strategy # Auchan products - Ultra Robust Strategy
# The DOM is flat and dynamic. We rely on finding product links first. # The DOM is flat and dynamic. We rely on finding product links first.
# Links usually contain '/p-' in the href. # Links usually contain '/p-' or '/pr-' in the href.
links = soup.select("a[href*='/p-']") links = soup.select("a[href*='/p-'], a[href*='/pr-']")
seen_urls = set() seen_urls = set()
for link in links: for link in links:
@@ -26,7 +26,7 @@ class AuchanParser(BaseParser):
continue continue
# Filter out non-product links if any (e.g. facets) # Filter out non-product links if any (e.g. facets)
if '/p-' not in href: if '/p-' not in href and '/pr-' not in href:
continue continue
seen_urls.add(href) seen_urls.add(href)
@@ -50,7 +50,7 @@ class AuchanParser(BaseParser):
# Title # Title
title = link.get('title') title = link.get('title')
if not title: if not title:
title_el = container.select_one("h3, div[class*='title'], span[class*='title']") title_el = container.select_one("p.product-thumbnail__description, h3, div[class*='title'], span[class*='title']")
if title_el: if title_el:
title = title_el.get_text(strip=True) title = title_el.get_text(strip=True)
if not title: if not title:
@@ -61,7 +61,7 @@ class AuchanParser(BaseParser):
# Image # Image
img_url = None img_url = None
img_el = container.select_one("img") img_el = container.select_one(".product-thumbnail__picture img, img")
if img_el: if img_el:
img_url = self._get_image_src(img_el) img_url = self._get_image_src(img_el)
if img_url: if img_url:
@@ -69,7 +69,7 @@ class AuchanParser(BaseParser):
# Price # Price
price = None price = None
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price") price_el = container.select_one("div.product-price, div[class*='price'], span[class*='price'], .product-price")
if price_el: if price_el:
price = self.parse_price_text(price_el.get_text()) price = self.parse_price_text(price_el.get_text())
+8 -6
View File
@@ -13,15 +13,17 @@ class CarrefourParser(BaseParser):
results = [] results = []
# Carrefour products # Carrefour products
# New container: div containing both image and title link # New container: article.large-horizontal or div.product-list-card-plp-grid-new
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']") cards = soup.select("article.large-horizontal, div.product-list-card-plp-grid-new, article, div[class*='product-card']")
# If no cards, try finding by link # If no cards, try finding by link
if not cards: if not cards:
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']") links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
cards = [] cards = []
for link in links: for link in links:
parent = link.find_parent("div", class_=lambda x: x and "product" in x) parent = link.find_parent("article")
if not parent:
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
if parent: if parent:
cards.append(parent) cards.append(parent)
else: else:
@@ -36,14 +38,14 @@ class CarrefourParser(BaseParser):
href = link_el.get('href') href = link_el.get('href')
url = self.make_absolute_url(href) url = self.make_absolute_url(href)
title_el = card.select_one("h3, h2, [class*='title']") title_el = card.select_one("h3.product-card-title__text, h3, h2, [class*='title']")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title: if not title:
continue continue
img_url = None img_url = None
img_el = card.select_one("img") img_el = card.select_one("img.product-card-image-new__content, img")
if img_el: if img_el:
img_url = self._get_image_src(img_el) img_url = self._get_image_src(img_el)
if img_url: if img_url:
@@ -51,7 +53,7 @@ class CarrefourParser(BaseParser):
price = None price = None
# Price is often text node near h3 or in a specific price element # Price is often text node near h3 or in a specific price element
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']") price_el = card.select_one("div.product-price__amount--main, [class*='price'], .product-card-price, span[class*='amount']")
if price_el: if price_el:
price = self.parse_price_text(price_el.get_text()) price = self.parse_price_text(price_el.get_text())
+12 -24
View File
@@ -13,51 +13,39 @@ class LIncroyableParser(BaseParser):
results = [] results = []
# L'Incroyable products # L'Incroyable products
# Config: a.product-link # Config: div.tailleBlocProdNew
# Try finding cards first cards = soup.select("div.tailleBlocProdNew")
cards = soup.select("div.product-card, div.product-miniature, article")
# If no cards found, try finding by link and getting parent
if not cards:
links = soup.select("a[href*='/p/'], a[href*='/produit/']")
cards = []
for link in links:
# Try to find a container div
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x))
if parent:
cards.append(parent)
else:
cards.append(link.parent) # Fallback to immediate parent
for card in cards: for card in cards:
try: try:
link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link") # Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links)
# We look for the main link to the product
link_el = card.select_one("a[href*='/p']")
if not link_el: if not link_el:
# If card is the link itself continue
if card.name == 'a' and card.get('href'):
link_el = card
else:
continue
href = link_el.get('href') href = link_el.get('href')
url = self.make_absolute_url(href) url = self.make_absolute_url(href)
title_el = card.select_one(".product-title, h3, h2, [class*='title']") # Title
title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title: if not title:
continue continue
# Image
img_url = None img_url = None
img_el = card.select_one("img") img_el = card.select_one("img.imgCoup2coeur, img")
if img_el: if img_el:
img_url = self._get_image_src(img_el) img_url = self._get_image_src(img_el)
if img_url: if img_url:
img_url = self.make_absolute_url(img_url) img_url = self.make_absolute_url(img_url)
# Price
price = None price = None
price_el = card.select_one(".price, .product-price, [class*='price']") price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew")
if price_el: if price_el:
price = self.parse_price_text(price_el.get_text()) price = self.parse_price_text(price_el.get_text())
+27 -43
View File
@@ -13,62 +13,46 @@ class StokomaniParser(BaseParser):
soup = BeautifulSoup(html, "html.parser") soup = BeautifulSoup(html, "html.parser")
results = [] results = []
# Find product title links # Stokomani products
links = soup.select("a.reversed-link.block") # Config: div.product-card
seen_urls = set() cards = soup.select("div.product-card")
for link in links: for card in cards:
try: try:
href = link.get('href') # Link
if not href or href in seen_urls: link_el = card.select_one("h3.product-card__title a, a[href*='/products/']")
if not link_el:
continue continue
if '/products/' not in href: href = link_el.get('href')
continue
seen_urls.add(href)
url = self.make_absolute_url(href) url = self.make_absolute_url(href)
title = link.get_text(strip=True) # Title
title_el = card.select_one("span.reversed-link__text, h3.product-card__title")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title: if not title:
continue continue
# Image: Look for the preceding <a> with aria-label # Image
img_url = None img_url = None
prev_a = link.find_previous_sibling("a", attrs={"aria-label": True}) img_el = card.select_one("div.media-wrapper img, img")
if prev_a and prev_a.get('href') == href: if img_el:
# Try specific selector first img_url = self._get_image_src(img_el)
img_el = prev_a.select_one("motion-element img, img") if img_url:
if img_el: img_url = self.make_absolute_url(img_url)
# Check for lazy loading attributes explicitly
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
if img_url and " " in img_url:
# Handle srcset: take the first URL
img_url = img_url.split(" ")[0]
if not img_url: # Price
img_url = self.extract_image_url(prev_a)
if img_url:
img_url = self.make_absolute_url(img_url)
# Price: Look for text node after the link
price = None price = None
next_sibling = link.next_sibling price_el = card.select_one("span.f-price-item--regular, .f-price-item, [class*='price']")
while next_sibling: if price_el:
if isinstance(next_sibling, NavigableString): price = self.parse_price_text(price_el.get_text())
price_text = next_sibling.strip()
if "€" in price_text: # Fallback price search in text
price = self.parse_price_text(price_text) if not price:
if price: text = card.get_text(" ", strip=True)
break price = self.parse_price_text(text)
elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])):
# Try finding price in next div if it's a price container
price = self.parse_price_text(next_sibling.get_text())
if price:
break
next_sibling = next_sibling.next_sibling
results.append(ProductResult( results.append(ProductResult(
title=title, title=title,
+47 -12
View File
@@ -108,20 +108,44 @@ class NewSearchService:
return result return result
# Use AI to analyze # Use AI to analyze
from app.services.ai_service import AIService try:
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text) from app.services.ai_service import AIService
# Check if AI service is available/configured before calling?
# For now, just try/except the call
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
if ai_result: if ai_result:
extraction, _ = ai_result extraction, _ = ai_result
result.price = extraction.price result.price = extraction.price
result.currency = extraction.currency or "EUR" result.currency = extraction.currency or "EUR"
result.in_stock = extraction.in_stock result.in_stock = extraction.in_stock
# Update image URL to point to our local screenshot # Update image URL to point to our local screenshot
# The frontend expects /screenshots/filename import os
import os filename = os.path.basename(screenshot_path)
filename = os.path.basename(screenshot_path) result.image_url = f"/screenshots/{filename}"
result.image_url = f"/screenshots/{filename}" else:
raise Exception("AI returned no result")
except Exception as e:
logger.warning(f"AI Analysis failed for {result.url}: {e}")
# Fallback: Try to extract price from page_text if Browserless found it
if page_text and "PRIX DÉTECTÉ:" in page_text:
try:
import re
price_match = re.search(r"PRIX DÉTECTÉ:\s*([\d\.]+)", page_text)
if price_match:
price_val = float(price_match.group(1))
result.price = price_val
logger.info(f"💰 Fallback: Extracted price {price_val} from text for {result.url}")
except Exception as parse_e:
logger.error(f"Error parsing fallback price: {parse_e}")
# Still use the screenshot if we have it
if screenshot_path:
import os
filename = os.path.basename(screenshot_path)
result.image_url = f"/screenshots/{filename}"
except Exception as e: except Exception as e:
logger.error(f"Error scraping item {result.url}: {e}") logger.error(f"Error scraping item {result.url}: {e}")
@@ -191,6 +215,17 @@ class NewSearchService:
# break # break
href = link.get("href") href = link.get("href")
# Special handling for sites where selector targets a container (Carrefour, Stokomani)
if not href and config.get("name") in ["Carrefour", "Stokomani"]:
# Try to find the main product link inside the container
# For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one
child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a")
if child_link:
href = child_link.get("href")
# Update link to point to the anchor for title/image extraction
link = child_link
if not href: if not href:
continue continue
+1 -1
View File
@@ -5,7 +5,7 @@ echo "Setting up screenshots directory..."
mkdir -p screenshots mkdir -p screenshots
echo "Running database migrations..." echo "Running database migrations..."
alembic upgrade heads # alembic upgrade heads
echo "Starting application..." echo "Starting application..."
exec uvicorn app.main:app --host 0.0.0.0 --port 8555 exec uvicorn app.main:app --host 0.0.0.0 --port 8555
+27 -22
View File
@@ -6,7 +6,7 @@ import sys
from pathlib import Path from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent)) sys.path.insert(0, str(Path(__file__).parent))
from app.services.improved_search_service import ImprovedSearchService from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS from app.core.search_config import SITE_CONFIGS
async def dump_search_html(site_key: str, query: str = "chaise"): async def dump_search_html(site_key: str, query: str = "chaise"):
@@ -19,31 +19,37 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
print(f"\n🔍 Dumping HTML for: {config['name']}") print(f"\n🔍 Dumping HTML for: {config['name']}")
# Ensure browser is initialized # Ensure browser is initialized
await ImprovedSearchService.initialize() await browserless_service.initialize()
# Create context manually to get HTML
context = await ImprovedSearchService._create_context(ImprovedSearchService._browser)
page = await context.new_page()
try: try:
search_url = config["search_url"].format(query=query) search_url = config["search_url"].format(query=query)
print(f" URL: {search_url}") print(f" URL: {search_url}")
await page.goto(search_url, wait_until="networkidle", timeout=30000) # Override wait_selector for La Foir'Fouille
await ImprovedSearchService._handle_popups(page) wait_selector = config.get("wait_selector")
await page.wait_for_timeout(3000) if site_key == "lafoirfouille.fr":
wait_selector = ".sf-grid-vignet"
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
html = await page.content() html_content, screenshot_path = await browserless_service.get_page_content(
search_url,
wait_selector=wait_selector,
use_proxy=config.get("requires_proxy", False)
)
if not html_content:
print(" ❌ No HTML content returned")
return
filename = f"dump_{site_key.replace('.', '_')}.html" filename = f"dump_{site_key.replace('.', '_')}.html"
with open(filename, "w", encoding="utf-8") as f: with open(filename, "w", encoding="utf-8") as f:
f.write(html) f.write(html_content)
print(f" ✅ Saved to: {filename} ({len(html)} bytes)") print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
# Quick analysis # Quick analysis
from bs4 import BeautifulSoup from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser") soup = BeautifulSoup(html_content, "html.parser")
# Try current selector # Try current selector
current_selector = config.get("product_selector") current_selector = config.get("product_selector")
@@ -56,18 +62,17 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
img_matches = soup.select(img_selector) img_matches = soup.select(img_selector)
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}") print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
except Exception as e:
print(f" ❌ Error during dump: {e}")
finally: finally:
await context.close() # We don't close the browser here to allow reuse if needed,
# but main() will shut it down.
pass
async def main(): async def main():
sites = [ sites = [
"e-leclerc.com", "stokomani.fr"
"auchan.fr",
"carrefour.fr",
"stokomani.fr",
"centrakor.com",
"cdiscount.com",
"lincroyable.fr"
] ]
for site_key in sites: for site_key in sites:
@@ -77,7 +82,7 @@ async def main():
print(f"❌ Error: {e}") print(f"❌ Error: {e}")
await asyncio.sleep(1) await asyncio.sleep(1)
await ImprovedSearchService.shutdown() await browserless_service.shutdown()
if __name__ == "__main__": if __name__ == "__main__":
asyncio.run(main()) asyncio.run(main())
+33
View File
@@ -0,0 +1,33 @@
from bs4 import BeautifulSoup
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
articles = soup.select("article.product-list-card-plp-grid-new")
print(f"Found {len(articles)} articles")
if articles:
first = articles[0]
print("\n--- First Article Structure ---")
print(first.prettify()[:1000]) # Print first 1000 chars
# Check for link
link = first.select_one("a.product-card-click-wrapper")
if link:
print(f"\nLink found: {link.get('href')}")
print(f"Link classes: {link.get('class')}")
# Check for image INSIDE link
img = link.select_one("img.product-card-image-new__content")
if img:
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
else:
print(f"\n❌ Image NOT found inside link")
# Check if image is elsewhere in article
img_article = first.select_one("img.product-card-image-new__content")
if img_article:
print(f" But image exists in article: {img_article.get('src')}")
else:
print("\nNo link found with selector a.product-card-click-wrapper")
+53
View File
@@ -0,0 +1,53 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
# Try to find product containers
print("Searching for product containers...")
potential_selectors = [
"div.product-miniature",
"article",
"div[class*='product']",
"div.product-card",
"div.item"
]
for selector in potential_selectors:
matches = soup.select(selector)
print(f"Selector '{selector}' matches: {len(matches)}")
if len(matches) > 0 and len(matches) < 5:
# If few matches, print classes to see if it's a wrapper
print(f" Classes: {matches[0].get('class')}")
# Print structure of first potential product
products = soup.select("div.product-miniature")
if not products:
products = soup.select("div[class*='product-item']")
if products:
first = products[0]
print("\n--- First Product Structure ---")
print(first.prettify()[:1000])
link = first.find("a")
if link:
print(f"\nLink found: {link.get('href')}")
img = first.find("img")
if img:
print(f"\nImage found: {img.get('src')}")
else:
print("\nNo obvious products found. Dumping generic structure...")
# Find any div with many children
divs = soup.find_all("div")
for div in divs:
if len(div.find_all("div", recursive=False)) > 10:
print(f"Found container with many children: {div.get('class')}")
# Print first child
child = div.find("div")
if child:
print(child.prettify()[:500])
break
+25
View File
@@ -0,0 +1,25 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
images = soup.select("img")
print(f"Found {len(images)} images")
for i, img in enumerate(images[:10]):
print(f"\n--- Image {i+1} ---")
print(f"Src: {img.get('src')}")
print(f"Classes: {img.get('class')}")
parent = img.parent
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
grandparent = parent.parent
if grandparent:
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
greatgrandparent = grandparent.parent
if greatgrandparent:
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
+45
View File
@@ -0,0 +1,45 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.core.search_config import SITE_CONFIGS
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
async def test_specific_sites():
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
query = "chaise"
print(f"Testing {target_sites} with query '{query}'...")
await browserless_service.initialize()
for site_key in target_sites:
if site_key not in SITE_CONFIGS:
print(f"Skipping {site_key} (not in config)")
continue
print(f"\n--- Testing {site_key} ---")
try:
results = await new_search_service.search_site(site_key, query)
print(f"Found {len(results)} results")
for r in results[:3]:
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
except Exception as e:
print(f"Error testing {site_key}: {e}")
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(test_specific_sites())