feat: implement new search service with multiple site parsers, search configuration, and supporting inspection/verification scripts.

This commit is contained in:
Michael committed 2025-11-30 22:04:14 +01:00
1 parent 95e38287cb
commit 64067347d1
13 files changed
+300 -129

No files matched your search

+1
View File
@@ -65,6 +65,7 @@ COPY alembic.ini ./
COPY docker-entrypoint.sh ./
# Make entrypoint executable
RUN sed -i 's/\r$//' docker-entrypoint.sh
RUN chmod +x docker-entrypoint.sh
# Copy built frontend static files
+12 -12
View File
@@ -65,9 +65,9 @@ SITE_CONFIGS = {
"stokomani.fr": {
"name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "a.reversed-link.block, a[href*='/products/']",
"product_image_selector": "img[loading='lazy'], img[class*='object'], img[sizes]",
"wait_selector": "a.reversed-link.block, .product-card",
"product_selector": "div.product-card",
"product_image_selector": "div.media-wrapper img, img[class*='product-card__image']",
"wait_selector": "div.product-card",
"category": "Discount",
"requires_proxy": False,
},
@@ -130,9 +130,9 @@ SITE_CONFIGS = {
"lincroyable.fr": {
"name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
"product_selector": "div.product-card, div.product-miniature, article",
"product_image_selector": "img.product-image, img[class*='product'], picture img",
"wait_selector": "div.product-card, div.product-miniature, article",
"product_selector": "div.tailleBlocProdNew",
"product_image_selector": "img.imgCoup2coeur, img[class*='product']",
"wait_selector": "div.tailleBlocProdNew",
"category": "Discount",
"requires_proxy": False,
},
@@ -157,18 +157,18 @@ SITE_CONFIGS = {
"auchan.fr": {
"name": "Auchan",
"search_url": "https://www.auchan.fr/recherche?text={query}",
"product_selector": "article, div[class*='product-card'], div[class*='list__item']",
"product_image_selector": "img[class*='product'], img[src*='auchan']",
"wait_selector": "article, div[class*='product-card'], div[class*='list__item']",
"product_selector": "article.product-thumbnail a.product-thumbnail__details-wrapper, div[class*='product-card'] a",
"product_image_selector": ".product-thumbnail__picture img, img[class*='product']",
"wait_selector": "article.product-thumbnail, div[class*='product-card']",
"category": "Grande Surface",
"requires_proxy": False,
},
"carrefour.fr": {
"name": "Carrefour",
"search_url": "https://www.carrefour.fr/s?q={query}",
"product_selector": "a[href*='/p/'], a[href*='/produit'], article a, div[class*='product'] a",
"product_image_selector": "img, picture source",
"wait_selector": None,
"product_selector": "article.product-list-card-plp-grid-new",
"product_image_selector": "img.product-card-image-new__content",
"wait_selector": "article.product-list-card-plp-grid-new",
"category": "Grande Surface",
"requires_proxy": False,
},
+6 -6
View File
@@ -14,9 +14,9 @@ class AuchanParser(BaseParser):
# Auchan products - Ultra Robust Strategy
# The DOM is flat and dynamic. We rely on finding product links first.
# Links usually contain '/p-' in the href.
# Links usually contain '/p-' or '/pr-' in the href.
links = soup.select("a[href*='/p-']")
links = soup.select("a[href*='/p-'], a[href*='/pr-']")
seen_urls = set()
for link in links:
@@ -26,7 +26,7 @@ class AuchanParser(BaseParser):
continue
# Filter out non-product links if any (e.g. facets)
if '/p-' not in href:
if '/p-' not in href and '/pr-' not in href:
continue
seen_urls.add(href)
@@ -50,7 +50,7 @@ class AuchanParser(BaseParser):
# Title
title = link.get('title')
if not title:
title_el = container.select_one("h3, div[class*='title'], span[class*='title']")
title_el = container.select_one("p.product-thumbnail__description, h3, div[class*='title'], span[class*='title']")
if title_el:
title = title_el.get_text(strip=True)
if not title:
@@ -61,7 +61,7 @@ class AuchanParser(BaseParser):
# Image
img_url = None
img_el = container.select_one("img")
img_el = container.select_one(".product-thumbnail__picture img, img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
@@ -69,7 +69,7 @@ class AuchanParser(BaseParser):
# Price
price = None
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price")
price_el = container.select_one("div.product-price, div[class*='price'], span[class*='price'], .product-price")
if price_el:
price = self.parse_price_text(price_el.get_text())
+8 -6
View File
@@ -13,15 +13,17 @@ class CarrefourParser(BaseParser):
results = []
# Carrefour products
# New container: div containing both image and title link
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']")
# New container: article.large-horizontal or div.product-list-card-plp-grid-new
cards = soup.select("article.large-horizontal, div.product-list-card-plp-grid-new, article, div[class*='product-card']")
# If no cards, try finding by link
if not cards:
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
cards = []
for link in links:
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
parent = link.find_parent("article")
if not parent:
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
if parent:
cards.append(parent)
else:
@@ -36,14 +38,14 @@ class CarrefourParser(BaseParser):
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = card.select_one("h3, h2, [class*='title']")
title_el = card.select_one("h3.product-card-title__text, h3, h2, [class*='title']")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
img_url = None
img_el = card.select_one("img")
img_el = card.select_one("img.product-card-image-new__content, img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
@@ -51,7 +53,7 @@ class CarrefourParser(BaseParser):
price = None
# Price is often text node near h3 or in a specific price element
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']")
price_el = card.select_one("div.product-price__amount--main, [class*='price'], .product-card-price, span[class*='amount']")
if price_el:
price = self.parse_price_text(price_el.get_text())
+12 -24
View File
@@ -13,51 +13,39 @@ class LIncroyableParser(BaseParser):
results = []
# L'Incroyable products
# Config: a.product-link
# Config: div.tailleBlocProdNew
# Try finding cards first
cards = soup.select("div.product-card, div.product-miniature, article")
cards = soup.select("div.tailleBlocProdNew")
# If no cards found, try finding by link and getting parent
if not cards:
links = soup.select("a[href*='/p/'], a[href*='/produit/']")
cards = []
for link in links:
# Try to find a container div
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x))
if parent:
cards.append(parent)
else:
cards.append(link.parent) # Fallback to immediate parent
for card in cards:
try:
link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link")
# Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links)
# We look for the main link to the product
link_el = card.select_one("a[href*='/p']")
if not link_el:
# If card is the link itself
if card.name == 'a' and card.get('href'):
link_el = card
else:
continue
continue
href = link_el.get('href')
url = self.make_absolute_url(href)
title_el = card.select_one(".product-title, h3, h2, [class*='title']")
# Title
title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
# Image
img_url = None
img_el = card.select_one("img")
img_el = card.select_one("img.imgCoup2coeur, img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
# Price
price = None
price_el = card.select_one(".price, .product-price, [class*='price']")
price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew")
if price_el:
price = self.parse_price_text(price_el.get_text())
+28 -44
View File
@@ -13,63 +13,47 @@ class StokomaniParser(BaseParser):
soup = BeautifulSoup(html, "html.parser")
results = []
# Find product title links
links = soup.select("a.reversed-link.block")
# Stokomani products
# Config: div.product-card
seen_urls = set()
cards = soup.select("div.product-card")
for link in links:
for card in cards:
try:
href = link.get('href')
if not href or href in seen_urls:
continue
if '/products/' not in href:
# Link
link_el = card.select_one("h3.product-card__title a, a[href*='/products/']")
if not link_el:
continue
seen_urls.add(href)
href = link_el.get('href')
url = self.make_absolute_url(href)
title = link.get_text(strip=True)
# Title
title_el = card.select_one("span.reversed-link__text, h3.product-card__title")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
# Image: Look for the preceding <a> with aria-label
# Image
img_url = None
prev_a = link.find_previous_sibling("a", attrs={"aria-label": True})
if prev_a and prev_a.get('href') == href:
# Try specific selector first
img_el = prev_a.select_one("motion-element img, img")
if img_el:
# Check for lazy loading attributes explicitly
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
if img_url and " " in img_url:
# Handle srcset: take the first URL
img_url = img_url.split(" ")[0]
if not img_url:
img_url = self.extract_image_url(prev_a)
img_el = card.select_one("div.media-wrapper img, img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
if img_url:
img_url = self.make_absolute_url(img_url)
# Price: Look for text node after the link
# Price
price = None
next_sibling = link.next_sibling
while next_sibling:
if isinstance(next_sibling, NavigableString):
price_text = next_sibling.strip()
if "€" in price_text:
price = self.parse_price_text(price_text)
if price:
break
elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])):
# Try finding price in next div if it's a price container
price = self.parse_price_text(next_sibling.get_text())
if price:
break
next_sibling = next_sibling.next_sibling
price_el = card.select_one("span.f-price-item--regular, .f-price-item, [class*='price']")
if price_el:
price = self.parse_price_text(price_el.get_text())
# Fallback price search in text
if not price:
text = card.get_text(" ", strip=True)
price = self.parse_price_text(text)
results.append(ProductResult(
title=title,
url=url,
+49 -14
View File
@@ -108,21 +108,45 @@ class NewSearchService:
return result
# Use AI to analyze
from app.services.ai_service import AIService
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
try:
from app.services.ai_service import AIService
# Check if AI service is available/configured before calling?
# For now, just try/except the call
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
if ai_result:
extraction, _ = ai_result
result.price = extraction.price
result.currency = extraction.currency or "EUR"
result.in_stock = extraction.in_stock
# Update image URL to point to our local screenshot
import os
filename = os.path.basename(screenshot_path)
result.image_url = f"/screenshots/{filename}"
else:
raise Exception("AI returned no result")
except Exception as e:
logger.warning(f"AI Analysis failed for {result.url}: {e}")
# Fallback: Try to extract price from page_text if Browserless found it
if page_text and "PRIX DÉTECTÉ:" in page_text:
try:
import re
price_match = re.search(r"PRIX DÉTECTÉ:\s*([\d\.]+)", page_text)
if price_match:
price_val = float(price_match.group(1))
result.price = price_val
logger.info(f"💰 Fallback: Extracted price {price_val} from text for {result.url}")
except Exception as parse_e:
logger.error(f"Error parsing fallback price: {parse_e}")
# Still use the screenshot if we have it
if screenshot_path:
import os
filename = os.path.basename(screenshot_path)
result.image_url = f"/screenshots/{filename}"
if ai_result:
extraction, _ = ai_result
result.price = extraction.price
result.currency = extraction.currency or "EUR"
result.in_stock = extraction.in_stock
# Update image URL to point to our local screenshot
# The frontend expects /screenshots/filename
import os
filename = os.path.basename(screenshot_path)
result.image_url = f"/screenshots/{filename}"
except Exception as e:
logger.error(f"Error scraping item {result.url}: {e}")
@@ -191,6 +215,17 @@ class NewSearchService:
# break
href = link.get("href")
# Special handling for sites where selector targets a container (Carrefour, Stokomani)
if not href and config.get("name") in ["Carrefour", "Stokomani"]:
# Try to find the main product link inside the container
# For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one
child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a")
if child_link:
href = child_link.get("href")
# Update link to point to the anchor for title/image extraction
link = child_link
if not href:
continue
+1 -1
View File
@@ -5,7 +5,7 @@ echo "Setting up screenshots directory..."
mkdir -p screenshots
echo "Running database migrations..."
alembic upgrade heads
# alembic upgrade heads
echo "Starting application..."
exec uvicorn app.main:app --host 0.0.0.0 --port 8555
+27 -22
View File
@@ -6,7 +6,7 @@ import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.improved_search_service import ImprovedSearchService
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
async def dump_search_html(site_key: str, query: str = "chaise"):
@@ -19,31 +19,37 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
print(f"\n🔍 Dumping HTML for: {config['name']}")
# Ensure browser is initialized
await ImprovedSearchService.initialize()
# Create context manually to get HTML
context = await ImprovedSearchService._create_context(ImprovedSearchService._browser)
page = await context.new_page()
await browserless_service.initialize()
try:
search_url = config["search_url"].format(query=query)
print(f" URL: {search_url}")
await page.goto(search_url, wait_until="networkidle", timeout=30000)
await ImprovedSearchService._handle_popups(page)
await page.wait_for_timeout(3000)
# Override wait_selector for La Foir'Fouille
wait_selector = config.get("wait_selector")
if site_key == "lafoirfouille.fr":
wait_selector = ".sf-grid-vignet"
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
html = await page.content()
html_content, screenshot_path = await browserless_service.get_page_content(
search_url,
wait_selector=wait_selector,
use_proxy=config.get("requires_proxy", False)
)
if not html_content:
print(" ❌ No HTML content returned")
return
filename = f"dump_{site_key.replace('.', '_')}.html"
with open(filename, "w", encoding="utf-8") as f:
f.write(html)
f.write(html_content)
print(f" ✅ Saved to: {filename} ({len(html)} bytes)")
print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
# Quick analysis
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
soup = BeautifulSoup(html_content, "html.parser")
# Try current selector
current_selector = config.get("product_selector")
@@ -55,19 +61,18 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
img_selector = config["product_image_selector"]
img_matches = soup.select(img_selector)
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
except Exception as e:
print(f" ❌ Error during dump: {e}")
finally:
await context.close()
# We don't close the browser here to allow reuse if needed,
# but main() will shut it down.
pass
async def main():
sites = [
"e-leclerc.com",
"auchan.fr",
"carrefour.fr",
"stokomani.fr",
"centrakor.com",
"cdiscount.com",
"lincroyable.fr"
"stokomani.fr"
]
for site_key in sites:
@@ -77,7 +82,7 @@ async def main():
print(f"❌ Error: {e}")
await asyncio.sleep(1)
await ImprovedSearchService.shutdown()
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+33
View File
@@ -0,0 +1,33 @@
from bs4 import BeautifulSoup
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
articles = soup.select("article.product-list-card-plp-grid-new")
print(f"Found {len(articles)} articles")
if articles:
first = articles[0]
print("\n--- First Article Structure ---")
print(first.prettify()[:1000]) # Print first 1000 chars
# Check for link
link = first.select_one("a.product-card-click-wrapper")
if link:
print(f"\nLink found: {link.get('href')}")
print(f"Link classes: {link.get('class')}")
# Check for image INSIDE link
img = link.select_one("img.product-card-image-new__content")
if img:
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
else:
print(f"\n❌ Image NOT found inside link")
# Check if image is elsewhere in article
img_article = first.select_one("img.product-card-image-new__content")
if img_article:
print(f" But image exists in article: {img_article.get('src')}")
else:
print("\nNo link found with selector a.product-card-click-wrapper")
+53
View File
@@ -0,0 +1,53 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
# Try to find product containers
print("Searching for product containers...")
potential_selectors = [
"div.product-miniature",
"article",
"div[class*='product']",
"div.product-card",
"div.item"
]
for selector in potential_selectors:
matches = soup.select(selector)
print(f"Selector '{selector}' matches: {len(matches)}")
if len(matches) > 0 and len(matches) < 5:
# If few matches, print classes to see if it's a wrapper
print(f" Classes: {matches[0].get('class')}")
# Print structure of first potential product
products = soup.select("div.product-miniature")
if not products:
products = soup.select("div[class*='product-item']")
if products:
first = products[0]
print("\n--- First Product Structure ---")
print(first.prettify()[:1000])
link = first.find("a")
if link:
print(f"\nLink found: {link.get('href')}")
img = first.find("img")
if img:
print(f"\nImage found: {img.get('src')}")
else:
print("\nNo obvious products found. Dumping generic structure...")
# Find any div with many children
divs = soup.find_all("div")
for div in divs:
if len(div.find_all("div", recursive=False)) > 10:
print(f"Found container with many children: {div.get('class')}")
# Print first child
child = div.find("div")
if child:
print(child.prettify()[:500])
break
+25
View File
@@ -0,0 +1,25 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
images = soup.select("img")
print(f"Found {len(images)} images")
for i, img in enumerate(images[:10]):
print(f"\n--- Image {i+1} ---")
print(f"Src: {img.get('src')}")
print(f"Classes: {img.get('class')}")
parent = img.parent
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
grandparent = parent.parent
if grandparent:
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
greatgrandparent = grandparent.parent
if greatgrandparent:
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
+45
View File
@@ -0,0 +1,45 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.core.search_config import SITE_CONFIGS
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
async def test_specific_sites():
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
query = "chaise"
print(f"Testing {target_sites} with query '{query}'...")
await browserless_service.initialize()
for site_key in target_sites:
if site_key not in SITE_CONFIGS:
print(f"Skipping {site_key} (not in config)")
continue
print(f"\n--- Testing {site_key} ---")
try:
results = await new_search_service.search_site(site_key, query)
print(f"Found {len(results)} results")
for r in results[:3]:
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
except Exception as e:
print(f"Error testing {site_key}: {e}")
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(test_specific_sites())