feat: Implement centralized search configuration with site definitions, proxy settings, and user agents, and add a debugging script.

This commit is contained in:
Michael committed 2025-11-30 15:58:27 +01:00
1 parent 7477d505e5
commit 79dc4bfafc
3 files changed
+199 -14

No files matched your search

+32 -4
View File
@@ -66,7 +66,7 @@ SITE_CONFIGS = {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
"product_selector": "a.link",
"product_image_selector": "img.tile-image",
"product_image_selector": "img.tile-image, img[class*='product'], picture img",
"wait_selector": ".product-tile",
"category": "Discount",
"requires_proxy": False,
@@ -75,7 +75,7 @@ SITE_CONFIGS = {
"name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "a.reversed-link.block",
"product_image_selector": "img.object-cover",
"product_image_selector": "img.object-cover, img[loading='lazy'], picture img",
"wait_selector": "a.reversed-link.block",
"category": "Discount",
"requires_proxy": False,
@@ -84,7 +84,7 @@ SITE_CONFIGS = {
"name": "Action",
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
"product_selector": "a.group[href^='/fr-fr/p/']",
"product_image_selector": "img[loading='lazy']",
"product_image_selector": "img[loading='lazy'], img[src*='product'], picture img",
"wait_selector": "a.group[href^='/fr-fr/p/']",
"category": "Discount",
"requires_proxy": False,
@@ -121,7 +121,7 @@ SITE_CONFIGS = {
"name": "Fnac",
"search_url": "https://www.fnac.com/SearchResult/ResultList.aspx?Search={query}",
"product_selector": "a.Article-title[href*='/a']",
"product_image_selector": "img.Article-itemVisual",
"product_image_selector": "img.Article-itemVisual, img[data-src], img[src*='product'], picture img",
"wait_selector": ".Article-item, .Article-list",
"category": "Culture & Tech",
"requires_proxy": False,
@@ -154,6 +154,34 @@ SITE_CONFIGS = {
"category": "Discount",
"requires_proxy": False,
},
# === GRANDES SURFACES ===
"e-leclerc.com": {
"name": "E.Leclerc",
"search_url": "https://www.e-leclerc.com/recherche?text={query}",
"product_selector": "a[href*='/produit'], a.product-card, article a",
"product_image_selector": "img[loading='lazy'], img[src*='scene7'], picture img",
"wait_selector": ".product-card, .search-results, article",
"category": "Grande Surface",
"requires_proxy": False,
},
"auchan.fr": {
"name": "Auchan",
"search_url": "https://www.auchan.fr/search?text={query}",
"product_selector": "a[href*='/p-'], a.product-card, a[data-testid*='product']",
"product_image_selector": "img.product-image, img[loading='lazy'], picture img",
"wait_selector": ".product-card, .product-item, [data-testid*='product']",
"category": "Grande Surface",
"requires_proxy": False,
},
"carrefour.fr": {
"name": "Carrefour",
"search_url": "https://www.carrefour.fr/s?q={query}",
"product_selector": "a[data-testid='product-card-link'], a[href*='/p/']",
"product_image_selector": "img[data-testid='product-card-image'], img[loading='lazy'], picture img",
"wait_selector": "[data-testid='product-card'], .product-card",
"category": "Grande Surface",
"requires_proxy": False,
},
}
# === COOKIE BANNERS ===
+47 -10
View File
@@ -240,17 +240,26 @@ class NewSearchService:
# logger.debug(f"Skipping result '{title}' - does not contain all query words: {query_words}")
continue
# Extract Image URL (Enhanced)
# Extract Image URL (Enhanced with multi-selector support)
image_url = None
if "product_image_selector" in config:
# Search in the link itself first
img_el = link.select_one(config["product_image_selector"])
# Split selectors by comma to support multiple fallback selectors
selectors = [s.strip() for s in config["product_image_selector"].split(",")]
# If not found in link, search in parent container
if not img_el:
img_el = None
# Try each selector in order
for selector in selectors:
# Search in the link itself first
img_el = link.select_one(selector)
if img_el:
break
# If not found in link, search in parent container
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img_el = container.select_one(config["product_image_selector"])
img_el = container.select_one(selector)
if img_el:
break
if img_el:
# Try multiple image attributes in order of priority
@@ -258,7 +267,8 @@ class NewSearchService:
img_el.get("src") or
img_el.get("data-src") or
img_el.get("data-lazy-src") or
img_el.get("data-original")
img_el.get("data-original") or
img_el.get("data-lazy")
)
# Handle srcset (use first URL)
@@ -274,17 +284,44 @@ class NewSearchService:
img.get("src") or
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original")
img.get("data-original") or
img.get("data-lazy")
)
# Handle srcset
if not image_url and img.get("srcset"):
srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Fallback: Search in parent container
if not image_url:
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img = container.find("img")
if img:
image_url = (
img.get("src") or
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original") or
img.get("data-lazy")
)
if not image_url and img.get("srcset"):
srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Make absolute URL
if image_url and not image_url.startswith("http"):
image_url = urljoin(base_url, image_url)
if image_url:
# Clean up data URIs or invalid URLs
if image_url.startswith("data:"):
logger.debug(f"Skipping data URI for: {title[:30]}")
image_url = None
elif not image_url.startswith("http"):
image_url = urljoin(base_url, image_url)
# Log if image not found
if not image_url:
logger.warning(f"No image found for: {title[:50]} | {site_key}")
# Create result
+120
View File
@@ -0,0 +1,120 @@
"""
Debug script to identify correct selectors for missing sites
"""
import asyncio
import sys
import os
from pathlib import Path
# Add app directory to path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.browserless_service import browserless_service
from bs4 import BeautifulSoup
async def analyze_site(name: str, url: str, wait_selector: str = None):
"""Analyze a search results page to identify selectors"""
print(f"\n{'='*80}")
print(f"Analyzing: {name}")
print(f"URL: {url}")
print(f"{'='*80}\n")
html, screenshot = await browserless_service.get_page_content(
url,
use_proxy=False,
wait_selector=wait_selector,
wait_timeout=10000
)
if not html:
print(f"❌ Failed to get content for {name}")
return
soup = BeautifulSoup(html, "html.parser")
# Save HTML for manual inspection
output_file = f"debug_{name.lower().replace(' ', '_')}.html"
with open(output_file, "w", encoding="utf-8") as f:
f.write(html)
print(f"💾 HTML saved to: {output_file}")
# Common product link patterns
product_patterns = [
"a[href*='/product']",
"a[href*='/p/']",
"a[href*='/produit']",
"a.product-card",
"a.product-link",
"[data-product-id]",
"article a",
"div[data-testid*='product'] a",
]
print("\n🔍 Searching for product links...")
for pattern in product_patterns:
links = soup.select(pattern)
if links and len(links) >= 3:
print(f"✅ Found {len(links)} matches for: {pattern}")
# Show first 3 examples
for i, link in enumerate(links[:3], 1):
href = link.get('href', 'NO_HREF')
text = link.get_text(strip=True)[:50]
print(f" {i}. {href[:60]} | {text}")
elif links:
print(f"⚠️ Found {len(links)} matches for: {pattern} (too few)")
# Common image patterns
image_patterns = [
"img[src*='product']",
"img.product-image",
"img.product-img",
"img[loading='lazy']",
"picture img",
"img[data-src]",
]
print("\n🖼️ Searching for product images...")
for pattern in image_patterns:
images = soup.select(pattern)
if images and len(images) >= 3:
print(f"✅ Found {len(images)} matches for: {pattern}")
for i, img in enumerate(images[:3], 1):
src = img.get('src') or img.get('data-src', 'NO_SRC')
alt = img.get('alt', 'NO_ALT')[:50]
print(f" {i}. {src[:60]} | {alt}")
elif images:
print(f"⚠️ Found {len(images)} matches for: {pattern} (too few)")
print(f"\n✅ Analysis complete for {name}\n")
async def main():
"""Test all missing sites"""
sites = [
{
"name": "E.Leclerc",
"url": "https://www.e-leclerc.com/recherche?text=chaise",
"wait": ".product-card, .search-results"
},
{
"name": "Auchan",
"url": "https://www.auchan.fr/search?text=chaise",
"wait": ".product-card, .product-item"
},
{
"name": "Carrefour",
"url": "https://www.carrefour.fr/s?q=chaise",
"wait": "[data-testid*='product'], .product"
},
]
for site in sites:
try:
await analyze_site(site["name"], site["url"], site.get("wait"))
except Exception as e:
print(f"❌ Error analyzing {site['name']}: {e}")
# Small delay between sites
await asyncio.sleep(2)
if __name__ == "__main__":
asyncio.run(main())