Merge pull request #172 from R0m1k3/antigravity

feat: add `search_config` module centralizing site definitions, proxy…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 14:16:02 +01:00
commit 45538afee2
3 files changed
+138 -40

No files matched your search

+10
View File
@@ -66,6 +66,7 @@ SITE_CONFIGS = {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
"product_selector": "a.link",
"product_image_selector": "img.tile-image",
"wait_selector": ".product-tile",
"category": "Discount",
"requires_proxy": False,
@@ -74,6 +75,7 @@ SITE_CONFIGS = {
"name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "a.reversed-link.block",
"product_image_selector": "img.object-cover",
"wait_selector": "a.reversed-link.block",
"category": "Discount",
"requires_proxy": False,
@@ -82,6 +84,7 @@ SITE_CONFIGS = {
"name": "Action",
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
"product_selector": "a.group[href^='/fr-fr/p/']",
"product_image_selector": "img[loading='lazy']",
"wait_selector": "a.group[href^='/fr-fr/p/']",
"category": "Discount",
"requires_proxy": False,
@@ -90,6 +93,7 @@ SITE_CONFIGS = {
"name": "Cdiscount",
"search_url": "https://www.cdiscount.com/search/10/{query}.html",
"product_selector": "a.prdtBImg[href*='/f-'][href*='.html']",
"product_image_selector": "img.prdtBImg",
"wait_selector": ".prdtBILDetails, .product-list",
"category": "E-commerce",
"requires_proxy": False,
@@ -99,6 +103,7 @@ SITE_CONFIGS = {
"name": "Darty",
"search_url": "https://www.darty.com/nav/recherche?text={query}",
"product_selector": "a[href*='/nav/achat/'][href*='.html']",
"product_image_selector": "img.product_img",
"wait_selector": ".product-card, .product-list",
"category": "Électronique",
"requires_proxy": False,
@@ -107,6 +112,7 @@ SITE_CONFIGS = {
"name": "Boulanger",
"search_url": "https://www.boulanger.com/resultats?tr={query}",
"product_selector": "a[href*='/ref/'][href*='_']",
"product_image_selector": "img.product-image",
"wait_selector": ".product-list, .product-card",
"category": "Électronique",
"requires_proxy": False,
@@ -115,6 +121,7 @@ SITE_CONFIGS = {
"name": "Fnac",
"search_url": "https://www.fnac.com/SearchResult/ResultList.aspx?Search={query}",
"product_selector": "a.Article-title[href*='/a']",
"product_image_selector": "img.Article-itemVisual",
"wait_selector": ".Article-item, .Article-list",
"category": "Culture & Tech",
"requires_proxy": False,
@@ -124,6 +131,7 @@ SITE_CONFIGS = {
"name": "B&M",
"search_url": "https://bmstores.fr/module/ambjolisearch/jolisearch?s={query}",
"product_selector": "a.thumbnail.product-thumbnail",
"product_image_selector": "img.product-thumbnail",
"wait_selector": ".products, .product-miniature",
"category": "Discount",
"requires_proxy": False,
@@ -132,6 +140,7 @@ SITE_CONFIGS = {
"name": "Centrakor",
"search_url": "https://www.centrakor.com/search/{query}",
"product_selector": "a.link.link--block",
"product_image_selector": "img.product-card__image",
"wait_selector": ".product-list, .product-item",
"category": "Discount",
"requires_proxy": False,
@@ -140,6 +149,7 @@ SITE_CONFIGS = {
"name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
"product_selector": "a[href^='/p']",
"product_image_selector": "img.product-image",
"wait_selector": ".product-item, .product-list",
"category": "Discount",
"requires_proxy": False,
+91 -29
View File
@@ -211,41 +211,103 @@ class BrowserlessService:
@staticmethod
async def _extract_amazon_price(page: Page) -> str:
"""Extract price from Amazon product page with strict validation."""
price_selectors = [
".a-price .a-offscreen", # Most reliable
"#corePrice_desktop .a-price .a-offscreen",
"#corePriceDisplay_desktop_feature_div .a-price .a-offscreen",
".a-price[data-a-color='price'] .a-offscreen",
"#priceblock_ourprice",
"#priceblock_dealprice",
"span.a-price-whole",
"""
Extract price from Amazon product page with strict validation.
Prioritize main price block to avoid false positives from recommendations.
"""
# STRATEGY: Target main price block FIRST to avoid picking up
# prices from "Other sellers", "Recommendations", etc.
# Priority 1: Main price display area (most specific)
main_price_selectors = [
"#corePriceDisplay_desktop_feature_div .a-price .a-offscreen", # Desktop main price
"#corePrice_desktop .a-price .a-offscreen", # Desktop main price (variant)
"#corePrice_feature_div .a-price .a-offscreen", # Mobile main price
"#priceblock_ourprice", # Legacy our price
"#priceblock_dealprice", # Legacy deal price
"#priceblock_saleprice", # Legacy sale price
]
for selector in price_selectors:
for selector in main_price_selectors:
try:
element = page.locator(selector).first
price_text = await element.inner_text(timeout=2000)
if price_text and price_text.strip():
# Validate format (Amazon uses "XX,YY €" or "XX,YY")
if re.search(r'\d+[.,]\d{2}', price_text):
logger.info(f"💰 Amazon price via {selector}: {price_text}")
return price_text.strip()
except Exception:
elements = page.locator(selector)
count = await elements.count()
for i in range(count):
element = elements.nth(i)
# Check if element is visible
if not await element.is_visible(timeout=1000):
continue
# Check if strikethrough (old price)
try:
text_decoration = await element.evaluate("el => window.getComputedStyle(el).textDecoration")
if "line-through" in text_decoration:
continue
except Exception:
pass
price_text = await element.inner_text(timeout=1000)
if price_text and price_text.strip():
# Validate format (Amazon uses "XX,YY €" or "XX,YY")
if re.search(r'\d+[.,]\d{2}', price_text):
# Extract numeric value for validation
numeric_match = re.search(r'(\d+)[.,](\d{2})', price_text)
if numeric_match:
price_val = float(f"{numeric_match.group(1)}.{numeric_match.group(2)}")
# Reasonable range: 0.01€ to 10,000€ (avoid crazy values)
if 0.01 <= price_val <= 10000:
logger.info(f"💰 Amazon main price via {selector}: {price_text} ({price_val}€)")
return price_text.strip()
except Exception as e:
logger.debug(f"Selector {selector} failed: {e}")
continue
# Try combination: whole + fraction
# Priority 2: Generic .a-price but ONLY in buybox area
try:
whole = await page.locator("span.a-price-whole").first.inner_text()
fraction = await page.locator("span.a-price-fraction").first.inner_text()
if whole and fraction:
price_text = f"{whole}{fraction}"
logger.info(f"💰 Amazon price from whole+fraction: {price_text}")
return price_text
# Try to find buybox first
buybox = page.locator("#buybox, #buybox_feature_div, #desktop_buybox")
if await buybox.count() > 0:
price_elem = buybox.locator(".a-price .a-offscreen").first
if await price_elem.is_visible(timeout=1000):
price_text = await price_elem.inner_text()
if price_text and re.search(r'\d+[.,]\d{2}', price_text):
numeric_match = re.search(r'(\d+)[.,](\d{2})', price_text)
if numeric_match:
price_val = float(f"{numeric_match.group(1)}.{numeric_match.group(2)}")
if 0.01 <= price_val <= 10000:
logger.info(f"💰 Amazon buybox price: {price_text} ({price_val}€)")
return price_text.strip()
except Exception:
pass
logger.warning("⚠️ Could not extract Amazon price")
# Priority 3: Whole + Fraction (common pattern)
try:
# Look specifically in price display area
price_container = page.locator("#corePriceDisplay_desktop_feature_div, #corePrice_desktop, #price")
whole_elem = price_container.locator("span.a-price-whole").first
fraction_elem = price_container.locator("span.a-price-fraction").first
whole = await whole_elem.inner_text(timeout=1000)
fraction = await fraction_elem.inner_text(timeout=1000)
if whole and fraction:
# Clean up (remove trailing comma/period from whole)
whole = whole.rstrip('.,')
price_text = f"{whole},{fraction}"
# Validate
numeric_match = re.search(r'(\d+)[.,](\d{2})', price_text)
if numeric_match:
price_val = float(f"{numeric_match.group(1)}.{numeric_match.group(2)}")
if 0.01 <= price_val <= 10000:
logger.info(f"💰 Amazon whole+fraction: {price_text} ({price_val}€)")
return price_text
except Exception:
pass
logger.warning("⚠️ Could not extract Amazon price with any method")
return ""
@staticmethod
+37 -11
View File
@@ -240,27 +240,53 @@ class NewSearchService:
# logger.debug(f"Skipping result '{title}' - does not contain all query words: {query_words}")
continue
# Extract Image URL (New)
# Extract Image URL (Enhanced)
image_url = None
if "product_image_selector" in config:
# Try to find image relative to the link or its container
# This is tricky because 'link' is just the <a> tag.
# We might need to look up to a container.
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img_el = container.select_one(config["product_image_selector"])
if img_el:
image_url = img_el.get("src") or img_el.get("data-src")
# Search in the link itself first
img_el = link.select_one(config["product_image_selector"])
# If not found in link, search in parent container
if not img_el:
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img_el = container.select_one(config["product_image_selector"])
if img_el:
# Try multiple image attributes in order of priority
image_url = (
img_el.get("src") or
img_el.get("data-src") or
img_el.get("data-lazy-src") or
img_el.get("data-original")
)
# Handle srcset (use first URL)
if not image_url and img_el.get("srcset"):
srcset = img_el.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Fallback image extraction
# Fallback: Find any img in the link
if not image_url:
img = link.find("img")
if img:
image_url = img.get("src") or img.get("data-src")
image_url = (
img.get("src") or
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original")
)
# Handle srcset
if not image_url and img.get("srcset"):
srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Make absolute URL
if image_url and not image_url.startswith("http"):
image_url = urljoin(base_url, image_url)
# Create result
results.append(SearchResult(
url=full_url,