mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Implement centralized search configuration with site definitions, proxy settings, and user agents, and add a debugging script.
This commit is contained in:
1 parent
7477d505e5
commit
79dc4bfafc
3 files changed
+199
-14
No files matched your search
@@ -66,7 +66,7 @@ SITE_CONFIGS = {
|
||||
"name": "Gifi",
|
||||
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
|
||||
"product_selector": "a.link",
|
||||
"product_image_selector": "img.tile-image",
|
||||
"product_image_selector": "img.tile-image, img[class*='product'], picture img",
|
||||
"wait_selector": ".product-tile",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
@@ -75,7 +75,7 @@ SITE_CONFIGS = {
|
||||
"name": "Stokomani",
|
||||
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
|
||||
"product_selector": "a.reversed-link.block",
|
||||
"product_image_selector": "img.object-cover",
|
||||
"product_image_selector": "img.object-cover, img[loading='lazy'], picture img",
|
||||
"wait_selector": "a.reversed-link.block",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
@@ -84,7 +84,7 @@ SITE_CONFIGS = {
|
||||
"name": "Action",
|
||||
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
|
||||
"product_selector": "a.group[href^='/fr-fr/p/']",
|
||||
"product_image_selector": "img[loading='lazy']",
|
||||
"product_image_selector": "img[loading='lazy'], img[src*='product'], picture img",
|
||||
"wait_selector": "a.group[href^='/fr-fr/p/']",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
@@ -121,7 +121,7 @@ SITE_CONFIGS = {
|
||||
"name": "Fnac",
|
||||
"search_url": "https://www.fnac.com/SearchResult/ResultList.aspx?Search={query}",
|
||||
"product_selector": "a.Article-title[href*='/a']",
|
||||
"product_image_selector": "img.Article-itemVisual",
|
||||
"product_image_selector": "img.Article-itemVisual, img[data-src], img[src*='product'], picture img",
|
||||
"wait_selector": ".Article-item, .Article-list",
|
||||
"category": "Culture & Tech",
|
||||
"requires_proxy": False,
|
||||
@@ -154,6 +154,34 @@ SITE_CONFIGS = {
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
# === GRANDES SURFACES ===
|
||||
"e-leclerc.com": {
|
||||
"name": "E.Leclerc",
|
||||
"search_url": "https://www.e-leclerc.com/recherche?text={query}",
|
||||
"product_selector": "a[href*='/produit'], a.product-card, article a",
|
||||
"product_image_selector": "img[loading='lazy'], img[src*='scene7'], picture img",
|
||||
"wait_selector": ".product-card, .search-results, article",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
"auchan.fr": {
|
||||
"name": "Auchan",
|
||||
"search_url": "https://www.auchan.fr/search?text={query}",
|
||||
"product_selector": "a[href*='/p-'], a.product-card, a[data-testid*='product']",
|
||||
"product_image_selector": "img.product-image, img[loading='lazy'], picture img",
|
||||
"wait_selector": ".product-card, .product-item, [data-testid*='product']",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
"carrefour.fr": {
|
||||
"name": "Carrefour",
|
||||
"search_url": "https://www.carrefour.fr/s?q={query}",
|
||||
"product_selector": "a[data-testid='product-card-link'], a[href*='/p/']",
|
||||
"product_image_selector": "img[data-testid='product-card-image'], img[loading='lazy'], picture img",
|
||||
"wait_selector": "[data-testid='product-card'], .product-card",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
}
|
||||
|
||||
# === COOKIE BANNERS ===
|
||||
|
||||
@@ -240,17 +240,26 @@ class NewSearchService:
|
||||
# logger.debug(f"Skipping result '{title}' - does not contain all query words: {query_words}")
|
||||
continue
|
||||
|
||||
# Extract Image URL (Enhanced)
|
||||
# Extract Image URL (Enhanced with multi-selector support)
|
||||
image_url = None
|
||||
if "product_image_selector" in config:
|
||||
# Search in the link itself first
|
||||
img_el = link.select_one(config["product_image_selector"])
|
||||
# Split selectors by comma to support multiple fallback selectors
|
||||
selectors = [s.strip() for s in config["product_image_selector"].split(",")]
|
||||
|
||||
# If not found in link, search in parent container
|
||||
if not img_el:
|
||||
img_el = None
|
||||
# Try each selector in order
|
||||
for selector in selectors:
|
||||
# Search in the link itself first
|
||||
img_el = link.select_one(selector)
|
||||
if img_el:
|
||||
break
|
||||
|
||||
# If not found in link, search in parent container
|
||||
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
if container:
|
||||
img_el = container.select_one(config["product_image_selector"])
|
||||
img_el = container.select_one(selector)
|
||||
if img_el:
|
||||
break
|
||||
|
||||
if img_el:
|
||||
# Try multiple image attributes in order of priority
|
||||
@@ -258,7 +267,8 @@ class NewSearchService:
|
||||
img_el.get("src") or
|
||||
img_el.get("data-src") or
|
||||
img_el.get("data-lazy-src") or
|
||||
img_el.get("data-original")
|
||||
img_el.get("data-original") or
|
||||
img_el.get("data-lazy")
|
||||
)
|
||||
|
||||
# Handle srcset (use first URL)
|
||||
@@ -274,17 +284,44 @@ class NewSearchService:
|
||||
img.get("src") or
|
||||
img.get("data-src") or
|
||||
img.get("data-lazy-src") or
|
||||
img.get("data-original")
|
||||
img.get("data-original") or
|
||||
img.get("data-lazy")
|
||||
)
|
||||
|
||||
# Handle srcset
|
||||
if not image_url and img.get("srcset"):
|
||||
srcset = img.get("srcset")
|
||||
image_url = srcset.split(",")[0].split()[0]
|
||||
|
||||
# Fallback: Search in parent container
|
||||
if not image_url:
|
||||
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
if container:
|
||||
img = container.find("img")
|
||||
if img:
|
||||
image_url = (
|
||||
img.get("src") or
|
||||
img.get("data-src") or
|
||||
img.get("data-lazy-src") or
|
||||
img.get("data-original") or
|
||||
img.get("data-lazy")
|
||||
)
|
||||
if not image_url and img.get("srcset"):
|
||||
srcset = img.get("srcset")
|
||||
image_url = srcset.split(",")[0].split()[0]
|
||||
|
||||
# Make absolute URL
|
||||
if image_url and not image_url.startswith("http"):
|
||||
image_url = urljoin(base_url, image_url)
|
||||
if image_url:
|
||||
# Clean up data URIs or invalid URLs
|
||||
if image_url.startswith("data:"):
|
||||
logger.debug(f"Skipping data URI for: {title[:30]}")
|
||||
image_url = None
|
||||
elif not image_url.startswith("http"):
|
||||
image_url = urljoin(base_url, image_url)
|
||||
|
||||
# Log if image not found
|
||||
if not image_url:
|
||||
logger.warning(f"No image found for: {title[:50]} | {site_key}")
|
||||
|
||||
|
||||
# Create result
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""
|
||||
Debug script to identify correct selectors for missing sites
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Add app directory to path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.browserless_service import browserless_service
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def analyze_site(name: str, url: str, wait_selector: str = None):
|
||||
"""Analyze a search results page to identify selectors"""
|
||||
print(f"\n{'='*80}")
|
||||
print(f"Analyzing: {name}")
|
||||
print(f"URL: {url}")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
html, screenshot = await browserless_service.get_page_content(
|
||||
url,
|
||||
use_proxy=False,
|
||||
wait_selector=wait_selector,
|
||||
wait_timeout=10000
|
||||
)
|
||||
|
||||
if not html:
|
||||
print(f"❌ Failed to get content for {name}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
# Save HTML for manual inspection
|
||||
output_file = f"debug_{name.lower().replace(' ', '_')}.html"
|
||||
with open(output_file, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
print(f"💾 HTML saved to: {output_file}")
|
||||
|
||||
# Common product link patterns
|
||||
product_patterns = [
|
||||
"a[href*='/product']",
|
||||
"a[href*='/p/']",
|
||||
"a[href*='/produit']",
|
||||
"a.product-card",
|
||||
"a.product-link",
|
||||
"[data-product-id]",
|
||||
"article a",
|
||||
"div[data-testid*='product'] a",
|
||||
]
|
||||
|
||||
print("\n🔍 Searching for product links...")
|
||||
for pattern in product_patterns:
|
||||
links = soup.select(pattern)
|
||||
if links and len(links) >= 3:
|
||||
print(f"✅ Found {len(links)} matches for: {pattern}")
|
||||
# Show first 3 examples
|
||||
for i, link in enumerate(links[:3], 1):
|
||||
href = link.get('href', 'NO_HREF')
|
||||
text = link.get_text(strip=True)[:50]
|
||||
print(f" {i}. {href[:60]} | {text}")
|
||||
elif links:
|
||||
print(f"⚠️ Found {len(links)} matches for: {pattern} (too few)")
|
||||
|
||||
# Common image patterns
|
||||
image_patterns = [
|
||||
"img[src*='product']",
|
||||
"img.product-image",
|
||||
"img.product-img",
|
||||
"img[loading='lazy']",
|
||||
"picture img",
|
||||
"img[data-src]",
|
||||
]
|
||||
|
||||
print("\n🖼️ Searching for product images...")
|
||||
for pattern in image_patterns:
|
||||
images = soup.select(pattern)
|
||||
if images and len(images) >= 3:
|
||||
print(f"✅ Found {len(images)} matches for: {pattern}")
|
||||
for i, img in enumerate(images[:3], 1):
|
||||
src = img.get('src') or img.get('data-src', 'NO_SRC')
|
||||
alt = img.get('alt', 'NO_ALT')[:50]
|
||||
print(f" {i}. {src[:60]} | {alt}")
|
||||
elif images:
|
||||
print(f"⚠️ Found {len(images)} matches for: {pattern} (too few)")
|
||||
|
||||
print(f"\n✅ Analysis complete for {name}\n")
|
||||
|
||||
async def main():
|
||||
"""Test all missing sites"""
|
||||
sites = [
|
||||
{
|
||||
"name": "E.Leclerc",
|
||||
"url": "https://www.e-leclerc.com/recherche?text=chaise",
|
||||
"wait": ".product-card, .search-results"
|
||||
},
|
||||
{
|
||||
"name": "Auchan",
|
||||
"url": "https://www.auchan.fr/search?text=chaise",
|
||||
"wait": ".product-card, .product-item"
|
||||
},
|
||||
{
|
||||
"name": "Carrefour",
|
||||
"url": "https://www.carrefour.fr/s?q=chaise",
|
||||
"wait": "[data-testid*='product'], .product"
|
||||
},
|
||||
]
|
||||
|
||||
for site in sites:
|
||||
try:
|
||||
await analyze_site(site["name"], site["url"], site.get("wait"))
|
||||
except Exception as e:
|
||||
print(f"❌ Error analyzing {site['name']}: {e}")
|
||||
|
||||
# Small delay between sites
|
||||
await asyncio.sleep(2)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
Reference in new issue
Block a user