From f976ae11273087ff1e55c12539b71cc0a185c6fb Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sat, 29 Nov 2025 13:09:53 +0100 Subject: [PATCH] feat: Implement web scraping and search system with centralized configuration, browserless integration, and a verification script. --- app/core/search_config.py | 40 ++++++++++++++++----------- app/services/browserless_service.py | 3 --- app/services/search_service.py | 27 ++++++++++++++----- verify_scrapers.py | 42 +++++++++++++++++++++++++++++ 4 files changed, 87 insertions(+), 25 deletions(-) create mode 100644 verify_scrapers.py diff --git a/app/core/search_config.py b/app/core/search_config.py index 4b98be9..a0c9f8f 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -51,6 +51,9 @@ USER_AGENT_POOL = [ # Firefox Windows "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0", + # Recent Chrome + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36", ] def get_random_user_agent() -> str: @@ -62,16 +65,16 @@ SITE_CONFIGS = { "gifi.fr": { "name": "Gifi", "search_url": "https://www.gifi.fr/resultat-recherche?q={query}", - "product_selector": "a.link, .product-item a.link, .product-tile a.link", - "wait_selector": ".product-item, .product-tile, .products-grid", + "product_selector": "a.link", + "wait_selector": ".product-tile", "category": "Discount", "requires_proxy": False, }, "stokomani.fr": { "name": "Stokomani", "search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}", - "product_selector": ".product-card__title a", - "wait_selector": ".product-card__title a", + "product_selector": "a.reversed-link.block", + "wait_selector": "a.reversed-link.block", "category": "Discount", "requires_proxy": False, }, @@ -83,12 +86,20 @@ SITE_CONFIGS = { }, "lafoirfouille.fr": { "name": "La Foir'Fouille", - "search_url": "https://www.lafoirfouille.fr/catalogsearch/result/?q={query}", + "search_url": "https://www.lafoirfouille.fr/?q={query}", "product_selector": ".product-item a.product-item-link, .product-item-info a", "wait_selector": ".products-grid, .product-items", "category": "Discount", "requires_proxy": False, }, + "auchan.fr": { + "name": "Auchan", + "search_url": "https://www.auchan.fr/recherche?text={query}", + "product_selector": "a[href*='/p-']", + "wait_selector": "a[href*='/p-']", + "category": "Grande Surface", + "requires_proxy": False, + }, "centrakor.com": { "name": "Centrakor", "search_url": "https://www.centrakor.com/catalogsearch/result/?q={query}", @@ -96,7 +107,7 @@ SITE_CONFIGS = { "wait_selector": ".products-grid, .product-items", "category": "Déco & Maison", "requires_proxy": False, - "pre_search_selector": "text=\"C'est parti !\"", + "pre_search_selector": "button.storelocator-search__button--go", }, "lincroyable.fr": { "name": "L'Incroyable", @@ -106,15 +117,6 @@ SITE_CONFIGS = { "category": "Déco & Maison", "requires_proxy": False, }, - "action.com": { - "name": "Action", - "search_url": "https://www.action.com/fr-fr/search/?q={query}", - "product_selector": "a[href^='/fr-fr/p/']", - "wait_selector": "a[href^='/fr-fr/p/']", - "category": "Discount", - "requires_proxy": False, - }, - # === GRANDES SURFACES === "e.leclerc": { "name": "E.Leclerc", "search_url": "https://www.e.leclerc/recherche?q={query}", @@ -123,6 +125,14 @@ SITE_CONFIGS = { "category": "Grande Surface", "requires_proxy": False, }, + "carrefour.fr": { + "name": "Carrefour", + "search_url": "https://www.carrefour.fr/s?q={query}", + "product_selector": "a.product-card-click-wrapper", + "wait_selector": "a.product-card-click-wrapper", + "category": "Grande Surface", + "requires_proxy": True, + }, # === E-COMMERCE GÉNÉRALISTE === "amazon.fr": { "name": "Amazon France", diff --git a/app/services/browserless_service.py b/app/services/browserless_service.py index 6c4d9a0..9e6baf7 100644 --- a/app/services/browserless_service.py +++ b/app/services/browserless_service.py @@ -93,9 +93,6 @@ class BrowserlessService: context = await self._browser.new_context(**options) - # Inject Stealth JS - await context.add_init_script(STEALTH_JS) - # Block aggressive tracking but keep images for visual verification if needed # (Optimized for speed vs detection) await context.route("**/*", lambda route: route.continue_()) diff --git a/app/services/search_service.py b/app/services/search_service.py index 9f6b084..e8a5307 100644 --- a/app/services/search_service.py +++ b/app/services/search_service.py @@ -185,14 +185,27 @@ class NewSearchService: # Extract title title = link.get_text(strip=True) - if not title: - # Try finding title in children or attributes - img = link.find("img") - if img and img.get("alt"): - title = img.get("alt") - elif link.get("title"): - title = link.get("title") + # If no text, check title attribute or nested image alt + if not title: + if link.get("title"): + title = link.get("title") + else: + img = link.find("img") + if img and img.get("alt"): + title = img.get("alt") + + # Special case for Stokomani or similar where link might be wrapping text but get_text failed or we selected a container + if not title and config.get("name") == "Stokomani": + # If we selected the container .product-card__title, the link is inside + child_link = link.find("a") + if child_link: + title = child_link.get_text(strip=True) + if not href: # Update href if we selected a container + href = child_link.get("href") + if href: + full_url = urljoin(base_url, href) + if not title or len(title) < 3: continue diff --git a/verify_scrapers.py b/verify_scrapers.py new file mode 100644 index 0000000..f30ff37 --- /dev/null +++ b/verify_scrapers.py @@ -0,0 +1,42 @@ +import asyncio +import logging +import sys +import os + +# Add project root to path +sys.path.append(os.getcwd()) + +from app.services.direct_search_service import direct_search_service +from app.core.search_config import SITE_CONFIGS + +# Configure logging +logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') +logger = logging.getLogger(__name__) + +async def verify_site(site_key): + logger.info(f"Verifying {site_key}...") + try: + # direct_search_service.search_site might need to be called differently if it's an instance method + # checking previous usage or assuming standard service pattern + results = await direct_search_service.search_site(site_key, "chaise") + if results: + logger.info(f"✅ {site_key}: Found {len(results)} results") + for i, res in enumerate(results[:3]): + title = res.get('title', 'No Title') + price = res.get('price', 'No Price') + url = res.get('url', 'No URL') + logger.info(f" {i+1}. {title[:50]}... - {price} - {url[:50]}...") + else: + logger.error(f"❌ {site_key}: No results found") + except Exception as e: + logger.error(f"❌ {site_key}: Error - {e}") + +async def main(): + sites_to_test = ["gifi.fr", "stokomani.fr", "auchan.fr", "carrefour.fr", "amazon.fr"] + + # Run sequentially to avoid overwhelming resources/logs + for site in sites_to_test: + await verify_site(site) + +if __name__ == "__main__": + asyncio.run(main())