From 63073d03b5f852f95ba4511c73ae3a5f62eaecb4 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 30 Nov 2025 23:54:48 +0100 Subject: [PATCH] feat: introduce new search configuration, improved search service, and Carrefour search integration test. --- app/core/search_config.py | 3 ++ app/services/improved_search_service.py | 72 +++++++++++++++++++++---- test_carrefour_search.py | 26 +++++++++ 3 files changed, 91 insertions(+), 10 deletions(-) create mode 100644 test_carrefour_search.py diff --git a/app/core/search_config.py b/app/core/search_config.py index 2c289ec..b3ea6e4 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -56,6 +56,9 @@ SITE_CONFIGS = { "gifi.fr": { "name": "Gifi", "search_url": "https://www.gifi.fr/resultat-recherche?q={query}", + "product_selector": "article.product-miniature, div.product-item, div[class*='product']", + "product_image_selector": "img.product-thumbnail, img[class*='product'], img", + "wait_selector": "article.product-miniature, div.product-item", "category": "Discount", "requires_proxy": False, }, diff --git a/app/services/improved_search_service.py b/app/services/improved_search_service.py index f8d2d0d..57b01b0 100644 --- a/app/services/improved_search_service.py +++ b/app/services/improved_search_service.py @@ -395,35 +395,79 @@ class ImprovedSearchService: @staticmethod async def _extract_price(page: Page) -> float | None: """Extract price from product page using multiple selectors""" + import re + + # PRIORITY 1: Selectors for sale/promotional prices (highest priority) + sale_price_selectors = [ + '.price-current', + '.prix-actuel', + '.sale-price', + '.promo-price', + '[class*="promo"]', + '[class*="sale"]', + '[class*="discount"]', + ] + + # PRIORITY 2: Standard price selectors price_selectors = [ '.price', '[data-testid="price"]', - '.prix-actuel', - '.price-current', '[itemprop="price"]', '.product-price', '.a-price .a-offscreen', '.a-price-whole', 'span[class*="price"]', ] - - for selector in price_selectors: + + # Try sale prices first + for selector in sale_price_selectors: try: elements = await page.query_selector_all(selector) for elem in elements: + # Skip if element is strikethrough (old price) + parent_html = await elem.evaluate('el => el.parentElement.outerHTML') + if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html: + continue + price_text = await elem.inner_text() if price_text: - # Parse price - import re cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() - cleaned = cleaned.replace(' ', '').replace('\xa0', '') - cleaned = cleaned.replace(',', '.') - + cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') + + match = re.search(r'(\d+\.?\d*)', cleaned) + if match: + try: + price_val = float(match.group(1)) + if 0.01 < price_val < 100000: + logger.debug(f"Found sale price: {price_val}€ from {selector}") + return price_val + except ValueError: + continue + except Exception: + continue + + # Fallback to standard price selectors + for selector in price_selectors: + try: + elements = await page.query_selector_all(selector) + for elem in elements: + # Skip if element is strikethrough (old price) + try: + parent_html = await elem.evaluate('el => el.parentElement.outerHTML') + if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html: + continue + except: + pass + + price_text = await elem.inner_text() + if price_text: + cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() + cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') + match = re.search(r'(\d+\.?\d*)', cleaned) if match: try: price_val = float(match.group(1)) - # Validate price if 0.01 < price_val < 100000: logger.debug(f"Found price: {price_val}€ from {selector}") return price_val @@ -517,6 +561,14 @@ class ImprovedSearchService: if not initial_results: logger.warning(f"No results found for {site_key}") + # Dump HTML for debugging + import os + dump_dir = "/app/debug_dumps" + os.makedirs(dump_dir, exist_ok=True) + dump_path = f"{dump_dir}/{site_key.replace('.', '_')}_no_results.html" + with open(dump_path, 'w', encoding='utf-8') as f: + f.write(content) + logger.warning(f"HTML dumped to {dump_path} for inspection") return # Phase 2: Scrape details (Streaming) diff --git a/test_carrefour_search.py b/test_carrefour_search.py new file mode 100644 index 0000000..def6b68 --- /dev/null +++ b/test_carrefour_search.py @@ -0,0 +1,26 @@ +""" +Test script to trigger Carrefour search and generate HTML dump +""" +import asyncio +import sys +sys.path.insert(0, '/app') + +from app.services.improved_search_service import ImprovedSearchService + +async def main(): + print("Initializing browser...") + await ImprovedSearchService.initialize() + + print("Searching Carrefour for 'chaise'...") + results = [] + async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"): + results.append(result) + print(f"Found: {result.title}") + + print(f"\nTotal results: {len(results)}") + + print("Shutting down...") + await ImprovedSearchService.shutdown() + +if __name__ == "__main__": + asyncio.run(main())