diff --git a/Dockerfile b/Dockerfile index e5d3f13..5550cc7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -65,6 +65,7 @@ COPY alembic.ini ./ COPY docker-entrypoint.sh ./ # Make entrypoint executable +RUN sed -i 's/\r$//' docker-entrypoint.sh RUN chmod +x docker-entrypoint.sh # Copy built frontend static files diff --git a/app/core/search_config.py b/app/core/search_config.py index 70bd2bc..cde55ec 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -65,9 +65,9 @@ SITE_CONFIGS = { "stokomani.fr": { "name": "Stokomani", "search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}", - "product_selector": "a.reversed-link.block, a[href*='/products/']", - "product_image_selector": "img[loading='lazy'], img[class*='object'], img[sizes]", - "wait_selector": "a.reversed-link.block, .product-card", + "product_selector": "div.product-card", + "product_image_selector": "div.media-wrapper img, img[class*='product-card__image']", + "wait_selector": "div.product-card", "category": "Discount", "requires_proxy": False, }, @@ -130,9 +130,9 @@ SITE_CONFIGS = { "lincroyable.fr": { "name": "L'Incroyable", "search_url": "https://www.lincroyable.fr/recherche-query={query}/", - "product_selector": "div.product-card, div.product-miniature, article", - "product_image_selector": "img.product-image, img[class*='product'], picture img", - "wait_selector": "div.product-card, div.product-miniature, article", + "product_selector": "div.tailleBlocProdNew", + "product_image_selector": "img.imgCoup2coeur, img[class*='product']", + "wait_selector": "div.tailleBlocProdNew", "category": "Discount", "requires_proxy": False, }, @@ -157,18 +157,18 @@ SITE_CONFIGS = { "auchan.fr": { "name": "Auchan", "search_url": "https://www.auchan.fr/recherche?text={query}", - "product_selector": "article, div[class*='product-card'], div[class*='list__item']", - "product_image_selector": "img[class*='product'], img[src*='auchan']", - "wait_selector": "article, div[class*='product-card'], div[class*='list__item']", + "product_selector": "article.product-thumbnail a.product-thumbnail__details-wrapper, div[class*='product-card'] a", + "product_image_selector": ".product-thumbnail__picture img, img[class*='product']", + "wait_selector": "article.product-thumbnail, div[class*='product-card']", "category": "Grande Surface", "requires_proxy": False, }, "carrefour.fr": { "name": "Carrefour", "search_url": "https://www.carrefour.fr/s?q={query}", - "product_selector": "a[href*='/p/'], a[href*='/produit'], article a, div[class*='product'] a", - "product_image_selector": "img, picture source", - "wait_selector": None, + "product_selector": "article.product-list-card-plp-grid-new", + "product_image_selector": "img.product-card-image-new__content", + "wait_selector": "article.product-list-card-plp-grid-new", "category": "Grande Surface", "requires_proxy": False, }, diff --git a/app/services/parsers/auchan_parser.py b/app/services/parsers/auchan_parser.py index a8e9221..7a8545f 100644 --- a/app/services/parsers/auchan_parser.py +++ b/app/services/parsers/auchan_parser.py @@ -14,9 +14,9 @@ class AuchanParser(BaseParser): # Auchan products - Ultra Robust Strategy # The DOM is flat and dynamic. We rely on finding product links first. - # Links usually contain '/p-' in the href. + # Links usually contain '/p-' or '/pr-' in the href. - links = soup.select("a[href*='/p-']") + links = soup.select("a[href*='/p-'], a[href*='/pr-']") seen_urls = set() for link in links: @@ -26,7 +26,7 @@ class AuchanParser(BaseParser): continue # Filter out non-product links if any (e.g. facets) - if '/p-' not in href: + if '/p-' not in href and '/pr-' not in href: continue seen_urls.add(href) @@ -50,7 +50,7 @@ class AuchanParser(BaseParser): # Title title = link.get('title') if not title: - title_el = container.select_one("h3, div[class*='title'], span[class*='title']") + title_el = container.select_one("p.product-thumbnail__description, h3, div[class*='title'], span[class*='title']") if title_el: title = title_el.get_text(strip=True) if not title: @@ -61,7 +61,7 @@ class AuchanParser(BaseParser): # Image img_url = None - img_el = container.select_one("img") + img_el = container.select_one(".product-thumbnail__picture img, img") if img_el: img_url = self._get_image_src(img_el) if img_url: @@ -69,7 +69,7 @@ class AuchanParser(BaseParser): # Price price = None - price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price") + price_el = container.select_one("div.product-price, div[class*='price'], span[class*='price'], .product-price") if price_el: price = self.parse_price_text(price_el.get_text()) diff --git a/app/services/parsers/carrefour_parser.py b/app/services/parsers/carrefour_parser.py index e5e0ffb..891509b 100644 --- a/app/services/parsers/carrefour_parser.py +++ b/app/services/parsers/carrefour_parser.py @@ -13,15 +13,17 @@ class CarrefourParser(BaseParser): results = [] # Carrefour products - # New container: div containing both image and title link - cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']") + # New container: article.large-horizontal or div.product-list-card-plp-grid-new + cards = soup.select("article.large-horizontal, div.product-list-card-plp-grid-new, article, div[class*='product-card']") # If no cards, try finding by link if not cards: links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']") cards = [] for link in links: - parent = link.find_parent("div", class_=lambda x: x and "product" in x) + parent = link.find_parent("article") + if not parent: + parent = link.find_parent("div", class_=lambda x: x and "product" in x) if parent: cards.append(parent) else: @@ -36,14 +38,14 @@ class CarrefourParser(BaseParser): href = link_el.get('href') url = self.make_absolute_url(href) - title_el = card.select_one("h3, h2, [class*='title']") + title_el = card.select_one("h3.product-card-title__text, h3, h2, [class*='title']") title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) if not title: continue img_url = None - img_el = card.select_one("img") + img_el = card.select_one("img.product-card-image-new__content, img") if img_el: img_url = self._get_image_src(img_el) if img_url: @@ -51,7 +53,7 @@ class CarrefourParser(BaseParser): price = None # Price is often text node near h3 or in a specific price element - price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']") + price_el = card.select_one("div.product-price__amount--main, [class*='price'], .product-card-price, span[class*='amount']") if price_el: price = self.parse_price_text(price_el.get_text()) diff --git a/app/services/parsers/lincroyable_parser.py b/app/services/parsers/lincroyable_parser.py index 1c0397d..c7a85dc 100644 --- a/app/services/parsers/lincroyable_parser.py +++ b/app/services/parsers/lincroyable_parser.py @@ -13,51 +13,39 @@ class LIncroyableParser(BaseParser): results = [] # L'Incroyable products - # Config: a.product-link + # Config: div.tailleBlocProdNew - # Try finding cards first - cards = soup.select("div.product-card, div.product-miniature, article") + cards = soup.select("div.tailleBlocProdNew") - # If no cards found, try finding by link and getting parent - if not cards: - links = soup.select("a[href*='/p/'], a[href*='/produit/']") - cards = [] - for link in links: - # Try to find a container div - parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x)) - if parent: - cards.append(parent) - else: - cards.append(link.parent) # Fallback to immediate parent - for card in cards: try: - link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link") + # Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links) + # We look for the main link to the product + link_el = card.select_one("a[href*='/p']") if not link_el: - # If card is the link itself - if card.name == 'a' and card.get('href'): - link_el = card - else: - continue + continue href = link_el.get('href') url = self.make_absolute_url(href) - title_el = card.select_one(".product-title, h3, h2, [class*='title']") + # Title + title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew") title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) if not title: continue + # Image img_url = None - img_el = card.select_one("img") + img_el = card.select_one("img.imgCoup2coeur, img") if img_el: img_url = self._get_image_src(img_el) if img_url: img_url = self.make_absolute_url(img_url) + # Price price = None - price_el = card.select_one(".price, .product-price, [class*='price']") + price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew") if price_el: price = self.parse_price_text(price_el.get_text()) diff --git a/app/services/parsers/stokomani_parser.py b/app/services/parsers/stokomani_parser.py index d9d944e..ba86a19 100644 --- a/app/services/parsers/stokomani_parser.py +++ b/app/services/parsers/stokomani_parser.py @@ -13,63 +13,47 @@ class StokomaniParser(BaseParser): soup = BeautifulSoup(html, "html.parser") results = [] - # Find product title links - links = soup.select("a.reversed-link.block") + # Stokomani products + # Config: div.product-card - seen_urls = set() + cards = soup.select("div.product-card") - for link in links: + for card in cards: try: - href = link.get('href') - if not href or href in seen_urls: - continue - - if '/products/' not in href: + # Link + link_el = card.select_one("h3.product-card__title a, a[href*='/products/']") + if not link_el: continue - seen_urls.add(href) + href = link_el.get('href') url = self.make_absolute_url(href) - title = link.get_text(strip=True) + # Title + title_el = card.select_one("span.reversed-link__text, h3.product-card__title") + title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) + if not title: continue - # Image: Look for the preceding with aria-label + # Image img_url = None - prev_a = link.find_previous_sibling("a", attrs={"aria-label": True}) - if prev_a and prev_a.get('href') == href: - # Try specific selector first - img_el = prev_a.select_one("motion-element img, img") - if img_el: - # Check for lazy loading attributes explicitly - img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src') - if img_url and " " in img_url: - # Handle srcset: take the first URL - img_url = img_url.split(" ")[0] - - if not img_url: - img_url = self.extract_image_url(prev_a) + img_el = card.select_one("div.media-wrapper img, img") + if img_el: + img_url = self._get_image_src(img_el) + if img_url: + img_url = self.make_absolute_url(img_url) - if img_url: - img_url = self.make_absolute_url(img_url) - - # Price: Look for text node after the link + # Price price = None - next_sibling = link.next_sibling - while next_sibling: - if isinstance(next_sibling, NavigableString): - price_text = next_sibling.strip() - if "€" in price_text: - price = self.parse_price_text(price_text) - if price: - break - elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])): - # Try finding price in next div if it's a price container - price = self.parse_price_text(next_sibling.get_text()) - if price: - break - next_sibling = next_sibling.next_sibling - + price_el = card.select_one("span.f-price-item--regular, .f-price-item, [class*='price']") + if price_el: + price = self.parse_price_text(price_el.get_text()) + + # Fallback price search in text + if not price: + text = card.get_text(" ", strip=True) + price = self.parse_price_text(text) + results.append(ProductResult( title=title, url=url, diff --git a/app/services/search_service.py b/app/services/search_service.py index a7cea28..34cba5c 100644 --- a/app/services/search_service.py +++ b/app/services/search_service.py @@ -108,21 +108,45 @@ class NewSearchService: return result # Use AI to analyze - from app.services.ai_service import AIService - ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text) + try: + from app.services.ai_service import AIService + # Check if AI service is available/configured before calling? + # For now, just try/except the call + ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text) + + if ai_result: + extraction, _ = ai_result + result.price = extraction.price + result.currency = extraction.currency or "EUR" + result.in_stock = extraction.in_stock + + # Update image URL to point to our local screenshot + import os + filename = os.path.basename(screenshot_path) + result.image_url = f"/screenshots/{filename}" + else: + raise Exception("AI returned no result") + + except Exception as e: + logger.warning(f"AI Analysis failed for {result.url}: {e}") + # Fallback: Try to extract price from page_text if Browserless found it + if page_text and "PRIX DÉTECTÉ:" in page_text: + try: + import re + price_match = re.search(r"PRIX DÉTECTÉ:\s*([\d\.]+)", page_text) + if price_match: + price_val = float(price_match.group(1)) + result.price = price_val + logger.info(f"💰 Fallback: Extracted price {price_val} from text for {result.url}") + except Exception as parse_e: + logger.error(f"Error parsing fallback price: {parse_e}") + + # Still use the screenshot if we have it + if screenshot_path: + import os + filename = os.path.basename(screenshot_path) + result.image_url = f"/screenshots/{filename}" - if ai_result: - extraction, _ = ai_result - result.price = extraction.price - result.currency = extraction.currency or "EUR" - result.in_stock = extraction.in_stock - - # Update image URL to point to our local screenshot - # The frontend expects /screenshots/filename - import os - filename = os.path.basename(screenshot_path) - result.image_url = f"/screenshots/{filename}" - except Exception as e: logger.error(f"Error scraping item {result.url}: {e}") @@ -191,6 +215,17 @@ class NewSearchService: # break href = link.get("href") + + # Special handling for sites where selector targets a container (Carrefour, Stokomani) + if not href and config.get("name") in ["Carrefour", "Stokomani"]: + # Try to find the main product link inside the container + # For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one + child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a") + if child_link: + href = child_link.get("href") + # Update link to point to the anchor for title/image extraction + link = child_link + if not href: continue diff --git a/docker-entrypoint.sh b/docker-entrypoint.sh index f3a7502..3bb4834 100644 --- a/docker-entrypoint.sh +++ b/docker-entrypoint.sh @@ -5,7 +5,7 @@ echo "Setting up screenshots directory..." mkdir -p screenshots echo "Running database migrations..." -alembic upgrade heads +# alembic upgrade heads echo "Starting application..." exec uvicorn app.main:app --host 0.0.0.0 --port 8555 diff --git a/dump_html.py b/dump_html.py index f0d0d72..3c3742f 100644 --- a/dump_html.py +++ b/dump_html.py @@ -6,7 +6,7 @@ import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).parent)) -from app.services.improved_search_service import ImprovedSearchService +from app.services.browserless_service import browserless_service from app.core.search_config import SITE_CONFIGS async def dump_search_html(site_key: str, query: str = "chaise"): @@ -19,31 +19,37 @@ async def dump_search_html(site_key: str, query: str = "chaise"): print(f"\n🔍 Dumping HTML for: {config['name']}") # Ensure browser is initialized - await ImprovedSearchService.initialize() - - # Create context manually to get HTML - context = await ImprovedSearchService._create_context(ImprovedSearchService._browser) - page = await context.new_page() + await browserless_service.initialize() try: search_url = config["search_url"].format(query=query) print(f" URL: {search_url}") - await page.goto(search_url, wait_until="networkidle", timeout=30000) - await ImprovedSearchService._handle_popups(page) - await page.wait_for_timeout(3000) + # Override wait_selector for La Foir'Fouille + wait_selector = config.get("wait_selector") + if site_key == "lafoirfouille.fr": + wait_selector = ".sf-grid-vignet" + print(f" ⚠️ Overriding wait_selector to: {wait_selector}") - html = await page.content() + html_content, screenshot_path = await browserless_service.get_page_content( + search_url, + wait_selector=wait_selector, + use_proxy=config.get("requires_proxy", False) + ) + if not html_content: + print(" ❌ No HTML content returned") + return + filename = f"dump_{site_key.replace('.', '_')}.html" with open(filename, "w", encoding="utf-8") as f: - f.write(html) + f.write(html_content) - print(f" ✅ Saved to: {filename} ({len(html)} bytes)") + print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)") # Quick analysis from bs4 import BeautifulSoup - soup = BeautifulSoup(html, "html.parser") + soup = BeautifulSoup(html_content, "html.parser") # Try current selector current_selector = config.get("product_selector") @@ -55,19 +61,18 @@ async def dump_search_html(site_key: str, query: str = "chaise"): img_selector = config["product_image_selector"] img_matches = soup.select(img_selector) print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}") + + except Exception as e: + print(f" ❌ Error during dump: {e}") finally: - await context.close() + # We don't close the browser here to allow reuse if needed, + # but main() will shut it down. + pass async def main(): sites = [ - "e-leclerc.com", - "auchan.fr", - "carrefour.fr", - "stokomani.fr", - "centrakor.com", - "cdiscount.com", - "lincroyable.fr" + "stokomani.fr" ] for site_key in sites: @@ -77,7 +82,7 @@ async def main(): print(f"❌ Error: {e}") await asyncio.sleep(1) - await ImprovedSearchService.shutdown() + await browserless_service.shutdown() if __name__ == "__main__": asyncio.run(main()) diff --git a/inspect_carrefour.py b/inspect_carrefour.py new file mode 100644 index 0000000..91a0b10 --- /dev/null +++ b/inspect_carrefour.py @@ -0,0 +1,33 @@ +from bs4 import BeautifulSoup + +with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f: + html = f.read() + +soup = BeautifulSoup(html, "html.parser") +articles = soup.select("article.product-list-card-plp-grid-new") + +print(f"Found {len(articles)} articles") + +if articles: + first = articles[0] + print("\n--- First Article Structure ---") + print(first.prettify()[:1000]) # Print first 1000 chars + + # Check for link + link = first.select_one("a.product-card-click-wrapper") + if link: + print(f"\nLink found: {link.get('href')}") + print(f"Link classes: {link.get('class')}") + + # Check for image INSIDE link + img = link.select_one("img.product-card-image-new__content") + if img: + print(f"\n✅ Image found INSIDE link: {img.get('src')}") + else: + print(f"\n❌ Image NOT found inside link") + # Check if image is elsewhere in article + img_article = first.select_one("img.product-card-image-new__content") + if img_article: + print(f" But image exists in article: {img_article.get('src')}") + else: + print("\nNo link found with selector a.product-card-click-wrapper") diff --git a/inspect_lafoirfouille.py b/inspect_lafoirfouille.py new file mode 100644 index 0000000..55d43de --- /dev/null +++ b/inspect_lafoirfouille.py @@ -0,0 +1,53 @@ +from bs4 import BeautifulSoup + +with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f: + html = f.read() + +soup = BeautifulSoup(html, "html.parser") + +# Try to find product containers +print("Searching for product containers...") +potential_selectors = [ + "div.product-miniature", + "article", + "div[class*='product']", + "div.product-card", + "div.item" +] + +for selector in potential_selectors: + matches = soup.select(selector) + print(f"Selector '{selector}' matches: {len(matches)}") + if len(matches) > 0 and len(matches) < 5: + # If few matches, print classes to see if it's a wrapper + print(f" Classes: {matches[0].get('class')}") + +# Print structure of first potential product +products = soup.select("div.product-miniature") +if not products: + products = soup.select("div[class*='product-item']") + +if products: + first = products[0] + print("\n--- First Product Structure ---") + print(first.prettify()[:1000]) + + link = first.find("a") + if link: + print(f"\nLink found: {link.get('href')}") + + img = first.find("img") + if img: + print(f"\nImage found: {img.get('src')}") +else: + print("\nNo obvious products found. Dumping generic structure...") + # Find any div with many children + divs = soup.find_all("div") + for div in divs: + if len(div.find_all("div", recursive=False)) > 10: + print(f"Found container with many children: {div.get('class')}") + # Print first child + child = div.find("div") + if child: + print(child.prettify()[:500]) + break diff --git a/inspect_lafoirfouille_images.py b/inspect_lafoirfouille_images.py new file mode 100644 index 0000000..b3e1cb3 --- /dev/null +++ b/inspect_lafoirfouille_images.py @@ -0,0 +1,25 @@ +from bs4 import BeautifulSoup + +with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f: + html = f.read() + +soup = BeautifulSoup(html, "html.parser") +images = soup.select("img") + +print(f"Found {len(images)} images") + +for i, img in enumerate(images[:10]): + print(f"\n--- Image {i+1} ---") + print(f"Src: {img.get('src')}") + print(f"Classes: {img.get('class')}") + + parent = img.parent + print(f"Parent: {parent.name} (Classes: {parent.get('class')})") + + grandparent = parent.parent + if grandparent: + print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})") + + greatgrandparent = grandparent.parent + if greatgrandparent: + print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})") diff --git a/verify_targets.py b/verify_targets.py new file mode 100644 index 0000000..ee74aea --- /dev/null +++ b/verify_targets.py @@ -0,0 +1,45 @@ +import asyncio +import logging +import sys +import os + +# Add project root to path +sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) + +from app.core.search_config import SITE_CONFIGS +from app.services.search_service import new_search_service +from app.services.browserless_service import browserless_service + +# Configure logging +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s - %(name)s - %(levelname)s - %(message)s", + handlers=[logging.StreamHandler()] +) + +async def test_specific_sites(): + target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"] + query = "chaise" + + print(f"Testing {target_sites} with query '{query}'...") + + await browserless_service.initialize() + + for site_key in target_sites: + if site_key not in SITE_CONFIGS: + print(f"Skipping {site_key} (not in config)") + continue + + print(f"\n--- Testing {site_key} ---") + try: + results = await new_search_service.search_site(site_key, query) + print(f"Found {len(results)} results") + for r in results[:3]: + print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]") + except Exception as e: + print(f"Error testing {site_key}: {e}") + + await browserless_service.shutdown() + +if __name__ == "__main__": + asyncio.run(test_specific_sites())