diff --git a/analyze_carrefour_prices.py b/analyze_carrefour_prices.py
new file mode 100644
index 0000000..c7aa92d
--- /dev/null
+++ b/analyze_carrefour_prices.py
@@ -0,0 +1,59 @@
+"""
+Analyze Carrefour price extraction
+"""
+import asyncio
+import sys
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+
+async def main():
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ print("Loading Carrefour search...")
+ await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
+
+ content = await page.content()
+ soup = BeautifulSoup(content, "html.parser")
+
+ products = soup.select("article.product-list-card-plp-grid-new")
+ print(f"Found {len(products)} products\n")
+
+ for i, product in enumerate(products[:3]):
+ print(f"=== Product {i+1} ===")
+
+ # Title
+ title_el = product.select_one("h3, h2, a")
+ title = title_el.get_text(strip=True) if title_el else "N/A"
+ print(f"Title: {title[:60]}")
+
+ # Find all text with € symbol
+ import re
+ product_html = str(product)
+ prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
+ print(f"Prices found in HTML: {prices}")
+
+ # Look for price elements
+ price_els = product.find_all(string=re.compile('€'))
+ if price_els:
+ print(f"Elements with €:")
+ for el in price_els[:3]:
+ print(f" - {el.strip()[:50]}")
+
+ print()
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/analyze_carrefour_product.py b/analyze_carrefour_product.py
new file mode 100644
index 0000000..96d18b3
--- /dev/null
+++ b/analyze_carrefour_product.py
@@ -0,0 +1,70 @@
+"""
+Analyze a Carrefour product page for price selectors
+"""
+import asyncio
+import sys
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+import re
+
+async def main():
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ # Visit a Carrefour product page (from the screenshot)
+ url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
+ print(f"Loading: {url}")
+ await page.goto(url, wait_until="networkidle")
+
+ content = await page.content()
+ soup = BeautifulSoup(content, "html.parser")
+
+ # Find all elements with € symbol
+ price_els = soup.find_all(string=re.compile('€'))
+ print(f"\nFound {len(price_els)} elements with '€'")
+
+ prices_found = set()
+ for el in price_els[:20]:
+ text = el.strip()
+ if text and len(text) < 50:
+ prices_found.add(text)
+ parent = el.find_parent()
+ print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
+
+ # Try common price selectors
+ selectors = [
+ ".product-price",
+ "[class*='price']",
+ ".price",
+ "span.price",
+ "div.price",
+ "[data-price]"
+ ]
+
+ print("\nTrying specific selectors:")
+ for selector in selectors:
+ try:
+ els = soup.select(selector)
+ if els:
+ for el in els[:2]:
+ text = el.get_text(strip=True)
+ if '€' in text:
+ print(f" {selector}: {text}")
+ except:
+ pass
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/analyze_centrakor.py b/analyze_centrakor.py
new file mode 100644
index 0000000..ea34efb
--- /dev/null
+++ b/analyze_centrakor.py
@@ -0,0 +1,69 @@
+"""
+Analyze Centrakor HTML structure for image selectors
+"""
+import asyncio
+import sys
+import re
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+
+async def main():
+ print("Connecting to browserless...")
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ print("Loading Centrakor search...")
+ await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
+
+ content = await page.content()
+
+ soup = BeautifulSoup(content, "html.parser")
+
+ # Try to find product containers
+ selectors = [
+ "div.product-item",
+ "div.product-card",
+ "article",
+ "div[class*='product']",
+ "li[class*='product']"
+ ]
+
+ for selector in selectors:
+ products = soup.select(selector)
+ if products:
+ print(f"\n✓ Found {len(products)} products with selector: {selector}")
+
+ # Analyze first product
+ first = products[0]
+ print(f"\nFirst product HTML snippet:")
+ print(str(first)[:500])
+ print("\n...")
+
+ # Find all images
+ images = first.find_all('img')
+ print(f"\nFound {len(images)} images in first product:")
+ for i, img in enumerate(images):
+ print(f"\n Image {i+1}:")
+ print(f" Class: {img.get('class')}")
+ print(f" Src: {img.get('src', '')[:80]}")
+ print(f" Data-src: {img.get('data-src', '')[:80]}")
+ print(f" Alt: {img.get('alt', '')[:50]}")
+
+ break
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+ print("\nDone")
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/analyze_centrakor_detailed.py b/analyze_centrakor_detailed.py
new file mode 100644
index 0000000..0d205b5
--- /dev/null
+++ b/analyze_centrakor_detailed.py
@@ -0,0 +1,58 @@
+"""
+Detailed analysis of Centrakor image structure
+"""
+import asyncio
+import sys
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+
+async def main():
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
+ content = await page.content()
+
+ soup = BeautifulSoup(content, "html.parser")
+ products = soup.select("div.product-item")
+
+ print(f"Analyzing {min(5, len(products))} products:\n")
+
+ for i, product in enumerate(products[:5]):
+ print(f"=== Product {i+1} ===")
+
+ # Title
+ title_el = product.select_one("a.product-item__name")
+ title = title_el.get_text(strip=True) if title_el else "N/A"
+ print(f"Title: {title}")
+
+ # All images
+ images = product.find_all('img')
+ print(f"Found {len(images)} img tags")
+
+ for j, img in enumerate(images):
+ print(f"\n Image {j+1}:")
+ print(f" tag: {img.name}")
+ print(f" class: {img.get('class')}")
+ for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
+ val = img.get(attr)
+ if val:
+ print(f" {attr}: {val[:80]}")
+
+ print()
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/analyze_gifi.py b/analyze_gifi.py
new file mode 100644
index 0000000..64bf2b0
--- /dev/null
+++ b/analyze_gifi.py
@@ -0,0 +1,24 @@
+"""
+Analyze Gifi HTML to find correct selectors
+"""
+with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
+ html = f.read()
+
+# Find product-related divs
+import re
+matches = re.findall(r'
]*class="[^"]*"[^>]*>.*?
', html[:50000], re.DOTALL)
+
+print(f"Total HTML size: {len(html)} bytes")
+
+# Search for price patterns
+price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
+print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
+
+# Find all class names containing specific keywords
+for keyword in ['product', 'article', 'item', 'card']:
+ classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
+ unique_classes = set(classes)
+ if unique_classes:
+ print(f"\nClasses containing '{keyword}':")
+ for cls in sorted(unique_classes):
+ print(f" - {cls}")
diff --git a/analyze_gifi_price.py b/analyze_gifi_price.py
new file mode 100644
index 0000000..2c9c2a5
--- /dev/null
+++ b/analyze_gifi_price.py
@@ -0,0 +1,69 @@
+"""
+Analyze a Gifi product page to understand price structure
+"""
+import asyncio
+import sys
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+
+async def main():
+ print("Connecting to browserless...")
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ # Visit a Gifi product page
+ url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
+ print(f"Loading: {url}")
+ await page.goto(url, wait_until="networkidle")
+
+ # Get page title
+ title = await page.title()
+ print(f"Title: {title}")
+
+ # Find all elements with price-like text
+ price_els = await page.query_selector_all("*:has-text('€')")
+ print(f"\nFound {len(price_els)} elements with '€'")
+
+ # Get first 10 price elements
+ for i, el in enumerate(price_els[:10]):
+ text = await el.inner_text()
+ tag = await el.evaluate("el => el.tagName")
+ classes = await el.evaluate("el => el.className")
+ print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
+
+ # Try specific selectors
+ selectors = [
+ ".price",
+ ".product-price",
+ "[class*='price']",
+ "[data-price]",
+ "span.price",
+ "div.price"
+ ]
+
+ print("\nTrying specific selectors:")
+ for selector in selectors:
+ try:
+ els = await page.query_selector_all(selector)
+ if els:
+ for el in els[:2]:
+ text = await el.inner_text()
+ print(f" {selector}: {text}")
+ except:
+ pass
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+ print("\nDone")
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/analyze_gifi_search_prices.py b/analyze_gifi_search_prices.py
new file mode 100644
index 0000000..c2018a9
--- /dev/null
+++ b/analyze_gifi_search_prices.py
@@ -0,0 +1,71 @@
+"""
+Test extracting price from Gifi search page HTML
+"""
+import asyncio
+import sys
+import re
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+
+async def main():
+ print("Connecting to browserless...")
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ print("Loading Gifi search...")
+ await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
+
+ content = await page.content()
+
+ soup = BeautifulSoup(content, "html.parser")
+ products = soup.select("div.product-tile")
+
+ print(f"Found {len(products)} products\n")
+
+ for i, product in enumerate(products[:3]):
+ print(f"\\n=== Product {i+1} ===")
+
+ # Get title
+ title_el = product.select_one("div.pdp-link a")
+ title = title_el.get_text(strip=True) if title_el else "N/A"
+ print(f"Title: {title}")
+
+ # Try to find price in product HTML
+ product_html = product.prettify()
+
+ # Look for price patterns
+ price_patterns = [
+ r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
+ r'€\s*(\d+)[,.](\d+)', # € 19,99
+ r'(\d+)€(\d+)', # 19€99
+ r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
+ ]
+
+ for pattern in price_patterns:
+ matches = re.findall(pattern, product_html)
+ if matches:
+ print(f"Pattern '{pattern}': {matches[:3]}")
+
+ # Find all text with €
+ euro_texts = product.find_all(string=re.compile('€'))
+ if euro_texts:
+ print(f"Texts with €:")
+ for text in euro_texts[:5]:
+ print(f" - {text.strip()[:80]}")
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+ print("\nDone")
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/app/core/search_config.py b/app/core/search_config.py
index b3ea6e4..8e1cc4a 100644
--- a/app/core/search_config.py
+++ b/app/core/search_config.py
@@ -56,9 +56,11 @@ SITE_CONFIGS = {
"gifi.fr": {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
- "product_selector": "article.product-miniature, div.product-item, div[class*='product']",
- "product_image_selector": "img.product-thumbnail, img[class*='product'], img",
- "wait_selector": "article.product-miniature, div.product-item",
+ "product_selector": "div.product-tile",
+ "product_link_selector": "a.link",
+ "product_title_selector": "div.pdp-link a",
+ "product_image_selector": "img.tile-image",
+ "wait_selector": "div.product-tile",
"category": "Discount",
"requires_proxy": False,
},
@@ -150,9 +152,9 @@ SITE_CONFIGS = {
"centrakor.com": {
"name": "Centrakor",
"search_url": "https://www.centrakor.com/search/{query}",
- "product_selector": "div.product-item, div.product-card, article",
- "product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']",
- "wait_selector": "div.product-item, div.product-card, article",
+ "product_selector": "div.product-item",
+ "product_image_selector": "img.responsive-image__actual",
+ "wait_selector": "div.product-item",
"category": "Discount",
"requires_proxy": False,
},
diff --git a/app/services/improved_search_service.py b/app/services/improved_search_service.py
index 57b01b0..325ed8e 100644
--- a/app/services/improved_search_service.py
+++ b/app/services/improved_search_service.py
@@ -202,6 +202,8 @@ class ImprovedSearchService:
for item in links:
# Keep reference to original container for image search
container = item
+ if site_key == "centrakor.com":
+ logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}")
# If selector targets the container (div.product-card), we need to find the link inside
if item.name != 'a':
@@ -263,12 +265,13 @@ class ImprovedSearchService:
# PRIORITY 1: Use product_image_selector if configured (site-specific)
if "product_image_selector" in config:
- # Search in the original container first
- img_el = container.select_one(config["product_image_selector"])
+ # Search in the original container first - get ALL matches
+ img_els = container.select(config["product_image_selector"])
- if img_el:
+ # Filter and find first valid image
+ for img_el in img_els:
# Try multiple attributes in order of priority
- image_url= (
+ candidate_url = (
img_el.get("src") or
img_el.get("data-src") or
img_el.get("data-lazy-src") or
@@ -276,13 +279,46 @@ class ImprovedSearchService:
)
# Handle srcset (use first URL)
- if not image_url and img_el.get("srcset"):
+ if not candidate_url and img_el.get("srcset"):
srcset = img_el.get("srcset")
- # srcset format: "url1 size1, url2 size2"
- image_url = srcset.split(",")[0].split()[0]
+ candidate_url = srcset.split(",")[0].split()[0]
- if image_url:
+ # Skip invalid images (pictos, icons, etc.)
+ if candidate_url:
+ # SPECIAL: No filtering for Centrakor (debugging)
+ if site_key == "centrakor.com":
+ # Skip placeholders
+ if "placeholder" in candidate_url.lower():
+ logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}")
+ continue
+ # Filter only tiny pictos
+ if 'picto' in candidate_url.lower() and ('width=60' in candidate_url or 'height=80' in candidate_url):
+ logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}")
+ continue
+ image_url = candidate_url
+ logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}")
+ break
+
+ # Normal filtering for other sites
+ # Filter out obvious pictos and small icons
+ if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
+ logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}")
+ continue
+ # Filter out VERY small images (less than 100px)
+ import re
+ width_match = re.search(r'width=(\d+)', candidate_url)
+ height_match = re.search(r'height=(\d+)', candidate_url)
+ if width_match and height_match:
+ width = int(width_match.group(1))
+ height = int(height_match.group(1))
+ if width < 100 and height < 100:
+ logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}")
+ continue
+
+ # This is a valid product image
+ image_url = candidate_url
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
+ break
# PRIORITY 2: Fallback - Look for any img directly in the link
if not image_url:
@@ -336,12 +372,26 @@ class ImprovedSearchService:
if not image_url:
logger.warning(f" ⚠️ No image found for: {title[:50]}")
+ # Extract price from search results for sites where product pages are unavailable
+ product_price = None
+ if site_key == "gifi.fr":
+ # Gifi: Extract price from product tile HTML
+ import re
+ container_html = str(container)
+ price_match = re.search(r'(\d+)[,.](\d+)\s*€', container_html)
+ if price_match:
+ euros = int(price_match.group(1))
+ cents = int(price_match.group(2))
+ product_price = float(f"{euros}.{cents}")
+ logger.debug(f" 💰 Extracted price from search: {product_price}€")
+
# Create result
results.append(SearchResult(
url=full_url,
title=title,
snippet=f"Product from {config['name']}",
source=config["name"],
+ price=product_price, # Set price if extracted from search
image_url=image_url
))
@@ -352,6 +402,28 @@ class ImprovedSearchService:
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
"""Scrape price and details for a single item using same context"""
try:
+ # SPECIAL CASE: L'Incroyable - Price is in the title
+ if "lincroyable.fr" in result.url:
+ import re
+ # Extract price from title (e.g., "34€99" or "59€99")
+ price_match = re.search(r'(\d+)€(\d+)', result.title)
+ if price_match:
+ # Convert to float (e.g., "34€99" -> 34.99)
+ price_euros = int(price_match.group(1))
+ price_cents = int(price_match.group(2))
+ result.price = float(f"{price_euros}.{price_cents}")
+
+ # Clean title by removing price
+ result.title = re.sub(r'\d+€\d+', '', result.title).strip()
+ logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title")
+
+ return result
+
+ # SPECIAL CASE: Gifi - Price extracted from search, no need to visit page
+ if "gifi.fr" in result.url and result.price is not None:
+ logger.debug(f"Gifi - Price already extracted from search: {result.price}€")
+ return result
+
page = await context.new_page()
try:
logger.debug(f"Scraping details for: {result.title[:50]}...")
diff --git a/deep_centrakor_analysis.py b/deep_centrakor_analysis.py
new file mode 100644
index 0000000..2be63d0
--- /dev/null
+++ b/deep_centrakor_analysis.py
@@ -0,0 +1,56 @@
+"""
+Deep analysis - compare products WITH images vs WITHOUT images
+"""
+import asyncio
+import sys
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+from bs4 import BeautifulSoup
+
+async def main():
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
+ content = await page.content()
+
+ soup = BeautifulSoup(content, "html.parser")
+ products = soup.select("div.product-item")
+
+ print(f"Analyzing {len(products)} products for image patterns\n")
+
+ for i, product in enumerate(products[:10]):
+ # Get title
+ title_el = product.select_one("a.product-item__name")
+ title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
+
+ # Get ALL images
+ all_imgs = product.select("img.responsive-image__actual")
+
+ print(f"\n=== {i+1}. {title[:50]} ===")
+ print(f"Found {len(all_imgs)} images")
+
+ for j, img in enumerate(all_imgs):
+ src = img.get('src', '')
+ print(f" Image {j+1}: {src if src else '(no src)'}")
+ if not src:
+ # Check other attributes
+ for attr in ['data-src', 'data-lazy-src', 'srcset']:
+ val = img.get(attr)
+ if val:
+ print(f" {attr}: {val[:80]}")
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/dump_gifi.py b/dump_gifi.py
new file mode 100644
index 0000000..bcb8c31
--- /dev/null
+++ b/dump_gifi.py
@@ -0,0 +1,56 @@
+"""
+Script to dump Gifi HTML and analyze structure
+"""
+import asyncio
+import sys
+import os
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+
+async def main():
+ print("Connecting to browserless...")
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ print("Loading Gifi search page...")
+ await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
+
+ # Wait for products
+ try:
+ await page.wait_for_selector("article.product-miniature", timeout=10000)
+ except:
+ pass
+
+ # Save HTML
+ content = await page.content()
+ os.makedirs("/app/debug_dumps", exist_ok=True)
+ with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
+ f.write(content)
+
+ print(f"HTML saved ({len(content)} bytes)")
+
+ # Extract first product structure
+ products = await page.query_selector_all("article.product-miniature")
+ print(f"Found {len(products)} products")
+
+ if products:
+ first_html = await products[0].evaluate("el => el.outerHTML")
+ with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
+ f.write(first_html)
+ print(f"First product HTML saved")
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+ print("Done")
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/dump_gifi_v2.py b/dump_gifi_v2.py
new file mode 100644
index 0000000..2594580
--- /dev/null
+++ b/dump_gifi_v2.py
@@ -0,0 +1,76 @@
+"""
+Dump Gifi with longer wait for JavaScript
+"""
+import asyncio
+import sys
+import os
+sys.path.insert(0, '/app')
+
+from playwright.async_api import async_playwright
+
+async def main():
+ print("Connecting to browserless...")
+ playwright = await async_playwright().start()
+ browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
+
+ context = await browser.new_context(
+ viewport={"width": 1920, "height": 1080},
+ user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
+ )
+
+ page = await context.new_page()
+
+ print("Loading Gifi search page...")
+ await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
+
+ # Wait for ANY content
+ print("Waiting for content...")
+ await page.wait_for_timeout(5000)
+
+ # Save HTML
+ content = await page.content()
+ os.makedirs("/app/debug_dumps", exist_ok=True)
+ with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
+ f.write(content)
+
+ print(f"HTML saved ({len(content)} bytes)")
+
+ # Find any elements with price
+ price_els = await page.query_selector_all("*:has-text('€')")
+ print(f"Elements with € symbol: {len(price_els)}")
+
+ # Find all divs/articles
+ all_divs = await page.query_selector_all("div, article, li")
+ print(f"Total divs/articles/li: {len(all_divs)}")
+
+ # Screenshot
+ await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
+ print("Screenshot saved")
+
+ # Get all classes
+ all_classes = await page.evaluate("""() => {
+ const elements = document.querySelectorAll('*');
+ const classes = new Set();
+ elements.forEach(el => {
+ if (el.className && typeof el.className === 'string') {
+ el.className.split(' ').forEach(cls => {
+ if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
+ classes.add(cls);
+ }
+ });
+ }
+ });
+ return Array.from(classes);
+ }""")
+
+ print(f"\\nProduct-related classes found:")
+ for cls in all_classes:
+ print(f" - {cls}")
+
+ await context.close()
+ await browser.close()
+ await playwright.stop()
+ print("Done")
+
+if __name__ == "__main__":
+ asyncio.run(main())
diff --git a/gifi_dump.html b/gifi_dump.html
new file mode 100644
index 0000000..e69de29
diff --git a/gifi_full.html b/gifi_full.html
new file mode 100644
index 0000000..0455d34
--- /dev/null
+++ b/gifi_full.html
@@ -0,0 +1,18821 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ chaise pas cher | GIFI
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
Continuer sans accepter →
Avec votre consentement, nos partenaires et nous-mêmes utilisons des cookies ou des technologies similaires afin de mesurer l’audience et la performance des contenus du site gifi.fr, personnaliser les contenus et diffuser des publicités adaptées à vos centres d’intérêt.
+Certains cookies sont indispensables au fonctionnement du site. Ils permettent notamment l’accès aux fonctionnalités principales et garantissent la sécurité de votre navigation. Ces cookies strictement nécessaires ne peuvent pas être désactivés.
+D’autres cookies, dits facultatifs, sous réserve de votre consentement, peuvent être utilisés pour analyser, améliorer et personnaliser nos contenus, nos publicités ainsi que votre parcours sur le site. Le refus de ces cookies facultatifs pourrait affecter votre expérience sur notre site, ainsi vous n’aurez peut-être pas accès à certaines offres ou contenus personnalisés ou à des publicités pertinentes.
+Pour en savoir plus sur la finalité de chaque traceur, vous pouvez consulter notre tableau des cookies.
+Nous conservons votre choix pendant 6 mois. À tout moment, vous pouvez acco
+rder ou retirer votre consentement en cliquant sur le lien "Gérer mes cookies" situé en bas à droite de chaque page de notre site.
+Pour plus d’informations, vous pouvez cliquer sur « En savoir plus » ou consulter notre politique de confidentialité