feat: Implement comprehensive search configurations for discount stores and add Gifi-specific search and analysis tools.

This commit is contained in:
Michael committed 2025-12-01 00:36:54 +01:00
1 parent 63073d03b5
commit c159036583
22 files changed
+19849 -14

No files matched your search

+59
View File
@@ -0,0 +1,59 @@
"""
Analyze Carrefour price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Carrefour search...")
await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("article.product-list-card-plp-grid-new")
print(f"Found {len(products)} products\n")
for i, product in enumerate(products[:3]):
print(f"=== Product {i+1} ===")
# Title
title_el = product.select_one("h3, h2, a")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title[:60]}")
# Find all text with € symbol
import re
product_html = str(product)
prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
print(f"Prices found in HTML: {prices}")
# Look for price elements
price_els = product.find_all(string=re.compile('€'))
if price_els:
print(f"Elements with €:")
for el in price_els[:3]:
print(f" - {el.strip()[:50]}")
print()
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+70
View File
@@ -0,0 +1,70 @@
"""
Analyze a Carrefour product page for price selectors
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
import re
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# Visit a Carrefour product page (from the screenshot)
url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
print(f"Loading: {url}")
await page.goto(url, wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Find all elements with € symbol
price_els = soup.find_all(string=re.compile('€'))
print(f"\nFound {len(price_els)} elements with '€'")
prices_found = set()
for el in price_els[:20]:
text = el.strip()
if text and len(text) < 50:
prices_found.add(text)
parent = el.find_parent()
print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
# Try common price selectors
selectors = [
".product-price",
"[class*='price']",
".price",
"span.price",
"div.price",
"[data-price]"
]
print("\nTrying specific selectors:")
for selector in selectors:
try:
els = soup.select(selector)
if els:
for el in els[:2]:
text = el.get_text(strip=True)
if '€' in text:
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+69
View File
@@ -0,0 +1,69 @@
"""
Analyze Centrakor HTML structure for image selectors
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Centrakor search...")
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Try to find product containers
selectors = [
"div.product-item",
"div.product-card",
"article",
"div[class*='product']",
"li[class*='product']"
]
for selector in selectors:
products = soup.select(selector)
if products:
print(f"\n✓ Found {len(products)} products with selector: {selector}")
# Analyze first product
first = products[0]
print(f"\nFirst product HTML snippet:")
print(str(first)[:500])
print("\n...")
# Find all images
images = first.find_all('img')
print(f"\nFound {len(images)} images in first product:")
for i, img in enumerate(images):
print(f"\n Image {i+1}:")
print(f" Class: {img.get('class')}")
print(f" Src: {img.get('src', '')[:80]}")
print(f" Data-src: {img.get('data-src', '')[:80]}")
print(f" Alt: {img.get('alt', '')[:50]}")
break
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
+58
View File
@@ -0,0 +1,58 @@
"""
Detailed analysis of Centrakor image structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Analyzing {min(5, len(products))} products:\n")
for i, product in enumerate(products[:5]):
print(f"=== Product {i+1} ===")
# Title
title_el = product.select_one("a.product-item__name")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title}")
# All images
images = product.find_all('img')
print(f"Found {len(images)} img tags")
for j, img in enumerate(images):
print(f"\n Image {j+1}:")
print(f" tag: {img.name}")
print(f" class: {img.get('class')}")
for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
val = img.get(attr)
if val:
print(f" {attr}: {val[:80]}")
print()
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+24
View File
@@ -0,0 +1,24 @@
"""
Analyze Gifi HTML to find correct selectors
"""
with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
html = f.read()
# Find product-related divs
import re
matches = re.findall(r'<div[^>]*class="[^"]*"[^>]*>.*?</div>', html[:50000], re.DOTALL)
print(f"Total HTML size: {len(html)} bytes")
# Search for price patterns
price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
# Find all class names containing specific keywords
for keyword in ['product', 'article', 'item', 'card']:
classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
unique_classes = set(classes)
if unique_classes:
print(f"\nClasses containing '{keyword}':")
for cls in sorted(unique_classes):
print(f" - {cls}")
+69
View File
@@ -0,0 +1,69 @@
"""
Analyze a Gifi product page to understand price structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# Visit a Gifi product page
url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
print(f"Loading: {url}")
await page.goto(url, wait_until="networkidle")
# Get page title
title = await page.title()
print(f"Title: {title}")
# Find all elements with price-like text
price_els = await page.query_selector_all("*:has-text('€')")
print(f"\nFound {len(price_els)} elements with '€'")
# Get first 10 price elements
for i, el in enumerate(price_els[:10]):
text = await el.inner_text()
tag = await el.evaluate("el => el.tagName")
classes = await el.evaluate("el => el.className")
print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
# Try specific selectors
selectors = [
".price",
".product-price",
"[class*='price']",
"[data-price]",
"span.price",
"div.price"
]
print("\nTrying specific selectors:")
for selector in selectors:
try:
els = await page.query_selector_all(selector)
if els:
for el in els[:2]:
text = await el.inner_text()
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
+71
View File
@@ -0,0 +1,71 @@
"""
Test extracting price from Gifi search page HTML
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-tile")
print(f"Found {len(products)} products\n")
for i, product in enumerate(products[:3]):
print(f"\\n=== Product {i+1} ===")
# Get title
title_el = product.select_one("div.pdp-link a")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title}")
# Try to find price in product HTML
product_html = product.prettify()
# Look for price patterns
price_patterns = [
r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
r'€\s*(\d+)[,.](\d+)', # € 19,99
r'(\d+)€(\d+)', # 19€99
r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
]
for pattern in price_patterns:
matches = re.findall(pattern, product_html)
if matches:
print(f"Pattern '{pattern}': {matches[:3]}")
# Find all text with €
euro_texts = product.find_all(string=re.compile('€'))
if euro_texts:
print(f"Texts with €:")
for text in euro_texts[:5]:
print(f" - {text.strip()[:80]}")
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
+8 -6
View File
@@ -56,9 +56,11 @@ SITE_CONFIGS = {
"gifi.fr": {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
"product_selector": "article.product-miniature, div.product-item, div[class*='product']",
"product_image_selector": "img.product-thumbnail, img[class*='product'], img",
"wait_selector": "article.product-miniature, div.product-item",
"product_selector": "div.product-tile",
"product_link_selector": "a.link",
"product_title_selector": "div.pdp-link a",
"product_image_selector": "img.tile-image",
"wait_selector": "div.product-tile",
"category": "Discount",
"requires_proxy": False,
},
@@ -150,9 +152,9 @@ SITE_CONFIGS = {
"centrakor.com": {
"name": "Centrakor",
"search_url": "https://www.centrakor.com/search/{query}",
"product_selector": "div.product-item, div.product-card, article",
"product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']",
"wait_selector": "div.product-item, div.product-card, article",
"product_selector": "div.product-item",
"product_image_selector": "img.responsive-image__actual",
"wait_selector": "div.product-item",
"category": "Discount",
"requires_proxy": False,
},
+80 -8
View File
@@ -202,6 +202,8 @@ class ImprovedSearchService:
for item in links:
# Keep reference to original container for image search
container = item
if site_key == "centrakor.com":
logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}")
# If selector targets the container (div.product-card), we need to find the link inside
if item.name != 'a':
@@ -263,12 +265,13 @@ class ImprovedSearchService:
# PRIORITY 1: Use product_image_selector if configured (site-specific)
if "product_image_selector" in config:
# Search in the original container first
img_el = container.select_one(config["product_image_selector"])
# Search in the original container first - get ALL matches
img_els = container.select(config["product_image_selector"])
if img_el:
# Filter and find first valid image
for img_el in img_els:
# Try multiple attributes in order of priority
image_url= (
candidate_url = (
img_el.get("src") or
img_el.get("data-src") or
img_el.get("data-lazy-src") or
@@ -276,13 +279,46 @@ class ImprovedSearchService:
)
# Handle srcset (use first URL)
if not image_url and img_el.get("srcset"):
if not candidate_url and img_el.get("srcset"):
srcset = img_el.get("srcset")
# srcset format: "url1 size1, url2 size2"
image_url = srcset.split(",")[0].split()[0]
candidate_url = srcset.split(",")[0].split()[0]
if image_url:
# Skip invalid images (pictos, icons, etc.)
if candidate_url:
# SPECIAL: No filtering for Centrakor (debugging)
if site_key == "centrakor.com":
# Skip placeholders
if "placeholder" in candidate_url.lower():
logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}")
continue
# Filter only tiny pictos
if 'picto' in candidate_url.lower() and ('width=60' in candidate_url or 'height=80' in candidate_url):
logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}")
continue
image_url = candidate_url
logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}")
break
# Normal filtering for other sites
# Filter out obvious pictos and small icons
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}")
continue
# Filter out VERY small images (less than 100px)
import re
width_match = re.search(r'width=(\d+)', candidate_url)
height_match = re.search(r'height=(\d+)', candidate_url)
if width_match and height_match:
width = int(width_match.group(1))
height = int(height_match.group(1))
if width < 100 and height < 100:
logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}")
continue
# This is a valid product image
image_url = candidate_url
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
break
# PRIORITY 2: Fallback - Look for any img directly in the link
if not image_url:
@@ -336,12 +372,26 @@ class ImprovedSearchService:
if not image_url:
logger.warning(f" ⚠️ No image found for: {title[:50]}")
# Extract price from search results for sites where product pages are unavailable
product_price = None
if site_key == "gifi.fr":
# Gifi: Extract price from product tile HTML
import re
container_html = str(container)
price_match = re.search(r'(\d+)[,.](\d+)\s*€', container_html)
if price_match:
euros = int(price_match.group(1))
cents = int(price_match.group(2))
product_price = float(f"{euros}.{cents}")
logger.debug(f" 💰 Extracted price from search: {product_price}€")
# Create result
results.append(SearchResult(
url=full_url,
title=title,
snippet=f"Product from {config['name']}",
source=config["name"],
price=product_price, # Set price if extracted from search
image_url=image_url
))
@@ -352,6 +402,28 @@ class ImprovedSearchService:
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
"""Scrape price and details for a single item using same context"""
try:
# SPECIAL CASE: L'Incroyable - Price is in the title
if "lincroyable.fr" in result.url:
import re
# Extract price from title (e.g., "34€99" or "59€99")
price_match = re.search(r'(\d+)€(\d+)', result.title)
if price_match:
# Convert to float (e.g., "34€99" -> 34.99)
price_euros = int(price_match.group(1))
price_cents = int(price_match.group(2))
result.price = float(f"{price_euros}.{price_cents}")
# Clean title by removing price
result.title = re.sub(r'\d+€\d+', '', result.title).strip()
logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title")
return result
# SPECIAL CASE: Gifi - Price extracted from search, no need to visit page
if "gifi.fr" in result.url and result.price is not None:
logger.debug(f"Gifi - Price already extracted from search: {result.price}€")
return result
page = await context.new_page()
try:
logger.debug(f"Scraping details for: {result.title[:50]}...")
+56
View File
@@ -0,0 +1,56 @@
"""
Deep analysis - compare products WITH images vs WITHOUT images
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Analyzing {len(products)} products for image patterns\n")
for i, product in enumerate(products[:10]):
# Get title
title_el = product.select_one("a.product-item__name")
title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
# Get ALL images
all_imgs = product.select("img.responsive-image__actual")
print(f"\n=== {i+1}. {title[:50]} ===")
print(f"Found {len(all_imgs)} images")
for j, img in enumerate(all_imgs):
src = img.get('src', '')
print(f" Image {j+1}: {src if src else '(no src)'}")
if not src:
# Check other attributes
for attr in ['data-src', 'data-lazy-src', 'srcset']:
val = img.get(attr)
if val:
print(f" {attr}: {val[:80]}")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+56
View File
@@ -0,0 +1,56 @@
"""
Script to dump Gifi HTML and analyze structure
"""
import asyncio
import sys
import os
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search page...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
# Wait for products
try:
await page.wait_for_selector("article.product-miniature", timeout=10000)
except:
pass
# Save HTML
content = await page.content()
os.makedirs("/app/debug_dumps", exist_ok=True)
with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"HTML saved ({len(content)} bytes)")
# Extract first product structure
products = await page.query_selector_all("article.product-miniature")
print(f"Found {len(products)} products")
if products:
first_html = await products[0].evaluate("el => el.outerHTML")
with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
f.write(first_html)
print(f"First product HTML saved")
await context.close()
await browser.close()
await playwright.stop()
print("Done")
if __name__ == "__main__":
asyncio.run(main())
+76
View File
@@ -0,0 +1,76 @@
"""
Dump Gifi with longer wait for JavaScript
"""
import asyncio
import sys
import os
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search page...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
# Wait for ANY content
print("Waiting for content...")
await page.wait_for_timeout(5000)
# Save HTML
content = await page.content()
os.makedirs("/app/debug_dumps", exist_ok=True)
with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"HTML saved ({len(content)} bytes)")
# Find any elements with price
price_els = await page.query_selector_all("*:has-text('€')")
print(f"Elements with € symbol: {len(price_els)}")
# Find all divs/articles
all_divs = await page.query_selector_all("div, article, li")
print(f"Total divs/articles/li: {len(all_divs)}")
# Screenshot
await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
print("Screenshot saved")
# Get all classes
all_classes = await page.evaluate("""() => {
const elements = document.querySelectorAll('*');
const classes = new Set();
elements.forEach(el => {
if (el.className && typeof el.className === 'string') {
el.className.split(' ').forEach(cls => {
if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
classes.add(cls);
}
});
}
});
return Array.from(classes);
}""")
print(f"\\nProduct-related classes found:")
for cls in all_classes:
print(f" - {cls}")
await context.close()
await browser.close()
await playwright.stop()
print("Done")
if __name__ == "__main__":
asyncio.run(main())
View File
Whitespace-only changes.
+18821
View File
File diff suppressed because it is too large. Load diff
+42
View File
@@ -0,0 +1,42 @@
"""
Test Carrefour via API endpoint (production-like)
"""
import requests
import json
url = "http://localhost:8555/api/search"
params = {
"q": "chaise",
"sites": "7", # Carrefour site ID
"max_results": 10
}
print(f"Testing: {url}")
print(f"Params: {params}\n")
response = requests.get(url, params=params, stream=True)
print(f"Status: {response.status_code}")
print(f"Headers: {dict(response.headers)}\n")
count = 0
for line in response.iter_lines():
if line:
try:
# Each line should be JSON
data = json.loads(line.decode('utf-8'))
count += 1
title = data.get('title', 'N/A')[:60]
price = data.get('price', 'N/A')
image = "✅" if data.get('image_url') else "❌"
print(f"{count}. [{image}] {title} - {price}€")
if count >= 10:
break
except json.JSONDecodeError as e:
print(f"JSON Error: {e}")
print(f"Line: {line[:100]}")
print(f"\nTotal: {count} results")
+40
View File
@@ -0,0 +1,40 @@
"""
Test if current Carrefour price extraction works
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Carrefour for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
results.append(result)
count += 1
price_status = f"{result.price}€" if result.price else "N/A"
print(f"{count}. {result.title[:55]} - {price_status}")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Count prices
with_prices = sum(1 for r in results if r.price)
print(f"Products with prices: {with_prices}/{len(results)}")
if with_prices == len(results):
print("✅ SUCCESS: All products have prices!")
else:
print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+69
View File
@@ -0,0 +1,69 @@
"""
Test detailed logging for Centrakor image extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
import re
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Testing first product:\n")
first = products[0]
# Get all images
img_els = first.select("img.responsive-image__actual")
print(f"Found {len(img_els)} images with selector")
for i, img_el in enumerate(img_els):
print(f"\n=== Image {i+1} ===")
candidate_url = img_el.get("src")
print(f"URL: {candidate_url}")
# Test filters
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
print(" ❌ Filtered: Contains picto/icon/logo/badge keyword")
continue
width_match = re.search(r'width=(\d+)', candidate_url)
height_match = re.search(r'height=(\d+)', candidate_url)
print(f" Width match: {width_match.group(1) if width_match else None}")
print(f" Height match: {height_match.group(1) if height_match else None}")
if width_match and height_match:
width = int(width_match.group(1))
height = int(height_match.group(1))
print(f" Dimensions: {width}x{height}")
if width < 100 and height < 100:
print(f" ❌ Filtered: Too small ({width}x{height})")
continue
else:
print(f" ✅ PASS: Large enough ({width}x{height})")
else:
print(" ✅ PASS: No dimensions in URL")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+42
View File
@@ -0,0 +1,42 @@
"""
Test Centrakor image extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Centrakor for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"):
results.append(result)
count += 1
has_image = "✅" if result.image_url else "❌"
print(f"{count}. {has_image} {result.title[:55]} - {result.price}€")
if result.image_url:
print(f" Image: {result.image_url[:70]}...")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Count images
with_images = sum(1 for r in results if r.image_url)
print(f"Products with images: {with_images}/{len(results)}")
if with_images == len(results):
print("✅ SUCCESS: All products have images!")
else:
print(f"⚠️ WARNING: {len(results) - with_images} products missing images")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+34
View File
@@ -0,0 +1,34 @@
"""
Final test of Gifi with new selectors
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
print(f"✓ {result.title[:50]} - {result.price}€")
print(f"\n==> Total: {len(results)} results")
if results:
print("\nFirst 3 products:")
for i, r in enumerate(results[:3]):
print(f"{i+1}. Title: {r.title}")
print(f" Price: {r.price}€")
print(f" URL: {r.url[:80]}...")
print()
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+38
View File
@@ -0,0 +1,38 @@
"""
Test Gifi price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
count += 1
print(f"{count}. {result.title[:60]} - Price: {result.price}€")
if count >= 5: # Only test first 5
break
print(f"\n==> Got {len(results)} results")
# Check if all prices are the same
prices = [r.price for r in results if r.price]
if prices:
unique_prices = set(prices)
print(f"Unique prices: {unique_prices}")
if len(unique_prices) == 1:
print("⚠️ WARNING: All prices are the same!")
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+41
View File
@@ -0,0 +1,41 @@
"""
Test fixed Gifi price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
count += 1
print(f"{count}. {result.title[:60]} - {result.price}€")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Check price diversity
prices = [r.price for r in results if r.price]
if prices:
unique_prices = set(prices)
print(f"Unique prices: {sorted(unique_prices)}")
print(f"Price range: {min(prices)}€ - {max(prices)}€")
if len(unique_prices) > 1:
print("✅ SUCCESS: Multiple different prices found!")
else:
print("⚠️ WARNING: All prices are the same")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
+26
View File
@@ -0,0 +1,26 @@
"""
Test script to analyze Gifi HTML structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
print(f"Found: {result.title} - Price: {result.price}")
print(f"\nTotal results: {len(results)}")
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())