mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #200 from R0m1k3/antigravity
feat: Implement comprehensive search configurations for discount stor…
This commit is contained in:
22 files changed
+19849
-14
No files matched your search
@@ -0,0 +1,59 @@
|
||||
"""
|
||||
Analyze Carrefour price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Carrefour search...")
|
||||
await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
products = soup.select("article.product-list-card-plp-grid-new")
|
||||
print(f"Found {len(products)} products\n")
|
||||
|
||||
for i, product in enumerate(products[:3]):
|
||||
print(f"=== Product {i+1} ===")
|
||||
|
||||
# Title
|
||||
title_el = product.select_one("h3, h2, a")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title[:60]}")
|
||||
|
||||
# Find all text with € symbol
|
||||
import re
|
||||
product_html = str(product)
|
||||
prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
|
||||
print(f"Prices found in HTML: {prices}")
|
||||
|
||||
# Look for price elements
|
||||
price_els = product.find_all(string=re.compile('€'))
|
||||
if price_els:
|
||||
print(f"Elements with €:")
|
||||
for el in price_els[:3]:
|
||||
print(f" - {el.strip()[:50]}")
|
||||
|
||||
print()
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,70 @@
|
||||
"""
|
||||
Analyze a Carrefour product page for price selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
import re
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# Visit a Carrefour product page (from the screenshot)
|
||||
url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
|
||||
print(f"Loading: {url}")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Find all elements with € symbol
|
||||
price_els = soup.find_all(string=re.compile('€'))
|
||||
print(f"\nFound {len(price_els)} elements with '€'")
|
||||
|
||||
prices_found = set()
|
||||
for el in price_els[:20]:
|
||||
text = el.strip()
|
||||
if text and len(text) < 50:
|
||||
prices_found.add(text)
|
||||
parent = el.find_parent()
|
||||
print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
|
||||
|
||||
# Try common price selectors
|
||||
selectors = [
|
||||
".product-price",
|
||||
"[class*='price']",
|
||||
".price",
|
||||
"span.price",
|
||||
"div.price",
|
||||
"[data-price]"
|
||||
]
|
||||
|
||||
print("\nTrying specific selectors:")
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = soup.select(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = el.get_text(strip=True)
|
||||
if '€' in text:
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,69 @@
|
||||
"""
|
||||
Analyze Centrakor HTML structure for image selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Centrakor search...")
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Try to find product containers
|
||||
selectors = [
|
||||
"div.product-item",
|
||||
"div.product-card",
|
||||
"article",
|
||||
"div[class*='product']",
|
||||
"li[class*='product']"
|
||||
]
|
||||
|
||||
for selector in selectors:
|
||||
products = soup.select(selector)
|
||||
if products:
|
||||
print(f"\n✓ Found {len(products)} products with selector: {selector}")
|
||||
|
||||
# Analyze first product
|
||||
first = products[0]
|
||||
print(f"\nFirst product HTML snippet:")
|
||||
print(str(first)[:500])
|
||||
print("\n...")
|
||||
|
||||
# Find all images
|
||||
images = first.find_all('img')
|
||||
print(f"\nFound {len(images)} images in first product:")
|
||||
for i, img in enumerate(images):
|
||||
print(f"\n Image {i+1}:")
|
||||
print(f" Class: {img.get('class')}")
|
||||
print(f" Src: {img.get('src', '')[:80]}")
|
||||
print(f" Data-src: {img.get('data-src', '')[:80]}")
|
||||
print(f" Alt: {img.get('alt', '')[:50]}")
|
||||
|
||||
break
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,58 @@
|
||||
"""
|
||||
Detailed analysis of Centrakor image structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Analyzing {min(5, len(products))} products:\n")
|
||||
|
||||
for i, product in enumerate(products[:5]):
|
||||
print(f"=== Product {i+1} ===")
|
||||
|
||||
# Title
|
||||
title_el = product.select_one("a.product-item__name")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title}")
|
||||
|
||||
# All images
|
||||
images = product.find_all('img')
|
||||
print(f"Found {len(images)} img tags")
|
||||
|
||||
for j, img in enumerate(images):
|
||||
print(f"\n Image {j+1}:")
|
||||
print(f" tag: {img.name}")
|
||||
print(f" class: {img.get('class')}")
|
||||
for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
|
||||
val = img.get(attr)
|
||||
if val:
|
||||
print(f" {attr}: {val[:80]}")
|
||||
|
||||
print()
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,24 @@
|
||||
"""
|
||||
Analyze Gifi HTML to find correct selectors
|
||||
"""
|
||||
with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
# Find product-related divs
|
||||
import re
|
||||
matches = re.findall(r'<div[^>]*class="[^"]*"[^>]*>.*?</div>', html[:50000], re.DOTALL)
|
||||
|
||||
print(f"Total HTML size: {len(html)} bytes")
|
||||
|
||||
# Search for price patterns
|
||||
price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
|
||||
print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
|
||||
|
||||
# Find all class names containing specific keywords
|
||||
for keyword in ['product', 'article', 'item', 'card']:
|
||||
classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
|
||||
unique_classes = set(classes)
|
||||
if unique_classes:
|
||||
print(f"\nClasses containing '{keyword}':")
|
||||
for cls in sorted(unique_classes):
|
||||
print(f" - {cls}")
|
||||
@@ -0,0 +1,69 @@
|
||||
"""
|
||||
Analyze a Gifi product page to understand price structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# Visit a Gifi product page
|
||||
url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
|
||||
print(f"Loading: {url}")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
# Get page title
|
||||
title = await page.title()
|
||||
print(f"Title: {title}")
|
||||
|
||||
# Find all elements with price-like text
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"\nFound {len(price_els)} elements with '€'")
|
||||
|
||||
# Get first 10 price elements
|
||||
for i, el in enumerate(price_els[:10]):
|
||||
text = await el.inner_text()
|
||||
tag = await el.evaluate("el => el.tagName")
|
||||
classes = await el.evaluate("el => el.className")
|
||||
print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
|
||||
|
||||
# Try specific selectors
|
||||
selectors = [
|
||||
".price",
|
||||
".product-price",
|
||||
"[class*='price']",
|
||||
"[data-price]",
|
||||
"span.price",
|
||||
"div.price"
|
||||
]
|
||||
|
||||
print("\nTrying specific selectors:")
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = await page.query_selector_all(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = await el.inner_text()
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,71 @@
|
||||
"""
|
||||
Test extracting price from Gifi search page HTML
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-tile")
|
||||
|
||||
print(f"Found {len(products)} products\n")
|
||||
|
||||
for i, product in enumerate(products[:3]):
|
||||
print(f"\\n=== Product {i+1} ===")
|
||||
|
||||
# Get title
|
||||
title_el = product.select_one("div.pdp-link a")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title}")
|
||||
|
||||
# Try to find price in product HTML
|
||||
product_html = product.prettify()
|
||||
|
||||
# Look for price patterns
|
||||
price_patterns = [
|
||||
r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
|
||||
r'€\s*(\d+)[,.](\d+)', # € 19,99
|
||||
r'(\d+)€(\d+)', # 19€99
|
||||
r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
|
||||
]
|
||||
|
||||
for pattern in price_patterns:
|
||||
matches = re.findall(pattern, product_html)
|
||||
if matches:
|
||||
print(f"Pattern '{pattern}': {matches[:3]}")
|
||||
|
||||
# Find all text with €
|
||||
euro_texts = product.find_all(string=re.compile('€'))
|
||||
if euro_texts:
|
||||
print(f"Texts with €:")
|
||||
for text in euro_texts[:5]:
|
||||
print(f" - {text.strip()[:80]}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -56,9 +56,11 @@ SITE_CONFIGS = {
|
||||
"gifi.fr": {
|
||||
"name": "Gifi",
|
||||
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
|
||||
"product_selector": "article.product-miniature, div.product-item, div[class*='product']",
|
||||
"product_image_selector": "img.product-thumbnail, img[class*='product'], img",
|
||||
"wait_selector": "article.product-miniature, div.product-item",
|
||||
"product_selector": "div.product-tile",
|
||||
"product_link_selector": "a.link",
|
||||
"product_title_selector": "div.pdp-link a",
|
||||
"product_image_selector": "img.tile-image",
|
||||
"wait_selector": "div.product-tile",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
@@ -150,9 +152,9 @@ SITE_CONFIGS = {
|
||||
"centrakor.com": {
|
||||
"name": "Centrakor",
|
||||
"search_url": "https://www.centrakor.com/search/{query}",
|
||||
"product_selector": "div.product-item, div.product-card, article",
|
||||
"product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']",
|
||||
"wait_selector": "div.product-item, div.product-card, article",
|
||||
"product_selector": "div.product-item",
|
||||
"product_image_selector": "img.responsive-image__actual",
|
||||
"wait_selector": "div.product-item",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
|
||||
@@ -202,6 +202,8 @@ class ImprovedSearchService:
|
||||
for item in links:
|
||||
# Keep reference to original container for image search
|
||||
container = item
|
||||
if site_key == "centrakor.com":
|
||||
logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}")
|
||||
|
||||
# If selector targets the container (div.product-card), we need to find the link inside
|
||||
if item.name != 'a':
|
||||
@@ -263,12 +265,13 @@ class ImprovedSearchService:
|
||||
|
||||
# PRIORITY 1: Use product_image_selector if configured (site-specific)
|
||||
if "product_image_selector" in config:
|
||||
# Search in the original container first
|
||||
img_el = container.select_one(config["product_image_selector"])
|
||||
# Search in the original container first - get ALL matches
|
||||
img_els = container.select(config["product_image_selector"])
|
||||
|
||||
if img_el:
|
||||
# Filter and find first valid image
|
||||
for img_el in img_els:
|
||||
# Try multiple attributes in order of priority
|
||||
image_url= (
|
||||
candidate_url = (
|
||||
img_el.get("src") or
|
||||
img_el.get("data-src") or
|
||||
img_el.get("data-lazy-src") or
|
||||
@@ -276,13 +279,46 @@ class ImprovedSearchService:
|
||||
)
|
||||
|
||||
# Handle srcset (use first URL)
|
||||
if not image_url and img_el.get("srcset"):
|
||||
if not candidate_url and img_el.get("srcset"):
|
||||
srcset = img_el.get("srcset")
|
||||
# srcset format: "url1 size1, url2 size2"
|
||||
image_url = srcset.split(",")[0].split()[0]
|
||||
candidate_url = srcset.split(",")[0].split()[0]
|
||||
|
||||
if image_url:
|
||||
# Skip invalid images (pictos, icons, etc.)
|
||||
if candidate_url:
|
||||
# SPECIAL: No filtering for Centrakor (debugging)
|
||||
if site_key == "centrakor.com":
|
||||
# Skip placeholders
|
||||
if "placeholder" in candidate_url.lower():
|
||||
logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}")
|
||||
continue
|
||||
# Filter only tiny pictos
|
||||
if 'picto' in candidate_url.lower() and ('width=60' in candidate_url or 'height=80' in candidate_url):
|
||||
logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}")
|
||||
continue
|
||||
image_url = candidate_url
|
||||
logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}")
|
||||
break
|
||||
|
||||
# Normal filtering for other sites
|
||||
# Filter out obvious pictos and small icons
|
||||
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
|
||||
logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}")
|
||||
continue
|
||||
# Filter out VERY small images (less than 100px)
|
||||
import re
|
||||
width_match = re.search(r'width=(\d+)', candidate_url)
|
||||
height_match = re.search(r'height=(\d+)', candidate_url)
|
||||
if width_match and height_match:
|
||||
width = int(width_match.group(1))
|
||||
height = int(height_match.group(1))
|
||||
if width < 100 and height < 100:
|
||||
logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}")
|
||||
continue
|
||||
|
||||
# This is a valid product image
|
||||
image_url = candidate_url
|
||||
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
|
||||
break
|
||||
|
||||
# PRIORITY 2: Fallback - Look for any img directly in the link
|
||||
if not image_url:
|
||||
@@ -336,12 +372,26 @@ class ImprovedSearchService:
|
||||
if not image_url:
|
||||
logger.warning(f" ⚠️ No image found for: {title[:50]}")
|
||||
|
||||
# Extract price from search results for sites where product pages are unavailable
|
||||
product_price = None
|
||||
if site_key == "gifi.fr":
|
||||
# Gifi: Extract price from product tile HTML
|
||||
import re
|
||||
container_html = str(container)
|
||||
price_match = re.search(r'(\d+)[,.](\d+)\s*€', container_html)
|
||||
if price_match:
|
||||
euros = int(price_match.group(1))
|
||||
cents = int(price_match.group(2))
|
||||
product_price = float(f"{euros}.{cents}")
|
||||
logger.debug(f" 💰 Extracted price from search: {product_price}€")
|
||||
|
||||
# Create result
|
||||
results.append(SearchResult(
|
||||
url=full_url,
|
||||
title=title,
|
||||
snippet=f"Product from {config['name']}",
|
||||
source=config["name"],
|
||||
price=product_price, # Set price if extracted from search
|
||||
image_url=image_url
|
||||
))
|
||||
|
||||
@@ -352,6 +402,28 @@ class ImprovedSearchService:
|
||||
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
|
||||
"""Scrape price and details for a single item using same context"""
|
||||
try:
|
||||
# SPECIAL CASE: L'Incroyable - Price is in the title
|
||||
if "lincroyable.fr" in result.url:
|
||||
import re
|
||||
# Extract price from title (e.g., "34€99" or "59€99")
|
||||
price_match = re.search(r'(\d+)€(\d+)', result.title)
|
||||
if price_match:
|
||||
# Convert to float (e.g., "34€99" -> 34.99)
|
||||
price_euros = int(price_match.group(1))
|
||||
price_cents = int(price_match.group(2))
|
||||
result.price = float(f"{price_euros}.{price_cents}")
|
||||
|
||||
# Clean title by removing price
|
||||
result.title = re.sub(r'\d+€\d+', '', result.title).strip()
|
||||
logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title")
|
||||
|
||||
return result
|
||||
|
||||
# SPECIAL CASE: Gifi - Price extracted from search, no need to visit page
|
||||
if "gifi.fr" in result.url and result.price is not None:
|
||||
logger.debug(f"Gifi - Price already extracted from search: {result.price}€")
|
||||
return result
|
||||
|
||||
page = await context.new_page()
|
||||
try:
|
||||
logger.debug(f"Scraping details for: {result.title[:50]}...")
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
"""
|
||||
Deep analysis - compare products WITH images vs WITHOUT images
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Analyzing {len(products)} products for image patterns\n")
|
||||
|
||||
for i, product in enumerate(products[:10]):
|
||||
# Get title
|
||||
title_el = product.select_one("a.product-item__name")
|
||||
title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
|
||||
|
||||
# Get ALL images
|
||||
all_imgs = product.select("img.responsive-image__actual")
|
||||
|
||||
print(f"\n=== {i+1}. {title[:50]} ===")
|
||||
print(f"Found {len(all_imgs)} images")
|
||||
|
||||
for j, img in enumerate(all_imgs):
|
||||
src = img.get('src', '')
|
||||
print(f" Image {j+1}: {src if src else '(no src)'}")
|
||||
if not src:
|
||||
# Check other attributes
|
||||
for attr in ['data-src', 'data-lazy-src', 'srcset']:
|
||||
val = img.get(attr)
|
||||
if val:
|
||||
print(f" {attr}: {val[:80]}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,56 @@
|
||||
"""
|
||||
Script to dump Gifi HTML and analyze structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search page...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
|
||||
|
||||
# Wait for products
|
||||
try:
|
||||
await page.wait_for_selector("article.product-miniature", timeout=10000)
|
||||
except:
|
||||
pass
|
||||
|
||||
# Save HTML
|
||||
content = await page.content()
|
||||
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||
with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
|
||||
print(f"HTML saved ({len(content)} bytes)")
|
||||
|
||||
# Extract first product structure
|
||||
products = await page.query_selector_all("article.product-miniature")
|
||||
print(f"Found {len(products)} products")
|
||||
|
||||
if products:
|
||||
first_html = await products[0].evaluate("el => el.outerHTML")
|
||||
with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
|
||||
f.write(first_html)
|
||||
print(f"First product HTML saved")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("Done")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,76 @@
|
||||
"""
|
||||
Dump Gifi with longer wait for JavaScript
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search page...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||
|
||||
# Wait for ANY content
|
||||
print("Waiting for content...")
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Save HTML
|
||||
content = await page.content()
|
||||
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||
with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
|
||||
print(f"HTML saved ({len(content)} bytes)")
|
||||
|
||||
# Find any elements with price
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"Elements with € symbol: {len(price_els)}")
|
||||
|
||||
# Find all divs/articles
|
||||
all_divs = await page.query_selector_all("div, article, li")
|
||||
print(f"Total divs/articles/li: {len(all_divs)}")
|
||||
|
||||
# Screenshot
|
||||
await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
|
||||
print("Screenshot saved")
|
||||
|
||||
# Get all classes
|
||||
all_classes = await page.evaluate("""() => {
|
||||
const elements = document.querySelectorAll('*');
|
||||
const classes = new Set();
|
||||
elements.forEach(el => {
|
||||
if (el.className && typeof el.className === 'string') {
|
||||
el.className.split(' ').forEach(cls => {
|
||||
if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
|
||||
classes.add(cls);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
return Array.from(classes);
|
||||
}""")
|
||||
|
||||
print(f"\\nProduct-related classes found:")
|
||||
for cls in all_classes:
|
||||
print(f" - {cls}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("Done")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
Whitespace-only changes.
+18821
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,42 @@
|
||||
"""
|
||||
Test Carrefour via API endpoint (production-like)
|
||||
"""
|
||||
import requests
|
||||
import json
|
||||
|
||||
url = "http://localhost:8555/api/search"
|
||||
params = {
|
||||
"q": "chaise",
|
||||
"sites": "7", # Carrefour site ID
|
||||
"max_results": 10
|
||||
}
|
||||
|
||||
print(f"Testing: {url}")
|
||||
print(f"Params: {params}\n")
|
||||
|
||||
response = requests.get(url, params=params, stream=True)
|
||||
|
||||
print(f"Status: {response.status_code}")
|
||||
print(f"Headers: {dict(response.headers)}\n")
|
||||
|
||||
count = 0
|
||||
for line in response.iter_lines():
|
||||
if line:
|
||||
try:
|
||||
# Each line should be JSON
|
||||
data = json.loads(line.decode('utf-8'))
|
||||
count += 1
|
||||
|
||||
title = data.get('title', 'N/A')[:60]
|
||||
price = data.get('price', 'N/A')
|
||||
image = "✅" if data.get('image_url') else "❌"
|
||||
|
||||
print(f"{count}. [{image}] {title} - {price}€")
|
||||
|
||||
if count >= 10:
|
||||
break
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"JSON Error: {e}")
|
||||
print(f"Line: {line[:100]}")
|
||||
|
||||
print(f"\nTotal: {count} results")
|
||||
@@ -0,0 +1,40 @@
|
||||
"""
|
||||
Test if current Carrefour price extraction works
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Carrefour for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
price_status = f"{result.price}€" if result.price else "N/A"
|
||||
print(f"{count}. {result.title[:55]} - {price_status}")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Count prices
|
||||
with_prices = sum(1 for r in results if r.price)
|
||||
print(f"Products with prices: {with_prices}/{len(results)}")
|
||||
|
||||
if with_prices == len(results):
|
||||
print("✅ SUCCESS: All products have prices!")
|
||||
else:
|
||||
print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,69 @@
|
||||
"""
|
||||
Test detailed logging for Centrakor image extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
import re
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Testing first product:\n")
|
||||
|
||||
first = products[0]
|
||||
|
||||
# Get all images
|
||||
img_els = first.select("img.responsive-image__actual")
|
||||
print(f"Found {len(img_els)} images with selector")
|
||||
|
||||
for i, img_el in enumerate(img_els):
|
||||
print(f"\n=== Image {i+1} ===")
|
||||
candidate_url = img_el.get("src")
|
||||
print(f"URL: {candidate_url}")
|
||||
|
||||
# Test filters
|
||||
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
|
||||
print(" ❌ Filtered: Contains picto/icon/logo/badge keyword")
|
||||
continue
|
||||
|
||||
width_match = re.search(r'width=(\d+)', candidate_url)
|
||||
height_match = re.search(r'height=(\d+)', candidate_url)
|
||||
print(f" Width match: {width_match.group(1) if width_match else None}")
|
||||
print(f" Height match: {height_match.group(1) if height_match else None}")
|
||||
|
||||
if width_match and height_match:
|
||||
width = int(width_match.group(1))
|
||||
height = int(height_match.group(1))
|
||||
print(f" Dimensions: {width}x{height}")
|
||||
if width < 100 and height < 100:
|
||||
print(f" ❌ Filtered: Too small ({width}x{height})")
|
||||
continue
|
||||
else:
|
||||
print(f" ✅ PASS: Large enough ({width}x{height})")
|
||||
else:
|
||||
print(" ✅ PASS: No dimensions in URL")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,42 @@
|
||||
"""
|
||||
Test Centrakor image extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Centrakor for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
has_image = "✅" if result.image_url else "❌"
|
||||
print(f"{count}. {has_image} {result.title[:55]} - {result.price}€")
|
||||
if result.image_url:
|
||||
print(f" Image: {result.image_url[:70]}...")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Count images
|
||||
with_images = sum(1 for r in results if r.image_url)
|
||||
print(f"Products with images: {with_images}/{len(results)}")
|
||||
|
||||
if with_images == len(results):
|
||||
print("✅ SUCCESS: All products have images!")
|
||||
else:
|
||||
print(f"⚠️ WARNING: {len(results) - with_images} products missing images")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,34 @@
|
||||
"""
|
||||
Final test of Gifi with new selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
print(f"✓ {result.title[:50]} - {result.price}€")
|
||||
|
||||
print(f"\n==> Total: {len(results)} results")
|
||||
|
||||
if results:
|
||||
print("\nFirst 3 products:")
|
||||
for i, r in enumerate(results[:3]):
|
||||
print(f"{i+1}. Title: {r.title}")
|
||||
print(f" Price: {r.price}€")
|
||||
print(f" URL: {r.url[:80]}...")
|
||||
print()
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,38 @@
|
||||
"""
|
||||
Test Gifi price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
print(f"{count}. {result.title[:60]} - Price: {result.price}€")
|
||||
if count >= 5: # Only test first 5
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Check if all prices are the same
|
||||
prices = [r.price for r in results if r.price]
|
||||
if prices:
|
||||
unique_prices = set(prices)
|
||||
print(f"Unique prices: {unique_prices}")
|
||||
if len(unique_prices) == 1:
|
||||
print("⚠️ WARNING: All prices are the same!")
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,41 @@
|
||||
"""
|
||||
Test fixed Gifi price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
print(f"{count}. {result.title[:60]} - {result.price}€")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Check price diversity
|
||||
prices = [r.price for r in results if r.price]
|
||||
if prices:
|
||||
unique_prices = set(prices)
|
||||
print(f"Unique prices: {sorted(unique_prices)}")
|
||||
print(f"Price range: {min(prices)}€ - {max(prices)}€")
|
||||
if len(unique_prices) > 1:
|
||||
print("✅ SUCCESS: Multiple different prices found!")
|
||||
else:
|
||||
print("⚠️ WARNING: All prices are the same")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,26 @@
|
||||
"""
|
||||
Test script to analyze Gifi HTML structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
print(f"Found: {result.title} - Price: {result.price}")
|
||||
|
||||
print(f"\nTotal results: {len(results)}")
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
Reference in new issue
Block a user