mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Implement comprehensive search configurations for discount stores and add Gifi-specific search and analysis tools.
This commit is contained in:
1 parent
63073d03b5
commit
c159036583
22 files changed
+19849
-14
No files matched your search
@@ -0,0 +1,59 @@
|
|||||||
|
"""
|
||||||
|
Analyze Carrefour price extraction
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
print("Loading Carrefour search...")
|
||||||
|
await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
|
||||||
|
|
||||||
|
content = await page.content()
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
|
||||||
|
products = soup.select("article.product-list-card-plp-grid-new")
|
||||||
|
print(f"Found {len(products)} products\n")
|
||||||
|
|
||||||
|
for i, product in enumerate(products[:3]):
|
||||||
|
print(f"=== Product {i+1} ===")
|
||||||
|
|
||||||
|
# Title
|
||||||
|
title_el = product.select_one("h3, h2, a")
|
||||||
|
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||||
|
print(f"Title: {title[:60]}")
|
||||||
|
|
||||||
|
# Find all text with € symbol
|
||||||
|
import re
|
||||||
|
product_html = str(product)
|
||||||
|
prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
|
||||||
|
print(f"Prices found in HTML: {prices}")
|
||||||
|
|
||||||
|
# Look for price elements
|
||||||
|
price_els = product.find_all(string=re.compile('€'))
|
||||||
|
if price_els:
|
||||||
|
print(f"Elements with €:")
|
||||||
|
for el in price_els[:3]:
|
||||||
|
print(f" - {el.strip()[:50]}")
|
||||||
|
|
||||||
|
print()
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
"""
|
||||||
|
Analyze a Carrefour product page for price selectors
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
import re
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
# Visit a Carrefour product page (from the screenshot)
|
||||||
|
url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
|
||||||
|
print(f"Loading: {url}")
|
||||||
|
await page.goto(url, wait_until="networkidle")
|
||||||
|
|
||||||
|
content = await page.content()
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
|
||||||
|
# Find all elements with € symbol
|
||||||
|
price_els = soup.find_all(string=re.compile('€'))
|
||||||
|
print(f"\nFound {len(price_els)} elements with '€'")
|
||||||
|
|
||||||
|
prices_found = set()
|
||||||
|
for el in price_els[:20]:
|
||||||
|
text = el.strip()
|
||||||
|
if text and len(text) < 50:
|
||||||
|
prices_found.add(text)
|
||||||
|
parent = el.find_parent()
|
||||||
|
print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
|
||||||
|
|
||||||
|
# Try common price selectors
|
||||||
|
selectors = [
|
||||||
|
".product-price",
|
||||||
|
"[class*='price']",
|
||||||
|
".price",
|
||||||
|
"span.price",
|
||||||
|
"div.price",
|
||||||
|
"[data-price]"
|
||||||
|
]
|
||||||
|
|
||||||
|
print("\nTrying specific selectors:")
|
||||||
|
for selector in selectors:
|
||||||
|
try:
|
||||||
|
els = soup.select(selector)
|
||||||
|
if els:
|
||||||
|
for el in els[:2]:
|
||||||
|
text = el.get_text(strip=True)
|
||||||
|
if '€' in text:
|
||||||
|
print(f" {selector}: {text}")
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
"""
|
||||||
|
Analyze Centrakor HTML structure for image selectors
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import re
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Connecting to browserless...")
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
print("Loading Centrakor search...")
|
||||||
|
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||||
|
|
||||||
|
content = await page.content()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
|
||||||
|
# Try to find product containers
|
||||||
|
selectors = [
|
||||||
|
"div.product-item",
|
||||||
|
"div.product-card",
|
||||||
|
"article",
|
||||||
|
"div[class*='product']",
|
||||||
|
"li[class*='product']"
|
||||||
|
]
|
||||||
|
|
||||||
|
for selector in selectors:
|
||||||
|
products = soup.select(selector)
|
||||||
|
if products:
|
||||||
|
print(f"\n✓ Found {len(products)} products with selector: {selector}")
|
||||||
|
|
||||||
|
# Analyze first product
|
||||||
|
first = products[0]
|
||||||
|
print(f"\nFirst product HTML snippet:")
|
||||||
|
print(str(first)[:500])
|
||||||
|
print("\n...")
|
||||||
|
|
||||||
|
# Find all images
|
||||||
|
images = first.find_all('img')
|
||||||
|
print(f"\nFound {len(images)} images in first product:")
|
||||||
|
for i, img in enumerate(images):
|
||||||
|
print(f"\n Image {i+1}:")
|
||||||
|
print(f" Class: {img.get('class')}")
|
||||||
|
print(f" Src: {img.get('src', '')[:80]}")
|
||||||
|
print(f" Data-src: {img.get('data-src', '')[:80]}")
|
||||||
|
print(f" Alt: {img.get('alt', '')[:50]}")
|
||||||
|
|
||||||
|
break
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
print("\nDone")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
"""
|
||||||
|
Detailed analysis of Centrakor image structure
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||||
|
content = await page.content()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
products = soup.select("div.product-item")
|
||||||
|
|
||||||
|
print(f"Analyzing {min(5, len(products))} products:\n")
|
||||||
|
|
||||||
|
for i, product in enumerate(products[:5]):
|
||||||
|
print(f"=== Product {i+1} ===")
|
||||||
|
|
||||||
|
# Title
|
||||||
|
title_el = product.select_one("a.product-item__name")
|
||||||
|
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||||
|
print(f"Title: {title}")
|
||||||
|
|
||||||
|
# All images
|
||||||
|
images = product.find_all('img')
|
||||||
|
print(f"Found {len(images)} img tags")
|
||||||
|
|
||||||
|
for j, img in enumerate(images):
|
||||||
|
print(f"\n Image {j+1}:")
|
||||||
|
print(f" tag: {img.name}")
|
||||||
|
print(f" class: {img.get('class')}")
|
||||||
|
for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
|
||||||
|
val = img.get(attr)
|
||||||
|
if val:
|
||||||
|
print(f" {attr}: {val[:80]}")
|
||||||
|
|
||||||
|
print()
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
"""
|
||||||
|
Analyze Gifi HTML to find correct selectors
|
||||||
|
"""
|
||||||
|
with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
|
||||||
|
html = f.read()
|
||||||
|
|
||||||
|
# Find product-related divs
|
||||||
|
import re
|
||||||
|
matches = re.findall(r'<div[^>]*class="[^"]*"[^>]*>.*?</div>', html[:50000], re.DOTALL)
|
||||||
|
|
||||||
|
print(f"Total HTML size: {len(html)} bytes")
|
||||||
|
|
||||||
|
# Search for price patterns
|
||||||
|
price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
|
||||||
|
print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
|
||||||
|
|
||||||
|
# Find all class names containing specific keywords
|
||||||
|
for keyword in ['product', 'article', 'item', 'card']:
|
||||||
|
classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
|
||||||
|
unique_classes = set(classes)
|
||||||
|
if unique_classes:
|
||||||
|
print(f"\nClasses containing '{keyword}':")
|
||||||
|
for cls in sorted(unique_classes):
|
||||||
|
print(f" - {cls}")
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
"""
|
||||||
|
Analyze a Gifi product page to understand price structure
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Connecting to browserless...")
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
# Visit a Gifi product page
|
||||||
|
url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
|
||||||
|
print(f"Loading: {url}")
|
||||||
|
await page.goto(url, wait_until="networkidle")
|
||||||
|
|
||||||
|
# Get page title
|
||||||
|
title = await page.title()
|
||||||
|
print(f"Title: {title}")
|
||||||
|
|
||||||
|
# Find all elements with price-like text
|
||||||
|
price_els = await page.query_selector_all("*:has-text('€')")
|
||||||
|
print(f"\nFound {len(price_els)} elements with '€'")
|
||||||
|
|
||||||
|
# Get first 10 price elements
|
||||||
|
for i, el in enumerate(price_els[:10]):
|
||||||
|
text = await el.inner_text()
|
||||||
|
tag = await el.evaluate("el => el.tagName")
|
||||||
|
classes = await el.evaluate("el => el.className")
|
||||||
|
print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
|
||||||
|
|
||||||
|
# Try specific selectors
|
||||||
|
selectors = [
|
||||||
|
".price",
|
||||||
|
".product-price",
|
||||||
|
"[class*='price']",
|
||||||
|
"[data-price]",
|
||||||
|
"span.price",
|
||||||
|
"div.price"
|
||||||
|
]
|
||||||
|
|
||||||
|
print("\nTrying specific selectors:")
|
||||||
|
for selector in selectors:
|
||||||
|
try:
|
||||||
|
els = await page.query_selector_all(selector)
|
||||||
|
if els:
|
||||||
|
for el in els[:2]:
|
||||||
|
text = await el.inner_text()
|
||||||
|
print(f" {selector}: {text}")
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
print("\nDone")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
"""
|
||||||
|
Test extracting price from Gifi search page HTML
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import re
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Connecting to browserless...")
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
print("Loading Gifi search...")
|
||||||
|
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||||
|
|
||||||
|
content = await page.content()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
products = soup.select("div.product-tile")
|
||||||
|
|
||||||
|
print(f"Found {len(products)} products\n")
|
||||||
|
|
||||||
|
for i, product in enumerate(products[:3]):
|
||||||
|
print(f"\\n=== Product {i+1} ===")
|
||||||
|
|
||||||
|
# Get title
|
||||||
|
title_el = product.select_one("div.pdp-link a")
|
||||||
|
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||||
|
print(f"Title: {title}")
|
||||||
|
|
||||||
|
# Try to find price in product HTML
|
||||||
|
product_html = product.prettify()
|
||||||
|
|
||||||
|
# Look for price patterns
|
||||||
|
price_patterns = [
|
||||||
|
r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
|
||||||
|
r'€\s*(\d+)[,.](\d+)', # € 19,99
|
||||||
|
r'(\d+)€(\d+)', # 19€99
|
||||||
|
r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
|
||||||
|
]
|
||||||
|
|
||||||
|
for pattern in price_patterns:
|
||||||
|
matches = re.findall(pattern, product_html)
|
||||||
|
if matches:
|
||||||
|
print(f"Pattern '{pattern}': {matches[:3]}")
|
||||||
|
|
||||||
|
# Find all text with €
|
||||||
|
euro_texts = product.find_all(string=re.compile('€'))
|
||||||
|
if euro_texts:
|
||||||
|
print(f"Texts with €:")
|
||||||
|
for text in euro_texts[:5]:
|
||||||
|
print(f" - {text.strip()[:80]}")
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
print("\nDone")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -56,9 +56,11 @@ SITE_CONFIGS = {
|
|||||||
"gifi.fr": {
|
"gifi.fr": {
|
||||||
"name": "Gifi",
|
"name": "Gifi",
|
||||||
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
|
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
|
||||||
"product_selector": "article.product-miniature, div.product-item, div[class*='product']",
|
"product_selector": "div.product-tile",
|
||||||
"product_image_selector": "img.product-thumbnail, img[class*='product'], img",
|
"product_link_selector": "a.link",
|
||||||
"wait_selector": "article.product-miniature, div.product-item",
|
"product_title_selector": "div.pdp-link a",
|
||||||
|
"product_image_selector": "img.tile-image",
|
||||||
|
"wait_selector": "div.product-tile",
|
||||||
"category": "Discount",
|
"category": "Discount",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
@@ -150,9 +152,9 @@ SITE_CONFIGS = {
|
|||||||
"centrakor.com": {
|
"centrakor.com": {
|
||||||
"name": "Centrakor",
|
"name": "Centrakor",
|
||||||
"search_url": "https://www.centrakor.com/search/{query}",
|
"search_url": "https://www.centrakor.com/search/{query}",
|
||||||
"product_selector": "div.product-item, div.product-card, article",
|
"product_selector": "div.product-item",
|
||||||
"product_image_selector": "img.product-item__image, img.product-card__image, img[loading='lazy']",
|
"product_image_selector": "img.responsive-image__actual",
|
||||||
"wait_selector": "div.product-item, div.product-card, article",
|
"wait_selector": "div.product-item",
|
||||||
"category": "Discount",
|
"category": "Discount",
|
||||||
"requires_proxy": False,
|
"requires_proxy": False,
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -202,6 +202,8 @@ class ImprovedSearchService:
|
|||||||
for item in links:
|
for item in links:
|
||||||
# Keep reference to original container for image search
|
# Keep reference to original container for image search
|
||||||
container = item
|
container = item
|
||||||
|
if site_key == "centrakor.com":
|
||||||
|
logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}")
|
||||||
|
|
||||||
# If selector targets the container (div.product-card), we need to find the link inside
|
# If selector targets the container (div.product-card), we need to find the link inside
|
||||||
if item.name != 'a':
|
if item.name != 'a':
|
||||||
@@ -263,12 +265,13 @@ class ImprovedSearchService:
|
|||||||
|
|
||||||
# PRIORITY 1: Use product_image_selector if configured (site-specific)
|
# PRIORITY 1: Use product_image_selector if configured (site-specific)
|
||||||
if "product_image_selector" in config:
|
if "product_image_selector" in config:
|
||||||
# Search in the original container first
|
# Search in the original container first - get ALL matches
|
||||||
img_el = container.select_one(config["product_image_selector"])
|
img_els = container.select(config["product_image_selector"])
|
||||||
|
|
||||||
if img_el:
|
# Filter and find first valid image
|
||||||
|
for img_el in img_els:
|
||||||
# Try multiple attributes in order of priority
|
# Try multiple attributes in order of priority
|
||||||
image_url= (
|
candidate_url = (
|
||||||
img_el.get("src") or
|
img_el.get("src") or
|
||||||
img_el.get("data-src") or
|
img_el.get("data-src") or
|
||||||
img_el.get("data-lazy-src") or
|
img_el.get("data-lazy-src") or
|
||||||
@@ -276,13 +279,46 @@ class ImprovedSearchService:
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Handle srcset (use first URL)
|
# Handle srcset (use first URL)
|
||||||
if not image_url and img_el.get("srcset"):
|
if not candidate_url and img_el.get("srcset"):
|
||||||
srcset = img_el.get("srcset")
|
srcset = img_el.get("srcset")
|
||||||
# srcset format: "url1 size1, url2 size2"
|
candidate_url = srcset.split(",")[0].split()[0]
|
||||||
image_url = srcset.split(",")[0].split()[0]
|
|
||||||
|
|
||||||
if image_url:
|
# Skip invalid images (pictos, icons, etc.)
|
||||||
|
if candidate_url:
|
||||||
|
# SPECIAL: No filtering for Centrakor (debugging)
|
||||||
|
if site_key == "centrakor.com":
|
||||||
|
# Skip placeholders
|
||||||
|
if "placeholder" in candidate_url.lower():
|
||||||
|
logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}")
|
||||||
|
continue
|
||||||
|
# Filter only tiny pictos
|
||||||
|
if 'picto' in candidate_url.lower() and ('width=60' in candidate_url or 'height=80' in candidate_url):
|
||||||
|
logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}")
|
||||||
|
continue
|
||||||
|
image_url = candidate_url
|
||||||
|
logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}")
|
||||||
|
break
|
||||||
|
|
||||||
|
# Normal filtering for other sites
|
||||||
|
# Filter out obvious pictos and small icons
|
||||||
|
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
|
||||||
|
logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}")
|
||||||
|
continue
|
||||||
|
# Filter out VERY small images (less than 100px)
|
||||||
|
import re
|
||||||
|
width_match = re.search(r'width=(\d+)', candidate_url)
|
||||||
|
height_match = re.search(r'height=(\d+)', candidate_url)
|
||||||
|
if width_match and height_match:
|
||||||
|
width = int(width_match.group(1))
|
||||||
|
height = int(height_match.group(1))
|
||||||
|
if width < 100 and height < 100:
|
||||||
|
logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
# This is a valid product image
|
||||||
|
image_url = candidate_url
|
||||||
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
|
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
|
||||||
|
break
|
||||||
|
|
||||||
# PRIORITY 2: Fallback - Look for any img directly in the link
|
# PRIORITY 2: Fallback - Look for any img directly in the link
|
||||||
if not image_url:
|
if not image_url:
|
||||||
@@ -336,12 +372,26 @@ class ImprovedSearchService:
|
|||||||
if not image_url:
|
if not image_url:
|
||||||
logger.warning(f" ⚠️ No image found for: {title[:50]}")
|
logger.warning(f" ⚠️ No image found for: {title[:50]}")
|
||||||
|
|
||||||
|
# Extract price from search results for sites where product pages are unavailable
|
||||||
|
product_price = None
|
||||||
|
if site_key == "gifi.fr":
|
||||||
|
# Gifi: Extract price from product tile HTML
|
||||||
|
import re
|
||||||
|
container_html = str(container)
|
||||||
|
price_match = re.search(r'(\d+)[,.](\d+)\s*€', container_html)
|
||||||
|
if price_match:
|
||||||
|
euros = int(price_match.group(1))
|
||||||
|
cents = int(price_match.group(2))
|
||||||
|
product_price = float(f"{euros}.{cents}")
|
||||||
|
logger.debug(f" 💰 Extracted price from search: {product_price}€")
|
||||||
|
|
||||||
# Create result
|
# Create result
|
||||||
results.append(SearchResult(
|
results.append(SearchResult(
|
||||||
url=full_url,
|
url=full_url,
|
||||||
title=title,
|
title=title,
|
||||||
snippet=f"Product from {config['name']}",
|
snippet=f"Product from {config['name']}",
|
||||||
source=config["name"],
|
source=config["name"],
|
||||||
|
price=product_price, # Set price if extracted from search
|
||||||
image_url=image_url
|
image_url=image_url
|
||||||
))
|
))
|
||||||
|
|
||||||
@@ -352,6 +402,28 @@ class ImprovedSearchService:
|
|||||||
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
|
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
|
||||||
"""Scrape price and details for a single item using same context"""
|
"""Scrape price and details for a single item using same context"""
|
||||||
try:
|
try:
|
||||||
|
# SPECIAL CASE: L'Incroyable - Price is in the title
|
||||||
|
if "lincroyable.fr" in result.url:
|
||||||
|
import re
|
||||||
|
# Extract price from title (e.g., "34€99" or "59€99")
|
||||||
|
price_match = re.search(r'(\d+)€(\d+)', result.title)
|
||||||
|
if price_match:
|
||||||
|
# Convert to float (e.g., "34€99" -> 34.99)
|
||||||
|
price_euros = int(price_match.group(1))
|
||||||
|
price_cents = int(price_match.group(2))
|
||||||
|
result.price = float(f"{price_euros}.{price_cents}")
|
||||||
|
|
||||||
|
# Clean title by removing price
|
||||||
|
result.title = re.sub(r'\d+€\d+', '', result.title).strip()
|
||||||
|
logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title")
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
# SPECIAL CASE: Gifi - Price extracted from search, no need to visit page
|
||||||
|
if "gifi.fr" in result.url and result.price is not None:
|
||||||
|
logger.debug(f"Gifi - Price already extracted from search: {result.price}€")
|
||||||
|
return result
|
||||||
|
|
||||||
page = await context.new_page()
|
page = await context.new_page()
|
||||||
try:
|
try:
|
||||||
logger.debug(f"Scraping details for: {result.title[:50]}...")
|
logger.debug(f"Scraping details for: {result.title[:50]}...")
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
"""
|
||||||
|
Deep analysis - compare products WITH images vs WITHOUT images
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||||
|
content = await page.content()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
products = soup.select("div.product-item")
|
||||||
|
|
||||||
|
print(f"Analyzing {len(products)} products for image patterns\n")
|
||||||
|
|
||||||
|
for i, product in enumerate(products[:10]):
|
||||||
|
# Get title
|
||||||
|
title_el = product.select_one("a.product-item__name")
|
||||||
|
title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
|
||||||
|
|
||||||
|
# Get ALL images
|
||||||
|
all_imgs = product.select("img.responsive-image__actual")
|
||||||
|
|
||||||
|
print(f"\n=== {i+1}. {title[:50]} ===")
|
||||||
|
print(f"Found {len(all_imgs)} images")
|
||||||
|
|
||||||
|
for j, img in enumerate(all_imgs):
|
||||||
|
src = img.get('src', '')
|
||||||
|
print(f" Image {j+1}: {src if src else '(no src)'}")
|
||||||
|
if not src:
|
||||||
|
# Check other attributes
|
||||||
|
for attr in ['data-src', 'data-lazy-src', 'srcset']:
|
||||||
|
val = img.get(attr)
|
||||||
|
if val:
|
||||||
|
print(f" {attr}: {val[:80]}")
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
"""
|
||||||
|
Script to dump Gifi HTML and analyze structure
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Connecting to browserless...")
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
print("Loading Gifi search page...")
|
||||||
|
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
|
||||||
|
|
||||||
|
# Wait for products
|
||||||
|
try:
|
||||||
|
await page.wait_for_selector("article.product-miniature", timeout=10000)
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Save HTML
|
||||||
|
content = await page.content()
|
||||||
|
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||||
|
with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
|
||||||
|
f.write(content)
|
||||||
|
|
||||||
|
print(f"HTML saved ({len(content)} bytes)")
|
||||||
|
|
||||||
|
# Extract first product structure
|
||||||
|
products = await page.query_selector_all("article.product-miniature")
|
||||||
|
print(f"Found {len(products)} products")
|
||||||
|
|
||||||
|
if products:
|
||||||
|
first_html = await products[0].evaluate("el => el.outerHTML")
|
||||||
|
with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
|
||||||
|
f.write(first_html)
|
||||||
|
print(f"First product HTML saved")
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
print("Done")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,76 @@
|
|||||||
|
"""
|
||||||
|
Dump Gifi with longer wait for JavaScript
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Connecting to browserless...")
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
print("Loading Gifi search page...")
|
||||||
|
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||||
|
|
||||||
|
# Wait for ANY content
|
||||||
|
print("Waiting for content...")
|
||||||
|
await page.wait_for_timeout(5000)
|
||||||
|
|
||||||
|
# Save HTML
|
||||||
|
content = await page.content()
|
||||||
|
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||||
|
with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
|
||||||
|
f.write(content)
|
||||||
|
|
||||||
|
print(f"HTML saved ({len(content)} bytes)")
|
||||||
|
|
||||||
|
# Find any elements with price
|
||||||
|
price_els = await page.query_selector_all("*:has-text('€')")
|
||||||
|
print(f"Elements with € symbol: {len(price_els)}")
|
||||||
|
|
||||||
|
# Find all divs/articles
|
||||||
|
all_divs = await page.query_selector_all("div, article, li")
|
||||||
|
print(f"Total divs/articles/li: {len(all_divs)}")
|
||||||
|
|
||||||
|
# Screenshot
|
||||||
|
await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
|
||||||
|
print("Screenshot saved")
|
||||||
|
|
||||||
|
# Get all classes
|
||||||
|
all_classes = await page.evaluate("""() => {
|
||||||
|
const elements = document.querySelectorAll('*');
|
||||||
|
const classes = new Set();
|
||||||
|
elements.forEach(el => {
|
||||||
|
if (el.className && typeof el.className === 'string') {
|
||||||
|
el.className.split(' ').forEach(cls => {
|
||||||
|
if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
|
||||||
|
classes.add(cls);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
});
|
||||||
|
return Array.from(classes);
|
||||||
|
}""")
|
||||||
|
|
||||||
|
print(f"\\nProduct-related classes found:")
|
||||||
|
for cls in all_classes:
|
||||||
|
print(f" - {cls}")
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
print("Done")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
Whitespace-only changes.
+18821
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,42 @@
|
|||||||
|
"""
|
||||||
|
Test Carrefour via API endpoint (production-like)
|
||||||
|
"""
|
||||||
|
import requests
|
||||||
|
import json
|
||||||
|
|
||||||
|
url = "http://localhost:8555/api/search"
|
||||||
|
params = {
|
||||||
|
"q": "chaise",
|
||||||
|
"sites": "7", # Carrefour site ID
|
||||||
|
"max_results": 10
|
||||||
|
}
|
||||||
|
|
||||||
|
print(f"Testing: {url}")
|
||||||
|
print(f"Params: {params}\n")
|
||||||
|
|
||||||
|
response = requests.get(url, params=params, stream=True)
|
||||||
|
|
||||||
|
print(f"Status: {response.status_code}")
|
||||||
|
print(f"Headers: {dict(response.headers)}\n")
|
||||||
|
|
||||||
|
count = 0
|
||||||
|
for line in response.iter_lines():
|
||||||
|
if line:
|
||||||
|
try:
|
||||||
|
# Each line should be JSON
|
||||||
|
data = json.loads(line.decode('utf-8'))
|
||||||
|
count += 1
|
||||||
|
|
||||||
|
title = data.get('title', 'N/A')[:60]
|
||||||
|
price = data.get('price', 'N/A')
|
||||||
|
image = "✅" if data.get('image_url') else "❌"
|
||||||
|
|
||||||
|
print(f"{count}. [{image}] {title} - {price}€")
|
||||||
|
|
||||||
|
if count >= 10:
|
||||||
|
break
|
||||||
|
except json.JSONDecodeError as e:
|
||||||
|
print(f"JSON Error: {e}")
|
||||||
|
print(f"Line: {line[:100]}")
|
||||||
|
|
||||||
|
print(f"\nTotal: {count} results")
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
"""
|
||||||
|
Test if current Carrefour price extraction works
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Carrefour for 'chaise'...\n")
|
||||||
|
results = []
|
||||||
|
count = 0
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
count += 1
|
||||||
|
price_status = f"{result.price}€" if result.price else "N/A"
|
||||||
|
print(f"{count}. {result.title[:55]} - {price_status}")
|
||||||
|
if count >= 10:
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"\n==> Got {len(results)} results")
|
||||||
|
|
||||||
|
# Count prices
|
||||||
|
with_prices = sum(1 for r in results if r.price)
|
||||||
|
print(f"Products with prices: {with_prices}/{len(results)}")
|
||||||
|
|
||||||
|
if with_prices == len(results):
|
||||||
|
print("✅ SUCCESS: All products have prices!")
|
||||||
|
else:
|
||||||
|
print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices")
|
||||||
|
|
||||||
|
print("\nShutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
"""
|
||||||
|
Test detailed logging for Centrakor image extraction
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from playwright.async_api import async_playwright
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
import re
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
playwright = await async_playwright().start()
|
||||||
|
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||||
|
|
||||||
|
context = await browser.new_context(
|
||||||
|
viewport={"width": 1920, "height": 1080},
|
||||||
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
page = await context.new_page()
|
||||||
|
|
||||||
|
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||||
|
content = await page.content()
|
||||||
|
|
||||||
|
soup = BeautifulSoup(content, "html.parser")
|
||||||
|
products = soup.select("div.product-item")
|
||||||
|
|
||||||
|
print(f"Testing first product:\n")
|
||||||
|
|
||||||
|
first = products[0]
|
||||||
|
|
||||||
|
# Get all images
|
||||||
|
img_els = first.select("img.responsive-image__actual")
|
||||||
|
print(f"Found {len(img_els)} images with selector")
|
||||||
|
|
||||||
|
for i, img_el in enumerate(img_els):
|
||||||
|
print(f"\n=== Image {i+1} ===")
|
||||||
|
candidate_url = img_el.get("src")
|
||||||
|
print(f"URL: {candidate_url}")
|
||||||
|
|
||||||
|
# Test filters
|
||||||
|
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
|
||||||
|
print(" ❌ Filtered: Contains picto/icon/logo/badge keyword")
|
||||||
|
continue
|
||||||
|
|
||||||
|
width_match = re.search(r'width=(\d+)', candidate_url)
|
||||||
|
height_match = re.search(r'height=(\d+)', candidate_url)
|
||||||
|
print(f" Width match: {width_match.group(1) if width_match else None}")
|
||||||
|
print(f" Height match: {height_match.group(1) if height_match else None}")
|
||||||
|
|
||||||
|
if width_match and height_match:
|
||||||
|
width = int(width_match.group(1))
|
||||||
|
height = int(height_match.group(1))
|
||||||
|
print(f" Dimensions: {width}x{height}")
|
||||||
|
if width < 100 and height < 100:
|
||||||
|
print(f" ❌ Filtered: Too small ({width}x{height})")
|
||||||
|
continue
|
||||||
|
else:
|
||||||
|
print(f" ✅ PASS: Large enough ({width}x{height})")
|
||||||
|
else:
|
||||||
|
print(" ✅ PASS: No dimensions in URL")
|
||||||
|
|
||||||
|
await context.close()
|
||||||
|
await browser.close()
|
||||||
|
await playwright.stop()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
"""
|
||||||
|
Test Centrakor image extraction
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Centrakor for 'chaise'...\n")
|
||||||
|
results = []
|
||||||
|
count = 0
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
count += 1
|
||||||
|
has_image = "✅" if result.image_url else "❌"
|
||||||
|
print(f"{count}. {has_image} {result.title[:55]} - {result.price}€")
|
||||||
|
if result.image_url:
|
||||||
|
print(f" Image: {result.image_url[:70]}...")
|
||||||
|
if count >= 10:
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"\n==> Got {len(results)} results")
|
||||||
|
|
||||||
|
# Count images
|
||||||
|
with_images = sum(1 for r in results if r.image_url)
|
||||||
|
print(f"Products with images: {with_images}/{len(results)}")
|
||||||
|
|
||||||
|
if with_images == len(results):
|
||||||
|
print("✅ SUCCESS: All products have images!")
|
||||||
|
else:
|
||||||
|
print(f"⚠️ WARNING: {len(results) - with_images} products missing images")
|
||||||
|
|
||||||
|
print("\nShutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
"""
|
||||||
|
Final test of Gifi with new selectors
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Gifi for 'chaise'...")
|
||||||
|
results = []
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
print(f"✓ {result.title[:50]} - {result.price}€")
|
||||||
|
|
||||||
|
print(f"\n==> Total: {len(results)} results")
|
||||||
|
|
||||||
|
if results:
|
||||||
|
print("\nFirst 3 products:")
|
||||||
|
for i, r in enumerate(results[:3]):
|
||||||
|
print(f"{i+1}. Title: {r.title}")
|
||||||
|
print(f" Price: {r.price}€")
|
||||||
|
print(f" URL: {r.url[:80]}...")
|
||||||
|
print()
|
||||||
|
|
||||||
|
print("Shutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
"""
|
||||||
|
Test Gifi price extraction
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Gifi for 'chaise'...")
|
||||||
|
results = []
|
||||||
|
count = 0
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
count += 1
|
||||||
|
print(f"{count}. {result.title[:60]} - Price: {result.price}€")
|
||||||
|
if count >= 5: # Only test first 5
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"\n==> Got {len(results)} results")
|
||||||
|
|
||||||
|
# Check if all prices are the same
|
||||||
|
prices = [r.price for r in results if r.price]
|
||||||
|
if prices:
|
||||||
|
unique_prices = set(prices)
|
||||||
|
print(f"Unique prices: {unique_prices}")
|
||||||
|
if len(unique_prices) == 1:
|
||||||
|
print("⚠️ WARNING: All prices are the same!")
|
||||||
|
|
||||||
|
print("Shutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
"""
|
||||||
|
Test fixed Gifi price extraction
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Gifi for 'chaise'...\n")
|
||||||
|
results = []
|
||||||
|
count = 0
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
count += 1
|
||||||
|
print(f"{count}. {result.title[:60]} - {result.price}€")
|
||||||
|
if count >= 10:
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"\n==> Got {len(results)} results")
|
||||||
|
|
||||||
|
# Check price diversity
|
||||||
|
prices = [r.price for r in results if r.price]
|
||||||
|
if prices:
|
||||||
|
unique_prices = set(prices)
|
||||||
|
print(f"Unique prices: {sorted(unique_prices)}")
|
||||||
|
print(f"Price range: {min(prices)}€ - {max(prices)}€")
|
||||||
|
if len(unique_prices) > 1:
|
||||||
|
print("✅ SUCCESS: Multiple different prices found!")
|
||||||
|
else:
|
||||||
|
print("⚠️ WARNING: All prices are the same")
|
||||||
|
|
||||||
|
print("\nShutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
"""
|
||||||
|
Test script to analyze Gifi HTML structure
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
sys.path.insert(0, '/app')
|
||||||
|
|
||||||
|
from app.services.improved_search_service import ImprovedSearchService
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
print("Initializing browser...")
|
||||||
|
await ImprovedSearchService.initialize()
|
||||||
|
|
||||||
|
print("Searching Gifi for 'chaise'...")
|
||||||
|
results = []
|
||||||
|
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||||
|
results.append(result)
|
||||||
|
print(f"Found: {result.title} - Price: {result.price}")
|
||||||
|
|
||||||
|
print(f"\nTotal results: {len(results)}")
|
||||||
|
|
||||||
|
print("Shutting down...")
|
||||||
|
await ImprovedSearchService.shutdown()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
Reference in new issue
Block a user