mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
14 files changed
+302
-131
No files matched your search
@@ -65,6 +65,7 @@ COPY alembic.ini ./
|
||||
COPY docker-entrypoint.sh ./
|
||||
|
||||
# Make entrypoint executable
|
||||
RUN sed -i 's/\r$//' docker-entrypoint.sh
|
||||
RUN chmod +x docker-entrypoint.sh
|
||||
|
||||
# Copy built frontend static files
|
||||
|
||||
+12
-12
@@ -65,9 +65,9 @@ SITE_CONFIGS = {
|
||||
"stokomani.fr": {
|
||||
"name": "Stokomani",
|
||||
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
|
||||
"product_selector": "a.reversed-link.block, a[href*='/products/']",
|
||||
"product_image_selector": "img[loading='lazy'], img[class*='object'], img[sizes]",
|
||||
"wait_selector": "a.reversed-link.block, .product-card",
|
||||
"product_selector": "div.product-card",
|
||||
"product_image_selector": "div.media-wrapper img, img[class*='product-card__image']",
|
||||
"wait_selector": "div.product-card",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
@@ -130,9 +130,9 @@ SITE_CONFIGS = {
|
||||
"lincroyable.fr": {
|
||||
"name": "L'Incroyable",
|
||||
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
|
||||
"product_selector": "div.product-card, div.product-miniature, article",
|
||||
"product_image_selector": "img.product-image, img[class*='product'], picture img",
|
||||
"wait_selector": "div.product-card, div.product-miniature, article",
|
||||
"product_selector": "div.tailleBlocProdNew",
|
||||
"product_image_selector": "img.imgCoup2coeur, img[class*='product']",
|
||||
"wait_selector": "div.tailleBlocProdNew",
|
||||
"category": "Discount",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
@@ -157,18 +157,18 @@ SITE_CONFIGS = {
|
||||
"auchan.fr": {
|
||||
"name": "Auchan",
|
||||
"search_url": "https://www.auchan.fr/recherche?text={query}",
|
||||
"product_selector": "article, div[class*='product-card'], div[class*='list__item']",
|
||||
"product_image_selector": "img[class*='product'], img[src*='auchan']",
|
||||
"wait_selector": "article, div[class*='product-card'], div[class*='list__item']",
|
||||
"product_selector": "article.product-thumbnail a.product-thumbnail__details-wrapper, div[class*='product-card'] a",
|
||||
"product_image_selector": ".product-thumbnail__picture img, img[class*='product']",
|
||||
"wait_selector": "article.product-thumbnail, div[class*='product-card']",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
"carrefour.fr": {
|
||||
"name": "Carrefour",
|
||||
"search_url": "https://www.carrefour.fr/s?q={query}",
|
||||
"product_selector": "a[href*='/p/'], a[href*='/produit'], article a, div[class*='product'] a",
|
||||
"product_image_selector": "img, picture source",
|
||||
"wait_selector": None,
|
||||
"product_selector": "article.product-list-card-plp-grid-new",
|
||||
"product_image_selector": "img.product-card-image-new__content",
|
||||
"wait_selector": "article.product-list-card-plp-grid-new",
|
||||
"category": "Grande Surface",
|
||||
"requires_proxy": False,
|
||||
},
|
||||
|
||||
@@ -14,9 +14,9 @@ class AuchanParser(BaseParser):
|
||||
|
||||
# Auchan products - Ultra Robust Strategy
|
||||
# The DOM is flat and dynamic. We rely on finding product links first.
|
||||
# Links usually contain '/p-' in the href.
|
||||
# Links usually contain '/p-' or '/pr-' in the href.
|
||||
|
||||
links = soup.select("a[href*='/p-']")
|
||||
links = soup.select("a[href*='/p-'], a[href*='/pr-']")
|
||||
seen_urls = set()
|
||||
|
||||
for link in links:
|
||||
@@ -26,7 +26,7 @@ class AuchanParser(BaseParser):
|
||||
continue
|
||||
|
||||
# Filter out non-product links if any (e.g. facets)
|
||||
if '/p-' not in href:
|
||||
if '/p-' not in href and '/pr-' not in href:
|
||||
continue
|
||||
|
||||
seen_urls.add(href)
|
||||
@@ -50,7 +50,7 @@ class AuchanParser(BaseParser):
|
||||
# Title
|
||||
title = link.get('title')
|
||||
if not title:
|
||||
title_el = container.select_one("h3, div[class*='title'], span[class*='title']")
|
||||
title_el = container.select_one("p.product-thumbnail__description, h3, div[class*='title'], span[class*='title']")
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
if not title:
|
||||
@@ -61,7 +61,7 @@ class AuchanParser(BaseParser):
|
||||
|
||||
# Image
|
||||
img_url = None
|
||||
img_el = container.select_one("img")
|
||||
img_el = container.select_one(".product-thumbnail__picture img, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
@@ -69,7 +69,7 @@ class AuchanParser(BaseParser):
|
||||
|
||||
# Price
|
||||
price = None
|
||||
price_el = container.select_one("div[class*='price'], span[class*='price'], .product-price")
|
||||
price_el = container.select_one("div.product-price, div[class*='price'], span[class*='price'], .product-price")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
|
||||
@@ -13,15 +13,17 @@ class CarrefourParser(BaseParser):
|
||||
results = []
|
||||
|
||||
# Carrefour products
|
||||
# New container: div containing both image and title link
|
||||
cards = soup.select("div.product-list-card-plp-grid-new, article, div[class*='product-card']")
|
||||
# New container: article.large-horizontal or div.product-list-card-plp-grid-new
|
||||
cards = soup.select("article.large-horizontal, div.product-list-card-plp-grid-new, article, div[class*='product-card']")
|
||||
|
||||
# If no cards, try finding by link
|
||||
if not cards:
|
||||
links = soup.select("a.c-link.product-card-click-wrapper, a[href*='/p/'], a[href*='/produit']")
|
||||
cards = []
|
||||
for link in links:
|
||||
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
parent = link.find_parent("article")
|
||||
if not parent:
|
||||
parent = link.find_parent("div", class_=lambda x: x and "product" in x)
|
||||
if parent:
|
||||
cards.append(parent)
|
||||
else:
|
||||
@@ -36,14 +38,14 @@ class CarrefourParser(BaseParser):
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = card.select_one("h3, h2, [class*='title']")
|
||||
title_el = card.select_one("h3.product-card-title__text, h3, h2, [class*='title']")
|
||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = None
|
||||
img_el = card.select_one("img")
|
||||
img_el = card.select_one("img.product-card-image-new__content, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
@@ -51,7 +53,7 @@ class CarrefourParser(BaseParser):
|
||||
|
||||
price = None
|
||||
# Price is often text node near h3 or in a specific price element
|
||||
price_el = card.select_one("[class*='price'], .product-card-price, span[class*='amount']")
|
||||
price_el = card.select_one("div.product-price__amount--main, [class*='price'], .product-card-price, span[class*='amount']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
|
||||
@@ -13,51 +13,39 @@ class LIncroyableParser(BaseParser):
|
||||
results = []
|
||||
|
||||
# L'Incroyable products
|
||||
# Config: a.product-link
|
||||
# Config: div.tailleBlocProdNew
|
||||
|
||||
# Try finding cards first
|
||||
cards = soup.select("div.product-card, div.product-miniature, article")
|
||||
cards = soup.select("div.tailleBlocProdNew")
|
||||
|
||||
# If no cards found, try finding by link and getting parent
|
||||
if not cards:
|
||||
links = soup.select("a[href*='/p/'], a[href*='/produit/']")
|
||||
cards = []
|
||||
for link in links:
|
||||
# Try to find a container div
|
||||
parent = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x))
|
||||
if parent:
|
||||
cards.append(parent)
|
||||
else:
|
||||
cards.append(link.parent) # Fallback to immediate parent
|
||||
|
||||
for card in cards:
|
||||
try:
|
||||
link_el = card.select_one("a[href*='/p/'], a[href*='/produit/'], a.product-link")
|
||||
# Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links)
|
||||
# We look for the main link to the product
|
||||
link_el = card.select_one("a[href*='/p']")
|
||||
if not link_el:
|
||||
# If card is the link itself
|
||||
if card.name == 'a' and card.get('href'):
|
||||
link_el = card
|
||||
else:
|
||||
continue
|
||||
continue
|
||||
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title_el = card.select_one(".product-title, h3, h2, [class*='title']")
|
||||
# Title
|
||||
title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew")
|
||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image
|
||||
img_url = None
|
||||
img_el = card.select_one("img")
|
||||
img_el = card.select_one("img.imgCoup2coeur, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
price_el = card.select_one(".price, .product-price, [class*='price']")
|
||||
price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
|
||||
@@ -13,63 +13,47 @@ class StokomaniParser(BaseParser):
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Find product title links
|
||||
links = soup.select("a.reversed-link.block")
|
||||
# Stokomani products
|
||||
# Config: div.product-card
|
||||
|
||||
seen_urls = set()
|
||||
cards = soup.select("div.product-card")
|
||||
|
||||
for link in links:
|
||||
for card in cards:
|
||||
try:
|
||||
href = link.get('href')
|
||||
if not href or href in seen_urls:
|
||||
continue
|
||||
|
||||
if '/products/' not in href:
|
||||
# Link
|
||||
link_el = card.select_one("h3.product-card__title a, a[href*='/products/']")
|
||||
if not link_el:
|
||||
continue
|
||||
|
||||
seen_urls.add(href)
|
||||
href = link_el.get('href')
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
title = link.get_text(strip=True)
|
||||
# Title
|
||||
title_el = card.select_one("span.reversed-link__text, h3.product-card__title")
|
||||
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image: Look for the preceding <a> with aria-label
|
||||
# Image
|
||||
img_url = None
|
||||
prev_a = link.find_previous_sibling("a", attrs={"aria-label": True})
|
||||
if prev_a and prev_a.get('href') == href:
|
||||
# Try specific selector first
|
||||
img_el = prev_a.select_one("motion-element img, img")
|
||||
if img_el:
|
||||
# Check for lazy loading attributes explicitly
|
||||
img_url = img_el.get('data-src') or img_el.get('data-srcset') or img_el.get('srcset') or img_el.get('src')
|
||||
if img_url and " " in img_url:
|
||||
# Handle srcset: take the first URL
|
||||
img_url = img_url.split(" ")[0]
|
||||
|
||||
if not img_url:
|
||||
img_url = self.extract_image_url(prev_a)
|
||||
img_el = card.select_one("div.media-wrapper img, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
# Price: Look for text node after the link
|
||||
# Price
|
||||
price = None
|
||||
next_sibling = link.next_sibling
|
||||
while next_sibling:
|
||||
if isinstance(next_sibling, NavigableString):
|
||||
price_text = next_sibling.strip()
|
||||
if "€" in price_text:
|
||||
price = self.parse_price_text(price_text)
|
||||
if price:
|
||||
break
|
||||
elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])):
|
||||
# Try finding price in next div if it's a price container
|
||||
price = self.parse_price_text(next_sibling.get_text())
|
||||
if price:
|
||||
break
|
||||
next_sibling = next_sibling.next_sibling
|
||||
|
||||
price_el = card.select_one("span.f-price-item--regular, .f-price-item, [class*='price']")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
# Fallback price search in text
|
||||
if not price:
|
||||
text = card.get_text(" ", strip=True)
|
||||
price = self.parse_price_text(text)
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
|
||||
@@ -108,21 +108,45 @@ class NewSearchService:
|
||||
return result
|
||||
|
||||
# Use AI to analyze
|
||||
from app.services.ai_service import AIService
|
||||
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
|
||||
try:
|
||||
from app.services.ai_service import AIService
|
||||
# Check if AI service is available/configured before calling?
|
||||
# For now, just try/except the call
|
||||
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
|
||||
|
||||
if ai_result:
|
||||
extraction, _ = ai_result
|
||||
result.price = extraction.price
|
||||
result.currency = extraction.currency or "EUR"
|
||||
result.in_stock = extraction.in_stock
|
||||
|
||||
# Update image URL to point to our local screenshot
|
||||
import os
|
||||
filename = os.path.basename(screenshot_path)
|
||||
result.image_url = f"/screenshots/{filename}"
|
||||
else:
|
||||
raise Exception("AI returned no result")
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"AI Analysis failed for {result.url}: {e}")
|
||||
# Fallback: Try to extract price from page_text if Browserless found it
|
||||
if page_text and "PRIX DÉTECTÉ:" in page_text:
|
||||
try:
|
||||
import re
|
||||
price_match = re.search(r"PRIX DÉTECTÉ:\s*([\d\.]+)", page_text)
|
||||
if price_match:
|
||||
price_val = float(price_match.group(1))
|
||||
result.price = price_val
|
||||
logger.info(f"💰 Fallback: Extracted price {price_val} from text for {result.url}")
|
||||
except Exception as parse_e:
|
||||
logger.error(f"Error parsing fallback price: {parse_e}")
|
||||
|
||||
# Still use the screenshot if we have it
|
||||
if screenshot_path:
|
||||
import os
|
||||
filename = os.path.basename(screenshot_path)
|
||||
result.image_url = f"/screenshots/{filename}"
|
||||
|
||||
if ai_result:
|
||||
extraction, _ = ai_result
|
||||
result.price = extraction.price
|
||||
result.currency = extraction.currency or "EUR"
|
||||
result.in_stock = extraction.in_stock
|
||||
|
||||
# Update image URL to point to our local screenshot
|
||||
# The frontend expects /screenshots/filename
|
||||
import os
|
||||
filename = os.path.basename(screenshot_path)
|
||||
result.image_url = f"/screenshots/{filename}"
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error scraping item {result.url}: {e}")
|
||||
|
||||
@@ -191,6 +215,17 @@ class NewSearchService:
|
||||
# break
|
||||
|
||||
href = link.get("href")
|
||||
|
||||
# Special handling for sites where selector targets a container (Carrefour, Stokomani)
|
||||
if not href and config.get("name") in ["Carrefour", "Stokomani"]:
|
||||
# Try to find the main product link inside the container
|
||||
# For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one
|
||||
child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a")
|
||||
if child_link:
|
||||
href = child_link.get("href")
|
||||
# Update link to point to the anchor for title/image extraction
|
||||
link = child_link
|
||||
|
||||
if not href:
|
||||
continue
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@ echo "Setting up screenshots directory..."
|
||||
mkdir -p screenshots
|
||||
|
||||
echo "Running database migrations..."
|
||||
alembic upgrade heads
|
||||
# alembic upgrade heads
|
||||
|
||||
echo "Starting application..."
|
||||
exec uvicorn app.main:app --host 0.0.0.0 --port 8555
|
||||
+27
-22
@@ -6,7 +6,7 @@ import sys
|
||||
from pathlib import Path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
from app.services.browserless_service import browserless_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
async def dump_search_html(site_key: str, query: str = "chaise"):
|
||||
@@ -19,31 +19,37 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
|
||||
print(f"\n🔍 Dumping HTML for: {config['name']}")
|
||||
|
||||
# Ensure browser is initialized
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
# Create context manually to get HTML
|
||||
context = await ImprovedSearchService._create_context(ImprovedSearchService._browser)
|
||||
page = await context.new_page()
|
||||
await browserless_service.initialize()
|
||||
|
||||
try:
|
||||
search_url = config["search_url"].format(query=query)
|
||||
print(f" URL: {search_url}")
|
||||
|
||||
await page.goto(search_url, wait_until="networkidle", timeout=30000)
|
||||
await ImprovedSearchService._handle_popups(page)
|
||||
await page.wait_for_timeout(3000)
|
||||
# Override wait_selector for La Foir'Fouille
|
||||
wait_selector = config.get("wait_selector")
|
||||
if site_key == "lafoirfouille.fr":
|
||||
wait_selector = ".sf-grid-vignet"
|
||||
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
|
||||
|
||||
html = await page.content()
|
||||
html_content, screenshot_path = await browserless_service.get_page_content(
|
||||
search_url,
|
||||
wait_selector=wait_selector,
|
||||
use_proxy=config.get("requires_proxy", False)
|
||||
)
|
||||
|
||||
if not html_content:
|
||||
print(" ❌ No HTML content returned")
|
||||
return
|
||||
|
||||
filename = f"dump_{site_key.replace('.', '_')}.html"
|
||||
with open(filename, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
f.write(html_content)
|
||||
|
||||
print(f" ✅ Saved to: {filename} ({len(html)} bytes)")
|
||||
print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
|
||||
|
||||
# Quick analysis
|
||||
from bs4 import BeautifulSoup
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
|
||||
# Try current selector
|
||||
current_selector = config.get("product_selector")
|
||||
@@ -55,19 +61,18 @@ async def dump_search_html(site_key: str, query: str = "chaise"):
|
||||
img_selector = config["product_image_selector"]
|
||||
img_matches = soup.select(img_selector)
|
||||
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
|
||||
|
||||
except Exception as e:
|
||||
print(f" ❌ Error during dump: {e}")
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
# We don't close the browser here to allow reuse if needed,
|
||||
# but main() will shut it down.
|
||||
pass
|
||||
|
||||
async def main():
|
||||
sites = [
|
||||
"e-leclerc.com",
|
||||
"auchan.fr",
|
||||
"carrefour.fr",
|
||||
"stokomani.fr",
|
||||
"centrakor.com",
|
||||
"cdiscount.com",
|
||||
"lincroyable.fr"
|
||||
"stokomani.fr"
|
||||
]
|
||||
|
||||
for site_key in sites:
|
||||
@@ -77,7 +82,7 @@ async def main():
|
||||
print(f"❌ Error: {e}")
|
||||
await asyncio.sleep(1)
|
||||
|
||||
await ImprovedSearchService.shutdown()
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,33 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
articles = soup.select("article.product-list-card-plp-grid-new")
|
||||
|
||||
print(f"Found {len(articles)} articles")
|
||||
|
||||
if articles:
|
||||
first = articles[0]
|
||||
print("\n--- First Article Structure ---")
|
||||
print(first.prettify()[:1000]) # Print first 1000 chars
|
||||
|
||||
# Check for link
|
||||
link = first.select_one("a.product-card-click-wrapper")
|
||||
if link:
|
||||
print(f"\nLink found: {link.get('href')}")
|
||||
print(f"Link classes: {link.get('class')}")
|
||||
|
||||
# Check for image INSIDE link
|
||||
img = link.select_one("img.product-card-image-new__content")
|
||||
if img:
|
||||
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
|
||||
else:
|
||||
print(f"\n❌ Image NOT found inside link")
|
||||
# Check if image is elsewhere in article
|
||||
img_article = first.select_one("img.product-card-image-new__content")
|
||||
if img_article:
|
||||
print(f" But image exists in article: {img_article.get('src')}")
|
||||
else:
|
||||
print("\nNo link found with selector a.product-card-click-wrapper")
|
||||
@@ -0,0 +1,53 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
# Try to find product containers
|
||||
print("Searching for product containers...")
|
||||
potential_selectors = [
|
||||
"div.product-miniature",
|
||||
"article",
|
||||
"div[class*='product']",
|
||||
"div.product-card",
|
||||
"div.item"
|
||||
]
|
||||
|
||||
for selector in potential_selectors:
|
||||
matches = soup.select(selector)
|
||||
print(f"Selector '{selector}' matches: {len(matches)}")
|
||||
if len(matches) > 0 and len(matches) < 5:
|
||||
# If few matches, print classes to see if it's a wrapper
|
||||
print(f" Classes: {matches[0].get('class')}")
|
||||
|
||||
# Print structure of first potential product
|
||||
products = soup.select("div.product-miniature")
|
||||
if not products:
|
||||
products = soup.select("div[class*='product-item']")
|
||||
|
||||
if products:
|
||||
first = products[0]
|
||||
print("\n--- First Product Structure ---")
|
||||
print(first.prettify()[:1000])
|
||||
|
||||
link = first.find("a")
|
||||
if link:
|
||||
print(f"\nLink found: {link.get('href')}")
|
||||
|
||||
img = first.find("img")
|
||||
if img:
|
||||
print(f"\nImage found: {img.get('src')}")
|
||||
else:
|
||||
print("\nNo obvious products found. Dumping generic structure...")
|
||||
# Find any div with many children
|
||||
divs = soup.find_all("div")
|
||||
for div in divs:
|
||||
if len(div.find_all("div", recursive=False)) > 10:
|
||||
print(f"Found container with many children: {div.get('class')}")
|
||||
# Print first child
|
||||
child = div.find("div")
|
||||
if child:
|
||||
print(child.prettify()[:500])
|
||||
break
|
||||
@@ -0,0 +1,25 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
images = soup.select("img")
|
||||
|
||||
print(f"Found {len(images)} images")
|
||||
|
||||
for i, img in enumerate(images[:10]):
|
||||
print(f"\n--- Image {i+1} ---")
|
||||
print(f"Src: {img.get('src')}")
|
||||
print(f"Classes: {img.get('class')}")
|
||||
|
||||
parent = img.parent
|
||||
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
|
||||
|
||||
grandparent = parent.parent
|
||||
if grandparent:
|
||||
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
|
||||
|
||||
greatgrandparent = grandparent.parent
|
||||
if greatgrandparent:
|
||||
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
|
||||
+2
-2
@@ -80,7 +80,7 @@ async def verify_all_sites():
|
||||
# 1. Test Browserless Connection
|
||||
print("\n[1] Testing Browserless Connection...")
|
||||
try:
|
||||
await browserless_service.start()
|
||||
await browserless_service.initialize()
|
||||
print("✅ Browserless connected successfully")
|
||||
except Exception as e:
|
||||
print(f"❌ Browserless connection failed: {e}")
|
||||
@@ -135,7 +135,7 @@ async def verify_all_sites():
|
||||
print(f" {status} {site['site']:20s} {count}")
|
||||
|
||||
# 5. Cleanup
|
||||
await browserless_service.stop()
|
||||
await browserless_service.shutdown()
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print(f"Verification completed at: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
from app.services.search_service import new_search_service
|
||||
from app.services.browserless_service import browserless_service
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler()]
|
||||
)
|
||||
|
||||
async def test_specific_sites():
|
||||
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
|
||||
query = "chaise"
|
||||
|
||||
print(f"Testing {target_sites} with query '{query}'...")
|
||||
|
||||
await browserless_service.initialize()
|
||||
|
||||
for site_key in target_sites:
|
||||
if site_key not in SITE_CONFIGS:
|
||||
print(f"Skipping {site_key} (not in config)")
|
||||
continue
|
||||
|
||||
print(f"\n--- Testing {site_key} ---")
|
||||
try:
|
||||
results = await new_search_service.search_site(site_key, query)
|
||||
print(f"Found {len(results)} results")
|
||||
for r in results[:3]:
|
||||
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
|
||||
except Exception as e:
|
||||
print(f"Error testing {site_key}: {e}")
|
||||
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_specific_sites())
|
||||
Reference in new issue
Block a user