feat: Add Amazon product scraping service using Playwright and Browserless for persistent browser connections.

This commit is contained in:
Michael committed 2025-11-30 14:56:01 +01:00
1 parent 66d6f14d13
commit 510904d899
1 file changed
+18 -3
+18 -3
View File
@@ -353,14 +353,29 @@ class AmazonScraperService:
logger.debug(f" ⏭️ Card {idx}: No title")
return None
# URL
# URL
link_elem = card.select_one('h2 a') or card.select_one('a.s-link-style')
if not link_elem:
logger.debug(f" ⏭️ Card {idx}: No link")
logger.debug(f" ⏭️ Card {idx}: No link element")
return None
href = link_elem.get('href', '')
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href
# CRITICAL: Validate href is not empty or just '#'
if not href or href == '#' or href.strip() == '':
logger.warning(f" ⏭️ Card {idx}: Invalid href '{href}' for {title[:30] if title else 'unknown'}")
return None
# Build absolute URL
if href.startswith('/'):
product_url = f"{AMAZON_FR_BASE_URL}{href}"
elif href.startswith('http'):
product_url = href
else:
logger.warning(f" ⏭️ Card {idx}: Unexpected href format '{href[:50]}' for {title[:30] if title else 'unknown'}")
return None
logger.debug(f" ✓ Card {idx}: URL = {product_url[:80]}...")
# Price
price = None