feat: add Amazon scraper service using Playwright, Browserless, and stealth for persistent sessions.

This commit is contained in:
Michael committed 2025-12-24 12:04:58 +01:00
1 parent fd34bc66b2
commit 0ab7ac5a38
3 files changed
+102 -47

No files matched your search

+69 -30
View File
@@ -39,8 +39,10 @@ AMAZON_POPUP_SELECTORS = [
# PYDANTIC SCHEMAS
# ============================================================================
class AmazonProduct(BaseModel):
"""Schema for Amazon product extraction"""
title: str = Field(description="Product title")
url: str = Field(description="Product URL")
price: float | None = Field(default=None, description="Price in EUR")
@@ -57,16 +59,17 @@ class AmazonProduct(BaseModel):
# PARSING HELPERS
# ============================================================================
def parse_amazon_price(price_text: str) -> float | None:
"""Parse Amazon price formats"""
if not price_text:
return None
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
cleaned = cleaned.replace(',', '.')
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(" ", "").replace("\xa0", "")
cleaned = cleaned.replace(",", ".")
match = re.search(r'(\d+\.?\d*)', cleaned)
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
return float(match.group(1))
@@ -80,10 +83,10 @@ def parse_rating(rating_text: str) -> float | None:
if not rating_text:
return None
match = re.search(r'(\d+[,.]\d+)', rating_text)
match = re.search(r"(\d+[,.]\d+)", rating_text)
if match:
try:
return float(match.group(1).replace(',', '.'))
return float(match.group(1).replace(",", "."))
except ValueError:
return None
return None
@@ -94,8 +97,8 @@ def parse_reviews_count(reviews_text: str) -> int | None:
if not reviews_text:
return None
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
cleaned = re.sub(r"[^\d\s]", "", reviews_text)
cleaned = cleaned.replace(" ", "").replace("\xa0", "")
try:
return int(cleaned)
@@ -107,6 +110,7 @@ def parse_reviews_count(reviews_text: str) -> int | None:
# AMAZON SCRAPER SERVICE
# ============================================================================
class AmazonScraperService:
"""Persistent browser service for Amazon scraping"""
@@ -327,14 +331,44 @@ class AmazonScraperService:
# Small delay
await asyncio.sleep(random.uniform(1.0, 3.0))
# NOW navigate to search in SAME context (cookies preserved)
# NOW interact with the search bar naturally
try:
logger.info(f"⌨️ Typing search query: {query}")
search_input_selector = "#twotabsearchtextbox"
# Wait for input to be visible and editable
await page.wait_for_selector(search_input_selector, state="visible", timeout=10000)
search_input = page.locator(search_input_selector)
# Click and clear first
await search_input.click()
await search_input.fill("")
# Type slowly like a human
await search_input.type(query, delay=100)
await asyncio.sleep(random.uniform(0.5, 1.5))
# Click search button
submit_selector = "#nav-search-submit-button"
await page.wait_for_selector(submit_selector, state="visible", timeout=5000)
await page.click(submit_selector)
logger.info("🖱️ Clicked search button, waiting for results...")
except Exception as e:
logger.warning(f"⚠️ Search bar interaction failed: {e}. Fallback to direct URL.")
# Fallback to direct navigation
logger.info(f"🔍 Navigating to search: {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=60000)
logger.info("Page loaded (domcontentloaded)")
# Wait for content or block
try:
await page.wait_for_selector('div[data-component-type="s-search-result"], .s-result-list', timeout=10000)
await page.wait_for_selector(
'div[data-component-type="s-search-result"], .s-result-list', timeout=10000
)
logger.info("✅ Search results detected")
await cls._simulate_human_behavior(page)
except Exception:
@@ -344,7 +378,10 @@ class AmazonScraperService:
html_content = await page.content()
# Proactive block detection
if "Type the characters you see in this image" in html_content or "Saisissez les caractères que vous voyez" in html_content:
if (
"Type the characters you see in this image" in html_content
or "Saisissez les caractères que vous voyez" in html_content
):
logger.error("🚫 CAPTCHA / Bot detection triggered")
return []
@@ -360,19 +397,19 @@ class AmazonScraperService:
return []
# Parse with BeautifulSoup
soup = BeautifulSoup(html_content, 'html.parser')
soup = BeautifulSoup(html_content, "html.parser")
# Find product cards
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
product_cards = soup.find_all("div", {"data-component-type": "s-search-result"})
if not product_cards:
# Try alternative
product_cards = soup.find_all('div', {'data-asin': True, 'data-index': True})
product_cards = soup.find_all("div", {"data-asin": True, "data-index": True})
if not product_cards:
logger.warning("⚠️ No products found")
# Check for blocks
if '503' in html_content or 'robot' in html_content.lower():
if "503" in html_content or "robot" in html_content.lower():
logger.error("🚫 Amazon blocked request")
return []
@@ -407,7 +444,7 @@ class AmazonScraperService:
def _extract_product(card, idx: int) -> AmazonProduct | None:
"""Extract product data from card"""
# ASIN
asin = card.get('data-asin', '')
asin = card.get("data-asin", "")
if not asin:
logger.debug(f" ⏭️ Card {idx}: No ASIN")
return None
@@ -417,7 +454,7 @@ class AmazonScraperService:
# Title
title = None
for selector in ['h2 a span', 'h2 span', 'h2.s-line-clamp-2 span']:
for selector in ["h2 a span", "h2 span", "h2.s-line-clamp-2 span"]:
elem = card.select_one(selector)
if elem:
title = elem.get_text(strip=True)
@@ -429,32 +466,34 @@ class AmazonScraperService:
return None
# URL
link_elem = card.select_one('h2 a') or card.select_one('a.s-link-style')
link_elem = card.select_one("h2 a") or card.select_one("a.s-link-style")
if not link_elem:
logger.debug(f" ⏭️ Card {idx}: No link element")
return None
href = link_elem.get('href', '')
href = link_elem.get("href", "")
# CRITICAL: Validate href is not empty or just '#'
if not href or href == '#' or href.strip() == '':
if not href or href == "#" or href.strip() == "":
logger.warning(f" ⏭️ Card {idx}: Invalid href '{href}' for {title[:30] if title else 'unknown'}")
return None
# Build absolute URL
if href.startswith('/'):
if href.startswith("/"):
product_url = f"{AMAZON_FR_BASE_URL}{href}"
elif href.startswith('http'):
elif href.startswith("http"):
product_url = href
else:
logger.warning(f" ⏭️ Card {idx}: Unexpected href format '{href[:50]}' for {title[:30] if title else 'unknown'}")
logger.warning(
f" ⏭️ Card {idx}: Unexpected href format '{href[:50]}' for {title[:30] if title else 'unknown'}"
)
return None
logger.debug(f" ✓ Card {idx}: URL = {product_url[:80]}...")
# Price
price = None
for selector in ['.a-price .a-offscreen', '.a-price-whole', 'span.a-price span.a-offscreen']:
for selector in [".a-price .a-offscreen", ".a-price-whole", "span.a-price span.a-offscreen"]:
elem = card.select_one(selector)
if elem:
price = parse_amazon_price(elem.get_text(strip=True))
@@ -463,7 +502,7 @@ class AmazonScraperService:
# Original price
original_price = None
elem = card.select_one('.a-price.a-text-price .a-offscreen')
elem = card.select_one(".a-price.a-text-price .a-offscreen")
if elem:
original_price = parse_amazon_price(elem.get_text(strip=True))
@@ -472,13 +511,13 @@ class AmazonScraperService:
for selector in ['[aria-label*="étoile"]', '[aria-label*="star"]']:
elem = card.select_one(selector)
if elem:
rating = parse_rating(elem.get('aria-label', ''))
rating = parse_rating(elem.get("aria-label", ""))
if rating:
break
# Reviews
reviews_count = None
for selector in ['[aria-label*="étoile"] + span', 'span.s-underline-text']:
for selector in ['[aria-label*="étoile"] + span', "span.s-underline-text"]:
elem = card.select_one(selector)
if elem:
reviews_count = parse_reviews_count(elem.get_text(strip=True))
@@ -487,15 +526,15 @@ class AmazonScraperService:
# Image
image_url = None
for selector in ['img.s-image', 'img']:
for selector in ["img.s-image", "img"]:
elem = card.select_one(selector)
if elem:
image_url = elem.get('src') or elem.get('data-src')
image_url = elem.get("src") or elem.get("data-src")
if image_url:
break
# Prime
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one('i.a-icon-prime'))
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one("i.a-icon-prime"))
# Stock
in_stock = True
+3 -1
View File
@@ -4,6 +4,7 @@
- [ ] Investigate Comparator Price Extraction (User reported false prices)
- [x] Fix Catalog Images (Generic/Missing icons)
- [/] **Amazon Login Wall Fix** (Redirected by User)
## 📋 Master Plan
@@ -38,4 +39,5 @@
- **2025-12-22**: Fixed Catalog Retrieval using HTTP fallback.
- **2025-12-22**: Investigating generic icon issue. Found images are served via Thumbor.
- **2025-12-22**: **FIXED**: Scraper now targets Thumbor images. Wiped bad catalogs.
- **2025-12-22**: User reported incorrect prices in Comparator. Investigating.
- **2025-12-24**: User prioritized Amazon Login Wall fix.
- **2025-12-24**: Reproduction script `test_amazon_scraper.py` patched to use local browser. Verifying issue.
+30 -16
View File
@@ -12,27 +12,36 @@ from pathlib import Path
# Ajouter le répertoire app au path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.amazon_scraper import (
scrape_amazon_search,
test_amazon_scraper,
)
from app.services.amazon_scraper_service import amazon_scraper_service, AmazonScraperService
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_DATA
# Define missing variable for test compatibility
AMAZON_USER_AGENTS = USER_AGENT_DATA
USER_AGENT_POOL = USER_AGENT_DATA
# Configuration du logging
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
)
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s")
logger = logging.getLogger(__name__)
# Monkey patch _connect_browser to use local launch for testing
async def _connect_browser_local(p):
logger.info("Launching local browser (headless)...")
return await p.chromium.launch(headless=True)
AmazonScraperService._connect_browser = _connect_browser_local
async def test_basic_search():
"""Test basique de recherche"""
logger.info("=" * 80)
logger.info("TEST 1: Recherche basique - 'aspirateur'")
logger.info("=" * 80)
products = await scrape_amazon_search("aspirateur", max_results=5)
products = await amazon_scraper_service.scrape_search("aspirateur", max_results=5)
if not products:
logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau")
@@ -42,7 +51,9 @@ async def test_basic_search():
for idx, product in enumerate(products, 1):
logger.info(f"\n{idx}. {product.title[:60]}...")
logger.info(f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else ""))
logger.info(
f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else "")
)
logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note")
logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}")
logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}")
@@ -62,7 +73,7 @@ async def test_multiple_queries():
for query in queries:
logger.info(f"\n🔍 Recherche: '{query}'")
products = await scrape_amazon_search(query, max_results=3)
products = await amazon_scraper_service.scrape_search(query, max_results=3)
results[query] = len(products)
logger.info(f" ✅ {len(products)} produits trouvés")
@@ -88,18 +99,21 @@ async def test_anti_detection():
logger.info("TEST 3: Vérification anti-détection")
logger.info("=" * 80)
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_POOL
from app.services.amazon_scraper import get_random_proxy, AMAZON_USER_AGENTS
# Updated to just check if we can run
logger.info("Skipping specific proxy/agent checks for this service as it handles them internally")
logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles")
logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards")
logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques")
# Test proxy
proxy = get_random_proxy()
# Test proxy
import random
proxy = random.choice(AMAZON_PROXY_LIST_RAW) if AMAZON_PROXY_LIST_RAW else None
if proxy:
# Extract just the IP for logging (hide credentials)
proxy_parts = proxy.split('@')
proxy_parts = proxy.split("@")
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
logger.info(f"✓ Proxy test: {proxy_server}")
else:
@@ -107,7 +121,7 @@ async def test_anti_detection():
# Test d'une recherche simple
logger.info("\n🧪 Test de recherche avec anti-détection...")
products = await scrape_amazon_search("livre", max_results=3)
products = await amazon_scraper_service.scrape_search("livre", max_results=3)
if products:
logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits")