mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: add Amazon scraper service using Playwright, Browserless, and stealth for persistent sessions.
This commit is contained in:
1 parent
fd34bc66b2
commit
0ab7ac5a38
3 files changed
+102
-47
No files matched your search
@@ -39,8 +39,10 @@ AMAZON_POPUP_SELECTORS = [
|
||||
# PYDANTIC SCHEMAS
|
||||
# ============================================================================
|
||||
|
||||
|
||||
class AmazonProduct(BaseModel):
|
||||
"""Schema for Amazon product extraction"""
|
||||
|
||||
title: str = Field(description="Product title")
|
||||
url: str = Field(description="Product URL")
|
||||
price: float | None = Field(default=None, description="Price in EUR")
|
||||
@@ -57,16 +59,17 @@ class AmazonProduct(BaseModel):
|
||||
# PARSING HELPERS
|
||||
# ============================================================================
|
||||
|
||||
|
||||
def parse_amazon_price(price_text: str) -> float | None:
|
||||
"""Parse Amazon price formats"""
|
||||
if not price_text:
|
||||
return None
|
||||
|
||||
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
|
||||
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
|
||||
cleaned = cleaned.replace(',', '.')
|
||||
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
|
||||
cleaned = cleaned.replace(" ", "").replace("\xa0", "")
|
||||
cleaned = cleaned.replace(",", ".")
|
||||
|
||||
match = re.search(r'(\d+\.?\d*)', cleaned)
|
||||
match = re.search(r"(\d+\.?\d*)", cleaned)
|
||||
if match:
|
||||
try:
|
||||
return float(match.group(1))
|
||||
@@ -80,10 +83,10 @@ def parse_rating(rating_text: str) -> float | None:
|
||||
if not rating_text:
|
||||
return None
|
||||
|
||||
match = re.search(r'(\d+[,.]\d+)', rating_text)
|
||||
match = re.search(r"(\d+[,.]\d+)", rating_text)
|
||||
if match:
|
||||
try:
|
||||
return float(match.group(1).replace(',', '.'))
|
||||
return float(match.group(1).replace(",", "."))
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
@@ -94,8 +97,8 @@ def parse_reviews_count(reviews_text: str) -> int | None:
|
||||
if not reviews_text:
|
||||
return None
|
||||
|
||||
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
|
||||
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
|
||||
cleaned = re.sub(r"[^\d\s]", "", reviews_text)
|
||||
cleaned = cleaned.replace(" ", "").replace("\xa0", "")
|
||||
|
||||
try:
|
||||
return int(cleaned)
|
||||
@@ -107,6 +110,7 @@ def parse_reviews_count(reviews_text: str) -> int | None:
|
||||
# AMAZON SCRAPER SERVICE
|
||||
# ============================================================================
|
||||
|
||||
|
||||
class AmazonScraperService:
|
||||
"""Persistent browser service for Amazon scraping"""
|
||||
|
||||
@@ -327,14 +331,44 @@ class AmazonScraperService:
|
||||
# Small delay
|
||||
await asyncio.sleep(random.uniform(1.0, 3.0))
|
||||
|
||||
# NOW navigate to search in SAME context (cookies preserved)
|
||||
# NOW interact with the search bar naturally
|
||||
try:
|
||||
logger.info(f"⌨️ Typing search query: {query}")
|
||||
search_input_selector = "#twotabsearchtextbox"
|
||||
|
||||
# Wait for input to be visible and editable
|
||||
await page.wait_for_selector(search_input_selector, state="visible", timeout=10000)
|
||||
search_input = page.locator(search_input_selector)
|
||||
|
||||
# Click and clear first
|
||||
await search_input.click()
|
||||
await search_input.fill("")
|
||||
|
||||
# Type slowly like a human
|
||||
await search_input.type(query, delay=100)
|
||||
|
||||
await asyncio.sleep(random.uniform(0.5, 1.5))
|
||||
|
||||
# Click search button
|
||||
submit_selector = "#nav-search-submit-button"
|
||||
await page.wait_for_selector(submit_selector, state="visible", timeout=5000)
|
||||
await page.click(submit_selector)
|
||||
|
||||
logger.info("🖱️ Clicked search button, waiting for results...")
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"⚠️ Search bar interaction failed: {e}. Fallback to direct URL.")
|
||||
# Fallback to direct navigation
|
||||
logger.info(f"🔍 Navigating to search: {search_url}")
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=60000)
|
||||
|
||||
logger.info("Page loaded (domcontentloaded)")
|
||||
|
||||
# Wait for content or block
|
||||
try:
|
||||
await page.wait_for_selector('div[data-component-type="s-search-result"], .s-result-list', timeout=10000)
|
||||
await page.wait_for_selector(
|
||||
'div[data-component-type="s-search-result"], .s-result-list', timeout=10000
|
||||
)
|
||||
logger.info("✅ Search results detected")
|
||||
await cls._simulate_human_behavior(page)
|
||||
except Exception:
|
||||
@@ -344,7 +378,10 @@ class AmazonScraperService:
|
||||
html_content = await page.content()
|
||||
|
||||
# Proactive block detection
|
||||
if "Type the characters you see in this image" in html_content or "Saisissez les caractères que vous voyez" in html_content:
|
||||
if (
|
||||
"Type the characters you see in this image" in html_content
|
||||
or "Saisissez les caractères que vous voyez" in html_content
|
||||
):
|
||||
logger.error("🚫 CAPTCHA / Bot detection triggered")
|
||||
return []
|
||||
|
||||
@@ -360,19 +397,19 @@ class AmazonScraperService:
|
||||
return []
|
||||
|
||||
# Parse with BeautifulSoup
|
||||
soup = BeautifulSoup(html_content, 'html.parser')
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
|
||||
# Find product cards
|
||||
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
|
||||
product_cards = soup.find_all("div", {"data-component-type": "s-search-result"})
|
||||
|
||||
if not product_cards:
|
||||
# Try alternative
|
||||
product_cards = soup.find_all('div', {'data-asin': True, 'data-index': True})
|
||||
product_cards = soup.find_all("div", {"data-asin": True, "data-index": True})
|
||||
|
||||
if not product_cards:
|
||||
logger.warning("⚠️ No products found")
|
||||
# Check for blocks
|
||||
if '503' in html_content or 'robot' in html_content.lower():
|
||||
if "503" in html_content or "robot" in html_content.lower():
|
||||
logger.error("🚫 Amazon blocked request")
|
||||
return []
|
||||
|
||||
@@ -407,7 +444,7 @@ class AmazonScraperService:
|
||||
def _extract_product(card, idx: int) -> AmazonProduct | None:
|
||||
"""Extract product data from card"""
|
||||
# ASIN
|
||||
asin = card.get('data-asin', '')
|
||||
asin = card.get("data-asin", "")
|
||||
if not asin:
|
||||
logger.debug(f" ⏭️ Card {idx}: No ASIN")
|
||||
return None
|
||||
@@ -417,7 +454,7 @@ class AmazonScraperService:
|
||||
|
||||
# Title
|
||||
title = None
|
||||
for selector in ['h2 a span', 'h2 span', 'h2.s-line-clamp-2 span']:
|
||||
for selector in ["h2 a span", "h2 span", "h2.s-line-clamp-2 span"]:
|
||||
elem = card.select_one(selector)
|
||||
if elem:
|
||||
title = elem.get_text(strip=True)
|
||||
@@ -429,32 +466,34 @@ class AmazonScraperService:
|
||||
return None
|
||||
|
||||
# URL
|
||||
link_elem = card.select_one('h2 a') or card.select_one('a.s-link-style')
|
||||
link_elem = card.select_one("h2 a") or card.select_one("a.s-link-style")
|
||||
if not link_elem:
|
||||
logger.debug(f" ⏭️ Card {idx}: No link element")
|
||||
return None
|
||||
|
||||
href = link_elem.get('href', '')
|
||||
href = link_elem.get("href", "")
|
||||
|
||||
# CRITICAL: Validate href is not empty or just '#'
|
||||
if not href or href == '#' or href.strip() == '':
|
||||
if not href or href == "#" or href.strip() == "":
|
||||
logger.warning(f" ⏭️ Card {idx}: Invalid href '{href}' for {title[:30] if title else 'unknown'}")
|
||||
return None
|
||||
|
||||
# Build absolute URL
|
||||
if href.startswith('/'):
|
||||
if href.startswith("/"):
|
||||
product_url = f"{AMAZON_FR_BASE_URL}{href}"
|
||||
elif href.startswith('http'):
|
||||
elif href.startswith("http"):
|
||||
product_url = href
|
||||
else:
|
||||
logger.warning(f" ⏭️ Card {idx}: Unexpected href format '{href[:50]}' for {title[:30] if title else 'unknown'}")
|
||||
logger.warning(
|
||||
f" ⏭️ Card {idx}: Unexpected href format '{href[:50]}' for {title[:30] if title else 'unknown'}"
|
||||
)
|
||||
return None
|
||||
|
||||
logger.debug(f" ✓ Card {idx}: URL = {product_url[:80]}...")
|
||||
|
||||
# Price
|
||||
price = None
|
||||
for selector in ['.a-price .a-offscreen', '.a-price-whole', 'span.a-price span.a-offscreen']:
|
||||
for selector in [".a-price .a-offscreen", ".a-price-whole", "span.a-price span.a-offscreen"]:
|
||||
elem = card.select_one(selector)
|
||||
if elem:
|
||||
price = parse_amazon_price(elem.get_text(strip=True))
|
||||
@@ -463,7 +502,7 @@ class AmazonScraperService:
|
||||
|
||||
# Original price
|
||||
original_price = None
|
||||
elem = card.select_one('.a-price.a-text-price .a-offscreen')
|
||||
elem = card.select_one(".a-price.a-text-price .a-offscreen")
|
||||
if elem:
|
||||
original_price = parse_amazon_price(elem.get_text(strip=True))
|
||||
|
||||
@@ -472,13 +511,13 @@ class AmazonScraperService:
|
||||
for selector in ['[aria-label*="étoile"]', '[aria-label*="star"]']:
|
||||
elem = card.select_one(selector)
|
||||
if elem:
|
||||
rating = parse_rating(elem.get('aria-label', ''))
|
||||
rating = parse_rating(elem.get("aria-label", ""))
|
||||
if rating:
|
||||
break
|
||||
|
||||
# Reviews
|
||||
reviews_count = None
|
||||
for selector in ['[aria-label*="étoile"] + span', 'span.s-underline-text']:
|
||||
for selector in ['[aria-label*="étoile"] + span', "span.s-underline-text"]:
|
||||
elem = card.select_one(selector)
|
||||
if elem:
|
||||
reviews_count = parse_reviews_count(elem.get_text(strip=True))
|
||||
@@ -487,15 +526,15 @@ class AmazonScraperService:
|
||||
|
||||
# Image
|
||||
image_url = None
|
||||
for selector in ['img.s-image', 'img']:
|
||||
for selector in ["img.s-image", "img"]:
|
||||
elem = card.select_one(selector)
|
||||
if elem:
|
||||
image_url = elem.get('src') or elem.get('data-src')
|
||||
image_url = elem.get("src") or elem.get("data-src")
|
||||
if image_url:
|
||||
break
|
||||
|
||||
# Prime
|
||||
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one('i.a-icon-prime'))
|
||||
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one("i.a-icon-prime"))
|
||||
|
||||
# Stock
|
||||
in_stock = True
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
|
||||
- [ ] Investigate Comparator Price Extraction (User reported false prices)
|
||||
- [x] Fix Catalog Images (Generic/Missing icons)
|
||||
- [/] **Amazon Login Wall Fix** (Redirected by User)
|
||||
|
||||
## 📋 Master Plan
|
||||
|
||||
@@ -38,4 +39,5 @@
|
||||
- **2025-12-22**: Fixed Catalog Retrieval using HTTP fallback.
|
||||
- **2025-12-22**: Investigating generic icon issue. Found images are served via Thumbor.
|
||||
- **2025-12-22**: **FIXED**: Scraper now targets Thumbor images. Wiped bad catalogs.
|
||||
- **2025-12-22**: User reported incorrect prices in Comparator. Investigating.
|
||||
- **2025-12-24**: User prioritized Amazon Login Wall fix.
|
||||
- **2025-12-24**: Reproduction script `test_amazon_scraper.py` patched to use local browser. Verifying issue.
|
||||
+30
-16
@@ -12,27 +12,36 @@ from pathlib import Path
|
||||
# Ajouter le répertoire app au path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.amazon_scraper import (
|
||||
scrape_amazon_search,
|
||||
test_amazon_scraper,
|
||||
)
|
||||
from app.services.amazon_scraper_service import amazon_scraper_service, AmazonScraperService
|
||||
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_DATA
|
||||
|
||||
# Define missing variable for test compatibility
|
||||
AMAZON_USER_AGENTS = USER_AGENT_DATA
|
||||
USER_AGENT_POOL = USER_AGENT_DATA
|
||||
|
||||
|
||||
# Configuration du logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s")
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Monkey patch _connect_browser to use local launch for testing
|
||||
async def _connect_browser_local(p):
|
||||
logger.info("Launching local browser (headless)...")
|
||||
return await p.chromium.launch(headless=True)
|
||||
|
||||
|
||||
AmazonScraperService._connect_browser = _connect_browser_local
|
||||
|
||||
|
||||
async def test_basic_search():
|
||||
"""Test basique de recherche"""
|
||||
logger.info("=" * 80)
|
||||
logger.info("TEST 1: Recherche basique - 'aspirateur'")
|
||||
logger.info("=" * 80)
|
||||
|
||||
products = await scrape_amazon_search("aspirateur", max_results=5)
|
||||
products = await amazon_scraper_service.scrape_search("aspirateur", max_results=5)
|
||||
|
||||
if not products:
|
||||
logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau")
|
||||
@@ -42,7 +51,9 @@ async def test_basic_search():
|
||||
|
||||
for idx, product in enumerate(products, 1):
|
||||
logger.info(f"\n{idx}. {product.title[:60]}...")
|
||||
logger.info(f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else ""))
|
||||
logger.info(
|
||||
f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else "")
|
||||
)
|
||||
logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note")
|
||||
logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}")
|
||||
logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}")
|
||||
@@ -62,7 +73,7 @@ async def test_multiple_queries():
|
||||
|
||||
for query in queries:
|
||||
logger.info(f"\n🔍 Recherche: '{query}'")
|
||||
products = await scrape_amazon_search(query, max_results=3)
|
||||
products = await amazon_scraper_service.scrape_search(query, max_results=3)
|
||||
results[query] = len(products)
|
||||
logger.info(f" ✅ {len(products)} produits trouvés")
|
||||
|
||||
@@ -88,18 +99,21 @@ async def test_anti_detection():
|
||||
logger.info("TEST 3: Vérification anti-détection")
|
||||
logger.info("=" * 80)
|
||||
|
||||
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_POOL
|
||||
from app.services.amazon_scraper import get_random_proxy, AMAZON_USER_AGENTS
|
||||
# Updated to just check if we can run
|
||||
logger.info("Skipping specific proxy/agent checks for this service as it handles them internally")
|
||||
|
||||
logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles")
|
||||
logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards")
|
||||
logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques")
|
||||
|
||||
# Test proxy
|
||||
proxy = get_random_proxy()
|
||||
# Test proxy
|
||||
import random
|
||||
|
||||
proxy = random.choice(AMAZON_PROXY_LIST_RAW) if AMAZON_PROXY_LIST_RAW else None
|
||||
if proxy:
|
||||
# Extract just the IP for logging (hide credentials)
|
||||
proxy_parts = proxy.split('@')
|
||||
proxy_parts = proxy.split("@")
|
||||
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
|
||||
logger.info(f"✓ Proxy test: {proxy_server}")
|
||||
else:
|
||||
@@ -107,7 +121,7 @@ async def test_anti_detection():
|
||||
|
||||
# Test d'une recherche simple
|
||||
logger.info("\n🧪 Test de recherche avec anti-détection...")
|
||||
products = await scrape_amazon_search("livre", max_results=3)
|
||||
products = await amazon_scraper_service.scrape_search("livre", max_results=3)
|
||||
|
||||
if products:
|
||||
logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits")
|
||||
|
||||
Reference in new issue
Block a user