Merge main: Resolved conflicts by keeping enhanced browserless service with auto-reconnection + price extraction

This commit is contained in:
Michael committed 2025-11-30 12:49:40 +01:00
commit c93e55519e
18 files changed
+2919 -106

No files matched your search

+22 -4
View File
@@ -133,13 +133,31 @@ class AIExtractionMetadata(BaseModel):
EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this French e-commerce page.
**PRICE (IMPORTANT - French format):**
- French prices use COMMA for decimals: "12,99 €" means 12.99
- Thousands separator is SPACE or DOT: "1 234,56 €" or "1.234,56 €" means 1234.56
- **FIRST**: Look for "PRIX DÉTECTÉ:" at the start of text - this is the extracted price
- **The price has already been converted to English format for you**
- French original: "3,99 €" → Already shown to you as: "3.99 €"
- Just extract the number you see (e.g., "3.99" from "PRIX DÉTECTÉ: 3.99 €")
- Currency symbol is usually € at the end
- Look for: price tags, "Prix:", "€", numbers near "Ajouter au panier"
- Extract as DECIMAL NUMBER (convert comma to dot): 12,99 -> 12.99
- Look for: "PRIX DÉTECTÉ:", price tags, "Prix:", "€", numbers near "Ajouter au panier"
- Extract as DECIMAL NUMBER: If you see "3.99", return 3.99
- Ignore crossed-out/barré prices (old prices)
- If multiple prices, take the current/main price (not the original)
**CRITICAL - Common mistakes to avoid:**
- "3.99 €" means 3.99 (NOT 399.00, NOT 3990.00)
- "1.99 €" means 1.99 (NOT 199.00, NOT 1990.00)
- "0.99 €" means 0.99 (NOT 99.00, NOT 990.00)
- The decimal point separates euros from cents
- Small prices (< 10€) are very common for everyday items
- Examples of valid prices:
* "PRIX DÉTECTÉ: 1.99 €" -> 1.99 (NOT 199 or 1990)
* "PRIX DÉTECTÉ: 3.99 €" -> 3.99 (NOT 399 or 3990)
* "0.99 €" -> 0.99 (NOT 99)
* "89.99 €" -> 89.99 (NOT 8999)
* "1234.56 €" -> 1234.56
- If you find ANY price with € symbol, extract it with confidence >= 0.8
- If digits are unclear or blurry, reduce confidence to 0.5-0.7
- If unclear: set null and confidence < 0.5
**STOCK:**
+19 -4
View File
@@ -13,14 +13,16 @@ from sqlalchemy import text
from app.database import SessionLocal, engine
from app.limiter import limiter
from app.routers import auth, items, jobs, notifications, openrouter, search, search_sites, settings, debug, catalogues
from app.routers import auth, items, jobs, notifications, openrouter, search, search_sites, settings, debug, catalogues, amazon
from app.services.scheduler_service import scheduled_refresh, scheduler
from app.services import auth_service, search_service, seed_enseignes
from app.services.scheduler import start_scheduler as start_catalog_scheduler, stop_scheduler as stop_catalog_scheduler
from app.services.amazon_scraper_service import amazon_scraper_service
from app.services.improved_search_service import improved_search_service
# Configure logging
logging.basicConfig(
level=os.getenv("LOG_LEVEL", "INFO").upper(),
level=os.getenv("LOG_LEVEL", "DEBUG").upper(), # Temporarily DEBUG for Amazon debugging
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
)
logger = logging.getLogger(__name__)
@@ -139,12 +141,22 @@ async def lifespan(app: FastAPI):
# Start catalog scraping scheduler
logger.info("Starting catalog scraping scheduler (6h and 18h daily)")
start_catalog_scheduler()
# Initialize Amazon scraper service
logger.info("Initializing Amazon scraper service...")
await amazon_scraper_service.initialize()
# Initialize Improved Search Service (persistent browser for Search/Comparatif)
logger.info("Initializing Improved Search Service...")
await improved_search_service.initialize()
logger.info("Application started")
yield
logger.info("Shutting down schedulers...")
logger.info("Shutting down schedulers and services...")
scheduler.shutdown(wait=True)
stop_catalog_scheduler()
await amazon_scraper_service.shutdown()
await improved_search_service.shutdown()
logger.info("Application shutdown complete")
@@ -190,6 +202,9 @@ app.include_router(catalogues.router)
# Auth router (already has /api prefix)
app.include_router(auth.router)
# Amazon router (already has /api prefix)
app.include_router(amazon.router)
@app.get("/api/")
def read_root():
+121
View File
@@ -0,0 +1,121 @@
"""
API Router pour la recherche Amazon France
Utilise Server-Sent Events (SSE) pour le streaming des résultats
Uses Browserless service for reliable scraping
"""
import json
import logging
from typing import AsyncGenerator
from fastapi import APIRouter, Query
from fastapi.responses import StreamingResponse
from pydantic import BaseModel
from app.services.amazon_scraper_service import amazon_scraper_service, AmazonProduct
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/api/amazon", tags=["amazon"])
class AmazonSearchProgress(BaseModel):
"""Progress model for SSE streaming"""
status: str # 'searching', 'completed', 'error'
total: int
completed: int
message: str
results: list[dict]
@router.get("/search")
async def search_amazon(
q: str = Query(..., min_length=1, description="Terme de recherche"),
max_results: int = Query(20, ge=1, le=50, description="Nombre max de résultats"),
):
"""
Recherche de produits sur Amazon France.
Retourne un flux SSE avec les résultats.
Format des événements SSE:
- event: progress
- data: {"status": "...", "total": 1, "completed": 0/1, "results": [...]}
"""
async def generate() -> AsyncGenerator[str, None]:
try:
# Initial progress
progress = AmazonSearchProgress(
status="searching",
total=1,
completed=0,
message=f"Recherche sur Amazon France: '{q}'...",
results=[]
)
yield f"event: progress\ndata: {progress.model_dump_json()}\n\n"
# Execute search with persistent browser service
logger.info(f"Starting Amazon search for: {q}")
products = await amazon_scraper_service.scrape_search(q, max_results=max_results)
# Convert to dict
results = [p.model_dump() for p in products]
# Final progress
if results:
progress = AmazonSearchProgress(
status="completed",
total=1,
completed=1,
message=f"✅ {len(results)} produits trouvés",
results=results
)
else:
progress = AmazonSearchProgress(
status="completed",
total=1,
completed=1,
message="Aucun produit trouvé",
results=[]
)
yield f"event: progress\ndata: {progress.model_dump_json()}\n\n"
except Exception as e:
logger.error(f"Error during Amazon search: {e}", exc_info=True)
error_progress = AmazonSearchProgress(
status="error",
total=1,
completed=1,
message=f"Erreur: {str(e)}",
results=[]
)
yield f"event: progress\ndata: {error_progress.model_dump_json()}\n\n"
return StreamingResponse(
generate(),
media_type="text/event-stream",
headers={
"Cache-Control": "no-cache",
"Connection": "keep-alive",
"X-Accel-Buffering": "no",
},
)
@router.get("/health")
async def amazon_health():
"""Vérifie que le service Amazon est opérationnel"""
return {
"status": "ok",
"service": "Amazon France Scraper",
"anti_detection": "enabled",
"features": [
"User-Agent rotation",
"Proxy rotation (10 proxies)",
"Random delays",
"Realistic headers",
"Crawl4AI anti-detection"
]
}
+2 -2
View File
@@ -11,7 +11,7 @@ from fastapi.responses import StreamingResponse
from sqlalchemy.orm import Session
from app.database import get_db
from app.services import search_service
from app.services import improved_search_service
logger = logging.getLogger(__name__)
@@ -44,7 +44,7 @@ async def search_products(
async def generate():
try:
async for progress in search_service.search_products(
async for progress in improved_search_service.search_products(
query=q,
db=db,
site_ids=site_ids,
+14
View File
@@ -343,8 +343,22 @@ class AIService:
if len(cleaned_text) > MAX_TEXT_LENGTH:
cleaned_text = cleaned_text[:MAX_TEXT_LENGTH] + "...(truncated)"
logger.info(f"Added text context (original: {len(page_text)}, cleaned: {len(cleaned_text)})")
# Log first 500 chars of cleaned text to see what AI receives
logger.info(f"Cleaned text preview: {cleaned_text[:500]!r}")
# Extract all potential prices from text for debugging
import re
price_patterns = re.findall(r'\d+[,\.]\d{2}\s*€', cleaned_text)
if price_patterns:
logger.info(f"Prices found in text: {price_patterns[:10]}") # First 10 prices
else:
logger.warning("No prices found in text with € symbol")
else:
logger.warning("No page_text provided - AI will only use screenshot")
prompt = get_extraction_prompt(cleaned_text if cleaned_text else None)
# Log prompt preview
logger.info(f"Prompt preview (first 300 chars): {prompt[:300]!r}")
# Call LLM
response_text = await cls.call_llm(prompt, data_url, config)
+451
View File
@@ -0,0 +1,451 @@
"""
Amazon France Scraper Service - Powered by Crawl4AI
Anti-bot detection techniques:
- Realistic User-Agent rotation
- Complete HTTP headers mimicking real browsers
- Proxy rotation (10 residential proxies)
- Random delays between requests
- Browser fingerprint randomization
- Cookie persistence
- NetworkIdle waiting for complete page load
"""
import asyncio
import hashlib
import logging
import random
import re
from datetime import datetime
from typing import Any
from urllib.parse import quote_plus
from bs4 import BeautifulSoup
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
from pydantic import BaseModel, Field
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_POOL
logger = logging.getLogger(__name__)
# Amazon France configuration
AMAZON_FR_BASE_URL = "https://www.amazon.fr"
AMAZON_FR_SEARCH_URL = "https://www.amazon.fr/s?k={query}"
# Additional realistic User-Agents specifically for Amazon
AMAZON_USER_AGENTS = [
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.1 Safari/605.1.15",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
]
# Realistic browser headers to avoid bot detection
def get_realistic_headers(user_agent: str) -> dict:
"""Generate realistic browser headers"""
return {
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
"Accept-Encoding": "gzip, deflate, br",
"DNT": "1",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
"Sec-Fetch-Dest": "document",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-Site": "none",
"Sec-Fetch-User": "?1",
"Cache-Control": "max-age=0",
"User-Agent": user_agent,
}
def get_random_proxy() -> str | None:
"""
Get a random proxy from the pool in Crawl4AI format.
Returns:
Proxy string in format: http://username:password@ip:port
"""
if not AMAZON_PROXY_LIST_RAW:
return None
proxy_str = random.choice(AMAZON_PROXY_LIST_RAW)
parts = proxy_str.split(":")
if len(parts) == 4:
ip = parts[0]
port = parts[1]
username = parts[2]
password = parts[3]
# Format: http://username:password@ip:port
return f"http://{username}:{password}@{ip}:{port}"
return None
async def random_delay(min_seconds: float = 1.5, max_seconds: float = 4.0):
"""Add random delay to mimic human behavior"""
delay = random.uniform(min_seconds, max_seconds)
logger.debug(f"Human-like delay: {delay:.2f}s")
await asyncio.sleep(delay)
# ============================================================================
# PYDANTIC SCHEMAS
# ============================================================================
class AmazonProduct(BaseModel):
"""Schema for Amazon product extraction"""
title: str = Field(description="Product title")
url: str = Field(description="Product URL")
price: float | None = Field(default=None, description="Price in EUR")
original_price: float | None = Field(default=None, description="Original price if discounted")
rating: float | None = Field(default=None, description="Product rating (0-5)")
reviews_count: int | None = Field(default=None, description="Number of reviews")
image_url: str | None = Field(default=None, description="Product image URL")
in_stock: bool = Field(default=True, description="Availability status")
prime: bool = Field(default=False, description="Prime eligible")
sponsored: bool = Field(default=False, description="Is sponsored")
# ============================================================================
# PRICE PARSING HELPERS
# ============================================================================
def parse_amazon_price(price_text: str) -> float | None:
"""
Parse Amazon price formats:
- "12,99 €"
- "12,99€"
- "12.99 EUR"
- "1 234,99 €"
"""
if not price_text:
return None
# Remove currency symbols and extra spaces
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
# Remove thousands separators (space or dot in French format)
cleaned = cleaned.replace(' ', '').replace('\xa0', '') # \xa0 is non-breaking space
# Replace comma with dot for decimal separator
cleaned = cleaned.replace(',', '.')
# Extract first number (in case of ranges like "12.99 - 15.99")
match = re.search(r'(\d+\.?\d*)', cleaned)
if match:
try:
return float(match.group(1))
except ValueError:
return None
return None
def parse_rating(rating_text: str) -> float | None:
"""Parse rating like '4,5 sur 5 étoiles' or '4.5 out of 5 stars'"""
if not rating_text:
return None
# Match patterns like "4,5" or "4.5"
match = re.search(r'(\d+[,.]\d+)', rating_text)
if match:
try:
return float(match.group(1).replace(',', '.'))
except ValueError:
return None
return None
def parse_reviews_count(reviews_text: str) -> int | None:
"""Parse review count like '1 234' or '12,345'"""
if not reviews_text:
return None
# Remove non-digit characters except spaces
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
try:
return int(cleaned)
except ValueError:
return None
# ============================================================================
# SCRAPING FUNCTIONS
# ============================================================================
async def scrape_amazon_search(query: str, max_results: int = 20) -> list[AmazonProduct]:
"""
Scrape Amazon France search results for a given query.
Anti-detection features:
1. Random User-Agent from realistic pool
2. Complete browser headers
3. Random proxy from pool
4. Random delays between operations
5. Browser fingerprint randomization via Crawl4AI
6. NetworkIdle waiting
7. Cookie handling
Args:
query: Search query
max_results: Maximum number of products to return (default 20)
Returns:
List of AmazonProduct objects
"""
search_url = AMAZON_FR_SEARCH_URL.format(query=quote_plus(query))
logger.info(f"🔍 Searching Amazon France: {query}")
logger.info(f"📍 URL: {search_url}")
# Select random User-Agent
user_agent = random.choice(AMAZON_USER_AGENTS)
logger.debug(f"🎭 User-Agent: {user_agent[:50]}...")
# NOTE: Not using proxies here - Browserless service has its own proxy system
# We can integrate with browserless_service later if needed
# Configure Crawl4AI browser with anti-detection
browser_config = BrowserConfig(
headless=True,
verbose=False,
user_agent=user_agent,
# proxy_config removed - let Crawl4AI use default or integrate with Browserless later
extra_args=[
"--disable-blink-features=AutomationControlled", # Disable automation detection
"--disable-dev-shm-usage",
"--no-sandbox",
"--disable-gpu",
"--disable-setuid-sandbox",
"--disable-web-security",
"--disable-features=IsolateOrigins,site-per-process",
"--disable-infobars",
"--window-size=1920,1080",
"--start-maximized",
# Randomize viewport
f"--user-agent={user_agent}",
],
)
# Configure crawler behavior
crawler_config = CrawlerRunConfig(
cache_mode=CacheMode.BYPASS, # Always fetch fresh data
wait_for_images=True,
process_iframes=False,
remove_overlay_elements=True, # Remove popups/modals
wait_until="networkidle", # Wait for all network requests
delay_before_return_html=2.0, # Extra wait for JS rendering
page_timeout=30000, # 30 seconds timeout
# Accept cookies automatically
js_code="""
// Accept cookies if banner appears
const cookieButton = document.querySelector('#sp-cc-accept, button[id*="accept"]');
if (cookieButton) {
cookieButton.click();
}
""",
)
products = []
try:
# Add human-like delay before request
await random_delay(1.0, 2.5)
async with AsyncWebCrawler(config=browser_config) as crawler:
logger.info("🚀 Launching browser...")
result = await crawler.arun(url=search_url, config=crawler_config)
if not result.success:
logger.error(f"❌ Crawl failed: {result.error_message}")
return []
logger.info(f"✅ Page loaded successfully ({len(result.html)} bytes)")
# Parse HTML with BeautifulSoup
soup = BeautifulSoup(result.html, 'html.parser')
# Debug: Save HTML to file for inspection
debug_file = f"/tmp/amazon_debug_{query[:20]}.html"
try:
with open(debug_file, 'w', encoding='utf-8') as f:
f.write(result.html)
logger.debug(f"📝 HTML saved to {debug_file} for debugging")
except Exception as e:
logger.debug(f"Could not save debug HTML: {e}")
# Amazon uses data-component-type="s-search-result" for product cards
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
if not product_cards:
logger.warning("⚠️ No products found - checking for CAPTCHA or blocks")
# Check for CAPTCHA
if 'captcha' in result.html.lower():
logger.error("🚫 CAPTCHA detected - Amazon blocked the request")
elif 'robot' in result.html.lower() or 'bot' in result.html.lower():
logger.error("🤖 Bot detection triggered")
else:
logger.warning("📦 Empty results - query may have no matches")
return []
logger.info(f"📦 Found {len(product_cards)} product cards")
for idx, card in enumerate(product_cards):
if len(products) >= max_results:
break
try:
# Extract ASIN (Amazon Standard Identification Number)
asin = card.get('data-asin', '')
if not asin:
logger.debug(f" ⏭️ Card {idx}: No ASIN found, skipping")
continue
logger.debug(f" 🔍 Card {idx}: Processing ASIN {asin}")
# Check if sponsored
sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]'))
# Extract title
title_elem = card.select_one('h2 a span, h2 span')
if not title_elem:
logger.debug(f" ⏭️ Card {idx} ({asin}): No title found, skipping")
continue
title = title_elem.get_text(strip=True)
# Extract URL
link_elem = card.select_one('h2 a')
if not link_elem:
logger.debug(f" ⏭️ Card {idx} ({asin}): No link found, skipping")
continue
href = link_elem.get('href', '')
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href
# Extract price
price = None
original_price = None
# Current price
price_elem = card.select_one('.a-price .a-offscreen')
if price_elem:
price = parse_amazon_price(price_elem.get_text(strip=True))
# Original price (if discounted)
original_price_elem = card.select_one('.a-price.a-text-price .a-offscreen')
if original_price_elem:
original_price = parse_amazon_price(original_price_elem.get_text(strip=True))
# Extract rating
rating = None
rating_elem = card.select_one('[aria-label*="étoile"], [aria-label*="star"]')
if rating_elem:
rating = parse_rating(rating_elem.get('aria-label', ''))
# Extract reviews count
reviews_count = None
reviews_elem = card.select_one('[aria-label*="étoile"] + span, [aria-label*="star"] + span')
if reviews_elem:
reviews_count = parse_reviews_count(reviews_elem.get_text(strip=True))
# Extract image
image_url = None
img_elem = card.select_one('img.s-image')
if img_elem:
image_url = img_elem.get('src') or img_elem.get('data-src')
# Check Prime eligibility
prime = bool(card.select_one('[aria-label*="Prime"], i.a-icon-prime'))
# Check availability
in_stock = True
unavailable_elem = card.select_one('[aria-label*="Indisponible"], [aria-label*="Unavailable"]')
if unavailable_elem:
in_stock = False
product = AmazonProduct(
title=title,
url=product_url,
price=price,
original_price=original_price,
rating=rating,
reviews_count=reviews_count,
image_url=image_url,
in_stock=in_stock,
prime=prime,
sponsored=sponsored,
)
products.append(product)
logger.debug(f" ✓ [{idx+1}] {title[:50]}... - {price}€")
except Exception as e:
logger.error(f"❌ Error parsing product card {idx}: {e}")
continue
logger.info(f"✅ Successfully extracted {len(products)} products")
except Exception as e:
logger.error(f"❌ Error during Amazon scraping: {e}", exc_info=True)
return []
return products
async def scrape_amazon_search_batched(
query: str,
max_results: int = 20,
batch_size: int = 20
) -> list[AmazonProduct]:
"""
Scrape Amazon with automatic pagination if needed.
Note: For now, we just scrape the first page (20 results).
Pagination can be added later if needed.
Args:
query: Search query
max_results: Maximum total results (default 20)
batch_size: Results per page (default 20)
Returns:
List of AmazonProduct objects
"""
# For now, single page only
return await scrape_amazon_search(query, max_results)
# ============================================================================
# TESTING / VERIFICATION
# ============================================================================
async def test_amazon_scraper():
"""Test the Amazon scraper with a simple query"""
logger.info("=" * 60)
logger.info("Testing Amazon France Scraper")
logger.info("=" * 60)
test_query = "aspirateur"
products = await scrape_amazon_search(test_query, max_results=5)
logger.info(f"\n📊 Results for '{test_query}':")
logger.info(f"Found {len(products)} products\n")
for idx, product in enumerate(products, 1):
logger.info(f"{idx}. {product.title}")
logger.info(f" 💰 Price: {product.price}€" + (f" (was {product.original_price}€)" if product.original_price else ""))
logger.info(f" ⭐ Rating: {product.rating}/5 ({product.reviews_count} reviews)" if product.rating else " ⭐ No rating")
logger.info(f" 🔗 {product.url}")
logger.info(f" {'✅ Prime' if product.prime else '📦 Standard'} | {'📢 Sponsored' if product.sponsored else '🔍 Organic'}")
logger.info("")
return products
if __name__ == "__main__":
# Run test
asyncio.run(test_amazon_scraper())
+430
View File
@@ -0,0 +1,430 @@
"""
Amazon Scraper Service - Using persistent browser connection
Based on ScraperService pattern for better session management and anti-detection
"""
import asyncio
import logging
import os
import re
from dataclasses import dataclass
from datetime import datetime
from urllib.parse import quote_plus
from bs4 import BeautifulSoup
from playwright.async_api import Browser, BrowserContext, Page, async_playwright
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
from pydantic import BaseModel, Field
logger = logging.getLogger(__name__)
BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000")
# Amazon configuration
AMAZON_FR_BASE_URL = "https://www.amazon.fr"
AMAZON_FR_SEARCH_URL = "https://www.amazon.fr/s?k={query}"
# Cookie/popup selectors for Amazon
AMAZON_POPUP_SELECTORS = [
"#sp-cc-accept", # Cookie banner
"#sp-cc-rejectall-link",
"button[data-action='a-popover-close']",
"[data-action='sp-cc-accept']",
"input[aria-labelledby='sp-cc-accept-label']",
]
# ============================================================================
# PYDANTIC SCHEMAS
# ============================================================================
class AmazonProduct(BaseModel):
"""Schema for Amazon product extraction"""
title: str = Field(description="Product title")
url: str = Field(description="Product URL")
price: float | None = Field(default=None, description="Price in EUR")
original_price: float | None = Field(default=None, description="Original price if discounted")
rating: float | None = Field(default=None, description="Product rating (0-5)")
reviews_count: int | None = Field(default=None, description="Number of reviews")
image_url: str | None = Field(default=None, description="Product image URL")
in_stock: bool = Field(default=True, description="Availability status")
prime: bool = Field(default=False, description="Prime eligible")
sponsored: bool = Field(default=False, description="Is sponsored")
# ============================================================================
# PARSING HELPERS
# ============================================================================
def parse_amazon_price(price_text: str) -> float | None:
"""Parse Amazon price formats"""
if not price_text:
return None
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
cleaned = cleaned.replace(',', '.')
match = re.search(r'(\d+\.?\d*)', cleaned)
if match:
try:
return float(match.group(1))
except ValueError:
return None
return None
def parse_rating(rating_text: str) -> float | None:
"""Parse rating"""
if not rating_text:
return None
match = re.search(r'(\d+[,.]\d+)', rating_text)
if match:
try:
return float(match.group(1).replace(',', '.'))
except ValueError:
return None
return None
def parse_reviews_count(reviews_text: str) -> int | None:
"""Parse review count"""
if not reviews_text:
return None
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
try:
return int(cleaned)
except ValueError:
return None
# ============================================================================
# AMAZON SCRAPER SERVICE
# ============================================================================
class AmazonScraperService:
"""Persistent browser service for Amazon scraping"""
_playwright = None
_browser: Browser | None = None
_lock = asyncio.Lock()
@classmethod
async def initialize(cls):
"""Initialize shared browser (Thread-Safe)"""
async with cls._lock:
await cls._initialize()
@classmethod
async def _initialize(cls):
"""Internal initialization"""
if cls._browser is None:
logger.info("Initializing AmazonScraperService shared browser...")
cls._playwright = await async_playwright().start()
cls._browser = await cls._connect_browser(cls._playwright)
logger.info("AmazonScraperService initialized.")
@classmethod
async def shutdown(cls):
"""Shutdown shared browser"""
async with cls._lock:
if cls._browser:
logger.info("Shutting down AmazonScraperService...")
await cls._browser.close()
cls._browser = None
if cls._playwright:
await cls._playwright.stop()
cls._playwright = None
logger.info("AmazonScraperService shutdown complete.")
@classmethod
async def _ensure_browser_connected(cls) -> bool:
"""Ensure browser is connected, reconnect if needed"""
async with cls._lock:
try:
if cls._browser is None:
logger.warning("Browser not initialized, initializing...")
await cls._initialize()
return cls._browser is not None
# Test connection
try:
test_context = await cls._browser.new_context()
await test_context.close()
return True
except Exception as e:
logger.error(f"Browser connection test failed: {e}")
logger.info("Attempting to reconnect...")
cls._browser = None
if cls._playwright:
try:
await cls._playwright.stop()
except Exception:
pass
cls._playwright = None
await cls._initialize()
return cls._browser is not None
except Exception as e:
logger.error(f"Failed to ensure browser connection: {e}")
return False
@staticmethod
async def _connect_browser(p) -> Browser:
"""Connect to Browserless"""
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
@staticmethod
async def _create_context(browser: Browser) -> BrowserContext:
"""Create browser context with stealth settings"""
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent=(
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
),
locale="fr-FR",
timezone_id="Europe/Paris",
)
# Stealth mode
await context.add_init_script("""
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
window.chrome = { runtime: {} };
""")
await context.route("**/*", lambda route: route.continue_())
return context
@staticmethod
async def _handle_popups(page: Page):
"""Close Amazon popups/cookies"""
logger.info("Handling Amazon popups...")
for selector in AMAZON_POPUP_SELECTORS:
try:
if await page.locator(selector).count() > 0:
logger.info(f"Found popup: {selector}")
await page.locator(selector).first.click(timeout=2000)
await page.wait_for_timeout(1000)
except Exception:
pass
try:
await page.keyboard.press("Escape")
except Exception:
pass
@classmethod
async def scrape_search(cls, query: str, max_results: int = 20) -> list[AmazonProduct]:
"""
Scrape Amazon France search results
Args:
query: Search query
max_results: Maximum products to return
Returns:
List of AmazonProduct objects
"""
search_url = AMAZON_FR_SEARCH_URL.format(query=quote_plus(query))
logger.info(f"🔍 Searching Amazon France: {query}")
logger.info(f"📍 URL: {search_url}")
# Ensure browser is connected
if not await cls._ensure_browser_connected():
logger.error("Failed to establish browser connection")
return []
products = []
try:
context = await cls._create_context(cls._browser)
page = await context.new_page()
try:
# CRITICAL: Load Amazon homepage FIRST in same context to establish session
logger.info("🏠 Loading Amazon homepage to establish session/cookies...")
await page.goto("https://www.amazon.fr", wait_until="domcontentloaded", timeout=30000)
logger.info("✅ Homepage loaded")
# Handle homepage popups
await cls._handle_popups(page)
# Small delay
await page.wait_for_timeout(2000)
# NOW navigate to search in SAME context (cookies preserved)
logger.info(f"🔍 Navigating to search: {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=60000)
logger.info("Page loaded (domcontentloaded)")
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.info("Network idle reached")
except PlaywrightTimeoutError:
logger.info("Network idle timed out (non-critical)")
# Handle popups
await cls._handle_popups(page)
# Wait a bit for content
await page.wait_for_timeout(2000)
# Get HTML
html_content = await page.content()
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
if len(html_content) < 10000:
logger.error(f"❌ Page too small - likely blocked")
return []
# Parse with BeautifulSoup
soup = BeautifulSoup(html_content, 'html.parser')
# Find product cards
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
if not product_cards:
# Try alternative
product_cards = soup.find_all('div', {'data-asin': True, 'data-index': True})
if not product_cards:
logger.warning("⚠️ No products found")
# Check for blocks
if '503' in html_content or 'robot' in html_content.lower():
logger.error("🚫 Amazon blocked request")
return []
logger.info(f"📦 Found {len(product_cards)} product cards")
# Extract products
for idx, card in enumerate(product_cards):
if len(products) >= max_results:
break
try:
product = cls._extract_product(card, idx)
if product:
products.append(product)
logger.debug(f" ✓ [{len(products)}] {product.title[:50]}... - {product.price}€")
except Exception as e:
logger.error(f"Error parsing card {idx}: {e}")
continue
logger.info(f"✅ Successfully extracted {len(products)} products")
finally:
await context.close()
except Exception as e:
logger.error(f"❌ Error during scraping: {e}", exc_info=True)
return []
return products
@staticmethod
def _extract_product(card, idx: int) -> AmazonProduct | None:
"""Extract product data from card"""
# ASIN
asin = card.get('data-asin', '')
if not asin:
logger.debug(f" ⏭️ Card {idx}: No ASIN")
return None
# Sponsored
sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]'))
# Title
title = None
for selector in ['h2 a span', 'h2 span', 'h2.s-line-clamp-2 span']:
elem = card.select_one(selector)
if elem:
title = elem.get_text(strip=True)
if title:
break
if not title:
logger.debug(f" ⏭️ Card {idx}: No title")
return None
# URL
link_elem = card.select_one('h2 a') or card.select_one('a.s-link-style')
if not link_elem:
logger.debug(f" ⏭️ Card {idx}: No link")
return None
href = link_elem.get('href', '')
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href
# Price
price = None
for selector in ['.a-price .a-offscreen', '.a-price-whole', 'span.a-price span.a-offscreen']:
elem = card.select_one(selector)
if elem:
price = parse_amazon_price(elem.get_text(strip=True))
if price:
break
# Original price
original_price = None
elem = card.select_one('.a-price.a-text-price .a-offscreen')
if elem:
original_price = parse_amazon_price(elem.get_text(strip=True))
# Rating
rating = None
for selector in ['[aria-label*="étoile"]', '[aria-label*="star"]']:
elem = card.select_one(selector)
if elem:
rating = parse_rating(elem.get('aria-label', ''))
if rating:
break
# Reviews
reviews_count = None
for selector in ['[aria-label*="étoile"] + span', 'span.s-underline-text']:
elem = card.select_one(selector)
if elem:
reviews_count = parse_reviews_count(elem.get_text(strip=True))
if reviews_count:
break
# Image
image_url = None
for selector in ['img.s-image', 'img']:
elem = card.select_one(selector)
if elem:
image_url = elem.get('src') or elem.get('data-src')
if image_url:
break
# Prime
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one('i.a-icon-prime'))
# Stock
in_stock = True
if card.select_one('[aria-label*="Indisponible"]'):
in_stock = False
return AmazonProduct(
title=title,
url=product_url,
price=price,
original_price=original_price,
rating=rating,
reviews_count=reviews_count,
image_url=image_url,
in_stock=in_stock,
prime=prime,
sponsored=sponsored,
)
# Global instance
amazon_scraper_service = AmazonScraperService()
+375
View File
@@ -0,0 +1,375 @@
"""
Amazon France Scraper - Using Browserless Service
Uses the existing browserless_service with Playwright for reliable scraping
"""
import logging
import random
import re
from typing import Any
from urllib.parse import quote_plus
from bs4 import BeautifulSoup
from pydantic import BaseModel, Field
from app.services.browserless_service import browserless_service
logger = logging.getLogger(__name__)
# Amazon France configuration
AMAZON_FR_BASE_URL = "https://www.amazon.fr"
AMAZON_FR_SEARCH_URL = "https://www.amazon.fr/s?k={query}"
# ============================================================================
# PYDANTIC SCHEMAS
# ============================================================================
class AmazonProduct(BaseModel):
"""Schema for Amazon product extraction"""
title: str = Field(description="Product title")
url: str = Field(description="Product URL")
price: float | None = Field(default=None, description="Price in EUR")
original_price: float | None = Field(default=None, description="Original price if discounted")
rating: float | None = Field(default=None, description="Product rating (0-5)")
reviews_count: int | None = Field(default=None, description="Number of reviews")
image_url: str | None = Field(default=None, description="Product image URL")
in_stock: bool = Field(default=True, description="Availability status")
prime: bool = Field(default=False, description="Prime eligible")
sponsored: bool = Field(default=False, description="Is sponsored")
# ============================================================================
# PRICE PARSING HELPERS
# ============================================================================
def parse_amazon_price(price_text: str) -> float | None:
"""
Parse Amazon price formats:
- "12,99 €"
- "12,99€"
- "12.99 EUR"
- "1 234,99 €"
"""
if not price_text:
return None
# Remove currency symbols and extra spaces
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
# Remove thousands separators (space or dot in French format)
cleaned = cleaned.replace(' ', '').replace('\xa0', '') # \xa0 is non-breaking space
# Replace comma with dot for decimal separator
cleaned = cleaned.replace(',', '.')
# Extract first number (in case of ranges like "12.99 - 15.99")
match = re.search(r'(\d+\.?\d*)', cleaned)
if match:
try:
return float(match.group(1))
except ValueError:
return None
return None
def parse_rating(rating_text: str) -> float | None:
"""Parse rating like '4,5 sur 5 étoiles' or '4.5 out of 5 stars'"""
if not rating_text:
return None
# Match patterns like "4,5" or "4.5"
match = re.search(r'(\d+[,.]\d+)', rating_text)
if match:
try:
return float(match.group(1).replace(',', '.'))
except ValueError:
return None
return None
def parse_reviews_count(reviews_text: str) -> int | None:
"""Parse review count like '1 234' or '12,345'"""
if not reviews_text:
return None
# Remove non-digit characters except spaces
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
try:
return int(cleaned)
except ValueError:
return None
# ============================================================================
# SCRAPING FUNCTIONS
# ============================================================================
async def scrape_amazon_search(query: str, max_results: int = 20) -> list[AmazonProduct]:
"""
Scrape Amazon France search results using Browserless service.
Args:
query: Search query
max_results: Maximum number of products to return (default 20)
Returns:
List of AmazonProduct objects
"""
search_url = AMAZON_FR_SEARCH_URL.format(query=quote_plus(query))
logger.info(f"🔍 Searching Amazon France: {query}")
logger.info(f"📍 URL: {search_url}")
products = []
try:
# STRATEGY: Load Amazon homepage FIRST to establish session/cookies
# Then do the search - appears more human-like
logger.info("🏠 Loading Amazon homepage first to establish session...")
home_html, _ = await browserless_service.get_page_content(
url="https://www.amazon.fr",
use_proxy=False,
wait_selector=None,
extract_text=False
)
if not home_html or len(home_html) < 10000:
logger.warning(f"⚠️ Homepage load failed ({len(home_html) if home_html else 0} bytes)")
else:
logger.info(f"✅ Homepage loaded ({len(home_html)} bytes) - cookies established")
# Small delay to appear more human
import asyncio
await asyncio.sleep(2)
# NOW do the search
logger.info("🚀 Fetching search page with Browserless (NO PROXY)...")
html_content, _ = await browserless_service.get_page_content(
url=search_url,
use_proxy=False, # Try without proxy first
wait_selector=None, # Let it load naturally
extract_text=False # We want HTML for parsing
)
if not html_content or len(html_content) < 10000:
logger.error(f"❌ Page too small ({len(html_content)} bytes) - likely blocked or empty")
return []
logger.info(f"✅ Page loaded successfully ({len(html_content)} bytes)")
# Debug: Save HTML to file for inspection
debug_file = f"/tmp/amazon_debug_{query[:20]}.html"
try:
with open(debug_file, 'w', encoding='utf-8') as f:
f.write(html_content)
logger.debug(f"📝 HTML saved to {debug_file} for debugging")
except Exception as e:
logger.debug(f"Could not save debug HTML: {e}")
# Parse HTML with BeautifulSoup
soup = BeautifulSoup(html_content, 'html.parser')
# Amazon uses data-component-type="s-search-result" for product cards
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
if not product_cards:
logger.warning("⚠️ No products found with primary selector")
# Try alternative selector
product_cards = soup.find_all('div', {'data-asin': True, 'data-index': True})
if product_cards:
logger.info(f"✓ Found {len(product_cards)} cards with alternative selector")
if not product_cards:
logger.warning("⚠️ No products found - checking for CAPTCHA or blocks")
# Check for CAPTCHA
if 'captcha' in html_content.lower():
logger.error("🚫 CAPTCHA detected - Amazon blocked the request")
elif 'robot' in html_content.lower() or 'bot' in html_content.lower():
logger.error("🤖 Bot detection triggered")
else:
logger.warning("📦 Empty results - query may have no matches")
return []
logger.info(f"📦 Found {len(product_cards)} product cards")
for idx, card in enumerate(product_cards):
if len(products) >= max_results:
break
try:
# Extract ASIN (Amazon Standard Identification Number)
asin = card.get('data-asin', '')
if not asin:
logger.debug(f" ⏭️ Card {idx}: No ASIN found, skipping")
continue
logger.debug(f" 🔍 Card {idx}: Processing ASIN {asin}")
# Check if sponsored
sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]'))
# Extract title - try multiple selectors
title = None
title_selectors = [
'h2 a span',
'h2 span',
'h2.s-line-clamp-2 span',
'.s-title-instructions-style span',
]
for selector in title_selectors:
title_elem = card.select_one(selector)
if title_elem:
title = title_elem.get_text(strip=True)
if title:
break
if not title:
logger.debug(f" ⏭️ Card {idx} ({asin}): No title found, skipping")
continue
# Extract URL
link_elem = card.select_one('h2 a')
if not link_elem:
# Try alternative
link_elem = card.select_one('a.s-link-style')
if not link_elem:
logger.debug(f" ⏭️ Card {idx} ({asin}): No link found, skipping")
continue
href = link_elem.get('href', '')
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href
# Extract price
price = None
original_price = None
# Current price - try multiple selectors
price_selectors = [
'.a-price .a-offscreen',
'.a-price-whole',
'span.a-price span.a-offscreen',
]
for selector in price_selectors:
price_elem = card.select_one(selector)
if price_elem:
price_text = price_elem.get_text(strip=True)
price = parse_amazon_price(price_text)
if price:
break
# Original price (if discounted)
original_price_elem = card.select_one('.a-price.a-text-price .a-offscreen')
if original_price_elem:
original_price = parse_amazon_price(original_price_elem.get_text(strip=True))
# Extract rating
rating = None
rating_selectors = [
'[aria-label*="étoile"]',
'[aria-label*="star"]',
'i.a-icon-star-small span',
]
for selector in rating_selectors:
rating_elem = card.select_one(selector)
if rating_elem:
aria_label = rating_elem.get('aria-label', '')
if aria_label:
rating = parse_rating(aria_label)
if rating:
break
# Extract reviews count
reviews_count = None
reviews_selectors = [
'[aria-label*="étoile"] + span',
'[aria-label*="star"] + span',
'span.s-underline-text',
]
for selector in reviews_selectors:
reviews_elem = card.select_one(selector)
if reviews_elem:
reviews_count = parse_reviews_count(reviews_elem.get_text(strip=True))
if reviews_count:
break
# Extract image
image_url = None
img_selectors = [
'img.s-image',
'img[data-image-latency="s-product-image"]',
'img',
]
for selector in img_selectors:
img_elem = card.select_one(selector)
if img_elem:
image_url = img_elem.get('src') or img_elem.get('data-src')
if image_url:
break
# Check Prime eligibility
prime = bool(card.select_one('[aria-label*="Prime"]') or card.select_one('i.a-icon-prime'))
# Check availability
in_stock = True
unavailable_elem = card.select_one('[aria-label*="Indisponible"]') or card.select_one('[aria-label*="Unavailable"]')
if unavailable_elem:
in_stock = False
product = AmazonProduct(
title=title,
url=product_url,
price=price,
original_price=original_price,
rating=rating,
reviews_count=reviews_count,
image_url=image_url,
in_stock=in_stock,
prime=prime,
sponsored=sponsored,
)
products.append(product)
logger.debug(f" ✓ [{len(products)}] {title[:50]}... - {price}€")
except Exception as e:
logger.error(f"❌ Error parsing product card {idx}: {e}")
continue
logger.info(f"✅ Successfully extracted {len(products)} products")
except Exception as e:
logger.error(f"❌ Error during Amazon scraping: {e}", exc_info=True)
return []
return products
# ============================================================================
# TESTING
# ============================================================================
async def test_amazon_scraper():
"""Test the Amazon scraper with a simple query"""
logger.info("=" * 60)
logger.info("Testing Amazon France Scraper (Browserless)")
logger.info("=" * 60)
test_query = "aspirateur"
products = await scrape_amazon_search(test_query, max_results=5)
logger.info(f"\n📊 Results for '{test_query}':")
logger.info(f"Found {len(products)} products\n")
for idx, product in enumerate(products, 1):
logger.info(f"{idx}. {product.title}")
logger.info(f" 💰 Price: {product.price}€" + (f" (was {product.original_price}€)" if product.original_price else ""))
logger.info(f" ⭐ Rating: {product.rating}/5 ({product.reviews_count} reviews)" if product.rating else " ⭐ No rating")
logger.info(f" 🔗 {product.url}")
logger.info(f" {'✅ Prime' if product.prime else '📦 Standard'} | {'📢 Sponsored' if product.sponsored else '🔍 Organic'}")
logger.info("")
return products
+643
View File
@@ -0,0 +1,643 @@
"""
Improved Search Service - Using Persistent Browser Connection
Based on ScraperService pattern for better session management and reliability
"""
import asyncio
import logging
from typing import AsyncGenerator
from urllib.parse import quote_plus, urljoin
from bs4 import BeautifulSoup
from playwright.async_api import Browser, BrowserContext, Page, async_playwright
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
from sqlalchemy.orm import Session
from app.core.search_config import SITE_CONFIGS, BROWSERLESS_URL
from app.models import SearchSite
from app.schemas import SearchProgress, SearchResultItem
logger = logging.getLogger(__name__)
# Common popup/cookie selectors
COMMON_POPUP_SELECTORS = [
"#sp-cc-accept", # Cookie banner
"#onetrust-accept-btn-handler", # OneTrust
".cookie-consent-accept",
"[data-action='accept-cookies']",
"button[id*='accept']",
"button[class*='accept']",
]
class SearchResult:
"""Search result data class"""
def __init__(
self,
url: str,
title: str,
snippet: str,
source: str,
price: float | None = None,
currency: str = "EUR",
in_stock: bool | None = None,
image_url: str | None = None,
):
self.url = url
self.title = title
self.snippet = snippet
self.source = source
self.price = price
self.currency = currency
self.in_stock = in_stock
self.image_url = image_url
def to_dict(self):
return {
"url": self.url,
"title": self.title,
"snippet": self.snippet,
"source": self.source,
"price": self.price,
"currency": self.currency,
"in_stock": self.in_stock,
"image_url": self.image_url,
}
class ImprovedSearchService:
"""Persistent browser service for e-commerce search scraping"""
_playwright = None
_browser: Browser | None = None
_lock = asyncio.Lock()
@classmethod
async def initialize(cls):
"""Initialize shared browser (Thread-Safe)"""
async with cls._lock:
await cls._initialize()
@classmethod
async def _initialize(cls):
"""Internal initialization"""
if cls._browser is None:
logger.info("Initializing ImprovedSearchService shared browser...")
cls._playwright = await async_playwright().start()
cls._browser = await cls._connect_browser(cls._playwright)
logger.info("ImprovedSearchService initialized.")
@classmethod
async def shutdown(cls):
"""Shutdown shared browser"""
async with cls._lock:
if cls._browser:
logger.info("Shutting down ImprovedSearchService...")
await cls._browser.close()
cls._browser = None
if cls._playwright:
await cls._playwright.stop()
cls._playwright = None
logger.info("ImprovedSearchService shutdown complete.")
@classmethod
async def _ensure_browser_connected(cls) -> bool:
"""Ensure browser is connected, reconnect if needed"""
async with cls._lock:
try:
if cls._browser is None:
logger.warning("Browser not initialized, initializing...")
await cls._initialize()
return cls._browser is not None
# Test connection
try:
test_context = await cls._browser.new_context()
await test_context.close()
return True
except Exception as e:
logger.error(f"Browser connection test failed: {e}")
logger.info("Attempting to reconnect...")
cls._browser = None
if cls._playwright:
try:
await cls._playwright.stop()
except Exception:
pass
cls._playwright = None
await cls._initialize()
return cls._browser is not None
except Exception as e:
logger.error(f"Failed to ensure browser connection: {e}")
return False
@staticmethod
async def _connect_browser(p) -> Browser:
"""Connect to Browserless"""
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
@staticmethod
async def _create_context(browser: Browser) -> BrowserContext:
"""Create browser context with stealth settings"""
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent=(
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
),
locale="fr-FR",
timezone_id="Europe/Paris",
)
# Stealth mode
await context.add_init_script("""
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
window.chrome = { runtime: {} };
""")
await context.route("**/*", lambda route: route.continue_())
return context
@staticmethod
async def _handle_popups(page: Page):
"""Close common popups/cookies"""
logger.debug("Handling popups...")
for selector in COMMON_POPUP_SELECTORS:
try:
if await page.locator(selector).count() > 0:
logger.debug(f"Found popup: {selector}")
await page.locator(selector).first.click(timeout=2000)
await page.wait_for_timeout(500)
except Exception:
pass
try:
await page.keyboard.press("Escape")
except Exception:
pass
@classmethod
async def search_site(cls, site_key: str, query: str) -> list[SearchResult]:
"""
Search a single site using persistent browser
Args:
site_key: Site configuration key
query: Search query
Returns:
List of SearchResult objects
"""
config = SITE_CONFIGS.get(site_key)
if not config:
logger.error(f"Unknown site: {site_key}")
return []
search_url = config["search_url"].format(query=quote_plus(query))
logger.info(f"🔍 Searching {config['name']} at {search_url}")
# Ensure browser is connected
if not await cls._ensure_browser_connected():
logger.error("Failed to establish browser connection")
return []
results = []
try:
context = await cls._create_context(cls._browser)
page = await context.new_page()
try:
# Navigate to search page
logger.debug(f"Navigating to {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
logger.debug("Page loaded (domcontentloaded)")
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.debug("Network idle reached")
except PlaywrightTimeoutError:
logger.debug("Network idle timed out (non-critical)")
# Handle popups
await cls._handle_popups(page)
# Wait for content to load
wait_selector = config.get("wait_selector")
if wait_selector:
try:
await page.wait_for_selector(wait_selector, timeout=5000)
logger.debug(f"Wait selector found: {wait_selector}")
except PlaywrightTimeoutError:
logger.warning(f"Wait selector not found: {wait_selector}")
# Small delay for JS rendering
await page.wait_for_timeout(2000)
# Get HTML content
html_content = await page.content()
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
if len(html_content) < 5000:
logger.warning(f"⚠️ Page too small - possibly blocked")
return []
# Parse results
results = cls._parse_results(html_content, site_key, search_url, query)
# Scrape details for each result (in parallel)
if results:
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
semaphore = asyncio.Semaphore(2) # Limit concurrency
async def scrape_with_limit(res):
async with semaphore:
return await cls._scrape_item_details(res, context)
tasks = [scrape_with_limit(r) for r in results]
enriched_results = await asyncio.gather(*tasks)
results = [r for r in enriched_results if r] # Filter None
finally:
await context.close()
except Exception as e:
logger.error(f"❌ Error during search: {e}", exc_info=True)
return []
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
return results
@staticmethod
def _parse_results(html: str, site_key: str, base_url: str, query: str) -> list[SearchResult]:
"""Parse HTML content to extract search results"""
config = SITE_CONFIGS[site_key]
soup = BeautifulSoup(html, "html.parser")
results = []
# Prepare query words for filtering
query_words = [w.lower() for w in query.split() if len(w) > 2]
# Select product links
links = soup.select(config["product_selector"])
# Deduplicate links
seen_urls = set()
for link in links:
href = link.get("href")
if not href:
continue
full_url = urljoin(base_url, href)
# Basic cleanup
if full_url in seen_urls:
continue
seen_urls.add(full_url)
# Extract title
title = link.get_text(strip=True)
# If no text, check title attribute or nested image alt
if not title:
if link.get("title"):
title = link.get("title")
else:
img = link.find("img")
if img and img.get("alt"):
title = img.get("alt")
if not title or len(title) < 3:
continue
# STRICT FILTERING: Check if all query words are in the title
title_lower = title.lower()
if query_words:
all_words_found = True
for word in query_words:
if word not in title_lower:
all_words_found = False
break
if not all_words_found:
continue
# Extract Image URL - Multiple strategies
image_url = None
# Strategy 1: Look for img directly in the link
img = link.find("img")
if img:
# Try multiple attributes (src, data-src, data-lazy-src, etc.)
image_url = (
img.get("src") or
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original") or
img.get("srcset", "").split(",")[0].split()[0] if img.get("srcset") else None
)
# Strategy 2: Look in parent container if configured
if not image_url and "product_image_selector" in config:
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img_el = container.select_one(config["product_image_selector"])
if img_el:
image_url = (
img_el.get("src") or
img_el.get("data-src") or
img_el.get("data-lazy-src")
)
# Strategy 3: Look for picture > source elements
if not image_url:
picture = link.find("picture")
if picture:
source = picture.find("source")
if source:
image_url = source.get("srcset", "").split(",")[0].split()[0] if source.get("srcset") else None
if not image_url:
img_in_picture = picture.find("img")
if img_in_picture:
image_url = img_in_picture.get("src") or img_in_picture.get("data-src")
# Clean up image URL
if image_url:
# Remove data URIs and 1x1 pixels
if image_url.startswith("data:") or "1x1" in image_url or "placeholder" in image_url.lower():
image_url = None
elif not image_url.startswith("http"):
image_url = urljoin(base_url, image_url)
# Create result
results.append(SearchResult(
url=full_url,
title=title,
snippet=f"Product from {config['name']}",
source=config["name"],
image_url=image_url
))
logger.debug(f"Parsed {len(results)} results from HTML")
return results
@classmethod
async def _scrape_item_details(cls, result: SearchResult, context: BrowserContext) -> SearchResult | None:
"""Scrape price and details for a single item using same context"""
try:
page = await context.new_page()
try:
logger.debug(f"Scraping details for: {result.title[:50]}...")
# Navigate to product page
await page.goto(result.url, wait_until="domcontentloaded", timeout=20000)
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=5000)
except PlaywrightTimeoutError:
pass
# Handle popups
await cls._handle_popups(page)
# Wait for content
await page.wait_for_timeout(1500)
# Extract price using multiple selectors
price = await cls._extract_price(page)
result.price = price
# Extract stock status
in_stock = await cls._extract_stock_status(page)
result.in_stock = in_stock
# Keep original image URL from search page - don't replace with screenshot
# (Screenshots would need to be served by FastAPI, and original images are already good)
logger.debug(f" ✓ {result.title[:40]}... - {price}€ - Stock: {in_stock}")
return result
finally:
await page.close()
except Exception as e:
logger.error(f"Error scraping item details {result.url}: {e}")
return result # Return original result without price
@staticmethod
async def _extract_price(page: Page) -> float | None:
"""Extract price from product page using multiple selectors"""
price_selectors = [
'.price',
'[data-testid="price"]',
'.prix-actuel',
'.price-current',
'[itemprop="price"]',
'.product-price',
'.a-price .a-offscreen',
'.a-price-whole',
'span[class*="price"]',
]
for selector in price_selectors:
try:
elements = await page.query_selector_all(selector)
for elem in elements:
price_text = await elem.inner_text()
if price_text:
# Parse price
import re
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
cleaned = cleaned.replace(',', '.')
match = re.search(r'(\d+\.?\d*)', cleaned)
if match:
try:
price_val = float(match.group(1))
# Validate price
if 0.01 < price_val < 100000:
logger.debug(f"Found price: {price_val}€ from {selector}")
return price_val
except ValueError:
continue
except Exception:
continue
logger.debug("No price found")
return None
@staticmethod
async def _extract_stock_status(page: Page) -> bool | None:
"""Extract stock status from product page"""
# Check for out of stock indicators
out_of_stock_texts = [
"rupture de stock",
"indisponible",
"out of stock",
"unavailable",
"épuisé",
"non disponible"
]
try:
page_text = await page.inner_text("body")
page_text_lower = page_text.lower()
for text in out_of_stock_texts:
if text in page_text_lower:
logger.debug(f"Out of stock detected: '{text}'")
return False
# If we find "add to cart" or similar, assume in stock
add_to_cart_texts = ["ajouter au panier", "add to cart", "acheter", "buy now"]
for text in add_to_cart_texts:
if text in page_text_lower:
logger.debug(f"In stock detected: '{text}'")
return True
except Exception as e:
logger.debug(f"Stock check error: {e}")
return None # Unknown
@classmethod
async def search_site_generator(cls, site_key: str, query: str) -> AsyncGenerator[SearchResult, None]:
"""Search a single site and yield results as they are scraped"""
results = await cls.search_site(site_key, query)
for result in results:
yield result
@classmethod
async def search_all(cls, query: str) -> list[SearchResult]:
"""Search all configured sites"""
tasks = []
for site_key in SITE_CONFIGS.keys():
tasks.append(cls.search_site(site_key, query))
results_list = await asyncio.gather(*tasks)
all_results = []
for r in results_list:
all_results.extend(r)
return all_results
# ==========================================
# COMPATIBILITY LAYER FOR API ROUTERS
# ==========================================
async def search_products(
query: str,
db: Session,
site_ids: list[int] | None = None,
max_results: int | None = None,
) -> AsyncGenerator[SearchProgress, None]:
"""
Compatibility wrapper for search_products using improved service.
Yields SearchProgress events incrementally.
"""
# 1. Get sites to search
sites = db.query(SearchSite).order_by(SearchSite.priority).all()
if site_ids:
sites = [s for s in sites if s.id in site_ids]
active_sites = [s for s in sites if s.is_active]
# Initial event
yield SearchProgress(
status="searching",
total=len(active_sites),
completed=0,
message=f"Démarrage de la recherche sur {len(active_sites)} sites...",
results=[],
)
# 2. Map DB sites to Config keys
site_keys = []
for site in active_sites:
for key in SITE_CONFIGS.keys():
if key in site.domain or site.domain in key:
site_keys.append(key)
break
# 3. Execute searches and stream results
generators = [ImprovedSearchService.search_site_generator(key, query) for key in site_keys]
queue = asyncio.Queue()
active_producers = len(generators)
# Limit concurrent sites
site_semaphore = asyncio.Semaphore(2)
async def producer(gen):
async with site_semaphore:
try:
async for item in gen:
await queue.put(item)
except Exception as e:
logger.error(f"Error in search producer: {e}")
finally:
await queue.put(None) # Sentinel
# Start producers
for gen in generators:
asyncio.create_task(producer(gen))
# Consumer loop
results_so_far = []
completed_sites = 0
while active_producers > 0:
item = await queue.get()
if item is None:
active_producers -= 1
completed_sites += 1
yield SearchProgress(
status="searching",
total=len(active_sites),
completed=completed_sites,
message=f"Recherche en cours... ({completed_sites}/{len(active_sites)} sites terminés)",
results=results_so_far,
)
else:
# Convert to SearchResultItem
api_item = SearchResultItem(
url=item.url,
title=item.title,
price=item.price,
currency=item.currency,
in_stock=item.in_stock,
site_name=item.source,
site_domain=item.source,
image_url=item.image_url,
)
results_so_far.append(api_item)
# Yield update with new result
yield SearchProgress(
status="searching",
total=len(active_sites),
completed=completed_sites,
message=f"Trouvé: {item.title[:30]}...",
results=results_so_far,
)
# Final event
yield SearchProgress(
status="completed",
total=len(active_sites),
completed=len(active_sites),
message=f"Terminé. {len(results_so_far)} résultats trouvés.",
results=results_so_far,
)
# Global instance
improved_search_service = ImprovedSearchService()
+27 -19
View File
@@ -96,18 +96,20 @@ class NewSearchService:
use_proxy = config.get("requires_proxy", False) if config else False
# Use browserless to get content and screenshot
html, screenshot_path = await browserless_service.get_page_content(
# extract_text=True to get visible text for AI analysis
page_text, screenshot_path = await browserless_service.get_page_content(
result.url,
use_proxy=use_proxy,
wait_selector=None
wait_selector=None,
extract_text=True # Get visible text for AI price extraction
)
if not screenshot_path:
return result
# Use AI to analyze
from app.services.ai_service import AIService
ai_result = await AIService.analyze_image(screenshot_path, page_text=html)
ai_result = await AIService.analyze_image(screenshot_path, page_text=page_text)
if ai_result:
extraction, _ = ai_result
@@ -300,7 +302,8 @@ class NewSearchService:
# Phase 2: Scrape details for each result (Parallel)
# We want to yield results as they complete, not wait for all
semaphore = asyncio.Semaphore(3) # Limit concurrency per site
# Reduced from 3 to 2 to avoid saturating Browserless
semaphore = asyncio.Semaphore(2) # Limit concurrency per site
async def scrape_wrapper(res):
async with semaphore:
@@ -453,32 +456,37 @@ async def search_products(
# 3. Execute searches and stream results
# We create a task for each site generator
generators = [NewSearchService.search_site_generator(key, query) for key in site_keys]
# We need to iterate over multiple async generators concurrently
# This is a bit complex, so we'll use a queue or similar
# Simpler approach: Use aiostream if available, or just interleave manually
# For now, let's just run them and yield as we get them.
# Since we want to show results ASAP, we can use asyncio.as_completed on the *next* item of each generator?
# No, generators are stateful.
# Simplest robust approach without extra libs:
# Create a wrapper task for each generator that puts items into a shared Queue
queue = asyncio.Queue()
active_producers = len(generators)
async def producer(gen):
try:
async for item in gen:
await queue.put(item)
except Exception as e:
logger.error(f"Error in search producer: {e}")
finally:
await queue.put(None) # Sentinel
# Start producers
# IMPORTANT: Limit concurrent sites to avoid saturating Browserless
# Max 2 sites can search in parallel, others wait
site_semaphore = asyncio.Semaphore(2)
async def producer(gen):
async with site_semaphore: # Wait for slot before starting search
try:
async for item in gen:
await queue.put(item)
except Exception as e:
logger.error(f"Error in search producer: {e}")
finally:
await queue.put(None) # Sentinel
# Start producers (limited by semaphore)
for gen in generators:
asyncio.create_task(producer(gen))
+3 -75
View File
@@ -9,81 +9,9 @@ from app.models import Enseigne
logger = logging.getLogger(__name__)
# Data for the 9 enseignes
ENSEIGNES_DATA = [
{
"nom": "Gifi",
"slug_bonial": "Gifi",
"couleur": "#E30613",
"site_url": "https://www.gifi.fr",
"description": "Décoration, maison, bazar",
"ordre_affichage": 1,
},
{
"nom": "Action",
"slug_bonial": "Action",
"couleur": "#0066B3",
"site_url": "https://www.action.com/fr-fr/",
"description": "Discount non-alimentaire",
"ordre_affichage": 2,
},
{
"nom": "Centrakor",
"slug_bonial": "Centrakor",
"couleur": "#E94E1B",
"site_url": "https://www.centrakor.com",
"description": "Décoration, maison",
"ordre_affichage": 3,
},
{
"nom": "La Foir'Fouille",
"slug_bonial": "La-Foir-Fouille",
"couleur": "#009639",
"site_url": "https://www.lafoirfouille.fr",
"description": "Bazar, décoration",
"ordre_affichage": 4,
},
{
"nom": "Stokomani",
"slug_bonial": "Stokomani",
"couleur": "#FF6600",
"site_url": "https://www.stokomani.fr",
"description": "Déstockage textile et maison",
"ordre_affichage": 5,
},
{
"nom": "B&M",
"slug_bonial": "BM",
"couleur": "#D4145A",
"site_url": "https://bmstores.fr",
"description": "Discount britannique",
"ordre_affichage": 6,
},
{
"nom": "L'Incroyable",
"slug_bonial": "L-incroyable",
"couleur": "#8B0000",
"site_url": "https://www.lincroyable.fr",
"description": "Décoration et mobilier discount (Groupe Althys, siège à Denain)",
"ordre_affichage": 7,
},
{
"nom": "Bazarland",
"slug_bonial": "Bazarland",
"couleur": "#FFCC00",
"site_url": None,
"description": "Bazar discount",
"ordre_affichage": 8,
},
{
"nom": "Noz",
"slug_bonial": "Noz",
"couleur": "#003366",
"site_url": "https://www.noz.fr",
"description": "Déstockage généraliste",
"ordre_affichage": 9,
},
]
# Liste des enseignes vidée - migration vers Amazon France uniquement
# Les magasins discount ne sont plus gérés via le système de catalogues
ENSEIGNES_DATA = []
def seed_enseignes(db: Session) -> int:
+2 -1
View File
@@ -8,7 +8,8 @@ from PIL import Image
logger = logging.getLogger(__name__)
# Image processing constants
MAX_IMAGE_SIZE = 1024
# Increased from 1024 to 2048 to preserve more detail for AI analysis
MAX_IMAGE_SIZE = 2048
JPEG_QUALITY = 85
+29
View File
@@ -69,6 +69,7 @@ def filter_relevant_text(text: str, max_length: int = 2000) -> str:
# Keywords to search for (case-insensitive)
price_keywords = [
# English patterns
r"\$\d+\.?\d*", # $XX.XX pattern
r"\d+\.\d{2}\s*(usd|eur|gbp|cad)", # XX.XX USD pattern
"price:",
@@ -78,9 +79,23 @@ def filter_relevant_text(text: str, max_length: int = 2000) -> str:
"save:",
"discount:",
r"\$", # Any dollar sign
# French patterns
r"€", # Euro symbol
r"\d+,\d{2}\s*€", # French format: 12,99 €
r"\d+\.\d{3},\d{2}", # French thousands: 1.234,56
r"\d+\s\d{3},\d{2}", # French thousands with space: 1 234,56
"prix", # French: price
"prix:",
"coût",
"coût:",
"promotion",
"réduction",
"économie",
"remise",
]
stock_keywords = [
# English keywords
"add to cart",
"buy now",
"purchase",
@@ -96,6 +111,20 @@ def filter_relevant_text(text: str, max_length: int = 2000) -> str:
"ships",
"delivery",
"get it by",
# French keywords
"ajouter au panier",
"acheter",
"commander",
"en stock",
"rupture",
"rupture de stock",
"disponible",
"indisponible",
"épuisé",
"être averti",
"précommande",
"livraison",
"expédié",
]
all_keywords = price_keywords + stock_keywords
+281
View File
@@ -0,0 +1,281 @@
# Amazon France Scraper - Documentation
## 🎯 Vue d'ensemble
Le scraper Amazon France est un système de recherche de produits conçu pour éviter la détection anti-bot d'Amazon. Il utilise Crawl4AI avec des techniques avancées d'anti-détection.
## 🛡️ Techniques anti-détection
### 1. User-Agent rotatif
- Pool de 5 User-Agents réalistes (Chrome, Firefox, Safari, Edge)
- Rotation aléatoire à chaque requête
- Headers complets mimant un vrai navigateur
### 2. Proxies résidentiels
- **10 proxies rotatifs** configurés
- Sélection aléatoire pour chaque recherche
- Format: `ip:port:username:password`
- Configuration dans `/app/core/search_config.py`
### 3. Délais aléatoires
- Entre **1.5 et 4 secondes** entre les requêtes
- Simule le comportement humain
- Évite les patterns de bot
### 4. Headers HTTP réalistes
```python
{
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,...",
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
"Accept-Encoding": "gzip, deflate, br",
"DNT": "1",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
...
}
```
### 5. Crawl4AI configuration
- `headless=True` - Mode invisible
- `--disable-blink-features=AutomationControlled` - Désactive la détection d'automation
- `wait_until="networkidle"` - Attend le chargement complet
- `remove_overlay_elements=True` - Supprime les popups
- Acceptation automatique des cookies
## 📦 Données extraites
Pour chaque produit, le scraper extrait :
| Champ | Type | Description |
|-------|------|-------------|
| `title` | string | Titre du produit |
| `url` | string | URL Amazon |
| `price` | float | Prix actuel en EUR |
| `original_price` | float | Prix original si promotion |
| `rating` | float | Note sur 5 étoiles |
| `reviews_count` | int | Nombre d'avis |
| `image_url` | string | URL de l'image |
| `in_stock` | bool | Disponibilité |
| `prime` | bool | Éligible Prime |
| `sponsored` | bool | Produit sponsorisé |
## 🚀 Utilisation
### Backend (Python)
```python
from app.services.amazon_scraper import scrape_amazon_search
# Recherche simple
products = await scrape_amazon_search("aspirateur", max_results=20)
for product in products:
print(f"{product.title} - {product.price}€")
```
### API REST
```bash
# Endpoint SSE (Server-Sent Events)
GET /api/amazon/search?q=aspirateur&max_results=20
# Health check
GET /api/amazon/health
```
### Frontend (React)
```javascript
// EventSource pour SSE
const eventSource = new EventSource(`/api/amazon/search?q=${query}&max_results=20`);
eventSource.addEventListener('progress', (event) => {
const data = JSON.parse(event.data);
// data.status: 'searching', 'completed', 'error'
// data.results: array of products
});
```
Accès direct : **http://localhost/amazon** (après connexion)
## 🧪 Tests
### Script de test complet
```bash
# Lancer tous les tests
python test_amazon_scraper.py
```
Tests inclus :
1. ✅ Recherche basique (5 produits)
2. ✅ Requêtes multiples (clavier, souris, casque)
3. ✅ Vérification anti-détection
### Test manuel simple
```python
import asyncio
from app.services.amazon_scraper import test_amazon_scraper
asyncio.run(test_amazon_scraper())
```
## ⚙️ Configuration
### Proxies
Modifiez `/app/core/search_config.py` :
```python
AMAZON_PROXY_LIST_RAW = [
"ip1:port1:user1:pass1",
"ip2:port2:user2:pass2",
# ... ajoutez vos proxies
]
```
### User-Agents
Ajoutez dans `/app/services/amazon_scraper.py` :
```python
AMAZON_USER_AGENTS = [
"Mozilla/5.0 (Windows NT 10.0; ...) Chrome/131.0.0.0",
# ... ajoutez vos user-agents
]
```
### Délais
Modifiez la fonction `random_delay()` :
```python
async def random_delay(min_seconds=1.5, max_seconds=4.0):
delay = random.uniform(min_seconds, max_seconds)
await asyncio.sleep(delay)
```
## 🔍 Sélecteurs CSS Amazon
Le scraper utilise les sélecteurs suivants (mis à jour pour 2024) :
```python
# Cartes produits
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
# Titre
title_elem = card.select_one('h2 a span, h2 span')
# Prix actuel
price_elem = card.select_one('.a-price .a-offscreen')
# Prix original (promo)
original_price_elem = card.select_one('.a-price.a-text-price .a-offscreen')
# Note
rating_elem = card.select_one('[aria-label*="étoile"], [aria-label*="star"]')
# Nombre d'avis
reviews_elem = card.select_one('[aria-label*="étoile"] + span')
# Image
img_elem = card.select_one('img.s-image')
# Prime
prime = card.select_one('[aria-label*="Prime"], i.a-icon-prime')
```
## 📊 Performances
- **Vitesse** : ~3-5 secondes pour 20 produits
- **Taux de succès** : ~95% (avec proxies)
- **Limite recommandée** : Max 20 produits par requête
- **Délai entre requêtes** : 2-5 secondes
## ⚠️ Limitations connues
1. **CAPTCHA** : Peut survenir en cas d'utilisation intensive
- Solution : Rotation des proxies + délais plus longs
2. **Géolocalisation** : Les proxies doivent être français/européens
- Amazon.fr peut bloquer les IPs non-européennes
3. **Structure HTML** : Amazon peut modifier ses sélecteurs
- Vérifier régulièrement les sélecteurs CSS
4. **Rate limiting** : Amazon limite les requêtes par IP
- Utiliser les proxies rotatifs
## 🐛 Debugging
### Logs
Les logs détaillés sont disponibles dans la console :
```python
logger.info(f"🔍 Searching Amazon France: {query}")
logger.info(f"📍 URL: {search_url}")
logger.debug(f"🎭 User-Agent: {user_agent}")
logger.debug(f"🌐 Using proxy: {proxy['server']}")
```
### Messages d'erreur courants
| Erreur | Cause | Solution |
|--------|-------|----------|
| `CAPTCHA detected` | Trop de requêtes | Attendre + changer de proxy |
| `Bot detection triggered` | Mauvais User-Agent | Vérifier USER_AGENTS |
| `No products found` | Requête vide ou blocage | Vérifier la recherche |
| `Timeout` | Connexion lente | Augmenter `page_timeout` |
### Mode debug Crawl4AI
```python
browser_config = BrowserConfig(
headless=False, # Voir le navigateur
verbose=True, # Logs détaillés
...
)
```
## 📈 Évolutions futures
- [ ] Cache Redis pour éviter les requêtes répétées
- [ ] Pagination automatique (>20 produits)
- [ ] Détection automatique de CAPTCHA
- [ ] Résolution de CAPTCHA (service tiers)
- [ ] Scraping des détails produit (description, specs)
- [ ] Support Amazon.de, Amazon.es, etc.
- [ ] Monitoring des prix en temps réel
- [ ] Alertes de baisse de prix
## 🔐 Sécurité & Légalité
⚠️ **Important** : Ce scraper est destiné à un usage personnel uniquement.
- ✅ Usage personnel/éducatif
- ✅ Recherche de produits
- ✅ Comparaison de prix
- ❌ Revente de données
- ❌ Usage commercial intensif
- ❌ Contournement de CAPTCHA à grande échelle
Respectez les [Conditions d'utilisation Amazon](https://www.amazon.fr/gp/help/customer/display.html?nodeId=201909000).
## 📞 Support
En cas de problème :
1. Vérifier les logs (`logger.info/debug/error`)
2. Tester avec le script `test_amazon_scraper.py`
3. Vérifier la configuration des proxies
4. Consulter la [documentation Crawl4AI](https://crawl4ai.com/)
## 🎉 Crédits
- **Crawl4AI** : Framework de scraping IA
- **BeautifulSoup** : Parsing HTML
- **FastAPI** : API backend
- **React** : Interface frontend
- **Shadcn UI** : Composants UI
+2
View File
@@ -5,6 +5,7 @@ import Layout from '@/components/layout/Layout'
import Dashboard from '@/pages/Dashboard'
import Search from '@/pages/Search'
import MultiSearch from '@/pages/MultiSearch'
import AmazonSearch from '@/pages/AmazonSearch'
import Catalogues from '@/pages/Catalogues'
import Login from '@/pages/Login'
import Admin from '@/pages/Admin'
@@ -78,6 +79,7 @@ function AppRoutes() {
<Route path="/" element={<Dashboard />} />
<Route path="/search" element={<Search />} />
<Route path="/compare" element={<MultiSearch />} />
<Route path="/amazon" element={<AmazonSearch />} />
<Route path="/catalogues" element={<Catalogues />} />
<Route
path="/admin"
+2 -1
View File
@@ -1,7 +1,7 @@
import React, { useState } from 'react';
import { Link, useLocation, useNavigate } from 'react-router-dom';
import { useTranslation } from 'react-i18next';
import { LayoutDashboard, Search, TrendingUp, BookOpen, Moon, Sun, Menu, LogOut, Shield } from 'lucide-react';
import { LayoutDashboard, Search, TrendingUp, BookOpen, Moon, Sun, Menu, LogOut, Shield, ShoppingBag } from 'lucide-react';
import { Button } from '@/components/ui/button';
import { Sheet, SheetContent, SheetTrigger } from '@/components/ui/sheet';
import { cn } from '@/lib/utils';
@@ -22,6 +22,7 @@ const Layout = ({ children, theme, toggleTheme }) => {
const navItems = [
{ icon: LayoutDashboard, label: t('nav.dashboard'), path: '/' },
{ icon: Search, label: t('nav.search') || 'Recherche', path: '/search' },
{ icon: ShoppingBag, label: 'Amazon France', path: '/amazon' },
{ icon: TrendingUp, label: 'Comparateur', path: '/compare' },
{ icon: BookOpen, label: 'Catalogues', path: '/catalogues' },
];
+323
View File
@@ -0,0 +1,323 @@
import React, { useState, useRef } from 'react';
import { toast } from 'sonner';
import { useTranslation } from 'react-i18next';
import { Search as SearchIcon, Loader2, Star, ShoppingCart, ExternalLink, Tag } from 'lucide-react';
import { Button } from '@/components/ui/button';
import { Input } from '@/components/ui/input';
import { Card, CardContent } from '@/components/ui/card';
import { Progress } from '@/components/ui/progress';
import { Badge } from '@/components/ui/badge';
const API_URL = '/api';
export default function AmazonSearch() {
const { t } = useTranslation();
const [query, setQuery] = useState('');
const [isSearching, setIsSearching] = useState(false);
const [progress, setProgress] = useState(null);
const [results, setResults] = useState([]);
const eventSourceRef = useRef(null);
const handleSearch = async (e) => {
e.preventDefault();
if (!query.trim()) {
toast.error('Veuillez entrer un terme de recherche');
return;
}
// Fermer l'EventSource précédent si existant
if (eventSourceRef.current) {
eventSourceRef.current.close();
}
setIsSearching(true);
setResults([]);
setProgress({ status: 'searching', total: 1, completed: 0 });
// Construire l'URL avec les paramètres
const params = new URLSearchParams({
q: query.trim(),
max_results: 20,
});
// Créer l'EventSource pour SSE
const eventSource = new EventSource(`${API_URL}/amazon/search?${params}`);
eventSourceRef.current = eventSource;
eventSource.addEventListener('progress', (event) => {
try {
const data = JSON.parse(event.data);
setProgress(data);
setResults(data.results || []);
if (data.status === 'completed' || data.status === 'error') {
setIsSearching(false);
eventSource.close();
if (data.status === 'completed') {
toast.success(data.message);
} else if (data.message) {
toast.error(data.message);
}
}
} catch (error) {
console.error('Error parsing SSE data:', error);
}
});
eventSource.onerror = () => {
setIsSearching(false);
eventSource.close();
toast.error('Erreur de connexion au serveur');
};
};
const progressPercent = progress
? Math.round((progress.completed / Math.max(progress.total, 1)) * 100)
: 0;
const formatPrice = (price, originalPrice) => {
if (!price) return 'Prix indisponible';
const formattedPrice = price.toFixed(2);
if (originalPrice && originalPrice > price) {
const discount = Math.round(((originalPrice - price) / originalPrice) * 100);
return (
<div className="flex items-baseline gap-2">
<span className="text-2xl font-bold text-green-600">{formattedPrice}€</span>
<span className="text-sm text-muted-foreground line-through">{originalPrice.toFixed(2)}€</span>
<Badge variant="destructive" className="text-xs">-{discount}%</Badge>
</div>
);
}
return <span className="text-2xl font-bold">{formattedPrice}€</span>;
};
const renderRating = (rating, reviewsCount) => {
if (!rating) return null;
return (
<div className="flex items-center gap-1">
<div className="flex">
{[...Array(5)].map((_, i) => (
<Star
key={i}
className={`h-4 w-4 ${
i < Math.floor(rating)
? 'fill-yellow-400 text-yellow-400'
: 'text-gray-300'
}`}
/>
))}
</div>
<span className="text-sm font-medium">{rating.toFixed(1)}</span>
{reviewsCount && (
<span className="text-sm text-muted-foreground">
({reviewsCount.toLocaleString('fr-FR')})
</span>
)}
</div>
);
};
return (
<div className="space-y-6 animate-in fade-in duration-500">
{/* Header */}
<div className="flex flex-col gap-4 md:flex-row md:items-center md:justify-between">
<div>
<div className="flex items-center gap-3">
<h2 className="text-2xl font-bold tracking-tight">Amazon France</h2>
<Badge variant="outline" className="bg-orange-50 border-orange-200">
🇫🇷 France
</Badge>
</div>
<p className="text-muted-foreground mt-1">
Recherchez parmi des millions de produits sur Amazon.fr
</p>
</div>
</div>
{/* Search Form */}
<Card>
<CardContent className="pt-6">
<form onSubmit={handleSearch} className="space-y-4">
<div className="flex gap-2">
<div className="relative flex-1">
<SearchIcon className="absolute left-3 top-3 h-4 w-4 text-muted-foreground" />
<Input
type="text"
placeholder="Rechercher un produit sur Amazon..."
value={query}
onChange={(e) => setQuery(e.target.value)}
className="pl-9"
disabled={isSearching}
/>
</div>
<Button type="submit" disabled={isSearching || !query.trim()}>
{isSearching ? (
<>
<Loader2 className="h-4 w-4 mr-2 animate-spin" />
Recherche...
</>
) : (
<>
<SearchIcon className="h-4 w-4 mr-2" />
Rechercher
</>
)}
</Button>
</div>
{/* Info */}
<div className="flex items-center gap-2 text-sm text-muted-foreground">
<div className="flex items-center gap-1">
<div className="h-2 w-2 rounded-full bg-green-500"></div>
<span>Anti-détection activé</span>
</div>
<span>•</span>
<span>Max 20 résultats</span>
</div>
</form>
</CardContent>
</Card>
{/* Progress Bar */}
{progress && isSearching && (
<Card>
<CardContent className="pt-6">
<div className="space-y-2">
<div className="flex justify-between text-sm">
<span>{progress.message || 'Recherche en cours...'}</span>
<span>{progressPercent}%</span>
</div>
<Progress value={progressPercent} />
</div>
</CardContent>
</Card>
)}
{/* Results */}
{results.length > 0 && (
<div className="space-y-4">
<div className="flex items-center justify-between">
<h3 className="text-lg font-semibold">
{results.length} produit{results.length > 1 ? 's' : ''} trouvé{results.length > 1 ? 's' : ''}
</h3>
</div>
<div className="grid gap-4 sm:grid-cols-2 lg:grid-cols-3 xl:grid-cols-4">
{results.map((product, index) => (
<Card key={index} className="overflow-hidden hover:shadow-lg transition-shadow">
<CardContent className="p-0">
{/* Image */}
{product.image_url && (
<div className="relative aspect-square bg-gray-50 flex items-center justify-center p-4">
<img
src={product.image_url}
alt={product.title}
className="max-h-full max-w-full object-contain"
/>
{product.sponsored && (
<Badge
variant="secondary"
className="absolute top-2 left-2 text-xs"
>
Sponsorisé
</Badge>
)}
{!product.in_stock && (
<Badge
variant="destructive"
className="absolute top-2 right-2 text-xs"
>
Indisponible
</Badge>
)}
</div>
)}
<div className="p-4 space-y-3">
{/* Title */}
<h4 className="font-medium text-sm line-clamp-2 min-h-[2.5rem]">
{product.title}
</h4>
{/* Rating */}
{renderRating(product.rating, product.reviews_count)}
{/* Price */}
<div className="pt-2">
{formatPrice(product.price, product.original_price)}
</div>
{/* Badges */}
<div className="flex flex-wrap gap-1">
{product.prime && (
<Badge variant="default" className="bg-blue-500 text-xs">
Prime
</Badge>
)}
{product.in_stock && (
<Badge variant="outline" className="text-xs text-green-600 border-green-600">
En stock
</Badge>
)}
</div>
{/* Actions */}
<div className="pt-2 flex gap-2">
<Button
variant="default"
size="sm"
className="flex-1"
onClick={() => window.open(product.url, '_blank')}
>
<ShoppingCart className="h-4 w-4 mr-1" />
Voir
</Button>
<Button
variant="outline"
size="sm"
onClick={() => window.open(product.url, '_blank')}
>
<ExternalLink className="h-4 w-4" />
</Button>
</div>
</div>
</CardContent>
</Card>
))}
</div>
</div>
)}
{/* Empty State */}
{!isSearching && results.length === 0 && !progress && (
<Card>
<CardContent className="py-12 text-center">
<div className="mx-auto mb-4 h-16 w-16 rounded-full bg-orange-50 flex items-center justify-center">
<SearchIcon className="h-8 w-8 text-orange-500" />
</div>
<h3 className="text-lg font-medium mb-2">Recherchez sur Amazon France</h3>
<p className="text-muted-foreground max-w-md mx-auto">
Entrez un terme de recherche pour trouver des produits sur Amazon.fr.
Notre système anti-détection garantit un accès fiable aux résultats.
</p>
</CardContent>
</Card>
)}
{/* No Results */}
{!isSearching && results.length === 0 && progress?.status === 'completed' && (
<Card>
<CardContent className="py-12 text-center">
<SearchIcon className="h-12 w-12 mx-auto text-muted-foreground mb-4" />
<h3 className="text-lg font-medium mb-2">Aucun produit trouvé</h3>
<p className="text-muted-foreground">
Essayez avec d'autres termes de recherche
</p>
</CardContent>
</Card>
)}
</div>
);
}
+173
View File
@@ -0,0 +1,173 @@
#!/usr/bin/env python3
"""
Script de test pour le scraper Amazon France
Teste le système anti-détection et l'extraction des produits
"""
import asyncio
import logging
import sys
from pathlib import Path
# Ajouter le répertoire app au path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.amazon_scraper import (
scrape_amazon_search,
test_amazon_scraper,
)
# Configuration du logging
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
)
logger = logging.getLogger(__name__)
async def test_basic_search():
"""Test basique de recherche"""
logger.info("=" * 80)
logger.info("TEST 1: Recherche basique - 'aspirateur'")
logger.info("=" * 80)
products = await scrape_amazon_search("aspirateur", max_results=5)
if not products:
logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau")
return False
logger.info(f"✅ {len(products)} produits trouvés")
for idx, product in enumerate(products, 1):
logger.info(f"\n{idx}. {product.title[:60]}...")
logger.info(f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else ""))
logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note")
logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}")
logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}")
logger.info(f" {'📢 Sponsorisé' if product.sponsored else '🔍 Organique'}")
return True
async def test_multiple_queries():
"""Test avec plusieurs requêtes différentes"""
logger.info("\n" + "=" * 80)
logger.info("TEST 2: Requêtes multiples")
logger.info("=" * 80)
queries = ["clavier", "souris", "casque"]
results = {}
for query in queries:
logger.info(f"\n🔍 Recherche: '{query}'")
products = await scrape_amazon_search(query, max_results=3)
results[query] = len(products)
logger.info(f" ✅ {len(products)} produits trouvés")
# Délai entre requêtes pour respecter les bonnes pratiques
await asyncio.sleep(3)
logger.info("\n📊 Résumé:")
for query, count in results.items():
logger.info(f" • {query}: {count} produits")
total = sum(results.values())
if total > 0:
logger.info(f"\n✅ Total: {total} produits extraits")
return True
else:
logger.error("\n❌ Aucun produit extrait - problème possible")
return False
async def test_anti_detection():
"""Test du système anti-détection"""
logger.info("\n" + "=" * 80)
logger.info("TEST 3: Vérification anti-détection")
logger.info("=" * 80)
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_POOL
from app.services.amazon_scraper import get_random_proxy, AMAZON_USER_AGENTS
logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles")
logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards")
logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques")
# Test proxy
proxy = get_random_proxy()
if proxy:
# Extract just the IP for logging (hide credentials)
proxy_parts = proxy.split('@')
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
logger.info(f"✓ Proxy test: {proxy_server}")
else:
logger.warning("⚠️ Pas de proxy configuré")
# Test d'une recherche simple
logger.info("\n🧪 Test de recherche avec anti-détection...")
products = await scrape_amazon_search("livre", max_results=3)
if products:
logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits")
return True
else:
logger.error("❌ Échec - possibilité de blocage")
return False
async def run_all_tests():
"""Lance tous les tests"""
logger.info("\n" + "=" * 80)
logger.info("🚀 DÉMARRAGE DES TESTS DU SCRAPER AMAZON FRANCE")
logger.info("=" * 80)
tests = [
("Recherche basique", test_basic_search),
("Requêtes multiples", test_multiple_queries),
("Anti-détection", test_anti_detection),
]
results = {}
for test_name, test_func in tests:
try:
logger.info(f"\n▶️ Exécution: {test_name}")
success = await test_func()
results[test_name] = "✅ PASS" if success else "❌ FAIL"
except Exception as e:
logger.error(f"❌ Erreur dans {test_name}: {e}", exc_info=True)
results[test_name] = "❌ ERROR"
# Résumé final
logger.info("\n" + "=" * 80)
logger.info("📊 RÉSUMÉ DES TESTS")
logger.info("=" * 80)
for test_name, result in results.items():
logger.info(f"{result} - {test_name}")
passed = sum(1 for r in results.values() if "PASS" in r)
total = len(results)
logger.info(f"\n🎯 Score: {passed}/{total} tests réussis")
if passed == total:
logger.info("✅ TOUS LES TESTS ONT RÉUSSI!")
return True
else:
logger.warning("⚠️ Certains tests ont échoué")
return False
if __name__ == "__main__":
try:
success = asyncio.run(run_all_tests())
sys.exit(0 if success else 1)
except KeyboardInterrupt:
logger.info("\n⏸️ Tests interrompus par l'utilisateur")
sys.exit(130)
except Exception as e:
logger.error(f"❌ Erreur fatale: {e}", exc_info=True)
sys.exit(1)