Merge pull request #158 from R0m1k3/claude/amazon-france-search-page-01KmPqbdPCqxWXxEoFw9jgFo

Claude/amazon france search page 01 km pqbd p cqx w xx eo fw9jg fo
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 10:36:26 +01:00
commit 7959e4e08a
2 files changed
+18 -11

No files matched your search

+1 -1
View File
@@ -20,7 +20,7 @@ from app.services.scheduler import start_scheduler as start_catalog_scheduler, s
# Configure logging
logging.basicConfig(
level=os.getenv("LOG_LEVEL", "INFO").upper(),
level=os.getenv("LOG_LEVEL", "DEBUG").upper(), # Temporarily DEBUG for Amazon debugging
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
)
logger = logging.getLogger(__name__)
+17 -10
View File
@@ -207,22 +207,15 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
user_agent = random.choice(AMAZON_USER_AGENTS)
logger.debug(f"🎭 User-Agent: {user_agent[:50]}...")
# Select random proxy
proxy = get_random_proxy()
if proxy:
# Extract just the IP for logging (hide credentials)
proxy_parts = proxy.split('@')
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
logger.debug(f"🌐 Using proxy: {proxy_server}")
else:
logger.warning("⚠️ No proxy available - may face rate limiting")
# NOTE: Not using proxies here - Browserless service has its own proxy system
# We can integrate with browserless_service later if needed
# Configure Crawl4AI browser with anti-detection
browser_config = BrowserConfig(
headless=True,
verbose=False,
user_agent=user_agent,
proxy_config=proxy, # Use proxy_config instead of deprecated proxy
# proxy_config removed - let Crawl4AI use default or integrate with Browserless later
extra_args=[
"--disable-blink-features=AutomationControlled", # Disable automation detection
"--disable-dev-shm-usage",
@@ -277,6 +270,15 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
# Parse HTML with BeautifulSoup
soup = BeautifulSoup(result.html, 'html.parser')
# Debug: Save HTML to file for inspection
debug_file = f"/tmp/amazon_debug_{query[:20]}.html"
try:
with open(debug_file, 'w', encoding='utf-8') as f:
f.write(result.html)
logger.debug(f"📝 HTML saved to {debug_file} for debugging")
except Exception as e:
logger.debug(f"Could not save debug HTML: {e}")
# Amazon uses data-component-type="s-search-result" for product cards
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
@@ -301,20 +303,25 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
# Extract ASIN (Amazon Standard Identification Number)
asin = card.get('data-asin', '')
if not asin:
logger.debug(f" ⏭️ Card {idx}: No ASIN found, skipping")
continue
logger.debug(f" 🔍 Card {idx}: Processing ASIN {asin}")
# Check if sponsored
sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]'))
# Extract title
title_elem = card.select_one('h2 a span, h2 span')
if not title_elem:
logger.debug(f" ⏭️ Card {idx} ({asin}): No title found, skipping")
continue
title = title_elem.get_text(strip=True)
# Extract URL
link_elem = card.select_one('h2 a')
if not link_elem:
logger.debug(f" ⏭️ Card {idx} ({asin}): No link found, skipping")
continue
href = link_elem.get('href', '')
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href