From 55276149a4db295b15a3e4c84556e65fb37edbb5 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 30 Nov 2025 09:32:45 +0000 Subject: [PATCH 1/2] debug: Add detailed logging and HTML dump for Amazon scraping - Add debug logs for each step of product extraction - Log ASIN, title, link extraction failures - Save HTML to /tmp/amazon_debug_*.html for inspection - Enable DEBUG logging level temporarily - Will help identify why 48 cards found but 0 products extracted --- app/main.py | 2 +- app/services/amazon_scraper.py | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/app/main.py b/app/main.py index 0e35c95..ea4d695 100644 --- a/app/main.py +++ b/app/main.py @@ -20,7 +20,7 @@ from app.services.scheduler import start_scheduler as start_catalog_scheduler, s # Configure logging logging.basicConfig( - level=os.getenv("LOG_LEVEL", "INFO").upper(), + level=os.getenv("LOG_LEVEL", "DEBUG").upper(), # Temporarily DEBUG for Amazon debugging format="%(asctime)s - %(name)s - %(levelname)s - %(message)s", ) logger = logging.getLogger(__name__) diff --git a/app/services/amazon_scraper.py b/app/services/amazon_scraper.py index 69ddb8c..cc70e1b 100644 --- a/app/services/amazon_scraper.py +++ b/app/services/amazon_scraper.py @@ -277,6 +277,15 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon # Parse HTML with BeautifulSoup soup = BeautifulSoup(result.html, 'html.parser') + # Debug: Save HTML to file for inspection + debug_file = f"/tmp/amazon_debug_{query[:20]}.html" + try: + with open(debug_file, 'w', encoding='utf-8') as f: + f.write(result.html) + logger.debug(f"📝 HTML saved to {debug_file} for debugging") + except Exception as e: + logger.debug(f"Could not save debug HTML: {e}") + # Amazon uses data-component-type="s-search-result" for product cards product_cards = soup.find_all('div', {'data-component-type': 's-search-result'}) @@ -301,20 +310,25 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon # Extract ASIN (Amazon Standard Identification Number) asin = card.get('data-asin', '') if not asin: + logger.debug(f" ⏭️ Card {idx}: No ASIN found, skipping") continue + logger.debug(f" 🔍 Card {idx}: Processing ASIN {asin}") + # Check if sponsored sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]')) # Extract title title_elem = card.select_one('h2 a span, h2 span') if not title_elem: + logger.debug(f" ⏭️ Card {idx} ({asin}): No title found, skipping") continue title = title_elem.get_text(strip=True) # Extract URL link_elem = card.select_one('h2 a') if not link_elem: + logger.debug(f" ⏭️ Card {idx} ({asin}): No link found, skipping") continue href = link_elem.get('href', '') product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href From a9480a47b3081b3beb682b02afc000400a65986c Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 30 Nov 2025 09:33:41 +0000 Subject: [PATCH 2/2] refactor: Remove Crawl4AI proxy config, rely on Browserless system - Removed proxy_config from BrowserConfig to avoid conflicts - Browserless service already has integrated proxy rotation - Will integrate with browserless_service later if needed - Focus on fixing CSS selectors first --- app/services/amazon_scraper.py | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/app/services/amazon_scraper.py b/app/services/amazon_scraper.py index cc70e1b..b64dc50 100644 --- a/app/services/amazon_scraper.py +++ b/app/services/amazon_scraper.py @@ -207,22 +207,15 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon user_agent = random.choice(AMAZON_USER_AGENTS) logger.debug(f"🎭 User-Agent: {user_agent[:50]}...") - # Select random proxy - proxy = get_random_proxy() - if proxy: - # Extract just the IP for logging (hide credentials) - proxy_parts = proxy.split('@') - proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy - logger.debug(f"🌐 Using proxy: {proxy_server}") - else: - logger.warning("⚠️ No proxy available - may face rate limiting") + # NOTE: Not using proxies here - Browserless service has its own proxy system + # We can integrate with browserless_service later if needed # Configure Crawl4AI browser with anti-detection browser_config = BrowserConfig( headless=True, verbose=False, user_agent=user_agent, - proxy_config=proxy, # Use proxy_config instead of deprecated proxy + # proxy_config removed - let Crawl4AI use default or integrate with Browserless later extra_args=[ "--disable-blink-features=AutomationControlled", # Disable automation detection "--disable-dev-shm-usage",