mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-12 01:39:25 +02:00
debug: Add detailed logging and HTML dump for Amazon scraping
- Add debug logs for each step of product extraction - Log ASIN, title, link extraction failures - Save HTML to /tmp/amazon_debug_*.html for inspection - Enable DEBUG logging level temporarily - Will help identify why 48 cards found but 0 products extracted
This commit is contained in:
2 files changed
+15
-1
No files matched your search
+1
-1
@@ -20,7 +20,7 @@ from app.services.scheduler import start_scheduler as start_catalog_scheduler, s
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=os.getenv("LOG_LEVEL", "INFO").upper(),
|
||||
level=os.getenv("LOG_LEVEL", "DEBUG").upper(), # Temporarily DEBUG for Amazon debugging
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -277,6 +277,15 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
|
||||
# Parse HTML with BeautifulSoup
|
||||
soup = BeautifulSoup(result.html, 'html.parser')
|
||||
|
||||
# Debug: Save HTML to file for inspection
|
||||
debug_file = f"/tmp/amazon_debug_{query[:20]}.html"
|
||||
try:
|
||||
with open(debug_file, 'w', encoding='utf-8') as f:
|
||||
f.write(result.html)
|
||||
logger.debug(f"📝 HTML saved to {debug_file} for debugging")
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not save debug HTML: {e}")
|
||||
|
||||
# Amazon uses data-component-type="s-search-result" for product cards
|
||||
product_cards = soup.find_all('div', {'data-component-type': 's-search-result'})
|
||||
|
||||
@@ -301,20 +310,25 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
|
||||
# Extract ASIN (Amazon Standard Identification Number)
|
||||
asin = card.get('data-asin', '')
|
||||
if not asin:
|
||||
logger.debug(f" ⏭️ Card {idx}: No ASIN found, skipping")
|
||||
continue
|
||||
|
||||
logger.debug(f" 🔍 Card {idx}: Processing ASIN {asin}")
|
||||
|
||||
# Check if sponsored
|
||||
sponsored = bool(card.select_one('[data-component-type="sp-sponsored-result"]'))
|
||||
|
||||
# Extract title
|
||||
title_elem = card.select_one('h2 a span, h2 span')
|
||||
if not title_elem:
|
||||
logger.debug(f" ⏭️ Card {idx} ({asin}): No title found, skipping")
|
||||
continue
|
||||
title = title_elem.get_text(strip=True)
|
||||
|
||||
# Extract URL
|
||||
link_elem = card.select_one('h2 a')
|
||||
if not link_elem:
|
||||
logger.debug(f" ⏭️ Card {idx} ({asin}): No link found, skipping")
|
||||
continue
|
||||
href = link_elem.get('href', '')
|
||||
product_url = f"{AMAZON_FR_BASE_URL}{href}" if href.startswith('/') else href
|
||||
|
||||
Reference in new issue
Block a user