mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
2 files changed
+150
-17
No files matched your search
@@ -220,28 +220,56 @@ async def process_item_check(item_id: int):
|
||||
# Check for notifications
|
||||
if item_data["notification_channel"] and extraction.price:
|
||||
channel = item_data["notification_channel"]
|
||||
logger.info(f"📢 Notification check for item {item_id}: old_price={old_price}, new_price={extraction.price}, target_price={item_data['target_price']}")
|
||||
|
||||
# 1. Target Price Reached
|
||||
if item_data["target_price"] and extraction.price <= item_data["target_price"]:
|
||||
# Only notify if we haven't already notified for this price (or if price dropped further)
|
||||
# For simplicity, we notify if current price <= target.
|
||||
# Ideally we should check if we already notified recently, but let's keep it simple.
|
||||
if not old_price or old_price > item_data["target_price"]:
|
||||
await NotificationService.send_notification(
|
||||
channel,
|
||||
title=f"🎯 Prix cible atteint : {item_data['name']}",
|
||||
body=f"Le prix de {item_data['name']} est passé à {extraction.price}€ (Cible: {item_data['target_price']}€)\n{item_data['url']}"
|
||||
)
|
||||
notification_sent = False
|
||||
|
||||
# 2. Price Drop
|
||||
elif old_price and extraction.price < old_price:
|
||||
drop_percent = (old_price - extraction.price) / old_price * 100
|
||||
if drop_percent >= 5: # Notify only for significant drops (> 5%)
|
||||
# CASE 1: Target price is set -> notify only when target is reached
|
||||
if item_data["target_price"]:
|
||||
if extraction.price <= item_data["target_price"]:
|
||||
# Only notify if we haven't already notified for this price
|
||||
if not old_price or old_price > item_data["target_price"]:
|
||||
logger.info(f"✅ Target price reached for {item_data['name']}: {extraction.price}€ <= {item_data['target_price']}€")
|
||||
await NotificationService.send_notification(
|
||||
channel,
|
||||
title=f"🎯 Prix cible atteint : {item_data['name']}",
|
||||
body=f"Le prix de {item_data['name']} est passé à {extraction.price}€ (Cible: {item_data['target_price']}€)\n{item_data['url']}"
|
||||
)
|
||||
notification_sent = True
|
||||
else:
|
||||
logger.info(f"ℹ️ Price {extraction.price}€ not yet at target {item_data['target_price']}€")
|
||||
|
||||
# CASE 2: No target price -> notify on ANY price drop
|
||||
else:
|
||||
if old_price and extraction.price < old_price:
|
||||
drop_percent = (old_price - extraction.price) / old_price * 100
|
||||
logger.info(f"📉 Price drop detected for {item_data['name']}: {drop_percent:.1f}%")
|
||||
await NotificationService.send_notification(
|
||||
channel,
|
||||
title=f"📉 Baisse de prix : {item_data['name']}",
|
||||
body=f"Le prix de {item_data['name']} a baissé de {drop_percent:.1f}% !\nNouveau prix : {extraction.price}€ (Ancien: {old_price}€)\n{item_data['url']}"
|
||||
)
|
||||
notification_sent = True
|
||||
|
||||
# Also notify on significant price increases (>= 5%)
|
||||
elif old_price and extraction.price > old_price:
|
||||
increase_percent = (extraction.price - old_price) / old_price * 100
|
||||
if increase_percent >= 5:
|
||||
logger.info(f"📈 Price increase detected for {item_data['name']}: {increase_percent:.1f}%")
|
||||
await NotificationService.send_notification(
|
||||
channel,
|
||||
title=f"📈 Hausse de prix : {item_data['name']}",
|
||||
body=f"Le prix de {item_data['name']} a augmenté de {increase_percent:.1f}%.\nNouveau prix : {extraction.price}€ (Ancien: {old_price}€)\n{item_data['url']}"
|
||||
)
|
||||
notification_sent = True
|
||||
|
||||
if not notification_sent:
|
||||
logger.info(f"ℹ️ No notification triggered for item {item_id} (no significant change)")
|
||||
else:
|
||||
if not item_data["notification_channel"]:
|
||||
logger.debug(f"⚠️ Item {item_id} has no notification channel assigned.")
|
||||
if not extraction.price:
|
||||
logger.debug(f"⚠️ Item {item_id} has no extracted price.")
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ POPUP_SELECTORS = [
|
||||
"button:has-text('No thanks')",
|
||||
"a:has-text('No, thanks')",
|
||||
"div[role='dialog'] button[aria-label='Close']",
|
||||
# Amazon Interstitials (Added for robustness)
|
||||
# Amazon Interstitials & Cookies
|
||||
"button:has-text('Continuer les achats')",
|
||||
"span:has-text('Continuer les achats')",
|
||||
"a:has-text('Continuer les achats')",
|
||||
@@ -32,13 +32,17 @@ POPUP_SELECTORS = [
|
||||
"span.a-button-inner > input.a-button-input[type='submit']",
|
||||
"form:has-text('Continuer les achats') input[type='submit']",
|
||||
"[aria-labelledby='continue-shopping-label']",
|
||||
"#sp-cc-accept",
|
||||
"#sp-cc-rejectall-link",
|
||||
"button[data-action='a-popover-close']",
|
||||
"[data-action='sp-cc-accept']",
|
||||
"input[aria-labelledby='sp-cc-accept-label']",
|
||||
# Didomi / Gifi
|
||||
"#didomi-notice-agree-button",
|
||||
"button[id='didomi-notice-agree-button']",
|
||||
"span:has-text('Accepter & Fermer')",
|
||||
"button:has-text('Accepter & Fermer')",
|
||||
# Common banners
|
||||
"#sp-cc-accept",
|
||||
"#onetrust-accept-btn-handler",
|
||||
".cookie-consent-accept",
|
||||
"[data-action='accept-cookies']",
|
||||
@@ -148,6 +152,10 @@ class ScraperService:
|
||||
return None, "", url, ""
|
||||
|
||||
try:
|
||||
# SPECIALIZED AMAZON HANDLING
|
||||
if "amazon" in url:
|
||||
return await ScraperService._scrape_amazon_specific(url, item_id, config, return_html)
|
||||
|
||||
context = await ScraperService._create_context(ScraperService._browser, url)
|
||||
page = await context.new_page()
|
||||
|
||||
@@ -222,6 +230,103 @@ class ScraperService:
|
||||
logger.error(f"Error scraping {url}: {e}")
|
||||
return None, "", url, ""
|
||||
|
||||
@staticmethod
|
||||
async def _scrape_amazon_specific(
|
||||
url: str,
|
||||
item_id: int | None,
|
||||
config: ScrapeConfig,
|
||||
return_html: bool
|
||||
) -> tuple[str | None, str, str, str]:
|
||||
"""
|
||||
Specialized scraping flow for Amazon to avoid bot detection and ensure good screenshots.
|
||||
Strategy:
|
||||
1. Emulate human visiting homepage first
|
||||
2. Navigate to product
|
||||
3. Aggressively handle popups
|
||||
4. Wait for main image to be visible
|
||||
"""
|
||||
logger.info(f"🛒 Starting specialized Amazon scrape for: {url}")
|
||||
|
||||
# Determine base domain
|
||||
from urllib.parse import urlparse
|
||||
parsed = urlparse(url)
|
||||
base_domain = f"{parsed.scheme}://{parsed.netloc}"
|
||||
|
||||
context = await ScraperService._create_context(ScraperService._browser, url)
|
||||
page = await context.new_page()
|
||||
|
||||
try:
|
||||
# 1. Warm-up: Visit Homepage to get cookies/session
|
||||
try:
|
||||
logger.info(f"🏠 Visiting {base_domain} to establish authentic session...")
|
||||
await page.goto(base_domain, wait_until="domcontentloaded", timeout=30000)
|
||||
await asyncio.sleep(2)
|
||||
await ScraperService._handle_popups(page)
|
||||
except Exception as e:
|
||||
logger.warning(f"Homepage warm-up failed (continuing anyway): {e}")
|
||||
|
||||
# 2. Navigate to Product
|
||||
logger.info(f"➡️ Navigating to product page: {url}")
|
||||
await page.goto(url, wait_until="domcontentloaded", timeout=60000)
|
||||
|
||||
final_url = page.url
|
||||
try:
|
||||
page_title = await page.title()
|
||||
except:
|
||||
page_title = "Amazon Product"
|
||||
|
||||
# 3. Check for Bot Detection / CAPTCHA
|
||||
content_check = await page.content()
|
||||
if "Type the characters you see in this image" in content_check or "Saisissez les caractères que vous voyez" in content_check:
|
||||
logger.error("🚫 Amazon CAPTCHA detected!")
|
||||
# Attempt refresh once
|
||||
logger.info("Retrying with refresh...")
|
||||
await page.reload()
|
||||
await asyncio.sleep(3)
|
||||
|
||||
# 4. Handle Popups & Location Selectors
|
||||
await ScraperService._handle_popups(page)
|
||||
|
||||
# Dismiss "Change Address" or specific Amazon location modals if any
|
||||
try:
|
||||
await page.evaluate("document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')")
|
||||
except: pass
|
||||
|
||||
# 5. Wait for Main Image (Critical for screenshot)
|
||||
logger.info("🖼️ Waiting for product image...")
|
||||
try:
|
||||
# Main image container on desktop
|
||||
await page.wait_for_selector(
|
||||
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper",
|
||||
timeout=10000
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not find main image container: {e}")
|
||||
|
||||
# 6. Smart User Behavior (Scroll to trigger lazy loading)
|
||||
if config.smart_scroll:
|
||||
await ScraperService._smart_scroll(page, config.scroll_pixels)
|
||||
# Scroll back up to header for good screenshot
|
||||
await page.evaluate("window.scrollTo(0, 0)")
|
||||
await asyncio.sleep(1)
|
||||
|
||||
# 7. Extract Data
|
||||
if return_html:
|
||||
content_data = await page.content()
|
||||
else:
|
||||
content_data = await ScraperService._extract_text(page, config.text_length)
|
||||
|
||||
# 8. Screenshot
|
||||
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
|
||||
|
||||
return screenshot_path, content_data, final_url, page_title
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"❌ Amazon specific scrape failed: {e}")
|
||||
return None, "", url, ""
|
||||
finally:
|
||||
await context.close()
|
||||
|
||||
@staticmethod
|
||||
async def _connect_browser(p) -> Browser:
|
||||
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
|
||||
|
||||
Reference in new issue
Block a user