Merge pull request #249 from R0m1k3/antigravity

Antigravity
This commit is contained in:
LogiFlow authored and GitHub committed 2025-12-23 19:56:30 +01:00
commit b079ae71e0
2 files changed
+150 -17

No files matched your search

+43 -15
View File
@@ -220,28 +220,56 @@ async def process_item_check(item_id: int):
# Check for notifications
if item_data["notification_channel"] and extraction.price:
channel = item_data["notification_channel"]
logger.info(f"📢 Notification check for item {item_id}: old_price={old_price}, new_price={extraction.price}, target_price={item_data['target_price']}")
# 1. Target Price Reached
if item_data["target_price"] and extraction.price <= item_data["target_price"]:
# Only notify if we haven't already notified for this price (or if price dropped further)
# For simplicity, we notify if current price <= target.
# Ideally we should check if we already notified recently, but let's keep it simple.
if not old_price or old_price > item_data["target_price"]:
await NotificationService.send_notification(
channel,
title=f"🎯 Prix cible atteint : {item_data['name']}",
body=f"Le prix de {item_data['name']} est passé à {extraction.price}€ (Cible: {item_data['target_price']}€)\n{item_data['url']}"
)
notification_sent = False
# 2. Price Drop
elif old_price and extraction.price < old_price:
drop_percent = (old_price - extraction.price) / old_price * 100
if drop_percent >= 5: # Notify only for significant drops (> 5%)
# CASE 1: Target price is set -> notify only when target is reached
if item_data["target_price"]:
if extraction.price <= item_data["target_price"]:
# Only notify if we haven't already notified for this price
if not old_price or old_price > item_data["target_price"]:
logger.info(f"✅ Target price reached for {item_data['name']}: {extraction.price}€ <= {item_data['target_price']}€")
await NotificationService.send_notification(
channel,
title=f"🎯 Prix cible atteint : {item_data['name']}",
body=f"Le prix de {item_data['name']} est passé à {extraction.price}€ (Cible: {item_data['target_price']}€)\n{item_data['url']}"
)
notification_sent = True
else:
logger.info(f"ℹ️ Price {extraction.price}€ not yet at target {item_data['target_price']}€")
# CASE 2: No target price -> notify on ANY price drop
else:
if old_price and extraction.price < old_price:
drop_percent = (old_price - extraction.price) / old_price * 100
logger.info(f"📉 Price drop detected for {item_data['name']}: {drop_percent:.1f}%")
await NotificationService.send_notification(
channel,
title=f"📉 Baisse de prix : {item_data['name']}",
body=f"Le prix de {item_data['name']} a baissé de {drop_percent:.1f}% !\nNouveau prix : {extraction.price}€ (Ancien: {old_price}€)\n{item_data['url']}"
)
notification_sent = True
# Also notify on significant price increases (>= 5%)
elif old_price and extraction.price > old_price:
increase_percent = (extraction.price - old_price) / old_price * 100
if increase_percent >= 5:
logger.info(f"📈 Price increase detected for {item_data['name']}: {increase_percent:.1f}%")
await NotificationService.send_notification(
channel,
title=f"📈 Hausse de prix : {item_data['name']}",
body=f"Le prix de {item_data['name']} a augmenté de {increase_percent:.1f}%.\nNouveau prix : {extraction.price}€ (Ancien: {old_price}€)\n{item_data['url']}"
)
notification_sent = True
if not notification_sent:
logger.info(f"ℹ️ No notification triggered for item {item_id} (no significant change)")
else:
if not item_data["notification_channel"]:
logger.debug(f"⚠️ Item {item_id} has no notification channel assigned.")
if not extraction.price:
logger.debug(f"⚠️ Item {item_id} has no extracted price.")
+107 -2
View File
@@ -23,7 +23,7 @@ POPUP_SELECTORS = [
"button:has-text('No thanks')",
"a:has-text('No, thanks')",
"div[role='dialog'] button[aria-label='Close']",
# Amazon Interstitials (Added for robustness)
# Amazon Interstitials & Cookies
"button:has-text('Continuer les achats')",
"span:has-text('Continuer les achats')",
"a:has-text('Continuer les achats')",
@@ -32,13 +32,17 @@ POPUP_SELECTORS = [
"span.a-button-inner > input.a-button-input[type='submit']",
"form:has-text('Continuer les achats') input[type='submit']",
"[aria-labelledby='continue-shopping-label']",
"#sp-cc-accept",
"#sp-cc-rejectall-link",
"button[data-action='a-popover-close']",
"[data-action='sp-cc-accept']",
"input[aria-labelledby='sp-cc-accept-label']",
# Didomi / Gifi
"#didomi-notice-agree-button",
"button[id='didomi-notice-agree-button']",
"span:has-text('Accepter & Fermer')",
"button:has-text('Accepter & Fermer')",
# Common banners
"#sp-cc-accept",
"#onetrust-accept-btn-handler",
".cookie-consent-accept",
"[data-action='accept-cookies']",
@@ -148,6 +152,10 @@ class ScraperService:
return None, "", url, ""
try:
# SPECIALIZED AMAZON HANDLING
if "amazon" in url:
return await ScraperService._scrape_amazon_specific(url, item_id, config, return_html)
context = await ScraperService._create_context(ScraperService._browser, url)
page = await context.new_page()
@@ -222,6 +230,103 @@ class ScraperService:
logger.error(f"Error scraping {url}: {e}")
return None, "", url, ""
@staticmethod
async def _scrape_amazon_specific(
url: str,
item_id: int | None,
config: ScrapeConfig,
return_html: bool
) -> tuple[str | None, str, str, str]:
"""
Specialized scraping flow for Amazon to avoid bot detection and ensure good screenshots.
Strategy:
1. Emulate human visiting homepage first
2. Navigate to product
3. Aggressively handle popups
4. Wait for main image to be visible
"""
logger.info(f"🛒 Starting specialized Amazon scrape for: {url}")
# Determine base domain
from urllib.parse import urlparse
parsed = urlparse(url)
base_domain = f"{parsed.scheme}://{parsed.netloc}"
context = await ScraperService._create_context(ScraperService._browser, url)
page = await context.new_page()
try:
# 1. Warm-up: Visit Homepage to get cookies/session
try:
logger.info(f"🏠 Visiting {base_domain} to establish authentic session...")
await page.goto(base_domain, wait_until="domcontentloaded", timeout=30000)
await asyncio.sleep(2)
await ScraperService._handle_popups(page)
except Exception as e:
logger.warning(f"Homepage warm-up failed (continuing anyway): {e}")
# 2. Navigate to Product
logger.info(f"➡️ Navigating to product page: {url}")
await page.goto(url, wait_until="domcontentloaded", timeout=60000)
final_url = page.url
try:
page_title = await page.title()
except:
page_title = "Amazon Product"
# 3. Check for Bot Detection / CAPTCHA
content_check = await page.content()
if "Type the characters you see in this image" in content_check or "Saisissez les caractères que vous voyez" in content_check:
logger.error("🚫 Amazon CAPTCHA detected!")
# Attempt refresh once
logger.info("Retrying with refresh...")
await page.reload()
await asyncio.sleep(3)
# 4. Handle Popups & Location Selectors
await ScraperService._handle_popups(page)
# Dismiss "Change Address" or specific Amazon location modals if any
try:
await page.evaluate("document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')")
except: pass
# 5. Wait for Main Image (Critical for screenshot)
logger.info("🖼️ Waiting for product image...")
try:
# Main image container on desktop
await page.wait_for_selector(
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper",
timeout=10000
)
except Exception as e:
logger.warning(f"Could not find main image container: {e}")
# 6. Smart User Behavior (Scroll to trigger lazy loading)
if config.smart_scroll:
await ScraperService._smart_scroll(page, config.scroll_pixels)
# Scroll back up to header for good screenshot
await page.evaluate("window.scrollTo(0, 0)")
await asyncio.sleep(1)
# 7. Extract Data
if return_html:
content_data = await page.content()
else:
content_data = await ScraperService._extract_text(page, config.text_length)
# 8. Screenshot
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
return screenshot_path, content_data, final_url, page_title
except Exception as e:
logger.error(f"❌ Amazon specific scrape failed: {e}")
return None, "", url, ""
finally:
await context.close()
@staticmethod
async def _connect_browser(p) -> Browser:
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")