mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
10 files changed
+487
-114
No files matched your search
+4
-2
@@ -141,7 +141,9 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
|
||||
- Look for: "PRIX DÉTECTÉ:", price tags, "Prix:", "€", numbers near "Ajouter au panier"
|
||||
- Extract as DECIMAL NUMBER: If you see "3.99", return 3.99
|
||||
- Ignore crossed-out/barré prices (old prices)
|
||||
- If multiple prices, take the current/main price (not the original)
|
||||
- **CRITICAL:** Ignore "Prix au litre", "Prix au kg", "P.U.", or unit prices usually shown in smaller text/parentheses (e.g., "4.60 € / L").
|
||||
- If multiple prices, take the current/main price (not the original, not the unit price)
|
||||
- **B&M STORES Specific:** The main price is often large and bold (e.g. "1.15€"), while unit price is small (e.g. "4.60 €/L"). ALWAYS take the main price.
|
||||
|
||||
**CRITICAL - Common mistakes to avoid:**
|
||||
- "3.99 €" means 3.99 (NOT 399.00, NOT 3990.00)
|
||||
@@ -161,7 +163,7 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
|
||||
- If unclear: set null and confidence < 0.5
|
||||
|
||||
**STOCK:**
|
||||
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart"
|
||||
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart", "Retrait 2h", "Click & Collect"
|
||||
- FALSE if: "Rupture", "Indisponible", "Épuisé", "Out of Stock", "Notify Me"
|
||||
- NULL if unclear or not shown
|
||||
|
||||
|
||||
+33
-10
@@ -36,19 +36,42 @@ def get_amazon_proxies() -> list[dict]:
|
||||
})
|
||||
return proxies
|
||||
|
||||
# === USER AGENTS ===
|
||||
USER_AGENT_POOL = [
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36",
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36",
|
||||
# === USER AGENTS & STEALTH ===
|
||||
USER_AGENT_DATA = [
|
||||
{
|
||||
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
|
||||
"platform": '"Windows"'
|
||||
},
|
||||
{
|
||||
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
|
||||
"ch": '"Google Chrome";v="130", "Chromium";v="130", "Not_A Brand";v="99"',
|
||||
"platform": '"Windows"'
|
||||
},
|
||||
{
|
||||
"ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
|
||||
"platform": '"macOS"'
|
||||
},
|
||||
{
|
||||
"ua": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
|
||||
"platform": '"Linux"'
|
||||
},
|
||||
{
|
||||
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
|
||||
"ch": '"Microsoft Edge";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
|
||||
"platform": '"Windows"'
|
||||
}
|
||||
]
|
||||
|
||||
def get_random_stealth_config() -> dict:
|
||||
"""Returns a random User-Agent and its corresponding Client Hints"""
|
||||
return random.choice(USER_AGENT_DATA)
|
||||
|
||||
def get_random_user_agent() -> str:
|
||||
return random.choice(USER_AGENT_POOL)
|
||||
"""Legacy helper for backward compatibility"""
|
||||
return get_random_stealth_config()["ua"]
|
||||
|
||||
# === SITE CONFIGURATIONS ===
|
||||
SITE_CONFIGS = {
|
||||
|
||||
@@ -180,17 +180,21 @@ class AmazonScraperService:
|
||||
|
||||
@staticmethod
|
||||
async def _create_context(browser: Browser) -> BrowserContext:
|
||||
"""Create browser context with stealth settings"""
|
||||
"""Create browser context with dynamic stealth settings"""
|
||||
from app.core.search_config import get_random_stealth_config
|
||||
|
||||
stealth_config = get_random_stealth_config()
|
||||
ua = stealth_config["ua"]
|
||||
ch = stealth_config["ch"]
|
||||
platform = stealth_config["platform"]
|
||||
|
||||
logger.info(f"🎭 Using stealth profile: {ua[:50]}...")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent=(
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/131.0.0.0 Safari/537.36"
|
||||
),
|
||||
user_agent=ua,
|
||||
locale="fr-FR",
|
||||
timezone_id="Europe/Paris",
|
||||
# Additional stealth parameters
|
||||
extra_http_headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
@@ -198,22 +202,24 @@ class AmazonScraperService:
|
||||
"DNT": "1",
|
||||
"Connection": "keep-alive",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Sec-Ch-Ua": ch,
|
||||
"Sec-Ch-Ua-Mobile": "?0",
|
||||
"Sec-Ch-Ua-Platform": platform,
|
||||
"Sec-Fetch-Dest": "document",
|
||||
"Sec-Fetch-Mode": "navigate",
|
||||
"Sec-Fetch-Site": "none",
|
||||
"Sec-Fetch-User": "?1",
|
||||
"Cache-Control": "max-age=0",
|
||||
"Referer": "https://www.google.fr/",
|
||||
},
|
||||
)
|
||||
|
||||
# Comprehensive stealth mode
|
||||
# Comprehensive stealth mode injector
|
||||
await context.add_init_script("""
|
||||
// Remove webdriver property
|
||||
Object.defineProperty(navigator, 'webdriver', {
|
||||
get: () => undefined
|
||||
});
|
||||
// Reset webdriver
|
||||
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
|
||||
|
||||
// Add chrome property
|
||||
// Re-mock chrome object
|
||||
window.chrome = {
|
||||
runtime: {},
|
||||
loadTimes: function() {},
|
||||
@@ -221,43 +227,46 @@ class AmazonScraperService:
|
||||
app: {}
|
||||
};
|
||||
|
||||
// Override permissions
|
||||
const originalQuery = window.navigator.permissions.query;
|
||||
window.navigator.permissions.query = (parameters) => (
|
||||
parameters.name === 'notifications' ?
|
||||
Promise.resolve({ state: Notification.permission }) :
|
||||
originalQuery(parameters)
|
||||
);
|
||||
|
||||
// Add plugins
|
||||
// Consistent plugins
|
||||
Object.defineProperty(navigator, 'plugins', {
|
||||
get: () => [1, 2, 3, 4, 5]
|
||||
get: () => [
|
||||
{ name: 'PDF Viewer', filename: 'internal-pdf-viewer' },
|
||||
{ name: 'Chrome PDF Viewer', filename: 'internal-pdf-viewer' },
|
||||
{ name: 'Chromium PDF Viewer', filename: 'internal-pdf-viewer' }
|
||||
]
|
||||
});
|
||||
|
||||
// Add languages
|
||||
// Languages
|
||||
Object.defineProperty(navigator, 'languages', {
|
||||
get: () => ['fr-FR', 'fr', 'en-US', 'en']
|
||||
});
|
||||
|
||||
// Override platform
|
||||
Object.defineProperty(navigator, 'platform', {
|
||||
get: () => 'Win32'
|
||||
});
|
||||
|
||||
// Mock battery API
|
||||
Object.defineProperty(navigator, 'getBattery', {
|
||||
get: () => () => Promise.resolve({
|
||||
charging: true,
|
||||
chargingTime: 0,
|
||||
dischargingTime: Infinity,
|
||||
level: 1
|
||||
})
|
||||
});
|
||||
// Device Memory (randomized)
|
||||
Object.defineProperty(navigator, 'deviceMemory', { get: () => 8 });
|
||||
""")
|
||||
|
||||
await context.route("**/*", lambda route: route.continue_())
|
||||
return context
|
||||
|
||||
@staticmethod
|
||||
async def _simulate_human_behavior(page: Page):
|
||||
"""Perform subtle human-like interactions"""
|
||||
import random
|
||||
try:
|
||||
# Random mouse movements
|
||||
for _ in range(3):
|
||||
x = random.randint(100, 800)
|
||||
y = random.randint(100, 600)
|
||||
await page.mouse.move(x, y, steps=10)
|
||||
await asyncio.sleep(random.uniform(0.1, 0.3))
|
||||
|
||||
# Subtle scroll
|
||||
await page.evaluate("window.scrollBy(0, window.innerHeight / 4)")
|
||||
await asyncio.sleep(random.uniform(0.5, 1.0))
|
||||
await page.evaluate("window.scrollBy(0, -window.innerHeight / 5)")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to simulate human behavior: {e}")
|
||||
|
||||
@staticmethod
|
||||
async def _handle_popups(page: Page):
|
||||
"""Close Amazon popups/cookies"""
|
||||
@@ -306,35 +315,44 @@ class AmazonScraperService:
|
||||
try:
|
||||
# CRITICAL: Load Amazon homepage FIRST in same context to establish session
|
||||
logger.info("🏠 Loading Amazon homepage to establish session/cookies...")
|
||||
await page.goto("https://www.amazon.fr", wait_until="domcontentloaded", timeout=30000)
|
||||
await page.goto("https://www.amazon.fr", wait_until="networkidle", timeout=30000)
|
||||
logger.info("✅ Homepage loaded")
|
||||
|
||||
# Handle homepage popups
|
||||
await cls._handle_popups(page)
|
||||
|
||||
# Simulate human behavior on homepage
|
||||
await cls._simulate_human_behavior(page)
|
||||
|
||||
# Small delay
|
||||
await page.wait_for_timeout(2000)
|
||||
await asyncio.sleep(random.uniform(1.0, 3.0))
|
||||
|
||||
# NOW navigate to search in SAME context (cookies preserved)
|
||||
logger.info(f"🔍 Navigating to search: {search_url}")
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=60000)
|
||||
logger.info("Page loaded (domcontentloaded)")
|
||||
|
||||
# Wait for network idle
|
||||
# Wait for content or block
|
||||
try:
|
||||
await page.wait_for_load_state("networkidle", timeout=10000)
|
||||
logger.info("Network idle reached")
|
||||
except PlaywrightTimeoutError:
|
||||
logger.info("Network idle timed out (non-critical)")
|
||||
|
||||
# Handle popups
|
||||
await cls._handle_popups(page)
|
||||
|
||||
# Wait a bit for content
|
||||
await page.wait_for_timeout(2000)
|
||||
await page.wait_for_selector('div[data-component-type="s-search-result"], .s-result-list', timeout=10000)
|
||||
logger.info("✅ Search results detected")
|
||||
await cls._simulate_human_behavior(page)
|
||||
except Exception:
|
||||
logger.warning("🕒 Search results not found immediately, checking for blocks...")
|
||||
|
||||
# Get HTML
|
||||
html_content = await page.content()
|
||||
|
||||
# Proactive block detection
|
||||
if "Type the characters you see in this image" in html_content or "Saisissez les caractères que vous voyez" in html_content:
|
||||
logger.error("🚫 CAPTCHA / Bot detection triggered")
|
||||
return []
|
||||
|
||||
if "Identifiez-vous" in html_content and "commander" not in html_content:
|
||||
logger.warning("⚠️ Redirected to login wall")
|
||||
# Try one more time with a different behavior or just fail
|
||||
return []
|
||||
|
||||
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
|
||||
|
||||
if len(html_content) < 10000:
|
||||
|
||||
@@ -172,20 +172,31 @@ class BrowserlessService:
|
||||
|
||||
@staticmethod
|
||||
async def _create_context(browser: Browser, use_proxy: bool = False) -> BrowserContext:
|
||||
"""Create a new browser context with stealth settings."""
|
||||
user_agent = get_random_user_agent()
|
||||
"""Create a new browser context with dynamic stealth settings."""
|
||||
from app.core.search_config import get_random_stealth_config
|
||||
|
||||
stealth_config = get_random_stealth_config()
|
||||
ua = stealth_config["ua"]
|
||||
ch = stealth_config["ch"]
|
||||
platform = stealth_config["platform"]
|
||||
|
||||
logger.info(f"🎭 Using stealth profile: {ua[:50]}...")
|
||||
|
||||
options = {
|
||||
"viewport": {"width": 1920, "height": 1080},
|
||||
"user_agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
|
||||
"user_agent": ua,
|
||||
"locale": "fr-FR",
|
||||
"timezone_id": "Europe/Paris",
|
||||
"java_script_enabled": True,
|
||||
"bypass_csp": True,
|
||||
"extra_http_headers": {
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"Sec-Ch-Ua-Platform": '"Windows"',
|
||||
"Sec-Ch-Ua": ch,
|
||||
"Sec-Ch-Ua-Mobile": "?0",
|
||||
"Sec-Ch-Ua-Platform": platform,
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Referer": "https://www.google.fr/",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -194,26 +205,35 @@ class BrowserlessService:
|
||||
|
||||
context = await browser.new_context(**options)
|
||||
|
||||
# Inject stealth scripts
|
||||
# Comprehensive stealth mode injector
|
||||
await context.add_init_script("""
|
||||
Object.defineProperty(navigator, 'webdriver', {
|
||||
get: () => undefined
|
||||
// Reset webdriver
|
||||
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
|
||||
|
||||
// Re-mock chrome object
|
||||
window.chrome = {
|
||||
runtime: {},
|
||||
loadTimes: function() {},
|
||||
csi: function() {},
|
||||
app: {}
|
||||
};
|
||||
|
||||
// Consistent plugins
|
||||
Object.defineProperty(navigator, 'plugins', {
|
||||
get: () => [
|
||||
{ name: 'PDF Viewer', filename: 'internal-pdf-viewer' },
|
||||
{ name: 'Chrome PDF Viewer', filename: 'internal-pdf-viewer' },
|
||||
{ name: 'Chromium PDF Viewer', filename: 'internal-pdf-viewer' }
|
||||
]
|
||||
});
|
||||
|
||||
// Languages
|
||||
Object.defineProperty(navigator, 'languages', {
|
||||
get: () => ['fr-FR', 'fr', 'en-US', 'en']
|
||||
});
|
||||
Object.defineProperty(navigator, 'plugins', {
|
||||
get: () => [1, 2, 3, 4, 5]
|
||||
});
|
||||
window.chrome = { runtime: {} };
|
||||
|
||||
// Mock permissions
|
||||
const originalQuery = window.navigator.permissions.query;
|
||||
window.navigator.permissions.query = (parameters) => (
|
||||
parameters.name === 'notifications' ?
|
||||
Promise.resolve({ state: Notification.permission }) :
|
||||
originalQuery(parameters)
|
||||
);
|
||||
// Device Memory (randomized)
|
||||
Object.defineProperty(navigator, 'deviceMemory', { get: () => 8 });
|
||||
""")
|
||||
|
||||
await context.route("**/*", lambda route: route.continue_())
|
||||
@@ -517,28 +537,37 @@ class BrowserlessService:
|
||||
page = await context.new_page()
|
||||
|
||||
try:
|
||||
# Random human-like lead-in delay
|
||||
import random
|
||||
await asyncio.sleep(random.uniform(0.5, 2.0))
|
||||
|
||||
await cls._navigate_and_wait(page, url, 30000)
|
||||
await cls._handle_popups(page)
|
||||
|
||||
# Check for Amazon Captcha / Login Wall
|
||||
# Check for Amazon Captcha / Login Wall / Blocking
|
||||
if "amazon" in url.lower():
|
||||
content_check = await page.content()
|
||||
if (
|
||||
is_blocked = (
|
||||
"Type the characters you see in this image" in content_check
|
||||
or "Saisissez les caractères que vous voyez" in content_check
|
||||
or "Sign in or create an account" in content_check
|
||||
or "Identifiez-vous ou créez un compte" in content_check
|
||||
):
|
||||
logger.warning(f"⚠️ Amazon Captcha/Login Wall detected (Attempt {attempt+1}/{retries})")
|
||||
or ("Sign in or create an account" in content_check and "orders" not in url.lower())
|
||||
or ("Identifiez-vous" in content_check and "commande" not in url.lower())
|
||||
or "api-services-support@amazon.com" in content_check
|
||||
or "service-unavailable" in content_check
|
||||
)
|
||||
|
||||
if is_blocked:
|
||||
logger.warning(f"⚠️ Amazon Blocking/Login Wall detected (Attempt {attempt+1}/{retries})")
|
||||
if attempt < retries - 1:
|
||||
await context.close()
|
||||
await asyncio.sleep(2 + attempt * 2) # Backoff
|
||||
# Exponential backoff with jitter
|
||||
await asyncio.sleep(2 * (attempt + 1) + random.uniform(0.5, 1.5))
|
||||
continue
|
||||
else:
|
||||
logger.error("❌ Amazon blocked all attempts")
|
||||
return "", ""
|
||||
|
||||
# Amazon-specific wait
|
||||
# Amazon-specific wait for price or content
|
||||
if "amazon" in url.lower() and "/dp/" in url:
|
||||
amazon_selectors = [".a-price .a-offscreen", "#corePriceDisplay_desktop_feature_div"]
|
||||
for selector in amazon_selectors:
|
||||
|
||||
@@ -56,3 +56,23 @@ class BMStoresParser(BaseParser):
|
||||
|
||||
logger.info(f"BMStoresParser found {len(results)} results")
|
||||
return results
|
||||
|
||||
def parse_product_details(self, html: str, product_url: str) -> dict:
|
||||
"""
|
||||
Extract price and stock from B&M product page
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
details = {"price": None, "in_stock": True}
|
||||
|
||||
# 1. Look for the main price (avoid suggested products prices)
|
||||
# B&M uses .current-price .price on detail pages
|
||||
price_el = soup.select_one(".product-prices .current-price .price, .product-price-and-shipping .price, [itemprop='price']")
|
||||
if price_el:
|
||||
details["price"] = self.parse_price_text(price_el.get_text())
|
||||
|
||||
# 2. Check stock
|
||||
stock_text = soup.get_text().lower()
|
||||
if "épuisé" in stock_text or "indisponible" in stock_text:
|
||||
details["in_stock"] = False
|
||||
|
||||
return details
|
||||
@@ -107,22 +107,33 @@ async def process_item_check(item_id: int):
|
||||
# Use independent ScraperService for tracking
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
screenshot_path, html_content, final_url = await ScraperService.scrape_item(
|
||||
screenshot_path, html_content, final_url, page_title = await ScraperService.scrape_item(
|
||||
url=item_data["url"],
|
||||
selector=item_data["selector"],
|
||||
item_id=item_id,
|
||||
return_html=True
|
||||
)
|
||||
|
||||
# Detect Amazon login wall/redirection
|
||||
if "amazon" in item_data["url"] and ("signin" in final_url or "captcha" in final_url):
|
||||
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. URL redirected to: {final_url}")
|
||||
await loop.run_in_executor(None, _update_db_error, item_id, "Amazon Bot Detection: Redirected to login/captcha")
|
||||
return
|
||||
|
||||
if not screenshot_path:
|
||||
raise Exception("Failed to capture screenshot")
|
||||
|
||||
# Detect Amazon login wall/redirection/block
|
||||
is_amazon = "amazon" in item_data["url"]
|
||||
is_blocked = False
|
||||
|
||||
if is_amazon:
|
||||
login_terms = ["signin", "captcha", "s'identifier", "log in", "login"]
|
||||
title_lower = page_title.lower()
|
||||
if any(term in final_url.lower() for term in login_terms) or \
|
||||
any(term in title_lower for term in login_terms) or \
|
||||
"amazon.fr: s'identifier" in title_lower:
|
||||
is_blocked = True
|
||||
|
||||
if is_blocked:
|
||||
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. Title: {page_title}, URL: {final_url}")
|
||||
await loop.run_in_executor(None, _update_db_error, item_id, f"Amazon Bot Detection: {page_title}")
|
||||
return
|
||||
|
||||
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
|
||||
# 1. Try specialized parser (if available) or JSON-LD
|
||||
price = None
|
||||
@@ -193,8 +204,10 @@ async def process_item_check(item_id: int):
|
||||
)
|
||||
else:
|
||||
# Fallback to Vision AI if text extraction failed
|
||||
# Use limited text context for Vision prompt to avoid token bloat
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
|
||||
# Clean text properly before sending to AI
|
||||
from app.utils.text import clean_text
|
||||
cleaned_html = clean_text(html_content)
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=cleaned_html[:10000])):
|
||||
raise Exception("AI analysis (Vision) failed")
|
||||
extraction, metadata = ai_result
|
||||
|
||||
|
||||
@@ -130,10 +130,10 @@ class ScraperService:
|
||||
item_id: int | None = None,
|
||||
config: ScrapeConfig | None = None,
|
||||
return_html: bool = False,
|
||||
) -> tuple[str | None, str, str]:
|
||||
) -> tuple[str | None, str, str, str]:
|
||||
"""
|
||||
Scrapes the given URL using Browserless and Playwright.
|
||||
Returns a tuple: (screenshot_path, page_text_or_html, final_url)
|
||||
Returns a tuple: (screenshot_path, page_text_or_html, final_url, page_title)
|
||||
"""
|
||||
if config is None:
|
||||
config = ScrapeConfig()
|
||||
@@ -145,18 +145,59 @@ class ScraperService:
|
||||
# Ensure browser is connected and healthy
|
||||
if not await ScraperService._ensure_browser_connected():
|
||||
logger.error("Failed to establish browser connection")
|
||||
return None, "", url
|
||||
return None, "", url, ""
|
||||
|
||||
try:
|
||||
context = await ScraperService._create_context(ScraperService._browser, url)
|
||||
page = await context.new_page()
|
||||
|
||||
try:
|
||||
# Random delay to simulate human lead-in
|
||||
import random
|
||||
await asyncio.sleep(random.uniform(0.5, 2.0))
|
||||
|
||||
await ScraperService._navigate_and_wait(page, url, timeout)
|
||||
final_url = page.url
|
||||
page_title = await page.title()
|
||||
|
||||
# Humanize: scroll a bit and back
|
||||
await page.mouse.move(random.randint(100, 500), random.randint(100, 500))
|
||||
await page.evaluate("window.scrollBy(0, 100)")
|
||||
await asyncio.sleep(0.5)
|
||||
await page.evaluate("window.scrollBy(0, -100)")
|
||||
|
||||
await ScraperService._handle_popups(page)
|
||||
|
||||
# FORCER LE MAGASIN NANCY POUR B&M STORES
|
||||
if "bmstores.fr" in url:
|
||||
try:
|
||||
logger.info("B&M Stores detected: checking store selection (Nancy)...")
|
||||
# Check if store is already selected or if modal is needed
|
||||
store_btn = page.locator(".mod-shops .btn-shop, .js-show-shops")
|
||||
if await store_btn.count() > 0:
|
||||
logger.info("Opening store selector...")
|
||||
await store_btn.first.click(timeout=2000)
|
||||
await page.wait_for_timeout(1000)
|
||||
|
||||
# Type Nancy in the search input
|
||||
search_input = page.locator("#search_mag")
|
||||
if await search_input.count() > 0:
|
||||
await search_input.fill("Nancy")
|
||||
await page.keyboard.press("Enter")
|
||||
await page.wait_for_timeout(1500)
|
||||
|
||||
# Select the first Nancy store (Nancy Essey or Nancy Centre)
|
||||
select_btn = page.locator(".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')")
|
||||
if await select_btn.count() > 0:
|
||||
logger.info("Selecting Nancy store...")
|
||||
await select_btn.first.click()
|
||||
await page.wait_for_timeout(2000)
|
||||
# Refresh page or wait for update
|
||||
await page.reload(wait_until="domcontentloaded")
|
||||
await page.wait_for_timeout(1000)
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to force B&M store Nancy: {e}")
|
||||
|
||||
if selector:
|
||||
await ScraperService._wait_for_selector(page, selector)
|
||||
else:
|
||||
@@ -172,19 +213,20 @@ class ScraperService:
|
||||
|
||||
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
|
||||
|
||||
return screenshot_path, content_data, final_url
|
||||
return screenshot_path, content_data, final_url, page_title
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error scraping {url}: {e}")
|
||||
return None, "", url
|
||||
return None, "", url, ""
|
||||
|
||||
@staticmethod
|
||||
async def _connect_browser(p) -> Browser:
|
||||
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
|
||||
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
|
||||
# Note: Added timeout for connection
|
||||
return await p.chromium.connect_over_cdp(BROWSERLESS_URL, timeout=30000)
|
||||
|
||||
@staticmethod
|
||||
async def _create_context(browser: Browser, url: str) -> BrowserContext:
|
||||
@@ -374,9 +416,40 @@ class ScraperService:
|
||||
|
||||
try:
|
||||
logger.info(f"Extracting text (limit: {text_length} chars)...")
|
||||
raw_text = await page.inner_text("body")
|
||||
page_text = raw_text[:text_length]
|
||||
logger.info(f"Extracted {len(page_text)} characters")
|
||||
|
||||
# Smart extraction: remove noise (nav, footer, scripts) before getting text
|
||||
clean_text = await page.evaluate("""
|
||||
() => {
|
||||
// Clone body to not affect the visual page
|
||||
const clone = document.body.cloneNode(true);
|
||||
|
||||
// Remove noise selectors
|
||||
const noiseSelectors = [
|
||||
'nav', 'header', 'footer', 'script', 'style', 'noscript', 'iframe',
|
||||
'.cookie-banner', '.popup', '#menu', '.menu', '.sidebar',
|
||||
'[role="navigation"]', '[role="banner"]', '[role="contentinfo"]'
|
||||
];
|
||||
|
||||
noiseSelectors.forEach(selector => {
|
||||
const elements = clone.querySelectorAll(selector);
|
||||
elements.forEach(el => el.remove());
|
||||
});
|
||||
|
||||
return clone.innerText;
|
||||
}
|
||||
""")
|
||||
|
||||
# Fallback if cleaning removed everything (unlikely but safe)
|
||||
if not clean_text or len(clean_text) < 100:
|
||||
logger.warning("Cleaned text too short, falling back to full body text")
|
||||
clean_text = await page.inner_text("body")
|
||||
|
||||
# Collapse whitespace
|
||||
import re
|
||||
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
|
||||
|
||||
page_text = clean_text[:text_length]
|
||||
logger.info(f"Extracted {len(page_text)} chars")
|
||||
return page_text
|
||||
except Exception as e:
|
||||
logger.error(f"Text extraction failed: {e}")
|
||||
|
||||
+10
-7
@@ -8,21 +8,24 @@ SNIPPET_CONTEXT_WINDOW = 100
|
||||
|
||||
def clean_text(text: str) -> str:
|
||||
"""
|
||||
Cleans the text by removing code blocks, HTML tags, and excessive whitespace.
|
||||
Cleans the text by removing script/style content, code blocks, HTML tags, and excessive whitespace.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
# Remove code blocks (```...```)
|
||||
# 1. Remove script and style elements entirely (including content)
|
||||
text = re.sub(r'<(script|style|header|footer|nav)[\s\S]*?>[\s\S]*?<\/\1>', '', text, flags=re.IGNORECASE)
|
||||
|
||||
# 2. Remove other HTML tags (basic) but keep content
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
|
||||
# 3. Remove code blocks (```...```)
|
||||
text = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
|
||||
|
||||
# Remove HTML tags (basic)
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
|
||||
# Remove non-printable characters (keep newlines and tabs)
|
||||
# 4. Remove non-printable characters (keep newlines and tabs)
|
||||
text = re.sub(r"[^\x20-\x7E\n\t]", "", text)
|
||||
|
||||
# Collapse excessive whitespace
|
||||
# 5. Collapse excessive whitespace
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
return text
|
||||
|
||||
@@ -18,6 +18,15 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
|
||||
- [x] Update `ItemService.delete_item` to clean up all related screenshots <!-- id: 6 -->
|
||||
- [x] Create `verify_fix_item_service.py` to test the new logic <!-- id: 5 -->
|
||||
|
||||
## Current Focus
|
||||
|
||||
Improving price extraction reliability for B&M Stores and others.
|
||||
|
||||
- [x] Analyze `ai_schema.py` to check the extraction prompt <!-- id: 7 -->
|
||||
- [x] Improve `ScraperService._extract_text` to be more targeted (e.g. main content only) <!-- id: 8 -->
|
||||
- [x] Update AI prompt to better handle multiple prices (unit vs package) <!-- id: 9 -->
|
||||
- [x] Verify extraction logic with simulation script <!-- id: 10 -->
|
||||
|
||||
## Progress Log
|
||||
|
||||
- Identified the issue: `ScraperService` overwrites `item_{id}.png`.
|
||||
@@ -25,3 +34,6 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
|
||||
- User approved plan.
|
||||
- Implemented filesystem scanning in `ItemService`.
|
||||
- Verified fix with `verify_fix_item_service.py` successfully.
|
||||
- Improved AI prompt to ignore unit prices (like "Prix au litre").
|
||||
- Enhanced text extraction to remove menu/footer noise.
|
||||
- Verified logic with `verify_extraction_logic.py` (simulated).
|
||||
@@ -0,0 +1,180 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Mock sqlalchemy
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
|
||||
# Mock pydantic
|
||||
mock_pydantic = MagicMock()
|
||||
# Mock BaseModel
|
||||
class MockBaseModel:
|
||||
def __init__(self, **kwargs):
|
||||
for k, v in kwargs.items():
|
||||
setattr(self, k, v)
|
||||
mock_pydantic.BaseModel = MockBaseModel
|
||||
mock_pydantic.Field = MagicMock(return_value=None)
|
||||
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
|
||||
|
||||
sys.modules["pydantic"] = mock_pydantic
|
||||
|
||||
# Mock app.utils.text which is imported by ai_schema
|
||||
mock_utils_text = MagicMock()
|
||||
sys.modules["app.utils.text"] = mock_utils_text
|
||||
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
|
||||
mock_utils_text.clean_text = lambda text: text.strip()
|
||||
|
||||
# Mock app.strings (if used) or other utils
|
||||
sys.modules["app.utils"] = MagicMock()
|
||||
|
||||
# Mock app.utils.image
|
||||
sys.modules["app.utils.image"] = MagicMock()
|
||||
|
||||
# Mock app.database
|
||||
sys.modules["app.database"] = MagicMock()
|
||||
|
||||
# Mock playwright
|
||||
mock_playwright = MagicMock()
|
||||
sys.modules["playwright"] = mock_playwright
|
||||
sys.modules["playwright.async_api"] = mock_playwright
|
||||
|
||||
# Mock generic types for type hints if needed
|
||||
mock_playwright.Browser = MagicMock
|
||||
mock_playwright.BrowserContext = MagicMock
|
||||
mock_playwright.Page = MagicMock
|
||||
mock_playwright.TimeoutError = Exception
|
||||
|
||||
# Now import the schema
|
||||
from app.ai_schema import get_extraction_prompt, get_repair_prompt
|
||||
|
||||
# We can't import ScraperService easily if it inherits from things or uses decorators
|
||||
# But for this test we only need get_extraction_prompt which is in ai_schema
|
||||
# So we can skip importing ScraperService if it causes issues,
|
||||
# BUT we wanted to verify ScraperService text cleaning logic...
|
||||
# Let's mock ScraperService dependencies completely.
|
||||
|
||||
try:
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
except ImportError:
|
||||
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
|
||||
ScraperService = None
|
||||
|
||||
# Mock litellm and tenacity
|
||||
sys.modules["litellm"] = MagicMock()
|
||||
sys.modules["tenacity"] = MagicMock()
|
||||
mock_retry = MagicMock()
|
||||
sys.modules["tenacity.retry"] = mock_retry
|
||||
|
||||
# Make sure imports inside ai_service don't fail
|
||||
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
|
||||
# We need to mock these specifically if the module imports them directly
|
||||
mock_tenacity = MagicMock()
|
||||
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
|
||||
mock_tenacity.retry_if_exception_type = MagicMock()
|
||||
mock_tenacity.stop_after_attempt = MagicMock()
|
||||
mock_tenacity.wait_exponential = MagicMock()
|
||||
sys.modules["tenacity"] = mock_tenacity
|
||||
|
||||
# Now import AIService
|
||||
# We will mock the AI response to verify the parsing logic
|
||||
from app.services.ai_service import AIService
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_extraction_logic():
|
||||
print("Verifying Extraction Logic...")
|
||||
|
||||
# 1. Test Text Cleaning in ScraperService
|
||||
# We can't mock Playwright page easily in a simple script without launching a browser.
|
||||
# But we can test the AI prompt generation which is critical.
|
||||
|
||||
# Simulate B&M text
|
||||
dirty_text = """
|
||||
Menu
|
||||
Accueil
|
||||
Panier
|
||||
|
||||
Boisson energisante ice 25cl
|
||||
Red Bull
|
||||
|
||||
1.15 €
|
||||
Prix au litre : 4,60 € / L
|
||||
|
||||
En stock
|
||||
Ajouter au panier
|
||||
|
||||
Footer
|
||||
Mentions légales
|
||||
"""
|
||||
|
||||
print("\n--- Testing Prompt Generation ---")
|
||||
prompt = get_extraction_prompt(dirty_text)
|
||||
|
||||
# Verify strict instructions are present
|
||||
checks = [
|
||||
"CRITICAL",
|
||||
"Ignore \"Prix au litre\"",
|
||||
"B&M STORES Specific",
|
||||
"Extract as DECIMAL NUMBER"
|
||||
]
|
||||
|
||||
all_passed = True
|
||||
for check in checks:
|
||||
if check in prompt:
|
||||
print(f"[OK] Prompt contains: {check}")
|
||||
else:
|
||||
print(f"[FAIL] Prompt missing: {check}")
|
||||
all_passed = False
|
||||
|
||||
if not all_passed:
|
||||
print("Prompt verification failed!")
|
||||
exit(1)
|
||||
|
||||
print("\n--- Testing Response Parsing (Mock AI) ---")
|
||||
|
||||
# Case 1: AI returns Main Price correctly
|
||||
mock_response_1 = """
|
||||
```json
|
||||
{
|
||||
"price": 1.15,
|
||||
"currency": "EUR",
|
||||
"in_stock": true,
|
||||
"price_confidence": 0.95,
|
||||
"in_stock_confidence": 1.0,
|
||||
"source_type": "text"
|
||||
}
|
||||
```
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_1)
|
||||
if result.price == 1.15 and result.in_stock is True:
|
||||
print("[OK] Parsed correct mocked response.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse correct response: {result}")
|
||||
exit(1)
|
||||
|
||||
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
|
||||
# If AI returns explicit null because it's confused
|
||||
mock_response_2 = """
|
||||
{
|
||||
"price": null,
|
||||
"currency": "EUR",
|
||||
"in_stock": null,
|
||||
"price_confidence": 0.0,
|
||||
"in_stock_confidence": 0.0,
|
||||
"source_type": "image"
|
||||
}
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_2)
|
||||
if result.price is None:
|
||||
print("[OK] Parsed null response correctly.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse null response.")
|
||||
|
||||
print("\nVerification of Logic Flow Complete (Simulated).")
|
||||
print("Real-world verification requires running the full scraper.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_extraction_logic())
|
||||
Reference in new issue
Block a user