Merge pull request #235 from R0m1k3/antigravity

Antigravity
This commit is contained in:
LogiFlow authored and GitHub committed 2025-12-18 16:32:56 +01:00
commit 36ce7e8e9f
10 files changed
+487 -114

No files matched your search

+4 -2
View File
@@ -141,7 +141,9 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
- Look for: "PRIX DÉTECTÉ:", price tags, "Prix:", "€", numbers near "Ajouter au panier"
- Extract as DECIMAL NUMBER: If you see "3.99", return 3.99
- Ignore crossed-out/barré prices (old prices)
- If multiple prices, take the current/main price (not the original)
- **CRITICAL:** Ignore "Prix au litre", "Prix au kg", "P.U.", or unit prices usually shown in smaller text/parentheses (e.g., "4.60 € / L").
- If multiple prices, take the current/main price (not the original, not the unit price)
- **B&M STORES Specific:** The main price is often large and bold (e.g. "1.15€"), while unit price is small (e.g. "4.60 €/L"). ALWAYS take the main price.
**CRITICAL - Common mistakes to avoid:**
- "3.99 €" means 3.99 (NOT 399.00, NOT 3990.00)
@@ -161,7 +163,7 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
- If unclear: set null and confidence < 0.5
**STOCK:**
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart"
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart", "Retrait 2h", "Click & Collect"
- FALSE if: "Rupture", "Indisponible", "Épuisé", "Out of Stock", "Notify Me"
- NULL if unclear or not shown
+33 -10
View File
@@ -36,19 +36,42 @@ def get_amazon_proxies() -> list[dict]:
})
return proxies
# === USER AGENTS ===
USER_AGENT_POOL = [
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36",
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/129.0.0.0 Safari/537.36",
# === USER AGENTS & STEALTH ===
USER_AGENT_DATA = [
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Windows"'
},
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="130", "Chromium";v="130", "Not_A Brand";v="99"',
"platform": '"Windows"'
},
{
"ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"macOS"'
},
{
"ua": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Linux"'
},
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
"ch": '"Microsoft Edge";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Windows"'
}
]
def get_random_stealth_config() -> dict:
"""Returns a random User-Agent and its corresponding Client Hints"""
return random.choice(USER_AGENT_DATA)
def get_random_user_agent() -> str:
return random.choice(USER_AGENT_POOL)
"""Legacy helper for backward compatibility"""
return get_random_stealth_config()["ua"]
# === SITE CONFIGURATIONS ===
SITE_CONFIGS = {
+69 -51
View File
@@ -180,17 +180,21 @@ class AmazonScraperService:
@staticmethod
async def _create_context(browser: Browser) -> BrowserContext:
"""Create browser context with stealth settings"""
"""Create browser context with dynamic stealth settings"""
from app.core.search_config import get_random_stealth_config
stealth_config = get_random_stealth_config()
ua = stealth_config["ua"]
ch = stealth_config["ch"]
platform = stealth_config["platform"]
logger.info(f"🎭 Using stealth profile: {ua[:50]}...")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent=(
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
),
user_agent=ua,
locale="fr-FR",
timezone_id="Europe/Paris",
# Additional stealth parameters
extra_http_headers={
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
@@ -198,22 +202,24 @@ class AmazonScraperService:
"DNT": "1",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
"Sec-Ch-Ua": ch,
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": platform,
"Sec-Fetch-Dest": "document",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-Site": "none",
"Sec-Fetch-User": "?1",
"Cache-Control": "max-age=0",
"Referer": "https://www.google.fr/",
},
)
# Comprehensive stealth mode
# Comprehensive stealth mode injector
await context.add_init_script("""
// Remove webdriver property
Object.defineProperty(navigator, 'webdriver', {
get: () => undefined
});
// Reset webdriver
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
// Add chrome property
// Re-mock chrome object
window.chrome = {
runtime: {},
loadTimes: function() {},
@@ -221,43 +227,46 @@ class AmazonScraperService:
app: {}
};
// Override permissions
const originalQuery = window.navigator.permissions.query;
window.navigator.permissions.query = (parameters) => (
parameters.name === 'notifications' ?
Promise.resolve({ state: Notification.permission }) :
originalQuery(parameters)
);
// Add plugins
// Consistent plugins
Object.defineProperty(navigator, 'plugins', {
get: () => [1, 2, 3, 4, 5]
get: () => [
{ name: 'PDF Viewer', filename: 'internal-pdf-viewer' },
{ name: 'Chrome PDF Viewer', filename: 'internal-pdf-viewer' },
{ name: 'Chromium PDF Viewer', filename: 'internal-pdf-viewer' }
]
});
// Add languages
// Languages
Object.defineProperty(navigator, 'languages', {
get: () => ['fr-FR', 'fr', 'en-US', 'en']
});
// Override platform
Object.defineProperty(navigator, 'platform', {
get: () => 'Win32'
});
// Mock battery API
Object.defineProperty(navigator, 'getBattery', {
get: () => () => Promise.resolve({
charging: true,
chargingTime: 0,
dischargingTime: Infinity,
level: 1
})
});
// Device Memory (randomized)
Object.defineProperty(navigator, 'deviceMemory', { get: () => 8 });
""")
await context.route("**/*", lambda route: route.continue_())
return context
@staticmethod
async def _simulate_human_behavior(page: Page):
"""Perform subtle human-like interactions"""
import random
try:
# Random mouse movements
for _ in range(3):
x = random.randint(100, 800)
y = random.randint(100, 600)
await page.mouse.move(x, y, steps=10)
await asyncio.sleep(random.uniform(0.1, 0.3))
# Subtle scroll
await page.evaluate("window.scrollBy(0, window.innerHeight / 4)")
await asyncio.sleep(random.uniform(0.5, 1.0))
await page.evaluate("window.scrollBy(0, -window.innerHeight / 5)")
except Exception as e:
logger.warning(f"Failed to simulate human behavior: {e}")
@staticmethod
async def _handle_popups(page: Page):
"""Close Amazon popups/cookies"""
@@ -306,35 +315,44 @@ class AmazonScraperService:
try:
# CRITICAL: Load Amazon homepage FIRST in same context to establish session
logger.info("🏠 Loading Amazon homepage to establish session/cookies...")
await page.goto("https://www.amazon.fr", wait_until="domcontentloaded", timeout=30000)
await page.goto("https://www.amazon.fr", wait_until="networkidle", timeout=30000)
logger.info("✅ Homepage loaded")
# Handle homepage popups
await cls._handle_popups(page)
# Simulate human behavior on homepage
await cls._simulate_human_behavior(page)
# Small delay
await page.wait_for_timeout(2000)
await asyncio.sleep(random.uniform(1.0, 3.0))
# NOW navigate to search in SAME context (cookies preserved)
logger.info(f"🔍 Navigating to search: {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=60000)
logger.info("Page loaded (domcontentloaded)")
# Wait for network idle
# Wait for content or block
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.info("Network idle reached")
except PlaywrightTimeoutError:
logger.info("Network idle timed out (non-critical)")
# Handle popups
await cls._handle_popups(page)
# Wait a bit for content
await page.wait_for_timeout(2000)
await page.wait_for_selector('div[data-component-type="s-search-result"], .s-result-list', timeout=10000)
logger.info("✅ Search results detected")
await cls._simulate_human_behavior(page)
except Exception:
logger.warning("🕒 Search results not found immediately, checking for blocks...")
# Get HTML
html_content = await page.content()
# Proactive block detection
if "Type the characters you see in this image" in html_content or "Saisissez les caractères que vous voyez" in html_content:
logger.error("🚫 CAPTCHA / Bot detection triggered")
return []
if "Identifiez-vous" in html_content and "commander" not in html_content:
logger.warning("⚠️ Redirected to login wall")
# Try one more time with a different behavior or just fail
return []
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
if len(html_content) < 10000:
+55 -26
View File
@@ -172,20 +172,31 @@ class BrowserlessService:
@staticmethod
async def _create_context(browser: Browser, use_proxy: bool = False) -> BrowserContext:
"""Create a new browser context with stealth settings."""
user_agent = get_random_user_agent()
"""Create a new browser context with dynamic stealth settings."""
from app.core.search_config import get_random_stealth_config
stealth_config = get_random_stealth_config()
ua = stealth_config["ua"]
ch = stealth_config["ch"]
platform = stealth_config["platform"]
logger.info(f"🎭 Using stealth profile: {ua[:50]}...")
options = {
"viewport": {"width": 1920, "height": 1080},
"user_agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
"user_agent": ua,
"locale": "fr-FR",
"timezone_id": "Europe/Paris",
"java_script_enabled": True,
"bypass_csp": True,
"extra_http_headers": {
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
"Sec-Ch-Ua-Platform": '"Windows"',
"Sec-Ch-Ua": ch,
"Sec-Ch-Ua-Mobile": "?0",
"Sec-Ch-Ua-Platform": platform,
"Upgrade-Insecure-Requests": "1",
"Referer": "https://www.google.fr/",
},
}
@@ -194,26 +205,35 @@ class BrowserlessService:
context = await browser.new_context(**options)
# Inject stealth scripts
# Comprehensive stealth mode injector
await context.add_init_script("""
Object.defineProperty(navigator, 'webdriver', {
get: () => undefined
// Reset webdriver
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
// Re-mock chrome object
window.chrome = {
runtime: {},
loadTimes: function() {},
csi: function() {},
app: {}
};
// Consistent plugins
Object.defineProperty(navigator, 'plugins', {
get: () => [
{ name: 'PDF Viewer', filename: 'internal-pdf-viewer' },
{ name: 'Chrome PDF Viewer', filename: 'internal-pdf-viewer' },
{ name: 'Chromium PDF Viewer', filename: 'internal-pdf-viewer' }
]
});
// Languages
Object.defineProperty(navigator, 'languages', {
get: () => ['fr-FR', 'fr', 'en-US', 'en']
});
Object.defineProperty(navigator, 'plugins', {
get: () => [1, 2, 3, 4, 5]
});
window.chrome = { runtime: {} };
// Mock permissions
const originalQuery = window.navigator.permissions.query;
window.navigator.permissions.query = (parameters) => (
parameters.name === 'notifications' ?
Promise.resolve({ state: Notification.permission }) :
originalQuery(parameters)
);
// Device Memory (randomized)
Object.defineProperty(navigator, 'deviceMemory', { get: () => 8 });
""")
await context.route("**/*", lambda route: route.continue_())
@@ -517,28 +537,37 @@ class BrowserlessService:
page = await context.new_page()
try:
# Random human-like lead-in delay
import random
await asyncio.sleep(random.uniform(0.5, 2.0))
await cls._navigate_and_wait(page, url, 30000)
await cls._handle_popups(page)
# Check for Amazon Captcha / Login Wall
# Check for Amazon Captcha / Login Wall / Blocking
if "amazon" in url.lower():
content_check = await page.content()
if (
is_blocked = (
"Type the characters you see in this image" in content_check
or "Saisissez les caractères que vous voyez" in content_check
or "Sign in or create an account" in content_check
or "Identifiez-vous ou créez un compte" in content_check
):
logger.warning(f"⚠️ Amazon Captcha/Login Wall detected (Attempt {attempt+1}/{retries})")
or ("Sign in or create an account" in content_check and "orders" not in url.lower())
or ("Identifiez-vous" in content_check and "commande" not in url.lower())
or "api-services-support@amazon.com" in content_check
or "service-unavailable" in content_check
)
if is_blocked:
logger.warning(f"⚠️ Amazon Blocking/Login Wall detected (Attempt {attempt+1}/{retries})")
if attempt < retries - 1:
await context.close()
await asyncio.sleep(2 + attempt * 2) # Backoff
# Exponential backoff with jitter
await asyncio.sleep(2 * (attempt + 1) + random.uniform(0.5, 1.5))
continue
else:
logger.error("❌ Amazon blocked all attempts")
return "", ""
# Amazon-specific wait
# Amazon-specific wait for price or content
if "amazon" in url.lower() and "/dp/" in url:
amazon_selectors = [".a-price .a-offscreen", "#corePriceDisplay_desktop_feature_div"]
for selector in amazon_selectors:
+20
View File
@@ -56,3 +56,23 @@ class BMStoresParser(BaseParser):
logger.info(f"BMStoresParser found {len(results)} results")
return results
def parse_product_details(self, html: str, product_url: str) -> dict:
"""
Extract price and stock from B&M product page
"""
soup = BeautifulSoup(html, "html.parser")
details = {"price": None, "in_stock": True}
# 1. Look for the main price (avoid suggested products prices)
# B&M uses .current-price .price on detail pages
price_el = soup.select_one(".product-prices .current-price .price, .product-price-and-shipping .price, [itemprop='price']")
if price_el:
details["price"] = self.parse_price_text(price_el.get_text())
# 2. Check stock
stock_text = soup.get_text().lower()
if "épuisé" in stock_text or "indisponible" in stock_text:
details["in_stock"] = False
return details
+22 -9
View File
@@ -107,22 +107,33 @@ async def process_item_check(item_id: int):
# Use independent ScraperService for tracking
from app.services.tracking_scraper_service import ScraperService
screenshot_path, html_content, final_url = await ScraperService.scrape_item(
screenshot_path, html_content, final_url, page_title = await ScraperService.scrape_item(
url=item_data["url"],
selector=item_data["selector"],
item_id=item_id,
return_html=True
)
# Detect Amazon login wall/redirection
if "amazon" in item_data["url"] and ("signin" in final_url or "captcha" in final_url):
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. URL redirected to: {final_url}")
await loop.run_in_executor(None, _update_db_error, item_id, "Amazon Bot Detection: Redirected to login/captcha")
return
if not screenshot_path:
raise Exception("Failed to capture screenshot")
# Detect Amazon login wall/redirection/block
is_amazon = "amazon" in item_data["url"]
is_blocked = False
if is_amazon:
login_terms = ["signin", "captcha", "s'identifier", "log in", "login"]
title_lower = page_title.lower()
if any(term in final_url.lower() for term in login_terms) or \
any(term in title_lower for term in login_terms) or \
"amazon.fr: s'identifier" in title_lower:
is_blocked = True
if is_blocked:
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. Title: {page_title}, URL: {final_url}")
await loop.run_in_executor(None, _update_db_error, item_id, f"Amazon Bot Detection: {page_title}")
return
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
# 1. Try specialized parser (if available) or JSON-LD
price = None
@@ -193,8 +204,10 @@ async def process_item_check(item_id: int):
)
else:
# Fallback to Vision AI if text extraction failed
# Use limited text context for Vision prompt to avoid token bloat
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
# Clean text properly before sending to AI
from app.utils.text import clean_text
cleaned_html = clean_text(html_content)
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=cleaned_html[:10000])):
raise Exception("AI analysis (Vision) failed")
extraction, metadata = ai_result
+82 -9
View File
@@ -130,10 +130,10 @@ class ScraperService:
item_id: int | None = None,
config: ScrapeConfig | None = None,
return_html: bool = False,
) -> tuple[str | None, str, str]:
) -> tuple[str | None, str, str, str]:
"""
Scrapes the given URL using Browserless and Playwright.
Returns a tuple: (screenshot_path, page_text_or_html, final_url)
Returns a tuple: (screenshot_path, page_text_or_html, final_url, page_title)
"""
if config is None:
config = ScrapeConfig()
@@ -145,18 +145,59 @@ class ScraperService:
# Ensure browser is connected and healthy
if not await ScraperService._ensure_browser_connected():
logger.error("Failed to establish browser connection")
return None, "", url
return None, "", url, ""
try:
context = await ScraperService._create_context(ScraperService._browser, url)
page = await context.new_page()
try:
# Random delay to simulate human lead-in
import random
await asyncio.sleep(random.uniform(0.5, 2.0))
await ScraperService._navigate_and_wait(page, url, timeout)
final_url = page.url
page_title = await page.title()
# Humanize: scroll a bit and back
await page.mouse.move(random.randint(100, 500), random.randint(100, 500))
await page.evaluate("window.scrollBy(0, 100)")
await asyncio.sleep(0.5)
await page.evaluate("window.scrollBy(0, -100)")
await ScraperService._handle_popups(page)
# FORCER LE MAGASIN NANCY POUR B&M STORES
if "bmstores.fr" in url:
try:
logger.info("B&M Stores detected: checking store selection (Nancy)...")
# Check if store is already selected or if modal is needed
store_btn = page.locator(".mod-shops .btn-shop, .js-show-shops")
if await store_btn.count() > 0:
logger.info("Opening store selector...")
await store_btn.first.click(timeout=2000)
await page.wait_for_timeout(1000)
# Type Nancy in the search input
search_input = page.locator("#search_mag")
if await search_input.count() > 0:
await search_input.fill("Nancy")
await page.keyboard.press("Enter")
await page.wait_for_timeout(1500)
# Select the first Nancy store (Nancy Essey or Nancy Centre)
select_btn = page.locator(".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')")
if await select_btn.count() > 0:
logger.info("Selecting Nancy store...")
await select_btn.first.click()
await page.wait_for_timeout(2000)
# Refresh page or wait for update
await page.reload(wait_until="domcontentloaded")
await page.wait_for_timeout(1000)
except Exception as e:
logger.warning(f"Failed to force B&M store Nancy: {e}")
if selector:
await ScraperService._wait_for_selector(page, selector)
else:
@@ -172,19 +213,20 @@ class ScraperService:
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
return screenshot_path, content_data, final_url
return screenshot_path, content_data, final_url, page_title
finally:
await context.close()
except Exception as e:
logger.error(f"Error scraping {url}: {e}")
return None, "", url
return None, "", url, ""
@staticmethod
async def _connect_browser(p) -> Browser:
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
# Note: Added timeout for connection
return await p.chromium.connect_over_cdp(BROWSERLESS_URL, timeout=30000)
@staticmethod
async def _create_context(browser: Browser, url: str) -> BrowserContext:
@@ -374,9 +416,40 @@ class ScraperService:
try:
logger.info(f"Extracting text (limit: {text_length} chars)...")
raw_text = await page.inner_text("body")
page_text = raw_text[:text_length]
logger.info(f"Extracted {len(page_text)} characters")
# Smart extraction: remove noise (nav, footer, scripts) before getting text
clean_text = await page.evaluate("""
() => {
// Clone body to not affect the visual page
const clone = document.body.cloneNode(true);
// Remove noise selectors
const noiseSelectors = [
'nav', 'header', 'footer', 'script', 'style', 'noscript', 'iframe',
'.cookie-banner', '.popup', '#menu', '.menu', '.sidebar',
'[role="navigation"]', '[role="banner"]', '[role="contentinfo"]'
];
noiseSelectors.forEach(selector => {
const elements = clone.querySelectorAll(selector);
elements.forEach(el => el.remove());
});
return clone.innerText;
}
""")
# Fallback if cleaning removed everything (unlikely but safe)
if not clean_text or len(clean_text) < 100:
logger.warning("Cleaned text too short, falling back to full body text")
clean_text = await page.inner_text("body")
# Collapse whitespace
import re
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
page_text = clean_text[:text_length]
logger.info(f"Extracted {len(page_text)} chars")
return page_text
except Exception as e:
logger.error(f"Text extraction failed: {e}")
+10 -7
View File
@@ -8,21 +8,24 @@ SNIPPET_CONTEXT_WINDOW = 100
def clean_text(text: str) -> str:
"""
Cleans the text by removing code blocks, HTML tags, and excessive whitespace.
Cleans the text by removing script/style content, code blocks, HTML tags, and excessive whitespace.
"""
if not text:
return ""
# Remove code blocks (```...```)
# 1. Remove script and style elements entirely (including content)
text = re.sub(r'<(script|style|header|footer|nav)[\s\S]*?>[\s\S]*?<\/\1>', '', text, flags=re.IGNORECASE)
# 2. Remove other HTML tags (basic) but keep content
text = re.sub(r"<[^>]+>", " ", text)
# 3. Remove code blocks (```...```)
text = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
# Remove HTML tags (basic)
text = re.sub(r"<[^>]+>", "", text)
# Remove non-printable characters (keep newlines and tabs)
# 4. Remove non-printable characters (keep newlines and tabs)
text = re.sub(r"[^\x20-\x7E\n\t]", "", text)
# Collapse excessive whitespace
# 5. Collapse excessive whitespace
text = re.sub(r"\s+", " ", text).strip()
return text
+12
View File
@@ -18,6 +18,15 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
- [x] Update `ItemService.delete_item` to clean up all related screenshots <!-- id: 6 -->
- [x] Create `verify_fix_item_service.py` to test the new logic <!-- id: 5 -->
## Current Focus
Improving price extraction reliability for B&M Stores and others.
- [x] Analyze `ai_schema.py` to check the extraction prompt <!-- id: 7 -->
- [x] Improve `ScraperService._extract_text` to be more targeted (e.g. main content only) <!-- id: 8 -->
- [x] Update AI prompt to better handle multiple prices (unit vs package) <!-- id: 9 -->
- [x] Verify extraction logic with simulation script <!-- id: 10 -->
## Progress Log
- Identified the issue: `ScraperService` overwrites `item_{id}.png`.
@@ -25,3 +34,6 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
- User approved plan.
- Implemented filesystem scanning in `ItemService`.
- Verified fix with `verify_fix_item_service.py` successfully.
- Improved AI prompt to ignore unit prices (like "Prix au litre").
- Enhanced text extraction to remove menu/footer noise.
- Verified logic with `verify_extraction_logic.py` (simulated).
+180
View File
@@ -0,0 +1,180 @@
import asyncio
import logging
import sys
from unittest.mock import MagicMock, AsyncMock
# Mock sqlalchemy
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
# Mock pydantic
mock_pydantic = MagicMock()
# Mock BaseModel
class MockBaseModel:
def __init__(self, **kwargs):
for k, v in kwargs.items():
setattr(self, k, v)
mock_pydantic.BaseModel = MockBaseModel
mock_pydantic.Field = MagicMock(return_value=None)
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
sys.modules["pydantic"] = mock_pydantic
# Mock app.utils.text which is imported by ai_schema
mock_utils_text = MagicMock()
sys.modules["app.utils.text"] = mock_utils_text
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
mock_utils_text.clean_text = lambda text: text.strip()
# Mock app.strings (if used) or other utils
sys.modules["app.utils"] = MagicMock()
# Mock app.utils.image
sys.modules["app.utils.image"] = MagicMock()
# Mock app.database
sys.modules["app.database"] = MagicMock()
# Mock playwright
mock_playwright = MagicMock()
sys.modules["playwright"] = mock_playwright
sys.modules["playwright.async_api"] = mock_playwright
# Mock generic types for type hints if needed
mock_playwright.Browser = MagicMock
mock_playwright.BrowserContext = MagicMock
mock_playwright.Page = MagicMock
mock_playwright.TimeoutError = Exception
# Now import the schema
from app.ai_schema import get_extraction_prompt, get_repair_prompt
# We can't import ScraperService easily if it inherits from things or uses decorators
# But for this test we only need get_extraction_prompt which is in ai_schema
# So we can skip importing ScraperService if it causes issues,
# BUT we wanted to verify ScraperService text cleaning logic...
# Let's mock ScraperService dependencies completely.
try:
from app.services.tracking_scraper_service import ScraperService
except ImportError:
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
ScraperService = None
# Mock litellm and tenacity
sys.modules["litellm"] = MagicMock()
sys.modules["tenacity"] = MagicMock()
mock_retry = MagicMock()
sys.modules["tenacity.retry"] = mock_retry
# Make sure imports inside ai_service don't fail
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
# We need to mock these specifically if the module imports them directly
mock_tenacity = MagicMock()
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
mock_tenacity.retry_if_exception_type = MagicMock()
mock_tenacity.stop_after_attempt = MagicMock()
mock_tenacity.wait_exponential = MagicMock()
sys.modules["tenacity"] = mock_tenacity
# Now import AIService
# We will mock the AI response to verify the parsing logic
from app.services.ai_service import AIService
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_extraction_logic():
print("Verifying Extraction Logic...")
# 1. Test Text Cleaning in ScraperService
# We can't mock Playwright page easily in a simple script without launching a browser.
# But we can test the AI prompt generation which is critical.
# Simulate B&M text
dirty_text = """
Menu
Accueil
Panier
Boisson energisante ice 25cl
Red Bull
1.15 €
Prix au litre : 4,60 € / L
En stock
Ajouter au panier
Footer
Mentions légales
"""
print("\n--- Testing Prompt Generation ---")
prompt = get_extraction_prompt(dirty_text)
# Verify strict instructions are present
checks = [
"CRITICAL",
"Ignore \"Prix au litre\"",
"B&M STORES Specific",
"Extract as DECIMAL NUMBER"
]
all_passed = True
for check in checks:
if check in prompt:
print(f"[OK] Prompt contains: {check}")
else:
print(f"[FAIL] Prompt missing: {check}")
all_passed = False
if not all_passed:
print("Prompt verification failed!")
exit(1)
print("\n--- Testing Response Parsing (Mock AI) ---")
# Case 1: AI returns Main Price correctly
mock_response_1 = """
```json
{
"price": 1.15,
"currency": "EUR",
"in_stock": true,
"price_confidence": 0.95,
"in_stock_confidence": 1.0,
"source_type": "text"
}
```
"""
result = AIService.parse_and_validate_response(mock_response_1)
if result.price == 1.15 and result.in_stock is True:
print("[OK] Parsed correct mocked response.")
else:
print(f"[FAIL] Failed to parse correct response: {result}")
exit(1)
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
# If AI returns explicit null because it's confused
mock_response_2 = """
{
"price": null,
"currency": "EUR",
"in_stock": null,
"price_confidence": 0.0,
"in_stock_confidence": 0.0,
"source_type": "image"
}
"""
result = AIService.parse_and_validate_response(mock_response_2)
if result.price is None:
print("[OK] Parsed null response correctly.")
else:
print(f"[FAIL] Failed to parse null response.")
print("\nVerification of Logic Flow Complete (Simulated).")
print("Real-world verification requires running the full scraper.")
if __name__ == "__main__":
asyncio.run(verify_extraction_logic())