diff --git a/app/services/ai_price_extractor.py b/app/services/ai_price_extractor.py index 76f856b..6eaf2ac 100644 --- a/app/services/ai_price_extractor.py +++ b/app/services/ai_price_extractor.py @@ -31,9 +31,9 @@ class AIPriceExtractor: logger.warning("No OpenRouter API key found for AI Price Extraction") return None - # Smart truncation: keep first 3000 chars which usually contain the price - # and product info. - clean_html = html_content[:3000] + # Smart truncation: keep first 15000 chars to ensure we capture the price + # (3000 was too short for B&M) + clean_html = html_content[:15000] prompt = f""" You are a price extraction expert. @@ -52,13 +52,13 @@ class AIPriceExtractor: try: # Try Primary Model (Gemma 3) + # Note: Removed response_format={"type": "json_object"} as it's not supported by all models response = await acompletion( model=cls.PRIMARY_MODEL, messages=[{"role": "user", "content": prompt}], api_key=api_key, api_base="https://openrouter.ai/api/v1", - temperature=0.1, - response_format={"type": "json_object"} + temperature=0.1 ) except Exception as e: logger.warning(f"Primary model {cls.PRIMARY_MODEL} failed ({e}), trying fallback {cls.FALLBACK_MODEL}...") @@ -68,24 +68,28 @@ class AIPriceExtractor: messages=[{"role": "user", "content": prompt}], api_key=api_key, api_base="https://openrouter.ai/api/v1", - temperature=0.1, - response_format={"type": "json_object"} + temperature=0.1 ) - content = response.choices[0].message.content - content = response.choices[0].message.content if not content: return None - - data = json.loads(content) - price = data.get("price") - if price: - # Handle string price "24,95" or "24.95" - if isinstance(price, str): - price = price.replace(',', '.').replace('€', '').strip() - return float(price) + # Clean markdown code blocks if present (common with LLMs) + content = content.replace("```json", "").replace("```", "").strip() + + try: + data = json.loads(content) + price = data.get("price") + + if price and str(price).lower() != "null": + # Handle string price "24,95" or "24.95" + if isinstance(price, str): + price = price.replace(',', '.').replace('€', '').strip() + return float(price) + except json.JSONDecodeError: + logger.warning(f"Failed to parse AI response as JSON: {content}") + return None return None diff --git a/app/services/improved_search_service.py b/app/services/improved_search_service.py index c9ef522..83bd87e 100644 --- a/app/services/improved_search_service.py +++ b/app/services/improved_search_service.py @@ -467,7 +467,7 @@ class ImprovedSearchService: @staticmethod async def _extract_price(page: Page) -> float | None: - """Extract price using Hybrid Strategy: JSON-LD -> AI -> CSS""" + """Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS""" import re # STRATEGY 1: JSON-LD (Most reliable) @@ -499,7 +499,8 @@ class ImprovedSearchService: # STRATEGY 2: AI Extraction (Priority over CSS as requested) try: - logger.info(" 🤖 Attempting AI extraction (Priority Strategy)...") + # Only try AI if API key is configured (check implicitly via helper) + # We log info but don't block if it fails html_content = await page.content() title = await page.title() @@ -510,9 +511,60 @@ class ImprovedSearchService: except Exception as e: logger.error(f" AI extraction failed: {e}") - # STRATEGY 3: CSS Selectors (Fallback) + # STRATEGY 3: Strict CSS Selectors (High Confidence) + # These selectors usually point to the main product price. + # If found, we trust them and return immediately. + strict_selectors = [ + '[itemprop="price"]', # Schema.org standard + '.current-price-value', + '.product-price .price', + 'meta[property="product:price:amount"]', + 'meta[name="twitter:data1"]', # Sometimes used for price + ] - # PRIORITY 1: Selectors for sale/promotional prices (highest priority) + for selector in strict_selectors: + try: + # Check meta tags first + if selector.startswith('meta'): + element = await page.query_selector(selector) + if element: + content = await element.get_attribute('content') + if content: + try: + price_val = float(content.strip().replace(',', '.')) + logger.debug(f" ✅ Found price via Meta Tag {selector}: {price_val}€") + return price_val + except: + pass + continue + + # Standard elements + elements = await page.query_selector_all(selector) + for elem in elements: + # Skip hidden + if not await elem.is_visible(): + continue + + price_text = await elem.inner_text() + if price_text: + cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() + cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') + match = re.search(r'(\d+\.?\d*)', cleaned) + if match: + try: + price_val = float(match.group(1)) + if 0.01 < price_val < 100000: + logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€") + return price_val + except: + continue + except: + continue + + # STRATEGY 4: Loose CSS Selectors (Fallback - Lowest Price Logic) + # Used when strict selectors fail. We collect ALL prices and pick the lowest. + + # PRIORITY 1: Selectors for sale/promotional prices sale_price_selectors = [ '.price-current', '.prix-actuel', @@ -527,7 +579,6 @@ class ImprovedSearchService: price_selectors = [ '.price', '[data-testid="price"]', - '[itemprop="price"]', '.product-price', '.a-price .a-offscreen', '.a-price-whole', @@ -563,7 +614,6 @@ class ImprovedSearchService: # Fallback to standard price selectors - # Collect ALL valid prices and return the lowest (usually the promotional price) all_prices = [] for selector in price_selectors: @@ -573,14 +623,10 @@ class ImprovedSearchService: # Skip if element is strikethrough (old price) try: parent_html = await elem.evaluate('el => el.parentElement.outerHTML') - # Check for strikethrough in parent if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html: - logger.debug(f" Skipping strikethrough price from {selector}") continue - # Also check the element itself elem_style = await elem.evaluate('el => window.getComputedStyle(el).textDecoration') if 'line-through' in elem_style: - logger.debug(f" Skipping element with line-through style") continue except: pass @@ -603,7 +649,7 @@ class ImprovedSearchService: logger.debug(f" Error with selector {selector}: {e}") continue - # Return the LOWEST price found (promotional price is usually lower) + # Return the LOWEST price found if all_prices: lowest_price = min(all_prices) logger.debug(f"Selected lowest price from {len(all_prices)} candidates: {lowest_price}€") diff --git a/docker-compose.yml b/docker-compose.yml index 95af9db..b0d6c26 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -15,6 +15,7 @@ services: - BROWSERLESS_URL=ws://browserless:3000 - LOG_LEVEL=INFO - CORS_ORIGINS=* + - OPENROUTER_API_KEY=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 depends_on: db: condition: service_healthy diff --git a/verify_bm_extraction.py b/verify_bm_extraction.py new file mode 100644 index 0000000..f3d12f1 --- /dev/null +++ b/verify_bm_extraction.py @@ -0,0 +1,91 @@ +import asyncio +import logging +import sys +from playwright.async_api import async_playwright +from app.services.ai_price_extractor import AIPriceExtractor +from app.services.improved_search_service import ImprovedSearchService +from app.core.search_config import SITE_CONFIGS + +# Configure logging +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +async def verify_bm(): + async with async_playwright() as p: + browser = await p.chromium.launch(headless=True) + page = await browser.new_page() + + # 1. Perform Search to get a product URL + logger.info("--- Step 1: Searching for 'Chaise pliante pu creme' on B&M ---") + # Use specific search to find the problematic product + search_url = "https://bmstores.fr/module/ambjolisearch/jolisearch?s=Chaise+pliante+pu+creme" + await page.goto(search_url) + await page.wait_for_load_state("networkidle") + + # Take screenshot of search results + await page.screenshot(path="bm_search_results.png") + logger.info("Screenshot saved: bm_search_results.png") + + # Get first product link + product_link = await page.get_attribute("a.thumbnail.product-thumbnail", "href") + if not product_link: + logger.error("No product found in search") + return + + if not product_link.startswith("http"): + product_link = "https://www.bmstores.fr" + product_link + + logger.info(f"Testing Product URL: {product_link}") + + # 2. Go to Product Page + await page.goto(product_link) + await page.wait_for_load_state("networkidle") + await page.screenshot(path="bm_product_page.png") + logger.info("Screenshot saved: bm_product_page.png") + + # 3. Dump HTML snippet (price area) + content = await page.content() + logger.info(f"HTML Content Length: {len(content)}") + + # Check for 12.95 + if "12,95" in content or "12.95" in content: + logger.info("✅ Price 12.95 found in raw HTML") + else: + logger.warning("❌ Price 12.95 NOT found in raw HTML") + + # 4. Test JSON-LD + logger.info("\n--- Step 2: Testing JSON-LD ---") + json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]') + for i, script in enumerate(json_ld_scripts): + text = await script.inner_text() + logger.info(f"JSON-LD #{i}: {text[:500]}...") + + # 5. Test CSS Selectors + logger.info("\n--- Step 3: Testing CSS Selectors ---") + selectors = [ + '.price-current', '.prix-actuel', '.sale-price', '.promo-price', + '.price', '[data-testid="price"]', '[itemprop="price"]', '.product-price', + '.current-price-value' + ] + for sel in selectors: + elements = await page.query_selector_all(sel) + for el in elements: + text = await el.inner_text() + logger.info(f"Selector '{sel}': {text.strip()}") + + # 6. Test AI Extraction + logger.info("\n--- Step 4: Testing AI Extraction (Gemma 3) ---") + title = await page.title() + + # Ensure API key is available + import os + if not os.getenv("OPENROUTER_API_KEY"): + logger.warning("OPENROUTER_API_KEY not set in env, AI might fail") + + ai_price = await AIPriceExtractor.extract_price(content, title) + logger.info(f"AI Extracted Price: {ai_price}") + + await browser.close() + +if __name__ == "__main__": + asyncio.run(verify_bm())