feat: Add improved search and AI price extraction services, update Docker Compose, and introduce benchmark verification script.

This commit is contained in:
Michael committed 2025-12-01 01:20:39 +01:00
1 parent 8303e767f7
commit b35401a87b
4 files changed
+170 -28

No files matched your search

+21 -17
View File
@@ -31,9 +31,9 @@ class AIPriceExtractor:
logger.warning("No OpenRouter API key found for AI Price Extraction")
return None
# Smart truncation: keep first 3000 chars which usually contain the price
# and product info.
clean_html = html_content[:3000]
# Smart truncation: keep first 15000 chars to ensure we capture the price
# (3000 was too short for B&M)
clean_html = html_content[:15000]
prompt = f"""
You are a price extraction expert.
@@ -52,13 +52,13 @@ class AIPriceExtractor:
try:
# Try Primary Model (Gemma 3)
# Note: Removed response_format={"type": "json_object"} as it's not supported by all models
response = await acompletion(
model=cls.PRIMARY_MODEL,
messages=[{"role": "user", "content": prompt}],
api_key=api_key,
api_base="https://openrouter.ai/api/v1",
temperature=0.1,
response_format={"type": "json_object"}
temperature=0.1
)
except Exception as e:
logger.warning(f"Primary model {cls.PRIMARY_MODEL} failed ({e}), trying fallback {cls.FALLBACK_MODEL}...")
@@ -68,24 +68,28 @@ class AIPriceExtractor:
messages=[{"role": "user", "content": prompt}],
api_key=api_key,
api_base="https://openrouter.ai/api/v1",
temperature=0.1,
response_format={"type": "json_object"}
temperature=0.1
)
content = response.choices[0].message.content
content = response.choices[0].message.content
if not content:
return None
data = json.loads(content)
price = data.get("price")
if price:
# Handle string price "24,95" or "24.95"
if isinstance(price, str):
price = price.replace(',', '.').replace('€', '').strip()
return float(price)
# Clean markdown code blocks if present (common with LLMs)
content = content.replace("```json", "").replace("```", "").strip()
try:
data = json.loads(content)
price = data.get("price")
if price and str(price).lower() != "null":
# Handle string price "24,95" or "24.95"
if isinstance(price, str):
price = price.replace(',', '.').replace('€', '').strip()
return float(price)
except json.JSONDecodeError:
logger.warning(f"Failed to parse AI response as JSON: {content}")
return None
return None
+57 -11
View File
@@ -467,7 +467,7 @@ class ImprovedSearchService:
@staticmethod
async def _extract_price(page: Page) -> float | None:
"""Extract price using Hybrid Strategy: JSON-LD -> AI -> CSS"""
"""Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS"""
import re
# STRATEGY 1: JSON-LD (Most reliable)
@@ -499,7 +499,8 @@ class ImprovedSearchService:
# STRATEGY 2: AI Extraction (Priority over CSS as requested)
try:
logger.info(" 🤖 Attempting AI extraction (Priority Strategy)...")
# Only try AI if API key is configured (check implicitly via helper)
# We log info but don't block if it fails
html_content = await page.content()
title = await page.title()
@@ -510,9 +511,60 @@ class ImprovedSearchService:
except Exception as e:
logger.error(f" AI extraction failed: {e}")
# STRATEGY 3: CSS Selectors (Fallback)
# STRATEGY 3: Strict CSS Selectors (High Confidence)
# These selectors usually point to the main product price.
# If found, we trust them and return immediately.
strict_selectors = [
'[itemprop="price"]', # Schema.org standard
'.current-price-value',
'.product-price .price',
'meta[property="product:price:amount"]',
'meta[name="twitter:data1"]', # Sometimes used for price
]
# PRIORITY 1: Selectors for sale/promotional prices (highest priority)
for selector in strict_selectors:
try:
# Check meta tags first
if selector.startswith('meta'):
element = await page.query_selector(selector)
if element:
content = await element.get_attribute('content')
if content:
try:
price_val = float(content.strip().replace(',', '.'))
logger.debug(f" ✅ Found price via Meta Tag {selector}: {price_val}€")
return price_val
except:
pass
continue
# Standard elements
elements = await page.query_selector_all(selector)
for elem in elements:
# Skip hidden
if not await elem.is_visible():
continue
price_text = await elem.inner_text()
if price_text:
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.')
match = re.search(r'(\d+\.?\d*)', cleaned)
if match:
try:
price_val = float(match.group(1))
if 0.01 < price_val < 100000:
logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€")
return price_val
except:
continue
except:
continue
# STRATEGY 4: Loose CSS Selectors (Fallback - Lowest Price Logic)
# Used when strict selectors fail. We collect ALL prices and pick the lowest.
# PRIORITY 1: Selectors for sale/promotional prices
sale_price_selectors = [
'.price-current',
'.prix-actuel',
@@ -527,7 +579,6 @@ class ImprovedSearchService:
price_selectors = [
'.price',
'[data-testid="price"]',
'[itemprop="price"]',
'.product-price',
'.a-price .a-offscreen',
'.a-price-whole',
@@ -563,7 +614,6 @@ class ImprovedSearchService:
# Fallback to standard price selectors
# Collect ALL valid prices and return the lowest (usually the promotional price)
all_prices = []
for selector in price_selectors:
@@ -573,14 +623,10 @@ class ImprovedSearchService:
# Skip if element is strikethrough (old price)
try:
parent_html = await elem.evaluate('el => el.parentElement.outerHTML')
# Check for strikethrough in parent
if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html:
logger.debug(f" Skipping strikethrough price from {selector}")
continue
# Also check the element itself
elem_style = await elem.evaluate('el => window.getComputedStyle(el).textDecoration')
if 'line-through' in elem_style:
logger.debug(f" Skipping element with line-through style")
continue
except:
pass
@@ -603,7 +649,7 @@ class ImprovedSearchService:
logger.debug(f" Error with selector {selector}: {e}")
continue
# Return the LOWEST price found (promotional price is usually lower)
# Return the LOWEST price found
if all_prices:
lowest_price = min(all_prices)
logger.debug(f"Selected lowest price from {len(all_prices)} candidates: {lowest_price}€")
+1
View File
@@ -15,6 +15,7 @@ services:
- BROWSERLESS_URL=ws://browserless:3000
- LOG_LEVEL=INFO
- CORS_ORIGINS=*
- OPENROUTER_API_KEY=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
depends_on:
db:
condition: service_healthy
+91
View File
@@ -0,0 +1,91 @@
import asyncio
import logging
import sys
from playwright.async_api import async_playwright
from app.services.ai_price_extractor import AIPriceExtractor
from app.services.improved_search_service import ImprovedSearchService
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_bm():
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
page = await browser.new_page()
# 1. Perform Search to get a product URL
logger.info("--- Step 1: Searching for 'Chaise pliante pu creme' on B&M ---")
# Use specific search to find the problematic product
search_url = "https://bmstores.fr/module/ambjolisearch/jolisearch?s=Chaise+pliante+pu+creme"
await page.goto(search_url)
await page.wait_for_load_state("networkidle")
# Take screenshot of search results
await page.screenshot(path="bm_search_results.png")
logger.info("Screenshot saved: bm_search_results.png")
# Get first product link
product_link = await page.get_attribute("a.thumbnail.product-thumbnail", "href")
if not product_link:
logger.error("No product found in search")
return
if not product_link.startswith("http"):
product_link = "https://www.bmstores.fr" + product_link
logger.info(f"Testing Product URL: {product_link}")
# 2. Go to Product Page
await page.goto(product_link)
await page.wait_for_load_state("networkidle")
await page.screenshot(path="bm_product_page.png")
logger.info("Screenshot saved: bm_product_page.png")
# 3. Dump HTML snippet (price area)
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# Check for 12.95
if "12,95" in content or "12.95" in content:
logger.info("✅ Price 12.95 found in raw HTML")
else:
logger.warning("❌ Price 12.95 NOT found in raw HTML")
# 4. Test JSON-LD
logger.info("\n--- Step 2: Testing JSON-LD ---")
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
for i, script in enumerate(json_ld_scripts):
text = await script.inner_text()
logger.info(f"JSON-LD #{i}: {text[:500]}...")
# 5. Test CSS Selectors
logger.info("\n--- Step 3: Testing CSS Selectors ---")
selectors = [
'.price-current', '.prix-actuel', '.sale-price', '.promo-price',
'.price', '[data-testid="price"]', '[itemprop="price"]', '.product-price',
'.current-price-value'
]
for sel in selectors:
elements = await page.query_selector_all(sel)
for el in elements:
text = await el.inner_text()
logger.info(f"Selector '{sel}': {text.strip()}")
# 6. Test AI Extraction
logger.info("\n--- Step 4: Testing AI Extraction (Gemma 3) ---")
title = await page.title()
# Ensure API key is available
import os
if not os.getenv("OPENROUTER_API_KEY"):
logger.warning("OPENROUTER_API_KEY not set in env, AI might fail")
ai_price = await AIPriceExtractor.extract_price(content, title)
logger.info(f"AI Extracted Price: {ai_price}")
await browser.close()
if __name__ == "__main__":
asyncio.run(verify_bm())