Merge pull request #201 from R0m1k3/antigravity

feat: Implement AI-powered price extraction using Gemma 2 and an impr…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-12-01 00:57:27 +01:00
commit 548cd296f4
4 files changed
+290 -22

No files matched your search

+82
View File
@@ -0,0 +1,82 @@
"""
Analyze B&M product page for price extraction
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# URL from the screenshot
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
print(f"Loading: {url}\n")
await page.goto(url, wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Find all elements with € symbol
price_els = soup.find_all(string=re.compile('€'))
print(f"Found {len(price_els)} elements with '€'\n")
prices_found = {}
for el in price_els[:30]:
text = el.strip()
if text and len(text) < 100:
parent = el.find_parent()
if parent:
parent_class = parent.get('class', [])
parent_class_str = ' '.join(parent_class) if isinstance(parent_class, list) else str(parent_class)
# Extract price value
price_match = re.search(r'(\d+[.,]\d+)\s*€', text)
if price_match:
price_val = price_match.group(1)
key = f"{price_val}€ in .{parent_class_str[:50]}"
if key not in prices_found:
prices_found[key] = text
print("Prices found:")
for key, text in prices_found.items():
print(f" {key}: '{text}'")
# Try common price selectors
print("\nTrying specific selectors:")
selectors = [
".price",
"[class*='price']",
"[data-price]",
".product-price",
"span[class*='price']"
]
for selector in selectors:
try:
els = soup.select(selector)
if els:
for el in els[:2]:
text = el.get_text(strip=True)
if '€' in text:
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+50
View File
@@ -0,0 +1,50 @@
"""
Analyze B&M with longer wait and playwright evaluation
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
print(f"Loading: {url}\n")
await page.goto(url, wait_until="networkidle")
# Wait extra time for JS
await page.wait_for_timeout(5000)
# Look for elements containing price
print("Searching for price elements...\n")
# Try to find any text with €
price_els = await page.query_selector_all("*:has-text('€')")
print(f"Found {len(price_els)} elements with €\n")
for i, el in enumerate(price_els[:10]):
text = await el.inner_text()
tag = await el.evaluate("el => el.tagName")
classes = await el.evaluate("el => el.className")
print(f"{i+1}. <{tag} class='{classes}'> {text[:100]}")
# Screenshot for debugging
await page.screenshot(path="/app/debug_dumps/bm_screenshot.png", full_page=True)
print("\nScreenshot saved to /app/debug_dumps/bm_screenshot.png")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
+94
View File
@@ -0,0 +1,94 @@
import logging
import json
import os
from typing import Optional
from app.services.ai_service import AIService
from litellm import acompletion
logger = logging.getLogger(__name__)
class AIPriceExtractor:
# Primary model requested by user (Gemma 3 Free)
PRIMARY_MODEL = "openrouter/google/gemma-3-27b-it:free"
# Fallback model (known to work/be free)
FALLBACK_MODEL = "openrouter/google/gemma-2-9b-it:free"
@classmethod
async def extract_price(cls, html_content: str, product_title: str) -> Optional[float]:
"""
Extract price from HTML using AI (Gemma 3 via OpenRouter with fallback).
"""
try:
# Get config to retrieve API key
config = AIService.get_ai_config()
api_key = config.get("api_key")
# Fallback to env var if not in DB config
if not api_key:
api_key = os.getenv("OPENROUTER_API_KEY")
if not api_key:
logger.warning("No OpenRouter API key found for AI Price Extraction")
return None
# Smart truncation: keep first 3000 chars which usually contain the price
# and product info.
clean_html = html_content[:3000]
prompt = f"""
You are a price extraction expert.
Task: Extract the CURRENT SELLING PRICE for the product "{product_title}" from the HTML snippet below.
Rules:
1. Return ONLY the numeric value (e.g., 24.95).
2. Ignore crossed-out prices (old prices).
3. If multiple prices exist, choose the one that seems to be the current effective price (usually the lowest non-crossed-out one).
4. If no price is found, return "null".
5. Output format: JSON {{ "price": 24.95 }}
HTML Snippet:
{clean_html}
"""
try:
# Try Primary Model (Gemma 3)
response = await acompletion(
model=cls.PRIMARY_MODEL,
messages=[{"role": "user", "content": prompt}],
api_key=api_key,
api_base="https://openrouter.ai/api/v1",
temperature=0.1,
response_format={"type": "json_object"}
)
except Exception as e:
logger.warning(f"Primary model {cls.PRIMARY_MODEL} failed ({e}), trying fallback {cls.FALLBACK_MODEL}...")
# Try Fallback Model (Gemma 2)
response = await acompletion(
model=cls.FALLBACK_MODEL,
messages=[{"role": "user", "content": prompt}],
api_key=api_key,
api_base="https://openrouter.ai/api/v1",
temperature=0.1,
response_format={"type": "json_object"}
)
content = response.choices[0].message.content
content = response.choices[0].message.content
if not content:
return None
data = json.loads(content)
price = data.get("price")
if price:
# Handle string price "24,95" or "24.95"
if isinstance(price, str):
price = price.replace(',', '.').replace('€', '').strip()
return float(price)
return None
except Exception as e:
logger.error(f"AI Price Extraction failed: {e}")
return None
+64 -22
View File
@@ -1,23 +1,5 @@
"""
Improved Search Service - Using Persistent Browser Connection
Based on ScraperService pattern for better session management and reliability
UPDATED: Now uses modular, site-specific parsers for robust product extraction
"""
import asyncio
import logging
from typing import AsyncGenerator
from urllib.parse import quote_plus, urljoin
from bs4 import BeautifulSoup
from playwright.async_api import Browser, BrowserContext, Page, async_playwright
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
from sqlalchemy.orm import Session
from app.core.search_config import SITE_CONFIGS, BROWSERLESS_URL
from app.models import SearchSite
from app.schemas import SearchProgress, SearchResultItem
from app.services.parsers import ParserFactory, ProductResult
logger = logging.getLogger(__name__)
@@ -466,9 +448,38 @@ class ImprovedSearchService:
@staticmethod
async def _extract_price(page: Page) -> float | None:
"""Extract price from product page using multiple selectors"""
"""Extract price using Hybrid Strategy: JSON-LD -> CSS -> AI"""
import re
# STRATEGY 1: JSON-LD (Most reliable)
try:
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
for script in json_ld_scripts:
try:
content = await script.inner_text()
data = json.loads(content)
# Handle list of objects
if isinstance(data, list):
data = data[0] if data else {}
# Check for Product type
if data.get('@type') == 'Product':
offers = data.get('offers')
if isinstance(offers, list):
offers = offers[0]
if offers and 'price' in offers:
price = float(offers['price'])
logger.debug(f" ✅ Found price via JSON-LD: {price}€")
return price
except:
continue
except Exception as e:
logger.debug(f" JSON-LD extraction failed: {e}")
# STRATEGY 2: CSS Selectors (Standard)
# PRIORITY 1: Selectors for sale/promotional prices (highest priority)
sale_price_selectors = [
'.price-current',
@@ -518,7 +529,11 @@ class ImprovedSearchService:
except Exception:
continue
# Fallback to standard price selectors
# Collect ALL valid prices and return the lowest (usually the promotional price)
all_prices = []
for selector in price_selectors:
try:
elements = await page.query_selector_all(selector)
@@ -526,7 +541,14 @@ class ImprovedSearchService:
# Skip if element is strikethrough (old price)
try:
parent_html = await elem.evaluate('el => el.parentElement.outerHTML')
# Check for strikethrough in parent
if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html:
logger.debug(f" Skipping strikethrough price from {selector}")
continue
# Also check the element itself
elem_style = await elem.evaluate('el => window.getComputedStyle(el).textDecoration')
if 'line-through' in elem_style:
logger.debug(f" Skipping element with line-through style")
continue
except:
pass
@@ -541,12 +563,32 @@ class ImprovedSearchService:
try:
price_val = float(match.group(1))
if 0.01 < price_val < 100000:
logger.debug(f"Found price: {price_val}€ from {selector}")
return price_val
logger.debug(f" Found candidate price: {price_val}€ from {selector}")
all_prices.append(price_val)
except ValueError:
continue
except Exception:
except Exception as e:
logger.debug(f" Error with selector {selector}: {e}")
continue
# Return the LOWEST price found (promotional price is usually lower)
if all_prices:
lowest_price = min(all_prices)
logger.debug(f"Selected lowest price from {len(all_prices)} candidates: {lowest_price}€")
return lowest_price
# STRATEGY 3: AI Fallback (Gemma 2 / OpenRouter)
try:
logger.info(" 🤖 CSS extraction failed, attempting AI extraction...")
html_content = await page.content()
title = await page.title()
ai_price = await AIPriceExtractor.extract_price(html_content, title)
if ai_price:
logger.info(f" 🤖 AI found price: {ai_price}€")
return ai_price
except Exception as e:
logger.error(f" AI fallback failed: {e}")
logger.debug("No price found")
return None