mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #201 from R0m1k3/antigravity
feat: Implement AI-powered price extraction using Gemma 2 and an impr…
This commit is contained in:
4 files changed
+290
-22
No files matched your search
@@ -0,0 +1,82 @@
|
||||
"""
|
||||
Analyze B&M product page for price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# URL from the screenshot
|
||||
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
|
||||
print(f"Loading: {url}\n")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Find all elements with € symbol
|
||||
price_els = soup.find_all(string=re.compile('€'))
|
||||
print(f"Found {len(price_els)} elements with '€'\n")
|
||||
|
||||
prices_found = {}
|
||||
for el in price_els[:30]:
|
||||
text = el.strip()
|
||||
if text and len(text) < 100:
|
||||
parent = el.find_parent()
|
||||
if parent:
|
||||
parent_class = parent.get('class', [])
|
||||
parent_class_str = ' '.join(parent_class) if isinstance(parent_class, list) else str(parent_class)
|
||||
|
||||
# Extract price value
|
||||
price_match = re.search(r'(\d+[.,]\d+)\s*€', text)
|
||||
if price_match:
|
||||
price_val = price_match.group(1)
|
||||
key = f"{price_val}€ in .{parent_class_str[:50]}"
|
||||
if key not in prices_found:
|
||||
prices_found[key] = text
|
||||
|
||||
print("Prices found:")
|
||||
for key, text in prices_found.items():
|
||||
print(f" {key}: '{text}'")
|
||||
|
||||
# Try common price selectors
|
||||
print("\nTrying specific selectors:")
|
||||
selectors = [
|
||||
".price",
|
||||
"[class*='price']",
|
||||
"[data-price]",
|
||||
".product-price",
|
||||
"span[class*='price']"
|
||||
]
|
||||
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = soup.select(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = el.get_text(strip=True)
|
||||
if '€' in text:
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,50 @@
|
||||
"""
|
||||
Analyze B&M with longer wait and playwright evaluation
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
|
||||
print(f"Loading: {url}\n")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
# Wait extra time for JS
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Look for elements containing price
|
||||
print("Searching for price elements...\n")
|
||||
|
||||
# Try to find any text with €
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"Found {len(price_els)} elements with €\n")
|
||||
|
||||
for i, el in enumerate(price_els[:10]):
|
||||
text = await el.inner_text()
|
||||
tag = await el.evaluate("el => el.tagName")
|
||||
classes = await el.evaluate("el => el.className")
|
||||
print(f"{i+1}. <{tag} class='{classes}'> {text[:100]}")
|
||||
|
||||
# Screenshot for debugging
|
||||
await page.screenshot(path="/app/debug_dumps/bm_screenshot.png", full_page=True)
|
||||
print("\nScreenshot saved to /app/debug_dumps/bm_screenshot.png")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,94 @@
|
||||
import logging
|
||||
import json
|
||||
import os
|
||||
from typing import Optional
|
||||
from app.services.ai_service import AIService
|
||||
from litellm import acompletion
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class AIPriceExtractor:
|
||||
# Primary model requested by user (Gemma 3 Free)
|
||||
PRIMARY_MODEL = "openrouter/google/gemma-3-27b-it:free"
|
||||
# Fallback model (known to work/be free)
|
||||
FALLBACK_MODEL = "openrouter/google/gemma-2-9b-it:free"
|
||||
|
||||
@classmethod
|
||||
async def extract_price(cls, html_content: str, product_title: str) -> Optional[float]:
|
||||
"""
|
||||
Extract price from HTML using AI (Gemma 3 via OpenRouter with fallback).
|
||||
"""
|
||||
try:
|
||||
# Get config to retrieve API key
|
||||
config = AIService.get_ai_config()
|
||||
api_key = config.get("api_key")
|
||||
|
||||
# Fallback to env var if not in DB config
|
||||
if not api_key:
|
||||
api_key = os.getenv("OPENROUTER_API_KEY")
|
||||
|
||||
if not api_key:
|
||||
logger.warning("No OpenRouter API key found for AI Price Extraction")
|
||||
return None
|
||||
|
||||
# Smart truncation: keep first 3000 chars which usually contain the price
|
||||
# and product info.
|
||||
clean_html = html_content[:3000]
|
||||
|
||||
prompt = f"""
|
||||
You are a price extraction expert.
|
||||
Task: Extract the CURRENT SELLING PRICE for the product "{product_title}" from the HTML snippet below.
|
||||
|
||||
Rules:
|
||||
1. Return ONLY the numeric value (e.g., 24.95).
|
||||
2. Ignore crossed-out prices (old prices).
|
||||
3. If multiple prices exist, choose the one that seems to be the current effective price (usually the lowest non-crossed-out one).
|
||||
4. If no price is found, return "null".
|
||||
5. Output format: JSON {{ "price": 24.95 }}
|
||||
|
||||
HTML Snippet:
|
||||
{clean_html}
|
||||
"""
|
||||
|
||||
try:
|
||||
# Try Primary Model (Gemma 3)
|
||||
response = await acompletion(
|
||||
model=cls.PRIMARY_MODEL,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
api_key=api_key,
|
||||
api_base="https://openrouter.ai/api/v1",
|
||||
temperature=0.1,
|
||||
response_format={"type": "json_object"}
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"Primary model {cls.PRIMARY_MODEL} failed ({e}), trying fallback {cls.FALLBACK_MODEL}...")
|
||||
# Try Fallback Model (Gemma 2)
|
||||
response = await acompletion(
|
||||
model=cls.FALLBACK_MODEL,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
api_key=api_key,
|
||||
api_base="https://openrouter.ai/api/v1",
|
||||
temperature=0.1,
|
||||
response_format={"type": "json_object"}
|
||||
)
|
||||
|
||||
content = response.choices[0].message.content
|
||||
|
||||
content = response.choices[0].message.content
|
||||
if not content:
|
||||
return None
|
||||
|
||||
data = json.loads(content)
|
||||
price = data.get("price")
|
||||
|
||||
if price:
|
||||
# Handle string price "24,95" or "24.95"
|
||||
if isinstance(price, str):
|
||||
price = price.replace(',', '.').replace('€', '').strip()
|
||||
return float(price)
|
||||
|
||||
return None
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"AI Price Extraction failed: {e}")
|
||||
return None
|
||||
@@ -1,23 +1,5 @@
|
||||
"""
|
||||
Improved Search Service - Using Persistent Browser Connection
|
||||
Based on ScraperService pattern for better session management and reliability
|
||||
|
||||
UPDATED: Now uses modular, site-specific parsers for robust product extraction
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import AsyncGenerator
|
||||
from urllib.parse import quote_plus, urljoin
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from playwright.async_api import Browser, BrowserContext, Page, async_playwright
|
||||
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.core.search_config import SITE_CONFIGS, BROWSERLESS_URL
|
||||
from app.models import SearchSite
|
||||
from app.schemas import SearchProgress, SearchResultItem
|
||||
from app.services.parsers import ParserFactory, ProductResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -466,9 +448,38 @@ class ImprovedSearchService:
|
||||
|
||||
@staticmethod
|
||||
async def _extract_price(page: Page) -> float | None:
|
||||
"""Extract price from product page using multiple selectors"""
|
||||
"""Extract price using Hybrid Strategy: JSON-LD -> CSS -> AI"""
|
||||
import re
|
||||
|
||||
# STRATEGY 1: JSON-LD (Most reliable)
|
||||
try:
|
||||
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
|
||||
for script in json_ld_scripts:
|
||||
try:
|
||||
content = await script.inner_text()
|
||||
data = json.loads(content)
|
||||
|
||||
# Handle list of objects
|
||||
if isinstance(data, list):
|
||||
data = data[0] if data else {}
|
||||
|
||||
# Check for Product type
|
||||
if data.get('@type') == 'Product':
|
||||
offers = data.get('offers')
|
||||
if isinstance(offers, list):
|
||||
offers = offers[0]
|
||||
|
||||
if offers and 'price' in offers:
|
||||
price = float(offers['price'])
|
||||
logger.debug(f" ✅ Found price via JSON-LD: {price}€")
|
||||
return price
|
||||
except:
|
||||
continue
|
||||
except Exception as e:
|
||||
logger.debug(f" JSON-LD extraction failed: {e}")
|
||||
|
||||
# STRATEGY 2: CSS Selectors (Standard)
|
||||
|
||||
# PRIORITY 1: Selectors for sale/promotional prices (highest priority)
|
||||
sale_price_selectors = [
|
||||
'.price-current',
|
||||
@@ -518,7 +529,11 @@ class ImprovedSearchService:
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
|
||||
# Fallback to standard price selectors
|
||||
# Collect ALL valid prices and return the lowest (usually the promotional price)
|
||||
all_prices = []
|
||||
|
||||
for selector in price_selectors:
|
||||
try:
|
||||
elements = await page.query_selector_all(selector)
|
||||
@@ -526,7 +541,14 @@ class ImprovedSearchService:
|
||||
# Skip if element is strikethrough (old price)
|
||||
try:
|
||||
parent_html = await elem.evaluate('el => el.parentElement.outerHTML')
|
||||
# Check for strikethrough in parent
|
||||
if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html:
|
||||
logger.debug(f" Skipping strikethrough price from {selector}")
|
||||
continue
|
||||
# Also check the element itself
|
||||
elem_style = await elem.evaluate('el => window.getComputedStyle(el).textDecoration')
|
||||
if 'line-through' in elem_style:
|
||||
logger.debug(f" Skipping element with line-through style")
|
||||
continue
|
||||
except:
|
||||
pass
|
||||
@@ -541,12 +563,32 @@ class ImprovedSearchService:
|
||||
try:
|
||||
price_val = float(match.group(1))
|
||||
if 0.01 < price_val < 100000:
|
||||
logger.debug(f"Found price: {price_val}€ from {selector}")
|
||||
return price_val
|
||||
logger.debug(f" Found candidate price: {price_val}€ from {selector}")
|
||||
all_prices.append(price_val)
|
||||
except ValueError:
|
||||
continue
|
||||
except Exception:
|
||||
except Exception as e:
|
||||
logger.debug(f" Error with selector {selector}: {e}")
|
||||
continue
|
||||
|
||||
# Return the LOWEST price found (promotional price is usually lower)
|
||||
if all_prices:
|
||||
lowest_price = min(all_prices)
|
||||
logger.debug(f"Selected lowest price from {len(all_prices)} candidates: {lowest_price}€")
|
||||
return lowest_price
|
||||
|
||||
# STRATEGY 3: AI Fallback (Gemma 2 / OpenRouter)
|
||||
try:
|
||||
logger.info(" 🤖 CSS extraction failed, attempting AI extraction...")
|
||||
html_content = await page.content()
|
||||
title = await page.title()
|
||||
|
||||
ai_price = await AIPriceExtractor.extract_price(html_content, title)
|
||||
if ai_price:
|
||||
logger.info(f" 🤖 AI found price: {ai_price}€")
|
||||
return ai_price
|
||||
except Exception as e:
|
||||
logger.error(f" AI fallback failed: {e}")
|
||||
|
||||
logger.debug("No price found")
|
||||
return None
|
||||
|
||||
Reference in new issue
Block a user