mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Add web scraping service for tracking and AI extraction verification script.
This commit is contained in:
1 parent
6fab64b229
commit
5428459f36
3 files changed
+143
-27
No files matched your search
@@ -107,36 +107,91 @@ async def process_item_check(item_id: int):
|
||||
# Use independent ScraperService for tracking
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
screenshot_path, page_text = await ScraperService.scrape_item(
|
||||
screenshot_path, html_content = await ScraperService.scrape_item(
|
||||
url=item_data["url"],
|
||||
selector=item_data["selector"],
|
||||
item_id=item_id
|
||||
item_id=item_id,
|
||||
return_html=True
|
||||
)
|
||||
|
||||
|
||||
# Determine availability based on content presence
|
||||
# If we got content, we assume available unless proven otherwise
|
||||
is_available = bool(page_text and len(page_text) > 500)
|
||||
|
||||
# Update availability status in database
|
||||
with database.SessionLocal() as session:
|
||||
if item := session.query(models.Item).filter(models.Item.id == item_id).first():
|
||||
item.is_available = is_available
|
||||
if not is_available:
|
||||
logger.warning(f"Product {item_id} marked as unavailable")
|
||||
item.last_error = "Product no longer available (404 or not found)"
|
||||
session.commit()
|
||||
|
||||
if not screenshot_path:
|
||||
raise Exception("Failed to capture screenshot")
|
||||
|
||||
# NOTE: ScraperService already saves to screenshots/item_{item_id}.png
|
||||
# so we don't need to copy it manually anymore.
|
||||
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=page_text)):
|
||||
raise Exception("AI analysis failed")
|
||||
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
|
||||
# 1. Try specialized parser (if available) or JSON-LD
|
||||
price = None
|
||||
in_stock = True
|
||||
|
||||
# Site-specific parser (Gifi)
|
||||
from urllib.parse import urlparse
|
||||
domain = urlparse(item_data["url"]).netloc
|
||||
if "gifi.fr" in domain:
|
||||
from app.services.parsers.gifi_parser import GifiParser
|
||||
try:
|
||||
parser = GifiParser()
|
||||
details = parser.parse_product_details(html_content, item_data["url"])
|
||||
if details.get("price") is not None:
|
||||
price = details["price"]
|
||||
in_stock = details.get("in_stock", True)
|
||||
logger.info(f"GifiParser found price: {price}€")
|
||||
except Exception as e:
|
||||
logger.debug(f"GifiParser failed: {e}")
|
||||
|
||||
# Simple JSON-LD extract (if not found by specific parser)
|
||||
if price is None:
|
||||
import json
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
scripts = soup.find_all("script", type="application/ld+json")
|
||||
for script in scripts:
|
||||
if script.string:
|
||||
try:
|
||||
data = json.loads(script.string)
|
||||
if isinstance(data, list): data = data[0]
|
||||
if data.get("@type") == "Product" and "offers" in data:
|
||||
offers = data["offers"]
|
||||
if isinstance(offers, list) and offers: offers = offers[0]
|
||||
if "price" in offers:
|
||||
price = float(str(offers["price"]).replace(',', '.'))
|
||||
break
|
||||
except: pass
|
||||
except Exception as e:
|
||||
logger.debug(f"JSON-LD extraction failed: {e}")
|
||||
|
||||
# 2. Try AIPriceExtractor (Text AI - Gemma 3) - Very reliable for search
|
||||
from app.services.ai_price_extractor import AIPriceExtractor
|
||||
if price is None:
|
||||
price = await AIPriceExtractor.extract_price(html_content, item_data["name"])
|
||||
if price:
|
||||
logger.info(f"AIPriceExtractor (Text) found price: {price}€")
|
||||
|
||||
# 3. Fallback/Verification with AIService (Vision AI - Gemini)
|
||||
extraction = None
|
||||
metadata = None
|
||||
|
||||
if price is not None:
|
||||
# Create a synthetic AI response if we already have a high-confidence price
|
||||
extraction = AIExtractionResponse(
|
||||
price=price,
|
||||
in_stock=in_stock,
|
||||
price_confidence=0.95,
|
||||
in_stock_confidence=0.9,
|
||||
source_type="text"
|
||||
)
|
||||
metadata = AIExtractionMetadata(
|
||||
model_name="hybrid-text-parser",
|
||||
provider="internal-hybrid",
|
||||
prompt_version="hybrid-v2",
|
||||
repair_used=False
|
||||
)
|
||||
else:
|
||||
# Fallback to Vision AI if text extraction failed
|
||||
# Use limited text context for Vision prompt to avoid token bloat
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
|
||||
raise Exception("AI analysis (Vision) failed")
|
||||
extraction, metadata = ai_result
|
||||
|
||||
extraction, metadata = ai_result
|
||||
thresholds = await loop.run_in_executor(None, _get_thresholds)
|
||||
old_price, old_stock = await loop.run_in_executor(
|
||||
None, _update_db_result, item_id, extraction, metadata, thresholds, screenshot_path
|
||||
|
||||
@@ -37,6 +37,13 @@ POPUP_SELECTORS = [
|
||||
"button[id='didomi-notice-agree-button']",
|
||||
"span:has-text('Accepter & Fermer')",
|
||||
"button:has-text('Accepter & Fermer')",
|
||||
# Common banners
|
||||
"#sp-cc-accept",
|
||||
"#onetrust-accept-btn-handler",
|
||||
".cookie-consent-accept",
|
||||
"[data-action='accept-cookies']",
|
||||
"button[id*='accept']",
|
||||
"button[class*='accept']",
|
||||
]
|
||||
|
||||
|
||||
@@ -122,10 +129,11 @@ class ScraperService:
|
||||
selector: str | None = None,
|
||||
item_id: int | None = None,
|
||||
config: ScrapeConfig | None = None,
|
||||
return_html: bool = False,
|
||||
) -> tuple[str | None, str]:
|
||||
"""
|
||||
Scrapes the given URL using Browserless and Playwright.
|
||||
Returns a tuple: (screenshot_path, page_text)
|
||||
Returns a tuple: (screenshot_path, page_text_or_html)
|
||||
"""
|
||||
if config is None:
|
||||
config = ScrapeConfig()
|
||||
@@ -155,10 +163,14 @@ class ScraperService:
|
||||
if config.smart_scroll:
|
||||
await ScraperService._smart_scroll(page, scroll_pixels)
|
||||
|
||||
page_text = await ScraperService._extract_text(page, config.text_length)
|
||||
if return_html:
|
||||
content_data = await page.content()
|
||||
else:
|
||||
content_data = await ScraperService._extract_text(page, config.text_length)
|
||||
|
||||
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
|
||||
|
||||
return screenshot_path, page_text
|
||||
return screenshot_path, content_data
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
@@ -174,15 +186,24 @@ class ScraperService:
|
||||
|
||||
@staticmethod
|
||||
async def _create_context(browser: Browser) -> BrowserContext:
|
||||
"""Create context with stealth and locale (aligned with ImprovedSearchService)"""
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent=(
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/120.0.0.0 Safari/537.36"
|
||||
"Chrome/131.0.0.0 Safari/537.36"
|
||||
),
|
||||
locale="fr-FR",
|
||||
timezone_id="Europe/Paris",
|
||||
)
|
||||
# Stealth mode / Ad blocking attempts
|
||||
|
||||
# Stealth mode
|
||||
await context.add_init_script("""
|
||||
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
|
||||
window.chrome = { runtime: {} };
|
||||
""")
|
||||
|
||||
await context.route("**/*", lambda route: route.continue_())
|
||||
return context
|
||||
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.ai_price_extractor import AIPriceExtractor
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_ai_extractor():
|
||||
print("Verifying AIPriceExtractor on Gifi dump...")
|
||||
|
||||
# Load the dump file
|
||||
dump_path = "gifi_full.html"
|
||||
if not os.path.exists(dump_path):
|
||||
print(f"Error: {dump_path} not found.")
|
||||
return
|
||||
|
||||
with open(dump_path, "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
title = "Gifi Product Test"
|
||||
|
||||
print("Calling AIPriceExtractor...")
|
||||
price = await AIPriceExtractor.extract_price(html, title)
|
||||
|
||||
print("\n--- AI Extraction Results ---")
|
||||
print(f"Price: {price}€")
|
||||
|
||||
if price is not None:
|
||||
print("\nSUCCESS: AIPriceExtractor worked!")
|
||||
else:
|
||||
print("\nFAILURE: AIPriceExtractor returned None.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_ai_extractor())
|
||||
Reference in new issue
Block a user