feat: Add web scraping service for tracking and AI extraction verification script.

This commit is contained in:
Michael committed 2025-12-18 11:05:09 +01:00
1 parent 6fab64b229
commit 5428459f36
3 files changed
+143 -27

No files matched your search

+77 -22
View File
@@ -107,36 +107,91 @@ async def process_item_check(item_id: int):
# Use independent ScraperService for tracking
from app.services.tracking_scraper_service import ScraperService
screenshot_path, page_text = await ScraperService.scrape_item(
screenshot_path, html_content = await ScraperService.scrape_item(
url=item_data["url"],
selector=item_data["selector"],
item_id=item_id
item_id=item_id,
return_html=True
)
# Determine availability based on content presence
# If we got content, we assume available unless proven otherwise
is_available = bool(page_text and len(page_text) > 500)
# Update availability status in database
with database.SessionLocal() as session:
if item := session.query(models.Item).filter(models.Item.id == item_id).first():
item.is_available = is_available
if not is_available:
logger.warning(f"Product {item_id} marked as unavailable")
item.last_error = "Product no longer available (404 or not found)"
session.commit()
if not screenshot_path:
raise Exception("Failed to capture screenshot")
# NOTE: ScraperService already saves to screenshots/item_{item_id}.png
# so we don't need to copy it manually anymore.
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=page_text)):
raise Exception("AI analysis failed")
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
# 1. Try specialized parser (if available) or JSON-LD
price = None
in_stock = True
# Site-specific parser (Gifi)
from urllib.parse import urlparse
domain = urlparse(item_data["url"]).netloc
if "gifi.fr" in domain:
from app.services.parsers.gifi_parser import GifiParser
try:
parser = GifiParser()
details = parser.parse_product_details(html_content, item_data["url"])
if details.get("price") is not None:
price = details["price"]
in_stock = details.get("in_stock", True)
logger.info(f"GifiParser found price: {price}€")
except Exception as e:
logger.debug(f"GifiParser failed: {e}")
# Simple JSON-LD extract (if not found by specific parser)
if price is None:
import json
try:
from bs4 import BeautifulSoup
soup = BeautifulSoup(html_content, "html.parser")
scripts = soup.find_all("script", type="application/ld+json")
for script in scripts:
if script.string:
try:
data = json.loads(script.string)
if isinstance(data, list): data = data[0]
if data.get("@type") == "Product" and "offers" in data:
offers = data["offers"]
if isinstance(offers, list) and offers: offers = offers[0]
if "price" in offers:
price = float(str(offers["price"]).replace(',', '.'))
break
except: pass
except Exception as e:
logger.debug(f"JSON-LD extraction failed: {e}")
# 2. Try AIPriceExtractor (Text AI - Gemma 3) - Very reliable for search
from app.services.ai_price_extractor import AIPriceExtractor
if price is None:
price = await AIPriceExtractor.extract_price(html_content, item_data["name"])
if price:
logger.info(f"AIPriceExtractor (Text) found price: {price}€")
# 3. Fallback/Verification with AIService (Vision AI - Gemini)
extraction = None
metadata = None
if price is not None:
# Create a synthetic AI response if we already have a high-confidence price
extraction = AIExtractionResponse(
price=price,
in_stock=in_stock,
price_confidence=0.95,
in_stock_confidence=0.9,
source_type="text"
)
metadata = AIExtractionMetadata(
model_name="hybrid-text-parser",
provider="internal-hybrid",
prompt_version="hybrid-v2",
repair_used=False
)
else:
# Fallback to Vision AI if text extraction failed
# Use limited text context for Vision prompt to avoid token bloat
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
raise Exception("AI analysis (Vision) failed")
extraction, metadata = ai_result
extraction, metadata = ai_result
thresholds = await loop.run_in_executor(None, _get_thresholds)
old_price, old_stock = await loop.run_in_executor(
None, _update_db_result, item_id, extraction, metadata, thresholds, screenshot_path
+26 -5
View File
@@ -37,6 +37,13 @@ POPUP_SELECTORS = [
"button[id='didomi-notice-agree-button']",
"span:has-text('Accepter & Fermer')",
"button:has-text('Accepter & Fermer')",
# Common banners
"#sp-cc-accept",
"#onetrust-accept-btn-handler",
".cookie-consent-accept",
"[data-action='accept-cookies']",
"button[id*='accept']",
"button[class*='accept']",
]
@@ -122,10 +129,11 @@ class ScraperService:
selector: str | None = None,
item_id: int | None = None,
config: ScrapeConfig | None = None,
return_html: bool = False,
) -> tuple[str | None, str]:
"""
Scrapes the given URL using Browserless and Playwright.
Returns a tuple: (screenshot_path, page_text)
Returns a tuple: (screenshot_path, page_text_or_html)
"""
if config is None:
config = ScrapeConfig()
@@ -155,10 +163,14 @@ class ScraperService:
if config.smart_scroll:
await ScraperService._smart_scroll(page, scroll_pixels)
page_text = await ScraperService._extract_text(page, config.text_length)
if return_html:
content_data = await page.content()
else:
content_data = await ScraperService._extract_text(page, config.text_length)
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
return screenshot_path, page_text
return screenshot_path, content_data
finally:
await context.close()
@@ -174,15 +186,24 @@ class ScraperService:
@staticmethod
async def _create_context(browser: Browser) -> BrowserContext:
"""Create context with stealth and locale (aligned with ImprovedSearchService)"""
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent=(
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36"
"Chrome/131.0.0.0 Safari/537.36"
),
locale="fr-FR",
timezone_id="Europe/Paris",
)
# Stealth mode / Ad blocking attempts
# Stealth mode
await context.add_init_script("""
Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
window.chrome = { runtime: {} };
""")
await context.route("**/*", lambda route: route.continue_())
return context
+40
View File
@@ -0,0 +1,40 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.ai_price_extractor import AIPriceExtractor
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_ai_extractor():
print("Verifying AIPriceExtractor on Gifi dump...")
# Load the dump file
dump_path = "gifi_full.html"
if not os.path.exists(dump_path):
print(f"Error: {dump_path} not found.")
return
with open(dump_path, "r", encoding="utf-8") as f:
html = f.read()
title = "Gifi Product Test"
print("Calling AIPriceExtractor...")
price = await AIPriceExtractor.extract_price(html, title)
print("\n--- AI Extraction Results ---")
print(f"Price: {price}€")
if price is not None:
print("\nSUCCESS: AIPriceExtractor worked!")
else:
print("\nFAILURE: AIPriceExtractor returned None.")
if __name__ == "__main__":
asyncio.run(verify_ai_extractor())