diff --git a/app/services/scheduler_service.py b/app/services/scheduler_service.py index 43045ba..29a4d25 100644 --- a/app/services/scheduler_service.py +++ b/app/services/scheduler_service.py @@ -107,36 +107,91 @@ async def process_item_check(item_id: int): # Use independent ScraperService for tracking from app.services.tracking_scraper_service import ScraperService - screenshot_path, page_text = await ScraperService.scrape_item( + screenshot_path, html_content = await ScraperService.scrape_item( url=item_data["url"], selector=item_data["selector"], - item_id=item_id + item_id=item_id, + return_html=True ) - - # Determine availability based on content presence - # If we got content, we assume available unless proven otherwise - is_available = bool(page_text and len(page_text) > 500) - - # Update availability status in database - with database.SessionLocal() as session: - if item := session.query(models.Item).filter(models.Item.id == item_id).first(): - item.is_available = is_available - if not is_available: - logger.warning(f"Product {item_id} marked as unavailable") - item.last_error = "Product no longer available (404 or not found)" - session.commit() - if not screenshot_path: raise Exception("Failed to capture screenshot") - - # NOTE: ScraperService already saves to screenshots/item_{item_id}.png - # so we don't need to copy it manually anymore. - if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=page_text)): - raise Exception("AI analysis failed") + # HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService) + # 1. Try specialized parser (if available) or JSON-LD + price = None + in_stock = True + + # Site-specific parser (Gifi) + from urllib.parse import urlparse + domain = urlparse(item_data["url"]).netloc + if "gifi.fr" in domain: + from app.services.parsers.gifi_parser import GifiParser + try: + parser = GifiParser() + details = parser.parse_product_details(html_content, item_data["url"]) + if details.get("price") is not None: + price = details["price"] + in_stock = details.get("in_stock", True) + logger.info(f"GifiParser found price: {price}€") + except Exception as e: + logger.debug(f"GifiParser failed: {e}") + + # Simple JSON-LD extract (if not found by specific parser) + if price is None: + import json + try: + from bs4 import BeautifulSoup + soup = BeautifulSoup(html_content, "html.parser") + scripts = soup.find_all("script", type="application/ld+json") + for script in scripts: + if script.string: + try: + data = json.loads(script.string) + if isinstance(data, list): data = data[0] + if data.get("@type") == "Product" and "offers" in data: + offers = data["offers"] + if isinstance(offers, list) and offers: offers = offers[0] + if "price" in offers: + price = float(str(offers["price"]).replace(',', '.')) + break + except: pass + except Exception as e: + logger.debug(f"JSON-LD extraction failed: {e}") + + # 2. Try AIPriceExtractor (Text AI - Gemma 3) - Very reliable for search + from app.services.ai_price_extractor import AIPriceExtractor + if price is None: + price = await AIPriceExtractor.extract_price(html_content, item_data["name"]) + if price: + logger.info(f"AIPriceExtractor (Text) found price: {price}€") + + # 3. Fallback/Verification with AIService (Vision AI - Gemini) + extraction = None + metadata = None + + if price is not None: + # Create a synthetic AI response if we already have a high-confidence price + extraction = AIExtractionResponse( + price=price, + in_stock=in_stock, + price_confidence=0.95, + in_stock_confidence=0.9, + source_type="text" + ) + metadata = AIExtractionMetadata( + model_name="hybrid-text-parser", + provider="internal-hybrid", + prompt_version="hybrid-v2", + repair_used=False + ) + else: + # Fallback to Vision AI if text extraction failed + # Use limited text context for Vision prompt to avoid token bloat + if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])): + raise Exception("AI analysis (Vision) failed") + extraction, metadata = ai_result - extraction, metadata = ai_result thresholds = await loop.run_in_executor(None, _get_thresholds) old_price, old_stock = await loop.run_in_executor( None, _update_db_result, item_id, extraction, metadata, thresholds, screenshot_path diff --git a/app/services/tracking_scraper_service.py b/app/services/tracking_scraper_service.py index 8d0e767..65a7de5 100644 --- a/app/services/tracking_scraper_service.py +++ b/app/services/tracking_scraper_service.py @@ -37,6 +37,13 @@ POPUP_SELECTORS = [ "button[id='didomi-notice-agree-button']", "span:has-text('Accepter & Fermer')", "button:has-text('Accepter & Fermer')", + # Common banners + "#sp-cc-accept", + "#onetrust-accept-btn-handler", + ".cookie-consent-accept", + "[data-action='accept-cookies']", + "button[id*='accept']", + "button[class*='accept']", ] @@ -122,10 +129,11 @@ class ScraperService: selector: str | None = None, item_id: int | None = None, config: ScrapeConfig | None = None, + return_html: bool = False, ) -> tuple[str | None, str]: """ Scrapes the given URL using Browserless and Playwright. - Returns a tuple: (screenshot_path, page_text) + Returns a tuple: (screenshot_path, page_text_or_html) """ if config is None: config = ScrapeConfig() @@ -155,10 +163,14 @@ class ScraperService: if config.smart_scroll: await ScraperService._smart_scroll(page, scroll_pixels) - page_text = await ScraperService._extract_text(page, config.text_length) + if return_html: + content_data = await page.content() + else: + content_data = await ScraperService._extract_text(page, config.text_length) + screenshot_path = await ScraperService._take_screenshot(page, url, item_id) - return screenshot_path, page_text + return screenshot_path, content_data finally: await context.close() @@ -174,15 +186,24 @@ class ScraperService: @staticmethod async def _create_context(browser: Browser) -> BrowserContext: + """Create context with stealth and locale (aligned with ImprovedSearchService)""" context = await browser.new_context( viewport={"width": 1920, "height": 1080}, user_agent=( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/120.0.0.0 Safari/537.36" + "Chrome/131.0.0.0 Safari/537.36" ), + locale="fr-FR", + timezone_id="Europe/Paris", ) - # Stealth mode / Ad blocking attempts + + # Stealth mode + await context.add_init_script(""" + Object.defineProperty(navigator, 'webdriver', { get: () => undefined }); + window.chrome = { runtime: {} }; + """) + await context.route("**/*", lambda route: route.continue_()) return context diff --git a/verify_ai_extractor.py b/verify_ai_extractor.py new file mode 100644 index 0000000..1af1bd7 --- /dev/null +++ b/verify_ai_extractor.py @@ -0,0 +1,40 @@ +import asyncio +import logging +import sys +import os + +# Add project root to path +sys.path.append(os.getcwd()) + +from app.services.ai_price_extractor import AIPriceExtractor + +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +async def verify_ai_extractor(): + print("Verifying AIPriceExtractor on Gifi dump...") + + # Load the dump file + dump_path = "gifi_full.html" + if not os.path.exists(dump_path): + print(f"Error: {dump_path} not found.") + return + + with open(dump_path, "r", encoding="utf-8") as f: + html = f.read() + + title = "Gifi Product Test" + + print("Calling AIPriceExtractor...") + price = await AIPriceExtractor.extract_price(html, title) + + print("\n--- AI Extraction Results ---") + print(f"Price: {price}€") + + if price is not None: + print("\nSUCCESS: AIPriceExtractor worked!") + else: + print("\nFAILURE: AIPriceExtractor returned None.") + +if __name__ == "__main__": + asyncio.run(verify_ai_extractor())