diff --git a/debug_amazon.py b/debug_amazon.py deleted file mode 100644 index 2497fbf..0000000 --- a/debug_amazon.py +++ /dev/null @@ -1,60 +0,0 @@ -import asyncio -import logging -import os -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - -BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000") - -async def debug_amazon(): - logger.info("Starting Amazon Debug Script") - - async with async_playwright() as p: - try: - logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}") - browser = await p.chromium.connect_over_cdp(BROWSERLESS_URL) - - # Use a very standard, recent User-Agent - user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" - - context = await browser.new_context( - user_agent=user_agent, - viewport={"width": 1920, "height": 1080}, - locale="fr-FR", - timezone_id="Europe/Paris" - ) - - page = await context.new_page() - - url = "https://www.amazon.fr/s?k=iphone" - logger.info(f"Navigating to {url}") - - response = await page.goto(url, wait_until="domcontentloaded", timeout=30000) - - if response: - status = response.status - logger.info(f"Response Status: {status}") - - content = await page.content() - if "api-services-support@amazon.com" in content or "Toutes nos excuses" in content: - logger.error("BLOCK DETECTED: Found blocking message in content") - else: - logger.info("No obvious blocking message found") - - # Save screenshot - await page.screenshot(path="debug_amazon_screenshot.png") - logger.info("Screenshot saved to debug_amazon_screenshot.png") - - else: - logger.error("No response received") - - await browser.close() - - except Exception as e: - logger.error(f"An error occurred: {e}") - -if __name__ == "__main__": - asyncio.run(debug_amazon()) diff --git a/debug_cataloguemate.py b/debug_cataloguemate.py deleted file mode 100644 index 78adf2c..0000000 --- a/debug_cataloguemate.py +++ /dev/null @@ -1,121 +0,0 @@ -import asyncio -import logging -import sys -from bs4 import BeautifulSoup -from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s - %(levelname)s - %(message)s", - handlers=[logging.StreamHandler(sys.stdout)] -) -logger = logging.getLogger(__name__) - -BASE_URL = "https://www.cataloguemate.fr" - -async def debug_catalog_list(): - """Debug the catalog list extraction""" - # Test with Gifi - slug = "gifi" - url = f"{BASE_URL}/{slug}/" - - logger.info(f"--- DEBUGGING LIST: {url} ---") - - browser_config = BrowserConfig(headless=True) - run_config = CrawlerRunConfig( - cache_mode=CacheMode.BYPASS, - wait_for_images=True, - ) - - async with AsyncWebCrawler(config=browser_config) as crawler: - result = await crawler.arun(url=url, config=run_config) - - if not result.success: - logger.error(f"Failed to fetch {url}: {result.error_message}") - return None - - logger.info(f"Successfully fetched {url} ({len(result.html)} chars)") - - soup = BeautifulSoup(result.html, 'html.parser') - - # 1. Dump all links to see what we have - links = soup.find_all('a', href=True) - logger.info(f"Found {len(links)} links total") - - potential_catalogs = [] - - for i, link in enumerate(links): - href = link['href'] - text = link.get_text(strip=True) - - # Normalize - if href.startswith(BASE_URL): - href = href.replace(BASE_URL, "") - - # Log interesting links - if slug in href or "catalogue" in href.lower(): - logger.info(f"Link {i}: {href} | Text: '{text}'") - - # Apply our filter logic to see if it passes - if href.startswith(f"/{slug}/") and href != f"/{slug}/": - if not any(x in href for x in ["offres", "magasins", "rechercher"]): - potential_catalogs.append(href) - logger.info(f" -> MATCHES FILTER!") - - logger.info(f"Total matching catalogs: {len(potential_catalogs)}") - return potential_catalogs[0] if potential_catalogs else None - -async def debug_catalog_page(catalog_rel_url): - """Debug the catalog page extraction""" - if not catalog_rel_url: - logger.error("No catalog URL to debug") - return - - full_url = f"{BASE_URL}{catalog_rel_url}" - logger.info(f"\n--- DEBUGGING PAGE: {full_url} ---") - - browser_config = BrowserConfig(headless=True) - run_config = CrawlerRunConfig( - cache_mode=CacheMode.BYPASS, - wait_for_images=True, - delay_before_return_html=2.0 # Wait a bit more - ) - - async with AsyncWebCrawler(config=browser_config) as crawler: - result = await crawler.arun(url=full_url, config=run_config) - - if not result.success: - logger.error(f"Failed to fetch {full_url}") - return - - soup = BeautifulSoup(result.html, 'html.parser') - - # 1. Dump all images - images = soup.find_all('img') - logger.info(f"Found {len(images)} images") - - for i, img in enumerate(images): - src = img.get('src', '') - width = img.get('width', '?') - height = img.get('height', '?') - alt = img.get('alt', '') - - # Filter noise - if "logo" in src or "icon" in src: - continue - - logger.info(f"Img {i}: {src} | {width}x{height} | Alt: {alt}") - - # Check our heuristic - is_likely = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images']) - if is_likely: - logger.info(" -> LIKELY CATALOG IMAGE") - -if __name__ == "__main__": - async def main(): - cat_url = await debug_catalog_list() - if cat_url: - await debug_catalog_page(cat_url) - - asyncio.run(main()) diff --git a/debug_cataloguemate_v2.py b/debug_cataloguemate_v2.py deleted file mode 100644 index 7973abc..0000000 --- a/debug_cataloguemate_v2.py +++ /dev/null @@ -1,96 +0,0 @@ -import asyncio -import logging -import sys -# from bs4 import BeautifulSoup # Removed -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s - %(levelname)s - %(message)s", - handlers=[logging.StreamHandler(sys.stdout)] -) -logger = logging.getLogger(__name__) - -BASE_URL = "https://www.cataloguemate.fr" - -async def debug_catalog_list(): - """Debug the catalog list extraction using Playwright directly""" - slug = "gifi" - url = f"{BASE_URL}/offres/paris/{slug}/" - - logger.info(f"--- DEBUGGING LIST: {url} ---") - - async with async_playwright() as p: - # Use browserless or local depending on connection - # For this script we use local headless for simplicity if browserless is not available, - # but since we are in the container we might need browserless. - # Let's try to simulate what browserless_service does but simplified. - - try: - browser = await p.chromium.launch(headless=True) # Try local first - except: - logger.info("Local browser failed, trying browserless...") - browser = await p.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" - ) - page = await context.new_page() - - logger.info(f"Navigating to {url}") - await page.goto(url, wait_until="domcontentloaded") - - # Wait a bit - await page.wait_for_timeout(2000) - - content = await page.content() - # soup = BeautifulSoup(content, 'html.parser') - - # 1. Dump all links to see what we have - # Use JS to extract links - links_data = await page.evaluate(""" - () => { - return Array.from(document.querySelectorAll('a[href]')).map(a => ({ - href: a.getAttribute('href'), - text: a.innerText.trim() - })); - } - """) - - logger.info(f"Found {len(links_data)} links total") - - potential_catalogs = [] - - for i, link in enumerate(links_data): - href = link['href'] - text = link['text'] - - # Normalize - if href.startswith(BASE_URL): - href = href.replace(BASE_URL, "") - - # Log interesting links - if slug in href or "catalogue" in href.lower(): - logger.info(f"Link {i}: {href} | Text: '{text}'") - - # Apply our filter logic to see if it passes - if f"/{slug}/" in href: - if not any(x in href for x in ["/offres/", "/magasins/", "/rechercher/", "page="]): - # Check ID pattern - import re - if re.search(r'-\d+/?$', href) or "catalogue" in href.lower(): - potential_catalogs.append(href) - logger.info(f" -> MATCHES FILTER! (Found catalog)") - else: - logger.info(f" -> Rejected (no ID/keyword)") - else: - logger.info(f" -> Rejected (invalid pattern)") - - logger.info(f"Total matching catalogs: {len(potential_catalogs)}") - - await browser.close() - return potential_catalogs[0] if potential_catalogs else None - -if __name__ == "__main__": - asyncio.run(debug_catalog_list()) diff --git a/debug_gifi.py b/debug_gifi.py deleted file mode 100644 index be24243..0000000 --- a/debug_gifi.py +++ /dev/null @@ -1,50 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.services.improved_search_service import ImprovedSearchService - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def debug_gifi(): - print("Initializing search service...") - await ImprovedSearchService.initialize() - - try: - print("Searching Gifi for 'chaise'...") - results = await ImprovedSearchService.search_site("gifi.fr", "chaise") - - print(f"\nFound {len(results)} results.") - - if results: - print("\n--- First 5 Results ---") - for i, res in enumerate(results[:5]): - print(f"\nItem {i+1}:") - print(f" Title: {res.title}") - print(f" Price: {res.price} {res.currency}") - print(f" Image: {res.image_url}") - print(f" URL: {res.url}") - print(f" In Stock: {res.in_stock}") - - # Check for missing critical data - missing_price = sum(1 for r in results if r.price is None) - missing_image = sum(1 for r in results if not r.image_url) - - print(f"\nStats:") - print(f" Total: {len(results)}") - print(f" Missing Price: {missing_price}") - print(f" Missing Image: {missing_image}") - - except Exception as e: - logger.error(f"Error: {e}", exc_info=True) - finally: - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(debug_gifi()) diff --git a/debug_missing_sites.py b/debug_missing_sites.py deleted file mode 100644 index 730e92f..0000000 --- a/debug_missing_sites.py +++ /dev/null @@ -1,120 +0,0 @@ -""" -Debug script to identify correct selectors for missing sites -""" -import asyncio -import sys -import os -from pathlib import Path - -# Add app directory to path -sys.path.insert(0, str(Path(__file__).parent)) - -from app.services.browserless_service import browserless_service -from bs4 import BeautifulSoup - -async def analyze_site(name: str, url: str, wait_selector: str = None): - """Analyze a search results page to identify selectors""" - print(f"\n{'='*80}") - print(f"Analyzing: {name}") - print(f"URL: {url}") - print(f"{'='*80}\n") - - html, screenshot = await browserless_service.get_page_content( - url, - use_proxy=False, - wait_selector=wait_selector, - wait_timeout=10000 - ) - - if not html: - print(f"❌ Failed to get content for {name}") - return - - soup = BeautifulSoup(html, "html.parser") - - # Save HTML for manual inspection - output_file = f"debug_{name.lower().replace(' ', '_')}.html" - with open(output_file, "w", encoding="utf-8") as f: - f.write(html) - print(f"💾 HTML saved to: {output_file}") - - # Common product link patterns - product_patterns = [ - "a[href*='/product']", - "a[href*='/p/']", - "a[href*='/produit']", - "a.product-card", - "a.product-link", - "[data-product-id]", - "article a", - "div[data-testid*='product'] a", - ] - - print("\n🔍 Searching for product links...") - for pattern in product_patterns: - links = soup.select(pattern) - if links and len(links) >= 3: - print(f"✅ Found {len(links)} matches for: {pattern}") - # Show first 3 examples - for i, link in enumerate(links[:3], 1): - href = link.get('href', 'NO_HREF') - text = link.get_text(strip=True)[:50] - print(f" {i}. {href[:60]} | {text}") - elif links: - print(f"⚠️ Found {len(links)} matches for: {pattern} (too few)") - - # Common image patterns - image_patterns = [ - "img[src*='product']", - "img.product-image", - "img.product-img", - "img[loading='lazy']", - "picture img", - "img[data-src]", - ] - - print("\n🖼️ Searching for product images...") - for pattern in image_patterns: - images = soup.select(pattern) - if images and len(images) >= 3: - print(f"✅ Found {len(images)} matches for: {pattern}") - for i, img in enumerate(images[:3], 1): - src = img.get('src') or img.get('data-src', 'NO_SRC') - alt = img.get('alt', 'NO_ALT')[:50] - print(f" {i}. {src[:60]} | {alt}") - elif images: - print(f"⚠️ Found {len(images)} matches for: {pattern} (too few)") - - print(f"\n✅ Analysis complete for {name}\n") - -async def main(): - """Test all missing sites""" - sites = [ - { - "name": "E.Leclerc", - "url": "https://www.e-leclerc.com/recherche?text=chaise", - "wait": ".product-card, .search-results" - }, - { - "name": "Auchan", - "url": "https://www.auchan.fr/search?text=chaise", - "wait": ".product-card, .product-item" - }, - { - "name": "Carrefour", - "url": "https://www.carrefour.fr/s?q=chaise", - "wait": "[data-testid*='product'], .product" - }, - ] - - for site in sites: - try: - await analyze_site(site["name"], site["url"], site.get("wait")) - except Exception as e: - print(f"❌ Error analyzing {site['name']}: {e}") - - # Small delay between sites - await asyncio.sleep(2) - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/debug_scraper.py b/debug_scraper.py deleted file mode 100644 index fd0d846..0000000 --- a/debug_scraper.py +++ /dev/null @@ -1,67 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.services.browserless_service import browserless_service -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def debug_site(site_key: str, query: str = "chaise"): - print(f"--- Debugging {site_key} ---") - config = SITE_CONFIGS.get(site_key) - if not config: - print(f"Site {site_key} not found in config") - return - - search_url = config["search_url"].format(query=query) - print(f"URL: {search_url}") - - print("Fetching content...") - try: - html, screenshot_path = await browserless_service.get_page_content( - search_url, - use_proxy=config.get("requires_proxy", False), - wait_selector=config.get("wait_selector") - ) - - print(f"Screenshot saved to: {screenshot_path}") - print(f"HTML length: {len(html)}") - - # Save HTML for inspection - with open(f"debug_{site_key}.html", "w", encoding="utf-8") as f: - f.write(html) - print(f"HTML saved to debug_{site_key}.html") - - # Check if wait selector is present in HTML - if config.get("wait_selector"): - from bs4 import BeautifulSoup - soup = BeautifulSoup(html, "html.parser") - found = soup.select(config["wait_selector"]) - print(f"Wait selector '{config['wait_selector']}' found: {len(found)} elements") - - except Exception as e: - print(f"Error: {e}") - -async def main(): - if len(sys.argv) < 2: - print("Usage: python debug_scraper.py [query]") - return - - site_key = sys.argv[1] - query = sys.argv[2] if len(sys.argv) > 2 else "chaise" - - await browserless_service.start() - try: - await debug_site(site_key, query) - finally: - await browserless_service.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/debug_search.py b/debug_search.py deleted file mode 100644 index 3d447aa..0000000 --- a/debug_search.py +++ /dev/null @@ -1,54 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) - -from app.services.search_service import new_search_service -from app.services.browserless_service import browserless_service -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s - %(name)s - %(levelname)s - %(message)s", - handlers=[logging.StreamHandler()] -) - -async def debug_search(): - print("--- Debugging Search ---") - - # Check Config - print(f"Available Config Keys: {list(SITE_CONFIGS.keys())}") - - sites = ["amazon.fr", "stokomani.fr", "lincroyable.fr"] - query = "iphone" - - try: - await browserless_service.initialize() - - for site in sites: - print(f"\nTesting {site}...") - if site not in SITE_CONFIGS: - print(f"❌ Site {site} NOT found in SITE_CONFIGS") - continue - - try: - count = 0 - async for r in new_search_service.search_site_generator(site, query): - count += 1 - print(f"✅ Found: {r.title} - {r.price} {r.currency}") - if count >= 1: - break - if count == 0: - print(f"⚠️ No results for {site}") - except Exception as e: - print(f"❌ Error searching {site}: {e}") - - finally: - await browserless_service.shutdown() - -if __name__ == "__main__": - asyncio.run(debug_search()) diff --git a/debug_sites.py b/debug_sites.py deleted file mode 100644 index 4b45bcc..0000000 --- a/debug_sites.py +++ /dev/null @@ -1,30 +0,0 @@ -import sys -import os - -# Add app to path -sys.path.append(os.getcwd()) - -from app.core.database import SessionLocal -from app.models import SearchSite -from app.core.search_config import SITE_CONFIGS - -def debug_sites(): - db = SessionLocal() - try: - print("=== SITE_CONFIGS Keys ===") - for key in SITE_CONFIGS.keys(): - print(f"- {key}") - - print("\n=== DB SearchSites ===") - sites = db.query(SearchSite).all() - for site in sites: - print(f"ID: {site.id} | Name: {site.name} | Domain: {site.domain} | Active: {site.is_active}") - print(f" URL: {site.search_url}") - print(f" Selector: {site.product_link_selector}") - print("-" * 20) - - finally: - db.close() - -if __name__ == "__main__": - debug_sites() diff --git a/task.md b/task.md index 0c498a9..1fe2169 100644 --- a/task.md +++ b/task.md @@ -16,6 +16,9 @@ - [x] Reproduce failure (Confirmed Browserless issue via analysis). - [x] Fix `cataloguemate_scraper.py` with HTTP fallback. +- [x] **Cleanup**: + - [x] Deleted `debug_*.py` and `verify_*.py` files. + ## 📝 Progress Log - **2025-12-22**: Vision Priority implemented and verified. diff --git a/verify_action.py b/verify_action.py deleted file mode 100644 index e7236d2..0000000 --- a/verify_action.py +++ /dev/null @@ -1,79 +0,0 @@ -import asyncio -import logging -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_action(): - async with async_playwright() as p: - # Use a standard User Agent - ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" - - browser = await p.chromium.launch(headless=True) - context = await browser.new_context(user_agent=ua) - page = await context.new_page() - - # 1. Perform Search - logger.info("--- Step 1: Searching for 'Chaise' on Action ---") - search_url = "https://www.action.com/fr-fr/search/?q=Chaise" - - try: - await page.goto(search_url, wait_until="domcontentloaded", timeout=30000) - # Wait a bit for any JS redirects or challenges - await page.wait_for_timeout(5000) - except Exception as e: - logger.error(f"Navigation error: {e}") - - # 2. Take Screenshot - await page.screenshot(path="action_debug.png") - logger.info("Screenshot saved: action_debug.png") - - # 3. Dump Content - content = await page.content() - logger.info(f"HTML Content Length: {len(content)}") - - # 4. Check for specific text - text = await page.inner_text("body") - logger.info(f"Page Text (first 500 chars): {text[:500]}") - - if "Challenge" in text or "human" in text or "Cloudflare" in text: - logger.warning("⚠️ Cloudflare Challenge detected in text!") - - # 5. Analyze Content for Product Links - # Action product URLs contain "/p/" - links = await page.evaluate(""" - Array.from(document.querySelectorAll('a')) - .map(a => a.href) - .filter(href => href.includes('/p/')) - """) - - logger.info(f"Found {len(links)} product links") - if links: - logger.info(f"First 5 links: {links[:5]}") - - # Find the parent container of the first link - parent_html = await page.evaluate(""" - (() => { - const link = document.querySelector("a[href*='/p/']"); - return link ? link.parentElement.outerHTML : "Not found"; - })() - """) - logger.info(f"Parent HTML of first link: {parent_html[:500]}") - - # Find the class of the link itself - link_class = await page.evaluate(""" - (() => { - const link = document.querySelector("a[href*='/p/']"); - return link ? link.className : "Not found"; - })() - """) - logger.info(f"Class of first link: {link_class}") - - - - await browser.close() - -if __name__ == "__main__": - asyncio.run(verify_action()) diff --git a/verify_ai_extractor.py b/verify_ai_extractor.py deleted file mode 100644 index 1af1bd7..0000000 --- a/verify_ai_extractor.py +++ /dev/null @@ -1,40 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.services.ai_price_extractor import AIPriceExtractor - -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_ai_extractor(): - print("Verifying AIPriceExtractor on Gifi dump...") - - # Load the dump file - dump_path = "gifi_full.html" - if not os.path.exists(dump_path): - print(f"Error: {dump_path} not found.") - return - - with open(dump_path, "r", encoding="utf-8") as f: - html = f.read() - - title = "Gifi Product Test" - - print("Calling AIPriceExtractor...") - price = await AIPriceExtractor.extract_price(html, title) - - print("\n--- AI Extraction Results ---") - print(f"Price: {price}€") - - if price is not None: - print("\nSUCCESS: AIPriceExtractor worked!") - else: - print("\nFAILURE: AIPriceExtractor returned None.") - -if __name__ == "__main__": - asyncio.run(verify_ai_extractor()) diff --git a/verify_ai_fallback.py b/verify_ai_fallback.py deleted file mode 100644 index c662953..0000000 --- a/verify_ai_fallback.py +++ /dev/null @@ -1,70 +0,0 @@ -import asyncio -import logging -from unittest.mock import MagicMock, patch -from app.services.ai_service import AIService - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -# Mock BadRequestError since we might not have litellm installed in the environment running this script -class MockBadRequestError(Exception): - pass - -async def verify_fallback(): - logger.info("Starting AI fallback verification...") - - # Mock config - config = { - "provider": "openrouter", - "model": "google/gemini-2.5-flash-image-preview", - "api_key": "fake-key", - "api_base": "https://openrouter.ai/api/v1", - "temperature": 0.1, - "max_tokens": 100, - "timeout": 30, - } - - # Mock success response - mock_response = MagicMock() - mock_response.choices = [MagicMock(message=MagicMock(content='{"price": 10.0}'))] - - # Patch acompletion - with patch("app.services.ai_service.acompletion") as mock_acompletion: - # Setup side effect: First call raises BadRequestError, second call succeeds - mock_acompletion.side_effect = [ - MockBadRequestError("400 Bad Request: The request is not supported by this model."), - mock_response - ] - - try: - # We need to patch the exception check in the code if we can't import the real exception - # But the code uses string check "BadRequestError" in str(type(e).__name__) - # So MockBadRequestError should work if we name it right or if the code checks "400" - - # Actually, let's just run it and see if our logic catches it. - # The code checks: is_bad_request = "BadRequestError" in str(type(e).__name__) or "400" in str(e) - # Our MockBadRequestError has "400" in the message, so it should be caught. - - logger.info("Calling call_llm...") - result = await AIService.call_llm("test prompt", "data:image/...", config) - - logger.info(f"Result: {result}") - - # Verify calls - assert mock_acompletion.call_count == 2 - logger.info("SUCCESS: acompletion was called twice (retry worked)") - - # Verify second call didn't have response_format - call_args = mock_acompletion.call_args_list[1] - kwargs = call_args.kwargs - if "response_format" not in kwargs: - logger.info("SUCCESS: Second call did not have response_format") - else: - logger.error("FAILURE: Second call still had response_format") - - except Exception as e: - logger.error(f"FAILURE: Exception raised: {e}") - -if __name__ == "__main__": - asyncio.run(verify_fallback()) diff --git a/verify_all_sites.py b/verify_all_sites.py deleted file mode 100644 index b891aba..0000000 --- a/verify_all_sites.py +++ /dev/null @@ -1,69 +0,0 @@ -import asyncio -import logging -import time -from app.services.improved_search_service import ImprovedSearchService -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - -async def verify_site(site_key: str): - print(f"\n{'='*50}") - print(f"🔍 Verifying {site_key}...") - print(f"{'='*50}") - - start_time = time.time() - first_result_time = None - count = 0 - - try: - async for result in ImprovedSearchService.search_site_generator(site_key, "chaise"): - count += 1 - if count == 1: - first_result_time = time.time() - elapsed = first_result_time - start_time - print(f"🚀 First result in {elapsed:.2f}s") - print(f" Title: {result.title}") - print(f" Price: {result.price} {result.currency}") - print(f" Image: {result.image_url}") - - # Print a dot for each result to show progress - print(".", end="", flush=True) - - total_time = time.time() - start_time - print(f"\n✅ Finished {site_key}: {count} results in {total_time:.2f}s") - - if count == 0: - print(f"❌ WARNING: 0 results found for {site_key}") - return False - return True - - except Exception as e: - print(f"\n❌ ERROR verifying {site_key}: {e}") - return False - -async def main(): - await ImprovedSearchService.initialize() - - sites = list(SITE_CONFIGS.keys()) - results = {} - - # Test all sites sequentially to avoid overwhelming the browser/network - for site in sites: - success = await verify_site(site) - results[site] = success - # Small pause between sites - await asyncio.sleep(2) - - await ImprovedSearchService.shutdown() - - print("\n" + "="*50) - print("SUMMARY") - print("="*50) - for site, success in results.items(): - status = "✅ PASS" if success else "❌ FAIL" - print(f"{status} - {site}") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/verify_amazon_popup.py b/verify_amazon_popup.py deleted file mode 100644 index 69a2270..0000000 --- a/verify_amazon_popup.py +++ /dev/null @@ -1,62 +0,0 @@ -import asyncio -from playwright.async_api import async_playwright - -async def main(): - async with async_playwright() as p: - # Launch with headless=True to mimic server environment - browser = await p.chromium.launch(headless=True) - context = await browser.new_context( - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36" - ) - page = await context.new_page() - - # Try a product URL that might trigger the check - url = "https://www.amazon.fr/dp/B07S58MPKW" - print(f"Navigating to {url}") - - try: - await page.goto(url) - await page.wait_for_timeout(5000) - - content = await page.content() - - if "Continuer les achats" in content: - print("🚨 Popup detected!") - with open("amazon_popup.html", "w", encoding="utf-8") as f: - f.write(content) - await page.screenshot(path="amazon_popup.png") - - # Analyze the button - print("Searching for button...") - - # Try various locators - locators = [ - "button", - "input[type='submit']", - "a.a-button-text", - "span.a-button-inner" - ] - - for sel in locators: - elements = page.locator(sel) - count = await elements.count() - for i in range(count): - el = elements.nth(i) - if await el.is_visible(): - txt = await el.inner_text() - val = await el.get_attribute("value") or "" - if "Continuer" in txt or "Continuer" in val: - print(f"✅ Found candidate: {sel}") - print(f" Text: {txt}") - print(f" Value: {val}") - print(f" OuterHTML: {await el.evaluate('el => el.outerHTML')}") - else: - print("No popup detected. Page title:", await page.title()) - - except Exception as e: - print(f"Error: {e}") - - await browser.close() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/verify_bm_extraction.py b/verify_bm_extraction.py deleted file mode 100644 index f3d12f1..0000000 --- a/verify_bm_extraction.py +++ /dev/null @@ -1,91 +0,0 @@ -import asyncio -import logging -import sys -from playwright.async_api import async_playwright -from app.services.ai_price_extractor import AIPriceExtractor -from app.services.improved_search_service import ImprovedSearchService -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_bm(): - async with async_playwright() as p: - browser = await p.chromium.launch(headless=True) - page = await browser.new_page() - - # 1. Perform Search to get a product URL - logger.info("--- Step 1: Searching for 'Chaise pliante pu creme' on B&M ---") - # Use specific search to find the problematic product - search_url = "https://bmstores.fr/module/ambjolisearch/jolisearch?s=Chaise+pliante+pu+creme" - await page.goto(search_url) - await page.wait_for_load_state("networkidle") - - # Take screenshot of search results - await page.screenshot(path="bm_search_results.png") - logger.info("Screenshot saved: bm_search_results.png") - - # Get first product link - product_link = await page.get_attribute("a.thumbnail.product-thumbnail", "href") - if not product_link: - logger.error("No product found in search") - return - - if not product_link.startswith("http"): - product_link = "https://www.bmstores.fr" + product_link - - logger.info(f"Testing Product URL: {product_link}") - - # 2. Go to Product Page - await page.goto(product_link) - await page.wait_for_load_state("networkidle") - await page.screenshot(path="bm_product_page.png") - logger.info("Screenshot saved: bm_product_page.png") - - # 3. Dump HTML snippet (price area) - content = await page.content() - logger.info(f"HTML Content Length: {len(content)}") - - # Check for 12.95 - if "12,95" in content or "12.95" in content: - logger.info("✅ Price 12.95 found in raw HTML") - else: - logger.warning("❌ Price 12.95 NOT found in raw HTML") - - # 4. Test JSON-LD - logger.info("\n--- Step 2: Testing JSON-LD ---") - json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]') - for i, script in enumerate(json_ld_scripts): - text = await script.inner_text() - logger.info(f"JSON-LD #{i}: {text[:500]}...") - - # 5. Test CSS Selectors - logger.info("\n--- Step 3: Testing CSS Selectors ---") - selectors = [ - '.price-current', '.prix-actuel', '.sale-price', '.promo-price', - '.price', '[data-testid="price"]', '[itemprop="price"]', '.product-price', - '.current-price-value' - ] - for sel in selectors: - elements = await page.query_selector_all(sel) - for el in elements: - text = await el.inner_text() - logger.info(f"Selector '{sel}': {text.strip()}") - - # 6. Test AI Extraction - logger.info("\n--- Step 4: Testing AI Extraction (Gemma 3) ---") - title = await page.title() - - # Ensure API key is available - import os - if not os.getenv("OPENROUTER_API_KEY"): - logger.warning("OPENROUTER_API_KEY not set in env, AI might fail") - - ai_price = await AIPriceExtractor.extract_price(content, title) - logger.info(f"AI Extracted Price: {ai_price}") - - await browser.close() - -if __name__ == "__main__": - asyncio.run(verify_bm()) diff --git a/verify_bonial.py b/verify_bonial.py deleted file mode 100644 index 90ba055..0000000 --- a/verify_bonial.py +++ /dev/null @@ -1,294 +0,0 @@ -""" -Verification Script for Bonial Catalog Module - -Tests database setup, scraper functionality, and API endpoints. -""" - -import asyncio -import logging -import sys -from datetime import datetime - -import httpx -from sqlalchemy import inspect, text - -from app.database import SessionLocal, engine -from app.models import Catalogue, CataloguePage, Enseigne, ScrapingLog -from app.services.bonial_scraper import scrape_enseigne -from app.services.seed_enseignes import seed_enseignes - -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -API_BASE_URL = "http://localhost:8555/api" - - -def check_database_tables(): - """Verify that catalog module tables exist.""" - logger.info("=" * 60) - logger.info("Phase 1: Checking Database Tables") - logger.info("=" * 60) - - inspector = inspect(engine) - required_tables = [ - "enseignes", - "catalogues", - "catalogue_pages", - "scraping_logs", - ] - - existing_tables = inspector.get_table_names() - - all_exist = True - for table in required_tables: - exists = table in existing_tables - status = "✅" if exists else "❌" - logger.info(f"{status} Table '{table}': {'EXISTS' if exists else 'MISSING'}") - if not exists: - all_exist = False - - if all_exist: - logger.info("\n✅ All required tables exist!\n") - return True - else: - logger.error("\n❌ Some tables are missing. Run migration: alembic upgrade head\n") - return False - - -def check_enseignes_seeding(): - """Verify that enseignes are seeded.""" - logger.info("=" * 60) - logger.info("Phase 2: Checking Enseignes Seeding") - logger.info("=" * 60) - - db = SessionLocal() - try: - count = db.query(Enseigne).count() - logger.info(f"Found {count} enseignes in database") - - if count == 0: - logger.info("Seeding enseignes...") - created = seed_enseignes(db) - logger.info(f"✅ Created {created} enseignes") - count = created - - if count >= 9: - logger.info("\n✅ All 9 enseignes are seeded!\n") - - # Display enseignes - enseignes = db.query(Enseigne).order_by(Enseigne.ordre_affichage).all() - for ens in enseignes: - active = "✅" if ens.is_active else "⚠️" - logger.info(f" {active} {ens.ordre_affichage}. {ens.nom} (slug: {ens.slug_bonial})") - - return True - else: - logger.warning(f"\n⚠️ Expected 9 enseignes, found {count}\n") - return False - - finally: - db.close() - - -async def test_scraper(): - """Test the Bonial scraper on one enseigne.""" - logger.info("=" * 60) - logger.info("Phase 3: Testing Bonial Scraper (Gifi)") - logger.info("=" * 60) - - db = SessionLocal() - try: - # Get Gifi enseigne - gifi = db.query(Enseigne).filter_by(slug_bonial="Gifi").first() - - if not gifi: - logger.error("❌ Gifi enseigne not found") - return False - - logger.info(f"Testing scraper for: {gifi.nom}") - logger.info("This may take 30-60 seconds...") - - # Run scraper - log = await scrape_enseigne(gifi, db) - - # Display results - logger.info(f"\nScraping completed:") - logger.info(f" Status: {log.statut}") - logger.info(f" Catalogues found: {log.catalogues_trouves}") - logger.info(f" New catalogues: {log.catalogues_nouveaux}") - logger.info(f" Duration: {log.duree_secondes:.2f}s") - - if log.message_erreur: - logger.warning(f" Error: {log.message_erreur}") - - if log.statut in ["success", "partial"] and log.catalogues_trouves > 0: - logger.info("\n✅ Scraper is working!\n") - - # Display sample catalog - cat = db.query(Catalogue).filter_by(enseigne_id=gifi.id).first() - if cat: - logger.info(f"Sample catalog:") - logger.info(f" Titre: {cat.titre}") - logger.info(f" Dates: {cat.date_debut.date()} → {cat.date_fin.date()}") - logger.info(f" Pages: {cat.nombre_pages}") - - return True - else: - logger.error("\n❌ Scraper failed or found no catalogs\n") - return False - - except Exception as e: - logger.error(f"❌ Error testing scraper: {e}") - return False - finally: - db.close() - - -async def test_api_endpoints(): - """Test API endpoints.""" - logger.info("=" * 60) - logger.info("Phase 4: Testing API Endpoints") - logger.info("=" * 60) - - async with httpx.AsyncClient(timeout=30.0) as client: - # Test 1: Get enseignes - logger.info("\n1. Testing GET /api/catalogues/enseignes") - try: - response = await client.get(f"{API_BASE_URL}/catalogues/enseignes") - if response.status_code == 200: - enseignes = response.json() - logger.info(f" ✅ Status 200 - Found {len(enseignes)} enseignes") - if enseignes: - logger.info(f" Sample: {enseignes[0]['nom']} ({enseignes[0]['catalogues_actifs_count']} catalogues)") - else: - logger.error(f" ❌ Status {response.status_code}") - return False - except Exception as e: - logger.error(f" ❌ Error: {e}") - return False - - # Test 2: Get catalogues - logger.info("\n2. Testing GET /api/catalogues") - try: - response = await client.get(f"{API_BASE_URL}/catalogues?page=1&limit=5") - if response.status_code == 200: - data = response.json() - catalogues = data.get("data", []) - pagination = data.get("pagination", {}) - logger.info(f" ✅ Status 200 - Found {pagination.get('total', 0)} catalogues") - logger.info(f" Page: {pagination.get('page')}/{pagination.get('pages_total')}") - if catalogues: - logger.info(f" Sample: {catalogues[0]['titre']}") - else: - logger.error(f" ❌ Status {response.status_code}") - return False - except Exception as e: - logger.error(f" ❌ Error: {e}") - return False - - # Test 3: Get catalogue detail - logger.info("\n3. Testing GET /api/catalogues/{id}") - try: - # Get first catalog ID - db = SessionLocal() - cat = db.query(Catalogue).first() - db.close() - - if cat: - response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}") - if response.status_code == 200: - detail = response.json() - logger.info(f" ✅ Status 200 - Catalogue: {detail['titre']}") - logger.info(f" Pages: {detail['nombre_pages']}") - else: - logger.error(f" ❌ Status {response.status_code}") - return False - else: - logger.warning(" ⚠️ No catalogues in DB to test detail endpoint") - except Exception as e: - logger.error(f" ❌ Error: {e}") - return False - - # Test 4: Get catalogue pages - logger.info("\n4. Testing GET /api/catalogues/{id}/pages") - try: - if cat: - response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}/pages") - if response.status_code == 200: - pages = response.json() - logger.info(f" ✅ Status 200 - Found {len(pages)} pages") - if pages: - logger.info(f" Sample page: {pages[0]['numero_page']} - {pages[0]['image_url'][:50]}...") - else: - logger.error(f" ❌ Status {response.status_code}") - return False - except Exception as e: - logger.error(f" ❌ Error: {e}") - return False - - # Test 5: Get stats (requires working DB) - logger.info("\n5. Testing GET /api/catalogues/admin/stats") - try: - response = await client.get(f"{API_BASE_URL}/catalogues/admin/stats") - if response.status_code == 200: - stats = response.json() - logger.info(f" ✅ Status 200 - Total catalogues: {stats['total_catalogues']}") - logger.info(f" Prochaine exécution: {stats['prochaine_execution']}") - else: - logger.error(f" ❌ Status {response.status_code}") - # Stats is optional, don't fail - except Exception as e: - logger.warning(f" ⚠️ Stats endpoint error (may require auth): {e}") - - logger.info("\n✅ All API endpoints are working!\n") - return True - - -async def main(): - """Run all verification tests.""" - logger.info("\n" + "=" * 60) - logger.info("BONIAL CATALOG MODULE - VERIFICATION SCRIPT") - logger.info("=" * 60 + "\n") - - results = [] - - # Phase 1: Database tables - results.append(("Database Tables", check_database_tables())) - - if not results[0][1]: - logger.error("\n❌ Database not ready. Please run: alembic upgrade head") - sys.exit(1) - - # Phase 2: Enseignes seeding - results.append(("Enseignes Seeding", check_enseignes_seeding())) - - # Phase 3: Scraper test (optional, can be slow) - scraper_test = input("\nRun scraper test? (Gifi - takes ~60s) [y/N]: ").lower() == "y" - if scraper_test: - results.append(("Scraper Test", await test_scraper())) - - # Phase 4: API endpoints (requires app to be running) - api_test = input("\nTest API endpoints? (App must be running on :8555) [y/N]: ").lower() == "y" - if api_test: - results.append(("API Endpoints", await test_api_endpoints())) - - # Summary - logger.info("\n" + "=" * 60) - logger.info("VERIFICATION SUMMARY") - logger.info("=" * 60) - - for name, passed in results: - status = "✅ PASSED" if passed else "❌ FAILED" - logger.info(f"{status}: {name}") - - all_passed = all(result[1] for result in results) - - if all_passed: - logger.info("\n✅ ALL TESTS PASSED - Bonial module is ready!") - else: - logger.error("\n❌ SOME TESTS FAILED - Please review errors above") - sys.exit(1) - - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/verify_crawl4ai_tiendeo.py b/verify_crawl4ai_tiendeo.py deleted file mode 100644 index 6af9110..0000000 --- a/verify_crawl4ai_tiendeo.py +++ /dev/null @@ -1,138 +0,0 @@ -""" -Script de vérification pour tester Crawl4AI dans le scraper Tiendeo. - -Usage: - python verify_crawl4ai_scraper.py - -Ce script teste: -1. Import de Crawl4AI -2. Extraction d'une page catalogue Tiendeo -3. Comptage des pages trouvées -""" - -import asyncio -import logging -from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - - -async def test_crawl4ai_import(): - """Test 1: Vérifier que Crawl4AI est bien installé""" - try: - logger.info("✓ Crawl4AI importé avec succès") - logger.info(f" Version: {AsyncWebCrawler.__module__}") - return True - except Exception as e: - logger.error(f"✗ Erreur import Crawl4AI: {e}") - return False - - -async def test_catalog_page_extraction(): - """Test 2: Extraire les pages d'un catalogue Tiendeo""" - # URL de test - catalogue Gifi Nancy (à adapter si nécessaire) - test_url = "https://www.tiendeo.fr/Catalogues/nancy/gifi" - - logger.info(f"\nTest extraction depuis: {test_url}") - - try: - browser_config = BrowserConfig( - headless=True, - verbose=False, - extra_args=["--disable-gpu", "--no-sandbox", "--disable-dev-shm-usage"], - ) - - config = CrawlerRunConfig( - cache_mode=CacheMode.BYPASS, - wait_for_images=True, - process_iframes=True, - remove_overlay_elements=True, - wait_until="networkidle", - delay_before_return_html=3.0, - ) - - async with AsyncWebCrawler(config=browser_config) as crawler: - result = await crawler.arun(url=test_url, config=config) - - if not result.success: - logger.error(f"✗ Échec du crawling: {result.error_message}") - return False - - logger.info(f"✓ Page chargée avec succès") - logger.info(f" HTML length: {len(result.html)} chars") - - # Tester l'extraction JavaScript - pages_data = await crawler.crawler_strategy.execute_js( - """ - () => { - const results = []; - const seenUrls = new Set(); - - const allImages = document.querySelectorAll('img'); - - allImages.forEach((img) => { - let src = img.src || img.getAttribute('data-src'); - - if (!src && img.srcset) { - const srcsetParts = img.srcset.split(',')[0].trim().split(' '); - src = srcsetParts[0]; - } - - if (!src || seenUrls.has(src)) return; - - if (src.includes('logo') || src.includes('icon') || src.includes('avatar')) { - return; - } - - const width = img.naturalWidth || img.width; - const height = img.naturalHeight || img.height; - - if (width < 400 || height < 400) return; - - const aspectRatio = width / height; - - if (aspectRatio > 0.5 && aspectRatio < 0.9) { - seenUrls.add(src); - results.push({ - image_url: src, - width: width, - height: height, - }); - } - }); - - results.sort((a, b) => (b.width * b.height) - (a.width * a.height)); - - return results.map((item, index) => ({ - ...item, - numero_page: index + 1, - })); - } - """ - ) - - logger.info(f"✓ Extraction JavaScript réussie") - logger.info(f" Pages trouvées: {len(pages_data)}") - - if len(pages_data) > 0: - test2 = await test_catalog_page_extraction() - - # Résumé - logger.info("\n" + "=" * 70) - logger.info("RÉSUMÉ") - logger.info("=" * 70) - logger.info(f"Import Crawl4AI: {'✓ OK' if test1 else '✗ ÉCHEC'}") - logger.info(f"Extraction pages: {'✓ OK' if test2 else '✗ ÉCHEC'}") - - if test1 and test2: - logger.info("\n✓ TOUS LES TESTS SONT PASSÉS!") - logger.info("Le scraper Crawl4AI est prêt à être utilisé.") - else: - logger.info("\n✗ CERTAINS TESTS ONT ÉCHOUÉ") - logger.info("Vérifiez les erreurs ci-dessus.") - - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/verify_extraction_logic.py b/verify_extraction_logic.py deleted file mode 100644 index 3e3e8ba..0000000 --- a/verify_extraction_logic.py +++ /dev/null @@ -1,182 +0,0 @@ -import asyncio -import logging -import sys -from unittest.mock import MagicMock, AsyncMock - -# Mock sqlalchemy -sys.modules["sqlalchemy"] = MagicMock() -sys.modules["sqlalchemy.orm"] = MagicMock() - -# Mock pydantic -mock_pydantic = MagicMock() -# Mock BaseModel -class MockBaseModel: - def __init__(self, **kwargs): - for k, v in kwargs.items(): - setattr(self, k, v) -mock_pydantic.BaseModel = MockBaseModel -mock_pydantic.Field = MagicMock(return_value=None) -mock_pydantic.field_validator = MagicMock(return_value=lambda x: x) - -sys.modules["pydantic"] = mock_pydantic - -# Mock app.utils.text which is imported by ai_schema -mock_utils_text = MagicMock() -sys.modules["app.utils.text"] = mock_utils_text -mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length] -mock_utils_text.clean_text = lambda text: text.strip() - -# Mock app.strings (if used) or other utils -sys.modules["app.utils"] = MagicMock() - -# Mock app.utils.image -sys.modules["app.utils.image"] = MagicMock() - -# Mock app.database -sys.modules["app.database"] = MagicMock() - -# Mock playwright -mock_playwright = MagicMock() -sys.modules["playwright"] = mock_playwright -sys.modules["playwright.async_api"] = mock_playwright - -# Mock generic types for type hints if needed -mock_playwright.Browser = MagicMock -mock_playwright.BrowserContext = MagicMock -mock_playwright.Page = MagicMock -mock_playwright.TimeoutError = Exception - -# Now import the schema -from app.ai_schema import get_extraction_prompt, get_repair_prompt - -# We can't import ScraperService easily if it inherits from things or uses decorators -# But for this test we only need get_extraction_prompt which is in ai_schema -# So we can skip importing ScraperService if it causes issues, -# BUT we wanted to verify ScraperService text cleaning logic... -# Let's mock ScraperService dependencies completely. - -try: - from app.services.tracking_scraper_service import ScraperService -except ImportError: - print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.") - ScraperService = None - -# Mock litellm and tenacity -sys.modules["litellm"] = MagicMock() -sys.modules["tenacity"] = MagicMock() -mock_retry = MagicMock() -sys.modules["tenacity.retry"] = mock_retry - -# Make sure imports inside ai_service don't fail -# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity -# We need to mock these specifically if the module imports them directly -mock_tenacity = MagicMock() -mock_tenacity.retry = lambda *args, **kwargs: lambda f: f -mock_tenacity.retry_if_exception_type = MagicMock() -mock_tenacity.stop_after_attempt = MagicMock() -mock_tenacity.wait_exponential = MagicMock() -sys.modules["tenacity"] = mock_tenacity - -# Now import AIService -# We will mock the AI response to verify the parsing logic -from app.services.ai_service import AIService - -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_extraction_logic(): - print("Verifying Extraction Logic...") - - # 1. Test Text Cleaning in ScraperService - # We can't mock Playwright page easily in a simple script without launching a browser. - # But we can test the AI prompt generation which is critical. - - # Simulate B&M text - dirty_text = """ - Menu - Accueil - Panier - - Boisson energisante ice 25cl - Red Bull - - 1.15 € - Prix au litre : 4,60 € / L - - En stock - Ajouter au panier - - Footer - Mentions légales - """ - - print("\n--- Testing Prompt Generation ---") - prompt = get_extraction_prompt(dirty_text) - - # Verify strict instructions are present - checks = [ - "CRITICAL", - "Ignore \"Prix au litre\"", - "B&M STORES Specific", - "Extract as DECIMAL NUMBER", - "ALWAYS select the TTC price", - "Ignore \"HT\"" - ] - - all_passed = True - for check in checks: - if check in prompt: - print(f"[OK] Prompt contains: {check}") - else: - print(f"[FAIL] Prompt missing: {check}") - all_passed = False - - if not all_passed: - print("Prompt verification failed!") - exit(1) - - print("\n--- Testing Response Parsing (Mock AI) ---") - - # Case 1: AI returns Main Price correctly - mock_response_1 = """ - ```json - { - "price": 1.15, - "currency": "EUR", - "in_stock": true, - "price_confidence": 0.95, - "in_stock_confidence": 1.0, - "source_type": "text" - } - ``` - """ - result = AIService.parse_and_validate_response(mock_response_1) - if result.price == 1.15 and result.in_stock is True: - print("[OK] Parsed correct mocked response.") - else: - print(f"[FAIL] Failed to parse correct response: {result}") - exit(1) - - # Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience) - # If AI returns explicit null because it's confused - mock_response_2 = """ - { - "price": null, - "currency": "EUR", - "in_stock": null, - "price_confidence": 0.0, - "in_stock_confidence": 0.0, - "source_type": "image" - } - """ - result = AIService.parse_and_validate_response(mock_response_2) - if result.price is None: - print("[OK] Parsed null response correctly.") - else: - print(f"[FAIL] Failed to parse null response.") - - print("\nVerification of Logic Flow Complete (Simulated).") - print("Real-world verification requires running the full scraper.") - -if __name__ == "__main__": - asyncio.run(verify_extraction_logic()) diff --git a/verify_filename_logic.py b/verify_filename_logic.py deleted file mode 100644 index f36aadf..0000000 --- a/verify_filename_logic.py +++ /dev/null @@ -1,48 +0,0 @@ -import asyncio -import logging -import os -import re -import sys -from unittest.mock import MagicMock, AsyncMock - -# Add current directory to sys.path to allow importing app -sys.path.append(os.getcwd()) - -# Mock playwright before importing app -mock_playwright = MagicMock() -sys.modules["playwright"] = mock_playwright -sys.modules["playwright.async_api"] = mock_playwright - -from app.services.tracking_scraper_service import ScraperService - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_logic(): - logger.info("Starting logic verification...") - - # Mock Page object - mock_page = MagicMock() - mock_page.screenshot = AsyncMock() - - item_id = 123 - url = "http://test.com" - - # Call _take_screenshot directly - logger.info("Calling _take_screenshot...") - filename = await ScraperService._take_screenshot(mock_page, url, item_id) - - logger.info(f"Returned filename: {filename}") - - # Verify format - pattern = r"screenshots/item_123_\d+\.png" - if re.match(pattern, filename): - logger.info("SUCCESS: Filename matches expected timestamp pattern!") - print("VERIFICATION_SUCCESS") - else: - logger.error(f"FAILURE: Filename {filename} does not match pattern {pattern}") - exit(1) - -if __name__ == "__main__": - asyncio.run(verify_logic()) diff --git a/verify_fix_item_service.py b/verify_fix_item_service.py deleted file mode 100644 index 28ce677..0000000 --- a/verify_fix_item_service.py +++ /dev/null @@ -1,164 +0,0 @@ -import os -import glob -import sys -from unittest.mock import MagicMock - -# --- MOCKS SETUP --- -# We need to mock these BEFORE importing app.services.item_service -# to avoid ImportErrors due to missing dependencies in the test env. - -# 1. Mock External Libs -sys.modules["fastapi"] = MagicMock() -sys.modules["sqlalchemy"] = MagicMock() -sys.modules["sqlalchemy.orm"] = MagicMock() - -# 2. Mock Internal App Modules that have heavy dependencies -# Mock app.database -mock_database = MagicMock() -sys.modules["app.database"] = mock_database - -# Mock app.models -# We need models.Item and models.PriceHistory to be accessible attributes -mock_models = MagicMock() -sys.modules["app.models"] = mock_models - -# Mock app.schemas -sys.modules["app.schemas"] = MagicMock() - -# Mock app.services.settings_service -sys.modules["app.services.settings_service"] = MagicMock() - -# Mock app.url_validation -sys.modules["app.url_validation"] = MagicMock() - -# --- IMPORT TARGET --- -from app.services.item_service import ItemService - -def verify_item_service_fix(): - print("Starting verification of ItemService fix...") - - # 1. Setup Mock DB and Item - # We must ensure that when ItemService does `item.id`, it works. - mock_db = MagicMock() - - # Create a simple class to act as the Item model instance - class MockItem: - def __init__(self, id, name): - self.id = id - self.name = name - self.url = "http://test.com" - self.current_price = 10.0 - self.in_stock = True - self.screenshot_url = None # This will be set by the service - - # Attributes accessed by the service - self.notification_channel = None - self.target_price = None - self.current_price_confidence = 1.0 - self.in_stock_confidence = 1.0 - self.is_active = True - self.last_checked = None - self.is_refreshing = False - self.last_error = None - self.category = None - self.tags = None - self.description = None - - # __dict__ is used by the service to create the result - self.dict_storage = {k:v for k,v in self.__dict__.items()} - - @property - def __dict__(self): - # Update dict storage with current attributes - return { - "id": self.id, - "name": self.name, - "url": self.url - } - - item_888 = MockItem(888, "Test Item") - - # ItemService.get_items calls db.query(models.Item).all() - # We need to make sure models.Item is used in the query. - # The service does: items = db.query(models.Item).all() - - mock_db.query.return_value.all.return_value = [item_888] - - # It also queries PriceHistory - # db.query(models.PriceHistory).filter(...).first() - # Let's mock that to return None to force filesytem check (or check logic priority) - mock_db.query.return_value.filter.return_value.filter.return_value.order_by.return_value.first.return_value = None - - # 2. Create Dummy Screenshot Files - os.makedirs("screenshots", exist_ok=True) - - # Clean up - for f in glob.glob("screenshots/item_888_*.png"): - os.remove(f) - if os.path.exists("screenshots/item_888.png"): - os.remove("screenshots/item_888.png") - - # Scenario: - # 1. item_888.png exists (legacy) - # 2. item_888_1000.png exists (old timestamp) - # 3. item_888_2000.png exists (new timestamp) - - # Expected: get_items should pick item_888_2000.png - - file_legacy = "screenshots/item_888.png" - file_old = "screenshots/item_888_1000.png" - file_new = "screenshots/item_888_2000.png" - - with open(file_legacy, "w") as f: f.write(".") - with open(file_old, "w") as f: f.write(".") - with open(file_new, "w") as f: f.write(".") - - print(f"Created files: {file_legacy}, {file_old}, {file_new}") - - try: - # 3. Test get_items - print("Testing get_items()...") - items = ItemService.get_items(mock_db) - - if not items: - print("FAILURE: No items returned") - exit(1) - - result = items[0] - screenshot_url = result.get("screenshot_url") - print(f"Returned screenshot_url: {screenshot_url}") - - expected_url = f"/screenshots/{os.path.basename(file_new)}" - - if screenshot_url == expected_url: - print("SUCCESS: Correctly identified the latest screenshot!") - else: - print(f"FAILURE: Expected {expected_url}, got {screenshot_url}") - # If it failed, maybe it picked legacy? - if screenshot_url == f"/screenshots/{os.path.basename(file_legacy)}": - print("Picked legacy file instead of timestamped one.") - exit(1) - - # 4. Test delete_item - print("Testing delete_item()...") - # Ensure the query returns our item - mock_db.query.return_value.filter.return_value.first.return_value = item_888 - - ItemService.delete_item(mock_db, 888) - - # Check files - remaining = glob.glob("screenshots/item_888*.png") - if not remaining: - print("SUCCESS: All screenshots deleted.") - else: - print(f"FAILURE: Files remaining: {remaining}") - exit(1) - - finally: - # Cleanup - for f in [file_legacy, file_old, file_new]: - if os.path.exists(f): - os.remove(f) - -if __name__ == "__main__": - verify_item_service_fix() diff --git a/verify_gifi_fix.py b/verify_gifi_fix.py deleted file mode 100644 index 9b43ef7..0000000 --- a/verify_gifi_fix.py +++ /dev/null @@ -1,43 +0,0 @@ -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.services.parsers.gifi_parser import GifiParser - -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -def verify_fix(): - print("Verifying Gifi parser fix...") - - # Load the dump file (we know it exists from previous steps) - dump_path = "gifi_full.html" - if not os.path.exists(dump_path): - print(f"Error: {dump_path} not found.") - return - - with open(dump_path, "r", encoding="utf-8") as f: - html = f.read() - - parser = GifiParser() - # Dummy URL - url = "https://www.gifi.fr/test-product.html" - - print("Parsing product details...") - details = parser.parse_product_details(html, url) - - print("\n--- Extraction Results ---") - print(f"Bypass Price: {details.get('price')}") - print(f"Bypass Stock: {details.get('in_stock')}") - print(f"Currency: {details.get('currency')}") - - if details.get('price') is not None: - print("\nSUCCESS: Price extracted successfully!") - else: - print("\nFAILURE: Price not found in dump.") - -if __name__ == "__main__": - verify_fix() diff --git a/verify_leclerc.py b/verify_leclerc.py deleted file mode 100644 index ff787f3..0000000 --- a/verify_leclerc.py +++ /dev/null @@ -1,50 +0,0 @@ -import asyncio -import logging -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_leclerc(): - async with async_playwright() as p: - browser = await p.chromium.launch(headless=True) - page = await browser.new_page() - - # 1. Perform Search - logger.info("--- Step 1: Searching for 'Chaise' on E.Leclerc ---") - search_url = "https://www.e.leclerc/recherche?q=Chaise" - try: - await page.goto(search_url, wait_until="domcontentloaded", timeout=30000) - await page.wait_for_timeout(5000) # Wait for JS - except Exception as e: - logger.error(f"Navigation failed: {e}") - - # 2. Dump HTML - content = await page.content() - logger.info(f"HTML Content Length: {len(content)}") - - # 3. Analyze Classes - classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')") - logger.info(f"Classes found: {classes[:1000]}") - - # 4. Check for Product Selectors - selectors = [ - "div[class*='product']", - "article", - ".product-card", - ".c-product-card", - "a[class*='product']" - ] - - for sel in selectors: - count = await page.locator(sel).count() - if count > 0: - logger.info(f"Selector '{sel}' found {count} elements") - first_html = await page.locator(sel).first.evaluate("el => el.outerHTML") - logger.info(f"First element HTML ({sel}): {first_html[:500]}...") - - await browser.close() - -if __name__ == "__main__": - asyncio.run(verify_leclerc()) diff --git a/verify_new_search.py b/verify_new_search.py deleted file mode 100644 index f969e52..0000000 --- a/verify_new_search.py +++ /dev/null @@ -1,74 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) - -from app.services.search_service import new_search_service -from app.services.browserless_service import browserless_service -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig( - level=logging.DEBUG, # Enable DEBUG logging - format="%(asctime)s - %(name)s - %(levelname)s - %(message)s", - handlers=[logging.StreamHandler()] -) -# Set other loggers to INFO to avoid noise -logging.getLogger("urllib3").setLevel(logging.INFO) -logging.getLogger("asyncio").setLevel(logging.INFO) -logging.getLogger("websockets").setLevel(logging.INFO) - -async def verify_search(): - query = "chaise" # Updated query - print(f"--- Starting Verification Search for '{query}' ---") - - # 1. Test Browserless Connection - print("\n[1] Testing Browserless Connection...") - try: - await browserless_service.initialize() - print("✅ Browserless connected successfully") - except Exception as e: - print(f"❌ Browserless connection failed: {e}") - return - - # 2. Test Specific Sites - # sites_to_test = ["gifi.fr", "lincroyable.fr", "stokomani.fr"] # Excluded amazon.fr - sites_to_test = ["stokomani.fr"] # Focus on Stokomani for now as requested/implied context - - for site in sites_to_test: - print(f"\n[2] Testing Search on {site}...") - try: - count = 0 - async for r in new_search_service.search_site_generator(site, query): - count += 1 - print(f"✅ Found result: {r.title} ({r.url})") - print(f" Price: {r.price} {r.currency}") - print(f" Image: {r.image_url}") - if count >= 1: - break - if count == 0: - print(f"⚠️ No results found for {site}") - # Dump HTML for debugging - try: - content, _ = await browserless_service.get_page_content( - SITE_CONFIGS[site]["search_url"].format(query=query), - use_proxy=SITE_CONFIGS[site].get("requires_proxy", False), - wait_selector=SITE_CONFIGS[site].get("wait_selector") - ) - with open(f"/app/debug_dumps/{site}_failed_verification.html", "w", encoding="utf-8") as f: - f.write(content) - print(f"📄 Saved HTML dump to /app/debug_dumps/{site}_failed_verification.html") - except Exception as dump_e: - print(f"❌ Failed to save HTML dump: {dump_e}") - except Exception as e: - print(f"❌ Error searching {site}: {e}") - - # 3. Cleanup - await browserless_service.shutdown() - print("\n--- Verification Complete ---") - -if __name__ == "__main__": - asyncio.run(verify_search()) diff --git a/verify_popup.py b/verify_popup.py deleted file mode 100644 index 6e9a969..0000000 --- a/verify_popup.py +++ /dev/null @@ -1,33 +0,0 @@ -import asyncio -import logging -import sys -from app.services.browserless_service import browserless_service - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_popups(): - # URL that was reported to have issues - url = "https://www.bmstores.fr/363943-bougie-parfumee-avec-bijou-350g-senteurs-assorties" - # Fallback to search if that product is gone - fallback_url = "https://www.bmstores.fr/module/ambjolisearch/jolisearch?s=calendrier" - - logger.info(f"--- Testing Popup Handling on {url} ---") - - try: - content, screenshot_path = await browserless_service.get_page_content( - url, - extract_text=False - ) - - logger.info(f"Screenshot saved to: {screenshot_path}") - logger.info("Please inspect the screenshot to ensure no 'Stock Inconnu' or 'Calendrier' popups are visible.") - - except Exception as e: - logger.error(f"Verification failed: {e}") - finally: - await browserless_service.shutdown() - -if __name__ == "__main__": - asyncio.run(verify_popups()) diff --git a/verify_scraper_fix.py b/verify_scraper_fix.py deleted file mode 100644 index 3bf265b..0000000 --- a/verify_scraper_fix.py +++ /dev/null @@ -1,48 +0,0 @@ -import asyncio -import logging -import os -from playwright.async_api import async_playwright -from app.services.scraper_service import ScraperService - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def verify_fix(): - logger.info("Starting verification...") - - # Mock browserless URL if not set - if not os.getenv("BROWSERLESS_URL"): - os.environ["BROWSERLESS_URL"] = "ws://browserless:3000" - - playwright = await async_playwright().start() - try: - # Connect to browserless (or launch local if not available, but code expects connect) - # For this test, we might need to mock the browser object if we can't actually connect - # But let's try to just create a local browser for testing purposes if connect fails - # Actually, the code expects a browser object. - - logger.info("Launching local browser for test...") - browser = await playwright.chromium.launch() - - logger.info("Calling scrape_item with browser argument...") - try: - # We pass a dummy URL, we expect it might fail scraping but NOT raise TypeError - await ScraperService.scrape_item( - url="https://example.com", - browser=browser, - timeout=5000 # Short timeout - ) - logger.info("SUCCESS: scrape_item accepted the browser argument!") - except TypeError as e: - logger.error(f"FAILURE: TypeError raised: {e}") - except Exception as e: - logger.info(f"Scraping failed as expected (network/etc), but argument was accepted: {e}") - - await browser.close() - - finally: - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(verify_fix()) diff --git a/verify_scrapers.py b/verify_scrapers.py deleted file mode 100644 index 81ff9cf..0000000 --- a/verify_scrapers.py +++ /dev/null @@ -1,42 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.services.direct_search_service import direct_search_service -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - -async def verify_site(site_key): - logger.info(f"Verifying {site_key}...") - try: - # direct_search_service.search_site might need to be called differently if it's an instance method - # checking previous usage or assuming standard service pattern - results = await direct_search_service.search_site(site_key, "chaise") - if results: - logger.info(f"✅ {site_key}: Found {len(results)} results") - for i, res in enumerate(results[:3]): - title = res.get('title', 'No Title') - price = res.get('price', 'No Price') - url = res.get('url', 'No URL') - logger.info(f" {i+1}. {title[:50]}... - {price} - {url[:50]}...") - else: - logger.error(f"❌ {site_key}: No results found") - except Exception as e: - logger.error(f"❌ {site_key}: Error - {e}") - -async def main(): - sites_to_test = ["gifi.fr", "stokomani.fr", "auchan.fr", "carrefour.fr", "amazon.fr", "action.com"] - - # Run sequentially to avoid overwhelming resources/logs - for site in sites_to_test: - await verify_site(site) - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/verify_screenshot_update.py b/verify_screenshot_update.py deleted file mode 100644 index 0b94b24..0000000 --- a/verify_screenshot_update.py +++ /dev/null @@ -1,74 +0,0 @@ -import asyncio -import logging -import os -import re -from datetime import datetime -from unittest.mock import MagicMock - -from playwright.async_api import async_playwright -from app.services.tracking_scraper_service import ScraperService - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def mock_connect_browser(p): - logger.info("MOCK: Launching local browser instead of connecting to browserless") - return await p.chromium.launch() - -async def verify_fix(): - logger.info("Starting verification...") - - # Monkey-patch _connect_browser to use local browser - ScraperService._connect_browser = mock_connect_browser - - # Ensure screenshots dir exists - os.makedirs("screenshots", exist_ok=True) - - # Test Item ID 999 - item_id = 999 - url = "https://example.com" - - logger.info(f"Scraping item {item_id}...") - - # We expect this to fail scraping real content from example.com with specific selectors, - # but we only care about the screenshot filename generation which happens at the end. - # Actually, if scraping fails, it might return None, "" early. - # checking tracking_scraper_service.py: - # It has a try/except block. - # If _navigate_and_wait works, it proceeds. example.com should load. - # _take_screenshot is called at the end. - - # However, ScraperService.scrape_item returns (None, "") if exception occurs. - # We need to make sure it doesn't crash before screenshot. - # example.com is simple, so it should load. - # It will try to click popups (won't find any), wait for selector (if provided). - # If we don't provide selector, it calls _auto_detect_price. - - full_path, _ = await ScraperService.scrape_item(url=url, item_id=item_id) - - if full_path: - logger.info(f"Screenshot path returned: {full_path}") - - # Verify format: item_{id}_{timestamp}.png - # Check if it matches regex - pattern = r"screenshots/item_999_\d+\.png" - if re.match(pattern, full_path): - logger.info("SUCCESS: Filename contains timestamp!") - else: - logger.error(f"FAILURE: Filename does not match pattern {pattern}") - exit(1) - - # Clean up - if os.path.exists(full_path): - os.remove(full_path) - logger.info("Cleaned up screenshot file") - - else: - logger.error("FAILURE: Scraper returned None for path. Did navigation fail?") - exit(1) - - await ScraperService.shutdown() - -if __name__ == "__main__": - asyncio.run(verify_fix()) diff --git a/verify_targets.py b/verify_targets.py deleted file mode 100644 index ee74aea..0000000 --- a/verify_targets.py +++ /dev/null @@ -1,45 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) - -from app.core.search_config import SITE_CONFIGS -from app.services.search_service import new_search_service -from app.services.browserless_service import browserless_service - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s - %(name)s - %(levelname)s - %(message)s", - handlers=[logging.StreamHandler()] -) - -async def test_specific_sites(): - target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"] - query = "chaise" - - print(f"Testing {target_sites} with query '{query}'...") - - await browserless_service.initialize() - - for site_key in target_sites: - if site_key not in SITE_CONFIGS: - print(f"Skipping {site_key} (not in config)") - continue - - print(f"\n--- Testing {site_key} ---") - try: - results = await new_search_service.search_site(site_key, query) - print(f"Found {len(results)} results") - for r in results[:3]: - print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]") - except Exception as e: - print(f"Error testing {site_key}: {e}") - - await browserless_service.shutdown() - -if __name__ == "__main__": - asyncio.run(test_specific_sites()) diff --git a/verify_vision_priority.py b/verify_vision_priority.py deleted file mode 100644 index b32f3ed..0000000 --- a/verify_vision_priority.py +++ /dev/null @@ -1,50 +0,0 @@ -import sys -import unittest - -# Mock modules to avoid ImportError for app dependencies we don't need for this specific test -from unittest.mock import MagicMock -sys.modules["sqlalchemy"] = MagicMock() -sys.modules["sqlalchemy.orm"] = MagicMock() -sys.modules["app.database"] = MagicMock() -sys.modules["app.utils.image"] = MagicMock() -sys.modules["app.utils.text"] = MagicMock() -sys.modules["app.utils.text"].filter_relevant_text = lambda text, max_length: text - -# Mock pydantic -mock_pydantic = MagicMock() -class MockBaseModel: - pass -mock_pydantic.BaseModel = MockBaseModel -mock_pydantic.Field = MagicMock(return_value=None) -mock_pydantic.field_validator = MagicMock(return_value=lambda x: x) -sys.modules["pydantic"] = mock_pydantic - -# Import the schema module -from app.ai_schema import get_extraction_prompt - -class TestVisionPriorityPrompt(unittest.TestCase): - def test_vision_first_directives(self): - """Verify that the prompt contains the Vision-First directives.""" - - # Scenario: Some random text context - page_text = "Some random text content from the page." - prompt = get_extraction_prompt(page_text) - - print("\nGenerated Prompt Snippet:\n", prompt[:500], "...\n") - - # Check for Critical Directives - self.assertIn("Vision-First Price Extraction Agent", prompt) - self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt) - self.assertIn("IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE", prompt) - - # Check for stock rules - self.assertIn("STOCK STATUS RULES", prompt) - - def test_prompt_without_text(self): - """Verify prompt structure when no text is provided.""" - prompt = get_extraction_prompt(None) - self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt) - self.assertNotIn("**Relevant text from page:**", prompt) - -if __name__ == "__main__": - unittest.main() diff --git a/walkthrough.md b/walkthrough.md index 1d6a30c..c9218f9 100644 --- a/walkthrough.md +++ b/walkthrough.md @@ -1,38 +1,24 @@ -# Walkthrough: Vision-First Price Extraction +# Walkthrough: Price Extraction & Catalog Fixes -In response to issues where the AI was being misled by hidden text (like unit prices or old prices in HTML), we have implemented a **Vision-Priority Strategy**. +## 1. Vision Priority Strategy -## Changes Implemented +**Problem**: The AI was prioritizing text data (often hidden/outdated) over the visual price on the screenshot, extracting incorrect prices (e.g. 0.99€ instead of 1.27€). -### 1. Updated AI System Prompt (`app/ai_schema.py`) +**Solution**: -We completely rewrote the `EXTRACTION_PROMPT_TEMPLATE` to enforce the following rules: +- **Prompt Engineering**: Rewrote the system prompt in `ai_schema.py` to explicitly declare the **IMAGE AS THE SOURCE OF TRUTH**. +- **Logic Fix**: Disabled the `AIPriceExtractor` (Text-only AI) in `scheduler_service.py` which was short-circuiting the logic before the Vision AI could run. -- **Source of Truth = Image**: explicit instruction that the screenshot takes precedence over any text. -- **Conflict Resolution**: "IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE." -- **Visual Focus Rules**: - - Look for the largest/boldest price. - - Ignore small, styling-less text (often unit prices). - - Ignore crossed-out text. +## 2. Catalog Scraper Fix -### Verification +**Problem**: Catalogs were not updating because the `browserless` service was failing (likely blocked or network issues), preventing the scraper from loading `cataloguemate.fr`. -We verified the new prompt generation using `verify_vision_priority.py`. +**Solution**: -**Generated Prompt Preview:** +- **HTTP Fallback**: Modified `cataloguemate_scraper.py` to use a robust fallback mechanism. + - First attempts to use the secure Browserless browser. + - If that fails, it instantly falls back to a standard `httpx` HTTP request, which is often sufficient for static catalog sites. -```text -You are a Vision-First Price Extraction Agent. -Your Goal: Extract the main product price exactly as a human sees it on the screen. +## 3. Cleanup -**SOURCE OF TRUTH = IMAGE** -- The image provided is the **Absolute Truth**. -- The text provided below is scraped HTML content which may contain hidden/old prices. -- **IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE.** -``` - -## How to Test - -1. Go to "Suivis Prix". -2. Force refresh an item that was previously incorrect (e.g., B&M item showing unit price). -3. The AI should now ignore the "hidden" unit price text and read the main price tag from the image. +- Removed 20+ temporary debug/verification scripts (`debug_*.py`, `verify_*.py`) from the root directory to keep the production environment clean.