Merge pull request #240 from R0m1k3/antigravity

chore: remove obsolete debug and verification scripts and streamline …
This commit is contained in:
LogiFlow authored and GitHub committed 2025-12-22 12:22:34 +01:00
commit f1fe4ef09c
30 files changed
+17 -2322

No files matched your search

-60
View File
@@ -1,60 +0,0 @@
import asyncio
import logging
import os
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000")
async def debug_amazon():
logger.info("Starting Amazon Debug Script")
async with async_playwright() as p:
try:
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
browser = await p.chromium.connect_over_cdp(BROWSERLESS_URL)
# Use a very standard, recent User-Agent
user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
context = await browser.new_context(
user_agent=user_agent,
viewport={"width": 1920, "height": 1080},
locale="fr-FR",
timezone_id="Europe/Paris"
)
page = await context.new_page()
url = "https://www.amazon.fr/s?k=iphone"
logger.info(f"Navigating to {url}")
response = await page.goto(url, wait_until="domcontentloaded", timeout=30000)
if response:
status = response.status
logger.info(f"Response Status: {status}")
content = await page.content()
if "api-services-support@amazon.com" in content or "Toutes nos excuses" in content:
logger.error("BLOCK DETECTED: Found blocking message in content")
else:
logger.info("No obvious blocking message found")
# Save screenshot
await page.screenshot(path="debug_amazon_screenshot.png")
logger.info("Screenshot saved to debug_amazon_screenshot.png")
else:
logger.error("No response received")
await browser.close()
except Exception as e:
logger.error(f"An error occurred: {e}")
if __name__ == "__main__":
asyncio.run(debug_amazon())
-121
View File
@@ -1,121 +0,0 @@
import asyncio
import logging
import sys
from bs4 import BeautifulSoup
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler(sys.stdout)]
)
logger = logging.getLogger(__name__)
BASE_URL = "https://www.cataloguemate.fr"
async def debug_catalog_list():
"""Debug the catalog list extraction"""
# Test with Gifi
slug = "gifi"
url = f"{BASE_URL}/{slug}/"
logger.info(f"--- DEBUGGING LIST: {url} ---")
browser_config = BrowserConfig(headless=True)
run_config = CrawlerRunConfig(
cache_mode=CacheMode.BYPASS,
wait_for_images=True,
)
async with AsyncWebCrawler(config=browser_config) as crawler:
result = await crawler.arun(url=url, config=run_config)
if not result.success:
logger.error(f"Failed to fetch {url}: {result.error_message}")
return None
logger.info(f"Successfully fetched {url} ({len(result.html)} chars)")
soup = BeautifulSoup(result.html, 'html.parser')
# 1. Dump all links to see what we have
links = soup.find_all('a', href=True)
logger.info(f"Found {len(links)} links total")
potential_catalogs = []
for i, link in enumerate(links):
href = link['href']
text = link.get_text(strip=True)
# Normalize
if href.startswith(BASE_URL):
href = href.replace(BASE_URL, "")
# Log interesting links
if slug in href or "catalogue" in href.lower():
logger.info(f"Link {i}: {href} | Text: '{text}'")
# Apply our filter logic to see if it passes
if href.startswith(f"/{slug}/") and href != f"/{slug}/":
if not any(x in href for x in ["offres", "magasins", "rechercher"]):
potential_catalogs.append(href)
logger.info(f" -> MATCHES FILTER!")
logger.info(f"Total matching catalogs: {len(potential_catalogs)}")
return potential_catalogs[0] if potential_catalogs else None
async def debug_catalog_page(catalog_rel_url):
"""Debug the catalog page extraction"""
if not catalog_rel_url:
logger.error("No catalog URL to debug")
return
full_url = f"{BASE_URL}{catalog_rel_url}"
logger.info(f"\n--- DEBUGGING PAGE: {full_url} ---")
browser_config = BrowserConfig(headless=True)
run_config = CrawlerRunConfig(
cache_mode=CacheMode.BYPASS,
wait_for_images=True,
delay_before_return_html=2.0 # Wait a bit more
)
async with AsyncWebCrawler(config=browser_config) as crawler:
result = await crawler.arun(url=full_url, config=run_config)
if not result.success:
logger.error(f"Failed to fetch {full_url}")
return
soup = BeautifulSoup(result.html, 'html.parser')
# 1. Dump all images
images = soup.find_all('img')
logger.info(f"Found {len(images)} images")
for i, img in enumerate(images):
src = img.get('src', '')
width = img.get('width', '?')
height = img.get('height', '?')
alt = img.get('alt', '')
# Filter noise
if "logo" in src or "icon" in src:
continue
logger.info(f"Img {i}: {src} | {width}x{height} | Alt: {alt}")
# Check our heuristic
is_likely = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
if is_likely:
logger.info(" -> LIKELY CATALOG IMAGE")
if __name__ == "__main__":
async def main():
cat_url = await debug_catalog_list()
if cat_url:
await debug_catalog_page(cat_url)
asyncio.run(main())
-96
View File
@@ -1,96 +0,0 @@
import asyncio
import logging
import sys
# from bs4 import BeautifulSoup # Removed
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler(sys.stdout)]
)
logger = logging.getLogger(__name__)
BASE_URL = "https://www.cataloguemate.fr"
async def debug_catalog_list():
"""Debug the catalog list extraction using Playwright directly"""
slug = "gifi"
url = f"{BASE_URL}/offres/paris/{slug}/"
logger.info(f"--- DEBUGGING LIST: {url} ---")
async with async_playwright() as p:
# Use browserless or local depending on connection
# For this script we use local headless for simplicity if browserless is not available,
# but since we are in the container we might need browserless.
# Let's try to simulate what browserless_service does but simplified.
try:
browser = await p.chromium.launch(headless=True) # Try local first
except:
logger.info("Local browser failed, trying browserless...")
browser = await p.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
)
page = await context.new_page()
logger.info(f"Navigating to {url}")
await page.goto(url, wait_until="domcontentloaded")
# Wait a bit
await page.wait_for_timeout(2000)
content = await page.content()
# soup = BeautifulSoup(content, 'html.parser')
# 1. Dump all links to see what we have
# Use JS to extract links
links_data = await page.evaluate("""
() => {
return Array.from(document.querySelectorAll('a[href]')).map(a => ({
href: a.getAttribute('href'),
text: a.innerText.trim()
}));
}
""")
logger.info(f"Found {len(links_data)} links total")
potential_catalogs = []
for i, link in enumerate(links_data):
href = link['href']
text = link['text']
# Normalize
if href.startswith(BASE_URL):
href = href.replace(BASE_URL, "")
# Log interesting links
if slug in href or "catalogue" in href.lower():
logger.info(f"Link {i}: {href} | Text: '{text}'")
# Apply our filter logic to see if it passes
if f"/{slug}/" in href:
if not any(x in href for x in ["/offres/", "/magasins/", "/rechercher/", "page="]):
# Check ID pattern
import re
if re.search(r'-\d+/?$', href) or "catalogue" in href.lower():
potential_catalogs.append(href)
logger.info(f" -> MATCHES FILTER! (Found catalog)")
else:
logger.info(f" -> Rejected (no ID/keyword)")
else:
logger.info(f" -> Rejected (invalid pattern)")
logger.info(f"Total matching catalogs: {len(potential_catalogs)}")
await browser.close()
return potential_catalogs[0] if potential_catalogs else None
if __name__ == "__main__":
asyncio.run(debug_catalog_list())
-50
View File
@@ -1,50 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.improved_search_service import ImprovedSearchService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def debug_gifi():
print("Initializing search service...")
await ImprovedSearchService.initialize()
try:
print("Searching Gifi for 'chaise'...")
results = await ImprovedSearchService.search_site("gifi.fr", "chaise")
print(f"\nFound {len(results)} results.")
if results:
print("\n--- First 5 Results ---")
for i, res in enumerate(results[:5]):
print(f"\nItem {i+1}:")
print(f" Title: {res.title}")
print(f" Price: {res.price} {res.currency}")
print(f" Image: {res.image_url}")
print(f" URL: {res.url}")
print(f" In Stock: {res.in_stock}")
# Check for missing critical data
missing_price = sum(1 for r in results if r.price is None)
missing_image = sum(1 for r in results if not r.image_url)
print(f"\nStats:")
print(f" Total: {len(results)}")
print(f" Missing Price: {missing_price}")
print(f" Missing Image: {missing_image}")
except Exception as e:
logger.error(f"Error: {e}", exc_info=True)
finally:
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(debug_gifi())
-120
View File
@@ -1,120 +0,0 @@
"""
Debug script to identify correct selectors for missing sites
"""
import asyncio
import sys
import os
from pathlib import Path
# Add app directory to path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.browserless_service import browserless_service
from bs4 import BeautifulSoup
async def analyze_site(name: str, url: str, wait_selector: str = None):
"""Analyze a search results page to identify selectors"""
print(f"\n{'='*80}")
print(f"Analyzing: {name}")
print(f"URL: {url}")
print(f"{'='*80}\n")
html, screenshot = await browserless_service.get_page_content(
url,
use_proxy=False,
wait_selector=wait_selector,
wait_timeout=10000
)
if not html:
print(f"❌ Failed to get content for {name}")
return
soup = BeautifulSoup(html, "html.parser")
# Save HTML for manual inspection
output_file = f"debug_{name.lower().replace(' ', '_')}.html"
with open(output_file, "w", encoding="utf-8") as f:
f.write(html)
print(f"💾 HTML saved to: {output_file}")
# Common product link patterns
product_patterns = [
"a[href*='/product']",
"a[href*='/p/']",
"a[href*='/produit']",
"a.product-card",
"a.product-link",
"[data-product-id]",
"article a",
"div[data-testid*='product'] a",
]
print("\n🔍 Searching for product links...")
for pattern in product_patterns:
links = soup.select(pattern)
if links and len(links) >= 3:
print(f"✅ Found {len(links)} matches for: {pattern}")
# Show first 3 examples
for i, link in enumerate(links[:3], 1):
href = link.get('href', 'NO_HREF')
text = link.get_text(strip=True)[:50]
print(f" {i}. {href[:60]} | {text}")
elif links:
print(f"⚠️ Found {len(links)} matches for: {pattern} (too few)")
# Common image patterns
image_patterns = [
"img[src*='product']",
"img.product-image",
"img.product-img",
"img[loading='lazy']",
"picture img",
"img[data-src]",
]
print("\n🖼️ Searching for product images...")
for pattern in image_patterns:
images = soup.select(pattern)
if images and len(images) >= 3:
print(f"✅ Found {len(images)} matches for: {pattern}")
for i, img in enumerate(images[:3], 1):
src = img.get('src') or img.get('data-src', 'NO_SRC')
alt = img.get('alt', 'NO_ALT')[:50]
print(f" {i}. {src[:60]} | {alt}")
elif images:
print(f"⚠️ Found {len(images)} matches for: {pattern} (too few)")
print(f"\n✅ Analysis complete for {name}\n")
async def main():
"""Test all missing sites"""
sites = [
{
"name": "E.Leclerc",
"url": "https://www.e-leclerc.com/recherche?text=chaise",
"wait": ".product-card, .search-results"
},
{
"name": "Auchan",
"url": "https://www.auchan.fr/search?text=chaise",
"wait": ".product-card, .product-item"
},
{
"name": "Carrefour",
"url": "https://www.carrefour.fr/s?q=chaise",
"wait": "[data-testid*='product'], .product"
},
]
for site in sites:
try:
await analyze_site(site["name"], site["url"], site.get("wait"))
except Exception as e:
print(f"❌ Error analyzing {site['name']}: {e}")
# Small delay between sites
await asyncio.sleep(2)
if __name__ == "__main__":
asyncio.run(main())
-67
View File
@@ -1,67 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def debug_site(site_key: str, query: str = "chaise"):
print(f"--- Debugging {site_key} ---")
config = SITE_CONFIGS.get(site_key)
if not config:
print(f"Site {site_key} not found in config")
return
search_url = config["search_url"].format(query=query)
print(f"URL: {search_url}")
print("Fetching content...")
try:
html, screenshot_path = await browserless_service.get_page_content(
search_url,
use_proxy=config.get("requires_proxy", False),
wait_selector=config.get("wait_selector")
)
print(f"Screenshot saved to: {screenshot_path}")
print(f"HTML length: {len(html)}")
# Save HTML for inspection
with open(f"debug_{site_key}.html", "w", encoding="utf-8") as f:
f.write(html)
print(f"HTML saved to debug_{site_key}.html")
# Check if wait selector is present in HTML
if config.get("wait_selector"):
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
found = soup.select(config["wait_selector"])
print(f"Wait selector '{config['wait_selector']}' found: {len(found)} elements")
except Exception as e:
print(f"Error: {e}")
async def main():
if len(sys.argv) < 2:
print("Usage: python debug_scraper.py <site_key> [query]")
return
site_key = sys.argv[1]
query = sys.argv[2] if len(sys.argv) > 2 else "chaise"
await browserless_service.start()
try:
await debug_site(site_key, query)
finally:
await browserless_service.stop()
if __name__ == "__main__":
asyncio.run(main())
-54
View File
@@ -1,54 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
async def debug_search():
print("--- Debugging Search ---")
# Check Config
print(f"Available Config Keys: {list(SITE_CONFIGS.keys())}")
sites = ["amazon.fr", "stokomani.fr", "lincroyable.fr"]
query = "iphone"
try:
await browserless_service.initialize()
for site in sites:
print(f"\nTesting {site}...")
if site not in SITE_CONFIGS:
print(f"❌ Site {site} NOT found in SITE_CONFIGS")
continue
try:
count = 0
async for r in new_search_service.search_site_generator(site, query):
count += 1
print(f"✅ Found: {r.title} - {r.price} {r.currency}")
if count >= 1:
break
if count == 0:
print(f"⚠️ No results for {site}")
except Exception as e:
print(f"❌ Error searching {site}: {e}")
finally:
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(debug_search())
-30
View File
@@ -1,30 +0,0 @@
import sys
import os
# Add app to path
sys.path.append(os.getcwd())
from app.core.database import SessionLocal
from app.models import SearchSite
from app.core.search_config import SITE_CONFIGS
def debug_sites():
db = SessionLocal()
try:
print("=== SITE_CONFIGS Keys ===")
for key in SITE_CONFIGS.keys():
print(f"- {key}")
print("\n=== DB SearchSites ===")
sites = db.query(SearchSite).all()
for site in sites:
print(f"ID: {site.id} | Name: {site.name} | Domain: {site.domain} | Active: {site.is_active}")
print(f" URL: {site.search_url}")
print(f" Selector: {site.product_link_selector}")
print("-" * 20)
finally:
db.close()
if __name__ == "__main__":
debug_sites()
+3
View File
@@ -16,6 +16,9 @@
- [x] Reproduce failure (Confirmed Browserless issue via analysis).
- [x] Fix `cataloguemate_scraper.py` with HTTP fallback.
- [x] **Cleanup**:
- [x] Deleted `debug_*.py` and `verify_*.py` files.
## 📝 Progress Log
- **2025-12-22**: Vision Priority implemented and verified.
-79
View File
@@ -1,79 +0,0 @@
import asyncio
import logging
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_action():
async with async_playwright() as p:
# Use a standard User Agent
ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
browser = await p.chromium.launch(headless=True)
context = await browser.new_context(user_agent=ua)
page = await context.new_page()
# 1. Perform Search
logger.info("--- Step 1: Searching for 'Chaise' on Action ---")
search_url = "https://www.action.com/fr-fr/search/?q=Chaise"
try:
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
# Wait a bit for any JS redirects or challenges
await page.wait_for_timeout(5000)
except Exception as e:
logger.error(f"Navigation error: {e}")
# 2. Take Screenshot
await page.screenshot(path="action_debug.png")
logger.info("Screenshot saved: action_debug.png")
# 3. Dump Content
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# 4. Check for specific text
text = await page.inner_text("body")
logger.info(f"Page Text (first 500 chars): {text[:500]}")
if "Challenge" in text or "human" in text or "Cloudflare" in text:
logger.warning("⚠️ Cloudflare Challenge detected in text!")
# 5. Analyze Content for Product Links
# Action product URLs contain "/p/"
links = await page.evaluate("""
Array.from(document.querySelectorAll('a'))
.map(a => a.href)
.filter(href => href.includes('/p/'))
""")
logger.info(f"Found {len(links)} product links")
if links:
logger.info(f"First 5 links: {links[:5]}")
# Find the parent container of the first link
parent_html = await page.evaluate("""
(() => {
const link = document.querySelector("a[href*='/p/']");
return link ? link.parentElement.outerHTML : "Not found";
})()
""")
logger.info(f"Parent HTML of first link: {parent_html[:500]}")
# Find the class of the link itself
link_class = await page.evaluate("""
(() => {
const link = document.querySelector("a[href*='/p/']");
return link ? link.className : "Not found";
})()
""")
logger.info(f"Class of first link: {link_class}")
await browser.close()
if __name__ == "__main__":
asyncio.run(verify_action())
-40
View File
@@ -1,40 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.ai_price_extractor import AIPriceExtractor
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_ai_extractor():
print("Verifying AIPriceExtractor on Gifi dump...")
# Load the dump file
dump_path = "gifi_full.html"
if not os.path.exists(dump_path):
print(f"Error: {dump_path} not found.")
return
with open(dump_path, "r", encoding="utf-8") as f:
html = f.read()
title = "Gifi Product Test"
print("Calling AIPriceExtractor...")
price = await AIPriceExtractor.extract_price(html, title)
print("\n--- AI Extraction Results ---")
print(f"Price: {price}€")
if price is not None:
print("\nSUCCESS: AIPriceExtractor worked!")
else:
print("\nFAILURE: AIPriceExtractor returned None.")
if __name__ == "__main__":
asyncio.run(verify_ai_extractor())
-70
View File
@@ -1,70 +0,0 @@
import asyncio
import logging
from unittest.mock import MagicMock, patch
from app.services.ai_service import AIService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# Mock BadRequestError since we might not have litellm installed in the environment running this script
class MockBadRequestError(Exception):
pass
async def verify_fallback():
logger.info("Starting AI fallback verification...")
# Mock config
config = {
"provider": "openrouter",
"model": "google/gemini-2.5-flash-image-preview",
"api_key": "fake-key",
"api_base": "https://openrouter.ai/api/v1",
"temperature": 0.1,
"max_tokens": 100,
"timeout": 30,
}
# Mock success response
mock_response = MagicMock()
mock_response.choices = [MagicMock(message=MagicMock(content='{"price": 10.0}'))]
# Patch acompletion
with patch("app.services.ai_service.acompletion") as mock_acompletion:
# Setup side effect: First call raises BadRequestError, second call succeeds
mock_acompletion.side_effect = [
MockBadRequestError("400 Bad Request: The request is not supported by this model."),
mock_response
]
try:
# We need to patch the exception check in the code if we can't import the real exception
# But the code uses string check "BadRequestError" in str(type(e).__name__)
# So MockBadRequestError should work if we name it right or if the code checks "400"
# Actually, let's just run it and see if our logic catches it.
# The code checks: is_bad_request = "BadRequestError" in str(type(e).__name__) or "400" in str(e)
# Our MockBadRequestError has "400" in the message, so it should be caught.
logger.info("Calling call_llm...")
result = await AIService.call_llm("test prompt", "data:image/...", config)
logger.info(f"Result: {result}")
# Verify calls
assert mock_acompletion.call_count == 2
logger.info("SUCCESS: acompletion was called twice (retry worked)")
# Verify second call didn't have response_format
call_args = mock_acompletion.call_args_list[1]
kwargs = call_args.kwargs
if "response_format" not in kwargs:
logger.info("SUCCESS: Second call did not have response_format")
else:
logger.error("FAILURE: Second call still had response_format")
except Exception as e:
logger.error(f"FAILURE: Exception raised: {e}")
if __name__ == "__main__":
asyncio.run(verify_fallback())
-69
View File
@@ -1,69 +0,0 @@
import asyncio
import logging
import time
from app.services.improved_search_service import ImprovedSearchService
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
async def verify_site(site_key: str):
print(f"\n{'='*50}")
print(f"🔍 Verifying {site_key}...")
print(f"{'='*50}")
start_time = time.time()
first_result_time = None
count = 0
try:
async for result in ImprovedSearchService.search_site_generator(site_key, "chaise"):
count += 1
if count == 1:
first_result_time = time.time()
elapsed = first_result_time - start_time
print(f"🚀 First result in {elapsed:.2f}s")
print(f" Title: {result.title}")
print(f" Price: {result.price} {result.currency}")
print(f" Image: {result.image_url}")
# Print a dot for each result to show progress
print(".", end="", flush=True)
total_time = time.time() - start_time
print(f"\n✅ Finished {site_key}: {count} results in {total_time:.2f}s")
if count == 0:
print(f"❌ WARNING: 0 results found for {site_key}")
return False
return True
except Exception as e:
print(f"\n❌ ERROR verifying {site_key}: {e}")
return False
async def main():
await ImprovedSearchService.initialize()
sites = list(SITE_CONFIGS.keys())
results = {}
# Test all sites sequentially to avoid overwhelming the browser/network
for site in sites:
success = await verify_site(site)
results[site] = success
# Small pause between sites
await asyncio.sleep(2)
await ImprovedSearchService.shutdown()
print("\n" + "="*50)
print("SUMMARY")
print("="*50)
for site, success in results.items():
status = "✅ PASS" if success else "❌ FAIL"
print(f"{status} - {site}")
if __name__ == "__main__":
asyncio.run(main())
-62
View File
@@ -1,62 +0,0 @@
import asyncio
from playwright.async_api import async_playwright
async def main():
async with async_playwright() as p:
# Launch with headless=True to mimic server environment
browser = await p.chromium.launch(headless=True)
context = await browser.new_context(
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
)
page = await context.new_page()
# Try a product URL that might trigger the check
url = "https://www.amazon.fr/dp/B07S58MPKW"
print(f"Navigating to {url}")
try:
await page.goto(url)
await page.wait_for_timeout(5000)
content = await page.content()
if "Continuer les achats" in content:
print("🚨 Popup detected!")
with open("amazon_popup.html", "w", encoding="utf-8") as f:
f.write(content)
await page.screenshot(path="amazon_popup.png")
# Analyze the button
print("Searching for button...")
# Try various locators
locators = [
"button",
"input[type='submit']",
"a.a-button-text",
"span.a-button-inner"
]
for sel in locators:
elements = page.locator(sel)
count = await elements.count()
for i in range(count):
el = elements.nth(i)
if await el.is_visible():
txt = await el.inner_text()
val = await el.get_attribute("value") or ""
if "Continuer" in txt or "Continuer" in val:
print(f"✅ Found candidate: {sel}")
print(f" Text: {txt}")
print(f" Value: {val}")
print(f" OuterHTML: {await el.evaluate('el => el.outerHTML')}")
else:
print("No popup detected. Page title:", await page.title())
except Exception as e:
print(f"Error: {e}")
await browser.close()
if __name__ == "__main__":
asyncio.run(main())
-91
View File
@@ -1,91 +0,0 @@
import asyncio
import logging
import sys
from playwright.async_api import async_playwright
from app.services.ai_price_extractor import AIPriceExtractor
from app.services.improved_search_service import ImprovedSearchService
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_bm():
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
page = await browser.new_page()
# 1. Perform Search to get a product URL
logger.info("--- Step 1: Searching for 'Chaise pliante pu creme' on B&M ---")
# Use specific search to find the problematic product
search_url = "https://bmstores.fr/module/ambjolisearch/jolisearch?s=Chaise+pliante+pu+creme"
await page.goto(search_url)
await page.wait_for_load_state("networkidle")
# Take screenshot of search results
await page.screenshot(path="bm_search_results.png")
logger.info("Screenshot saved: bm_search_results.png")
# Get first product link
product_link = await page.get_attribute("a.thumbnail.product-thumbnail", "href")
if not product_link:
logger.error("No product found in search")
return
if not product_link.startswith("http"):
product_link = "https://www.bmstores.fr" + product_link
logger.info(f"Testing Product URL: {product_link}")
# 2. Go to Product Page
await page.goto(product_link)
await page.wait_for_load_state("networkidle")
await page.screenshot(path="bm_product_page.png")
logger.info("Screenshot saved: bm_product_page.png")
# 3. Dump HTML snippet (price area)
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# Check for 12.95
if "12,95" in content or "12.95" in content:
logger.info("✅ Price 12.95 found in raw HTML")
else:
logger.warning("❌ Price 12.95 NOT found in raw HTML")
# 4. Test JSON-LD
logger.info("\n--- Step 2: Testing JSON-LD ---")
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
for i, script in enumerate(json_ld_scripts):
text = await script.inner_text()
logger.info(f"JSON-LD #{i}: {text[:500]}...")
# 5. Test CSS Selectors
logger.info("\n--- Step 3: Testing CSS Selectors ---")
selectors = [
'.price-current', '.prix-actuel', '.sale-price', '.promo-price',
'.price', '[data-testid="price"]', '[itemprop="price"]', '.product-price',
'.current-price-value'
]
for sel in selectors:
elements = await page.query_selector_all(sel)
for el in elements:
text = await el.inner_text()
logger.info(f"Selector '{sel}': {text.strip()}")
# 6. Test AI Extraction
logger.info("\n--- Step 4: Testing AI Extraction (Gemma 3) ---")
title = await page.title()
# Ensure API key is available
import os
if not os.getenv("OPENROUTER_API_KEY"):
logger.warning("OPENROUTER_API_KEY not set in env, AI might fail")
ai_price = await AIPriceExtractor.extract_price(content, title)
logger.info(f"AI Extracted Price: {ai_price}")
await browser.close()
if __name__ == "__main__":
asyncio.run(verify_bm())
-294
View File
@@ -1,294 +0,0 @@
"""
Verification Script for Bonial Catalog Module
Tests database setup, scraper functionality, and API endpoints.
"""
import asyncio
import logging
import sys
from datetime import datetime
import httpx
from sqlalchemy import inspect, text
from app.database import SessionLocal, engine
from app.models import Catalogue, CataloguePage, Enseigne, ScrapingLog
from app.services.bonial_scraper import scrape_enseigne
from app.services.seed_enseignes import seed_enseignes
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
API_BASE_URL = "http://localhost:8555/api"
def check_database_tables():
"""Verify that catalog module tables exist."""
logger.info("=" * 60)
logger.info("Phase 1: Checking Database Tables")
logger.info("=" * 60)
inspector = inspect(engine)
required_tables = [
"enseignes",
"catalogues",
"catalogue_pages",
"scraping_logs",
]
existing_tables = inspector.get_table_names()
all_exist = True
for table in required_tables:
exists = table in existing_tables
status = "✅" if exists else "❌"
logger.info(f"{status} Table '{table}': {'EXISTS' if exists else 'MISSING'}")
if not exists:
all_exist = False
if all_exist:
logger.info("\n✅ All required tables exist!\n")
return True
else:
logger.error("\n❌ Some tables are missing. Run migration: alembic upgrade head\n")
return False
def check_enseignes_seeding():
"""Verify that enseignes are seeded."""
logger.info("=" * 60)
logger.info("Phase 2: Checking Enseignes Seeding")
logger.info("=" * 60)
db = SessionLocal()
try:
count = db.query(Enseigne).count()
logger.info(f"Found {count} enseignes in database")
if count == 0:
logger.info("Seeding enseignes...")
created = seed_enseignes(db)
logger.info(f"✅ Created {created} enseignes")
count = created
if count >= 9:
logger.info("\n✅ All 9 enseignes are seeded!\n")
# Display enseignes
enseignes = db.query(Enseigne).order_by(Enseigne.ordre_affichage).all()
for ens in enseignes:
active = "✅" if ens.is_active else "⚠️"
logger.info(f" {active} {ens.ordre_affichage}. {ens.nom} (slug: {ens.slug_bonial})")
return True
else:
logger.warning(f"\n⚠️ Expected 9 enseignes, found {count}\n")
return False
finally:
db.close()
async def test_scraper():
"""Test the Bonial scraper on one enseigne."""
logger.info("=" * 60)
logger.info("Phase 3: Testing Bonial Scraper (Gifi)")
logger.info("=" * 60)
db = SessionLocal()
try:
# Get Gifi enseigne
gifi = db.query(Enseigne).filter_by(slug_bonial="Gifi").first()
if not gifi:
logger.error("❌ Gifi enseigne not found")
return False
logger.info(f"Testing scraper for: {gifi.nom}")
logger.info("This may take 30-60 seconds...")
# Run scraper
log = await scrape_enseigne(gifi, db)
# Display results
logger.info(f"\nScraping completed:")
logger.info(f" Status: {log.statut}")
logger.info(f" Catalogues found: {log.catalogues_trouves}")
logger.info(f" New catalogues: {log.catalogues_nouveaux}")
logger.info(f" Duration: {log.duree_secondes:.2f}s")
if log.message_erreur:
logger.warning(f" Error: {log.message_erreur}")
if log.statut in ["success", "partial"] and log.catalogues_trouves > 0:
logger.info("\n✅ Scraper is working!\n")
# Display sample catalog
cat = db.query(Catalogue).filter_by(enseigne_id=gifi.id).first()
if cat:
logger.info(f"Sample catalog:")
logger.info(f" Titre: {cat.titre}")
logger.info(f" Dates: {cat.date_debut.date()} → {cat.date_fin.date()}")
logger.info(f" Pages: {cat.nombre_pages}")
return True
else:
logger.error("\n❌ Scraper failed or found no catalogs\n")
return False
except Exception as e:
logger.error(f"❌ Error testing scraper: {e}")
return False
finally:
db.close()
async def test_api_endpoints():
"""Test API endpoints."""
logger.info("=" * 60)
logger.info("Phase 4: Testing API Endpoints")
logger.info("=" * 60)
async with httpx.AsyncClient(timeout=30.0) as client:
# Test 1: Get enseignes
logger.info("\n1. Testing GET /api/catalogues/enseignes")
try:
response = await client.get(f"{API_BASE_URL}/catalogues/enseignes")
if response.status_code == 200:
enseignes = response.json()
logger.info(f" ✅ Status 200 - Found {len(enseignes)} enseignes")
if enseignes:
logger.info(f" Sample: {enseignes[0]['nom']} ({enseignes[0]['catalogues_actifs_count']} catalogues)")
else:
logger.error(f" ❌ Status {response.status_code}")
return False
except Exception as e:
logger.error(f" ❌ Error: {e}")
return False
# Test 2: Get catalogues
logger.info("\n2. Testing GET /api/catalogues")
try:
response = await client.get(f"{API_BASE_URL}/catalogues?page=1&limit=5")
if response.status_code == 200:
data = response.json()
catalogues = data.get("data", [])
pagination = data.get("pagination", {})
logger.info(f" ✅ Status 200 - Found {pagination.get('total', 0)} catalogues")
logger.info(f" Page: {pagination.get('page')}/{pagination.get('pages_total')}")
if catalogues:
logger.info(f" Sample: {catalogues[0]['titre']}")
else:
logger.error(f" ❌ Status {response.status_code}")
return False
except Exception as e:
logger.error(f" ❌ Error: {e}")
return False
# Test 3: Get catalogue detail
logger.info("\n3. Testing GET /api/catalogues/{id}")
try:
# Get first catalog ID
db = SessionLocal()
cat = db.query(Catalogue).first()
db.close()
if cat:
response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}")
if response.status_code == 200:
detail = response.json()
logger.info(f" ✅ Status 200 - Catalogue: {detail['titre']}")
logger.info(f" Pages: {detail['nombre_pages']}")
else:
logger.error(f" ❌ Status {response.status_code}")
return False
else:
logger.warning(" ⚠️ No catalogues in DB to test detail endpoint")
except Exception as e:
logger.error(f" ❌ Error: {e}")
return False
# Test 4: Get catalogue pages
logger.info("\n4. Testing GET /api/catalogues/{id}/pages")
try:
if cat:
response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}/pages")
if response.status_code == 200:
pages = response.json()
logger.info(f" ✅ Status 200 - Found {len(pages)} pages")
if pages:
logger.info(f" Sample page: {pages[0]['numero_page']} - {pages[0]['image_url'][:50]}...")
else:
logger.error(f" ❌ Status {response.status_code}")
return False
except Exception as e:
logger.error(f" ❌ Error: {e}")
return False
# Test 5: Get stats (requires working DB)
logger.info("\n5. Testing GET /api/catalogues/admin/stats")
try:
response = await client.get(f"{API_BASE_URL}/catalogues/admin/stats")
if response.status_code == 200:
stats = response.json()
logger.info(f" ✅ Status 200 - Total catalogues: {stats['total_catalogues']}")
logger.info(f" Prochaine exécution: {stats['prochaine_execution']}")
else:
logger.error(f" ❌ Status {response.status_code}")
# Stats is optional, don't fail
except Exception as e:
logger.warning(f" ⚠️ Stats endpoint error (may require auth): {e}")
logger.info("\n✅ All API endpoints are working!\n")
return True
async def main():
"""Run all verification tests."""
logger.info("\n" + "=" * 60)
logger.info("BONIAL CATALOG MODULE - VERIFICATION SCRIPT")
logger.info("=" * 60 + "\n")
results = []
# Phase 1: Database tables
results.append(("Database Tables", check_database_tables()))
if not results[0][1]:
logger.error("\n❌ Database not ready. Please run: alembic upgrade head")
sys.exit(1)
# Phase 2: Enseignes seeding
results.append(("Enseignes Seeding", check_enseignes_seeding()))
# Phase 3: Scraper test (optional, can be slow)
scraper_test = input("\nRun scraper test? (Gifi - takes ~60s) [y/N]: ").lower() == "y"
if scraper_test:
results.append(("Scraper Test", await test_scraper()))
# Phase 4: API endpoints (requires app to be running)
api_test = input("\nTest API endpoints? (App must be running on :8555) [y/N]: ").lower() == "y"
if api_test:
results.append(("API Endpoints", await test_api_endpoints()))
# Summary
logger.info("\n" + "=" * 60)
logger.info("VERIFICATION SUMMARY")
logger.info("=" * 60)
for name, passed in results:
status = "✅ PASSED" if passed else "❌ FAILED"
logger.info(f"{status}: {name}")
all_passed = all(result[1] for result in results)
if all_passed:
logger.info("\n✅ ALL TESTS PASSED - Bonial module is ready!")
else:
logger.error("\n❌ SOME TESTS FAILED - Please review errors above")
sys.exit(1)
if __name__ == "__main__":
asyncio.run(main())
-138
View File
@@ -1,138 +0,0 @@
"""
Script de vérification pour tester Crawl4AI dans le scraper Tiendeo.
Usage:
python verify_crawl4ai_scraper.py
Ce script teste:
1. Import de Crawl4AI
2. Extraction d'une page catalogue Tiendeo
3. Comptage des pages trouvées
"""
import asyncio
import logging
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def test_crawl4ai_import():
"""Test 1: Vérifier que Crawl4AI est bien installé"""
try:
logger.info("✓ Crawl4AI importé avec succès")
logger.info(f" Version: {AsyncWebCrawler.__module__}")
return True
except Exception as e:
logger.error(f"✗ Erreur import Crawl4AI: {e}")
return False
async def test_catalog_page_extraction():
"""Test 2: Extraire les pages d'un catalogue Tiendeo"""
# URL de test - catalogue Gifi Nancy (à adapter si nécessaire)
test_url = "https://www.tiendeo.fr/Catalogues/nancy/gifi"
logger.info(f"\nTest extraction depuis: {test_url}")
try:
browser_config = BrowserConfig(
headless=True,
verbose=False,
extra_args=["--disable-gpu", "--no-sandbox", "--disable-dev-shm-usage"],
)
config = CrawlerRunConfig(
cache_mode=CacheMode.BYPASS,
wait_for_images=True,
process_iframes=True,
remove_overlay_elements=True,
wait_until="networkidle",
delay_before_return_html=3.0,
)
async with AsyncWebCrawler(config=browser_config) as crawler:
result = await crawler.arun(url=test_url, config=config)
if not result.success:
logger.error(f"✗ Échec du crawling: {result.error_message}")
return False
logger.info(f"✓ Page chargée avec succès")
logger.info(f" HTML length: {len(result.html)} chars")
# Tester l'extraction JavaScript
pages_data = await crawler.crawler_strategy.execute_js(
"""
() => {
const results = [];
const seenUrls = new Set();
const allImages = document.querySelectorAll('img');
allImages.forEach((img) => {
let src = img.src || img.getAttribute('data-src');
if (!src && img.srcset) {
const srcsetParts = img.srcset.split(',')[0].trim().split(' ');
src = srcsetParts[0];
}
if (!src || seenUrls.has(src)) return;
if (src.includes('logo') || src.includes('icon') || src.includes('avatar')) {
return;
}
const width = img.naturalWidth || img.width;
const height = img.naturalHeight || img.height;
if (width < 400 || height < 400) return;
const aspectRatio = width / height;
if (aspectRatio > 0.5 && aspectRatio < 0.9) {
seenUrls.add(src);
results.push({
image_url: src,
width: width,
height: height,
});
}
});
results.sort((a, b) => (b.width * b.height) - (a.width * a.height));
return results.map((item, index) => ({
...item,
numero_page: index + 1,
}));
}
"""
)
logger.info(f"✓ Extraction JavaScript réussie")
logger.info(f" Pages trouvées: {len(pages_data)}")
if len(pages_data) > 0:
test2 = await test_catalog_page_extraction()
# Résumé
logger.info("\n" + "=" * 70)
logger.info("RÉSUMÉ")
logger.info("=" * 70)
logger.info(f"Import Crawl4AI: {'✓ OK' if test1 else '✗ ÉCHEC'}")
logger.info(f"Extraction pages: {'✓ OK' if test2 else '✗ ÉCHEC'}")
if test1 and test2:
logger.info("\n✓ TOUS LES TESTS SONT PASSÉS!")
logger.info("Le scraper Crawl4AI est prêt à être utilisé.")
else:
logger.info("\n✗ CERTAINS TESTS ONT ÉCHOUÉ")
logger.info("Vérifiez les erreurs ci-dessus.")
if __name__ == "__main__":
asyncio.run(main())
-182
View File
@@ -1,182 +0,0 @@
import asyncio
import logging
import sys
from unittest.mock import MagicMock, AsyncMock
# Mock sqlalchemy
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
# Mock pydantic
mock_pydantic = MagicMock()
# Mock BaseModel
class MockBaseModel:
def __init__(self, **kwargs):
for k, v in kwargs.items():
setattr(self, k, v)
mock_pydantic.BaseModel = MockBaseModel
mock_pydantic.Field = MagicMock(return_value=None)
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
sys.modules["pydantic"] = mock_pydantic
# Mock app.utils.text which is imported by ai_schema
mock_utils_text = MagicMock()
sys.modules["app.utils.text"] = mock_utils_text
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
mock_utils_text.clean_text = lambda text: text.strip()
# Mock app.strings (if used) or other utils
sys.modules["app.utils"] = MagicMock()
# Mock app.utils.image
sys.modules["app.utils.image"] = MagicMock()
# Mock app.database
sys.modules["app.database"] = MagicMock()
# Mock playwright
mock_playwright = MagicMock()
sys.modules["playwright"] = mock_playwright
sys.modules["playwright.async_api"] = mock_playwright
# Mock generic types for type hints if needed
mock_playwright.Browser = MagicMock
mock_playwright.BrowserContext = MagicMock
mock_playwright.Page = MagicMock
mock_playwright.TimeoutError = Exception
# Now import the schema
from app.ai_schema import get_extraction_prompt, get_repair_prompt
# We can't import ScraperService easily if it inherits from things or uses decorators
# But for this test we only need get_extraction_prompt which is in ai_schema
# So we can skip importing ScraperService if it causes issues,
# BUT we wanted to verify ScraperService text cleaning logic...
# Let's mock ScraperService dependencies completely.
try:
from app.services.tracking_scraper_service import ScraperService
except ImportError:
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
ScraperService = None
# Mock litellm and tenacity
sys.modules["litellm"] = MagicMock()
sys.modules["tenacity"] = MagicMock()
mock_retry = MagicMock()
sys.modules["tenacity.retry"] = mock_retry
# Make sure imports inside ai_service don't fail
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
# We need to mock these specifically if the module imports them directly
mock_tenacity = MagicMock()
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
mock_tenacity.retry_if_exception_type = MagicMock()
mock_tenacity.stop_after_attempt = MagicMock()
mock_tenacity.wait_exponential = MagicMock()
sys.modules["tenacity"] = mock_tenacity
# Now import AIService
# We will mock the AI response to verify the parsing logic
from app.services.ai_service import AIService
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_extraction_logic():
print("Verifying Extraction Logic...")
# 1. Test Text Cleaning in ScraperService
# We can't mock Playwright page easily in a simple script without launching a browser.
# But we can test the AI prompt generation which is critical.
# Simulate B&M text
dirty_text = """
Menu
Accueil
Panier
Boisson energisante ice 25cl
Red Bull
1.15 €
Prix au litre : 4,60 € / L
En stock
Ajouter au panier
Footer
Mentions légales
"""
print("\n--- Testing Prompt Generation ---")
prompt = get_extraction_prompt(dirty_text)
# Verify strict instructions are present
checks = [
"CRITICAL",
"Ignore \"Prix au litre\"",
"B&M STORES Specific",
"Extract as DECIMAL NUMBER",
"ALWAYS select the TTC price",
"Ignore \"HT\""
]
all_passed = True
for check in checks:
if check in prompt:
print(f"[OK] Prompt contains: {check}")
else:
print(f"[FAIL] Prompt missing: {check}")
all_passed = False
if not all_passed:
print("Prompt verification failed!")
exit(1)
print("\n--- Testing Response Parsing (Mock AI) ---")
# Case 1: AI returns Main Price correctly
mock_response_1 = """
```json
{
"price": 1.15,
"currency": "EUR",
"in_stock": true,
"price_confidence": 0.95,
"in_stock_confidence": 1.0,
"source_type": "text"
}
```
"""
result = AIService.parse_and_validate_response(mock_response_1)
if result.price == 1.15 and result.in_stock is True:
print("[OK] Parsed correct mocked response.")
else:
print(f"[FAIL] Failed to parse correct response: {result}")
exit(1)
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
# If AI returns explicit null because it's confused
mock_response_2 = """
{
"price": null,
"currency": "EUR",
"in_stock": null,
"price_confidence": 0.0,
"in_stock_confidence": 0.0,
"source_type": "image"
}
"""
result = AIService.parse_and_validate_response(mock_response_2)
if result.price is None:
print("[OK] Parsed null response correctly.")
else:
print(f"[FAIL] Failed to parse null response.")
print("\nVerification of Logic Flow Complete (Simulated).")
print("Real-world verification requires running the full scraper.")
if __name__ == "__main__":
asyncio.run(verify_extraction_logic())
-48
View File
@@ -1,48 +0,0 @@
import asyncio
import logging
import os
import re
import sys
from unittest.mock import MagicMock, AsyncMock
# Add current directory to sys.path to allow importing app
sys.path.append(os.getcwd())
# Mock playwright before importing app
mock_playwright = MagicMock()
sys.modules["playwright"] = mock_playwright
sys.modules["playwright.async_api"] = mock_playwright
from app.services.tracking_scraper_service import ScraperService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_logic():
logger.info("Starting logic verification...")
# Mock Page object
mock_page = MagicMock()
mock_page.screenshot = AsyncMock()
item_id = 123
url = "http://test.com"
# Call _take_screenshot directly
logger.info("Calling _take_screenshot...")
filename = await ScraperService._take_screenshot(mock_page, url, item_id)
logger.info(f"Returned filename: {filename}")
# Verify format
pattern = r"screenshots/item_123_\d+\.png"
if re.match(pattern, filename):
logger.info("SUCCESS: Filename matches expected timestamp pattern!")
print("VERIFICATION_SUCCESS")
else:
logger.error(f"FAILURE: Filename {filename} does not match pattern {pattern}")
exit(1)
if __name__ == "__main__":
asyncio.run(verify_logic())
-164
View File
@@ -1,164 +0,0 @@
import os
import glob
import sys
from unittest.mock import MagicMock
# --- MOCKS SETUP ---
# We need to mock these BEFORE importing app.services.item_service
# to avoid ImportErrors due to missing dependencies in the test env.
# 1. Mock External Libs
sys.modules["fastapi"] = MagicMock()
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
# 2. Mock Internal App Modules that have heavy dependencies
# Mock app.database
mock_database = MagicMock()
sys.modules["app.database"] = mock_database
# Mock app.models
# We need models.Item and models.PriceHistory to be accessible attributes
mock_models = MagicMock()
sys.modules["app.models"] = mock_models
# Mock app.schemas
sys.modules["app.schemas"] = MagicMock()
# Mock app.services.settings_service
sys.modules["app.services.settings_service"] = MagicMock()
# Mock app.url_validation
sys.modules["app.url_validation"] = MagicMock()
# --- IMPORT TARGET ---
from app.services.item_service import ItemService
def verify_item_service_fix():
print("Starting verification of ItemService fix...")
# 1. Setup Mock DB and Item
# We must ensure that when ItemService does `item.id`, it works.
mock_db = MagicMock()
# Create a simple class to act as the Item model instance
class MockItem:
def __init__(self, id, name):
self.id = id
self.name = name
self.url = "http://test.com"
self.current_price = 10.0
self.in_stock = True
self.screenshot_url = None # This will be set by the service
# Attributes accessed by the service
self.notification_channel = None
self.target_price = None
self.current_price_confidence = 1.0
self.in_stock_confidence = 1.0
self.is_active = True
self.last_checked = None
self.is_refreshing = False
self.last_error = None
self.category = None
self.tags = None
self.description = None
# __dict__ is used by the service to create the result
self.dict_storage = {k:v for k,v in self.__dict__.items()}
@property
def __dict__(self):
# Update dict storage with current attributes
return {
"id": self.id,
"name": self.name,
"url": self.url
}
item_888 = MockItem(888, "Test Item")
# ItemService.get_items calls db.query(models.Item).all()
# We need to make sure models.Item is used in the query.
# The service does: items = db.query(models.Item).all()
mock_db.query.return_value.all.return_value = [item_888]
# It also queries PriceHistory
# db.query(models.PriceHistory).filter(...).first()
# Let's mock that to return None to force filesytem check (or check logic priority)
mock_db.query.return_value.filter.return_value.filter.return_value.order_by.return_value.first.return_value = None
# 2. Create Dummy Screenshot Files
os.makedirs("screenshots", exist_ok=True)
# Clean up
for f in glob.glob("screenshots/item_888_*.png"):
os.remove(f)
if os.path.exists("screenshots/item_888.png"):
os.remove("screenshots/item_888.png")
# Scenario:
# 1. item_888.png exists (legacy)
# 2. item_888_1000.png exists (old timestamp)
# 3. item_888_2000.png exists (new timestamp)
# Expected: get_items should pick item_888_2000.png
file_legacy = "screenshots/item_888.png"
file_old = "screenshots/item_888_1000.png"
file_new = "screenshots/item_888_2000.png"
with open(file_legacy, "w") as f: f.write(".")
with open(file_old, "w") as f: f.write(".")
with open(file_new, "w") as f: f.write(".")
print(f"Created files: {file_legacy}, {file_old}, {file_new}")
try:
# 3. Test get_items
print("Testing get_items()...")
items = ItemService.get_items(mock_db)
if not items:
print("FAILURE: No items returned")
exit(1)
result = items[0]
screenshot_url = result.get("screenshot_url")
print(f"Returned screenshot_url: {screenshot_url}")
expected_url = f"/screenshots/{os.path.basename(file_new)}"
if screenshot_url == expected_url:
print("SUCCESS: Correctly identified the latest screenshot!")
else:
print(f"FAILURE: Expected {expected_url}, got {screenshot_url}")
# If it failed, maybe it picked legacy?
if screenshot_url == f"/screenshots/{os.path.basename(file_legacy)}":
print("Picked legacy file instead of timestamped one.")
exit(1)
# 4. Test delete_item
print("Testing delete_item()...")
# Ensure the query returns our item
mock_db.query.return_value.filter.return_value.first.return_value = item_888
ItemService.delete_item(mock_db, 888)
# Check files
remaining = glob.glob("screenshots/item_888*.png")
if not remaining:
print("SUCCESS: All screenshots deleted.")
else:
print(f"FAILURE: Files remaining: {remaining}")
exit(1)
finally:
# Cleanup
for f in [file_legacy, file_old, file_new]:
if os.path.exists(f):
os.remove(f)
if __name__ == "__main__":
verify_item_service_fix()
-43
View File
@@ -1,43 +0,0 @@
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.parsers.gifi_parser import GifiParser
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
def verify_fix():
print("Verifying Gifi parser fix...")
# Load the dump file (we know it exists from previous steps)
dump_path = "gifi_full.html"
if not os.path.exists(dump_path):
print(f"Error: {dump_path} not found.")
return
with open(dump_path, "r", encoding="utf-8") as f:
html = f.read()
parser = GifiParser()
# Dummy URL
url = "https://www.gifi.fr/test-product.html"
print("Parsing product details...")
details = parser.parse_product_details(html, url)
print("\n--- Extraction Results ---")
print(f"Bypass Price: {details.get('price')}")
print(f"Bypass Stock: {details.get('in_stock')}")
print(f"Currency: {details.get('currency')}")
if details.get('price') is not None:
print("\nSUCCESS: Price extracted successfully!")
else:
print("\nFAILURE: Price not found in dump.")
if __name__ == "__main__":
verify_fix()
-50
View File
@@ -1,50 +0,0 @@
import asyncio
import logging
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_leclerc():
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
page = await browser.new_page()
# 1. Perform Search
logger.info("--- Step 1: Searching for 'Chaise' on E.Leclerc ---")
search_url = "https://www.e.leclerc/recherche?q=Chaise"
try:
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(5000) # Wait for JS
except Exception as e:
logger.error(f"Navigation failed: {e}")
# 2. Dump HTML
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# 3. Analyze Classes
classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')")
logger.info(f"Classes found: {classes[:1000]}")
# 4. Check for Product Selectors
selectors = [
"div[class*='product']",
"article",
".product-card",
".c-product-card",
"a[class*='product']"
]
for sel in selectors:
count = await page.locator(sel).count()
if count > 0:
logger.info(f"Selector '{sel}' found {count} elements")
first_html = await page.locator(sel).first.evaluate("el => el.outerHTML")
logger.info(f"First element HTML ({sel}): {first_html[:500]}...")
await browser.close()
if __name__ == "__main__":
asyncio.run(verify_leclerc())
-74
View File
@@ -1,74 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(
level=logging.DEBUG, # Enable DEBUG logging
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
# Set other loggers to INFO to avoid noise
logging.getLogger("urllib3").setLevel(logging.INFO)
logging.getLogger("asyncio").setLevel(logging.INFO)
logging.getLogger("websockets").setLevel(logging.INFO)
async def verify_search():
query = "chaise" # Updated query
print(f"--- Starting Verification Search for '{query}' ---")
# 1. Test Browserless Connection
print("\n[1] Testing Browserless Connection...")
try:
await browserless_service.initialize()
print("✅ Browserless connected successfully")
except Exception as e:
print(f"❌ Browserless connection failed: {e}")
return
# 2. Test Specific Sites
# sites_to_test = ["gifi.fr", "lincroyable.fr", "stokomani.fr"] # Excluded amazon.fr
sites_to_test = ["stokomani.fr"] # Focus on Stokomani for now as requested/implied context
for site in sites_to_test:
print(f"\n[2] Testing Search on {site}...")
try:
count = 0
async for r in new_search_service.search_site_generator(site, query):
count += 1
print(f"✅ Found result: {r.title} ({r.url})")
print(f" Price: {r.price} {r.currency}")
print(f" Image: {r.image_url}")
if count >= 1:
break
if count == 0:
print(f"⚠️ No results found for {site}")
# Dump HTML for debugging
try:
content, _ = await browserless_service.get_page_content(
SITE_CONFIGS[site]["search_url"].format(query=query),
use_proxy=SITE_CONFIGS[site].get("requires_proxy", False),
wait_selector=SITE_CONFIGS[site].get("wait_selector")
)
with open(f"/app/debug_dumps/{site}_failed_verification.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"📄 Saved HTML dump to /app/debug_dumps/{site}_failed_verification.html")
except Exception as dump_e:
print(f"❌ Failed to save HTML dump: {dump_e}")
except Exception as e:
print(f"❌ Error searching {site}: {e}")
# 3. Cleanup
await browserless_service.shutdown()
print("\n--- Verification Complete ---")
if __name__ == "__main__":
asyncio.run(verify_search())
-33
View File
@@ -1,33 +0,0 @@
import asyncio
import logging
import sys
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_popups():
# URL that was reported to have issues
url = "https://www.bmstores.fr/363943-bougie-parfumee-avec-bijou-350g-senteurs-assorties"
# Fallback to search if that product is gone
fallback_url = "https://www.bmstores.fr/module/ambjolisearch/jolisearch?s=calendrier"
logger.info(f"--- Testing Popup Handling on {url} ---")
try:
content, screenshot_path = await browserless_service.get_page_content(
url,
extract_text=False
)
logger.info(f"Screenshot saved to: {screenshot_path}")
logger.info("Please inspect the screenshot to ensure no 'Stock Inconnu' or 'Calendrier' popups are visible.")
except Exception as e:
logger.error(f"Verification failed: {e}")
finally:
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(verify_popups())
-48
View File
@@ -1,48 +0,0 @@
import asyncio
import logging
import os
from playwright.async_api import async_playwright
from app.services.scraper_service import ScraperService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_fix():
logger.info("Starting verification...")
# Mock browserless URL if not set
if not os.getenv("BROWSERLESS_URL"):
os.environ["BROWSERLESS_URL"] = "ws://browserless:3000"
playwright = await async_playwright().start()
try:
# Connect to browserless (or launch local if not available, but code expects connect)
# For this test, we might need to mock the browser object if we can't actually connect
# But let's try to just create a local browser for testing purposes if connect fails
# Actually, the code expects a browser object.
logger.info("Launching local browser for test...")
browser = await playwright.chromium.launch()
logger.info("Calling scrape_item with browser argument...")
try:
# We pass a dummy URL, we expect it might fail scraping but NOT raise TypeError
await ScraperService.scrape_item(
url="https://example.com",
browser=browser,
timeout=5000 # Short timeout
)
logger.info("SUCCESS: scrape_item accepted the browser argument!")
except TypeError as e:
logger.error(f"FAILURE: TypeError raised: {e}")
except Exception as e:
logger.info(f"Scraping failed as expected (network/etc), but argument was accepted: {e}")
await browser.close()
finally:
await playwright.stop()
if __name__ == "__main__":
asyncio.run(verify_fix())
-42
View File
@@ -1,42 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.direct_search_service import direct_search_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
async def verify_site(site_key):
logger.info(f"Verifying {site_key}...")
try:
# direct_search_service.search_site might need to be called differently if it's an instance method
# checking previous usage or assuming standard service pattern
results = await direct_search_service.search_site(site_key, "chaise")
if results:
logger.info(f"✅ {site_key}: Found {len(results)} results")
for i, res in enumerate(results[:3]):
title = res.get('title', 'No Title')
price = res.get('price', 'No Price')
url = res.get('url', 'No URL')
logger.info(f" {i+1}. {title[:50]}... - {price} - {url[:50]}...")
else:
logger.error(f"❌ {site_key}: No results found")
except Exception as e:
logger.error(f"❌ {site_key}: Error - {e}")
async def main():
sites_to_test = ["gifi.fr", "stokomani.fr", "auchan.fr", "carrefour.fr", "amazon.fr", "action.com"]
# Run sequentially to avoid overwhelming resources/logs
for site in sites_to_test:
await verify_site(site)
if __name__ == "__main__":
asyncio.run(main())
-74
View File
@@ -1,74 +0,0 @@
import asyncio
import logging
import os
import re
from datetime import datetime
from unittest.mock import MagicMock
from playwright.async_api import async_playwright
from app.services.tracking_scraper_service import ScraperService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def mock_connect_browser(p):
logger.info("MOCK: Launching local browser instead of connecting to browserless")
return await p.chromium.launch()
async def verify_fix():
logger.info("Starting verification...")
# Monkey-patch _connect_browser to use local browser
ScraperService._connect_browser = mock_connect_browser
# Ensure screenshots dir exists
os.makedirs("screenshots", exist_ok=True)
# Test Item ID 999
item_id = 999
url = "https://example.com"
logger.info(f"Scraping item {item_id}...")
# We expect this to fail scraping real content from example.com with specific selectors,
# but we only care about the screenshot filename generation which happens at the end.
# Actually, if scraping fails, it might return None, "" early.
# checking tracking_scraper_service.py:
# It has a try/except block.
# If _navigate_and_wait works, it proceeds. example.com should load.
# _take_screenshot is called at the end.
# However, ScraperService.scrape_item returns (None, "") if exception occurs.
# We need to make sure it doesn't crash before screenshot.
# example.com is simple, so it should load.
# It will try to click popups (won't find any), wait for selector (if provided).
# If we don't provide selector, it calls _auto_detect_price.
full_path, _ = await ScraperService.scrape_item(url=url, item_id=item_id)
if full_path:
logger.info(f"Screenshot path returned: {full_path}")
# Verify format: item_{id}_{timestamp}.png
# Check if it matches regex
pattern = r"screenshots/item_999_\d+\.png"
if re.match(pattern, full_path):
logger.info("SUCCESS: Filename contains timestamp!")
else:
logger.error(f"FAILURE: Filename does not match pattern {pattern}")
exit(1)
# Clean up
if os.path.exists(full_path):
os.remove(full_path)
logger.info("Cleaned up screenshot file")
else:
logger.error("FAILURE: Scraper returned None for path. Did navigation fail?")
exit(1)
await ScraperService.shutdown()
if __name__ == "__main__":
asyncio.run(verify_fix())
-45
View File
@@ -1,45 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.core.search_config import SITE_CONFIGS
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
async def test_specific_sites():
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
query = "chaise"
print(f"Testing {target_sites} with query '{query}'...")
await browserless_service.initialize()
for site_key in target_sites:
if site_key not in SITE_CONFIGS:
print(f"Skipping {site_key} (not in config)")
continue
print(f"\n--- Testing {site_key} ---")
try:
results = await new_search_service.search_site(site_key, query)
print(f"Found {len(results)} results")
for r in results[:3]:
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
except Exception as e:
print(f"Error testing {site_key}: {e}")
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(test_specific_sites())
-50
View File
@@ -1,50 +0,0 @@
import sys
import unittest
# Mock modules to avoid ImportError for app dependencies we don't need for this specific test
from unittest.mock import MagicMock
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
sys.modules["app.database"] = MagicMock()
sys.modules["app.utils.image"] = MagicMock()
sys.modules["app.utils.text"] = MagicMock()
sys.modules["app.utils.text"].filter_relevant_text = lambda text, max_length: text
# Mock pydantic
mock_pydantic = MagicMock()
class MockBaseModel:
pass
mock_pydantic.BaseModel = MockBaseModel
mock_pydantic.Field = MagicMock(return_value=None)
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
sys.modules["pydantic"] = mock_pydantic
# Import the schema module
from app.ai_schema import get_extraction_prompt
class TestVisionPriorityPrompt(unittest.TestCase):
def test_vision_first_directives(self):
"""Verify that the prompt contains the Vision-First directives."""
# Scenario: Some random text context
page_text = "Some random text content from the page."
prompt = get_extraction_prompt(page_text)
print("\nGenerated Prompt Snippet:\n", prompt[:500], "...\n")
# Check for Critical Directives
self.assertIn("Vision-First Price Extraction Agent", prompt)
self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt)
self.assertIn("IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE", prompt)
# Check for stock rules
self.assertIn("STOCK STATUS RULES", prompt)
def test_prompt_without_text(self):
"""Verify prompt structure when no text is provided."""
prompt = get_extraction_prompt(None)
self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt)
self.assertNotIn("**Relevant text from page:**", prompt)
if __name__ == "__main__":
unittest.main()
+14 -28
View File
@@ -1,38 +1,24 @@
# Walkthrough: Vision-First Price Extraction
# Walkthrough: Price Extraction & Catalog Fixes
In response to issues where the AI was being misled by hidden text (like unit prices or old prices in HTML), we have implemented a **Vision-Priority Strategy**.
## 1. Vision Priority Strategy
## Changes Implemented
**Problem**: The AI was prioritizing text data (often hidden/outdated) over the visual price on the screenshot, extracting incorrect prices (e.g. 0.99€ instead of 1.27€).
### 1. Updated AI System Prompt (`app/ai_schema.py`)
**Solution**:
We completely rewrote the `EXTRACTION_PROMPT_TEMPLATE` to enforce the following rules:
- **Prompt Engineering**: Rewrote the system prompt in `ai_schema.py` to explicitly declare the **IMAGE AS THE SOURCE OF TRUTH**.
- **Logic Fix**: Disabled the `AIPriceExtractor` (Text-only AI) in `scheduler_service.py` which was short-circuiting the logic before the Vision AI could run.
- **Source of Truth = Image**: explicit instruction that the screenshot takes precedence over any text.
- **Conflict Resolution**: "IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE."
- **Visual Focus Rules**:
- Look for the largest/boldest price.
- Ignore small, styling-less text (often unit prices).
- Ignore crossed-out text.
## 2. Catalog Scraper Fix
### Verification
**Problem**: Catalogs were not updating because the `browserless` service was failing (likely blocked or network issues), preventing the scraper from loading `cataloguemate.fr`.
We verified the new prompt generation using `verify_vision_priority.py`.
**Solution**:
**Generated Prompt Preview:**
- **HTTP Fallback**: Modified `cataloguemate_scraper.py` to use a robust fallback mechanism.
- First attempts to use the secure Browserless browser.
- If that fails, it instantly falls back to a standard `httpx` HTTP request, which is often sufficient for static catalog sites.
```text
You are a Vision-First Price Extraction Agent.
Your Goal: Extract the main product price exactly as a human sees it on the screen.
## 3. Cleanup
**SOURCE OF TRUTH = IMAGE**
- The image provided is the **Absolute Truth**.
- The text provided below is scraped HTML content which may contain hidden/old prices.
- **IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE.**
```
## How to Test
1. Go to "Suivis Prix".
2. Force refresh an item that was previously incorrect (e.g., B&M item showing unit price).
3. The AI should now ignore the "hidden" unit price text and read the main price tag from the image.
- Removed 20+ temporary debug/verification scripts (`debug_*.py`, `verify_*.py`) from the root directory to keep the production environment clean.