mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #240 from R0m1k3/antigravity
chore: remove obsolete debug and verification scripts and streamline …
This commit is contained in:
30 files changed
+17
-2322
No files matched your search
@@ -1,60 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000")
|
||||
|
||||
async def debug_amazon():
|
||||
logger.info("Starting Amazon Debug Script")
|
||||
|
||||
async with async_playwright() as p:
|
||||
try:
|
||||
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
|
||||
browser = await p.chromium.connect_over_cdp(BROWSERLESS_URL)
|
||||
|
||||
# Use a very standard, recent User-Agent
|
||||
user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
|
||||
context = await browser.new_context(
|
||||
user_agent=user_agent,
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
locale="fr-FR",
|
||||
timezone_id="Europe/Paris"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
url = "https://www.amazon.fr/s?k=iphone"
|
||||
logger.info(f"Navigating to {url}")
|
||||
|
||||
response = await page.goto(url, wait_until="domcontentloaded", timeout=30000)
|
||||
|
||||
if response:
|
||||
status = response.status
|
||||
logger.info(f"Response Status: {status}")
|
||||
|
||||
content = await page.content()
|
||||
if "api-services-support@amazon.com" in content or "Toutes nos excuses" in content:
|
||||
logger.error("BLOCK DETECTED: Found blocking message in content")
|
||||
else:
|
||||
logger.info("No obvious blocking message found")
|
||||
|
||||
# Save screenshot
|
||||
await page.screenshot(path="debug_amazon_screenshot.png")
|
||||
logger.info("Screenshot saved to debug_amazon_screenshot.png")
|
||||
|
||||
else:
|
||||
logger.error("No response received")
|
||||
|
||||
await browser.close()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"An error occurred: {e}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(debug_amazon())
|
||||
@@ -1,121 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from bs4 import BeautifulSoup
|
||||
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler(sys.stdout)]
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
BASE_URL = "https://www.cataloguemate.fr"
|
||||
|
||||
async def debug_catalog_list():
|
||||
"""Debug the catalog list extraction"""
|
||||
# Test with Gifi
|
||||
slug = "gifi"
|
||||
url = f"{BASE_URL}/{slug}/"
|
||||
|
||||
logger.info(f"--- DEBUGGING LIST: {url} ---")
|
||||
|
||||
browser_config = BrowserConfig(headless=True)
|
||||
run_config = CrawlerRunConfig(
|
||||
cache_mode=CacheMode.BYPASS,
|
||||
wait_for_images=True,
|
||||
)
|
||||
|
||||
async with AsyncWebCrawler(config=browser_config) as crawler:
|
||||
result = await crawler.arun(url=url, config=run_config)
|
||||
|
||||
if not result.success:
|
||||
logger.error(f"Failed to fetch {url}: {result.error_message}")
|
||||
return None
|
||||
|
||||
logger.info(f"Successfully fetched {url} ({len(result.html)} chars)")
|
||||
|
||||
soup = BeautifulSoup(result.html, 'html.parser')
|
||||
|
||||
# 1. Dump all links to see what we have
|
||||
links = soup.find_all('a', href=True)
|
||||
logger.info(f"Found {len(links)} links total")
|
||||
|
||||
potential_catalogs = []
|
||||
|
||||
for i, link in enumerate(links):
|
||||
href = link['href']
|
||||
text = link.get_text(strip=True)
|
||||
|
||||
# Normalize
|
||||
if href.startswith(BASE_URL):
|
||||
href = href.replace(BASE_URL, "")
|
||||
|
||||
# Log interesting links
|
||||
if slug in href or "catalogue" in href.lower():
|
||||
logger.info(f"Link {i}: {href} | Text: '{text}'")
|
||||
|
||||
# Apply our filter logic to see if it passes
|
||||
if href.startswith(f"/{slug}/") and href != f"/{slug}/":
|
||||
if not any(x in href for x in ["offres", "magasins", "rechercher"]):
|
||||
potential_catalogs.append(href)
|
||||
logger.info(f" -> MATCHES FILTER!")
|
||||
|
||||
logger.info(f"Total matching catalogs: {len(potential_catalogs)}")
|
||||
return potential_catalogs[0] if potential_catalogs else None
|
||||
|
||||
async def debug_catalog_page(catalog_rel_url):
|
||||
"""Debug the catalog page extraction"""
|
||||
if not catalog_rel_url:
|
||||
logger.error("No catalog URL to debug")
|
||||
return
|
||||
|
||||
full_url = f"{BASE_URL}{catalog_rel_url}"
|
||||
logger.info(f"\n--- DEBUGGING PAGE: {full_url} ---")
|
||||
|
||||
browser_config = BrowserConfig(headless=True)
|
||||
run_config = CrawlerRunConfig(
|
||||
cache_mode=CacheMode.BYPASS,
|
||||
wait_for_images=True,
|
||||
delay_before_return_html=2.0 # Wait a bit more
|
||||
)
|
||||
|
||||
async with AsyncWebCrawler(config=browser_config) as crawler:
|
||||
result = await crawler.arun(url=full_url, config=run_config)
|
||||
|
||||
if not result.success:
|
||||
logger.error(f"Failed to fetch {full_url}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(result.html, 'html.parser')
|
||||
|
||||
# 1. Dump all images
|
||||
images = soup.find_all('img')
|
||||
logger.info(f"Found {len(images)} images")
|
||||
|
||||
for i, img in enumerate(images):
|
||||
src = img.get('src', '')
|
||||
width = img.get('width', '?')
|
||||
height = img.get('height', '?')
|
||||
alt = img.get('alt', '')
|
||||
|
||||
# Filter noise
|
||||
if "logo" in src or "icon" in src:
|
||||
continue
|
||||
|
||||
logger.info(f"Img {i}: {src} | {width}x{height} | Alt: {alt}")
|
||||
|
||||
# Check our heuristic
|
||||
is_likely = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
|
||||
if is_likely:
|
||||
logger.info(" -> LIKELY CATALOG IMAGE")
|
||||
|
||||
if __name__ == "__main__":
|
||||
async def main():
|
||||
cat_url = await debug_catalog_list()
|
||||
if cat_url:
|
||||
await debug_catalog_page(cat_url)
|
||||
|
||||
asyncio.run(main())
|
||||
@@ -1,96 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
# from bs4 import BeautifulSoup # Removed
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler(sys.stdout)]
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
BASE_URL = "https://www.cataloguemate.fr"
|
||||
|
||||
async def debug_catalog_list():
|
||||
"""Debug the catalog list extraction using Playwright directly"""
|
||||
slug = "gifi"
|
||||
url = f"{BASE_URL}/offres/paris/{slug}/"
|
||||
|
||||
logger.info(f"--- DEBUGGING LIST: {url} ---")
|
||||
|
||||
async with async_playwright() as p:
|
||||
# Use browserless or local depending on connection
|
||||
# For this script we use local headless for simplicity if browserless is not available,
|
||||
# but since we are in the container we might need browserless.
|
||||
# Let's try to simulate what browserless_service does but simplified.
|
||||
|
||||
try:
|
||||
browser = await p.chromium.launch(headless=True) # Try local first
|
||||
except:
|
||||
logger.info("Local browser failed, trying browserless...")
|
||||
browser = await p.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
|
||||
)
|
||||
page = await context.new_page()
|
||||
|
||||
logger.info(f"Navigating to {url}")
|
||||
await page.goto(url, wait_until="domcontentloaded")
|
||||
|
||||
# Wait a bit
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
content = await page.content()
|
||||
# soup = BeautifulSoup(content, 'html.parser')
|
||||
|
||||
# 1. Dump all links to see what we have
|
||||
# Use JS to extract links
|
||||
links_data = await page.evaluate("""
|
||||
() => {
|
||||
return Array.from(document.querySelectorAll('a[href]')).map(a => ({
|
||||
href: a.getAttribute('href'),
|
||||
text: a.innerText.trim()
|
||||
}));
|
||||
}
|
||||
""")
|
||||
|
||||
logger.info(f"Found {len(links_data)} links total")
|
||||
|
||||
potential_catalogs = []
|
||||
|
||||
for i, link in enumerate(links_data):
|
||||
href = link['href']
|
||||
text = link['text']
|
||||
|
||||
# Normalize
|
||||
if href.startswith(BASE_URL):
|
||||
href = href.replace(BASE_URL, "")
|
||||
|
||||
# Log interesting links
|
||||
if slug in href or "catalogue" in href.lower():
|
||||
logger.info(f"Link {i}: {href} | Text: '{text}'")
|
||||
|
||||
# Apply our filter logic to see if it passes
|
||||
if f"/{slug}/" in href:
|
||||
if not any(x in href for x in ["/offres/", "/magasins/", "/rechercher/", "page="]):
|
||||
# Check ID pattern
|
||||
import re
|
||||
if re.search(r'-\d+/?$', href) or "catalogue" in href.lower():
|
||||
potential_catalogs.append(href)
|
||||
logger.info(f" -> MATCHES FILTER! (Found catalog)")
|
||||
else:
|
||||
logger.info(f" -> Rejected (no ID/keyword)")
|
||||
else:
|
||||
logger.info(f" -> Rejected (invalid pattern)")
|
||||
|
||||
logger.info(f"Total matching catalogs: {len(potential_catalogs)}")
|
||||
|
||||
await browser.close()
|
||||
return potential_catalogs[0] if potential_catalogs else None
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(debug_catalog_list())
|
||||
@@ -1,50 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def debug_gifi():
|
||||
print("Initializing search service...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
try:
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = await ImprovedSearchService.search_site("gifi.fr", "chaise")
|
||||
|
||||
print(f"\nFound {len(results)} results.")
|
||||
|
||||
if results:
|
||||
print("\n--- First 5 Results ---")
|
||||
for i, res in enumerate(results[:5]):
|
||||
print(f"\nItem {i+1}:")
|
||||
print(f" Title: {res.title}")
|
||||
print(f" Price: {res.price} {res.currency}")
|
||||
print(f" Image: {res.image_url}")
|
||||
print(f" URL: {res.url}")
|
||||
print(f" In Stock: {res.in_stock}")
|
||||
|
||||
# Check for missing critical data
|
||||
missing_price = sum(1 for r in results if r.price is None)
|
||||
missing_image = sum(1 for r in results if not r.image_url)
|
||||
|
||||
print(f"\nStats:")
|
||||
print(f" Total: {len(results)}")
|
||||
print(f" Missing Price: {missing_price}")
|
||||
print(f" Missing Image: {missing_image}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error: {e}", exc_info=True)
|
||||
finally:
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(debug_gifi())
|
||||
@@ -1,120 +0,0 @@
|
||||
"""
|
||||
Debug script to identify correct selectors for missing sites
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Add app directory to path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.browserless_service import browserless_service
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def analyze_site(name: str, url: str, wait_selector: str = None):
|
||||
"""Analyze a search results page to identify selectors"""
|
||||
print(f"\n{'='*80}")
|
||||
print(f"Analyzing: {name}")
|
||||
print(f"URL: {url}")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
html, screenshot = await browserless_service.get_page_content(
|
||||
url,
|
||||
use_proxy=False,
|
||||
wait_selector=wait_selector,
|
||||
wait_timeout=10000
|
||||
)
|
||||
|
||||
if not html:
|
||||
print(f"❌ Failed to get content for {name}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
# Save HTML for manual inspection
|
||||
output_file = f"debug_{name.lower().replace(' ', '_')}.html"
|
||||
with open(output_file, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
print(f"💾 HTML saved to: {output_file}")
|
||||
|
||||
# Common product link patterns
|
||||
product_patterns = [
|
||||
"a[href*='/product']",
|
||||
"a[href*='/p/']",
|
||||
"a[href*='/produit']",
|
||||
"a.product-card",
|
||||
"a.product-link",
|
||||
"[data-product-id]",
|
||||
"article a",
|
||||
"div[data-testid*='product'] a",
|
||||
]
|
||||
|
||||
print("\n🔍 Searching for product links...")
|
||||
for pattern in product_patterns:
|
||||
links = soup.select(pattern)
|
||||
if links and len(links) >= 3:
|
||||
print(f"✅ Found {len(links)} matches for: {pattern}")
|
||||
# Show first 3 examples
|
||||
for i, link in enumerate(links[:3], 1):
|
||||
href = link.get('href', 'NO_HREF')
|
||||
text = link.get_text(strip=True)[:50]
|
||||
print(f" {i}. {href[:60]} | {text}")
|
||||
elif links:
|
||||
print(f"⚠️ Found {len(links)} matches for: {pattern} (too few)")
|
||||
|
||||
# Common image patterns
|
||||
image_patterns = [
|
||||
"img[src*='product']",
|
||||
"img.product-image",
|
||||
"img.product-img",
|
||||
"img[loading='lazy']",
|
||||
"picture img",
|
||||
"img[data-src]",
|
||||
]
|
||||
|
||||
print("\n🖼️ Searching for product images...")
|
||||
for pattern in image_patterns:
|
||||
images = soup.select(pattern)
|
||||
if images and len(images) >= 3:
|
||||
print(f"✅ Found {len(images)} matches for: {pattern}")
|
||||
for i, img in enumerate(images[:3], 1):
|
||||
src = img.get('src') or img.get('data-src', 'NO_SRC')
|
||||
alt = img.get('alt', 'NO_ALT')[:50]
|
||||
print(f" {i}. {src[:60]} | {alt}")
|
||||
elif images:
|
||||
print(f"⚠️ Found {len(images)} matches for: {pattern} (too few)")
|
||||
|
||||
print(f"\n✅ Analysis complete for {name}\n")
|
||||
|
||||
async def main():
|
||||
"""Test all missing sites"""
|
||||
sites = [
|
||||
{
|
||||
"name": "E.Leclerc",
|
||||
"url": "https://www.e-leclerc.com/recherche?text=chaise",
|
||||
"wait": ".product-card, .search-results"
|
||||
},
|
||||
{
|
||||
"name": "Auchan",
|
||||
"url": "https://www.auchan.fr/search?text=chaise",
|
||||
"wait": ".product-card, .product-item"
|
||||
},
|
||||
{
|
||||
"name": "Carrefour",
|
||||
"url": "https://www.carrefour.fr/s?q=chaise",
|
||||
"wait": "[data-testid*='product'], .product"
|
||||
},
|
||||
]
|
||||
|
||||
for site in sites:
|
||||
try:
|
||||
await analyze_site(site["name"], site["url"], site.get("wait"))
|
||||
except Exception as e:
|
||||
print(f"❌ Error analyzing {site['name']}: {e}")
|
||||
|
||||
# Small delay between sites
|
||||
await asyncio.sleep(2)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,67 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.browserless_service import browserless_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def debug_site(site_key: str, query: str = "chaise"):
|
||||
print(f"--- Debugging {site_key} ---")
|
||||
config = SITE_CONFIGS.get(site_key)
|
||||
if not config:
|
||||
print(f"Site {site_key} not found in config")
|
||||
return
|
||||
|
||||
search_url = config["search_url"].format(query=query)
|
||||
print(f"URL: {search_url}")
|
||||
|
||||
print("Fetching content...")
|
||||
try:
|
||||
html, screenshot_path = await browserless_service.get_page_content(
|
||||
search_url,
|
||||
use_proxy=config.get("requires_proxy", False),
|
||||
wait_selector=config.get("wait_selector")
|
||||
)
|
||||
|
||||
print(f"Screenshot saved to: {screenshot_path}")
|
||||
print(f"HTML length: {len(html)}")
|
||||
|
||||
# Save HTML for inspection
|
||||
with open(f"debug_{site_key}.html", "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
print(f"HTML saved to debug_{site_key}.html")
|
||||
|
||||
# Check if wait selector is present in HTML
|
||||
if config.get("wait_selector"):
|
||||
from bs4 import BeautifulSoup
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
found = soup.select(config["wait_selector"])
|
||||
print(f"Wait selector '{config['wait_selector']}' found: {len(found)} elements")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
|
||||
async def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python debug_scraper.py <site_key> [query]")
|
||||
return
|
||||
|
||||
site_key = sys.argv[1]
|
||||
query = sys.argv[2] if len(sys.argv) > 2 else "chaise"
|
||||
|
||||
await browserless_service.start()
|
||||
try:
|
||||
await debug_site(site_key, query)
|
||||
finally:
|
||||
await browserless_service.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,54 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
|
||||
from app.services.search_service import new_search_service
|
||||
from app.services.browserless_service import browserless_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler()]
|
||||
)
|
||||
|
||||
async def debug_search():
|
||||
print("--- Debugging Search ---")
|
||||
|
||||
# Check Config
|
||||
print(f"Available Config Keys: {list(SITE_CONFIGS.keys())}")
|
||||
|
||||
sites = ["amazon.fr", "stokomani.fr", "lincroyable.fr"]
|
||||
query = "iphone"
|
||||
|
||||
try:
|
||||
await browserless_service.initialize()
|
||||
|
||||
for site in sites:
|
||||
print(f"\nTesting {site}...")
|
||||
if site not in SITE_CONFIGS:
|
||||
print(f"❌ Site {site} NOT found in SITE_CONFIGS")
|
||||
continue
|
||||
|
||||
try:
|
||||
count = 0
|
||||
async for r in new_search_service.search_site_generator(site, query):
|
||||
count += 1
|
||||
print(f"✅ Found: {r.title} - {r.price} {r.currency}")
|
||||
if count >= 1:
|
||||
break
|
||||
if count == 0:
|
||||
print(f"⚠️ No results for {site}")
|
||||
except Exception as e:
|
||||
print(f"❌ Error searching {site}: {e}")
|
||||
|
||||
finally:
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(debug_search())
|
||||
@@ -1,30 +0,0 @@
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add app to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.core.database import SessionLocal
|
||||
from app.models import SearchSite
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
def debug_sites():
|
||||
db = SessionLocal()
|
||||
try:
|
||||
print("=== SITE_CONFIGS Keys ===")
|
||||
for key in SITE_CONFIGS.keys():
|
||||
print(f"- {key}")
|
||||
|
||||
print("\n=== DB SearchSites ===")
|
||||
sites = db.query(SearchSite).all()
|
||||
for site in sites:
|
||||
print(f"ID: {site.id} | Name: {site.name} | Domain: {site.domain} | Active: {site.is_active}")
|
||||
print(f" URL: {site.search_url}")
|
||||
print(f" Selector: {site.product_link_selector}")
|
||||
print("-" * 20)
|
||||
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
debug_sites()
|
||||
@@ -16,6 +16,9 @@
|
||||
- [x] Reproduce failure (Confirmed Browserless issue via analysis).
|
||||
- [x] Fix `cataloguemate_scraper.py` with HTTP fallback.
|
||||
|
||||
- [x] **Cleanup**:
|
||||
- [x] Deleted `debug_*.py` and `verify_*.py` files.
|
||||
|
||||
## 📝 Progress Log
|
||||
|
||||
- **2025-12-22**: Vision Priority implemented and verified.
|
||||
|
||||
@@ -1,79 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_action():
|
||||
async with async_playwright() as p:
|
||||
# Use a standard User Agent
|
||||
ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(user_agent=ua)
|
||||
page = await context.new_page()
|
||||
|
||||
# 1. Perform Search
|
||||
logger.info("--- Step 1: Searching for 'Chaise' on Action ---")
|
||||
search_url = "https://www.action.com/fr-fr/search/?q=Chaise"
|
||||
|
||||
try:
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
|
||||
# Wait a bit for any JS redirects or challenges
|
||||
await page.wait_for_timeout(5000)
|
||||
except Exception as e:
|
||||
logger.error(f"Navigation error: {e}")
|
||||
|
||||
# 2. Take Screenshot
|
||||
await page.screenshot(path="action_debug.png")
|
||||
logger.info("Screenshot saved: action_debug.png")
|
||||
|
||||
# 3. Dump Content
|
||||
content = await page.content()
|
||||
logger.info(f"HTML Content Length: {len(content)}")
|
||||
|
||||
# 4. Check for specific text
|
||||
text = await page.inner_text("body")
|
||||
logger.info(f"Page Text (first 500 chars): {text[:500]}")
|
||||
|
||||
if "Challenge" in text or "human" in text or "Cloudflare" in text:
|
||||
logger.warning("⚠️ Cloudflare Challenge detected in text!")
|
||||
|
||||
# 5. Analyze Content for Product Links
|
||||
# Action product URLs contain "/p/"
|
||||
links = await page.evaluate("""
|
||||
Array.from(document.querySelectorAll('a'))
|
||||
.map(a => a.href)
|
||||
.filter(href => href.includes('/p/'))
|
||||
""")
|
||||
|
||||
logger.info(f"Found {len(links)} product links")
|
||||
if links:
|
||||
logger.info(f"First 5 links: {links[:5]}")
|
||||
|
||||
# Find the parent container of the first link
|
||||
parent_html = await page.evaluate("""
|
||||
(() => {
|
||||
const link = document.querySelector("a[href*='/p/']");
|
||||
return link ? link.parentElement.outerHTML : "Not found";
|
||||
})()
|
||||
""")
|
||||
logger.info(f"Parent HTML of first link: {parent_html[:500]}")
|
||||
|
||||
# Find the class of the link itself
|
||||
link_class = await page.evaluate("""
|
||||
(() => {
|
||||
const link = document.querySelector("a[href*='/p/']");
|
||||
return link ? link.className : "Not found";
|
||||
})()
|
||||
""")
|
||||
logger.info(f"Class of first link: {link_class}")
|
||||
|
||||
|
||||
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_action())
|
||||
@@ -1,40 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.ai_price_extractor import AIPriceExtractor
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_ai_extractor():
|
||||
print("Verifying AIPriceExtractor on Gifi dump...")
|
||||
|
||||
# Load the dump file
|
||||
dump_path = "gifi_full.html"
|
||||
if not os.path.exists(dump_path):
|
||||
print(f"Error: {dump_path} not found.")
|
||||
return
|
||||
|
||||
with open(dump_path, "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
title = "Gifi Product Test"
|
||||
|
||||
print("Calling AIPriceExtractor...")
|
||||
price = await AIPriceExtractor.extract_price(html, title)
|
||||
|
||||
print("\n--- AI Extraction Results ---")
|
||||
print(f"Price: {price}€")
|
||||
|
||||
if price is not None:
|
||||
print("\nSUCCESS: AIPriceExtractor worked!")
|
||||
else:
|
||||
print("\nFAILURE: AIPriceExtractor returned None.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_ai_extractor())
|
||||
@@ -1,70 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
from unittest.mock import MagicMock, patch
|
||||
from app.services.ai_service import AIService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Mock BadRequestError since we might not have litellm installed in the environment running this script
|
||||
class MockBadRequestError(Exception):
|
||||
pass
|
||||
|
||||
async def verify_fallback():
|
||||
logger.info("Starting AI fallback verification...")
|
||||
|
||||
# Mock config
|
||||
config = {
|
||||
"provider": "openrouter",
|
||||
"model": "google/gemini-2.5-flash-image-preview",
|
||||
"api_key": "fake-key",
|
||||
"api_base": "https://openrouter.ai/api/v1",
|
||||
"temperature": 0.1,
|
||||
"max_tokens": 100,
|
||||
"timeout": 30,
|
||||
}
|
||||
|
||||
# Mock success response
|
||||
mock_response = MagicMock()
|
||||
mock_response.choices = [MagicMock(message=MagicMock(content='{"price": 10.0}'))]
|
||||
|
||||
# Patch acompletion
|
||||
with patch("app.services.ai_service.acompletion") as mock_acompletion:
|
||||
# Setup side effect: First call raises BadRequestError, second call succeeds
|
||||
mock_acompletion.side_effect = [
|
||||
MockBadRequestError("400 Bad Request: The request is not supported by this model."),
|
||||
mock_response
|
||||
]
|
||||
|
||||
try:
|
||||
# We need to patch the exception check in the code if we can't import the real exception
|
||||
# But the code uses string check "BadRequestError" in str(type(e).__name__)
|
||||
# So MockBadRequestError should work if we name it right or if the code checks "400"
|
||||
|
||||
# Actually, let's just run it and see if our logic catches it.
|
||||
# The code checks: is_bad_request = "BadRequestError" in str(type(e).__name__) or "400" in str(e)
|
||||
# Our MockBadRequestError has "400" in the message, so it should be caught.
|
||||
|
||||
logger.info("Calling call_llm...")
|
||||
result = await AIService.call_llm("test prompt", "data:image/...", config)
|
||||
|
||||
logger.info(f"Result: {result}")
|
||||
|
||||
# Verify calls
|
||||
assert mock_acompletion.call_count == 2
|
||||
logger.info("SUCCESS: acompletion was called twice (retry worked)")
|
||||
|
||||
# Verify second call didn't have response_format
|
||||
call_args = mock_acompletion.call_args_list[1]
|
||||
kwargs = call_args.kwargs
|
||||
if "response_format" not in kwargs:
|
||||
logger.info("SUCCESS: Second call did not have response_format")
|
||||
else:
|
||||
logger.error("FAILURE: Second call still had response_format")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"FAILURE: Exception raised: {e}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_fallback())
|
||||
@@ -1,69 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_site(site_key: str):
|
||||
print(f"\n{'='*50}")
|
||||
print(f"🔍 Verifying {site_key}...")
|
||||
print(f"{'='*50}")
|
||||
|
||||
start_time = time.time()
|
||||
first_result_time = None
|
||||
count = 0
|
||||
|
||||
try:
|
||||
async for result in ImprovedSearchService.search_site_generator(site_key, "chaise"):
|
||||
count += 1
|
||||
if count == 1:
|
||||
first_result_time = time.time()
|
||||
elapsed = first_result_time - start_time
|
||||
print(f"🚀 First result in {elapsed:.2f}s")
|
||||
print(f" Title: {result.title}")
|
||||
print(f" Price: {result.price} {result.currency}")
|
||||
print(f" Image: {result.image_url}")
|
||||
|
||||
# Print a dot for each result to show progress
|
||||
print(".", end="", flush=True)
|
||||
|
||||
total_time = time.time() - start_time
|
||||
print(f"\n✅ Finished {site_key}: {count} results in {total_time:.2f}s")
|
||||
|
||||
if count == 0:
|
||||
print(f"❌ WARNING: 0 results found for {site_key}")
|
||||
return False
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
print(f"\n❌ ERROR verifying {site_key}: {e}")
|
||||
return False
|
||||
|
||||
async def main():
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
sites = list(SITE_CONFIGS.keys())
|
||||
results = {}
|
||||
|
||||
# Test all sites sequentially to avoid overwhelming the browser/network
|
||||
for site in sites:
|
||||
success = await verify_site(site)
|
||||
results[site] = success
|
||||
# Small pause between sites
|
||||
await asyncio.sleep(2)
|
||||
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
print("\n" + "="*50)
|
||||
print("SUMMARY")
|
||||
print("="*50)
|
||||
for site, success in results.items():
|
||||
status = "✅ PASS" if success else "❌ FAIL"
|
||||
print(f"{status} - {site}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,62 +0,0 @@
|
||||
import asyncio
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
async with async_playwright() as p:
|
||||
# Launch with headless=True to mimic server environment
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
|
||||
)
|
||||
page = await context.new_page()
|
||||
|
||||
# Try a product URL that might trigger the check
|
||||
url = "https://www.amazon.fr/dp/B07S58MPKW"
|
||||
print(f"Navigating to {url}")
|
||||
|
||||
try:
|
||||
await page.goto(url)
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
content = await page.content()
|
||||
|
||||
if "Continuer les achats" in content:
|
||||
print("🚨 Popup detected!")
|
||||
with open("amazon_popup.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
await page.screenshot(path="amazon_popup.png")
|
||||
|
||||
# Analyze the button
|
||||
print("Searching for button...")
|
||||
|
||||
# Try various locators
|
||||
locators = [
|
||||
"button",
|
||||
"input[type='submit']",
|
||||
"a.a-button-text",
|
||||
"span.a-button-inner"
|
||||
]
|
||||
|
||||
for sel in locators:
|
||||
elements = page.locator(sel)
|
||||
count = await elements.count()
|
||||
for i in range(count):
|
||||
el = elements.nth(i)
|
||||
if await el.is_visible():
|
||||
txt = await el.inner_text()
|
||||
val = await el.get_attribute("value") or ""
|
||||
if "Continuer" in txt or "Continuer" in val:
|
||||
print(f"✅ Found candidate: {sel}")
|
||||
print(f" Text: {txt}")
|
||||
print(f" Value: {val}")
|
||||
print(f" OuterHTML: {await el.evaluate('el => el.outerHTML')}")
|
||||
else:
|
||||
print("No popup detected. Page title:", await page.title())
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,91 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from playwright.async_api import async_playwright
|
||||
from app.services.ai_price_extractor import AIPriceExtractor
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_bm():
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
page = await browser.new_page()
|
||||
|
||||
# 1. Perform Search to get a product URL
|
||||
logger.info("--- Step 1: Searching for 'Chaise pliante pu creme' on B&M ---")
|
||||
# Use specific search to find the problematic product
|
||||
search_url = "https://bmstores.fr/module/ambjolisearch/jolisearch?s=Chaise+pliante+pu+creme"
|
||||
await page.goto(search_url)
|
||||
await page.wait_for_load_state("networkidle")
|
||||
|
||||
# Take screenshot of search results
|
||||
await page.screenshot(path="bm_search_results.png")
|
||||
logger.info("Screenshot saved: bm_search_results.png")
|
||||
|
||||
# Get first product link
|
||||
product_link = await page.get_attribute("a.thumbnail.product-thumbnail", "href")
|
||||
if not product_link:
|
||||
logger.error("No product found in search")
|
||||
return
|
||||
|
||||
if not product_link.startswith("http"):
|
||||
product_link = "https://www.bmstores.fr" + product_link
|
||||
|
||||
logger.info(f"Testing Product URL: {product_link}")
|
||||
|
||||
# 2. Go to Product Page
|
||||
await page.goto(product_link)
|
||||
await page.wait_for_load_state("networkidle")
|
||||
await page.screenshot(path="bm_product_page.png")
|
||||
logger.info("Screenshot saved: bm_product_page.png")
|
||||
|
||||
# 3. Dump HTML snippet (price area)
|
||||
content = await page.content()
|
||||
logger.info(f"HTML Content Length: {len(content)}")
|
||||
|
||||
# Check for 12.95
|
||||
if "12,95" in content or "12.95" in content:
|
||||
logger.info("✅ Price 12.95 found in raw HTML")
|
||||
else:
|
||||
logger.warning("❌ Price 12.95 NOT found in raw HTML")
|
||||
|
||||
# 4. Test JSON-LD
|
||||
logger.info("\n--- Step 2: Testing JSON-LD ---")
|
||||
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
|
||||
for i, script in enumerate(json_ld_scripts):
|
||||
text = await script.inner_text()
|
||||
logger.info(f"JSON-LD #{i}: {text[:500]}...")
|
||||
|
||||
# 5. Test CSS Selectors
|
||||
logger.info("\n--- Step 3: Testing CSS Selectors ---")
|
||||
selectors = [
|
||||
'.price-current', '.prix-actuel', '.sale-price', '.promo-price',
|
||||
'.price', '[data-testid="price"]', '[itemprop="price"]', '.product-price',
|
||||
'.current-price-value'
|
||||
]
|
||||
for sel in selectors:
|
||||
elements = await page.query_selector_all(sel)
|
||||
for el in elements:
|
||||
text = await el.inner_text()
|
||||
logger.info(f"Selector '{sel}': {text.strip()}")
|
||||
|
||||
# 6. Test AI Extraction
|
||||
logger.info("\n--- Step 4: Testing AI Extraction (Gemma 3) ---")
|
||||
title = await page.title()
|
||||
|
||||
# Ensure API key is available
|
||||
import os
|
||||
if not os.getenv("OPENROUTER_API_KEY"):
|
||||
logger.warning("OPENROUTER_API_KEY not set in env, AI might fail")
|
||||
|
||||
ai_price = await AIPriceExtractor.extract_price(content, title)
|
||||
logger.info(f"AI Extracted Price: {ai_price}")
|
||||
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_bm())
|
||||
@@ -1,294 +0,0 @@
|
||||
"""
|
||||
Verification Script for Bonial Catalog Module
|
||||
|
||||
Tests database setup, scraper functionality, and API endpoints.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from datetime import datetime
|
||||
|
||||
import httpx
|
||||
from sqlalchemy import inspect, text
|
||||
|
||||
from app.database import SessionLocal, engine
|
||||
from app.models import Catalogue, CataloguePage, Enseigne, ScrapingLog
|
||||
from app.services.bonial_scraper import scrape_enseigne
|
||||
from app.services.seed_enseignes import seed_enseignes
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
API_BASE_URL = "http://localhost:8555/api"
|
||||
|
||||
|
||||
def check_database_tables():
|
||||
"""Verify that catalog module tables exist."""
|
||||
logger.info("=" * 60)
|
||||
logger.info("Phase 1: Checking Database Tables")
|
||||
logger.info("=" * 60)
|
||||
|
||||
inspector = inspect(engine)
|
||||
required_tables = [
|
||||
"enseignes",
|
||||
"catalogues",
|
||||
"catalogue_pages",
|
||||
"scraping_logs",
|
||||
]
|
||||
|
||||
existing_tables = inspector.get_table_names()
|
||||
|
||||
all_exist = True
|
||||
for table in required_tables:
|
||||
exists = table in existing_tables
|
||||
status = "✅" if exists else "❌"
|
||||
logger.info(f"{status} Table '{table}': {'EXISTS' if exists else 'MISSING'}")
|
||||
if not exists:
|
||||
all_exist = False
|
||||
|
||||
if all_exist:
|
||||
logger.info("\n✅ All required tables exist!\n")
|
||||
return True
|
||||
else:
|
||||
logger.error("\n❌ Some tables are missing. Run migration: alembic upgrade head\n")
|
||||
return False
|
||||
|
||||
|
||||
def check_enseignes_seeding():
|
||||
"""Verify that enseignes are seeded."""
|
||||
logger.info("=" * 60)
|
||||
logger.info("Phase 2: Checking Enseignes Seeding")
|
||||
logger.info("=" * 60)
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
count = db.query(Enseigne).count()
|
||||
logger.info(f"Found {count} enseignes in database")
|
||||
|
||||
if count == 0:
|
||||
logger.info("Seeding enseignes...")
|
||||
created = seed_enseignes(db)
|
||||
logger.info(f"✅ Created {created} enseignes")
|
||||
count = created
|
||||
|
||||
if count >= 9:
|
||||
logger.info("\n✅ All 9 enseignes are seeded!\n")
|
||||
|
||||
# Display enseignes
|
||||
enseignes = db.query(Enseigne).order_by(Enseigne.ordre_affichage).all()
|
||||
for ens in enseignes:
|
||||
active = "✅" if ens.is_active else "⚠️"
|
||||
logger.info(f" {active} {ens.ordre_affichage}. {ens.nom} (slug: {ens.slug_bonial})")
|
||||
|
||||
return True
|
||||
else:
|
||||
logger.warning(f"\n⚠️ Expected 9 enseignes, found {count}\n")
|
||||
return False
|
||||
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
async def test_scraper():
|
||||
"""Test the Bonial scraper on one enseigne."""
|
||||
logger.info("=" * 60)
|
||||
logger.info("Phase 3: Testing Bonial Scraper (Gifi)")
|
||||
logger.info("=" * 60)
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
# Get Gifi enseigne
|
||||
gifi = db.query(Enseigne).filter_by(slug_bonial="Gifi").first()
|
||||
|
||||
if not gifi:
|
||||
logger.error("❌ Gifi enseigne not found")
|
||||
return False
|
||||
|
||||
logger.info(f"Testing scraper for: {gifi.nom}")
|
||||
logger.info("This may take 30-60 seconds...")
|
||||
|
||||
# Run scraper
|
||||
log = await scrape_enseigne(gifi, db)
|
||||
|
||||
# Display results
|
||||
logger.info(f"\nScraping completed:")
|
||||
logger.info(f" Status: {log.statut}")
|
||||
logger.info(f" Catalogues found: {log.catalogues_trouves}")
|
||||
logger.info(f" New catalogues: {log.catalogues_nouveaux}")
|
||||
logger.info(f" Duration: {log.duree_secondes:.2f}s")
|
||||
|
||||
if log.message_erreur:
|
||||
logger.warning(f" Error: {log.message_erreur}")
|
||||
|
||||
if log.statut in ["success", "partial"] and log.catalogues_trouves > 0:
|
||||
logger.info("\n✅ Scraper is working!\n")
|
||||
|
||||
# Display sample catalog
|
||||
cat = db.query(Catalogue).filter_by(enseigne_id=gifi.id).first()
|
||||
if cat:
|
||||
logger.info(f"Sample catalog:")
|
||||
logger.info(f" Titre: {cat.titre}")
|
||||
logger.info(f" Dates: {cat.date_debut.date()} → {cat.date_fin.date()}")
|
||||
logger.info(f" Pages: {cat.nombre_pages}")
|
||||
|
||||
return True
|
||||
else:
|
||||
logger.error("\n❌ Scraper failed or found no catalogs\n")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"❌ Error testing scraper: {e}")
|
||||
return False
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
async def test_api_endpoints():
|
||||
"""Test API endpoints."""
|
||||
logger.info("=" * 60)
|
||||
logger.info("Phase 4: Testing API Endpoints")
|
||||
logger.info("=" * 60)
|
||||
|
||||
async with httpx.AsyncClient(timeout=30.0) as client:
|
||||
# Test 1: Get enseignes
|
||||
logger.info("\n1. Testing GET /api/catalogues/enseignes")
|
||||
try:
|
||||
response = await client.get(f"{API_BASE_URL}/catalogues/enseignes")
|
||||
if response.status_code == 200:
|
||||
enseignes = response.json()
|
||||
logger.info(f" ✅ Status 200 - Found {len(enseignes)} enseignes")
|
||||
if enseignes:
|
||||
logger.info(f" Sample: {enseignes[0]['nom']} ({enseignes[0]['catalogues_actifs_count']} catalogues)")
|
||||
else:
|
||||
logger.error(f" ❌ Status {response.status_code}")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f" ❌ Error: {e}")
|
||||
return False
|
||||
|
||||
# Test 2: Get catalogues
|
||||
logger.info("\n2. Testing GET /api/catalogues")
|
||||
try:
|
||||
response = await client.get(f"{API_BASE_URL}/catalogues?page=1&limit=5")
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
catalogues = data.get("data", [])
|
||||
pagination = data.get("pagination", {})
|
||||
logger.info(f" ✅ Status 200 - Found {pagination.get('total', 0)} catalogues")
|
||||
logger.info(f" Page: {pagination.get('page')}/{pagination.get('pages_total')}")
|
||||
if catalogues:
|
||||
logger.info(f" Sample: {catalogues[0]['titre']}")
|
||||
else:
|
||||
logger.error(f" ❌ Status {response.status_code}")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f" ❌ Error: {e}")
|
||||
return False
|
||||
|
||||
# Test 3: Get catalogue detail
|
||||
logger.info("\n3. Testing GET /api/catalogues/{id}")
|
||||
try:
|
||||
# Get first catalog ID
|
||||
db = SessionLocal()
|
||||
cat = db.query(Catalogue).first()
|
||||
db.close()
|
||||
|
||||
if cat:
|
||||
response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}")
|
||||
if response.status_code == 200:
|
||||
detail = response.json()
|
||||
logger.info(f" ✅ Status 200 - Catalogue: {detail['titre']}")
|
||||
logger.info(f" Pages: {detail['nombre_pages']}")
|
||||
else:
|
||||
logger.error(f" ❌ Status {response.status_code}")
|
||||
return False
|
||||
else:
|
||||
logger.warning(" ⚠️ No catalogues in DB to test detail endpoint")
|
||||
except Exception as e:
|
||||
logger.error(f" ❌ Error: {e}")
|
||||
return False
|
||||
|
||||
# Test 4: Get catalogue pages
|
||||
logger.info("\n4. Testing GET /api/catalogues/{id}/pages")
|
||||
try:
|
||||
if cat:
|
||||
response = await client.get(f"{API_BASE_URL}/catalogues/{cat.id}/pages")
|
||||
if response.status_code == 200:
|
||||
pages = response.json()
|
||||
logger.info(f" ✅ Status 200 - Found {len(pages)} pages")
|
||||
if pages:
|
||||
logger.info(f" Sample page: {pages[0]['numero_page']} - {pages[0]['image_url'][:50]}...")
|
||||
else:
|
||||
logger.error(f" ❌ Status {response.status_code}")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f" ❌ Error: {e}")
|
||||
return False
|
||||
|
||||
# Test 5: Get stats (requires working DB)
|
||||
logger.info("\n5. Testing GET /api/catalogues/admin/stats")
|
||||
try:
|
||||
response = await client.get(f"{API_BASE_URL}/catalogues/admin/stats")
|
||||
if response.status_code == 200:
|
||||
stats = response.json()
|
||||
logger.info(f" ✅ Status 200 - Total catalogues: {stats['total_catalogues']}")
|
||||
logger.info(f" Prochaine exécution: {stats['prochaine_execution']}")
|
||||
else:
|
||||
logger.error(f" ❌ Status {response.status_code}")
|
||||
# Stats is optional, don't fail
|
||||
except Exception as e:
|
||||
logger.warning(f" ⚠️ Stats endpoint error (may require auth): {e}")
|
||||
|
||||
logger.info("\n✅ All API endpoints are working!\n")
|
||||
return True
|
||||
|
||||
|
||||
async def main():
|
||||
"""Run all verification tests."""
|
||||
logger.info("\n" + "=" * 60)
|
||||
logger.info("BONIAL CATALOG MODULE - VERIFICATION SCRIPT")
|
||||
logger.info("=" * 60 + "\n")
|
||||
|
||||
results = []
|
||||
|
||||
# Phase 1: Database tables
|
||||
results.append(("Database Tables", check_database_tables()))
|
||||
|
||||
if not results[0][1]:
|
||||
logger.error("\n❌ Database not ready. Please run: alembic upgrade head")
|
||||
sys.exit(1)
|
||||
|
||||
# Phase 2: Enseignes seeding
|
||||
results.append(("Enseignes Seeding", check_enseignes_seeding()))
|
||||
|
||||
# Phase 3: Scraper test (optional, can be slow)
|
||||
scraper_test = input("\nRun scraper test? (Gifi - takes ~60s) [y/N]: ").lower() == "y"
|
||||
if scraper_test:
|
||||
results.append(("Scraper Test", await test_scraper()))
|
||||
|
||||
# Phase 4: API endpoints (requires app to be running)
|
||||
api_test = input("\nTest API endpoints? (App must be running on :8555) [y/N]: ").lower() == "y"
|
||||
if api_test:
|
||||
results.append(("API Endpoints", await test_api_endpoints()))
|
||||
|
||||
# Summary
|
||||
logger.info("\n" + "=" * 60)
|
||||
logger.info("VERIFICATION SUMMARY")
|
||||
logger.info("=" * 60)
|
||||
|
||||
for name, passed in results:
|
||||
status = "✅ PASSED" if passed else "❌ FAILED"
|
||||
logger.info(f"{status}: {name}")
|
||||
|
||||
all_passed = all(result[1] for result in results)
|
||||
|
||||
if all_passed:
|
||||
logger.info("\n✅ ALL TESTS PASSED - Bonial module is ready!")
|
||||
else:
|
||||
logger.error("\n❌ SOME TESTS FAILED - Please review errors above")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,138 +0,0 @@
|
||||
"""
|
||||
Script de vérification pour tester Crawl4AI dans le scraper Tiendeo.
|
||||
|
||||
Usage:
|
||||
python verify_crawl4ai_scraper.py
|
||||
|
||||
Ce script teste:
|
||||
1. Import de Crawl4AI
|
||||
2. Extraction d'une page catalogue Tiendeo
|
||||
3. Comptage des pages trouvées
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
async def test_crawl4ai_import():
|
||||
"""Test 1: Vérifier que Crawl4AI est bien installé"""
|
||||
try:
|
||||
logger.info("✓ Crawl4AI importé avec succès")
|
||||
logger.info(f" Version: {AsyncWebCrawler.__module__}")
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error(f"✗ Erreur import Crawl4AI: {e}")
|
||||
return False
|
||||
|
||||
|
||||
async def test_catalog_page_extraction():
|
||||
"""Test 2: Extraire les pages d'un catalogue Tiendeo"""
|
||||
# URL de test - catalogue Gifi Nancy (à adapter si nécessaire)
|
||||
test_url = "https://www.tiendeo.fr/Catalogues/nancy/gifi"
|
||||
|
||||
logger.info(f"\nTest extraction depuis: {test_url}")
|
||||
|
||||
try:
|
||||
browser_config = BrowserConfig(
|
||||
headless=True,
|
||||
verbose=False,
|
||||
extra_args=["--disable-gpu", "--no-sandbox", "--disable-dev-shm-usage"],
|
||||
)
|
||||
|
||||
config = CrawlerRunConfig(
|
||||
cache_mode=CacheMode.BYPASS,
|
||||
wait_for_images=True,
|
||||
process_iframes=True,
|
||||
remove_overlay_elements=True,
|
||||
wait_until="networkidle",
|
||||
delay_before_return_html=3.0,
|
||||
)
|
||||
|
||||
async with AsyncWebCrawler(config=browser_config) as crawler:
|
||||
result = await crawler.arun(url=test_url, config=config)
|
||||
|
||||
if not result.success:
|
||||
logger.error(f"✗ Échec du crawling: {result.error_message}")
|
||||
return False
|
||||
|
||||
logger.info(f"✓ Page chargée avec succès")
|
||||
logger.info(f" HTML length: {len(result.html)} chars")
|
||||
|
||||
# Tester l'extraction JavaScript
|
||||
pages_data = await crawler.crawler_strategy.execute_js(
|
||||
"""
|
||||
() => {
|
||||
const results = [];
|
||||
const seenUrls = new Set();
|
||||
|
||||
const allImages = document.querySelectorAll('img');
|
||||
|
||||
allImages.forEach((img) => {
|
||||
let src = img.src || img.getAttribute('data-src');
|
||||
|
||||
if (!src && img.srcset) {
|
||||
const srcsetParts = img.srcset.split(',')[0].trim().split(' ');
|
||||
src = srcsetParts[0];
|
||||
}
|
||||
|
||||
if (!src || seenUrls.has(src)) return;
|
||||
|
||||
if (src.includes('logo') || src.includes('icon') || src.includes('avatar')) {
|
||||
return;
|
||||
}
|
||||
|
||||
const width = img.naturalWidth || img.width;
|
||||
const height = img.naturalHeight || img.height;
|
||||
|
||||
if (width < 400 || height < 400) return;
|
||||
|
||||
const aspectRatio = width / height;
|
||||
|
||||
if (aspectRatio > 0.5 && aspectRatio < 0.9) {
|
||||
seenUrls.add(src);
|
||||
results.push({
|
||||
image_url: src,
|
||||
width: width,
|
||||
height: height,
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
results.sort((a, b) => (b.width * b.height) - (a.width * a.height));
|
||||
|
||||
return results.map((item, index) => ({
|
||||
...item,
|
||||
numero_page: index + 1,
|
||||
}));
|
||||
}
|
||||
"""
|
||||
)
|
||||
|
||||
logger.info(f"✓ Extraction JavaScript réussie")
|
||||
logger.info(f" Pages trouvées: {len(pages_data)}")
|
||||
|
||||
if len(pages_data) > 0:
|
||||
test2 = await test_catalog_page_extraction()
|
||||
|
||||
# Résumé
|
||||
logger.info("\n" + "=" * 70)
|
||||
logger.info("RÉSUMÉ")
|
||||
logger.info("=" * 70)
|
||||
logger.info(f"Import Crawl4AI: {'✓ OK' if test1 else '✗ ÉCHEC'}")
|
||||
logger.info(f"Extraction pages: {'✓ OK' if test2 else '✗ ÉCHEC'}")
|
||||
|
||||
if test1 and test2:
|
||||
logger.info("\n✓ TOUS LES TESTS SONT PASSÉS!")
|
||||
logger.info("Le scraper Crawl4AI est prêt à être utilisé.")
|
||||
else:
|
||||
logger.info("\n✗ CERTAINS TESTS ONT ÉCHOUÉ")
|
||||
logger.info("Vérifiez les erreurs ci-dessus.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,182 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Mock sqlalchemy
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
|
||||
# Mock pydantic
|
||||
mock_pydantic = MagicMock()
|
||||
# Mock BaseModel
|
||||
class MockBaseModel:
|
||||
def __init__(self, **kwargs):
|
||||
for k, v in kwargs.items():
|
||||
setattr(self, k, v)
|
||||
mock_pydantic.BaseModel = MockBaseModel
|
||||
mock_pydantic.Field = MagicMock(return_value=None)
|
||||
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
|
||||
|
||||
sys.modules["pydantic"] = mock_pydantic
|
||||
|
||||
# Mock app.utils.text which is imported by ai_schema
|
||||
mock_utils_text = MagicMock()
|
||||
sys.modules["app.utils.text"] = mock_utils_text
|
||||
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
|
||||
mock_utils_text.clean_text = lambda text: text.strip()
|
||||
|
||||
# Mock app.strings (if used) or other utils
|
||||
sys.modules["app.utils"] = MagicMock()
|
||||
|
||||
# Mock app.utils.image
|
||||
sys.modules["app.utils.image"] = MagicMock()
|
||||
|
||||
# Mock app.database
|
||||
sys.modules["app.database"] = MagicMock()
|
||||
|
||||
# Mock playwright
|
||||
mock_playwright = MagicMock()
|
||||
sys.modules["playwright"] = mock_playwright
|
||||
sys.modules["playwright.async_api"] = mock_playwright
|
||||
|
||||
# Mock generic types for type hints if needed
|
||||
mock_playwright.Browser = MagicMock
|
||||
mock_playwright.BrowserContext = MagicMock
|
||||
mock_playwright.Page = MagicMock
|
||||
mock_playwright.TimeoutError = Exception
|
||||
|
||||
# Now import the schema
|
||||
from app.ai_schema import get_extraction_prompt, get_repair_prompt
|
||||
|
||||
# We can't import ScraperService easily if it inherits from things or uses decorators
|
||||
# But for this test we only need get_extraction_prompt which is in ai_schema
|
||||
# So we can skip importing ScraperService if it causes issues,
|
||||
# BUT we wanted to verify ScraperService text cleaning logic...
|
||||
# Let's mock ScraperService dependencies completely.
|
||||
|
||||
try:
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
except ImportError:
|
||||
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
|
||||
ScraperService = None
|
||||
|
||||
# Mock litellm and tenacity
|
||||
sys.modules["litellm"] = MagicMock()
|
||||
sys.modules["tenacity"] = MagicMock()
|
||||
mock_retry = MagicMock()
|
||||
sys.modules["tenacity.retry"] = mock_retry
|
||||
|
||||
# Make sure imports inside ai_service don't fail
|
||||
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
|
||||
# We need to mock these specifically if the module imports them directly
|
||||
mock_tenacity = MagicMock()
|
||||
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
|
||||
mock_tenacity.retry_if_exception_type = MagicMock()
|
||||
mock_tenacity.stop_after_attempt = MagicMock()
|
||||
mock_tenacity.wait_exponential = MagicMock()
|
||||
sys.modules["tenacity"] = mock_tenacity
|
||||
|
||||
# Now import AIService
|
||||
# We will mock the AI response to verify the parsing logic
|
||||
from app.services.ai_service import AIService
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_extraction_logic():
|
||||
print("Verifying Extraction Logic...")
|
||||
|
||||
# 1. Test Text Cleaning in ScraperService
|
||||
# We can't mock Playwright page easily in a simple script without launching a browser.
|
||||
# But we can test the AI prompt generation which is critical.
|
||||
|
||||
# Simulate B&M text
|
||||
dirty_text = """
|
||||
Menu
|
||||
Accueil
|
||||
Panier
|
||||
|
||||
Boisson energisante ice 25cl
|
||||
Red Bull
|
||||
|
||||
1.15 €
|
||||
Prix au litre : 4,60 € / L
|
||||
|
||||
En stock
|
||||
Ajouter au panier
|
||||
|
||||
Footer
|
||||
Mentions légales
|
||||
"""
|
||||
|
||||
print("\n--- Testing Prompt Generation ---")
|
||||
prompt = get_extraction_prompt(dirty_text)
|
||||
|
||||
# Verify strict instructions are present
|
||||
checks = [
|
||||
"CRITICAL",
|
||||
"Ignore \"Prix au litre\"",
|
||||
"B&M STORES Specific",
|
||||
"Extract as DECIMAL NUMBER",
|
||||
"ALWAYS select the TTC price",
|
||||
"Ignore \"HT\""
|
||||
]
|
||||
|
||||
all_passed = True
|
||||
for check in checks:
|
||||
if check in prompt:
|
||||
print(f"[OK] Prompt contains: {check}")
|
||||
else:
|
||||
print(f"[FAIL] Prompt missing: {check}")
|
||||
all_passed = False
|
||||
|
||||
if not all_passed:
|
||||
print("Prompt verification failed!")
|
||||
exit(1)
|
||||
|
||||
print("\n--- Testing Response Parsing (Mock AI) ---")
|
||||
|
||||
# Case 1: AI returns Main Price correctly
|
||||
mock_response_1 = """
|
||||
```json
|
||||
{
|
||||
"price": 1.15,
|
||||
"currency": "EUR",
|
||||
"in_stock": true,
|
||||
"price_confidence": 0.95,
|
||||
"in_stock_confidence": 1.0,
|
||||
"source_type": "text"
|
||||
}
|
||||
```
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_1)
|
||||
if result.price == 1.15 and result.in_stock is True:
|
||||
print("[OK] Parsed correct mocked response.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse correct response: {result}")
|
||||
exit(1)
|
||||
|
||||
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
|
||||
# If AI returns explicit null because it's confused
|
||||
mock_response_2 = """
|
||||
{
|
||||
"price": null,
|
||||
"currency": "EUR",
|
||||
"in_stock": null,
|
||||
"price_confidence": 0.0,
|
||||
"in_stock_confidence": 0.0,
|
||||
"source_type": "image"
|
||||
}
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_2)
|
||||
if result.price is None:
|
||||
print("[OK] Parsed null response correctly.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse null response.")
|
||||
|
||||
print("\nVerification of Logic Flow Complete (Simulated).")
|
||||
print("Real-world verification requires running the full scraper.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_extraction_logic())
|
||||
@@ -1,48 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Add current directory to sys.path to allow importing app
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
# Mock playwright before importing app
|
||||
mock_playwright = MagicMock()
|
||||
sys.modules["playwright"] = mock_playwright
|
||||
sys.modules["playwright.async_api"] = mock_playwright
|
||||
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_logic():
|
||||
logger.info("Starting logic verification...")
|
||||
|
||||
# Mock Page object
|
||||
mock_page = MagicMock()
|
||||
mock_page.screenshot = AsyncMock()
|
||||
|
||||
item_id = 123
|
||||
url = "http://test.com"
|
||||
|
||||
# Call _take_screenshot directly
|
||||
logger.info("Calling _take_screenshot...")
|
||||
filename = await ScraperService._take_screenshot(mock_page, url, item_id)
|
||||
|
||||
logger.info(f"Returned filename: {filename}")
|
||||
|
||||
# Verify format
|
||||
pattern = r"screenshots/item_123_\d+\.png"
|
||||
if re.match(pattern, filename):
|
||||
logger.info("SUCCESS: Filename matches expected timestamp pattern!")
|
||||
print("VERIFICATION_SUCCESS")
|
||||
else:
|
||||
logger.error(f"FAILURE: Filename {filename} does not match pattern {pattern}")
|
||||
exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_logic())
|
||||
@@ -1,164 +0,0 @@
|
||||
import os
|
||||
import glob
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
# --- MOCKS SETUP ---
|
||||
# We need to mock these BEFORE importing app.services.item_service
|
||||
# to avoid ImportErrors due to missing dependencies in the test env.
|
||||
|
||||
# 1. Mock External Libs
|
||||
sys.modules["fastapi"] = MagicMock()
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
|
||||
# 2. Mock Internal App Modules that have heavy dependencies
|
||||
# Mock app.database
|
||||
mock_database = MagicMock()
|
||||
sys.modules["app.database"] = mock_database
|
||||
|
||||
# Mock app.models
|
||||
# We need models.Item and models.PriceHistory to be accessible attributes
|
||||
mock_models = MagicMock()
|
||||
sys.modules["app.models"] = mock_models
|
||||
|
||||
# Mock app.schemas
|
||||
sys.modules["app.schemas"] = MagicMock()
|
||||
|
||||
# Mock app.services.settings_service
|
||||
sys.modules["app.services.settings_service"] = MagicMock()
|
||||
|
||||
# Mock app.url_validation
|
||||
sys.modules["app.url_validation"] = MagicMock()
|
||||
|
||||
# --- IMPORT TARGET ---
|
||||
from app.services.item_service import ItemService
|
||||
|
||||
def verify_item_service_fix():
|
||||
print("Starting verification of ItemService fix...")
|
||||
|
||||
# 1. Setup Mock DB and Item
|
||||
# We must ensure that when ItemService does `item.id`, it works.
|
||||
mock_db = MagicMock()
|
||||
|
||||
# Create a simple class to act as the Item model instance
|
||||
class MockItem:
|
||||
def __init__(self, id, name):
|
||||
self.id = id
|
||||
self.name = name
|
||||
self.url = "http://test.com"
|
||||
self.current_price = 10.0
|
||||
self.in_stock = True
|
||||
self.screenshot_url = None # This will be set by the service
|
||||
|
||||
# Attributes accessed by the service
|
||||
self.notification_channel = None
|
||||
self.target_price = None
|
||||
self.current_price_confidence = 1.0
|
||||
self.in_stock_confidence = 1.0
|
||||
self.is_active = True
|
||||
self.last_checked = None
|
||||
self.is_refreshing = False
|
||||
self.last_error = None
|
||||
self.category = None
|
||||
self.tags = None
|
||||
self.description = None
|
||||
|
||||
# __dict__ is used by the service to create the result
|
||||
self.dict_storage = {k:v for k,v in self.__dict__.items()}
|
||||
|
||||
@property
|
||||
def __dict__(self):
|
||||
# Update dict storage with current attributes
|
||||
return {
|
||||
"id": self.id,
|
||||
"name": self.name,
|
||||
"url": self.url
|
||||
}
|
||||
|
||||
item_888 = MockItem(888, "Test Item")
|
||||
|
||||
# ItemService.get_items calls db.query(models.Item).all()
|
||||
# We need to make sure models.Item is used in the query.
|
||||
# The service does: items = db.query(models.Item).all()
|
||||
|
||||
mock_db.query.return_value.all.return_value = [item_888]
|
||||
|
||||
# It also queries PriceHistory
|
||||
# db.query(models.PriceHistory).filter(...).first()
|
||||
# Let's mock that to return None to force filesytem check (or check logic priority)
|
||||
mock_db.query.return_value.filter.return_value.filter.return_value.order_by.return_value.first.return_value = None
|
||||
|
||||
# 2. Create Dummy Screenshot Files
|
||||
os.makedirs("screenshots", exist_ok=True)
|
||||
|
||||
# Clean up
|
||||
for f in glob.glob("screenshots/item_888_*.png"):
|
||||
os.remove(f)
|
||||
if os.path.exists("screenshots/item_888.png"):
|
||||
os.remove("screenshots/item_888.png")
|
||||
|
||||
# Scenario:
|
||||
# 1. item_888.png exists (legacy)
|
||||
# 2. item_888_1000.png exists (old timestamp)
|
||||
# 3. item_888_2000.png exists (new timestamp)
|
||||
|
||||
# Expected: get_items should pick item_888_2000.png
|
||||
|
||||
file_legacy = "screenshots/item_888.png"
|
||||
file_old = "screenshots/item_888_1000.png"
|
||||
file_new = "screenshots/item_888_2000.png"
|
||||
|
||||
with open(file_legacy, "w") as f: f.write(".")
|
||||
with open(file_old, "w") as f: f.write(".")
|
||||
with open(file_new, "w") as f: f.write(".")
|
||||
|
||||
print(f"Created files: {file_legacy}, {file_old}, {file_new}")
|
||||
|
||||
try:
|
||||
# 3. Test get_items
|
||||
print("Testing get_items()...")
|
||||
items = ItemService.get_items(mock_db)
|
||||
|
||||
if not items:
|
||||
print("FAILURE: No items returned")
|
||||
exit(1)
|
||||
|
||||
result = items[0]
|
||||
screenshot_url = result.get("screenshot_url")
|
||||
print(f"Returned screenshot_url: {screenshot_url}")
|
||||
|
||||
expected_url = f"/screenshots/{os.path.basename(file_new)}"
|
||||
|
||||
if screenshot_url == expected_url:
|
||||
print("SUCCESS: Correctly identified the latest screenshot!")
|
||||
else:
|
||||
print(f"FAILURE: Expected {expected_url}, got {screenshot_url}")
|
||||
# If it failed, maybe it picked legacy?
|
||||
if screenshot_url == f"/screenshots/{os.path.basename(file_legacy)}":
|
||||
print("Picked legacy file instead of timestamped one.")
|
||||
exit(1)
|
||||
|
||||
# 4. Test delete_item
|
||||
print("Testing delete_item()...")
|
||||
# Ensure the query returns our item
|
||||
mock_db.query.return_value.filter.return_value.first.return_value = item_888
|
||||
|
||||
ItemService.delete_item(mock_db, 888)
|
||||
|
||||
# Check files
|
||||
remaining = glob.glob("screenshots/item_888*.png")
|
||||
if not remaining:
|
||||
print("SUCCESS: All screenshots deleted.")
|
||||
else:
|
||||
print(f"FAILURE: Files remaining: {remaining}")
|
||||
exit(1)
|
||||
|
||||
finally:
|
||||
# Cleanup
|
||||
for f in [file_legacy, file_old, file_new]:
|
||||
if os.path.exists(f):
|
||||
os.remove(f)
|
||||
|
||||
if __name__ == "__main__":
|
||||
verify_item_service_fix()
|
||||
@@ -1,43 +0,0 @@
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.parsers.gifi_parser import GifiParser
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
def verify_fix():
|
||||
print("Verifying Gifi parser fix...")
|
||||
|
||||
# Load the dump file (we know it exists from previous steps)
|
||||
dump_path = "gifi_full.html"
|
||||
if not os.path.exists(dump_path):
|
||||
print(f"Error: {dump_path} not found.")
|
||||
return
|
||||
|
||||
with open(dump_path, "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
parser = GifiParser()
|
||||
# Dummy URL
|
||||
url = "https://www.gifi.fr/test-product.html"
|
||||
|
||||
print("Parsing product details...")
|
||||
details = parser.parse_product_details(html, url)
|
||||
|
||||
print("\n--- Extraction Results ---")
|
||||
print(f"Bypass Price: {details.get('price')}")
|
||||
print(f"Bypass Stock: {details.get('in_stock')}")
|
||||
print(f"Currency: {details.get('currency')}")
|
||||
|
||||
if details.get('price') is not None:
|
||||
print("\nSUCCESS: Price extracted successfully!")
|
||||
else:
|
||||
print("\nFAILURE: Price not found in dump.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
verify_fix()
|
||||
@@ -1,50 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_leclerc():
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
page = await browser.new_page()
|
||||
|
||||
# 1. Perform Search
|
||||
logger.info("--- Step 1: Searching for 'Chaise' on E.Leclerc ---")
|
||||
search_url = "https://www.e.leclerc/recherche?q=Chaise"
|
||||
try:
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
|
||||
await page.wait_for_timeout(5000) # Wait for JS
|
||||
except Exception as e:
|
||||
logger.error(f"Navigation failed: {e}")
|
||||
|
||||
# 2. Dump HTML
|
||||
content = await page.content()
|
||||
logger.info(f"HTML Content Length: {len(content)}")
|
||||
|
||||
# 3. Analyze Classes
|
||||
classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')")
|
||||
logger.info(f"Classes found: {classes[:1000]}")
|
||||
|
||||
# 4. Check for Product Selectors
|
||||
selectors = [
|
||||
"div[class*='product']",
|
||||
"article",
|
||||
".product-card",
|
||||
".c-product-card",
|
||||
"a[class*='product']"
|
||||
]
|
||||
|
||||
for sel in selectors:
|
||||
count = await page.locator(sel).count()
|
||||
if count > 0:
|
||||
logger.info(f"Selector '{sel}' found {count} elements")
|
||||
first_html = await page.locator(sel).first.evaluate("el => el.outerHTML")
|
||||
logger.info(f"First element HTML ({sel}): {first_html[:500]}...")
|
||||
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_leclerc())
|
||||
@@ -1,74 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
|
||||
from app.services.search_service import new_search_service
|
||||
from app.services.browserless_service import browserless_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.DEBUG, # Enable DEBUG logging
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler()]
|
||||
)
|
||||
# Set other loggers to INFO to avoid noise
|
||||
logging.getLogger("urllib3").setLevel(logging.INFO)
|
||||
logging.getLogger("asyncio").setLevel(logging.INFO)
|
||||
logging.getLogger("websockets").setLevel(logging.INFO)
|
||||
|
||||
async def verify_search():
|
||||
query = "chaise" # Updated query
|
||||
print(f"--- Starting Verification Search for '{query}' ---")
|
||||
|
||||
# 1. Test Browserless Connection
|
||||
print("\n[1] Testing Browserless Connection...")
|
||||
try:
|
||||
await browserless_service.initialize()
|
||||
print("✅ Browserless connected successfully")
|
||||
except Exception as e:
|
||||
print(f"❌ Browserless connection failed: {e}")
|
||||
return
|
||||
|
||||
# 2. Test Specific Sites
|
||||
# sites_to_test = ["gifi.fr", "lincroyable.fr", "stokomani.fr"] # Excluded amazon.fr
|
||||
sites_to_test = ["stokomani.fr"] # Focus on Stokomani for now as requested/implied context
|
||||
|
||||
for site in sites_to_test:
|
||||
print(f"\n[2] Testing Search on {site}...")
|
||||
try:
|
||||
count = 0
|
||||
async for r in new_search_service.search_site_generator(site, query):
|
||||
count += 1
|
||||
print(f"✅ Found result: {r.title} ({r.url})")
|
||||
print(f" Price: {r.price} {r.currency}")
|
||||
print(f" Image: {r.image_url}")
|
||||
if count >= 1:
|
||||
break
|
||||
if count == 0:
|
||||
print(f"⚠️ No results found for {site}")
|
||||
# Dump HTML for debugging
|
||||
try:
|
||||
content, _ = await browserless_service.get_page_content(
|
||||
SITE_CONFIGS[site]["search_url"].format(query=query),
|
||||
use_proxy=SITE_CONFIGS[site].get("requires_proxy", False),
|
||||
wait_selector=SITE_CONFIGS[site].get("wait_selector")
|
||||
)
|
||||
with open(f"/app/debug_dumps/{site}_failed_verification.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
print(f"📄 Saved HTML dump to /app/debug_dumps/{site}_failed_verification.html")
|
||||
except Exception as dump_e:
|
||||
print(f"❌ Failed to save HTML dump: {dump_e}")
|
||||
except Exception as e:
|
||||
print(f"❌ Error searching {site}: {e}")
|
||||
|
||||
# 3. Cleanup
|
||||
await browserless_service.shutdown()
|
||||
print("\n--- Verification Complete ---")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_search())
|
||||
@@ -1,33 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from app.services.browserless_service import browserless_service
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_popups():
|
||||
# URL that was reported to have issues
|
||||
url = "https://www.bmstores.fr/363943-bougie-parfumee-avec-bijou-350g-senteurs-assorties"
|
||||
# Fallback to search if that product is gone
|
||||
fallback_url = "https://www.bmstores.fr/module/ambjolisearch/jolisearch?s=calendrier"
|
||||
|
||||
logger.info(f"--- Testing Popup Handling on {url} ---")
|
||||
|
||||
try:
|
||||
content, screenshot_path = await browserless_service.get_page_content(
|
||||
url,
|
||||
extract_text=False
|
||||
)
|
||||
|
||||
logger.info(f"Screenshot saved to: {screenshot_path}")
|
||||
logger.info("Please inspect the screenshot to ensure no 'Stock Inconnu' or 'Calendrier' popups are visible.")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Verification failed: {e}")
|
||||
finally:
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_popups())
|
||||
@@ -1,48 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
from playwright.async_api import async_playwright
|
||||
from app.services.scraper_service import ScraperService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_fix():
|
||||
logger.info("Starting verification...")
|
||||
|
||||
# Mock browserless URL if not set
|
||||
if not os.getenv("BROWSERLESS_URL"):
|
||||
os.environ["BROWSERLESS_URL"] = "ws://browserless:3000"
|
||||
|
||||
playwright = await async_playwright().start()
|
||||
try:
|
||||
# Connect to browserless (or launch local if not available, but code expects connect)
|
||||
# For this test, we might need to mock the browser object if we can't actually connect
|
||||
# But let's try to just create a local browser for testing purposes if connect fails
|
||||
# Actually, the code expects a browser object.
|
||||
|
||||
logger.info("Launching local browser for test...")
|
||||
browser = await playwright.chromium.launch()
|
||||
|
||||
logger.info("Calling scrape_item with browser argument...")
|
||||
try:
|
||||
# We pass a dummy URL, we expect it might fail scraping but NOT raise TypeError
|
||||
await ScraperService.scrape_item(
|
||||
url="https://example.com",
|
||||
browser=browser,
|
||||
timeout=5000 # Short timeout
|
||||
)
|
||||
logger.info("SUCCESS: scrape_item accepted the browser argument!")
|
||||
except TypeError as e:
|
||||
logger.error(f"FAILURE: TypeError raised: {e}")
|
||||
except Exception as e:
|
||||
logger.info(f"Scraping failed as expected (network/etc), but argument was accepted: {e}")
|
||||
|
||||
await browser.close()
|
||||
|
||||
finally:
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_fix())
|
||||
@@ -1,42 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.direct_search_service import direct_search_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_site(site_key):
|
||||
logger.info(f"Verifying {site_key}...")
|
||||
try:
|
||||
# direct_search_service.search_site might need to be called differently if it's an instance method
|
||||
# checking previous usage or assuming standard service pattern
|
||||
results = await direct_search_service.search_site(site_key, "chaise")
|
||||
if results:
|
||||
logger.info(f"✅ {site_key}: Found {len(results)} results")
|
||||
for i, res in enumerate(results[:3]):
|
||||
title = res.get('title', 'No Title')
|
||||
price = res.get('price', 'No Price')
|
||||
url = res.get('url', 'No URL')
|
||||
logger.info(f" {i+1}. {title[:50]}... - {price} - {url[:50]}...")
|
||||
else:
|
||||
logger.error(f"❌ {site_key}: No results found")
|
||||
except Exception as e:
|
||||
logger.error(f"❌ {site_key}: Error - {e}")
|
||||
|
||||
async def main():
|
||||
sites_to_test = ["gifi.fr", "stokomani.fr", "auchan.fr", "carrefour.fr", "amazon.fr", "action.com"]
|
||||
|
||||
# Run sequentially to avoid overwhelming resources/logs
|
||||
for site in sites_to_test:
|
||||
await verify_site(site)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,74 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def mock_connect_browser(p):
|
||||
logger.info("MOCK: Launching local browser instead of connecting to browserless")
|
||||
return await p.chromium.launch()
|
||||
|
||||
async def verify_fix():
|
||||
logger.info("Starting verification...")
|
||||
|
||||
# Monkey-patch _connect_browser to use local browser
|
||||
ScraperService._connect_browser = mock_connect_browser
|
||||
|
||||
# Ensure screenshots dir exists
|
||||
os.makedirs("screenshots", exist_ok=True)
|
||||
|
||||
# Test Item ID 999
|
||||
item_id = 999
|
||||
url = "https://example.com"
|
||||
|
||||
logger.info(f"Scraping item {item_id}...")
|
||||
|
||||
# We expect this to fail scraping real content from example.com with specific selectors,
|
||||
# but we only care about the screenshot filename generation which happens at the end.
|
||||
# Actually, if scraping fails, it might return None, "" early.
|
||||
# checking tracking_scraper_service.py:
|
||||
# It has a try/except block.
|
||||
# If _navigate_and_wait works, it proceeds. example.com should load.
|
||||
# _take_screenshot is called at the end.
|
||||
|
||||
# However, ScraperService.scrape_item returns (None, "") if exception occurs.
|
||||
# We need to make sure it doesn't crash before screenshot.
|
||||
# example.com is simple, so it should load.
|
||||
# It will try to click popups (won't find any), wait for selector (if provided).
|
||||
# If we don't provide selector, it calls _auto_detect_price.
|
||||
|
||||
full_path, _ = await ScraperService.scrape_item(url=url, item_id=item_id)
|
||||
|
||||
if full_path:
|
||||
logger.info(f"Screenshot path returned: {full_path}")
|
||||
|
||||
# Verify format: item_{id}_{timestamp}.png
|
||||
# Check if it matches regex
|
||||
pattern = r"screenshots/item_999_\d+\.png"
|
||||
if re.match(pattern, full_path):
|
||||
logger.info("SUCCESS: Filename contains timestamp!")
|
||||
else:
|
||||
logger.error(f"FAILURE: Filename does not match pattern {pattern}")
|
||||
exit(1)
|
||||
|
||||
# Clean up
|
||||
if os.path.exists(full_path):
|
||||
os.remove(full_path)
|
||||
logger.info("Cleaned up screenshot file")
|
||||
|
||||
else:
|
||||
logger.error("FAILURE: Scraper returned None for path. Did navigation fail?")
|
||||
exit(1)
|
||||
|
||||
await ScraperService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_fix())
|
||||
@@ -1,45 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
from app.services.search_service import new_search_service
|
||||
from app.services.browserless_service import browserless_service
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
handlers=[logging.StreamHandler()]
|
||||
)
|
||||
|
||||
async def test_specific_sites():
|
||||
target_sites = ["auchan.fr", "carrefour.fr", "lafoirfouille.fr", "stokomani.fr"]
|
||||
query = "chaise"
|
||||
|
||||
print(f"Testing {target_sites} with query '{query}'...")
|
||||
|
||||
await browserless_service.initialize()
|
||||
|
||||
for site_key in target_sites:
|
||||
if site_key not in SITE_CONFIGS:
|
||||
print(f"Skipping {site_key} (not in config)")
|
||||
continue
|
||||
|
||||
print(f"\n--- Testing {site_key} ---")
|
||||
try:
|
||||
results = await new_search_service.search_site(site_key, query)
|
||||
print(f"Found {len(results)} results")
|
||||
for r in results[:3]:
|
||||
print(f" - {r.title} ({r.price}€) [Image: {r.image_url}]")
|
||||
except Exception as e:
|
||||
print(f"Error testing {site_key}: {e}")
|
||||
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_specific_sites())
|
||||
@@ -1,50 +0,0 @@
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
# Mock modules to avoid ImportError for app dependencies we don't need for this specific test
|
||||
from unittest.mock import MagicMock
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
sys.modules["app.database"] = MagicMock()
|
||||
sys.modules["app.utils.image"] = MagicMock()
|
||||
sys.modules["app.utils.text"] = MagicMock()
|
||||
sys.modules["app.utils.text"].filter_relevant_text = lambda text, max_length: text
|
||||
|
||||
# Mock pydantic
|
||||
mock_pydantic = MagicMock()
|
||||
class MockBaseModel:
|
||||
pass
|
||||
mock_pydantic.BaseModel = MockBaseModel
|
||||
mock_pydantic.Field = MagicMock(return_value=None)
|
||||
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
|
||||
sys.modules["pydantic"] = mock_pydantic
|
||||
|
||||
# Import the schema module
|
||||
from app.ai_schema import get_extraction_prompt
|
||||
|
||||
class TestVisionPriorityPrompt(unittest.TestCase):
|
||||
def test_vision_first_directives(self):
|
||||
"""Verify that the prompt contains the Vision-First directives."""
|
||||
|
||||
# Scenario: Some random text context
|
||||
page_text = "Some random text content from the page."
|
||||
prompt = get_extraction_prompt(page_text)
|
||||
|
||||
print("\nGenerated Prompt Snippet:\n", prompt[:500], "...\n")
|
||||
|
||||
# Check for Critical Directives
|
||||
self.assertIn("Vision-First Price Extraction Agent", prompt)
|
||||
self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt)
|
||||
self.assertIn("IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE", prompt)
|
||||
|
||||
# Check for stock rules
|
||||
self.assertIn("STOCK STATUS RULES", prompt)
|
||||
|
||||
def test_prompt_without_text(self):
|
||||
"""Verify prompt structure when no text is provided."""
|
||||
prompt = get_extraction_prompt(None)
|
||||
self.assertIn("**SOURCE OF TRUTH = IMAGE**", prompt)
|
||||
self.assertNotIn("**Relevant text from page:**", prompt)
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+14
-28
@@ -1,38 +1,24 @@
|
||||
# Walkthrough: Vision-First Price Extraction
|
||||
# Walkthrough: Price Extraction & Catalog Fixes
|
||||
|
||||
In response to issues where the AI was being misled by hidden text (like unit prices or old prices in HTML), we have implemented a **Vision-Priority Strategy**.
|
||||
## 1. Vision Priority Strategy
|
||||
|
||||
## Changes Implemented
|
||||
**Problem**: The AI was prioritizing text data (often hidden/outdated) over the visual price on the screenshot, extracting incorrect prices (e.g. 0.99€ instead of 1.27€).
|
||||
|
||||
### 1. Updated AI System Prompt (`app/ai_schema.py`)
|
||||
**Solution**:
|
||||
|
||||
We completely rewrote the `EXTRACTION_PROMPT_TEMPLATE` to enforce the following rules:
|
||||
- **Prompt Engineering**: Rewrote the system prompt in `ai_schema.py` to explicitly declare the **IMAGE AS THE SOURCE OF TRUTH**.
|
||||
- **Logic Fix**: Disabled the `AIPriceExtractor` (Text-only AI) in `scheduler_service.py` which was short-circuiting the logic before the Vision AI could run.
|
||||
|
||||
- **Source of Truth = Image**: explicit instruction that the screenshot takes precedence over any text.
|
||||
- **Conflict Resolution**: "IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE."
|
||||
- **Visual Focus Rules**:
|
||||
- Look for the largest/boldest price.
|
||||
- Ignore small, styling-less text (often unit prices).
|
||||
- Ignore crossed-out text.
|
||||
## 2. Catalog Scraper Fix
|
||||
|
||||
### Verification
|
||||
**Problem**: Catalogs were not updating because the `browserless` service was failing (likely blocked or network issues), preventing the scraper from loading `cataloguemate.fr`.
|
||||
|
||||
We verified the new prompt generation using `verify_vision_priority.py`.
|
||||
**Solution**:
|
||||
|
||||
**Generated Prompt Preview:**
|
||||
- **HTTP Fallback**: Modified `cataloguemate_scraper.py` to use a robust fallback mechanism.
|
||||
- First attempts to use the secure Browserless browser.
|
||||
- If that fails, it instantly falls back to a standard `httpx` HTTP request, which is often sufficient for static catalog sites.
|
||||
|
||||
```text
|
||||
You are a Vision-First Price Extraction Agent.
|
||||
Your Goal: Extract the main product price exactly as a human sees it on the screen.
|
||||
## 3. Cleanup
|
||||
|
||||
**SOURCE OF TRUTH = IMAGE**
|
||||
- The image provided is the **Absolute Truth**.
|
||||
- The text provided below is scraped HTML content which may contain hidden/old prices.
|
||||
- **IF IMAGE AND TEXT CONFLICT, TRUST THE IMAGE.**
|
||||
```
|
||||
|
||||
## How to Test
|
||||
|
||||
1. Go to "Suivis Prix".
|
||||
2. Force refresh an item that was previously incorrect (e.g., B&M item showing unit price).
|
||||
3. The AI should now ignore the "hidden" unit price text and read the main price tag from the image.
|
||||
- Removed 20+ temporary debug/verification scripts (`debug_*.py`, `verify_*.py`) from the root directory to keep the production environment clean.
|
||||
Reference in new issue
Block a user