mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
122 lines
4.0 KiB
Python
122 lines
4.0 KiB
Python
import asyncio
|
|
import logging
|
|
import sys
|
|
from bs4 import BeautifulSoup
|
|
from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode
|
|
|
|
# Configure logging
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format="%(asctime)s - %(levelname)s - %(message)s",
|
|
handlers=[logging.StreamHandler(sys.stdout)]
|
|
)
|
|
logger = logging.getLogger(__name__)
|
|
|
|
BASE_URL = "https://www.cataloguemate.fr"
|
|
|
|
async def debug_catalog_list():
|
|
"""Debug the catalog list extraction"""
|
|
# Test with Gifi
|
|
slug = "gifi"
|
|
url = f"{BASE_URL}/{slug}/"
|
|
|
|
logger.info(f"--- DEBUGGING LIST: {url} ---")
|
|
|
|
browser_config = BrowserConfig(headless=True)
|
|
run_config = CrawlerRunConfig(
|
|
cache_mode=CacheMode.BYPASS,
|
|
wait_for_images=True,
|
|
)
|
|
|
|
async with AsyncWebCrawler(config=browser_config) as crawler:
|
|
result = await crawler.arun(url=url, config=run_config)
|
|
|
|
if not result.success:
|
|
logger.error(f"Failed to fetch {url}: {result.error_message}")
|
|
return None
|
|
|
|
logger.info(f"Successfully fetched {url} ({len(result.html)} chars)")
|
|
|
|
soup = BeautifulSoup(result.html, 'html.parser')
|
|
|
|
# 1. Dump all links to see what we have
|
|
links = soup.find_all('a', href=True)
|
|
logger.info(f"Found {len(links)} links total")
|
|
|
|
potential_catalogs = []
|
|
|
|
for i, link in enumerate(links):
|
|
href = link['href']
|
|
text = link.get_text(strip=True)
|
|
|
|
# Normalize
|
|
if href.startswith(BASE_URL):
|
|
href = href.replace(BASE_URL, "")
|
|
|
|
# Log interesting links
|
|
if slug in href or "catalogue" in href.lower():
|
|
logger.info(f"Link {i}: {href} | Text: '{text}'")
|
|
|
|
# Apply our filter logic to see if it passes
|
|
if href.startswith(f"/{slug}/") and href != f"/{slug}/":
|
|
if not any(x in href for x in ["offres", "magasins", "rechercher"]):
|
|
potential_catalogs.append(href)
|
|
logger.info(f" -> MATCHES FILTER!")
|
|
|
|
logger.info(f"Total matching catalogs: {len(potential_catalogs)}")
|
|
return potential_catalogs[0] if potential_catalogs else None
|
|
|
|
async def debug_catalog_page(catalog_rel_url):
|
|
"""Debug the catalog page extraction"""
|
|
if not catalog_rel_url:
|
|
logger.error("No catalog URL to debug")
|
|
return
|
|
|
|
full_url = f"{BASE_URL}{catalog_rel_url}"
|
|
logger.info(f"\n--- DEBUGGING PAGE: {full_url} ---")
|
|
|
|
browser_config = BrowserConfig(headless=True)
|
|
run_config = CrawlerRunConfig(
|
|
cache_mode=CacheMode.BYPASS,
|
|
wait_for_images=True,
|
|
delay_before_return_html=2.0 # Wait a bit more
|
|
)
|
|
|
|
async with AsyncWebCrawler(config=browser_config) as crawler:
|
|
result = await crawler.arun(url=full_url, config=run_config)
|
|
|
|
if not result.success:
|
|
logger.error(f"Failed to fetch {full_url}")
|
|
return
|
|
|
|
soup = BeautifulSoup(result.html, 'html.parser')
|
|
|
|
# 1. Dump all images
|
|
images = soup.find_all('img')
|
|
logger.info(f"Found {len(images)} images")
|
|
|
|
for i, img in enumerate(images):
|
|
src = img.get('src', '')
|
|
width = img.get('width', '?')
|
|
height = img.get('height', '?')
|
|
alt = img.get('alt', '')
|
|
|
|
# Filter noise
|
|
if "logo" in src or "icon" in src:
|
|
continue
|
|
|
|
logger.info(f"Img {i}: {src} | {width}x{height} | Alt: {alt}")
|
|
|
|
# Check our heuristic
|
|
is_likely = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
|
|
if is_likely:
|
|
logger.info(" -> LIKELY CATALOG IMAGE")
|
|
|
|
if __name__ == "__main__":
|
|
async def main():
|
|
cat_url = await debug_catalog_list()
|
|
if cat_url:
|
|
await debug_catalog_page(cat_url)
|
|
|
|
asyncio.run(main())
|