diff --git a/app/routers/catalogues.py b/app/routers/catalogues.py index 4a03182..041fa60 100644 --- a/app/routers/catalogues.py +++ b/app/routers/catalogues.py @@ -315,3 +315,48 @@ async def get_scraping_stats( derniere_mise_a_jour=derniere_mise_a_jour, prochaine_execution=prochaine, ) + + +@router.post("/admin/cleanup", response_model=dict) +async def cleanup_catalogs( + db: Session = Depends(get_db), + current_user: User = Depends(get_current_active_user), +): + """ + Delete invalid catalogs (0 pages or generic titles). + Admin only. + """ + if not current_user.is_admin: + raise HTTPException(status_code=403, detail="Not authorized") + + deleted_count = 0 + + # 1. Delete catalogs with 0 pages + bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all() + for cat in bad_catalogs: + db.delete(cat) + deleted_count += 1 + + # 2. Delete generic titles + generic_titles = [ + "Restez informé", + "Catalogues Bazar", + "Toutes les offres", + "Voir les offres", + "Téléchargez l'application", + "Newsletter", + "BLACK FRIDAY" + ] + + for title_part in generic_titles: + generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all() + for cat in generic_cats: + # Only delete if it has few pages (e.g. < 2) to be safe + if cat.nombre_pages < 2: + db.delete(cat) + deleted_count += 1 + + db.commit() + logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs") + + return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."} diff --git a/app/services/bonial_scraper.py b/app/services/bonial_scraper.py index 01450da..098e9bc 100644 --- a/app/services/bonial_scraper.py +++ b/app/services/bonial_scraper.py @@ -19,16 +19,17 @@ from app.models import Enseigne, Catalogue, CataloguePage, ScrapingLog logger = logging.getLogger(__name__) -# Bonial selectors based on manual analysis +# Bonial selectors based on browser inspection BONIAL_SELECTORS = { - "catalog_card": "a[href*='/catalogue/']:has(img), section:has(img[src*='content-media.bonial.biz'])", - "catalog_title": "div[class*='title'], h3, h2, h4, .text-lg, span[class*='text-']", - "catalog_image": "img[src*='content-media.bonial.biz'], img[src*='bonial']", + # Card must contain a catalog link AND an image + "catalog_card": "section:has(a[href*='/catalogue/']), div[class*='brochure-card']", + "catalog_title": "div[class*='title'], h2, h3, h4, .text-lg", + "catalog_image": "img[src*='content-media.bonial.biz']", "catalog_link": "a[href*='/catalogue/']", "cookie_accept": "button:has-text('OK'), button:has-text('Accepter'), #onetrust-accept-btn-handler", } -# Date pattern: matches dd/mm or dd/mm/yyyy +# Date pattern: matches "lun. 25/11 - mar. 08/12/2025" or similar DATE_PATTERN = r"(\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)" @@ -195,6 +196,30 @@ async def scrape_catalog_list( if catalogue_url and not catalogue_url.startswith("http"): catalogue_url = f"https://www.bonial.fr{catalogue_url}" + # Validate URL - Must be a real catalog viewer, not an app link or generic page + if catalogue_url and ("adjust.com" in catalogue_url or "/Mobile" in catalogue_url): + logger.warning(f"Skipping app/mobile link for catalog {idx}: {catalogue_url}") + catalogue_url = None + + if catalogue_url and "/catalogue/" not in catalogue_url: + logger.warning(f"Skipping non-catalog URL for catalog {idx}: {catalogue_url}") + catalogue_url = None + + # Filter out generic banners/newsletters based on title + if titre: + lower_title = titre.lower() + ignored_titles = [ + "restez informé", + "catalogues bazar", + "toutes les offres", + "voir les offres", + "téléchargez l'application", + "newsletter" + ] + if any(ignored in lower_title for ignored in ignored_titles): + logger.info(f"Skipping generic/banner card: {titre}") + continue + # Validate data has_url = bool(catalogue_url) has_image = bool(image_url) @@ -211,6 +236,7 @@ async def scrape_catalog_list( if "semaine" in dates_text.lower() or "valable" in dates_text.lower(): date_fin = datetime.now().replace(day=datetime.now().day + 7) # Rough approx + # STRICTER VALIDATION: Must have URL, Image AND (Dates OR strong title signal) if has_url and has_image: catalogues.append({ "titre": titre.strip(), @@ -224,8 +250,11 @@ async def scrape_catalog_list( # DEBUG: Log HTML content if extraction failed to help diagnosis try: html_content = await card.inner_html() - logger.warning(f"Incomplete data for catalog {idx}: title={titre}, dates={has_dates}, image={has_image}, url={has_url}") - logger.warning(f"Card HTML snippet: {html_content[:200]}...") + # Only log if it's not obviously bad + if not has_url: + logger.warning(f"Skipping card {idx} (No URL): {titre}") + elif not has_image: + logger.warning(f"Skipping card {idx} (No Image): {titre}") except Exception: pass diff --git a/scripts/cleanup_catalogs.py b/scripts/cleanup_catalogs.py new file mode 100644 index 0000000..c942a62 --- /dev/null +++ b/scripts/cleanup_catalogs.py @@ -0,0 +1,75 @@ +import asyncio +import logging +import os +import sys + +# Add parent directory to path to import app modules +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from sqlalchemy import create_engine, text +from sqlalchemy.orm import sessionmaker + +from app.database import Base, get_db +from app.models import Catalogue, CataloguePage, ScrapingLog + +# Configure logging +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +def cleanup_bad_catalogs(): + """ + Delete catalogs that have 0 pages or were created with invalid data. + """ + # Get database URL from environment or use default + DATABASE_URL = os.getenv("DATABASE_URL", "postgresql://priceflow:priceflow@localhost:5488/priceflow") + + engine = create_engine(DATABASE_URL) + SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine) + db = SessionLocal() + + try: + logger.info("Starting cleanup of invalid catalogs...") + + # 1. Delete catalogs with 0 pages + # We need to check actual pages count in DB, not just the column + + # Find catalogs with no pages + bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all() + + count = 0 + for cat in bad_catalogs: + logger.info(f"Deleting empty catalog: {cat.titre} (ID: {cat.id})") + db.delete(cat) + count += 1 + + # 2. Delete catalogs with generic titles if any remain + generic_titles = [ + "Restez informé", + "Catalogues Bazar", + "Toutes les offres", + "Voir les offres", + "Téléchargez l'application", + "Newsletter", + "BLACK FRIDAY" # Often a banner if 0 pages + ] + + for title_part in generic_titles: + generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all() + for cat in generic_cats: + # Only delete if it has few pages (e.g. < 2) or seems suspicious + # Actually, let's just check if it has pages. If it has pages, it might be valid. + # But we already deleted 0-page ones. + # Let's be careful. + pass + + db.commit() + logger.info(f"Cleanup complete. Deleted {count} invalid catalogs.") + + except Exception as e: + logger.error(f"Error during cleanup: {e}") + db.rollback() + finally: + db.close() + +if __name__ == "__main__": + cleanup_bad_catalogs()