Merge pull request #121 from R0m1k3/antigravity

feat: implement Bonial.fr scraper service for promotional catalogs us…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-29 22:32:27 +01:00
commit 2fa99b4b4d
3 files changed
+156 -7

No files matched your search

+45
View File
@@ -315,3 +315,48 @@ async def get_scraping_stats(
derniere_mise_a_jour=derniere_mise_a_jour,
prochaine_execution=prochaine,
)
@router.post("/admin/cleanup", response_model=dict)
async def cleanup_catalogs(
db: Session = Depends(get_db),
current_user: User = Depends(get_current_active_user),
):
"""
Delete invalid catalogs (0 pages or generic titles).
Admin only.
"""
if not current_user.is_admin:
raise HTTPException(status_code=403, detail="Not authorized")
deleted_count = 0
# 1. Delete catalogs with 0 pages
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
for cat in bad_catalogs:
db.delete(cat)
deleted_count += 1
# 2. Delete generic titles
generic_titles = [
"Restez informé",
"Catalogues Bazar",
"Toutes les offres",
"Voir les offres",
"Téléchargez l'application",
"Newsletter",
"BLACK FRIDAY"
]
for title_part in generic_titles:
generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all()
for cat in generic_cats:
# Only delete if it has few pages (e.g. < 2) to be safe
if cat.nombre_pages < 2:
db.delete(cat)
deleted_count += 1
db.commit()
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
+36 -7
View File
@@ -19,16 +19,17 @@ from app.models import Enseigne, Catalogue, CataloguePage, ScrapingLog
logger = logging.getLogger(__name__)
# Bonial selectors based on manual analysis
# Bonial selectors based on browser inspection
BONIAL_SELECTORS = {
"catalog_card": "a[href*='/catalogue/']:has(img), section:has(img[src*='content-media.bonial.biz'])",
"catalog_title": "div[class*='title'], h3, h2, h4, .text-lg, span[class*='text-']",
"catalog_image": "img[src*='content-media.bonial.biz'], img[src*='bonial']",
# Card must contain a catalog link AND an image
"catalog_card": "section:has(a[href*='/catalogue/']), div[class*='brochure-card']",
"catalog_title": "div[class*='title'], h2, h3, h4, .text-lg",
"catalog_image": "img[src*='content-media.bonial.biz']",
"catalog_link": "a[href*='/catalogue/']",
"cookie_accept": "button:has-text('OK'), button:has-text('Accepter'), #onetrust-accept-btn-handler",
}
# Date pattern: matches dd/mm or dd/mm/yyyy
# Date pattern: matches "lun. 25/11 - mar. 08/12/2025" or similar
DATE_PATTERN = r"(\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)"
@@ -195,6 +196,30 @@ async def scrape_catalog_list(
if catalogue_url and not catalogue_url.startswith("http"):
catalogue_url = f"https://www.bonial.fr{catalogue_url}"
# Validate URL - Must be a real catalog viewer, not an app link or generic page
if catalogue_url and ("adjust.com" in catalogue_url or "/Mobile" in catalogue_url):
logger.warning(f"Skipping app/mobile link for catalog {idx}: {catalogue_url}")
catalogue_url = None
if catalogue_url and "/catalogue/" not in catalogue_url:
logger.warning(f"Skipping non-catalog URL for catalog {idx}: {catalogue_url}")
catalogue_url = None
# Filter out generic banners/newsletters based on title
if titre:
lower_title = titre.lower()
ignored_titles = [
"restez informé",
"catalogues bazar",
"toutes les offres",
"voir les offres",
"téléchargez l'application",
"newsletter"
]
if any(ignored in lower_title for ignored in ignored_titles):
logger.info(f"Skipping generic/banner card: {titre}")
continue
# Validate data
has_url = bool(catalogue_url)
has_image = bool(image_url)
@@ -211,6 +236,7 @@ async def scrape_catalog_list(
if "semaine" in dates_text.lower() or "valable" in dates_text.lower():
date_fin = datetime.now().replace(day=datetime.now().day + 7) # Rough approx
# STRICTER VALIDATION: Must have URL, Image AND (Dates OR strong title signal)
if has_url and has_image:
catalogues.append({
"titre": titre.strip(),
@@ -224,8 +250,11 @@ async def scrape_catalog_list(
# DEBUG: Log HTML content if extraction failed to help diagnosis
try:
html_content = await card.inner_html()
logger.warning(f"Incomplete data for catalog {idx}: title={titre}, dates={has_dates}, image={has_image}, url={has_url}")
logger.warning(f"Card HTML snippet: {html_content[:200]}...")
# Only log if it's not obviously bad
if not has_url:
logger.warning(f"Skipping card {idx} (No URL): {titre}")
elif not has_image:
logger.warning(f"Skipping card {idx} (No Image): {titre}")
except Exception:
pass
+75
View File
@@ -0,0 +1,75 @@
import asyncio
import logging
import os
import sys
# Add parent directory to path to import app modules
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from sqlalchemy import create_engine, text
from sqlalchemy.orm import sessionmaker
from app.database import Base, get_db
from app.models import Catalogue, CataloguePage, ScrapingLog
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
def cleanup_bad_catalogs():
"""
Delete catalogs that have 0 pages or were created with invalid data.
"""
# Get database URL from environment or use default
DATABASE_URL = os.getenv("DATABASE_URL", "postgresql://priceflow:priceflow@localhost:5488/priceflow")
engine = create_engine(DATABASE_URL)
SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine)
db = SessionLocal()
try:
logger.info("Starting cleanup of invalid catalogs...")
# 1. Delete catalogs with 0 pages
# We need to check actual pages count in DB, not just the column
# Find catalogs with no pages
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
count = 0
for cat in bad_catalogs:
logger.info(f"Deleting empty catalog: {cat.titre} (ID: {cat.id})")
db.delete(cat)
count += 1
# 2. Delete catalogs with generic titles if any remain
generic_titles = [
"Restez informé",
"Catalogues Bazar",
"Toutes les offres",
"Voir les offres",
"Téléchargez l'application",
"Newsletter",
"BLACK FRIDAY" # Often a banner if 0 pages
]
for title_part in generic_titles:
generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all()
for cat in generic_cats:
# Only delete if it has few pages (e.g. < 2) or seems suspicious
# Actually, let's just check if it has pages. If it has pages, it might be valid.
# But we already deleted 0-page ones.
# Let's be careful.
pass
db.commit()
logger.info(f"Cleanup complete. Deleted {count} invalid catalogs.")
except Exception as e:
logger.error(f"Error during cleanup: {e}")
db.rollback()
finally:
db.close()
if __name__ == "__main__":
cleanup_bad_catalogs()