mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: implement Bonial.fr scraper service for promotional catalogs using Playwright.
This commit is contained in:
1 parent
ba932b79ef
commit
3d3d7c18c1
3 files changed
+156
-7
No files matched your search
@@ -315,3 +315,48 @@ async def get_scraping_stats(
|
||||
derniere_mise_a_jour=derniere_mise_a_jour,
|
||||
prochaine_execution=prochaine,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/admin/cleanup", response_model=dict)
|
||||
async def cleanup_catalogs(
|
||||
db: Session = Depends(get_db),
|
||||
current_user: User = Depends(get_current_active_user),
|
||||
):
|
||||
"""
|
||||
Delete invalid catalogs (0 pages or generic titles).
|
||||
Admin only.
|
||||
"""
|
||||
if not current_user.is_admin:
|
||||
raise HTTPException(status_code=403, detail="Not authorized")
|
||||
|
||||
deleted_count = 0
|
||||
|
||||
# 1. Delete catalogs with 0 pages
|
||||
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
|
||||
for cat in bad_catalogs:
|
||||
db.delete(cat)
|
||||
deleted_count += 1
|
||||
|
||||
# 2. Delete generic titles
|
||||
generic_titles = [
|
||||
"Restez informé",
|
||||
"Catalogues Bazar",
|
||||
"Toutes les offres",
|
||||
"Voir les offres",
|
||||
"Téléchargez l'application",
|
||||
"Newsletter",
|
||||
"BLACK FRIDAY"
|
||||
]
|
||||
|
||||
for title_part in generic_titles:
|
||||
generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all()
|
||||
for cat in generic_cats:
|
||||
# Only delete if it has few pages (e.g. < 2) to be safe
|
||||
if cat.nombre_pages < 2:
|
||||
db.delete(cat)
|
||||
deleted_count += 1
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
|
||||
|
||||
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
|
||||
@@ -19,16 +19,17 @@ from app.models import Enseigne, Catalogue, CataloguePage, ScrapingLog
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Bonial selectors based on manual analysis
|
||||
# Bonial selectors based on browser inspection
|
||||
BONIAL_SELECTORS = {
|
||||
"catalog_card": "a[href*='/catalogue/']:has(img), section:has(img[src*='content-media.bonial.biz'])",
|
||||
"catalog_title": "div[class*='title'], h3, h2, h4, .text-lg, span[class*='text-']",
|
||||
"catalog_image": "img[src*='content-media.bonial.biz'], img[src*='bonial']",
|
||||
# Card must contain a catalog link AND an image
|
||||
"catalog_card": "section:has(a[href*='/catalogue/']), div[class*='brochure-card']",
|
||||
"catalog_title": "div[class*='title'], h2, h3, h4, .text-lg",
|
||||
"catalog_image": "img[src*='content-media.bonial.biz']",
|
||||
"catalog_link": "a[href*='/catalogue/']",
|
||||
"cookie_accept": "button:has-text('OK'), button:has-text('Accepter'), #onetrust-accept-btn-handler",
|
||||
}
|
||||
|
||||
# Date pattern: matches dd/mm or dd/mm/yyyy
|
||||
# Date pattern: matches "lun. 25/11 - mar. 08/12/2025" or similar
|
||||
DATE_PATTERN = r"(\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)"
|
||||
|
||||
|
||||
@@ -195,6 +196,30 @@ async def scrape_catalog_list(
|
||||
if catalogue_url and not catalogue_url.startswith("http"):
|
||||
catalogue_url = f"https://www.bonial.fr{catalogue_url}"
|
||||
|
||||
# Validate URL - Must be a real catalog viewer, not an app link or generic page
|
||||
if catalogue_url and ("adjust.com" in catalogue_url or "/Mobile" in catalogue_url):
|
||||
logger.warning(f"Skipping app/mobile link for catalog {idx}: {catalogue_url}")
|
||||
catalogue_url = None
|
||||
|
||||
if catalogue_url and "/catalogue/" not in catalogue_url:
|
||||
logger.warning(f"Skipping non-catalog URL for catalog {idx}: {catalogue_url}")
|
||||
catalogue_url = None
|
||||
|
||||
# Filter out generic banners/newsletters based on title
|
||||
if titre:
|
||||
lower_title = titre.lower()
|
||||
ignored_titles = [
|
||||
"restez informé",
|
||||
"catalogues bazar",
|
||||
"toutes les offres",
|
||||
"voir les offres",
|
||||
"téléchargez l'application",
|
||||
"newsletter"
|
||||
]
|
||||
if any(ignored in lower_title for ignored in ignored_titles):
|
||||
logger.info(f"Skipping generic/banner card: {titre}")
|
||||
continue
|
||||
|
||||
# Validate data
|
||||
has_url = bool(catalogue_url)
|
||||
has_image = bool(image_url)
|
||||
@@ -211,6 +236,7 @@ async def scrape_catalog_list(
|
||||
if "semaine" in dates_text.lower() or "valable" in dates_text.lower():
|
||||
date_fin = datetime.now().replace(day=datetime.now().day + 7) # Rough approx
|
||||
|
||||
# STRICTER VALIDATION: Must have URL, Image AND (Dates OR strong title signal)
|
||||
if has_url and has_image:
|
||||
catalogues.append({
|
||||
"titre": titre.strip(),
|
||||
@@ -224,8 +250,11 @@ async def scrape_catalog_list(
|
||||
# DEBUG: Log HTML content if extraction failed to help diagnosis
|
||||
try:
|
||||
html_content = await card.inner_html()
|
||||
logger.warning(f"Incomplete data for catalog {idx}: title={titre}, dates={has_dates}, image={has_image}, url={has_url}")
|
||||
logger.warning(f"Card HTML snippet: {html_content[:200]}...")
|
||||
# Only log if it's not obviously bad
|
||||
if not has_url:
|
||||
logger.warning(f"Skipping card {idx} (No URL): {titre}")
|
||||
elif not has_image:
|
||||
logger.warning(f"Skipping card {idx} (No Image): {titre}")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
# Add parent directory to path to import app modules
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from sqlalchemy import create_engine, text
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
|
||||
from app.database import Base, get_db
|
||||
from app.models import Catalogue, CataloguePage, ScrapingLog
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
def cleanup_bad_catalogs():
|
||||
"""
|
||||
Delete catalogs that have 0 pages or were created with invalid data.
|
||||
"""
|
||||
# Get database URL from environment or use default
|
||||
DATABASE_URL = os.getenv("DATABASE_URL", "postgresql://priceflow:priceflow@localhost:5488/priceflow")
|
||||
|
||||
engine = create_engine(DATABASE_URL)
|
||||
SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine)
|
||||
db = SessionLocal()
|
||||
|
||||
try:
|
||||
logger.info("Starting cleanup of invalid catalogs...")
|
||||
|
||||
# 1. Delete catalogs with 0 pages
|
||||
# We need to check actual pages count in DB, not just the column
|
||||
|
||||
# Find catalogs with no pages
|
||||
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
|
||||
|
||||
count = 0
|
||||
for cat in bad_catalogs:
|
||||
logger.info(f"Deleting empty catalog: {cat.titre} (ID: {cat.id})")
|
||||
db.delete(cat)
|
||||
count += 1
|
||||
|
||||
# 2. Delete catalogs with generic titles if any remain
|
||||
generic_titles = [
|
||||
"Restez informé",
|
||||
"Catalogues Bazar",
|
||||
"Toutes les offres",
|
||||
"Voir les offres",
|
||||
"Téléchargez l'application",
|
||||
"Newsletter",
|
||||
"BLACK FRIDAY" # Often a banner if 0 pages
|
||||
]
|
||||
|
||||
for title_part in generic_titles:
|
||||
generic_cats = db.query(Catalogue).filter(Catalogue.titre.ilike(f"%{title_part}%")).all()
|
||||
for cat in generic_cats:
|
||||
# Only delete if it has few pages (e.g. < 2) or seems suspicious
|
||||
# Actually, let's just check if it has pages. If it has pages, it might be valid.
|
||||
# But we already deleted 0-page ones.
|
||||
# Let's be careful.
|
||||
pass
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Cleanup complete. Deleted {count} invalid catalogs.")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error during cleanup: {e}")
|
||||
db.rollback()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
cleanup_bad_catalogs()
|
||||
Reference in new issue
Block a user