From d772e85ae4b58522bce7c750e1a7c3615b90b287 Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Sun, 30 Nov 2025 10:39:16 +0100 Subject: [PATCH] feat: Add catalogue and enseigne API endpoints with scraping and scheduling services. --- app/routers/catalogues.py | 52 +++++++++++++++++++++ app/services/cataloguemate_scraper.py | 66 +++++++++++++++++++++------ app/services/scheduler.py | 40 +++++++++++++++- 3 files changed, 141 insertions(+), 17 deletions(-) diff --git a/app/routers/catalogues.py b/app/routers/catalogues.py index a2d399f..7227b85 100644 --- a/app/routers/catalogues.py +++ b/app/routers/catalogues.py @@ -322,3 +322,55 @@ async def cleanup_catalogs( logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs") return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."} + + +@router.delete("/admin/{catalogue_id}") +async def delete_catalogue( + catalogue_id: int, + db: Session = Depends(get_db), + current_user: User = Depends(get_current_user), +): + """Delete a specific catalogue (admin only)""" + if not current_user.is_admin: + raise HTTPException(status_code=403, detail="Not authorized") + + catalogue = db.query(Catalogue).filter(Catalogue.id == catalogue_id).first() + if not catalogue: + raise HTTPException(status_code=404, detail="Catalogue not found") + + # Pages are cascade-deleted automatically due to relationship + db.delete(catalogue) + db.commit() + logger.info(f"Deleted catalogue {catalogue_id} ({catalogue.titre})") + + return {"message": f"Catalogue '{catalogue.titre}' deleted successfully"} + + +@router.post("/admin/purge") +async def purge_old_catalogues( + db: Session = Depends(get_db), + current_user: User = Depends(get_current_user), +): + """Purge catalogues expired more than 3 months ago (admin only)""" + if not current_user.is_admin: + raise HTTPException(status_code=403, detail="Not authorized") + + from datetime import timedelta + cutoff_date = datetime.now() - timedelta(days=90) + + old_catalogues = db.query(Catalogue).filter( + Catalogue.date_fin < cutoff_date + ).all() + + count = len(old_catalogues) + for cat in old_catalogues: + db.delete(cat) + + db.commit() + logger.info(f"Purged {count} catalogues with date_fin < {cutoff_date}") + + return { + "message": f"Purged {count} catalogues expired more than 3 months ago", + "cutoff_date": cutoff_date.isoformat(), + "deleted_count": count + } diff --git a/app/services/cataloguemate_scraper.py b/app/services/cataloguemate_scraper.py index 51828f1..3b41ad0 100644 --- a/app/services/cataloguemate_scraper.py +++ b/app/services/cataloguemate_scraper.py @@ -102,20 +102,60 @@ async def scrape_catalog_list(enseigne: Enseigne) -> list[dict[str, Any]]: async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]: """ Scrape pages from a specific catalog. - Iterates through ?page=1, ?page=2 etc. + Detects total pages from pagination links on first page. """ pages = [] page_num = 1 - max_pages = 100 # Safety limit + max_pages = 100 # Default safety limit + # First page: detect total number of pages from pagination + first_page_html, _ = await browserless_service.get_page_content(catalog_url) + + if not first_page_html: + logger.error(f"Failed to fetch first page for {catalog_url}") + return [] + + soup = BeautifulSoup(first_page_html, 'html.parser') + + # Try to detect max pages from pagination links + # Look for links with ?page=X or just numeric text + pagination_links = soup.find_all('a', href=True) + page_numbers = [] + + for link in pagination_links: + href = link.get('href', '') + text = link.get_text(strip=True) + + # Check if it's a page number link + if 'page=' in href: + try: + page_match = re.search(r'page=(\d+)', href) + if page_match: + page_numbers.append(int(page_match.group(1))) + except: + pass + elif text.isdigit(): + # Direct numeric link (e.g., "11") + page_numbers.append(int(text)) + + if page_numbers: + max_pages = max(page_numbers) + logger.info(f"Detected {max_pages} total pages from pagination") + else: + logger.warning(f"Could not detect page count, using default limit of {max_pages}") + + # Now scrape all pages while page_num <= max_pages: # Construct URL for specific page current_url = catalog_url if page_num == 1 else f"{catalog_url}?page={page_num}" - logger.info(f"Scraping page {page_num}: {current_url}") + logger.info(f"Scraping page {page_num}/{max_pages}: {current_url}") - # Use browserless service - html_content, _ = await browserless_service.get_page_content(current_url) + # Reuse first page HTML for page 1 + if page_num == 1: + html_content = first_page_html + else: + html_content, _ = await browserless_service.get_page_content(current_url) if not html_content: logger.warning(f"Failed to fetch page {page_num}") @@ -148,18 +188,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]: # Heuristic: Catalog pages are usually large vertical images # Check src for keywords - is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images']) + is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet']) # Logic: # 1. If keyword match AND decent size -> Strong candidate # 2. If no keyword match but HUGE size -> Candidate if is_likely_catalog: - if area > max_area: + if area > max_area or (area == 0 and max_area == 0): max_area = area main_image_url = src - elif area == 0 and not main_image_url: - main_image_url = src elif area > 100000: # Arbitrary large size (e.g. 300x333) if area > max_area: max_area = area @@ -179,18 +217,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]: logger.info(f"Found image for page {page_num}: {main_image_url}") else: logger.warning(f"No catalog image found on page {page_num}") - # If we can't find an image, we might have reached the end or it's a bad page + # If we can't find an image after page 1, we've likely reached the end if page_num > 1: + logger.info(f"Stopping pagination at page {page_num - 1} (no image found)") break - # Check for pagination "Next" to know if we should continue - if not main_image_url: - break - page_num += 1 - # Small delay handled by browserless service already, but adding a tiny one here + # Small delay to avoid overwhelming the server await asyncio.sleep(0.5) + logger.info(f"Scraped {len(pages)} pages total") return pages async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog: diff --git a/app/services/scheduler.py b/app/services/scheduler.py index c61bd3d..67ed213 100644 --- a/app/services/scheduler.py +++ b/app/services/scheduler.py @@ -41,6 +41,33 @@ async def scheduled_scrape_job(): db.close() +async def scheduled_purge_job(): + """Scheduled job to purge old catalogs (expired > 3 months)""" + logger.info("Starting scheduled catalog purge job") + + from datetime import datetime, timedelta + from app.models import Catalogue + + db = SessionLocal() + try: + cutoff_date = datetime.now() - timedelta(days=90) + + old_catalogues = db.query(Catalogue).filter( + Catalogue.date_fin < cutoff_date + ).all() + + count = len(old_catalogues) + for cat in old_catalogues: + db.delete(cat) + + db.commit() + logger.info(f"Scheduled purge complete: deleted {count} catalogues with date_fin < {cutoff_date}") + except Exception as e: + logger.error(f"Error in scheduled purge job: {e}") + finally: + db.close() + + def start_scheduler(): """Start the scheduler with configured jobs""" # Schedule scraping twice daily at 6:00 and 18:00 @@ -48,12 +75,21 @@ def start_scheduler(): scheduled_scrape_job, CronTrigger(hour="6,18", minute=0), id="catalog_scraping", - name="Scrape Bonial catalogs", + name="Scrape Cataloguemate catalogs", + replace_existing=True, + ) + + # Schedule automatic purge on the 1st of each month at 3:00 AM + scheduler.add_job( + scheduled_purge_job, + CronTrigger(day=1, hour=3, minute=0), + id="catalog_purge", + name="Purge old catalogs", replace_existing=True, ) scheduler.start() - logger.info("Catalog scraping scheduler started (runs at 6:00 and 18:00)") + logger.info("Catalog scraping scheduler started (scraping at 6:00 and 18:00, purge on 1st of month at 3:00)") def stop_scheduler():