feat: Add catalogue and enseigne API endpoints with scraping and scheduling services.

This commit is contained in:
Michael committed 2025-11-30 10:39:16 +01:00
1 parent df8fbded92
commit d772e85ae4
3 files changed
+141 -17

No files matched your search

+52
View File
@@ -322,3 +322,55 @@ async def cleanup_catalogs(
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
@router.delete("/admin/{catalogue_id}")
async def delete_catalogue(
catalogue_id: int,
db: Session = Depends(get_db),
current_user: User = Depends(get_current_user),
):
"""Delete a specific catalogue (admin only)"""
if not current_user.is_admin:
raise HTTPException(status_code=403, detail="Not authorized")
catalogue = db.query(Catalogue).filter(Catalogue.id == catalogue_id).first()
if not catalogue:
raise HTTPException(status_code=404, detail="Catalogue not found")
# Pages are cascade-deleted automatically due to relationship
db.delete(catalogue)
db.commit()
logger.info(f"Deleted catalogue {catalogue_id} ({catalogue.titre})")
return {"message": f"Catalogue '{catalogue.titre}' deleted successfully"}
@router.post("/admin/purge")
async def purge_old_catalogues(
db: Session = Depends(get_db),
current_user: User = Depends(get_current_user),
):
"""Purge catalogues expired more than 3 months ago (admin only)"""
if not current_user.is_admin:
raise HTTPException(status_code=403, detail="Not authorized")
from datetime import timedelta
cutoff_date = datetime.now() - timedelta(days=90)
old_catalogues = db.query(Catalogue).filter(
Catalogue.date_fin < cutoff_date
).all()
count = len(old_catalogues)
for cat in old_catalogues:
db.delete(cat)
db.commit()
logger.info(f"Purged {count} catalogues with date_fin < {cutoff_date}")
return {
"message": f"Purged {count} catalogues expired more than 3 months ago",
"cutoff_date": cutoff_date.isoformat(),
"deleted_count": count
}
+51 -15
View File
@@ -102,20 +102,60 @@ async def scrape_catalog_list(enseigne: Enseigne) -> list[dict[str, Any]]:
async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
"""
Scrape pages from a specific catalog.
Iterates through ?page=1, ?page=2 etc.
Detects total pages from pagination links on first page.
"""
pages = []
page_num = 1
max_pages = 100 # Safety limit
max_pages = 100 # Default safety limit
# First page: detect total number of pages from pagination
first_page_html, _ = await browserless_service.get_page_content(catalog_url)
if not first_page_html:
logger.error(f"Failed to fetch first page for {catalog_url}")
return []
soup = BeautifulSoup(first_page_html, 'html.parser')
# Try to detect max pages from pagination links
# Look for links with ?page=X or just numeric text
pagination_links = soup.find_all('a', href=True)
page_numbers = []
for link in pagination_links:
href = link.get('href', '')
text = link.get_text(strip=True)
# Check if it's a page number link
if 'page=' in href:
try:
page_match = re.search(r'page=(\d+)', href)
if page_match:
page_numbers.append(int(page_match.group(1)))
except:
pass
elif text.isdigit():
# Direct numeric link (e.g., "11")
page_numbers.append(int(text))
if page_numbers:
max_pages = max(page_numbers)
logger.info(f"Detected {max_pages} total pages from pagination")
else:
logger.warning(f"Could not detect page count, using default limit of {max_pages}")
# Now scrape all pages
while page_num <= max_pages:
# Construct URL for specific page
current_url = catalog_url if page_num == 1 else f"{catalog_url}?page={page_num}"
logger.info(f"Scraping page {page_num}: {current_url}")
logger.info(f"Scraping page {page_num}/{max_pages}: {current_url}")
# Use browserless service
html_content, _ = await browserless_service.get_page_content(current_url)
# Reuse first page HTML for page 1
if page_num == 1:
html_content = first_page_html
else:
html_content, _ = await browserless_service.get_page_content(current_url)
if not html_content:
logger.warning(f"Failed to fetch page {page_num}")
@@ -148,18 +188,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
# Heuristic: Catalog pages are usually large vertical images
# Check src for keywords
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet'])
# Logic:
# 1. If keyword match AND decent size -> Strong candidate
# 2. If no keyword match but HUGE size -> Candidate
if is_likely_catalog:
if area > max_area:
if area > max_area or (area == 0 and max_area == 0):
max_area = area
main_image_url = src
elif area == 0 and not main_image_url:
main_image_url = src
elif area > 100000: # Arbitrary large size (e.g. 300x333)
if area > max_area:
max_area = area
@@ -179,18 +217,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
logger.info(f"Found image for page {page_num}: {main_image_url}")
else:
logger.warning(f"No catalog image found on page {page_num}")
# If we can't find an image, we might have reached the end or it's a bad page
# If we can't find an image after page 1, we've likely reached the end
if page_num > 1:
logger.info(f"Stopping pagination at page {page_num - 1} (no image found)")
break
# Check for pagination "Next" to know if we should continue
if not main_image_url:
break
page_num += 1
# Small delay handled by browserless service already, but adding a tiny one here
# Small delay to avoid overwhelming the server
await asyncio.sleep(0.5)
logger.info(f"Scraped {len(pages)} pages total")
return pages
async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
+38 -2
View File
@@ -41,6 +41,33 @@ async def scheduled_scrape_job():
db.close()
async def scheduled_purge_job():
"""Scheduled job to purge old catalogs (expired > 3 months)"""
logger.info("Starting scheduled catalog purge job")
from datetime import datetime, timedelta
from app.models import Catalogue
db = SessionLocal()
try:
cutoff_date = datetime.now() - timedelta(days=90)
old_catalogues = db.query(Catalogue).filter(
Catalogue.date_fin < cutoff_date
).all()
count = len(old_catalogues)
for cat in old_catalogues:
db.delete(cat)
db.commit()
logger.info(f"Scheduled purge complete: deleted {count} catalogues with date_fin < {cutoff_date}")
except Exception as e:
logger.error(f"Error in scheduled purge job: {e}")
finally:
db.close()
def start_scheduler():
"""Start the scheduler with configured jobs"""
# Schedule scraping twice daily at 6:00 and 18:00
@@ -48,12 +75,21 @@ def start_scheduler():
scheduled_scrape_job,
CronTrigger(hour="6,18", minute=0),
id="catalog_scraping",
name="Scrape Bonial catalogs",
name="Scrape Cataloguemate catalogs",
replace_existing=True,
)
# Schedule automatic purge on the 1st of each month at 3:00 AM
scheduler.add_job(
scheduled_purge_job,
CronTrigger(day=1, hour=3, minute=0),
id="catalog_purge",
name="Purge old catalogs",
replace_existing=True,
)
scheduler.start()
logger.info("Catalog scraping scheduler started (runs at 6:00 and 18:00)")
logger.info("Catalog scraping scheduler started (scraping at 6:00 and 18:00, purge on 1st of month at 3:00)")
def stop_scheduler():