mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Add catalogue and enseigne API endpoints with scraping and scheduling services.
This commit is contained in:
1 parent
df8fbded92
commit
d772e85ae4
3 files changed
+141
-17
No files matched your search
@@ -322,3 +322,55 @@ async def cleanup_catalogs(
|
||||
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
|
||||
|
||||
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
|
||||
|
||||
|
||||
@router.delete("/admin/{catalogue_id}")
|
||||
async def delete_catalogue(
|
||||
catalogue_id: int,
|
||||
db: Session = Depends(get_db),
|
||||
current_user: User = Depends(get_current_user),
|
||||
):
|
||||
"""Delete a specific catalogue (admin only)"""
|
||||
if not current_user.is_admin:
|
||||
raise HTTPException(status_code=403, detail="Not authorized")
|
||||
|
||||
catalogue = db.query(Catalogue).filter(Catalogue.id == catalogue_id).first()
|
||||
if not catalogue:
|
||||
raise HTTPException(status_code=404, detail="Catalogue not found")
|
||||
|
||||
# Pages are cascade-deleted automatically due to relationship
|
||||
db.delete(catalogue)
|
||||
db.commit()
|
||||
logger.info(f"Deleted catalogue {catalogue_id} ({catalogue.titre})")
|
||||
|
||||
return {"message": f"Catalogue '{catalogue.titre}' deleted successfully"}
|
||||
|
||||
|
||||
@router.post("/admin/purge")
|
||||
async def purge_old_catalogues(
|
||||
db: Session = Depends(get_db),
|
||||
current_user: User = Depends(get_current_user),
|
||||
):
|
||||
"""Purge catalogues expired more than 3 months ago (admin only)"""
|
||||
if not current_user.is_admin:
|
||||
raise HTTPException(status_code=403, detail="Not authorized")
|
||||
|
||||
from datetime import timedelta
|
||||
cutoff_date = datetime.now() - timedelta(days=90)
|
||||
|
||||
old_catalogues = db.query(Catalogue).filter(
|
||||
Catalogue.date_fin < cutoff_date
|
||||
).all()
|
||||
|
||||
count = len(old_catalogues)
|
||||
for cat in old_catalogues:
|
||||
db.delete(cat)
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Purged {count} catalogues with date_fin < {cutoff_date}")
|
||||
|
||||
return {
|
||||
"message": f"Purged {count} catalogues expired more than 3 months ago",
|
||||
"cutoff_date": cutoff_date.isoformat(),
|
||||
"deleted_count": count
|
||||
}
|
||||
@@ -102,20 +102,60 @@ async def scrape_catalog_list(enseigne: Enseigne) -> list[dict[str, Any]]:
|
||||
async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Scrape pages from a specific catalog.
|
||||
Iterates through ?page=1, ?page=2 etc.
|
||||
Detects total pages from pagination links on first page.
|
||||
"""
|
||||
pages = []
|
||||
page_num = 1
|
||||
max_pages = 100 # Safety limit
|
||||
max_pages = 100 # Default safety limit
|
||||
|
||||
# First page: detect total number of pages from pagination
|
||||
first_page_html, _ = await browserless_service.get_page_content(catalog_url)
|
||||
|
||||
if not first_page_html:
|
||||
logger.error(f"Failed to fetch first page for {catalog_url}")
|
||||
return []
|
||||
|
||||
soup = BeautifulSoup(first_page_html, 'html.parser')
|
||||
|
||||
# Try to detect max pages from pagination links
|
||||
# Look for links with ?page=X or just numeric text
|
||||
pagination_links = soup.find_all('a', href=True)
|
||||
page_numbers = []
|
||||
|
||||
for link in pagination_links:
|
||||
href = link.get('href', '')
|
||||
text = link.get_text(strip=True)
|
||||
|
||||
# Check if it's a page number link
|
||||
if 'page=' in href:
|
||||
try:
|
||||
page_match = re.search(r'page=(\d+)', href)
|
||||
if page_match:
|
||||
page_numbers.append(int(page_match.group(1)))
|
||||
except:
|
||||
pass
|
||||
elif text.isdigit():
|
||||
# Direct numeric link (e.g., "11")
|
||||
page_numbers.append(int(text))
|
||||
|
||||
if page_numbers:
|
||||
max_pages = max(page_numbers)
|
||||
logger.info(f"Detected {max_pages} total pages from pagination")
|
||||
else:
|
||||
logger.warning(f"Could not detect page count, using default limit of {max_pages}")
|
||||
|
||||
# Now scrape all pages
|
||||
while page_num <= max_pages:
|
||||
# Construct URL for specific page
|
||||
current_url = catalog_url if page_num == 1 else f"{catalog_url}?page={page_num}"
|
||||
|
||||
logger.info(f"Scraping page {page_num}: {current_url}")
|
||||
logger.info(f"Scraping page {page_num}/{max_pages}: {current_url}")
|
||||
|
||||
# Use browserless service
|
||||
html_content, _ = await browserless_service.get_page_content(current_url)
|
||||
# Reuse first page HTML for page 1
|
||||
if page_num == 1:
|
||||
html_content = first_page_html
|
||||
else:
|
||||
html_content, _ = await browserless_service.get_page_content(current_url)
|
||||
|
||||
if not html_content:
|
||||
logger.warning(f"Failed to fetch page {page_num}")
|
||||
@@ -148,18 +188,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
|
||||
# Heuristic: Catalog pages are usually large vertical images
|
||||
# Check src for keywords
|
||||
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
|
||||
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet'])
|
||||
|
||||
# Logic:
|
||||
# 1. If keyword match AND decent size -> Strong candidate
|
||||
# 2. If no keyword match but HUGE size -> Candidate
|
||||
|
||||
if is_likely_catalog:
|
||||
if area > max_area:
|
||||
if area > max_area or (area == 0 and max_area == 0):
|
||||
max_area = area
|
||||
main_image_url = src
|
||||
elif area == 0 and not main_image_url:
|
||||
main_image_url = src
|
||||
elif area > 100000: # Arbitrary large size (e.g. 300x333)
|
||||
if area > max_area:
|
||||
max_area = area
|
||||
@@ -179,18 +217,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
logger.info(f"Found image for page {page_num}: {main_image_url}")
|
||||
else:
|
||||
logger.warning(f"No catalog image found on page {page_num}")
|
||||
# If we can't find an image, we might have reached the end or it's a bad page
|
||||
# If we can't find an image after page 1, we've likely reached the end
|
||||
if page_num > 1:
|
||||
logger.info(f"Stopping pagination at page {page_num - 1} (no image found)")
|
||||
break
|
||||
|
||||
# Check for pagination "Next" to know if we should continue
|
||||
if not main_image_url:
|
||||
break
|
||||
|
||||
page_num += 1
|
||||
# Small delay handled by browserless service already, but adding a tiny one here
|
||||
# Small delay to avoid overwhelming the server
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
logger.info(f"Scraped {len(pages)} pages total")
|
||||
return pages
|
||||
|
||||
async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
|
||||
|
||||
@@ -41,6 +41,33 @@ async def scheduled_scrape_job():
|
||||
db.close()
|
||||
|
||||
|
||||
async def scheduled_purge_job():
|
||||
"""Scheduled job to purge old catalogs (expired > 3 months)"""
|
||||
logger.info("Starting scheduled catalog purge job")
|
||||
|
||||
from datetime import datetime, timedelta
|
||||
from app.models import Catalogue
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
cutoff_date = datetime.now() - timedelta(days=90)
|
||||
|
||||
old_catalogues = db.query(Catalogue).filter(
|
||||
Catalogue.date_fin < cutoff_date
|
||||
).all()
|
||||
|
||||
count = len(old_catalogues)
|
||||
for cat in old_catalogues:
|
||||
db.delete(cat)
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Scheduled purge complete: deleted {count} catalogues with date_fin < {cutoff_date}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error in scheduled purge job: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def start_scheduler():
|
||||
"""Start the scheduler with configured jobs"""
|
||||
# Schedule scraping twice daily at 6:00 and 18:00
|
||||
@@ -48,12 +75,21 @@ def start_scheduler():
|
||||
scheduled_scrape_job,
|
||||
CronTrigger(hour="6,18", minute=0),
|
||||
id="catalog_scraping",
|
||||
name="Scrape Bonial catalogs",
|
||||
name="Scrape Cataloguemate catalogs",
|
||||
replace_existing=True,
|
||||
)
|
||||
|
||||
# Schedule automatic purge on the 1st of each month at 3:00 AM
|
||||
scheduler.add_job(
|
||||
scheduled_purge_job,
|
||||
CronTrigger(day=1, hour=3, minute=0),
|
||||
id="catalog_purge",
|
||||
name="Purge old catalogs",
|
||||
replace_existing=True,
|
||||
)
|
||||
|
||||
scheduler.start()
|
||||
logger.info("Catalog scraping scheduler started (runs at 6:00 and 18:00)")
|
||||
logger.info("Catalog scraping scheduler started (scraping at 6:00 and 18:00, purge on 1st of month at 3:00)")
|
||||
|
||||
|
||||
def stop_scheduler():
|
||||
|
||||
Reference in new issue
Block a user