mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
5 files changed
+227
-26
No files matched your search
@@ -322,3 +322,55 @@ async def cleanup_catalogs(
|
||||
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
|
||||
|
||||
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
|
||||
|
||||
|
||||
@router.delete("/admin/{catalogue_id}")
|
||||
async def delete_catalogue(
|
||||
catalogue_id: int,
|
||||
db: Session = Depends(get_db),
|
||||
current_user: User = Depends(get_current_user),
|
||||
):
|
||||
"""Delete a specific catalogue (admin only)"""
|
||||
if not current_user.is_admin:
|
||||
raise HTTPException(status_code=403, detail="Not authorized")
|
||||
|
||||
catalogue = db.query(Catalogue).filter(Catalogue.id == catalogue_id).first()
|
||||
if not catalogue:
|
||||
raise HTTPException(status_code=404, detail="Catalogue not found")
|
||||
|
||||
# Pages are cascade-deleted automatically due to relationship
|
||||
db.delete(catalogue)
|
||||
db.commit()
|
||||
logger.info(f"Deleted catalogue {catalogue_id} ({catalogue.titre})")
|
||||
|
||||
return {"message": f"Catalogue '{catalogue.titre}' deleted successfully"}
|
||||
|
||||
|
||||
@router.post("/admin/purge")
|
||||
async def purge_old_catalogues(
|
||||
db: Session = Depends(get_db),
|
||||
current_user: User = Depends(get_current_user),
|
||||
):
|
||||
"""Purge catalogues expired more than 3 months ago (admin only)"""
|
||||
if not current_user.is_admin:
|
||||
raise HTTPException(status_code=403, detail="Not authorized")
|
||||
|
||||
from datetime import timedelta
|
||||
cutoff_date = datetime.now() - timedelta(days=90)
|
||||
|
||||
old_catalogues = db.query(Catalogue).filter(
|
||||
Catalogue.date_fin < cutoff_date
|
||||
).all()
|
||||
|
||||
count = len(old_catalogues)
|
||||
for cat in old_catalogues:
|
||||
db.delete(cat)
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Purged {count} catalogues with date_fin < {cutoff_date}")
|
||||
|
||||
return {
|
||||
"message": f"Purged {count} catalogues expired more than 3 months ago",
|
||||
"cutoff_date": cutoff_date.isoformat(),
|
||||
"deleted_count": count
|
||||
}
|
||||
@@ -102,20 +102,60 @@ async def scrape_catalog_list(enseigne: Enseigne) -> list[dict[str, Any]]:
|
||||
async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Scrape pages from a specific catalog.
|
||||
Iterates through ?page=1, ?page=2 etc.
|
||||
Detects total pages from pagination links on first page.
|
||||
"""
|
||||
pages = []
|
||||
page_num = 1
|
||||
max_pages = 100 # Safety limit
|
||||
max_pages = 100 # Default safety limit
|
||||
|
||||
# First page: detect total number of pages from pagination
|
||||
first_page_html, _ = await browserless_service.get_page_content(catalog_url)
|
||||
|
||||
if not first_page_html:
|
||||
logger.error(f"Failed to fetch first page for {catalog_url}")
|
||||
return []
|
||||
|
||||
soup = BeautifulSoup(first_page_html, 'html.parser')
|
||||
|
||||
# Try to detect max pages from pagination links
|
||||
# Look for links with ?page=X or just numeric text
|
||||
pagination_links = soup.find_all('a', href=True)
|
||||
page_numbers = []
|
||||
|
||||
for link in pagination_links:
|
||||
href = link.get('href', '')
|
||||
text = link.get_text(strip=True)
|
||||
|
||||
# Check if it's a page number link
|
||||
if 'page=' in href:
|
||||
try:
|
||||
page_match = re.search(r'page=(\d+)', href)
|
||||
if page_match:
|
||||
page_numbers.append(int(page_match.group(1)))
|
||||
except:
|
||||
pass
|
||||
elif text.isdigit():
|
||||
# Direct numeric link (e.g., "11")
|
||||
page_numbers.append(int(text))
|
||||
|
||||
if page_numbers:
|
||||
max_pages = max(page_numbers)
|
||||
logger.info(f"Detected {max_pages} total pages from pagination")
|
||||
else:
|
||||
logger.warning(f"Could not detect page count, using default limit of {max_pages}")
|
||||
|
||||
# Now scrape all pages
|
||||
while page_num <= max_pages:
|
||||
# Construct URL for specific page
|
||||
current_url = catalog_url if page_num == 1 else f"{catalog_url}?page={page_num}"
|
||||
|
||||
logger.info(f"Scraping page {page_num}: {current_url}")
|
||||
logger.info(f"Scraping page {page_num}/{max_pages}: {current_url}")
|
||||
|
||||
# Use browserless service
|
||||
html_content, _ = await browserless_service.get_page_content(current_url)
|
||||
# Reuse first page HTML for page 1
|
||||
if page_num == 1:
|
||||
html_content = first_page_html
|
||||
else:
|
||||
html_content, _ = await browserless_service.get_page_content(current_url)
|
||||
|
||||
if not html_content:
|
||||
logger.warning(f"Failed to fetch page {page_num}")
|
||||
@@ -148,18 +188,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
|
||||
# Heuristic: Catalog pages are usually large vertical images
|
||||
# Check src for keywords
|
||||
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
|
||||
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet'])
|
||||
|
||||
# Logic:
|
||||
# 1. If keyword match AND decent size -> Strong candidate
|
||||
# 2. If no keyword match but HUGE size -> Candidate
|
||||
|
||||
if is_likely_catalog:
|
||||
if area > max_area:
|
||||
if area > max_area or (area == 0 and max_area == 0):
|
||||
max_area = area
|
||||
main_image_url = src
|
||||
elif area == 0 and not main_image_url:
|
||||
main_image_url = src
|
||||
elif area > 100000: # Arbitrary large size (e.g. 300x333)
|
||||
if area > max_area:
|
||||
max_area = area
|
||||
@@ -179,18 +217,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
|
||||
logger.info(f"Found image for page {page_num}: {main_image_url}")
|
||||
else:
|
||||
logger.warning(f"No catalog image found on page {page_num}")
|
||||
# If we can't find an image, we might have reached the end or it's a bad page
|
||||
# If we can't find an image after page 1, we've likely reached the end
|
||||
if page_num > 1:
|
||||
logger.info(f"Stopping pagination at page {page_num - 1} (no image found)")
|
||||
break
|
||||
|
||||
# Check for pagination "Next" to know if we should continue
|
||||
if not main_image_url:
|
||||
break
|
||||
|
||||
page_num += 1
|
||||
# Small delay handled by browserless service already, but adding a tiny one here
|
||||
# Small delay to avoid overwhelming the server
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
logger.info(f"Scraped {len(pages)} pages total")
|
||||
return pages
|
||||
|
||||
async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
|
||||
@@ -242,7 +278,8 @@ async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
|
||||
date_debut=datetime.now(), # Todo: extract real dates
|
||||
date_fin=datetime.now() + timedelta(days=14),
|
||||
image_couverture_url=pages[0]['image_url'],
|
||||
content_hash=content_hash
|
||||
content_hash=content_hash,
|
||||
nombre_pages=len(pages) # Set actual page count
|
||||
)
|
||||
db.add(new_cat)
|
||||
db.commit()
|
||||
|
||||
@@ -41,6 +41,33 @@ async def scheduled_scrape_job():
|
||||
db.close()
|
||||
|
||||
|
||||
async def scheduled_purge_job():
|
||||
"""Scheduled job to purge old catalogs (expired > 3 months)"""
|
||||
logger.info("Starting scheduled catalog purge job")
|
||||
|
||||
from datetime import datetime, timedelta
|
||||
from app.models import Catalogue
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
cutoff_date = datetime.now() - timedelta(days=90)
|
||||
|
||||
old_catalogues = db.query(Catalogue).filter(
|
||||
Catalogue.date_fin < cutoff_date
|
||||
).all()
|
||||
|
||||
count = len(old_catalogues)
|
||||
for cat in old_catalogues:
|
||||
db.delete(cat)
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Scheduled purge complete: deleted {count} catalogues with date_fin < {cutoff_date}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error in scheduled purge job: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def start_scheduler():
|
||||
"""Start the scheduler with configured jobs"""
|
||||
# Schedule scraping twice daily at 6:00 and 18:00
|
||||
@@ -48,12 +75,21 @@ def start_scheduler():
|
||||
scheduled_scrape_job,
|
||||
CronTrigger(hour="6,18", minute=0),
|
||||
id="catalog_scraping",
|
||||
name="Scrape Bonial catalogs",
|
||||
name="Scrape Cataloguemate catalogs",
|
||||
replace_existing=True,
|
||||
)
|
||||
|
||||
# Schedule automatic purge on the 1st of each month at 3:00 AM
|
||||
scheduler.add_job(
|
||||
scheduled_purge_job,
|
||||
CronTrigger(day=1, hour=3, minute=0),
|
||||
id="catalog_purge",
|
||||
name="Purge old catalogs",
|
||||
replace_existing=True,
|
||||
)
|
||||
|
||||
scheduler.start()
|
||||
logger.info("Catalog scraping scheduler started (runs at 6:00 and 18:00)")
|
||||
logger.info("Catalog scraping scheduler started (scraping at 6:00 and 18:00, purge on 1st of month at 3:00)")
|
||||
|
||||
|
||||
def stop_scheduler():
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import React, { useState, useEffect } from 'react';
|
||||
import axios from 'axios';
|
||||
import { toast } from 'sonner';
|
||||
import { Loader2, Play, Clock, CheckCircle2, XCircle, AlertCircle } from 'lucide-react';
|
||||
import { Loader2, Play, Clock, CheckCircle2, XCircle, AlertCircle, Trash2 } from 'lucide-react';
|
||||
import { useAuth } from '../../hooks/use-auth';
|
||||
|
||||
const API_BASE = '/api/catalogues/admin';
|
||||
@@ -77,6 +77,30 @@ export default function CatalogueAdmin() {
|
||||
}
|
||||
};
|
||||
|
||||
const triggerPurge = async () => {
|
||||
if (!confirm('Êtes-vous sûr de vouloir purger les catalogues expirés depuis plus de 3 mois ?')) {
|
||||
return;
|
||||
}
|
||||
|
||||
setLoading(true);
|
||||
try {
|
||||
const response = await axios.post(`${API_BASE}/purge`, null, {
|
||||
headers: getAuthHeaders()
|
||||
});
|
||||
|
||||
toast.success(response.data.message);
|
||||
toast.info(`${response.data.deleted_count} catalogue(s) supprimé(s)`);
|
||||
|
||||
// Reload data
|
||||
await Promise.all([loadStats(), loadLogs()]);
|
||||
} catch (error) {
|
||||
console.error('Error triggering purge:', error);
|
||||
toast.error('Erreur lors de la purge');
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const getStatusIcon = (status) => {
|
||||
switch (status) {
|
||||
case 'success':
|
||||
@@ -97,9 +121,28 @@ export default function CatalogueAdmin() {
|
||||
|
||||
return (
|
||||
<div className="space-y-6">
|
||||
<div>
|
||||
<h2 className="text-2xl font-bold mb-2">Administration Catalogues</h2>
|
||||
<p className="text-muted-foreground">Gérer le scraping des catalogues Bonial</p>
|
||||
<div className="flex justify-between items-center">
|
||||
<div>
|
||||
<h2 className="text-2xl font-bold mb-2">Administration Catalogues</h2>
|
||||
<p className="text-muted-foreground">Gérer le scraping des catalogues Cataloguemate</p>
|
||||
</div>
|
||||
<button
|
||||
onClick={triggerPurge}
|
||||
disabled={loading}
|
||||
className="px-4 py-2 bg-red-600 text-white rounded-lg hover:bg-red-700 disabled:opacity-50 flex items-center gap-2"
|
||||
>
|
||||
{loading ? (
|
||||
<>
|
||||
<Loader2 className="h-4 w-4 animate-spin" />
|
||||
Purge...
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<Trash2 className="h-4 w-4" />
|
||||
Purger (> 3 mois)
|
||||
</>
|
||||
)}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Stats Cards */}
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
import React, { useEffect, useState } from 'react';
|
||||
import { toast } from 'sonner';
|
||||
import { Loader2, Calendar, BookOpen, Search, ChevronRight, Eye } from 'lucide-react';
|
||||
import { Loader2, Calendar, BookOpen, Search, ChevronRight, Eye, Trash2 } from 'lucide-react';
|
||||
import axios from 'axios';
|
||||
import { useAuth } from '../hooks/use-auth';
|
||||
|
||||
const API_BASE = '/api/catalogues';
|
||||
|
||||
@@ -14,6 +15,7 @@ export default function Catalogues() {
|
||||
const [selectedCatalogue, setSelectedCatalogue] = useState(null);
|
||||
const [cataloguePages, setCataloguePages] = useState([]);
|
||||
const [searchQuery, setSearchQuery] = useState('');
|
||||
const { user, getAuthHeaders } = useAuth();
|
||||
|
||||
useEffect(() => {
|
||||
loadEnseignes();
|
||||
@@ -104,6 +106,26 @@ export default function Catalogues() {
|
||||
return now >= debut && now <= fin;
|
||||
};
|
||||
|
||||
const deleteCatalogue = async (catalogueId, event) => {
|
||||
event.stopPropagation(); // Prevent opening the catalogue
|
||||
|
||||
if (!confirm('Êtes-vous sûr de vouloir supprimer ce catalogue ?')) {
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
await axios.delete(`${API_BASE}/admin/${catalogueId}`, {
|
||||
headers: getAuthHeaders()
|
||||
});
|
||||
|
||||
toast.success('Catalogue supprimé');
|
||||
loadCatalogues(); // Reload the list
|
||||
} catch (error) {
|
||||
console.error('Error deleting catalogue:', error);
|
||||
toast.error('Erreur lors de la suppression');
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="p-6 max-w-7xl mx-auto">
|
||||
{/* Header */}
|
||||
@@ -244,9 +266,20 @@ export default function Catalogues() {
|
||||
</div>
|
||||
|
||||
{/* Pages */}
|
||||
<div className="flex items-center gap-1 text-sm text-muted-foreground">
|
||||
<BookOpen className="h-4 w-4" />
|
||||
<span>{catalogue.nombre_pages} pages</span>
|
||||
<div className="flex items-center justify-between">
|
||||
<div className="flex items-center gap-1 text-sm text-muted-foreground">
|
||||
<BookOpen className="h-4 w-4" />
|
||||
<span>{catalogue.nombre_pages} pages</span>
|
||||
</div>
|
||||
{user?.is_admin && (
|
||||
<button
|
||||
onClick={(e) => deleteCatalogue(catalogue.id, e)}
|
||||
className="p-2 text-red-500 hover:bg-red-50 rounded-lg transition-colors"
|
||||
title="Supprimer le catalogue"
|
||||
>
|
||||
<Trash2 className="h-4 w-4" />
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
Reference in new issue
Block a user