Merge pull request #159 from R0m1k3/antigravity

Antigravity
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 10:50:10 +01:00
commit 15955edfa0
5 files changed
+227 -26

No files matched your search

+52
View File
@@ -322,3 +322,55 @@ async def cleanup_catalogs(
logger.info(f"Cleanup: deleted {deleted_count} invalid catalogs")
return {"message": f"Cleanup complete. Deleted {deleted_count} invalid catalogs."}
@router.delete("/admin/{catalogue_id}")
async def delete_catalogue(
catalogue_id: int,
db: Session = Depends(get_db),
current_user: User = Depends(get_current_user),
):
"""Delete a specific catalogue (admin only)"""
if not current_user.is_admin:
raise HTTPException(status_code=403, detail="Not authorized")
catalogue = db.query(Catalogue).filter(Catalogue.id == catalogue_id).first()
if not catalogue:
raise HTTPException(status_code=404, detail="Catalogue not found")
# Pages are cascade-deleted automatically due to relationship
db.delete(catalogue)
db.commit()
logger.info(f"Deleted catalogue {catalogue_id} ({catalogue.titre})")
return {"message": f"Catalogue '{catalogue.titre}' deleted successfully"}
@router.post("/admin/purge")
async def purge_old_catalogues(
db: Session = Depends(get_db),
current_user: User = Depends(get_current_user),
):
"""Purge catalogues expired more than 3 months ago (admin only)"""
if not current_user.is_admin:
raise HTTPException(status_code=403, detail="Not authorized")
from datetime import timedelta
cutoff_date = datetime.now() - timedelta(days=90)
old_catalogues = db.query(Catalogue).filter(
Catalogue.date_fin < cutoff_date
).all()
count = len(old_catalogues)
for cat in old_catalogues:
db.delete(cat)
db.commit()
logger.info(f"Purged {count} catalogues with date_fin < {cutoff_date}")
return {
"message": f"Purged {count} catalogues expired more than 3 months ago",
"cutoff_date": cutoff_date.isoformat(),
"deleted_count": count
}
+53 -16
View File
@@ -102,20 +102,60 @@ async def scrape_catalog_list(enseigne: Enseigne) -> list[dict[str, Any]]:
async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
"""
Scrape pages from a specific catalog.
Iterates through ?page=1, ?page=2 etc.
Detects total pages from pagination links on first page.
"""
pages = []
page_num = 1
max_pages = 100 # Safety limit
max_pages = 100 # Default safety limit
# First page: detect total number of pages from pagination
first_page_html, _ = await browserless_service.get_page_content(catalog_url)
if not first_page_html:
logger.error(f"Failed to fetch first page for {catalog_url}")
return []
soup = BeautifulSoup(first_page_html, 'html.parser')
# Try to detect max pages from pagination links
# Look for links with ?page=X or just numeric text
pagination_links = soup.find_all('a', href=True)
page_numbers = []
for link in pagination_links:
href = link.get('href', '')
text = link.get_text(strip=True)
# Check if it's a page number link
if 'page=' in href:
try:
page_match = re.search(r'page=(\d+)', href)
if page_match:
page_numbers.append(int(page_match.group(1)))
except:
pass
elif text.isdigit():
# Direct numeric link (e.g., "11")
page_numbers.append(int(text))
if page_numbers:
max_pages = max(page_numbers)
logger.info(f"Detected {max_pages} total pages from pagination")
else:
logger.warning(f"Could not detect page count, using default limit of {max_pages}")
# Now scrape all pages
while page_num <= max_pages:
# Construct URL for specific page
current_url = catalog_url if page_num == 1 else f"{catalog_url}?page={page_num}"
logger.info(f"Scraping page {page_num}: {current_url}")
logger.info(f"Scraping page {page_num}/{max_pages}: {current_url}")
# Use browserless service
html_content, _ = await browserless_service.get_page_content(current_url)
# Reuse first page HTML for page 1
if page_num == 1:
html_content = first_page_html
else:
html_content, _ = await browserless_service.get_page_content(current_url)
if not html_content:
logger.warning(f"Failed to fetch page {page_num}")
@@ -148,18 +188,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
# Heuristic: Catalog pages are usually large vertical images
# Check src for keywords
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images'])
is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet'])
# Logic:
# 1. If keyword match AND decent size -> Strong candidate
# 2. If no keyword match but HUGE size -> Candidate
if is_likely_catalog:
if area > max_area:
if area > max_area or (area == 0 and max_area == 0):
max_area = area
main_image_url = src
elif area == 0 and not main_image_url:
main_image_url = src
elif area > 100000: # Arbitrary large size (e.g. 300x333)
if area > max_area:
max_area = area
@@ -179,18 +217,16 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]:
logger.info(f"Found image for page {page_num}: {main_image_url}")
else:
logger.warning(f"No catalog image found on page {page_num}")
# If we can't find an image, we might have reached the end or it's a bad page
# If we can't find an image after page 1, we've likely reached the end
if page_num > 1:
logger.info(f"Stopping pagination at page {page_num - 1} (no image found)")
break
# Check for pagination "Next" to know if we should continue
if not main_image_url:
break
page_num += 1
# Small delay handled by browserless service already, but adding a tiny one here
# Small delay to avoid overwhelming the server
await asyncio.sleep(0.5)
logger.info(f"Scraped {len(pages)} pages total")
return pages
async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
@@ -242,7 +278,8 @@ async def scrape_enseigne(enseigne: Enseigne, db: Session) -> ScrapingLog:
date_debut=datetime.now(), # Todo: extract real dates
date_fin=datetime.now() + timedelta(days=14),
image_couverture_url=pages[0]['image_url'],
content_hash=content_hash
content_hash=content_hash,
nombre_pages=len(pages) # Set actual page count
)
db.add(new_cat)
db.commit()
+38 -2
View File
@@ -41,6 +41,33 @@ async def scheduled_scrape_job():
db.close()
async def scheduled_purge_job():
"""Scheduled job to purge old catalogs (expired > 3 months)"""
logger.info("Starting scheduled catalog purge job")
from datetime import datetime, timedelta
from app.models import Catalogue
db = SessionLocal()
try:
cutoff_date = datetime.now() - timedelta(days=90)
old_catalogues = db.query(Catalogue).filter(
Catalogue.date_fin < cutoff_date
).all()
count = len(old_catalogues)
for cat in old_catalogues:
db.delete(cat)
db.commit()
logger.info(f"Scheduled purge complete: deleted {count} catalogues with date_fin < {cutoff_date}")
except Exception as e:
logger.error(f"Error in scheduled purge job: {e}")
finally:
db.close()
def start_scheduler():
"""Start the scheduler with configured jobs"""
# Schedule scraping twice daily at 6:00 and 18:00
@@ -48,12 +75,21 @@ def start_scheduler():
scheduled_scrape_job,
CronTrigger(hour="6,18", minute=0),
id="catalog_scraping",
name="Scrape Bonial catalogs",
name="Scrape Cataloguemate catalogs",
replace_existing=True,
)
# Schedule automatic purge on the 1st of each month at 3:00 AM
scheduler.add_job(
scheduled_purge_job,
CronTrigger(day=1, hour=3, minute=0),
id="catalog_purge",
name="Purge old catalogs",
replace_existing=True,
)
scheduler.start()
logger.info("Catalog scraping scheduler started (runs at 6:00 and 18:00)")
logger.info("Catalog scraping scheduler started (scraping at 6:00 and 18:00, purge on 1st of month at 3:00)")
def stop_scheduler():
@@ -1,7 +1,7 @@
import React, { useState, useEffect } from 'react';
import axios from 'axios';
import { toast } from 'sonner';
import { Loader2, Play, Clock, CheckCircle2, XCircle, AlertCircle } from 'lucide-react';
import { Loader2, Play, Clock, CheckCircle2, XCircle, AlertCircle, Trash2 } from 'lucide-react';
import { useAuth } from '../../hooks/use-auth';
const API_BASE = '/api/catalogues/admin';
@@ -77,6 +77,30 @@ export default function CatalogueAdmin() {
}
};
const triggerPurge = async () => {
if (!confirm('Êtes-vous sûr de vouloir purger les catalogues expirés depuis plus de 3 mois ?')) {
return;
}
setLoading(true);
try {
const response = await axios.post(`${API_BASE}/purge`, null, {
headers: getAuthHeaders()
});
toast.success(response.data.message);
toast.info(`${response.data.deleted_count} catalogue(s) supprimé(s)`);
// Reload data
await Promise.all([loadStats(), loadLogs()]);
} catch (error) {
console.error('Error triggering purge:', error);
toast.error('Erreur lors de la purge');
} finally {
setLoading(false);
}
};
const getStatusIcon = (status) => {
switch (status) {
case 'success':
@@ -97,9 +121,28 @@ export default function CatalogueAdmin() {
return (
<div className="space-y-6">
<div>
<h2 className="text-2xl font-bold mb-2">Administration Catalogues</h2>
<p className="text-muted-foreground">Gérer le scraping des catalogues Bonial</p>
<div className="flex justify-between items-center">
<div>
<h2 className="text-2xl font-bold mb-2">Administration Catalogues</h2>
<p className="text-muted-foreground">Gérer le scraping des catalogues Cataloguemate</p>
</div>
<button
onClick={triggerPurge}
disabled={loading}
className="px-4 py-2 bg-red-600 text-white rounded-lg hover:bg-red-700 disabled:opacity-50 flex items-center gap-2"
>
{loading ? (
<>
<Loader2 className="h-4 w-4 animate-spin" />
Purge...
</>
) : (
<>
<Trash2 className="h-4 w-4" />
Purger (&gt; 3 mois)
</>
)}
</button>
</div>
{/* Stats Cards */}
+37 -4
View File
@@ -1,7 +1,8 @@
import React, { useEffect, useState } from 'react';
import { toast } from 'sonner';
import { Loader2, Calendar, BookOpen, Search, ChevronRight, Eye } from 'lucide-react';
import { Loader2, Calendar, BookOpen, Search, ChevronRight, Eye, Trash2 } from 'lucide-react';
import axios from 'axios';
import { useAuth } from '../hooks/use-auth';
const API_BASE = '/api/catalogues';
@@ -14,6 +15,7 @@ export default function Catalogues() {
const [selectedCatalogue, setSelectedCatalogue] = useState(null);
const [cataloguePages, setCataloguePages] = useState([]);
const [searchQuery, setSearchQuery] = useState('');
const { user, getAuthHeaders } = useAuth();
useEffect(() => {
loadEnseignes();
@@ -104,6 +106,26 @@ export default function Catalogues() {
return now >= debut && now <= fin;
};
const deleteCatalogue = async (catalogueId, event) => {
event.stopPropagation(); // Prevent opening the catalogue
if (!confirm('Êtes-vous sûr de vouloir supprimer ce catalogue ?')) {
return;
}
try {
await axios.delete(`${API_BASE}/admin/${catalogueId}`, {
headers: getAuthHeaders()
});
toast.success('Catalogue supprimé');
loadCatalogues(); // Reload the list
} catch (error) {
console.error('Error deleting catalogue:', error);
toast.error('Erreur lors de la suppression');
}
};
return (
<div className="p-6 max-w-7xl mx-auto">
{/* Header */}
@@ -244,9 +266,20 @@ export default function Catalogues() {
</div>
{/* Pages */}
<div className="flex items-center gap-1 text-sm text-muted-foreground">
<BookOpen className="h-4 w-4" />
<span>{catalogue.nombre_pages} pages</span>
<div className="flex items-center justify-between">
<div className="flex items-center gap-1 text-sm text-muted-foreground">
<BookOpen className="h-4 w-4" />
<span>{catalogue.nombre_pages} pages</span>
</div>
{user?.is_admin && (
<button
onClick={(e) => deleteCatalogue(catalogue.id, e)}
className="p-2 text-red-500 hover:bg-red-50 rounded-lg transition-colors"
title="Supprimer le catalogue"
>
<Trash2 className="h-4 w-4" />
</button>
)}
</div>
</div>
</div>