diff --git a/app/routers/catalogues.py b/app/routers/catalogues.py index f00644b..34d0acf 100644 --- a/app/routers/catalogues.py +++ b/app/routers/catalogues.py @@ -24,7 +24,7 @@ from app.schemas_catalogues import ( ScrapingLogResponse, ScrapingStatsResponse, ) -from app.services.bonial_scraper import scrape_all_enseignes, scrape_enseigne +from app.services.tiendeo_scraper import scrape_all_enseignes, scrape_enseigne logger = logging.getLogger(__name__) diff --git a/app/services/bonial_scraper.py b/app/services/bonial_scraper.py index 366efd8..536e5b6 100644 --- a/app/services/bonial_scraper.py +++ b/app/services/bonial_scraper.py @@ -106,180 +106,117 @@ async def accept_cookies(page: Page) -> None: async def scrape_catalog_list(page: Page, enseigne: Enseigne) -> list[dict[str, Any]]: - """Scrape the list of catalogs for an enseigne from Bonial.""" + """Scrape the list of catalogs for an enseigne from Bonial using full-page scroll.""" url = f"https://www.bonial.fr/Enseignes/{enseigne.slug_bonial}" logger.info(f"Scraping catalog list for {enseigne.nom} from {url}") await page.goto(url, wait_until="networkidle") await accept_cookies(page) - # Find all "Ouvrir le catalogue" buttons/links using Playwright's get_by_text - try: - catalog_elements = await page.get_by_text("Ouvrir le catalogue").all() - logger.info(f"Found {len(catalog_elements)} catalog text elements for {enseigne.nom}") - - # Extract links from these elements - catalog_links = [] - for element in catalog_elements: - link_el = None - - # Strategy 1: Element itself is a link - tag_name = await element.evaluate("el => el.tagName.toLowerCase()") - if tag_name == "a": - link_el = element - - # Strategy 2: Find link parent - if not link_el: - link_parent = await element.evaluate_handle("el => el.closest('a')") - if link_parent: - link_el = link_parent.as_element() - - # Strategy 3: Find link child - if not link_el: - try: - link_child = await element.query_selector("a") - if link_child: - link_el = link_child - except: - pass - - # Strategy 4: Find link sibling (next or previous) - if not link_el: - try: - # Try next sibling - next_sibling = await element.evaluate_handle("el => el.nextElementSibling") - if next_sibling: - sibling_el = next_sibling.as_element() - if sibling_el: - sibling_tag = await sibling_el.evaluate("el => el.tagName.toLowerCase()") - if sibling_tag == "a": - link_el = sibling_el - except: - pass - - # Strategy 5: Go up to parent container and find ANY link - if not link_el: - try: - parent_container = await element.evaluate_handle("el => el.parentElement") - if parent_container: - parent_el = parent_container.as_element() - if parent_el: - link_in_parent = await parent_el.query_selector("a") - if link_in_parent: - link_el = link_in_parent - except: - pass - - if link_el: - catalog_links.append(link_el) - else: - logger.debug(f"Could not find link for element with text 'Ouvrir le catalogue'") - - logger.info(f"Found {len(catalog_links)} actual catalog links for {enseigne.nom}") - except Exception as e: - logger.error(f"No catalog links found for {enseigne.nom}: {e}") - return [] + # Scroll the page multiple times to trigger lazy loading + logger.info("Scrolling page to load all dynamic content...") + for i in range(5): # Scroll 5 times + await page.evaluate("window.scrollTo(0, document.body.scrollHeight)") + await asyncio.sleep(2) # Wait for content to load - if not catalog_links: - logger.warning(f"No catalogs found for {enseigne.nom}") - return [] + # Scroll back to top + await page.evaluate("window.scrollTo(0, 0)") + await asyncio.sleep(1) + + # Extract all catalog data using JavaScript to find patterns + catalogs_data = await page.evaluate(""" + () => { + const results = []; + + // Find all h2 elements which typically contain catalog titles + const h2Elements = document.querySelectorAll('h2'); + + h2Elements.forEach((h2, index) => { + const title = h2.textContent.trim(); + + // Skip generic titles + const genericTitles = ['restez informé', 'catalogues bazar', 'une enseigne', 'optimisez']; + if (genericTitles.some(generic => title.toLowerCase().includes(generic))) { + return; + } + + // Find parent container + let container = h2.parentElement; + for (let i = 0; i < 10 && container; i++) { + // Look for image in this container + const img = container.querySelector('img'); + // Look for link in this container + const link = container.querySelector('a[href*="/contentViewer/"]'); + + if (img && link) { + // Found a complete catalog card! + const imgSrc = img.src || img.getAttribute('data-src') || img.getAttribute('srcset')?.split(' ')[0]; + const href = link.href; + + // Try to extract dates from container text + const containerText = container.textContent; + const dateMatch = containerText.match(/(\d{1,2}[\/\-]\d{1,2}(?:[\/\-]\d{2,4})?)/g); + + results.push({ + title: title, + image: imgSrc, + url: href, + dates: dateMatch || [], + containerText: containerText.substring(0, 500) // First 500 chars for debugging + }); + break; // Found it, stop going up + } + + container = container.parentElement; + } + }); + + return results; + } + """) + + logger.info(f"Found {len(catalogs_data)} potential catalog cards for {enseigne.nom}") catalogues = [] - for idx, link in enumerate(catalog_links): + for idx, cat_data in enumerate(catalogs_data): try: - # Get the catalog URL - catalogue_url = await link.get_attribute("href") - if not catalogue_url: - logger.warning(f"Link {idx} has no href") - continue - - if not catalogue_url.startswith("http"): - catalogue_url = f"https://www.bonial.fr{catalogue_url}" + titre = cat_data.get('title', 'Catalogue') + image_url = cat_data.get('image') + catalogue_url = cat_data.get('url') + dates_list = cat_data.get('dates', []) + container_text = cat_data.get('containerText', '') - # Find the parent container that has the title and image - # Go up to find a section/div that contains both the link and the title/image - parent = link - for _ in range(5): # Try going up max 5 levels - parent = await parent.evaluate_handle("el => el.parentElement") - parent = parent.as_element() - if not parent: - break - - # Try to find title in this parent - try: - title_el = await parent.query_selector("h2, h3, [class*='title']") - if title_el: - break - except: - continue - - if not parent: - logger.warning(f"Could not find parent container for link {idx}") + if not catalogue_url or not image_url: + logger.warning(f"Skipping catalog {idx}: missing URL or image") continue - # Extract title from parent - titre = "Catalogue" - try: - title_el = await parent.query_selector("h2, h3, [class*='title']") - if title_el: - titre = await title_el.inner_text() - except: - pass - - # Extract dates from parent text - dates_text = await parent.inner_text() if parent else "" - - # Extract image from parent - image_url = None - try: - image_el = await parent.query_selector("img") - if image_el: - image_url = await image_el.get_attribute("src") - if image_url and image_url.startswith("//"): - image_url = "https:" + image_url - except: - pass - - # Filter generic titles - if titre: - lower_title = titre.lower() - ignored_titles = [ - "restez informé", "catalogues bazar", "toutes les offres", - "voir les offres", "téléchargez l'application", "newsletter" - ] - if any(ignored in lower_title for ignored in ignored_titles): - logger.info(f"Skipping generic/banner card: {titre}") - continue - - has_url = bool(catalogue_url) - has_image = bool(image_url) + # Ensure absolute URLs + if image_url and image_url.startswith('//'): + image_url = 'https:' + image_url + # Parse dates date_debut, date_fin = datetime.now(), datetime.now() try: - date_debut, date_fin = parse_bonial_dates(dates_text) + if dates_list: + dates_str = ' - '.join(dates_list) + date_debut, date_fin = parse_bonial_dates(dates_str) except ValueError: - if "semaine" in dates_text.lower() or "valable" in dates_text.lower(): - from datetime import timedelta - date_fin = datetime.now() + timedelta(days=7) + # Default to 1 week validity + from datetime import timedelta + date_fin = datetime.now() + timedelta(days=7) - if has_url and has_image: - catalogues.append({ - "titre": titre.strip(), - "date_debut": date_debut, - "date_fin": date_fin, - "image_couverture_url": image_url, - "catalogue_url": catalogue_url, - }) - logger.debug(f"Extracted catalog: {titre}") - else: - if not has_url: - logger.warning(f"Skipping {idx} (No URL): {titre}") - elif not has_image: - logger.warning(f"Skipping {idx} (No Image): {titre}") + catalogues.append({ + "titre": titre.strip(), + "date_debut": date_debut, + "date_fin": date_fin, + "image_couverture_url": image_url, + "catalogue_url": catalogue_url, + }) + logger.info(f"Extracted catalog: {titre}") except Exception as e: - logger.error(f"Error extracting catalog {idx} for {enseigne.nom}: {e}") + logger.error(f"Error processing catalog {idx} for {enseigne.nom}: {e}") continue logger.info(f"Successfully extracted {len(catalogues)} catalogs for {enseigne.nom}") diff --git a/app/services/tiendeo_scraper.py b/app/services/tiendeo_scraper.py new file mode 100644 index 0000000..ec6aff4 --- /dev/null +++ b/app/services/tiendeo_scraper.py @@ -0,0 +1,381 @@ +""" +Tiendeo.fr Scraper Service + +Scrapes promotional catalogs from Tiendeo.fr for configured enseignes. +Much simpler structure than Bonial - catalog cards are clearly defined with direct links. +""" + +import asyncio +import hashlib +import logging +import re +import os +from datetime import datetime, timedelta +from typing import Any + +from playwright.async_api import async_playwright, Page, Browser +from sqlalchemy.orm import Session + +from app.models import Enseigne, Catalogue, CataloguePage, ScrapingLog + +logger = logging.getLogger(__name__) + +# Tiendeo site base URL +TIENDEO_BASE_URL = "https://www.tiendeo.fr" + +# Date pattern for Tiendeo: "Expire le 31/12" or "mar. 25/11 - lun. 08/12" +DATE_PATTERN = r"(\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)" + + +def parse_tiendeo_dates(date_str: str) -> tuple[datetime, datetime]: + """ + Parse Tiendeo date format to datetime objects. + + Args: + date_str: Date string like "Expire le 31/12" or "mar. 25/11 - lun. 08/12" + + Returns: + Tuple of (date_debut, date_fin) + """ + matches = re.findall(DATE_PATTERN, date_str) + + if not matches: + logger.warning(f"No dates found in '{date_str}', using default duration") + raise ValueError(f"Invalid date format: {date_str}") + + today = datetime.now() + current_year = today.year + + parsed_dates = [] + for date_match in matches: + date_match = date_match.replace("-", "/") + parts = date_match.split("/") + + day = int(parts[0]) + month = int(parts[1]) + year = int(parts[2]) if len(parts) > 2 else current_year + + if year < 100: + year += 2000 + + # Handle year rollover + if len(parts) == 2: + if month < today.month - 3: + year += 1 + + try: + parsed_dates.append(datetime(year, month, day)) + except ValueError: + continue + + if not parsed_dates: + raise ValueError("No valid dates parsed") + + if len(parsed_dates) >= 2: + return min(parsed_dates), max(parsed_dates) + elif len(parsed_dates) == 1: + # If only one date (expire date), use today as start + return today, parsed_dates[0] + + return parsed_dates[0], parsed_dates[0] + + +def compute_catalog_hash(enseigne_id: int, titre: str, date_debut: datetime) -> str: + """Compute SHA256 hash for duplicate detection.""" + content = f"{enseigne_id}|{titre}|{date_debut.isoformat()}" + return hashlib.sha256(content.encode()).hexdigest() + + +async def scrape_catalog_list(page: Page, enseigne: Enseigne) -> list[dict[str, Any]]: + """Scrape the list of catalogs for an enseigne from Tiendeo.""" + # Tiendeo uses store URLs like: /Magasins/{city}/{enseigne-slug} + # We'll use Nancy as default city for now + url = f"{TIENDEO_BASE_URL}/Magasins/nancy/{enseigne.slug_bonial.lower()}" + logger.info(f"Scraping catalog list for {enseigne.nom} from {url}") + + await page.goto(url, wait_until="networkidle") + await asyncio.sleep(2) # Wait for dynamic content + + # Extract catalog data using JavaScript + catalogs_data = await page.evaluate(""" + () => { + const results = []; + + // Find all links to catalog pages (format: /Catalogues/{id}) + const catalogLinks = document.querySelectorAll('a[href*="/Catalogues/"]'); + + catalogLinks.forEach((link) => { + const href = link.href; + const catalogId = href.split('/Catalogues/')[1]; + + if (!catalogId) return; + + // Extract title from link text or nearby heading + let title = link.textContent.trim(); + + // Try to find h3 or h4 near this link for better title + const parentContainer = link.closest('div, section, article'); + if (parentContainer) { + const heading = parentContainer.querySelector('h3, h4, h2'); + if (heading && heading.textContent.trim().length > title.length) { + title = heading.textContent.trim(); + } + } + + // Look for image in the same container + let imgSrc = null; + if (parentContainer) { + const img = parentContainer.querySelector('img'); + if (img) { + imgSrc = img.src || img.getAttribute('data-src') || img.getAttribute('srcset')?.split(' ')[0]; + } + } + + // Extract date information from container text + const containerText = parentContainer ? parentContainer.textContent : link.textContent; + + // Avoid duplicates + if (!results.some(r => r.url === href)) { + results.push({ + title: title, + url: href, + image: imgSrc, + containerText: containerText.substring(0, 300), // For date parsing + catalogId: catalogId + }); + } + }); + + return results; + } + """) + + logger.info(f"Found {len(catalogs_data)} catalog links for {enseigne.nom}") + + catalogues = [] + + for idx, cat_data in enumerate(catalogs_data): + try: + titre = cat_data.get('title', 'Catalogue') + image_url = cat_data.get('image') + catalogue_url = cat_data.get('url') + container_text = cat_data.get('containerText', '') + + if not catalogue_url: + logger.warning(f"Skipping catalog {idx}: missing URL") + continue + + # Clean title - remove extra info like distances, "Gifi", etc. + titre = titre.replace('Gifi', '').replace('Action', '').replace('Centrakor', '') + titre = re.sub(r'\d+\.\d+\s*km', '', titre) # Remove "3.2 km" + titre = re.sub(r'Nancy', '', titre, flags=re.IGNORECASE) + titre = re.sub(r'Expire\s+le.*', '', titre) # Remove expire info from title + titre = titre.strip() + + # Skip if title is empty or too short + if len(titre) < 3: + logger.warning(f"Skipping catalog {idx}: title too short after cleaning") + continue + + # Ensure absolute URL + if image_url and image_url.startswith('//'): + image_url = 'https:' + image_url + + # Parse dates from container text + date_debut, date_fin = datetime.now(), datetime.now() + timedelta(days=30) + try: + date_debut, date_fin = parse_tiendeo_dates(container_text) + except ValueError: + # Default to 30 days validity + logger.debug(f"Could not parse dates for '{titre}', using 30 days default") + + catalogues.append({ + "titre": titre, + "date_debut": date_debut, + "date_fin": date_fin, + "image_couverture_url": image_url or "", + "catalogue_url": catalogue_url, + }) + logger.info(f"Extracted catalog: {titre}") + + except Exception as e: + logger.error(f"Error processing catalog {idx} for {enseigne.nom}: {e}") + continue + + logger.info(f"Successfully extracted {len(catalogues)} catalogs for {enseigne.nom}") + return catalogues + + +async def scrape_catalog_pages(page: Page, catalogue_url: str) -> list[dict[str, Any]]: + """Scrape all pages of a catalog from the Tiendeo viewer.""" + logger.info(f"Scraping catalog pages from {catalogue_url}") + + await page.goto(catalogue_url, wait_until="networkidle") + await asyncio.sleep(3) # Wait for viewer to load + + # Find all catalog page images + images = await page.query_selector_all("img[src*='tiendeo'], img[src*='cdn']") + + if not images: + logger.warning(f"No images found in catalog viewer: {catalogue_url}") + return [] + + seen_urls = set() + pages = [] + + for idx, img in enumerate(images): + try: + src = await img.get_attribute("src") + + if src and src not in seen_urls: + width = await img.evaluate("el => el.naturalWidth") + height = await img.evaluate("el => el.naturalHeight") + + # Skip small images (thumbnails) + if width < 200 or height < 200: + continue + + seen_urls.add(src) + pages.append({ + "numero_page": len(pages) + 1, + "image_url": src, + "largeur": width, + "hauteur": height, + }) + + except Exception as e: + logger.debug(f"Error processing image {idx}: {e}") + continue + + logger.info(f"Found {len(pages)} pages in catalog") + return pages + + +async def scrape_enseigne(enseigne: Enseigne, db: Session, browser: Browser | None = None) -> ScrapingLog: + """Scrape all catalogs for a specific enseigne from Tiendeo.""" + start_time = datetime.now() + log = ScrapingLog( + enseigne_id=enseigne.id, + statut="error", + catalogues_trouves=0, + catalogues_nouveaux=0, + catalogues_mis_a_jour=0, + ) + + own_browser = browser is None + + try: + if own_browser: + browserless_url = os.environ.get("BROWSERLESS_URL", "ws://browserless:3000") + async with async_playwright() as p: + browser = await p.chromium.connect_over_cdp(browserless_url) + + page = await browser.new_page() + + await page.set_extra_http_headers({ + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" + }) + + catalogues_data = await scrape_catalog_list(page, enseigne) + log.catalogues_trouves = len(catalogues_data) + + if not catalogues_data: + log.statut = "success" + log.message_erreur = "No catalogs found (may be normal)" + await page.close() + return log + + for cat_data in catalogues_data: + try: + content_hash = compute_catalog_hash( + enseigne.id, + cat_data["titre"], + cat_data["date_debut"] + ) + + existing = db.query(Catalogue).filter_by(content_hash=content_hash).first() + + if existing: + logger.debug(f"Catalog already exists: {cat_data['titre']}") + continue + + await asyncio.sleep(2) # Respectful delay + pages_data = await scrape_catalog_pages(page, cat_data["catalogue_url"]) + + catalogue = Catalogue( + enseigne_id=enseigne.id, + titre=cat_data["titre"], + date_debut=cat_data["date_debut"], + date_fin=cat_data["date_fin"], + image_couverture_url=cat_data["image_couverture_url"], + catalogue_url=cat_data["catalogue_url"], + statut="actif", + nombre_pages=len(pages_data), + content_hash=content_hash, + ) + db.add(catalogue) + db.flush() + + for page_data in pages_data: + catalogue_page = CataloguePage( + catalogue_id=catalogue.id, + numero_page=page_data["numero_page"], + image_url=page_data["image_url"], + largeur=page_data["largeur"], + hauteur=page_data["hauteur"], + ) + db.add(catalogue_page) + + log.catalogues_nouveaux += 1 + logger.info(f"Added new catalog: {cat_data['titre']}") + + except Exception as e: + logger.error(f"Error processing catalog: {e}") + continue + + db.commit() + log.statut = "success" + await page.close() + + except Exception as e: + logger.error(f"Error scraping {enseigne.nom}: {e}") + log.statut = "error" + log.message_erreur = str(e) + db.rollback() + + finally: + if own_browser and browser: + await browser.close() + + log.duree_secondes = (datetime.now() - start_time).total_seconds() + db.add(log) + db.commit() + + return log + + +async def scrape_all_enseignes(db: Session) -> list[ScrapingLog]: + """Scrape all active enseignes from Tiendeo.""" + enseignes = db.query(Enseigne).filter_by(is_active=True).all() + logger.info(f"Starting Tiendeo scraping for {len(enseignes)} active enseignes") + + logs = [] + + browserless_url = os.environ.get("BROWSERLESS_URL", "ws://browserless:3000") + logger.info(f"Connecting to Browserless at {browserless_url}") + + async with async_playwright() as p: + browser = await p.chromium.connect_over_cdp(browserless_url) + + for enseigne in enseignes: + try: + await asyncio.sleep(2) + log = await scrape_enseigne(enseigne, db, browser) + logs.append(log) + except Exception as e: + logger.error(f"Failed to scrape {enseigne.nom}: {e}") + + await browser.close() + + logger.info(f"Tiendeo scraping complete: {len(logs)} enseignes processed") + return logs