From 7842eead4a1c7b3ee9ff58b02684b25cb5bef3dd Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Mon, 22 Dec 2025 13:49:27 +0100 Subject: [PATCH] feat: add `debug_catalog_images.py` to analyze catalog image selectors and update `task.md` to reflect the new focus on catalog image fixes. --- app/services/cataloguemate_scraper.py | 8 +++++--- task.md | 19 ++++++++++--------- walkthrough.md | 21 ++++++++++++++------- 3 files changed, 29 insertions(+), 19 deletions(-) diff --git a/app/services/cataloguemate_scraper.py b/app/services/cataloguemate_scraper.py index 7410514..098a1b7 100644 --- a/app/services/cataloguemate_scraper.py +++ b/app/services/cataloguemate_scraper.py @@ -207,11 +207,13 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]: all_imgs = soup.find_all('img') for img in all_imgs: - src = img.get('src') + # Check multiple attributes for the real image URL (lazy loading) + src = img.get('data-src') or img.get('data-original') or img.get('src') + if not src: continue # Skip common UI elements - if any(x in src.lower() for x in ['logo', 'icon', 'facebook', 'twitter', 'instagram']): + if any(x in src.lower() for x in ['logo', 'icon', 'facebook', 'twitter', 'instagram', 'loader']): continue # Calculate area if dimensions exist @@ -226,7 +228,7 @@ async def scrape_catalog_pages(catalog_url: str) -> list[dict[str, Any]]: # Heuristic: Catalog pages are usually large vertical images # Check src for keywords - is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet']) + is_likely_catalog = any(k in src.lower() for k in ['page', 'flyer', 'catalog', 'upload', 'images', 'leaflet', 'thumbor']) # Logic: # 1. If keyword match AND decent size -> Strong candidate diff --git a/task.md b/task.md index 1fe2169..5474fbe 100644 --- a/task.md +++ b/task.md @@ -2,25 +2,26 @@ ## 🚀 Current Focus -- [ ] Debug Catalog Scraper +- [ ] Fix Catalog Images (Generic/Missing icons) ## 📋 Master Plan - [x] Analyze `verify_extraction_logic.py`, `ai_service.py`, `ai_schema.py`, `tracking_scraper_service.py`. - [x] Proposal Phase: Vision Priority accepted. -- [ ] Implementation Phase: +- [x] Implementation Phase: - [x] Update `ai_schema.py`. - [x] Fix `scheduler_service.py` (Disable conflicting Text AI). - [x] Verification Phase: Verified with `verify_vision_priority.py`. -- [ ] **Catalog Issue**: - - [x] Reproduce failure (Confirmed Browserless issue via analysis). +- [x] **Catalog Retrieval Fix**: + - [x] Reproduce failure (Confirmed Browserless issue). - [x] Fix `cataloguemate_scraper.py` with HTTP fallback. - -- [x] **Cleanup**: - - [x] Deleted `debug_*.py` and `verify_*.py` files. + - [x] Cleanup debug files. +- [ ] **Catalog Image Fix**: + - [x] Analyze page HTML for correct image selectors (Found `data-src`). + - [x] Refine `cataloguemate_scraper.py` image extraction logic. ## 📝 Progress Log - **2025-12-22**: Vision Priority implemented and verified. -- **2025-12-22**: User reported Catalog issue. Logic identified as `cataloguemate.fr`. -- **2025-12-22**: Initial debug script failed (missing dependencies). Creating v2 using internal services. +- **2025-12-22**: Fixed Catalog Retrieval using HTTP fallback. +- **2025-12-22**: User reports images are generic icons. Investigating image selectors. diff --git a/walkthrough.md b/walkthrough.md index c9218f9..c34c243 100644 --- a/walkthrough.md +++ b/walkthrough.md @@ -9,16 +9,23 @@ - **Prompt Engineering**: Rewrote the system prompt in `ai_schema.py` to explicitly declare the **IMAGE AS THE SOURCE OF TRUTH**. - **Logic Fix**: Disabled the `AIPriceExtractor` (Text-only AI) in `scheduler_service.py` which was short-circuiting the logic before the Vision AI could run. -## 2. Catalog Scraper Fix +## 2. Catalog Scraper Fix (Connectivity) -**Problem**: Catalogs were not updating because the `browserless` service was failing (likely blocked or network issues), preventing the scraper from loading `cataloguemate.fr`. +**Problem**: Catalogs were not updating because the `browserless` service was failing (likely blocked or network issues). **Solution**: -- **HTTP Fallback**: Modified `cataloguemate_scraper.py` to use a robust fallback mechanism. - - First attempts to use the secure Browserless browser. - - If that fails, it instantly falls back to a standard `httpx` HTTP request, which is often sufficient for static catalog sites. +- **HTTP Fallback**: Modified `cataloguemate_scraper.py` to use a robust fallback mechanism (Browserless -> Fallback to HTTPX). -## 3. Cleanup +## 3. Catalog Images Fix (Lazy Loading) -- Removed 20+ temporary debug/verification scripts (`debug_*.py`, `verify_*.py`) from the root directory to keep the production environment clean. +**Problem**: Catalog pages displayed generic icons or placeholders instead of the actual catalog images. +**Analysis**: The website uses lazy loading. Structure: ``. +**Solution**: + +- **Smart Extraction**: Updated `cataloguemate_scraper.py` to checking `data-src` and `data-original` attributes first. +- **Improved Filtering**: Added `loader` to the ignore list and `thumbor` to the whitelist to ensure only high-quality catalog images are selected. + +## 4. Cleanup + +- Removed temporary debug/verification scripts to keep the production environment clean.