From bc3d3f6e07f0d5ecab65441a24cd42a19ef7817a Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Tue, 13 Jan 2026 14:37:23 +0100 Subject: [PATCH] =?UTF-8?q?feat:=20am=C3=A9lioration=20de=20la=20robustess?= =?UTF-8?q?e=20de=20l'extraction=20des=20prix=20pour=20le=20comparateur?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- app/services/improved_search_service.py | 109 +++++++++++++-------- app/services/parsers/base_parser.py | 77 +++++++++++---- app/services/parsers/gifi_parser.py | 97 +++++++++--------- app/services/parsers/lincroyable_parser.py | 77 +++++++-------- task.md | 26 ++--- 5 files changed, 221 insertions(+), 165 deletions(-) diff --git a/app/services/improved_search_service.py b/app/services/improved_search_service.py index 0e724a3..0aa02c6 100644 --- a/app/services/improved_search_service.py +++ b/app/services/improved_search_service.py @@ -455,7 +455,61 @@ class ImprovedSearchService: return result # Return original result without price @staticmethod - async def _extract_price(page: Page) -> float | None: + def _clean_price_text(price_text: str) -> float | None: + """ + Robust price parsing logic shared with BaseParser. + Handles French formats, hybrid notations (9€99), and noise (eco-part). + """ + if not price_text: + return None + + import re + + # Normalization + cleaned = price_text.strip().replace("\xa0", " ").replace("\u202f", " ") + + # Hybrid notation "9€99" -> "9.99" + cleaned = re.sub(r"(\d+)\s*€\s*(\d+)", r"\1.\2", cleaned) + cleaned = cleaned.replace("€", "").replace("EUR", "") + + # Noise removal + noise_patterns = [ + r"\s*[/]\s*.*$", # / pce, / kg + r"\bsoit\b.*$", + r"\bdont\b.*$", + r"\béco-part\b.*$", + r"\beco-part\b.*$", + ] + for pattern in noise_patterns: + cleaned = re.sub(pattern, "", cleaned, flags=re.IGNORECASE) + + # Thousands separators + if "." in cleaned and "," in cleaned: + if cleaned.rfind(".") < cleaned.rfind(","): + cleaned = cleaned.replace(".", "") + else: + cleaned = cleaned.replace(",", "") + if " " in cleaned and ("," in cleaned or "." in cleaned): + cleaned = cleaned.replace(" ", "") + + cleaned = cleaned.replace(",", ".") + + # Extraction + match = re.search(r"(\d+\.?\d*)", cleaned) + if match: + try: + val_str = match.group(1) + if val_str.endswith("."): + val_str = val_str[:-1] + price = float(val_str) + if 0.00 <= price <= 1000000: + return price + except ValueError: + pass + return None + + @classmethod + async def _extract_price(cls, page: Page) -> float | None: """Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS""" import re @@ -535,18 +589,10 @@ class ImprovedSearchService: continue price_text = await elem.inner_text() - if price_text: - cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip() - cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".") - match = re.search(r"(\d+\.?\d*)", cleaned) - if match: - try: - price_val = float(match.group(1)) - if 0.01 < price_val < 100000: - logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€") - return price_val - except: - continue + price_val = cls._clean_price_text(price_text) + if price_val is not None: + logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€") + return price_val except: continue @@ -591,24 +637,14 @@ class ImprovedSearchService: continue price_text = await elem.inner_text() - if price_text: + price_val = cls._clean_price_text(price_text) + if price_val is not None: text_lower = price_text.lower() - # Filter out forbidden words + # Filter out forbidden words (secondary safety) if any(w in text_lower for w in FORBIDDEN_WORDS): continue - - cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip() - cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".") - - match = re.search(r"(\d+\.?\d*)", cleaned) - if match: - try: - price_val = float(match.group(1)) - if 0.01 < price_val < 100000: - logger.debug(f"Found sale price: {price_val}€ from {selector}") - return price_val - except ValueError: - continue + logger.debug(f"Found sale price: {price_val}€ from {selector}") + return price_val except Exception: continue @@ -634,19 +670,10 @@ class ImprovedSearchService: pass price_text = await elem.inner_text() - if price_text: - cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip() - cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".") - - match = re.search(r"(\d+\.?\d*)", cleaned) - if match: - try: - price_val = float(match.group(1)) - if 0.01 < price_val < 100000: - logger.debug(f" Found candidate price: {price_val}€ from {selector}") - all_prices.append(price_val) - except ValueError: - continue + price_val = cls._clean_price_text(price_text) + if price_val is not None: + logger.debug(f" Found candidate price: {price_val}€ from {selector}") + all_prices.append(price_val) except Exception as e: logger.debug(f" Error with selector {selector}: {e}") continue diff --git a/app/services/parsers/base_parser.py b/app/services/parsers/base_parser.py index c94e879..d759162 100644 --- a/app/services/parsers/base_parser.py +++ b/app/services/parsers/base_parser.py @@ -9,14 +9,13 @@ from abc import ABC, abstractmethod from dataclasses import dataclass from urllib.parse import urljoin -from bs4 import BeautifulSoup - logger = logging.getLogger(__name__) @dataclass class ProductResult: """Unified product result data structure""" + title: str url: str source: str @@ -100,22 +99,60 @@ class BaseParser(ABC): if not price_text: return None - # Remove currency symbols - cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() + # Normalization phase + # Replace non-breaking spaces and specific French formatting + cleaned = price_text.strip().replace("\xa0", " ").replace("\u202f", " ") - # Remove thousands separators - cleaned = cleaned.replace(' ', '').replace('\xa0', '') + # Handle the "9€99" case: "9€99" -> "9.99" + cleaned = re.sub(r"(\d+)\s*€\s*(\d+)", r"\1.\2", cleaned) - # Replace French decimal separator - cleaned = cleaned.replace(',', '.') + # Remove currency symbols (already partially handled by regex above but for safety) + cleaned = cleaned.replace("€", "").replace("EUR", "") - # Extract first number - match = re.search(r'(\d+\.?\d*)', cleaned) + # Handle noise like "Eco-part", "/ pce", "soit ..." + # Use regex with word boundaries to avoid catching segments of words + noise_patterns = [ + r"\s*[/]\s*.*$", # / pce, / kg + r"\bsoit\b.*$", # soit ... + r"\bdont\b.*$", # dont ... eco-part + r"\béco-part\b.*$", + r"\beco-part\b.*$", + ] + + for pattern in noise_patterns: + cleaned = re.sub(pattern, "", cleaned, flags=re.IGNORECASE) + + # Handle thousands separators and decimal points + # If it contains both dot and comma, we need to decide which is which + if "." in cleaned and "," in cleaned: + # Usually the one closer to the end is the decimal separator + if cleaned.rfind(".") < cleaned.rfind(","): + # Format: 1.234,56 + cleaned = cleaned.replace(".", "") + else: + # Format: 1,234.56 + cleaned = cleaned.replace(",", "") + + # If it has a space as thousands separator + if " " in cleaned and ("," in cleaned or "." in cleaned): + # Format: 1 234,56 or 1 234.56 + cleaned = cleaned.replace(" ", "") + + # Standardize decimal separator to dot + cleaned = cleaned.replace(",", ".") + + # Final extraction: find the first number that looks like a price + match = re.search(r"(\d+\.?\d*)", cleaned) if match: try: - price = float(match.group(1)) - # Validate range - if 0.01 <= price <= 100000: + price_str = match.group(1) + # Remove trailing dot if any (e.g. from a sentence end) + if price_str.endswith("."): + price_str = price_str[:-1] + + price = float(price_str) + # Validate range: up to 1,000,000 for high-end items + if 0.00 <= price <= 1000000: return price except ValueError: pass @@ -127,10 +164,10 @@ class BaseParser(ABC): if not rating_text: return None - match = re.search(r'(\d+[,.]\d+)', rating_text) + match = re.search(r"(\d+[,.]\d+)", rating_text) if match: try: - return float(match.group(1).replace(',', '.')) + return float(match.group(1).replace(",", ".")) except ValueError: pass @@ -141,8 +178,8 @@ class BaseParser(ABC): if not reviews_text: return None - cleaned = re.sub(r'[^\d\s]', '', reviews_text) - cleaned = cleaned.replace(' ', '').replace('\xa0', '') + cleaned = re.sub(r"[^\d\s]", "", reviews_text) + cleaned = cleaned.replace(" ", "").replace("\xa0", "") try: return int(cleaned) @@ -209,15 +246,15 @@ class BaseParser(ABC): def _get_image_src(self, img_element) -> str | None: """Extract image source from img element""" - attrs = ['src', 'data-src', 'data-lazy-src', 'data-original', 'data-lazy'] + attrs = ["src", "data-src", "data-lazy-src", "data-original", "data-lazy"] for attr in attrs: url = img_element.get(attr) - if url and not url.startswith('data:'): + if url and not url.startswith("data:"): return url # Try srcset - srcset = img_element.get('srcset') + srcset = img_element.get("srcset") if srcset: return srcset.split(",")[0].split()[0] diff --git a/app/services/parsers/gifi_parser.py b/app/services/parsers/gifi_parser.py index 44e8ed1..01f8cc6 100644 --- a/app/services/parsers/gifi_parser.py +++ b/app/services/parsers/gifi_parser.py @@ -1,10 +1,10 @@ from bs4 import BeautifulSoup from app.services.parsers.base_parser import BaseParser, ProductResult import logging -import re logger = logging.getLogger(__name__) + class GifiParser(BaseParser): def __init__(self): super().__init__("gifi.fr", "https://www.gifi.fr") @@ -12,104 +12,101 @@ class GifiParser(BaseParser): def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: soup = BeautifulSoup(html, "html.parser") results = [] - + # Gifi products # Config: a.link # Wait selector: .product-tile - + products = soup.select(".product-tile, div[class*='product-tile']") - + for product in products: try: link_el = product.select_one("a.link") or product.find("a") if not link_el: continue - - href = link_el.get('href') + + href = link_el.get("href") url = self.make_absolute_url(href) - + # Title: Try specific classes first to avoid getting rating text etc. title = None title_el = product.select_one(".pdp-link > a, .link, [class*='name'], [class*='title']") if title_el: title = title_el.get_text(strip=True) - + if not title: title = link_el.get_text(strip=True) - + if not title: continue # Image extraction # Priority: picture img -> img with class -> any img img_url = None - + # 1. Try picture source (often high res) picture = product.select_one("picture") if picture: source = picture.find("source") if source and source.get("srcset"): img_url = source.get("srcset").split(",")[0].split()[0] - + if not img_url: img = picture.find("img") if img: img_url = self._get_image_src(img) - + # 2. Try direct image selectors if not img_url: - img_url = self.extract_image_url(product, [ - "img.tile-image", - "img[class*='product']", - ".image-container img" - ]) - + img_url = self.extract_image_url( + product, ["img.tile-image", "img[class*='product']", ".image-container img"] + ) + price = None # 1. Try specific price selectors price_el = product.select_one(".price, .value, [class*='price'], .sales .value") if price_el: price = self.parse_price_text(price_el.get_text()) - - # 2. Fallback: Regex on the entire product text + + # 2. Fallback: Use improved base logic on the entire product text if price is None: product_text = product.get_text(separator=" ", strip=True) - # Look for price pattern: number followed by € or EUR - # e.g. "7,99 €", "12 €", "12.50€" - price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', product_text, re.IGNORECASE) - if price_match: - price = self.parse_price_text(price_match.group(0)) - - results.append(ProductResult( - title=title, - url=url, - source="Gifi", - price=price, - currency="EUR", - in_stock=True, - image_url=img_url, - snippet=f"Product from Gifi" - )) - + price = self.parse_price_text(product_text) + + results.append( + ProductResult( + title=title, + url=url, + source="Gifi", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from Gifi", + ) + ) + except Exception as e: logger.error(f"Error parsing Gifi product: {e}") continue - + logger.info(f"GifiParser found {len(results)} results") return results def parse_product_details(self, html: str, product_url: str) -> dict: soup = BeautifulSoup(html, "html.parser") - + # 1. Price extraction price = None # Specific Gifi product page price selectors price_el = soup.select_one(".prices .price .value, .product-price .price .value, .price-sales .value") if price_el: price = self.parse_price_text(price_el.get_text()) - + if price is None: # Fallback to schema.org data if present import json + scripts = soup.find_all("script", type="application/ld+json") for script in scripts: if script.string: @@ -128,23 +125,17 @@ class GifiParser(BaseParser): break except: pass - + if price is None: - # Text fallback - price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', soup.get_text(), re.IGNORECASE) - if price_match: - price = self.parse_price_text(price_match.group(0)) + # Text fallback using improved base logic + price = self.parse_price_text(soup.get_text()) # 2. Stock extraction - in_stock = True # specific availability check might be complex, default true if page loads - + in_stock = True # specific availability check might be complex, default true if page loads + # Check for "Out of stock" messages exhausted_el = soup.select_one(".availability-msg.exhausted, .availability-msg.out-of-stock") if exhausted_el: in_stock = False - - return { - "price": price, - "in_stock": in_stock, - "currency": "EUR" - } + + return {"price": price, "in_stock": in_stock, "currency": "EUR"} diff --git a/app/services/parsers/lincroyable_parser.py b/app/services/parsers/lincroyable_parser.py index 16da973..eff534d 100644 --- a/app/services/parsers/lincroyable_parser.py +++ b/app/services/parsers/lincroyable_parser.py @@ -1,10 +1,10 @@ from bs4 import BeautifulSoup from app.services.parsers.base_parser import BaseParser, ProductResult import logging -import re logger = logging.getLogger(__name__) + class LIncroyableParser(BaseParser): def __init__(self): super().__init__("lincroyable.fr", "https://www.lincroyable.fr") @@ -12,12 +12,12 @@ class LIncroyableParser(BaseParser): def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]: soup = BeautifulSoup(html, "html.parser") results = [] - + # L'Incroyable products # Config: div.tailleBlocProdNew - + cards = soup.select("div.tailleBlocProdNew") - + for card in cards: try: # Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links) @@ -25,78 +25,77 @@ class LIncroyableParser(BaseParser): link_el = card.select_one("a[href*='/p']") if not link_el: continue - - href = link_el.get('href') + + href = link_el.get("href") url = self.make_absolute_url(href) - + # Title title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew") title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True) - + if not title: continue # Image - Robust extraction img_url = None - + # Exclude heart/wishlist icons explicitly # Select all images and filter images = card.select("img") valid_images = [] for img in images: - src = img.get('src', '') or img.get('data-src', '') - classes = img.get('class', []) - + src = img.get("src", "") or img.get("data-src", "") + classes = img.get("class", []) + # Skip heart/wishlist icons - if 'coup2coeur' in str(classes).lower() or 'wishlist' in str(classes).lower(): + if "coup2coeur" in str(classes).lower() or "wishlist" in str(classes).lower(): continue - if 'coeur' in src.lower() or 'heart' in src.lower(): + if "coeur" in src.lower() or "heart" in src.lower(): continue - + valid_images.append(img) - + # Prioritize product images for img in valid_images: - src = img.get('src', '') or img.get('data-src', '') - if 'product' in src.lower() or 'p/' in src.lower(): + src = img.get("src", "") or img.get("data-src", "") + if "product" in src.lower() or "p/" in src.lower(): img_url = self._get_image_src(img) break - + # Fallback to first valid image if no specific product image found if not img_url and valid_images: img_url = self._get_image_src(valid_images[0]) if img_url: img_url = self.make_absolute_url(img_url) - + # Price price = None price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew") if price_el: price = self.parse_price_text(price_el.get_text()) - - # Fallback price search in text + + # Fallback price search in text using the improved base logic if not price: text = card.get_text(" ", strip=True) - # Look for price pattern: number followed by € - price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*€', text) - if price_match: - price = self.parse_price_text(price_match.group(0)) - - results.append(ProductResult( - title=title, - url=url, - source="L'Incroyable", - price=price, - currency="EUR", - in_stock=True, - image_url=img_url, - snippet=f"Product from L'Incroyable" - )) - + price = self.parse_price_text(text) + + results.append( + ProductResult( + title=title, + url=url, + source="L'Incroyable", + price=price, + currency="EUR", + in_stock=True, + image_url=img_url, + snippet=f"Product from L'Incroyable", + ) + ) + except Exception as e: logger.error(f"Error parsing L'Incroyable product: {e}") continue - + logger.info(f"LIncroyableParser found {len(results)} results") return results diff --git a/task.md b/task.md index 6806f87..1edb2be 100644 --- a/task.md +++ b/task.md @@ -1,21 +1,23 @@ -# Deep Debugging Action.com Availability +# Investigation et Correction du Comparateur -## Context +## Contexte -Some products on Action.com are still marked as unavailable even after the first round of fixes. +Le comparateur de prix récupère parfois des prix incorrects sur certains sites (B&M, L'Incroyable, Gifi, etc.). Il faut identifier les causes (mauvais sélecteurs, parsing regex trop large) et stabiliser l'extraction. -## Current Focus +## Focus Actuel -Identifying why specific Action.com products fail the availability check. +Identification des routes et services impliqués dans la recherche du comparateur. ## Master Plan -- [ ] List currently unavailable Action.com items from the database (if possible) or logs -- [ ] Reproduce the check for a specific problematic item -- [ ] Analyze the HTML and Title for these items -- [ ] Refine the matching logic or the unavailability detection -- [ ] Verify fix with multiple Action.com items +- [x] Identifier les services et parsers utilisés par le comparateur +- [x] Identifier les sites posant problème (B&M, Gifi, L'Incroyable, etc.) +- [x] Reproduire les erreurs d'extraction de prix avec des scripts de diagnostic +- [x] Améliorer la logique de `parse_price_text` dans `BaseParser` +- [x] Mettre à jour les sélecteurs CSS spécifiques dans les parsers si nécessaire +- [x] Vérifier les corrections sur une sélection de produits +- [x] Nettoyage et synchronisation Git -## Progress Log +## Log de Progression -- [/] Task started. +- [/] Initialisation de la tâche.