feat: amélioration de la robustesse de l'extraction des prix pour le comparateur

This commit is contained in:
Michael committed 2026-01-13 14:37:23 +01:00
1 parent 4a686e1b86
commit bc3d3f6e07
5 files changed
+221 -165

No files matched your search

+68 -41
View File
@@ -455,7 +455,61 @@ class ImprovedSearchService:
return result # Return original result without price
@staticmethod
async def _extract_price(page: Page) -> float | None:
def _clean_price_text(price_text: str) -> float | None:
"""
Robust price parsing logic shared with BaseParser.
Handles French formats, hybrid notations (9€99), and noise (eco-part).
"""
if not price_text:
return None
import re
# Normalization
cleaned = price_text.strip().replace("\xa0", " ").replace("\u202f", " ")
# Hybrid notation "9€99" -> "9.99"
cleaned = re.sub(r"(\d+)\s*€\s*(\d+)", r"\1.\2", cleaned)
cleaned = cleaned.replace("€", "").replace("EUR", "")
# Noise removal
noise_patterns = [
r"\s*[/]\s*.*$", # / pce, / kg
r"\bsoit\b.*$",
r"\bdont\b.*$",
r"\béco-part\b.*$",
r"\beco-part\b.*$",
]
for pattern in noise_patterns:
cleaned = re.sub(pattern, "", cleaned, flags=re.IGNORECASE)
# Thousands separators
if "." in cleaned and "," in cleaned:
if cleaned.rfind(".") < cleaned.rfind(","):
cleaned = cleaned.replace(".", "")
else:
cleaned = cleaned.replace(",", "")
if " " in cleaned and ("," in cleaned or "." in cleaned):
cleaned = cleaned.replace(" ", "")
cleaned = cleaned.replace(",", ".")
# Extraction
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
val_str = match.group(1)
if val_str.endswith("."):
val_str = val_str[:-1]
price = float(val_str)
if 0.00 <= price <= 1000000:
return price
except ValueError:
pass
return None
@classmethod
async def _extract_price(cls, page: Page) -> float | None:
"""Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS"""
import re
@@ -535,18 +589,10 @@ class ImprovedSearchService:
continue
price_text = await elem.inner_text()
if price_text:
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
price_val = float(match.group(1))
if 0.01 < price_val < 100000:
logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€")
return price_val
except:
continue
price_val = cls._clean_price_text(price_text)
if price_val is not None:
logger.debug(f" ✅ Found price via Strict CSS {selector}: {price_val}€")
return price_val
except:
continue
@@ -591,24 +637,14 @@ class ImprovedSearchService:
continue
price_text = await elem.inner_text()
if price_text:
price_val = cls._clean_price_text(price_text)
if price_val is not None:
text_lower = price_text.lower()
# Filter out forbidden words
# Filter out forbidden words (secondary safety)
if any(w in text_lower for w in FORBIDDEN_WORDS):
continue
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
price_val = float(match.group(1))
if 0.01 < price_val < 100000:
logger.debug(f"Found sale price: {price_val}€ from {selector}")
return price_val
except ValueError:
continue
logger.debug(f"Found sale price: {price_val}€ from {selector}")
return price_val
except Exception:
continue
@@ -634,19 +670,10 @@ class ImprovedSearchService:
pass
price_text = await elem.inner_text()
if price_text:
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
price_val = float(match.group(1))
if 0.01 < price_val < 100000:
logger.debug(f" Found candidate price: {price_val}€ from {selector}")
all_prices.append(price_val)
except ValueError:
continue
price_val = cls._clean_price_text(price_text)
if price_val is not None:
logger.debug(f" Found candidate price: {price_val}€ from {selector}")
all_prices.append(price_val)
except Exception as e:
logger.debug(f" Error with selector {selector}: {e}")
continue
+57 -20
View File
@@ -9,14 +9,13 @@ from abc import ABC, abstractmethod
from dataclasses import dataclass
from urllib.parse import urljoin
from bs4 import BeautifulSoup
logger = logging.getLogger(__name__)
@dataclass
class ProductResult:
"""Unified product result data structure"""
title: str
url: str
source: str
@@ -100,22 +99,60 @@ class BaseParser(ABC):
if not price_text:
return None
# Remove currency symbols
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip()
# Normalization phase
# Replace non-breaking spaces and specific French formatting
cleaned = price_text.strip().replace("\xa0", " ").replace("\u202f", " ")
# Remove thousands separators
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
# Handle the "9€99" case: "9€99" -> "9.99"
cleaned = re.sub(r"(\d+)\s*€\s*(\d+)", r"\1.\2", cleaned)
# Replace French decimal separator
cleaned = cleaned.replace(',', '.')
# Remove currency symbols (already partially handled by regex above but for safety)
cleaned = cleaned.replace("€", "").replace("EUR", "")
# Extract first number
match = re.search(r'(\d+\.?\d*)', cleaned)
# Handle noise like "Eco-part", "/ pce", "soit ..."
# Use regex with word boundaries to avoid catching segments of words
noise_patterns = [
r"\s*[/]\s*.*$", # / pce, / kg
r"\bsoit\b.*$", # soit ...
r"\bdont\b.*$", # dont ... eco-part
r"\béco-part\b.*$",
r"\beco-part\b.*$",
]
for pattern in noise_patterns:
cleaned = re.sub(pattern, "", cleaned, flags=re.IGNORECASE)
# Handle thousands separators and decimal points
# If it contains both dot and comma, we need to decide which is which
if "." in cleaned and "," in cleaned:
# Usually the one closer to the end is the decimal separator
if cleaned.rfind(".") < cleaned.rfind(","):
# Format: 1.234,56
cleaned = cleaned.replace(".", "")
else:
# Format: 1,234.56
cleaned = cleaned.replace(",", "")
# If it has a space as thousands separator
if " " in cleaned and ("," in cleaned or "." in cleaned):
# Format: 1 234,56 or 1 234.56
cleaned = cleaned.replace(" ", "")
# Standardize decimal separator to dot
cleaned = cleaned.replace(",", ".")
# Final extraction: find the first number that looks like a price
match = re.search(r"(\d+\.?\d*)", cleaned)
if match:
try:
price = float(match.group(1))
# Validate range
if 0.01 <= price <= 100000:
price_str = match.group(1)
# Remove trailing dot if any (e.g. from a sentence end)
if price_str.endswith("."):
price_str = price_str[:-1]
price = float(price_str)
# Validate range: up to 1,000,000 for high-end items
if 0.00 <= price <= 1000000:
return price
except ValueError:
pass
@@ -127,10 +164,10 @@ class BaseParser(ABC):
if not rating_text:
return None
match = re.search(r'(\d+[,.]\d+)', rating_text)
match = re.search(r"(\d+[,.]\d+)", rating_text)
if match:
try:
return float(match.group(1).replace(',', '.'))
return float(match.group(1).replace(",", "."))
except ValueError:
pass
@@ -141,8 +178,8 @@ class BaseParser(ABC):
if not reviews_text:
return None
cleaned = re.sub(r'[^\d\s]', '', reviews_text)
cleaned = cleaned.replace(' ', '').replace('\xa0', '')
cleaned = re.sub(r"[^\d\s]", "", reviews_text)
cleaned = cleaned.replace(" ", "").replace("\xa0", "")
try:
return int(cleaned)
@@ -209,15 +246,15 @@ class BaseParser(ABC):
def _get_image_src(self, img_element) -> str | None:
"""Extract image source from img element"""
attrs = ['src', 'data-src', 'data-lazy-src', 'data-original', 'data-lazy']
attrs = ["src", "data-src", "data-lazy-src", "data-original", "data-lazy"]
for attr in attrs:
url = img_element.get(attr)
if url and not url.startswith('data:'):
if url and not url.startswith("data:"):
return url
# Try srcset
srcset = img_element.get('srcset')
srcset = img_element.get("srcset")
if srcset:
return srcset.split(",")[0].split()[0]
+44 -53
View File
@@ -1,10 +1,10 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
import re
logger = logging.getLogger(__name__)
class GifiParser(BaseParser):
def __init__(self):
super().__init__("gifi.fr", "https://www.gifi.fr")
@@ -12,104 +12,101 @@ class GifiParser(BaseParser):
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# Gifi products
# Config: a.link
# Wait selector: .product-tile
products = soup.select(".product-tile, div[class*='product-tile']")
for product in products:
try:
link_el = product.select_one("a.link") or product.find("a")
if not link_el:
continue
href = link_el.get('href')
href = link_el.get("href")
url = self.make_absolute_url(href)
# Title: Try specific classes first to avoid getting rating text etc.
title = None
title_el = product.select_one(".pdp-link > a, .link, [class*='name'], [class*='title']")
if title_el:
title = title_el.get_text(strip=True)
if not title:
title = link_el.get_text(strip=True)
if not title:
continue
# Image extraction
# Priority: picture img -> img with class -> any img
img_url = None
# 1. Try picture source (often high res)
picture = product.select_one("picture")
if picture:
source = picture.find("source")
if source and source.get("srcset"):
img_url = source.get("srcset").split(",")[0].split()[0]
if not img_url:
img = picture.find("img")
if img:
img_url = self._get_image_src(img)
# 2. Try direct image selectors
if not img_url:
img_url = self.extract_image_url(product, [
"img.tile-image",
"img[class*='product']",
".image-container img"
])
img_url = self.extract_image_url(
product, ["img.tile-image", "img[class*='product']", ".image-container img"]
)
price = None
# 1. Try specific price selectors
price_el = product.select_one(".price, .value, [class*='price'], .sales .value")
if price_el:
price = self.parse_price_text(price_el.get_text())
# 2. Fallback: Regex on the entire product text
# 2. Fallback: Use improved base logic on the entire product text
if price is None:
product_text = product.get_text(separator=" ", strip=True)
# Look for price pattern: number followed by € or EUR
# e.g. "7,99 €", "12 €", "12.50€"
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', product_text, re.IGNORECASE)
if price_match:
price = self.parse_price_text(price_match.group(0))
results.append(ProductResult(
title=title,
url=url,
source="Gifi",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Gifi"
))
price = self.parse_price_text(product_text)
results.append(
ProductResult(
title=title,
url=url,
source="Gifi",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from Gifi",
)
)
except Exception as e:
logger.error(f"Error parsing Gifi product: {e}")
continue
logger.info(f"GifiParser found {len(results)} results")
return results
def parse_product_details(self, html: str, product_url: str) -> dict:
soup = BeautifulSoup(html, "html.parser")
# 1. Price extraction
price = None
# Specific Gifi product page price selectors
price_el = soup.select_one(".prices .price .value, .product-price .price .value, .price-sales .value")
if price_el:
price = self.parse_price_text(price_el.get_text())
if price is None:
# Fallback to schema.org data if present
import json
scripts = soup.find_all("script", type="application/ld+json")
for script in scripts:
if script.string:
@@ -128,23 +125,17 @@ class GifiParser(BaseParser):
break
except:
pass
if price is None:
# Text fallback
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', soup.get_text(), re.IGNORECASE)
if price_match:
price = self.parse_price_text(price_match.group(0))
# Text fallback using improved base logic
price = self.parse_price_text(soup.get_text())
# 2. Stock extraction
in_stock = True # specific availability check might be complex, default true if page loads
in_stock = True # specific availability check might be complex, default true if page loads
# Check for "Out of stock" messages
exhausted_el = soup.select_one(".availability-msg.exhausted, .availability-msg.out-of-stock")
if exhausted_el:
in_stock = False
return {
"price": price,
"in_stock": in_stock,
"currency": "EUR"
}
return {"price": price, "in_stock": in_stock, "currency": "EUR"}
+38 -39
View File
@@ -1,10 +1,10 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
import re
logger = logging.getLogger(__name__)
class LIncroyableParser(BaseParser):
def __init__(self):
super().__init__("lincroyable.fr", "https://www.lincroyable.fr")
@@ -12,12 +12,12 @@ class LIncroyableParser(BaseParser):
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
soup = BeautifulSoup(html, "html.parser")
results = []
# L'Incroyable products
# Config: div.tailleBlocProdNew
cards = soup.select("div.tailleBlocProdNew")
for card in cards:
try:
# Link is usually in an 'a' tag inside, or the card itself might be clickable (but here we see multiple links)
@@ -25,78 +25,77 @@ class LIncroyableParser(BaseParser):
link_el = card.select_one("a[href*='/p']")
if not link_el:
continue
href = link_el.get('href')
href = link_el.get("href")
url = self.make_absolute_url(href)
# Title
title_el = card.select_one("h3.nomCoupDeCoeurNew, .nomCoupDeCoeurNew")
title = title_el.get_text(strip=True) if title_el else link_el.get_text(strip=True)
if not title:
continue
# Image - Robust extraction
img_url = None
# Exclude heart/wishlist icons explicitly
# Select all images and filter
images = card.select("img")
valid_images = []
for img in images:
src = img.get('src', '') or img.get('data-src', '')
classes = img.get('class', [])
src = img.get("src", "") or img.get("data-src", "")
classes = img.get("class", [])
# Skip heart/wishlist icons
if 'coup2coeur' in str(classes).lower() or 'wishlist' in str(classes).lower():
if "coup2coeur" in str(classes).lower() or "wishlist" in str(classes).lower():
continue
if 'coeur' in src.lower() or 'heart' in src.lower():
if "coeur" in src.lower() or "heart" in src.lower():
continue
valid_images.append(img)
# Prioritize product images
for img in valid_images:
src = img.get('src', '') or img.get('data-src', '')
if 'product' in src.lower() or 'p/' in src.lower():
src = img.get("src", "") or img.get("data-src", "")
if "product" in src.lower() or "p/" in src.lower():
img_url = self._get_image_src(img)
break
# Fallback to first valid image if no specific product image found
if not img_url and valid_images:
img_url = self._get_image_src(valid_images[0])
if img_url:
img_url = self.make_absolute_url(img_url)
# Price
price = None
price_el = card.select_one("p.prixCoupDeCoeurNew, .prixCoupDeCoeurNew")
if price_el:
price = self.parse_price_text(price_el.get_text())
# Fallback price search in text
# Fallback price search in text using the improved base logic
if not price:
text = card.get_text(" ", strip=True)
# Look for price pattern: number followed by €
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*€', text)
if price_match:
price = self.parse_price_text(price_match.group(0))
results.append(ProductResult(
title=title,
url=url,
source="L'Incroyable",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from L'Incroyable"
))
price = self.parse_price_text(text)
results.append(
ProductResult(
title=title,
url=url,
source="L'Incroyable",
price=price,
currency="EUR",
in_stock=True,
image_url=img_url,
snippet=f"Product from L'Incroyable",
)
)
except Exception as e:
logger.error(f"Error parsing L'Incroyable product: {e}")
continue
logger.info(f"LIncroyableParser found {len(results)} results")
return results
+14 -12
View File
@@ -1,21 +1,23 @@
# Deep Debugging Action.com Availability
# Investigation et Correction du Comparateur
## Context
## Contexte
Some products on Action.com are still marked as unavailable even after the first round of fixes.
Le comparateur de prix récupère parfois des prix incorrects sur certains sites (B&M, L'Incroyable, Gifi, etc.). Il faut identifier les causes (mauvais sélecteurs, parsing regex trop large) et stabiliser l'extraction.
## Current Focus
## Focus Actuel
Identifying why specific Action.com products fail the availability check.
Identification des routes et services impliqués dans la recherche du comparateur.
## Master Plan
- [ ] List currently unavailable Action.com items from the database (if possible) or logs
- [ ] Reproduce the check for a specific problematic item
- [ ] Analyze the HTML and Title for these items
- [ ] Refine the matching logic or the unavailability detection
- [ ] Verify fix with multiple Action.com items
- [x] Identifier les services et parsers utilisés par le comparateur
- [x] Identifier les sites posant problème (B&M, Gifi, L'Incroyable, etc.)
- [x] Reproduire les erreurs d'extraction de prix avec des scripts de diagnostic
- [x] Améliorer la logique de `parse_price_text` dans `BaseParser`
- [x] Mettre à jour les sélecteurs CSS spécifiques dans les parsers si nécessaire
- [x] Vérifier les corrections sur une sélection de produits
- [x] Nettoyage et synchronisation Git
## Progress Log
## Log de Progression
- [/] Task started.
- [/] Initialisation de la tâche.