From b4dd4b436550055458e89eb10b12ee5414c12864 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 29 Nov 2025 23:43:42 +0000 Subject: [PATCH] feat: Add generic price extraction for all e-commerce sites MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit **Problem:** Price extraction was only working for Amazon. Other sites (B&M, Stokomani, etc.) had prices missing from extracted text, causing AI to rely solely on screenshots. Example error: ``` WARNING - No prices found in text with € symbol AI Response: {"price": 9.99} // Wrong! Real price was 1.99€ ``` **Root Cause:** - `page.inner_text('body')` doesn't capture all price elements - Some sites use hidden elements, iframes, or CSS pseudo-elements - Only Amazon had dedicated price extraction (_extract_amazon_price) - Other sites got: `WARNING - Could not extract price from DOM` **Solution: Universal Price Extraction** 1. **New Method: _extract_generic_price()** (browserless_service.py:145-208) - Tries common CSS selectors across all e-commerce sites: * Class-based: `.price`, `.product-price`, `.current-price` * Attribute-based: `[data-price]`, `[itemprop='price']` * French-specific: `[class*='prix']`, `[class*='tarif']` * Excludes old prices: `:not([class*='old'])` - Validates price format (contains € or comma with digits) - Falls back to regex: `\d+[,\.]\d{2}\s*€` on all text - Returns first valid price found 2. **Updated Extraction Logic** (browserless_service.py:319-338) - Amazon → uses `_extract_amazon_price()` (specific) - All other sites → uses `_extract_generic_price()` (universal) - Prepends "PRIX DÉTECTÉ: X,XX €" to text context - Logs success/failure for debugging **Flow:** ``` Page load (bmstores.fr) ↓ Try selectors: .price, .product-price, [itemprop='price']... ↓ Found: "1,99 €" via selector .price ↓ Prepend: "PRIX DÉTECTÉ: 1,99 €\n\n[page text]" ↓ Send to AI ↓ AI sees explicit price → {"price": 1.99, "confidence": 0.95} ``` **Expected Results:** - Before: `WARNING - No prices found in text with € symbol` - After: `INFO - Found price via selector .price: 1,99 €` - AI confidence improves (text + image vs image only) - Fewer digit recognition errors (1 vs 9) **Tested with:** - https://bmstores.fr/produits/accessoire-noel-enfant/125651-lutin-farceur-35cm - Price: 1,99 € Fixes issue where non-Amazon sites had missing prices in text context, causing AI to misread prices from screenshots only. --- app/services/browserless_service.py | 90 +++++++++++++++++++++++++++-- 1 file changed, 84 insertions(+), 6 deletions(-) diff --git a/app/services/browserless_service.py b/app/services/browserless_service.py index f894119..a55e98b 100644 --- a/app/services/browserless_service.py +++ b/app/services/browserless_service.py @@ -142,6 +142,71 @@ class BrowserlessService: except Exception: pass + async def _extract_generic_price(self, page: Page) -> str: + """ + Extract price from generic e-commerce pages using common selectors and patterns. + Returns price text (e.g., "1,99 €") or empty string if not found. + """ + # Common price selectors used across e-commerce sites + price_selectors = [ + # Common class names + "[class*='price']:not([class*='old']):not([class*='was']):not([class*='original'])", + ".product-price", + ".price", + ".current-price", + ".sale-price", + ".final-price", + # Common data attributes + "[data-price]", + "[itemprop='price']", + # ID-based + "#price", + "#product-price", + "#our-price", + # Specific patterns + "span[class*='prix']", + "div[class*='prix']", + "span[class*='tarif']", + ] + + for selector in price_selectors: + try: + elements = page.locator(selector) + count = await elements.count() + + # Try first visible element + for i in range(min(count, 3)): # Check max 3 elements + try: + element = elements.nth(i) + if await element.is_visible(timeout=1000): + price_text = await element.inner_text() + # Check if it looks like a price (contains € or digits with comma/dot) + if price_text and ('€' in price_text or (',' in price_text and any(c.isdigit() for c in price_text))): + # Clean up price text + price_text = price_text.strip() + logger.info(f"Found price via selector {selector}: {price_text}") + return price_text + except Exception: + continue + except Exception: + continue + + # Fallback: Use regex to find price patterns in all text + try: + all_text = await page.inner_text('body') + import re + # Match French price formats: "1,99 €", "12,99€", "1.234,56 €" + price_matches = re.findall(r'\d+[,\.]\d{2}\s*€', all_text) + if price_matches: + # Return first price found + logger.info(f"Found price via regex: {price_matches[0]}") + return price_matches[0] + except Exception: + pass + + logger.warning("Could not extract generic price with any method") + return "" + async def _extract_amazon_price(self, page: Page) -> str: """ Extract price directly from Amazon product page using CSS selectors. @@ -316,13 +381,26 @@ class BrowserlessService: logger.warning(f"Failed to extract inner_text, falling back to HTML: {e}") content = await page.content() - # For Amazon, extract price directly and prepend to content - # This helps AI by providing explicit price information + # Extract price directly from DOM for better accuracy + extracted_price = None + + # For Amazon, use specific selectors if "amazon" in url.lower() and "/dp/" in url: - amazon_price_text = await self._extract_amazon_price(page) - if amazon_price_text: - content = f"PRIX DÉTECTÉ: {amazon_price_text}\n\n{content}" - logger.info(f"Prepended Amazon price to content: {amazon_price_text}") + extracted_price = await self._extract_amazon_price(page) + if extracted_price: + logger.info(f"Extracted Amazon price: {extracted_price}") + else: + # For other sites, try common price selectors + extracted_price = await self._extract_generic_price(page) + if extracted_price: + logger.info(f"Extracted generic price: {extracted_price}") + + # Prepend extracted price to content if found + if extracted_price: + content = f"PRIX DÉTECTÉ: {extracted_price}\n\n{content}" + logger.info(f"Prepended price to content: {extracted_price}") + else: + logger.warning("Could not extract price from DOM, relying on text/image only") else: # Extract raw HTML for parsing (search use case) content = await page.content()