test: Update search verification query and target site.

This commit is contained in:
Michael committed 2025-12-22 15:42:33 +01:00
1 parent 0ff229de13
commit 39d4107e6d
2 files changed
+148 -123

No files matched your search

+146 -121
View File
@@ -36,6 +36,7 @@ COMMON_POPUP_SELECTORS = [
class SearchResult: class SearchResult:
"""Search result data class""" """Search result data class"""
def __init__( def __init__(
self, self,
url: str, url: str,
@@ -177,8 +178,6 @@ class ImprovedSearchService:
except Exception: except Exception:
pass pass
@staticmethod @staticmethod
def _parse_results(html: str, site_key: str, base_url: str, query: str) -> list[SearchResult]: def _parse_results(html: str, site_key: str, base_url: str, query: str) -> list[SearchResult]:
""" """
@@ -193,7 +192,9 @@ class ImprovedSearchService:
query_words = [w.lower() for w in query.split() if len(w) > 2] query_words = [w.lower() for w in query.split() if len(w) > 2]
# Select product links # Select product links
logger.debug(f"Parsing content for {site_key} (length: {len(html)}) with selector: {config['product_selector']}") logger.debug(
f"Parsing content for {site_key} (length: {len(html)}) with selector: {config['product_selector']}"
)
links = soup.select(config["product_selector"]) links = soup.select(config["product_selector"])
logger.debug(f"Found {len(links)} raw items for {site_key}") logger.debug(f"Found {len(links)} raw items for {site_key}")
@@ -205,9 +206,9 @@ class ImprovedSearchService:
container = item container = item
if site_key == "centrakor.com": if site_key == "centrakor.com":
logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}") logger.debug(f" Processing Centrakor item: {item.name}, classes: {item.get('class')}")
# If selector targets the container (div.product-card), we need to find the link inside # If selector targets the container (div.product-card), we need to find the link inside
if item.name != 'a': if item.name != "a":
# Try to find the link using product_link_selector if available # Try to find the link using product_link_selector if available
if "product_link_selector" in config: if "product_link_selector" in config:
found_link = item.select_one(config["product_link_selector"]) found_link = item.select_one(config["product_link_selector"])
@@ -215,7 +216,7 @@ class ImprovedSearchService:
link = found_link link = found_link
else: else:
# Fallback: find first 'a' tag # Fallback: find first 'a' tag
found_link = item.find('a') found_link = item.find("a")
if found_link: if found_link:
link = found_link link = found_link
else: else:
@@ -223,7 +224,7 @@ class ImprovedSearchService:
continue continue
else: else:
# No link selector specified, try to find first 'a' # No link selector specified, try to find first 'a'
found_link = item.find('a') found_link = item.find("a")
if found_link: if found_link:
link = found_link link = found_link
else: else:
@@ -232,7 +233,7 @@ class ImprovedSearchService:
else: else:
# Item is already a link # Item is already a link
link = item link = item
href = link.get("href") href = link.get("href")
if not href: if not href:
logger.debug(f" ⚠️ No href in link for {site_key}") logger.debug(f" ⚠️ No href in link for {site_key}")
@@ -268,22 +269,22 @@ class ImprovedSearchService:
if "product_image_selector" in config: if "product_image_selector" in config:
# Search in the original container first - get ALL matches # Search in the original container first - get ALL matches
img_els = container.select(config["product_image_selector"]) img_els = container.select(config["product_image_selector"])
# Filter and find first valid image # Filter and find first valid image
for img_el in img_els: for img_el in img_els:
# Try multiple attributes in order of priority # Try multiple attributes in order of priority
candidate_url = ( candidate_url = (
img_el.get("src") or img_el.get("src")
img_el.get("data-src") or or img_el.get("data-src")
img_el.get("data-lazy-src") or or img_el.get("data-lazy-src")
img_el.get("data-original") or img_el.get("data-original")
) )
# Handle srcset (use first URL) # Handle srcset (use first URL)
if not candidate_url and img_el.get("srcset"): if not candidate_url and img_el.get("srcset"):
srcset = img_el.get("srcset") srcset = img_el.get("srcset")
candidate_url = srcset.split(",")[0].split()[0] candidate_url = srcset.split(",")[0].split()[0]
# Skip invalid images (pictos, icons, etc.) # Skip invalid images (pictos, icons, etc.)
if candidate_url: if candidate_url:
# SPECIAL: No filtering for Centrakor (debugging) # SPECIAL: No filtering for Centrakor (debugging)
@@ -293,29 +294,32 @@ class ImprovedSearchService:
logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}") logger.debug(f" ⏭️ Skipping placeholder: {candidate_url[:50]}")
continue continue
# Filter only tiny pictos # Filter only tiny pictos
if 'picto' in candidate_url.lower() and ('width=60' in candidate_url or 'height=80' in candidate_url): if "picto" in candidate_url.lower() and (
"width=60" in candidate_url or "height=80" in candidate_url
):
logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}") logger.debug(f" ⏭️ Skipping tiny picto: {candidate_url[:50]}")
continue continue
image_url = candidate_url image_url = candidate_url
logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}") logger.debug(f" 🖼️ Centrakor image: {image_url[:70]}")
break break
# Normal filtering for other sites # Normal filtering for other sites
# Filter out obvious pictos and small icons # Filter out obvious pictos and small icons
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']): if any(keyword in candidate_url.lower() for keyword in ["picto", "icon", "logo", "badge"]):
logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}") logger.debug(f" ⏭️ Skipping picto/icon: {candidate_url[:50]}")
continue continue
# Filter out VERY small images (less than 100px) # Filter out VERY small images (less than 100px)
import re import re
width_match = re.search(r'width=(\d+)', candidate_url)
height_match = re.search(r'height=(\d+)', candidate_url) width_match = re.search(r"width=(\d+)", candidate_url)
height_match = re.search(r"height=(\d+)", candidate_url)
if width_match and height_match: if width_match and height_match:
width = int(width_match.group(1)) width = int(width_match.group(1))
height = int(height_match.group(1)) height = int(height_match.group(1))
if width < 100 and height < 100: if width < 100 and height < 100:
logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}") logger.debug(f" ⏭️ Skipping small image ({width}x{height}): {candidate_url[:50]}")
continue continue
# This is a valid product image # This is a valid product image
image_url = candidate_url image_url = candidate_url
logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...") logger.debug(f" 🖼️ Image found via product_image_selector: {image_url[:50]}...")
@@ -326,17 +330,14 @@ class ImprovedSearchService:
img = link.find("img") img = link.find("img")
if img: if img:
image_url = ( image_url = (
img.get("src") or img.get("src") or img.get("data-src") or img.get("data-lazy-src") or img.get("data-original")
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original")
) )
# Handle srcset # Handle srcset
if not image_url and img.get("srcset"): if not image_url and img.get("srcset"):
srcset = img.get("srcset") srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0] image_url = srcset.split(",")[0].split()[0]
if image_url: if image_url:
logger.debug(f" 🖼️ Image found via link.find('img'): {image_url[:50]}...") logger.debug(f" 🖼️ Image found via link.find('img'): {image_url[:50]}...")
@@ -348,28 +349,30 @@ class ImprovedSearchService:
if source and source.get("srcset"): if source and source.get("srcset"):
srcset = source.get("srcset") srcset = source.get("srcset")
image_url = srcset.split(",")[0].split()[0] image_url = srcset.split(",")[0].split()[0]
if not image_url: if not image_url:
img_in_picture = picture.find("img") img_in_picture = picture.find("img")
if img_in_picture: if img_in_picture:
image_url = img_in_picture.get("src") or img_in_picture.get("data-src") image_url = img_in_picture.get("src") or img_in_picture.get("data-src")
if image_url: if image_url:
logger.debug(f" 🖼️ Image found via <picture>: {image_url[:50]}...") logger.debug(f" 🖼️ Image found via <picture>: {image_url[:50]}...")
# Clean up and validate image URL # Clean up and validate image URL
if image_url: if image_url:
# Remove data URIs, 1x1 pixels, placeholders # Remove data URIs, 1x1 pixels, placeholders
if (image_url.startswith("data:") or if (
"1x1" in image_url or image_url.startswith("data:")
"placeholder" in image_url.lower() or or "1x1" in image_url
image_url.strip() == ""): or "placeholder" in image_url.lower()
or image_url.strip() == ""
):
logger.debug(f" ⏭️ Skipping invalid image: {image_url[:50]}") logger.debug(f" ⏭️ Skipping invalid image: {image_url[:50]}")
image_url = None image_url = None
elif not image_url.startswith("http"): elif not image_url.startswith("http"):
image_url = urljoin(base_url, image_url) image_url = urljoin(base_url, image_url)
logger.debug(f" 🔗 Made image URL absolute: {image_url[:80]}...") logger.debug(f" 🔗 Made image URL absolute: {image_url[:80]}...")
if not image_url: if not image_url:
logger.warning(f" ⚠️ No image found for: {title[:50]}") logger.warning(f" ⚠️ No image found for: {title[:50]}")
@@ -378,8 +381,9 @@ class ImprovedSearchService:
if site_key == "gifi.fr": if site_key == "gifi.fr":
# Gifi: Extract price from product tile HTML # Gifi: Extract price from product tile HTML
import re import re
container_html = str(container) container_html = str(container)
price_match = re.search(r'(\d+)[,.](\d+)\s*€', container_html) price_match = re.search(r"(\d+)[,.](\d+)\s*€", container_html)
if price_match: if price_match:
euros = int(price_match.group(1)) euros = int(price_match.group(1))
cents = int(price_match.group(2)) cents = int(price_match.group(2))
@@ -387,14 +391,16 @@ class ImprovedSearchService:
logger.debug(f" 💰 Extracted price from search: {product_price}€") logger.debug(f" 💰 Extracted price from search: {product_price}€")
# Create result # Create result
results.append(SearchResult( results.append(
url=full_url, SearchResult(
title=title, url=full_url,
snippet=f"Product from {config['name']}", title=title,
source=config["name"], snippet=f"Product from {config['name']}",
price=product_price, # Set price if extracted from search source=config["name"],
image_url=image_url price=product_price, # Set price if extracted from search
)) image_url=image_url,
)
)
logger.debug(f"Parsed {len(results)} results from HTML") logger.debug(f"Parsed {len(results)} results from HTML")
return results return results
@@ -406,25 +412,28 @@ class ImprovedSearchService:
# SPECIAL CASE: L'Incroyable - Price is in the title # SPECIAL CASE: L'Incroyable - Price is in the title
if "lincroyable.fr" in result.url: if "lincroyable.fr" in result.url:
import re import re
# Extract price from title (e.g., "34€99" or "59€99") # Extract price from title (e.g., "34€99" or "59€99")
price_match = re.search(r'(\d+)€(\d+)', result.title) # Extract price from title (e.g., "34€99" or "59€99")
# Fix: Use non-greedy regex and ensure only 2 decimals
price_match = re.search(r"(\d+)€(\d{2})(?!\d)", result.title)
if price_match: if price_match:
# Convert to float (e.g., "34€99" -> 34.99) # Convert to float (e.g., "34€99" -> 34.99)
price_euros = int(price_match.group(1)) price_euros = int(price_match.group(1))
price_cents = int(price_match.group(2)) price_cents = int(price_match.group(2))
result.price = float(f"{price_euros}.{price_cents}") result.price = float(f"{price_euros}.{price_cents}")
# Clean title by removing price # Clean title by removing price
result.title = re.sub(r'\d+€\d+', '', result.title).strip() result.title = re.sub(r"\d+€\d+", "", result.title).strip()
logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title") logger.debug(f"L'Incroyable - Extracted price {result.price}€ from title")
return result return result
# SPECIAL CASE: Gifi - Price extracted from search, no need to visit page # SPECIAL CASE: Gifi - Price extracted from search, no need to visit page
if "gifi.fr" in result.url and result.price is not None: if "gifi.fr" in result.url and result.price is not None:
logger.debug(f"Gifi - Price already extracted from search: {result.price}€") logger.debug(f"Gifi - Price already extracted from search: {result.price}€")
return result return result
page = await context.new_page() page = await context.new_page()
try: try:
logger.debug(f"Scraping details for: {result.title[:50]}...") logger.debug(f"Scraping details for: {result.title[:50]}...")
@@ -469,7 +478,7 @@ class ImprovedSearchService:
async def _extract_price(page: Page) -> float | None: async def _extract_price(page: Page) -> float | None:
"""Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS""" """Extract price using Hybrid Strategy: JSON-LD -> AI -> Strict CSS -> Loose CSS"""
import re import re
# STRATEGY 1: JSON-LD (Most reliable) # STRATEGY 1: JSON-LD (Most reliable)
try: try:
json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]') json_ld_scripts = await page.query_selector_all('script[type="application/ld+json"]')
@@ -477,19 +486,19 @@ class ImprovedSearchService:
try: try:
content = await script.inner_text() content = await script.inner_text()
data = json.loads(content) data = json.loads(content)
# Handle list of objects # Handle list of objects
if isinstance(data, list): if isinstance(data, list):
data = data[0] if data else {} data = data[0] if data else {}
# Check for Product type # Check for Product type
if data.get('@type') == 'Product': if data.get("@type") == "Product":
offers = data.get('offers') offers = data.get("offers")
if isinstance(offers, list): if isinstance(offers, list):
offers = offers[0] offers = offers[0]
if offers and 'price' in offers: if offers and "price" in offers:
price = float(offers['price']) price = float(offers["price"])
logger.debug(f" ✅ Found price via JSON-LD: {price}€") logger.debug(f" ✅ Found price via JSON-LD: {price}€")
return price return price
except: except:
@@ -503,7 +512,7 @@ class ImprovedSearchService:
# We log info but don't block if it fails # We log info but don't block if it fails
html_content = await page.content() html_content = await page.content()
title = await page.title() title = await page.title()
ai_price = await AIPriceExtractor.extract_price(html_content, title) ai_price = await AIPriceExtractor.extract_price(html_content, title)
if ai_price: if ai_price:
logger.info(f" 🤖 AI found price: {ai_price}€") logger.info(f" 🤖 AI found price: {ai_price}€")
@@ -512,26 +521,26 @@ class ImprovedSearchService:
logger.error(f" AI extraction failed: {e}") logger.error(f" AI extraction failed: {e}")
# STRATEGY 3: Strict CSS Selectors (High Confidence) # STRATEGY 3: Strict CSS Selectors (High Confidence)
# These selectors usually point to the main product price. # These selectors usually point to the main product price.
# If found, we trust them and return immediately. # If found, we trust them and return immediately.
strict_selectors = [ strict_selectors = [
'[itemprop="price"]', # Schema.org standard '[itemprop="price"]', # Schema.org standard
'.current-price-value', ".current-price-value",
'.product-price .price', ".product-price .price",
'meta[property="product:price:amount"]', 'meta[property="product:price:amount"]',
'meta[name="twitter:data1"]', # Sometimes used for price 'meta[name="twitter:data1"]', # Sometimes used for price
] ]
for selector in strict_selectors: for selector in strict_selectors:
try: try:
# Check meta tags first # Check meta tags first
if selector.startswith('meta'): if selector.startswith("meta"):
element = await page.query_selector(selector) element = await page.query_selector(selector)
if element: if element:
content = await element.get_attribute('content') content = await element.get_attribute("content")
if content: if content:
try: try:
price_val = float(content.strip().replace(',', '.')) price_val = float(content.strip().replace(",", "."))
logger.debug(f" ✅ Found price via Meta Tag {selector}: {price_val}€") logger.debug(f" ✅ Found price via Meta Tag {selector}: {price_val}€")
return price_val return price_val
except: except:
@@ -544,12 +553,12 @@ class ImprovedSearchService:
# Skip hidden # Skip hidden
if not await elem.is_visible(): if not await elem.is_visible():
continue continue
price_text = await elem.inner_text() price_text = await elem.inner_text()
if price_text: if price_text:
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r'(\d+\.?\d*)', cleaned) match = re.search(r"(\d+\.?\d*)", cleaned)
if match: if match:
try: try:
price_val = float(match.group(1)) price_val = float(match.group(1))
@@ -563,44 +572,55 @@ class ImprovedSearchService:
# STRATEGY 4: Loose CSS Selectors (Fallback - Lowest Price Logic) # STRATEGY 4: Loose CSS Selectors (Fallback - Lowest Price Logic)
# Used when strict selectors fail. We collect ALL prices and pick the lowest. # Used when strict selectors fail. We collect ALL prices and pick the lowest.
# PRIORITY 1: Selectors for sale/promotional prices # PRIORITY 1: Selectors for sale/promotional prices
sale_price_selectors = [ sale_price_selectors = [
'.price-current', ".price-current",
'.prix-actuel', ".prix-actuel",
'.sale-price', ".sale-price",
'.promo-price', ".promo-price",
'[class*="promo"]', '[class*="promo"]',
'[class*="sale"]', '[class*="sale"]',
'[class*="discount"]', '[class*="discount"]',
] ]
# PRIORITY 2: Standard price selectors # PRIORITY 2: Standard price selectors
price_selectors = [ price_selectors = [
'.price', ".price",
'[data-testid="price"]', '[data-testid="price"]',
'.product-price', ".product-price",
'.a-price .a-offscreen', ".a-price .a-offscreen",
'.a-price-whole', ".a-price-whole",
'span[class*="price"]', # REMOVED: 'span[class*="price"]' - Too broad, picks up 'price-tax-excluded' (HT)
] ]
# Blacklist words that indicate installments or taxes
FORBIDDEN_WORDS = ["mois", "fois", "x", "ht", "hors tax", "eco-part", "sans"]
# Try sale prices first # Try sale prices first
for selector in sale_price_selectors: for selector in sale_price_selectors:
try: try:
elements = await page.query_selector_all(selector) elements = await page.query_selector_all(selector)
for elem in elements: for elem in elements:
# Skip if element is strikethrough (old price) # Skip if element is strikethrough (old price)
parent_html = await elem.evaluate('el => el.parentElement.outerHTML') parent_html = await elem.evaluate("el => el.parentElement.outerHTML")
if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html: if (
"text-decoration: line-through" in parent_html
or "text-decoration-line: line-through" in parent_html
):
continue continue
price_text = await elem.inner_text() price_text = await elem.inner_text()
if price_text: if price_text:
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() text_lower = price_text.lower()
cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') # Filter out forbidden words
if any(w in text_lower for w in FORBIDDEN_WORDS):
match = re.search(r'(\d+\.?\d*)', cleaned) continue
cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r"(\d+\.?\d*)", cleaned)
if match: if match:
try: try:
price_val = float(match.group(1)) price_val = float(match.group(1))
@@ -611,32 +631,34 @@ class ImprovedSearchService:
continue continue
except Exception: except Exception:
continue continue
# Fallback to standard price selectors # Fallback to standard price selectors
all_prices = [] all_prices = []
for selector in price_selectors: for selector in price_selectors:
try: try:
elements = await page.query_selector_all(selector) elements = await page.query_selector_all(selector)
for elem in elements: for elem in elements:
# Skip if element is strikethrough (old price) # Skip if element is strikethrough (old price)
try: try:
parent_html = await elem.evaluate('el => el.parentElement.outerHTML') parent_html = await elem.evaluate("el => el.parentElement.outerHTML")
if 'text-decoration: line-through' in parent_html or 'text-decoration-line: line-through' in parent_html: if (
"text-decoration: line-through" in parent_html
or "text-decoration-line: line-through" in parent_html
):
continue continue
elem_style = await elem.evaluate('el => window.getComputedStyle(el).textDecoration') elem_style = await elem.evaluate("el => window.getComputedStyle(el).textDecoration")
if 'line-through' in elem_style: if "line-through" in elem_style:
continue continue
except: except:
pass pass
price_text = await elem.inner_text() price_text = await elem.inner_text()
if price_text: if price_text:
cleaned = price_text.strip().replace('€', '').replace('EUR', '').strip() cleaned = price_text.strip().replace("€", "").replace("EUR", "").strip()
cleaned = cleaned.replace(' ', '').replace('\xa0', '').replace(',', '.') cleaned = cleaned.replace(" ", "").replace("\xa0", "").replace(",", ".")
match = re.search(r'(\d+\.?\d*)', cleaned) match = re.search(r"(\d+\.?\d*)", cleaned)
if match: if match:
try: try:
price_val = float(match.group(1)) price_val = float(match.group(1))
@@ -648,7 +670,7 @@ class ImprovedSearchService:
except Exception as e: except Exception as e:
logger.debug(f" Error with selector {selector}: {e}") logger.debug(f" Error with selector {selector}: {e}")
continue continue
# Return the LOWEST price found # Return the LOWEST price found
if all_prices: if all_prices:
lowest_price = min(all_prices) lowest_price = min(all_prices)
@@ -668,7 +690,7 @@ class ImprovedSearchService:
"out of stock", "out of stock",
"unavailable", "unavailable",
"épuisé", "épuisé",
"non disponible" "non disponible",
] ]
try: try:
@@ -692,9 +714,6 @@ class ImprovedSearchService:
return None # Unknown return None # Unknown
@classmethod @classmethod
async def search_site_generator(cls, site_key: str, query: str) -> AsyncGenerator[SearchResult, None]: async def search_site_generator(cls, site_key: str, query: str) -> AsyncGenerator[SearchResult, None]:
"""Search a single site and yield results as they are scraped""" """Search a single site and yield results as they are scraped"""
@@ -712,13 +731,13 @@ class ImprovedSearchService:
try: try:
context = await cls._create_context(cls._browser) context = await cls._create_context(cls._browser)
try: try:
page = await context.new_page() page = await context.new_page()
# Navigate # Navigate
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000) await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
# Wait for selector # Wait for selector
wait_selector = config.get("wait_selector") or config.get("product_selector") wait_selector = config.get("wait_selector") or config.get("product_selector")
try: try:
@@ -729,36 +748,37 @@ class ImprovedSearchService:
# Get content # Get content
content = await page.content() content = await page.content()
await page.close() # Close search page to free resources await page.close() # Close search page to free resources
# Parse results (Phase 1) # Parse results (Phase 1)
base_url = search_url.split("/search")[0] base_url = search_url.split("/search")[0]
if "amazon" in site_key: if "amazon" in site_key:
base_url = "https://www.amazon.fr" base_url = "https://www.amazon.fr"
initial_results = cls._parse_results(content, site_key, base_url, query) initial_results = cls._parse_results(content, site_key, base_url, query)
if not initial_results: if not initial_results:
logger.warning(f"No results found for {site_key}") logger.warning(f"No results found for {site_key}")
# Dump HTML for debugging # Dump HTML for debugging
import os import os
dump_dir = "/app/debug_dumps" dump_dir = "/app/debug_dumps"
os.makedirs(dump_dir, exist_ok=True) os.makedirs(dump_dir, exist_ok=True)
dump_path = f"{dump_dir}/{site_key.replace('.', '_')}_no_results.html" dump_path = f"{dump_dir}/{site_key.replace('.', '_')}_no_results.html"
with open(dump_path, 'w', encoding='utf-8') as f: with open(dump_path, "w", encoding="utf-8") as f:
f.write(content) f.write(content)
logger.warning(f"HTML dumped to {dump_path} for inspection") logger.warning(f"HTML dumped to {dump_path} for inspection")
return return
# Phase 2: Scrape details (Streaming) # Phase 2: Scrape details (Streaming)
semaphore = asyncio.Semaphore(3) semaphore = asyncio.Semaphore(3)
async def scrape_wrapper(res): async def scrape_wrapper(res):
async with semaphore: async with semaphore:
return await cls._scrape_item_details(res, context) return await cls._scrape_item_details(res, context)
tasks = [scrape_wrapper(r) for r in initial_results] tasks = [scrape_wrapper(r) for r in initial_results]
for future in asyncio.as_completed(tasks): for future in asyncio.as_completed(tasks):
enriched_res = await future enriched_res = await future
if enriched_res: if enriched_res:
@@ -788,6 +808,7 @@ class ImprovedSearchService:
# COMPATIBILITY LAYER FOR API ROUTERS # COMPATIBILITY LAYER FOR API ROUTERS
# ========================================== # ==========================================
async def search_products( async def search_products(
query: str, query: str,
db: Session, db: Session,
@@ -818,9 +839,11 @@ async def search_products(
site_keys = [] site_keys = []
for site in active_sites: for site in active_sites:
matched_key = None matched_key = None
# Normalize domain for comparison (remove www., lowercase, etc.) # Normalize domain for comparison (remove www., lowercase, etc.)
site_domain_normalized = site.domain.lower().replace("www.", "").replace("http://", "").replace("https://", "").strip("/") site_domain_normalized = (
site.domain.lower().replace("www.", "").replace("http://", "").replace("https://", "").strip("/")
)
# Try multiple matching strategies: # Try multiple matching strategies:
for key in SITE_CONFIGS.keys(): for key in SITE_CONFIGS.keys():
@@ -855,11 +878,13 @@ async def search_products(
matched_key = key matched_key = key
logger.info(f"✅ Mapped {site.name} ({site.domain}) → {key} (normalized contains)") logger.info(f"✅ Mapped {site.name} ({site.domain}) → {key} (normalized contains)")
break break
if matched_key: if matched_key:
site_keys.append(matched_key) site_keys.append(matched_key)
else: else:
logger.warning(f"❌ No config found for {site.name} (domain: {site.domain}, normalized: {site_domain_normalized})") logger.warning(
f"❌ No config found for {site.name} (domain: {site.domain}, normalized: {site_domain_normalized})"
)
# 3. Execute searches and stream results # 3. Execute searches and stream results
generators = [ImprovedSearchService.search_site_generator(key, query) for key in site_keys] generators = [ImprovedSearchService.search_site_generator(key, query) for key in site_keys]
+2 -2
View File
@@ -30,8 +30,8 @@ ImprovedSearchService._connect_browser = _connect_browser_local
async def test_search(): async def test_search():
await ImprovedSearchService.initialize() await ImprovedSearchService.initialize()
query = "nintendo switch" query = "Mange-debout" # Common item on their site
target_site = "stokomani.fr" target_site = "lincroyable.fr"
print(f"Searching for: {query} on {target_site}") print(f"Searching for: {query} on {target_site}")