mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Merge pull request #192 from R0m1k3/antigravity
feat: Implement improved search service with persistent browser and m…
This commit is contained in:
4 files changed
+203
-120
No files matched your search
@@ -200,89 +200,113 @@ class ImprovedSearchService:
|
||||
search_url = config["search_url"].format(query=quote_plus(query))
|
||||
logger.info(f"🔍 Searching {config['name']} at {search_url}")
|
||||
|
||||
# Ensure browser is connected
|
||||
if not await cls._ensure_browser_connected():
|
||||
logger.error("Failed to establish browser connection")
|
||||
return []
|
||||
|
||||
results = []
|
||||
|
||||
try:
|
||||
context = await cls._create_context(cls._browser)
|
||||
page = await context.new_page()
|
||||
|
||||
# Retry logic
|
||||
max_retries = 1
|
||||
for attempt in range(max_retries + 1):
|
||||
try:
|
||||
# Navigate to search page
|
||||
logger.debug(f"Navigating to {search_url}")
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
|
||||
logger.debug("Page loaded (domcontentloaded)")
|
||||
|
||||
# Wait for network idle
|
||||
try:
|
||||
await page.wait_for_load_state("networkidle", timeout=10000)
|
||||
logger.debug("Network idle reached")
|
||||
except PlaywrightTimeoutError:
|
||||
logger.debug("Network idle timed out (non-critical)")
|
||||
|
||||
# Handle popups
|
||||
await cls._handle_popups(page)
|
||||
|
||||
# Wait for content to load
|
||||
wait_selector = config.get("wait_selector")
|
||||
if wait_selector:
|
||||
try:
|
||||
await page.wait_for_selector(wait_selector, timeout=5000)
|
||||
logger.debug(f"Wait selector found: {wait_selector}")
|
||||
except PlaywrightTimeoutError:
|
||||
logger.warning(f"Wait selector not found: {wait_selector}")
|
||||
|
||||
# Small delay for JS rendering
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
# Get HTML content
|
||||
html_content = await page.content()
|
||||
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
|
||||
|
||||
if len(html_content) < 5000:
|
||||
logger.warning(f"⚠️ Page too small - possibly blocked")
|
||||
# Ensure browser is connected (check before each attempt)
|
||||
if not await cls._ensure_browser_connected():
|
||||
logger.error("Failed to establish browser connection")
|
||||
if attempt < max_retries:
|
||||
continue
|
||||
return []
|
||||
|
||||
# Parse results using specialized parser
|
||||
parser = ParserFactory.get_parser(site_key)
|
||||
parsed_products = parser.parse_search_results(html_content, query, search_url)
|
||||
context = await cls._create_context(cls._browser)
|
||||
try:
|
||||
page = await context.new_page()
|
||||
|
||||
# Convert ProductResult to SearchResult
|
||||
results = [cls._convert_to_search_result(p) for p in parsed_products]
|
||||
try:
|
||||
# Navigate to search page
|
||||
logger.debug(f"Navigating to {search_url}")
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
|
||||
logger.debug("Page loaded (domcontentloaded)")
|
||||
|
||||
# Scrape details for each result (in parallel)
|
||||
if results:
|
||||
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
|
||||
|
||||
# Enrich all results
|
||||
results_to_enrich = results
|
||||
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
|
||||
|
||||
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
|
||||
# Wait for network idle
|
||||
try:
|
||||
await page.wait_for_load_state("networkidle", timeout=10000)
|
||||
logger.debug("Network idle reached")
|
||||
except PlaywrightTimeoutError:
|
||||
logger.debug("Network idle timed out (non-critical)")
|
||||
|
||||
async def scrape_with_limit(res):
|
||||
async with semaphore:
|
||||
return await cls._scrape_item_details(res, context)
|
||||
# Handle popups
|
||||
await cls._handle_popups(page)
|
||||
|
||||
tasks = [scrape_with_limit(r) for r in results_to_enrich]
|
||||
enriched_results = await asyncio.gather(*tasks)
|
||||
|
||||
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
|
||||
results = [r for r in enriched_results if r]
|
||||
# Wait for content to load
|
||||
wait_selector = config.get("wait_selector")
|
||||
if wait_selector:
|
||||
try:
|
||||
await page.wait_for_selector(wait_selector, timeout=5000)
|
||||
logger.debug(f"Wait selector found: {wait_selector}")
|
||||
except PlaywrightTimeoutError:
|
||||
logger.warning(f"Wait selector not found: {wait_selector}")
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
# Small delay for JS rendering
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"❌ Error during search: {e}", exc_info=True)
|
||||
return []
|
||||
# Get HTML content
|
||||
html_content = await page.content()
|
||||
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
|
||||
|
||||
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
|
||||
return results
|
||||
if len(html_content) < 5000:
|
||||
logger.warning(f"⚠️ Page too small - possibly blocked")
|
||||
return []
|
||||
|
||||
# Parse results using specialized parser
|
||||
parser = ParserFactory.get_parser(site_key)
|
||||
parsed_products = parser.parse_search_results(html_content, query, search_url)
|
||||
|
||||
# Convert ProductResult to SearchResult
|
||||
results = [cls._convert_to_search_result(p) for p in parsed_products]
|
||||
|
||||
# Scrape details for each result (in parallel)
|
||||
if results:
|
||||
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
|
||||
|
||||
# Enrich all results
|
||||
results_to_enrich = results
|
||||
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
|
||||
|
||||
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
|
||||
|
||||
async def scrape_with_limit(res):
|
||||
async with semaphore:
|
||||
return await cls._scrape_item_details(res, context)
|
||||
|
||||
tasks = [scrape_with_limit(r) for r in results_to_enrich]
|
||||
enriched_results = await asyncio.gather(*tasks)
|
||||
|
||||
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
|
||||
results = [r for r in enriched_results if r]
|
||||
|
||||
# If we got here, success!
|
||||
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
|
||||
return results
|
||||
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
finally:
|
||||
try:
|
||||
await context.close()
|
||||
except Exception as e:
|
||||
logger.debug(f"Error closing context (ignored): {e}")
|
||||
|
||||
except (PlaywrightTimeoutError, Exception) as e:
|
||||
# Check if it's a connection error or target closed
|
||||
error_msg = str(e).lower()
|
||||
is_connection_error = "target closed" in error_msg or "connection" in error_msg or "browser has been closed" in error_msg
|
||||
|
||||
if is_connection_error and attempt < max_retries:
|
||||
logger.warning(f"⚠️ Connection error during search for {site_key}: {e}. Retrying ({attempt+1}/{max_retries})...")
|
||||
# Force reconnect
|
||||
cls._browser = None
|
||||
await asyncio.sleep(1)
|
||||
continue
|
||||
|
||||
logger.error(f"❌ Error during search: {e}", exc_info=True)
|
||||
return []
|
||||
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def _convert_to_search_result(product: ProductResult) -> SearchResult:
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
import re
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -39,13 +40,45 @@ class GifiParser(BaseParser):
|
||||
if not title:
|
||||
continue
|
||||
|
||||
img_url = self.extract_image_url(product)
|
||||
# Image extraction
|
||||
# Priority: picture img -> img with class -> any img
|
||||
img_url = None
|
||||
|
||||
# 1. Try picture source (often high res)
|
||||
picture = product.select_one("picture")
|
||||
if picture:
|
||||
source = picture.find("source")
|
||||
if source and source.get("srcset"):
|
||||
img_url = source.get("srcset").split(",")[0].split()[0]
|
||||
|
||||
if not img_url:
|
||||
img = picture.find("img")
|
||||
if img:
|
||||
img_url = self._get_image_src(img)
|
||||
|
||||
# 2. Try direct image selectors
|
||||
if not img_url:
|
||||
img_url = self.extract_image_url(product, [
|
||||
"img.tile-image",
|
||||
"img[class*='product']",
|
||||
".image-container img"
|
||||
])
|
||||
|
||||
price = None
|
||||
price_el = product.select_one(".price, .value, [class*='price']")
|
||||
# 1. Try specific price selectors
|
||||
price_el = product.select_one(".price, .value, [class*='price'], .sales .value")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
# 2. Fallback: Regex on the entire product text
|
||||
if price is None:
|
||||
product_text = product.get_text(separator=" ", strip=True)
|
||||
# Look for price pattern: number followed by € or EUR
|
||||
# e.g. "7,99 €", "12 €", "12.50€"
|
||||
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', product_text, re.IGNORECASE)
|
||||
if price_match:
|
||||
price = self.parse_price_text(price_match.group(0))
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
url=url,
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from bs4 import BeautifulSoup
|
||||
from app.services.parsers.base_parser import BaseParser, ProductResult
|
||||
import logging
|
||||
import re
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -35,13 +36,38 @@ class LIncroyableParser(BaseParser):
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Image
|
||||
# Image - Robust extraction
|
||||
img_url = None
|
||||
img_el = card.select_one("img.imgCoup2coeur, img")
|
||||
if img_el:
|
||||
img_url = self._get_image_src(img_el)
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
# Exclude heart/wishlist icons explicitly
|
||||
# Select all images and filter
|
||||
images = card.select("img")
|
||||
valid_images = []
|
||||
for img in images:
|
||||
src = img.get('src', '') or img.get('data-src', '')
|
||||
classes = img.get('class', [])
|
||||
|
||||
# Skip heart/wishlist icons
|
||||
if 'coup2coeur' in str(classes).lower() or 'wishlist' in str(classes).lower():
|
||||
continue
|
||||
if 'coeur' in src.lower() or 'heart' in src.lower():
|
||||
continue
|
||||
|
||||
valid_images.append(img)
|
||||
|
||||
# Prioritize product images
|
||||
for img in valid_images:
|
||||
src = img.get('src', '') or img.get('data-src', '')
|
||||
if 'product' in src.lower() or 'p/' in src.lower():
|
||||
img_url = self._get_image_src(img)
|
||||
break
|
||||
|
||||
# Fallback to first valid image if no specific product image found
|
||||
if not img_url and valid_images:
|
||||
img_url = self._get_image_src(valid_images[0])
|
||||
|
||||
if img_url:
|
||||
img_url = self.make_absolute_url(img_url)
|
||||
|
||||
# Price
|
||||
price = None
|
||||
@@ -52,7 +78,10 @@ class LIncroyableParser(BaseParser):
|
||||
# Fallback price search in text
|
||||
if not price:
|
||||
text = card.get_text(" ", strip=True)
|
||||
price = self.parse_price_text(text)
|
||||
# Look for price pattern: number followed by €
|
||||
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*€', text)
|
||||
if price_match:
|
||||
price = self.parse_price_text(price_match.group(0))
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
|
||||
+35
-38
@@ -1,53 +1,50 @@
|
||||
import asyncio
|
||||
import logging
|
||||
from playwright.async_api import async_playwright
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def debug_gifi():
|
||||
url = "https://www.gifi.fr/catalogsearch/result/?q=lutin"
|
||||
print("Initializing search service...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
page = await browser.new_page()
|
||||
try:
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = await ImprovedSearchService.search_site("gifi.fr", "chaise")
|
||||
|
||||
logger.info(f"Navigating to {url}...")
|
||||
await page.goto(url, timeout=60000)
|
||||
print(f"\nFound {len(results)} results.")
|
||||
|
||||
# Wait for results
|
||||
try:
|
||||
await page.wait_for_selector(".products-grid, .product-items, .search-results, .products", timeout=10000)
|
||||
logger.info("Results container found.")
|
||||
except:
|
||||
logger.warning("Results container NOT found (timeout).")
|
||||
if results:
|
||||
print("\n--- First 5 Results ---")
|
||||
for i, res in enumerate(results[:5]):
|
||||
print(f"\nItem {i+1}:")
|
||||
print(f" Title: {res.title}")
|
||||
print(f" Price: {res.price} {res.currency}")
|
||||
print(f" Image: {res.image_url}")
|
||||
print(f" URL: {res.url}")
|
||||
print(f" In Stock: {res.in_stock}")
|
||||
|
||||
# Check for missing critical data
|
||||
missing_price = sum(1 for r in results if r.price is None)
|
||||
missing_image = sum(1 for r in results if not r.image_url)
|
||||
|
||||
# Get all links matching the current selector
|
||||
selector = ".product-item a.product-item-link, .product-item-info a, .product-item a, .products-grid a, a[href$='.html']"
|
||||
|
||||
# Let's try to be more specific and see what we get with different parts of the selector
|
||||
|
||||
# 1. Current full selector
|
||||
elements = await page.query_selector_all(selector)
|
||||
logger.info(f"Found {len(elements)} elements with current selector.")
|
||||
|
||||
for i, el in enumerate(elements[:10]):
|
||||
href = await el.get_attribute("href")
|
||||
text = await el.inner_text()
|
||||
logger.info(f"Link {i}: {href} | Text: {text.strip()[:50]}")
|
||||
|
||||
# 2. Try to find product specific classes
|
||||
logger.info("--- Inspecting specific product classes ---")
|
||||
product_items = await page.query_selector_all(".product-item")
|
||||
logger.info(f"Found {len(product_items)} .product-item elements")
|
||||
|
||||
if product_items:
|
||||
first_item = product_items[0]
|
||||
html = await first_item.inner_html()
|
||||
logger.info(f"First product item HTML snippet: {html[:500]}")
|
||||
|
||||
await browser.close()
|
||||
print(f"\nStats:")
|
||||
print(f" Total: {len(results)}")
|
||||
print(f" Missing Price: {missing_price}")
|
||||
print(f" Missing Image: {missing_image}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error: {e}", exc_info=True)
|
||||
finally:
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(debug_gifi())
|
||||
Reference in new issue
Block a user