Merge pull request #192 from R0m1k3/antigravity

feat: Implement improved search service with persistent browser and m…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-11-30 22:21:33 +01:00
commit 5e377e5d30
4 files changed
+203 -120

No files matched your search

+97 -73
View File
@@ -200,89 +200,113 @@ class ImprovedSearchService:
search_url = config["search_url"].format(query=quote_plus(query))
logger.info(f"🔍 Searching {config['name']} at {search_url}")
# Ensure browser is connected
if not await cls._ensure_browser_connected():
logger.error("Failed to establish browser connection")
return []
results = []
try:
context = await cls._create_context(cls._browser)
page = await context.new_page()
# Retry logic
max_retries = 1
for attempt in range(max_retries + 1):
try:
# Navigate to search page
logger.debug(f"Navigating to {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
logger.debug("Page loaded (domcontentloaded)")
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.debug("Network idle reached")
except PlaywrightTimeoutError:
logger.debug("Network idle timed out (non-critical)")
# Handle popups
await cls._handle_popups(page)
# Wait for content to load
wait_selector = config.get("wait_selector")
if wait_selector:
try:
await page.wait_for_selector(wait_selector, timeout=5000)
logger.debug(f"Wait selector found: {wait_selector}")
except PlaywrightTimeoutError:
logger.warning(f"Wait selector not found: {wait_selector}")
# Small delay for JS rendering
await page.wait_for_timeout(2000)
# Get HTML content
html_content = await page.content()
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
if len(html_content) < 5000:
logger.warning(f"⚠️ Page too small - possibly blocked")
# Ensure browser is connected (check before each attempt)
if not await cls._ensure_browser_connected():
logger.error("Failed to establish browser connection")
if attempt < max_retries:
continue
return []
# Parse results using specialized parser
parser = ParserFactory.get_parser(site_key)
parsed_products = parser.parse_search_results(html_content, query, search_url)
context = await cls._create_context(cls._browser)
try:
page = await context.new_page()
# Convert ProductResult to SearchResult
results = [cls._convert_to_search_result(p) for p in parsed_products]
try:
# Navigate to search page
logger.debug(f"Navigating to {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
logger.debug("Page loaded (domcontentloaded)")
# Scrape details for each result (in parallel)
if results:
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
# Enrich all results
results_to_enrich = results
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.debug("Network idle reached")
except PlaywrightTimeoutError:
logger.debug("Network idle timed out (non-critical)")
async def scrape_with_limit(res):
async with semaphore:
return await cls._scrape_item_details(res, context)
# Handle popups
await cls._handle_popups(page)
tasks = [scrape_with_limit(r) for r in results_to_enrich]
enriched_results = await asyncio.gather(*tasks)
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
results = [r for r in enriched_results if r]
# Wait for content to load
wait_selector = config.get("wait_selector")
if wait_selector:
try:
await page.wait_for_selector(wait_selector, timeout=5000)
logger.debug(f"Wait selector found: {wait_selector}")
except PlaywrightTimeoutError:
logger.warning(f"Wait selector not found: {wait_selector}")
finally:
await context.close()
# Small delay for JS rendering
await page.wait_for_timeout(2000)
except Exception as e:
logger.error(f"❌ Error during search: {e}", exc_info=True)
return []
# Get HTML content
html_content = await page.content()
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
return results
if len(html_content) < 5000:
logger.warning(f"⚠️ Page too small - possibly blocked")
return []
# Parse results using specialized parser
parser = ParserFactory.get_parser(site_key)
parsed_products = parser.parse_search_results(html_content, query, search_url)
# Convert ProductResult to SearchResult
results = [cls._convert_to_search_result(p) for p in parsed_products]
# Scrape details for each result (in parallel)
if results:
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
# Enrich all results
results_to_enrich = results
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
async def scrape_with_limit(res):
async with semaphore:
return await cls._scrape_item_details(res, context)
tasks = [scrape_with_limit(r) for r in results_to_enrich]
enriched_results = await asyncio.gather(*tasks)
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
results = [r for r in enriched_results if r]
# If we got here, success!
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
return results
finally:
await page.close()
finally:
try:
await context.close()
except Exception as e:
logger.debug(f"Error closing context (ignored): {e}")
except (PlaywrightTimeoutError, Exception) as e:
# Check if it's a connection error or target closed
error_msg = str(e).lower()
is_connection_error = "target closed" in error_msg or "connection" in error_msg or "browser has been closed" in error_msg
if is_connection_error and attempt < max_retries:
logger.warning(f"⚠️ Connection error during search for {site_key}: {e}. Retrying ({attempt+1}/{max_retries})...")
# Force reconnect
cls._browser = None
await asyncio.sleep(1)
continue
logger.error(f"❌ Error during search: {e}", exc_info=True)
return []
return []
@staticmethod
def _convert_to_search_result(product: ProductResult) -> SearchResult:
+35 -2
View File
@@ -1,6 +1,7 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
import re
logger = logging.getLogger(__name__)
@@ -39,13 +40,45 @@ class GifiParser(BaseParser):
if not title:
continue
img_url = self.extract_image_url(product)
# Image extraction
# Priority: picture img -> img with class -> any img
img_url = None
# 1. Try picture source (often high res)
picture = product.select_one("picture")
if picture:
source = picture.find("source")
if source and source.get("srcset"):
img_url = source.get("srcset").split(",")[0].split()[0]
if not img_url:
img = picture.find("img")
if img:
img_url = self._get_image_src(img)
# 2. Try direct image selectors
if not img_url:
img_url = self.extract_image_url(product, [
"img.tile-image",
"img[class*='product']",
".image-container img"
])
price = None
price_el = product.select_one(".price, .value, [class*='price']")
# 1. Try specific price selectors
price_el = product.select_one(".price, .value, [class*='price'], .sales .value")
if price_el:
price = self.parse_price_text(price_el.get_text())
# 2. Fallback: Regex on the entire product text
if price is None:
product_text = product.get_text(separator=" ", strip=True)
# Look for price pattern: number followed by € or EUR
# e.g. "7,99 €", "12 €", "12.50€"
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', product_text, re.IGNORECASE)
if price_match:
price = self.parse_price_text(price_match.group(0))
results.append(ProductResult(
title=title,
url=url,
+36 -7
View File
@@ -1,6 +1,7 @@
from bs4 import BeautifulSoup
from app.services.parsers.base_parser import BaseParser, ProductResult
import logging
import re
logger = logging.getLogger(__name__)
@@ -35,13 +36,38 @@ class LIncroyableParser(BaseParser):
if not title:
continue
# Image
# Image - Robust extraction
img_url = None
img_el = card.select_one("img.imgCoup2coeur, img")
if img_el:
img_url = self._get_image_src(img_el)
if img_url:
img_url = self.make_absolute_url(img_url)
# Exclude heart/wishlist icons explicitly
# Select all images and filter
images = card.select("img")
valid_images = []
for img in images:
src = img.get('src', '') or img.get('data-src', '')
classes = img.get('class', [])
# Skip heart/wishlist icons
if 'coup2coeur' in str(classes).lower() or 'wishlist' in str(classes).lower():
continue
if 'coeur' in src.lower() or 'heart' in src.lower():
continue
valid_images.append(img)
# Prioritize product images
for img in valid_images:
src = img.get('src', '') or img.get('data-src', '')
if 'product' in src.lower() or 'p/' in src.lower():
img_url = self._get_image_src(img)
break
# Fallback to first valid image if no specific product image found
if not img_url and valid_images:
img_url = self._get_image_src(valid_images[0])
if img_url:
img_url = self.make_absolute_url(img_url)
# Price
price = None
@@ -52,7 +78,10 @@ class LIncroyableParser(BaseParser):
# Fallback price search in text
if not price:
text = card.get_text(" ", strip=True)
price = self.parse_price_text(text)
# Look for price pattern: number followed by €
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*€', text)
if price_match:
price = self.parse_price_text(price_match.group(0))
results.append(ProductResult(
title=title,
+35 -38
View File
@@ -1,53 +1,50 @@
import asyncio
import logging
from playwright.async_api import async_playwright
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.improved_search_service import ImprovedSearchService
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def debug_gifi():
url = "https://www.gifi.fr/catalogsearch/result/?q=lutin"
print("Initializing search service...")
await ImprovedSearchService.initialize()
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
page = await browser.new_page()
try:
print("Searching Gifi for 'chaise'...")
results = await ImprovedSearchService.search_site("gifi.fr", "chaise")
logger.info(f"Navigating to {url}...")
await page.goto(url, timeout=60000)
print(f"\nFound {len(results)} results.")
# Wait for results
try:
await page.wait_for_selector(".products-grid, .product-items, .search-results, .products", timeout=10000)
logger.info("Results container found.")
except:
logger.warning("Results container NOT found (timeout).")
if results:
print("\n--- First 5 Results ---")
for i, res in enumerate(results[:5]):
print(f"\nItem {i+1}:")
print(f" Title: {res.title}")
print(f" Price: {res.price} {res.currency}")
print(f" Image: {res.image_url}")
print(f" URL: {res.url}")
print(f" In Stock: {res.in_stock}")
# Check for missing critical data
missing_price = sum(1 for r in results if r.price is None)
missing_image = sum(1 for r in results if not r.image_url)
# Get all links matching the current selector
selector = ".product-item a.product-item-link, .product-item-info a, .product-item a, .products-grid a, a[href$='.html']"
# Let's try to be more specific and see what we get with different parts of the selector
# 1. Current full selector
elements = await page.query_selector_all(selector)
logger.info(f"Found {len(elements)} elements with current selector.")
for i, el in enumerate(elements[:10]):
href = await el.get_attribute("href")
text = await el.inner_text()
logger.info(f"Link {i}: {href} | Text: {text.strip()[:50]}")
# 2. Try to find product specific classes
logger.info("--- Inspecting specific product classes ---")
product_items = await page.query_selector_all(".product-item")
logger.info(f"Found {len(product_items)} .product-item elements")
if product_items:
first_item = product_items[0]
html = await first_item.inner_html()
logger.info(f"First product item HTML snippet: {html[:500]}")
await browser.close()
print(f"\nStats:")
print(f" Total: {len(results)}")
print(f" Missing Price: {missing_price}")
print(f" Missing Image: {missing_image}")
except Exception as e:
logger.error(f"Error: {e}", exc_info=True)
finally:
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(debug_gifi())