feat: Add ScraperService with Playwright for robust web scraping, including browser management, configurable scraping options, and automated popup handling.

This commit is contained in:
Michael committed 2025-12-18 10:56:06 +01:00
1 parent 0d68859a5b
commit 6fab64b229
3 files changed
+100

No files matched your search

+52
View File
@@ -96,3 +96,55 @@ class GifiParser(BaseParser):
logger.info(f"GifiParser found {len(results)} results")
return results
def parse_product_details(self, html: str, product_url: str) -> dict:
soup = BeautifulSoup(html, "html.parser")
# 1. Price extraction
price = None
# Specific Gifi product page price selectors
price_el = soup.select_one(".prices .price .value, .product-price .price .value, .price-sales .value")
if price_el:
price = self.parse_price_text(price_el.get_text())
if price is None:
# Fallback to schema.org data if present
import json
scripts = soup.find_all("script", type="application/ld+json")
for script in scripts:
if script.string:
try:
data = json.loads(script.string)
if isinstance(data, dict):
if data.get("@type") == "Product" and "offers" in data:
offers = data["offers"]
if isinstance(offers, list) and offers:
offers = offers[0]
if "price" in offers:
price = float(offers["price"])
break
elif data.get("@type") == "Offer" and "price" in data:
price = float(data["price"])
break
except:
pass
if price is None:
# Text fallback
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', soup.get_text(), re.IGNORECASE)
if price_match:
price = self.parse_price_text(price_match.group(0))
# 2. Stock extraction
in_stock = True # specific availability check might be complex, default true if page loads
# Check for "Out of stock" messages
exhausted_el = soup.select_one(".availability-msg.exhausted, .availability-msg.out-of-stock")
if exhausted_el:
in_stock = False
return {
"price": price,
"in_stock": in_stock,
"currency": "EUR"
}
+5
View File
@@ -32,6 +32,11 @@ POPUP_SELECTORS = [
"span.a-button-inner > input.a-button-input[type='submit']",
"form:has-text('Continuer les achats') input[type='submit']",
"[aria-labelledby='continue-shopping-label']",
# Didomi / Gifi
"#didomi-notice-agree-button",
"button[id='didomi-notice-agree-button']",
"span:has-text('Accepter & Fermer')",
"button:has-text('Accepter & Fermer')",
]
+43
View File
@@ -0,0 +1,43 @@
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.parsers.gifi_parser import GifiParser
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
def verify_fix():
print("Verifying Gifi parser fix...")
# Load the dump file (we know it exists from previous steps)
dump_path = "gifi_full.html"
if not os.path.exists(dump_path):
print(f"Error: {dump_path} not found.")
return
with open(dump_path, "r", encoding="utf-8") as f:
html = f.read()
parser = GifiParser()
# Dummy URL
url = "https://www.gifi.fr/test-product.html"
print("Parsing product details...")
details = parser.parse_product_details(html, url)
print("\n--- Extraction Results ---")
print(f"Bypass Price: {details.get('price')}")
print(f"Bypass Stock: {details.get('in_stock')}")
print(f"Currency: {details.get('currency')}")
if details.get('price') is not None:
print("\nSUCCESS: Price extracted successfully!")
else:
print("\nFAILURE: Price not found in dump.")
if __name__ == "__main__":
verify_fix()