mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
4 files changed
+167
No files matched your search
@@ -96,3 +96,55 @@ class GifiParser(BaseParser):
|
||||
|
||||
logger.info(f"GifiParser found {len(results)} results")
|
||||
return results
|
||||
|
||||
def parse_product_details(self, html: str, product_url: str) -> dict:
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
# 1. Price extraction
|
||||
price = None
|
||||
# Specific Gifi product page price selectors
|
||||
price_el = soup.select_one(".prices .price .value, .product-price .price .value, .price-sales .value")
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
|
||||
if price is None:
|
||||
# Fallback to schema.org data if present
|
||||
import json
|
||||
scripts = soup.find_all("script", type="application/ld+json")
|
||||
for script in scripts:
|
||||
if script.string:
|
||||
try:
|
||||
data = json.loads(script.string)
|
||||
if isinstance(data, dict):
|
||||
if data.get("@type") == "Product" and "offers" in data:
|
||||
offers = data["offers"]
|
||||
if isinstance(offers, list) and offers:
|
||||
offers = offers[0]
|
||||
if "price" in offers:
|
||||
price = float(offers["price"])
|
||||
break
|
||||
elif data.get("@type") == "Offer" and "price" in data:
|
||||
price = float(data["price"])
|
||||
break
|
||||
except:
|
||||
pass
|
||||
|
||||
if price is None:
|
||||
# Text fallback
|
||||
price_match = re.search(r'(\d+(?:[.,]\d+)?)\s*(?:€|EUR)', soup.get_text(), re.IGNORECASE)
|
||||
if price_match:
|
||||
price = self.parse_price_text(price_match.group(0))
|
||||
|
||||
# 2. Stock extraction
|
||||
in_stock = True # specific availability check might be complex, default true if page loads
|
||||
|
||||
# Check for "Out of stock" messages
|
||||
exhausted_el = soup.select_one(".availability-msg.exhausted, .availability-msg.out-of-stock")
|
||||
if exhausted_el:
|
||||
in_stock = False
|
||||
|
||||
return {
|
||||
"price": price,
|
||||
"in_stock": in_stock,
|
||||
"currency": "EUR"
|
||||
}
|
||||
@@ -32,6 +32,11 @@ POPUP_SELECTORS = [
|
||||
"span.a-button-inner > input.a-button-input[type='submit']",
|
||||
"form:has-text('Continuer les achats') input[type='submit']",
|
||||
"[aria-labelledby='continue-shopping-label']",
|
||||
# Didomi / Gifi
|
||||
"#didomi-notice-agree-button",
|
||||
"button[id='didomi-notice-agree-button']",
|
||||
"span:has-text('Accepter & Fermer')",
|
||||
"button:has-text('Accepter & Fermer')",
|
||||
]
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
URL = "https://www.gifi.fr/meuble-et-deco/decoration/bougie-et-senteur/diffuseur-et-senteur/encens-nag-champa-15-g/000000000000540823.html"
|
||||
|
||||
async def reproduce_scrape():
|
||||
logger.info("Starting reproduction script...")
|
||||
async with async_playwright() as p:
|
||||
# Launch browser (headless=True by default which is what we want for reproduction usually)
|
||||
# But for debugging blocking, sometimes headless=False helps. Let's start with True (default)
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
logger.info(f"Navigating to {URL}...")
|
||||
try:
|
||||
await page.goto(URL, wait_until="domcontentloaded", timeout=60000)
|
||||
logger.info("Page loaded.")
|
||||
|
||||
# Wait a bit for dynamic content
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Extract title
|
||||
title = await page.title()
|
||||
logger.info(f"Page Title: {title}")
|
||||
|
||||
# Extract body text
|
||||
content = await page.content()
|
||||
body_text = await page.inner_text("body")
|
||||
|
||||
logger.info(f"Content Length: {len(content)}")
|
||||
logger.info(f"Body Text Length: {len(body_text)}")
|
||||
|
||||
# Check for price
|
||||
if "€" in body_text:
|
||||
logger.info("Found '€' in body text.")
|
||||
else:
|
||||
logger.warning("'€' NOT found in body text.")
|
||||
|
||||
# specific check for likely price
|
||||
import re
|
||||
prices = re.findall(r'\d+[,\.]\d{2}\s*€', body_text)
|
||||
logger.info(f"Prices found in text: {prices}")
|
||||
|
||||
# Save content for review
|
||||
with open("gifi_reproduction.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
logger.info("Saved gifi_reproduction.html")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error during navigation/scraping: {e}")
|
||||
|
||||
finally:
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(reproduce_scrape())
|
||||
@@ -0,0 +1,43 @@
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.parsers.gifi_parser import GifiParser
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
def verify_fix():
|
||||
print("Verifying Gifi parser fix...")
|
||||
|
||||
# Load the dump file (we know it exists from previous steps)
|
||||
dump_path = "gifi_full.html"
|
||||
if not os.path.exists(dump_path):
|
||||
print(f"Error: {dump_path} not found.")
|
||||
return
|
||||
|
||||
with open(dump_path, "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
parser = GifiParser()
|
||||
# Dummy URL
|
||||
url = "https://www.gifi.fr/test-product.html"
|
||||
|
||||
print("Parsing product details...")
|
||||
details = parser.parse_product_details(html, url)
|
||||
|
||||
print("\n--- Extraction Results ---")
|
||||
print(f"Bypass Price: {details.get('price')}")
|
||||
print(f"Bypass Stock: {details.get('in_stock')}")
|
||||
print(f"Currency: {details.get('currency')}")
|
||||
|
||||
if details.get('price') is not None:
|
||||
print("\nSUCCESS: Price extracted successfully!")
|
||||
else:
|
||||
print("\nFAILURE: Price not found in dump.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
verify_fix()
|
||||
Reference in new issue
Block a user