feat: implement web scraping debugger and consolidate search site configurations.

This commit is contained in:
Michael committed 2025-11-29 10:56:43 +01:00
1 parent bcc3b0900f
commit 97a8e7a951
3 files changed
+84 -60

No files matched your search

-60
View File
@@ -51,66 +51,6 @@ USER_AGENT_POOL = [
# Firefox Windows
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
# Edge Windows
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
]
def get_random_user_agent() -> str:
return random.choice(USER_AGENT_POOL)
# === SITE CONFIGURATIONS ===
SITE_CONFIGS = {
# === MAGASINS DISCOUNT ===
"gifi.fr": {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
"product_selector": "a.link, .product-item a.link, .product-tile a.link",
"wait_selector": ".product-item, .product-tile, .products-grid",
"category": "Discount",
"requires_proxy": False,
},
"stokomani.fr": {
"name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "a[href^='/products/']",
"wait_selector": "a[href^='/products/']",
"category": "Discount",
"requires_proxy": False,
},
"bmstores.fr": {
"name": "B&M",
"wait_selector": ".product-card, .search-results",
"category": "Discount",
"requires_proxy": False,
},
"lafoirfouille.fr": {
"name": "La Foir'Fouille",
"search_url": "https://www.lafoirfouille.fr/catalogsearch/result/?q={query}",
"product_selector": ".product-item a.product-item-link, .product-item-info a",
"wait_selector": ".products-grid, .product-items",
"category": "Discount",
"requires_proxy": False,
},
# === GRANDES SURFACES ===
"e.leclerc": {
"name": "E.Leclerc",
"search_url": "https://www.e.leclerc/recherche?q={query}",
"product_selector": "a[href*='/fp/'][href*='-']:not([href*='promo'])",
"wait_selector": "[data-testid='product-grid'], .search-results-list",
"category": "Grande Surface",
"requires_proxy": False,
},
"auchan.fr": {
"name": "Auchan",
"search_url": "https://www.auchan.fr/recherche?text={query}",
"product_selector": "a[href*='/p/'][href$='.html']",
"wait_selector": ".search-results, .product-grid",
"category": "Grande Surface",
"requires_proxy": False,
},
"carrefour.fr": {
"name": "Carrefour",
"search_url": "https://www.carrefour.fr/s?q={query}",
"product_selector": "a.product-card-click-wrapper[href^='/p/']",
"wait_selector": "a.product-card-click-wrapper[href^='/p/']",
"category": "Grande Surface",
"requires_proxy": False,
+17
View File
@@ -165,6 +165,23 @@ class BrowserlessService:
# Retry logic could be handled here or by caller,
# for now we return what we have, caller decides
# Handle pre-search interaction if configured (e.g. Centrakor store selection)
# We can't easily pass the config here without changing the signature,
# but we can check if the URL matches a site with pre_search_selector
from app.core.search_config import SITE_CONFIGS
for config in SITE_CONFIGS.values():
if config.get("pre_search_selector") and config["search_url"].split("/")[2] in url:
try:
selector = config["pre_search_selector"]
logger.info(f"Attempting pre-search interaction: {selector}")
if await page.locator(selector).is_visible(timeout=5000):
await page.click(selector)
logger.info("Clicked pre-search selector")
await asyncio.sleep(2) # Wait for transition
except Exception as e:
logger.warning(f"Pre-search interaction failed: {e}")
break
await self.handle_popups(page)
await self.simulate_human_behavior(page)
+67
View File
@@ -0,0 +1,67 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def debug_site(site_key: str, query: str = "chaise"):
print(f"--- Debugging {site_key} ---")
config = SITE_CONFIGS.get(site_key)
if not config:
print(f"Site {site_key} not found in config")
return
search_url = config["search_url"].format(query=query)
print(f"URL: {search_url}")
print("Fetching content...")
try:
html, screenshot_path = await browserless_service.get_page_content(
search_url,
use_proxy=config.get("requires_proxy", False),
wait_selector=config.get("wait_selector")
)
print(f"Screenshot saved to: {screenshot_path}")
print(f"HTML length: {len(html)}")
# Save HTML for inspection
with open(f"debug_{site_key}.html", "w", encoding="utf-8") as f:
f.write(html)
print(f"HTML saved to debug_{site_key}.html")
# Check if wait selector is present in HTML
if config.get("wait_selector"):
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
found = soup.select(config["wait_selector"])
print(f"Wait selector '{config['wait_selector']}' found: {len(found)} elements")
except Exception as e:
print(f"Error: {e}")
async def main():
if len(sys.argv) < 2:
print("Usage: python debug_scraper.py <site_key> [query]")
return
site_key = sys.argv[1]
query = sys.argv[2] if len(sys.argv) > 2 else "chaise"
await browserless_service.start()
try:
await debug_site(site_key, query)
finally:
await browserless_service.stop()
if __name__ == "__main__":
asyncio.run(main())