Merge pull request #209 from R0m1k3/antigravity

feat: Introduce `ImprovedSearchService` utilizing Playwright for pers…
This commit is contained in:
LogiFlow authored and GitHub committed 2025-12-01 01:30:27 +01:00
commit cf621da596
4 files changed
+127 -28

No files matched your search

+18
View File
@@ -176,6 +176,24 @@ SITE_CONFIGS = {
"category": "Grande Surface",
"requires_proxy": False,
},
"action.com": {
"name": "Action",
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
"product_selector": "div.product-card, div.card",
"product_image_selector": "img",
"wait_selector": "div.product-card, div.card",
"category": "Discount",
"requires_proxy": True,
},
"e-leclerc.com": {
"name": "E.Leclerc",
"search_url": "https://www.e.leclerc/recherche?q={query}",
"product_selector": "a.product-card-link",
"product_image_selector": "img",
"wait_selector": "a.product-card-link",
"category": "Grande Surface",
"requires_proxy": False,
},
}
# === COOKIE BANNERS ===
+1 -1
View File
@@ -11,7 +11,7 @@ from typing import AsyncGenerator, Optional
from urllib.parse import quote_plus, urljoin, urlparse
from bs4 import BeautifulSoup
from playwright.async_api import Browser, BrowserContext, Page, async_playwright
from playwright.async_api import Browser, BrowserContext, Page, async_playwright, TimeoutError as PlaywrightTimeoutError
from sqlalchemy.orm import Session
from app.models import SearchSite
+58 -27
View File
@@ -1,36 +1,67 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.getcwd())
from app.services.direct_search_service import direct_search_service
from app.core.search_config import SITE_CONFIGS
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_site(site_key):
logger.info(f"Verifying {site_key}...")
try:
results = await direct_search_service.search_site(site_key, "chaise")
if results:
logger.info(f"✅ {site_key}: Found {len(results)} results")
for i, res in enumerate(results[:3]):
title = res.get('title', 'No Title')
price = res.get('price', 'No Price')
url = res.get('url', 'No URL')
logger.info(f" {i+1}. {title[:50]}... - {price} - {url[:50]}...")
else:
logger.error(f"❌ {site_key}: No results found")
except Exception as e:
logger.error(f"❌ {site_key}: Error - {e}")
async def verify_action():
async with async_playwright() as p:
browser = await p.chromium.launch(
headless=True,
proxy={
"server": "http://142.111.48.253:7030",
"username": "jasuwwjr",
"password": "elbsx170nmnl"
}
)
page = await browser.new_page()
# 1. Perform Search
logger.info("--- Step 1: Searching for 'Chaise' on Action ---")
search_url = "https://www.action.com/fr-fr/search/?q=Chaise"
await page.goto(search_url, wait_until="domcontentloaded")
await page.wait_for_timeout(5000) # Wait for JS to load
async def main():
await verify_site("action.com")
# 2. Dump HTML
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# Save HTML to file for analysis (optional, but good for debugging)
with open("action_search.html", "w", encoding="utf-8") as f:
f.write(content)
# 3. Try to find product containers
# Common selectors to test
selectors = [
".product-card",
".product-item",
".card",
"div[class*='product']",
"a[class*='product']"
]
# Print all classes found
classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')")
logger.info(f"Classes found: {classes[:1000]}")
# Check for specific text "Chaise" to see if results loaded
if "Chaise" in content:
logger.info("✅ 'Chaise' found in content")
else:
logger.warning("❌ 'Chaise' NOT found in content")
for sel in selectors:
count = await page.locator(sel).count()
if count > 0:
logger.info(f"Selector '{sel}' found {count} elements")
# Print first element HTML
first_html = await page.locator(sel).first.evaluate("el => el.outerHTML")
logger.info(f"First element HTML ({sel}): {first_html[:500]}...")
await browser.close()
if __name__ == "__main__":
asyncio.run(main())
asyncio.run(verify_action())
+50
View File
@@ -0,0 +1,50 @@
import asyncio
import logging
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_leclerc():
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True)
page = await browser.new_page()
# 1. Perform Search
logger.info("--- Step 1: Searching for 'Chaise' on E.Leclerc ---")
search_url = "https://www.e.leclerc/recherche?q=Chaise"
try:
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(5000) # Wait for JS
except Exception as e:
logger.error(f"Navigation failed: {e}")
# 2. Dump HTML
content = await page.content()
logger.info(f"HTML Content Length: {len(content)}")
# 3. Analyze Classes
classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')")
logger.info(f"Classes found: {classes[:1000]}")
# 4. Check for Product Selectors
selectors = [
"div[class*='product']",
"article",
".product-card",
".c-product-card",
"a[class*='product']"
]
for sel in selectors:
count = await page.locator(sel).count()
if count > 0:
logger.info(f"Selector '{sel}' found {count} elements")
first_html = await page.locator(sel).first.evaluate("el => el.outerHTML")
logger.info(f"First element HTML ({sel}): {first_html[:500]}...")
await browser.close()
if __name__ == "__main__":
asyncio.run(verify_leclerc())