mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: introduce centralized search configuration for scraping, including site selectors, proxies, and user agents.
This commit is contained in:
1 parent
36ae466583
commit
ffc2f0d632
2 files changed
+56
-44
No files matched your search
@@ -179,9 +179,9 @@ SITE_CONFIGS = {
|
||||
"action.com": {
|
||||
"name": "Action",
|
||||
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
|
||||
"product_selector": "div.product-card, div.card",
|
||||
"product_image_selector": "img",
|
||||
"wait_selector": "div.product-card, div.card",
|
||||
"product_selector": "div[data-testid='product-card']",
|
||||
"product_image_selector": "img[data-testid='product-card-image']",
|
||||
"wait_selector": "div[data-testid='product-card']",
|
||||
"category": "Discount",
|
||||
"requires_proxy": True,
|
||||
},
|
||||
|
||||
+53
-41
@@ -8,58 +8,70 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_action():
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(
|
||||
headless=True,
|
||||
proxy={
|
||||
"server": "http://142.111.48.253:7030",
|
||||
"username": "jasuwwjr",
|
||||
"password": "elbsx170nmnl"
|
||||
}
|
||||
)
|
||||
page = await browser.new_page()
|
||||
# Use a standard User Agent
|
||||
ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(user_agent=ua)
|
||||
page = await context.new_page()
|
||||
|
||||
# 1. Perform Search
|
||||
logger.info("--- Step 1: Searching for 'Chaise' on Action ---")
|
||||
search_url = "https://www.action.com/fr-fr/search/?q=Chaise"
|
||||
await page.goto(search_url, wait_until="domcontentloaded")
|
||||
await page.wait_for_timeout(5000) # Wait for JS to load
|
||||
|
||||
|
||||
# 2. Dump HTML
|
||||
try:
|
||||
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
|
||||
# Wait a bit for any JS redirects or challenges
|
||||
await page.wait_for_timeout(5000)
|
||||
except Exception as e:
|
||||
logger.error(f"Navigation error: {e}")
|
||||
|
||||
# 2. Take Screenshot
|
||||
await page.screenshot(path="action_debug.png")
|
||||
logger.info("Screenshot saved: action_debug.png")
|
||||
|
||||
# 3. Dump Content
|
||||
content = await page.content()
|
||||
logger.info(f"HTML Content Length: {len(content)}")
|
||||
|
||||
# Save HTML to file for analysis (optional, but good for debugging)
|
||||
with open("action_search.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
# 4. Check for specific text
|
||||
text = await page.inner_text("body")
|
||||
logger.info(f"Page Text (first 500 chars): {text[:500]}")
|
||||
|
||||
if "Challenge" in text or "human" in text or "Cloudflare" in text:
|
||||
logger.warning("⚠️ Cloudflare Challenge detected in text!")
|
||||
|
||||
# 3. Try to find product containers
|
||||
# Common selectors to test
|
||||
selectors = [
|
||||
".product-card",
|
||||
".product-item",
|
||||
".card",
|
||||
"div[class*='product']",
|
||||
"a[class*='product']"
|
||||
]
|
||||
# 5. Analyze Content for Product Links
|
||||
# Action product URLs contain "/p/"
|
||||
links = await page.evaluate("""
|
||||
Array.from(document.querySelectorAll('a'))
|
||||
.map(a => a.href)
|
||||
.filter(href => href.includes('/p/'))
|
||||
""")
|
||||
|
||||
# Print all classes found
|
||||
classes = await page.evaluate("Array.from(document.querySelectorAll('*')).map(e => e.className).filter(c => c).join(' ')")
|
||||
logger.info(f"Classes found: {classes[:1000]}")
|
||||
|
||||
# Check for specific text "Chaise" to see if results loaded
|
||||
if "Chaise" in content:
|
||||
logger.info("✅ 'Chaise' found in content")
|
||||
else:
|
||||
logger.warning("❌ 'Chaise' NOT found in content")
|
||||
logger.info(f"Found {len(links)} product links")
|
||||
if links:
|
||||
logger.info(f"First 5 links: {links[:5]}")
|
||||
|
||||
# Find the parent container of the first link
|
||||
parent_html = await page.evaluate("""
|
||||
(() => {
|
||||
const link = document.querySelector("a[href*='/p/']");
|
||||
return link ? link.parentElement.outerHTML : "Not found";
|
||||
})()
|
||||
""")
|
||||
logger.info(f"Parent HTML of first link: {parent_html[:500]}")
|
||||
|
||||
# Find the class of the link itself
|
||||
link_class = await page.evaluate("""
|
||||
(() => {
|
||||
const link = document.querySelector("a[href*='/p/']");
|
||||
return link ? link.className : "Not found";
|
||||
})()
|
||||
""")
|
||||
logger.info(f"Class of first link: {link_class}")
|
||||
|
||||
|
||||
for sel in selectors:
|
||||
count = await page.locator(sel).count()
|
||||
if count > 0:
|
||||
logger.info(f"Selector '{sel}' found {count} elements")
|
||||
# Print first element HTML
|
||||
first_html = await page.locator(sel).first.evaluate("el => el.outerHTML")
|
||||
logger.info(f"First element HTML ({sel}): {first_html[:500]}...")
|
||||
|
||||
await browser.close()
|
||||
|
||||
|
||||
Reference in new issue
Block a user