mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
Fix: Action.com availability false positives and Amazon login detection optimization. Cleaned up diagnostic scripts.
This commit is contained in:
1 parent
7804e817c3
commit
8329eb28a3
58 files changed
+64
-112706
No files matched your search
-19490
File diff suppressed because it is too large.
Load diff
@@ -1,82 +0,0 @@
|
||||
"""
|
||||
Analyze B&M product page for price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# URL from the screenshot
|
||||
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
|
||||
print(f"Loading: {url}\n")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Find all elements with € symbol
|
||||
price_els = soup.find_all(string=re.compile('€'))
|
||||
print(f"Found {len(price_els)} elements with '€'\n")
|
||||
|
||||
prices_found = {}
|
||||
for el in price_els[:30]:
|
||||
text = el.strip()
|
||||
if text and len(text) < 100:
|
||||
parent = el.find_parent()
|
||||
if parent:
|
||||
parent_class = parent.get('class', [])
|
||||
parent_class_str = ' '.join(parent_class) if isinstance(parent_class, list) else str(parent_class)
|
||||
|
||||
# Extract price value
|
||||
price_match = re.search(r'(\d+[.,]\d+)\s*€', text)
|
||||
if price_match:
|
||||
price_val = price_match.group(1)
|
||||
key = f"{price_val}€ in .{parent_class_str[:50]}"
|
||||
if key not in prices_found:
|
||||
prices_found[key] = text
|
||||
|
||||
print("Prices found:")
|
||||
for key, text in prices_found.items():
|
||||
print(f" {key}: '{text}'")
|
||||
|
||||
# Try common price selectors
|
||||
print("\nTrying specific selectors:")
|
||||
selectors = [
|
||||
".price",
|
||||
"[class*='price']",
|
||||
"[data-price]",
|
||||
".product-price",
|
||||
"span[class*='price']"
|
||||
]
|
||||
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = soup.select(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = el.get_text(strip=True)
|
||||
if '€' in text:
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,50 +0,0 @@
|
||||
"""
|
||||
Analyze B&M with longer wait and playwright evaluation
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
|
||||
print(f"Loading: {url}\n")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
# Wait extra time for JS
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Look for elements containing price
|
||||
print("Searching for price elements...\n")
|
||||
|
||||
# Try to find any text with €
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"Found {len(price_els)} elements with €\n")
|
||||
|
||||
for i, el in enumerate(price_els[:10]):
|
||||
text = await el.inner_text()
|
||||
tag = await el.evaluate("el => el.tagName")
|
||||
classes = await el.evaluate("el => el.className")
|
||||
print(f"{i+1}. <{tag} class='{classes}'> {text[:100]}")
|
||||
|
||||
# Screenshot for debugging
|
||||
await page.screenshot(path="/app/debug_dumps/bm_screenshot.png", full_page=True)
|
||||
print("\nScreenshot saved to /app/debug_dumps/bm_screenshot.png")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,59 +0,0 @@
|
||||
"""
|
||||
Analyze Carrefour price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Carrefour search...")
|
||||
await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
products = soup.select("article.product-list-card-plp-grid-new")
|
||||
print(f"Found {len(products)} products\n")
|
||||
|
||||
for i, product in enumerate(products[:3]):
|
||||
print(f"=== Product {i+1} ===")
|
||||
|
||||
# Title
|
||||
title_el = product.select_one("h3, h2, a")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title[:60]}")
|
||||
|
||||
# Find all text with € symbol
|
||||
import re
|
||||
product_html = str(product)
|
||||
prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
|
||||
print(f"Prices found in HTML: {prices}")
|
||||
|
||||
# Look for price elements
|
||||
price_els = product.find_all(string=re.compile('€'))
|
||||
if price_els:
|
||||
print(f"Elements with €:")
|
||||
for el in price_els[:3]:
|
||||
print(f" - {el.strip()[:50]}")
|
||||
|
||||
print()
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,70 +0,0 @@
|
||||
"""
|
||||
Analyze a Carrefour product page for price selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
import re
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# Visit a Carrefour product page (from the screenshot)
|
||||
url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
|
||||
print(f"Loading: {url}")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Find all elements with € symbol
|
||||
price_els = soup.find_all(string=re.compile('€'))
|
||||
print(f"\nFound {len(price_els)} elements with '€'")
|
||||
|
||||
prices_found = set()
|
||||
for el in price_els[:20]:
|
||||
text = el.strip()
|
||||
if text and len(text) < 50:
|
||||
prices_found.add(text)
|
||||
parent = el.find_parent()
|
||||
print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
|
||||
|
||||
# Try common price selectors
|
||||
selectors = [
|
||||
".product-price",
|
||||
"[class*='price']",
|
||||
".price",
|
||||
"span.price",
|
||||
"div.price",
|
||||
"[data-price]"
|
||||
]
|
||||
|
||||
print("\nTrying specific selectors:")
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = soup.select(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = el.get_text(strip=True)
|
||||
if '€' in text:
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,69 +0,0 @@
|
||||
"""
|
||||
Analyze Centrakor HTML structure for image selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Centrakor search...")
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
|
||||
# Try to find product containers
|
||||
selectors = [
|
||||
"div.product-item",
|
||||
"div.product-card",
|
||||
"article",
|
||||
"div[class*='product']",
|
||||
"li[class*='product']"
|
||||
]
|
||||
|
||||
for selector in selectors:
|
||||
products = soup.select(selector)
|
||||
if products:
|
||||
print(f"\n✓ Found {len(products)} products with selector: {selector}")
|
||||
|
||||
# Analyze first product
|
||||
first = products[0]
|
||||
print(f"\nFirst product HTML snippet:")
|
||||
print(str(first)[:500])
|
||||
print("\n...")
|
||||
|
||||
# Find all images
|
||||
images = first.find_all('img')
|
||||
print(f"\nFound {len(images)} images in first product:")
|
||||
for i, img in enumerate(images):
|
||||
print(f"\n Image {i+1}:")
|
||||
print(f" Class: {img.get('class')}")
|
||||
print(f" Src: {img.get('src', '')[:80]}")
|
||||
print(f" Data-src: {img.get('data-src', '')[:80]}")
|
||||
print(f" Alt: {img.get('alt', '')[:50]}")
|
||||
|
||||
break
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,58 +0,0 @@
|
||||
"""
|
||||
Detailed analysis of Centrakor image structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Analyzing {min(5, len(products))} products:\n")
|
||||
|
||||
for i, product in enumerate(products[:5]):
|
||||
print(f"=== Product {i+1} ===")
|
||||
|
||||
# Title
|
||||
title_el = product.select_one("a.product-item__name")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title}")
|
||||
|
||||
# All images
|
||||
images = product.find_all('img')
|
||||
print(f"Found {len(images)} img tags")
|
||||
|
||||
for j, img in enumerate(images):
|
||||
print(f"\n Image {j+1}:")
|
||||
print(f" tag: {img.name}")
|
||||
print(f" class: {img.get('class')}")
|
||||
for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
|
||||
val = img.get(attr)
|
||||
if val:
|
||||
print(f" {attr}: {val[:80]}")
|
||||
|
||||
print()
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,24 +0,0 @@
|
||||
"""
|
||||
Analyze Gifi HTML to find correct selectors
|
||||
"""
|
||||
with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
# Find product-related divs
|
||||
import re
|
||||
matches = re.findall(r'<div[^>]*class="[^"]*"[^>]*>.*?</div>', html[:50000], re.DOTALL)
|
||||
|
||||
print(f"Total HTML size: {len(html)} bytes")
|
||||
|
||||
# Search for price patterns
|
||||
price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
|
||||
print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
|
||||
|
||||
# Find all class names containing specific keywords
|
||||
for keyword in ['product', 'article', 'item', 'card']:
|
||||
classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
|
||||
unique_classes = set(classes)
|
||||
if unique_classes:
|
||||
print(f"\nClasses containing '{keyword}':")
|
||||
for cls in sorted(unique_classes):
|
||||
print(f" - {cls}")
|
||||
@@ -1,69 +0,0 @@
|
||||
"""
|
||||
Analyze a Gifi product page to understand price structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# Visit a Gifi product page
|
||||
url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
|
||||
print(f"Loading: {url}")
|
||||
await page.goto(url, wait_until="networkidle")
|
||||
|
||||
# Get page title
|
||||
title = await page.title()
|
||||
print(f"Title: {title}")
|
||||
|
||||
# Find all elements with price-like text
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"\nFound {len(price_els)} elements with '€'")
|
||||
|
||||
# Get first 10 price elements
|
||||
for i, el in enumerate(price_els[:10]):
|
||||
text = await el.inner_text()
|
||||
tag = await el.evaluate("el => el.tagName")
|
||||
classes = await el.evaluate("el => el.className")
|
||||
print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
|
||||
|
||||
# Try specific selectors
|
||||
selectors = [
|
||||
".price",
|
||||
".product-price",
|
||||
"[class*='price']",
|
||||
"[data-price]",
|
||||
"span.price",
|
||||
"div.price"
|
||||
]
|
||||
|
||||
print("\nTrying specific selectors:")
|
||||
for selector in selectors:
|
||||
try:
|
||||
els = await page.query_selector_all(selector)
|
||||
if els:
|
||||
for el in els[:2]:
|
||||
text = await el.inner_text()
|
||||
print(f" {selector}: {text}")
|
||||
except:
|
||||
pass
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,71 +0,0 @@
|
||||
"""
|
||||
Test extracting price from Gifi search page HTML
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import re
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-tile")
|
||||
|
||||
print(f"Found {len(products)} products\n")
|
||||
|
||||
for i, product in enumerate(products[:3]):
|
||||
print(f"\\n=== Product {i+1} ===")
|
||||
|
||||
# Get title
|
||||
title_el = product.select_one("div.pdp-link a")
|
||||
title = title_el.get_text(strip=True) if title_el else "N/A"
|
||||
print(f"Title: {title}")
|
||||
|
||||
# Try to find price in product HTML
|
||||
product_html = product.prettify()
|
||||
|
||||
# Look for price patterns
|
||||
price_patterns = [
|
||||
r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
|
||||
r'€\s*(\d+)[,.](\d+)', # € 19,99
|
||||
r'(\d+)€(\d+)', # 19€99
|
||||
r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
|
||||
]
|
||||
|
||||
for pattern in price_patterns:
|
||||
matches = re.findall(pattern, product_html)
|
||||
if matches:
|
||||
print(f"Pattern '{pattern}': {matches[:3]}")
|
||||
|
||||
# Find all text with €
|
||||
euro_texts = product.find_all(string=re.compile('€'))
|
||||
if euro_texts:
|
||||
print(f"Texts with €:")
|
||||
for text in euro_texts[:5]:
|
||||
print(f" - {text.strip()[:80]}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("\nDone")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,93 +0,0 @@
|
||||
"""
|
||||
Script pour analyser les sélecteurs CSS des sites manquants
|
||||
"""
|
||||
import asyncio
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def analyze_site(url: str, site_name: str):
|
||||
print(f"\n=== Analyse de {site_name} ===")
|
||||
print(f"URL: {url}")
|
||||
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=False)
|
||||
page = await browser.new_page()
|
||||
|
||||
try:
|
||||
await page.goto(url, wait_until="networkidle", timeout=30000)
|
||||
await page.wait_for_timeout(3000)
|
||||
|
||||
# Try cookie banners
|
||||
cookie_selectors = [
|
||||
"button:has-text('Accepter')",
|
||||
"button:has-text('Tout accepter')",
|
||||
"#didomi-notice-agree-button",
|
||||
".didomi-continue-without-agreeing"
|
||||
]
|
||||
for selector in cookie_selectors:
|
||||
try:
|
||||
await page.click(selector, timeout=2000)
|
||||
print(f"✓ Cookie banner fermé: {selector}")
|
||||
break
|
||||
except:
|
||||
pass
|
||||
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
# Save HTML
|
||||
html = await page.content()
|
||||
filename = f"{site_name.lower().replace(' ', '_')}_search.html"
|
||||
with open(filename, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
print(f"✓ HTML sauvegardé: {filename}")
|
||||
|
||||
# Take screenshot
|
||||
screenshot_path = f"{site_name.lower().replace(' ', '_')}_search.png"
|
||||
await page.screenshot(path=screenshot_path, full_page=True)
|
||||
print(f"✓ Screenshot: {screenshot_path}")
|
||||
|
||||
# Test common product link selectors
|
||||
test_selectors = [
|
||||
"a.product-link",
|
||||
"a.product-name",
|
||||
"a[href*='/produit']",
|
||||
"a[href*='/product']",
|
||||
"a[href*='/p/']",
|
||||
".product-title a",
|
||||
".product-item a",
|
||||
"article a",
|
||||
"a.product",
|
||||
"a[itemprop='url']",
|
||||
]
|
||||
|
||||
print("\n--- Test de sélecteurs ---")
|
||||
for selector in test_selectors:
|
||||
try:
|
||||
elements = await page.query_selector_all(selector)
|
||||
if elements:
|
||||
print(f"✓ {selector}: {len(elements)} éléments trouvés")
|
||||
# Get first few hrefs
|
||||
for i, elem in enumerate(elements[:3]):
|
||||
href = await elem.get_attribute("href")
|
||||
text = await elem.inner_text()
|
||||
print(f" [{i+1}] {text[:50]} -> {href}")
|
||||
except Exception as e:
|
||||
pass
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Erreur: {e}")
|
||||
finally:
|
||||
await browser.close()
|
||||
|
||||
async def main():
|
||||
sites = [
|
||||
("https://bmstores.fr/module/ambjolisearch/jolisearch?s=chaise", "BM"),
|
||||
("https://www.centrakor.com/recherche?controller=search&s=chaise", "Centrakor"),
|
||||
("https://www.lincroyable.fr/recherche?query=chaise", "L'Incroyable"),
|
||||
]
|
||||
|
||||
for url, name in sites:
|
||||
await analyze_site(url, name)
|
||||
await asyncio.sleep(2)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -50,6 +50,10 @@ def _normalize_title(title: str) -> str:
|
||||
for pattern in patterns:
|
||||
title = re.sub(pattern, "", title)
|
||||
|
||||
# Separate numbers from letters (handles cases like 'Blanc40' or '100pièces')
|
||||
title = re.sub(r"([a-zA-Z])(\d)", r"\1 \2", title)
|
||||
title = re.sub(r"(\d)([a-zA-Z])", r"\1 \2", title)
|
||||
|
||||
# Remove extra spaces and punctuation that might differ between sites
|
||||
title = re.sub(r"[^\w\s]", " ", title)
|
||||
title = " ".join(title.split())
|
||||
@@ -323,7 +327,14 @@ async def process_item_check(item_id: int):
|
||||
html_lower = html_content.lower() if html_content else ""
|
||||
|
||||
if any(term in final_url_lower for term in unavailable_terms) or any(
|
||||
term in html_lower for term in unavailable_terms
|
||||
term in html_lower
|
||||
for term in [
|
||||
# Use more specific terms for HTML to avoid false positives in random text or meta
|
||||
"malheureusement, ce produit est actuellement indisponible",
|
||||
">produit indisponible<",
|
||||
">ce produit est indisponible<",
|
||||
"product-unavailable-message",
|
||||
]
|
||||
):
|
||||
logger.warning(f"Product unavailable confirmed for item {item_id} on Action.com")
|
||||
await loop.run_in_executor(
|
||||
|
||||
@@ -31,7 +31,6 @@ POPUP_SELECTORS = [
|
||||
"input[value='Continuer les achats']",
|
||||
"input[value='Continue shopping']",
|
||||
"input[value='Continue shopping']",
|
||||
|
||||
"form:has-text('Continuer les achats') input[type='submit']",
|
||||
"[aria-labelledby='continue-shopping-label']",
|
||||
"#sp-cc-accept",
|
||||
@@ -53,7 +52,6 @@ POPUP_SELECTORS = [
|
||||
]
|
||||
|
||||
|
||||
|
||||
# Random User-Agents to alternate fingerprint
|
||||
AMAZON_USER_AGENTS = [
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
||||
@@ -63,6 +61,7 @@ AMAZON_USER_AGENTS = [
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
|
||||
]
|
||||
|
||||
|
||||
@dataclass
|
||||
class ScrapeConfig:
|
||||
"""Configuration for scraping parameters."""
|
||||
@@ -174,6 +173,7 @@ class ScraperService:
|
||||
try:
|
||||
# Random delay to simulate human lead-in
|
||||
import random
|
||||
|
||||
await asyncio.sleep(random.uniform(0.5, 2.0))
|
||||
|
||||
await ScraperService._navigate_and_wait(page, url, timeout)
|
||||
@@ -207,7 +207,9 @@ class ScraperService:
|
||||
await page.wait_for_timeout(1500)
|
||||
|
||||
# Select the first Nancy store (Nancy Essey or Nancy Centre)
|
||||
select_btn = page.locator(".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')")
|
||||
select_btn = page.locator(
|
||||
".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')"
|
||||
)
|
||||
if await select_btn.count() > 0:
|
||||
logger.info("Selecting Nancy store...")
|
||||
await select_btn.first.click()
|
||||
@@ -244,10 +246,7 @@ class ScraperService:
|
||||
|
||||
@staticmethod
|
||||
async def _scrape_amazon_specific(
|
||||
url: str,
|
||||
item_id: int | None,
|
||||
config: ScrapeConfig,
|
||||
return_html: bool
|
||||
url: str, item_id: int | None, config: ScrapeConfig, return_html: bool
|
||||
) -> tuple[str | None, str, str, str]:
|
||||
"""
|
||||
Specialized scraping flow for Amazon to avoid bot detection and ensure good screenshots.
|
||||
@@ -261,6 +260,7 @@ class ScraperService:
|
||||
|
||||
# Determine base domain
|
||||
from urllib.parse import urlparse
|
||||
|
||||
parsed = urlparse(url)
|
||||
base_domain = f"{parsed.scheme}://{parsed.netloc}"
|
||||
|
||||
@@ -297,14 +297,20 @@ class ScraperService:
|
||||
|
||||
# Dismiss "Change Address" or specific Amazon location modals if any
|
||||
try:
|
||||
await page.evaluate("document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')")
|
||||
except: pass
|
||||
await page.evaluate(
|
||||
"document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')"
|
||||
)
|
||||
except:
|
||||
pass
|
||||
|
||||
# 5. Check for Bot Detection / CAPTCHA / Login (Content-based)
|
||||
# We do this AFTER popup removal because sometimes "Identifiez-vous" is in a dismissible modal
|
||||
content_check = await page.content()
|
||||
|
||||
if "Type the characters you see in this image" in content_check or "Saisissez les caractères que vous voyez" in content_check:
|
||||
if (
|
||||
"Type the characters you see in this image" in content_check
|
||||
or "Saisissez les caractères que vous voyez" in content_check
|
||||
):
|
||||
logger.error("🚫 Amazon CAPTCHA detected!")
|
||||
# Attempt refresh once
|
||||
logger.info("Retrying with refresh...")
|
||||
@@ -324,14 +330,8 @@ class ScraperService:
|
||||
await close_btn.first.click()
|
||||
await page.wait_for_timeout(1000)
|
||||
content_check = await page.content() # Refresh content
|
||||
except: pass
|
||||
|
||||
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
|
||||
# Retry once for Login wall too
|
||||
logger.info("⚠️ Amazon Login/Auth detected. Retrying with refresh...")
|
||||
await page.reload()
|
||||
await asyncio.sleep(3)
|
||||
content_check = await page.content()
|
||||
except:
|
||||
pass
|
||||
|
||||
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
|
||||
# Final Check: Do we have a product title?
|
||||
@@ -339,15 +339,28 @@ class ScraperService:
|
||||
# If NO title, it's a hard redirect/gate. Fail.
|
||||
try:
|
||||
title_check = page.locator("#productTitle, #title")
|
||||
if await title_check.count() > 0:
|
||||
if await title_check.count() > 0 and await title_check.first.is_visible():
|
||||
logger.info("⚠️ Login prompt detected but Product Title found. Ignoring/Hiding modal...")
|
||||
# Attempt to brute-force remove the modal overlay again just in case
|
||||
await page.evaluate("document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())")
|
||||
await page.evaluate(
|
||||
"() => document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())"
|
||||
)
|
||||
else:
|
||||
# No title found, now we can try to reload or fail
|
||||
logger.info("⚠️ Amazon Login/Auth detected and No Title found. Retrying with refresh...")
|
||||
await page.reload()
|
||||
await asyncio.sleep(3)
|
||||
content_check = await page.content()
|
||||
|
||||
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
|
||||
# Re-check title after reload
|
||||
if await title_check.count() > 0 and await title_check.first.is_visible():
|
||||
logger.info("⚠️ Title appeared after refresh despite login prompt.")
|
||||
else:
|
||||
logger.error("🚫 Amazon Login Prompt detected (Blocking)!")
|
||||
return None, "LOGIN_REQUIRED", final_url, page_title
|
||||
except:
|
||||
logger.error("🚫 Amazon Login Prompt detected (Error Checking Title)!")
|
||||
except Exception as e:
|
||||
logger.error(f"🚫 Amazon Login Prompt detected (Error Checking Title: {e})")
|
||||
return None, "LOGIN_REQUIRED", final_url, page_title
|
||||
|
||||
# 6. Wait for Main Image (Critical for screenshot)
|
||||
@@ -355,8 +368,7 @@ class ScraperService:
|
||||
try:
|
||||
# Main image container on desktop
|
||||
await page.wait_for_selector(
|
||||
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper",
|
||||
timeout=10000
|
||||
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper", timeout=10000
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not find main image container: {e}")
|
||||
@@ -396,6 +408,7 @@ class ScraperService:
|
||||
"""Create context with advanced stealth and headers (specifically for Amazon)"""
|
||||
# Determine base domain for referer
|
||||
from urllib.parse import urlparse
|
||||
|
||||
parsed = urlparse(url)
|
||||
base_domain = f"{parsed.scheme}://{parsed.netloc}/"
|
||||
|
||||
@@ -422,8 +435,8 @@ class ScraperService:
|
||||
"Sec-Fetch-Site": "none",
|
||||
"Sec-Fetch-User": "?1",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Referer": base_domain if "amazon" in url else "https://www.google.com/"
|
||||
}
|
||||
"Referer": base_domain if "amazon" in url else "https://www.google.com/",
|
||||
},
|
||||
)
|
||||
|
||||
# Advanced Stealth mode
|
||||
@@ -497,7 +510,8 @@ class ScraperService:
|
||||
try:
|
||||
await page.keyboard.press("Escape")
|
||||
await page.wait_for_timeout(200)
|
||||
except: pass
|
||||
except:
|
||||
pass
|
||||
|
||||
# 2. Javascript cleanup (Hide pesky overlays and CMPs that won't close)
|
||||
logger.info("Injecting CSS/JS cleanup for persistent overlays...")
|
||||
@@ -612,7 +626,8 @@ class ScraperService:
|
||||
|
||||
# Collapse whitespace
|
||||
import re
|
||||
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
|
||||
|
||||
clean_text = re.sub(r"\s+", " ", clean_text).strip()
|
||||
|
||||
page_text = clean_text[:text_length]
|
||||
logger.info(f"Extracted {len(page_text)} chars")
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
from app.services.improved_search_service import BeautifulSoup
|
||||
print("BeautifulSoup imported successfully")
|
||||
@@ -1,12 +0,0 @@
|
||||
import sys
|
||||
import os
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
import app.core.search_config
|
||||
print(f"Search Config File: {app.core.search_config.__file__}")
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
print(f"Keys: {list(SITE_CONFIGS.keys())}")
|
||||
|
||||
with open(app.core.search_config.__file__, 'r') as f:
|
||||
content = f.read()
|
||||
print(f"File content length: {len(content)}")
|
||||
print(f"Contains stokomani.fr: {'stokomani.fr' in content}")
|
||||
@@ -1,10 +0,0 @@
|
||||
import sys
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
print("Attributes of ImprovedSearchService:")
|
||||
attrs = dir(ImprovedSearchService)
|
||||
if 'search_site_generator' in attrs:
|
||||
print("SUCCESS: search_site_generator found")
|
||||
else:
|
||||
print("FAILURE: search_site_generator NOT found")
|
||||
print("Available attributes:", [a for a in attrs if not a.startswith('__')])
|
||||
@@ -1,52 +0,0 @@
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
|
||||
# Add app to path
|
||||
sys.path.insert(0, os.getcwd())
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from app.core.search_config import get_amazon_proxies
|
||||
|
||||
|
||||
async def check_proxy(proxy, semaphore):
|
||||
async with semaphore:
|
||||
proxy_url = proxy["server"]
|
||||
print(f"Testing {proxy_url}...")
|
||||
|
||||
async with async_playwright() as p:
|
||||
# Connect to browserless or launch local
|
||||
# Using launch local for simpler testing without ws dependency if possible
|
||||
# But the app uses browserless. Let's try launch first.
|
||||
try:
|
||||
browser = await p.chromium.launch(headless=True, proxy=proxy)
|
||||
page = await browser.new_page()
|
||||
try:
|
||||
# amazon.fr might block, use httpbin for connectivity check
|
||||
await page.goto("http://httpbin.org/ip", timeout=15000)
|
||||
content = await page.content()
|
||||
print(f"✅ {proxy_url}: Success")
|
||||
await browser.close()
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f"❌ {proxy_url}: Failed - {str(e)[:100]}")
|
||||
await browser.close()
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f"❌ {proxy_url}: Launch Failed - {str(e)[:100]}")
|
||||
return False
|
||||
|
||||
|
||||
async def main():
|
||||
proxies = get_amazon_proxies()
|
||||
print(f"Checking {len(proxies)} proxies...")
|
||||
|
||||
semaphore = asyncio.Semaphore(3) # Limit concurrency
|
||||
results = await asyncio.gather(*[check_proxy(p, semaphore) for p in proxies])
|
||||
|
||||
working = sum(results)
|
||||
print(f"\nSummary: {working}/{len(proxies)} working.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,7 +0,0 @@
|
||||
import sys
|
||||
try:
|
||||
from app.services import improved_search_service
|
||||
print("Import successful")
|
||||
except Exception as e:
|
||||
print(f"Import failed: {e}")
|
||||
sys.exit(1)
|
||||
@@ -1,56 +0,0 @@
|
||||
"""
|
||||
Deep analysis - compare products WITH images vs WITHOUT images
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Analyzing {len(products)} products for image patterns\n")
|
||||
|
||||
for i, product in enumerate(products[:10]):
|
||||
# Get title
|
||||
title_el = product.select_one("a.product-item__name")
|
||||
title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
|
||||
|
||||
# Get ALL images
|
||||
all_imgs = product.select("img.responsive-image__actual")
|
||||
|
||||
print(f"\n=== {i+1}. {title[:50]} ===")
|
||||
print(f"Found {len(all_imgs)} images")
|
||||
|
||||
for j, img in enumerate(all_imgs):
|
||||
src = img.get('src', '')
|
||||
print(f" Image {j+1}: {src if src else '(no src)'}")
|
||||
if not src:
|
||||
# Check other attributes
|
||||
for attr in ['data-src', 'data-lazy-src', 'srcset']:
|
||||
val = img.get(attr)
|
||||
if val:
|
||||
print(f" {attr}: {val[:80]}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,71 +0,0 @@
|
||||
"""
|
||||
Quick diagnostic script to analyze actual HTML structure of search pages
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
from pathlib import Path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.search_service import NewSearchService
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
async def diagnose_site(site_key: str, query: str = "chaise"):
|
||||
"""Run a search and show detailed diagnostics"""
|
||||
config = SITE_CONFIGS.get(site_key)
|
||||
if not config:
|
||||
print(f"❌ Site '{site_key}' not found in config")
|
||||
return
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"🔍 Diagnosing: {config['name']} ({site_key})")
|
||||
print(f"{'='*80}")
|
||||
print(f"Search URL: {config['search_url'].format(query=query)}")
|
||||
print(f"Product Selector: {config['product_selector']}")
|
||||
print(f"Image Selector: {config.get('product_image_selector', 'NONE')}")
|
||||
|
||||
try:
|
||||
results = await NewSearchService.search_site(site_key, query)
|
||||
|
||||
print(f"\n📊 Results: {len(results)} products found")
|
||||
|
||||
if len(results) == 0:
|
||||
print("⚠️ NO RESULTS - Check if product_selector is correct")
|
||||
else:
|
||||
print("\n✅ Sample Results:")
|
||||
for i, result in enumerate(results[:3], 1):
|
||||
print(f"\n {i}. {result.title[:60]}")
|
||||
print(f" URL: {result.url[:80]}")
|
||||
print(f" Image: {result.image_url[:80] if result.image_url else '❌ NONE'}")
|
||||
print(f" Price: {result.price}€" if result.price else " Price: ❌ NONE")
|
||||
|
||||
# Count images
|
||||
with_images = sum(1 for r in results if r.image_url)
|
||||
print(f"\n📈 Images: {with_images}/{len(results)} ({with_images/len(results)*100:.0f}%)")
|
||||
|
||||
with_prices = sum(1 for r in results if r.price)
|
||||
print(f"💰 Prices: {with_prices}/{len(results)} ({with_prices/len(results)*100:.0f}%)")
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ ERROR: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
|
||||
async def main():
|
||||
"""Diagnose problematic sites"""
|
||||
sites_to_check = [
|
||||
"e-leclerc.com",
|
||||
"auchan.fr",
|
||||
"carrefour.fr",
|
||||
"stokomani.fr",
|
||||
"centrakor.com",
|
||||
"cdiscount.com",
|
||||
"lincroyable.fr"
|
||||
]
|
||||
|
||||
for site_key in sites_to_check:
|
||||
await diagnose_site(site_key)
|
||||
print("\n" + "="*80 + "\n")
|
||||
await asyncio.sleep(1) # Rate limiting
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,56 +0,0 @@
|
||||
"""
|
||||
Script to dump Gifi HTML and analyze structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search page...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
|
||||
|
||||
# Wait for products
|
||||
try:
|
||||
await page.wait_for_selector("article.product-miniature", timeout=10000)
|
||||
except:
|
||||
pass
|
||||
|
||||
# Save HTML
|
||||
content = await page.content()
|
||||
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||
with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
|
||||
print(f"HTML saved ({len(content)} bytes)")
|
||||
|
||||
# Extract first product structure
|
||||
products = await page.query_selector_all("article.product-miniature")
|
||||
print(f"Found {len(products)} products")
|
||||
|
||||
if products:
|
||||
first_html = await products[0].evaluate("el => el.outerHTML")
|
||||
with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
|
||||
f.write(first_html)
|
||||
print(f"First product HTML saved")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("Done")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,76 +0,0 @@
|
||||
"""
|
||||
Dump Gifi with longer wait for JavaScript
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def main():
|
||||
print("Connecting to browserless...")
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
print("Loading Gifi search page...")
|
||||
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
|
||||
|
||||
# Wait for ANY content
|
||||
print("Waiting for content...")
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Save HTML
|
||||
content = await page.content()
|
||||
os.makedirs("/app/debug_dumps", exist_ok=True)
|
||||
with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
|
||||
print(f"HTML saved ({len(content)} bytes)")
|
||||
|
||||
# Find any elements with price
|
||||
price_els = await page.query_selector_all("*:has-text('€')")
|
||||
print(f"Elements with € symbol: {len(price_els)}")
|
||||
|
||||
# Find all divs/articles
|
||||
all_divs = await page.query_selector_all("div, article, li")
|
||||
print(f"Total divs/articles/li: {len(all_divs)}")
|
||||
|
||||
# Screenshot
|
||||
await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
|
||||
print("Screenshot saved")
|
||||
|
||||
# Get all classes
|
||||
all_classes = await page.evaluate("""() => {
|
||||
const elements = document.querySelectorAll('*');
|
||||
const classes = new Set();
|
||||
elements.forEach(el => {
|
||||
if (el.className && typeof el.className === 'string') {
|
||||
el.className.split(' ').forEach(cls => {
|
||||
if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
|
||||
classes.add(cls);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
return Array.from(classes);
|
||||
}""")
|
||||
|
||||
print(f"\\nProduct-related classes found:")
|
||||
for cls in all_classes:
|
||||
print(f" - {cls}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
print("Done")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,88 +0,0 @@
|
||||
"""
|
||||
Quick HTML Dumper - Saves raw HTML from search pages for manual analysis
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
from pathlib import Path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.browserless_service import browserless_service
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
async def dump_search_html(site_key: str, query: str = "chaise"):
|
||||
"""Download and save raw HTML for manual inspection"""
|
||||
config = SITE_CONFIGS.get(site_key)
|
||||
if not config:
|
||||
print(f"❌ Site '{site_key}' not found")
|
||||
return
|
||||
|
||||
print(f"\n🔍 Dumping HTML for: {config['name']}")
|
||||
|
||||
# Ensure browser is initialized
|
||||
await browserless_service.initialize()
|
||||
|
||||
try:
|
||||
search_url = config["search_url"].format(query=query)
|
||||
print(f" URL: {search_url}")
|
||||
|
||||
# Override wait_selector for La Foir'Fouille
|
||||
wait_selector = config.get("wait_selector")
|
||||
if site_key == "lafoirfouille.fr":
|
||||
wait_selector = ".sf-grid-vignet"
|
||||
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
|
||||
|
||||
html_content, screenshot_path = await browserless_service.get_page_content(
|
||||
search_url,
|
||||
wait_selector=wait_selector,
|
||||
use_proxy=config.get("requires_proxy", False)
|
||||
)
|
||||
|
||||
if not html_content:
|
||||
print(" ❌ No HTML content returned")
|
||||
return
|
||||
|
||||
filename = f"dump_{site_key.replace('.', '_')}.html"
|
||||
with open(filename, "w", encoding="utf-8") as f:
|
||||
f.write(html_content)
|
||||
|
||||
print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
|
||||
|
||||
# Quick analysis
|
||||
from bs4 import BeautifulSoup
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
|
||||
# Try current selector
|
||||
current_selector = config.get("product_selector")
|
||||
matches = soup.select(current_selector)
|
||||
print(f" 📊 Current selector '{current_selector}' matches: {len(matches)}")
|
||||
|
||||
# Try image selector
|
||||
if "product_image_selector" in config:
|
||||
img_selector = config["product_image_selector"]
|
||||
img_matches = soup.select(img_selector)
|
||||
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
|
||||
|
||||
except Exception as e:
|
||||
print(f" ❌ Error during dump: {e}")
|
||||
|
||||
finally:
|
||||
# We don't close the browser here to allow reuse if needed,
|
||||
# but main() will shut it down.
|
||||
pass
|
||||
|
||||
async def main():
|
||||
sites = [
|
||||
"stokomani.fr"
|
||||
]
|
||||
|
||||
for site_key in sites:
|
||||
try:
|
||||
await dump_search_html(site_key)
|
||||
except Exception as e:
|
||||
print(f"❌ Error: {e}")
|
||||
await asyncio.sleep(1)
|
||||
|
||||
await browserless_service.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,103 +0,0 @@
|
||||
import json
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
|
||||
proxies_raw = """
|
||||
46.161.6.165:8080
|
||||
78.47.219.204:3128
|
||||
134.209.29.120:8080
|
||||
161.35.70.249:80
|
||||
134.209.29.120:80
|
||||
52.188.28.218:3128
|
||||
209.97.150.167:3128
|
||||
62.60.151.128:80
|
||||
68.235.35.171:3128
|
||||
209.97.150.167:80
|
||||
159.203.61.169:8080
|
||||
209.97.150.167:8080
|
||||
195.158.8.123:3128
|
||||
208.87.243.199:7878
|
||||
144.76.42.215:8118
|
||||
216.229.112.25:8080
|
||||
159.203.61.169:80
|
||||
103.3.246.71:3128
|
||||
138.68.60.8:80
|
||||
139.59.1.14:80
|
||||
8.243.68.11:8080
|
||||
41.223.119.156:3128
|
||||
34.96.238.40:8080
|
||||
59.6.25.118:3128
|
||||
129.150.39.251:8000
|
||||
162.240.154.26:3128
|
||||
35.152.252.253:8080
|
||||
144.125.164.158:8081
|
||||
47.81.14.7:3129
|
||||
144.125.164.222:8080
|
||||
175.99.220.171:80
|
||||
8.219.97.248:80
|
||||
144.125.164.158:8080
|
||||
164.68.110.241:8091
|
||||
144.125.164.222:8081
|
||||
140.238.184.182:3128
|
||||
139.59.1.14:3128
|
||||
8.212.160.196:8080
|
||||
164.68.110.241:9992
|
||||
173.212.246.157:3128
|
||||
47.236.130.95:3128
|
||||
103.147.246.18:8080
|
||||
128.199.202.122:80
|
||||
200.24.159.230:8080
|
||||
128.199.202.122:8080
|
||||
103.166.158.251:1111
|
||||
59.153.16.214:1120
|
||||
43.224.118.155:1121
|
||||
89.43.132.247:8080
|
||||
182.253.62.190:8080
|
||||
193.95.53.131:8077
|
||||
203.196.8.6:3128
|
||||
103.245.110.198:1452
|
||||
45.180.140.241:8080
|
||||
212.2.254.246:3128
|
||||
103.220.206.110:8585
|
||||
103.157.79.145:1080
|
||||
45.87.140.155:8080
|
||||
164.138.205.119:8080
|
||||
137.59.51.243:1120
|
||||
38.210.179.77:999
|
||||
27.147.163.188:40544
|
||||
194.87.77.22:80
|
||||
20.27.219.85:8080
|
||||
49.254.245.70:15648
|
||||
115.144.173.67:15648
|
||||
"""
|
||||
|
||||
proxy_list = [p.strip() for p in proxies_raw.strip().split("\n") if p.strip()]
|
||||
ips = [p.split(":")[0] for p in proxy_list]
|
||||
|
||||
chunk_size = 100
|
||||
fr_proxies = []
|
||||
|
||||
for i in range(0, len(ips), chunk_size):
|
||||
chunk = ips[i : i + chunk_size]
|
||||
try:
|
||||
req = urllib.request.Request("http://ip-api.com/batch", data=json.dumps(chunk).encode("utf-8"))
|
||||
with urllib.request.urlopen(req) as response:
|
||||
data = json.loads(response.read().decode("utf-8"))
|
||||
|
||||
for idx, result in enumerate(data):
|
||||
if result.get("countryCode") == "FR":
|
||||
full_proxy = proxy_list[i + idx]
|
||||
fr_proxies.append(full_proxy)
|
||||
print(f"Found FR proxy: {full_proxy}")
|
||||
except Exception as e:
|
||||
print(f"Error querying batch: {e}")
|
||||
|
||||
print(f"Total FR proxies found: {len(fr_proxies)}")
|
||||
|
||||
if len(fr_proxies) > 0:
|
||||
for p in fr_proxies:
|
||||
print(f"PROXY:{p}")
|
||||
else:
|
||||
print("No French proxies found. Printing first 10 generic ones as backup:")
|
||||
for p in proxy_list[:10]:
|
||||
print(f"PROXY:{p}")
|
||||
@@ -1,118 +0,0 @@
|
||||
import urllib.request
|
||||
import logging
|
||||
import concurrent.futures
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("proxy_finder")
|
||||
|
||||
# Target URL for verification
|
||||
TARGET_URL = "https://www.amazon.fr"
|
||||
|
||||
|
||||
def fetch_proxy_list(url):
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
url, data=None, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=10) as response:
|
||||
if response.status == 200:
|
||||
text = response.read().decode("utf-8")
|
||||
proxies = [p.strip() for p in text.splitlines() if p.strip() and ":" in p]
|
||||
logger.info(f"Fetched {len(proxies)} proxies from {url}")
|
||||
return proxies
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to fetch from {url}: {e}")
|
||||
return []
|
||||
|
||||
|
||||
def check_proxy_fast(proxy):
|
||||
try:
|
||||
proxy_handler = urllib.request.ProxyHandler({"http": proxy, "https": proxy})
|
||||
opener = urllib.request.build_opener(proxy_handler)
|
||||
opener.addheaders = [("User-Agent", "Mozilla/5.0")]
|
||||
with opener.open(TARGET_URL, timeout=5) as response:
|
||||
# 200, 403, 503 all mean the proxy is alive (amazon may block but proxy works)
|
||||
if response.status in [200, 403, 503]:
|
||||
return proxy
|
||||
except:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def main():
|
||||
logger.info("Starting proxy finder (FAST MODE)...")
|
||||
|
||||
sources = [
|
||||
"https://api.proxyscrape.com/v2/?request=getproxies&protocol=http&timeout=10000&country=all&ssl=all&anonymity=all",
|
||||
"https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt",
|
||||
"https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt",
|
||||
"https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt",
|
||||
]
|
||||
|
||||
# 1. Fetch all proxies
|
||||
all_proxies = set()
|
||||
for url in sources:
|
||||
proxies = fetch_proxy_list(url)
|
||||
if proxies:
|
||||
all_proxies.update(proxies)
|
||||
|
||||
# Add local raw proxies
|
||||
local_raw = [
|
||||
"164.68.110.241:8091",
|
||||
"164.68.110.241:9992",
|
||||
"173.212.246.157:3128",
|
||||
"142.111.48.253:7030",
|
||||
"31.59.20.176:6754",
|
||||
"23.95.150.145:6114",
|
||||
"198.23.239.134:6540",
|
||||
"107.172.163.27:6543",
|
||||
"198.105.121.200:6462",
|
||||
"64.137.96.74:6641",
|
||||
"84.247.60.125:6095",
|
||||
"216.10.27.159:6837",
|
||||
"142.111.67.146:5611",
|
||||
]
|
||||
all_proxies.update(local_raw)
|
||||
|
||||
print(f"\nTesting {len(all_proxies)} unique proxies against {TARGET_URL}...")
|
||||
|
||||
working_proxies = []
|
||||
|
||||
# Use ThreadPoolExecutor for speed
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor:
|
||||
future_to_proxy = {executor.submit(check_proxy_fast, p): p for p in all_proxies}
|
||||
|
||||
count = 0
|
||||
total = len(all_proxies)
|
||||
|
||||
for future in concurrent.futures.as_completed(future_to_proxy):
|
||||
count += 1
|
||||
if count % 500 == 0:
|
||||
print(f"Processed {count}/{total} - Found {len(working_proxies)} so far")
|
||||
|
||||
res = future.result()
|
||||
if res:
|
||||
print(f"ALIVE: {res}")
|
||||
working_proxies.append(res)
|
||||
# Stop if we have enough
|
||||
if len(working_proxies) >= 30:
|
||||
print("Found 30 proxies, stopping.")
|
||||
executor.shutdown(wait=False, cancel_futures=True)
|
||||
break
|
||||
|
||||
print("\n" + "=" * 50)
|
||||
print(f"FOUND {len(working_proxies)} WORKING PROXIES")
|
||||
print("=" * 50)
|
||||
|
||||
# Format for python list
|
||||
formatted_list = "[\n" + ",\n".join([f' "{p}"' for p in working_proxies]) + "\n]"
|
||||
print(formatted_list)
|
||||
|
||||
# Save to file
|
||||
with open("working_proxies.txt", "w") as f:
|
||||
f.write("\n".join(working_proxies))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Whitespace-only changes.
-18821
File diff suppressed because it is too large.
Load diff
@@ -1,33 +0,0 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
articles = soup.select("article.product-list-card-plp-grid-new")
|
||||
|
||||
print(f"Found {len(articles)} articles")
|
||||
|
||||
if articles:
|
||||
first = articles[0]
|
||||
print("\n--- First Article Structure ---")
|
||||
print(first.prettify()[:1000]) # Print first 1000 chars
|
||||
|
||||
# Check for link
|
||||
link = first.select_one("a.product-card-click-wrapper")
|
||||
if link:
|
||||
print(f"\nLink found: {link.get('href')}")
|
||||
print(f"Link classes: {link.get('class')}")
|
||||
|
||||
# Check for image INSIDE link
|
||||
img = link.select_one("img.product-card-image-new__content")
|
||||
if img:
|
||||
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
|
||||
else:
|
||||
print(f"\n❌ Image NOT found inside link")
|
||||
# Check if image is elsewhere in article
|
||||
img_article = first.select_one("img.product-card-image-new__content")
|
||||
if img_article:
|
||||
print(f" But image exists in article: {img_article.get('src')}")
|
||||
else:
|
||||
print("\nNo link found with selector a.product-card-click-wrapper")
|
||||
@@ -1,53 +0,0 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
# Try to find product containers
|
||||
print("Searching for product containers...")
|
||||
potential_selectors = [
|
||||
"div.product-miniature",
|
||||
"article",
|
||||
"div[class*='product']",
|
||||
"div.product-card",
|
||||
"div.item"
|
||||
]
|
||||
|
||||
for selector in potential_selectors:
|
||||
matches = soup.select(selector)
|
||||
print(f"Selector '{selector}' matches: {len(matches)}")
|
||||
if len(matches) > 0 and len(matches) < 5:
|
||||
# If few matches, print classes to see if it's a wrapper
|
||||
print(f" Classes: {matches[0].get('class')}")
|
||||
|
||||
# Print structure of first potential product
|
||||
products = soup.select("div.product-miniature")
|
||||
if not products:
|
||||
products = soup.select("div[class*='product-item']")
|
||||
|
||||
if products:
|
||||
first = products[0]
|
||||
print("\n--- First Product Structure ---")
|
||||
print(first.prettify()[:1000])
|
||||
|
||||
link = first.find("a")
|
||||
if link:
|
||||
print(f"\nLink found: {link.get('href')}")
|
||||
|
||||
img = first.find("img")
|
||||
if img:
|
||||
print(f"\nImage found: {img.get('src')}")
|
||||
else:
|
||||
print("\nNo obvious products found. Dumping generic structure...")
|
||||
# Find any div with many children
|
||||
divs = soup.find_all("div")
|
||||
for div in divs:
|
||||
if len(div.find_all("div", recursive=False)) > 10:
|
||||
print(f"Found container with many children: {div.get('class')}")
|
||||
# Print first child
|
||||
child = div.find("div")
|
||||
if child:
|
||||
print(child.prettify()[:500])
|
||||
break
|
||||
@@ -1,25 +0,0 @@
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
|
||||
html = f.read()
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
images = soup.select("img")
|
||||
|
||||
print(f"Found {len(images)} images")
|
||||
|
||||
for i, img in enumerate(images[:10]):
|
||||
print(f"\n--- Image {i+1} ---")
|
||||
print(f"Src: {img.get('src')}")
|
||||
print(f"Classes: {img.get('class')}")
|
||||
|
||||
parent = img.parent
|
||||
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
|
||||
|
||||
grandparent = parent.parent
|
||||
if grandparent:
|
||||
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
|
||||
|
||||
greatgrandparent = grandparent.parent
|
||||
if greatgrandparent:
|
||||
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
|
||||
@@ -1,33 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
|
||||
|
||||
from app.services.browserless_service import browserless_service
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
|
||||
|
||||
async def inspect():
|
||||
url = "https://www.lincroyable.fr/recherche-query=iphone/"
|
||||
print(f"Inspecting {url}...")
|
||||
|
||||
# Use host port for local debugging
|
||||
os.environ["BROWSERLESS_URL"] = "ws://localhost:3012"
|
||||
|
||||
await browserless_service.initialize()
|
||||
try:
|
||||
content, _ = await browserless_service.get_page_content(url, wait_selector="body")
|
||||
with open("lincroyable_dump.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
print("HTML dumped to lincroyable_dump.html")
|
||||
finally:
|
||||
await browserless_service.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(inspect())
|
||||
@@ -1,47 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
async def inspect_local():
|
||||
url = "https://www.lincroyable.fr/recherche-query=iphone/"
|
||||
print(f"Inspecting {url} using local Playwright...")
|
||||
|
||||
async with async_playwright() as p:
|
||||
# Launch local browser (headless=True matches server environment usually, but we can try False to see)
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
page = await context.new_page()
|
||||
|
||||
try:
|
||||
logging.info(f"Navigating to {url}")
|
||||
await page.goto(url, timeout=30000)
|
||||
logging.info("Navigation successful")
|
||||
|
||||
content = await page.content()
|
||||
with open("lincroyable_local_dump.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
print("HTML dumped to lincroyable_local_dump.html")
|
||||
|
||||
await page.screenshot(path="lincroyable_local.png")
|
||||
print("Screenshot saved to lincroyable_local.png")
|
||||
|
||||
except Exception as e:
|
||||
logging.error(f"Error: {e}")
|
||||
try:
|
||||
await page.screenshot(path="lincroyable_error.png")
|
||||
except:
|
||||
pass
|
||||
finally:
|
||||
await browser.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(inspect_local())
|
||||
@@ -1,47 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
from app.services.browserless_service import BrowserlessService
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def dump_site(site_key, query="chaise"):
|
||||
config = SITE_CONFIGS.get(site_key)
|
||||
if not config:
|
||||
logger.error(f"Site {site_key} not found in config")
|
||||
return
|
||||
|
||||
search_url = config["search_url"].format(query=query)
|
||||
logger.info(f"Dumping {site_key} from {search_url}")
|
||||
|
||||
try:
|
||||
content, _ = await BrowserlessService.get_page_content(
|
||||
search_url,
|
||||
wait_selector=config.get("wait_selector"),
|
||||
use_proxy=config.get("requires_proxy", False)
|
||||
)
|
||||
|
||||
if content:
|
||||
filename = f"/app/{site_key}_dump.html"
|
||||
with open(filename, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
logger.info(f"Successfully dumped to {filename}")
|
||||
else:
|
||||
logger.error(f"Failed to get content for {site_key}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error dumping {site_key}: {e}")
|
||||
|
||||
async def main():
|
||||
await BrowserlessService.initialize()
|
||||
try:
|
||||
await dump_site("amazon.fr")
|
||||
await dump_site("stokomani.fr")
|
||||
# await dump_site("lincroyable.fr") # Already have this
|
||||
finally:
|
||||
await BrowserlessService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
File diff suppressed because it is too large.
Load diff
Binary file not shown.
|
Before Width: | Height: | Size: 48 KiB |
File diff suppressed because it is too large.
Load diff
@@ -1,62 +0,0 @@
|
||||
"""
|
||||
Script to list all sites in database and check which ones are missing from search_config.py
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.database import SessionLocal
|
||||
from app.models import SearchSite
|
||||
from app.core.search_config import SITE_CONFIGS
|
||||
|
||||
def main():
|
||||
"""List all sites in DB and identify missing configurations"""
|
||||
db = SessionLocal()
|
||||
|
||||
try:
|
||||
# Get all sites from database
|
||||
sites = db.query(SearchSite).order_by(SearchSite.name).all()
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"Sites in Database: {len(sites)}")
|
||||
print(f"Sites in SITE_CONFIGS: {len(SITE_CONFIGS)}")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
# Check each DB site
|
||||
missing = []
|
||||
configured = []
|
||||
|
||||
for site in sites:
|
||||
domain_clean = site.domain.replace("www.", "").lower()
|
||||
|
||||
# Check if configured
|
||||
is_configured = False
|
||||
for key in SITE_CONFIGS.keys():
|
||||
if key in domain_clean or domain_clean in key:
|
||||
configured.append((site.id, site.name, site.domain, key))
|
||||
is_configured = True
|
||||
break
|
||||
|
||||
if not is_configured:
|
||||
missing.append((site.id, site.name, site.domain))
|
||||
|
||||
# Display results
|
||||
if configured:
|
||||
print("✅ CONFIGURED SITES:")
|
||||
for sid, name, domain, key in configured:
|
||||
print(f" [{sid:2d}] {name:20s} ({domain:25s}) → {key}")
|
||||
|
||||
if missing:
|
||||
print(f"\n❌ MISSING {len(missing)} SITES:")
|
||||
for sid, name, domain in missing:
|
||||
print(f" [{sid:2d}] {name:20s} ({domain})")
|
||||
else:
|
||||
print("\n✅ All sites are configured!")
|
||||
|
||||
print(f"\n{'='*80}\n")
|
||||
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,67 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
URL = "https://www.gifi.fr/meuble-et-deco/decoration/bougie-et-senteur/diffuseur-et-senteur/encens-nag-champa-15-g/000000000000540823.html"
|
||||
|
||||
async def reproduce_scrape():
|
||||
logger.info("Starting reproduction script...")
|
||||
async with async_playwright() as p:
|
||||
# Launch browser (headless=True by default which is what we want for reproduction usually)
|
||||
# But for debugging blocking, sometimes headless=False helps. Let's start with True (default)
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
logger.info(f"Navigating to {URL}...")
|
||||
try:
|
||||
await page.goto(URL, wait_until="domcontentloaded", timeout=60000)
|
||||
logger.info("Page loaded.")
|
||||
|
||||
# Wait a bit for dynamic content
|
||||
await page.wait_for_timeout(5000)
|
||||
|
||||
# Extract title
|
||||
title = await page.title()
|
||||
logger.info(f"Page Title: {title}")
|
||||
|
||||
# Extract body text
|
||||
content = await page.content()
|
||||
body_text = await page.inner_text("body")
|
||||
|
||||
logger.info(f"Content Length: {len(content)}")
|
||||
logger.info(f"Body Text Length: {len(body_text)}")
|
||||
|
||||
# Check for price
|
||||
if "€" in body_text:
|
||||
logger.info("Found '€' in body text.")
|
||||
else:
|
||||
logger.warning("'€' NOT found in body text.")
|
||||
|
||||
# specific check for likely price
|
||||
import re
|
||||
prices = re.findall(r'\d+[,\.]\d{2}\s*€', body_text)
|
||||
logger.info(f"Prices found in text: {prices}")
|
||||
|
||||
# Save content for review
|
||||
with open("gifi_reproduction.html", "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
logger.info("Saved gifi_reproduction.html")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error during navigation/scraping: {e}")
|
||||
|
||||
finally:
|
||||
await browser.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(reproduce_scrape())
|
||||
@@ -1,31 +0,0 @@
|
||||
from app.database import SessionLocal
|
||||
from app.services import auth_service
|
||||
from app import models
|
||||
|
||||
def reset_admin():
|
||||
db = SessionLocal()
|
||||
try:
|
||||
user = auth_service.get_user_by_username(db, "admin")
|
||||
if user:
|
||||
print("Found admin user. Resetting password...")
|
||||
auth_service.update_password(db, user, "admin")
|
||||
print("Password reset to 'admin'")
|
||||
|
||||
# Verify
|
||||
print("Verifying login...")
|
||||
auth_user = auth_service.authenticate_user(db, "admin", "admin")
|
||||
if auth_user:
|
||||
print("SUCCESS: Login verified!")
|
||||
else:
|
||||
print("ERROR: Login failed after reset!")
|
||||
else:
|
||||
print("Admin user not found. Creating...")
|
||||
auth_service.create_user(db, "admin", "admin", is_admin=True)
|
||||
print("Admin user created with password 'admin'")
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
reset_admin()
|
||||
@@ -1,53 +0,0 @@
|
||||
import sys
|
||||
import os
|
||||
import logging
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
# Add project root to path
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.database import SessionLocal, engine
|
||||
from app.models import Catalogue
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def cleanup_catalogs():
|
||||
db = SessionLocal()
|
||||
try:
|
||||
logger.info("Starting cleanup...")
|
||||
deleted_count = 0
|
||||
|
||||
# 1. Delete catalogs with 0 pages
|
||||
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
|
||||
for cat in bad_catalogs:
|
||||
logger.info(f"Deleting empty catalog: {cat.titre}")
|
||||
db.delete(cat)
|
||||
deleted_count += 1
|
||||
|
||||
# 2. Delete catalogs with iconic/bad images
|
||||
all_catalogs = db.query(Catalogue).all()
|
||||
for cat in all_catalogs:
|
||||
if not cat.image_couverture_url:
|
||||
continue
|
||||
|
||||
if any(
|
||||
x in cat.image_couverture_url.lower()
|
||||
for x in ["icon", "logo", "loader", "facebook", "twitter", "assets/img"]
|
||||
):
|
||||
logger.info(f"Deleting catalog with bad image: {cat.titre} ({cat.image_couverture_url})")
|
||||
db.delete(cat)
|
||||
deleted_count += 1
|
||||
|
||||
db.commit()
|
||||
logger.info(f"Cleanup complete. Deleted {deleted_count} catalogs.")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error during cleanup: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
cleanup_catalogs()
|
||||
-18100
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
-32156
File diff suppressed because it is too large.
Load diff
@@ -1,29 +1,21 @@
|
||||
# Investigation et Correction du Statut "Produit Retiré"
|
||||
# Nettoyage et Commit du Projet
|
||||
|
||||
## Contexte
|
||||
|
||||
Le système de suivi des prix marquait incorrectement certains produits comme "retirés" alors qu'ils étaient toujours disponibles, souvent à cause de blocages (bot detection) ou de changements mineurs de titre.
|
||||
Après avoir corrigé les problèmes de disponibilité d'Action.com et Amazon, il est nécessaire de nettoyer le répertoire racine des scripts de diagnostic et fichiers temporaires avant de committer les changements.
|
||||
|
||||
## Focus Actuel
|
||||
|
||||
Finalisation et vérification.
|
||||
Identification et suppression des fichiers inutiles, suivi du commit et push.
|
||||
|
||||
## Master Plan
|
||||
|
||||
- [x] Analyser la logique de comparaison de titres dans `app/services/scheduler_service.py`
|
||||
- [x] Vérifier si les scrapers extraient correctement les titres lors des mises à jour
|
||||
- [x] Identifier les cas limites (edge cases) où la similarité de titre échoue
|
||||
- [x] Corriger la logique pour éviter les faux positifs d'indisponibilité
|
||||
- [x] Vérifier la correction avec un exemple concret (test_logic.py)
|
||||
- [x] Implémenter le reset automatique de disponibilité si un prix est trouvé
|
||||
- [ ] Identifier les fichiers non essentiels à l'application
|
||||
- [ ] Supprimer les fichiers de diagnostic et scripts temporaires
|
||||
- [ ] Vérifier que l'application fonctionne toujours (build/syntaxe)
|
||||
- [ ] Committer les changements vers le dépôt Git
|
||||
- [ ] Pusher les changements
|
||||
|
||||
## Log de Progression
|
||||
|
||||
- [x] Investigation terminée : identification des faux positifs dus aux titres de blocage (Cloudflare, etc.) et aux placeholders ("Loading").
|
||||
- [x] Logique de normalisation renforcée.
|
||||
- [x] Détection des bots ajoutée pour tous les sites.
|
||||
- [x] Auto-reset de `is_available` implémenté dans `_update_db_result`.
|
||||
- [x] Correction déployée dans `scheduler_service.py`.
|
||||
- [x] Affinage de la détection de bot pour Stokomani/L'Incroyable (détection conditionnelle au titre).
|
||||
- [x] Support des versions "V2" et matching par mots pour les noms courts sur Amazon.
|
||||
- [x] Sécurisation du flux Action.com (check d'indisponibilité déplacé après le titre).
|
||||
- [ ] Planification du nettoyage commencée.
|
||||
@@ -1,187 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Script de test pour le scraper Amazon France
|
||||
Teste le système anti-détection et l'extraction des produits
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Ajouter le répertoire app au path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from app.services.amazon_scraper_service import amazon_scraper_service, AmazonScraperService
|
||||
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_DATA
|
||||
|
||||
# Define missing variable for test compatibility
|
||||
AMAZON_USER_AGENTS = USER_AGENT_DATA
|
||||
USER_AGENT_POOL = USER_AGENT_DATA
|
||||
|
||||
|
||||
# Configuration du logging
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s")
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Monkey patch _connect_browser to use local launch for testing
|
||||
async def _connect_browser_local(p):
|
||||
logger.info("Launching local browser (headless)...")
|
||||
return await p.chromium.launch(headless=True)
|
||||
|
||||
|
||||
AmazonScraperService._connect_browser = _connect_browser_local
|
||||
|
||||
|
||||
async def test_basic_search():
|
||||
"""Test basique de recherche"""
|
||||
logger.info("=" * 80)
|
||||
logger.info("TEST 1: Recherche basique - 'aspirateur'")
|
||||
logger.info("=" * 80)
|
||||
|
||||
products = await amazon_scraper_service.scrape_search("aspirateur", max_results=5)
|
||||
|
||||
if not products:
|
||||
logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau")
|
||||
return False
|
||||
|
||||
logger.info(f"✅ {len(products)} produits trouvés")
|
||||
|
||||
for idx, product in enumerate(products, 1):
|
||||
logger.info(f"\n{idx}. {product.title[:60]}...")
|
||||
logger.info(
|
||||
f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else "")
|
||||
)
|
||||
logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note")
|
||||
logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}")
|
||||
logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}")
|
||||
logger.info(f" {'📢 Sponsorisé' if product.sponsored else '🔍 Organique'}")
|
||||
|
||||
return True
|
||||
|
||||
|
||||
async def test_multiple_queries():
|
||||
"""Test avec plusieurs requêtes différentes"""
|
||||
logger.info("\n" + "=" * 80)
|
||||
logger.info("TEST 2: Requêtes multiples")
|
||||
logger.info("=" * 80)
|
||||
|
||||
queries = ["clavier", "souris", "casque"]
|
||||
results = {}
|
||||
|
||||
for query in queries:
|
||||
logger.info(f"\n🔍 Recherche: '{query}'")
|
||||
products = await amazon_scraper_service.scrape_search(query, max_results=3)
|
||||
results[query] = len(products)
|
||||
logger.info(f" ✅ {len(products)} produits trouvés")
|
||||
|
||||
# Délai entre requêtes pour respecter les bonnes pratiques
|
||||
await asyncio.sleep(3)
|
||||
|
||||
logger.info("\n📊 Résumé:")
|
||||
for query, count in results.items():
|
||||
logger.info(f" • {query}: {count} produits")
|
||||
|
||||
total = sum(results.values())
|
||||
if total > 0:
|
||||
logger.info(f"\n✅ Total: {total} produits extraits")
|
||||
return True
|
||||
else:
|
||||
logger.error("\n❌ Aucun produit extrait - problème possible")
|
||||
return False
|
||||
|
||||
|
||||
async def test_anti_detection():
|
||||
"""Test du système anti-détection"""
|
||||
logger.info("\n" + "=" * 80)
|
||||
logger.info("TEST 3: Vérification anti-détection")
|
||||
logger.info("=" * 80)
|
||||
|
||||
# Updated to just check if we can run
|
||||
logger.info("Skipping specific proxy/agent checks for this service as it handles them internally")
|
||||
|
||||
logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles")
|
||||
logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards")
|
||||
logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques")
|
||||
|
||||
# Test proxy
|
||||
# Test proxy
|
||||
import random
|
||||
|
||||
proxy = random.choice(AMAZON_PROXY_LIST_RAW) if AMAZON_PROXY_LIST_RAW else None
|
||||
if proxy:
|
||||
# Extract just the IP for logging (hide credentials)
|
||||
proxy_parts = proxy.split("@")
|
||||
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
|
||||
logger.info(f"✓ Proxy test: {proxy_server}")
|
||||
else:
|
||||
logger.warning("⚠️ Pas de proxy configuré")
|
||||
|
||||
# Test d'une recherche simple
|
||||
logger.info("\n🧪 Test de recherche avec anti-détection...")
|
||||
products = await amazon_scraper_service.scrape_search("livre", max_results=3)
|
||||
|
||||
if products:
|
||||
logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits")
|
||||
return True
|
||||
else:
|
||||
logger.error("❌ Échec - possibilité de blocage")
|
||||
return False
|
||||
|
||||
|
||||
async def run_all_tests():
|
||||
"""Lance tous les tests"""
|
||||
logger.info("\n" + "=" * 80)
|
||||
logger.info("🚀 DÉMARRAGE DES TESTS DU SCRAPER AMAZON FRANCE")
|
||||
logger.info("=" * 80)
|
||||
|
||||
tests = [
|
||||
("Recherche basique", test_basic_search),
|
||||
("Requêtes multiples", test_multiple_queries),
|
||||
("Anti-détection", test_anti_detection),
|
||||
]
|
||||
|
||||
results = {}
|
||||
|
||||
for test_name, test_func in tests:
|
||||
try:
|
||||
logger.info(f"\n▶️ Exécution: {test_name}")
|
||||
success = await test_func()
|
||||
results[test_name] = "✅ PASS" if success else "❌ FAIL"
|
||||
except Exception as e:
|
||||
logger.error(f"❌ Erreur dans {test_name}: {e}", exc_info=True)
|
||||
results[test_name] = "❌ ERROR"
|
||||
|
||||
# Résumé final
|
||||
logger.info("\n" + "=" * 80)
|
||||
logger.info("📊 RÉSUMÉ DES TESTS")
|
||||
logger.info("=" * 80)
|
||||
|
||||
for test_name, result in results.items():
|
||||
logger.info(f"{result} - {test_name}")
|
||||
|
||||
passed = sum(1 for r in results.values() if "PASS" in r)
|
||||
total = len(results)
|
||||
|
||||
logger.info(f"\n🎯 Score: {passed}/{total} tests réussis")
|
||||
|
||||
if passed == total:
|
||||
logger.info("✅ TOUS LES TESTS ONT RÉUSSI!")
|
||||
return True
|
||||
else:
|
||||
logger.warning("⚠️ Certains tests ont échoué")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
success = asyncio.run(run_all_tests())
|
||||
sys.exit(0 if success else 1)
|
||||
except KeyboardInterrupt:
|
||||
logger.info("\n⏸️ Tests interrompus par l'utilisateur")
|
||||
sys.exit(130)
|
||||
except Exception as e:
|
||||
logger.error(f"❌ Erreur fatale: {e}", exc_info=True)
|
||||
sys.exit(1)
|
||||
@@ -1,42 +0,0 @@
|
||||
"""
|
||||
Test Carrefour via API endpoint (production-like)
|
||||
"""
|
||||
import requests
|
||||
import json
|
||||
|
||||
url = "http://localhost:8555/api/search"
|
||||
params = {
|
||||
"q": "chaise",
|
||||
"sites": "7", # Carrefour site ID
|
||||
"max_results": 10
|
||||
}
|
||||
|
||||
print(f"Testing: {url}")
|
||||
print(f"Params: {params}\n")
|
||||
|
||||
response = requests.get(url, params=params, stream=True)
|
||||
|
||||
print(f"Status: {response.status_code}")
|
||||
print(f"Headers: {dict(response.headers)}\n")
|
||||
|
||||
count = 0
|
||||
for line in response.iter_lines():
|
||||
if line:
|
||||
try:
|
||||
# Each line should be JSON
|
||||
data = json.loads(line.decode('utf-8'))
|
||||
count += 1
|
||||
|
||||
title = data.get('title', 'N/A')[:60]
|
||||
price = data.get('price', 'N/A')
|
||||
image = "✅" if data.get('image_url') else "❌"
|
||||
|
||||
print(f"{count}. [{image}] {title} - {price}€")
|
||||
|
||||
if count >= 10:
|
||||
break
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"JSON Error: {e}")
|
||||
print(f"Line: {line[:100]}")
|
||||
|
||||
print(f"\nTotal: {count} results")
|
||||
@@ -1,40 +0,0 @@
|
||||
"""
|
||||
Test if current Carrefour price extraction works
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Carrefour for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
price_status = f"{result.price}€" if result.price else "N/A"
|
||||
print(f"{count}. {result.title[:55]} - {price_status}")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Count prices
|
||||
with_prices = sum(1 for r in results if r.price)
|
||||
print(f"Products with prices: {with_prices}/{len(results)}")
|
||||
|
||||
if with_prices == len(results):
|
||||
print("✅ SUCCESS: All products have prices!")
|
||||
else:
|
||||
print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,26 +0,0 @@
|
||||
"""
|
||||
Test script to trigger Carrefour search and generate HTML dump
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Carrefour for 'chaise'...")
|
||||
results = []
|
||||
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
|
||||
results.append(result)
|
||||
print(f"Found: {result.title}")
|
||||
|
||||
print(f"\nTotal results: {len(results)}")
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,69 +0,0 @@
|
||||
"""
|
||||
Test detailed logging for Centrakor image extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from bs4 import BeautifulSoup
|
||||
import re
|
||||
|
||||
async def main():
|
||||
playwright = await async_playwright().start()
|
||||
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
)
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
|
||||
content = await page.content()
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
products = soup.select("div.product-item")
|
||||
|
||||
print(f"Testing first product:\n")
|
||||
|
||||
first = products[0]
|
||||
|
||||
# Get all images
|
||||
img_els = first.select("img.responsive-image__actual")
|
||||
print(f"Found {len(img_els)} images with selector")
|
||||
|
||||
for i, img_el in enumerate(img_els):
|
||||
print(f"\n=== Image {i+1} ===")
|
||||
candidate_url = img_el.get("src")
|
||||
print(f"URL: {candidate_url}")
|
||||
|
||||
# Test filters
|
||||
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
|
||||
print(" ❌ Filtered: Contains picto/icon/logo/badge keyword")
|
||||
continue
|
||||
|
||||
width_match = re.search(r'width=(\d+)', candidate_url)
|
||||
height_match = re.search(r'height=(\d+)', candidate_url)
|
||||
print(f" Width match: {width_match.group(1) if width_match else None}")
|
||||
print(f" Height match: {height_match.group(1) if height_match else None}")
|
||||
|
||||
if width_match and height_match:
|
||||
width = int(width_match.group(1))
|
||||
height = int(height_match.group(1))
|
||||
print(f" Dimensions: {width}x{height}")
|
||||
if width < 100 and height < 100:
|
||||
print(f" ❌ Filtered: Too small ({width}x{height})")
|
||||
continue
|
||||
else:
|
||||
print(f" ✅ PASS: Large enough ({width}x{height})")
|
||||
else:
|
||||
print(" ✅ PASS: No dimensions in URL")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
await playwright.stop()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,42 +0,0 @@
|
||||
"""
|
||||
Test Centrakor image extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Centrakor for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
has_image = "✅" if result.image_url else "❌"
|
||||
print(f"{count}. {has_image} {result.title[:55]} - {result.price}€")
|
||||
if result.image_url:
|
||||
print(f" Image: {result.image_url[:70]}...")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Count images
|
||||
with_images = sum(1 for r in results if r.image_url)
|
||||
print(f"Products with images: {with_images}/{len(results)}")
|
||||
|
||||
if with_images == len(results):
|
||||
print("✅ SUCCESS: All products have images!")
|
||||
else:
|
||||
print(f"⚠️ WARNING: {len(results) - with_images} products missing images")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,34 +0,0 @@
|
||||
"""
|
||||
Final test of Gifi with new selectors
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
print(f"✓ {result.title[:50]} - {result.price}€")
|
||||
|
||||
print(f"\n==> Total: {len(results)} results")
|
||||
|
||||
if results:
|
||||
print("\nFirst 3 products:")
|
||||
for i, r in enumerate(results[:3]):
|
||||
print(f"{i+1}. Title: {r.title}")
|
||||
print(f" Price: {r.price}€")
|
||||
print(f" URL: {r.url[:80]}...")
|
||||
print()
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,38 +0,0 @@
|
||||
"""
|
||||
Test Gifi price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
print(f"{count}. {result.title[:60]} - Price: {result.price}€")
|
||||
if count >= 5: # Only test first 5
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Check if all prices are the same
|
||||
prices = [r.price for r in results if r.price]
|
||||
if prices:
|
||||
unique_prices = set(prices)
|
||||
print(f"Unique prices: {unique_prices}")
|
||||
if len(unique_prices) == 1:
|
||||
print("⚠️ WARNING: All prices are the same!")
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,41 +0,0 @@
|
||||
"""
|
||||
Test fixed Gifi price extraction
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...\n")
|
||||
results = []
|
||||
count = 0
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
count += 1
|
||||
print(f"{count}. {result.title[:60]} - {result.price}€")
|
||||
if count >= 10:
|
||||
break
|
||||
|
||||
print(f"\n==> Got {len(results)} results")
|
||||
|
||||
# Check price diversity
|
||||
prices = [r.price for r in results if r.price]
|
||||
if prices:
|
||||
unique_prices = set(prices)
|
||||
print(f"Unique prices: {sorted(unique_prices)}")
|
||||
print(f"Price range: {min(prices)}€ - {max(prices)}€")
|
||||
if len(unique_prices) > 1:
|
||||
print("✅ SUCCESS: Multiple different prices found!")
|
||||
else:
|
||||
print("⚠️ WARNING: All prices are the same")
|
||||
|
||||
print("\nShutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,26 +0,0 @@
|
||||
"""
|
||||
Test script to analyze Gifi HTML structure
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
sys.path.insert(0, '/app')
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
|
||||
async def main():
|
||||
print("Initializing browser...")
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
print("Searching Gifi for 'chaise'...")
|
||||
results = []
|
||||
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
|
||||
results.append(result)
|
||||
print(f"Found: {result.title} - Price: {result.price}")
|
||||
|
||||
print(f"\nTotal results: {len(results)}")
|
||||
|
||||
print("Shutting down...")
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,29 +0,0 @@
|
||||
from bs4 import BeautifulSoup
|
||||
import sys
|
||||
|
||||
try:
|
||||
with open("/app/debug_dumps/stokomani.fr_failed_verification.html", "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
|
||||
print(f"Read {len(content)} bytes")
|
||||
|
||||
soup = BeautifulSoup(content, "html.parser")
|
||||
print("Soup created")
|
||||
|
||||
selector = "div.product-card"
|
||||
items = soup.select(selector)
|
||||
print(f"Found {len(items)} items with selector '{selector}'")
|
||||
|
||||
if items:
|
||||
item = items[0]
|
||||
print("First item classes:", item.get("class"))
|
||||
|
||||
title_selector = "h3.product-card__title a"
|
||||
title_el = item.select_one(title_selector)
|
||||
if title_el:
|
||||
print("Title found:", title_el.get_text(strip=True))
|
||||
else:
|
||||
print(f"Title NOT found with '{title_selector}'")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
@@ -1,43 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Identify workspace root
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
from app.services.cataloguemate_scraper import scrape_catalog_pages
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
async def verify_scraper():
|
||||
# Gifi catalog URL from the browser session
|
||||
url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/"
|
||||
|
||||
print(f"Verifying scraper on: {url}")
|
||||
|
||||
pages = await scrape_catalog_pages(url)
|
||||
|
||||
print(f"Found {len(pages)} pages.")
|
||||
|
||||
if not pages:
|
||||
print("FAIL: No pages found.")
|
||||
return
|
||||
|
||||
# check first page image
|
||||
first_img = pages[0]["image_url"]
|
||||
print(f"Page 1 Image: {first_img}")
|
||||
|
||||
if "thumbor" in first_img or "leafletscdns" in first_img:
|
||||
print("SUCCESS: Image is a Thumbor/Leaflet URL.")
|
||||
elif "icon" in first_img or "logo" in first_img:
|
||||
print("FAIL: Image appears to be an icon/logo.")
|
||||
else:
|
||||
print(f"WARNING: Image URL is: {first_img}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_scraper())
|
||||
@@ -1,45 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Identify workspace root
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
# Mock Env Vars
|
||||
os.environ["DATABASE_URL"] = "postgresql://user:password@localhost:5432/pricewatch"
|
||||
os.environ["BROWSERLESS_URL"] = "ws://localhost:3012" # Ignored by patch, but good for completeness
|
||||
|
||||
from app.services.improved_search_service import ImprovedSearchService
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
# Setup logging
|
||||
logging.basicConfig(level=logging.DEBUG) # DEBUG level to see price extraction logic
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Monkey patch _connect_browser to use local launch
|
||||
async def _connect_browser_local(p):
|
||||
logger.info("Launching local browser (headless)...")
|
||||
return await p.chromium.launch(headless=True)
|
||||
|
||||
|
||||
ImprovedSearchService._connect_browser = _connect_browser_local
|
||||
|
||||
|
||||
async def test_search():
|
||||
await ImprovedSearchService.initialize()
|
||||
|
||||
query = "nintendo switch"
|
||||
target_site = "gifi.fr"
|
||||
|
||||
print(f"Searching for: {query} on {target_site}")
|
||||
|
||||
async for result in ImprovedSearchService.search_site_generator(target_site, query):
|
||||
print(f"[{result.source}] {result.title}\n -> Price: {result.price}€\n -> URL: {result.url}")
|
||||
|
||||
await ImprovedSearchService.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_search())
|
||||
@@ -1,108 +0,0 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
import sys
|
||||
|
||||
# Mock logger
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Fallback implementation of _fetch_with_fallback for standalone test
|
||||
async def _fetch_with_fallback(url):
|
||||
# We need to install httpx for this to work
|
||||
try:
|
||||
import httpx
|
||||
|
||||
headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
}
|
||||
async with httpx.AsyncClient(verify=False, timeout=30.0) as client:
|
||||
response = await client.get(url, headers=headers)
|
||||
return response.text
|
||||
except ImportError:
|
||||
print("Please pip install httpx strict")
|
||||
return ""
|
||||
|
||||
|
||||
async def scrape_catalog_pages_standalone(catalog_url: str):
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
print(f"Scraping: {catalog_url}")
|
||||
html_content = await _fetch_with_fallback(catalog_url)
|
||||
|
||||
if not html_content:
|
||||
print("Failed to fetch content")
|
||||
return []
|
||||
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
|
||||
# --- COPIED LOGIC FROM cataloguemate_scraper.py ---
|
||||
main_image_url = None
|
||||
max_area = 0
|
||||
|
||||
# Strategy 1: Look for specific container/class identified in browser inspection
|
||||
candidates = soup.select(".letaky-grid-preview img")
|
||||
|
||||
# Strategy 2: Fallback to all images if specific container not found
|
||||
if not candidates:
|
||||
candidates = soup.find_all("img")
|
||||
|
||||
print(f"Found {len(candidates)} candidates")
|
||||
|
||||
for img in candidates:
|
||||
# Check multiple attributes for the real image URL
|
||||
src = img.get("src") or img.get("data-src") or img.get("data-original")
|
||||
|
||||
if not src:
|
||||
continue
|
||||
|
||||
# Skip common UI elements - refined list
|
||||
if any(
|
||||
x in src.lower()
|
||||
for x in [
|
||||
"logo",
|
||||
"icon",
|
||||
"facebook",
|
||||
"twitter",
|
||||
"instagram",
|
||||
"loader",
|
||||
"spinner",
|
||||
"market",
|
||||
"googleplay",
|
||||
"appstore",
|
||||
]
|
||||
):
|
||||
continue
|
||||
|
||||
# Strong Signal: URL contains 'thumbor' or 'leafletscdns' (host for catalog images)
|
||||
is_thumbor = "thumbor" in src.lower() or "leafletscdns" in src.lower()
|
||||
|
||||
# Calculate area if dimensions exist
|
||||
width = img.get("width")
|
||||
height = img.get("height")
|
||||
area = 0
|
||||
if width and height:
|
||||
try:
|
||||
area = int(width) * int(height)
|
||||
except:
|
||||
pass
|
||||
|
||||
if is_thumbor:
|
||||
if area > max_area or (area == 0 and max_area == 0):
|
||||
max_area = area
|
||||
main_image_url = src
|
||||
print(f"Match (Thumbor): {src}")
|
||||
elif area > 50000:
|
||||
if area > max_area:
|
||||
max_area = area
|
||||
main_image_url = src
|
||||
print(f"Match (Size): {src}")
|
||||
|
||||
return main_image_url
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/"
|
||||
result = asyncio.run(scrape_catalog_pages_standalone(url))
|
||||
print(f"FINAL RESULT: {result}")
|
||||
Reference in new issue
Block a user