Fix: Action.com availability false positives and Amazon login detection optimization. Cleaned up diagnostic scripts.

This commit is contained in:
Michael committed 2026-01-13 13:37:54 +01:00
1 parent 7804e817c3
commit 8329eb28a3
58 files changed
+65 -112707

No files matched your search

-19490
View File
File diff suppressed because it is too large. Load diff
-82
View File
@@ -1,82 +0,0 @@
"""
Analyze B&M product page for price extraction
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# URL from the screenshot
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
print(f"Loading: {url}\n")
await page.goto(url, wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Find all elements with € symbol
price_els = soup.find_all(string=re.compile('€'))
print(f"Found {len(price_els)} elements with '€'\n")
prices_found = {}
for el in price_els[:30]:
text = el.strip()
if text and len(text) < 100:
parent = el.find_parent()
if parent:
parent_class = parent.get('class', [])
parent_class_str = ' '.join(parent_class) if isinstance(parent_class, list) else str(parent_class)
# Extract price value
price_match = re.search(r'(\d+[.,]\d+)\s*€', text)
if price_match:
price_val = price_match.group(1)
key = f"{price_val}€ in .{parent_class_str[:50]}"
if key not in prices_found:
prices_found[key] = text
print("Prices found:")
for key, text in prices_found.items():
print(f" {key}: '{text}'")
# Try common price selectors
print("\nTrying specific selectors:")
selectors = [
".price",
"[class*='price']",
"[data-price]",
".product-price",
"span[class*='price']"
]
for selector in selectors:
try:
els = soup.select(selector)
if els:
for el in els[:2]:
text = el.get_text(strip=True)
if '€' in text:
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-50
View File
@@ -1,50 +0,0 @@
"""
Analyze B&M with longer wait and playwright evaluation
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966"
print(f"Loading: {url}\n")
await page.goto(url, wait_until="networkidle")
# Wait extra time for JS
await page.wait_for_timeout(5000)
# Look for elements containing price
print("Searching for price elements...\n")
# Try to find any text with €
price_els = await page.query_selector_all("*:has-text('€')")
print(f"Found {len(price_els)} elements with €\n")
for i, el in enumerate(price_els[:10]):
text = await el.inner_text()
tag = await el.evaluate("el => el.tagName")
classes = await el.evaluate("el => el.className")
print(f"{i+1}. <{tag} class='{classes}'> {text[:100]}")
# Screenshot for debugging
await page.screenshot(path="/app/debug_dumps/bm_screenshot.png", full_page=True)
print("\nScreenshot saved to /app/debug_dumps/bm_screenshot.png")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-59
View File
@@ -1,59 +0,0 @@
"""
Analyze Carrefour price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Carrefour search...")
await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("article.product-list-card-plp-grid-new")
print(f"Found {len(products)} products\n")
for i, product in enumerate(products[:3]):
print(f"=== Product {i+1} ===")
# Title
title_el = product.select_one("h3, h2, a")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title[:60]}")
# Find all text with € symbol
import re
product_html = str(product)
prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html)
print(f"Prices found in HTML: {prices}")
# Look for price elements
price_els = product.find_all(string=re.compile('€'))
if price_els:
print(f"Elements with €:")
for el in price_els[:3]:
print(f" - {el.strip()[:50]}")
print()
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-70
View File
@@ -1,70 +0,0 @@
"""
Analyze a Carrefour product page for price selectors
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
import re
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# Visit a Carrefour product page (from the screenshot)
url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010"
print(f"Loading: {url}")
await page.goto(url, wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Find all elements with € symbol
price_els = soup.find_all(string=re.compile('€'))
print(f"\nFound {len(price_els)} elements with '€'")
prices_found = set()
for el in price_els[:20]:
text = el.strip()
if text and len(text) < 50:
prices_found.add(text)
parent = el.find_parent()
print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>")
# Try common price selectors
selectors = [
".product-price",
"[class*='price']",
".price",
"span.price",
"div.price",
"[data-price]"
]
print("\nTrying specific selectors:")
for selector in selectors:
try:
els = soup.select(selector)
if els:
for el in els[:2]:
text = el.get_text(strip=True)
if '€' in text:
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-69
View File
@@ -1,69 +0,0 @@
"""
Analyze Centrakor HTML structure for image selectors
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Centrakor search...")
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
# Try to find product containers
selectors = [
"div.product-item",
"div.product-card",
"article",
"div[class*='product']",
"li[class*='product']"
]
for selector in selectors:
products = soup.select(selector)
if products:
print(f"\n✓ Found {len(products)} products with selector: {selector}")
# Analyze first product
first = products[0]
print(f"\nFirst product HTML snippet:")
print(str(first)[:500])
print("\n...")
# Find all images
images = first.find_all('img')
print(f"\nFound {len(images)} images in first product:")
for i, img in enumerate(images):
print(f"\n Image {i+1}:")
print(f" Class: {img.get('class')}")
print(f" Src: {img.get('src', '')[:80]}")
print(f" Data-src: {img.get('data-src', '')[:80]}")
print(f" Alt: {img.get('alt', '')[:50]}")
break
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
-58
View File
@@ -1,58 +0,0 @@
"""
Detailed analysis of Centrakor image structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Analyzing {min(5, len(products))} products:\n")
for i, product in enumerate(products[:5]):
print(f"=== Product {i+1} ===")
# Title
title_el = product.select_one("a.product-item__name")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title}")
# All images
images = product.find_all('img')
print(f"Found {len(images)} img tags")
for j, img in enumerate(images):
print(f"\n Image {j+1}:")
print(f" tag: {img.name}")
print(f" class: {img.get('class')}")
for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']:
val = img.get(attr)
if val:
print(f" {attr}: {val[:80]}")
print()
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-24
View File
@@ -1,24 +0,0 @@
"""
Analyze Gifi HTML to find correct selectors
"""
with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f:
html = f.read()
# Find product-related divs
import re
matches = re.findall(r'<div[^>]*class="[^"]*"[^>]*>.*?</div>', html[:50000], re.DOTALL)
print(f"Total HTML size: {len(html)} bytes")
# Search for price patterns
price_patterns = re.findall(r'50[.,]00\s*€', html[:50000])
print(f"\nFound {len(price_patterns)} instances of '50,00 €'")
# Find all class names containing specific keywords
for keyword in ['product', 'article', 'item', 'card']:
classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE)
unique_classes = set(classes)
if unique_classes:
print(f"\nClasses containing '{keyword}':")
for cls in sorted(unique_classes):
print(f" - {cls}")
-69
View File
@@ -1,69 +0,0 @@
"""
Analyze a Gifi product page to understand price structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
# Visit a Gifi product page
url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html"
print(f"Loading: {url}")
await page.goto(url, wait_until="networkidle")
# Get page title
title = await page.title()
print(f"Title: {title}")
# Find all elements with price-like text
price_els = await page.query_selector_all("*:has-text('€')")
print(f"\nFound {len(price_els)} elements with '€'")
# Get first 10 price elements
for i, el in enumerate(price_els[:10]):
text = await el.inner_text()
tag = await el.evaluate("el => el.tagName")
classes = await el.evaluate("el => el.className")
print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}")
# Try specific selectors
selectors = [
".price",
".product-price",
"[class*='price']",
"[data-price]",
"span.price",
"div.price"
]
print("\nTrying specific selectors:")
for selector in selectors:
try:
els = await page.query_selector_all(selector)
if els:
for el in els[:2]:
text = await el.inner_text()
print(f" {selector}: {text}")
except:
pass
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
-71
View File
@@ -1,71 +0,0 @@
"""
Test extracting price from Gifi search page HTML
"""
import asyncio
import sys
import re
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-tile")
print(f"Found {len(products)} products\n")
for i, product in enumerate(products[:3]):
print(f"\\n=== Product {i+1} ===")
# Get title
title_el = product.select_one("div.pdp-link a")
title = title_el.get_text(strip=True) if title_el else "N/A"
print(f"Title: {title}")
# Try to find price in product HTML
product_html = product.prettify()
# Look for price patterns
price_patterns = [
r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 €
r'€\s*(\d+)[,.](\d+)', # € 19,99
r'(\d+)€(\d+)', # 19€99
r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price
]
for pattern in price_patterns:
matches = re.findall(pattern, product_html)
if matches:
print(f"Pattern '{pattern}': {matches[:3]}")
# Find all text with €
euro_texts = product.find_all(string=re.compile('€'))
if euro_texts:
print(f"Texts with €:")
for text in euro_texts[:5]:
print(f" - {text.strip()[:80]}")
await context.close()
await browser.close()
await playwright.stop()
print("\nDone")
if __name__ == "__main__":
asyncio.run(main())
-93
View File
@@ -1,93 +0,0 @@
"""
Script pour analyser les sélecteurs CSS des sites manquants
"""
import asyncio
from playwright.async_api import async_playwright
async def analyze_site(url: str, site_name: str):
print(f"\n=== Analyse de {site_name} ===")
print(f"URL: {url}")
async with async_playwright() as p:
browser = await p.chromium.launch(headless=False)
page = await browser.new_page()
try:
await page.goto(url, wait_until="networkidle", timeout=30000)
await page.wait_for_timeout(3000)
# Try cookie banners
cookie_selectors = [
"button:has-text('Accepter')",
"button:has-text('Tout accepter')",
"#didomi-notice-agree-button",
".didomi-continue-without-agreeing"
]
for selector in cookie_selectors:
try:
await page.click(selector, timeout=2000)
print(f"✓ Cookie banner fermé: {selector}")
break
except:
pass
await page.wait_for_timeout(2000)
# Save HTML
html = await page.content()
filename = f"{site_name.lower().replace(' ', '_')}_search.html"
with open(filename, "w", encoding="utf-8") as f:
f.write(html)
print(f"✓ HTML sauvegardé: {filename}")
# Take screenshot
screenshot_path = f"{site_name.lower().replace(' ', '_')}_search.png"
await page.screenshot(path=screenshot_path, full_page=True)
print(f"✓ Screenshot: {screenshot_path}")
# Test common product link selectors
test_selectors = [
"a.product-link",
"a.product-name",
"a[href*='/produit']",
"a[href*='/product']",
"a[href*='/p/']",
".product-title a",
".product-item a",
"article a",
"a.product",
"a[itemprop='url']",
]
print("\n--- Test de sélecteurs ---")
for selector in test_selectors:
try:
elements = await page.query_selector_all(selector)
if elements:
print(f"✓ {selector}: {len(elements)} éléments trouvés")
# Get first few hrefs
for i, elem in enumerate(elements[:3]):
href = await elem.get_attribute("href")
text = await elem.inner_text()
print(f" [{i+1}] {text[:50]} -> {href}")
except Exception as e:
pass
except Exception as e:
print(f"❌ Erreur: {e}")
finally:
await browser.close()
async def main():
sites = [
("https://bmstores.fr/module/ambjolisearch/jolisearch?s=chaise", "BM"),
("https://www.centrakor.com/recherche?controller=search&s=chaise", "Centrakor"),
("https://www.lincroyable.fr/recherche?query=chaise", "L'Incroyable"),
]
for url, name in sites:
await analyze_site(url, name)
await asyncio.sleep(2)
if __name__ == "__main__":
asyncio.run(main())
+12 -1
View File
@@ -50,6 +50,10 @@ def _normalize_title(title: str) -> str:
for pattern in patterns:
title = re.sub(pattern, "", title)
# Separate numbers from letters (handles cases like 'Blanc40' or '100pièces')
title = re.sub(r"([a-zA-Z])(\d)", r"\1 \2", title)
title = re.sub(r"(\d)([a-zA-Z])", r"\1 \2", title)
# Remove extra spaces and punctuation that might differ between sites
title = re.sub(r"[^\w\s]", " ", title)
title = " ".join(title.split())
@@ -323,7 +327,14 @@ async def process_item_check(item_id: int):
html_lower = html_content.lower() if html_content else ""
if any(term in final_url_lower for term in unavailable_terms) or any(
term in html_lower for term in unavailable_terms
term in html_lower
for term in [
# Use more specific terms for HTML to avoid false positives in random text or meta
"malheureusement, ce produit est actuellement indisponible",
">produit indisponible<",
">ce produit est indisponible<",
"product-unavailable-message",
]
):
logger.warning(f"Product unavailable confirmed for item {item_id} on Action.com")
await loop.run_in_executor(
+44 -29
View File
@@ -31,7 +31,6 @@ POPUP_SELECTORS = [
"input[value='Continuer les achats']",
"input[value='Continue shopping']",
"input[value='Continue shopping']",
"form:has-text('Continuer les achats') input[type='submit']",
"[aria-labelledby='continue-shopping-label']",
"#sp-cc-accept",
@@ -53,7 +52,6 @@ POPUP_SELECTORS = [
]
# Random User-Agents to alternate fingerprint
AMAZON_USER_AGENTS = [
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
@@ -63,6 +61,7 @@ AMAZON_USER_AGENTS = [
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
]
@dataclass
class ScrapeConfig:
"""Configuration for scraping parameters."""
@@ -174,6 +173,7 @@ class ScraperService:
try:
# Random delay to simulate human lead-in
import random
await asyncio.sleep(random.uniform(0.5, 2.0))
await ScraperService._navigate_and_wait(page, url, timeout)
@@ -207,7 +207,9 @@ class ScraperService:
await page.wait_for_timeout(1500)
# Select the first Nancy store (Nancy Essey or Nancy Centre)
select_btn = page.locator(".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')")
select_btn = page.locator(
".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')"
)
if await select_btn.count() > 0:
logger.info("Selecting Nancy store...")
await select_btn.first.click()
@@ -244,10 +246,7 @@ class ScraperService:
@staticmethod
async def _scrape_amazon_specific(
url: str,
item_id: int | None,
config: ScrapeConfig,
return_html: bool
url: str, item_id: int | None, config: ScrapeConfig, return_html: bool
) -> tuple[str | None, str, str, str]:
"""
Specialized scraping flow for Amazon to avoid bot detection and ensure good screenshots.
@@ -261,6 +260,7 @@ class ScraperService:
# Determine base domain
from urllib.parse import urlparse
parsed = urlparse(url)
base_domain = f"{parsed.scheme}://{parsed.netloc}"
@@ -297,14 +297,20 @@ class ScraperService:
# Dismiss "Change Address" or specific Amazon location modals if any
try:
await page.evaluate("document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')")
except: pass
await page.evaluate(
"document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')"
)
except:
pass
# 5. Check for Bot Detection / CAPTCHA / Login (Content-based)
# We do this AFTER popup removal because sometimes "Identifiez-vous" is in a dismissible modal
content_check = await page.content()
if "Type the characters you see in this image" in content_check or "Saisissez les caractères que vous voyez" in content_check:
if (
"Type the characters you see in this image" in content_check
or "Saisissez les caractères que vous voyez" in content_check
):
logger.error("🚫 Amazon CAPTCHA detected!")
# Attempt refresh once
logger.info("Retrying with refresh...")
@@ -324,14 +330,8 @@ class ScraperService:
await close_btn.first.click()
await page.wait_for_timeout(1000)
content_check = await page.content() # Refresh content
except: pass
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
# Retry once for Login wall too
logger.info("⚠️ Amazon Login/Auth detected. Retrying with refresh...")
await page.reload()
await asyncio.sleep(3)
content_check = await page.content()
except:
pass
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
# Final Check: Do we have a product title?
@@ -339,15 +339,28 @@ class ScraperService:
# If NO title, it's a hard redirect/gate. Fail.
try:
title_check = page.locator("#productTitle, #title")
if await title_check.count() > 0:
if await title_check.count() > 0 and await title_check.first.is_visible():
logger.info("⚠️ Login prompt detected but Product Title found. Ignoring/Hiding modal...")
# Attempt to brute-force remove the modal overlay again just in case
await page.evaluate("document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())")
await page.evaluate(
"() => document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())"
)
else:
# No title found, now we can try to reload or fail
logger.info("⚠️ Amazon Login/Auth detected and No Title found. Retrying with refresh...")
await page.reload()
await asyncio.sleep(3)
content_check = await page.content()
if "Identifiez-vous" in content_check or "ap_signin" in content_check:
# Re-check title after reload
if await title_check.count() > 0 and await title_check.first.is_visible():
logger.info("⚠️ Title appeared after refresh despite login prompt.")
else:
logger.error("🚫 Amazon Login Prompt detected (Blocking)!")
return None, "LOGIN_REQUIRED", final_url, page_title
except:
logger.error("🚫 Amazon Login Prompt detected (Error Checking Title)!")
except Exception as e:
logger.error(f"🚫 Amazon Login Prompt detected (Error Checking Title: {e})")
return None, "LOGIN_REQUIRED", final_url, page_title
# 6. Wait for Main Image (Critical for screenshot)
@@ -355,8 +368,7 @@ class ScraperService:
try:
# Main image container on desktop
await page.wait_for_selector(
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper",
timeout=10000
"#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper", timeout=10000
)
except Exception as e:
logger.warning(f"Could not find main image container: {e}")
@@ -396,6 +408,7 @@ class ScraperService:
"""Create context with advanced stealth and headers (specifically for Amazon)"""
# Determine base domain for referer
from urllib.parse import urlparse
parsed = urlparse(url)
base_domain = f"{parsed.scheme}://{parsed.netloc}/"
@@ -422,8 +435,8 @@ class ScraperService:
"Sec-Fetch-Site": "none",
"Sec-Fetch-User": "?1",
"Upgrade-Insecure-Requests": "1",
"Referer": base_domain if "amazon" in url else "https://www.google.com/"
}
"Referer": base_domain if "amazon" in url else "https://www.google.com/",
},
)
# Advanced Stealth mode
@@ -478,7 +491,7 @@ class ScraperService:
# 1. Multi-pass clicking (some popups appear after others are closed)
for i in range(2):
logger.debug(f"Popup removal pass {i+1}")
logger.debug(f"Popup removal pass {i + 1}")
for selector in POPUP_SELECTORS:
try:
locators = page.locator(selector)
@@ -497,7 +510,8 @@ class ScraperService:
try:
await page.keyboard.press("Escape")
await page.wait_for_timeout(200)
except: pass
except:
pass
# 2. Javascript cleanup (Hide pesky overlays and CMPs that won't close)
logger.info("Injecting CSS/JS cleanup for persistent overlays...")
@@ -612,7 +626,8 @@ class ScraperService:
# Collapse whitespace
import re
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
clean_text = re.sub(r"\s+", " ", clean_text).strip()
page_text = clean_text[:text_length]
logger.info(f"Extracted {len(page_text)} chars")
-2
View File
@@ -1,2 +0,0 @@
from app.services.improved_search_service import BeautifulSoup
print("BeautifulSoup imported successfully")
-12
View File
@@ -1,12 +0,0 @@
import sys
import os
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
import app.core.search_config
print(f"Search Config File: {app.core.search_config.__file__}")
from app.core.search_config import SITE_CONFIGS
print(f"Keys: {list(SITE_CONFIGS.keys())}")
with open(app.core.search_config.__file__, 'r') as f:
content = f.read()
print(f"File content length: {len(content)}")
print(f"Contains stokomani.fr: {'stokomani.fr' in content}")
-10
View File
@@ -1,10 +0,0 @@
import sys
from app.services.improved_search_service import ImprovedSearchService
print("Attributes of ImprovedSearchService:")
attrs = dir(ImprovedSearchService)
if 'search_site_generator' in attrs:
print("SUCCESS: search_site_generator found")
else:
print("FAILURE: search_site_generator NOT found")
print("Available attributes:", [a for a in attrs if not a.startswith('__')])
-52
View File
@@ -1,52 +0,0 @@
import asyncio
import os
import sys
# Add app to path
sys.path.insert(0, os.getcwd())
from playwright.async_api import async_playwright
from app.core.search_config import get_amazon_proxies
async def check_proxy(proxy, semaphore):
async with semaphore:
proxy_url = proxy["server"]
print(f"Testing {proxy_url}...")
async with async_playwright() as p:
# Connect to browserless or launch local
# Using launch local for simpler testing without ws dependency if possible
# But the app uses browserless. Let's try launch first.
try:
browser = await p.chromium.launch(headless=True, proxy=proxy)
page = await browser.new_page()
try:
# amazon.fr might block, use httpbin for connectivity check
await page.goto("http://httpbin.org/ip", timeout=15000)
content = await page.content()
print(f"✅ {proxy_url}: Success")
await browser.close()
return True
except Exception as e:
print(f"❌ {proxy_url}: Failed - {str(e)[:100]}")
await browser.close()
return False
except Exception as e:
print(f"❌ {proxy_url}: Launch Failed - {str(e)[:100]}")
return False
async def main():
proxies = get_amazon_proxies()
print(f"Checking {len(proxies)} proxies...")
semaphore = asyncio.Semaphore(3) # Limit concurrency
results = await asyncio.gather(*[check_proxy(p, semaphore) for p in proxies])
working = sum(results)
print(f"\nSummary: {working}/{len(proxies)} working.")
if __name__ == "__main__":
asyncio.run(main())
-7
View File
@@ -1,7 +0,0 @@
import sys
try:
from app.services import improved_search_service
print("Import successful")
except Exception as e:
print(f"Import failed: {e}")
sys.exit(1)
-56
View File
@@ -1,56 +0,0 @@
"""
Deep analysis - compare products WITH images vs WITHOUT images
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Analyzing {len(products)} products for image patterns\n")
for i, product in enumerate(products[:10]):
# Get title
title_el = product.select_one("a.product-item__name")
title = title_el.get_text(strip=True) if title_el else f"Product {i+1}"
# Get ALL images
all_imgs = product.select("img.responsive-image__actual")
print(f"\n=== {i+1}. {title[:50]} ===")
print(f"Found {len(all_imgs)} images")
for j, img in enumerate(all_imgs):
src = img.get('src', '')
print(f" Image {j+1}: {src if src else '(no src)'}")
if not src:
# Check other attributes
for attr in ['data-src', 'data-lazy-src', 'srcset']:
val = img.get(attr)
if val:
print(f" {attr}: {val[:80]}")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-71
View File
@@ -1,71 +0,0 @@
"""
Quick diagnostic script to analyze actual HTML structure of search pages
"""
import asyncio
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.search_service import NewSearchService
from app.core.search_config import SITE_CONFIGS
async def diagnose_site(site_key: str, query: str = "chaise"):
"""Run a search and show detailed diagnostics"""
config = SITE_CONFIGS.get(site_key)
if not config:
print(f"❌ Site '{site_key}' not found in config")
return
print(f"\n{'='*80}")
print(f"🔍 Diagnosing: {config['name']} ({site_key})")
print(f"{'='*80}")
print(f"Search URL: {config['search_url'].format(query=query)}")
print(f"Product Selector: {config['product_selector']}")
print(f"Image Selector: {config.get('product_image_selector', 'NONE')}")
try:
results = await NewSearchService.search_site(site_key, query)
print(f"\n📊 Results: {len(results)} products found")
if len(results) == 0:
print("⚠️ NO RESULTS - Check if product_selector is correct")
else:
print("\n✅ Sample Results:")
for i, result in enumerate(results[:3], 1):
print(f"\n {i}. {result.title[:60]}")
print(f" URL: {result.url[:80]}")
print(f" Image: {result.image_url[:80] if result.image_url else '❌ NONE'}")
print(f" Price: {result.price}€" if result.price else " Price: ❌ NONE")
# Count images
with_images = sum(1 for r in results if r.image_url)
print(f"\n📈 Images: {with_images}/{len(results)} ({with_images/len(results)*100:.0f}%)")
with_prices = sum(1 for r in results if r.price)
print(f"💰 Prices: {with_prices}/{len(results)} ({with_prices/len(results)*100:.0f}%)")
except Exception as e:
print(f"❌ ERROR: {e}")
import traceback
traceback.print_exc()
async def main():
"""Diagnose problematic sites"""
sites_to_check = [
"e-leclerc.com",
"auchan.fr",
"carrefour.fr",
"stokomani.fr",
"centrakor.com",
"cdiscount.com",
"lincroyable.fr"
]
for site_key in sites_to_check:
await diagnose_site(site_key)
print("\n" + "="*80 + "\n")
await asyncio.sleep(1) # Rate limiting
if __name__ == "__main__":
asyncio.run(main())
-56
View File
@@ -1,56 +0,0 @@
"""
Script to dump Gifi HTML and analyze structure
"""
import asyncio
import sys
import os
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search page...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded")
# Wait for products
try:
await page.wait_for_selector("article.product-miniature", timeout=10000)
except:
pass
# Save HTML
content = await page.content()
os.makedirs("/app/debug_dumps", exist_ok=True)
with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"HTML saved ({len(content)} bytes)")
# Extract first product structure
products = await page.query_selector_all("article.product-miniature")
print(f"Found {len(products)} products")
if products:
first_html = await products[0].evaluate("el => el.outerHTML")
with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f:
f.write(first_html)
print(f"First product HTML saved")
await context.close()
await browser.close()
await playwright.stop()
print("Done")
if __name__ == "__main__":
asyncio.run(main())
-76
View File
@@ -1,76 +0,0 @@
"""
Dump Gifi with longer wait for JavaScript
"""
import asyncio
import sys
import os
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
async def main():
print("Connecting to browserless...")
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
print("Loading Gifi search page...")
await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle")
# Wait for ANY content
print("Waiting for content...")
await page.wait_for_timeout(5000)
# Save HTML
content = await page.content()
os.makedirs("/app/debug_dumps", exist_ok=True)
with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"HTML saved ({len(content)} bytes)")
# Find any elements with price
price_els = await page.query_selector_all("*:has-text('€')")
print(f"Elements with € symbol: {len(price_els)}")
# Find all divs/articles
all_divs = await page.query_selector_all("div, article, li")
print(f"Total divs/articles/li: {len(all_divs)}")
# Screenshot
await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True)
print("Screenshot saved")
# Get all classes
all_classes = await page.evaluate("""() => {
const elements = document.querySelectorAll('*');
const classes = new Set();
elements.forEach(el => {
if (el.className && typeof el.className === 'string') {
el.className.split(' ').forEach(cls => {
if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) {
classes.add(cls);
}
});
}
});
return Array.from(classes);
}""")
print(f"\\nProduct-related classes found:")
for cls in all_classes:
print(f" - {cls}")
await context.close()
await browser.close()
await playwright.stop()
print("Done")
if __name__ == "__main__":
asyncio.run(main())
-88
View File
@@ -1,88 +0,0 @@
"""
Quick HTML Dumper - Saves raw HTML from search pages for manual analysis
"""
import asyncio
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
async def dump_search_html(site_key: str, query: str = "chaise"):
"""Download and save raw HTML for manual inspection"""
config = SITE_CONFIGS.get(site_key)
if not config:
print(f"❌ Site '{site_key}' not found")
return
print(f"\n🔍 Dumping HTML for: {config['name']}")
# Ensure browser is initialized
await browserless_service.initialize()
try:
search_url = config["search_url"].format(query=query)
print(f" URL: {search_url}")
# Override wait_selector for La Foir'Fouille
wait_selector = config.get("wait_selector")
if site_key == "lafoirfouille.fr":
wait_selector = ".sf-grid-vignet"
print(f" ⚠️ Overriding wait_selector to: {wait_selector}")
html_content, screenshot_path = await browserless_service.get_page_content(
search_url,
wait_selector=wait_selector,
use_proxy=config.get("requires_proxy", False)
)
if not html_content:
print(" ❌ No HTML content returned")
return
filename = f"dump_{site_key.replace('.', '_')}.html"
with open(filename, "w", encoding="utf-8") as f:
f.write(html_content)
print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)")
# Quick analysis
from bs4 import BeautifulSoup
soup = BeautifulSoup(html_content, "html.parser")
# Try current selector
current_selector = config.get("product_selector")
matches = soup.select(current_selector)
print(f" 📊 Current selector '{current_selector}' matches: {len(matches)}")
# Try image selector
if "product_image_selector" in config:
img_selector = config["product_image_selector"]
img_matches = soup.select(img_selector)
print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}")
except Exception as e:
print(f" ❌ Error during dump: {e}")
finally:
# We don't close the browser here to allow reuse if needed,
# but main() will shut it down.
pass
async def main():
sites = [
"stokomani.fr"
]
for site_key in sites:
try:
await dump_search_html(site_key)
except Exception as e:
print(f"❌ Error: {e}")
await asyncio.sleep(1)
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-103
View File
@@ -1,103 +0,0 @@
import json
import urllib.request
import urllib.error
proxies_raw = """
46.161.6.165:8080
78.47.219.204:3128
134.209.29.120:8080
161.35.70.249:80
134.209.29.120:80
52.188.28.218:3128
209.97.150.167:3128
62.60.151.128:80
68.235.35.171:3128
209.97.150.167:80
159.203.61.169:8080
209.97.150.167:8080
195.158.8.123:3128
208.87.243.199:7878
144.76.42.215:8118
216.229.112.25:8080
159.203.61.169:80
103.3.246.71:3128
138.68.60.8:80
139.59.1.14:80
8.243.68.11:8080
41.223.119.156:3128
34.96.238.40:8080
59.6.25.118:3128
129.150.39.251:8000
162.240.154.26:3128
35.152.252.253:8080
144.125.164.158:8081
47.81.14.7:3129
144.125.164.222:8080
175.99.220.171:80
8.219.97.248:80
144.125.164.158:8080
164.68.110.241:8091
144.125.164.222:8081
140.238.184.182:3128
139.59.1.14:3128
8.212.160.196:8080
164.68.110.241:9992
173.212.246.157:3128
47.236.130.95:3128
103.147.246.18:8080
128.199.202.122:80
200.24.159.230:8080
128.199.202.122:8080
103.166.158.251:1111
59.153.16.214:1120
43.224.118.155:1121
89.43.132.247:8080
182.253.62.190:8080
193.95.53.131:8077
203.196.8.6:3128
103.245.110.198:1452
45.180.140.241:8080
212.2.254.246:3128
103.220.206.110:8585
103.157.79.145:1080
45.87.140.155:8080
164.138.205.119:8080
137.59.51.243:1120
38.210.179.77:999
27.147.163.188:40544
194.87.77.22:80
20.27.219.85:8080
49.254.245.70:15648
115.144.173.67:15648
"""
proxy_list = [p.strip() for p in proxies_raw.strip().split("\n") if p.strip()]
ips = [p.split(":")[0] for p in proxy_list]
chunk_size = 100
fr_proxies = []
for i in range(0, len(ips), chunk_size):
chunk = ips[i : i + chunk_size]
try:
req = urllib.request.Request("http://ip-api.com/batch", data=json.dumps(chunk).encode("utf-8"))
with urllib.request.urlopen(req) as response:
data = json.loads(response.read().decode("utf-8"))
for idx, result in enumerate(data):
if result.get("countryCode") == "FR":
full_proxy = proxy_list[i + idx]
fr_proxies.append(full_proxy)
print(f"Found FR proxy: {full_proxy}")
except Exception as e:
print(f"Error querying batch: {e}")
print(f"Total FR proxies found: {len(fr_proxies)}")
if len(fr_proxies) > 0:
for p in fr_proxies:
print(f"PROXY:{p}")
else:
print("No French proxies found. Printing first 10 generic ones as backup:")
for p in proxy_list[:10]:
print(f"PROXY:{p}")
-118
View File
@@ -1,118 +0,0 @@
import urllib.request
import logging
import concurrent.futures
# Setup logging
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("proxy_finder")
# Target URL for verification
TARGET_URL = "https://www.amazon.fr"
def fetch_proxy_list(url):
try:
req = urllib.request.Request(
url, data=None, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
)
with urllib.request.urlopen(req, timeout=10) as response:
if response.status == 200:
text = response.read().decode("utf-8")
proxies = [p.strip() for p in text.splitlines() if p.strip() and ":" in p]
logger.info(f"Fetched {len(proxies)} proxies from {url}")
return proxies
except Exception as e:
logger.error(f"Failed to fetch from {url}: {e}")
return []
def check_proxy_fast(proxy):
try:
proxy_handler = urllib.request.ProxyHandler({"http": proxy, "https": proxy})
opener = urllib.request.build_opener(proxy_handler)
opener.addheaders = [("User-Agent", "Mozilla/5.0")]
with opener.open(TARGET_URL, timeout=5) as response:
# 200, 403, 503 all mean the proxy is alive (amazon may block but proxy works)
if response.status in [200, 403, 503]:
return proxy
except:
pass
return None
def main():
logger.info("Starting proxy finder (FAST MODE)...")
sources = [
"https://api.proxyscrape.com/v2/?request=getproxies&protocol=http&timeout=10000&country=all&ssl=all&anonymity=all",
"https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt",
"https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt",
"https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt",
]
# 1. Fetch all proxies
all_proxies = set()
for url in sources:
proxies = fetch_proxy_list(url)
if proxies:
all_proxies.update(proxies)
# Add local raw proxies
local_raw = [
"164.68.110.241:8091",
"164.68.110.241:9992",
"173.212.246.157:3128",
"142.111.48.253:7030",
"31.59.20.176:6754",
"23.95.150.145:6114",
"198.23.239.134:6540",
"107.172.163.27:6543",
"198.105.121.200:6462",
"64.137.96.74:6641",
"84.247.60.125:6095",
"216.10.27.159:6837",
"142.111.67.146:5611",
]
all_proxies.update(local_raw)
print(f"\nTesting {len(all_proxies)} unique proxies against {TARGET_URL}...")
working_proxies = []
# Use ThreadPoolExecutor for speed
with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor:
future_to_proxy = {executor.submit(check_proxy_fast, p): p for p in all_proxies}
count = 0
total = len(all_proxies)
for future in concurrent.futures.as_completed(future_to_proxy):
count += 1
if count % 500 == 0:
print(f"Processed {count}/{total} - Found {len(working_proxies)} so far")
res = future.result()
if res:
print(f"ALIVE: {res}")
working_proxies.append(res)
# Stop if we have enough
if len(working_proxies) >= 30:
print("Found 30 proxies, stopping.")
executor.shutdown(wait=False, cancel_futures=True)
break
print("\n" + "=" * 50)
print(f"FOUND {len(working_proxies)} WORKING PROXIES")
print("=" * 50)
# Format for python list
formatted_list = "[\n" + ",\n".join([f' "{p}"' for p in working_proxies]) + "\n]"
print(formatted_list)
# Save to file
with open("working_proxies.txt", "w") as f:
f.write("\n".join(working_proxies))
if __name__ == "__main__":
main()
View File
Whitespace-only changes.
-18821
View File
File diff suppressed because it is too large. Load diff
-33
View File
@@ -1,33 +0,0 @@
from bs4 import BeautifulSoup
with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
articles = soup.select("article.product-list-card-plp-grid-new")
print(f"Found {len(articles)} articles")
if articles:
first = articles[0]
print("\n--- First Article Structure ---")
print(first.prettify()[:1000]) # Print first 1000 chars
# Check for link
link = first.select_one("a.product-card-click-wrapper")
if link:
print(f"\nLink found: {link.get('href')}")
print(f"Link classes: {link.get('class')}")
# Check for image INSIDE link
img = link.select_one("img.product-card-image-new__content")
if img:
print(f"\n✅ Image found INSIDE link: {img.get('src')}")
else:
print(f"\n❌ Image NOT found inside link")
# Check if image is elsewhere in article
img_article = first.select_one("img.product-card-image-new__content")
if img_article:
print(f" But image exists in article: {img_article.get('src')}")
else:
print("\nNo link found with selector a.product-card-click-wrapper")
-53
View File
@@ -1,53 +0,0 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
# Try to find product containers
print("Searching for product containers...")
potential_selectors = [
"div.product-miniature",
"article",
"div[class*='product']",
"div.product-card",
"div.item"
]
for selector in potential_selectors:
matches = soup.select(selector)
print(f"Selector '{selector}' matches: {len(matches)}")
if len(matches) > 0 and len(matches) < 5:
# If few matches, print classes to see if it's a wrapper
print(f" Classes: {matches[0].get('class')}")
# Print structure of first potential product
products = soup.select("div.product-miniature")
if not products:
products = soup.select("div[class*='product-item']")
if products:
first = products[0]
print("\n--- First Product Structure ---")
print(first.prettify()[:1000])
link = first.find("a")
if link:
print(f"\nLink found: {link.get('href')}")
img = first.find("img")
if img:
print(f"\nImage found: {img.get('src')}")
else:
print("\nNo obvious products found. Dumping generic structure...")
# Find any div with many children
divs = soup.find_all("div")
for div in divs:
if len(div.find_all("div", recursive=False)) > 10:
print(f"Found container with many children: {div.get('class')}")
# Print first child
child = div.find("div")
if child:
print(child.prettify()[:500])
break
-25
View File
@@ -1,25 +0,0 @@
from bs4 import BeautifulSoup
with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f:
html = f.read()
soup = BeautifulSoup(html, "html.parser")
images = soup.select("img")
print(f"Found {len(images)} images")
for i, img in enumerate(images[:10]):
print(f"\n--- Image {i+1} ---")
print(f"Src: {img.get('src')}")
print(f"Classes: {img.get('class')}")
parent = img.parent
print(f"Parent: {parent.name} (Classes: {parent.get('class')})")
grandparent = parent.parent
if grandparent:
print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})")
greatgrandparent = grandparent.parent
if greatgrandparent:
print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})")
-33
View File
@@ -1,33 +0,0 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(level=logging.INFO)
async def inspect():
url = "https://www.lincroyable.fr/recherche-query=iphone/"
print(f"Inspecting {url}...")
# Use host port for local debugging
os.environ["BROWSERLESS_URL"] = "ws://localhost:3012"
await browserless_service.initialize()
try:
content, _ = await browserless_service.get_page_content(url, wait_selector="body")
with open("lincroyable_dump.html", "w", encoding="utf-8") as f:
f.write(content)
print("HTML dumped to lincroyable_dump.html")
finally:
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(inspect())
-47
View File
@@ -1,47 +0,0 @@
import asyncio
import logging
import os
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def inspect_local():
url = "https://www.lincroyable.fr/recherche-query=iphone/"
print(f"Inspecting {url} using local Playwright...")
async with async_playwright() as p:
# Launch local browser (headless=True matches server environment usually, but we can try False to see)
browser = await p.chromium.launch(headless=True)
context = await browser.new_context(
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
page = await context.new_page()
try:
logging.info(f"Navigating to {url}")
await page.goto(url, timeout=30000)
logging.info("Navigation successful")
content = await page.content()
with open("lincroyable_local_dump.html", "w", encoding="utf-8") as f:
f.write(content)
print("HTML dumped to lincroyable_local_dump.html")
await page.screenshot(path="lincroyable_local.png")
print("Screenshot saved to lincroyable_local.png")
except Exception as e:
logging.error(f"Error: {e}")
try:
await page.screenshot(path="lincroyable_error.png")
except:
pass
finally:
await browser.close()
if __name__ == "__main__":
asyncio.run(inspect_local())
-47
View File
@@ -1,47 +0,0 @@
import asyncio
import logging
from app.services.browserless_service import BrowserlessService
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def dump_site(site_key, query="chaise"):
config = SITE_CONFIGS.get(site_key)
if not config:
logger.error(f"Site {site_key} not found in config")
return
search_url = config["search_url"].format(query=query)
logger.info(f"Dumping {site_key} from {search_url}")
try:
content, _ = await BrowserlessService.get_page_content(
search_url,
wait_selector=config.get("wait_selector"),
use_proxy=config.get("requires_proxy", False)
)
if content:
filename = f"/app/{site_key}_dump.html"
with open(filename, "w", encoding="utf-8") as f:
f.write(content)
logger.info(f"Successfully dumped to {filename}")
else:
logger.error(f"Failed to get content for {site_key}")
except Exception as e:
logger.error(f"Error dumping {site_key}: {e}")
async def main():
await BrowserlessService.initialize()
try:
await dump_site("amazon.fr")
await dump_site("stokomani.fr")
# await dump_site("lincroyable.fr") # Already have this
finally:
await BrowserlessService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
File diff suppressed because it is too large. Load diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 48 KiB

File diff suppressed because it is too large. Load diff
-62
View File
@@ -1,62 +0,0 @@
"""
Script to list all sites in database and check which ones are missing from search_config.py
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
from app.database import SessionLocal
from app.models import SearchSite
from app.core.search_config import SITE_CONFIGS
def main():
"""List all sites in DB and identify missing configurations"""
db = SessionLocal()
try:
# Get all sites from database
sites = db.query(SearchSite).order_by(SearchSite.name).all()
print(f"\n{'='*80}")
print(f"Sites in Database: {len(sites)}")
print(f"Sites in SITE_CONFIGS: {len(SITE_CONFIGS)}")
print(f"{'='*80}\n")
# Check each DB site
missing = []
configured = []
for site in sites:
domain_clean = site.domain.replace("www.", "").lower()
# Check if configured
is_configured = False
for key in SITE_CONFIGS.keys():
if key in domain_clean or domain_clean in key:
configured.append((site.id, site.name, site.domain, key))
is_configured = True
break
if not is_configured:
missing.append((site.id, site.name, site.domain))
# Display results
if configured:
print("✅ CONFIGURED SITES:")
for sid, name, domain, key in configured:
print(f" [{sid:2d}] {name:20s} ({domain:25s}) → {key}")
if missing:
print(f"\n❌ MISSING {len(missing)} SITES:")
for sid, name, domain in missing:
print(f" [{sid:2d}] {name:20s} ({domain})")
else:
print("\n✅ All sites are configured!")
print(f"\n{'='*80}\n")
finally:
db.close()
if __name__ == "__main__":
main()
-67
View File
@@ -1,67 +0,0 @@
import asyncio
import logging
import sys
from playwright.async_api import async_playwright
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
URL = "https://www.gifi.fr/meuble-et-deco/decoration/bougie-et-senteur/diffuseur-et-senteur/encens-nag-champa-15-g/000000000000540823.html"
async def reproduce_scrape():
logger.info("Starting reproduction script...")
async with async_playwright() as p:
# Launch browser (headless=True by default which is what we want for reproduction usually)
# But for debugging blocking, sometimes headless=False helps. Let's start with True (default)
browser = await p.chromium.launch(headless=True)
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
page = await context.new_page()
logger.info(f"Navigating to {URL}...")
try:
await page.goto(URL, wait_until="domcontentloaded", timeout=60000)
logger.info("Page loaded.")
# Wait a bit for dynamic content
await page.wait_for_timeout(5000)
# Extract title
title = await page.title()
logger.info(f"Page Title: {title}")
# Extract body text
content = await page.content()
body_text = await page.inner_text("body")
logger.info(f"Content Length: {len(content)}")
logger.info(f"Body Text Length: {len(body_text)}")
# Check for price
if "€" in body_text:
logger.info("Found '€' in body text.")
else:
logger.warning("'€' NOT found in body text.")
# specific check for likely price
import re
prices = re.findall(r'\d+[,\.]\d{2}\s*€', body_text)
logger.info(f"Prices found in text: {prices}")
# Save content for review
with open("gifi_reproduction.html", "w", encoding="utf-8") as f:
f.write(content)
logger.info("Saved gifi_reproduction.html")
except Exception as e:
logger.error(f"Error during navigation/scraping: {e}")
finally:
await browser.close()
if __name__ == "__main__":
asyncio.run(reproduce_scrape())
-31
View File
@@ -1,31 +0,0 @@
from app.database import SessionLocal
from app.services import auth_service
from app import models
def reset_admin():
db = SessionLocal()
try:
user = auth_service.get_user_by_username(db, "admin")
if user:
print("Found admin user. Resetting password...")
auth_service.update_password(db, user, "admin")
print("Password reset to 'admin'")
# Verify
print("Verifying login...")
auth_user = auth_service.authenticate_user(db, "admin", "admin")
if auth_user:
print("SUCCESS: Login verified!")
else:
print("ERROR: Login failed after reset!")
else:
print("Admin user not found. Creating...")
auth_service.create_user(db, "admin", "admin", is_admin=True)
print("Admin user created with password 'admin'")
except Exception as e:
print(f"Error: {e}")
finally:
db.close()
if __name__ == "__main__":
reset_admin()
-53
View File
@@ -1,53 +0,0 @@
import sys
import os
import logging
from sqlalchemy.orm import Session
# Add project root to path
sys.path.append(os.getcwd())
from app.database import SessionLocal, engine
from app.models import Catalogue
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
def cleanup_catalogs():
db = SessionLocal()
try:
logger.info("Starting cleanup...")
deleted_count = 0
# 1. Delete catalogs with 0 pages
bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all()
for cat in bad_catalogs:
logger.info(f"Deleting empty catalog: {cat.titre}")
db.delete(cat)
deleted_count += 1
# 2. Delete catalogs with iconic/bad images
all_catalogs = db.query(Catalogue).all()
for cat in all_catalogs:
if not cat.image_couverture_url:
continue
if any(
x in cat.image_couverture_url.lower()
for x in ["icon", "logo", "loader", "facebook", "twitter", "assets/img"]
):
logger.info(f"Deleting catalog with bad image: {cat.titre} ({cat.image_couverture_url})")
db.delete(cat)
deleted_count += 1
db.commit()
logger.info(f"Cleanup complete. Deleted {deleted_count} catalogs.")
except Exception as e:
logger.error(f"Error during cleanup: {e}")
finally:
db.close()
if __name__ == "__main__":
cleanup_catalogs()
-18100
View File
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
-32156
View File
File diff suppressed because it is too large. Load diff
+9 -17
View File
@@ -1,29 +1,21 @@
# Investigation et Correction du Statut "Produit Retiré"
# Nettoyage et Commit du Projet
## Contexte
Le système de suivi des prix marquait incorrectement certains produits comme "retirés" alors qu'ils étaient toujours disponibles, souvent à cause de blocages (bot detection) ou de changements mineurs de titre.
Après avoir corrigé les problèmes de disponibilité d'Action.com et Amazon, il est nécessaire de nettoyer le répertoire racine des scripts de diagnostic et fichiers temporaires avant de committer les changements.
## Focus Actuel
Finalisation et vérification.
Identification et suppression des fichiers inutiles, suivi du commit et push.
## Master Plan
- [x] Analyser la logique de comparaison de titres dans `app/services/scheduler_service.py`
- [x] Vérifier si les scrapers extraient correctement les titres lors des mises à jour
- [x] Identifier les cas limites (edge cases) où la similarité de titre échoue
- [x] Corriger la logique pour éviter les faux positifs d'indisponibilité
- [x] Vérifier la correction avec un exemple concret (test_logic.py)
- [x] Implémenter le reset automatique de disponibilité si un prix est trouvé
- [ ] Identifier les fichiers non essentiels à l'application
- [ ] Supprimer les fichiers de diagnostic et scripts temporaires
- [ ] Vérifier que l'application fonctionne toujours (build/syntaxe)
- [ ] Committer les changements vers le dépôt Git
- [ ] Pusher les changements
## Log de Progression
- [x] Investigation terminée : identification des faux positifs dus aux titres de blocage (Cloudflare, etc.) et aux placeholders ("Loading").
- [x] Logique de normalisation renforcée.
- [x] Détection des bots ajoutée pour tous les sites.
- [x] Auto-reset de `is_available` implémenté dans `_update_db_result`.
- [x] Correction déployée dans `scheduler_service.py`.
- [x] Affinage de la détection de bot pour Stokomani/L'Incroyable (détection conditionnelle au titre).
- [x] Support des versions "V2" et matching par mots pour les noms courts sur Amazon.
- [x] Sécurisation du flux Action.com (check d'indisponibilité déplacé après le titre).
- [ ] Planification du nettoyage commencée.
-187
View File
@@ -1,187 +0,0 @@
#!/usr/bin/env python3
"""
Script de test pour le scraper Amazon France
Teste le système anti-détection et l'extraction des produits
"""
import asyncio
import logging
import sys
from pathlib import Path
# Ajouter le répertoire app au path
sys.path.insert(0, str(Path(__file__).parent))
from app.services.amazon_scraper_service import amazon_scraper_service, AmazonScraperService
from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_DATA
# Define missing variable for test compatibility
AMAZON_USER_AGENTS = USER_AGENT_DATA
USER_AGENT_POOL = USER_AGENT_DATA
# Configuration du logging
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s")
logger = logging.getLogger(__name__)
# Monkey patch _connect_browser to use local launch for testing
async def _connect_browser_local(p):
logger.info("Launching local browser (headless)...")
return await p.chromium.launch(headless=True)
AmazonScraperService._connect_browser = _connect_browser_local
async def test_basic_search():
"""Test basique de recherche"""
logger.info("=" * 80)
logger.info("TEST 1: Recherche basique - 'aspirateur'")
logger.info("=" * 80)
products = await amazon_scraper_service.scrape_search("aspirateur", max_results=5)
if not products:
logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau")
return False
logger.info(f"✅ {len(products)} produits trouvés")
for idx, product in enumerate(products, 1):
logger.info(f"\n{idx}. {product.title[:60]}...")
logger.info(
f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else "")
)
logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note")
logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}")
logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}")
logger.info(f" {'📢 Sponsorisé' if product.sponsored else '🔍 Organique'}")
return True
async def test_multiple_queries():
"""Test avec plusieurs requêtes différentes"""
logger.info("\n" + "=" * 80)
logger.info("TEST 2: Requêtes multiples")
logger.info("=" * 80)
queries = ["clavier", "souris", "casque"]
results = {}
for query in queries:
logger.info(f"\n🔍 Recherche: '{query}'")
products = await amazon_scraper_service.scrape_search(query, max_results=3)
results[query] = len(products)
logger.info(f" ✅ {len(products)} produits trouvés")
# Délai entre requêtes pour respecter les bonnes pratiques
await asyncio.sleep(3)
logger.info("\n📊 Résumé:")
for query, count in results.items():
logger.info(f" • {query}: {count} produits")
total = sum(results.values())
if total > 0:
logger.info(f"\n✅ Total: {total} produits extraits")
return True
else:
logger.error("\n❌ Aucun produit extrait - problème possible")
return False
async def test_anti_detection():
"""Test du système anti-détection"""
logger.info("\n" + "=" * 80)
logger.info("TEST 3: Vérification anti-détection")
logger.info("=" * 80)
# Updated to just check if we can run
logger.info("Skipping specific proxy/agent checks for this service as it handles them internally")
logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles")
logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards")
logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques")
# Test proxy
# Test proxy
import random
proxy = random.choice(AMAZON_PROXY_LIST_RAW) if AMAZON_PROXY_LIST_RAW else None
if proxy:
# Extract just the IP for logging (hide credentials)
proxy_parts = proxy.split("@")
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
logger.info(f"✓ Proxy test: {proxy_server}")
else:
logger.warning("⚠️ Pas de proxy configuré")
# Test d'une recherche simple
logger.info("\n🧪 Test de recherche avec anti-détection...")
products = await amazon_scraper_service.scrape_search("livre", max_results=3)
if products:
logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits")
return True
else:
logger.error("❌ Échec - possibilité de blocage")
return False
async def run_all_tests():
"""Lance tous les tests"""
logger.info("\n" + "=" * 80)
logger.info("🚀 DÉMARRAGE DES TESTS DU SCRAPER AMAZON FRANCE")
logger.info("=" * 80)
tests = [
("Recherche basique", test_basic_search),
("Requêtes multiples", test_multiple_queries),
("Anti-détection", test_anti_detection),
]
results = {}
for test_name, test_func in tests:
try:
logger.info(f"\n▶️ Exécution: {test_name}")
success = await test_func()
results[test_name] = "✅ PASS" if success else "❌ FAIL"
except Exception as e:
logger.error(f"❌ Erreur dans {test_name}: {e}", exc_info=True)
results[test_name] = "❌ ERROR"
# Résumé final
logger.info("\n" + "=" * 80)
logger.info("📊 RÉSUMÉ DES TESTS")
logger.info("=" * 80)
for test_name, result in results.items():
logger.info(f"{result} - {test_name}")
passed = sum(1 for r in results.values() if "PASS" in r)
total = len(results)
logger.info(f"\n🎯 Score: {passed}/{total} tests réussis")
if passed == total:
logger.info("✅ TOUS LES TESTS ONT RÉUSSI!")
return True
else:
logger.warning("⚠️ Certains tests ont échoué")
return False
if __name__ == "__main__":
try:
success = asyncio.run(run_all_tests())
sys.exit(0 if success else 1)
except KeyboardInterrupt:
logger.info("\n⏸️ Tests interrompus par l'utilisateur")
sys.exit(130)
except Exception as e:
logger.error(f"❌ Erreur fatale: {e}", exc_info=True)
sys.exit(1)
-42
View File
@@ -1,42 +0,0 @@
"""
Test Carrefour via API endpoint (production-like)
"""
import requests
import json
url = "http://localhost:8555/api/search"
params = {
"q": "chaise",
"sites": "7", # Carrefour site ID
"max_results": 10
}
print(f"Testing: {url}")
print(f"Params: {params}\n")
response = requests.get(url, params=params, stream=True)
print(f"Status: {response.status_code}")
print(f"Headers: {dict(response.headers)}\n")
count = 0
for line in response.iter_lines():
if line:
try:
# Each line should be JSON
data = json.loads(line.decode('utf-8'))
count += 1
title = data.get('title', 'N/A')[:60]
price = data.get('price', 'N/A')
image = "✅" if data.get('image_url') else "❌"
print(f"{count}. [{image}] {title} - {price}€")
if count >= 10:
break
except json.JSONDecodeError as e:
print(f"JSON Error: {e}")
print(f"Line: {line[:100]}")
print(f"\nTotal: {count} results")
-40
View File
@@ -1,40 +0,0 @@
"""
Test if current Carrefour price extraction works
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Carrefour for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
results.append(result)
count += 1
price_status = f"{result.price}€" if result.price else "N/A"
print(f"{count}. {result.title[:55]} - {price_status}")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Count prices
with_prices = sum(1 for r in results if r.price)
print(f"Products with prices: {with_prices}/{len(results)}")
if with_prices == len(results):
print("✅ SUCCESS: All products have prices!")
else:
print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-26
View File
@@ -1,26 +0,0 @@
"""
Test script to trigger Carrefour search and generate HTML dump
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Carrefour for 'chaise'...")
results = []
async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"):
results.append(result)
print(f"Found: {result.title}")
print(f"\nTotal results: {len(results)}")
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-69
View File
@@ -1,69 +0,0 @@
"""
Test detailed logging for Centrakor image extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from playwright.async_api import async_playwright
from bs4 import BeautifulSoup
import re
async def main():
playwright = await async_playwright().start()
browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000")
context = await browser.new_context(
viewport={"width": 1920, "height": 1080},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
)
page = await context.new_page()
await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle")
content = await page.content()
soup = BeautifulSoup(content, "html.parser")
products = soup.select("div.product-item")
print(f"Testing first product:\n")
first = products[0]
# Get all images
img_els = first.select("img.responsive-image__actual")
print(f"Found {len(img_els)} images with selector")
for i, img_el in enumerate(img_els):
print(f"\n=== Image {i+1} ===")
candidate_url = img_el.get("src")
print(f"URL: {candidate_url}")
# Test filters
if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']):
print(" ❌ Filtered: Contains picto/icon/logo/badge keyword")
continue
width_match = re.search(r'width=(\d+)', candidate_url)
height_match = re.search(r'height=(\d+)', candidate_url)
print(f" Width match: {width_match.group(1) if width_match else None}")
print(f" Height match: {height_match.group(1) if height_match else None}")
if width_match and height_match:
width = int(width_match.group(1))
height = int(height_match.group(1))
print(f" Dimensions: {width}x{height}")
if width < 100 and height < 100:
print(f" ❌ Filtered: Too small ({width}x{height})")
continue
else:
print(f" ✅ PASS: Large enough ({width}x{height})")
else:
print(" ✅ PASS: No dimensions in URL")
await context.close()
await browser.close()
await playwright.stop()
if __name__ == "__main__":
asyncio.run(main())
-42
View File
@@ -1,42 +0,0 @@
"""
Test Centrakor image extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Centrakor for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"):
results.append(result)
count += 1
has_image = "✅" if result.image_url else "❌"
print(f"{count}. {has_image} {result.title[:55]} - {result.price}€")
if result.image_url:
print(f" Image: {result.image_url[:70]}...")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Count images
with_images = sum(1 for r in results if r.image_url)
print(f"Products with images: {with_images}/{len(results)}")
if with_images == len(results):
print("✅ SUCCESS: All products have images!")
else:
print(f"⚠️ WARNING: {len(results) - with_images} products missing images")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-34
View File
@@ -1,34 +0,0 @@
"""
Final test of Gifi with new selectors
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
print(f"✓ {result.title[:50]} - {result.price}€")
print(f"\n==> Total: {len(results)} results")
if results:
print("\nFirst 3 products:")
for i, r in enumerate(results[:3]):
print(f"{i+1}. Title: {r.title}")
print(f" Price: {r.price}€")
print(f" URL: {r.url[:80]}...")
print()
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-38
View File
@@ -1,38 +0,0 @@
"""
Test Gifi price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
count += 1
print(f"{count}. {result.title[:60]} - Price: {result.price}€")
if count >= 5: # Only test first 5
break
print(f"\n==> Got {len(results)} results")
# Check if all prices are the same
prices = [r.price for r in results if r.price]
if prices:
unique_prices = set(prices)
print(f"Unique prices: {unique_prices}")
if len(unique_prices) == 1:
print("⚠️ WARNING: All prices are the same!")
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-41
View File
@@ -1,41 +0,0 @@
"""
Test fixed Gifi price extraction
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...\n")
results = []
count = 0
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
count += 1
print(f"{count}. {result.title[:60]} - {result.price}€")
if count >= 10:
break
print(f"\n==> Got {len(results)} results")
# Check price diversity
prices = [r.price for r in results if r.price]
if prices:
unique_prices = set(prices)
print(f"Unique prices: {sorted(unique_prices)}")
print(f"Price range: {min(prices)}€ - {max(prices)}€")
if len(unique_prices) > 1:
print("✅ SUCCESS: Multiple different prices found!")
else:
print("⚠️ WARNING: All prices are the same")
print("\nShutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-26
View File
@@ -1,26 +0,0 @@
"""
Test script to analyze Gifi HTML structure
"""
import asyncio
import sys
sys.path.insert(0, '/app')
from app.services.improved_search_service import ImprovedSearchService
async def main():
print("Initializing browser...")
await ImprovedSearchService.initialize()
print("Searching Gifi for 'chaise'...")
results = []
async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"):
results.append(result)
print(f"Found: {result.title} - Price: {result.price}")
print(f"\nTotal results: {len(results)}")
print("Shutting down...")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
-29
View File
@@ -1,29 +0,0 @@
from bs4 import BeautifulSoup
import sys
try:
with open("/app/debug_dumps/stokomani.fr_failed_verification.html", "r", encoding="utf-8") as f:
content = f.read()
print(f"Read {len(content)} bytes")
soup = BeautifulSoup(content, "html.parser")
print("Soup created")
selector = "div.product-card"
items = soup.select(selector)
print(f"Found {len(items)} items with selector '{selector}'")
if items:
item = items[0]
print("First item classes:", item.get("class"))
title_selector = "h3.product-card__title a"
title_el = item.select_one(title_selector)
if title_el:
print("Title found:", title_el.get_text(strip=True))
else:
print(f"Title NOT found with '{title_selector}'")
except Exception as e:
print(f"Error: {e}")
-43
View File
@@ -1,43 +0,0 @@
import asyncio
import logging
import sys
import os
# Identify workspace root
sys.path.append(os.getcwd())
from app.services.cataloguemate_scraper import scrape_catalog_pages
# Setup logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_scraper():
# Gifi catalog URL from the browser session
url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/"
print(f"Verifying scraper on: {url}")
pages = await scrape_catalog_pages(url)
print(f"Found {len(pages)} pages.")
if not pages:
print("FAIL: No pages found.")
return
# check first page image
first_img = pages[0]["image_url"]
print(f"Page 1 Image: {first_img}")
if "thumbor" in first_img or "leafletscdns" in first_img:
print("SUCCESS: Image is a Thumbor/Leaflet URL.")
elif "icon" in first_img or "logo" in first_img:
print("FAIL: Image appears to be an icon/logo.")
else:
print(f"WARNING: Image URL is: {first_img}")
if __name__ == "__main__":
asyncio.run(verify_scraper())
-45
View File
@@ -1,45 +0,0 @@
import asyncio
import logging
import sys
import os
# Identify workspace root
sys.path.append(os.getcwd())
# Mock Env Vars
os.environ["DATABASE_URL"] = "postgresql://user:password@localhost:5432/pricewatch"
os.environ["BROWSERLESS_URL"] = "ws://localhost:3012" # Ignored by patch, but good for completeness
from app.services.improved_search_service import ImprovedSearchService
from playwright.async_api import async_playwright
# Setup logging
logging.basicConfig(level=logging.DEBUG) # DEBUG level to see price extraction logic
logger = logging.getLogger(__name__)
# Monkey patch _connect_browser to use local launch
async def _connect_browser_local(p):
logger.info("Launching local browser (headless)...")
return await p.chromium.launch(headless=True)
ImprovedSearchService._connect_browser = _connect_browser_local
async def test_search():
await ImprovedSearchService.initialize()
query = "nintendo switch"
target_site = "gifi.fr"
print(f"Searching for: {query} on {target_site}")
async for result in ImprovedSearchService.search_site_generator(target_site, query):
print(f"[{result.source}] {result.title}\n -> Price: {result.price}€\n -> URL: {result.url}")
await ImprovedSearchService.shutdown()
if __name__ == "__main__":
asyncio.run(test_search())
-108
View File
@@ -1,108 +0,0 @@
import asyncio
import logging
import re
import sys
# Mock logger
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# Fallback implementation of _fetch_with_fallback for standalone test
async def _fetch_with_fallback(url):
# We need to install httpx for this to work
try:
import httpx
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
async with httpx.AsyncClient(verify=False, timeout=30.0) as client:
response = await client.get(url, headers=headers)
return response.text
except ImportError:
print("Please pip install httpx strict")
return ""
async def scrape_catalog_pages_standalone(catalog_url: str):
from bs4 import BeautifulSoup
print(f"Scraping: {catalog_url}")
html_content = await _fetch_with_fallback(catalog_url)
if not html_content:
print("Failed to fetch content")
return []
soup = BeautifulSoup(html_content, "html.parser")
# --- COPIED LOGIC FROM cataloguemate_scraper.py ---
main_image_url = None
max_area = 0
# Strategy 1: Look for specific container/class identified in browser inspection
candidates = soup.select(".letaky-grid-preview img")
# Strategy 2: Fallback to all images if specific container not found
if not candidates:
candidates = soup.find_all("img")
print(f"Found {len(candidates)} candidates")
for img in candidates:
# Check multiple attributes for the real image URL
src = img.get("src") or img.get("data-src") or img.get("data-original")
if not src:
continue
# Skip common UI elements - refined list
if any(
x in src.lower()
for x in [
"logo",
"icon",
"facebook",
"twitter",
"instagram",
"loader",
"spinner",
"market",
"googleplay",
"appstore",
]
):
continue
# Strong Signal: URL contains 'thumbor' or 'leafletscdns' (host for catalog images)
is_thumbor = "thumbor" in src.lower() or "leafletscdns" in src.lower()
# Calculate area if dimensions exist
width = img.get("width")
height = img.get("height")
area = 0
if width and height:
try:
area = int(width) * int(height)
except:
pass
if is_thumbor:
if area > max_area or (area == 0 and max_area == 0):
max_area = area
main_image_url = src
print(f"Match (Thumbor): {src}")
elif area > 50000:
if area > max_area:
max_area = area
main_image_url = src
print(f"Match (Size): {src}")
return main_image_url
if __name__ == "__main__":
url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/"
result = asyncio.run(scrape_catalog_pages_standalone(url))
print(f"FINAL RESULT: {result}")