diff --git a/amazon.fr_dump.html b/amazon.fr_dump.html deleted file mode 100644 index dab68f3..0000000 --- a/amazon.fr_dump.html +++ /dev/null @@ -1,19490 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -Amazon.fr : chaise - - - - - - - - - - - - - - - - - -
- - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
- - -
\ No newline at end of file diff --git a/analyze_bm_prices.py b/analyze_bm_prices.py deleted file mode 100644 index e1778ce..0000000 --- a/analyze_bm_prices.py +++ /dev/null @@ -1,82 +0,0 @@ -""" -Analyze B&M product page for price extraction -""" -import asyncio -import sys -import re -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - # URL from the screenshot - url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966" - print(f"Loading: {url}\n") - await page.goto(url, wait_until="networkidle") - - content = await page.content() - soup = BeautifulSoup(content, "html.parser") - - # Find all elements with € symbol - price_els = soup.find_all(string=re.compile('€')) - print(f"Found {len(price_els)} elements with '€'\n") - - prices_found = {} - for el in price_els[:30]: - text = el.strip() - if text and len(text) < 100: - parent = el.find_parent() - if parent: - parent_class = parent.get('class', []) - parent_class_str = ' '.join(parent_class) if isinstance(parent_class, list) else str(parent_class) - - # Extract price value - price_match = re.search(r'(\d+[.,]\d+)\s*€', text) - if price_match: - price_val = price_match.group(1) - key = f"{price_val}€ in .{parent_class_str[:50]}" - if key not in prices_found: - prices_found[key] = text - - print("Prices found:") - for key, text in prices_found.items(): - print(f" {key}: '{text}'") - - # Try common price selectors - print("\nTrying specific selectors:") - selectors = [ - ".price", - "[class*='price']", - "[data-price]", - ".product-price", - "span[class*='price']" - ] - - for selector in selectors: - try: - els = soup.select(selector) - if els: - for el in els[:2]: - text = el.get_text(strip=True) - if '€' in text: - print(f" {selector}: {text}") - except: - pass - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_bm_v2.py b/analyze_bm_v2.py deleted file mode 100644 index a7d9f74..0000000 --- a/analyze_bm_v2.py +++ /dev/null @@ -1,50 +0,0 @@ -""" -Analyze B&M with longer wait and playwright evaluation -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - url = "https://www.bmstores.fr/products/chaise-haute-pliante-bois-492966" - print(f"Loading: {url}\n") - await page.goto(url, wait_until="networkidle") - - # Wait extra time for JS - await page.wait_for_timeout(5000) - - # Look for elements containing price - print("Searching for price elements...\n") - - # Try to find any text with € - price_els = await page.query_selector_all("*:has-text('€')") - print(f"Found {len(price_els)} elements with €\n") - - for i, el in enumerate(price_els[:10]): - text = await el.inner_text() - tag = await el.evaluate("el => el.tagName") - classes = await el.evaluate("el => el.className") - print(f"{i+1}. <{tag} class='{classes}'> {text[:100]}") - - # Screenshot for debugging - await page.screenshot(path="/app/debug_dumps/bm_screenshot.png", full_page=True) - print("\nScreenshot saved to /app/debug_dumps/bm_screenshot.png") - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_carrefour_prices.py b/analyze_carrefour_prices.py deleted file mode 100644 index c7aa92d..0000000 --- a/analyze_carrefour_prices.py +++ /dev/null @@ -1,59 +0,0 @@ -""" -Analyze Carrefour price extraction -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - print("Loading Carrefour search...") - await page.goto("https://www.carrefour.fr/s?q=chaise", wait_until="networkidle") - - content = await page.content() - soup = BeautifulSoup(content, "html.parser") - - products = soup.select("article.product-list-card-plp-grid-new") - print(f"Found {len(products)} products\n") - - for i, product in enumerate(products[:3]): - print(f"=== Product {i+1} ===") - - # Title - title_el = product.select_one("h3, h2, a") - title = title_el.get_text(strip=True) if title_el else "N/A" - print(f"Title: {title[:60]}") - - # Find all text with € symbol - import re - product_html = str(product) - prices = re.findall(r'(\d+[.,]\d+)\s*€', product_html) - print(f"Prices found in HTML: {prices}") - - # Look for price elements - price_els = product.find_all(string=re.compile('€')) - if price_els: - print(f"Elements with €:") - for el in price_els[:3]: - print(f" - {el.strip()[:50]}") - - print() - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_carrefour_product.py b/analyze_carrefour_product.py deleted file mode 100644 index 96d18b3..0000000 --- a/analyze_carrefour_product.py +++ /dev/null @@ -1,70 +0,0 @@ -""" -Analyze a Carrefour product page for price selectors -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup -import re - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - # Visit a Carrefour product page (from the screenshot) - url = "https://www.carrefour.fr/p/chaise-pliante-44x45-7x79-cm-gris-carrefour-home-3245390032010" - print(f"Loading: {url}") - await page.goto(url, wait_until="networkidle") - - content = await page.content() - soup = BeautifulSoup(content, "html.parser") - - # Find all elements with € symbol - price_els = soup.find_all(string=re.compile('€')) - print(f"\nFound {len(price_els)} elements with '€'") - - prices_found = set() - for el in price_els[:20]: - text = el.strip() - if text and len(text) < 50: - prices_found.add(text) - parent = el.find_parent() - print(f" '{text}' in <{parent.name} class='{parent.get('class')}'>") - - # Try common price selectors - selectors = [ - ".product-price", - "[class*='price']", - ".price", - "span.price", - "div.price", - "[data-price]" - ] - - print("\nTrying specific selectors:") - for selector in selectors: - try: - els = soup.select(selector) - if els: - for el in els[:2]: - text = el.get_text(strip=True) - if '€' in text: - print(f" {selector}: {text}") - except: - pass - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_centrakor.py b/analyze_centrakor.py deleted file mode 100644 index ea34efb..0000000 --- a/analyze_centrakor.py +++ /dev/null @@ -1,69 +0,0 @@ -""" -Analyze Centrakor HTML structure for image selectors -""" -import asyncio -import sys -import re -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - print("Connecting to browserless...") - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - print("Loading Centrakor search...") - await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle") - - content = await page.content() - - soup = BeautifulSoup(content, "html.parser") - - # Try to find product containers - selectors = [ - "div.product-item", - "div.product-card", - "article", - "div[class*='product']", - "li[class*='product']" - ] - - for selector in selectors: - products = soup.select(selector) - if products: - print(f"\n✓ Found {len(products)} products with selector: {selector}") - - # Analyze first product - first = products[0] - print(f"\nFirst product HTML snippet:") - print(str(first)[:500]) - print("\n...") - - # Find all images - images = first.find_all('img') - print(f"\nFound {len(images)} images in first product:") - for i, img in enumerate(images): - print(f"\n Image {i+1}:") - print(f" Class: {img.get('class')}") - print(f" Src: {img.get('src', '')[:80]}") - print(f" Data-src: {img.get('data-src', '')[:80]}") - print(f" Alt: {img.get('alt', '')[:50]}") - - break - - await context.close() - await browser.close() - await playwright.stop() - print("\nDone") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_centrakor_detailed.py b/analyze_centrakor_detailed.py deleted file mode 100644 index 0d205b5..0000000 --- a/analyze_centrakor_detailed.py +++ /dev/null @@ -1,58 +0,0 @@ -""" -Detailed analysis of Centrakor image structure -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle") - content = await page.content() - - soup = BeautifulSoup(content, "html.parser") - products = soup.select("div.product-item") - - print(f"Analyzing {min(5, len(products))} products:\n") - - for i, product in enumerate(products[:5]): - print(f"=== Product {i+1} ===") - - # Title - title_el = product.select_one("a.product-item__name") - title = title_el.get_text(strip=True) if title_el else "N/A" - print(f"Title: {title}") - - # All images - images = product.find_all('img') - print(f"Found {len(images)} img tags") - - for j, img in enumerate(images): - print(f"\n Image {j+1}:") - print(f" tag: {img.name}") - print(f" class: {img.get('class')}") - for attr in ['src', 'data-src', 'data-lazy-src', 'srcset', 'data-srcset']: - val = img.get(attr) - if val: - print(f" {attr}: {val[:80]}") - - print() - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_gifi.py b/analyze_gifi.py deleted file mode 100644 index 64bf2b0..0000000 --- a/analyze_gifi.py +++ /dev/null @@ -1,24 +0,0 @@ -""" -Analyze Gifi HTML to find correct selectors -""" -with open("/app/debug_dumps/gifi_full.html", "r", encoding="utf-8") as f: - html = f.read() - -# Find product-related divs -import re -matches = re.findall(r']*class="[^"]*"[^>]*>.*?', html[:50000], re.DOTALL) - -print(f"Total HTML size: {len(html)} bytes") - -# Search for price patterns -price_patterns = re.findall(r'50[.,]00\s*€', html[:50000]) -print(f"\nFound {len(price_patterns)} instances of '50,00 €'") - -# Find all class names containing specific keywords -for keyword in ['product', 'article', 'item', 'card']: - classes = re.findall(rf'class="([^"]*{keyword}[^"]*)"', html[:100000], re.IGNORECASE) - unique_classes = set(classes) - if unique_classes: - print(f"\nClasses containing '{keyword}':") - for cls in sorted(unique_classes): - print(f" - {cls}") diff --git a/analyze_gifi_price.py b/analyze_gifi_price.py deleted file mode 100644 index 2c9c2a5..0000000 --- a/analyze_gifi_price.py +++ /dev/null @@ -1,69 +0,0 @@ -""" -Analyze a Gifi product page to understand price structure -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright - -async def main(): - print("Connecting to browserless...") - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - # Visit a Gifi product page - url = "https://www.gifi.fr/meuble-et-deco/linge-de-maison/coussin-plaid-et-tapis/housse-de-chaise-canape-ou-fauteuil/housse-de-chaise-uni-blanc/000000000000410028.html" - print(f"Loading: {url}") - await page.goto(url, wait_until="networkidle") - - # Get page title - title = await page.title() - print(f"Title: {title}") - - # Find all elements with price-like text - price_els = await page.query_selector_all("*:has-text('€')") - print(f"\nFound {len(price_els)} elements with '€'") - - # Get first 10 price elements - for i, el in enumerate(price_els[:10]): - text = await el.inner_text() - tag = await el.evaluate("el => el.tagName") - classes = await el.evaluate("el => el.className") - print(f"{i+1}. <{tag} class='{classes}'> {text[:50]}") - - # Try specific selectors - selectors = [ - ".price", - ".product-price", - "[class*='price']", - "[data-price]", - "span.price", - "div.price" - ] - - print("\nTrying specific selectors:") - for selector in selectors: - try: - els = await page.query_selector_all(selector) - if els: - for el in els[:2]: - text = await el.inner_text() - print(f" {selector}: {text}") - except: - pass - - await context.close() - await browser.close() - await playwright.stop() - print("\nDone") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_gifi_search_prices.py b/analyze_gifi_search_prices.py deleted file mode 100644 index c2018a9..0000000 --- a/analyze_gifi_search_prices.py +++ /dev/null @@ -1,71 +0,0 @@ -""" -Test extracting price from Gifi search page HTML -""" -import asyncio -import sys -import re -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - print("Connecting to browserless...") - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - print("Loading Gifi search...") - await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle") - - content = await page.content() - - soup = BeautifulSoup(content, "html.parser") - products = soup.select("div.product-tile") - - print(f"Found {len(products)} products\n") - - for i, product in enumerate(products[:3]): - print(f"\\n=== Product {i+1} ===") - - # Get title - title_el = product.select_one("div.pdp-link a") - title = title_el.get_text(strip=True) if title_el else "N/A" - print(f"Title: {title}") - - # Try to find price in product HTML - product_html = product.prettify() - - # Look for price patterns - price_patterns = [ - r'(\d+)[,.](\d+)\s*€', # 19,99 € or 19.99 € - r'€\s*(\d+)[,.](\d+)', # € 19,99 - r'(\d+)€(\d+)', # 19€99 - r'"price"\s*:\s*"?(\d+\.?\d*)"?', # JSON price - ] - - for pattern in price_patterns: - matches = re.findall(pattern, product_html) - if matches: - print(f"Pattern '{pattern}': {matches[:3]}") - - # Find all text with € - euro_texts = product.find_all(string=re.compile('€')) - if euro_texts: - print(f"Texts with €:") - for text in euro_texts[:5]: - print(f" - {text.strip()[:80]}") - - await context.close() - await browser.close() - await playwright.stop() - print("\nDone") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/analyze_missing_sites.py b/analyze_missing_sites.py deleted file mode 100644 index fa7eca5..0000000 --- a/analyze_missing_sites.py +++ /dev/null @@ -1,93 +0,0 @@ -""" -Script pour analyser les sélecteurs CSS des sites manquants -""" -import asyncio -from playwright.async_api import async_playwright - -async def analyze_site(url: str, site_name: str): - print(f"\n=== Analyse de {site_name} ===") - print(f"URL: {url}") - - async with async_playwright() as p: - browser = await p.chromium.launch(headless=False) - page = await browser.new_page() - - try: - await page.goto(url, wait_until="networkidle", timeout=30000) - await page.wait_for_timeout(3000) - - # Try cookie banners - cookie_selectors = [ - "button:has-text('Accepter')", - "button:has-text('Tout accepter')", - "#didomi-notice-agree-button", - ".didomi-continue-without-agreeing" - ] - for selector in cookie_selectors: - try: - await page.click(selector, timeout=2000) - print(f"✓ Cookie banner fermé: {selector}") - break - except: - pass - - await page.wait_for_timeout(2000) - - # Save HTML - html = await page.content() - filename = f"{site_name.lower().replace(' ', '_')}_search.html" - with open(filename, "w", encoding="utf-8") as f: - f.write(html) - print(f"✓ HTML sauvegardé: {filename}") - - # Take screenshot - screenshot_path = f"{site_name.lower().replace(' ', '_')}_search.png" - await page.screenshot(path=screenshot_path, full_page=True) - print(f"✓ Screenshot: {screenshot_path}") - - # Test common product link selectors - test_selectors = [ - "a.product-link", - "a.product-name", - "a[href*='/produit']", - "a[href*='/product']", - "a[href*='/p/']", - ".product-title a", - ".product-item a", - "article a", - "a.product", - "a[itemprop='url']", - ] - - print("\n--- Test de sélecteurs ---") - for selector in test_selectors: - try: - elements = await page.query_selector_all(selector) - if elements: - print(f"✓ {selector}: {len(elements)} éléments trouvés") - # Get first few hrefs - for i, elem in enumerate(elements[:3]): - href = await elem.get_attribute("href") - text = await elem.inner_text() - print(f" [{i+1}] {text[:50]} -> {href}") - except Exception as e: - pass - - except Exception as e: - print(f"❌ Erreur: {e}") - finally: - await browser.close() - -async def main(): - sites = [ - ("https://bmstores.fr/module/ambjolisearch/jolisearch?s=chaise", "BM"), - ("https://www.centrakor.com/recherche?controller=search&s=chaise", "Centrakor"), - ("https://www.lincroyable.fr/recherche?query=chaise", "L'Incroyable"), - ] - - for url, name in sites: - await analyze_site(url, name) - await asyncio.sleep(2) - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/app/services/scheduler_service.py b/app/services/scheduler_service.py index 459f7fe..e2808a3 100644 --- a/app/services/scheduler_service.py +++ b/app/services/scheduler_service.py @@ -50,6 +50,10 @@ def _normalize_title(title: str) -> str: for pattern in patterns: title = re.sub(pattern, "", title) + # Separate numbers from letters (handles cases like 'Blanc40' or '100pièces') + title = re.sub(r"([a-zA-Z])(\d)", r"\1 \2", title) + title = re.sub(r"(\d)([a-zA-Z])", r"\1 \2", title) + # Remove extra spaces and punctuation that might differ between sites title = re.sub(r"[^\w\s]", " ", title) title = " ".join(title.split()) @@ -323,7 +327,14 @@ async def process_item_check(item_id: int): html_lower = html_content.lower() if html_content else "" if any(term in final_url_lower for term in unavailable_terms) or any( - term in html_lower for term in unavailable_terms + term in html_lower + for term in [ + # Use more specific terms for HTML to avoid false positives in random text or meta + "malheureusement, ce produit est actuellement indisponible", + ">produit indisponible<", + ">ce produit est indisponible<", + "product-unavailable-message", + ] ): logger.warning(f"Product unavailable confirmed for item {item_id} on Action.com") await loop.run_in_executor( diff --git a/app/services/tracking_scraper_service.py b/app/services/tracking_scraper_service.py index f6bab4c..f7f9de9 100644 --- a/app/services/tracking_scraper_service.py +++ b/app/services/tracking_scraper_service.py @@ -31,7 +31,6 @@ POPUP_SELECTORS = [ "input[value='Continuer les achats']", "input[value='Continue shopping']", "input[value='Continue shopping']", - "form:has-text('Continuer les achats') input[type='submit']", "[aria-labelledby='continue-shopping-label']", "#sp-cc-accept", @@ -53,7 +52,6 @@ POPUP_SELECTORS = [ ] - # Random User-Agents to alternate fingerprint AMAZON_USER_AGENTS = [ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", @@ -63,6 +61,7 @@ AMAZON_USER_AGENTS = [ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0", ] + @dataclass class ScrapeConfig: """Configuration for scraping parameters.""" @@ -174,12 +173,13 @@ class ScraperService: try: # Random delay to simulate human lead-in import random + await asyncio.sleep(random.uniform(0.5, 2.0)) - + await ScraperService._navigate_and_wait(page, url, timeout) final_url = page.url page_title = await page.title() - + # Humanize: scroll a bit and back await page.mouse.move(random.randint(100, 500), random.randint(100, 500)) await page.evaluate("window.scrollBy(0, 100)") @@ -198,16 +198,18 @@ class ScraperService: logger.info("Opening store selector...") await store_btn.first.click(timeout=2000) await page.wait_for_timeout(1000) - + # Type Nancy in the search input search_input = page.locator("#search_mag") if await search_input.count() > 0: await search_input.fill("Nancy") await page.keyboard.press("Enter") await page.wait_for_timeout(1500) - + # Select the first Nancy store (Nancy Essey or Nancy Centre) - select_btn = page.locator(".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')") + select_btn = page.locator( + ".shop-list .btn-select-shop, button:has-text('Choisir ce magasin')" + ) if await select_btn.count() > 0: logger.info("Selecting Nancy store...") await select_btn.first.click() @@ -230,7 +232,7 @@ class ScraperService: content_data = await page.content() else: content_data = await ScraperService._extract_text(page, config.text_length) - + screenshot_path = await ScraperService._take_screenshot(page, url, item_id) return screenshot_path, content_data, final_url, page_title @@ -244,10 +246,7 @@ class ScraperService: @staticmethod async def _scrape_amazon_specific( - url: str, - item_id: int | None, - config: ScrapeConfig, - return_html: bool + url: str, item_id: int | None, config: ScrapeConfig, return_html: bool ) -> tuple[str | None, str, str, str]: """ Specialized scraping flow for Amazon to avoid bot detection and ensure good screenshots. @@ -258,15 +257,16 @@ class ScraperService: 4. Wait for main image to be visible """ logger.info(f"🛒 Starting specialized Amazon scrape for: {url}") - + # Determine base domain from urllib.parse import urlparse + parsed = urlparse(url) base_domain = f"{parsed.scheme}://{parsed.netloc}" - + context = await ScraperService._create_context(ScraperService._browser, url) page = await context.new_page() - + try: # 1. Warm-up: Visit Homepage to get cookies/session try: @@ -280,7 +280,7 @@ class ScraperService: # 2. Navigate to Product logger.info(f"➡️ Navigating to product page: {url}") await page.goto(url, wait_until="domcontentloaded", timeout=60000) - + final_url = page.url try: page_title = await page.title() @@ -289,22 +289,28 @@ class ScraperService: # 3. Check for Hard Redirects (URL-based) if "/ap/signin" in page.url: - logger.error("🚫 Amazon Login Redirect detected (URL)!") - return None, "LOGIN_REQUIRED", final_url, page_title + logger.error("🚫 Amazon Login Redirect detected (URL)!") + return None, "LOGIN_REQUIRED", final_url, page_title # 4. Handle Popups & Location Selectors await ScraperService._handle_popups(page) - + # Dismiss "Change Address" or specific Amazon location modals if any try: - await page.evaluate("document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')") - except: pass + await page.evaluate( + "document.getElementById('nav-main')?.classList.remove('nav-progressive-attribute')" + ) + except: + pass # 5. Check for Bot Detection / CAPTCHA / Login (Content-based) # We do this AFTER popup removal because sometimes "Identifiez-vous" is in a dismissible modal content_check = await page.content() - - if "Type the characters you see in this image" in content_check or "Saisissez les caractères que vous voyez" in content_check: + + if ( + "Type the characters you see in this image" in content_check + or "Saisissez les caractères que vous voyez" in content_check + ): logger.error("🚫 Amazon CAPTCHA detected!") # Attempt refresh once logger.info("Retrying with refresh...") @@ -313,7 +319,7 @@ class ScraperService: # Re-check content_check = await page.content() if "Type the characters you see in this image" in content_check: - return None, "CAPTCHA_DETECTED", final_url, page_title + return None, "CAPTCHA_DETECTED", final_url, page_title if "Identifiez-vous" in content_check or "ap_signin" in content_check: # Double check: is it a modal we missed? Try one last specific closure @@ -323,40 +329,46 @@ class ScraperService: logger.info("Found login modal close button, clicking...") await close_btn.first.click() await page.wait_for_timeout(1000) - content_check = await page.content() # Refresh content - except: pass + content_check = await page.content() # Refresh content + except: + pass if "Identifiez-vous" in content_check or "ap_signin" in content_check: - # Retry once for Login wall too - logger.info("⚠️ Amazon Login/Auth detected. Retrying with refresh...") - await page.reload() - await asyncio.sleep(3) - content_check = await page.content() + # Final Check: Do we have a product title? + # If we have a title, it's just a popup we missed/hid. Proceed. + # If NO title, it's a hard redirect/gate. Fail. + try: + title_check = page.locator("#productTitle, #title") + if await title_check.count() > 0 and await title_check.first.is_visible(): + logger.info("⚠️ Login prompt detected but Product Title found. Ignoring/Hiding modal...") + # Attempt to brute-force remove the modal overlay again just in case + await page.evaluate( + "() => document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())" + ) + else: + # No title found, now we can try to reload or fail + logger.info("⚠️ Amazon Login/Auth detected and No Title found. Retrying with refresh...") + await page.reload() + await asyncio.sleep(3) + content_check = await page.content() - if "Identifiez-vous" in content_check or "ap_signin" in content_check: - # Final Check: Do we have a product title? - # If we have a title, it's just a popup we missed/hid. Proceed. - # If NO title, it's a hard redirect/gate. Fail. - try: - title_check = page.locator("#productTitle, #title") - if await title_check.count() > 0: - logger.info("⚠️ Login prompt detected but Product Title found. Ignoring/Hiding modal...") - # Attempt to brute-force remove the modal overlay again just in case - await page.evaluate("document.querySelectorAll('.a-popover-modal, .a-modal-scroller').forEach(e => e.remove())") - else: - logger.error("🚫 Amazon Login Prompt detected (Blocking)!") - return None, "LOGIN_REQUIRED", final_url, page_title - except: - logger.error("🚫 Amazon Login Prompt detected (Error Checking Title)!") - return None, "LOGIN_REQUIRED", final_url, page_title + if "Identifiez-vous" in content_check or "ap_signin" in content_check: + # Re-check title after reload + if await title_check.count() > 0 and await title_check.first.is_visible(): + logger.info("⚠️ Title appeared after refresh despite login prompt.") + else: + logger.error("🚫 Amazon Login Prompt detected (Blocking)!") + return None, "LOGIN_REQUIRED", final_url, page_title + except Exception as e: + logger.error(f"🚫 Amazon Login Prompt detected (Error Checking Title: {e})") + return None, "LOGIN_REQUIRED", final_url, page_title # 6. Wait for Main Image (Critical for screenshot) logger.info("🖼️ Waiting for product image...") try: # Main image container on desktop await page.wait_for_selector( - "#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper", - timeout=10000 + "#imgTagWrapperId, #landingImage, #main-image-container, .imgTagWrapper", timeout=10000 ) except Exception as e: logger.warning(f"Could not find main image container: {e}") @@ -373,10 +385,10 @@ class ScraperService: content_data = await page.content() else: content_data = await ScraperService._extract_text(page, config.text_length) - + # 8. Screenshot screenshot_path = await ScraperService._take_screenshot(page, url, item_id) - + return screenshot_path, content_data, final_url, page_title except Exception as e: @@ -396,14 +408,15 @@ class ScraperService: """Create context with advanced stealth and headers (specifically for Amazon)""" # Determine base domain for referer from urllib.parse import urlparse + parsed = urlparse(url) base_domain = f"{parsed.scheme}://{parsed.netloc}/" if "amazon" in url: - ua = random.choice(AMAZON_USER_AGENTS) - logger.info(f"🎭 Using rotated User-Agent: {ua}") + ua = random.choice(AMAZON_USER_AGENTS) + logger.info(f"🎭 Using rotated User-Agent: {ua}") else: - ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" + ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" context = await browser.new_context( viewport={"width": 1920, "height": 1080}, @@ -422,8 +435,8 @@ class ScraperService: "Sec-Fetch-Site": "none", "Sec-Fetch-User": "?1", "Upgrade-Insecure-Requests": "1", - "Referer": base_domain if "amazon" in url else "https://www.google.com/" - } + "Referer": base_domain if "amazon" in url else "https://www.google.com/", + }, ) # Advanced Stealth mode @@ -475,10 +488,10 @@ class ScraperService: async def _handle_popups(page: Page): """Aggressive multi-pass popup and overlay removal""" logger.info("Starting aggressive popup removal...") - + # 1. Multi-pass clicking (some popups appear after others are closed) for i in range(2): - logger.debug(f"Popup removal pass {i+1}") + logger.debug(f"Popup removal pass {i + 1}") for selector in POPUP_SELECTORS: try: locators = page.locator(selector) @@ -492,12 +505,13 @@ class ScraperService: await page.wait_for_timeout(500) except Exception: pass - + # Hammer Escape key try: await page.keyboard.press("Escape") await page.wait_for_timeout(200) - except: pass + except: + pass # 2. Javascript cleanup (Hide pesky overlays and CMPs that won't close) logger.info("Injecting CSS/JS cleanup for persistent overlays...") @@ -540,7 +554,7 @@ class ScraperService: document.documentElement.style.overflow = 'auto'; } """) - + # Wait for any transitions await page.wait_for_timeout(1000) @@ -582,7 +596,7 @@ class ScraperService: try: logger.info(f"Extracting text (limit: {text_length} chars)...") - + # Smart extraction: remove noise (nav, footer, scripts) before getting text clean_text = await page.evaluate(""" () => { @@ -604,7 +618,7 @@ class ScraperService: return clone.innerText; } """) - + # Fallback if cleaning removed everything (unlikely but safe) if not clean_text or len(clean_text) < 100: logger.warning("Cleaned text too short, falling back to full body text") @@ -612,7 +626,8 @@ class ScraperService: # Collapse whitespace import re - clean_text = re.sub(r'\s+', ' ', clean_text).strip() + + clean_text = re.sub(r"\s+", " ", clean_text).strip() page_text = clean_text[:text_length] logger.info(f"Extracted {len(page_text)} chars") diff --git a/check_bs4.py b/check_bs4.py deleted file mode 100644 index cc10bf9..0000000 --- a/check_bs4.py +++ /dev/null @@ -1,2 +0,0 @@ -from app.services.improved_search_service import BeautifulSoup -print("BeautifulSoup imported successfully") diff --git a/check_import.py b/check_import.py deleted file mode 100644 index 8adb705..0000000 --- a/check_import.py +++ /dev/null @@ -1,12 +0,0 @@ -import sys -import os -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) -import app.core.search_config -print(f"Search Config File: {app.core.search_config.__file__}") -from app.core.search_config import SITE_CONFIGS -print(f"Keys: {list(SITE_CONFIGS.keys())}") - -with open(app.core.search_config.__file__, 'r') as f: - content = f.read() - print(f"File content length: {len(content)}") - print(f"Contains stokomani.fr: {'stokomani.fr' in content}") diff --git a/check_method.py b/check_method.py deleted file mode 100644 index aebd9e5..0000000 --- a/check_method.py +++ /dev/null @@ -1,10 +0,0 @@ -import sys -from app.services.improved_search_service import ImprovedSearchService - -print("Attributes of ImprovedSearchService:") -attrs = dir(ImprovedSearchService) -if 'search_site_generator' in attrs: - print("SUCCESS: search_site_generator found") -else: - print("FAILURE: search_site_generator NOT found") - print("Available attributes:", [a for a in attrs if not a.startswith('__')]) diff --git a/check_proxies.py b/check_proxies.py deleted file mode 100644 index 43ce566..0000000 --- a/check_proxies.py +++ /dev/null @@ -1,52 +0,0 @@ -import asyncio -import os -import sys - -# Add app to path -sys.path.insert(0, os.getcwd()) - -from playwright.async_api import async_playwright -from app.core.search_config import get_amazon_proxies - - -async def check_proxy(proxy, semaphore): - async with semaphore: - proxy_url = proxy["server"] - print(f"Testing {proxy_url}...") - - async with async_playwright() as p: - # Connect to browserless or launch local - # Using launch local for simpler testing without ws dependency if possible - # But the app uses browserless. Let's try launch first. - try: - browser = await p.chromium.launch(headless=True, proxy=proxy) - page = await browser.new_page() - try: - # amazon.fr might block, use httpbin for connectivity check - await page.goto("http://httpbin.org/ip", timeout=15000) - content = await page.content() - print(f"✅ {proxy_url}: Success") - await browser.close() - return True - except Exception as e: - print(f"❌ {proxy_url}: Failed - {str(e)[:100]}") - await browser.close() - return False - except Exception as e: - print(f"❌ {proxy_url}: Launch Failed - {str(e)[:100]}") - return False - - -async def main(): - proxies = get_amazon_proxies() - print(f"Checking {len(proxies)} proxies...") - - semaphore = asyncio.Semaphore(3) # Limit concurrency - results = await asyncio.gather(*[check_proxy(p, semaphore) for p in proxies]) - - working = sum(results) - print(f"\nSummary: {working}/{len(proxies)} working.") - - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/check_syntax.py b/check_syntax.py deleted file mode 100644 index 4cc2eb0..0000000 --- a/check_syntax.py +++ /dev/null @@ -1,7 +0,0 @@ -import sys -try: - from app.services import improved_search_service - print("Import successful") -except Exception as e: - print(f"Import failed: {e}") - sys.exit(1) diff --git a/deep_centrakor_analysis.py b/deep_centrakor_analysis.py deleted file mode 100644 index 2be63d0..0000000 --- a/deep_centrakor_analysis.py +++ /dev/null @@ -1,56 +0,0 @@ -""" -Deep analysis - compare products WITH images vs WITHOUT images -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle") - content = await page.content() - - soup = BeautifulSoup(content, "html.parser") - products = soup.select("div.product-item") - - print(f"Analyzing {len(products)} products for image patterns\n") - - for i, product in enumerate(products[:10]): - # Get title - title_el = product.select_one("a.product-item__name") - title = title_el.get_text(strip=True) if title_el else f"Product {i+1}" - - # Get ALL images - all_imgs = product.select("img.responsive-image__actual") - - print(f"\n=== {i+1}. {title[:50]} ===") - print(f"Found {len(all_imgs)} images") - - for j, img in enumerate(all_imgs): - src = img.get('src', '') - print(f" Image {j+1}: {src if src else '(no src)'}") - if not src: - # Check other attributes - for attr in ['data-src', 'data-lazy-src', 'srcset']: - val = img.get(attr) - if val: - print(f" {attr}: {val[:80]}") - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/diagnose_selectors.py b/diagnose_selectors.py deleted file mode 100644 index 362b255..0000000 --- a/diagnose_selectors.py +++ /dev/null @@ -1,71 +0,0 @@ -""" -Quick diagnostic script to analyze actual HTML structure of search pages -""" -import asyncio -import sys -from pathlib import Path -sys.path.insert(0, str(Path(__file__).parent)) - -from app.services.search_service import NewSearchService -from app.core.search_config import SITE_CONFIGS - -async def diagnose_site(site_key: str, query: str = "chaise"): - """Run a search and show detailed diagnostics""" - config = SITE_CONFIGS.get(site_key) - if not config: - print(f"❌ Site '{site_key}' not found in config") - return - - print(f"\n{'='*80}") - print(f"🔍 Diagnosing: {config['name']} ({site_key})") - print(f"{'='*80}") - print(f"Search URL: {config['search_url'].format(query=query)}") - print(f"Product Selector: {config['product_selector']}") - print(f"Image Selector: {config.get('product_image_selector', 'NONE')}") - - try: - results = await NewSearchService.search_site(site_key, query) - - print(f"\n📊 Results: {len(results)} products found") - - if len(results) == 0: - print("⚠️ NO RESULTS - Check if product_selector is correct") - else: - print("\n✅ Sample Results:") - for i, result in enumerate(results[:3], 1): - print(f"\n {i}. {result.title[:60]}") - print(f" URL: {result.url[:80]}") - print(f" Image: {result.image_url[:80] if result.image_url else '❌ NONE'}") - print(f" Price: {result.price}€" if result.price else " Price: ❌ NONE") - - # Count images - with_images = sum(1 for r in results if r.image_url) - print(f"\n📈 Images: {with_images}/{len(results)} ({with_images/len(results)*100:.0f}%)") - - with_prices = sum(1 for r in results if r.price) - print(f"💰 Prices: {with_prices}/{len(results)} ({with_prices/len(results)*100:.0f}%)") - - except Exception as e: - print(f"❌ ERROR: {e}") - import traceback - traceback.print_exc() - -async def main(): - """Diagnose problematic sites""" - sites_to_check = [ - "e-leclerc.com", - "auchan.fr", - "carrefour.fr", - "stokomani.fr", - "centrakor.com", - "cdiscount.com", - "lincroyable.fr" - ] - - for site_key in sites_to_check: - await diagnose_site(site_key) - print("\n" + "="*80 + "\n") - await asyncio.sleep(1) # Rate limiting - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/dump_gifi.py b/dump_gifi.py deleted file mode 100644 index bcb8c31..0000000 --- a/dump_gifi.py +++ /dev/null @@ -1,56 +0,0 @@ -""" -Script to dump Gifi HTML and analyze structure -""" -import asyncio -import sys -import os -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright - -async def main(): - print("Connecting to browserless...") - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - print("Loading Gifi search page...") - await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="domcontentloaded") - - # Wait for products - try: - await page.wait_for_selector("article.product-miniature", timeout=10000) - except: - pass - - # Save HTML - content = await page.content() - os.makedirs("/app/debug_dumps", exist_ok=True) - with open("/app/debug_dumps/gifi_full.html", "w", encoding="utf-8") as f: - f.write(content) - - print(f"HTML saved ({len(content)} bytes)") - - # Extract first product structure - products = await page.query_selector_all("article.product-miniature") - print(f"Found {len(products)} products") - - if products: - first_html = await products[0].evaluate("el => el.outerHTML") - with open("/app/debug_dumps/gifi_first_product.html", "w", encoding="utf-8") as f: - f.write(first_html) - print(f"First product HTML saved") - - await context.close() - await browser.close() - await playwright.stop() - print("Done") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/dump_gifi_v2.py b/dump_gifi_v2.py deleted file mode 100644 index 2594580..0000000 --- a/dump_gifi_v2.py +++ /dev/null @@ -1,76 +0,0 @@ -""" -Dump Gifi with longer wait for JavaScript -""" -import asyncio -import sys -import os -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright - -async def main(): - print("Connecting to browserless...") - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - print("Loading Gifi search page...") - await page.goto("https://www.gifi.fr/resultat-recherche?q=chaise", wait_until="networkidle") - - # Wait for ANY content - print("Waiting for content...") - await page.wait_for_timeout(5000) - - # Save HTML - content = await page.content() - os.makedirs("/app/debug_dumps", exist_ok=True) - with open("/app/debug_dumps/gifi_with_wait.html", "w", encoding="utf-8") as f: - f.write(content) - - print(f"HTML saved ({len(content)} bytes)") - - # Find any elements with price - price_els = await page.query_selector_all("*:has-text('€')") - print(f"Elements with € symbol: {len(price_els)}") - - # Find all divs/articles - all_divs = await page.query_selector_all("div, article, li") - print(f"Total divs/articles/li: {len(all_divs)}") - - # Screenshot - await page.screenshot(path="/app/debug_dumps/gifi_screenshot.png", full_page=True) - print("Screenshot saved") - - # Get all classes - all_classes = await page.evaluate("""() => { - const elements = document.querySelectorAll('*'); - const classes = new Set(); - elements.forEach(el => { - if (el.className && typeof el.className === 'string') { - el.className.split(' ').forEach(cls => { - if (cls && (cls.includes('product') || cls.includes('item') || cls.includes('card'))) { - classes.add(cls); - } - }); - } - }); - return Array.from(classes); - }""") - - print(f"\\nProduct-related classes found:") - for cls in all_classes: - print(f" - {cls}") - - await context.close() - await browser.close() - await playwright.stop() - print("Done") - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/dump_html.py b/dump_html.py deleted file mode 100644 index 3c3742f..0000000 --- a/dump_html.py +++ /dev/null @@ -1,88 +0,0 @@ -""" -Quick HTML Dumper - Saves raw HTML from search pages for manual analysis -""" -import asyncio -import sys -from pathlib import Path -sys.path.insert(0, str(Path(__file__).parent)) - -from app.services.browserless_service import browserless_service -from app.core.search_config import SITE_CONFIGS - -async def dump_search_html(site_key: str, query: str = "chaise"): - """Download and save raw HTML for manual inspection""" - config = SITE_CONFIGS.get(site_key) - if not config: - print(f"❌ Site '{site_key}' not found") - return - - print(f"\n🔍 Dumping HTML for: {config['name']}") - - # Ensure browser is initialized - await browserless_service.initialize() - - try: - search_url = config["search_url"].format(query=query) - print(f" URL: {search_url}") - - # Override wait_selector for La Foir'Fouille - wait_selector = config.get("wait_selector") - if site_key == "lafoirfouille.fr": - wait_selector = ".sf-grid-vignet" - print(f" ⚠️ Overriding wait_selector to: {wait_selector}") - - html_content, screenshot_path = await browserless_service.get_page_content( - search_url, - wait_selector=wait_selector, - use_proxy=config.get("requires_proxy", False) - ) - - if not html_content: - print(" ❌ No HTML content returned") - return - - filename = f"dump_{site_key.replace('.', '_')}.html" - with open(filename, "w", encoding="utf-8") as f: - f.write(html_content) - - print(f" ✅ Saved to: {filename} ({len(html_content)} bytes)") - - # Quick analysis - from bs4 import BeautifulSoup - soup = BeautifulSoup(html_content, "html.parser") - - # Try current selector - current_selector = config.get("product_selector") - matches = soup.select(current_selector) - print(f" 📊 Current selector '{current_selector}' matches: {len(matches)}") - - # Try image selector - if "product_image_selector" in config: - img_selector = config["product_image_selector"] - img_matches = soup.select(img_selector) - print(f" 🖼️ Current image selector '{img_selector}' matches: {len(img_matches)}") - - except Exception as e: - print(f" ❌ Error during dump: {e}") - - finally: - # We don't close the browser here to allow reuse if needed, - # but main() will shut it down. - pass - -async def main(): - sites = [ - "stokomani.fr" - ] - - for site_key in sites: - try: - await dump_search_html(site_key) - except Exception as e: - print(f"❌ Error: {e}") - await asyncio.sleep(1) - - await browserless_service.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/filter_proxies.py b/filter_proxies.py deleted file mode 100644 index c63526d..0000000 --- a/filter_proxies.py +++ /dev/null @@ -1,103 +0,0 @@ -import json -import urllib.request -import urllib.error - -proxies_raw = """ -46.161.6.165:8080 -78.47.219.204:3128 -134.209.29.120:8080 -161.35.70.249:80 -134.209.29.120:80 -52.188.28.218:3128 -209.97.150.167:3128 -62.60.151.128:80 -68.235.35.171:3128 -209.97.150.167:80 -159.203.61.169:8080 -209.97.150.167:8080 -195.158.8.123:3128 -208.87.243.199:7878 -144.76.42.215:8118 -216.229.112.25:8080 -159.203.61.169:80 -103.3.246.71:3128 -138.68.60.8:80 -139.59.1.14:80 -8.243.68.11:8080 -41.223.119.156:3128 -34.96.238.40:8080 -59.6.25.118:3128 -129.150.39.251:8000 -162.240.154.26:3128 -35.152.252.253:8080 -144.125.164.158:8081 -47.81.14.7:3129 -144.125.164.222:8080 -175.99.220.171:80 -8.219.97.248:80 -144.125.164.158:8080 -164.68.110.241:8091 -144.125.164.222:8081 -140.238.184.182:3128 -139.59.1.14:3128 -8.212.160.196:8080 -164.68.110.241:9992 -173.212.246.157:3128 -47.236.130.95:3128 -103.147.246.18:8080 -128.199.202.122:80 -200.24.159.230:8080 -128.199.202.122:8080 -103.166.158.251:1111 -59.153.16.214:1120 -43.224.118.155:1121 -89.43.132.247:8080 -182.253.62.190:8080 -193.95.53.131:8077 -203.196.8.6:3128 -103.245.110.198:1452 -45.180.140.241:8080 -212.2.254.246:3128 -103.220.206.110:8585 -103.157.79.145:1080 -45.87.140.155:8080 -164.138.205.119:8080 -137.59.51.243:1120 -38.210.179.77:999 -27.147.163.188:40544 -194.87.77.22:80 -20.27.219.85:8080 -49.254.245.70:15648 -115.144.173.67:15648 -""" - -proxy_list = [p.strip() for p in proxies_raw.strip().split("\n") if p.strip()] -ips = [p.split(":")[0] for p in proxy_list] - -chunk_size = 100 -fr_proxies = [] - -for i in range(0, len(ips), chunk_size): - chunk = ips[i : i + chunk_size] - try: - req = urllib.request.Request("http://ip-api.com/batch", data=json.dumps(chunk).encode("utf-8")) - with urllib.request.urlopen(req) as response: - data = json.loads(response.read().decode("utf-8")) - - for idx, result in enumerate(data): - if result.get("countryCode") == "FR": - full_proxy = proxy_list[i + idx] - fr_proxies.append(full_proxy) - print(f"Found FR proxy: {full_proxy}") - except Exception as e: - print(f"Error querying batch: {e}") - -print(f"Total FR proxies found: {len(fr_proxies)}") - -if len(fr_proxies) > 0: - for p in fr_proxies: - print(f"PROXY:{p}") -else: - print("No French proxies found. Printing first 10 generic ones as backup:") - for p in proxy_list[:10]: - print(f"PROXY:{p}") diff --git a/find_working_proxies.py b/find_working_proxies.py deleted file mode 100644 index be4b28d..0000000 --- a/find_working_proxies.py +++ /dev/null @@ -1,118 +0,0 @@ -import urllib.request -import logging -import concurrent.futures - -# Setup logging -logging.basicConfig(level=logging.INFO, format="%(message)s") -logger = logging.getLogger("proxy_finder") - -# Target URL for verification -TARGET_URL = "https://www.amazon.fr" - - -def fetch_proxy_list(url): - try: - req = urllib.request.Request( - url, data=None, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} - ) - with urllib.request.urlopen(req, timeout=10) as response: - if response.status == 200: - text = response.read().decode("utf-8") - proxies = [p.strip() for p in text.splitlines() if p.strip() and ":" in p] - logger.info(f"Fetched {len(proxies)} proxies from {url}") - return proxies - except Exception as e: - logger.error(f"Failed to fetch from {url}: {e}") - return [] - - -def check_proxy_fast(proxy): - try: - proxy_handler = urllib.request.ProxyHandler({"http": proxy, "https": proxy}) - opener = urllib.request.build_opener(proxy_handler) - opener.addheaders = [("User-Agent", "Mozilla/5.0")] - with opener.open(TARGET_URL, timeout=5) as response: - # 200, 403, 503 all mean the proxy is alive (amazon may block but proxy works) - if response.status in [200, 403, 503]: - return proxy - except: - pass - return None - - -def main(): - logger.info("Starting proxy finder (FAST MODE)...") - - sources = [ - "https://api.proxyscrape.com/v2/?request=getproxies&protocol=http&timeout=10000&country=all&ssl=all&anonymity=all", - "https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt", - "https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt", - "https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt", - ] - - # 1. Fetch all proxies - all_proxies = set() - for url in sources: - proxies = fetch_proxy_list(url) - if proxies: - all_proxies.update(proxies) - - # Add local raw proxies - local_raw = [ - "164.68.110.241:8091", - "164.68.110.241:9992", - "173.212.246.157:3128", - "142.111.48.253:7030", - "31.59.20.176:6754", - "23.95.150.145:6114", - "198.23.239.134:6540", - "107.172.163.27:6543", - "198.105.121.200:6462", - "64.137.96.74:6641", - "84.247.60.125:6095", - "216.10.27.159:6837", - "142.111.67.146:5611", - ] - all_proxies.update(local_raw) - - print(f"\nTesting {len(all_proxies)} unique proxies against {TARGET_URL}...") - - working_proxies = [] - - # Use ThreadPoolExecutor for speed - with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor: - future_to_proxy = {executor.submit(check_proxy_fast, p): p for p in all_proxies} - - count = 0 - total = len(all_proxies) - - for future in concurrent.futures.as_completed(future_to_proxy): - count += 1 - if count % 500 == 0: - print(f"Processed {count}/{total} - Found {len(working_proxies)} so far") - - res = future.result() - if res: - print(f"ALIVE: {res}") - working_proxies.append(res) - # Stop if we have enough - if len(working_proxies) >= 30: - print("Found 30 proxies, stopping.") - executor.shutdown(wait=False, cancel_futures=True) - break - - print("\n" + "=" * 50) - print(f"FOUND {len(working_proxies)} WORKING PROXIES") - print("=" * 50) - - # Format for python list - formatted_list = "[\n" + ",\n".join([f' "{p}"' for p in working_proxies]) + "\n]" - print(formatted_list) - - # Save to file - with open("working_proxies.txt", "w") as f: - f.write("\n".join(working_proxies)) - - -if __name__ == "__main__": - main() diff --git a/gifi_dump.html b/gifi_dump.html deleted file mode 100644 index e69de29..0000000 diff --git a/gifi_full.html b/gifi_full.html deleted file mode 100644 index 0455d34..0000000 --- a/gifi_full.html +++ /dev/null @@ -1,18821 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - chaise pas cher | GIFI - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
- - - - - - - - - - - - - - - - - - - - - - - - - - -
- - - - - - -
- - - gifi example - -
-
- - - gifi example - -
- - - - -
-
- - - -
- - - - - - - - - - - - - - - -
- - -
- - - - - - - - - -
-
-
-
- -
-
- - - - -

- - Résultats de recherche pour : - - - chaise - -

-
- - Nombre de produits trouvés : - - - 147 - -
- -
- - - - -
-
- -
-
- - - -
- -
-
- - - - - - - - -
-
-

- Filtrer -

- -
-
- -
- -
-
- -
-
- - - - - -
    - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - - - -
  • - -
  • - - - -
- - - - - -
-
- -
-
- -
-
- - - - - - - - -
    -
  • -
    -
    -
    -
    -
    - - -
    -
    -
    - - - - € - -
    -
    - - - - € - -
    -
    -
    - - - - - - -
  • -
- - -
-
- -
- - - -
- - -
- - - -
-
- - - - -
- - - - - - - - -
    - - -
- -
- - -
-
- - -
- -
- - -
- -
- - -
- -
- - -
- -
- - -
- -
- - -
- -
- - -
- -
- -
-
- -
- -
- - -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 7,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 16,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 7,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 79,90 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 25,00 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 25,00 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 39,95 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 39,90 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 45,00 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 17,90 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,29 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 17,90 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 2,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,29 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 5,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 1,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 40,27 € - - - - - - - - - - - - - -
- -
- -
- - - - - - - Price reduced from - - - -
- - 79,99 € - - - - - - - - - - - - - -
- - - to - -
-
- -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 3,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 2,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 2,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 2,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- -
- - - - - - -
- - - - - - - -
- - - - -
- - - - - - - - - - - -
- - - - - - - - - - - - - -
- - - - - - - - - - - - - - -
- - 2,99 € - - - - - - - - - - - - - -
- -
- -
- - -
- - -
- - - - - - - -
- - - -
- Vendu par GiFi -
- - - -
- - -
- - - - - - - - -
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
-
- - - 50% - - - - - En Bon d'achat - - -
-
-
- - - -
- - -
- -
- - -
- - - - - -
-
-
- -
-
-
- - -
- - - - - -
-
-
- - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/inspect_carrefour.py b/inspect_carrefour.py deleted file mode 100644 index 91a0b10..0000000 --- a/inspect_carrefour.py +++ /dev/null @@ -1,33 +0,0 @@ -from bs4 import BeautifulSoup - -with open("dump_carrefour_fr.html", "r", encoding="utf-8") as f: - html = f.read() - -soup = BeautifulSoup(html, "html.parser") -articles = soup.select("article.product-list-card-plp-grid-new") - -print(f"Found {len(articles)} articles") - -if articles: - first = articles[0] - print("\n--- First Article Structure ---") - print(first.prettify()[:1000]) # Print first 1000 chars - - # Check for link - link = first.select_one("a.product-card-click-wrapper") - if link: - print(f"\nLink found: {link.get('href')}") - print(f"Link classes: {link.get('class')}") - - # Check for image INSIDE link - img = link.select_one("img.product-card-image-new__content") - if img: - print(f"\n✅ Image found INSIDE link: {img.get('src')}") - else: - print(f"\n❌ Image NOT found inside link") - # Check if image is elsewhere in article - img_article = first.select_one("img.product-card-image-new__content") - if img_article: - print(f" But image exists in article: {img_article.get('src')}") - else: - print("\nNo link found with selector a.product-card-click-wrapper") diff --git a/inspect_lafoirfouille.py b/inspect_lafoirfouille.py deleted file mode 100644 index 55d43de..0000000 --- a/inspect_lafoirfouille.py +++ /dev/null @@ -1,53 +0,0 @@ -from bs4 import BeautifulSoup - -with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f: - html = f.read() - -soup = BeautifulSoup(html, "html.parser") - -# Try to find product containers -print("Searching for product containers...") -potential_selectors = [ - "div.product-miniature", - "article", - "div[class*='product']", - "div.product-card", - "div.item" -] - -for selector in potential_selectors: - matches = soup.select(selector) - print(f"Selector '{selector}' matches: {len(matches)}") - if len(matches) > 0 and len(matches) < 5: - # If few matches, print classes to see if it's a wrapper - print(f" Classes: {matches[0].get('class')}") - -# Print structure of first potential product -products = soup.select("div.product-miniature") -if not products: - products = soup.select("div[class*='product-item']") - -if products: - first = products[0] - print("\n--- First Product Structure ---") - print(first.prettify()[:1000]) - - link = first.find("a") - if link: - print(f"\nLink found: {link.get('href')}") - - img = first.find("img") - if img: - print(f"\nImage found: {img.get('src')}") -else: - print("\nNo obvious products found. Dumping generic structure...") - # Find any div with many children - divs = soup.find_all("div") - for div in divs: - if len(div.find_all("div", recursive=False)) > 10: - print(f"Found container with many children: {div.get('class')}") - # Print first child - child = div.find("div") - if child: - print(child.prettify()[:500]) - break diff --git a/inspect_lafoirfouille_images.py b/inspect_lafoirfouille_images.py deleted file mode 100644 index b3e1cb3..0000000 --- a/inspect_lafoirfouille_images.py +++ /dev/null @@ -1,25 +0,0 @@ -from bs4 import BeautifulSoup - -with open("dump_lafoirfouille_fr.html", "r", encoding="utf-8") as f: - html = f.read() - -soup = BeautifulSoup(html, "html.parser") -images = soup.select("img") - -print(f"Found {len(images)} images") - -for i, img in enumerate(images[:10]): - print(f"\n--- Image {i+1} ---") - print(f"Src: {img.get('src')}") - print(f"Classes: {img.get('class')}") - - parent = img.parent - print(f"Parent: {parent.name} (Classes: {parent.get('class')})") - - grandparent = parent.parent - if grandparent: - print(f"Grandparent: {grandparent.name} (Classes: {grandparent.get('class')})") - - greatgrandparent = grandparent.parent - if greatgrandparent: - print(f"Great Grandparent: {greatgrandparent.name} (Classes: {greatgrandparent.get('class')})") diff --git a/inspect_lincroyable.py b/inspect_lincroyable.py deleted file mode 100644 index 177ecd8..0000000 --- a/inspect_lincroyable.py +++ /dev/null @@ -1,33 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Add project root to path -sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "."))) - -from app.services.browserless_service import browserless_service - -# Configure logging -logging.basicConfig(level=logging.INFO) - - -async def inspect(): - url = "https://www.lincroyable.fr/recherche-query=iphone/" - print(f"Inspecting {url}...") - - # Use host port for local debugging - os.environ["BROWSERLESS_URL"] = "ws://localhost:3012" - - await browserless_service.initialize() - try: - content, _ = await browserless_service.get_page_content(url, wait_selector="body") - with open("lincroyable_dump.html", "w", encoding="utf-8") as f: - f.write(content) - print("HTML dumped to lincroyable_dump.html") - finally: - await browserless_service.shutdown() - - -if __name__ == "__main__": - asyncio.run(inspect()) diff --git a/inspect_local_playwright.py b/inspect_local_playwright.py deleted file mode 100644 index ed34dc8..0000000 --- a/inspect_local_playwright.py +++ /dev/null @@ -1,47 +0,0 @@ -import asyncio -import logging -import os -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - - -async def inspect_local(): - url = "https://www.lincroyable.fr/recherche-query=iphone/" - print(f"Inspecting {url} using local Playwright...") - - async with async_playwright() as p: - # Launch local browser (headless=True matches server environment usually, but we can try False to see) - browser = await p.chromium.launch(headless=True) - context = await browser.new_context( - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" - ) - page = await context.new_page() - - try: - logging.info(f"Navigating to {url}") - await page.goto(url, timeout=30000) - logging.info("Navigation successful") - - content = await page.content() - with open("lincroyable_local_dump.html", "w", encoding="utf-8") as f: - f.write(content) - print("HTML dumped to lincroyable_local_dump.html") - - await page.screenshot(path="lincroyable_local.png") - print("Screenshot saved to lincroyable_local.png") - - except Exception as e: - logging.error(f"Error: {e}") - try: - await page.screenshot(path="lincroyable_error.png") - except: - pass - finally: - await browser.close() - - -if __name__ == "__main__": - asyncio.run(inspect_local()) diff --git a/inspect_sites.py b/inspect_sites.py deleted file mode 100644 index 6b02d7d..0000000 --- a/inspect_sites.py +++ /dev/null @@ -1,47 +0,0 @@ -import asyncio -import logging -from app.services.browserless_service import BrowserlessService -from app.core.search_config import SITE_CONFIGS - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -async def dump_site(site_key, query="chaise"): - config = SITE_CONFIGS.get(site_key) - if not config: - logger.error(f"Site {site_key} not found in config") - return - - search_url = config["search_url"].format(query=query) - logger.info(f"Dumping {site_key} from {search_url}") - - try: - content, _ = await BrowserlessService.get_page_content( - search_url, - wait_selector=config.get("wait_selector"), - use_proxy=config.get("requires_proxy", False) - ) - - if content: - filename = f"/app/{site_key}_dump.html" - with open(filename, "w", encoding="utf-8") as f: - f.write(content) - logger.info(f"Successfully dumped to {filename}") - else: - logger.error(f"Failed to get content for {site_key}") - - except Exception as e: - logger.error(f"Error dumping {site_key}: {e}") - -async def main(): - await BrowserlessService.initialize() - try: - await dump_site("amazon.fr") - await dump_site("stokomani.fr") - # await dump_site("lincroyable.fr") # Already have this - finally: - await BrowserlessService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/lincroyable_dump.html b/lincroyable_dump.html deleted file mode 100644 index c13465b..0000000 --- a/lincroyable_dump.html +++ /dev/null @@ -1,1876 +0,0 @@ - - - - - - - - iphone - L'Incroyable - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
-
- - - - - - - - - - - - - -
- -
-
-
-

Recherche

26 résultats pour votre recherche "iphone"

- -
-
- -
-
-
- - - -
-
- -
-
- -
-
-

coup de coeur

-

-
-
- Poêle - plusieurs gammes
+3
- -
- -
-

Poêle 'Metallic' induction

-

En aluminium
-

d28cm

-

14€99

- -
- -

Uniquement en magasin

-
-
-
-
-
-

coup de coeur

-

-
-
- Chaise Florence - plusieurs gammes
- -
- -
-

Chaise Florence

-

En polypropylène
-

44.5x42/48.5xh.79cm

-

29€99

- -
- -

Ajouter au panier

-
-
-
-
-
-

coup de coeur

-

-
-
- Bac de rangement - plusieurs gammes
- -
- -
-

Bac de rangement 'Estanca' 75L gris

-

Certifié IP67
-

64x47xh33.3cm

-

24€99

- -
- -

Ajouter au panier

-
-
-
-
-
-

coup de coeur

-

-
-
- Poêle - plusieurs gammes
+3
- -
- -
-

Poêle 'Metallic' induction

-

En aluminium
-

d30cm

-

16€99

- -
- -

Uniquement en magasin

-
-
-
- -
-
-

coup de coeur

-

-
-
- Chaise Florence - plusieurs gammes
- -
- -
-

Chaise Florence

-

En polypropylène
-

44.5x42/48.5xh.79cm

-

29€99

- -
- -

Ajouter au panier

-
-
-
-
-
-

coup de coeur

-

-
-
- Poêle - plusieurs gammes
+3
- -
- -
-

Poêle 'Metallic' induction

-

En aluminium
-

d26cm

-

13€99

- -
- -

Uniquement en magasin

-
-
-
- -
-
-

coup de coeur

-

-
-
- Chaise Florence - plusieurs gammes
- -
- -
-

Chaise Florence

-

En polypropylène
-

44.5x42/48.5xh.79cm

-

29€99

- -
- -

Ajouter au panier

-
-
-
-
-
-

coup de coeur

-

-
-
- Bac de rangement - plusieurs gammes
- -
- -
-

Bac de rangement 'Estanca' 50L gris

-

Certifié IP67
-

57.8x38.2xh31.8cm

-

17€99

- -
- -

Uniquement en magasin

-
-
-
-
-
-
-
- retour Top -
-
- -
- - - - - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/lincroyable_local.png b/lincroyable_local.png deleted file mode 100644 index e470194..0000000 Binary files a/lincroyable_local.png and /dev/null differ diff --git a/lincroyable_local_dump.html b/lincroyable_local_dump.html deleted file mode 100644 index 4aac0b5..0000000 --- a/lincroyable_local_dump.html +++ /dev/null @@ -1,1600 +0,0 @@ - - - - - - - - iphone - L'Incroyable - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
-
- - - - - - - - - - - - - -
- -
-
-
-

Recherche

26 résultats pour votre recherche "iphone"

- -
-
- -
-
-
- - - -
-
- -
-
-
-
-
- retour Top -
-
- -
- - - - - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/list_missing_sites.py b/list_missing_sites.py deleted file mode 100644 index 6644217..0000000 --- a/list_missing_sites.py +++ /dev/null @@ -1,62 +0,0 @@ -""" -Script to list all sites in database and check which ones are missing from search_config.py -""" -import sys -from pathlib import Path -sys.path.insert(0, str(Path(__file__).parent)) - -from app.database import SessionLocal -from app.models import SearchSite -from app.core.search_config import SITE_CONFIGS - -def main(): - """List all sites in DB and identify missing configurations""" - db = SessionLocal() - - try: - # Get all sites from database - sites = db.query(SearchSite).order_by(SearchSite.name).all() - - print(f"\n{'='*80}") - print(f"Sites in Database: {len(sites)}") - print(f"Sites in SITE_CONFIGS: {len(SITE_CONFIGS)}") - print(f"{'='*80}\n") - - # Check each DB site - missing = [] - configured = [] - - for site in sites: - domain_clean = site.domain.replace("www.", "").lower() - - # Check if configured - is_configured = False - for key in SITE_CONFIGS.keys(): - if key in domain_clean or domain_clean in key: - configured.append((site.id, site.name, site.domain, key)) - is_configured = True - break - - if not is_configured: - missing.append((site.id, site.name, site.domain)) - - # Display results - if configured: - print("✅ CONFIGURED SITES:") - for sid, name, domain, key in configured: - print(f" [{sid:2d}] {name:20s} ({domain:25s}) → {key}") - - if missing: - print(f"\n❌ MISSING {len(missing)} SITES:") - for sid, name, domain in missing: - print(f" [{sid:2d}] {name:20s} ({domain})") - else: - print("\n✅ All sites are configured!") - - print(f"\n{'='*80}\n") - - finally: - db.close() - -if __name__ == "__main__": - main() diff --git a/reproduce_gifi_scrape.py b/reproduce_gifi_scrape.py deleted file mode 100644 index 360129e..0000000 --- a/reproduce_gifi_scrape.py +++ /dev/null @@ -1,67 +0,0 @@ -import asyncio -import logging -import sys -from playwright.async_api import async_playwright - -# Configure logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -URL = "https://www.gifi.fr/meuble-et-deco/decoration/bougie-et-senteur/diffuseur-et-senteur/encens-nag-champa-15-g/000000000000540823.html" - -async def reproduce_scrape(): - logger.info("Starting reproduction script...") - async with async_playwright() as p: - # Launch browser (headless=True by default which is what we want for reproduction usually) - # But for debugging blocking, sometimes headless=False helps. Let's start with True (default) - browser = await p.chromium.launch(headless=True) - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" - ) - - page = await context.new_page() - - logger.info(f"Navigating to {URL}...") - try: - await page.goto(URL, wait_until="domcontentloaded", timeout=60000) - logger.info("Page loaded.") - - # Wait a bit for dynamic content - await page.wait_for_timeout(5000) - - # Extract title - title = await page.title() - logger.info(f"Page Title: {title}") - - # Extract body text - content = await page.content() - body_text = await page.inner_text("body") - - logger.info(f"Content Length: {len(content)}") - logger.info(f"Body Text Length: {len(body_text)}") - - # Check for price - if "€" in body_text: - logger.info("Found '€' in body text.") - else: - logger.warning("'€' NOT found in body text.") - - # specific check for likely price - import re - prices = re.findall(r'\d+[,\.]\d{2}\s*€', body_text) - logger.info(f"Prices found in text: {prices}") - - # Save content for review - with open("gifi_reproduction.html", "w", encoding="utf-8") as f: - f.write(content) - logger.info("Saved gifi_reproduction.html") - - except Exception as e: - logger.error(f"Error during navigation/scraping: {e}") - - finally: - await browser.close() - -if __name__ == "__main__": - asyncio.run(reproduce_scrape()) diff --git a/reset_admin.py b/reset_admin.py deleted file mode 100644 index c99f940..0000000 --- a/reset_admin.py +++ /dev/null @@ -1,31 +0,0 @@ -from app.database import SessionLocal -from app.services import auth_service -from app import models - -def reset_admin(): - db = SessionLocal() - try: - user = auth_service.get_user_by_username(db, "admin") - if user: - print("Found admin user. Resetting password...") - auth_service.update_password(db, user, "admin") - print("Password reset to 'admin'") - - # Verify - print("Verifying login...") - auth_user = auth_service.authenticate_user(db, "admin", "admin") - if auth_user: - print("SUCCESS: Login verified!") - else: - print("ERROR: Login failed after reset!") - else: - print("Admin user not found. Creating...") - auth_service.create_user(db, "admin", "admin", is_admin=True) - print("Admin user created with password 'admin'") - except Exception as e: - print(f"Error: {e}") - finally: - db.close() - -if __name__ == "__main__": - reset_admin() diff --git a/run_cleanup.py b/run_cleanup.py deleted file mode 100644 index 9aaed22..0000000 --- a/run_cleanup.py +++ /dev/null @@ -1,53 +0,0 @@ -import sys -import os -import logging -from sqlalchemy.orm import Session - -# Add project root to path -sys.path.append(os.getcwd()) - -from app.database import SessionLocal, engine -from app.models import Catalogue - -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - - -def cleanup_catalogs(): - db = SessionLocal() - try: - logger.info("Starting cleanup...") - deleted_count = 0 - - # 1. Delete catalogs with 0 pages - bad_catalogs = db.query(Catalogue).filter(Catalogue.nombre_pages == 0).all() - for cat in bad_catalogs: - logger.info(f"Deleting empty catalog: {cat.titre}") - db.delete(cat) - deleted_count += 1 - - # 2. Delete catalogs with iconic/bad images - all_catalogs = db.query(Catalogue).all() - for cat in all_catalogs: - if not cat.image_couverture_url: - continue - - if any( - x in cat.image_couverture_url.lower() - for x in ["icon", "logo", "loader", "facebook", "twitter", "assets/img"] - ): - logger.info(f"Deleting catalog with bad image: {cat.titre} ({cat.image_couverture_url})") - db.delete(cat) - deleted_count += 1 - - db.commit() - logger.info(f"Cleanup complete. Deleted {deleted_count} catalogs.") - - except Exception as e: - logger.error(f"Error during cleanup: {e}") - finally: - db.close() - - -if __name__ == "__main__": - cleanup_catalogs() diff --git a/stokomani.fr_dump.html b/stokomani.fr_dump.html deleted file mode 100644 index 837301c..0000000 --- a/stokomani.fr_dump.html +++ /dev/null @@ -1,18100 +0,0 @@ - - - - - - - - - Recherche : 31 résultats trouvés pour « chaise » – Stokomani - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Passer au contenu -
- -
- -
-
- - -
-
-
- - -
-
-
- -
- - -
-
-
- -
-
- -
-
-
-
- -
-
-
- - - -
- - -
-
- - - - - - - -
-
- - - - -
- - -
-
- -
-
- - - - - - - - - - - - - - - -
- - -
- - - - - - - -
- - -
- -
\ No newline at end of file diff --git a/stokomani.fr_failed_verification.html b/stokomani.fr_failed_verification.html deleted file mode 100644 index c7dbbe9..0000000 --- a/stokomani.fr_failed_verification.html +++ /dev/null @@ -1,18100 +0,0 @@ - - - - - - - - - Recherche : 31 résultats trouvés pour « chaise » – Stokomani - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Passer au contenu -
- -
- -
-
- - -
-
-
- - -
-
-
- -
- - -
-
-
- -
-
- -
-
-
-
- -
-
-
- - - -
- - -
-
- - - - - - - -
-
- - - - -
- - -
-
- -
-
- - - - - - - - - - - - - - - -
- - -
- - - - - - - -
- - -
- -
\ No newline at end of file diff --git a/stokomani_raw.html b/stokomani_raw.html deleted file mode 100644 index 596b11f..0000000 --- a/stokomani_raw.html +++ /dev/null @@ -1,32156 +0,0 @@ - - - - - - - - - - - - Recherche : 31 résultats trouvés pour « chaise » – Stokomani - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Passer au contenu -
- -
- -
-
- - -
-
-
- - -
-
-
- -
- - -
-
-
- -
-
- -
-
-
-
- -
-
-
- - - -
- - -
-
- - - - - - - -
-
- - - - -
- - -
-
- -
-
- - - - - - - - - - - - - - - -
- - - - - - - -
- - -
- - -
- diff --git a/task.md b/task.md index d6628b5..a2f60bb 100644 --- a/task.md +++ b/task.md @@ -1,29 +1,21 @@ -# Investigation et Correction du Statut "Produit Retiré" +# Nettoyage et Commit du Projet ## Contexte -Le système de suivi des prix marquait incorrectement certains produits comme "retirés" alors qu'ils étaient toujours disponibles, souvent à cause de blocages (bot detection) ou de changements mineurs de titre. +Après avoir corrigé les problèmes de disponibilité d'Action.com et Amazon, il est nécessaire de nettoyer le répertoire racine des scripts de diagnostic et fichiers temporaires avant de committer les changements. ## Focus Actuel -Finalisation et vérification. +Identification et suppression des fichiers inutiles, suivi du commit et push. ## Master Plan -- [x] Analyser la logique de comparaison de titres dans `app/services/scheduler_service.py` -- [x] Vérifier si les scrapers extraient correctement les titres lors des mises à jour -- [x] Identifier les cas limites (edge cases) où la similarité de titre échoue -- [x] Corriger la logique pour éviter les faux positifs d'indisponibilité -- [x] Vérifier la correction avec un exemple concret (test_logic.py) -- [x] Implémenter le reset automatique de disponibilité si un prix est trouvé +- [ ] Identifier les fichiers non essentiels à l'application +- [ ] Supprimer les fichiers de diagnostic et scripts temporaires +- [ ] Vérifier que l'application fonctionne toujours (build/syntaxe) +- [ ] Committer les changements vers le dépôt Git +- [ ] Pusher les changements ## Log de Progression -- [x] Investigation terminée : identification des faux positifs dus aux titres de blocage (Cloudflare, etc.) et aux placeholders ("Loading"). -- [x] Logique de normalisation renforcée. -- [x] Détection des bots ajoutée pour tous les sites. -- [x] Auto-reset de `is_available` implémenté dans `_update_db_result`. -- [x] Correction déployée dans `scheduler_service.py`. -- [x] Affinage de la détection de bot pour Stokomani/L'Incroyable (détection conditionnelle au titre). -- [x] Support des versions "V2" et matching par mots pour les noms courts sur Amazon. -- [x] Sécurisation du flux Action.com (check d'indisponibilité déplacé après le titre). +- [ ] Planification du nettoyage commencée. diff --git a/test_amazon_scraper.py b/test_amazon_scraper.py deleted file mode 100755 index 9a008b1..0000000 --- a/test_amazon_scraper.py +++ /dev/null @@ -1,187 +0,0 @@ -#!/usr/bin/env python3 -""" -Script de test pour le scraper Amazon France -Teste le système anti-détection et l'extraction des produits -""" - -import asyncio -import logging -import sys -from pathlib import Path - -# Ajouter le répertoire app au path -sys.path.insert(0, str(Path(__file__).parent)) - -from app.services.amazon_scraper_service import amazon_scraper_service, AmazonScraperService -from app.core.search_config import AMAZON_PROXY_LIST_RAW, USER_AGENT_DATA - -# Define missing variable for test compatibility -AMAZON_USER_AGENTS = USER_AGENT_DATA -USER_AGENT_POOL = USER_AGENT_DATA - - -# Configuration du logging -logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s") - -logger = logging.getLogger(__name__) - - -# Monkey patch _connect_browser to use local launch for testing -async def _connect_browser_local(p): - logger.info("Launching local browser (headless)...") - return await p.chromium.launch(headless=True) - - -AmazonScraperService._connect_browser = _connect_browser_local - - -async def test_basic_search(): - """Test basique de recherche""" - logger.info("=" * 80) - logger.info("TEST 1: Recherche basique - 'aspirateur'") - logger.info("=" * 80) - - products = await amazon_scraper_service.scrape_search("aspirateur", max_results=5) - - if not products: - logger.error("❌ Aucun produit trouvé - possibilité de détection ou problème réseau") - return False - - logger.info(f"✅ {len(products)} produits trouvés") - - for idx, product in enumerate(products, 1): - logger.info(f"\n{idx}. {product.title[:60]}...") - logger.info( - f" 💰 Prix: {product.price}€" + (f" (était {product.original_price}€)" if product.original_price else "") - ) - logger.info(f" ⭐ Note: {product.rating}/5" if product.rating else " ⭐ Pas de note") - logger.info(f" 📦 {'En stock' if product.in_stock else 'Indisponible'}") - logger.info(f" {'🚚 Prime' if product.prime else '📮 Standard'}") - logger.info(f" {'📢 Sponsorisé' if product.sponsored else '🔍 Organique'}") - - return True - - -async def test_multiple_queries(): - """Test avec plusieurs requêtes différentes""" - logger.info("\n" + "=" * 80) - logger.info("TEST 2: Requêtes multiples") - logger.info("=" * 80) - - queries = ["clavier", "souris", "casque"] - results = {} - - for query in queries: - logger.info(f"\n🔍 Recherche: '{query}'") - products = await amazon_scraper_service.scrape_search(query, max_results=3) - results[query] = len(products) - logger.info(f" ✅ {len(products)} produits trouvés") - - # Délai entre requêtes pour respecter les bonnes pratiques - await asyncio.sleep(3) - - logger.info("\n📊 Résumé:") - for query, count in results.items(): - logger.info(f" • {query}: {count} produits") - - total = sum(results.values()) - if total > 0: - logger.info(f"\n✅ Total: {total} produits extraits") - return True - else: - logger.error("\n❌ Aucun produit extrait - problème possible") - return False - - -async def test_anti_detection(): - """Test du système anti-détection""" - logger.info("\n" + "=" * 80) - logger.info("TEST 3: Vérification anti-détection") - logger.info("=" * 80) - - # Updated to just check if we can run - logger.info("Skipping specific proxy/agent checks for this service as it handles them internally") - - logger.info(f"✓ {len(AMAZON_PROXY_LIST_RAW)} proxies disponibles") - logger.info(f"✓ {len(USER_AGENT_POOL)} User-Agents standards") - logger.info(f"✓ {len(AMAZON_USER_AGENTS)} User-Agents Amazon spécifiques") - - # Test proxy - # Test proxy - import random - - proxy = random.choice(AMAZON_PROXY_LIST_RAW) if AMAZON_PROXY_LIST_RAW else None - if proxy: - # Extract just the IP for logging (hide credentials) - proxy_parts = proxy.split("@") - proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy - logger.info(f"✓ Proxy test: {proxy_server}") - else: - logger.warning("⚠️ Pas de proxy configuré") - - # Test d'une recherche simple - logger.info("\n🧪 Test de recherche avec anti-détection...") - products = await amazon_scraper_service.scrape_search("livre", max_results=3) - - if products: - logger.info(f"✅ Anti-détection fonctionnel - {len(products)} produits extraits") - return True - else: - logger.error("❌ Échec - possibilité de blocage") - return False - - -async def run_all_tests(): - """Lance tous les tests""" - logger.info("\n" + "=" * 80) - logger.info("🚀 DÉMARRAGE DES TESTS DU SCRAPER AMAZON FRANCE") - logger.info("=" * 80) - - tests = [ - ("Recherche basique", test_basic_search), - ("Requêtes multiples", test_multiple_queries), - ("Anti-détection", test_anti_detection), - ] - - results = {} - - for test_name, test_func in tests: - try: - logger.info(f"\n▶️ Exécution: {test_name}") - success = await test_func() - results[test_name] = "✅ PASS" if success else "❌ FAIL" - except Exception as e: - logger.error(f"❌ Erreur dans {test_name}: {e}", exc_info=True) - results[test_name] = "❌ ERROR" - - # Résumé final - logger.info("\n" + "=" * 80) - logger.info("📊 RÉSUMÉ DES TESTS") - logger.info("=" * 80) - - for test_name, result in results.items(): - logger.info(f"{result} - {test_name}") - - passed = sum(1 for r in results.values() if "PASS" in r) - total = len(results) - - logger.info(f"\n🎯 Score: {passed}/{total} tests réussis") - - if passed == total: - logger.info("✅ TOUS LES TESTS ONT RÉUSSI!") - return True - else: - logger.warning("⚠️ Certains tests ont échoué") - return False - - -if __name__ == "__main__": - try: - success = asyncio.run(run_all_tests()) - sys.exit(0 if success else 1) - except KeyboardInterrupt: - logger.info("\n⏸️ Tests interrompus par l'utilisateur") - sys.exit(130) - except Exception as e: - logger.error(f"❌ Erreur fatale: {e}", exc_info=True) - sys.exit(1) diff --git a/test_carrefour_api.py b/test_carrefour_api.py deleted file mode 100644 index 9b53484..0000000 --- a/test_carrefour_api.py +++ /dev/null @@ -1,42 +0,0 @@ -""" -Test Carrefour via API endpoint (production-like) -""" -import requests -import json - -url = "http://localhost:8555/api/search" -params = { - "q": "chaise", - "sites": "7", # Carrefour site ID - "max_results": 10 -} - -print(f"Testing: {url}") -print(f"Params: {params}\n") - -response = requests.get(url, params=params, stream=True) - -print(f"Status: {response.status_code}") -print(f"Headers: {dict(response.headers)}\n") - -count = 0 -for line in response.iter_lines(): - if line: - try: - # Each line should be JSON - data = json.loads(line.decode('utf-8')) - count += 1 - - title = data.get('title', 'N/A')[:60] - price = data.get('price', 'N/A') - image = "✅" if data.get('image_url') else "❌" - - print(f"{count}. [{image}] {title} - {price}€") - - if count >= 10: - break - except json.JSONDecodeError as e: - print(f"JSON Error: {e}") - print(f"Line: {line[:100]}") - -print(f"\nTotal: {count} results") diff --git a/test_carrefour_prices.py b/test_carrefour_prices.py deleted file mode 100644 index 9347cf8..0000000 --- a/test_carrefour_prices.py +++ /dev/null @@ -1,40 +0,0 @@ -""" -Test if current Carrefour price extraction works -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Carrefour for 'chaise'...\n") - results = [] - count = 0 - async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"): - results.append(result) - count += 1 - price_status = f"{result.price}€" if result.price else "N/A" - print(f"{count}. {result.title[:55]} - {price_status}") - if count >= 10: - break - - print(f"\n==> Got {len(results)} results") - - # Count prices - with_prices = sum(1 for r in results if r.price) - print(f"Products with prices: {with_prices}/{len(results)}") - - if with_prices == len(results): - print("✅ SUCCESS: All products have prices!") - else: - print(f"⚠️ WARNING: {len(results) - with_prices} products missing prices") - - print("\nShutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_carrefour_search.py b/test_carrefour_search.py deleted file mode 100644 index def6b68..0000000 --- a/test_carrefour_search.py +++ /dev/null @@ -1,26 +0,0 @@ -""" -Test script to trigger Carrefour search and generate HTML dump -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Carrefour for 'chaise'...") - results = [] - async for result in ImprovedSearchService.search_site_generator("carrefour.fr", "chaise"): - results.append(result) - print(f"Found: {result.title}") - - print(f"\nTotal results: {len(results)}") - - print("Shutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_centrakor_filter.py b/test_centrakor_filter.py deleted file mode 100644 index cd4ceb5..0000000 --- a/test_centrakor_filter.py +++ /dev/null @@ -1,69 +0,0 @@ -""" -Test detailed logging for Centrakor image extraction -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from playwright.async_api import async_playwright -from bs4 import BeautifulSoup -import re - -async def main(): - playwright = await async_playwright().start() - browser = await playwright.chromium.connect_over_cdp("ws://browserless:3000") - - context = await browser.new_context( - viewport={"width": 1920, "height": 1080}, - user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" - ) - - page = await context.new_page() - - await page.goto("https://www.centrakor.com/search/chaise", wait_until="networkidle") - content = await page.content() - - soup = BeautifulSoup(content, "html.parser") - products = soup.select("div.product-item") - - print(f"Testing first product:\n") - - first = products[0] - - # Get all images - img_els = first.select("img.responsive-image__actual") - print(f"Found {len(img_els)} images with selector") - - for i, img_el in enumerate(img_els): - print(f"\n=== Image {i+1} ===") - candidate_url = img_el.get("src") - print(f"URL: {candidate_url}") - - # Test filters - if any(keyword in candidate_url.lower() for keyword in ['picto', 'icon', 'logo', 'badge']): - print(" ❌ Filtered: Contains picto/icon/logo/badge keyword") - continue - - width_match = re.search(r'width=(\d+)', candidate_url) - height_match = re.search(r'height=(\d+)', candidate_url) - print(f" Width match: {width_match.group(1) if width_match else None}") - print(f" Height match: {height_match.group(1) if height_match else None}") - - if width_match and height_match: - width = int(width_match.group(1)) - height = int(height_match.group(1)) - print(f" Dimensions: {width}x{height}") - if width < 100 and height < 100: - print(f" ❌ Filtered: Too small ({width}x{height})") - continue - else: - print(f" ✅ PASS: Large enough ({width}x{height})") - else: - print(" ✅ PASS: No dimensions in URL") - - await context.close() - await browser.close() - await playwright.stop() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_centrakor_images.py b/test_centrakor_images.py deleted file mode 100644 index e8eb106..0000000 --- a/test_centrakor_images.py +++ /dev/null @@ -1,42 +0,0 @@ -""" -Test Centrakor image extraction -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Centrakor for 'chaise'...\n") - results = [] - count = 0 - async for result in ImprovedSearchService.search_site_generator("centrakor.com", "chaise"): - results.append(result) - count += 1 - has_image = "✅" if result.image_url else "❌" - print(f"{count}. {has_image} {result.title[:55]} - {result.price}€") - if result.image_url: - print(f" Image: {result.image_url[:70]}...") - if count >= 10: - break - - print(f"\n==> Got {len(results)} results") - - # Count images - with_images = sum(1 for r in results if r.image_url) - print(f"Products with images: {with_images}/{len(results)}") - - if with_images == len(results): - print("✅ SUCCESS: All products have images!") - else: - print(f"⚠️ WARNING: {len(results) - with_images} products missing images") - - print("\nShutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_gifi_final.py b/test_gifi_final.py deleted file mode 100644 index a969b43..0000000 --- a/test_gifi_final.py +++ /dev/null @@ -1,34 +0,0 @@ -""" -Final test of Gifi with new selectors -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Gifi for 'chaise'...") - results = [] - async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"): - results.append(result) - print(f"✓ {result.title[:50]} - {result.price}€") - - print(f"\n==> Total: {len(results)} results") - - if results: - print("\nFirst 3 products:") - for i, r in enumerate(results[:3]): - print(f"{i+1}. Title: {r.title}") - print(f" Price: {r.price}€") - print(f" URL: {r.url[:80]}...") - print() - - print("Shutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_gifi_prices.py b/test_gifi_prices.py deleted file mode 100644 index ec8232c..0000000 --- a/test_gifi_prices.py +++ /dev/null @@ -1,38 +0,0 @@ -""" -Test Gifi price extraction -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Gifi for 'chaise'...") - results = [] - count = 0 - async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"): - results.append(result) - count += 1 - print(f"{count}. {result.title[:60]} - Price: {result.price}€") - if count >= 5: # Only test first 5 - break - - print(f"\n==> Got {len(results)} results") - - # Check if all prices are the same - prices = [r.price for r in results if r.price] - if prices: - unique_prices = set(prices) - print(f"Unique prices: {unique_prices}") - if len(unique_prices) == 1: - print("⚠️ WARNING: All prices are the same!") - - print("Shutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_gifi_prices_final.py b/test_gifi_prices_final.py deleted file mode 100644 index e5be347..0000000 --- a/test_gifi_prices_final.py +++ /dev/null @@ -1,41 +0,0 @@ -""" -Test fixed Gifi price extraction -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Gifi for 'chaise'...\n") - results = [] - count = 0 - async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"): - results.append(result) - count += 1 - print(f"{count}. {result.title[:60]} - {result.price}€") - if count >= 10: - break - - print(f"\n==> Got {len(results)} results") - - # Check price diversity - prices = [r.price for r in results if r.price] - if prices: - unique_prices = set(prices) - print(f"Unique prices: {sorted(unique_prices)}") - print(f"Price range: {min(prices)}€ - {max(prices)}€") - if len(unique_prices) > 1: - print("✅ SUCCESS: Multiple different prices found!") - else: - print("⚠️ WARNING: All prices are the same") - - print("\nShutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_gifi_search.py b/test_gifi_search.py deleted file mode 100644 index f8e8a37..0000000 --- a/test_gifi_search.py +++ /dev/null @@ -1,26 +0,0 @@ -""" -Test script to analyze Gifi HTML structure -""" -import asyncio -import sys -sys.path.insert(0, '/app') - -from app.services.improved_search_service import ImprovedSearchService - -async def main(): - print("Initializing browser...") - await ImprovedSearchService.initialize() - - print("Searching Gifi for 'chaise'...") - results = [] - async for result in ImprovedSearchService.search_site_generator("gifi.fr", "chaise"): - results.append(result) - print(f"Found: {result.title} - Price: {result.price}") - - print(f"\nTotal results: {len(results)}") - - print("Shutting down...") - await ImprovedSearchService.shutdown() - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/test_soup.py b/test_soup.py deleted file mode 100644 index 3df7bef..0000000 --- a/test_soup.py +++ /dev/null @@ -1,29 +0,0 @@ -from bs4 import BeautifulSoup -import sys - -try: - with open("/app/debug_dumps/stokomani.fr_failed_verification.html", "r", encoding="utf-8") as f: - content = f.read() - - print(f"Read {len(content)} bytes") - - soup = BeautifulSoup(content, "html.parser") - print("Soup created") - - selector = "div.product-card" - items = soup.select(selector) - print(f"Found {len(items)} items with selector '{selector}'") - - if items: - item = items[0] - print("First item classes:", item.get("class")) - - title_selector = "h3.product-card__title a" - title_el = item.select_one(title_selector) - if title_el: - print("Title found:", title_el.get_text(strip=True)) - else: - print(f"Title NOT found with '{title_selector}'") - -except Exception as e: - print(f"Error: {e}") diff --git a/verify_catalog_fix.py b/verify_catalog_fix.py deleted file mode 100644 index abc2f5e..0000000 --- a/verify_catalog_fix.py +++ /dev/null @@ -1,43 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Identify workspace root -sys.path.append(os.getcwd()) - -from app.services.cataloguemate_scraper import scrape_catalog_pages - -# Setup logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - - -async def verify_scraper(): - # Gifi catalog URL from the browser session - url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/" - - print(f"Verifying scraper on: {url}") - - pages = await scrape_catalog_pages(url) - - print(f"Found {len(pages)} pages.") - - if not pages: - print("FAIL: No pages found.") - return - - # check first page image - first_img = pages[0]["image_url"] - print(f"Page 1 Image: {first_img}") - - if "thumbor" in first_img or "leafletscdns" in first_img: - print("SUCCESS: Image is a Thumbor/Leaflet URL.") - elif "icon" in first_img or "logo" in first_img: - print("FAIL: Image appears to be an icon/logo.") - else: - print(f"WARNING: Image URL is: {first_img}") - - -if __name__ == "__main__": - asyncio.run(verify_scraper()) diff --git a/verify_search_price.py b/verify_search_price.py deleted file mode 100644 index 00bd5ca..0000000 --- a/verify_search_price.py +++ /dev/null @@ -1,45 +0,0 @@ -import asyncio -import logging -import sys -import os - -# Identify workspace root -sys.path.append(os.getcwd()) - -# Mock Env Vars -os.environ["DATABASE_URL"] = "postgresql://user:password@localhost:5432/pricewatch" -os.environ["BROWSERLESS_URL"] = "ws://localhost:3012" # Ignored by patch, but good for completeness - -from app.services.improved_search_service import ImprovedSearchService -from playwright.async_api import async_playwright - -# Setup logging -logging.basicConfig(level=logging.DEBUG) # DEBUG level to see price extraction logic -logger = logging.getLogger(__name__) - - -# Monkey patch _connect_browser to use local launch -async def _connect_browser_local(p): - logger.info("Launching local browser (headless)...") - return await p.chromium.launch(headless=True) - - -ImprovedSearchService._connect_browser = _connect_browser_local - - -async def test_search(): - await ImprovedSearchService.initialize() - - query = "nintendo switch" - target_site = "gifi.fr" - - print(f"Searching for: {query} on {target_site}") - - async for result in ImprovedSearchService.search_site_generator(target_site, query): - print(f"[{result.source}] {result.title}\n -> Price: {result.price}€\n -> URL: {result.url}") - - await ImprovedSearchService.shutdown() - - -if __name__ == "__main__": - asyncio.run(test_search()) diff --git a/verify_standalone.py b/verify_standalone.py deleted file mode 100644 index 072636d..0000000 --- a/verify_standalone.py +++ /dev/null @@ -1,108 +0,0 @@ -import asyncio -import logging -import re -import sys - -# Mock logger -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - - -# Fallback implementation of _fetch_with_fallback for standalone test -async def _fetch_with_fallback(url): - # We need to install httpx for this to work - try: - import httpx - - headers = { - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" - } - async with httpx.AsyncClient(verify=False, timeout=30.0) as client: - response = await client.get(url, headers=headers) - return response.text - except ImportError: - print("Please pip install httpx strict") - return "" - - -async def scrape_catalog_pages_standalone(catalog_url: str): - from bs4 import BeautifulSoup - - print(f"Scraping: {catalog_url}") - html_content = await _fetch_with_fallback(catalog_url) - - if not html_content: - print("Failed to fetch content") - return [] - - soup = BeautifulSoup(html_content, "html.parser") - - # --- COPIED LOGIC FROM cataloguemate_scraper.py --- - main_image_url = None - max_area = 0 - - # Strategy 1: Look for specific container/class identified in browser inspection - candidates = soup.select(".letaky-grid-preview img") - - # Strategy 2: Fallback to all images if specific container not found - if not candidates: - candidates = soup.find_all("img") - - print(f"Found {len(candidates)} candidates") - - for img in candidates: - # Check multiple attributes for the real image URL - src = img.get("src") or img.get("data-src") or img.get("data-original") - - if not src: - continue - - # Skip common UI elements - refined list - if any( - x in src.lower() - for x in [ - "logo", - "icon", - "facebook", - "twitter", - "instagram", - "loader", - "spinner", - "market", - "googleplay", - "appstore", - ] - ): - continue - - # Strong Signal: URL contains 'thumbor' or 'leafletscdns' (host for catalog images) - is_thumbor = "thumbor" in src.lower() or "leafletscdns" in src.lower() - - # Calculate area if dimensions exist - width = img.get("width") - height = img.get("height") - area = 0 - if width and height: - try: - area = int(width) * int(height) - except: - pass - - if is_thumbor: - if area > max_area or (area == 0 and max_area == 0): - max_area = area - main_image_url = src - print(f"Match (Thumbor): {src}") - elif area > 50000: - if area > max_area: - max_area = area - main_image_url = src - print(f"Match (Size): {src}") - - return main_image_url - - -if __name__ == "__main__": - url = "https://www.cataloguemate.fr/gifi/catalogue-du-mardi-16122025-61964/" - result = asyncio.run(scrape_catalog_pages_standalone(url)) - print(f"FINAL RESULT: {result}")