From cd18b47bc0c823c75f80d583f720df57c8a1257f Mon Sep 17 00:00:00 2001 From: Michael SCHAL Date: Wed, 24 Dec 2025 21:13:22 +0100 Subject: [PATCH] feat: Add core search configurations for various e-commerce sites, including proxy and user agent management, and introduce initial Amazon scraper service and proxy utilities. --- app/core/search_config.py | 42 ++++----- app/services/amazon_scraper_service.py | 9 +- check_proxies.py | 52 +++++++++++ filter_proxies.py | 103 +++++++++++++++++++++ find_working_proxies.py | 118 +++++++++++++++++++++++++ working_proxies.txt | 10 +++ 6 files changed, 309 insertions(+), 25 deletions(-) create mode 100644 check_proxies.py create mode 100644 filter_proxies.py create mode 100644 find_working_proxies.py create mode 100644 working_proxies.txt diff --git a/app/core/search_config.py b/app/core/search_config.py index 1be78dc..fda8d42 100644 --- a/app/core/search_config.py +++ b/app/core/search_config.py @@ -10,18 +10,10 @@ BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000") DEBUG_DUMPS_DIR = "/app/debug_dumps" # === PROXY CONFIGURATION === -AMAZON_PROXY_LIST_RAW = [ - "142.111.48.253:7030:jasuwwjr:elbsx170nmnl", - "31.59.20.176:6754:jasuwwjr:elbsx170nmnl", - "23.95.150.145:6114:jasuwwjr:elbsx170nmnl", - "198.23.239.134:6540:jasuwwjr:elbsx170nmnl", - "107.172.163.27:6543:jasuwwjr:elbsx170nmnl", - "198.105.121.200:6462:jasuwwjr:elbsx170nmnl", - "64.137.96.74:6641:jasuwwjr:elbsx170nmnl", - "84.247.60.125:6095:jasuwwjr:elbsx170nmnl", - "216.10.27.159:6837:jasuwwjr:elbsx170nmnl", - "142.111.67.146:5611:jasuwwjr:elbsx170nmnl", -] +# NOTE: All free proxies tested are non-functional. Direct connections will be used. +# Add working proxies here when available. +AMAZON_PROXY_LIST_RAW = [] + def get_amazon_proxies() -> list[dict]: """Convert raw proxy list to Playwright format""" @@ -29,50 +21,52 @@ def get_amazon_proxies() -> list[dict]: for proxy in AMAZON_PROXY_LIST_RAW: parts = proxy.split(":") if len(parts) == 4: - proxies.append({ - "server": f"http://{parts[0]}:{parts[1]}", - "username": parts[2], - "password": parts[3] - }) + proxies.append({"server": f"http://{parts[0]}:{parts[1]}", "username": parts[2], "password": parts[3]}) + elif len(parts) == 2: + proxies.append({"server": f"http://{parts[0]}:{parts[1]}"}) return proxies + # === USER AGENTS & STEALTH === USER_AGENT_DATA = [ { "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", "ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"', - "platform": '"Windows"' + "platform": '"Windows"', }, { "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36", "ch": '"Google Chrome";v="130", "Chromium";v="130", "Not_A Brand";v="99"', - "platform": '"Windows"' + "platform": '"Windows"', }, { "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", "ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"', - "platform": '"macOS"' + "platform": '"macOS"', }, { "ua": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", "ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"', - "platform": '"Linux"' + "platform": '"Linux"', }, { "ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0", "ch": '"Microsoft Edge";v="131", "Chromium";v="131", "Not_A Brand";v="24"', - "platform": '"Windows"' - } + "platform": '"Windows"', + }, ] + def get_random_stealth_config() -> dict: """Returns a random User-Agent and its corresponding Client Hints""" return random.choice(USER_AGENT_DATA) + def get_random_user_agent() -> str: """Legacy helper for backward compatibility""" return get_random_stealth_config()["ua"] + # === SITE CONFIGURATIONS === SITE_CONFIGS = { # === MAGASINS DISCOUNT === @@ -240,7 +234,7 @@ COOKIE_ACCEPT_SELECTORS = [ ".modal-close", ".close-modal", ".popup-close", - "button[aria-label='Close']", + "button[aria-label='Close']", "button[aria-label='Fermer']", ".js-modal-close", "div[class*='popup'] button[class*='close']", diff --git a/app/services/amazon_scraper_service.py b/app/services/amazon_scraper_service.py index beddc52..5b398aa 100644 --- a/app/services/amazon_scraper_service.py +++ b/app/services/amazon_scraper_service.py @@ -352,7 +352,14 @@ class AmazonScraperService: logger.warning(f"⏳ Attempt {attempt} failed, retrying in {delay}s with new identity...") await asyncio.sleep(delay) - logger.error(f"❌ All {MAX_RETRY_ATTEMPTS} attempts failed for query: {query}") + if not products and proxies: + logger.warning("⚠️ All proxy attempts failed. Attempting fallback to direct connection...") + products = await cls._try_scrape(query, max_results, None, MAX_RETRY_ATTEMPTS + 1) + if products: + logger.info("✅ Successfully extracted products using direct connection fallback") + return products + + logger.error(f"❌ All attempts failed for query: {query}") return [] @classmethod diff --git a/check_proxies.py b/check_proxies.py new file mode 100644 index 0000000..43ce566 --- /dev/null +++ b/check_proxies.py @@ -0,0 +1,52 @@ +import asyncio +import os +import sys + +# Add app to path +sys.path.insert(0, os.getcwd()) + +from playwright.async_api import async_playwright +from app.core.search_config import get_amazon_proxies + + +async def check_proxy(proxy, semaphore): + async with semaphore: + proxy_url = proxy["server"] + print(f"Testing {proxy_url}...") + + async with async_playwright() as p: + # Connect to browserless or launch local + # Using launch local for simpler testing without ws dependency if possible + # But the app uses browserless. Let's try launch first. + try: + browser = await p.chromium.launch(headless=True, proxy=proxy) + page = await browser.new_page() + try: + # amazon.fr might block, use httpbin for connectivity check + await page.goto("http://httpbin.org/ip", timeout=15000) + content = await page.content() + print(f"✅ {proxy_url}: Success") + await browser.close() + return True + except Exception as e: + print(f"❌ {proxy_url}: Failed - {str(e)[:100]}") + await browser.close() + return False + except Exception as e: + print(f"❌ {proxy_url}: Launch Failed - {str(e)[:100]}") + return False + + +async def main(): + proxies = get_amazon_proxies() + print(f"Checking {len(proxies)} proxies...") + + semaphore = asyncio.Semaphore(3) # Limit concurrency + results = await asyncio.gather(*[check_proxy(p, semaphore) for p in proxies]) + + working = sum(results) + print(f"\nSummary: {working}/{len(proxies)} working.") + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/filter_proxies.py b/filter_proxies.py new file mode 100644 index 0000000..c63526d --- /dev/null +++ b/filter_proxies.py @@ -0,0 +1,103 @@ +import json +import urllib.request +import urllib.error + +proxies_raw = """ +46.161.6.165:8080 +78.47.219.204:3128 +134.209.29.120:8080 +161.35.70.249:80 +134.209.29.120:80 +52.188.28.218:3128 +209.97.150.167:3128 +62.60.151.128:80 +68.235.35.171:3128 +209.97.150.167:80 +159.203.61.169:8080 +209.97.150.167:8080 +195.158.8.123:3128 +208.87.243.199:7878 +144.76.42.215:8118 +216.229.112.25:8080 +159.203.61.169:80 +103.3.246.71:3128 +138.68.60.8:80 +139.59.1.14:80 +8.243.68.11:8080 +41.223.119.156:3128 +34.96.238.40:8080 +59.6.25.118:3128 +129.150.39.251:8000 +162.240.154.26:3128 +35.152.252.253:8080 +144.125.164.158:8081 +47.81.14.7:3129 +144.125.164.222:8080 +175.99.220.171:80 +8.219.97.248:80 +144.125.164.158:8080 +164.68.110.241:8091 +144.125.164.222:8081 +140.238.184.182:3128 +139.59.1.14:3128 +8.212.160.196:8080 +164.68.110.241:9992 +173.212.246.157:3128 +47.236.130.95:3128 +103.147.246.18:8080 +128.199.202.122:80 +200.24.159.230:8080 +128.199.202.122:8080 +103.166.158.251:1111 +59.153.16.214:1120 +43.224.118.155:1121 +89.43.132.247:8080 +182.253.62.190:8080 +193.95.53.131:8077 +203.196.8.6:3128 +103.245.110.198:1452 +45.180.140.241:8080 +212.2.254.246:3128 +103.220.206.110:8585 +103.157.79.145:1080 +45.87.140.155:8080 +164.138.205.119:8080 +137.59.51.243:1120 +38.210.179.77:999 +27.147.163.188:40544 +194.87.77.22:80 +20.27.219.85:8080 +49.254.245.70:15648 +115.144.173.67:15648 +""" + +proxy_list = [p.strip() for p in proxies_raw.strip().split("\n") if p.strip()] +ips = [p.split(":")[0] for p in proxy_list] + +chunk_size = 100 +fr_proxies = [] + +for i in range(0, len(ips), chunk_size): + chunk = ips[i : i + chunk_size] + try: + req = urllib.request.Request("http://ip-api.com/batch", data=json.dumps(chunk).encode("utf-8")) + with urllib.request.urlopen(req) as response: + data = json.loads(response.read().decode("utf-8")) + + for idx, result in enumerate(data): + if result.get("countryCode") == "FR": + full_proxy = proxy_list[i + idx] + fr_proxies.append(full_proxy) + print(f"Found FR proxy: {full_proxy}") + except Exception as e: + print(f"Error querying batch: {e}") + +print(f"Total FR proxies found: {len(fr_proxies)}") + +if len(fr_proxies) > 0: + for p in fr_proxies: + print(f"PROXY:{p}") +else: + print("No French proxies found. Printing first 10 generic ones as backup:") + for p in proxy_list[:10]: + print(f"PROXY:{p}") diff --git a/find_working_proxies.py b/find_working_proxies.py new file mode 100644 index 0000000..be4b28d --- /dev/null +++ b/find_working_proxies.py @@ -0,0 +1,118 @@ +import urllib.request +import logging +import concurrent.futures + +# Setup logging +logging.basicConfig(level=logging.INFO, format="%(message)s") +logger = logging.getLogger("proxy_finder") + +# Target URL for verification +TARGET_URL = "https://www.amazon.fr" + + +def fetch_proxy_list(url): + try: + req = urllib.request.Request( + url, data=None, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} + ) + with urllib.request.urlopen(req, timeout=10) as response: + if response.status == 200: + text = response.read().decode("utf-8") + proxies = [p.strip() for p in text.splitlines() if p.strip() and ":" in p] + logger.info(f"Fetched {len(proxies)} proxies from {url}") + return proxies + except Exception as e: + logger.error(f"Failed to fetch from {url}: {e}") + return [] + + +def check_proxy_fast(proxy): + try: + proxy_handler = urllib.request.ProxyHandler({"http": proxy, "https": proxy}) + opener = urllib.request.build_opener(proxy_handler) + opener.addheaders = [("User-Agent", "Mozilla/5.0")] + with opener.open(TARGET_URL, timeout=5) as response: + # 200, 403, 503 all mean the proxy is alive (amazon may block but proxy works) + if response.status in [200, 403, 503]: + return proxy + except: + pass + return None + + +def main(): + logger.info("Starting proxy finder (FAST MODE)...") + + sources = [ + "https://api.proxyscrape.com/v2/?request=getproxies&protocol=http&timeout=10000&country=all&ssl=all&anonymity=all", + "https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt", + "https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt", + "https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt", + ] + + # 1. Fetch all proxies + all_proxies = set() + for url in sources: + proxies = fetch_proxy_list(url) + if proxies: + all_proxies.update(proxies) + + # Add local raw proxies + local_raw = [ + "164.68.110.241:8091", + "164.68.110.241:9992", + "173.212.246.157:3128", + "142.111.48.253:7030", + "31.59.20.176:6754", + "23.95.150.145:6114", + "198.23.239.134:6540", + "107.172.163.27:6543", + "198.105.121.200:6462", + "64.137.96.74:6641", + "84.247.60.125:6095", + "216.10.27.159:6837", + "142.111.67.146:5611", + ] + all_proxies.update(local_raw) + + print(f"\nTesting {len(all_proxies)} unique proxies against {TARGET_URL}...") + + working_proxies = [] + + # Use ThreadPoolExecutor for speed + with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor: + future_to_proxy = {executor.submit(check_proxy_fast, p): p for p in all_proxies} + + count = 0 + total = len(all_proxies) + + for future in concurrent.futures.as_completed(future_to_proxy): + count += 1 + if count % 500 == 0: + print(f"Processed {count}/{total} - Found {len(working_proxies)} so far") + + res = future.result() + if res: + print(f"ALIVE: {res}") + working_proxies.append(res) + # Stop if we have enough + if len(working_proxies) >= 30: + print("Found 30 proxies, stopping.") + executor.shutdown(wait=False, cancel_futures=True) + break + + print("\n" + "=" * 50) + print(f"FOUND {len(working_proxies)} WORKING PROXIES") + print("=" * 50) + + # Format for python list + formatted_list = "[\n" + ",\n".join([f' "{p}"' for p in working_proxies]) + "\n]" + print(formatted_list) + + # Save to file + with open("working_proxies.txt", "w") as f: + f.write("\n".join(working_proxies)) + + +if __name__ == "__main__": + main() diff --git a/working_proxies.txt b/working_proxies.txt new file mode 100644 index 0000000..c7fbfb6 --- /dev/null +++ b/working_proxies.txt @@ -0,0 +1,10 @@ +http://52.188.28.218:3128 +http://104.168.10.224:8888 +http://194.87.77.22:80 +http://43.225.148.210:1120 +http://68.235.35.171:3128 +http://211.230.49.122:3128 +http://59.6.25.118:3128 +http://144.125.164.222:8081 +http://91.229.91.52:50000 +http://195.225.109.132:3128 \ No newline at end of file