feat: Add core search configurations for various e-commerce sites, including proxy and user agent management, and introduce initial Amazon scraper service and proxy utilities.

This commit is contained in:
Michael committed 2025-12-24 21:13:22 +01:00
1 parent 43de609773
commit cd18b47bc0
6 files changed
+309 -25

No files matched your search

+18 -24
View File
@@ -10,18 +10,10 @@ BROWSERLESS_URL = os.getenv("BROWSERLESS_URL", "ws://browserless:3000")
DEBUG_DUMPS_DIR = "/app/debug_dumps"
# === PROXY CONFIGURATION ===
AMAZON_PROXY_LIST_RAW = [
"142.111.48.253:7030:jasuwwjr:elbsx170nmnl",
"31.59.20.176:6754:jasuwwjr:elbsx170nmnl",
"23.95.150.145:6114:jasuwwjr:elbsx170nmnl",
"198.23.239.134:6540:jasuwwjr:elbsx170nmnl",
"107.172.163.27:6543:jasuwwjr:elbsx170nmnl",
"198.105.121.200:6462:jasuwwjr:elbsx170nmnl",
"64.137.96.74:6641:jasuwwjr:elbsx170nmnl",
"84.247.60.125:6095:jasuwwjr:elbsx170nmnl",
"216.10.27.159:6837:jasuwwjr:elbsx170nmnl",
"142.111.67.146:5611:jasuwwjr:elbsx170nmnl",
]
# NOTE: All free proxies tested are non-functional. Direct connections will be used.
# Add working proxies here when available.
AMAZON_PROXY_LIST_RAW = []
def get_amazon_proxies() -> list[dict]:
"""Convert raw proxy list to Playwright format"""
@@ -29,50 +21,52 @@ def get_amazon_proxies() -> list[dict]:
for proxy in AMAZON_PROXY_LIST_RAW:
parts = proxy.split(":")
if len(parts) == 4:
proxies.append({
"server": f"http://{parts[0]}:{parts[1]}",
"username": parts[2],
"password": parts[3]
})
proxies.append({"server": f"http://{parts[0]}:{parts[1]}", "username": parts[2], "password": parts[3]})
elif len(parts) == 2:
proxies.append({"server": f"http://{parts[0]}:{parts[1]}"})
return proxies
# === USER AGENTS & STEALTH ===
USER_AGENT_DATA = [
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Windows"'
"platform": '"Windows"',
},
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="130", "Chromium";v="130", "Not_A Brand";v="99"',
"platform": '"Windows"'
"platform": '"Windows"',
},
{
"ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"macOS"'
"platform": '"macOS"',
},
{
"ua": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
"ch": '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Linux"'
"platform": '"Linux"',
},
{
"ua": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36 Edg/131.0.0.0",
"ch": '"Microsoft Edge";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
"platform": '"Windows"'
}
"platform": '"Windows"',
},
]
def get_random_stealth_config() -> dict:
"""Returns a random User-Agent and its corresponding Client Hints"""
return random.choice(USER_AGENT_DATA)
def get_random_user_agent() -> str:
"""Legacy helper for backward compatibility"""
return get_random_stealth_config()["ua"]
# === SITE CONFIGURATIONS ===
SITE_CONFIGS = {
# === MAGASINS DISCOUNT ===
@@ -240,7 +234,7 @@ COOKIE_ACCEPT_SELECTORS = [
".modal-close",
".close-modal",
".popup-close",
"button[aria-label='Close']",
"button[aria-label='Close']",
"button[aria-label='Fermer']",
".js-modal-close",
"div[class*='popup'] button[class*='close']",
+8 -1
View File
@@ -352,7 +352,14 @@ class AmazonScraperService:
logger.warning(f"⏳ Attempt {attempt} failed, retrying in {delay}s with new identity...")
await asyncio.sleep(delay)
logger.error(f"❌ All {MAX_RETRY_ATTEMPTS} attempts failed for query: {query}")
if not products and proxies:
logger.warning("⚠️ All proxy attempts failed. Attempting fallback to direct connection...")
products = await cls._try_scrape(query, max_results, None, MAX_RETRY_ATTEMPTS + 1)
if products:
logger.info("✅ Successfully extracted products using direct connection fallback")
return products
logger.error(f"❌ All attempts failed for query: {query}")
return []
@classmethod
+52
View File
@@ -0,0 +1,52 @@
import asyncio
import os
import sys
# Add app to path
sys.path.insert(0, os.getcwd())
from playwright.async_api import async_playwright
from app.core.search_config import get_amazon_proxies
async def check_proxy(proxy, semaphore):
async with semaphore:
proxy_url = proxy["server"]
print(f"Testing {proxy_url}...")
async with async_playwright() as p:
# Connect to browserless or launch local
# Using launch local for simpler testing without ws dependency if possible
# But the app uses browserless. Let's try launch first.
try:
browser = await p.chromium.launch(headless=True, proxy=proxy)
page = await browser.new_page()
try:
# amazon.fr might block, use httpbin for connectivity check
await page.goto("http://httpbin.org/ip", timeout=15000)
content = await page.content()
print(f"✅ {proxy_url}: Success")
await browser.close()
return True
except Exception as e:
print(f"❌ {proxy_url}: Failed - {str(e)[:100]}")
await browser.close()
return False
except Exception as e:
print(f"❌ {proxy_url}: Launch Failed - {str(e)[:100]}")
return False
async def main():
proxies = get_amazon_proxies()
print(f"Checking {len(proxies)} proxies...")
semaphore = asyncio.Semaphore(3) # Limit concurrency
results = await asyncio.gather(*[check_proxy(p, semaphore) for p in proxies])
working = sum(results)
print(f"\nSummary: {working}/{len(proxies)} working.")
if __name__ == "__main__":
asyncio.run(main())
+103
View File
@@ -0,0 +1,103 @@
import json
import urllib.request
import urllib.error
proxies_raw = """
46.161.6.165:8080
78.47.219.204:3128
134.209.29.120:8080
161.35.70.249:80
134.209.29.120:80
52.188.28.218:3128
209.97.150.167:3128
62.60.151.128:80
68.235.35.171:3128
209.97.150.167:80
159.203.61.169:8080
209.97.150.167:8080
195.158.8.123:3128
208.87.243.199:7878
144.76.42.215:8118
216.229.112.25:8080
159.203.61.169:80
103.3.246.71:3128
138.68.60.8:80
139.59.1.14:80
8.243.68.11:8080
41.223.119.156:3128
34.96.238.40:8080
59.6.25.118:3128
129.150.39.251:8000
162.240.154.26:3128
35.152.252.253:8080
144.125.164.158:8081
47.81.14.7:3129
144.125.164.222:8080
175.99.220.171:80
8.219.97.248:80
144.125.164.158:8080
164.68.110.241:8091
144.125.164.222:8081
140.238.184.182:3128
139.59.1.14:3128
8.212.160.196:8080
164.68.110.241:9992
173.212.246.157:3128
47.236.130.95:3128
103.147.246.18:8080
128.199.202.122:80
200.24.159.230:8080
128.199.202.122:8080
103.166.158.251:1111
59.153.16.214:1120
43.224.118.155:1121
89.43.132.247:8080
182.253.62.190:8080
193.95.53.131:8077
203.196.8.6:3128
103.245.110.198:1452
45.180.140.241:8080
212.2.254.246:3128
103.220.206.110:8585
103.157.79.145:1080
45.87.140.155:8080
164.138.205.119:8080
137.59.51.243:1120
38.210.179.77:999
27.147.163.188:40544
194.87.77.22:80
20.27.219.85:8080
49.254.245.70:15648
115.144.173.67:15648
"""
proxy_list = [p.strip() for p in proxies_raw.strip().split("\n") if p.strip()]
ips = [p.split(":")[0] for p in proxy_list]
chunk_size = 100
fr_proxies = []
for i in range(0, len(ips), chunk_size):
chunk = ips[i : i + chunk_size]
try:
req = urllib.request.Request("http://ip-api.com/batch", data=json.dumps(chunk).encode("utf-8"))
with urllib.request.urlopen(req) as response:
data = json.loads(response.read().decode("utf-8"))
for idx, result in enumerate(data):
if result.get("countryCode") == "FR":
full_proxy = proxy_list[i + idx]
fr_proxies.append(full_proxy)
print(f"Found FR proxy: {full_proxy}")
except Exception as e:
print(f"Error querying batch: {e}")
print(f"Total FR proxies found: {len(fr_proxies)}")
if len(fr_proxies) > 0:
for p in fr_proxies:
print(f"PROXY:{p}")
else:
print("No French proxies found. Printing first 10 generic ones as backup:")
for p in proxy_list[:10]:
print(f"PROXY:{p}")
+118
View File
@@ -0,0 +1,118 @@
import urllib.request
import logging
import concurrent.futures
# Setup logging
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("proxy_finder")
# Target URL for verification
TARGET_URL = "https://www.amazon.fr"
def fetch_proxy_list(url):
try:
req = urllib.request.Request(
url, data=None, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
)
with urllib.request.urlopen(req, timeout=10) as response:
if response.status == 200:
text = response.read().decode("utf-8")
proxies = [p.strip() for p in text.splitlines() if p.strip() and ":" in p]
logger.info(f"Fetched {len(proxies)} proxies from {url}")
return proxies
except Exception as e:
logger.error(f"Failed to fetch from {url}: {e}")
return []
def check_proxy_fast(proxy):
try:
proxy_handler = urllib.request.ProxyHandler({"http": proxy, "https": proxy})
opener = urllib.request.build_opener(proxy_handler)
opener.addheaders = [("User-Agent", "Mozilla/5.0")]
with opener.open(TARGET_URL, timeout=5) as response:
# 200, 403, 503 all mean the proxy is alive (amazon may block but proxy works)
if response.status in [200, 403, 503]:
return proxy
except:
pass
return None
def main():
logger.info("Starting proxy finder (FAST MODE)...")
sources = [
"https://api.proxyscrape.com/v2/?request=getproxies&protocol=http&timeout=10000&country=all&ssl=all&anonymity=all",
"https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt",
"https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt",
"https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt",
]
# 1. Fetch all proxies
all_proxies = set()
for url in sources:
proxies = fetch_proxy_list(url)
if proxies:
all_proxies.update(proxies)
# Add local raw proxies
local_raw = [
"164.68.110.241:8091",
"164.68.110.241:9992",
"173.212.246.157:3128",
"142.111.48.253:7030",
"31.59.20.176:6754",
"23.95.150.145:6114",
"198.23.239.134:6540",
"107.172.163.27:6543",
"198.105.121.200:6462",
"64.137.96.74:6641",
"84.247.60.125:6095",
"216.10.27.159:6837",
"142.111.67.146:5611",
]
all_proxies.update(local_raw)
print(f"\nTesting {len(all_proxies)} unique proxies against {TARGET_URL}...")
working_proxies = []
# Use ThreadPoolExecutor for speed
with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor:
future_to_proxy = {executor.submit(check_proxy_fast, p): p for p in all_proxies}
count = 0
total = len(all_proxies)
for future in concurrent.futures.as_completed(future_to_proxy):
count += 1
if count % 500 == 0:
print(f"Processed {count}/{total} - Found {len(working_proxies)} so far")
res = future.result()
if res:
print(f"ALIVE: {res}")
working_proxies.append(res)
# Stop if we have enough
if len(working_proxies) >= 30:
print("Found 30 proxies, stopping.")
executor.shutdown(wait=False, cancel_futures=True)
break
print("\n" + "=" * 50)
print(f"FOUND {len(working_proxies)} WORKING PROXIES")
print("=" * 50)
# Format for python list
formatted_list = "[\n" + ",\n".join([f' "{p}"' for p in working_proxies]) + "\n]"
print(formatted_list)
# Save to file
with open("working_proxies.txt", "w") as f:
f.write("\n".join(working_proxies))
if __name__ == "__main__":
main()
+10
View File
@@ -0,0 +1,10 @@
http://52.188.28.218:3128
http://104.168.10.224:8888
http://194.87.77.22:80
http://43.225.148.210:1120
http://68.235.35.171:3128
http://211.230.49.122:3128
http://59.6.25.118:3128
http://144.125.164.222:8081
http://91.229.91.52:50000
http://195.225.109.132:3128