mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
fix: Correct proxy configuration for Crawl4AI compatibility
- Changed proxy format from dict to string (http://user:pass@ip:port) - Updated get_random_proxy() to return Crawl4AI-compatible format - Replaced deprecated 'proxy' with 'proxy_config' in BrowserConfig - Fixed test script to handle new proxy format - Added credential hiding in proxy logging for security Fixes AttributeError: 'dict' object has no attribute 'strip'
This commit is contained in:
2 files changed
+24
-10
No files matched your search
@@ -59,8 +59,13 @@ def get_realistic_headers(user_agent: str) -> dict:
|
||||
}
|
||||
|
||||
|
||||
def get_random_proxy() -> dict | None:
|
||||
"""Get a random proxy from the pool"""
|
||||
def get_random_proxy() -> str | None:
|
||||
"""
|
||||
Get a random proxy from the pool in Crawl4AI format.
|
||||
|
||||
Returns:
|
||||
Proxy string in format: http://username:password@ip:port
|
||||
"""
|
||||
if not AMAZON_PROXY_LIST_RAW:
|
||||
return None
|
||||
|
||||
@@ -68,11 +73,14 @@ def get_random_proxy() -> dict | None:
|
||||
parts = proxy_str.split(":")
|
||||
|
||||
if len(parts) == 4:
|
||||
return {
|
||||
"server": f"http://{parts[0]}:{parts[1]}",
|
||||
"username": parts[2],
|
||||
"password": parts[3]
|
||||
}
|
||||
ip = parts[0]
|
||||
port = parts[1]
|
||||
username = parts[2]
|
||||
password = parts[3]
|
||||
|
||||
# Format: http://username:password@ip:port
|
||||
return f"http://{username}:{password}@{ip}:{port}"
|
||||
|
||||
return None
|
||||
|
||||
|
||||
@@ -202,7 +210,10 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
|
||||
# Select random proxy
|
||||
proxy = get_random_proxy()
|
||||
if proxy:
|
||||
logger.debug(f"🌐 Using proxy: {proxy['server']}")
|
||||
# Extract just the IP for logging (hide credentials)
|
||||
proxy_parts = proxy.split('@')
|
||||
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
|
||||
logger.debug(f"🌐 Using proxy: {proxy_server}")
|
||||
else:
|
||||
logger.warning("⚠️ No proxy available - may face rate limiting")
|
||||
|
||||
@@ -211,7 +222,7 @@ async def scrape_amazon_search(query: str, max_results: int = 20) -> list[Amazon
|
||||
headless=True,
|
||||
verbose=False,
|
||||
user_agent=user_agent,
|
||||
proxy=proxy,
|
||||
proxy_config=proxy, # Use proxy_config instead of deprecated proxy
|
||||
extra_args=[
|
||||
"--disable-blink-features=AutomationControlled", # Disable automation detection
|
||||
"--disable-dev-shm-usage",
|
||||
|
||||
@@ -98,7 +98,10 @@ async def test_anti_detection():
|
||||
# Test proxy
|
||||
proxy = get_random_proxy()
|
||||
if proxy:
|
||||
logger.info(f"✓ Proxy test: {proxy['server']}")
|
||||
# Extract just the IP for logging (hide credentials)
|
||||
proxy_parts = proxy.split('@')
|
||||
proxy_server = proxy_parts[1] if len(proxy_parts) > 1 else proxy
|
||||
logger.info(f"✓ Proxy test: {proxy_server}")
|
||||
else:
|
||||
logger.warning("⚠️ Pas de proxy configuré")
|
||||
|
||||
|
||||
Reference in new issue
Block a user