mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
fix: Stabilité Browserless v2 + headers anti-détection
Browserless (docker-compose.yml): - Image mise à jour: ghcr.io/browserless/chromium:latest - Timeout augmenté à 300s (5 minutes) - shm_size: 2gb (critique pour Chrome) - Mémoire: 3G limit, 1G reservation - Nouvelles variables: HEALTH, CHROME_REFRESH_TIME Scraper Service: - Ajout headers HTTP réalistes (Accept-Language, Accept) - Locale fr-FR et timezone Europe/Paris - Timeout connexion explicite (60s) - Fermeture propre des ressources (page, context, browser) - Meilleure gestion des erreurs de fermeture
This commit is contained in:
2 files changed
+54
-14
No files matched your search
@@ -154,9 +154,18 @@ class ScraperService:
|
||||
"""Exécute le scraping réel (appelé par scrape_item avec retries)."""
|
||||
async with async_playwright() as p:
|
||||
browser = None
|
||||
context = None
|
||||
page = None
|
||||
try:
|
||||
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
|
||||
browser = await p.chromium.connect_over_cdp(BROWSERLESS_URL)
|
||||
|
||||
# Utiliser l'endpoint WebSocket de Browserless
|
||||
# Format: ws://browserless:3000?token=xxx ou ws://browserless:3000
|
||||
browser = await p.chromium.connect_over_cdp(
|
||||
BROWSERLESS_URL,
|
||||
timeout=60000, # 60s pour la connexion
|
||||
)
|
||||
|
||||
context = await browser.new_context(
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
user_agent=(
|
||||
@@ -164,6 +173,13 @@ class ScraperService:
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/120.0.0.0 Safari/537.36"
|
||||
),
|
||||
locale="fr-FR",
|
||||
timezone_id="Europe/Paris",
|
||||
# Options supplémentaires pour éviter la détection
|
||||
extra_http_headers={
|
||||
"Accept-Language": "fr-FR,fr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
}
|
||||
)
|
||||
|
||||
# Stealth mode / Ad blocking attempts
|
||||
@@ -308,5 +324,21 @@ class ScraperService:
|
||||
return None, ""
|
||||
|
||||
finally:
|
||||
if browser:
|
||||
await browser.close()
|
||||
# Fermeture propre des ressources dans l'ordre inverse
|
||||
try:
|
||||
if page and not page.is_closed():
|
||||
await page.close()
|
||||
except Exception as e:
|
||||
logger.debug(f"Error closing page: {e}")
|
||||
|
||||
try:
|
||||
if context:
|
||||
await context.close()
|
||||
except Exception as e:
|
||||
logger.debug(f"Error closing context: {e}")
|
||||
|
||||
try:
|
||||
if browser and browser.is_connected():
|
||||
await browser.close()
|
||||
except Exception as e:
|
||||
logger.debug(f"Error closing browser: {e}")
|
||||
+19
-11
@@ -49,26 +49,34 @@ services:
|
||||
|
||||
# Browserless - Navigateur headless pour le scraping
|
||||
browserless:
|
||||
image: browserless/chrome:latest
|
||||
image: ghcr.io/browserless/chromium:latest
|
||||
container_name: priceflow-browserless
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "3012:3000"
|
||||
environment:
|
||||
- MAX_CONCURRENT_SESSIONS=3
|
||||
- CONNECTION_TIMEOUT=180000
|
||||
- MAX_QUEUE_LENGTH=5
|
||||
- PREBOOT_CHROME=true
|
||||
- KEEP_ALIVE=true
|
||||
- DEFAULT_BLOCK_ADS=true
|
||||
# Timeouts
|
||||
- TIMEOUT=300000
|
||||
- CONNECTION_TIMEOUT=300000
|
||||
# Sessions
|
||||
- CONCURRENT=3
|
||||
- QUEUED=5
|
||||
# Options de stabilité
|
||||
- HEALTH=true
|
||||
- EXIT_ON_HEALTH_FAILURE=true
|
||||
# Options Chrome
|
||||
- DEFAULT_HEADLESS=true
|
||||
- DEFAULT_STEALTH=true
|
||||
- ENABLE_DEBUGGER=false
|
||||
- DISABLED_FEATURES=gpu
|
||||
- DEFAULT_LAUNCH_ARGS=["--disable-dev-shm-usage","--no-sandbox","--disable-setuid-sandbox"]
|
||||
- DEFAULT_BLOCK_ADS=true
|
||||
- DEFAULT_IGNORE_HTTPS_ERRORS=true
|
||||
- CHROME_REFRESH_TIME=3600000
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 2G
|
||||
memory: 3G
|
||||
reservations:
|
||||
memory: 1G
|
||||
shm_size: '2gb'
|
||||
networks:
|
||||
- nginx_default
|
||||
|
||||
|
||||
Reference in new issue
Block a user