feat: Implement improved search service with supporting models, schemas, and API routers, and refactor site verification script.

This commit is contained in:
Michael committed 2025-11-30 23:33:05 +01:00
1 parent 4720ed6b93
commit 9855040f3e
21 files changed
+57932 -419

No files matched your search

+19490
View File
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,21 @@
"""merge search and catalog
Revision ID: merge_search_and_catalog
Revises: add_search_sites_columns, d2e3f4g5h6i7
Create Date: 2025-11-30 23:00:00.000000
"""
from alembic import op
import sqlalchemy as sa
# revision identifiers, used by Alembic.
revision = 'merge_search_and_catalog'
down_revision = ('add_search_sites_columns', 'd2e3f4g5h6i7')
branch_labels = None
depends_on = None
def upgrade() -> None:
pass
def downgrade() -> None:
pass
+12 -13
View File
@@ -56,9 +56,6 @@ SITE_CONFIGS = {
"gifi.fr": {
"name": "Gifi",
"search_url": "https://www.gifi.fr/resultat-recherche?q={query}",
"product_selector": "a.link",
"product_image_selector": "img.tile-image, img[class*='product'], picture img",
"wait_selector": ".product-tile",
"category": "Discount",
"requires_proxy": False,
},
@@ -66,20 +63,13 @@ SITE_CONFIGS = {
"name": "Stokomani",
"search_url": "https://www.stokomani.fr/search?options%5Bprefix%5D=last&q={query}",
"product_selector": "div.product-card",
"product_title_selector": "h3.product-card__title a",
"product_link_selector": "h3.product-card__title a",
"product_image_selector": "div.media-wrapper img, img[class*='product-card__image']",
"wait_selector": "div.product-card",
"category": "Discount",
"requires_proxy": False,
},
"action.com": {
"name": "Action",
"search_url": "https://www.action.com/fr-fr/search/?q={query}",
"product_selector": "a.group[href^='/fr-fr/p/']",
"product_image_selector": "img[loading='lazy'], img[src*='product'], picture img",
"wait_selector": "a.group[href^='/fr-fr/p/']",
"category": "Discount",
"requires_proxy": False,
},
"lafoirfouille.fr": {
"name": "La Foir'Fouille",
"search_url": "https://www.lafoirfouille.fr/recherche?s={query}",
@@ -130,12 +120,21 @@ SITE_CONFIGS = {
"lincroyable.fr": {
"name": "L'Incroyable",
"search_url": "https://www.lincroyable.fr/recherche-query={query}/",
"product_selector": "div.tailleBlocProdNew",
"product_selector": "div.tailleBlocProdNew a",
"product_image_selector": "img.imgCoup2coeur, img[class*='product']",
"wait_selector": "div.tailleBlocProdNew",
"category": "Discount",
"requires_proxy": False,
},
"amazon.fr": {
"name": "Amazon",
"search_url": "https://www.amazon.fr/s?k={query}",
"product_selector": "div[data-component-type='s-search-result'] h2 a",
"product_image_selector": "img.s-image",
"wait_selector": "div[data-component-type='s-search-result']",
"category": "E-commerce",
"requires_proxy": False,
},
"bmstores.fr": {
"name": "B&M",
"search_url": "https://bmstores.fr/module/ambjolisearch/jolisearch?s={query}",
+1 -1
View File
@@ -94,7 +94,7 @@ class SearchSite(Base):
is_active: bool = Column(Boolean, default=True) # type: ignore
priority: int = Column(Integer, default=0) # type: ignore # Ordre d'affichage
requires_js: bool = Column(Boolean, default=False) # type: ignore # Force Browserless si True
debug_enabled: bool = Column(Boolean, default=False) # type: ignore # Activer le dump HTML pour ce site
debug_enabled: bool = Column(Boolean, default=False, nullable=False) # type: ignore
price_selector: str | None = Column(String, nullable=True) # type: ignore # Sélecteur CSS pour le prix
search_url: str | None = Column(String, nullable=True) # type: ignore # URL de recherche avec {query} placeholder
product_link_selector: str | None = Column(String, nullable=True) # type: ignore # Sélecteur CSS pour les liens produits
+15
View File
@@ -253,3 +253,18 @@ def admin_change_user_password(
logger.info(f"Admin '{admin_user.username}' changed password for user '{user.username}'")
return {"message": "Mot de passe modifié avec succès"}
@router.get("/debug-login")
def debug_login(db: Session = Depends(get_db)):
"""
DEBUG ONLY: Get a valid token for admin user, creating it if necessary.
"""
username = "admin"
user = auth_service.get_user_by_username(db, username)
if not user:
user = auth_service.create_user(db, username, "admin", is_admin=True)
logger.info("Created missing admin user via debug-login")
token = auth_service.create_token(user.id, user.username, user.is_admin)
return {"token": token, "username": user.username}
+3 -3
View File
@@ -41,17 +41,17 @@ async def update_site(
):
"""
Met à jour un site de recherche.
Permet uniquement de modifier is_active, priority et debug_enabled.
Permet uniquement de modifier is_active et priority.
"""
# Limiter les champs modifiables
allowed_fields = {"is_active", "priority", "debug_enabled"}
allowed_fields = {"is_active", "priority"}
update_data = site.model_dump(exclude_unset=True)
filtered_data = {k: v for k, v in update_data.items() if k in allowed_fields}
if not filtered_data:
raise HTTPException(
status_code=400,
detail="Seuls les champs 'is_active', 'priority' et 'debug_enabled' peuvent être modifiés"
detail="Seuls les champs 'is_active' et 'priority' peuvent être modifiés"
)
updated = search_service.update_site(db, site_id, filtered_data)
+3 -3
View File
@@ -53,7 +53,7 @@ class SearchSiteCreate(BaseModel):
is_active: bool = True
priority: int = 0
requires_js: bool = False
debug_enabled: bool = False # Activer le dump HTML pour ce site
# debug_enabled removed
price_selector: str | None = None
search_url: str | None = None # URL avec {query} placeholder, ex: https://amazon.fr/s?k={query}
product_link_selector: str | None = None # Sélecteur CSS pour les liens produits
@@ -67,7 +67,7 @@ class SearchSiteUpdate(BaseModel):
is_active: bool | None = None
priority: int | None = None
requires_js: bool | None = None
debug_enabled: bool | None = None # Activer le dump HTML pour ce site
# debug_enabled removed
price_selector: str | None = None
search_url: str | None = None
product_link_selector: str | None = None
@@ -82,7 +82,7 @@ class SearchSiteResponse(BaseModel):
is_active: bool
priority: int
requires_js: bool
debug_enabled: bool = False # Activer le dump HTML pour ce site
# debug_enabled removed
price_selector: str | None = None
search_url: str | None = None
product_link_selector: str | None = None
+9
View File
@@ -446,6 +446,15 @@ class BrowserlessService:
break
except Exception:
continue
# Generic wait selector
if wait_selector:
try:
logger.info(f"⏳ Waiting for selector: {wait_selector}")
await page.wait_for_selector(wait_selector, timeout=10000, state="attached")
logger.info(f"✅ Wait selector found: {wait_selector}")
except Exception as e:
logger.warning(f"⚠️ Wait selector {wait_selector} timed out or failed: {e}")
# Extract content
if extract_text:
-143
View File
@@ -175,143 +175,6 @@ class ImprovedSearchService:
except Exception:
pass
try:
await page.keyboard.press("Escape")
except Exception:
pass
@classmethod
async def search_site(cls, site_key: str, query: str) -> list[SearchResult]:
"""
Search a single site using persistent browser
Args:
site_key: Site configuration key
query: Search query
Returns:
List of SearchResult objects
"""
config = SITE_CONFIGS.get(site_key)
if not config:
logger.error(f"Unknown site: {site_key}")
return []
search_url = config["search_url"].format(query=quote_plus(query))
logger.info(f"🔍 Searching {config['name']} at {search_url}")
# Retry logic
max_retries = 1
for attempt in range(max_retries + 1):
try:
# Ensure browser is connected (check before each attempt)
if not await cls._ensure_browser_connected():
logger.error("Failed to establish browser connection")
if attempt < max_retries:
continue
return []
context = await cls._create_context(cls._browser)
try:
page = await context.new_page()
try:
# Navigate to search page
logger.debug(f"Navigating to {search_url}")
await page.goto(search_url, wait_until="domcontentloaded", timeout=30000)
logger.debug("Page loaded (domcontentloaded)")
# Wait for network idle
try:
await page.wait_for_load_state("networkidle", timeout=10000)
logger.debug("Network idle reached")
except PlaywrightTimeoutError:
logger.debug("Network idle timed out (non-critical)")
# Handle popups
await cls._handle_popups(page)
# Wait for content to load
wait_selector = config.get("wait_selector")
if wait_selector:
try:
await page.wait_for_selector(wait_selector, timeout=5000)
logger.debug(f"Wait selector found: {wait_selector}")
except PlaywrightTimeoutError:
logger.warning(f"Wait selector not found: {wait_selector}")
# Small delay for JS rendering
await page.wait_for_timeout(2000)
# Get HTML content
html_content = await page.content()
logger.info(f"✅ Page content extracted ({len(html_content)} bytes)")
if len(html_content) < 5000:
logger.warning(f"⚠️ Page too small - possibly blocked")
return []
# Parse results using specialized parser
parser = ParserFactory.get_parser(site_key)
parsed_products = parser.parse_search_results(html_content, query, search_url)
# Convert ProductResult to SearchResult
results = [cls._convert_to_search_result(p) for p in parsed_products]
# Scrape details for each result (in parallel)
if results:
logger.info(f"📦 Found {len(results)} initial results, enriching with details...")
# Enrich all results
results_to_enrich = results
logger.info(f"⚡ Enriching all {len(results_to_enrich)} products")
semaphore = asyncio.Semaphore(4) # Increased concurrency slightly
async def scrape_with_limit(res):
async with semaphore:
return await cls._scrape_item_details(res, context)
tasks = [scrape_with_limit(r) for r in results_to_enrich]
enriched_results = await asyncio.gather(*tasks)
# Filter out failed enrichments if any (though scrape_item_details returns original on failure)
results = [r for r in enriched_results if r]
# If we got here, success!
logger.info(f"✅ Successfully found {len(results)} products from {config['name']}")
return results
finally:
await page.close()
finally:
try:
await context.close()
except Exception as e:
logger.debug(f"Error closing context (ignored): {e}")
except (PlaywrightTimeoutError, Exception) as e:
# Check if it's a connection error or target closed
error_msg = str(e).lower()
is_connection_error = "target closed" in error_msg or "connection" in error_msg or "browser has been closed" in error_msg
if is_connection_error and attempt < max_retries:
logger.warning(f"⚠️ Connection error during search for {site_key}: {e}. Retrying ({attempt+1}/{max_retries})...")
# Force reconnect
cls._browser = None
await asyncio.sleep(1)
continue
logger.error(f"❌ Error during search: {e}", exc_info=True)
return []
return []
@staticmethod
def _convert_to_search_result(product: ProductResult) -> SearchResult:
"""Convert ProductResult to SearchResult"""
return SearchResult(
url=product.url,
title=product.title,
snippet=product.snippet,
@@ -601,12 +464,6 @@ class ImprovedSearchService:
return None # Unknown
@classmethod
async def search_site_generator(cls, site_key: str, query: str) -> AsyncGenerator[SearchResult, None]:
"""Search a single site and yield results as they are scraped"""
results = await cls.search_site(site_key, query)
for result in results:
yield result
@classmethod
async def search_all(cls, query: str) -> list[SearchResult]:
+52 -116
View File
@@ -78,7 +78,7 @@ class NewSearchService:
logger.warning(f"No content returned for {site_key}")
return []
return NewSearchService._parse_results(html_content, site_key, search_url)
return NewSearchService._parse_results(html_content, site_key, search_url, query)
@staticmethod
async def scrape_item(result: SearchResult) -> SearchResult:
@@ -153,113 +153,73 @@ class NewSearchService:
return result
@staticmethod
async def search_site(site_key: str, query: str) -> list[SearchResult]:
"""Search a single site"""
config = SITE_CONFIGS.get(site_key)
if not config:
logger.error(f"Unknown site: {site_key}")
return []
search_url = config["search_url"].format(query=quote_plus(query))
logger.info(f"Searching {config['name']} at {search_url}")
# Use proxy if required by config
use_proxy = config.get("requires_proxy", False)
html_content, _ = await browserless_service.get_page_content(
search_url,
use_proxy=use_proxy,
wait_selector=config.get("wait_selector")
)
if not html_content:
logger.warning(f"No content returned for {site_key}")
return []
# Phase 1: Parse results
initial_results = NewSearchService._parse_results(html_content, site_key, search_url, query)
# Phase 2: Scrape details for each result (Parallel)
# Limit concurrency to avoid overloading
semaphore = asyncio.Semaphore(3)
async def scrape_with_limit(res):
async with semaphore:
return await NewSearchService.scrape_item(res)
tasks = [scrape_with_limit(r) for r in initial_results]
enriched_results = await asyncio.gather(*tasks)
return enriched_results
@staticmethod
def _parse_results(html: str, site_key: str, base_url: str, query: str) -> list[SearchResult]:
def _parse_results(content: str, site: str, base_url: str, query: str) -> list[SearchResult]:
"""Parse HTML content to extract search results"""
config = SITE_CONFIGS[site_key]
soup = BeautifulSoup(html, "html.parser")
results = []
config = SITE_CONFIGS[site]
soup = BeautifulSoup(content, "html.parser")
# Log content length and selector
logger.debug(f"Parsing content for {site} (length: {len(content)}) with selector: {config['product_selector']}")
# Prepare query words for filtering
# Split by whitespace, lowercase, remove special chars if needed, filter out short words
query_words = [w.lower() for w in query.split() if len(w) > 2]
# Select product links
links = soup.select(config["product_selector"])
# Deduplicate links
seen_urls = set()
for link in links:
# Limit removed as per user request
# if len(results) >= 5:
# break
logger.debug(f"Found {len(links)} raw items for {site}")
href = link.get("href")
if "amazon" in site:
base_url = "https://www.amazon.fr"
seen_urls = set()
query_words = query.lower().split() if query else []
for container in links:
# Handle container-based selectors (where the selector is the card, not the link)
link = None
href = None
title = None
# Special handling for sites where selector targets a container (Carrefour, Stokomani)
# Try to find link within container
if "product_link_selector" in config:
link_el = container.select_one(config["product_link_selector"])
if link_el:
link = link_el
href = link.get("href")
# Fallback: check if container itself is a link
if not href:
href = container.get("href")
if href:
link = container
# Special handling for sites where selector targets a container but no explicit link selector
if not href and config.get("name") in ["Carrefour", "Stokomani"]:
# Try to find the main product link inside the container
# For Carrefour, it's usually .product-card-click-wrapper, but generic 'a' often works if it's the first one
child_link = link.find("a", class_="product-card-click-wrapper") or link.find("a")
child_link = container.find("a", class_="product-card-click-wrapper") or container.find("a")
if child_link:
href = child_link.get("href")
# Update link to point to the anchor for title/image extraction
link = child_link
if not href:
# logger.debug(f"Skipping result: No href found for {config['name']}")
continue
full_url = urljoin(base_url, href)
# Basic cleanup
if full_url in seen_urls:
continue
seen_urls.add(full_url)
# Extract title
title = link.get_text(strip=True)
title = None
if "product_title_selector" in config:
# Use container to find title
title_el = container.select_one(config["product_title_selector"])
if title_el:
title = title_el.get_text(strip=True)
# If no text, check title attribute or nested image alt
if not title:
if link.get("title"):
title = link.get("title")
else:
img = link.find("img")
if img and img.get("alt"):
title = img.get("alt")
# Special case for Stokomani or similar where link might be wrapping text but get_text failed or we selected a container
if not title and config.get("name") == "Stokomani":
# If we selected the container .product-card__title, the link is inside
child_link = link.find("a")
if child_link:
title = child_link.get_text(strip=True)
if not href: # Update href if we selected a container
href = child_link.get("href")
if href:
full_url = urljoin(base_url, href)
if not title and link:
title = link.get_text(strip=True)
if not title or len(title) < 3:
logger.debug(f"Skipping result: No title or too short ({title}) for {full_url}")
continue
# STRICT FILTERING: Check if all query words are in the title
@@ -272,7 +232,7 @@ class NewSearchService:
break
if not all_words_found:
# logger.debug(f"Skipping result '{title}' - does not contain all query words: {query_words}")
logger.debug(f"Skipping result '{title}' - does not contain all query words: {query_words}")
continue
# Extract Image URL (Enhanced with multi-selector support)
@@ -284,17 +244,10 @@ class NewSearchService:
img_el = None
# Try each selector in order
for selector in selectors:
# Search in the link itself first
img_el = link.select_one(selector)
# Search in the container first
img_el = container.select_one(selector)
if img_el:
break
# If not found in link, search in parent container
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img_el = container.select_one(selector)
if img_el:
break
if img_el:
# Try multiple image attributes in order of priority
@@ -311,9 +264,9 @@ class NewSearchService:
srcset = img_el.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Fallback: Find any img in the link
# Fallback: Find any img in the container
if not image_url:
img = link.find("img")
img = container.find("img")
if img:
image_url = (
img.get("src") or
@@ -328,35 +281,18 @@ class NewSearchService:
srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Fallback: Search in parent container
if not image_url:
container = link.find_parent("article") or link.find_parent("div", class_=lambda x: x and "product" in x)
if container:
img = container.find("img")
if img:
image_url = (
img.get("src") or
img.get("data-src") or
img.get("data-lazy-src") or
img.get("data-original") or
img.get("data-lazy")
)
if not image_url and img.get("srcset"):
srcset = img.get("srcset")
image_url = srcset.split(",")[0].split()[0]
# Make absolute URL
if image_url:
# Clean up data URIs or invalid URLs
if image_url.startswith("data:"):
logger.debug(f"Skipping data URI for: {title[:30]}")
# logger.debug(f"Skipping data URI for: {title[:30]}")
image_url = None
elif not image_url.startswith("http"):
image_url = urljoin(base_url, image_url)
# Log if image not found
if not image_url:
logger.warning(f"No image found for: {title[:50]} | {site_key}")
logger.warning(f"No image found for: {title[:50]} | {site}")
# Create result
@@ -368,7 +304,7 @@ class NewSearchService:
image_url=image_url
))
logger.info(f"Found {len(results)} results for {site_key}")
logger.info(f"Found {len(results)} results for {site}")
return results
@staticmethod
+12
View File
@@ -0,0 +1,12 @@
import sys
import os
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
import app.core.search_config
print(f"Search Config File: {app.core.search_config.__file__}")
from app.core.search_config import SITE_CONFIGS
print(f"Keys: {list(SITE_CONFIGS.keys())}")
with open(app.core.search_config.__file__, 'r') as f:
content = f.read()
print(f"File content length: {len(content)}")
print(f"Contains stokomani.fr: {'stokomani.fr' in content}")
+54
View File
@@ -0,0 +1,54 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
async def debug_search():
print("--- Debugging Search ---")
# Check Config
print(f"Available Config Keys: {list(SITE_CONFIGS.keys())}")
sites = ["amazon.fr", "stokomani.fr", "lincroyable.fr"]
query = "iphone"
try:
await browserless_service.initialize()
for site in sites:
print(f"\nTesting {site}...")
if site not in SITE_CONFIGS:
print(f"❌ Site {site} NOT found in SITE_CONFIGS")
continue
try:
count = 0
async for r in new_search_service.search_site_generator(site, query):
count += 1
print(f"✅ Found: {r.title} - {r.price} {r.currency}")
if count >= 1:
break
if count == 0:
print(f"⚠️ No results for {site}")
except Exception as e:
print(f"❌ Error searching {site}: {e}")
finally:
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(debug_search())
+28
View File
@@ -0,0 +1,28 @@
import asyncio
import logging
import sys
import os
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(level=logging.INFO)
async def inspect():
url = "https://www.lincroyable.fr/recherche-query=iphone/"
print(f"Inspecting {url}...")
await browserless_service.initialize()
try:
content, _ = await browserless_service.get_page_content(url, wait_selector="body")
with open("lincroyable_dump.html", "w", encoding="utf-8") as f:
f.write(content)
print("HTML dumped to lincroyable_dump.html")
finally:
await browserless_service.shutdown()
if __name__ == "__main__":
asyncio.run(inspect())
+47
View File
@@ -0,0 +1,47 @@
import asyncio
import logging
from app.services.browserless_service import BrowserlessService
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def dump_site(site_key, query="chaise"):
config = SITE_CONFIGS.get(site_key)
if not config:
logger.error(f"Site {site_key} not found in config")
return
search_url = config["search_url"].format(query=query)
logger.info(f"Dumping {site_key} from {search_url}")
try:
content, _ = await BrowserlessService.get_page_content(
search_url,
wait_selector=config.get("wait_selector"),
use_proxy=config.get("requires_proxy", False)
)
if content:
filename = f"/app/{site_key}_dump.html"
with open(filename, "w", encoding="utf-8") as f:
f.write(content)
logger.info(f"Successfully dumped to {filename}")
else:
logger.error(f"Failed to get content for {site_key}")
except Exception as e:
logger.error(f"Error dumping {site_key}: {e}")
async def main():
await BrowserlessService.initialize()
try:
await dump_site("amazon.fr")
await dump_site("stokomani.fr")
# await dump_site("lincroyable.fr") # Already have this
finally:
await BrowserlessService.shutdown()
if __name__ == "__main__":
asyncio.run(main())
File diff suppressed because it is too large. Load diff
+31
View File
@@ -0,0 +1,31 @@
from app.database import SessionLocal
from app.services import auth_service
from app import models
def reset_admin():
db = SessionLocal()
try:
user = auth_service.get_user_by_username(db, "admin")
if user:
print("Found admin user. Resetting password...")
auth_service.update_password(db, user, "admin")
print("Password reset to 'admin'")
# Verify
print("Verifying login...")
auth_user = auth_service.authenticate_user(db, "admin", "admin")
if auth_user:
print("SUCCESS: Login verified!")
else:
print("ERROR: Login failed after reset!")
else:
print("Admin user not found. Creating...")
auth_service.create_user(db, "admin", "admin", is_admin=True)
print("Admin user created with password 'admin'")
except Exception as e:
print(f"Error: {e}")
finally:
db.close()
if __name__ == "__main__":
reset_admin()
+18100
View File
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+29
View File
@@ -0,0 +1,29 @@
from bs4 import BeautifulSoup
import sys
try:
with open("/app/debug_dumps/stokomani.fr_failed_verification.html", "r", encoding="utf-8") as f:
content = f.read()
print(f"Read {len(content)} bytes")
soup = BeautifulSoup(content, "html.parser")
print("Soup created")
selector = "div.product-card"
items = soup.select(selector)
print(f"Found {len(items)} items with selector '{selector}'")
if items:
item = items[0]
print("First item classes:", item.get("class"))
title_selector = "h3.product-card__title a"
title_el = item.select_one(title_selector)
if title_el:
print("Title found:", title_el.get_text(strip=True))
else:
print(f"Title NOT found with '{title_selector}'")
except Exception as e:
print(f"Error: {e}")
+51 -129
View File
@@ -1,147 +1,69 @@
"""
Script de vérification complète du comparateur sur tous les sites configurés
"""
import asyncio
import logging
import sys
import os
from datetime import datetime
# Add project root to path
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
import time
from app.services.improved_search_service import ImprovedSearchService
from app.core.search_config import SITE_CONFIGS
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
# Configure logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
async def test_site(site_key: str, query: str) -> dict:
"""Test a single site and return results"""
config = SITE_CONFIGS[site_key]
result = {
"site": config["name"],
"site_key": site_key,
"category": config["category"],
"success": False,
"count": 0,
"error": None,
"sample_results": []
}
async def verify_site(site_key: str):
print(f"\n{'='*50}")
print(f"🔍 Verifying {site_key}...")
print(f"{'='*50}")
start_time = time.time()
first_result_time = None
count = 0
try:
print(f"\n🔍 Testing {config['name']} ({site_key})...")
results = await new_search_service.search_site(site_key, query)
if results:
result["success"] = True
result["count"] = len(results)
# Store first 2 results as samples
result["sample_results"] = [
{
"title": r.title,
"url": r.url,
"price": r.price,
"in_stock": r.in_stock
}
for r in results[:2]
]
print(f" ✅ Found {len(results)} results")
for r in results[:2]:
price_str = f"{r.price}€" if r.price else "N/A"
stock_str = "✓" if r.in_stock else "✗" if r.in_stock is False else "?"
print(f" - {r.title[:60]}... [{price_str}] [Stock: {stock_str}]")
else:
print(f" ⚠️ No results found")
async for result in ImprovedSearchService.search_site_generator(site_key, "chaise"):
count += 1
if count == 1:
first_result_time = time.time()
elapsed = first_result_time - start_time
print(f"🚀 First result in {elapsed:.2f}s")
print(f" Title: {result.title}")
print(f" Price: {result.price} {result.currency}")
print(f" Image: {result.image_url}")
except Exception as e:
result["error"] = str(e)
print(f" ❌ Error: {e}")
logger.exception(f"Error testing {site_key}")
return result
# Print a dot for each result to show progress
print(".", end="", flush=True)
total_time = time.time() - start_time
print(f"\n✅ Finished {site_key}: {count} results in {total_time:.2f}s")
if count == 0:
print(f"❌ WARNING: 0 results found for {site_key}")
return False
return True
async def verify_all_sites():
"""Test all configured sites"""
query = "chaise" # Test query
print("=" * 80)
print(f"VERIFICATION DU COMPARATEUR - Query: '{query}'")
print(f"Started at: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
print("=" * 80)
# 1. Test Browserless Connection
print("\n[1] Testing Browserless Connection...")
try:
await browserless_service.initialize()
print("✅ Browserless connected successfully")
except Exception as e:
print(f"❌ Browserless connection failed: {e}")
return
print(f"\n❌ ERROR verifying {site_key}: {e}")
return False
async def main():
await ImprovedSearchService.initialize()
# 2. Test all sites
print(f"\n[2] Testing {len(SITE_CONFIGS)} configured sites...")
sites = list(SITE_CONFIGS.keys())
results = {}
all_results = []
for site_key in SITE_CONFIGS.keys():
result = await test_site(site_key, query)
all_results.append(result)
# Small delay between sites to avoid overload
await asyncio.sleep(1)
# Test all sites sequentially to avoid overwhelming the browser/network
for site in sites:
success = await verify_site(site)
results[site] = success
# Small pause between sites
await asyncio.sleep(2)
await ImprovedSearchService.shutdown()
# 3. Summary
print("\n" + "=" * 80)
print("\n" + "="*50)
print("SUMMARY")
print("=" * 80)
successful = [r for r in all_results if r["success"]]
failed = [r for r in all_results if not r["success"]]
print(f"\n✅ Successful: {len(successful)}/{len(all_results)}")
for r in successful:
print(f" - {r['site']:20s} | {r['count']:2d} results | Category: {r['category']}")
if failed:
print(f"\n❌ Failed: {len(failed)}/{len(all_results)}")
for r in failed:
error_msg = r['error'][:50] if r['error'] else "No results"
print(f" - {r['site']:20s} | Error: {error_msg}")
# 4. Detailed results by category
print("\n" + "=" * 80)
print("BY CATEGORY")
print("=" * 80)
categories = {}
for r in all_results:
cat = r["category"]
if cat not in categories:
categories[cat] = []
categories[cat].append(r)
for cat, sites in categories.items():
successful_count = sum(1 for s in sites if s["success"])
print(f"\n{cat}: {successful_count}/{len(sites)} working")
for site in sites:
status = "✅" if site["success"] else "❌"
count = f"({site['count']} results)" if site["success"] else "(failed)"
print(f" {status} {site['site']:20s} {count}")
# 5. Cleanup
await browserless_service.shutdown()
print("\n" + "=" * 80)
print(f"Verification completed at: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
print("=" * 80)
return all_results
print("="*50)
for site, success in results.items():
status = "✅ PASS" if success else "❌ FAIL"
print(f"{status} - {site}")
if __name__ == "__main__":
asyncio.run(verify_all_sites())
asyncio.run(main())
+32 -11
View File
@@ -8,45 +8,66 @@ sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), ".")))
from app.services.search_service import new_search_service
from app.services.browserless_service import browserless_service
from app.core.search_config import SITE_CONFIGS
# Configure logging
logging.basicConfig(
level=logging.INFO,
level=logging.DEBUG, # Enable DEBUG logging
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
handlers=[logging.StreamHandler()]
)
# Set other loggers to INFO to avoid noise
logging.getLogger("urllib3").setLevel(logging.INFO)
logging.getLogger("asyncio").setLevel(logging.INFO)
logging.getLogger("websockets").setLevel(logging.INFO)
async def verify_search():
query = "iphone 15"
query = "chaise" # Updated query
print(f"--- Starting Verification Search for '{query}' ---")
# 1. Test Browserless Connection
print("\n[1] Testing Browserless Connection...")
try:
await browserless_service.start()
await browserless_service.initialize()
print("✅ Browserless connected successfully")
except Exception as e:
print(f"❌ Browserless connection failed: {e}")
return
# 2. Test Specific Sites
sites_to_test = ["gifi.fr", "amazon.fr", "e.leclerc"]
# sites_to_test = ["gifi.fr", "lincroyable.fr", "stokomani.fr"] # Excluded amazon.fr
sites_to_test = ["stokomani.fr"] # Focus on Stokomani for now as requested/implied context
for site in sites_to_test:
print(f"\n[2] Testing Search on {site}...")
try:
results = await new_search_service.search_site(site, query)
if results:
print(f"✅ Found {len(results)} results for {site}")
for r in results[:2]:
print(f" - {r.title} ({r.url})")
else:
count = 0
async for r in new_search_service.search_site_generator(site, query):
count += 1
print(f"✅ Found result: {r.title} ({r.url})")
print(f" Price: {r.price} {r.currency}")
print(f" Image: {r.image_url}")
if count >= 1:
break
if count == 0:
print(f"⚠️ No results found for {site}")
# Dump HTML for debugging
try:
content, _ = await browserless_service.get_page_content(
SITE_CONFIGS[site]["search_url"].format(query=query),
use_proxy=SITE_CONFIGS[site].get("requires_proxy", False),
wait_selector=SITE_CONFIGS[site].get("wait_selector")
)
with open(f"/app/debug_dumps/{site}_failed_verification.html", "w", encoding="utf-8") as f:
f.write(content)
print(f"📄 Saved HTML dump to /app/debug_dumps/{site}_failed_verification.html")
except Exception as dump_e:
print(f"❌ Failed to save HTML dump: {dump_e}")
except Exception as e:
print(f"❌ Error searching {site}: {e}")
# 3. Cleanup
await browserless_service.stop()
await browserless_service.shutdown()
print("\n--- Verification Complete ---")
if __name__ == "__main__":