feat: Add scheduled item tracking service with dedicated web scraping and data extraction capabilities.

This commit is contained in:
Michael committed 2025-12-18 13:23:53 +01:00
1 parent d50b18690e
commit 51ede950ff
3 files changed
+50 -22

No files matched your search

+22 -9
View File
@@ -107,22 +107,33 @@ async def process_item_check(item_id: int):
# Use independent ScraperService for tracking
from app.services.tracking_scraper_service import ScraperService
screenshot_path, html_content, final_url = await ScraperService.scrape_item(
screenshot_path, html_content, final_url, page_title = await ScraperService.scrape_item(
url=item_data["url"],
selector=item_data["selector"],
item_id=item_id,
return_html=True
)
# Detect Amazon login wall/redirection
if "amazon" in item_data["url"] and ("signin" in final_url or "captcha" in final_url):
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. URL redirected to: {final_url}")
await loop.run_in_executor(None, _update_db_error, item_id, "Amazon Bot Detection: Redirected to login/captcha")
return
if not screenshot_path:
raise Exception("Failed to capture screenshot")
# Detect Amazon login wall/redirection/block
is_amazon = "amazon" in item_data["url"]
is_blocked = False
if is_amazon:
login_terms = ["signin", "captcha", "s'identifier", "log in", "login"]
title_lower = page_title.lower()
if any(term in final_url.lower() for term in login_terms) or \
any(term in title_lower for term in login_terms) or \
"amazon.fr: s'identifier" in title_lower:
is_blocked = True
if is_blocked:
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. Title: {page_title}, URL: {final_url}")
await loop.run_in_executor(None, _update_db_error, item_id, f"Amazon Bot Detection: {page_title}")
return
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
# 1. Try specialized parser (if available) or JSON-LD
price = None
@@ -193,8 +204,10 @@ async def process_item_check(item_id: int):
)
else:
# Fallback to Vision AI if text extraction failed
# Use limited text context for Vision prompt to avoid token bloat
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
# Clean text properly before sending to AI
from app.utils.text import clean_text
cleaned_html = clean_text(html_content)
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=cleaned_html[:10000])):
raise Exception("AI analysis (Vision) failed")
extraction, metadata = ai_result
+18 -6
View File
@@ -130,10 +130,10 @@ class ScraperService:
item_id: int | None = None,
config: ScrapeConfig | None = None,
return_html: bool = False,
) -> tuple[str | None, str, str]:
) -> tuple[str | None, str, str, str]:
"""
Scrapes the given URL using Browserless and Playwright.
Returns a tuple: (screenshot_path, page_text_or_html, final_url)
Returns a tuple: (screenshot_path, page_text_or_html, final_url, page_title)
"""
if config is None:
config = ScrapeConfig()
@@ -145,16 +145,27 @@ class ScraperService:
# Ensure browser is connected and healthy
if not await ScraperService._ensure_browser_connected():
logger.error("Failed to establish browser connection")
return None, "", url
return None, "", url, ""
try:
context = await ScraperService._create_context(ScraperService._browser, url)
page = await context.new_page()
try:
# Random delay to simulate human lead-in
import random
await asyncio.sleep(random.uniform(0.5, 2.0))
await ScraperService._navigate_and_wait(page, url, timeout)
final_url = page.url
page_title = await page.title()
# Humanize: scroll a bit and back
await page.mouse.move(random.randint(100, 500), random.randint(100, 500))
await page.evaluate("window.scrollBy(0, 100)")
await asyncio.sleep(0.5)
await page.evaluate("window.scrollBy(0, -100)")
await ScraperService._handle_popups(page)
if selector:
@@ -172,19 +183,20 @@ class ScraperService:
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
return screenshot_path, content_data, final_url
return screenshot_path, content_data, final_url, page_title
finally:
await context.close()
except Exception as e:
logger.error(f"Error scraping {url}: {e}")
return None, "", url
return None, "", url, ""
@staticmethod
async def _connect_browser(p) -> Browser:
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
# Note: Added timeout for connection
return await p.chromium.connect_over_cdp(BROWSERLESS_URL, timeout=30000)
@staticmethod
async def _create_context(browser: Browser, url: str) -> BrowserContext:
+10 -7
View File
@@ -8,21 +8,24 @@ SNIPPET_CONTEXT_WINDOW = 100
def clean_text(text: str) -> str:
"""
Cleans the text by removing code blocks, HTML tags, and excessive whitespace.
Cleans the text by removing script/style content, code blocks, HTML tags, and excessive whitespace.
"""
if not text:
return ""
# Remove code blocks (```...```)
# 1. Remove script and style elements entirely (including content)
text = re.sub(r'<(script|style|header|footer|nav)[\s\S]*?>[\s\S]*?<\/\1>', '', text, flags=re.IGNORECASE)
# 2. Remove other HTML tags (basic) but keep content
text = re.sub(r"<[^>]+>", " ", text)
# 3. Remove code blocks (```...```)
text = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
# Remove HTML tags (basic)
text = re.sub(r"<[^>]+>", "", text)
# Remove non-printable characters (keep newlines and tabs)
# 4. Remove non-printable characters (keep newlines and tabs)
text = re.sub(r"[^\x20-\x7E\n\t]", "", text)
# Collapse excessive whitespace
# 5. Collapse excessive whitespace
text = re.sub(r"\s+", " ", text).strip()
return text