mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Add scheduled item tracking service with dedicated web scraping and data extraction capabilities.
This commit is contained in:
1 parent
d50b18690e
commit
51ede950ff
3 files changed
+50
-22
No files matched your search
@@ -107,22 +107,33 @@ async def process_item_check(item_id: int):
|
||||
# Use independent ScraperService for tracking
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
screenshot_path, html_content, final_url = await ScraperService.scrape_item(
|
||||
screenshot_path, html_content, final_url, page_title = await ScraperService.scrape_item(
|
||||
url=item_data["url"],
|
||||
selector=item_data["selector"],
|
||||
item_id=item_id,
|
||||
return_html=True
|
||||
)
|
||||
|
||||
# Detect Amazon login wall/redirection
|
||||
if "amazon" in item_data["url"] and ("signin" in final_url or "captcha" in final_url):
|
||||
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. URL redirected to: {final_url}")
|
||||
await loop.run_in_executor(None, _update_db_error, item_id, "Amazon Bot Detection: Redirected to login/captcha")
|
||||
return
|
||||
|
||||
if not screenshot_path:
|
||||
raise Exception("Failed to capture screenshot")
|
||||
|
||||
# Detect Amazon login wall/redirection/block
|
||||
is_amazon = "amazon" in item_data["url"]
|
||||
is_blocked = False
|
||||
|
||||
if is_amazon:
|
||||
login_terms = ["signin", "captcha", "s'identifier", "log in", "login"]
|
||||
title_lower = page_title.lower()
|
||||
if any(term in final_url.lower() for term in login_terms) or \
|
||||
any(term in title_lower for term in login_terms) or \
|
||||
"amazon.fr: s'identifier" in title_lower:
|
||||
is_blocked = True
|
||||
|
||||
if is_blocked:
|
||||
logger.warning(f"Amazon Bot Detection triggered for item {item_id}. Title: {page_title}, URL: {final_url}")
|
||||
await loop.run_in_executor(None, _update_db_error, item_id, f"Amazon Bot Detection: {page_title}")
|
||||
return
|
||||
|
||||
# HYBRID EXTRACTION STRATEGY (Aligned with ImprovedSearchService)
|
||||
# 1. Try specialized parser (if available) or JSON-LD
|
||||
price = None
|
||||
@@ -193,8 +204,10 @@ async def process_item_check(item_id: int):
|
||||
)
|
||||
else:
|
||||
# Fallback to Vision AI if text extraction failed
|
||||
# Use limited text context for Vision prompt to avoid token bloat
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=html_content[:5000])):
|
||||
# Clean text properly before sending to AI
|
||||
from app.utils.text import clean_text
|
||||
cleaned_html = clean_text(html_content)
|
||||
if not (ai_result := await AIService.analyze_image(screenshot_path, page_text=cleaned_html[:10000])):
|
||||
raise Exception("AI analysis (Vision) failed")
|
||||
extraction, metadata = ai_result
|
||||
|
||||
|
||||
@@ -130,10 +130,10 @@ class ScraperService:
|
||||
item_id: int | None = None,
|
||||
config: ScrapeConfig | None = None,
|
||||
return_html: bool = False,
|
||||
) -> tuple[str | None, str, str]:
|
||||
) -> tuple[str | None, str, str, str]:
|
||||
"""
|
||||
Scrapes the given URL using Browserless and Playwright.
|
||||
Returns a tuple: (screenshot_path, page_text_or_html, final_url)
|
||||
Returns a tuple: (screenshot_path, page_text_or_html, final_url, page_title)
|
||||
"""
|
||||
if config is None:
|
||||
config = ScrapeConfig()
|
||||
@@ -145,16 +145,27 @@ class ScraperService:
|
||||
# Ensure browser is connected and healthy
|
||||
if not await ScraperService._ensure_browser_connected():
|
||||
logger.error("Failed to establish browser connection")
|
||||
return None, "", url
|
||||
return None, "", url, ""
|
||||
|
||||
try:
|
||||
context = await ScraperService._create_context(ScraperService._browser, url)
|
||||
page = await context.new_page()
|
||||
|
||||
try:
|
||||
# Random delay to simulate human lead-in
|
||||
import random
|
||||
await asyncio.sleep(random.uniform(0.5, 2.0))
|
||||
|
||||
await ScraperService._navigate_and_wait(page, url, timeout)
|
||||
final_url = page.url
|
||||
page_title = await page.title()
|
||||
|
||||
# Humanize: scroll a bit and back
|
||||
await page.mouse.move(random.randint(100, 500), random.randint(100, 500))
|
||||
await page.evaluate("window.scrollBy(0, 100)")
|
||||
await asyncio.sleep(0.5)
|
||||
await page.evaluate("window.scrollBy(0, -100)")
|
||||
|
||||
await ScraperService._handle_popups(page)
|
||||
|
||||
if selector:
|
||||
@@ -172,19 +183,20 @@ class ScraperService:
|
||||
|
||||
screenshot_path = await ScraperService._take_screenshot(page, url, item_id)
|
||||
|
||||
return screenshot_path, content_data, final_url
|
||||
return screenshot_path, content_data, final_url, page_title
|
||||
|
||||
finally:
|
||||
await context.close()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error scraping {url}: {e}")
|
||||
return None, "", url
|
||||
return None, "", url, ""
|
||||
|
||||
@staticmethod
|
||||
async def _connect_browser(p) -> Browser:
|
||||
logger.info(f"Connecting to Browserless at {BROWSERLESS_URL}")
|
||||
return await p.chromium.connect_over_cdp(BROWSERLESS_URL)
|
||||
# Note: Added timeout for connection
|
||||
return await p.chromium.connect_over_cdp(BROWSERLESS_URL, timeout=30000)
|
||||
|
||||
@staticmethod
|
||||
async def _create_context(browser: Browser, url: str) -> BrowserContext:
|
||||
|
||||
+10
-7
@@ -8,21 +8,24 @@ SNIPPET_CONTEXT_WINDOW = 100
|
||||
|
||||
def clean_text(text: str) -> str:
|
||||
"""
|
||||
Cleans the text by removing code blocks, HTML tags, and excessive whitespace.
|
||||
Cleans the text by removing script/style content, code blocks, HTML tags, and excessive whitespace.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
# Remove code blocks (```...```)
|
||||
# 1. Remove script and style elements entirely (including content)
|
||||
text = re.sub(r'<(script|style|header|footer|nav)[\s\S]*?>[\s\S]*?<\/\1>', '', text, flags=re.IGNORECASE)
|
||||
|
||||
# 2. Remove other HTML tags (basic) but keep content
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
|
||||
# 3. Remove code blocks (```...```)
|
||||
text = re.sub(r"```.*?```", "", text, flags=re.DOTALL)
|
||||
|
||||
# Remove HTML tags (basic)
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
|
||||
# Remove non-printable characters (keep newlines and tabs)
|
||||
# 4. Remove non-printable characters (keep newlines and tabs)
|
||||
text = re.sub(r"[^\x20-\x7E\n\t]", "", text)
|
||||
|
||||
# Collapse excessive whitespace
|
||||
# 5. Collapse excessive whitespace
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
return text
|
||||
|
||||
Reference in new issue
Block a user