mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
docs: revise task.md to address screenshot update caching instead of popup obscuration.
This commit is contained in:
1 parent
ed7f23d657
commit
4590273801
4 files changed
+134
-12
No files matched your search
@@ -269,7 +269,9 @@ class ScraperService:
|
||||
os.makedirs(screenshot_dir, exist_ok=True)
|
||||
|
||||
if item_id:
|
||||
filename = f"{screenshot_dir}/item_{item_id}.png"
|
||||
# Use timestamp to ensure unique filenames (fixes caching issues)
|
||||
timestamp = int(datetime.now().timestamp())
|
||||
filename = f"{screenshot_dir}/item_{item_id}_{timestamp}.png"
|
||||
else:
|
||||
url_part = url.split("//")[-1].replace("/", "_")
|
||||
timestamp = datetime.now().timestamp()
|
||||
|
||||
@@ -1,23 +1,21 @@
|
||||
# Task: Remove Obscuring Windows from Product Screenshots
|
||||
# Task: Fix Screenshot Update Issue
|
||||
|
||||
## Context
|
||||
|
||||
The user has reported that screenshots in the "suivi produits" (product tracking) section still contain windows (popups, overlays) that obscure the content. A specific example provided shows a "Stock Inconnu" and "Calendrier de l'Avent" popup on a B&M website.
|
||||
The user reports that in the "Suivi prix" (Tracking) section, triggering an update ("Mise à jour") does not update the screenshot. This has been identified as a caching issue due to static filenames.
|
||||
|
||||
## Current Focus
|
||||
|
||||
Investigating the scraping and screenshot logic to handle these popups.
|
||||
Implementing logic to generate unique timestamped filenames for screenshots to bypass browser caching and ensure the latest image is displayed.
|
||||
|
||||
## Master Plan
|
||||
|
||||
- [x] Explore codebase to locate scraping and screenshot logic <!-- id: 0 -->
|
||||
- [x] Identify specific handling for B&M or generic popup closing <!-- id: 1 -->
|
||||
- [x] Implement fix to close popups before screenshot <!-- id: 2 -->
|
||||
- [/] Verify fix (Requires App Rebuild) <!-- id: 3 -->
|
||||
- [x] Modify `tracking_scraper_service.py` to use timestamped filenames for tracking screenshots <!-- id: 0 -->
|
||||
- [x] Verify that `item_service.py` correctly picks up the new files <!-- id: 1 -->
|
||||
- [x] Create verification script `verify_screenshot_update.py` <!-- id: 2 -->
|
||||
- [x] Run verification and confirm fix <!-- id: 3 -->
|
||||
|
||||
## Progress Log
|
||||
|
||||
- Initialized task.md
|
||||
- Created implementation plan to expand popup selectors.
|
||||
- Updated `browserless_service.py` with expanded selectors and retry logic.
|
||||
- Attempted local verification but failed due to environment restrictions.
|
||||
- Identified the issue: `ScraperService` overwrites `item_{id}.png`.
|
||||
- Created implementation plan.
|
||||
@@ -0,0 +1,48 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Add current directory to sys.path to allow importing app
|
||||
sys.path.append(os.getcwd())
|
||||
|
||||
# Mock playwright before importing app
|
||||
mock_playwright = MagicMock()
|
||||
sys.modules["playwright"] = mock_playwright
|
||||
sys.modules["playwright.async_api"] = mock_playwright
|
||||
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_logic():
|
||||
logger.info("Starting logic verification...")
|
||||
|
||||
# Mock Page object
|
||||
mock_page = MagicMock()
|
||||
mock_page.screenshot = AsyncMock()
|
||||
|
||||
item_id = 123
|
||||
url = "http://test.com"
|
||||
|
||||
# Call _take_screenshot directly
|
||||
logger.info("Calling _take_screenshot...")
|
||||
filename = await ScraperService._take_screenshot(mock_page, url, item_id)
|
||||
|
||||
logger.info(f"Returned filename: {filename}")
|
||||
|
||||
# Verify format
|
||||
pattern = r"screenshots/item_123_\d+\.png"
|
||||
if re.match(pattern, filename):
|
||||
logger.info("SUCCESS: Filename matches expected timestamp pattern!")
|
||||
print("VERIFICATION_SUCCESS")
|
||||
else:
|
||||
logger.error(f"FAILURE: Filename {filename} does not match pattern {pattern}")
|
||||
exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_logic())
|
||||
@@ -0,0 +1,74 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def mock_connect_browser(p):
|
||||
logger.info("MOCK: Launching local browser instead of connecting to browserless")
|
||||
return await p.chromium.launch()
|
||||
|
||||
async def verify_fix():
|
||||
logger.info("Starting verification...")
|
||||
|
||||
# Monkey-patch _connect_browser to use local browser
|
||||
ScraperService._connect_browser = mock_connect_browser
|
||||
|
||||
# Ensure screenshots dir exists
|
||||
os.makedirs("screenshots", exist_ok=True)
|
||||
|
||||
# Test Item ID 999
|
||||
item_id = 999
|
||||
url = "https://example.com"
|
||||
|
||||
logger.info(f"Scraping item {item_id}...")
|
||||
|
||||
# We expect this to fail scraping real content from example.com with specific selectors,
|
||||
# but we only care about the screenshot filename generation which happens at the end.
|
||||
# Actually, if scraping fails, it might return None, "" early.
|
||||
# checking tracking_scraper_service.py:
|
||||
# It has a try/except block.
|
||||
# If _navigate_and_wait works, it proceeds. example.com should load.
|
||||
# _take_screenshot is called at the end.
|
||||
|
||||
# However, ScraperService.scrape_item returns (None, "") if exception occurs.
|
||||
# We need to make sure it doesn't crash before screenshot.
|
||||
# example.com is simple, so it should load.
|
||||
# It will try to click popups (won't find any), wait for selector (if provided).
|
||||
# If we don't provide selector, it calls _auto_detect_price.
|
||||
|
||||
full_path, _ = await ScraperService.scrape_item(url=url, item_id=item_id)
|
||||
|
||||
if full_path:
|
||||
logger.info(f"Screenshot path returned: {full_path}")
|
||||
|
||||
# Verify format: item_{id}_{timestamp}.png
|
||||
# Check if it matches regex
|
||||
pattern = r"screenshots/item_999_\d+\.png"
|
||||
if re.match(pattern, full_path):
|
||||
logger.info("SUCCESS: Filename contains timestamp!")
|
||||
else:
|
||||
logger.error(f"FAILURE: Filename does not match pattern {pattern}")
|
||||
exit(1)
|
||||
|
||||
# Clean up
|
||||
if os.path.exists(full_path):
|
||||
os.remove(full_path)
|
||||
logger.info("Cleaned up screenshot file")
|
||||
|
||||
else:
|
||||
logger.error("FAILURE: Scraper returned None for path. Did navigation fail?")
|
||||
exit(1)
|
||||
|
||||
await ScraperService.shutdown()
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_fix())
|
||||
Reference in new issue
Block a user