mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Introduce AI extraction schema with validation, prompt generation, and a verification script to improve price extraction reliability.
This commit is contained in:
1 parent
ef7cd7c1cf
commit
384baa3575
4 files changed
+230
-5
No files matched your search
+4
-2
@@ -141,7 +141,9 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
|
||||
- Look for: "PRIX DÉTECTÉ:", price tags, "Prix:", "€", numbers near "Ajouter au panier"
|
||||
- Extract as DECIMAL NUMBER: If you see "3.99", return 3.99
|
||||
- Ignore crossed-out/barré prices (old prices)
|
||||
- If multiple prices, take the current/main price (not the original)
|
||||
- **CRITICAL:** Ignore "Prix au litre", "Prix au kg", "P.U.", or unit prices usually shown in smaller text/parentheses (e.g., "4.60 € / L").
|
||||
- If multiple prices, take the current/main price (not the original, not the unit price)
|
||||
- **B&M STORES Specific:** The main price is often large and bold (e.g. "1.15€"), while unit price is small (e.g. "4.60 €/L"). ALWAYS take the main price.
|
||||
|
||||
**CRITICAL - Common mistakes to avoid:**
|
||||
- "3.99 €" means 3.99 (NOT 399.00, NOT 3990.00)
|
||||
@@ -161,7 +163,7 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
|
||||
- If unclear: set null and confidence < 0.5
|
||||
|
||||
**STOCK:**
|
||||
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart"
|
||||
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart", "Retrait 2h", "Click & Collect"
|
||||
- FALSE if: "Rupture", "Indisponible", "Épuisé", "Out of Stock", "Notify Me"
|
||||
- NULL if unclear or not shown
|
||||
|
||||
|
||||
@@ -416,9 +416,40 @@ class ScraperService:
|
||||
|
||||
try:
|
||||
logger.info(f"Extracting text (limit: {text_length} chars)...")
|
||||
raw_text = await page.inner_text("body")
|
||||
page_text = raw_text[:text_length]
|
||||
logger.info(f"Extracted {len(page_text)} characters")
|
||||
|
||||
# Smart extraction: remove noise (nav, footer, scripts) before getting text
|
||||
clean_text = await page.evaluate("""
|
||||
() => {
|
||||
// Clone body to not affect the visual page
|
||||
const clone = document.body.cloneNode(true);
|
||||
|
||||
// Remove noise selectors
|
||||
const noiseSelectors = [
|
||||
'nav', 'header', 'footer', 'script', 'style', 'noscript', 'iframe',
|
||||
'.cookie-banner', '.popup', '#menu', '.menu', '.sidebar',
|
||||
'[role="navigation"]', '[role="banner"]', '[role="contentinfo"]'
|
||||
];
|
||||
|
||||
noiseSelectors.forEach(selector => {
|
||||
const elements = clone.querySelectorAll(selector);
|
||||
elements.forEach(el => el.remove());
|
||||
});
|
||||
|
||||
return clone.innerText;
|
||||
}
|
||||
""")
|
||||
|
||||
# Fallback if cleaning removed everything (unlikely but safe)
|
||||
if not clean_text or len(clean_text) < 100:
|
||||
logger.warning("Cleaned text too short, falling back to full body text")
|
||||
clean_text = await page.inner_text("body")
|
||||
|
||||
# Collapse whitespace
|
||||
import re
|
||||
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
|
||||
|
||||
page_text = clean_text[:text_length]
|
||||
logger.info(f"Extracted {len(page_text)} chars")
|
||||
return page_text
|
||||
except Exception as e:
|
||||
logger.error(f"Text extraction failed: {e}")
|
||||
|
||||
@@ -18,6 +18,15 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
|
||||
- [x] Update `ItemService.delete_item` to clean up all related screenshots <!-- id: 6 -->
|
||||
- [x] Create `verify_fix_item_service.py` to test the new logic <!-- id: 5 -->
|
||||
|
||||
## Current Focus
|
||||
|
||||
Improving price extraction reliability for B&M Stores and others.
|
||||
|
||||
- [x] Analyze `ai_schema.py` to check the extraction prompt <!-- id: 7 -->
|
||||
- [x] Improve `ScraperService._extract_text` to be more targeted (e.g. main content only) <!-- id: 8 -->
|
||||
- [x] Update AI prompt to better handle multiple prices (unit vs package) <!-- id: 9 -->
|
||||
- [x] Verify extraction logic with simulation script <!-- id: 10 -->
|
||||
|
||||
## Progress Log
|
||||
|
||||
- Identified the issue: `ScraperService` overwrites `item_{id}.png`.
|
||||
@@ -25,3 +34,6 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
|
||||
- User approved plan.
|
||||
- Implemented filesystem scanning in `ItemService`.
|
||||
- Verified fix with `verify_fix_item_service.py` successfully.
|
||||
- Improved AI prompt to ignore unit prices (like "Prix au litre").
|
||||
- Enhanced text extraction to remove menu/footer noise.
|
||||
- Verified logic with `verify_extraction_logic.py` (simulated).
|
||||
@@ -0,0 +1,180 @@
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock
|
||||
|
||||
# Mock sqlalchemy
|
||||
sys.modules["sqlalchemy"] = MagicMock()
|
||||
sys.modules["sqlalchemy.orm"] = MagicMock()
|
||||
|
||||
# Mock pydantic
|
||||
mock_pydantic = MagicMock()
|
||||
# Mock BaseModel
|
||||
class MockBaseModel:
|
||||
def __init__(self, **kwargs):
|
||||
for k, v in kwargs.items():
|
||||
setattr(self, k, v)
|
||||
mock_pydantic.BaseModel = MockBaseModel
|
||||
mock_pydantic.Field = MagicMock(return_value=None)
|
||||
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
|
||||
|
||||
sys.modules["pydantic"] = mock_pydantic
|
||||
|
||||
# Mock app.utils.text which is imported by ai_schema
|
||||
mock_utils_text = MagicMock()
|
||||
sys.modules["app.utils.text"] = mock_utils_text
|
||||
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
|
||||
mock_utils_text.clean_text = lambda text: text.strip()
|
||||
|
||||
# Mock app.strings (if used) or other utils
|
||||
sys.modules["app.utils"] = MagicMock()
|
||||
|
||||
# Mock app.utils.image
|
||||
sys.modules["app.utils.image"] = MagicMock()
|
||||
|
||||
# Mock app.database
|
||||
sys.modules["app.database"] = MagicMock()
|
||||
|
||||
# Mock playwright
|
||||
mock_playwright = MagicMock()
|
||||
sys.modules["playwright"] = mock_playwright
|
||||
sys.modules["playwright.async_api"] = mock_playwright
|
||||
|
||||
# Mock generic types for type hints if needed
|
||||
mock_playwright.Browser = MagicMock
|
||||
mock_playwright.BrowserContext = MagicMock
|
||||
mock_playwright.Page = MagicMock
|
||||
mock_playwright.TimeoutError = Exception
|
||||
|
||||
# Now import the schema
|
||||
from app.ai_schema import get_extraction_prompt, get_repair_prompt
|
||||
|
||||
# We can't import ScraperService easily if it inherits from things or uses decorators
|
||||
# But for this test we only need get_extraction_prompt which is in ai_schema
|
||||
# So we can skip importing ScraperService if it causes issues,
|
||||
# BUT we wanted to verify ScraperService text cleaning logic...
|
||||
# Let's mock ScraperService dependencies completely.
|
||||
|
||||
try:
|
||||
from app.services.tracking_scraper_service import ScraperService
|
||||
except ImportError:
|
||||
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
|
||||
ScraperService = None
|
||||
|
||||
# Mock litellm and tenacity
|
||||
sys.modules["litellm"] = MagicMock()
|
||||
sys.modules["tenacity"] = MagicMock()
|
||||
mock_retry = MagicMock()
|
||||
sys.modules["tenacity.retry"] = mock_retry
|
||||
|
||||
# Make sure imports inside ai_service don't fail
|
||||
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
|
||||
# We need to mock these specifically if the module imports them directly
|
||||
mock_tenacity = MagicMock()
|
||||
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
|
||||
mock_tenacity.retry_if_exception_type = MagicMock()
|
||||
mock_tenacity.stop_after_attempt = MagicMock()
|
||||
mock_tenacity.wait_exponential = MagicMock()
|
||||
sys.modules["tenacity"] = mock_tenacity
|
||||
|
||||
# Now import AIService
|
||||
# We will mock the AI response to verify the parsing logic
|
||||
from app.services.ai_service import AIService
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
async def verify_extraction_logic():
|
||||
print("Verifying Extraction Logic...")
|
||||
|
||||
# 1. Test Text Cleaning in ScraperService
|
||||
# We can't mock Playwright page easily in a simple script without launching a browser.
|
||||
# But we can test the AI prompt generation which is critical.
|
||||
|
||||
# Simulate B&M text
|
||||
dirty_text = """
|
||||
Menu
|
||||
Accueil
|
||||
Panier
|
||||
|
||||
Boisson energisante ice 25cl
|
||||
Red Bull
|
||||
|
||||
1.15 €
|
||||
Prix au litre : 4,60 € / L
|
||||
|
||||
En stock
|
||||
Ajouter au panier
|
||||
|
||||
Footer
|
||||
Mentions légales
|
||||
"""
|
||||
|
||||
print("\n--- Testing Prompt Generation ---")
|
||||
prompt = get_extraction_prompt(dirty_text)
|
||||
|
||||
# Verify strict instructions are present
|
||||
checks = [
|
||||
"CRITICAL",
|
||||
"Ignore \"Prix au litre\"",
|
||||
"B&M STORES Specific",
|
||||
"Extract as DECIMAL NUMBER"
|
||||
]
|
||||
|
||||
all_passed = True
|
||||
for check in checks:
|
||||
if check in prompt:
|
||||
print(f"[OK] Prompt contains: {check}")
|
||||
else:
|
||||
print(f"[FAIL] Prompt missing: {check}")
|
||||
all_passed = False
|
||||
|
||||
if not all_passed:
|
||||
print("Prompt verification failed!")
|
||||
exit(1)
|
||||
|
||||
print("\n--- Testing Response Parsing (Mock AI) ---")
|
||||
|
||||
# Case 1: AI returns Main Price correctly
|
||||
mock_response_1 = """
|
||||
```json
|
||||
{
|
||||
"price": 1.15,
|
||||
"currency": "EUR",
|
||||
"in_stock": true,
|
||||
"price_confidence": 0.95,
|
||||
"in_stock_confidence": 1.0,
|
||||
"source_type": "text"
|
||||
}
|
||||
```
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_1)
|
||||
if result.price == 1.15 and result.in_stock is True:
|
||||
print("[OK] Parsed correct mocked response.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse correct response: {result}")
|
||||
exit(1)
|
||||
|
||||
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
|
||||
# If AI returns explicit null because it's confused
|
||||
mock_response_2 = """
|
||||
{
|
||||
"price": null,
|
||||
"currency": "EUR",
|
||||
"in_stock": null,
|
||||
"price_confidence": 0.0,
|
||||
"in_stock_confidence": 0.0,
|
||||
"source_type": "image"
|
||||
}
|
||||
"""
|
||||
result = AIService.parse_and_validate_response(mock_response_2)
|
||||
if result.price is None:
|
||||
print("[OK] Parsed null response correctly.")
|
||||
else:
|
||||
print(f"[FAIL] Failed to parse null response.")
|
||||
|
||||
print("\nVerification of Logic Flow Complete (Simulated).")
|
||||
print("Real-world verification requires running the full scraper.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(verify_extraction_logic())
|
||||
Reference in new issue
Block a user