feat: Introduce AI extraction schema with validation, prompt generation, and a verification script to improve price extraction reliability.

This commit is contained in:
Michael committed 2025-12-18 16:32:32 +01:00
1 parent ef7cd7c1cf
commit 384baa3575
4 files changed
+230 -5

No files matched your search

+4 -2
View File
@@ -141,7 +141,9 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
- Look for: "PRIX DÉTECTÉ:", price tags, "Prix:", "€", numbers near "Ajouter au panier"
- Extract as DECIMAL NUMBER: If you see "3.99", return 3.99
- Ignore crossed-out/barré prices (old prices)
- If multiple prices, take the current/main price (not the original)
- **CRITICAL:** Ignore "Prix au litre", "Prix au kg", "P.U.", or unit prices usually shown in smaller text/parentheses (e.g., "4.60 € / L").
- If multiple prices, take the current/main price (not the original, not the unit price)
- **B&M STORES Specific:** The main price is often large and bold (e.g. "1.15€"), while unit price is small (e.g. "4.60 €/L"). ALWAYS take the main price.
**CRITICAL - Common mistakes to avoid:**
- "3.99 €" means 3.99 (NOT 399.00, NOT 3990.00)
@@ -161,7 +163,7 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this
- If unclear: set null and confidence < 0.5
**STOCK:**
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart"
- TRUE if: "Ajouter au panier", "Acheter", "En stock", "Disponible", "Add to Cart", "Retrait 2h", "Click & Collect"
- FALSE if: "Rupture", "Indisponible", "Épuisé", "Out of Stock", "Notify Me"
- NULL if unclear or not shown
+34 -3
View File
@@ -416,9 +416,40 @@ class ScraperService:
try:
logger.info(f"Extracting text (limit: {text_length} chars)...")
raw_text = await page.inner_text("body")
page_text = raw_text[:text_length]
logger.info(f"Extracted {len(page_text)} characters")
# Smart extraction: remove noise (nav, footer, scripts) before getting text
clean_text = await page.evaluate("""
() => {
// Clone body to not affect the visual page
const clone = document.body.cloneNode(true);
// Remove noise selectors
const noiseSelectors = [
'nav', 'header', 'footer', 'script', 'style', 'noscript', 'iframe',
'.cookie-banner', '.popup', '#menu', '.menu', '.sidebar',
'[role="navigation"]', '[role="banner"]', '[role="contentinfo"]'
];
noiseSelectors.forEach(selector => {
const elements = clone.querySelectorAll(selector);
elements.forEach(el => el.remove());
});
return clone.innerText;
}
""")
# Fallback if cleaning removed everything (unlikely but safe)
if not clean_text or len(clean_text) < 100:
logger.warning("Cleaned text too short, falling back to full body text")
clean_text = await page.inner_text("body")
# Collapse whitespace
import re
clean_text = re.sub(r'\s+', ' ', clean_text).strip()
page_text = clean_text[:text_length]
logger.info(f"Extracted {len(page_text)} chars")
return page_text
except Exception as e:
logger.error(f"Text extraction failed: {e}")
+12
View File
@@ -18,6 +18,15 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
- [x] Update `ItemService.delete_item` to clean up all related screenshots <!-- id: 6 -->
- [x] Create `verify_fix_item_service.py` to test the new logic <!-- id: 5 -->
## Current Focus
Improving price extraction reliability for B&M Stores and others.
- [x] Analyze `ai_schema.py` to check the extraction prompt <!-- id: 7 -->
- [x] Improve `ScraperService._extract_text` to be more targeted (e.g. main content only) <!-- id: 8 -->
- [x] Update AI prompt to better handle multiple prices (unit vs package) <!-- id: 9 -->
- [x] Verify extraction logic with simulation script <!-- id: 10 -->
## Progress Log
- Identified the issue: `ScraperService` overwrites `item_{id}.png`.
@@ -25,3 +34,6 @@ Implementing logic to generate unique timestamped filenames for screenshots to b
- User approved plan.
- Implemented filesystem scanning in `ItemService`.
- Verified fix with `verify_fix_item_service.py` successfully.
- Improved AI prompt to ignore unit prices (like "Prix au litre").
- Enhanced text extraction to remove menu/footer noise.
- Verified logic with `verify_extraction_logic.py` (simulated).
+180
View File
@@ -0,0 +1,180 @@
import asyncio
import logging
import sys
from unittest.mock import MagicMock, AsyncMock
# Mock sqlalchemy
sys.modules["sqlalchemy"] = MagicMock()
sys.modules["sqlalchemy.orm"] = MagicMock()
# Mock pydantic
mock_pydantic = MagicMock()
# Mock BaseModel
class MockBaseModel:
def __init__(self, **kwargs):
for k, v in kwargs.items():
setattr(self, k, v)
mock_pydantic.BaseModel = MockBaseModel
mock_pydantic.Field = MagicMock(return_value=None)
mock_pydantic.field_validator = MagicMock(return_value=lambda x: x)
sys.modules["pydantic"] = mock_pydantic
# Mock app.utils.text which is imported by ai_schema
mock_utils_text = MagicMock()
sys.modules["app.utils.text"] = mock_utils_text
mock_utils_text.filter_relevant_text = lambda text, max_length: text[:max_length]
mock_utils_text.clean_text = lambda text: text.strip()
# Mock app.strings (if used) or other utils
sys.modules["app.utils"] = MagicMock()
# Mock app.utils.image
sys.modules["app.utils.image"] = MagicMock()
# Mock app.database
sys.modules["app.database"] = MagicMock()
# Mock playwright
mock_playwright = MagicMock()
sys.modules["playwright"] = mock_playwright
sys.modules["playwright.async_api"] = mock_playwright
# Mock generic types for type hints if needed
mock_playwright.Browser = MagicMock
mock_playwright.BrowserContext = MagicMock
mock_playwright.Page = MagicMock
mock_playwright.TimeoutError = Exception
# Now import the schema
from app.ai_schema import get_extraction_prompt, get_repair_prompt
# We can't import ScraperService easily if it inherits from things or uses decorators
# But for this test we only need get_extraction_prompt which is in ai_schema
# So we can skip importing ScraperService if it causes issues,
# BUT we wanted to verify ScraperService text cleaning logic...
# Let's mock ScraperService dependencies completely.
try:
from app.services.tracking_scraper_service import ScraperService
except ImportError:
print("Warning: Could not import ScraperService due to dependencies. Skipping Service tests.")
ScraperService = None
# Mock litellm and tenacity
sys.modules["litellm"] = MagicMock()
sys.modules["tenacity"] = MagicMock()
mock_retry = MagicMock()
sys.modules["tenacity.retry"] = mock_retry
# Make sure imports inside ai_service don't fail
# It imports: retry, retry_if_exception_type, stop_after_attempt, wait_exponential from tenacity
# We need to mock these specifically if the module imports them directly
mock_tenacity = MagicMock()
mock_tenacity.retry = lambda *args, **kwargs: lambda f: f
mock_tenacity.retry_if_exception_type = MagicMock()
mock_tenacity.stop_after_attempt = MagicMock()
mock_tenacity.wait_exponential = MagicMock()
sys.modules["tenacity"] = mock_tenacity
# Now import AIService
# We will mock the AI response to verify the parsing logic
from app.services.ai_service import AIService
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
async def verify_extraction_logic():
print("Verifying Extraction Logic...")
# 1. Test Text Cleaning in ScraperService
# We can't mock Playwright page easily in a simple script without launching a browser.
# But we can test the AI prompt generation which is critical.
# Simulate B&M text
dirty_text = """
Menu
Accueil
Panier
Boisson energisante ice 25cl
Red Bull
1.15 €
Prix au litre : 4,60 € / L
En stock
Ajouter au panier
Footer
Mentions légales
"""
print("\n--- Testing Prompt Generation ---")
prompt = get_extraction_prompt(dirty_text)
# Verify strict instructions are present
checks = [
"CRITICAL",
"Ignore \"Prix au litre\"",
"B&M STORES Specific",
"Extract as DECIMAL NUMBER"
]
all_passed = True
for check in checks:
if check in prompt:
print(f"[OK] Prompt contains: {check}")
else:
print(f"[FAIL] Prompt missing: {check}")
all_passed = False
if not all_passed:
print("Prompt verification failed!")
exit(1)
print("\n--- Testing Response Parsing (Mock AI) ---")
# Case 1: AI returns Main Price correctly
mock_response_1 = """
```json
{
"price": 1.15,
"currency": "EUR",
"in_stock": true,
"price_confidence": 0.95,
"in_stock_confidence": 1.0,
"source_type": "text"
}
```
"""
result = AIService.parse_and_validate_response(mock_response_1)
if result.price == 1.15 and result.in_stock is True:
print("[OK] Parsed correct mocked response.")
else:
print(f"[FAIL] Failed to parse correct response: {result}")
exit(1)
# Case 2: AI returns confusion (simulating what we want to avoid, but checking schema resilience)
# If AI returns explicit null because it's confused
mock_response_2 = """
{
"price": null,
"currency": "EUR",
"in_stock": null,
"price_confidence": 0.0,
"in_stock_confidence": 0.0,
"source_type": "image"
}
"""
result = AIService.parse_and_validate_response(mock_response_2)
if result.price is None:
print("[OK] Parsed null response correctly.")
else:
print(f"[FAIL] Failed to parse null response.")
print("\nVerification of Logic Flow Complete (Simulated).")
print("Real-world verification requires running the full scraper.")
if __name__ == "__main__":
asyncio.run(verify_extraction_logic())