diff --git a/app/ai_schema.py b/app/ai_schema.py index a3a67dc..65621d1 100644 --- a/app/ai_schema.py +++ b/app/ai_schema.py @@ -141,11 +141,21 @@ EXTRACTION_PROMPT_TEMPLATE = """Extract product price and stock status from this - Extract as DECIMAL NUMBER (convert comma to dot): 12,99 -> 12.99 - Ignore crossed-out/barré prices (old prices) - If multiple prices, take the current/main price (not the original) + +**CRITICAL - Read digits carefully:** +- PAY CLOSE ATTENTION to first digit: "1,99 €" is NOT "9,99 €" +- Double-check: Is it 1, 7, 8, or 9? These look similar +- Small prices (< 5€) are common: 0,99 €, 1,49 €, 1,99 €, 2,99 €, 3,99 €, 4,99 € +- VERIFY the price makes sense for the product type + - Examples of valid prices: + * "1,99 €" -> 1.99 (NOT 9.99) + * "0,99 €" -> 0.99 (NOT 9.99) * "89,99 €" -> 89.99 * "1 234,56 €" -> 1234.56 * "PRIX DÉTECTÉ: 89,99 €" -> 89.99 - If you find ANY price with € symbol, extract it with confidence >= 0.8 +- If digits are unclear or blurry, reduce confidence to 0.5-0.7 - If unclear: set null and confidence < 0.5 **STOCK:** diff --git a/app/services/ai_service.py b/app/services/ai_service.py index 60096ae..71da37a 100644 --- a/app/services/ai_service.py +++ b/app/services/ai_service.py @@ -343,14 +343,22 @@ class AIService: if len(cleaned_text) > MAX_TEXT_LENGTH: cleaned_text = cleaned_text[:MAX_TEXT_LENGTH] + "...(truncated)" logger.info(f"Added text context (original: {len(page_text)}, cleaned: {len(cleaned_text)})") - # Debug: Log first 500 chars of cleaned text to see what AI receives - logger.debug(f"Cleaned text preview: {cleaned_text[:500]!r}") + # Log first 500 chars of cleaned text to see what AI receives + logger.info(f"Cleaned text preview: {cleaned_text[:500]!r}") + + # Extract all potential prices from text for debugging + import re + price_patterns = re.findall(r'\d+[,\.]\d{2}\s*€', cleaned_text) + if price_patterns: + logger.info(f"Prices found in text: {price_patterns[:10]}") # First 10 prices + else: + logger.warning("No prices found in text with € symbol") else: logger.warning("No page_text provided - AI will only use screenshot") prompt = get_extraction_prompt(cleaned_text if cleaned_text else None) - # Debug: Log prompt preview - logger.debug(f"Prompt preview (first 300 chars): {prompt[:300]!r}") + # Log prompt preview + logger.info(f"Prompt preview (first 300 chars): {prompt[:300]!r}") # Call LLM response_text = await cls.call_llm(prompt, data_url, config)