feat: Introduce scheduler_service to automate product price and stock tracking, utilizing AI extraction and confidence thresholds.

This commit is contained in:
Michael committed 2026-01-30 15:26:09 +01:00
1 parent 06dcf0a7a5
commit ad0570be76
2 files changed
+16 -11

No files matched your search

+15 -10
View File
@@ -43,12 +43,19 @@ def _normalize_title(title: str) -> str:
r" \| gifi$",
r" - amazon$",
r" \| amazon$",
# Country specific
r" \| action fr$",
r" - action fr$",
r" \| action be$",
r" - action be$",
r" \| action nl$",
r" - action nl$",
r"^b&m stores : ",
r"^gifi : ",
r"^action : ",
]
for pattern in patterns:
title = re.sub(pattern, "", title)
title = re.sub(pattern, "", title, flags=re.IGNORECASE)
# Separate numbers from letters (handles cases like 'Blanc40' or '100pièces')
title = re.sub(r"([a-zA-Z])(\d)", r"\1 \2", title)
@@ -80,15 +87,14 @@ def _is_generic_or_error_title(title: str) -> bool:
"attention required",
"erreur 403",
"403 forbidden",
"security check",
"security check", # Warning: Can be in valid titles? Unlikely.
"robot or human",
"bot verification",
"loading",
"chargement",
"veuillez patienter",
# "loading", # Removed: too generic, handled by content check usually
# "chargement",
# "veuillez patienter",
"site non accessible",
"maintenance",
"un instant",
]
if any(term in t for term in error_terms):
return True
@@ -296,8 +302,6 @@ async def process_item_check(item_id: int):
is_generic_title = _is_generic_or_error_title(page_title)
if not titles_match:
# If titles don't match, check if it's because of a generic bot detection/blocker
# We only do this check if titles_match is False to avoid false positives on valid pages
# If titles don't match, check if it's because of a generic bot detection/blocker
# We only do this check if titles_match is False to avoid false positives on valid pages
bot_terms_title = [
@@ -314,10 +318,11 @@ async def process_item_check(item_id: int):
"captcha",
"robot or human",
"bot verification",
"security check",
# "security check", # Removed: too generic
"access denied",
"attention required",
"enable javascript",
# "enable javascript", # Removed: common in noscript tags
"cf-browser-verification",
]
html_lower = html_content.lower() if html_content else ""
+1 -1
View File
@@ -187,7 +187,7 @@ class ScraperService:
config_selector = cfg.get("price_selector")
# Use config selector if none provided
if selector is None and config_selector:
if not selector and config_selector:
logger.info(f"Using configured price selector for {domain}: {config_selector}")
selector = config_selector