mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Introduce scheduler_service to automate product price and stock tracking, utilizing AI extraction and confidence thresholds.
This commit is contained in:
1 parent
06dcf0a7a5
commit
ad0570be76
2 files changed
+16
-11
No files matched your search
@@ -43,12 +43,19 @@ def _normalize_title(title: str) -> str:
|
||||
r" \| gifi$",
|
||||
r" - amazon$",
|
||||
r" \| amazon$",
|
||||
# Country specific
|
||||
r" \| action fr$",
|
||||
r" - action fr$",
|
||||
r" \| action be$",
|
||||
r" - action be$",
|
||||
r" \| action nl$",
|
||||
r" - action nl$",
|
||||
r"^b&m stores : ",
|
||||
r"^gifi : ",
|
||||
r"^action : ",
|
||||
]
|
||||
for pattern in patterns:
|
||||
title = re.sub(pattern, "", title)
|
||||
title = re.sub(pattern, "", title, flags=re.IGNORECASE)
|
||||
|
||||
# Separate numbers from letters (handles cases like 'Blanc40' or '100pièces')
|
||||
title = re.sub(r"([a-zA-Z])(\d)", r"\1 \2", title)
|
||||
@@ -80,15 +87,14 @@ def _is_generic_or_error_title(title: str) -> bool:
|
||||
"attention required",
|
||||
"erreur 403",
|
||||
"403 forbidden",
|
||||
"security check",
|
||||
"security check", # Warning: Can be in valid titles? Unlikely.
|
||||
"robot or human",
|
||||
"bot verification",
|
||||
"loading",
|
||||
"chargement",
|
||||
"veuillez patienter",
|
||||
# "loading", # Removed: too generic, handled by content check usually
|
||||
# "chargement",
|
||||
# "veuillez patienter",
|
||||
"site non accessible",
|
||||
"maintenance",
|
||||
"un instant",
|
||||
]
|
||||
if any(term in t for term in error_terms):
|
||||
return True
|
||||
@@ -296,8 +302,6 @@ async def process_item_check(item_id: int):
|
||||
is_generic_title = _is_generic_or_error_title(page_title)
|
||||
|
||||
if not titles_match:
|
||||
# If titles don't match, check if it's because of a generic bot detection/blocker
|
||||
# We only do this check if titles_match is False to avoid false positives on valid pages
|
||||
# If titles don't match, check if it's because of a generic bot detection/blocker
|
||||
# We only do this check if titles_match is False to avoid false positives on valid pages
|
||||
bot_terms_title = [
|
||||
@@ -314,10 +318,11 @@ async def process_item_check(item_id: int):
|
||||
"captcha",
|
||||
"robot or human",
|
||||
"bot verification",
|
||||
"security check",
|
||||
# "security check", # Removed: too generic
|
||||
"access denied",
|
||||
"attention required",
|
||||
"enable javascript",
|
||||
# "enable javascript", # Removed: common in noscript tags
|
||||
"cf-browser-verification",
|
||||
]
|
||||
|
||||
html_lower = html_content.lower() if html_content else ""
|
||||
|
||||
@@ -187,7 +187,7 @@ class ScraperService:
|
||||
config_selector = cfg.get("price_selector")
|
||||
|
||||
# Use config selector if none provided
|
||||
if selector is None and config_selector:
|
||||
if not selector and config_selector:
|
||||
logger.info(f"Using configured price selector for {domain}: {config_selector}")
|
||||
selector = config_selector
|
||||
|
||||
|
||||
Reference in new issue
Block a user