mirror of
https://github.com/R0m1k3/Priceflow.git
synced 2026-10-11 17:29:14 +02:00
feat: Add Stokomani search results parser
This commit is contained in:
1 parent
c2f042beee
commit
b3d5c56707
1 file changed
+23
-40
@@ -9,13 +9,12 @@ class StokomaniParser(BaseParser):
|
||||
super().__init__("stokomani.fr", "https://www.stokomani.fr")
|
||||
|
||||
def parse_search_results(self, html: str, query: str, search_url: str) -> list[ProductResult]:
|
||||
from bs4 import BeautifulSoup, NavigableString
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
results = []
|
||||
|
||||
# Find product cards
|
||||
# Based on inspection, links are a.reversed-link.block
|
||||
# We'll look for the container of these links
|
||||
links = soup.select("a.reversed-link.block, a[href*='/products/']")
|
||||
# Find product title links
|
||||
links = soup.select("a.reversed-link.block")
|
||||
|
||||
seen_urls = set()
|
||||
|
||||
@@ -25,54 +24,38 @@ class StokomaniParser(BaseParser):
|
||||
if not href or href in seen_urls:
|
||||
continue
|
||||
|
||||
# Filter out non-product links
|
||||
if '/products/' not in href:
|
||||
continue
|
||||
|
||||
seen_urls.add(href)
|
||||
url = self.make_absolute_url(href)
|
||||
|
||||
# Title
|
||||
title = link.get_text(strip=True)
|
||||
if not title:
|
||||
# Try finding title in nested elements
|
||||
title_el = link.find(class_=lambda x: x and 'title' in x)
|
||||
if title_el:
|
||||
title = title_el.get_text(strip=True)
|
||||
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Find parent card to scope image and price search
|
||||
# Usually the card is a few levels up
|
||||
card = link.find_parent("div", class_=lambda x: x and ("product" in x or "card" in x or "item" in x))
|
||||
if not card:
|
||||
# Fallback: use the link's parent
|
||||
card = link.parent.parent
|
||||
# Image: Look for the preceding <a> with aria-label
|
||||
img_url = None
|
||||
prev_a = link.find_previous_sibling("a", attrs={"aria-label": True})
|
||||
if prev_a and prev_a.get('href') == href:
|
||||
img_url = self.extract_image_url(prev_a)
|
||||
|
||||
# Image
|
||||
img_url = self.extract_image_url(card)
|
||||
|
||||
# Price
|
||||
# Price: Look for text node after the link
|
||||
price = None
|
||||
# Try specific price selectors first
|
||||
price_selectors = [
|
||||
".price", ".money", ".current-price",
|
||||
"span[class*='price']", "div[class*='price']"
|
||||
]
|
||||
|
||||
for selector in price_selectors:
|
||||
price_el = card.select_one(selector)
|
||||
if price_el:
|
||||
price = self.parse_price_text(price_el.get_text())
|
||||
if price:
|
||||
break
|
||||
|
||||
# Fallback: Look for price pattern in the card text
|
||||
if not price:
|
||||
# Get text but exclude the title to avoid false positives if title has numbers
|
||||
card_text = card.get_text(" ", strip=True)
|
||||
price = self.parse_price_text(card_text)
|
||||
next_sibling = link.next_sibling
|
||||
while next_sibling:
|
||||
if isinstance(next_sibling, NavigableString):
|
||||
price_text = next_sibling.strip()
|
||||
if "€" in price_text:
|
||||
price = self.parse_price_text(price_text)
|
||||
if price:
|
||||
break
|
||||
elif next_sibling.name == 'div' and 'price' in str(next_sibling.get('class', [])):
|
||||
# Try finding price in next div if it's a price container
|
||||
price = self.parse_price_text(next_sibling.get_text())
|
||||
if price:
|
||||
break
|
||||
next_sibling = next_sibling.next_sibling
|
||||
|
||||
results.append(ProductResult(
|
||||
title=title,
|
||||
|
||||
Reference in new issue
Block a user