""" Base scraper class with common functionality """ import logging import re from abc import ABC, abstractmethod from dataclasses import dataclass, field from typing import List, Optional from urllib.parse import urlparse, urlunparse, parse_qs, urlencode logger = logging.getLogger(__name__) def clean_product_name(name: str) -> str: """Clean garbage from product names (SVG classes, CSS, etc.)""" if not name: return name import re # Remove SVG/CSS class garbage (pikachu_svg__st1{fill:#7c888f}.review-full-...) name = re.sub(r'[\w-]*_svg__[\w\s{}:#.;-]+', '', name, flags=re.IGNORECASE) name = re.sub(r'\.review-full-[\w\s{}:#.;-]+', '', name, flags=re.IGNORECASE) name = re.sub(r'\{[^}]*\}', '', name) # Remove any remaining {css} blocks name = re.sub(r'\d+\s*Reviews?$', '', name, flags=re.IGNORECASE) # Remove trailing "X Reviews" # Clean up multiple spaces name = re.sub(r'\s+', ' ', name) return name.strip() def normalize_url(url: str) -> str: """ Normalize a URL to prevent duplicates from minor variations. - Removes tracking/session query parameters - Removes trailing slashes - Lowercases the domain - Keeps only essential path components """ if not url: return url try: parsed = urlparse(url) # Lowercase the domain netloc = parsed.netloc.lower() # Remove trailing slash from path path = parsed.path.rstrip('/') # Filter out common tracking/non-essential query params query_params = parse_qs(parsed.query) exclude_params = { 'ref', 'utm_source', 'utm_medium', 'utm_campaign', 'utm_content', 'utm_term', 'gclid', 'fbclid', 'msclkid', 'dclid', 'zanpid', 'src', 'srsltid', 'skuId', 'preselect', 'intl', 'lnk', 'Nao', 'start', 'page', 'sort', 'sortBy', 'facets', } filtered_params = { k: v for k, v in query_params.items() if k.lower() not in {p.lower() for p in exclude_params} } # Rebuild query string (sorted for consistency) query = urlencode(sorted(filtered_params.items()), doseq=True) if filtered_params else '' # Rebuild URL without fragment normalized = urlunparse((parsed.scheme, netloc, path, '', query, '')) return normalized except Exception as e: logger.debug(f"Error normalizing URL {url}: {e}") return url @dataclass class Product: """Represents a product listing""" name: str url: str price: Optional[str] in_stock: bool image_url: Optional[str] = None site: str = "" product_id: Optional[str] = None # Unique identifier for tracking _normalized_url: str = field(default="", repr=False, compare=False) def __post_init__(self): self._normalized_url = normalize_url(self.url) # Clean the name of any SVG/CSS garbage if self.name: self.name = clean_product_name(self.name) @property def normalized_url(self) -> str: """Get the normalized URL for deduplication""" if not self._normalized_url: self._normalized_url = normalize_url(self.url) return self._normalized_url def __hash__(self): # Use normalized URL for hashing to prevent duplicates return hash(self.normalized_url) def __eq__(self, other): if isinstance(other, Product): # Compare by normalized URL OR by product_id if both have one if self.product_id and other.product_id and self.site == other.site: return self.product_id == other.product_id return self.normalized_url == other.normalized_url return False class BaseScraper(ABC): """Base class for site-specific scrapers""" site_name: str = "unknown" # Terms that indicate a Pokemon product POKEMON_TERMS = [ "pokemon", "pokémon", "poke", "tcg", "pikachu", "charizard", "mewtwo", "eevee", "snorlax", "booster", "elite trainer", "etb", "scarlet", "violet", "prismatic", "evolutions", ] # Terms that indicate NOT a Pokemon product (false positives from search) EXCLUDE_TERMS = [ "ice cube", "oven", "barbie", "hot wheels", "lego", "furniture", "appliance", "kitchen", "bedding", "glitter girls", "masters of the universe", "transformers", "room essentials", "threshold", ] def is_pokemon_product(self, product: Product) -> bool: """ Check if a product is actually a Pokemon product. Filters out false positives from search results. """ name_lower = product.name.lower() # Check for exclusion terms first for term in self.EXCLUDE_TERMS: if term in name_lower: return False # Check for Pokemon terms for term in self.POKEMON_TERMS: if term in name_lower: return True # If no Pokemon terms found, reject it return False def filter_pokemon_products(self, products: List[Product]) -> List[Product]: """Filter to only include valid Pokemon products""" filtered = [p for p in products if self.is_pokemon_product(p)] rejected = len(products) - len(filtered) if rejected > 0: logger.info(f"Filtered out {rejected} non-Pokemon products") return filtered @abstractmethod def scrape_category_page(self, url: str) -> List[Product]: """ Scrape a category/search page and return all products found Args: url: URL of the category page Returns: List of Product objects """ pass @abstractmethod def check_product_stock(self, product_url: str) -> tuple[bool, Optional[str]]: """ Check if a specific product is in stock Args: product_url: URL of the product page Returns: Tuple of (is_in_stock, price) """ pass def filter_by_keywords(self, products: List[Product], keywords: List[str]) -> List[Product]: """Filter products by keywords in name""" if not keywords: return products filtered = [] for product in products: name_lower = product.name.lower() if any(kw.lower() in name_lower for kw in keywords): filtered.append(product) return filtered