From 25cfa80f200c938527620c6c81f69bd607a53c78 Mon Sep 17 00:00:00 2001 From: Mike McGhen Date: Sat, 28 Mar 2026 12:45:18 -0400 Subject: [PATCH] feat: Enhance Best Buy scraper to support new product card formats and improve data extraction methods --- debug_bestbuy.html | 267 ++++++++++++++++++++++++++-- scrapers/bestbuy.py | 420 +++++++++++++++++++++++++++++++++++++------- 2 files changed, 609 insertions(+), 78 deletions(-) diff --git a/debug_bestbuy.html b/debug_bestbuy.html index cc4c922..816ad78 100644 --- a/debug_bestbuy.html +++ b/debug_bestbuy.html @@ -1,4 +1,4 @@ -pokemon tcg - Best Buypokemon trading cards - Best Buy
-
Skip to contentGo to Product Search
-
+
-
Main Content

pokemon tcg

in Trading Cards (267)
Sort by:

pokemon trading cards

(483)
Sort by:
1-18 of 267 items
Tell us about your search experience
Sponsored
Sponsored
Sponsored

Similar products from outside of Best Buy

sponsored

\ No newline at end of file +

\ No newline at end of file diff --git a/scrapers/bestbuy.py b/scrapers/bestbuy.py index 8dcb5cb..6a13a07 100644 --- a/scrapers/bestbuy.py +++ b/scrapers/bestbuy.py @@ -112,7 +112,11 @@ class BestBuyScraper(BaseScraper): # Try multiple selectors for product cards - Best Buy updates these frequently product_cards = ( - soup.select("li.sku-item") + soup.select("li.product-list-item") # Current Best Buy format (2026) + or soup.select("[data-testid='list-item']") # Modern React testid + or soup.select("[data-testid='product-card']") # Alternate testid + or soup.select("[class*='ProductCard']") # React component class + or soup.select("li.sku-item") # Legacy or soup.select("[data-sku-id]") or soup.select(".sku-item") or soup.select("[class*='sku-item']") @@ -120,6 +124,8 @@ class BestBuyScraper(BaseScraper): or soup.select("[class*='productCard']") or soup.select("[class*='product-card']") or soup.select(".list-item") + or soup.select("[class*='listItem']") # camelCase variant + or soup.select("article[class*='product']") # Semantic HTML ) logger.info(f"Found {len(product_cards)} product cards on page {page_num}") @@ -129,15 +135,33 @@ class BestBuyScraper(BaseScraper): if product: products.append(product) - # Fallback: parse product links - if not product_cards: - product_links = soup.select("a[href*='/site/'][href*='.p']") - logger.info(f"Fallback: Found {len(product_links)} product links") + # Fallback 1: Try to extract from JS state / __NEXT_DATA__ (most reliable) + if not product_cards or not products: + logger.info("Trying JS state extraction...") + js_products = self._extract_from_page_data(soup, browser) + if js_products: + logger.info(f"Extracted {len(js_products)} products from JS state") + products.extend(js_products) + + # Fallback 2: Parse product links (resilient to DOM changes) + if not products: + # Try multiple link patterns - Best Buy uses different URL formats + # New format: /product/pokemon-card-name/ABC123/sku/12345 + # Old format: /site/product-name/12345.p + product_links = ( + soup.select("a[href*='/product/'][href*='/sku/']") # New format with /sku/ + or soup.select("a[href*='/site/'][href*='.p']") # Legacy format + or soup.select("a[href*='skuId=']") # URL param format + ) + logger.info(f"Fallback links: Found {len(product_links)} product links") href_to_links = {} for link in product_links: href = link.get("href", "") - if not href or ".p" not in href: + if not href: + continue + # Accept links with /sku/, .p suffix, or skuId parameter + if "/sku/" not in href and ".p" not in href and "skuId=" not in href: continue if href not in href_to_links: href_to_links[href] = [] @@ -154,9 +178,15 @@ class BestBuyScraper(BaseScraper): if product: products.append(product) - # Also try parsing from data attributes and script tags - if not products: - products.extend(self._extract_from_page_data(soup, browser)) + # Debug: Save HTML if no products found for analysis + if not products and not product_cards: + try: + debug_path = "debug_bestbuy.html" + with open(debug_path, "w", encoding="utf-8") as f: + f.write(html) + logger.warning(f"No products found - saved HTML to {debug_path} for debugging") + except Exception as e: + logger.debug(f"Could not save debug HTML: {e}") # Don't close - reuse browser for next page @@ -185,58 +215,225 @@ class BestBuyScraper(BaseScraper): """Extract products from page data attributes and evaluate JS if needed""" products = [] - # Try to get product data from data attributes - items_with_data = soup.select("[data-testid][data-sku-id]") - for item in items_with_data: - sku_id = item.get("data-sku-id", "") - if sku_id: - # Find name and price within this element - name_elem = item.select_one("h4") or item.select_one("[class*='title']") or item.select_one("a") - name = name_elem.get_text(strip=True) if name_elem else "" + # Method 1: Parse Apollo SSR data from script tags (Best Buy's current format) + # Best Buy uses window[Symbol.for("ApolloSSRDataTransport")] format + for script in soup.find_all("script"): + if script.string and "ApolloSSRDataTransport" in script.string: + logger.info("Found Apollo SSR data in script tag") + products.extend(self._extract_from_apollo_data(script.string)) + if products: + break + + # Method 2: Try to parse __NEXT_DATA__ script tag (older format) + if not products: + next_data_script = soup.select_one("script#__NEXT_DATA__") + if next_data_script and next_data_script.string: + try: + data = json.loads(next_data_script.string) + logger.info("Found __NEXT_DATA__ script tag in HTML") + products.extend(self._extract_products_from_json(data)) + except (json.JSONDecodeError, TypeError) as e: + logger.debug(f"Could not parse __NEXT_DATA__ from HTML: {e}") + + # Method 3: Try multiple JS state sources via browser execution + if not products: + state_scripts = [ + "return window.__NEXT_DATA__ ? JSON.stringify(window.__NEXT_DATA__) : null", + "return window.__INITIAL_STATE__ ? JSON.stringify(window.__INITIAL_STATE__) : null", + "return window.__PRELOADED_STATE__ ? JSON.stringify(window.__PRELOADED_STATE__) : null", + "return window.__APP_STATE__ ? JSON.stringify(window.__APP_STATE__) : null", + ] + + for script in state_scripts: + try: + state_json = browser.driver.execute_script(script) + if state_json: + data = json.loads(state_json) + logger.info(f"Extracted state from JS: {script[:50]}...") + extracted = self._extract_products_from_json(data) + if extracted: + products.extend(extracted) + break # Stop if we found products + except Exception as e: + logger.debug(f"Could not extract from JS state ({script[:30]}): {e}") + + # Method 3: Try to get product data from data attributes + if not products: + items_with_data = soup.select("[data-testid][data-sku-id]") or soup.select("[data-sku-id]") + for item in items_with_data: + sku_id = item.get("data-sku-id", "") + if sku_id: + # Find name and price within this element + name_elem = item.select_one("h4") or item.select_one("[class*='title']") or item.select_one("a") + name = name_elem.get_text(strip=True) if name_elem else "" + + if name and len(name) > 5: + price_text = item.get_text() + price_match = re.search(r"\$[\d,]+\.?\d*", price_text) + price = price_match.group() if price_match else None + + products.append(Product( + name=name, + url=f"{self.base_url}/site/{sku_id}.p", + price=price, + in_stock=True, + image_url=None, + site=self.site_name, + product_id=sku_id, + )) + + return products + + def _extract_from_apollo_data(self, script_content: str) -> List[Product]: + """Extract products from Best Buy's Apollo SSR data format""" + products = [] + + try: + # Extract product URLs - new format: /product/name/ID/sku/skuId + url_pattern = r'"pdp":"(https://www\.bestbuy\.com/product/[^"]+)"' + url_matches = re.findall(url_pattern, script_content) + logger.info(f"Found {len(url_matches)} product URLs in Apollo data") + + # Extract product names (short format) + name_pattern = r'"short":"([^"]+)"' + name_matches = re.findall(name_pattern, script_content) + + # Extract SKU IDs + sku_pattern = r'"skuId":"(\d+)"' + sku_matches = re.findall(sku_pattern, script_content) + + # Extract prices - look for priceEventPrice or similar + # Prices in Apollo format: "priceEventPrice":29.99 or "regularPrice":39.99 + price_pattern = r'"(?:priceEventPrice|regularPrice|currentPrice)":(\d+\.?\d*)' + price_matches = re.findall(price_pattern, script_content) + + # Extract images + image_pattern = r'"piscesHref":"(https://pisces\.bbystatic\.com/[^"]+)"' + image_matches = re.findall(image_pattern, script_content) + + logger.info(f"Apollo extraction: {len(url_matches)} URLs, {len(name_matches)} names, {len(sku_matches)} SKUs, {len(price_matches)} prices") + + # Create products from URLs (most reliable source) + seen_urls = set() + for url in url_matches: + if url in seen_urls: + continue + seen_urls.add(url) + + # Extract SKU from URL: /product/.../sku/12345 + sku_match = re.search(r'/sku/(\d+)', url) + sku_id = sku_match.group(1) if sku_match else "" + + # Try to find name for this product + # Look for name near the URL in the data + name = None + url_pos = script_content.find(url) + if url_pos > 0: + # Look for "short":"..." within 2000 chars before the URL + context = script_content[max(0, url_pos-2000):url_pos] + name_match = re.search(r'"short":"([^"]+)"[^}]*$', context) + if name_match: + name = name_match.group(1) + + if not name and name_matches: + # Use any name that contains pokemon (fallback) + for n in name_matches: + if 'pok' in n.lower(): + name = n + break + + if not name: + # Extract from URL + url_parts = url.split('/') + if len(url_parts) > 4: + name = url_parts[4].replace('-', ' ').title() + + # Find price + price = None + if price_matches: + # Use first available price as default + price = f"${float(price_matches[0]):.2f}" + + # Find image + image_url = None + if image_matches: + image_url = image_matches[0] if name and len(name) > 5: - price_text = item.get_text() - price_match = re.search(r"\$[\d,]+\.?\d*", price_text) - price = price_match.group() if price_match else None - products.append(Product( name=name, - url=f"{self.base_url}/site/{sku_id}.p", + url=url, price=price, - in_stock=True, - image_url=None, + in_stock=True, # Assume in stock if listed + image_url=image_url, site=self.site_name, product_id=sku_id, )) - # Try to extract from window.__INITIAL_STATE__ or similar JS objects - try: - initial_state = browser.driver.execute_script(""" - if (window.__INITIAL_STATE__) return JSON.stringify(window.__INITIAL_STATE__); - if (window.__NEXT_DATA__) return JSON.stringify(window.__NEXT_DATA__); - return null; - """) - if initial_state: - data = json.loads(initial_state) - products.extend(self._extract_products_from_json(data)) - except Exception as e: - logger.debug(f"Could not extract from JS state: {e}") + # Deduplicate by SKU + seen_skus = set() + unique_products = [] + for p in products: + if p.product_id and p.product_id not in seen_skus: + seen_skus.add(p.product_id) + unique_products.append(p) + elif not p.product_id: + unique_products.append(p) - return products + logger.info(f"Extracted {len(unique_products)} unique products from Apollo data") + return unique_products + + except Exception as e: + logger.error(f"Error parsing Apollo data: {e}") + return [] def _extract_products_from_json(self, data, depth=0) -> List[Product]: """Recursively search JSON for product data""" products = [] - if depth > 10: + if depth > 15: # Increased depth for deeply nested structures return products if isinstance(data, dict): + # First check common Best Buy JSON paths (Next.js structure) + if depth == 0: + # Try common paths in __NEXT_DATA__ + common_paths = [ + ("props", "pageProps", "products"), + ("props", "pageProps", "initialData", "products"), + ("props", "pageProps", "searchResults", "products"), + ("props", "pageProps", "items"), + ("props", "pageProps", "initialData", "searchResult", "products"), + ("props", "initialState", "products"), + ("pageProps", "products"), + ("pageProps", "items"), + ] + for path in common_paths: + obj = data + for key in path: + if isinstance(obj, dict) and key in obj: + obj = obj[key] + else: + obj = None + break + if obj and isinstance(obj, list): + logger.info(f"Found products at path: {'.'.join(path)}") + for item in obj: + if isinstance(item, dict): + product = self._parse_product_json(item) + if product: + products.append(product) + # Check if this looks like a Best Buy product - if "skuId" in data or ("name" in data and "regularPrice" in data): + if "skuId" in data or "sku" in data: + product = self._parse_product_json(data) + if product: + products.append(product) + elif "name" in data and ("regularPrice" in data or "salePrice" in data or "price" in data): product = self._parse_product_json(data) if product: products.append(product) + # Continue recursive search for value in data.values(): products.extend(self._extract_products_from_json(value, depth + 1)) @@ -249,12 +446,34 @@ class BestBuyScraper(BaseScraper): def _parse_product_json(self, data: dict) -> Optional[Product]: """Parse a product from Best Buy's JSON data""" try: - name = data.get("name") or data.get("displayName", "") - if not name: + # Try multiple name fields + name = ( + data.get("name") + or data.get("displayName") + or data.get("title") + or data.get("productName") + or "" + ) + if not name or len(name) < 5: return None - sku_id = data.get("skuId") or data.get("sku", "") - url_slug = data.get("url") or "" + # Try multiple SKU fields + sku_id = ( + data.get("skuId") + or data.get("sku") + or data.get("productId") + or data.get("id") + or "" + ) + + # Try multiple URL fields + url_slug = ( + data.get("url") + or data.get("pdpUrl") + or data.get("productUrl") + or data.get("link") + or "" + ) if url_slug: url = url_slug if url_slug.startswith("http") else f"{self.base_url}{url_slug}" @@ -263,23 +482,55 @@ class BestBuyScraper(BaseScraper): else: return None - # Get price + # Get price - try multiple price structures price = None if "regularPrice" in data: - price = f"${data['regularPrice']:.2f}" + price = f"${data['regularPrice']:.2f}" if isinstance(data['regularPrice'], (int, float)) else data['regularPrice'] elif "salePrice" in data: - price = f"${data['salePrice']:.2f}" + price = f"${data['salePrice']:.2f}" if isinstance(data['salePrice'], (int, float)) else data['salePrice'] + elif "currentPrice" in data: + price = f"${data['currentPrice']:.2f}" if isinstance(data['currentPrice'], (int, float)) else data['currentPrice'] + elif "price" in data: + p = data['price'] + if isinstance(p, dict): + price = p.get("currentPrice") or p.get("regularPrice") or p.get("salePrice") + if isinstance(price, (int, float)): + price = f"${price:.2f}" + elif isinstance(p, (int, float)): + price = f"${p:.2f}" + else: + price = str(p) if p else None + elif "priceInfo" in data: + price_info = data["priceInfo"] + if isinstance(price_info, dict): + price = price_info.get("currentPrice") or price_info.get("price") + if isinstance(price, (int, float)): + price = f"${price:.2f}" - # Check availability + # Check availability - handle multiple formats in_stock = True - availability = data.get("availability", {}) - if isinstance(availability, dict): - in_stock = availability.get("isAvailable", True) - elif data.get("orderable") is False: + if "availability" in data: + availability = data["availability"] + if isinstance(availability, dict): + in_stock = availability.get("isAvailable", True) or availability.get("available", True) + elif isinstance(availability, bool): + in_stock = availability + elif isinstance(availability, str): + in_stock = availability.lower() not in ["unavailable", "sold out", "out of stock"] + if data.get("orderable") is False: + in_stock = False + if data.get("inStock") is False: in_stock = False - # Get image - image_url = data.get("image") or data.get("thumbnailImage") + # Get image - try multiple fields + image_url = ( + data.get("image") + or data.get("thumbnailImage") + or data.get("imageUrl") + or data.get("thumbnail") + ) + if isinstance(image_url, dict): + image_url = image_url.get("src") or image_url.get("url") return Product( name=name, @@ -288,7 +539,7 @@ class BestBuyScraper(BaseScraper): in_stock=in_stock, image_url=image_url, site=self.site_name, - product_id=str(sku_id), + product_id=str(sku_id) if sku_id else "", ) except Exception as e: @@ -298,10 +549,16 @@ class BestBuyScraper(BaseScraper): def _parse_product_card(self, card) -> Optional[Product]: """Parse a product card element""" try: - # Find link - link = card.select_one("a[href*='/site/'][href*='.p']") or card.select_one("a.image-link") - if not link: - link = card.select_one("a") + # Find link - try multiple patterns + link = ( + card.select_one("a[href*='/product/']") # New Best Buy format + or card.select_one("a[href*='/site/'][href*='.p']") # Legacy format + or card.select_one("a[href*='skuId=']") + or card.select_one("a.image-link") + or card.select_one("[data-testid='product-link']") + or card.select_one("a[data-track]") + or card.select_one("a") + ) if not link: return None @@ -310,11 +567,16 @@ class BestBuyScraper(BaseScraper): return None url = href if href.startswith("http") else f"{self.base_url}{href}" - # Get name + # Get name - try multiple modern selectors name_elem = ( - card.select_one(".sku-title a") - or card.select_one("h4.sku-header a") + card.select_one("h4") # Current Best Buy format + or card.select_one("h3") or card.select_one("[data-testid='product-title']") + or card.select_one("[data-testid='product-name']") + or card.select_one("[class*='productTitle']") + or card.select_one("[class*='ProductTitle']") + or card.select_one(".sku-title a") + or card.select_one("h4.sku-header a") or card.select_one(".sku-title") or link ) @@ -324,10 +586,15 @@ class BestBuyScraper(BaseScraper): if len(name) < 5: return None - # Get price + # Get price - try multiple modern selectors price_elem = ( - card.select_one(".priceView-customer-price span") + card.select_one("div.pricing") # Current Best Buy format + or card.select_one("[class*='pricing']") or card.select_one("[data-testid='customer-price']") + or card.select_one("[data-testid='current-price']") + or card.select_one("[class*='customerPrice']") + or card.select_one("[class*='CurrentPrice']") + or card.select_one(".priceView-customer-price span") or card.select_one(".pricing-price__regular-price") or card.select_one("[class*='price']") ) @@ -348,11 +615,24 @@ class BestBuyScraper(BaseScraper): if image_url and not image_url.startswith("http"): image_url = f"https:{image_url}" if image_url.startswith("//") else f"{self.base_url}{image_url}" - # Extract SKU ID from URL + # Extract product ID from URL - handle multiple formats + # New: /product/product-name/ABC123 or /product/.../sku/12345 + # Old: /site/.../12345.p product_id = "" - match = re.search(r"/(\d+)\.p", url) + # Try /sku/12345 format first + match = re.search(r"/sku/(\d+)", url) if match: product_id = match.group(1) + else: + # Try old .p format + match = re.search(r"/(\d+)\.p", url) + if match: + product_id = match.group(1) + else: + # New format: last path segment is the ID + url_parts = url.rstrip('/').split('/') + if url_parts: + product_id = url_parts[-1] return Product( name=name, @@ -382,11 +662,17 @@ class BestBuyScraper(BaseScraper): if not name or len(name) < 5: return None - # Extract SKU ID + # Extract SKU ID - handle both old and new formats + # New: /product/.../sku/12345 + # Old: /site/.../12345.p product_id = "" - match = re.search(r"/(\d+)\.p", url) + match = re.search(r"/sku/(\d+)", url) # New format first if match: product_id = match.group(1) + else: + match = re.search(r"/(\d+)\.p", url) # Old format + if match: + product_id = match.group(1) # Find price near link parent = link.find_parent()