feat: Enhance Best Buy scraper to support new product card formats and improve data extraction methods

This commit is contained in:
2026-03-28 12:45:18 -04:00
parent 297cb9b146
commit 25cfa80f20
2 changed files with 609 additions and 78 deletions
+256 -11
View File
File diff suppressed because one or more lines are too long
+353 -67
View File
@@ -112,7 +112,11 @@ class BestBuyScraper(BaseScraper):
# Try multiple selectors for product cards - Best Buy updates these frequently
product_cards = (
soup.select("li.sku-item")
soup.select("li.product-list-item") # Current Best Buy format (2026)
or soup.select("[data-testid='list-item']") # Modern React testid
or soup.select("[data-testid='product-card']") # Alternate testid
or soup.select("[class*='ProductCard']") # React component class
or soup.select("li.sku-item") # Legacy
or soup.select("[data-sku-id]")
or soup.select(".sku-item")
or soup.select("[class*='sku-item']")
@@ -120,6 +124,8 @@ class BestBuyScraper(BaseScraper):
or soup.select("[class*='productCard']")
or soup.select("[class*='product-card']")
or soup.select(".list-item")
or soup.select("[class*='listItem']") # camelCase variant
or soup.select("article[class*='product']") # Semantic HTML
)
logger.info(f"Found {len(product_cards)} product cards on page {page_num}")
@@ -129,15 +135,33 @@ class BestBuyScraper(BaseScraper):
if product:
products.append(product)
# Fallback: parse product links
if not product_cards:
product_links = soup.select("a[href*='/site/'][href*='.p']")
logger.info(f"Fallback: Found {len(product_links)} product links")
# Fallback 1: Try to extract from JS state / __NEXT_DATA__ (most reliable)
if not product_cards or not products:
logger.info("Trying JS state extraction...")
js_products = self._extract_from_page_data(soup, browser)
if js_products:
logger.info(f"Extracted {len(js_products)} products from JS state")
products.extend(js_products)
# Fallback 2: Parse product links (resilient to DOM changes)
if not products:
# Try multiple link patterns - Best Buy uses different URL formats
# New format: /product/pokemon-card-name/ABC123/sku/12345
# Old format: /site/product-name/12345.p
product_links = (
soup.select("a[href*='/product/'][href*='/sku/']") # New format with /sku/
or soup.select("a[href*='/site/'][href*='.p']") # Legacy format
or soup.select("a[href*='skuId=']") # URL param format
)
logger.info(f"Fallback links: Found {len(product_links)} product links")
href_to_links = {}
for link in product_links:
href = link.get("href", "")
if not href or ".p" not in href:
if not href:
continue
# Accept links with /sku/, .p suffix, or skuId parameter
if "/sku/" not in href and ".p" not in href and "skuId=" not in href:
continue
if href not in href_to_links:
href_to_links[href] = []
@@ -154,9 +178,15 @@ class BestBuyScraper(BaseScraper):
if product:
products.append(product)
# Also try parsing from data attributes and script tags
if not products:
products.extend(self._extract_from_page_data(soup, browser))
# Debug: Save HTML if no products found for analysis
if not products and not product_cards:
try:
debug_path = "debug_bestbuy.html"
with open(debug_path, "w", encoding="utf-8") as f:
f.write(html)
logger.warning(f"No products found - saved HTML to {debug_path} for debugging")
except Exception as e:
logger.debug(f"Could not save debug HTML: {e}")
# Don't close - reuse browser for next page
@@ -185,58 +215,225 @@ class BestBuyScraper(BaseScraper):
"""Extract products from page data attributes and evaluate JS if needed"""
products = []
# Try to get product data from data attributes
items_with_data = soup.select("[data-testid][data-sku-id]")
for item in items_with_data:
sku_id = item.get("data-sku-id", "")
if sku_id:
# Find name and price within this element
name_elem = item.select_one("h4") or item.select_one("[class*='title']") or item.select_one("a")
name = name_elem.get_text(strip=True) if name_elem else ""
# Method 1: Parse Apollo SSR data from script tags (Best Buy's current format)
# Best Buy uses window[Symbol.for("ApolloSSRDataTransport")] format
for script in soup.find_all("script"):
if script.string and "ApolloSSRDataTransport" in script.string:
logger.info("Found Apollo SSR data in script tag")
products.extend(self._extract_from_apollo_data(script.string))
if products:
break
# Method 2: Try to parse __NEXT_DATA__ script tag (older format)
if not products:
next_data_script = soup.select_one("script#__NEXT_DATA__")
if next_data_script and next_data_script.string:
try:
data = json.loads(next_data_script.string)
logger.info("Found __NEXT_DATA__ script tag in HTML")
products.extend(self._extract_products_from_json(data))
except (json.JSONDecodeError, TypeError) as e:
logger.debug(f"Could not parse __NEXT_DATA__ from HTML: {e}")
# Method 3: Try multiple JS state sources via browser execution
if not products:
state_scripts = [
"return window.__NEXT_DATA__ ? JSON.stringify(window.__NEXT_DATA__) : null",
"return window.__INITIAL_STATE__ ? JSON.stringify(window.__INITIAL_STATE__) : null",
"return window.__PRELOADED_STATE__ ? JSON.stringify(window.__PRELOADED_STATE__) : null",
"return window.__APP_STATE__ ? JSON.stringify(window.__APP_STATE__) : null",
]
for script in state_scripts:
try:
state_json = browser.driver.execute_script(script)
if state_json:
data = json.loads(state_json)
logger.info(f"Extracted state from JS: {script[:50]}...")
extracted = self._extract_products_from_json(data)
if extracted:
products.extend(extracted)
break # Stop if we found products
except Exception as e:
logger.debug(f"Could not extract from JS state ({script[:30]}): {e}")
# Method 3: Try to get product data from data attributes
if not products:
items_with_data = soup.select("[data-testid][data-sku-id]") or soup.select("[data-sku-id]")
for item in items_with_data:
sku_id = item.get("data-sku-id", "")
if sku_id:
# Find name and price within this element
name_elem = item.select_one("h4") or item.select_one("[class*='title']") or item.select_one("a")
name = name_elem.get_text(strip=True) if name_elem else ""
if name and len(name) > 5:
price_text = item.get_text()
price_match = re.search(r"\$[\d,]+\.?\d*", price_text)
price = price_match.group() if price_match else None
products.append(Product(
name=name,
url=f"{self.base_url}/site/{sku_id}.p",
price=price,
in_stock=True,
image_url=None,
site=self.site_name,
product_id=sku_id,
))
return products
def _extract_from_apollo_data(self, script_content: str) -> List[Product]:
"""Extract products from Best Buy's Apollo SSR data format"""
products = []
try:
# Extract product URLs - new format: /product/name/ID/sku/skuId
url_pattern = r'"pdp":"(https://www\.bestbuy\.com/product/[^"]+)"'
url_matches = re.findall(url_pattern, script_content)
logger.info(f"Found {len(url_matches)} product URLs in Apollo data")
# Extract product names (short format)
name_pattern = r'"short":"([^"]+)"'
name_matches = re.findall(name_pattern, script_content)
# Extract SKU IDs
sku_pattern = r'"skuId":"(\d+)"'
sku_matches = re.findall(sku_pattern, script_content)
# Extract prices - look for priceEventPrice or similar
# Prices in Apollo format: "priceEventPrice":29.99 or "regularPrice":39.99
price_pattern = r'"(?:priceEventPrice|regularPrice|currentPrice)":(\d+\.?\d*)'
price_matches = re.findall(price_pattern, script_content)
# Extract images
image_pattern = r'"piscesHref":"(https://pisces\.bbystatic\.com/[^"]+)"'
image_matches = re.findall(image_pattern, script_content)
logger.info(f"Apollo extraction: {len(url_matches)} URLs, {len(name_matches)} names, {len(sku_matches)} SKUs, {len(price_matches)} prices")
# Create products from URLs (most reliable source)
seen_urls = set()
for url in url_matches:
if url in seen_urls:
continue
seen_urls.add(url)
# Extract SKU from URL: /product/.../sku/12345
sku_match = re.search(r'/sku/(\d+)', url)
sku_id = sku_match.group(1) if sku_match else ""
# Try to find name for this product
# Look for name near the URL in the data
name = None
url_pos = script_content.find(url)
if url_pos > 0:
# Look for "short":"..." within 2000 chars before the URL
context = script_content[max(0, url_pos-2000):url_pos]
name_match = re.search(r'"short":"([^"]+)"[^}]*$', context)
if name_match:
name = name_match.group(1)
if not name and name_matches:
# Use any name that contains pokemon (fallback)
for n in name_matches:
if 'pok' in n.lower():
name = n
break
if not name:
# Extract from URL
url_parts = url.split('/')
if len(url_parts) > 4:
name = url_parts[4].replace('-', ' ').title()
# Find price
price = None
if price_matches:
# Use first available price as default
price = f"${float(price_matches[0]):.2f}"
# Find image
image_url = None
if image_matches:
image_url = image_matches[0]
if name and len(name) > 5:
price_text = item.get_text()
price_match = re.search(r"\$[\d,]+\.?\d*", price_text)
price = price_match.group() if price_match else None
products.append(Product(
name=name,
url=f"{self.base_url}/site/{sku_id}.p",
url=url,
price=price,
in_stock=True,
image_url=None,
in_stock=True, # Assume in stock if listed
image_url=image_url,
site=self.site_name,
product_id=sku_id,
))
# Try to extract from window.__INITIAL_STATE__ or similar JS objects
try:
initial_state = browser.driver.execute_script("""
if (window.__INITIAL_STATE__) return JSON.stringify(window.__INITIAL_STATE__);
if (window.__NEXT_DATA__) return JSON.stringify(window.__NEXT_DATA__);
return null;
""")
if initial_state:
data = json.loads(initial_state)
products.extend(self._extract_products_from_json(data))
except Exception as e:
logger.debug(f"Could not extract from JS state: {e}")
# Deduplicate by SKU
seen_skus = set()
unique_products = []
for p in products:
if p.product_id and p.product_id not in seen_skus:
seen_skus.add(p.product_id)
unique_products.append(p)
elif not p.product_id:
unique_products.append(p)
return products
logger.info(f"Extracted {len(unique_products)} unique products from Apollo data")
return unique_products
except Exception as e:
logger.error(f"Error parsing Apollo data: {e}")
return []
def _extract_products_from_json(self, data, depth=0) -> List[Product]:
"""Recursively search JSON for product data"""
products = []
if depth > 10:
if depth > 15: # Increased depth for deeply nested structures
return products
if isinstance(data, dict):
# First check common Best Buy JSON paths (Next.js structure)
if depth == 0:
# Try common paths in __NEXT_DATA__
common_paths = [
("props", "pageProps", "products"),
("props", "pageProps", "initialData", "products"),
("props", "pageProps", "searchResults", "products"),
("props", "pageProps", "items"),
("props", "pageProps", "initialData", "searchResult", "products"),
("props", "initialState", "products"),
("pageProps", "products"),
("pageProps", "items"),
]
for path in common_paths:
obj = data
for key in path:
if isinstance(obj, dict) and key in obj:
obj = obj[key]
else:
obj = None
break
if obj and isinstance(obj, list):
logger.info(f"Found products at path: {'.'.join(path)}")
for item in obj:
if isinstance(item, dict):
product = self._parse_product_json(item)
if product:
products.append(product)
# Check if this looks like a Best Buy product
if "skuId" in data or ("name" in data and "regularPrice" in data):
if "skuId" in data or "sku" in data:
product = self._parse_product_json(data)
if product:
products.append(product)
elif "name" in data and ("regularPrice" in data or "salePrice" in data or "price" in data):
product = self._parse_product_json(data)
if product:
products.append(product)
# Continue recursive search
for value in data.values():
products.extend(self._extract_products_from_json(value, depth + 1))
@@ -249,12 +446,34 @@ class BestBuyScraper(BaseScraper):
def _parse_product_json(self, data: dict) -> Optional[Product]:
"""Parse a product from Best Buy's JSON data"""
try:
name = data.get("name") or data.get("displayName", "")
if not name:
# Try multiple name fields
name = (
data.get("name")
or data.get("displayName")
or data.get("title")
or data.get("productName")
or ""
)
if not name or len(name) < 5:
return None
sku_id = data.get("skuId") or data.get("sku", "")
url_slug = data.get("url") or ""
# Try multiple SKU fields
sku_id = (
data.get("skuId")
or data.get("sku")
or data.get("productId")
or data.get("id")
or ""
)
# Try multiple URL fields
url_slug = (
data.get("url")
or data.get("pdpUrl")
or data.get("productUrl")
or data.get("link")
or ""
)
if url_slug:
url = url_slug if url_slug.startswith("http") else f"{self.base_url}{url_slug}"
@@ -263,23 +482,55 @@ class BestBuyScraper(BaseScraper):
else:
return None
# Get price
# Get price - try multiple price structures
price = None
if "regularPrice" in data:
price = f"${data['regularPrice']:.2f}"
price = f"${data['regularPrice']:.2f}" if isinstance(data['regularPrice'], (int, float)) else data['regularPrice']
elif "salePrice" in data:
price = f"${data['salePrice']:.2f}"
price = f"${data['salePrice']:.2f}" if isinstance(data['salePrice'], (int, float)) else data['salePrice']
elif "currentPrice" in data:
price = f"${data['currentPrice']:.2f}" if isinstance(data['currentPrice'], (int, float)) else data['currentPrice']
elif "price" in data:
p = data['price']
if isinstance(p, dict):
price = p.get("currentPrice") or p.get("regularPrice") or p.get("salePrice")
if isinstance(price, (int, float)):
price = f"${price:.2f}"
elif isinstance(p, (int, float)):
price = f"${p:.2f}"
else:
price = str(p) if p else None
elif "priceInfo" in data:
price_info = data["priceInfo"]
if isinstance(price_info, dict):
price = price_info.get("currentPrice") or price_info.get("price")
if isinstance(price, (int, float)):
price = f"${price:.2f}"
# Check availability
# Check availability - handle multiple formats
in_stock = True
availability = data.get("availability", {})
if isinstance(availability, dict):
in_stock = availability.get("isAvailable", True)
elif data.get("orderable") is False:
if "availability" in data:
availability = data["availability"]
if isinstance(availability, dict):
in_stock = availability.get("isAvailable", True) or availability.get("available", True)
elif isinstance(availability, bool):
in_stock = availability
elif isinstance(availability, str):
in_stock = availability.lower() not in ["unavailable", "sold out", "out of stock"]
if data.get("orderable") is False:
in_stock = False
if data.get("inStock") is False:
in_stock = False
# Get image
image_url = data.get("image") or data.get("thumbnailImage")
# Get image - try multiple fields
image_url = (
data.get("image")
or data.get("thumbnailImage")
or data.get("imageUrl")
or data.get("thumbnail")
)
if isinstance(image_url, dict):
image_url = image_url.get("src") or image_url.get("url")
return Product(
name=name,
@@ -288,7 +539,7 @@ class BestBuyScraper(BaseScraper):
in_stock=in_stock,
image_url=image_url,
site=self.site_name,
product_id=str(sku_id),
product_id=str(sku_id) if sku_id else "",
)
except Exception as e:
@@ -298,10 +549,16 @@ class BestBuyScraper(BaseScraper):
def _parse_product_card(self, card) -> Optional[Product]:
"""Parse a product card element"""
try:
# Find link
link = card.select_one("a[href*='/site/'][href*='.p']") or card.select_one("a.image-link")
if not link:
link = card.select_one("a")
# Find link - try multiple patterns
link = (
card.select_one("a[href*='/product/']") # New Best Buy format
or card.select_one("a[href*='/site/'][href*='.p']") # Legacy format
or card.select_one("a[href*='skuId=']")
or card.select_one("a.image-link")
or card.select_one("[data-testid='product-link']")
or card.select_one("a[data-track]")
or card.select_one("a")
)
if not link:
return None
@@ -310,11 +567,16 @@ class BestBuyScraper(BaseScraper):
return None
url = href if href.startswith("http") else f"{self.base_url}{href}"
# Get name
# Get name - try multiple modern selectors
name_elem = (
card.select_one(".sku-title a")
or card.select_one("h4.sku-header a")
card.select_one("h4") # Current Best Buy format
or card.select_one("h3")
or card.select_one("[data-testid='product-title']")
or card.select_one("[data-testid='product-name']")
or card.select_one("[class*='productTitle']")
or card.select_one("[class*='ProductTitle']")
or card.select_one(".sku-title a")
or card.select_one("h4.sku-header a")
or card.select_one(".sku-title")
or link
)
@@ -324,10 +586,15 @@ class BestBuyScraper(BaseScraper):
if len(name) < 5:
return None
# Get price
# Get price - try multiple modern selectors
price_elem = (
card.select_one(".priceView-customer-price span")
card.select_one("div.pricing") # Current Best Buy format
or card.select_one("[class*='pricing']")
or card.select_one("[data-testid='customer-price']")
or card.select_one("[data-testid='current-price']")
or card.select_one("[class*='customerPrice']")
or card.select_one("[class*='CurrentPrice']")
or card.select_one(".priceView-customer-price span")
or card.select_one(".pricing-price__regular-price")
or card.select_one("[class*='price']")
)
@@ -348,11 +615,24 @@ class BestBuyScraper(BaseScraper):
if image_url and not image_url.startswith("http"):
image_url = f"https:{image_url}" if image_url.startswith("//") else f"{self.base_url}{image_url}"
# Extract SKU ID from URL
# Extract product ID from URL - handle multiple formats
# New: /product/product-name/ABC123 or /product/.../sku/12345
# Old: /site/.../12345.p
product_id = ""
match = re.search(r"/(\d+)\.p", url)
# Try /sku/12345 format first
match = re.search(r"/sku/(\d+)", url)
if match:
product_id = match.group(1)
else:
# Try old .p format
match = re.search(r"/(\d+)\.p", url)
if match:
product_id = match.group(1)
else:
# New format: last path segment is the ID
url_parts = url.rstrip('/').split('/')
if url_parts:
product_id = url_parts[-1]
return Product(
name=name,
@@ -382,11 +662,17 @@ class BestBuyScraper(BaseScraper):
if not name or len(name) < 5:
return None
# Extract SKU ID
# Extract SKU ID - handle both old and new formats
# New: /product/.../sku/12345
# Old: /site/.../12345.p
product_id = ""
match = re.search(r"/(\d+)\.p", url)
match = re.search(r"/sku/(\d+)", url) # New format first
if match:
product_id = match.group(1)
else:
match = re.search(r"/(\d+)\.p", url) # Old format
if match:
product_id = match.group(1)
# Find price near link
parent = link.find_parent()