""" GameStop.com scraper """ import re import logging import time import sys import os from typing import List, Optional from bs4 import BeautifulSoup from .base import BaseScraper, Product from src.browser import get_browser from config import CAPTCHA_WAIT_TIMEOUT # Add parent directory to path for tools import sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) logger = logging.getLogger(__name__) def warmup_gamestop(): """ Warmup function to solve Cloudflare CAPTCHA manually. Uses undetected-chromedriver to bypass bot detection. """ from tools.stealth_browser import StealthBrowser print("\n" + "=" * 60) print("GAMESTOP WARMUP - Using Undetected Chrome") print("=" * 60) print("A browser window should open.") print("If you see a Cloudflare challenge, solve it manually.") print(f"Waiting up to {CAPTCHA_WAIT_TIMEOUT} seconds...") print("=" * 60 + "\n") logger.info("Starting GameStop warmup with undetected-chromedriver...") # Use stealth browser instead of Playwright browser = StealthBrowser(headless=False, session_name="gamestop") try: browser.start() logger.info("Stealth browser started") # Navigate to GameStop browser.driver.get("https://www.gamestop.com") time.sleep(3) # Check for Cloudflare challenge start_time = time.time() challenge_detected = False while time.time() - start_time < CAPTCHA_WAIT_TIMEOUT: try: html = browser.driver.page_source html_lower = html.lower() title = browser.driver.title.lower() # Cloudflare challenge indicators is_cloudflare_challenge = ( "just a moment" in title or "checking your browser" in html_lower or "cf-challenge" in html_lower or "turnstile" in html_lower or (len(html) < 5000 and "challenge" in html_lower) ) if is_cloudflare_challenge: if not challenge_detected: challenge_detected = True print(">>> Cloudflare challenge detected! Please solve it in the browser window.") logger.info("Cloudflare challenge detected - waiting for manual solve...") time.sleep(5) else: # Check if we're on actual GameStop content if len(html) > 10000 and "gamestop" in html_lower: print(">>> Challenge solved! GameStop page loaded successfully.") logger.info("GameStop warmup complete!") browser.stop() return True time.sleep(2) except Exception as e: logger.debug(f"Error checking page: {e}") time.sleep(2) logger.warning("GameStop warmup timed out - CAPTCHA may not be solved") print(">>> Warmup timed out. You may need to try again.") browser.stop() return False except Exception as e: logger.error(f"Error during warmup: {e}") try: browser.stop() except: pass return False class GameStopScraper(BaseScraper): """Scraper for GameStop.com - Uses undetected-chromedriver to bypass Cloudflare""" site_name = "gamestop" base_url = "https://www.gamestop.com" def __init__(self): self._stealth_browser = None self._restart_attempts = 0 self._max_restart_attempts = 3 def _get_stealth_browser(self): """Get or create stealth browser for GameStop, with auto-restart on session errors""" from tools.stealth_browser import StealthBrowser # Check if we need to create a new browser if self._stealth_browser is None: logger.info("Creating new GameStop stealth browser...") self._stealth_browser = StealthBrowser(headless=False, session_name="gamestop") self._stealth_browser.start() self._restart_attempts = 0 return self._stealth_browser # Check if existing session is still valid if not self._stealth_browser.is_session_valid(): if self._restart_attempts >= self._max_restart_attempts: logger.error(f"GameStop browser failed after {self._max_restart_attempts} restart attempts") # Reset counter and try one more time after clearing self._restart_attempts = 0 self._stealth_browser = None return self._get_stealth_browser() logger.warning("GameStop browser session invalid, restarting...") self._restart_attempts += 1 self._stealth_browser.restart() return self._stealth_browser def _handle_session_error(self, error: Exception) -> bool: """ Check if error is a session error and handle it. Returns True if browser was restarted and operation should be retried. """ error_msg = str(error).lower() session_errors = ["invalid session id", "session deleted", "no such session", "browser has closed"] if any(err in error_msg for err in session_errors): logger.warning(f"Session error detected: {error}") if self._stealth_browser and self._restart_attempts < self._max_restart_attempts: self._restart_attempts += 1 try: self._stealth_browser.restart() return True # Retry operation except Exception as e: logger.error(f"Failed to restart browser: {e}") return False def scrape_category_page(self, url: str, max_pages: int = 1) -> List[Product]: """ Scrape a GameStop search/category page for all products Uses undetected-chromedriver to bypass Cloudflare Args: url: Search/category page URL max_pages: Maximum number of pages to scrape (default 1) Returns: List of Product objects """ all_products = [] # Add sort by newest if not already in URL if "sort=" not in url.lower(): separator = "&" if "?" in url else "?" url = f"{url}{separator}sort=date-desc" # Use stealth browser for GameStop try: browser = self._get_stealth_browser() except Exception as e: logger.error(f"Failed to start stealth browser: {e}") return all_products for page_num in range(1, max_pages + 1): products = [] page_url = url if page_num == 1 else f"{url}&start={24 * (page_num - 1)}" retry_count = 0 max_retries = 2 while retry_count <= max_retries: try: logger.info(f"Navigating to: {page_url}") browser.driver.get(page_url) time.sleep(5) # Wait for page load html = browser.driver.page_source break # Success, exit retry loop except Exception as e: if self._handle_session_error(e) and retry_count < max_retries: retry_count += 1 logger.info(f"Retrying after browser restart (attempt {retry_count}/{max_retries})") browser = self._get_stealth_browser() # Get restarted browser continue else: logger.error(f"Error navigating to {page_url}: {e}") return all_products else: # while loop completed without break (all retries exhausted) logger.error(f"Failed to load page after {max_retries} retries") return all_products try: # Check for Cloudflare challenge html_lower = html.lower() title = browser.driver.title.lower() is_cloudflare = ( "just a moment" in title or "checking your browser" in html_lower or "cf-challenge" in html_lower ) if is_cloudflare: logger.warning("Cloudflare challenge detected! Waiting for manual solve...") start_time = time.time() while time.time() - start_time < CAPTCHA_WAIT_TIMEOUT: time.sleep(5) html = browser.driver.page_source html_lower = html.lower() title = browser.driver.title.lower() if "just a moment" not in title and "cf-challenge" not in html_lower: logger.info("Cloudflare challenge solved!") break else: logger.error("Cloudflare challenge not solved in time") return all_products # Save debug screenshot try: browser.driver.save_screenshot("debug_gamestop.png") logger.info("Saved debug screenshot to debug_gamestop.png") except: pass soup = BeautifulSoup(html, "html.parser") # Try multiple selectors for product cards product_cards = ( soup.select(".product-tile") or soup.select("[data-testid='product-tile']") or soup.select(".grid-tile") or soup.select(".product-grid-tile") or soup.select("[class*='ProductTile']") ) logger.info(f"Found {len(product_cards)} product cards") # If no cards found, try link-based extraction if not product_cards: product_links = soup.select("a[href*='/products/']") logger.info(f"Fallback: Found {len(product_links)} product links") # Group by href href_to_links = {} for link in product_links: href = link.get("href", "") if not href or "/products/" not in href: continue if href not in href_to_links: href_to_links[href] = [] href_to_links[href].append(link) for href, links in href_to_links.items(): # Pick best link best_link = links[0] for link in links: if link.get("aria-label") or len(link.get_text(strip=True)) > 10: best_link = link break product = self._parse_product_link(best_link, soup) if product: products.append(product) else: for card in product_cards: product = self._parse_product_card(card) if product: products.append(product) # Don't close - reuse browser for next page except Exception as e: logger.error(f"Error scraping GameStop page {page_num}: {e}") break # Don't raise - just return what we have all_products.extend(products) # Stop if no products found on this page if not products: break # Remove duplicates seen_urls = set() unique_products = [] for p in all_products: if p.url not in seen_urls: seen_urls.add(p.url) unique_products.append(p) logger.info(f"Scraped {len(unique_products)} unique products from GameStop") return unique_products def _parse_product_card(self, card) -> Optional[Product]: """Parse a product card element""" try: # Find link - try multiple patterns link = ( card.select_one("a[href*='/products/']") or card.select_one("a[href*='/video-games/']") or card.select_one("a[href*='/collectibles/']") or card.select_one("a.product-tile-link") or card.select_one("a") ) if not link: logger.debug("No link found in card") return None href = link.get("href", "") if not href: logger.debug("Empty href") return None url = href if href.startswith("http") else f"{self.base_url}{href}" # Get name - try multiple patterns name_elem = ( card.select_one(".product-tile-title") or card.select_one(".product-name a") or card.select_one(".product-name") or card.select_one(".product-title") or card.select_one("[data-testid='product-name']") or card.select_one("a[aria-label]") or link ) # Try to get name from aria-label first name = "" if name_elem: name = name_elem.get("aria-label", "") or name_elem.get_text(strip=True) # Skip if name too short if len(name) < 5: logger.debug(f"Name too short: '{name}' from {url[:50]}") return None # Get price price_elem = ( card.select_one(".price-sales") or card.select_one(".product-price") or card.select_one("[data-testid='price']") or card.select_one("[class*='price']") ) price = None if price_elem: price_match = re.search(r"\$[\d,]+\.?\d*", price_elem.get_text()) if price_match: price = price_match.group() # Check stock in_stock = self._check_card_stock_status(card) # Get image img = card.select_one("img") image_url = None if img: image_url = img.get("src") or img.get("data-src") or img.get("data-lazy") if image_url and not image_url.startswith("http"): image_url = f"{self.base_url}{image_url}" # Extract product ID from URL product_id = "" match = re.search(r"/products/[^/]+/(\d+)", url) if match: product_id = match.group(1) return Product( name=name, url=url, price=price, in_stock=in_stock, image_url=image_url, site=self.site_name, product_id=product_id, ) except Exception as e: logger.debug(f"Error parsing GameStop product card: {e}") return None def _parse_product_link(self, link, soup) -> Optional[Product]: """Parse a product from a product link element""" try: href = link.get("href", "") if not href: return None url = href if href.startswith("http") else f"{self.base_url}{href}" # Get name name = link.get("aria-label", "") or link.get_text(strip=True) # Skip if name too short if not name or len(name) < 5: return None # Extract product ID from URL product_id = "" match = re.search(r"/products/[^/]+/(\d+)", url) if match: product_id = match.group(1) # Try to find price near this link parent = link.find_parent() price = None for _ in range(5): if parent: price_text = parent.get_text() price_match = re.search(r"\$[\d,]+\.?\d*", price_text) if price_match: price = price_match.group() break parent = parent.find_parent() # Try to find image image_url = None parent = link.find_parent() for _ in range(5): if parent: img = parent.select_one("img") if img: image_url = img.get("src") or img.get("data-src") if image_url and not image_url.startswith("http"): image_url = f"{self.base_url}{image_url}" break parent = parent.find_parent() # Check stock in_stock = True parent = link.find_parent() for _ in range(5): if parent: text = parent.get_text().lower() if any(phrase in text for phrase in ["sold out", "out of stock", "not available"]): in_stock = False break parent = parent.find_parent() return Product( name=name, url=url, price=price, in_stock=in_stock, image_url=image_url, site=self.site_name, product_id=product_id, ) except Exception as e: logger.debug(f"Error parsing GameStop product link: {e}") return None def _check_card_stock_status(self, card) -> bool: """Check if a product card indicates in-stock status""" card_text = card.get_text().lower() if hasattr(card, "get_text") else str(card).lower() out_of_stock_phrases = [ "sold out", "out of stock", "not available", "unavailable", "currently unavailable", ] for phrase in out_of_stock_phrases: if phrase in card_text: return False in_stock_phrases = [ "add to cart", "add to bag", "available", "buy now", "in stock", ] for phrase in in_stock_phrases: if phrase in card_text: return True # Default: assume in stock if listed return True def check_product_stock(self, product_url: str) -> tuple[bool, Optional[str]]: """Check if a specific product is in stock""" try: browser = self._get_stealth_browser() browser.driver.get(product_url) time.sleep(3) html = browser.driver.page_source soup = BeautifulSoup(html, "html.parser") page_text = soup.get_text().lower() out_of_stock = any( phrase in page_text for phrase in ["sold out", "out of stock", "not available", "unavailable"] ) in_stock = not out_of_stock # Get price price = None price_elem = ( soup.select_one(".product-price") or soup.select_one(".price-sales") or soup.select_one("[data-testid='price']") ) if price_elem: price_match = re.search(r"\$[\d,]+\.?\d*", price_elem.get_text()) if price_match: price = price_match.group() return in_stock, price except Exception as e: logger.error(f"Error checking GameStop product stock: {e}") return False, None