使用Selenium爬取Steam独立游戏链接的异常问题求助
Steam独立游戏爬虫异常修复方案
问题现象
- 每次迭代预期获取12个游戏链接,实际仅能抓取3-6个
- 循环爬取时持续获取高度重复的内容,推测是Steam动态推荐机制导致
用户原代码
import csv from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.action_chains import ActionChains from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # Set up Chrome options for running in headless mode chrome_options = Options() chrome_options.add_argument("--headless") # Enable headless mode # Set up the Selenium webdriver with the specified options driver = webdriver.Chrome(options=chrome_options) # Replace with the path to your chromedriver executable # Define the base URL base_url = "https://store.steampowered.com/tags/en/Indie/?offset=" # Create a list to store the links and URLs data = [] # Iterate over the website IDs for website_id in range(12, 97, 12): url = base_url + str(website_id) driver.get(url) # Define the explicit wait with a maximum timeout of 10 seconds wait = WebDriverWait(driver, 10) # Scroll to the bottom of the page using JavaScript driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") # Find all the parent elements that trigger the mouse hover interaction parent_elements = wait.until(EC.visibility_of_all_elements_located((By.CSS_SELECTOR, '.salepreviewwidgets_TitleCtn_1F4bc'))) # Clear the data list data.clear() # Iterate over the parent elements for parent_element in parent_elements: # Move the mouse cursor to the parent element to trigger the mouse hover action actions = ActionChains(driver) actions.move_to_element(parent_element).perform() # Find the child element (link) within the parent element link_element = parent_element.find_element(By.CSS_SELECTOR, 'a') # Extract the link URL and add it to the data list link = link_element.get_attribute('href') data.append([link, url]) # Save the data to the CSV file by appending to existing content output_filename = "links.csv" with open(output_filename, 'a', newline='') as csvfile: writer = csv.writer(csvfile) writer.writerows(data) # Append the data to the file print("Data appended to", output_filename) # Close the webdriver driver.quit()
问题分析与修复
1. 抓取数量不足的修复
原因
- 滚动到底部后页面未完全加载就开始查找元素
- 无头模式下默认窗口尺寸过小,导致部分游戏卡片被隐藏
visibility_of_all_elements_located仅返回可见元素,未加载完成的元素会被过滤
修复措施
- 给无头模式设置固定窗口尺寸,确保所有卡片可见
- 滚动到底部后增加等待时间,等待新内容加载完成
- 先判断元素是否存在,再筛选可见元素进行提取
2. 重复抓取内容的修复
原因
Steam的offset参数会结合用户会话生成动态内容,直接修改URL访问无法准确获取对应分页的列表,页面会加载缓存或推荐内容
修复措施
- 模拟真实用户点击「加载更多」按钮的交互,而非直接修改URL
- 每次加载后等待新内容渲染完成,再提取当前页面所有游戏链接
修改后的完整代码
import csv import time from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # 配置Chrome无头模式 chrome_options = Options() chrome_options.add_argument("--headless") chrome_options.add_argument("--window-size=1920,1080") # 设置窗口尺寸,避免元素隐藏 driver = webdriver.Chrome(options=chrome_options) base_url = "https://store.steampowered.com/tags/en/Indie/" driver.get(base_url) output_filename = "links.csv" # 初始化CSV文件,写入表头 with open(output_filename, 'w', newline='') as csvfile: writer = csv.writer(csvfile) writer.writerow(["Game Link", "Source URL"]) wait = WebDriverWait(driver, 15) # 定义要抓取的页数(每页12个,这里抓8页共96个) total_pages = 8 current_page = 0 while current_page < total_pages: # 等待所有游戏卡片加载完成 game_cards = wait.until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, '.salepreviewwidgets_TitleCtn_1F4bc'))) # 提取当前页面所有可见卡片的链接 current_links = [] for card in game_cards: if card.is_displayed(): link_element = card.find_element(By.CSS_SELECTOR, 'a') link = link_element.get_attribute('href') current_links.append([link, driver.current_url]) # 写入CSV with open(output_filename, 'a', newline='') as csvfile: writer = csv.writer(csvfile) writer.writerows(current_links) print(f"已抓取第{current_page+1}页,共{len(current_links)}个链接") # 点击加载更多按钮 try: load_more_btn = wait.until(EC.element_to_be_clickable((By.CSS_SELECTOR, '.LoadMoreButton'))) driver.execute_script("arguments[0].scrollIntoView();", load_more_btn) time.sleep(1) load_more_btn.click() # 等待新内容加载 wait.until(EC.staleness_of(game_cards[0])) current_page += 1 time.sleep(2) except Exception as e: print(f"加载更多失败: {e}") break driver.quit() print("抓取完成,结果已保存到links.csv")
内容的提问来源于stack exchange,提问作者reresearchgames
相关产品推荐
相关产品推荐

