You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何为Steam网页爬虫实现多线程以提升爬取效率?

给Steam爬虫集成多线程的改造方案

核心问题分析

原代码基于单个Chrome浏览器实例串行处理链接,每个页面的加载、渲染都要等待,8万条链接的耗时自然会非常长。另外,WebDriver实例不是线程安全的,不能多个线程共用同一个浏览器实例,所以正确思路是让每个线程独立维护自己的浏览器实例,同时用线程池控制并发数量,避免资源耗尽。

改造步骤

1. 拆分详情页爬取逻辑

把原link_tester中处理单条链接的代码抽成独立函数,让每个线程可以单独调用,确保每个线程有自己的浏览器环境:

from selenium.webdriver.support.ui import Select
from selenium.webdriver.common.by import By
from unidecode import clean
import pandas as pd
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from selenium import webdriver
import os
import random

def scrape_game_detail(link, driver_path=r'PATH'):
    # 每个线程初始化独立的浏览器实例
    os.environ['PATH'] += driver_path
    driver = webdriver.Chrome()
    driver.implicitly_wait(1)
    
    try:
        driver.get(link)
        
        # 处理年龄验证弹窗
        try:
            dropdown = driver.find_element(By.XPATH, "//select[@id='ageYear']")
            dd = Select(dropdown)
            dd.select_by_value("1990")
            driver.find_element(By.XPATH, "//a[@id='view_product_page_btn']").click()
        except Exception:
            pass  # 无年龄验证则跳过
        
        driver.implicitly_wait(0)
        
        # 抓取评论数据
        RecentReview = "null"
        FullReview = "null"
        try:
            element = driver.find_element(By.CSS_SELECTOR, "#userReviews")
            text = element.text.split("\n")
            RecentReview = text[1]
            FullReview = text[3]
        except Exception:
            pass
        
        # 抓取游戏基础信息
        Title = ""
        Genre_s = ""
        Developer_s = ""
        Publisher_s = ""
        ReleaseDate = ""
        try:
            element = driver.find_element(By.CSS_SELECTOR, "#genresAndManufacturer")
            text = element.text.split("\n")
            text = [x for x in text if not x.startswith('FRANCHISE:')]
            Title = clean(text[0], fix_unicode=True, to_ascii=True).replace("title: ", "")
            Genre_s = clean(text[1], fix_unicode=True, to_ascii=True).replace("genre: ", "")
            Developer_s = clean(text[2], fix_unicode=True, to_ascii=True).replace("developer: ","")
            Publisher_s = clean(text[3], fix_unicode=True, to_ascii=True).replace("publisher: ","")
            ReleaseDate = clean(text[4], fix_unicode=True, to_ascii=True).replace("release date: ","")
        except Exception:
            pass
        
        # 抓取价格信息
        BasePrice = ""
        try:
            element = driver.find_element(By.CLASS_NAME, "game_purchase_action_bg")
            text = clean(element.text, no_line_breaks=True).replace("add to cart", "")
            BasePrice = text
        except Exception:
            pass
        
        # 随机延迟,降低反爬风险
        time.sleep(random.uniform(0.5, 1.5))
        
        return [Title, Genre_s, Developer_s, Publisher_s, ReleaseDate, RecentReview, FullReview, BasePrice]
    
    finally:
        # 确保浏览器关闭,释放资源
        driver.quit()

2. 修改主爬虫类,集成线程池

原类只负责获取游戏链接,然后通过线程池批量并行处理详情页爬取:

class Steamscrape(webdriver.Chrome):
    def __init__(self, driver_path=r'PATH', teardown=False):
        self.driver_path = driver_path
        self.teardown = teardown
        os.environ['PATH'] += self.driver_path
        super().__init__()
        self.implicitly_wait(1)
        self.maximize_window()

    def __exit__(self, exc_type, exc_val, exc_tb):
        if self.teardown:
            self.quit()

    def land_first_page(self):
        self.get(const.BASE_URL)  # 替换为你的Steam列表页地址

    def scroll_down(self):
        SCROLL_PAUSE_TIME = 0.5
        last_height = self.execute_script("return document.body.scrollHeight")
        while True:
            self.execute_script("window.scrollTo(0, document.body.scrollHeight);")
            time.sleep(SCROLL_PAUSE_TIME)
            new_height = self.execute_script("return document.body.scrollHeight")
            if new_height == last_height:
                break
            last_height = new_height

    def get_links(self):
        links = self.find_elements(By.CSS_SELECTOR, "a[href^='https://store.steampowered.com/app/']")
        link_list = []
        for link in links:
            href = link.get_attribute("href")
            if href not in link_list:  # 去重,避免重复处理同一链接
                link_list.append(href)
            # 原代码仅取12条,爬8万条请删除以下两行
            # if len(link_list) == 12:
            #     break
        return link_list

    def load_data(self, Gameslist):
        df = pd.DataFrame(Gameslist, columns=["Title", "Genre", "Developers", "Publisher", "Release Date", "Recent Reviews", "Total Reviews", "Price"])
        df.to_csv(r'PATH')  # 替换为你的数据保存路径
        print(df.head(10))

    def run(self):
        start_time = time.time()
        
        # 第一步:获取所有游戏链接
        self.land_first_page()
        self.scroll_down()
        link_list = self.get_links()
        print(f"共获取到 {len(link_list)} 条游戏链接")
        
        # 第二步:线程池并行处理链接
        GamesList = []
        # 控制并发数,根据机器配置调整,建议10-20之间
        with ThreadPoolExecutor(max_workers=12) as executor:
            # 提交所有爬取任务
            future_to_link = {executor.submit(scrape_game_detail, link, self.driver_path): link for link in link_list}
            
            # 逐个获取任务结果
            for future in as_completed(future_to_link):
                link = future_to_link[future]
                try:
                    game_data = future.result()
                    GamesList.append(game_data)
                    print(f"已完成 {len(GamesList)}/{len(link_list)} 条链接爬取")
                except Exception as exc:
                    print(f"链接 {link} 爬取失败: {str(exc)}")
        
        # 第三步:保存数据到CSV
        self.load_data(GamesList)
        
        total_time = time.time() - start_time
        print(f"总耗时: {total_time:.2f} 秒")

3. 关键注意事项

  • 并发数控制:max_workers不要设置过大,否则会同时打开大量Chrome窗口,导致CPU、内存占用过高,甚至触发Steam反爬机制。8核CPU建议设置10-15。
  • 反爬应对:加入随机延迟、避免固定请求频率,必要时可搭配代理IP池,降低被封禁的风险。
  • 资源释放:每个线程的浏览器实例必须在finally块中关闭,防止内存泄漏。
  • 链接去重:列表页可能存在重复链接,加入去重逻辑避免无效爬取。

额外优化方向

如果觉得多开Chrome资源占用过高,可以改用requests+BeautifulSoup组合直接请求页面,配合requests-html处理动态渲染内容,这种方式资源占用极低,并发数可设置更高。若Steam有API接口,直接调用API效率会远高于页面爬取。

内容的提问来源于stack exchange,提问作者Ornsteiner

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 11:51:58