如何为Steam网页爬虫实现多线程以提升爬取效率?
给Steam爬虫集成多线程的改造方案
核心问题分析
原代码基于单个Chrome浏览器实例串行处理链接,每个页面的加载、渲染都要等待,8万条链接的耗时自然会非常长。另外,WebDriver实例不是线程安全的,不能多个线程共用同一个浏览器实例,所以正确思路是让每个线程独立维护自己的浏览器实例,同时用线程池控制并发数量,避免资源耗尽。
改造步骤
1. 拆分详情页爬取逻辑
把原link_tester中处理单条链接的代码抽成独立函数,让每个线程可以单独调用,确保每个线程有自己的浏览器环境:
from selenium.webdriver.support.ui import Select from selenium.webdriver.common.by import By from unidecode import clean import pandas as pd import time from concurrent.futures import ThreadPoolExecutor, as_completed from selenium import webdriver import os import random def scrape_game_detail(link, driver_path=r'PATH'): # 每个线程初始化独立的浏览器实例 os.environ['PATH'] += driver_path driver = webdriver.Chrome() driver.implicitly_wait(1) try: driver.get(link) # 处理年龄验证弹窗 try: dropdown = driver.find_element(By.XPATH, "//select[@id='ageYear']") dd = Select(dropdown) dd.select_by_value("1990") driver.find_element(By.XPATH, "//a[@id='view_product_page_btn']").click() except Exception: pass # 无年龄验证则跳过 driver.implicitly_wait(0) # 抓取评论数据 RecentReview = "null" FullReview = "null" try: element = driver.find_element(By.CSS_SELECTOR, "#userReviews") text = element.text.split("\n") RecentReview = text[1] FullReview = text[3] except Exception: pass # 抓取游戏基础信息 Title = "" Genre_s = "" Developer_s = "" Publisher_s = "" ReleaseDate = "" try: element = driver.find_element(By.CSS_SELECTOR, "#genresAndManufacturer") text = element.text.split("\n") text = [x for x in text if not x.startswith('FRANCHISE:')] Title = clean(text[0], fix_unicode=True, to_ascii=True).replace("title: ", "") Genre_s = clean(text[1], fix_unicode=True, to_ascii=True).replace("genre: ", "") Developer_s = clean(text[2], fix_unicode=True, to_ascii=True).replace("developer: ","") Publisher_s = clean(text[3], fix_unicode=True, to_ascii=True).replace("publisher: ","") ReleaseDate = clean(text[4], fix_unicode=True, to_ascii=True).replace("release date: ","") except Exception: pass # 抓取价格信息 BasePrice = "" try: element = driver.find_element(By.CLASS_NAME, "game_purchase_action_bg") text = clean(element.text, no_line_breaks=True).replace("add to cart", "") BasePrice = text except Exception: pass # 随机延迟,降低反爬风险 time.sleep(random.uniform(0.5, 1.5)) return [Title, Genre_s, Developer_s, Publisher_s, ReleaseDate, RecentReview, FullReview, BasePrice] finally: # 确保浏览器关闭,释放资源 driver.quit()
2. 修改主爬虫类,集成线程池
原类只负责获取游戏链接,然后通过线程池批量并行处理详情页爬取:
class Steamscrape(webdriver.Chrome): def __init__(self, driver_path=r'PATH', teardown=False): self.driver_path = driver_path self.teardown = teardown os.environ['PATH'] += self.driver_path super().__init__() self.implicitly_wait(1) self.maximize_window() def __exit__(self, exc_type, exc_val, exc_tb): if self.teardown: self.quit() def land_first_page(self): self.get(const.BASE_URL) # 替换为你的Steam列表页地址 def scroll_down(self): SCROLL_PAUSE_TIME = 0.5 last_height = self.execute_script("return document.body.scrollHeight") while True: self.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(SCROLL_PAUSE_TIME) new_height = self.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height def get_links(self): links = self.find_elements(By.CSS_SELECTOR, "a[href^='https://store.steampowered.com/app/']") link_list = [] for link in links: href = link.get_attribute("href") if href not in link_list: # 去重,避免重复处理同一链接 link_list.append(href) # 原代码仅取12条,爬8万条请删除以下两行 # if len(link_list) == 12: # break return link_list def load_data(self, Gameslist): df = pd.DataFrame(Gameslist, columns=["Title", "Genre", "Developers", "Publisher", "Release Date", "Recent Reviews", "Total Reviews", "Price"]) df.to_csv(r'PATH') # 替换为你的数据保存路径 print(df.head(10)) def run(self): start_time = time.time() # 第一步:获取所有游戏链接 self.land_first_page() self.scroll_down() link_list = self.get_links() print(f"共获取到 {len(link_list)} 条游戏链接") # 第二步:线程池并行处理链接 GamesList = [] # 控制并发数,根据机器配置调整,建议10-20之间 with ThreadPoolExecutor(max_workers=12) as executor: # 提交所有爬取任务 future_to_link = {executor.submit(scrape_game_detail, link, self.driver_path): link for link in link_list} # 逐个获取任务结果 for future in as_completed(future_to_link): link = future_to_link[future] try: game_data = future.result() GamesList.append(game_data) print(f"已完成 {len(GamesList)}/{len(link_list)} 条链接爬取") except Exception as exc: print(f"链接 {link} 爬取失败: {str(exc)}") # 第三步:保存数据到CSV self.load_data(GamesList) total_time = time.time() - start_time print(f"总耗时: {total_time:.2f} 秒")
3. 关键注意事项
- 并发数控制:
max_workers不要设置过大,否则会同时打开大量Chrome窗口,导致CPU、内存占用过高,甚至触发Steam反爬机制。8核CPU建议设置10-15。 - 反爬应对:加入随机延迟、避免固定请求频率,必要时可搭配代理IP池,降低被封禁的风险。
- 资源释放:每个线程的浏览器实例必须在
finally块中关闭,防止内存泄漏。 - 链接去重:列表页可能存在重复链接,加入去重逻辑避免无效爬取。
额外优化方向
如果觉得多开Chrome资源占用过高,可以改用requests+BeautifulSoup组合直接请求页面,配合requests-html处理动态渲染内容,这种方式资源占用极低,并发数可设置更高。若Steam有API接口,直接调用API效率会远高于页面爬取。
内容的提问来源于stack exchange,提问作者Ornsteiner
相关产品推荐
相关产品推荐

