Python中Selenium(Chromedriver)for循环迭代耗时过长的优化咨询
问题描述
我是Python初学者,需要从一组结构一致的URL中提取数据,目标网站的JavaScript会修改初始HTML,所以用Selenium获取最终HTML。目前每次迭代耗时约4秒,其中wait.until(page_has_loaded)占一半时间,想优化性能。以下是我的代码:
import win32com.client as win32 import requests import openpyxl import time from bs4 import BeautifulSoup from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait headers = {"User-Agent": 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/110.0.0.0 Safari/537.36'} driver_path = 'C:\\webdrivers\\chromedriver.exe' dir = "C:\\Users\\Me\\OneDrive\\Dokumente_\\Notizen\\CSGOItems\\CSGOItems.xlsx" workbook = openpyxl.load_workbook(dir) sheet1 = workbook["Tabelle1"] sheet2 = workbook["AllPrices"] URLSkinBit = [ 'https://skinbid.com/auctions?mh=Gamma%20Case&sellType=fixed_price&skip=0&take=30&sort=price%23asc&ref=csgoskins', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=gamma&sellType=all', 'https://skinbid.com/auctions?mh=Danger%20Zone%20Case&sellType=fixed_price&skip=0&take=30&sort=price%23asc&ref=csgoskins', 'https://skinbid.com/auctions?mh=Dreams%20%26%20Nightmares%20Case&sellType=fixed_price&skip=0&take=30&sort=price%23asc&ref=csgoskins', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=vanguard&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=chroma%203&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=spectrum%202&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=clutch&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=snakebite&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=falchion&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=fracture&sellType=all', 'https://skinbid.com/listings?popular=false&goodDeals=false&sort=price%23asc&take=10&skip=0&search=prisma%202&sellType=all', ] def page_has_loaded(driver): return driver.execute_script("return document.readyState") == "complete" def SkinBitPrices(): global count3 count3 = 0 with webdriver.Chrome(executable_path=driver_path) as driver: for url in URLSkinBit: driver.get(url) wait = WebDriverWait(driver, 10) wait.until(page_has_loaded) html = driver.page_source soup = BeautifulSoup(html, 'html.parser') container = soup.find('div', {'class': 'price'}).text price = float(container.replace(' € ', '')) print("%.2f" % price) #Edit Excel-File cell = str(3 + count3) sheet2['B' + cell] = price count3 += 1 driver.quit() workbook.save(dir) workbook.close() return SkinBitPrices()
优化方案
替换等待逻辑,直接等待目标元素
原代码等待document.readyState == "complete"仅表示DOM加载完成,不代表动态数据渲染完毕。直接等待你要提取的.price元素出现,能大幅减少无效等待时间。需要导入额外模块:from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.common.by import By # 替换原等待代码 wait = WebDriverWait(driver, 10) # 等待第一个.price元素出现 price_element = wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'price')))启用Chrome无头模式
无头模式无需渲染可视化界面,能节省大量资源和时间。初始化driver时添加配置:options = Options() options.add_argument('--headless=new') # 新版无头模式,更接近正常Chrome行为 options.add_argument('--disable-gpu') # 禁用GPU加速,避免部分环境报错 with webdriver.Chrome(executable_path=driver_path, options=options) as driver: # 后续代码不变跳过BeautifulSoup,直接用Selenium提取数据
没必要将HTML传给BeautifulSoup二次解析,直接用Selenium的元素定位提取文本,减少中间步骤:# 替换原soup解析部分 price_text = price_element.text.strip() price = float(price_text.replace(' € ', ''))批量写入Excel,减少IO操作
原代码每次循环都修改Excel单元格,频繁IO会拖慢速度。先将所有价格存入列表,最后一次性写入:def SkinBitPrices(): prices = [] with webdriver.Chrome(executable_path=driver_path, options=options) as driver: for url in URLSkinBit: driver.get(url) wait = WebDriverWait(driver, 10) price_element = wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'price'))) price_text = price_element.text.strip() price = float(price_text.replace(' € ', '')) prices.append(price) print("%.2f" % price) # 最后批量写入Excel for idx, price in enumerate(prices): sheet2[f'B{3 + idx}'] = price workbook.save(dir) workbook.close()禁用不必要的浏览器特性
添加更多Chrome参数,进一步减少加载时间:options.add_argument('--no-sandbox') options.add_argument('--disable-dev-shm-usage') options.add_argument('--disable-images') # 不需要图片时禁用,能大幅提速 options.add_argument('--blink-settings=imagesEnabled=false')
内容的提问来源于stack exchange,提问作者Do0dl3r
相关产品推荐
相关产品推荐

