如何解决Selenium爬虫在Headless模式下无法找到滚动元素的问题?
问题描述
我正在为blur.io开发一款获取NFT借贷数据的Selenium网络爬虫,该爬虫在非Headless模式下运行完全正常,但在Headless模式下无法找到用于滚动加载内容的可滚动元素,导致脚本报错。
我已尝试添加以下修复参数:
options.add_argument("--headless=new") options.add_argument("--window-size=1440, 900") options.add_argument('--disable-gpu') options.add_argument('--no-sandbox') options.add_argument("--start-maximized")
同时我也使用WebDriverWait等待元素可见,但仍然无法找到该元素并报错:
WebDriverWait(driver,20).until(EC.visibility_of_element_located((By.CLASS_NAME, 'rows')))
完整代码如下:
from selenium import webdriver from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options import time from tkinter import * from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC global nftData def removePRCNT(string): return float(string.replace("%", "")) nftData = [] def execute_loan_checker(apyThreshold, ltvThreshold, ethThreshold): global nftData del nftData[:] path = "MYPATH/YOURPATH" service = Service(path) options = Options() #OPTIONS IVE TRIED, DIDNT WORK TO FIX HEADLESS ISSUE options.add_argument("--headless=new") #works fine without this line options.add_argument("--window-size=1440, 900") options.add_argument('--disable-gpu') options.add_argument('--no-sandbox') options.add_argument("--start-maximized") #OTHER MISC OPTIONS options.add_experimental_option("detach", True) options.add_experimental_option("excludeSwitches",["enable-automation"]) driver = webdriver.Chrome(service=service, options=options) collection_links = ["https://blur.io/eth/collection/wrapped-cryptopunks/loans", "https://blur.io/eth/collection/azuki/loans", "https://blur.io/eth/collection/milady/loans", "https://blur.io/eth/collection/degods-eth/loans", "https://blur.io/eth/collection/boredapeyachtclub/loans", "https://blur.io/eth/collection/mutant-ape-yacht-club/loans", "https://blur.io/eth/collection/kanpai-pandas/loans", "https://blur.io/eth/collection/remilio-babies/loans", "https://blur.io/eth/collection/pudgypenguins/loans", "https://blur.io/eth/collection/otherdeed/loans", "https://blur.io/eth/collection/bored-ape-kennel-club/loans", "https://blur.io/eth/collection/clonex/loans", "https://blur.io/eth/collection/beanzofficial/loans", "https://blur.io/eth/collection/azukielementalbeans/loans", "https://blur.io/eth/collection/azukielementals/loans", "https://blur.io/eth/collection/proof-moonbirds/loans", "https://blur.io/eth/collection/lilpudgys/loans"] def gatherLoanData(): addedNFTnames = [] for link in collection_links: driver.get(link) #waiting until element is clickable then click it loans_button = WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.XPATH, "//button[.='All Loans']"))) loans_button.click() time.sleep(.4) #might need to adjust sleep time based on computer speed, caused errors depending on wait timing #THIS IS WHERE ITS BEEN GETTING STUCK WebDriverWait(driver, 20).until(EC.visibility_of_element_located((By.CLASS_NAME, 'rows'))) scrollable_element = WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.CLASS_NAME, "rows"))) scroll_amount = 500 # Amount of pixels to scroll each time status = "AUCTION" #set status to auction for first iteration # Only scrolls while the status is AUCTION, to get live loans while status == "AUCTION": # Scroll down by scroll_amount pixels each time print("pp") for loan_row in driver.find_elements(By.XPATH, "//div[@id= 'COLLECTION_MAIN']//div[@role='rowgroup']//div[@role='row']"): nftName = loan_row.find_element(By.XPATH, "div[1]").text #get nft title status = loan_row.find_element(By.XPATH, "div[2]").text #get auction/active status to filter if status == "ACTIVE": break borrowAmount = loan_row.find_element(By.XPATH, "div[3]").text # get borrow amount ltv = loan_row.find_element(By.XPATH, "div[4]").text # get the ltv value apy = loan_row.find_element(By.XPATH, "div[5]").text # get the apy value if nftName not in addedNFTnames and ethThreshold > float(borrowAmount) and ltvThreshold > removePRCNT(ltv) and removePRCNT(apy) > apyThreshold: nftData.append([nftName, borrowAmount, ltv, apy]) addedNFTnames.append(nftName) #add to list of nfts, to check that it hasnt been added again driver.execute_script('arguments[0].scrollTop = arguments[0].scrollTop + {};'.format(scroll_amount), scrollable_element) time.sleep(.05) # Delay, might need to be increased based on load speed gatherLoanData() driver.close() return nftData execute_loan_checker(0,999,999) #CALLS SCRIPT WITH NO FILTERING OPTIONS FOR TESTING
解决方案
1. 添加真实用户代理
Headless模式下Chrome的UA会带有HeadlessChrome标识,网站可能返回不同DOM结构,添加真实UA模拟普通浏览器:
options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
2. 替换固定sleep为显式等待
点击"All Loans"后用固定sleep可能在Headless下加载不及时,改为等待目标元素出现:
# 替换原有的time.sleep(.4) WebDriverWait(driver, 10).until(EC.presence_of_element_located((By.CLASS_NAME, 'rows')))
3. 优化元素定位逻辑
Headless模式下元素class可能存在差异,改用更精确的XPath定位滚动容器:
# 替换原有的rows元素定位 scrollable_element = WebDriverWait(driver, 20).until( EC.element_to_be_clickable((By.XPATH, "//div[@id='COLLECTION_MAIN']//div[contains(@class, 'rows')]")) )
4. 调整滚动策略
如果滚动容器仍无法定位,直接滚动整个页面替代元素滚动:
# 替换原有的元素滚动代码 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
5. 优化Headless模式配置
移除可能导致进程残留的配置,添加禁用自动化扩展参数:
options = Options() options.add_argument("--headless=new") options.add_argument("--window-size=1920,1080") # 使用更贴近桌面的分辨率 options.add_argument('--disable-gpu') options.add_argument('--no-sandbox') options.add_argument("--start-maximized") options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") options.add_experimental_option("excludeSwitches",["enable-automation"]) options.add_experimental_option('useAutomationExtension', False) # 移除detach参数,避免Headless模式下进程残留 # options.add_experimental_option("detach", True)
6. 优化循环内元素获取
循环内重复调用find_elements效率低且不稳定,改为先一次性获取所有行元素:
# 替换原有的while循环内逻辑 while status == "AUCTION": # 等待所有行元素加载完成 rows = WebDriverWait(driver, 10).until( EC.presence_of_all_elements_located((By.XPATH, "//div[@id= 'COLLECTION_MAIN']//div[@role='rowgroup']//div[@role='row']")) ) for loan_row in rows: # 原有逻辑保持不变 nftName = loan_row.find_element(By.XPATH, "div[1]").text status = loan_row.find_element(By.XPATH, "div[2]").text if status == "ACTIVE": break borrowAmount = loan_row.find_element(By.XPATH, "div[3]").text ltv = loan_row.find_element(By.XPATH, "div[4]").text apy = loan_row.find_element(By.XPATH, "div[5]").text if nftName not in addedNFTnames and ethThreshold > float(borrowAmount) and ltvThreshold > removePRCNT(ltv) and removePRCNT(apy) > apyThreshold: nftData.append([nftName, borrowAmount, ltv, apy]) addedNFTnames.append(nftName) # 执行滚动操作 driver.execute_script('arguments[0].scrollTop = arguments[0].scrollTop + {};'.format(scroll_amount), scrollable_element) time.sleep(.05)
内容的提问来源于stack exchange,提问作者number2patrician
相关产品推荐
相关产品推荐

